diff --git a/CHANGELOG.md b/CHANGELOG.md index 3c07c3c..eecdceb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,30 @@ # Changelog +## 14.0.0 — 2026-09-05 + +### Changed + +- `ResearchDriver` gains optional `isComplete()`. +The research loop requires this result and storage readiness before it reports ready. +An unfinished driver receives steering rounds even when storage requirements pass. +Drivers without this method retain their storage readiness behavior. +The exported interface shape requires a major release under the package compatibility check. +- Requires `agent-eval` `>=0.174.0 <0.175.0` and tests against `0.174.0`. +This cohort uses Eval's corrected complete-method result and cost accounting contracts. +- Default knowledge evaluator version `2` averages only measured dimensions. +It omits `answer_quality` without answer evaluation, `promotion_decision` without a decision, and `blocking_readiness` without blocking requirements. +Consumers must handle absent dimension keys and compare scores using their evaluator version. +Structural-only results state that no task outcome evaluation occurred. +`candidate-ready` still leaves the candidate detached from the live knowledge base. + +### Fixed + +- Knowledge diagnosis runs before acquisition and updates. +One lifecycle carries findings, acquisition, and update results into final answer checks and the promotion decision. +Diagnosis alone does not consume final evaluation cases. +Disabling a required phase fails before candidate work starts. +Final measurement still uses frozen candidate bytes and cannot feed another adaptive update. + ## 13.0.1 — 2026-09-01 ### Changed diff --git a/README.md b/README.md index 4b1f699..f2191c0 100644 --- a/README.md +++ b/README.md @@ -309,10 +309,18 @@ Reusing a run ID with a different implementation reference fails before cached w Different run IDs create separate candidate workspaces, so workers can explore in parallel. Promotion checks the original base hash and rejects a stale candidate instead of replacing newer work. Candidate retries use `evaluateDevelopment` when provided, otherwise they use deterministic validation, readiness, and KB quality checks. +Diagnosis runs before acquisition and updates, using development data only. +Its findings and update results remain available to final answer checks and the promotion decision. Development evaluation must use only train or selection data. The configured `evaluate` callback and final RAG phases run once, on the first candidate that passes those development checks. A failed final evaluation ends the run instead of selecting another candidate against final data. +The default evaluator reports only measured dimensions and averages those dimensions with equal weight. +It omits `answer_quality` without answer evaluation, `promotion_decision` without a promotion decision, and `blocking_readiness` without blocking readiness requirements. +Default evaluator version `2` records this weighting. +A candidate can pass structural checks without any task outcome evaluation; the metric notes state this limit. +`candidate-ready` means the configured checks passed and the candidate remains detached from the live knowledge base. + Candidate promotion currently requires Linux because it relies on Linux directory descriptors for exact file identity. ## Evaluate and improve RAG diff --git a/api-surface.json b/api-surface.json index 91cda25..82a86aa 100644 --- a/api-surface.json +++ b/api-surface.json @@ -410,7 +410,7 @@ "ResearchClaimRecord": "value febfce5cc05e", "ResearchClaimRecordSchema": "value 373728f5643d", "ResearchContribution": "value 4ef9b903af78", - "ResearchDriver": "value 39a48e748009", + "ResearchDriver": "value 76db7bd73161", "ResearchDrivingDriver": "value 919454a2ef43", "ResearchDrivingDriverOptions": "value bf4180687f2d", "ResearchDrivingState": "value 55bc4328e4c3", diff --git a/docs/architecture.md b/docs/architecture.md index 75e3dc1..ebdb2e3 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -58,6 +58,10 @@ After the exact submitted bytes are durable, `runVerifiedResearchLoop` passes th The ledger materializes only observations whose complete source identity matches a confirmed record, so reusing one URI for different bytes cannot activate the wrong claims and a crash on either side resumes safely. Unversioned URI-only ledgers cannot prove which bytes produced their observations; reads and writes fail with `ClaimLedgerMigrationRequiredError` and preserve the original file for an explicit archive-and-reverify migration. Before synchronous question generation, the persistent driver records `preparedRounds`; a resume reconstructs and checkpoints any prepared round whose questions were interrupted, and the loop publishes its `research.iteration` event only after that checkpoint succeeds. +The research loop requires storage readiness and the driver's optional `isComplete()` result before it reports completion. +An unfinished driver can generate steering with no remaining storage gaps, so passing source requirements does not stop research prematurely. +Drivers without `isComplete()` use storage readiness alone. +Without readiness specifications, the loop runs to its round limit and never reports ready. Every write in this layer goes through `durable-fs` (`writeFileDurable`, `writeJsonDurableWithinRoot`): temp file, fsync, atomic rename, and parent fsync. `O_NOFOLLOW` descriptors anchored through `/proc/self/fd` prevent a directory swapped for a symlink during a write from redirecting it outside the root. diff --git a/package.json b/package.json index cb82121..a2e2e39 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@tangle-network/agent-knowledge", - "version": "13.0.1", + "version": "14.0.0", "description": "Build, search, evaluate, and improve source-backed knowledge bases.", "homepage": "https://github.com/tangle-network/agent-knowledge#readme", "repository": { @@ -84,14 +84,14 @@ "zod": "4.5.4" }, "peerDependencies": { - "@tangle-network/agent-eval": ">=0.173.0 <0.174.0", + "@tangle-network/agent-eval": ">=0.174.0 <0.175.0", "@tangle-network/agent-interface": "^2.0.0" }, "devDependencies": { "@arethetypeswrong/cli": "^0.18.5", "@biomejs/biome": "^2.5.11", "@neo4j-labs/agent-memory": "0.4.1", - "@tangle-network/agent-eval": "0.173.0", + "@tangle-network/agent-eval": "0.174.0", "@tangle-network/agent-interface": "2.0.0", "@types/node": "^26.4.0", "mem0ai": "3.1.7", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 93a6834..089787e 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -35,8 +35,8 @@ importers: specifier: 0.4.1 version: 0.4.1 '@tangle-network/agent-eval': - specifier: 0.173.0 - version: 0.173.0 + specifier: 0.174.0 + version: 0.174.0 '@tangle-network/agent-interface': specifier: 2.0.0 version: 2.0.0 @@ -755,8 +755,8 @@ packages: '@modelcontextprotocol/sdk': optional: true - '@tangle-network/agent-eval@0.173.0': - resolution: {integrity: sha512-rLz26KS66ZhwCrPpsMUdvCiwcbyUCGGyS+SCwDc+6YJNSlCm+66ivry/yTO/WBS7j8yuI6hk2cpDBKEeHU6+uA==} + '@tangle-network/agent-eval@0.174.0': + resolution: {integrity: sha512-JcGQzN4fHll5ItkhEcFs7z8HF7ikwMLnlQhnSZekwJqGy6JbsmjGK87adxOV5SO12x1YRiALNzrQm1Hq6u1YUA==} engines: {node: '>=20.19.0'} hasBin: true @@ -2959,7 +2959,7 @@ snapshots: '@tangle-network/agent-interface': 2.0.0 zod: 4.5.4 - '@tangle-network/agent-eval@0.173.0': + '@tangle-network/agent-eval@0.174.0': dependencies: '@asteasolutions/zod-to-openapi': 9.1.0(zod@4.5.4) '@hono/node-server': 2.0.12(hono@4.12.32) diff --git a/src/kb-improvement/contracts.ts b/src/kb-improvement/contracts.ts index b0781d7..3cafb39 100644 --- a/src/kb-improvement/contracts.ts +++ b/src/kb-improvement/contracts.ts @@ -554,6 +554,7 @@ export interface LeaseHandle { export const DEFAULT_LEASE_TTL_MS = 15 * 60 * 1000 export const UPDATE_PHASES: readonly RagKnowledgeImprovementPhase[] = [ + 'gap-diagnosis', 'knowledge-acquisition', 'knowledge-update', ] @@ -561,7 +562,6 @@ export const UPDATE_PHASES: readonly RagKnowledgeImprovementPhase[] = [ export const EVALUATION_PHASES: readonly RagKnowledgeImprovementPhase[] = [ 'rag-optimization', 'retrieval-tuning', - 'gap-diagnosis', 'answer-quality', 'promotion', ] diff --git a/src/kb-improvement/evaluation.ts b/src/kb-improvement/evaluation.ts index c9ee0cb..3514ccc 100644 --- a/src/kb-improvement/evaluation.ts +++ b/src/kb-improvement/evaluation.ts @@ -15,6 +15,7 @@ import { type RunRagKnowledgeImprovementLoopResult, runRagKnowledgeImprovementLoop, } from '../rag-improvement-loop' +import { runRagKnowledgeImprovementPhases } from '../rag-improvement-phases' import { readinessFor } from '../readiness-helpers' import { mean } from '../statistics' import { type ValidateKnowledgeResult, validateKnowledgeIndex } from '../validate' @@ -49,6 +50,11 @@ import { export function assertKnowledgeImprovementOptions(options: KnowledgeImprovementOptions): void { assertImmutableRef(options.implementationRef, 'knowledge improvement implementationRef') + for (const phase of options.requiredPhases ?? []) { + if (options.enabledPhases && !options.enabledPhases.includes(phase)) { + throw new Error(`required phase ${phase} is not enabled`) + } + } if ( options.answerQualityCostCeiling !== undefined && (!Number.isFinite(options.answerQualityCostCeiling) || options.answerQualityCostCeiling < 0) @@ -120,19 +126,11 @@ export async function measureCandidate( } clearCandidateMeasurement(candidate) - const lifecycles: RunRagKnowledgeImprovementLoopResult[] = [] + let lifecycle: RunRagKnowledgeImprovementLoopResult | undefined if (candidate.status === 'running') { - const updateLifecycle = await runCandidateUpdateLifecycle( - runId, - candidate, - candidateRoot, - options, - now, - ) - if (updateLifecycle) lifecycles.push(updateLifecycle) + lifecycle = await runCandidateUpdateLifecycle(runId, candidate, candidateRoot, options, now) } return withFrozenCandidateWorkspace(runDir, candidate, candidateRoot, async (snapshot) => { - let lifecycle = mergeLifecycleResults(options.goal, lifecycles) const development = await evaluateCandidate( runDir, state, @@ -160,17 +158,16 @@ export async function measureCandidate( runId: state.runId, candidateId: candidate.candidateId, }) - const evaluationLifecycle = await runCandidateEvaluationLifecycle( + lifecycle = await runCandidateEvaluationLifecycle( runId, runDir, candidate, snapshot.root, snapshot.hash, + lifecycle, options, now, ) - if (evaluationLifecycle) lifecycles.push(evaluationLifecycle) - lifecycle = mergeLifecycleResults(options.goal, lifecycles) const measured = await evaluateCandidate( runDir, state, @@ -200,6 +197,7 @@ async function runCandidateUpdateLifecycle( if (!shouldRunUpdateStage(options)) return undefined const lifecycle = await runRagKnowledgeImprovementLoop({ goal: options.goal, + diagnose: options.diagnose, acquireKnowledge: options.acquireKnowledge, knowledgeResearch: candidateKnowledgeResearchOptions(candidateRoot, options), updateKnowledge: candidateUpdateHook(runId, candidate, candidateRoot, options), @@ -217,55 +215,58 @@ async function runCandidateEvaluationLifecycle( candidate: KnowledgeImprovementCandidateRecord, candidateRoot: string, candidateHash: string, + lifecycle: RunRagKnowledgeImprovementLoopResult | undefined, options: KnowledgeImprovementOptions, now: () => Date, ): Promise { if (!shouldRunEvaluationStage(options)) return undefined const candidateIndex = await buildKnowledgeIndex(candidateRoot) return withBaselineSnapshot(runDir, candidate.baseHash, (baselineRoot) => - runRagKnowledgeImprovementLoop({ - goal: options.goal, - optimization: options.ragOptimization - ? { - ...options.ragOptimization, - executionRef: candidateExecutionRef( - options.ragOptimization.executionRef, - candidateHash, - ), - runDir: - options.ragOptimization.runDir ?? - join(runDir, 'rag-optimization', candidate.candidateId), - run: (input) => - options.ragOptimization!.run({ - ...input, - runId, - iteration: candidate.iteration, - candidateId: candidate.candidateId, - root: candidateRoot, - baselineRoot, - candidateRoot, - candidateIndex, - baseHash: candidate.baseHash, - }), - } - : undefined, - retrieval: options.retrieval - ? { - ...options.retrieval, - executionRef: candidateExecutionRef(options.retrieval.executionRef, candidateHash), - index: candidateIndex, - runDir: options.retrieval.runDir ?? join(runDir, 'retrieval', candidate.candidateId), - } - : undefined, - diagnose: options.diagnose, - evaluateAnswers: options.evaluateAnswers, - answerQualityCostCeiling: options.answerQualityCostCeiling, - decidePromotion: options.decidePromotion, - enabledPhases: selectedStagePhases(options, EVALUATION_PHASES), - requiredPhases: selectedStageRequiredPhases(options, EVALUATION_PHASES), - signal: options.signal, - now, - }), + runRagKnowledgeImprovementPhases( + { + goal: options.goal, + optimization: options.ragOptimization + ? { + ...options.ragOptimization, + executionRef: candidateExecutionRef( + options.ragOptimization.executionRef, + candidateHash, + ), + runDir: + options.ragOptimization.runDir ?? + join(runDir, 'rag-optimization', candidate.candidateId), + run: (input) => + options.ragOptimization!.run({ + ...input, + runId, + iteration: candidate.iteration, + candidateId: candidate.candidateId, + root: candidateRoot, + baselineRoot, + candidateRoot, + candidateIndex, + baseHash: candidate.baseHash, + }), + } + : undefined, + retrieval: options.retrieval + ? { + ...options.retrieval, + executionRef: candidateExecutionRef(options.retrieval.executionRef, candidateHash), + index: candidateIndex, + runDir: options.retrieval.runDir ?? join(runDir, 'retrieval', candidate.candidateId), + } + : undefined, + evaluateAnswers: options.evaluateAnswers, + answerQualityCostCeiling: options.answerQualityCostCeiling, + decidePromotion: options.decidePromotion, + enabledPhases: selectedStagePhases(options, EVALUATION_PHASES), + requiredPhases: selectedStageRequiredPhases(options, EVALUATION_PHASES), + signal: options.signal, + now, + }, + lifecycle, + ), ) } @@ -317,10 +318,10 @@ function shouldRunUpdateStage(options: KnowledgeImprovementOptions): boolean { const phases = selectedStagePhases(options, UPDATE_PHASES) if (phases.length === 0) return false return Boolean( - options.acquireKnowledge || - options.step || - options.knowledgeResearch || - options.updateKnowledge || + (phases.includes('gap-diagnosis') && options.diagnose) || + (phases.includes('knowledge-acquisition') && options.acquireKnowledge) || + (phases.includes('knowledge-update') && + (options.step || options.knowledgeResearch || options.updateKnowledge)) || selectedStageRequiredPhases(options, UPDATE_PHASES).length > 0, ) } @@ -329,11 +330,10 @@ function shouldRunEvaluationStage(options: KnowledgeImprovementOptions): boolean const phases = selectedStagePhases(options, EVALUATION_PHASES) if (phases.length === 0) return false return Boolean( - options.ragOptimization || - options.retrieval || - options.diagnose || - options.evaluateAnswers || - options.decidePromotion || + (phases.includes('rag-optimization') && options.ragOptimization) || + (phases.includes('retrieval-tuning') && options.retrieval) || + (phases.includes('answer-quality') && options.evaluateAnswers) || + (phases.includes('promotion') && options.decidePromotion) || options.evaluate || selectedStageRequiredPhases(options, EVALUATION_PHASES).length > 0, ) @@ -354,31 +354,6 @@ function selectedStageRequiredPhases( return (options.requiredPhases ?? []).filter((phase) => stagePhases.includes(phase)) } -function mergeLifecycleResults( - goal: string, - lifecycles: readonly RunRagKnowledgeImprovementLoopResult[], -): RunRagKnowledgeImprovementLoopResult | undefined { - if (lifecycles.length === 0) return undefined - return { - goal, - phases: lifecycles.flatMap((lifecycle) => lifecycle.phases), - optimization: lastDefined(lifecycles.map((lifecycle) => lifecycle.optimization)), - retrieval: lastDefined(lifecycles.map((lifecycle) => lifecycle.retrieval)), - findings: lifecycles.flatMap((lifecycle) => lifecycle.findings), - acquisition: lastDefined(lifecycles.map((lifecycle) => lifecycle.acquisition)), - knowledgeUpdate: lastDefined(lifecycles.map((lifecycle) => lifecycle.knowledgeUpdate)), - answerQuality: lastDefined(lifecycles.map((lifecycle) => lifecycle.answerQuality)), - promotion: lastDefined(lifecycles.map((lifecycle) => lifecycle.promotion)), - } -} - -function lastDefined(values: readonly (T | undefined)[]): T | undefined { - for (let index = values.length - 1; index >= 0; index -= 1) { - if (values[index] !== undefined) return values[index] - } - return undefined -} - async function evaluateCandidate( runDir: string, state: KnowledgeImprovementRunState, @@ -495,18 +470,16 @@ function defaultKnowledgeImprovementMetric( ): KnowledgeImprovementMetric { const blockingMissing = readiness?.report.blockingMissingRequirements.length ?? 0 const blockingTotal = readinessSpecs?.filter((spec) => spec.importance === 'blocking').length ?? 0 - const blockingReadiness = - blockingTotal === 0 ? 1 : Math.max(0, blockingTotal - blockingMissing) / blockingTotal - const answerQuality = lifecycle?.answerQuality - ? mean(Object.values(lifecycle.answerQuality.metrics)) - : 1 - const promotionDecision = lifecycle?.promotion ? (lifecycle.promotion.promoted ? 1 : 0) : 1 - const dimensions = { + const dimensions: Record = { validation: validation.ok ? 1 : 0, kb_quality: kbQuality.ok ? 1 : 0, - blocking_readiness: blockingReadiness, - answer_quality: answerQuality, - promotion_decision: promotionDecision, + ...(blockingTotal > 0 + ? { blocking_readiness: Math.max(0, blockingTotal - blockingMissing) / blockingTotal } + : {}), + ...(lifecycle?.answerQuality + ? { answer_quality: mean(Object.values(lifecycle.answerQuality.metrics)) } + : {}), + ...(lifecycle?.promotion ? { promotion_decision: lifecycle.promotion.promoted ? 1 : 0 } : {}), } const failedReasons = [ validation.ok ? undefined : 'candidate validation failed', @@ -519,11 +492,17 @@ function defaultKnowledgeImprovementMetric( score: mean(Object.values(dimensions)), passed: failedReasons.length === 0, dimensions, - notes: + notes: [ failedReasons.length === 0 ? 'candidate passed configured checks' : failedReasons.join('; '), + !lifecycle?.answerQuality && !lifecycle?.optimization && !lifecycle?.retrieval + ? 'structural checks only; no task outcome evaluation was performed' + : undefined, + ] + .filter((note): note is string => note !== undefined) + .join('; '), provenance: { evaluator: '@tangle-network/agent-knowledge/default-knowledge-improvement-metric', - version: '1', + version: '2', method: 'deterministic', }, } diff --git a/src/kb-improvement/selected-candidate.ts b/src/kb-improvement/selected-candidate.ts index 79789ae..dd23233 100644 --- a/src/kb-improvement/selected-candidate.ts +++ b/src/kb-improvement/selected-candidate.ts @@ -118,9 +118,9 @@ export interface ImproveSelectedKnowledgeCandidateOptions rationale?: string /** JSON-safe policy output retained in the selection receipt. */ selectionMetadata?: Record - /** Evaluation-only phases. `knowledge-update` is inserted by this helper. */ + /** Diagnosis and evaluation phases. `knowledge-update` is inserted by this helper. */ enabledEvaluationPhases?: readonly KnowledgeEvaluationPhase[] - /** Evaluation-only phases that must complete. `knowledge-update` is always required. */ + /** Diagnosis and evaluation phases that must complete. `knowledge-update` is always required. */ requiredEvaluationPhases?: readonly KnowledgeEvaluationPhase[] } @@ -149,7 +149,7 @@ export async function improveSelectedKnowledgeCandidate( ) const requestedImplementationRef = immutableRefSchema.parse(options.implementationRef) const enabledEvaluationPhases = normalizeEvaluationPhases( - options.enabledEvaluationPhases ?? EVALUATION_PHASES, + options.enabledEvaluationPhases ?? ['gap-diagnosis', ...EVALUATION_PHASES], 'enabledEvaluationPhases', ) const requiredEvaluationPhases = normalizeEvaluationPhases( diff --git a/src/rag-improvement-loop.ts b/src/rag-improvement-loop.ts index 8852d41..748ce52 100644 --- a/src/rag-improvement-loop.ts +++ b/src/rag-improvement-loop.ts @@ -1,21 +1,15 @@ import type { ComparisonCost } from '@tangle-network/agent-eval/campaign' import type { AgentCandidateJsonValue as JsonValue } from '@tangle-network/agent-interface' -import { assertRagAnswerEvidence, ragAnswerEvidenceRejectionReasons } from './rag-answer-evidence' -import { - type RunRagOptimizationOptions, - type RunRagOptimizationResult, - runRagOptimization, -} from './rag-optimization' -import { - type KnowledgeResearchLoopDecision, - type KnowledgeResearchLoopResult, - type RunKnowledgeResearchLoopOptions, - runKnowledgeResearchLoop, +import { runRagKnowledgeImprovementPhases } from './rag-improvement-phases' +import type { RunRagOptimizationOptions, RunRagOptimizationResult } from './rag-optimization' +import type { + KnowledgeResearchLoopDecision, + KnowledgeResearchLoopResult, + RunKnowledgeResearchLoopOptions, } from './research-loop' -import { - type RunRetrievalImprovementLoopOptions, - type RunRetrievalImprovementLoopResult, - runRetrievalImprovementLoop, +import type { + RunRetrievalImprovementLoopOptions, + RunRetrievalImprovementLoopResult, } from './retrieval-optimization' export type RagKnowledgeImprovementPhase = @@ -184,476 +178,5 @@ type MaybePromise = T | Promise export async function runRagKnowledgeImprovementLoop( options: RunRagKnowledgeImprovementLoopOptions, ): Promise { - assertConfiguredRequiredPhases(options) - assertOptionalCostCeiling(options.answerQualityCostCeiling, 'answerQualityCostCeiling') - const now = options.now ?? (() => new Date()) - const phases: RagKnowledgeImprovementPhaseResult[] = [] - let optimization: RunRagOptimizationResult | undefined - let retrieval: RunRetrievalImprovementLoopResult | undefined - let findings: RagGapFinding[] = [] - let acquisition: KnowledgeResearchLoopDecision | undefined - let knowledgeUpdate: RagKnowledgeUpdateResult | undefined - let answerQuality: RagAnswerQualityResult | undefined - let promotion: RagPromotionResult | undefined - - if (phaseEnabled(options, 'gap-diagnosis')) { - if (options.diagnose) { - findings = [ - ...(await runPhase( - phases, - now, - 'gap-diagnosis', - async () => { - assertNotAborted(options.signal) - return options.diagnose!({ - goal: options.goal, - phases, - optimization: selectRagOptimization(optimization), - signal: options.signal, - retrieval: selectRetrievalOptimization(retrieval), - }) - }, - (diagnosed) => `${diagnosed.length} finding(s)`, - )), - ] - } else { - skipPhase(phases, now, 'gap-diagnosis', 'no diagnosis hook provided') - } - } - - if (phaseEnabled(options, 'knowledge-acquisition')) { - if (options.acquireKnowledge) { - acquisition = await runPhase( - phases, - now, - 'knowledge-acquisition', - async () => { - assertNotAborted(options.signal) - return options.acquireKnowledge!({ - goal: options.goal, - phases, - optimization: selectRagOptimization(optimization), - signal: options.signal, - retrieval: selectRetrievalOptimization(retrieval), - findings, - }) - }, - summarizeAcquisitionDecision, - ) - } else { - skipPhase(phases, now, 'knowledge-acquisition', 'no acquisition hook provided') - } - } - - if (phaseEnabled(options, 'knowledge-update')) { - if (options.updateKnowledge) { - knowledgeUpdate = await runPhase( - phases, - now, - 'knowledge-update', - async () => { - assertNotAborted(options.signal) - return options.updateKnowledge!({ - goal: options.goal, - phases, - optimization: selectRagOptimization(optimization), - signal: options.signal, - retrieval: selectRetrievalOptimization(retrieval), - findings, - acquisition, - }) - }, - (result) => result.summary, - ) - } else if (options.knowledgeResearch) { - knowledgeUpdate = await runPhase( - phases, - now, - 'knowledge-update', - async () => { - assertNotAborted(options.signal) - return runKnowledgeResearchUpdate(options, acquisition) - }, - (result) => result.summary, - ) - } else { - skipPhase(phases, now, 'knowledge-update', 'no update hook or research loop provided') - } - } - - if ( - phaseEnabled(options, 'rag-optimization') && - (options.optimization || - options.enabledPhases?.includes('rag-optimization') || - options.requiredPhases?.includes('rag-optimization')) - ) { - if (options.optimization) { - optimization = await runPhase( - phases, - now, - 'rag-optimization', - async () => { - assertNotAborted(options.signal) - return runRagOptimization(options.optimization!) - }, - summarizeRagOptimization, - ) - } else { - skipPhase(phases, now, 'rag-optimization', 'no full RAG optimization options provided') - } - } - - if (phaseEnabled(options, 'retrieval-tuning')) { - if (options.retrieval) { - retrieval = await runPhase( - phases, - now, - 'retrieval-tuning', - async () => { - assertNotAborted(options.signal) - return runRetrievalImprovementLoop(options.retrieval!) - }, - summarizeRetrievalResult, - ) - } else { - skipPhase(phases, now, 'retrieval-tuning', 'no retrieval options provided') - } - } - - if (phaseEnabled(options, 'answer-quality')) { - if (options.evaluateAnswers) { - answerQuality = await runPhase( - phases, - now, - 'answer-quality', - async () => { - assertNotAborted(options.signal) - const result = await options.evaluateAnswers!({ - goal: options.goal, - phases, - optimization: selectRagOptimization(optimization), - signal: options.signal, - retrieval: selectRetrievalOptimization(retrieval), - findings, - acquisition, - knowledgeUpdate, - }) - assertRagAnswerEvidence(result) - return result - }, - summarizeAnswerQuality, - ) - findings = [...findings, ...(answerQuality.findings ?? [])] - } else { - skipPhase(phases, now, 'answer-quality', 'no answer-quality hook provided') - } - } - - if (phaseEnabled(options, 'promotion')) { - if (options.decidePromotion) { - promotion = await runPhase( - phases, - now, - 'promotion', - async () => { - assertNotAborted(options.signal) - const evidenceRejection = rejectUnsafePromotionEvidence({ - optimization: optimization?.comparison, - optimizationCostCeiling: options.optimization?.costCeiling, - retrieval: retrieval?.comparison, - retrievalCostCeiling: options.retrieval?.costCeiling, - answerQuality, - answerQualityCostCeiling: options.answerQualityCostCeiling, - }) - if (evidenceRejection) return evidenceRejection - return options.decidePromotion!({ - goal: options.goal, - phases, - optimization: selectRagOptimization(optimization), - optimizationComparison: optimization?.comparison, - signal: options.signal, - retrieval: selectRetrievalOptimization(retrieval), - retrievalComparison: retrieval?.comparison, - findings, - acquisition, - knowledgeUpdate, - answerQuality, - }) - }, - (result) => `${result.promoted ? 'promoted' : 'held'}: ${result.reason}`, - ) - } else { - skipPhase(phases, now, 'promotion', 'no promotion decision hook provided') - } - } - - return { - goal: options.goal, - phases, - optimization, - retrieval, - findings, - acquisition, - knowledgeUpdate, - answerQuality, - promotion, - } -} - -function rejectUnsafePromotionEvidence(evidence: { - optimization?: RunRagOptimizationResult['comparison'] - optimizationCostCeiling?: number - retrieval?: RunRetrievalImprovementLoopResult['comparison'] - retrievalCostCeiling?: number - answerQuality?: RagAnswerQualityResult - answerQualityCostCeiling?: number -}): RagPromotionResult | undefined { - const reasons: string[] = [] - if (!evidence.optimization && !evidence.retrieval && !evidence.answerQuality) { - reasons.push('promotion requires final RAG, retrieval, or answer-quality evidence') - } - for (const [label, comparison, costCeiling] of [ - ['RAG', evidence.optimization, evidence.optimizationCostCeiling], - ['retrieval', evidence.retrieval, evidence.retrievalCostCeiling], - ] as const) { - if (!comparison) continue - if (!comparison.totalCost.accountingComplete) { - reasons.push(`${label} final comparison has incomplete cost accounting`) - } - const optimizerSource = comparison.best.provenance?.source - if (optimizerSource && optimizerSource.evidence !== 'observed') { - reasons.push(`${label} optimizer package identity was not observed`) - } - if (comparison.best.liftCi.low < 0) { - reasons.push(`${label} final comparison does not rule out a regression`) - } - if ( - costCeiling !== undefined && - exceedsCostCeiling(comparison.totalCost.totalCostUsd, costCeiling) - ) { - reasons.push( - `${label} final comparison cost ${comparison.totalCost.totalCostUsd} exceeds ${costCeiling}`, - ) - } - } - if (evidence.answerQuality) { - reasons.push( - ...ragAnswerEvidenceRejectionReasons( - evidence.answerQuality, - evidence.answerQualityCostCeiling, - ), - ) - } - if (reasons.length === 0) return undefined - return { - promoted: false, - reason: reasons.join('; '), - } -} - -function assertOptionalCostCeiling(value: number | undefined, label: string): void { - if (value !== undefined && (!Number.isFinite(value) || value < 0)) { - throw new Error(`${label} must be a non-negative finite number`) - } -} - -function exceedsCostCeiling(totalCostUsd: number, costCeiling: number): boolean { - const tolerance = Number.EPSILON * Math.max(1, Math.abs(totalCostUsd), Math.abs(costCeiling)) * 8 - return totalCostUsd - costCeiling > tolerance -} - -function summarizeRagOptimization(result: RunRagOptimizationResult): string { - return `${result.methodName}; winner=${result.winner.surfaceHash}` -} - -async function runKnowledgeResearchUpdate( - options: RunRagKnowledgeImprovementLoopOptions, - acquisition: KnowledgeResearchLoopDecision | undefined, -): Promise { - const research = options.knowledgeResearch - if (!research) { - throw new Error('knowledgeResearch options are required to run the knowledge update phase') - } - const { goal, step, ...rest } = research - const researchStep = step ?? acquisitionBackedResearchStep(acquisition) - const maxIterations = step ? rest.maxIterations : 1 - const result = await runKnowledgeResearchLoop({ - ...rest, - goal: goal ?? options.goal, - maxIterations, - signal: options.signal, - step: researchStep, - }) - return { - applied: result.steps.some((stepResult) => { - return stepResult.addedSources.length > 0 || Boolean(stepResult.applied) - }), - summary: `${result.iterations} research iteration(s); done=${String(result.done)}`, - research: result, - } -} - -function acquisitionBackedResearchStep( - acquisition: KnowledgeResearchLoopDecision | undefined, -): RunKnowledgeResearchLoopOptions['step'] { - if (!acquisition) { - throw new Error( - 'knowledgeResearch requires either a step hook or a knowledge-acquisition result to apply', - ) - } - return () => ({ ...acquisition, done: acquisition.done ?? true }) -} - -async function runPhase( - phases: RagKnowledgeImprovementPhaseResult[], - now: () => Date, - phase: RagKnowledgeImprovementPhase, - action: () => MaybePromise, - summarize: (result: T) => string, -): Promise { - const startedAt = now().toISOString() - try { - const result = await action() - phases.push({ - phase, - status: 'completed', - summary: summarize(result), - startedAt, - finishedAt: now().toISOString(), - }) - return result - } catch (error) { - phases.push({ - phase, - status: 'failed', - summary: (error as Error).message, - startedAt, - finishedAt: now().toISOString(), - }) - throw error - } -} - -function skipPhase( - phases: RagKnowledgeImprovementPhaseResult[], - now: () => Date, - phase: RagKnowledgeImprovementPhase, - summary: string, -): void { - const timestamp = now().toISOString() - phases.push({ phase, status: 'skipped', summary, startedAt: timestamp, finishedAt: timestamp }) -} - -function summarizeRetrievalResult(result: RunRetrievalImprovementLoopResult): string { - return `${result.methodName}; winner=${result.winner.surfaceHash}` -} - -function selectRagOptimization( - result: RunRagOptimizationResult | undefined, -): RagOptimizationSelection | undefined { - if (!result) return undefined - return { - methodName: result.methodName, - baseline: structuredClone(result.baseline), - winner: structuredClone(result.winner), - baselineConfig: structuredClone(result.baselineConfig), - winnerConfig: structuredClone(result.winnerConfig), - } -} - -function selectRetrievalOptimization( - result: RunRetrievalImprovementLoopResult | undefined, -): RetrievalOptimizationSelection | undefined { - if (!result) return undefined - return { - methodName: result.methodName, - baseline: structuredClone(result.baseline), - winner: structuredClone(result.winner), - baselineConfig: structuredClone(result.baselineConfig), - winnerConfig: structuredClone(result.winnerConfig), - } -} - -function summarizeAcquisitionDecision(decision: KnowledgeResearchLoopDecision): string { - const sourcePathCount = decision.sourcePaths?.length ?? 0 - const sourceTextCount = decision.sourceTexts?.length ?? 0 - const proposal = decision.proposalText ? 'proposal' : 'no proposal' - return `${sourcePathCount} path source(s), ${sourceTextCount} text source(s), ${proposal}` -} - -function summarizeAnswerQuality(result: RagAnswerQualityResult): string { - const metrics = Object.entries(result.metrics) - .sort(([a], [b]) => a.localeCompare(b)) - .map(([key, value]) => `${key}=${formatMetric(value)}`) - .join(', ') - return `${result.passed ? 'passed' : 'failed'}${metrics ? `; ${metrics}` : ''}` -} - -function formatMetric(value: number): string { - return Number.isFinite(value) ? value.toFixed(3) : String(value) -} - -function phaseEnabled( - options: RunRagKnowledgeImprovementLoopOptions, - phase: RagKnowledgeImprovementPhase, -): boolean { - return !options.enabledPhases || options.enabledPhases.includes(phase) -} - -function assertConfiguredRequiredPhases(options: RunRagKnowledgeImprovementLoopOptions): void { - for (const phase of options.requiredPhases ?? []) { - if (!phaseEnabled(options, phase)) { - throw new Error(`required phase ${phase} is not enabled`) - } - if (!phaseConfigured(options, phase)) { - throw new Error(requiredPhaseMessage(phase)) - } - } -} - -function phaseConfigured( - options: RunRagKnowledgeImprovementLoopOptions, - phase: RagKnowledgeImprovementPhase, -): boolean { - switch (phase) { - case 'rag-optimization': - return Boolean(options.optimization) - case 'retrieval-tuning': - return Boolean(options.retrieval) - case 'gap-diagnosis': - return Boolean(options.diagnose) - case 'knowledge-acquisition': - return Boolean(options.acquireKnowledge) - case 'knowledge-update': - return Boolean(options.updateKnowledge ?? options.knowledgeResearch) - case 'answer-quality': - return Boolean(options.evaluateAnswers) - case 'promotion': - return Boolean(options.decidePromotion) - } -} - -function requiredPhaseMessage(phase: RagKnowledgeImprovementPhase): string { - switch (phase) { - case 'rag-optimization': - return 'required phase rag-optimization requires optimization options' - case 'retrieval-tuning': - return 'required phase retrieval-tuning requires retrieval options' - case 'gap-diagnosis': - return 'required phase gap-diagnosis requires a diagnose hook' - case 'knowledge-acquisition': - return 'required phase knowledge-acquisition requires an acquireKnowledge hook' - case 'knowledge-update': - return 'required phase knowledge-update requires updateKnowledge or knowledgeResearch' - case 'answer-quality': - return 'required phase answer-quality requires an evaluateAnswers hook' - case 'promotion': - return 'required phase promotion requires a decidePromotion hook' - } -} - -function assertNotAborted(signal: AbortSignal | undefined): void { - if (signal?.aborted) { - throw new Error('RAG knowledge improvement loop aborted') - } + return runRagKnowledgeImprovementPhases(options) } diff --git a/src/rag-improvement-phases.ts b/src/rag-improvement-phases.ts new file mode 100644 index 0000000..91ac3c9 --- /dev/null +++ b/src/rag-improvement-phases.ts @@ -0,0 +1,502 @@ +import { assertRagAnswerEvidence, ragAnswerEvidenceRejectionReasons } from './rag-answer-evidence' +import type { + RagAnswerQualityResult, + RagKnowledgeImprovementPhase, + RagKnowledgeImprovementPhaseResult, + RagKnowledgeUpdateResult, + RagOptimizationSelection, + RagPromotionResult, + RetrievalOptimizationSelection, + RunRagKnowledgeImprovementLoopOptions, + RunRagKnowledgeImprovementLoopResult, +} from './rag-improvement-loop' +import { type RunRagOptimizationResult, runRagOptimization } from './rag-optimization' +import { + type KnowledgeResearchLoopDecision, + type RunKnowledgeResearchLoopOptions, + runKnowledgeResearchLoop, +} from './research-loop' +import { + type RunRetrievalImprovementLoopResult, + runRetrievalImprovementLoop, +} from './retrieval-optimization' + +type MaybePromise = T | Promise + +export async function runRagKnowledgeImprovementPhases( + options: RunRagKnowledgeImprovementLoopOptions, + previous?: RunRagKnowledgeImprovementLoopResult, +): Promise { + assertConfiguredRequiredPhases(options) + assertOptionalCostCeiling(options.answerQualityCostCeiling, 'answerQualityCostCeiling') + const now = options.now ?? (() => new Date()) + const phases: RagKnowledgeImprovementPhaseResult[] = [...(previous?.phases ?? [])] + let optimization = previous?.optimization + let retrieval = previous?.retrieval + let findings = [...(previous?.findings ?? [])] + let acquisition = previous?.acquisition + let knowledgeUpdate = previous?.knowledgeUpdate + let answerQuality = previous?.answerQuality + let promotion = previous?.promotion + + if (phaseEnabled(options, 'gap-diagnosis')) { + if (options.diagnose) { + findings = [ + ...(await runPhase( + phases, + now, + 'gap-diagnosis', + async () => { + assertNotAborted(options.signal) + return options.diagnose!({ + goal: options.goal, + phases, + optimization: selectRagOptimization(optimization), + signal: options.signal, + retrieval: selectRetrievalOptimization(retrieval), + }) + }, + (diagnosed) => `${diagnosed.length} finding(s)`, + )), + ] + } else { + skipPhase(phases, now, 'gap-diagnosis', 'no diagnosis hook provided') + } + } + + if (phaseEnabled(options, 'knowledge-acquisition')) { + if (options.acquireKnowledge) { + acquisition = await runPhase( + phases, + now, + 'knowledge-acquisition', + async () => { + assertNotAborted(options.signal) + return options.acquireKnowledge!({ + goal: options.goal, + phases, + optimization: selectRagOptimization(optimization), + signal: options.signal, + retrieval: selectRetrievalOptimization(retrieval), + findings, + }) + }, + summarizeAcquisitionDecision, + ) + } else { + skipPhase(phases, now, 'knowledge-acquisition', 'no acquisition hook provided') + } + } + + if (phaseEnabled(options, 'knowledge-update')) { + if (options.updateKnowledge) { + knowledgeUpdate = await runPhase( + phases, + now, + 'knowledge-update', + async () => { + assertNotAborted(options.signal) + return options.updateKnowledge!({ + goal: options.goal, + phases, + optimization: selectRagOptimization(optimization), + signal: options.signal, + retrieval: selectRetrievalOptimization(retrieval), + findings, + acquisition, + }) + }, + (result) => result.summary, + ) + } else if (options.knowledgeResearch) { + knowledgeUpdate = await runPhase( + phases, + now, + 'knowledge-update', + async () => { + assertNotAborted(options.signal) + return runKnowledgeResearchUpdate(options, acquisition) + }, + (result) => result.summary, + ) + } else { + skipPhase(phases, now, 'knowledge-update', 'no update hook or research loop provided') + } + } + + if ( + phaseEnabled(options, 'rag-optimization') && + (options.optimization || + options.enabledPhases?.includes('rag-optimization') || + options.requiredPhases?.includes('rag-optimization')) + ) { + if (options.optimization) { + optimization = await runPhase( + phases, + now, + 'rag-optimization', + async () => { + assertNotAborted(options.signal) + return runRagOptimization(options.optimization!) + }, + summarizeRagOptimization, + ) + } else { + skipPhase(phases, now, 'rag-optimization', 'no full RAG optimization options provided') + } + } + + if (phaseEnabled(options, 'retrieval-tuning')) { + if (options.retrieval) { + retrieval = await runPhase( + phases, + now, + 'retrieval-tuning', + async () => { + assertNotAborted(options.signal) + return runRetrievalImprovementLoop(options.retrieval!) + }, + summarizeRetrievalResult, + ) + } else { + skipPhase(phases, now, 'retrieval-tuning', 'no retrieval options provided') + } + } + + if (phaseEnabled(options, 'answer-quality')) { + if (options.evaluateAnswers) { + answerQuality = await runPhase( + phases, + now, + 'answer-quality', + async () => { + assertNotAborted(options.signal) + const result = await options.evaluateAnswers!({ + goal: options.goal, + phases, + optimization: selectRagOptimization(optimization), + signal: options.signal, + retrieval: selectRetrievalOptimization(retrieval), + findings, + acquisition, + knowledgeUpdate, + }) + assertRagAnswerEvidence(result) + return result + }, + summarizeAnswerQuality, + ) + findings = [...findings, ...(answerQuality.findings ?? [])] + } else { + skipPhase(phases, now, 'answer-quality', 'no answer-quality hook provided') + } + } + + if (phaseEnabled(options, 'promotion')) { + if (options.decidePromotion) { + promotion = await runPhase( + phases, + now, + 'promotion', + async () => { + assertNotAborted(options.signal) + const evidenceRejection = rejectUnsafePromotionEvidence({ + optimization: optimization?.comparison, + optimizationCostCeiling: options.optimization?.costCeiling, + retrieval: retrieval?.comparison, + retrievalCostCeiling: options.retrieval?.costCeiling, + answerQuality, + answerQualityCostCeiling: options.answerQualityCostCeiling, + }) + if (evidenceRejection) return evidenceRejection + return options.decidePromotion!({ + goal: options.goal, + phases, + optimization: selectRagOptimization(optimization), + optimizationComparison: optimization?.comparison, + signal: options.signal, + retrieval: selectRetrievalOptimization(retrieval), + retrievalComparison: retrieval?.comparison, + findings, + acquisition, + knowledgeUpdate, + answerQuality, + }) + }, + (result) => `${result.promoted ? 'promoted' : 'held'}: ${result.reason}`, + ) + } else { + skipPhase(phases, now, 'promotion', 'no promotion decision hook provided') + } + } + + return { + goal: options.goal, + phases, + optimization, + retrieval, + findings, + acquisition, + knowledgeUpdate, + answerQuality, + promotion, + } +} + +function rejectUnsafePromotionEvidence(evidence: { + optimization?: RunRagOptimizationResult['comparison'] + optimizationCostCeiling?: number + retrieval?: RunRetrievalImprovementLoopResult['comparison'] + retrievalCostCeiling?: number + answerQuality?: RagAnswerQualityResult + answerQualityCostCeiling?: number +}): RagPromotionResult | undefined { + const reasons: string[] = [] + if (!evidence.optimization && !evidence.retrieval && !evidence.answerQuality) { + reasons.push('promotion requires final RAG, retrieval, or answer-quality evidence') + } + for (const [label, comparison, costCeiling] of [ + ['RAG', evidence.optimization, evidence.optimizationCostCeiling], + ['retrieval', evidence.retrieval, evidence.retrievalCostCeiling], + ] as const) { + if (!comparison) continue + if (!comparison.totalCost.accountingComplete) { + reasons.push(`${label} final comparison has incomplete cost accounting`) + } + const optimizerSource = comparison.best.provenance?.source + if (optimizerSource && optimizerSource.evidence !== 'observed') { + reasons.push(`${label} optimizer package identity was not observed`) + } + if (comparison.best.liftCi.low < 0) { + reasons.push(`${label} final comparison does not rule out a regression`) + } + if ( + costCeiling !== undefined && + exceedsCostCeiling(comparison.totalCost.totalCostUsd, costCeiling) + ) { + reasons.push( + `${label} final comparison cost ${comparison.totalCost.totalCostUsd} exceeds ${costCeiling}`, + ) + } + } + if (evidence.answerQuality) { + reasons.push( + ...ragAnswerEvidenceRejectionReasons( + evidence.answerQuality, + evidence.answerQualityCostCeiling, + ), + ) + } + if (reasons.length === 0) return undefined + return { + promoted: false, + reason: reasons.join('; '), + } +} + +function assertOptionalCostCeiling(value: number | undefined, label: string): void { + if (value !== undefined && (!Number.isFinite(value) || value < 0)) { + throw new Error(`${label} must be a non-negative finite number`) + } +} + +function exceedsCostCeiling(totalCostUsd: number, costCeiling: number): boolean { + const tolerance = Number.EPSILON * Math.max(1, Math.abs(totalCostUsd), Math.abs(costCeiling)) * 8 + return totalCostUsd - costCeiling > tolerance +} + +function summarizeRagOptimization(result: RunRagOptimizationResult): string { + return `${result.methodName}; winner=${result.winner.surfaceHash}` +} + +async function runKnowledgeResearchUpdate( + options: RunRagKnowledgeImprovementLoopOptions, + acquisition: KnowledgeResearchLoopDecision | undefined, +): Promise { + const research = options.knowledgeResearch + if (!research) { + throw new Error('knowledgeResearch options are required to run the knowledge update phase') + } + const { goal, step, ...rest } = research + const researchStep = step ?? acquisitionBackedResearchStep(acquisition) + const maxIterations = step ? rest.maxIterations : 1 + const result = await runKnowledgeResearchLoop({ + ...rest, + goal: goal ?? options.goal, + maxIterations, + signal: options.signal, + step: researchStep, + }) + return { + applied: result.steps.some((stepResult) => { + return stepResult.addedSources.length > 0 || Boolean(stepResult.applied) + }), + summary: `${result.iterations} research iteration(s); done=${String(result.done)}`, + research: result, + } +} + +function acquisitionBackedResearchStep( + acquisition: KnowledgeResearchLoopDecision | undefined, +): RunKnowledgeResearchLoopOptions['step'] { + if (!acquisition) { + throw new Error( + 'knowledgeResearch requires either a step hook or a knowledge-acquisition result to apply', + ) + } + return () => ({ ...acquisition, done: acquisition.done ?? true }) +} + +async function runPhase( + phases: RagKnowledgeImprovementPhaseResult[], + now: () => Date, + phase: RagKnowledgeImprovementPhase, + action: () => MaybePromise, + summarize: (result: T) => string, +): Promise { + const startedAt = now().toISOString() + try { + const result = await action() + phases.push({ + phase, + status: 'completed', + summary: summarize(result), + startedAt, + finishedAt: now().toISOString(), + }) + return result + } catch (error) { + phases.push({ + phase, + status: 'failed', + summary: (error as Error).message, + startedAt, + finishedAt: now().toISOString(), + }) + throw error + } +} + +function skipPhase( + phases: RagKnowledgeImprovementPhaseResult[], + now: () => Date, + phase: RagKnowledgeImprovementPhase, + summary: string, +): void { + const timestamp = now().toISOString() + phases.push({ phase, status: 'skipped', summary, startedAt: timestamp, finishedAt: timestamp }) +} + +function summarizeRetrievalResult(result: RunRetrievalImprovementLoopResult): string { + return `${result.methodName}; winner=${result.winner.surfaceHash}` +} + +function selectRagOptimization( + result: RunRagOptimizationResult | undefined, +): RagOptimizationSelection | undefined { + if (!result) return undefined + return { + methodName: result.methodName, + baseline: structuredClone(result.baseline), + winner: structuredClone(result.winner), + baselineConfig: structuredClone(result.baselineConfig), + winnerConfig: structuredClone(result.winnerConfig), + } +} + +function selectRetrievalOptimization( + result: RunRetrievalImprovementLoopResult | undefined, +): RetrievalOptimizationSelection | undefined { + if (!result) return undefined + return { + methodName: result.methodName, + baseline: structuredClone(result.baseline), + winner: structuredClone(result.winner), + baselineConfig: structuredClone(result.baselineConfig), + winnerConfig: structuredClone(result.winnerConfig), + } +} + +function summarizeAcquisitionDecision(decision: KnowledgeResearchLoopDecision): string { + const sourcePathCount = decision.sourcePaths?.length ?? 0 + const sourceTextCount = decision.sourceTexts?.length ?? 0 + const proposal = decision.proposalText ? 'proposal' : 'no proposal' + return `${sourcePathCount} path source(s), ${sourceTextCount} text source(s), ${proposal}` +} + +function summarizeAnswerQuality(result: RagAnswerQualityResult): string { + const metrics = Object.entries(result.metrics) + .sort(([a], [b]) => a.localeCompare(b)) + .map(([key, value]) => `${key}=${formatMetric(value)}`) + .join(', ') + return `${result.passed ? 'passed' : 'failed'}${metrics ? `; ${metrics}` : ''}` +} + +function formatMetric(value: number): string { + return Number.isFinite(value) ? value.toFixed(3) : String(value) +} + +function phaseEnabled( + options: RunRagKnowledgeImprovementLoopOptions, + phase: RagKnowledgeImprovementPhase, +): boolean { + return !options.enabledPhases || options.enabledPhases.includes(phase) +} + +function assertConfiguredRequiredPhases(options: RunRagKnowledgeImprovementLoopOptions): void { + for (const phase of options.requiredPhases ?? []) { + if (!phaseEnabled(options, phase)) { + throw new Error(`required phase ${phase} is not enabled`) + } + if (!phaseConfigured(options, phase)) { + throw new Error(requiredPhaseMessage(phase)) + } + } +} + +function phaseConfigured( + options: RunRagKnowledgeImprovementLoopOptions, + phase: RagKnowledgeImprovementPhase, +): boolean { + switch (phase) { + case 'rag-optimization': + return Boolean(options.optimization) + case 'retrieval-tuning': + return Boolean(options.retrieval) + case 'gap-diagnosis': + return Boolean(options.diagnose) + case 'knowledge-acquisition': + return Boolean(options.acquireKnowledge) + case 'knowledge-update': + return Boolean(options.updateKnowledge ?? options.knowledgeResearch) + case 'answer-quality': + return Boolean(options.evaluateAnswers) + case 'promotion': + return Boolean(options.decidePromotion) + } +} + +function requiredPhaseMessage(phase: RagKnowledgeImprovementPhase): string { + switch (phase) { + case 'rag-optimization': + return 'required phase rag-optimization requires optimization options' + case 'retrieval-tuning': + return 'required phase retrieval-tuning requires retrieval options' + case 'gap-diagnosis': + return 'required phase gap-diagnosis requires a diagnose hook' + case 'knowledge-acquisition': + return 'required phase knowledge-acquisition requires an acquireKnowledge hook' + case 'knowledge-update': + return 'required phase knowledge-update requires updateKnowledge or knowledgeResearch' + case 'answer-quality': + return 'required phase answer-quality requires an evaluateAnswers hook' + case 'promotion': + return 'required phase promotion requires a decidePromotion hook' + } +} + +function assertNotAborted(signal: AbortSignal | undefined): void { + if (signal?.aborted) { + throw new Error('RAG knowledge improvement loop aborted') + } +} diff --git a/src/verified-research-loop.ts b/src/verified-research-loop.ts index 018ffb8..691ff99 100644 --- a/src/verified-research-loop.ts +++ b/src/verified-research-loop.ts @@ -123,6 +123,7 @@ export interface DriverResearchContext { * open. Only invoked when `driverResearches` is true. * - `foldGaps` — turn the remaining gaps into a steer string for the worker's * next prompt. Defaults to a compact bulleted list when omitted. + * - `isComplete` — require the driver's research objectives as well as storage readiness. * - `checkpoint` — write whatever state the driver accumulated to durable * storage. Called at the end of every round, after `foldGaps`, so state that * a synchronous hook produced is on disk before the next round can crash. @@ -138,6 +139,7 @@ export interface ResearchDriver { ): Promise | SourceVerdict research?(ctx: DriverResearchContext): Promise | ResearchContribution foldGaps?(gaps: KnowledgeGap[]): string + isComplete?(): boolean prepareFold?(): Promise | void commitSources?(sources: readonly SourceRecord[]): Promise | void checkpoint?(): Promise | void @@ -186,7 +188,7 @@ export interface VerifiedResearchRound { /** Curated pages written this round (worker proposal + driver proposal). */ writtenPages: string[] readiness?: EvalKnowledgeBundleBuildResult - /** True once the readiness gate reports no blocking gaps. */ + /** True once readiness passes and the driver reports completion, when configured. */ ready: boolean event: KnowledgeEvent notes: { worker?: string; driver?: string } @@ -214,8 +216,8 @@ export interface VerifiedResearchLoopResult { * rejects ones that aren't real/relevant; (2) GAP-FILLS the gaps the worker * missed with its own research pass (when `driverResearches`); (3) folds the * remaining gaps into the worker's next prompt; and (4) GATES on - * `scoreKnowledgeReadiness` — the loop stops as soon as there are no blocking - * gaps. + * `scoreKnowledgeReadiness` and optional `isComplete` — the loop stops when + * readiness passes and the driver has no unfinished research. * * Set `driverResearches: false` (default) for the pure-coordinator mode: the * driver only verifies + gates and contributes no research itself. @@ -236,7 +238,7 @@ export async function runVerifiedResearchLoop( // confirming them to the driver. The records carry both original URI and hash. await confirmRegisteredSources(options.driver, index.sources) let readiness = readinessFor(options, index) - let ready = isReady(readiness?.report) + let ready = isReady(readiness?.report) && (options.driver.isComplete?.() ?? true) let steer: string | undefined for (let round = 1; round <= maxRounds && !ready; round++) { @@ -324,9 +326,10 @@ export async function runVerifiedResearchLoop( } // 4. DRIVER GATES on readiness and folds the remainder into the next prompt. - ready = isReady(readiness?.report) + const driverComplete = options.driver.isComplete?.() ?? true + ready = isReady(readiness?.report) && driverComplete const remainingGaps = gapsFromReadiness(readiness) - if (ready || remainingGaps.length === 0) { + if (ready || (remainingGaps.length === 0 && driverComplete)) { steer = undefined } else { await options.driver.prepareFold?.() @@ -409,8 +412,8 @@ function gapFor( } function foldGaps(driver: ResearchDriver, gaps: KnowledgeGap[]): string | undefined { - if (gaps.length === 0) return undefined if (driver.foldGaps) return driver.foldGaps(gaps) + if (gaps.length === 0) return undefined return [ 'The knowledge base is still missing the following. Prioritise these next round:', ...gaps.map( diff --git a/tests/kb-improvement/lifecycle.test.ts b/tests/kb-improvement/lifecycle.test.ts new file mode 100644 index 0000000..bd5e613 --- /dev/null +++ b/tests/kb-improvement/lifecycle.test.ts @@ -0,0 +1,190 @@ +import { writeFile } from 'node:fs/promises' +import { join } from 'node:path' +import { describe, expect, it } from 'vitest' +import type { + RagGapFinding, + RagKnowledgeAcquisitionInput, + RagKnowledgeUpdateInput, +} from '../../src/index' +import { + improveTestKnowledgeBase as improveKnowledgeBase, + passingMetric, + withKb, +} from '../support/kb-improvement' +import { testExecutionRef } from '../support/optimization' + +describe('knowledge candidate learning state', () => { + it.each(['gap-diagnosis', 'answer-quality'] as const)( + 'rejects disabled required phase %s before candidate work', + async (phase) => { + await withKb(async (root) => { + let updates = 0 + await expect( + improveKnowledgeBase({ + root, + runId: 'disabled-required-phase', + goal: 'Require every configured phase', + enabledPhases: [], + requiredPhases: [phase], + updateKnowledge: () => { + updates += 1 + return { applied: false, summary: 'Must not run.' } + }, + }), + ).rejects.toThrow(`required phase ${phase} is not enabled`) + expect(updates).toBe(0) + }) + }, + ) + + it('carries diagnosis and update results through development into final measurement', async () => { + await withKb(async (root) => { + const calls: string[] = [] + const findings: RagGapFinding[] = [ + { + id: 'policy-gap', + kind: 'missing-source', + severity: 'warning', + message: 'Missing policy.', + }, + ] + const finalFinding: RagGapFinding = { + id: 'answer-observation', + kind: 'unknown', + severity: 'info', + message: 'Final answer observed.', + } + const acquisition = { notes: 'Read the policy source.', done: true } + const update = { applied: true, summary: 'Updated the policy page.' } + let acquisitionInput: RagKnowledgeAcquisitionInput | undefined + let updateInput: RagKnowledgeUpdateInput | undefined + const result = await improveKnowledgeBase({ + root, + runId: 'shared-learning-state', + goal: 'Improve answers using diagnosed policy gaps', + answerQualityCostCeiling: 0, + enabledPhases: [ + 'gap-diagnosis', + 'knowledge-acquisition', + 'knowledge-update', + 'answer-quality', + 'promotion', + ], + diagnose(input) { + calls.push('diagnose') + expect(input.optimization).toBeUndefined() + expect(input.retrieval).toBeUndefined() + return findings + }, + acquireKnowledge(input) { + calls.push('acquire') + acquisitionInput = input + expect(input.findings).toEqual(findings) + return acquisition + }, + async updateKnowledge(input) { + calls.push('update') + updateInput = input + expect(input.findings).toEqual(findings) + expect(input.acquisition).toEqual(acquisition) + await writeFile(join(input.candidateRoot, 'knowledge', 'policy.md'), '# Policy\n') + return update + }, + evaluateDevelopment({ lifecycle }) { + calls.push('development') + expect(lifecycle?.findings).toEqual(findings) + expect(lifecycle?.knowledgeUpdate).toEqual(update) + expect(lifecycle?.answerQuality).toBeUndefined() + return passingMetric() + }, + evaluateAnswers(input) { + calls.push('answers') + expect(input.findings).toEqual(findings) + expect(input.acquisition).toEqual(acquisition) + expect(input.knowledgeUpdate).toEqual(update) + expect(input.phases.map((phase) => phase.phase)).toEqual([ + 'gap-diagnosis', + 'knowledge-acquisition', + 'knowledge-update', + ]) + return { + passed: true, + metrics: { quality: 0.4 }, + finalScenarioIds: ['policy-final-a', 'policy-final-b'], + datasetRef: testExecutionRef('shared-state-final-data'), + evaluatorRef: testExecutionRef('shared-state-final-evaluator'), + cost: { totalCostUsd: 0, accountingComplete: true, incompleteReasons: [] }, + findings: [finalFinding], + } + }, + decidePromotion(input) { + calls.push('promotion') + expect(input.findings).toEqual([...findings, finalFinding]) + expect(input.acquisition).toEqual(acquisition) + expect(input.knowledgeUpdate).toEqual(update) + expect(input.answerQuality?.metrics).toEqual({ quality: 0.4 }) + return { promoted: true, reason: 'Configured answer checks passed.' } + }, + }) + + expect(calls).toEqual([ + 'diagnose', + 'acquire', + 'update', + 'development', + 'answers', + 'promotion', + ]) + expect(acquisitionInput?.findings).toEqual(findings) + expect(updateInput?.findings).toEqual(findings) + expect(updateInput?.phases.some((phase) => phase.phase === 'answer-quality')).toBe(false) + expect(result.lifecycle?.findings).toEqual([...findings, finalFinding]) + expect(result.evaluation?.dimensions).toEqual({ + validation: 1, + kb_quality: 1, + answer_quality: 0.4, + promotion_decision: 1, + }) + expect(result.evaluation?.score).toBeCloseTo(0.85) + expect(result.evaluation?.notes).not.toContain('no task outcome evaluation') + expect(result.promoted).toBe(false) + expect(result.state.status).toBe('candidate-ready') + }) + }) + + it('runs diagnosis without consuming final cases or reporting unmeasured checks', async () => { + await withKb(async (root) => { + let diagnoses = 0 + const result = await improveKnowledgeBase({ + root, + runId: 'diagnosis-only', + goal: 'Inspect existing knowledge', + enabledPhases: ['gap-diagnosis'], + requiredPhases: ['gap-diagnosis'], + diagnose: () => { + diagnoses += 1 + return [] + }, + evaluateAnswers: () => { + throw new Error('disabled answer evaluation must not run') + }, + }) + + expect(diagnoses).toBe(1) + expect(result.lifecycle?.phases.map((phase) => phase.phase)).toEqual(['gap-diagnosis']) + expect(result.candidate?.finalEvaluationStartedAt).toBeUndefined() + expect(result.candidate?.candidateHash).toBe(result.candidate?.baseHash) + expect(result.evaluation).toMatchObject({ + score: 1, + passed: true, + provenance: { version: '2', method: 'deterministic' }, + }) + expect(result.evaluation?.dimensions).toEqual({ validation: 1, kb_quality: 1 }) + expect(result.evaluation?.notes).toContain( + 'structural checks only; no task outcome evaluation was performed', + ) + expect(result.promoted).toBe(false) + expect(result.state.status).toBe('candidate-ready') + }) + }) +}) diff --git a/tests/kb-improvement/selected-candidate.test.ts b/tests/kb-improvement/selected-candidate.test.ts index f32be67..40fd62a 100644 --- a/tests/kb-improvement/selected-candidate.test.ts +++ b/tests/kb-improvement/selected-candidate.test.ts @@ -35,6 +35,7 @@ describe('improveSelectedKnowledgeCandidate', () => { }) const sourceCandidate = knowledgeImprovementCandidateRef(source) let evaluatedSelectedSnapshot = false + let diagnosisCalls = 0 const selected = await improveSelectedKnowledgeCandidate({ root, @@ -44,7 +45,20 @@ describe('improveSelectedKnowledgeCandidate', () => { selectedPaths: ['knowledge/original.md', 'knowledge/keep.md'], rationale: 'The dropped page is redundant with an existing source.', selectionMetadata: { reviewer: 'test-reviewer', policyVersion: 1 }, + diagnose() { + diagnosisCalls += 1 + return [ + { + id: 'selected-gap', + kind: 'unknown', + severity: 'info', + message: 'Inspect selected pages.', + }, + ] + }, async evaluate(input) { + expect(input.lifecycle?.findings.map((finding) => finding.id)).toEqual(['selected-gap']) + expect(input.lifecycle?.knowledgeUpdate?.applied).toBe(true) expect(input.candidateRoot).not.toBe(input.baselineRoot) await expect( readFile(join(input.candidateRoot, 'knowledge', 'keep.md'), 'utf8'), @@ -62,6 +76,7 @@ describe('improveSelectedKnowledgeCandidate', () => { const candidate = knowledgeImprovementCandidateRef(selected) expect(evaluatedSelectedSnapshot).toBe(true) + expect(diagnosisCalls).toBe(1) expect(candidate.baseHash).toBe(baseHash) expect(selected.selection).toMatchObject({ kind: 'measured-knowledge-change-selection-receipt', diff --git a/tests/loops/research-driving-loop.test.ts b/tests/loops/research-driving-loop.test.ts index ad2a538..b8bb27f 100644 --- a/tests/loops/research-driving-loop.test.ts +++ b/tests/loops/research-driving-loop.test.ts @@ -76,10 +76,7 @@ const corpus: ScriptedSource[] = [ }, ] -// minSources: 2 keeps the readiness gate UNMET after a single source, so the -// loop stays not-ready and the driver folds steer (its depth-driving channel) -// across rounds. This also mirrors the driver's own bar: a claim is not settled -// until >= 2 independent sources back it. +// Storage readiness passes after one source; the driver still requires corroboration. const specs: KnowledgeReadinessSpec[] = [ defineReadinessSpec({ id: 'topic/definition', @@ -87,7 +84,7 @@ const specs: KnowledgeReadinessSpec[] = [ query: 'self speculative decoding how it works method', requiredFor: ['ResearchAgent'], importance: 'blocking', - minSources: 2, + minSources: 1, minHits: 1, }), ] @@ -187,11 +184,6 @@ describe('research-driving driver in the real two-agent loop (offline, scripted) goal: 'self-speculative decoding', worker: scriptedWorker(), driver, - // Readiness is satisfiable by one source, so the loop would otherwise stop - // early — we run multiple rounds to exercise the driver's depth-driving by - // NOT marking it ready until round 2 reveals the corroborating source. The - // readiness spec only needs one hit, so we drive >1 round via maxRounds and - // assert on the driver's own state, which is the real "done" signal. readinessSpecs: specs, maxRounds: 3, onRound: () => { @@ -207,7 +199,9 @@ describe('research-driving driver in the real two-agent loop (offline, scripted) }) // The loop ran and the KB grew. - expect(result.steps.length).toBeGreaterThanOrEqual(1) + expect(result.steps.length).toBeGreaterThanOrEqual(2) + expect(result.steps[0]?.readiness?.report.blockingMissingRequirements).toEqual([]) + expect(result.steps[0]?.ready).toBe(false) const index = await buildKnowledgeIndex(root) expect(index.sources.length).toBeGreaterThanOrEqual(1) @@ -288,7 +282,7 @@ describe('research-driving driver in the real two-agent loop (offline, scripted) } } - await runVerifiedResearchLoop({ + const result = await runVerifiedResearchLoop({ root, goal: 'self-speculative decoding', worker: floodWorker, @@ -304,5 +298,99 @@ describe('research-driving driver in the real two-agent loop (offline, scripted) expect(state.claims[0]?.supportingHosts.size).toBe(1) expect(state.weaklySupported).toHaveLength(1) expect(driver.isComplete()).toBe(false) + expect(result.rounds).toBe(2) + expect(result.readiness?.report.blockingMissingRequirements).toEqual([]) + expect(result.ready).toBe(false) + }) + + it('keeps storage-only drivers unchanged and skips work when both checks already pass', async () => { + const first = await runVerifiedResearchLoop({ + root, + goal: 'self-speculative decoding', + worker: scriptedWorker(), + driver: { verifySource: () => ({ accept: true }) }, + readinessSpecs: specs, + maxRounds: 3, + }) + expect(first.rounds).toBe(1) + expect(first.ready).toBe(true) + + const resumed = await runVerifiedResearchLoop({ + root, + goal: 'self-speculative decoding', + worker: () => { + throw new Error('completed research must not launch a worker') + }, + driver: { verifySource: () => ({ accept: true }), isComplete: () => true }, + readinessSpecs: specs, + }) + expect(resumed.rounds).toBe(0) + expect(resumed.ready).toBe(true) + }) + + it('prepares and checkpoints steering with no storage gaps until the driver completes', async () => { + const calls: string[] = [] + let complete = false + const worker = scriptedWorker() + const result = await runVerifiedResearchLoop({ + root, + goal: 'self-speculative decoding', + worker: async (context) => { + calls.push(`worker:${context.round}`) + if (context.round === 2) { + expect(context.gaps).toEqual([]) + expect(context.steer).toBe('Corroborate the policy claim.') + complete = true + } + return worker(context) + }, + driver: { + verifySource: () => ({ accept: true }), + isComplete: () => complete, + prepareFold: () => { + calls.push('prepare') + }, + foldGaps: (gaps) => { + expect(gaps).toEqual([]) + calls.push('fold') + return 'Corroborate the policy claim.' + }, + checkpoint: () => { + calls.push('checkpoint') + }, + }, + onRound: ({ round }) => { + calls.push(`published:${round}`) + }, + readinessSpecs: specs, + maxRounds: 3, + }) + + expect(result.rounds).toBe(2) + expect(result.ready).toBe(true) + expect(calls).toEqual([ + 'worker:1', + 'prepare', + 'fold', + 'checkpoint', + 'published:1', + 'worker:2', + 'checkpoint', + 'published:2', + ]) + }) + + it('never reports ready without readiness specs even when the driver is complete', async () => { + const result = await runVerifiedResearchLoop({ + root, + goal: 'self-speculative decoding', + worker: () => ({ sources: [] }), + driver: { verifySource: () => ({ accept: true }), isComplete: () => true }, + readinessSpecs: [], + maxRounds: 2, + }) + expect(result.rounds).toBe(2) + expect(result.ready).toBe(false) + expect(result.readiness).toBeUndefined() }) })