From 32a53741ff226158eb744faf84127fe1f5a18fa7 Mon Sep 17 00:00:00 2001 From: Drew Stone Date: Tue, 6 Oct 2026 01:16:39 -0600 Subject: [PATCH] refactor(knowledge)!: remove unused KB policy optimization, relation graph, and invalidation helpers No consumer imports optimizeKnowledgeBasePolicy, improveSelectedKnowledgeCandidate, buildKnowledgeRelationGraph, knowledgeCitationAuditFindings, planInvalidationPropagation, calibrateRagAnswerJudge, or createRagAnswerQualityHook. improveKnowledgeBase, the page graph, citation resolution, lint, and RAG answer scoring stay. BREAKING CHANGE: those root exports and the relation-graph types and schemas are removed. --- CHANGELOG.md | 2 + README.md | 21 +- api-surface.json | 34 -- docs/architecture.md | 1 - scripts/verify-package.mjs | 1 - src/citation-lint.test.ts | 48 -- src/citation-lint.ts | 59 -- src/graph.test.ts | 17 - src/index.ts | 3 - src/invalidation-propagation.test.ts | 111 ---- src/invalidation-propagation.ts | 104 ---- src/kb-improvement.ts | 13 - src/kb-improvement/optimization.ts | 191 ------- src/kb-improvement/selected-candidate.ts | 529 ------------------ src/rag-eval.ts | 1 - src/rag-eval/calibration.ts | 87 --- src/relation-graph.test.ts | 259 --------- src/relation-graph.ts | 376 ------------- src/schemas.ts | 12 - src/types.ts | 16 - tests/kb-improvement/optimization.test.ts | 420 -------------- .../kb-improvement/selected-candidate.test.ts | 242 -------- tests/kb-improvement/state-scope.test.ts | 39 +- tests/rag-eval.test.ts | 35 -- 24 files changed, 6 insertions(+), 2615 deletions(-) delete mode 100644 src/citation-lint.test.ts delete mode 100644 src/citation-lint.ts delete mode 100644 src/invalidation-propagation.test.ts delete mode 100644 src/invalidation-propagation.ts delete mode 100644 src/kb-improvement/optimization.ts delete mode 100644 src/kb-improvement/selected-candidate.ts delete mode 100644 src/rag-eval/calibration.ts delete mode 100644 src/relation-graph.test.ts delete mode 100644 src/relation-graph.ts delete mode 100644 tests/kb-improvement/optimization.test.ts delete mode 100644 tests/kb-improvement/selected-candidate.test.ts diff --git a/CHANGELOG.md b/CHANGELOG.md index 944283a..bb0fa57 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,6 +12,8 @@ Compare memory providers on product tasks with Agent Eval. Memory adapters, branches, holdout, lifecycle bounds, play memory tools, and the `/benchmarks` in-memory and no-op adapters are unchanged. Remove the `/sources` entry point and the unused authority adapters behind it (Cornell LII, IRS publications, state Secretary of State, polite HTTP fetch, HTML extraction), plus `detectChanges` and the freshness stores. No consumer imported them. The source registry (`addSourceText`, `addSourcePath`, `loadSourceRegistry`), source adapters such as `textSourceAdapter`, and readiness freshness scoring are unchanged. +Remove unused root APIs: `optimizeKnowledgeBasePolicy`, `improveSelectedKnowledgeCandidate`, `buildKnowledgeRelationGraph` with its queries and schemas, `knowledgeCitationAuditFindings`, `planInvalidationPropagation`, `formatKnowledgeInvalidationProposal`, `calibrateRagAnswerJudge`, and `createRagAnswerQualityHook`. +`improveKnowledgeBase`, `buildKnowledgeGraph`, citation resolution and audit, the lint `cites-invalidated` warning, and RAG answer scoring are unchanged. ## 19.1.5 diff --git a/README.md b/README.md index 99e56c6..317b131 100644 --- a/README.md +++ b/README.md @@ -29,7 +29,6 @@ Use Knowledge 17 with older Eval releases. | Prove what knowledge was visible, retrieved, and selected for use | `createKnowledgeRetrievalReceipt`, `createKnowledgeUseReceipt` | package root | | Improve a live knowledge base without editing it in place | `improveKnowledgeBase` | package root | | Optimize retrieval or a complete RAG configuration | `runRetrievalImprovementLoop`, `runRagOptimization` | package root | -| Optimize a KB maintenance policy | `optimizeKnowledgeBasePolicy` | package root | | Run retrieval, research, answer checks, and promotion as one process | `runRagKnowledgeImprovementLoop` | package root | | Connect a memory provider or branch its state | `AgentMemoryAdapter`, `createAgentMemoryBranch` | `/memory` | | Use live research or coding agents | `runKnowledgeImprovementJob` | `@tangle-network/agent-runtime` | @@ -88,7 +87,6 @@ Pass `refresh: 'always'` to rebuild its index before every query, or call `inval Use `asRetrievalEvalRetriever()` to send the same search path into retrieval tests. `knowledgePageRelations(pages)` lists the labeled relations between pages (`wikilink`, `citation`, `shared-source`, `contradicts`), and `buildKnowledgeGraph` collapses them into the weighted page graph stored in the index. -For caller-defined provenance (runs, claims, models, any predicate), `buildKnowledgeRelationGraph({ nodes, relations })` keeps one edge per `(sourceId, targetId, predicate)`, refuses a conflicting repeat or an undeclared endpoint, and `neighbors`, `walk`, and `isReachable` query it by predicate and direction; `KnowledgeRelationGraphSchema` round-trips a persisted graph with its metadata. Pages live under `knowledge/` unless you name another root-relative directory. `loadKnowledgePages`, `buildKnowledgeIndex`, `writeKnowledgeIndex`, `applyKnowledgeWriteBlocks`, `createFileSystemSearchProvider`, and `createRunScopedStores` all take one `pagesDirectory` option (`KnowledgePagesOptions`), so a store laid out as `kb/pages//` is read, indexed, searched, chained, and written through the same value. @@ -279,19 +277,9 @@ const receipt = createKnowledgeRetrievalReceipt({ `excludeInvalidated` defaults to **true** here, the opposite of `searchKnowledge`: a brief offers every page it names with an id ready to cite, so a refuted page in it invites a run to build on a dead claim. `maxChars` bounds the brief, and a page whose line does not fit is left out of `text`, `hits`, `citationIds`, and `results` alike, so all four always describe one identical set. -## Propagate an invalidation +## Invalidated pages -A page whose own evidence refuted it carries an `invalidation`. A reader who arrives through a citation never meets that verdict, so run the propagation pass after grading: - -```ts -const plan = planInvalidationPropagation(originatedPages(await loadKnowledgePages(root))) -if (plan.stamps.length > 0) { - await applyKnowledgeWriteBlocks(root, formatKnowledgeInvalidationProposal(plan)) -} -``` - -Each stamped page records `citesInvalidated: [ids]` in its frontmatter, and nothing else changes. -The plan is a diff, so a second pass over an already stamped store produces no mutation, and a citation whose target was revalidated has its stamp removed. +A page whose own evidence refuted it carries an `invalidation`. `agent-knowledge lint` reports a `cites-invalidated` warning for every live citation into a refuted page, and `searchKnowledge(index, query, { excludeInvalidated: true })` drops the refuted pages from a result set. The default stays `false`: a caller reading history needs them. @@ -379,9 +367,7 @@ Use the narrowest API that matches the job: |---|---| | `runRetrievalImprovementLoop` | Runs one complete `OptimizationMethod` over serialized retrieval configuration. | | `runRagOptimization` | Optimizes retrieval and answer behavior as one serialized RAG configuration. | -| `optimizeKnowledgeBasePolicy` | Optimizes a KB maintenance policy, then applies only the selected policy to an isolated candidate. | | `scoreKnowledgeBaseIndex` | Measures KB structure, citations, source freshness, and configured quality thresholds. | -| `createRagAnswerQualityHook` | Adapts answer-quality checks such as support, relevance, citations, and abstention. | | `runRagKnowledgeImprovementLoop` | Connects retrieval tuning, gap diagnosis, source acquisition, KB updates, answer checks, and a promotion decision. | | `improveKnowledgeBase` | Adds resumable state, isolated candidates, exact promotion, and conflict detection around that process. | @@ -430,7 +416,7 @@ Official external methods must report observed package identity. Custom in-process methods have no external package identity, so their behavior must be covered by `executionRef`. Treat `accountingComplete: false` as incomplete evidence for activation. -Retrieval, RAG, serialized-candidate, and KB-policy optimization accept Eval's optional `claim` and `finalEvidence` options. +Retrieval, RAG, and serialized-candidate optimization accept Eval's optional `claim` and `finalEvidence` options. Use `claim.independentUnit` to group questions from the same source. A `new-units` claim requires separate source units for development and final evaluation. Results retain scenario scores, source-unit scores, and both observation counts. @@ -449,7 +435,6 @@ Only answer evaluation, the terminal promotion decision, and the returned result Answer-quality evidence must name at least two final scenario IDs, immutable dataset and evaluator references, non-empty finite metrics, and observed cost accounting. Promotion also requires `answerQualityCostCeiling`. -`calibrateRagAnswerJudge()` checks supplied strong and weak fixtures; it does not measure an evaluator's error rates. For evaluator admission, use `auditEvaluator()` from `@tangle-network/agent-eval/meta-eval` with actual judgments of independently verified controls. The application must enforce evaluator and auditor separation and retain evidence for the labels. diff --git a/api-surface.json b/api-surface.json index 99d8202..843a772 100644 --- a/api-surface.json +++ b/api-surface.json @@ -31,10 +31,8 @@ "Bm25Hit": "type a6c513ea2447", "Bm25Options": "type 9da64a8973e6", "BuildEvalKnowledgeBundleOptions": "type 8951c7835e42", - "BuildKnowledgeRelationGraphInput": "type f08e22cb9d75", "BuildRetrievalEvalDispatchOptions": "type 2a1fe2b2da73", "CHECKABLE_RUNG_THRESHOLD": "value 534bcde62c80", - "CITES_INVALIDATED_FIELD": "value 3fff28ee6d1f", "CheckExecution": "type 884df3b223af", "ChunkingOptions": "type 00fb66d7d155", "ClaimEvidence": "type f780da49a3ef", @@ -83,8 +81,6 @@ "HindsightClientLike": "type e4b6e0bd1488", "HindsightMemoryAdapterOptions": "type 995d4468eb3c", "HindsightOperationUnknownError": "value 408246fe51ae", - "ImproveSelectedKnowledgeCandidateOptions": "type c738ecaa49fa", - "ImproveSelectedKnowledgeCandidateResult": "type 3a8454d1031d", "KB_CLAIM_LEDGER_DIR": "value 2e94ea580600", "KB_EVENTS_PATH": "value 7c4a3ddbac83", "KB_INDEX_PATH": "value f450045c9510", @@ -126,7 +122,6 @@ "KnowledgeDiscoveryWorker": "type 704c91abebe2", "KnowledgeDuplicateIntakeError": "value 1a079e39f277", "KnowledgeDuplicateIntakePair": "type 839de966eb41", - "KnowledgeEvaluationPhase": "type 1a50f18a958f", "KnowledgeEvent": "type 5c127ca4decf", "KnowledgeEventQuery": "type 748b7d9d7b93", "KnowledgeEventSchema": "value 373728f5643d", @@ -165,8 +160,6 @@ "KnowledgeIndex": "type 332f4c423f7d", "KnowledgeIndexSchema": "value 373728f5643d", "KnowledgeInspection": "type 112825b7bb54", - "KnowledgeInvalidationPlan": "type 62d21063bf4d", - "KnowledgeInvalidationStamp": "type a31367c54758", "KnowledgeLayout": "type beb95ce22863", "KnowledgeLexicalFieldBoosts": "type 019abb7897d0", "KnowledgeLexicalIndex": "type 81bf4d9fd93b", @@ -184,7 +177,6 @@ "KnowledgePageSchema": "value 373728f5643d", "KnowledgePagesOptions": "type 07b5e1cb6149", "KnowledgePolicy": "type 6e24b2884fd6", - "KnowledgePolicyDispatch": "type b5df3ac0d310", "KnowledgePromotionEntry": "type 307413f52c66", "KnowledgePromotionError": "value d1141cf28fad", "KnowledgePromotionErrorCode": "type cb7b16fdb529", @@ -195,18 +187,7 @@ "KnowledgeReadinessSpec": "type 946661b52948", "KnowledgeReceiptAttributeValue": "type af09de7ef965", "KnowledgeRelation": "type f930b03cefcf", - "KnowledgeRelationDirection": "type efbba5c61bea", - "KnowledgeRelationGraph": "type a4c42e720d21", - "KnowledgeRelationGraphError": "value cadf45a5ab8e", - "KnowledgeRelationGraphErrorCode": "type 62a06ddc01f3", - "KnowledgeRelationGraphSchema": "value 373728f5643d", - "KnowledgeRelationNeighbor": "type e463470d2186", - "KnowledgeRelationNode": "type 941a44f20696", - "KnowledgeRelationNodeSchema": "value 373728f5643d", - "KnowledgeRelationQuery": "type 59aad90a4bf4", "KnowledgeRelationSchema": "value 373728f5643d", - "KnowledgeRelationWalkOptions": "type fe4fbf78ea49", - "KnowledgeRelationWalkStep": "type dce4639198b3", "KnowledgeRelease": "type 28df725e17fa", "KnowledgeReleaseInput": "type 844c1ad8d42f", "KnowledgeReleaseReport": "type 5f72778544bd", @@ -238,7 +219,6 @@ "KnowledgeWriteIntakeRequest": "type 5074f0dab514", "KnowledgeWriteParseResult": "type 3ec744fcaf93", "LoadKnowledgeImprovementActivationResultOptions": "type 3c304bc52d1c", - "MeasuredKnowledgeSelectionReceipt": "type 28a9a57c6626", "Mem0ClientMode": "type abfe4225abaf", "Mem0HostedClient": "type 4d1434ac5a9a", "Mem0HostedMemoryAdapterOptions": "type 627ef089dd0c", @@ -250,8 +230,6 @@ "NearDuplicatePair": "type 31bc35ca4a75", "NearDuplicateReport": "type 0ddbba09b111", "Neo4jAgentMemoryAdapterOptions": "type c0746e451047", - "OptimizeKnowledgeBasePolicyOptions": "type 07a3ccb0b8e0", - "OptimizeKnowledgeBasePolicyResult": "type 1e3b03c7a0a3", "OriginatedKnowledgeSearchResult": "type 99110190f58f", "OriginatedPage": "type 62ba6f531a4a", "PageOrigin": "type df33a580b8a9", @@ -400,9 +378,7 @@ "buildKnowledgeGraph": "value f4bc0817e156", "buildKnowledgeIndex": "value 7a198b5e15bd", "buildKnowledgeLexicalIndex": "value 61e7f22cdce2", - "buildKnowledgeRelationGraph": "value 2c4502c049ed", "buildRetrievalEvalDispatch": "value 4f12d1f4c065", - "calibrateRagAnswerJudge": "value 74f91dafd1b1", "canonicalPathsEqual": "value 1a1d22b3b1c3", "canonicalRelativeWithinRoot": "value 9a5ffa3946f3", "chunkMarkdown": "value 3f930d4ba32f", @@ -431,7 +407,6 @@ "createPlayMemoryTools": "value cbd6422b065c", "createQmdKnowledgeTools": "value 136b3d258c5c", "createQmdSearchProvider": "value f1d2a2d4d631", - "createRagAnswerQualityHook": "value 7e18d330e76b", "createRunScopedStores": "value 6bfc535c7281", "decodeKnowledgeVisibilitySnapshot": "value 143bc2b124e6", "deepQuestionId": "value ffec05529ae9", @@ -449,7 +424,6 @@ "forkAgentMemoryBranchSnapshot": "value 0c258d0909ac", "formatFrontmatter": "value e0a4508d4ded", "formatKnowledgeCitationReference": "value 8c9301778549", - "formatKnowledgeInvalidationProposal": "value 0facdbad3e8e", "fromAgentCandidateKnowledgeRef": "value 247b544b764e", "gradeClaims": "value 5a3c1c162258", "gradeFor": "value 79db4a5137a3", @@ -458,7 +432,6 @@ "hashKnowledgeBase": "value bdb2d9eea5f9", "hindsightMemoryBankId": "value b828b7e949f0", "improveKnowledgeBase": "value a4ebce095bdf", - "improveSelectedKnowledgeCandidate": "value 6d48e36af283", "initKnowledgeBase": "value 65692669c5c8", "inspectKnowledgeIndex": "value 79b67f7de1cd", "inspectPendingKnowledgeMutation": "value 38b2e30827bd", @@ -466,12 +439,10 @@ "isKnowledgeMutationHeld": "value e8f0de09e802", "isKnowledgePagePath": "value 2819fa14db65", "isMissingFile": "value 73ca49f2b72a", - "isReachable": "value 980b56e89641", "isSafeKnowledgePath": "value 47d3e4caf0ee", "isScaffoldPath": "value 2819fa14db65", "jsonCandidateCodec": "value 6520c69bb348", "jsonObjectCandidateCodec": "value 8eb601cd6de7", - "knowledgeCitationAuditFindings": "value 639f15656570", "knowledgeImprovementCandidateRef": "value 2c85f9c2a1a0", "knowledgeImprovementRunDir": "value d3fdafeff129", "knowledgeImprovementRunId": "value b5b9477f7ef3", @@ -482,7 +453,6 @@ "knowledgeVisibilityArtifactRef": "value dfa0672dff59", "layoutFor": "value a4861d951cf1", "linkClaimContradictions": "value cf315a2deab2", - "lintCurrentRunCitations": "value 59004ed0538a", "lintKnowledgeIndex": "value c866a9cbb418", "listRegularFilesWithinRoot": "value f2b302b691fa", "loadKnowledgeImprovementActivationResult": "value c55a6c68e27a", @@ -499,20 +469,17 @@ "memoryWriteResultToSourceRecord": "value bd019f5d823f", "mergeClaimLedgers": "value d40bae24e22c", "mergeTrackedClaims": "value af96c4516708", - "neighbors": "value 29f67f6cea95", "normalizeClaimText": "value 4c9eca896a37", "normalizeExternalRagScores": "value 1a2d603c9282", "normalizeKnowledgeStateScope": "value beef86fc6f66", "normalizeLinkTarget": "value 0258f337b3b3", "normalizePageText": "value 9c20decc8888", "normalizePagesDirectory": "value 7af1695dc6cf", - "optimizeKnowledgeBasePolicy": "value 3cec15621ce3", "originatedPages": "value 38bee2f6341e", "parseFrontmatter": "value 62f43b77d35a", "parseKnowledgeCitationReference": "value 098a42106c17", "parseKnowledgeWriteBlocks": "value d6bd24d2858d", "partitionRetrievalScenarios": "value 363802debe8a", - "planInvalidationPropagation": "value 39da8e2c1a14", "promoteKnowledgeCandidate": "value ea1e94a3d6a2", "promoteRunScopedPages": "value 58ff1ec22528", "proposeFromFinding": "value 1dcecd58e869", @@ -578,7 +545,6 @@ "verifyKnowledgeRetrievalReceipt": "value e7dd569982cb", "verifyKnowledgeUseReceipt": "value 13e38b6c70c9", "verifyKnowledgeVisibilitySnapshot": "value cc5c6db30910", - "walk": "value 5035737ef563", "withKnowledgeImprovementCandidate": "value 1807f0708eaa", "withKnowledgeImprovementComparison": "value 209ea121fc27", "withKnowledgeMutation": "value 499371d01638", diff --git a/docs/architecture.md b/docs/architecture.md index 3931021..30cf6d0 100644 --- a/docs/architecture.md +++ b/docs/architecture.md @@ -8,7 +8,6 @@ It owns the small set of primitives every serious agent knowledge system needs: - generated knowledge pages and units - claims with source references - deterministic indexing, graph construction, search, and lint -- labeled relation graphs with one edge per `(source, target, predicate)` and neighbor, walk, and reachability queries - retrieval/RAG candidate surfaces, gold-target scoring, and eval-loop adapters - safe LLM write proposals - eval-gated release confidence through `@tangle-network/agent-eval` diff --git a/scripts/verify-package.mjs b/scripts/verify-package.mjs index 2a1932a..64f1901 100644 --- a/scripts/verify-package.mjs +++ b/scripts/verify-package.mjs @@ -24,7 +24,6 @@ const publicImports = [ const requiredRootExports = [ 'createFileSystemSearchProvider', 'normalizeKnowledgeStateScope', - 'optimizeKnowledgeBasePolicy', 'runRagOptimization', 'runRetrievalImprovementLoop', 'runSerializedKnowledgeOptimization', diff --git a/src/citation-lint.test.ts b/src/citation-lint.test.ts deleted file mode 100644 index 073451d..0000000 --- a/src/citation-lint.test.ts +++ /dev/null @@ -1,48 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { knowledgeCitationAuditFindings } from './citation-lint' -import { auditKnowledgeCitations } from './citation-resolution' -import type { OriginatedPage, PageOrigin } from './run-scoped' -import type { KnowledgePage } from './types' - -function page(id: string, origin: PageOrigin, path: string, cites?: string[]): OriginatedPage { - const value: KnowledgePage = { - id, - path, - title: id, - text: id, - frontmatter: { id }, - sourceIds: [], - tags: [], - outLinks: [], - ...(cites ? { cites } : {}), - } - return { page: value, origin } -} - -describe('knowledgeCitationAuditFindings', () => { - it('produces blocking missing, ambiguous, and self-citation findings', () => { - const visible = [ - page('author', 'here', 'knowledge/author.md', ['missing', 'reused', 'author']), - page('reused', 'inherited:parent', 'knowledge/parent.md'), - page('reused', 'shared', 'knowledge/shared.md'), - ] - - const findings = knowledgeCitationAuditFindings( - auditKnowledgeCitations(visible, { sourceOrigins: ['here'] }), - ) - - expect(findings.map((finding) => [finding.type, finding.severity])).toEqual([ - ['broken-citation', 'error'], - ['ambiguous-citation', 'error'], - ['broken-citation', 'error'], - ]) - expect(findings[1]?.message).toMatch(/qualify it as here::/) - expect(findings[1]?.metadata).toMatchObject({ - sourcePageId: 'author', - candidates: [ - { origin: 'inherited:parent', path: 'knowledge/parent.md' }, - { origin: 'shared', path: 'knowledge/shared.md' }, - ], - }) - }) -}) diff --git a/src/citation-lint.ts b/src/citation-lint.ts deleted file mode 100644 index 53c066a..0000000 --- a/src/citation-lint.ts +++ /dev/null @@ -1,59 +0,0 @@ -import { - auditCurrentRunCitations, - type KnowledgeCitationAuditIssue, - type KnowledgeCitationAuditReport, -} from './citation-resolution' -import type { RunScopedStores } from './run-scoped' -import type { KnowledgeLintFinding } from './types' - -/** Convert a chain-aware citation audit into the package lint vocabulary. */ -export function knowledgeCitationAuditFindings( - report: KnowledgeCitationAuditReport, -): KnowledgeLintFinding[] { - return report.issues.map(issueToFinding) -} - -/** Lint only current-run pages against their complete declared visibility chain. */ -export async function lintCurrentRunCitations( - stores: RunScopedStores, - runId: string, -): Promise { - return knowledgeCitationAuditFindings(await auditCurrentRunCitations(stores, runId)) -} - -function issueToFinding(issue: KnowledgeCitationAuditIssue): KnowledgeLintFinding { - const candidates = issue.candidates.map((candidate) => ({ - pageId: candidate.pageId, - path: candidate.page.path, - origin: candidate.origin, - })) - if (issue.kind === 'ambiguous') { - return { - type: 'ambiguous-citation', - severity: 'error', - page: issue.sourcePath, - message: - `Citation "${issue.persistedCitation}" resolves to ${issue.candidates.length} visible pages; ` + - 'qualify it as here::, inherited:::, or shared::.', - metadata: { - sourcePageId: issue.sourcePageId, - sourceOrigin: issue.sourceOrigin, - candidates, - }, - } - } - return { - type: 'broken-citation', - severity: 'error', - page: issue.sourcePath, - message: - issue.kind === 'self' - ? `Page "${issue.sourcePageId}" cites itself through "${issue.persistedCitation}".` - : `Citation "${issue.persistedCitation}" resolves to no visible page.`, - metadata: { - sourcePageId: issue.sourcePageId, - sourceOrigin: issue.sourceOrigin, - candidates, - }, - } -} diff --git a/src/graph.test.ts b/src/graph.test.ts index db3b608..a84bc72 100644 --- a/src/graph.test.ts +++ b/src/graph.test.ts @@ -1,6 +1,5 @@ import { describe, expect, it } from 'vitest' import { buildKnowledgeGraph, knowledgePageRelations } from './graph' -import { buildKnowledgeRelationGraph, neighbors } from './relation-graph' import type { KnowledgeGraph, KnowledgePage } from './types' function page( @@ -176,22 +175,6 @@ describe('knowledgePageRelations', () => { ]) }) - it('keeps two predicates between one pair as two relations in a relation graph', () => { - const relations = knowledgePageRelations(pages).filter( - (relation) => relation.predicate !== 'shared-source', - ) - const graph = buildKnowledgeRelationGraph({ relations }) - const out = neighbors(graph, 'attention', { direction: 'out' }) - expect(out.map((neighbor) => [neighbor.nodeId, neighbor.relation.predicate])).toEqual([ - ['flash-attention', 'wikilink'], - ['flash-attention', 'citation'], - ['orphan', 'contradicts'], - ]) - expect(neighbors(graph, 'orphan', { direction: 'in', predicate: 'contradicts' })).toHaveLength( - 2, - ) - }) - it('emits nothing for a missing, ambiguous, or self target', () => { const relations = knowledgePageRelations(pages) expect(relations.some((relation) => relation.targetId === 'missing-page')).toBe(false) diff --git a/src/index.ts b/src/index.ts index c12abaf..8eede23 100644 --- a/src/index.ts +++ b/src/index.ts @@ -2,7 +2,6 @@ export * from './adapters' export * from './agent-candidate' export * from './benchmarks/index' export * from './chunking' -export * from './citation-lint' export * from './citation-resolution' export * from './claim-evidence' export * from './claim-grounding' @@ -39,7 +38,6 @@ export * from './graph' export * from './ids' export * from './indexer' export * from './inspect' -export * from './invalidation-propagation' export * from './kb-improvement' export * from './kb-store' export { @@ -83,7 +81,6 @@ export * from './rag-eval' export * from './rag-improvement-loop' export * from './rag-optimization' export * from './readiness-check' -export * from './relation-graph' export * from './release' export * from './research-loop' export * from './retrieval-eval' diff --git a/src/invalidation-propagation.test.ts b/src/invalidation-propagation.test.ts deleted file mode 100644 index 472e1c4..0000000 --- a/src/invalidation-propagation.test.ts +++ /dev/null @@ -1,111 +0,0 @@ -import { mkdtemp, readFile, realpath, rm, writeFile } from 'node:fs/promises' -import { tmpdir } from 'node:os' -import { join } from 'node:path' -import { afterEach, beforeEach, describe, expect, it } from 'vitest' -import { - formatKnowledgeInvalidationProposal, - planInvalidationPropagation, -} from './invalidation-propagation' -import { applyKnowledgeWriteBlocks } from './proposals' -import { originatedPages } from './run-scoped' -import { initKnowledgeBase, loadKnowledgePages } from './store' -import type { KnowledgePage } from './types' - -const overturned = { - verdict: 'contradicted' as const, - observedAt: '2026-08-18T00:00:00.000Z', - reason: 'The replication measured the opposite direction.', -} - -function page(id: string, frontmatter: Record): KnowledgePage { - const cites = frontmatter.cites as string[] | undefined - return { - id, - path: `knowledge/${id}.md`, - title: id, - text: `Body of ${id}.`, - frontmatter: { id, ...frontmatter }, - sourceIds: [], - tags: [], - outLinks: [], - ...(cites ? { cites } : {}), - ...(frontmatter.invalidation ? { invalidation: overturned } : {}), - } -} - -describe('planInvalidationPropagation', () => { - it('stamps a citer of an invalidated page and clears the stamp when the verdict is gone', () => { - const refuted = page('refuted', { invalidation: overturned }) - const citer = page('citer', { cites: ['refuted'] }) - - const plan = planInvalidationPropagation(originatedPages([refuted, citer])) - - expect(plan.invalidatedPageIds).toEqual(['refuted']) - expect(plan.stamps.map((stamp) => [stamp.page.id, stamp.citesInvalidated])).toEqual([ - ['citer', ['refuted']], - ]) - - const stamped = page('citer', { cites: ['refuted'], citesInvalidated: ['refuted'] }) - const revalidated = page('refuted', {}) - expect( - planInvalidationPropagation(originatedPages([revalidated, stamped])).stamps.map((stamp) => [ - stamp.page.id, - stamp.citesInvalidated, - ]), - ).toEqual([['citer', []]]) - }) - - it('never stamps a page the store only inherits', () => { - const refuted = page('refuted', { invalidation: overturned }) - const inheritedCiter = page('inherited-citer', { cites: ['refuted'] }) - - const plan = planInvalidationPropagation([ - ...originatedPages([refuted]), - ...originatedPages([inheritedCiter], 'inherited:run-a'), - ]) - - expect(plan.stamps).toEqual([]) - }) -}) - -describe('applying an invalidation plan through the write path', () => { - let root: string - - beforeEach(async () => { - root = await realpath(await mkdtemp(join(tmpdir(), 'invalidation-'))) - await initKnowledgeBase(root) - }) - afterEach(async () => { - await rm(root, { recursive: true, force: true }) - }) - - it('is a no-op on the second pass over a store it already stamped', async () => { - await writeFile( - join(root, 'knowledge', 'refuted.md'), - `---\nid: refuted\ninvalidation: ${JSON.stringify(overturned)}\n---\n\n# Refuted\n\nA claim its own replication overturned.\n`, - ) - await writeFile( - join(root, 'knowledge', 'citer.md'), - '---\nid: citer\ntags:\n - live\ncites:\n - refuted\n---\n\n# Citer\n\nBuilt on the refuted claim.\n', - ) - - const runPass = async () => { - const plan = planInvalidationPropagation(originatedPages(await loadKnowledgePages(root))) - if (plan.stamps.length === 0) return { stamped: [] as string[] } - const applied = await applyKnowledgeWriteBlocks( - root, - formatKnowledgeInvalidationProposal(plan), - ) - return { stamped: applied.written } - } - - expect((await runPass()).stamped).toEqual(['knowledge/citer.md']) - const afterFirst = await readFile(join(root, 'knowledge', 'citer.md'), 'utf8') - expect(afterFirst).toContain('citesInvalidated:\n - refuted') - expect(afterFirst).toContain('- live') - expect(afterFirst).toContain('Built on the refuted claim.') - - expect((await runPass()).stamped).toEqual([]) - expect(await readFile(join(root, 'knowledge', 'citer.md'), 'utf8')).toBe(afterFirst) - }) -}) diff --git a/src/invalidation-propagation.ts b/src/invalidation-propagation.ts deleted file mode 100644 index 27f37b7..0000000 --- a/src/invalidation-propagation.ts +++ /dev/null @@ -1,104 +0,0 @@ -/** - * Invalidation propagation. - * - * A page whose own evidence refuted it carries an `invalidation`. That verdict - * is invisible to a reader who arrives through a citation, so a page that cites - * a refuted page records which of its citations are refuted. The pass is - * planned as a diff, so a store already stamped produces no mutation and the - * pass can run after every grading round. - */ -import { parseKnowledgeCitationReference, resolveKnowledgeCitation } from './citation-resolution' -import { formatFrontmatter } from './frontmatter' -import type { OriginatedPage } from './run-scoped' -import type { KnowledgeId, KnowledgePage } from './types' - -/** Frontmatter field naming the cited pages whose evidence refuted them. */ -export const CITES_INVALIDATED_FIELD = 'citesInvalidated' - -export interface KnowledgeInvalidationStamp { - readonly page: KnowledgePage - /** The value the field must hold, sorted and deduplicated. Empty removes the field. */ - readonly citesInvalidated: readonly KnowledgeId[] - /** The value the page holds now, in the order it is stored. */ - readonly current: readonly KnowledgeId[] -} - -export interface KnowledgeInvalidationPlan { - /** Every visible page carrying an invalidation, sorted by id. */ - readonly invalidatedPageIds: readonly KnowledgeId[] - /** Only the pages whose stamp differs from what they hold, in path order. */ - readonly stamps: readonly KnowledgeInvalidationStamp[] -} - -/** - * Plan the `citesInvalidated` stamp for every page authored in the target - * store. - * - * Citations resolve over the whole chain, so a page here may be stamped for - * citing a refuted inherited or shared page. Only `here` pages are stamped: a - * run does not write the stores it inherits or shares. - */ -export function planInvalidationPropagation( - visiblePages: readonly OriginatedPage[], -): KnowledgeInvalidationPlan { - if (!Array.isArray(visiblePages)) { - throw new TypeError('knowledge invalidation propagation requires the visible pages') - } - const invalidatedPageIds = [ - ...new Set( - visiblePages - .filter((entry) => entry.page.invalidation !== undefined) - .map((entry) => entry.page.id), - ), - ].sort() - - const stamps: KnowledgeInvalidationStamp[] = [] - for (const entry of visiblePages) { - if (entry.origin !== 'here') continue - const page = entry.page - const refuted = new Set() - for (const persisted of page.cites ?? []) { - const resolution = resolveKnowledgeCitation( - visiblePages, - parseKnowledgeCitationReference(persisted), - ) - if (resolution.resolved?.page.invalidation !== undefined) { - refuted.add(resolution.resolved.page.id) - } - } - const citesInvalidated = [...refuted].sort() - const current = idList(page.frontmatter[CITES_INVALIDATED_FIELD]) - if (sameOrder(current, citesInvalidated)) continue - stamps.push({ page, citesInvalidated, current }) - } - stamps.sort((left, right) => left.page.path.localeCompare(right.page.path)) - return { invalidatedPageIds, stamps } -} - -/** - * Render one plan as a write-block proposal for `applyKnowledgeWriteBlocks`. - * - * Only the stamped field changes. The page is rendered through - * `formatFrontmatter`, so its frontmatter is written in that writer's - * normalized form. - */ -export function formatKnowledgeInvalidationProposal(plan: KnowledgeInvalidationPlan): string { - return plan.stamps.map((stamp) => renderStampedBlock(stamp)).join('\n') -} - -function renderStampedBlock(stamp: KnowledgeInvalidationStamp): string { - const frontmatter: Record = { ...stamp.page.frontmatter } - if (stamp.citesInvalidated.length === 0) delete frontmatter[CITES_INVALIDATED_FIELD] - else frontmatter[CITES_INVALIDATED_FIELD] = [...stamp.citesInvalidated] - const content = formatFrontmatter(frontmatter, stamp.page.text) - return `---FILE: ${stamp.page.path}---\n${content.replace(/\n+$/, '')}\n---END FILE---` -} - -function idList(value: unknown): KnowledgeId[] { - const values = typeof value === 'string' ? [value] : Array.isArray(value) ? value : [] - return values.filter((item): item is string => typeof item === 'string' && item.trim() !== '') -} - -function sameOrder(left: readonly string[], right: readonly string[]): boolean { - return left.length === right.length && left.every((item, index) => item === right[index]) -} diff --git a/src/kb-improvement.ts b/src/kb-improvement.ts index da26469..fb1febc 100644 --- a/src/kb-improvement.ts +++ b/src/kb-improvement.ts @@ -33,20 +33,7 @@ export { KnowledgeImprovementEvidenceSchema, KnowledgeImprovementRunStateSchema, } from './kb-improvement/contracts' -export type { - KnowledgePolicyDispatch, - OptimizeKnowledgeBasePolicyOptions, - OptimizeKnowledgeBasePolicyResult, -} from './kb-improvement/optimization' -export { optimizeKnowledgeBasePolicy } from './kb-improvement/optimization' export { improveKnowledgeBase } from './kb-improvement/run' -export type { - ImproveSelectedKnowledgeCandidateOptions, - ImproveSelectedKnowledgeCandidateResult, - KnowledgeEvaluationPhase, - MeasuredKnowledgeSelectionReceipt, -} from './kb-improvement/selected-candidate' -export { improveSelectedKnowledgeCandidate } from './kb-improvement/selected-candidate' export type { KnowledgeImprovementEvent } from './kb-improvement/state' export { knowledgeImprovementRunDir, diff --git a/src/kb-improvement/optimization.ts b/src/kb-improvement/optimization.ts deleted file mode 100644 index 0748ca6..0000000 --- a/src/kb-improvement/optimization.ts +++ /dev/null @@ -1,191 +0,0 @@ -import type { - DispatchContext, - OptimizationMethod, - Scenario, -} from '@tangle-network/agent-eval/campaign' -import type { AgentCandidateJsonValue as JsonValue } from '@tangle-network/agent-interface' -import { sha256, stableId } from '../ids' -import { assertImmutableRef } from '../immutable-ref' -import { - type RunSerializedKnowledgeOptimizationOptions, - type RunSerializedKnowledgeOptimizationResult, - runSerializedKnowledgeOptimization, -} from '../optimization' -import type { - KnowledgeImprovementOptions, - KnowledgeImprovementResult, - KnowledgeImprovementUpdateInput, -} from './contracts' -import { improveKnowledgeBase } from './run' -import { hashKnowledgeBase } from './workspace' - -type PolicyCandidateOptions = Omit< - KnowledgeImprovementOptions, - | 'root' - | 'goal' - | 'implementationRef' - | 'runId' - | 'maxCandidates' - | 'step' - | 'knowledgeResearch' - | 'updateKnowledge' -> - -type PolicyOptimizationBaseOptions< - TPolicy extends JsonValue, - TScenario extends Scenario, - TArtifact, -> = Omit< - RunSerializedKnowledgeOptimizationOptions, - | 'baseline' - | 'method' - | 'trainScenarios' - | 'selectionScenarios' - | 'finalScenarios' - | 'executionRef' -> - -export interface OptimizeKnowledgeBasePolicyOptions< - TPolicy extends JsonValue, - TScenario extends Scenario, - TArtifact, -> extends PolicyOptimizationBaseOptions { - root: string - goal: string - baselinePolicy: TPolicy - method: OptimizationMethod - trainScenarios: readonly TScenario[] - selectionScenarios: readonly TScenario[] - finalScenarios: readonly TScenario[] - /** Commit or content identity for evaluation, applyPolicy, and external dependencies. */ - policyApplicationRef: string - /** Optional namespace for parallel materialization of the same measured policy. */ - candidateRunLabel?: string - candidate?: PolicyCandidateOptions - applyPolicy( - input: KnowledgeImprovementUpdateInput & { - policy: TPolicy - policySurface: string - policySurfaceHash: string - optimizationMethod: string - }, - ): Promise<{ - applied: boolean - summary: string - metadata?: Record - }> -} - -export interface OptimizeKnowledgeBasePolicyResult { - optimization: RunSerializedKnowledgeOptimizationResult - improvement: KnowledgeImprovementResult -} - -/** - * Optimizes a serialized KB-maintenance policy, then materializes the selected - * policy in one isolated knowledge candidate. Activation remains explicit. - */ -export async function optimizeKnowledgeBasePolicy< - TPolicy extends JsonValue, - TScenario extends Scenario, - TArtifact, ->( - options: OptimizeKnowledgeBasePolicyOptions, -): Promise> { - const { - root, - goal, - baselinePolicy, - method, - trainScenarios, - selectionScenarios, - finalScenarios, - policyApplicationRef, - candidateRunLabel, - candidate, - applyPolicy, - ...optimizationOptions - } = options - if (typeof root !== 'string' || !root.trim()) { - throw new Error('optimizeKnowledgeBasePolicy root must be non-empty') - } - if (typeof goal !== 'string' || !goal.trim()) { - throw new Error('optimizeKnowledgeBasePolicy goal must be non-empty') - } - assertImmutableRef(policyApplicationRef, 'optimizeKnowledgeBasePolicy policyApplicationRef') - if ( - candidateRunLabel !== undefined && - (typeof candidateRunLabel !== 'string' || !candidateRunLabel.trim()) - ) { - throw new Error('optimizeKnowledgeBasePolicy candidateRunLabel must be non-empty') - } - const baseHash = await hashKnowledgeBase(root, candidate?.stateScope) - const optimization = await runSerializedKnowledgeOptimization({ - ...optimizationOptions, - executionRef: policyApplicationRef, - baseline: baselinePolicy, - method, - trainScenarios, - selectionScenarios, - finalScenarios, - }) - const winner = optimization.winner - const currentBaseHash = await hashKnowledgeBase(root, candidate?.stateScope) - if (currentBaseHash !== baseHash) { - throw new Error( - `knowledge base changed during policy optimization: expected ${baseHash}, got ${currentBaseHash}`, - ) - } - const runId = stableId( - 'kbpolicy', - `${candidateRunLabel ?? 'default'}:${goal}:${optimization.methodName}:${winner.surfaceHash}:${policyApplicationRef}:${baseHash}`, - ) - const improvement = await improveKnowledgeBase({ - ...(candidate ?? {}), - root, - goal, - implementationRef: `sha256:${sha256( - `${policyApplicationRef}\n${winner.surfaceHash}\n${optimization.methodName}`, - )}`, - runId, - maxCandidates: 1, - updateKnowledge: async (input) => { - if (input.baseHash !== baseHash) { - throw new Error( - `knowledge base changed before policy materialization: expected ${baseHash}, got ${input.baseHash}`, - ) - } - const result = await applyPolicy({ - ...input, - policy: structuredClone(winner.value), - policySurface: winner.surface, - policySurfaceHash: winner.surfaceHash, - optimizationMethod: optimization.methodName, - }) - return { - ...result, - metadata: { - ...(result.metadata ?? {}), - optimization: { - method: optimization.methodName, - policySurfaceHash: winner.surfaceHash, - policyApplicationRef, - }, - }, - } - }, - }) - return { optimization, improvement } -} - -export type KnowledgePolicyDispatch< - TPolicy extends JsonValue, - TScenario extends Scenario, - TArtifact, -> = (input: { - candidate: TPolicy - candidateSurface: string - candidateSurfaceHash: string - scenario: TScenario - context: DispatchContext -}) => Promise diff --git a/src/kb-improvement/selected-candidate.ts b/src/kb-improvement/selected-candidate.ts deleted file mode 100644 index 7361b59..0000000 --- a/src/kb-improvement/selected-candidate.ts +++ /dev/null @@ -1,529 +0,0 @@ -import { createHash } from 'node:crypto' -import { join } from 'node:path' -import { canonicalJson, contentHash } from '@tangle-network/agent-eval' -import type { AgentCandidateJsonValue as JsonValue } from '@tangle-network/agent-interface' -import { z } from 'zod' -import { isMissingFile, readRegularFileWithinRoot, writeJsonDurableWithinRoot } from '../durable-fs' -import { - applyKnowledgeFileTransaction, - assertKnowledgeMutationPath, - finishKnowledgeFileTransaction, - type KnowledgeFileMutation, - type KnowledgeFileTransaction, - type KnowledgeFileTransactionPlanEntry, - knowledgeFileTransactionPlanHash, - prepareKnowledgeFileTransaction, -} from '../file-transaction' -import { stableId } from '../ids' -import { writeKnowledgeIndex } from '../indexer' -import { type KnowledgeStateScope, normalizeKnowledgeStateScope } from '../knowledge-state-scope' -import { withKnowledgeMutation } from '../mutation-lock' -import { normalizePagesDirectory } from '../pages-directory' -import type { RagKnowledgeImprovementPhase } from '../rag-improvement-loop' -import type { - KnowledgeImprovementCandidateRef, - KnowledgeImprovementOptions, - KnowledgeImprovementResult, -} from './contracts' -import { - EVALUATION_PHASES, - immutableRefSchema, - KB_IMPROVEMENT_PAGES_DIRECTORY, - KnowledgeImprovementCandidateRefSchema, - safePathSegmentSchema, -} from './contracts' -import { improveKnowledgeBase } from './run' -import { knowledgeImprovementRunDir } from './state' -import { knowledgeFilePlanEntries } from './transition' -import { hashKnowledgeBase, withKnowledgeImprovementComparison } from './workspace' - -const selectionPathSchema = z - .string() - .min(1) - .transform((path, context) => { - try { - try { - return assertKnowledgeMutationPath(path, KB_IMPROVEMENT_PAGES_DIRECTORY, true) - } catch { - return normalizePagesDirectory(path) - } - } catch (error) { - context.addIssue({ - code: 'custom', - message: error instanceof Error ? error.message : String(error), - }) - return z.NEVER - } - }) -const selectionMetadataSchema = z.record(z.string(), z.json()) -const measuredSelectionLifecycleSchema = z - .object({ - kind: z.literal('measured-knowledge-change-selection'), - version: z.literal(1), - selectionDigest: z.string().regex(/^[a-f0-9]{64}$/), - sourceCandidateId: z.string().min(1), - sourceEvidenceHash: z.string().regex(/^[a-f0-9]{64}$/), - sourcePlanHash: z.string().regex(/^[a-f0-9]{64}$/), - selectedPaths: z.array(selectionPathSchema), - selectedCandidateHash: z.string().regex(/^[a-f0-9]{64}$/), - selectedPlanHash: z.string().regex(/^[a-f0-9]{64}$/), - }) - .strict() -const measuredSelectionReceiptSchema = z - .object({ - kind: z.literal('measured-knowledge-change-selection-receipt'), - version: z.literal(1), - receiptHash: z.string().regex(/^[a-f0-9]{64}$/), - selectionDigest: z.string().regex(/^[a-f0-9]{64}$/), - sourceCandidate: KnowledgeImprovementCandidateRefSchema, - sourcePlanHash: z.string().regex(/^[a-f0-9]{64}$/), - selectedPaths: z.array(selectionPathSchema), - derivedImplementationRef: immutableRefSchema, - runId: z.string().min(1), - derivedCandidateId: z.string().min(1), - derivedCandidateStatus: z.string().min(1), - selectedCandidateHash: z.string().regex(/^[a-f0-9]{64}$/), - selectedPlanHash: z.string().regex(/^[a-f0-9]{64}$/), - selectedEvidenceHash: z.string().regex(/^[a-f0-9]{64}$/), - rationale: z.string().trim().min(1).optional(), - metadata: selectionMetadataSchema.optional(), - }) - .strict() - -export type KnowledgeEvaluationPhase = Exclude< - RagKnowledgeImprovementPhase, - 'knowledge-acquisition' | 'knowledge-update' -> - -export interface ImproveSelectedKnowledgeCandidateOptions - extends Omit< - KnowledgeImprovementOptions, - | 'root' - | 'goal' - | 'implementationRef' - | 'runId' - | 'step' - | 'knowledgeResearch' - | 'acquireKnowledge' - | 'updateKnowledge' - | 'enabledPhases' - | 'requiredPhases' - > { - root: string - goal: string - /** Identity of the helper, policy, or human procedure choosing the subset. */ - implementationRef: string - /** Previously measured whole candidate from which the subset is derived. */ - sourceCandidate: KnowledgeImprovementCandidateRef - /** Exact changed file paths to carry into the derived candidate. */ - selectedPaths: readonly string[] - /** Optional explicit run identity. The default includes the selection digest. */ - runId?: string - /** Human-readable reason for selecting this subset. */ - rationale?: string - /** JSON-safe policy output retained in the selection receipt. */ - selectionMetadata?: Record - /** Diagnosis and evaluation phases. `knowledge-update` is inserted by this helper. */ - enabledEvaluationPhases?: readonly KnowledgeEvaluationPhase[] - /** Diagnosis and evaluation phases that must complete. `knowledge-update` is always required. */ - requiredEvaluationPhases?: readonly KnowledgeEvaluationPhase[] -} - -export type MeasuredKnowledgeSelectionReceipt = z.infer - -export interface ImproveSelectedKnowledgeCandidateResult extends KnowledgeImprovementResult { - selection: MeasuredKnowledgeSelectionReceipt -} - -/** - * Derive a subset from an already measured candidate, then measure the subset as - * a new candidate before it can be promoted. - * - * Directly filtering an activation plan is unsafe: a whole candidate can pass - * while one page subset breaks links, removes supporting sources, or changes a - * readiness score. This helper instead applies the chosen changed paths to an - * isolated baseline, recomputes the index, runs the ordinary improvement - * evaluator, and returns the ordinary candidate reference. Promotion therefore - * remains one path and can never admit an unmeasured hybrid. - */ -export async function improveSelectedKnowledgeCandidate( - options: ImproveSelectedKnowledgeCandidateOptions, -): Promise { - const sourceCandidate = Object.freeze( - KnowledgeImprovementCandidateRefSchema.parse(options.sourceCandidate), - ) - const requestedImplementationRef = immutableRefSchema.parse(options.implementationRef) - const enabledEvaluationPhases = normalizeEvaluationPhases( - options.enabledEvaluationPhases ?? ['gap-diagnosis', ...EVALUATION_PHASES], - 'enabledEvaluationPhases', - ) - const requiredEvaluationPhases = normalizeEvaluationPhases( - options.requiredEvaluationPhases ?? [], - 'requiredEvaluationPhases', - ) - const metadata = options.selectionMetadata - ? selectionMetadataSchema.parse(structuredClone(options.selectionMetadata)) - : undefined - - return withKnowledgeImprovementComparison( - { root: options.root, candidate: sourceCandidate }, - async (source) => { - const scope = normalizeKnowledgeStateScope(source.stateScope) - if ( - options.stateScope !== undefined && - canonicalJson(normalizeKnowledgeStateScope(options.stateScope)) !== canonicalJson(scope) - ) { - throw new Error('selected knowledge candidate stateScope differs from its source') - } - const sourcePlan = await knowledgeFilePlanEntries( - source.baseline.root, - source.candidate.root, - scope, - ) - const sourcePlanHash = knowledgeFileTransactionPlanHash( - sourcePlan, - scope.pagesDirectory, - scope.researchState, - ) - if (sourcePlanHash !== sourceCandidate.promotionPlanHash) { - throw new Error('source knowledge candidate plan no longer matches its measured identity') - } - const changedSourcePlan = sourcePlan - .filter(planEntryChanged) - .filter((entry) => notDerivedPath(entry, scope)) - const selectedPaths = normalizeSelectedPaths(options.selectedPaths, changedSourcePlan, scope) - const selectionMaterial = immutableJson({ - kind: 'measured-knowledge-change-selection' as const, - version: 1 as const, - sourceCandidate, - sourcePlanHash, - selectedPaths, - ...(options.rationale?.trim() ? { rationale: options.rationale.trim() } : {}), - ...(metadata ? { metadata } : {}), - }) - const selectionDigest = contentHash(selectionMaterial) - const derivedImplementationRef = immutableRefSchema.parse( - `sha256:${contentHash({ - engine: 'agent-knowledge/measured-selection-v1', - implementationRef: requestedImplementationRef, - selectionDigest, - })}`, - ) - const runId = - options.runId ?? - stableId( - 'kimpsel', - canonicalJson({ - root: options.root, - goal: options.goal, - derivedImplementationRef, - selectionDigest, - }), - ) - const selectedEntries = selectedPaths.map((path) => { - const entry = changedSourcePlan.find((candidate) => candidate.path === path) - if (!entry) throw new Error(`selected knowledge path disappeared: ${path}`) - return entry - }) - const selectionMutationPlanHash = knowledgeFileTransactionPlanHash( - selectedEntries, - scope.pagesDirectory, - scope.researchState, - ) - let lifecycleSelection: z.infer | undefined - - const result = await improveKnowledgeBase({ - ...improvementOptions(options), - root: options.root, - stateScope: scope, - goal: options.goal, - implementationRef: derivedImplementationRef, - runId, - enabledPhases: ['knowledge-update', ...enabledEvaluationPhases], - requiredPhases: ['knowledge-update', ...requiredEvaluationPhases], - async updateKnowledge(input) { - const purpose = `knowledge-selected-candidate:${selectionDigest}` - const recoveryOwner = `knowledge-selected-candidate:${sourceCandidate.candidateId}` - await withKnowledgeMutation( - input.candidateRoot, - async (lock) => { - lock.assertOwned() - if (!lock.recovery && selectedEntries.length > 0) { - const transaction = await prepareKnowledgeFileTransaction({ - root: input.candidateRoot, - transactionRoot: lock.transactionRoot, - purpose, - ...scope, - recoveryOwner, - mutations: await selectionMutations(source.candidate.root, selectedEntries), - now: options.now, - }) - if (transaction) { - assertSelectionTransaction(transaction, selectionMutationPlanHash) - await applyKnowledgeFileTransaction({ - root: input.candidateRoot, - transactionRoot: lock.transactionRoot, - transaction, - beforeCommit: lock.assertOwned, - }) - await finishKnowledgeFileTransaction({ - root: input.candidateRoot, - transactionRoot: lock.transactionRoot, - transaction, - assertOwned: lock.assertOwned, - }) - } - } - await writeKnowledgeIndex(input.candidateRoot, scope) - lock.assertOwned() - }, - { - resumeTransaction: { - purpose, - recoveryOwner, - validate: (transaction) => - assertSelectionTransaction(transaction, selectionMutationPlanHash), - }, - }, - ) - - const selectedPlan = await knowledgeFilePlanEntries( - source.baseline.root, - input.candidateRoot, - scope, - ) - assertExactSelectedChanges(selectedPlan, selectedPaths, scope) - const selectedPlanHash = knowledgeFileTransactionPlanHash( - selectedPlan, - scope.pagesDirectory, - scope.researchState, - ) - const selectedCandidateHash = await hashKnowledgeBase(input.candidateRoot, scope) - lifecycleSelection = measuredSelectionLifecycleSchema.parse({ - kind: 'measured-knowledge-change-selection', - version: 1, - selectionDigest, - sourceCandidateId: sourceCandidate.candidateId, - sourceEvidenceHash: sourceCandidate.evidenceHash, - sourcePlanHash, - selectedPaths, - selectedCandidateHash, - selectedPlanHash, - }) - return { - applied: selectedPaths.length > 0, - summary: - selectedPaths.length > 0 - ? `Applied ${selectedPaths.length} selected measured change(s).` - : 'Selected no changes; measuring the exact baseline as a null candidate.', - metadata: { selection: lifecycleSelection }, - } - }, - }) - - const candidate = result.candidate - if (!candidate?.candidateHash || !candidate.promotionPlanHash || !candidate.evidenceHash) { - throw new Error('selected knowledge candidate did not produce measured candidate evidence') - } - const recordedSelection = measuredSelectionLifecycleSchema.parse( - result.lifecycle?.knowledgeUpdate?.metadata?.selection ?? lifecycleSelection, - ) - if ( - recordedSelection.selectionDigest !== selectionDigest || - recordedSelection.selectedCandidateHash !== candidate.candidateHash || - recordedSelection.selectedPlanHash !== candidate.promotionPlanHash || - canonicalJson(recordedSelection.selectedPaths) !== canonicalJson(selectedPaths) - ) { - throw new Error('selected knowledge candidate evidence does not bind the measured subset') - } - - const receiptWithoutHash = immutableJson({ - kind: 'measured-knowledge-change-selection-receipt' as const, - version: 1 as const, - selectionDigest, - sourceCandidate, - sourcePlanHash, - selectedPaths, - derivedImplementationRef, - runId: result.runId, - derivedCandidateId: candidate.candidateId, - derivedCandidateStatus: candidate.status, - selectedCandidateHash: candidate.candidateHash, - selectedPlanHash: candidate.promotionPlanHash, - selectedEvidenceHash: candidate.evidenceHash, - ...(options.rationale?.trim() ? { rationale: options.rationale.trim() } : {}), - ...(metadata ? { metadata } : {}), - }) - const receipt = measuredSelectionReceiptSchema.parse({ - ...receiptWithoutHash, - receiptHash: contentHash(receiptWithoutHash), - }) - await persistSelectionReceipt(options.root, receipt) - return { ...result, selection: receipt } - }, - ) -} - -function improvementOptions( - options: ImproveSelectedKnowledgeCandidateOptions, -): Omit< - KnowledgeImprovementOptions, - | 'root' - | 'goal' - | 'implementationRef' - | 'runId' - | 'step' - | 'knowledgeResearch' - | 'acquireKnowledge' - | 'updateKnowledge' - | 'enabledPhases' - | 'requiredPhases' -> { - const { - root: _root, - goal: _goal, - implementationRef: _implementationRef, - sourceCandidate: _sourceCandidate, - selectedPaths: _selectedPaths, - runId: _runId, - rationale: _rationale, - selectionMetadata: _selectionMetadata, - enabledEvaluationPhases: _enabledEvaluationPhases, - requiredEvaluationPhases: _requiredEvaluationPhases, - ...rest - } = options - return rest -} - -function normalizeEvaluationPhases( - phases: readonly RagKnowledgeImprovementPhase[], - field: string, -): KnowledgeEvaluationPhase[] { - const unique = [...new Set(phases)] - for (const phase of unique) { - if (phase === 'knowledge-acquisition' || phase === 'knowledge-update') { - throw new Error(`${field} cannot contain the selection-owned phase '${phase}'`) - } - } - return unique as KnowledgeEvaluationPhase[] -} - -function normalizeSelectedPaths( - paths: readonly string[], - changedPlan: readonly KnowledgeFileTransactionPlanEntry[], - scope: KnowledgeStateScope, -): string[] { - const available = new Set(changedPlan.map((entry) => entry.path)) - const selected: string[] = [] - const seen = new Set() - for (const input of paths) { - const path = assertKnowledgeMutationPath( - selectionPathSchema.parse(input), - normalizeKnowledgeStateScope(scope).pagesDirectory, - scope.researchState, - ) - if (!notDerivedPath({ path }, scope)) { - throw new Error(`derived knowledge path cannot be selected directly: ${path}`) - } - if (seen.has(path)) throw new Error(`selected knowledge path is repeated: ${path}`) - if (!available.has(path)) { - throw new Error(`selected knowledge path is not a changed source-candidate file: ${path}`) - } - seen.add(path) - selected.push(path) - } - return selected.sort((left, right) => left.localeCompare(right)) -} - -function planEntryChanged(entry: KnowledgeFileTransactionPlanEntry): boolean { - return ( - entry.beforeHash !== entry.afterHash || - (entry.beforeHash !== null && entry.afterHash !== null && entry.beforeMode !== entry.afterMode) - ) -} - -function notDerivedPath(entry: { path: string }, scope: KnowledgeStateScope): boolean { - return entry.path !== `${normalizeKnowledgeStateScope(scope).pagesDirectory}/index.md` -} - -async function selectionMutations( - sourceCandidateRoot: string, - entries: readonly KnowledgeFileTransactionPlanEntry[], -): Promise { - return Promise.all( - entries.map(async (entry) => { - if (entry.afterHash === null) return { path: entry.path, content: null } - const file = await readRegularFileWithinRoot(sourceCandidateRoot, entry.path) - const actualHash = createHash('sha256').update(file.bytes).digest('hex') - if (actualHash !== entry.afterHash || file.mode !== entry.afterMode) { - throw new Error(`source candidate changed before subset materialization: ${entry.path}`) - } - return { path: entry.path, content: file.bytes, mode: file.mode } - }), - ) -} - -function assertSelectionTransaction( - transaction: KnowledgeFileTransaction, - expectedPlanHash: string, -): void { - if ( - knowledgeFileTransactionPlanHash( - transaction.entries, - normalizePagesDirectory(transaction.pagesDirectory), - transaction.researchState, - ) !== expectedPlanHash - ) { - throw new Error('selected knowledge transaction does not match its approved path set') - } -} - -function assertExactSelectedChanges( - plan: readonly KnowledgeFileTransactionPlanEntry[], - selectedPaths: readonly string[], - scope: KnowledgeStateScope, -): void { - const actual = plan - .filter(planEntryChanged) - .filter((entry) => notDerivedPath(entry, scope)) - .map((entry) => entry.path) - .sort((left, right) => left.localeCompare(right)) - if (canonicalJson(actual) !== canonicalJson(selectedPaths)) { - throw new Error( - `selected knowledge candidate changed the wrong files: expected ${selectedPaths.join(', ') || '(none)'}, got ${actual.join(', ') || '(none)'}`, - ) - } -} - -async function persistSelectionReceipt( - root: string, - receipt: MeasuredKnowledgeSelectionReceipt, -): Promise { - const runDir = knowledgeImprovementRunDir(root, receipt.runId) - const relativePath = join( - 'candidates', - safePathSegmentSchema.parse(receipt.derivedCandidateId), - 'selection.json', - ).replace(/\\/g, '/') - try { - const existing = measuredSelectionReceiptSchema.parse( - JSON.parse((await readRegularFileWithinRoot(runDir, relativePath)).bytes.toString('utf8')), - ) - if (canonicalJson(existing) !== canonicalJson(receipt)) { - throw new Error('measured knowledge selection receipt conflicts with durable content') - } - return - } catch (error) { - if (!isMissingFile(error)) throw error - } - await writeJsonDurableWithinRoot(runDir, relativePath, receipt) -} - -function immutableJson(value: T): T { - if (value === null || typeof value !== 'object') return value - const children: readonly unknown[] = Array.isArray(value) - ? value - : Object.values(value as Record) - for (const child of children) immutableJson(child) - return Object.freeze(value) -} diff --git a/src/rag-eval.ts b/src/rag-eval.ts index da6ccd3..a12a7fa 100644 --- a/src/rag-eval.ts +++ b/src/rag-eval.ts @@ -1,4 +1,3 @@ -export { calibrateRagAnswerJudge, createRagAnswerQualityHook } from './rag-eval/calibration' export type { ExternalRagEvalScore, KnowledgeBaseQualityOptions, diff --git a/src/rag-eval/calibration.ts b/src/rag-eval/calibration.ts deleted file mode 100644 index e01a25d..0000000 --- a/src/rag-eval/calibration.ts +++ /dev/null @@ -1,87 +0,0 @@ -import { contentHash } from '@tangle-network/agent-eval' -import { assertImmutableRef } from '../immutable-ref' -import type { RagAnswerQualityResult, RagGapFinding } from '../rag-improvement-loop' -import type { - RagAnswerMetricSummary, - RagAnswerQualityHookOptions, - RagCalibrationOptions, - RagCalibrationResult, -} from './contracts' -import { aggregateRagAnswerMetrics, ragAnswerQualityJudge, scoreRagAnswerArtifact } from './scoring' - -export function createRagAnswerQualityHook( - options: RagAnswerQualityHookOptions, -): () => Promise { - assertImmutableRef(options.evaluatorRef, 'RAG answer evaluatorRef') - const finalScenarioIds = options.scenarios.map((scenario) => scenario.id) - if ( - finalScenarioIds.length < 2 || - new Set(finalScenarioIds).size !== finalScenarioIds.length || - finalScenarioIds.some((id) => !id.trim()) - ) { - throw new Error('RAG answer quality requires at least 2 unique final scenarios') - } - const datasetRef = `sha256:${contentHash(options.scenarios)}` - return async () => { - const summaries: RagAnswerMetricSummary[] = [] - const findings: RagGapFinding[] = [] - for (const scenario of options.scenarios) { - const initialArtifact = await options.run(scenario) - const external = await options.externalEvaluator?.({ scenario, artifact: initialArtifact }) - const artifact = external - ? { - ...initialArtifact, - externalScores: [ - ...(initialArtifact.externalScores ?? []), - ...(Array.isArray(external) ? external : [external]), - ], - } - : initialArtifact - const summary = scoreRagAnswerArtifact(artifact, scenario, { - thresholds: options.thresholds, - weights: options.weights, - }) - summaries.push(summary) - findings.push(...summary.findings) - } - const metrics = aggregateRagAnswerMetrics(summaries) - const cost = - typeof options.cost === 'function' ? await options.cost() : structuredClone(options.cost) - return { - passed: findings.length === 0, - metrics, - finalScenarioIds, - datasetRef, - evaluatorRef: options.evaluatorRef, - cost, - findings, - metadata: { scenarioCount: options.scenarios.length }, - } - } -} - -export async function calibrateRagAnswerJudge( - options: RagCalibrationOptions, -): Promise { - const judge = options.judge ?? ragAnswerQualityJudge() - const strong = await judge.score({ - artifact: options.strong, - scenario: options.scenario, - signal: options.signal ?? new AbortController().signal, - }) - const weak = await judge.score({ - artifact: options.weak, - scenario: options.scenario, - signal: options.signal ?? new AbortController().signal, - }) - const strongScore = strong.composite - const weakScore = weak.composite - const minStrongScore = options.minStrongScore ?? 0.7 - const maxWeakScore = options.maxWeakScore ?? 0.3 - return { - passed: strongScore >= minStrongScore && weakScore <= maxWeakScore, - strongScore, - weakScore, - gap: strongScore - weakScore, - } -} diff --git a/src/relation-graph.test.ts b/src/relation-graph.test.ts deleted file mode 100644 index 849e89b..0000000 --- a/src/relation-graph.test.ts +++ /dev/null @@ -1,259 +0,0 @@ -import { describe, expect, it } from 'vitest' -import { - buildKnowledgeRelationGraph, - isReachable, - KnowledgeRelationGraphError, - neighbors, - walk, -} from './relation-graph' -import { KnowledgeRelationGraphSchema } from './schemas' -import type { KnowledgeRelation, KnowledgeRelationNode } from './types' - -const nodes: KnowledgeRelationNode[] = [ - { id: 'run-1', kind: 'run', label: 'Run 1' }, - { id: 'run-2', kind: 'run', label: 'Run 2', metadata: { seed: 7, tags: ['baseline'] } }, - { id: 'run-3', kind: 'run' }, - { id: 'claim-a', kind: 'claim' }, - { id: 'model-x', kind: 'model', metadata: { provider: 'local' } }, -] - -const relations: KnowledgeRelation[] = [ - { sourceId: 'run-2', targetId: 'run-1', predicate: 'branched-from' }, - { sourceId: 'run-2', targetId: 'run-1', predicate: 'supersedes', weight: 1 }, - { sourceId: 'run-3', targetId: 'run-2', predicate: 'branched-from' }, - { sourceId: 'run-2', targetId: 'claim-a', predicate: 'cites-evidence', metadata: { at: 3 } }, - { sourceId: 'run-1', targetId: 'model-x', predicate: 'executed-with' }, - { sourceId: 'run-3', targetId: 'model-x', predicate: 'executed-with' }, -] - -function expectError(run: () => unknown, code: KnowledgeRelationGraphError['code']): void { - try { - run() - } catch (error) { - expect(error).toBeInstanceOf(KnowledgeRelationGraphError) - expect((error as KnowledgeRelationGraphError).code).toBe(code) - return - } - throw new Error(`expected KnowledgeRelationGraphError ${code}`) -} - -describe('buildKnowledgeRelationGraph', () => { - it('keeps one edge per (source, target, predicate) in first-seen order', () => { - const graph = buildKnowledgeRelationGraph({ nodes, relations }) - expect(graph.nodes).toEqual(nodes) - expect(graph.edges).toEqual(relations) - expect( - graph.edges.filter((edge) => edge.sourceId === 'run-2' && edge.targetId === 'run-1'), - ).toHaveLength(2) - }) - - it('accepts a repeated triple only when it is byte-identical', () => { - const repeated: KnowledgeRelation = { - sourceId: 'run-2', - targetId: 'claim-a', - predicate: 'cites-evidence', - metadata: { at: 3 }, - } - const graph = buildKnowledgeRelationGraph({ relations: [...relations, repeated] }) - expect(graph.edges).toEqual(relations) - - expectError( - () => - buildKnowledgeRelationGraph({ - relations: [...relations, { ...repeated, metadata: { at: 4 } }], - }), - 'duplicate-relation', - ) - expectError( - () => - buildKnowledgeRelationGraph({ - relations: [ - { sourceId: 'a', targetId: 'b', predicate: 'p', weight: 1 }, - { sourceId: 'a', targetId: 'b', predicate: 'p', weight: 2 }, - ], - }), - 'duplicate-relation', - ) - }) - - it('refuses an endpoint outside the declared nodes instead of adding a node', () => { - expectError( - () => - buildKnowledgeRelationGraph({ - nodes, - relations: [{ sourceId: 'run-1', targetId: 'worker-9', predicate: 'authored-by' }], - }), - 'undeclared-endpoint', - ) - }) - - it('carries edges only when no nodes are declared', () => { - const graph = buildKnowledgeRelationGraph({ relations }) - expect(graph.nodes).toEqual([]) - expect(neighbors(graph, 'run-1', { direction: 'in' }).map((n) => n.nodeId)).toEqual([ - 'run-2', - 'run-2', - ]) - }) - - it('refuses duplicate node ids, empty ids, and non-JSON metadata', () => { - expectError( - () => - buildKnowledgeRelationGraph({ - nodes: [ - { id: 'x', kind: 'run' }, - { id: 'x', kind: 'claim' }, - ], - relations: [], - }), - 'duplicate-node', - ) - expectError( - () => buildKnowledgeRelationGraph({ nodes: [{ id: '', kind: 'run' }], relations: [] }), - 'invalid-node', - ) - expectError( - () => buildKnowledgeRelationGraph({ nodes: [{ id: 'x', kind: '' }], relations: [] }), - 'invalid-node', - ) - expectError( - () => - buildKnowledgeRelationGraph({ - nodes: [{ id: 'x', kind: 'run', metadata: { at: Number.NaN } }], - relations: [], - }), - 'invalid-node', - ) - expectError( - () => - buildKnowledgeRelationGraph({ - relations: [{ sourceId: 'a', targetId: '', predicate: 'p' }], - }), - 'invalid-relation', - ) - expectError( - () => - buildKnowledgeRelationGraph({ - relations: [ - { sourceId: 'a', targetId: 'b', predicate: 'p', weight: Number.POSITIVE_INFINITY }, - ], - }), - 'invalid-relation', - ) - expectError( - () => - buildKnowledgeRelationGraph({ - relations: [ - { sourceId: 'a', targetId: 'b', predicate: 'p', metadata: { note: undefined } }, - ], - }), - 'invalid-relation', - ) - }) - - it('round-trips node and edge metadata through the schema', () => { - const graph = buildKnowledgeRelationGraph({ nodes, relations }) - const parsed = KnowledgeRelationGraphSchema.parse(JSON.parse(JSON.stringify(graph))) - expect(parsed).toEqual(graph) - expect(parsed.nodes[1]?.metadata).toEqual({ seed: 7, tags: ['baseline'] }) - expect(parsed.edges[3]?.metadata).toEqual({ at: 3 }) - expect(walk(parsed, 'run-3', { direction: 'out', predicate: 'branched-from' })).toEqual( - walk(graph, 'run-3', { direction: 'out', predicate: 'branched-from' }), - ) - }) -}) - -describe('relation graph queries', () => { - const graph = buildKnowledgeRelationGraph({ nodes, relations }) - - it('lists neighbors by predicate and direction', () => { - expect(neighbors(graph, 'run-2', { direction: 'out', predicate: 'branched-from' })).toEqual([ - { nodeId: 'run-1', relation: relations[0] }, - ]) - expect(neighbors(graph, 'run-1', { direction: 'in' }).map((n) => n.relation.predicate)).toEqual( - ['branched-from', 'supersedes'], - ) - expect(neighbors(graph, 'run-2', { direction: 'both' }).map((n) => n.nodeId)).toEqual([ - 'run-1', - 'run-1', - 'claim-a', - 'run-3', - ]) - expect(neighbors(graph, 'claim-a', { direction: 'out' })).toEqual([]) - }) - - it('walks ancestors over out edges and descendants over in edges', () => { - expect(walk(graph, 'run-3', { direction: 'out', predicate: 'branched-from' })).toEqual([ - { nodeId: 'run-2', depth: 1, from: 'run-3', relation: relations[2] }, - { nodeId: 'run-1', depth: 2, from: 'run-2', relation: relations[0] }, - ]) - expect( - walk(graph, 'run-1', { direction: 'in', predicate: 'branched-from' }).map((s) => [ - s.nodeId, - s.depth, - ]), - ).toEqual([ - ['run-2', 1], - ['run-3', 2], - ]) - expect( - walk(graph, 'run-3', { direction: 'out', predicate: 'branched-from', maxDepth: 1 }).map( - (s) => s.nodeId, - ), - ).toEqual(['run-2']) - expect(walk(graph, 'run-3', { direction: 'out', maxDepth: 0 })).toEqual([]) - }) - - it('reports each node once on a cycle and terminates', () => { - const cyclic = buildKnowledgeRelationGraph({ - relations: [ - { sourceId: 'a', targetId: 'b', predicate: 'next' }, - { sourceId: 'b', targetId: 'c', predicate: 'next' }, - { sourceId: 'c', targetId: 'a', predicate: 'next' }, - { sourceId: 'c', targetId: 'c', predicate: 'next' }, - ], - }) - expect(walk(cyclic, 'a', { direction: 'out' }).map((s) => s.nodeId)).toEqual(['b', 'c']) - expect(walk(cyclic, 'a', { direction: 'both' }).map((s) => s.nodeId)).toEqual(['b', 'c']) - expect(neighbors(cyclic, 'c', { direction: 'both' }).map((n) => n.nodeId)).toEqual([ - 'a', - 'c', - 'b', - ]) - expect(isReachable(cyclic, 'a', 'a', { direction: 'out' })).toBe(true) - expect(isReachable(cyclic, 'b', 'a', { direction: 'out' })).toBe(true) - }) - - it('answers reachability along the requested predicate and direction', () => { - expect( - isReachable(graph, 'run-3', 'run-1', { direction: 'out', predicate: 'branched-from' }), - ).toBe(true) - expect( - isReachable(graph, 'run-1', 'run-3', { direction: 'out', predicate: 'branched-from' }), - ).toBe(false) - expect( - isReachable(graph, 'run-1', 'run-3', { direction: 'in', predicate: 'branched-from' }), - ).toBe(true) - expect( - isReachable(graph, 'run-3', 'model-x', { direction: 'out', predicate: 'branched-from' }), - ).toBe(false) - expect(isReachable(graph, 'run-3', 'model-x', { direction: 'out' })).toBe(true) - expect(isReachable(graph, 'claim-a', 'model-x', { direction: 'both' })).toBe(true) - }) - - it('refuses an unknown node and a malformed query', () => { - expectError(() => neighbors(graph, 'ghost', { direction: 'out' }), 'unknown-node') - expectError(() => walk(graph, 'ghost', { direction: 'out' }), 'unknown-node') - expectError(() => isReachable(graph, 'run-1', 'ghost', { direction: 'out' }), 'unknown-node') - expectError( - () => neighbors(graph, 'run-1', { direction: 'sideways' as 'out' }), - 'invalid-query', - ) - expectError( - () => neighbors(graph, 'run-1', { direction: 'out', predicate: '' }), - 'invalid-query', - ) - expectError(() => walk(graph, 'run-1', { direction: 'out', maxDepth: -1 }), 'invalid-query') - expectError(() => walk(graph, 'run-1', { direction: 'out', maxDepth: 1.5 }), 'invalid-query') - }) -}) diff --git a/src/relation-graph.ts b/src/relation-graph.ts deleted file mode 100644 index 5f90218..0000000 --- a/src/relation-graph.ts +++ /dev/null @@ -1,376 +0,0 @@ -import { canonicalCandidateJson } from '@tangle-network/agent-interface' -import type { - KnowledgeId, - KnowledgeRelation, - KnowledgeRelationGraph, - KnowledgeRelationNode, -} from './types' - -export type KnowledgeRelationGraphErrorCode = - | 'invalid-node' - | 'duplicate-node' - | 'invalid-relation' - | 'duplicate-relation' - | 'undeclared-endpoint' - | 'unknown-node' - | 'invalid-query' - -/** A relation graph input or query that cannot be honored exactly. */ -export class KnowledgeRelationGraphError extends Error { - readonly code: KnowledgeRelationGraphErrorCode - - constructor(code: KnowledgeRelationGraphErrorCode, message: string) { - super(message) - this.name = 'KnowledgeRelationGraphError' - this.code = code - } -} - -export interface BuildKnowledgeRelationGraphInput { - /** - * Declared node inventory. When present, every relation endpoint must name - * one declared node; an undeclared endpoint is refused rather than added. - * When absent, the graph carries edges only and `nodes` is empty. - */ - nodes?: readonly KnowledgeRelationNode[] - relations: readonly KnowledgeRelation[] -} - -/** - * `out` follows `sourceId -> targetId`, `in` follows `targetId -> sourceId`, - * `both` follows a relation from either end. - */ -export type KnowledgeRelationDirection = 'out' | 'in' | 'both' - -export interface KnowledgeRelationQuery { - /** Restrict to relations with this predicate. Every predicate when absent. */ - predicate?: string - direction: KnowledgeRelationDirection -} - -export interface KnowledgeRelationWalkOptions extends KnowledgeRelationQuery { - /** Most relations between the start node and a reported node. Unbounded when absent. */ - maxDepth?: number -} - -export interface KnowledgeRelationNeighbor { - nodeId: KnowledgeId - relation: KnowledgeRelation -} - -export interface KnowledgeRelationWalkStep { - nodeId: KnowledgeId - /** Number of relations between the start node and this node. */ - depth: number - /** The node the search came from when it first reached this node. */ - from: KnowledgeId - relation: KnowledgeRelation -} - -/** - * Build a labeled multi-edge graph from caller relations. - * - * The graph keeps one edge per `(sourceId, targetId, predicate)`, in first-seen - * order. A repeated triple is accepted only when its weight and metadata are - * identical in canonical JSON; any other repeat is a conflict and is refused. - * Node and relation metadata must be finite, acyclic JSON so the graph - * round-trips through `KnowledgeRelationGraphSchema` unchanged. - */ -export function buildKnowledgeRelationGraph( - input: BuildKnowledgeRelationGraphInput, -): KnowledgeRelationGraph { - const declared = - input.nodes === undefined ? undefined : new Map() - const nodes: KnowledgeRelationNode[] = [] - if (declared !== undefined) { - for (const [position, node] of input.nodes!.entries()) { - const normalized = normalizeNode(node, position) - if (declared.has(normalized.id)) { - throw new KnowledgeRelationGraphError( - 'duplicate-node', - `node "${normalized.id}" is declared more than once`, - ) - } - declared.set(normalized.id, normalized) - nodes.push(normalized) - } - } - - const edges: KnowledgeRelation[] = [] - const canonicalByTriple = new Map() - for (const [position, relation] of input.relations.entries()) { - const normalized = normalizeRelation(relation, position) - if (declared !== undefined) { - for (const endpoint of [normalized.sourceId, normalized.targetId]) { - if (!declared.has(endpoint)) { - throw new KnowledgeRelationGraphError( - 'undeclared-endpoint', - `relation ${describeTriple(normalized)} names undeclared node "${endpoint}"`, - ) - } - } - } - const canonical = canonicalRelationJson(normalized) - const key = tripleKey(normalized) - const prior = canonicalByTriple.get(key) - if (prior !== undefined) { - if (prior === canonical) continue - throw new KnowledgeRelationGraphError( - 'duplicate-relation', - `relation ${describeTriple(normalized)} is repeated with different weight or metadata`, - ) - } - canonicalByTriple.set(key, canonical) - edges.push(normalized) - } - return { nodes, edges } -} - -/** Relations touching `id`, each with the node at the other end, in graph order. */ -export function neighbors( - graph: KnowledgeRelationGraph, - id: KnowledgeId, - query: KnowledgeRelationQuery, -): KnowledgeRelationNeighbor[] { - const index = indexOf(graph) - validateQuery(query) - assertKnownNode(index, id) - return neighborsOf(index, id, query) -} - -/** - * Breadth-first traversal from `id` over matching relations. Each reachable - * node is reported once, at the depth where the search first reached it; the - * start node is never reported, so a cycle back to it adds no step. - */ -export function walk( - graph: KnowledgeRelationGraph, - id: KnowledgeId, - options: KnowledgeRelationWalkOptions, -): KnowledgeRelationWalkStep[] { - const index = indexOf(graph) - validateQuery(options) - validateMaxDepth(options.maxDepth) - assertKnownNode(index, id) - const steps: KnowledgeRelationWalkStep[] = [] - for (const step of traverse(index, id, options, options.maxDepth)) steps.push(step) - return steps -} - -/** - * Whether `to` is `from` or lies on a path of matching relations out of `from`. - */ -export function isReachable( - graph: KnowledgeRelationGraph, - from: KnowledgeId, - to: KnowledgeId, - query: KnowledgeRelationQuery, -): boolean { - const index = indexOf(graph) - validateQuery(query) - assertKnownNode(index, from) - assertKnownNode(index, to) - if (from === to) return true - for (const step of traverse(index, from, query, undefined)) { - if (step.nodeId === to) return true - } - return false -} - -interface RelationIndex { - readonly outgoing: ReadonlyMap - readonly incoming: ReadonlyMap - readonly known: ReadonlySet -} - -// The query index is derived once per graph object. A graph is a value: build -// a new one instead of mutating `nodes` or `edges` after the first query. -const indexes = new WeakMap() - -function indexOf(graph: KnowledgeRelationGraph): RelationIndex { - const cached = indexes.get(graph) - if (cached !== undefined) return cached - const outgoing = new Map() - const incoming = new Map() - const known = new Set() - for (const node of graph.nodes) known.add(node.id) - for (const relation of graph.edges) { - known.add(relation.sourceId) - known.add(relation.targetId) - push(outgoing, relation.sourceId, relation) - push(incoming, relation.targetId, relation) - } - const index: RelationIndex = { outgoing, incoming, known } - indexes.set(graph, index) - return index -} - -function push( - map: Map, - key: KnowledgeId, - relation: KnowledgeRelation, -): void { - const list = map.get(key) - if (list === undefined) map.set(key, [relation]) - else list.push(relation) -} - -function neighborsOf( - index: RelationIndex, - id: KnowledgeId, - query: KnowledgeRelationQuery, -): KnowledgeRelationNeighbor[] { - const matches = (relation: KnowledgeRelation): boolean => - query.predicate === undefined || relation.predicate === query.predicate - const result: KnowledgeRelationNeighbor[] = [] - if (query.direction !== 'in') { - for (const relation of index.outgoing.get(id) ?? []) { - if (matches(relation)) result.push({ nodeId: relation.targetId, relation }) - } - } - if (query.direction !== 'out') { - for (const relation of index.incoming.get(id) ?? []) { - // A self-loop already appears in the outgoing pass when both ends are followed. - if (query.direction === 'both' && relation.sourceId === id) continue - if (matches(relation)) result.push({ nodeId: relation.sourceId, relation }) - } - } - return result -} - -function* traverse( - index: RelationIndex, - start: KnowledgeId, - query: KnowledgeRelationQuery, - maxDepth: number | undefined, -): Generator { - const visited = new Set([start]) - const queue: Array<{ nodeId: KnowledgeId; depth: number }> = [{ nodeId: start, depth: 0 }] - for (let head = 0; head < queue.length; head++) { - const current = queue[head]! - if (maxDepth !== undefined && current.depth >= maxDepth) continue - for (const neighbor of neighborsOf(index, current.nodeId, query)) { - if (visited.has(neighbor.nodeId)) continue - visited.add(neighbor.nodeId) - const depth = current.depth + 1 - yield { nodeId: neighbor.nodeId, depth, from: current.nodeId, relation: neighbor.relation } - queue.push({ nodeId: neighbor.nodeId, depth }) - } - } -} - -function assertKnownNode(index: RelationIndex, id: KnowledgeId): void { - if (!index.known.has(id)) { - throw new KnowledgeRelationGraphError('unknown-node', `node "${id}" is not in the graph`) - } -} - -function validateQuery(query: KnowledgeRelationQuery): void { - if (query.direction !== 'out' && query.direction !== 'in' && query.direction !== 'both') { - throw new KnowledgeRelationGraphError( - 'invalid-query', - `direction must be "out", "in", or "both", got ${JSON.stringify(query.direction)}`, - ) - } - if (query.predicate !== undefined && !isNonEmptyString(query.predicate)) { - throw new KnowledgeRelationGraphError('invalid-query', 'predicate must be a non-empty string') - } -} - -function validateMaxDepth(maxDepth: number | undefined): void { - if (maxDepth === undefined) return - if (!Number.isInteger(maxDepth) || maxDepth < 0) { - throw new KnowledgeRelationGraphError( - 'invalid-query', - `maxDepth must be a non-negative integer, got ${JSON.stringify(maxDepth)}`, - ) - } -} - -function normalizeNode(node: KnowledgeRelationNode, position: number): KnowledgeRelationNode { - if (!isNonEmptyString(node.id)) { - throw new KnowledgeRelationGraphError('invalid-node', `node at ${position} has no id`) - } - if (!isNonEmptyString(node.kind)) { - throw new KnowledgeRelationGraphError('invalid-node', `node "${node.id}" has no kind`) - } - if (node.label !== undefined && typeof node.label !== 'string') { - throw new KnowledgeRelationGraphError('invalid-node', `node "${node.id}" label is not a string`) - } - if (node.metadata !== undefined) - assertCanonicalMetadata(node.metadata, `node "${node.id}"`, 'invalid-node') - return { - id: node.id, - kind: node.kind, - ...(node.label !== undefined ? { label: node.label } : {}), - ...(node.metadata !== undefined ? { metadata: node.metadata } : {}), - } -} - -function normalizeRelation(relation: KnowledgeRelation, position: number): KnowledgeRelation { - for (const field of ['sourceId', 'targetId', 'predicate'] as const) { - if (!isNonEmptyString(relation[field])) { - throw new KnowledgeRelationGraphError( - 'invalid-relation', - `relation at ${position} has no ${field}`, - ) - } - } - if (relation.weight !== undefined && !Number.isFinite(relation.weight)) { - throw new KnowledgeRelationGraphError( - 'invalid-relation', - `relation ${describeTriple(relation)} weight must be a finite number`, - ) - } - return { - sourceId: relation.sourceId, - targetId: relation.targetId, - predicate: relation.predicate, - ...(relation.weight !== undefined ? { weight: relation.weight } : {}), - ...(relation.metadata !== undefined ? { metadata: relation.metadata } : {}), - } -} - -function canonicalRelationJson(relation: KnowledgeRelation): string { - if (relation.metadata !== undefined) { - assertCanonicalMetadata( - relation.metadata, - `relation ${describeTriple(relation)}`, - 'invalid-relation', - ) - } - return canonicalCandidateJson({ - ...(relation.weight !== undefined ? { weight: relation.weight } : {}), - ...(relation.metadata !== undefined ? { metadata: relation.metadata } : {}), - }) -} - -function assertCanonicalMetadata( - metadata: Record, - subject: string, - code: 'invalid-node' | 'invalid-relation', -): void { - if (metadata === null || typeof metadata !== 'object' || Array.isArray(metadata)) { - throw new KnowledgeRelationGraphError(code, `${subject} metadata must be an object`) - } - try { - canonicalCandidateJson(metadata) - } catch (error) { - throw new KnowledgeRelationGraphError( - code, - `${subject} metadata must be finite, acyclic JSON: ${error instanceof Error ? error.message : String(error)}`, - ) - } -} - -function tripleKey(relation: KnowledgeRelation): string { - return JSON.stringify([relation.sourceId, relation.targetId, relation.predicate]) -} - -function describeTriple(relation: KnowledgeRelation): string { - return `${relation.sourceId} -[${relation.predicate}]-> ${relation.targetId}` -} - -function isNonEmptyString(value: unknown): value is string { - return typeof value === 'string' && value.length > 0 -} diff --git a/src/schemas.ts b/src/schemas.ts index cf9c3e8..53bb8f9 100644 --- a/src/schemas.ts +++ b/src/schemas.ts @@ -84,18 +84,6 @@ export const KnowledgeRelationSchema = z.object({ metadata: z.record(z.string(), z.unknown()).optional(), }) -export const KnowledgeRelationNodeSchema = z.object({ - id: z.string(), - kind: z.string(), - label: z.string().optional(), - metadata: z.record(z.string(), z.unknown()).optional(), -}) - -export const KnowledgeRelationGraphSchema = z.object({ - nodes: z.array(KnowledgeRelationNodeSchema), - edges: z.array(KnowledgeRelationSchema), -}) - export const KnowledgeIndexSchema = z.object({ root: z.string(), generatedAt: z.string(), diff --git a/src/types.ts b/src/types.ts index 762b5da..ae2ad6b 100644 --- a/src/types.ts +++ b/src/types.ts @@ -57,22 +57,6 @@ export interface KnowledgeRelation { metadata?: Record } -/** A caller-declared vertex of a labeled relation graph. */ -export interface KnowledgeRelationNode { - id: KnowledgeId - /** Caller vocabulary, such as `run`, `claim`, `page`, or `model`. */ - kind: string - label?: string - metadata?: Record -} - -/** A labeled multi-edge graph with one edge per `(sourceId, targetId, predicate)`. */ -export interface KnowledgeRelationGraph { - /** Declared nodes in declaration order; empty when the graph was built from relations alone. */ - nodes: KnowledgeRelationNode[] - edges: KnowledgeRelation[] -} - export interface KnowledgeUnit { id: KnowledgeId title: string diff --git a/tests/kb-improvement/optimization.test.ts b/tests/kb-improvement/optimization.test.ts deleted file mode 100644 index f16f882..0000000 --- a/tests/kb-improvement/optimization.test.ts +++ /dev/null @@ -1,420 +0,0 @@ -import { createHash } from 'node:crypto' -import { mkdir, readFile, writeFile } from 'node:fs/promises' -import { dirname, join } from 'node:path' -import { - inMemoryCampaignStorage, - type OptimizationMethod, - type Scenario, -} from '@tangle-network/agent-eval/campaign' -import { describe, expect, it } from 'vitest' -import { - addSourceText, - applyKnowledgeWriteBlocks, - hashKnowledgeBase, - improveKnowledgeBase, - optimizeKnowledgeBasePolicy, - type RagAnswerEvalArtifact, - type RagAnswerEvalScenario, - scenarioContentFingerprint, -} from '../../src/index' -import { - mutableCandidateRoot, - passingMetric, - refundProposal, - refundSource, - refundSpec, - withKb, -} from '../support/kb-improvement' - -interface PolicyScenario extends Scenario { - kind: 'kb-policy-eval' - prompt: string -} - -interface PolicyArtifact { - score: number -} - -type Policy = { evidence: 'none' | 'required'; maxSources: number } - -function immutableRef(value: string): string { - return `sha256:${createHash('sha256').update(value).digest('hex')}` -} - -describe('optimizeKnowledgeBasePolicy', () => { - it.skipIf(process.platform !== 'linux')( - 'runs full RAG evaluation against the isolated candidate KB', - async () => { - await withKb(async (root) => { - const method: OptimizationMethod = { - name: 'fixture-candidate-rag-method', - async optimize(input) { - expect('testScenarios' in input).toBe(false) - return { - winnerSurface: '{"mode":"grounded"}', - cost: { - totalCostUsd: 0, - costProvenance: { kind: 'observed', usd: 0 }, - accountingComplete: true, - incompleteReasons: [], - }, - } - }, - } - const scenario = (id: string): RagAnswerEvalScenario => ({ - id, - kind: 'rag-answer-eval', - query: `${id} candidate policy`, - }) - const seenCandidateRoots = new Set() - - const result = await improveKnowledgeBase({ - root, - goal: 'Evaluate RAG against candidate knowledge', - implementationRef: immutableRef('candidate-rag-improvement'), - runId: 'candidate-rag-optimization', - async updateKnowledge({ candidateRoot }) { - const path = join(candidateRoot, 'knowledge', 'candidate-policy.md') - await mkdir(dirname(path), { recursive: true }) - await writeFile( - path, - [ - '---', - 'id: candidate-policy', - 'title: Candidate Policy', - '---', - '# Candidate Policy', - 'Candidate-only evidence.', - ].join('\n'), - ) - return { applied: true, summary: 'wrote candidate knowledge' } - }, - ragOptimization: { - executionRef: immutableRef('candidate-rag-execution'), - baseline: { mode: 'unsupported' }, - method, - trainScenarios: [scenario('candidate-rag-train')], - selectionScenarios: [scenario('candidate-rag-selection')], - finalScenarios: [scenario('candidate-rag-final-a'), scenario('candidate-rag-final-b')], - async run({ - config, - scenario: item, - baseHash, - baselineRoot, - candidateRoot, - candidateIndex, - }) { - seenCandidateRoots.add(candidateRoot) - expect(candidateRoot).not.toBe(root) - expect(baselineRoot).not.toBe(root) - expect(await hashKnowledgeBase(baselineRoot)).toBe(baseHash) - expect(candidateIndex.pages.map((page) => page.id)).toContain('candidate-policy') - const score = config.mode === 'grounded' ? 1 : 0 - return { - query: item.query, - answer: score ? 'Candidate-only evidence.' : 'Unsupported answer.', - contexts: [], - metadata: { score }, - } - }, - judges: [ - { - name: 'candidate-rag-quality', - dimensions: [{ key: 'quality', description: 'candidate RAG quality' }], - score: ({ artifact }) => { - const score = Number(artifact.metadata?.score ?? 0) - return { composite: score, dimensions: { quality: score } } - }, - }, - ], - storage: inMemoryCampaignStorage(), - expectUsage: 'off', - resamples: 200, - }, - requiredPhases: ['rag-optimization'], - evaluate: passingMetric, - }) - - expect(seenCandidateRoots.size).toBe(1) - expect(result.lifecycle?.optimization?.winner.value).toEqual({ mode: 'grounded' }) - expect(result.lifecycle?.optimization?.comparison.testScenarioIds).toEqual([ - 'candidate-rag-final-a', - 'candidate-rag-final-b', - ]) - }) - }, - ) - - it.skipIf(process.platform !== 'linux')( - 'uses development checks for retries and runs final evaluation once', - async () => { - await withKb(async (root) => { - let methodCalls = 0 - let promotionCalls = 0 - let developmentEvaluatorCalls = 0 - let finalEvaluatorCalls = 0 - const updatedIterations: number[] = [] - const finalDispatches: string[] = [] - const finalIds = ['a', 'b', 'c', 'd', 'e', 'f'].map((suffix) => `single-final-${suffix}`) - const scenario = (id: string): RagAnswerEvalScenario => ({ - id, - kind: 'rag-answer-eval', - query: id, - }) - const method: OptimizationMethod = { - name: 'single-final-method', - async optimize() { - methodCalls += 1 - return { - winnerSurface: '{"mode":"candidate"}', - cost: { - totalCostUsd: 0, - costProvenance: { kind: 'observed', usd: 0 }, - accountingComplete: true, - incompleteReasons: [], - }, - } - }, - } - - const result = await improveKnowledgeBase({ - root, - goal: 'Retry development candidates without reusing final cases', - implementationRef: immutableRef('single-final-improvement'), - runId: 'single-final-improvement', - maxCandidates: 3, - async updateKnowledge({ candidateRoot, iteration }) { - updatedIterations.push(iteration) - if (iteration === 1) - return { applied: false, summary: 'left required knowledge absent' } - const source = refundSource() - const added = await addSourceText(candidateRoot, source) - await applyKnowledgeWriteBlocks(candidateRoot, refundProposal(added.id)) - return { applied: true, summary: `updated candidate ${iteration}` } - }, - readinessSpecs: [refundSpec], - strict: true, - ragOptimization: { - executionRef: immutableRef('single-final-rag'), - baseline: { mode: 'baseline' }, - method, - trainScenarios: [scenario('single-final-train')], - selectionScenarios: [scenario('single-final-selection')], - finalScenarios: finalIds.map(scenario), - async run({ config, scenario: item }) { - if (finalIds.includes(item.id)) finalDispatches.push(item.id) - return { - query: item.query, - answer: 'answer', - contexts: [], - metadata: { score: config.mode === 'candidate' ? 1 : 0 }, - } - }, - judges: [ - { - name: 'single-final-quality', - dimensions: [{ key: 'quality', description: 'answer quality' }], - score: ({ artifact }) => { - const score = Number(artifact.metadata?.score ?? 0) - return { composite: score, dimensions: { quality: score } } - }, - }, - ], - storage: inMemoryCampaignStorage(), - expectUsage: 'off', - resamples: 200, - }, - requiredPhases: ['rag-optimization', 'promotion'], - evaluateDevelopment({ iteration }) { - developmentEvaluatorCalls += 1 - return { - score: iteration >= 2 ? 1 : 0, - passed: iteration >= 2, - provenance: { - evaluator: 'single-final-development', - version: '1', - method: 'deterministic', - }, - } - }, - evaluate() { - finalEvaluatorCalls += 1 - return passingMetric() - }, - decidePromotion() { - promotionCalls += 1 - return { promoted: false, reason: 'adversarial final rejection' } - }, - }) - - expect(updatedIterations).toEqual([1, 2]) - expect(methodCalls).toBe(1) - expect(promotionCalls).toBe(1) - expect(developmentEvaluatorCalls).toBe(2) - expect(finalEvaluatorCalls).toBe(1) - expect(new Set(finalDispatches)).toEqual(new Set(finalIds)) - expect(finalDispatches).toHaveLength(finalIds.length * 2) - expect(result.state.status).toBe('rejected') - expect(result.state.candidates).toHaveLength(2) - }) - }, - ) - - it.skipIf(process.platform !== 'linux')( - 'runs a complete method and applies only the exact winner to an isolated candidate', - async () => { - await withKb(async (root) => { - const methodInputs: string[][] = [] - const method: OptimizationMethod = { - name: 'fixture-kb-policy-method', - async optimize(input) { - methodInputs.push([ - ...input.trainScenarios.map((scenario) => scenario.id), - ...input.selectionScenarios.map((scenario) => scenario.id), - ]) - expect('testScenarios' in input).toBe(false) - return { - winnerSurface: '{"evidence":"required","maxSources":4}', - cost: { - totalCostUsd: 0, - costProvenance: { kind: 'observed', usd: 0 }, - accountingComplete: true, - incompleteReasons: [], - }, - } - }, - } - const scenario = (id: string): PolicyScenario => ({ - id, - kind: 'kb-policy-eval', - prompt: `${id} source-backed update`, - }) - - const result = await optimizeKnowledgeBasePolicy({ - root, - goal: 'Select a source-backed KB maintenance policy', - baselinePolicy: { evidence: 'none', maxSources: 1 }, - method, - trainScenarios: [scenario('policy-train')], - selectionScenarios: [scenario('policy-selection')], - finalScenarios: [scenario('policy-final-a'), scenario('policy-final-b')], - policyApplicationRef: immutableRef('write-maintenance-policy'), - dispatchCandidate: async ({ candidate }) => ({ - score: candidate.evidence === 'required' && candidate.maxSources >= 2 ? 1 : 0, - }), - judges: [ - { - name: 'policy-quality', - dimensions: [{ key: 'quality', description: 'policy satisfies evidence rules' }], - score: ({ artifact }) => ({ - composite: artifact.score, - dimensions: { quality: artifact.score }, - }), - }, - ], - scenarioFingerprint: scenarioContentFingerprint, - runDir: 'memory://kb-policy-optimization-test', - storage: inMemoryCampaignStorage(), - expectUsage: 'off', - resamples: 200, - candidate: { evaluate: passingMetric }, - async applyPolicy({ candidateRoot, policy, policySurfaceHash, optimizationMethod }) { - expect(policy).toEqual({ evidence: 'required', maxSources: 4 }) - expect(optimizationMethod).toBe('fixture-kb-policy-method') - const path = join(candidateRoot, 'knowledge', 'maintenance-policy.md') - await mkdir(dirname(path), { recursive: true }) - await writeFile( - path, - `# Maintenance Policy\n\n${policySurfaceHash}: require source evidence.\n`, - ) - return { applied: true, summary: 'wrote selected maintenance policy' } - }, - }) - - expect(methodInputs).toEqual([['policy-train', 'policy-selection']]) - expect(result.optimization.winner.value).toEqual({ - evidence: 'required', - maxSources: 4, - }) - expect(result.optimization.comparison.testScenarioIds).toEqual([ - 'policy-final-a', - 'policy-final-b', - ]) - expect(result.improvement.state.status).toBe('candidate-ready') - expect(result.improvement.promoted).toBe(false) - expect(result.improvement.lifecycle?.knowledgeUpdate?.metadata?.optimization).toEqual({ - method: 'fixture-kb-policy-method', - policySurfaceHash: result.optimization.winner.surfaceHash, - policyApplicationRef: immutableRef('write-maintenance-policy'), - }) - await expect( - readFile(join(root, 'knowledge', 'maintenance-policy.md'), 'utf8'), - ).rejects.toMatchObject({ code: 'ENOENT' }) - - const candidateRoot = mutableCandidateRoot(root, result.improvement) - await expect( - readFile(join(candidateRoot, 'knowledge', 'maintenance-policy.md'), 'utf8'), - ).resolves.toContain(result.optimization.winner.surfaceHash) - }) - }, - ) - - it('does not materialize a policy winner after the live knowledge base changes', async () => { - await withKb(async (root) => { - let applyCalls = 0 - const method: OptimizationMethod = { - name: 'concurrent-kb-change', - async optimize() { - await writeFile(join(root, 'knowledge', 'concurrent-change.md'), '# Concurrent change\n') - return { - winnerSurface: '{"evidence":"required","maxSources":2}', - cost: { - totalCostUsd: 0, - costProvenance: { kind: 'observed', usd: 0 }, - accountingComplete: true, - incompleteReasons: [], - }, - } - }, - } - const scenario = (id: string): PolicyScenario => ({ - id, - kind: 'kb-policy-eval', - prompt: `${id} policy`, - }) - - await expect( - optimizeKnowledgeBasePolicy({ - root, - goal: 'Reject a policy measured against changing knowledge', - baselinePolicy: { evidence: 'none', maxSources: 1 }, - method, - trainScenarios: [scenario('changing-train')], - selectionScenarios: [scenario('changing-selection')], - finalScenarios: [scenario('changing-final-a'), scenario('changing-final-b')], - policyApplicationRef: immutableRef('changing-policy'), - dispatchCandidate: async () => ({ score: 1 }), - judges: [ - { - name: 'changing-policy-quality', - dimensions: [{ key: 'quality', description: 'policy quality' }], - score: ({ artifact }) => ({ - composite: artifact.score, - dimensions: { quality: artifact.score }, - }), - }, - ], - runDir: 'memory://changing-kb-policy-test', - storage: inMemoryCampaignStorage(), - expectUsage: 'off', - resamples: 200, - async applyPolicy() { - applyCalls += 1 - return { applied: true, summary: 'must not run' } - }, - }), - ).rejects.toThrow('knowledge base changed during policy optimization') - expect(applyCalls).toBe(0) - }) - }) -}) diff --git a/tests/kb-improvement/selected-candidate.test.ts b/tests/kb-improvement/selected-candidate.test.ts deleted file mode 100644 index 40fd62a..0000000 --- a/tests/kb-improvement/selected-candidate.test.ts +++ /dev/null @@ -1,242 +0,0 @@ -import { readFile, writeFile } from 'node:fs/promises' -import { join } from 'node:path' -import { describe, expect, it } from 'vitest' -import { - hashKnowledgeBase, - improveSelectedKnowledgeCandidate, - knowledgeImprovementCandidateRef, - knowledgeImprovementRunDir, - promoteKnowledgeCandidate, - withKnowledgeImprovementComparison, -} from '../../src/index' -import { - improveTestKnowledgeBase as improveKnowledgeBase, - passingMetric, - TEST_KNOWLEDGE_IMPLEMENTATION_REF, - withKb, -} from '../support/kb-improvement' - -describe('improveSelectedKnowledgeCandidate', () => { - it('remeasures the exact selected subset before ordinary promotion', async () => { - await withKb(async (root) => { - await writeFile(join(root, 'knowledge', 'original.md'), '# Original\n') - const baseHash = await hashKnowledgeBase(root) - const source = await improveKnowledgeBase({ - root, - goal: 'Generate a broad candidate from which a measured subset will be selected', - runId: 'broad-source-candidate', - updateKnowledge: async ({ candidateRoot }) => { - await writeFile(join(candidateRoot, 'knowledge', 'keep.md'), '# Keep\n') - await writeFile(join(candidateRoot, 'knowledge', 'drop.md'), '# Drop\n') - await writeFile(join(candidateRoot, 'knowledge', 'original.md'), '# Changed\n') - return { applied: true, summary: 'created a three-file broad candidate' } - }, - evaluate: passingMetric, - }) - const sourceCandidate = knowledgeImprovementCandidateRef(source) - let evaluatedSelectedSnapshot = false - let diagnosisCalls = 0 - - const selected = await improveSelectedKnowledgeCandidate({ - root, - goal: 'Measure only the useful files from the broad candidate', - implementationRef: TEST_KNOWLEDGE_IMPLEMENTATION_REF, - sourceCandidate, - selectedPaths: ['knowledge/original.md', 'knowledge/keep.md'], - rationale: 'The dropped page is redundant with an existing source.', - selectionMetadata: { reviewer: 'test-reviewer', policyVersion: 1 }, - diagnose() { - diagnosisCalls += 1 - return [ - { - id: 'selected-gap', - kind: 'unknown', - severity: 'info', - message: 'Inspect selected pages.', - }, - ] - }, - async evaluate(input) { - expect(input.lifecycle?.findings.map((finding) => finding.id)).toEqual(['selected-gap']) - expect(input.lifecycle?.knowledgeUpdate?.applied).toBe(true) - expect(input.candidateRoot).not.toBe(input.baselineRoot) - await expect( - readFile(join(input.candidateRoot, 'knowledge', 'keep.md'), 'utf8'), - ).resolves.toBe('# Keep\n') - await expect( - readFile(join(input.candidateRoot, 'knowledge', 'original.md'), 'utf8'), - ).resolves.toBe('# Changed\n') - await expect( - readFile(join(input.candidateRoot, 'knowledge', 'drop.md'), 'utf8'), - ).rejects.toMatchObject({ code: 'ENOENT' }) - evaluatedSelectedSnapshot = true - return passingMetric() - }, - }) - const candidate = knowledgeImprovementCandidateRef(selected) - - expect(evaluatedSelectedSnapshot).toBe(true) - expect(diagnosisCalls).toBe(1) - expect(candidate.baseHash).toBe(baseHash) - expect(selected.selection).toMatchObject({ - kind: 'measured-knowledge-change-selection-receipt', - sourceCandidate, - selectedPaths: ['knowledge/keep.md', 'knowledge/original.md'], - selectedCandidateHash: candidate.candidateHash, - selectedPlanHash: candidate.promotionPlanHash, - selectedEvidenceHash: candidate.evidenceHash, - }) - - await withKnowledgeImprovementComparison({ root, candidate }, async (comparison) => { - await expect( - readFile(join(comparison.candidate.root, 'knowledge', 'keep.md'), 'utf8'), - ).resolves.toBe('# Keep\n') - await expect( - readFile(join(comparison.candidate.root, 'knowledge', 'drop.md'), 'utf8'), - ).rejects.toMatchObject({ code: 'ENOENT' }) - }) - - const promoted = await promoteKnowledgeCandidate({ root, candidate }) - expect(promoted).toMatchObject({ promoted: true, blocked: false }) - await expect(readFile(join(root, 'knowledge', 'keep.md'), 'utf8')).resolves.toBe('# Keep\n') - await expect(readFile(join(root, 'knowledge', 'original.md'), 'utf8')).resolves.toBe( - '# Changed\n', - ) - await expect(readFile(join(root, 'knowledge', 'drop.md'), 'utf8')).rejects.toMatchObject({ - code: 'ENOENT', - }) - - const receiptPath = join( - knowledgeImprovementRunDir(root, candidate.runId), - 'candidates', - candidate.candidateId, - 'selection.json', - ) - const storedReceipt = JSON.parse(await readFile(receiptPath, 'utf8')) - expect(storedReceipt).toEqual(selected.selection) - }) - }) - - it('makes a harmful subset fail its own evaluator instead of inheriting the whole-candidate pass', async () => { - await withKb(async (root) => { - const source = await improveKnowledgeBase({ - root, - goal: 'Create a whole candidate whose two pages work together', - runId: 'whole-pair-source', - updateKnowledge: async ({ candidateRoot }) => { - await writeFile(join(candidateRoot, 'knowledge', 'claim.md'), '# Claim\nUses support.\n') - await writeFile(join(candidateRoot, 'knowledge', 'support.md'), '# Support\nEvidence.\n') - return { applied: true, summary: 'created claim and support pages' } - }, - evaluate: passingMetric, - }) - const sourceCandidate = knowledgeImprovementCandidateRef(source) - - const selected = await improveSelectedKnowledgeCandidate({ - root, - goal: 'Test whether the claim page survives without its support page', - implementationRef: TEST_KNOWLEDGE_IMPLEMENTATION_REF, - sourceCandidate, - selectedPaths: ['knowledge/claim.md'], - async evaluate({ candidateRoot }) { - try { - await readFile(join(candidateRoot, 'knowledge', 'support.md'), 'utf8') - return passingMetric() - } catch { - return { - score: 0, - passed: false, - notes: 'claim is missing its supporting page', - provenance: { - evaluator: 'selected-candidate-dependency-check', - version: '1', - method: 'deterministic', - }, - } - } - }, - }) - - expect(selected.candidate).toMatchObject({ status: 'rejected' }) - expect(selected.evaluation).toMatchObject({ passed: false }) - expect(() => knowledgeImprovementCandidateRef(selected)).toThrow(/is not ready/) - await expect(readFile(join(root, 'knowledge', 'claim.md'), 'utf8')).rejects.toMatchObject({ - code: 'ENOENT', - }) - }) - }) - - it('rejects unknown, repeated, and generated paths before opening a derived run', async () => { - await withKb(async (root) => { - const source = await improveKnowledgeBase({ - root, - goal: 'Create one selectable page', - runId: 'selection-shape-source', - updateKnowledge: async ({ candidateRoot }) => { - await writeFile(join(candidateRoot, 'knowledge', 'page.md'), '# Page\n') - return { applied: true, summary: 'created one page' } - }, - evaluate: passingMetric, - }) - const sourceCandidate = knowledgeImprovementCandidateRef(source) - const common = { - root, - goal: 'Refuse a malformed selection', - implementationRef: TEST_KNOWLEDGE_IMPLEMENTATION_REF, - sourceCandidate, - evaluate: passingMetric, - } - - await expect( - improveSelectedKnowledgeCandidate({ ...common, selectedPaths: ['knowledge/missing.md'] }), - ).rejects.toThrow(/not a changed source-candidate file/) - await expect( - improveSelectedKnowledgeCandidate({ - ...common, - selectedPaths: ['knowledge/page.md', 'knowledge/page.md'], - }), - ).rejects.toThrow(/repeated/) - await expect( - improveSelectedKnowledgeCandidate({ ...common, selectedPaths: ['knowledge/index.md'] }), - ).rejects.toThrow(/derived knowledge path/) - }) - }) - - it('reopens the same measured selection without rerunning its evaluator', async () => { - await withKb(async (root) => { - const source = await improveKnowledgeBase({ - root, - goal: 'Create a resumable source candidate', - runId: 'selected-resume-source', - updateKnowledge: async ({ candidateRoot }) => { - await writeFile(join(candidateRoot, 'knowledge', 'selected.md'), '# Selected\n') - return { applied: true, summary: 'created selected page' } - }, - evaluate: passingMetric, - }) - let evaluationCalls = 0 - const options = { - root, - goal: 'Measure a stable selected candidate', - runId: 'selected-resume-derived', - implementationRef: TEST_KNOWLEDGE_IMPLEMENTATION_REF, - sourceCandidate: knowledgeImprovementCandidateRef(source), - selectedPaths: ['knowledge/selected.md'], - evaluate() { - evaluationCalls += 1 - return passingMetric() - }, - } - - const first = await improveSelectedKnowledgeCandidate(options) - expect(evaluationCalls).toBe(1) - const second = await improveSelectedKnowledgeCandidate(options) - - expect(evaluationCalls).toBe(1) - expect(second.selection).toEqual(first.selection) - expect(knowledgeImprovementCandidateRef(second)).toEqual( - knowledgeImprovementCandidateRef(first), - ) - }) - }) -}) diff --git a/tests/kb-improvement/state-scope.test.ts b/tests/kb-improvement/state-scope.test.ts index db18b94..7b8b972 100644 --- a/tests/kb-improvement/state-scope.test.ts +++ b/tests/kb-improvement/state-scope.test.ts @@ -6,19 +6,13 @@ import { createKnowledgeEvent, FileSystemKbStore, hashKnowledgeBase, - improveSelectedKnowledgeCandidate, knowledgeImprovementCandidateRef, promoteKnowledgeCandidate, restoreKnowledgeCandidateBaseline, withKnowledgeImprovementCandidate, withKnowledgeImprovementComparison, } from '../../src/index' -import { - improveTestKnowledgeBase, - passingMetric, - TEST_KNOWLEDGE_IMPLEMENTATION_REF, - withKb, -} from '../support/kb-improvement' +import { improveTestKnowledgeBase, passingMetric, withKb } from '../support/kb-improvement' const stateScope = { pagesDirectory: 'kb/pages', researchState: true } const ledgerPath = '.agent-knowledge/claim-ledgers/episode.json' @@ -173,37 +167,6 @@ describe('declared candidate state', () => { }, ) - it.skipIf(process.platform !== 'linux')( - 'remeasures selected custom pages and research records using the source scope', - async () => { - await withKb(async (root) => { - await prepare(root) - const result = await improveTestKnowledgeBase({ - root, - goal: 'Learn retry policy', - stateScope, - async updateKnowledge({ candidateRoot }) { - await recordResearch(candidateRoot, 2) - return { applied: true, summary: 'Second research round' } - }, - evaluate: passingMetric, - }) - const selected = await improveSelectedKnowledgeCandidate({ - root, - goal: 'Select the claim state', - implementationRef: TEST_KNOWLEDGE_IMPLEMENTATION_REF, - sourceCandidate: knowledgeImprovementCandidateRef(result), - selectedPaths: [ledgerPath], - evaluate: passingMetric, - }) - const reference = knowledgeImprovementCandidateRef(selected) - await promoteKnowledgeCandidate({ root, candidate: reference }) - expect(await readRounds(root)).toBe(2) - expect(await new FileSystemKbStore({ root }).listEvents()).toHaveLength(1) - }) - }, - ) - it('requires explicit research permission and rejects credentials and derived state in transaction paths', () => { expect(() => assertKnowledgeMutationPath(ledgerPath, 'kb/pages')).toThrow('unsupported') expect(assertKnowledgeMutationPath(ledgerPath, 'kb/pages', true)).toBe(ledgerPath) diff --git a/tests/rag-eval.test.ts b/tests/rag-eval.test.ts index 814abf0..9e1aede 100644 --- a/tests/rag-eval.test.ts +++ b/tests/rag-eval.test.ts @@ -1,7 +1,5 @@ import { describe, expect, it } from 'vitest' import { - calibrateRagAnswerJudge, - createRagAnswerQualityHook, type KnowledgeIndex, normalizeExternalRagScores, type RagAnswerEvalArtifact, @@ -72,19 +70,6 @@ describe('RAG answer evaluation', () => { expect(weak.findings.map((finding) => finding.kind)).toContain('citation-mismatch') }) - it('calibrates the metric on deliberately strong and weak examples', async () => { - const calibration = await calibrateRagAnswerJudge({ - scenario, - strong: strongArtifact, - weak: weakArtifact, - }) - - expect(calibration.passed).toBe(true) - expect(calibration.strongScore).toBeGreaterThanOrEqual(0.7) - expect(calibration.weakScore).toBeLessThanOrEqual(0.3) - expect(calibration.gap).toBeGreaterThan(0.5) - }) - it('normalizes external Ragas, DeepEval, TruLens, and RAGChecker scores', () => { const scores = normalizeExternalRagScores([ { provider: 'ragas', scores: { faithfulness: 0.91, answer_relevancy: 0.82 } }, @@ -127,26 +112,6 @@ describe('RAG answer evaluation', () => { }) }) - it('builds a lifecycle answer-quality hook over real answer cases', async () => { - const hook = createRagAnswerQualityHook({ - scenarios: [scenario, secondScenario], - evaluatorRef: `sha256:${'a'.repeat(64)}`, - cost: { totalCostUsd: 0, accountingComplete: true, incompleteReasons: [] }, - run: (item) => ({ ...strongArtifact, query: item.query }), - externalEvaluator: () => ({ - provider: 'trulens', - scores: { groundedness: 1, answer_relevance: 1, context_relevance: 1 }, - }), - }) - - const result = await hook() - expect(result.passed).toBe(true) - expect(result.metrics.composite).toBe(1) - expect(result.finalScenarioIds).toEqual(['refund-window', 'refund-window-paraphrase']) - expect(result.datasetRef).toMatch(/^sha256:[a-f0-9]{64}$/) - expect(result.metadata?.scenarioCount).toBe(2) - }) - it('returns an agent-eval judge for direct campaign wiring', async () => { const judge = ragAnswerQualityJudge() const result = await judge.score({