From 4a01a875d87bbfadf3d42e31b1580f412f8ecd04 Mon Sep 17 00:00:00 2001 From: Mao Nakamoto <41178744+maonakamoto@users.noreply.github.com> Date: Sat, 29 Aug 2026 13:35:53 +0200 Subject: [PATCH] refactor(agent): the grounding harness is imported, not mirrored FleetCrown half of the mirror retirement (fleet AI-engine plan, Phase 2). src/lib/agent/core was the CANONICAL copy of the harness, mirrored byte-for-byte into OrangeCat and policed by scripts/test/agent-core-drift.ts plus scripts/sync-agent-core.ts. Its own README called the duplication "deliberate and temporary" and named extraction to a package as the exit. ai-kit v0.6.2 is that exit; OrangeCat's twin PR (#846, merged) deletes the other side. Fourteen imports across nine files collapse from three deep paths to one subpath import. The sync script, the drift check and the npm script that ran it are deleted rather than kept green: with a single copy there is nothing left to drift, and a gate with no possible failure mode is noise wearing a uniform. ai-kit also moves v0.2.1 -> v0.6.2 (still the ai-ration name in the old lock entry); the three root imports this repo uses (chain, fair-share, limits) are unchanged surfaces across that jump, which typecheck confirms. The packaged copy differs from the deleted canonical by two mechanical deltas, both made where the code now lives: .js extensions on relative imports (pure ESM) and two null-guards under ai-kit's noUncheckedIndexedAccess. tsc clean; 124/124 unit test files pass (one fewer than before: the drift check no longer exists to run). Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01WqKqMnHQHSmkGFfc5t7Rxn --- package-lock.json | 41 +++- package.json | 3 +- scripts/sync-agent-core.ts | 41 ---- scripts/test/agent-core-drift.ts | 145 ------------- scripts/test/agent-grounding.ts | 6 +- scripts/test/agent-tool-loop.ts | 2 +- scripts/test/fact-budget.ts | 2 +- src/lib/agent/brief.ts | 2 +- src/lib/agent/context.ts | 4 +- src/lib/agent/core/README.md | 74 ------- src/lib/agent/core/contract.ts | 176 ---------------- src/lib/agent/core/facts.ts | 172 ---------------- src/lib/agent/core/verify.ts | 339 ------------------------------- src/lib/agent/fact-budget.ts | 2 +- src/lib/agent/loop.ts | 6 +- src/lib/agent/sources.ts | 2 +- src/lib/agent/tools/handlers.ts | 2 +- src/lib/agent/tools/registry.ts | 2 +- src/lib/loki-core.ts | 6 +- 19 files changed, 54 insertions(+), 973 deletions(-) delete mode 100644 scripts/sync-agent-core.ts delete mode 100644 scripts/test/agent-core-drift.ts delete mode 100644 src/lib/agent/core/README.md delete mode 100644 src/lib/agent/core/contract.ts delete mode 100644 src/lib/agent/core/facts.ts delete mode 100644 src/lib/agent/core/verify.ts diff --git a/package-lock.json b/package-lock.json index e7bb378d..fb8bec3f 100644 --- a/package-lock.json +++ b/package-lock.json @@ -25,7 +25,7 @@ "@xterm/addon-fit": "^0.11.0", "@xterm/addon-web-links": "^0.12.0", "@xterm/xterm": "^6.0.0", - "ai-kit": "github:bitbaum/ai-kit#v0.2.1", + "ai-kit": "github:bitbaum/ai-kit#v0.6.2", "bip-kit": "^0.1.0", "bs58check": "^4.0.0", "class-variance-authority": "^0.7.1", @@ -69,6 +69,9 @@ "tailwindcss": "^4", "tsx": "^4.23.12", "typescript": "^5" + }, + "engines": { + "node": ">=20" } }, "node_modules/@alloc/quick-lru": { @@ -6005,14 +6008,40 @@ "node": ">= 14" } }, - "node_modules/ai-kit": { - "name": "ai-ration", - "version": "0.2.1", - "resolved": "git+ssh://git@github.com/bitbaum/ai-kit.git#07c492ce27785d66d9623acf7b30874fb908da13", - "integrity": "sha512-FUhbTToHPXhJxiruyhr+1oyyYAKmqQ9yMONdD2CPL/o4/svhOuOV05M789Nfe2yhLgbWTjmh5Yc8k4Tu40umGw==", + "node_modules/ai-forms": { + "version": "0.1.2", + "resolved": "https://registry.npmjs.org/ai-forms/-/ai-forms-0.1.2.tgz", + "integrity": "sha512-agadca4pN0iGlZTlNy1Vo2SmnQB+0dNkrQSDE3wCJYeaVxKDs5KaxiwGj5+GL6mVYcI1xZ8pwnT9kL9WcxnBHA==", "license": "MIT", "engines": { "node": ">=18" + }, + "peerDependencies": { + "react": ">=18" + }, + "peerDependenciesMeta": { + "react": { + "optional": true + } + } + }, + "node_modules/ai-kit": { + "version": "0.6.2", + "resolved": "git+ssh://git@github.com/bitbaum/ai-kit.git#ace11f14d817079ae2bf4cdc010b6daa9faacbec", + "license": "MIT", + "dependencies": { + "ai-forms": "^0.1.2" + }, + "engines": { + "node": ">=20" + }, + "peerDependencies": { + "react": ">=18" + }, + "peerDependenciesMeta": { + "react": { + "optional": true + } } }, "node_modules/ajv": { diff --git a/package.json b/package.json index 77cdb54f..251f1376 100644 --- a/package.json +++ b/package.json @@ -93,7 +93,6 @@ "git:prune": "bash scripts/git/prune-merged.sh", "ship": "git push origin main", "prepare": "husky", - "sync:agent-core": "tsx scripts/sync-agent-core.ts", "probe:models": "tsx scripts/probe-models.ts", "check:models": "tsx scripts/check-model-ids.ts", "check:telemetry": "tsx scripts/check-telemetry.ts", @@ -118,7 +117,7 @@ "@xterm/addon-fit": "^0.11.0", "@xterm/addon-web-links": "^0.12.0", "@xterm/xterm": "^6.0.0", - "ai-kit": "github:bitbaum/ai-kit#v0.2.1", + "ai-kit": "github:bitbaum/ai-kit#v0.6.2", "bip-kit": "^0.1.0", "bs58check": "^4.0.0", "class-variance-authority": "^0.7.1", diff --git a/scripts/sync-agent-core.ts b/scripts/sync-agent-core.ts deleted file mode 100644 index fa70ec76..00000000 --- a/scripts/sync-agent-core.ts +++ /dev/null @@ -1,41 +0,0 @@ -/** - * Push the canonical agent/core into OrangeCat's mirror. - * Run: npx tsx scripts/sync-agent-core.ts (npm run sync:agent-core) - * - * FleetCrown owns the canonical copy; OrangeCat mirrors it byte-for-byte. See - * src/lib/agent/core/README.md for why the harness is duplicated rather than - * packaged, and scripts/test/agent-core-drift.ts for the check that makes the - * duplication safe. - * - * Deliberately a no-op when the sibling repo is absent — CI clones one repo at - * a time, and a sync script that fails there would block every unrelated build. - */ -import { readdirSync, readFileSync, writeFileSync, existsSync, mkdirSync } from "node:fs"; -import { join, dirname } from "node:path"; -import { fileURLToPath } from "node:url"; - -const HERE = dirname(fileURLToPath(import.meta.url)); -const SRC = join(HERE, "..", "src", "lib", "agent", "core"); -const DEST = process.env.ORANGECAT_DIR - ? join(process.env.ORANGECAT_DIR, "src", "services", "agent-core") - : join(HERE, "..", "..", "orangecat", "src", "services", "agent-core"); - -if (!existsSync(dirname(dirname(DEST)))) { - console.log(`↷ OrangeCat not found at ${DEST} — skipping mirror (set ORANGECAT_DIR to override)`); - process.exit(0); -} - -mkdirSync(DEST, { recursive: true }); - -let changed = 0; -for (const file of readdirSync(SRC).sort()) { - const body = readFileSync(join(SRC, file), "utf8"); - const target = join(DEST, file); - const current = existsSync(target) ? readFileSync(target, "utf8") : null; - if (current === body) continue; - writeFileSync(target, body); - console.log(` → ${file}`); - changed++; -} - -console.log(changed === 0 ? "✓ agent-core mirror already in sync" : `✓ mirrored ${changed} file(s) to ${DEST}`); diff --git a/scripts/test/agent-core-drift.ts b/scripts/test/agent-core-drift.ts deleted file mode 100644 index 79f7fb24..00000000 --- a/scripts/test/agent-core-drift.ts +++ /dev/null @@ -1,145 +0,0 @@ -/** - * Drift gate for the mirrored agent/core. - * Run: npx tsx scripts/test/agent-core-drift.ts - * - * The harness is duplicated into OrangeCat rather than packaged (see - * src/lib/agent/core/README.md). Duplication is only safe if divergence is - * impossible to commit accidentally — two copies of "what counts as grounded", - * quietly disagreeing, is a worse failure than the one the harness was built to - * fix, because it would make the two assistants wrong in different ways. - * - * So: SHA-256 per file, compared against the mirror. Any difference fails. - * - * SKIPS (exit 0) when OrangeCat is not checked out beside this repo, because CI - * clones one repo at a time. That means the gate is a LOCAL and pre-push - * guarantee, not a CI one — the honest boundary, stated rather than implied. - * The corresponding check on OrangeCat's side is what catches a mirror edited - * in isolation. - * - * READS THE MIRROR FROM OrangeCat's origin/main, not from its working tree. - * The working tree is checked out to whatever branch someone happens to be - * working on there, so a tree-based comparison answers a question nobody - * asked — "does my canonical match a sibling repo's in-progress feature - * branch?" — and goes red for reasons that have nothing to do with the commit - * being pushed. On 2026-08-25 the mirror was resynced and MERGED to OrangeCat - * main, and this gate still blocked every FleetCrown push on the machine, - * because the OrangeCat checkout sat on an unrelated branch cut before it. - * - * A gate that stays red about code that is fine is worse than no gate: the - * only way past it is --no-verify, which disables the checks that do work. - * origin/main is what "the mirror" actually means — the shared branch both - * repos deploy from — and it is also what the fix (`npm run sync:agent-core`, - * commit, merge) actually updates. - */ -import { createHash } from "node:crypto"; -import { execFileSync } from "node:child_process"; -import { readdirSync, readFileSync, existsSync } from "node:fs"; -import { join, dirname } from "node:path"; -import { fileURLToPath } from "node:url"; - -const HERE = dirname(fileURLToPath(import.meta.url)); -const SRC = join(HERE, "..", "..", "src", "lib", "agent", "core"); -const OC_REPO = process.env.ORANGECAT_DIR - ?? join(HERE, "..", "..", "..", "orangecat"); -const MIRROR_PATH = "src/services/agent-core"; -const REF = process.env.ORANGECAT_REF ?? "origin/main"; - - -/** - * Did THIS branch touch the canonical agent-core files? - * - * Drift can arise two ways: you changed canonical and did not sync, or somebody - * changed the mirror in the other repository. Only the first is your diff. The - * second is ambient — it was already true before you started, and blocking your - * push on it is how a gate teaches people to pass --no-verify. - * - * So: your change → fail. Somebody else's → warn and let the push through, with - * the fix printed. `desktop-release-drift` already works this way, which is why - * it is the one check in this suite that has never blocked an unrelated push. - */ -function branchTouchedCanonical(): boolean { - try { - // Resolve the default branch rather than assuming "main": three repos in - // this fleet use master and two have no origin/HEAD set. - let def = "main"; - try { - def = execFileSync("git", ["symbolic-ref", "--short", "refs/remotes/origin/HEAD"], { - encoding: "utf8", stdio: ["ignore", "pipe", "ignore"], - }).trim().replace(/^origin\//, "") || "main"; - } catch { /* fall through to main */ } - const base = execFileSync("git", ["merge-base", "HEAD", `origin/${def}`], { - encoding: "utf8", stdio: ["ignore", "pipe", "ignore"], - }).trim(); - const changed = execFileSync("git", ["diff", "--name-only", `${base}...HEAD`], { - encoding: "utf8", stdio: ["ignore", "pipe", "ignore"], - }); - return changed.split("\n").some(f => f.startsWith("src/lib/agent/core/")); - } catch { - // Cannot tell — assume it is yours. A gate that cannot establish innocence - // should not grant it. - return true; - } -} - -const sha = (s: string) => createHash("sha256").update(s).digest("hex").slice(0, 16); - -if (!existsSync(join(OC_REPO, ".git"))) { - console.log(`↷ agent-core drift: OrangeCat not checked out at ${OC_REPO} — skipped`); - process.exit(0); -} - -const git = (...args: string[]) => - execFileSync("git", ["-C", OC_REPO, ...args], { encoding: "utf8", stdio: ["ignore", "pipe", "ignore"] }); - -// A missing ref is an infrastructure fact (shallow clone, no fetch yet), not a -// verdict about the code — skip rather than block, same as a missing checkout. -let mirrorFiles: string[]; -try { - mirrorFiles = git("ls-tree", "--name-only", `${REF}:${MIRROR_PATH}`) - .split("\n") - .filter(Boolean) - .sort(); -} catch { - console.log(`↷ agent-core drift: ${REF} unavailable in ${OC_REPO} — skipped (run \`git fetch\` there)`); - process.exit(0); -} - -const srcFiles = readdirSync(SRC).sort(); -const problems: string[] = []; - -for (const f of srcFiles) { - if (!mirrorFiles.includes(f)) { - problems.push(`missing from mirror: ${f}`); - continue; - } - const a = sha(readFileSync(join(SRC, f), "utf8")); - const b = sha(git("show", `${REF}:${MIRROR_PATH}/${f}`)); - if (a !== b) problems.push(`content differs: ${f} (canonical ${a} vs mirror ${b})`); -} -for (const f of mirrorFiles) { - if (!srcFiles.includes(f)) problems.push(`extra file in mirror (not canonical): ${f}`); -} - -if (problems.length > 0) { - const yours = branchTouchedCanonical(); - if (!yours) { - console.warn("⚠ agent-core drift detected, but this branch did not touch src/lib/agent/core/ —"); - console.warn(" the mirror moved in OrangeCat, not here. Not blocking your push."); - for (const p of problems) console.warn(` ${p}`); - console.warn(" Fix separately: npm run sync:agent-core, then merge that in OrangeCat."); - process.exit(0); - } - - console.error("✗ agent-core drift detected:"); - for (const p of problems) console.error(` ${p}`); - console.error( - "\n Fix: edit the FleetCrown copy, run `npm run sync:agent-core`, then commit\n" + - ` and merge that change in OrangeCat — this compares against ${REF}, not a\n` + - " working tree, so an unmerged local sync will not clear it.", - ); - process.exit(1); -} - -console.log( - `✓ agent-core drift: ${srcFiles.length} file(s) identical between this repo and OrangeCat ${REF}`, -); diff --git a/scripts/test/agent-grounding.ts b/scripts/test/agent-grounding.ts index 3dcbe2b8..ffd46535 100644 --- a/scripts/test/agent-grounding.ts +++ b/scripts/test/agent-grounding.ts @@ -29,9 +29,9 @@ import { renderFacts, unrecordedFields, NOT_RECORDED, -} from "../../src/lib/agent/core/facts"; -import { buildContract, buildGroundedContext, renderDirectives, buildAssistantRules, NO_BASIS } from "../../src/lib/agent/core/contract"; -import { verifyAnswer, buildRepairPrompt } from "../../src/lib/agent/core/verify"; +} from "ai-kit/grounding"; +import { buildContract, buildGroundedContext, renderDirectives, buildAssistantRules, NO_BASIS } from "ai-kit/grounding"; +import { verifyAnswer, buildRepairPrompt } from "ai-kit/grounding"; // ── The real records, exactly as FleetCrown stores them ────────────────────── const FACTS = assignFactIds([ diff --git a/scripts/test/agent-tool-loop.ts b/scripts/test/agent-tool-loop.ts index 35613673..48aaa455 100644 --- a/scripts/test/agent-tool-loop.ts +++ b/scripts/test/agent-tool-loop.ts @@ -20,7 +20,7 @@ import { z } from "zod"; import { parseTextToolCalls, stripToolCallLines, type ModelTurn } from "../../src/lib/agent/llm"; import { defineTool, renderToolCatalog, toOpenAITools, type ToolRegistry } from "../../src/lib/agent/tools/registry"; import { runLokiTurn } from "../../src/lib/agent/loop"; -import { makeFact, assignFactIds } from "../../src/lib/agent/core/facts"; +import { makeFact, assignFactIds } from "ai-kit/grounding"; const NAMES = ["search_people", "list_projects", "propose_action"]; diff --git a/scripts/test/fact-budget.ts b/scripts/test/fact-budget.ts index 882f37e5..70e8000c 100644 --- a/scripts/test/fact-budget.ts +++ b/scripts/test/fact-budget.ts @@ -15,7 +15,7 @@ import { estimateTokens, omissionNotice, } from "@/lib/agent/fact-budget"; -import type { Fact } from "@/lib/agent/core/facts"; +import type { Fact } from "ai-kit/grounding"; function assert(condition: boolean, message: string): void { if (!condition) throw new Error(message); diff --git a/src/lib/agent/brief.ts b/src/lib/agent/brief.ts index cbd6b1c4..c8300d64 100644 --- a/src/lib/agent/brief.ts +++ b/src/lib/agent/brief.ts @@ -26,7 +26,7 @@ import { getStuckGoals, getGoalsDueSoon, listUpcomingCommitments } from "@/db/queries/today"; import { getEventsDueSoon } from "@/db/queries/events"; import { getTodayHabits } from "@/db/queries/habits"; -import type { Directive } from "@/lib/agent/core/contract"; +import type { Directive } from "ai-kit/grounding"; /** Days ahead treated as "imminent" for the day-planning brief. */ const IMMINENT_DAYS = 3; diff --git a/src/lib/agent/context.ts b/src/lib/agent/context.ts index e6f15906..1f15c9c8 100644 --- a/src/lib/agent/context.ts +++ b/src/lib/agent/context.ts @@ -26,8 +26,8 @@ * records answers about the wrong one. Cheap deterministic facts (people, * projects) are kept whole; retrieved documents are the elastic part. */ -import { assignFactIds, renderFacts, type Fact } from "@/lib/agent/core/facts"; -import { buildGroundedContext, type Directive } from "@/lib/agent/core/contract"; +import { assignFactIds, renderFacts, type Fact } from "ai-kit/grounding"; +import { buildGroundedContext, type Directive } from "ai-kit/grounding"; import { peopleFacts, projectFacts, diff --git a/src/lib/agent/core/README.md b/src/lib/agent/core/README.md deleted file mode 100644 index 1689bd4d..00000000 --- a/src/lib/agent/core/README.md +++ /dev/null @@ -1,74 +0,0 @@ -# agent/core — the grounding harness - -Pure, dependency-free TypeScript shared by **Loki** (FleetCrown) and **Cat** -(OrangeCat). No DB, no network, no framework, no imports outside this directory. -That constraint is what makes it mirrorable, and it is enforced by the drift -check — do not relax it. - -## What it is for - -Both assistants had the same class of failure: a model asked to fill a rigid -answer format against thin context invents the missing parts, and the invention -is indistinguishable from the truth because both arrive as confident prose. - -The harness makes unsupported claims **hard to express** rather than merely -discouraged: - -| Module | Mechanism | -|---|---| -| `facts.ts` | Records with a **declared field set**. Fields with no stored value render as an explicit ``, so absence is a stated negative rather than silence. Each record gets a citation id. | -| `contract.ts` | A rules block **generated from this turn's facts** — enumerating the legal citation ids and the concrete gaps — plus `Directive`, for answers the app computed in SQL and the model may only phrase. | -| `verify.ts` | A deterministic post-generation check. Flags citations that resolve to nothing, and proper nouns / numbers / paths with no source in the records or the user's message. No extra model call. | - -## Why absence must be explicit - -Loki once reported a contact as *"Ilya Druzhnikov (UZH)"*. The stored record had -no organisation field, and the string `UZH` appears nowhere in the operator's -data — it is the substring inside dr**UZH**nikov, surfaced by a keyword match and -then narrated as an affiliation. - -A field the model was never shown is easy to invent. A field it was shown as -`affiliation: ` is a specific negative it has to actively -contradict. That is the whole design. - -## Why the verifier is deterministic - -It runs on **every** turn, including free-tier ones on small models — which is -exactly where fabrication is most likely. A verifier that costs a frontier call -is one that gets disabled where it matters most. - -It works because fabrication is overwhelmingly *nominal*: models invent -organisations, titles, file paths, phone numbers and dates. Those are -mechanically recognisable and, if genuine, must appear in the retrieved records. - -## Mirroring — read before editing - -This directory is **duplicated verbatim** in two repos: - -``` -fleetcrown/src/lib/agent/core/ ← canonical -orangecat/src/services/agent-core/ ← mirror -``` - -`scripts/test/agent-core-drift.ts` in **both** repos compares SHA-256 per file -and fails CI on any difference. So: - -1. Edit the FleetCrown copy. -2. Run `npm run sync:agent-core` (FleetCrown) to push the mirror. -3. Commit both repos. - -This duplication is deliberate and temporary. The two apps have incompatible -data layers (Drizzle/Postgres vs Supabase), so a shared package was not worth -blocking on — but two silently-diverging copies of "what counts as grounded" -would be worse than either. The drift check buys SSOT-in-practice now; the exit -is extraction to `@fleet/agent-core` (the `@fleet/ai-forms` pattern), after -which both repos import instead of mirroring. - -## What belongs here vs in the app - -**Here:** anything that defines what grounding *means*. -**In the app:** anything that knows where data lives — the adapters that map -rows to `Fact`s (`fleetcrown/src/lib/agent/sources.ts`, -`orangecat/src/services/cat/sources.ts`) and the SQL behind `Directive`s. - -If you find yourself importing a DB client here, the code belongs in an adapter. diff --git a/src/lib/agent/core/contract.ts b/src/lib/agent/core/contract.ts deleted file mode 100644 index b277293a..00000000 --- a/src/lib/agent/core/contract.ts +++ /dev/null @@ -1,176 +0,0 @@ -/** - * The grounding contract — the rules block that ships with every turn's facts. - * MIRRORED MODULE (see core/README.md). - * - * Why this is generated rather than a hand-written constant: a standing prose - * rule ("only use provided context") is a weak signal that models trade away - * under format pressure. The failure that motivated this harness was exactly - * that — a prompt demanding "1 focus, 3 tasks, 1 person, under 150 words, no - * hedging" got four confidently-formatted answers, three of them invented, - * against a context block that already said "if a question falls outside this - * context, say so rather than guessing". - * - * The lesson: the model did not disobey a rule it forgot. It obeyed the - * STRONGER of two conflicting instructions — fill five slots — because nothing - * made the empty slot expressible. So this block does three things a static - * prompt cannot: - * - * 1. Names the exact citation handles that exist this turn, so "cite a fact" - * is a closed-set choice rather than free text. - * 2. Names the exact fields that are unrecorded THIS TURN, so the prohibition - * is concrete ("you have no affiliation for any person here") instead of - * abstract. - * 3. Supplies the escape hatch verbatim, so refusing a slot is a cheaper - * token path than inventing one. - */ -import { NOT_RECORDED, unrecordedFields, type Fact } from "./facts"; - -/** The exact string the model must emit when a slot cannot be filled. */ -export const NO_BASIS = "Not in your data."; - -/** - * Build the contract for a specific fact set. Empty fact sets get the strictest - * form — with nothing retrieved, EVERY answer must be a refusal, and saying so - * plainly beats hoping the model notices the context block is empty. - */ -export function buildContract(facts: Fact[], directives: Directive[] = []): string { - const ids = [ - ...facts.map((f) => `[${f.id}]`), - ...directives.map((_, i) => `[${directiveId(i)}]`), - ].join(" "); - const gaps = unrecordedFields(facts); - - const rules = [ - "## Grounding contract — this overrides every formatting instruction below", - "", - "You are answering from a fixed set of records. They are the ONLY things you know about the operator.", - "", - facts.length === 0 && directives.length === 0 - ? `1. NO records were retrieved for this turn. You therefore cannot answer any question about the operator's projects, people, goals, habits, commitments or events. Reply "${NO_BASIS}" and say what you would need.` - : `1. Every claim about the operator MUST cite a record id. Legal citations this turn, and no others: ${ids}`, - `2. A field shown as \`${NOT_RECORDED}\` means you DO NOT KNOW it. Never supply a value for it — not from the record's own wording, not from a name that looks like a place or an organisation, not from general knowledge about a similarly-named person. A surname is not an employer.`, - `3. If any part of the request has no supporting record, answer that part with exactly "${NO_BASIS}" and continue with the parts you can support. A requested format NEVER obliges you to invent an item. Returning three of five requested items, each cited, is a correct and complete answer.`, - "4. Do not describe a person's role, employer, seniority, or history unless a record field states it. Do not infer an organisation from a name.", - "5. You have not browsed the web this turn. If asked to research someone, say you cannot and report only what the records hold.", - "6. If you are correcting an earlier answer, the correction is subject to every rule above — cite the record, or say the record does not exist.", - ]; - - if (gaps.length > 0) { - rules.push( - "", - `Unrecorded in THIS turn's records — you have no value for any of these and must not state one: ${gaps.join(", ")}`, - ); - } - - return rules.join("\n"); -} - -/** - * The subset of the contract that needs no fact ids — for an assistant whose - * context is still prose (Cat) rather than typed records. - * - * Weaker than `buildContract` by construction: without ids there is nothing to - * cite, so rule 1 cannot exist and the verifier runs in entity-attribution - * mode. What survives is the part that stopped the worst failure — never state - * an attribute for someone in the user's data that their record does not carry, - * and never imply research you did not perform. - * - * This is a stepping stone, not the destination. It exists so a live product - * gets the protection now, without a same-day rewrite of its whole context - * layer; the destination is typed records here too. - */ -export function buildAssistantRules(opts: { subjectNoun: string }): string { - return [ - "## Grounding rules — these override formatting instructions", - "", - `1. Everything you state about the user's own ${opts.subjectNoun} must come from the context above. Do not add an organisation, role, employer, history, or relationship that the context does not state.`, - "2. Do not infer an affiliation from a name. A word inside someone's name is not their employer or their city.", - "3. You have not browsed the web in this turn. If asked to research a person or company, say you cannot, and report only what the context holds.", - `4. If part of the request has no support in the context, answer that part with exactly "${NO_BASIS}" and continue with the parts you can support. A requested format never obliges you to invent an item.`, - "5. General knowledge (how Bitcoin, Lightning, or a payment method works) is fine to use and is not covered by rules 1–2. The restriction is on facts about THIS user and the people and organisations in their data.", - "6. A correction is a claim too. If you are correcting yourself, it must be supported by the context or stated as unknown.", - ].join("\n"); -} - -/** - * A deterministic answer computed by the app, not the model. - * - * Some questions are not judgment calls at all. "Which goals are stuck at 0% - * for 30+ days", "what is due in the next 3 days", "which habit is at risk" - * are SQL predicates with exact answers, and asking a language model to derive - * them from injected prose is strictly worse than computing them: it can only - * introduce error. The model's job is to PHRASE the result, not to derive it. - * - * `answer` is empty when the query ran and found nothing — which is itself a - * real, citable answer ("nothing is due"), and crucially different from the - * query never having run. - */ -export type Directive = { - /** What was asked, in the app's words: "goals stuck 30+ days". */ - question: string; - /** Computed result lines. Empty array = ran, found nothing. */ - answer: string[]; - /** How it was computed, shown to the model so it can be honest about method. */ - method: string; -}; - -/** - * Citation handle for a computed answer, parallel to a Fact's [F1]. - * - * Directives used to be uncitable, and the contract demands a citation for - * every claim — so a model reporting a computed result had nothing legal to - * point at and wrote "[no record id]" into the user's answer. That is the - * harness leaking its own plumbing onto the screen. Give computed answers real - * ids and the sentence cites [D1] like anything else. - */ -export function directiveId(index: number): string { - return `D${index + 1}`; -} - -/** - * Render computed answers. These are stated as settled, because they are: the - * model must not re-derive, second-guess, or "improve" them, and an empty - * result must be reported as an empty result rather than backfilled from the - * fact set. - */ -export function renderDirectives(directives: Directive[]): string { - if (directives.length === 0) return ""; - const blocks = directives.map((d, i) => { - const body = - d.answer.length > 0 - ? d.answer.map((a) => ` - ${a}`).join("\n") - : " (none — the query ran and matched nothing)"; - return ` [${directiveId(i)}] ${d.question} [${d.method}]\n${body}`; - }); - return [ - "## Computed answers — already resolved, do not re-derive", - "These were computed directly from the database for this turn. They are exact.", - "Report them as given and cite their id, exactly as you would a record.", - "Where the result is empty, say so plainly — do not substitute a plausible item from the records.", - "", - ...blocks, - ].join("\n"); -} - -/** - * Assemble the full grounded context: contract, computed answers, then records. - * - * Order is deliberate and load-bearing. The contract comes FIRST so it frames - * everything read afterwards, and the records come LAST so they sit closest to - * the user's question — the position small models weight most heavily. - */ -export function buildGroundedContext(input: { - facts: Fact[]; - directives?: Directive[]; - renderedFacts: string; -}): string { - return [ - buildContract(input.facts, input.directives ?? []), - renderDirectives(input.directives ?? []), - input.facts.length > 0 - ? ["## Records", "", input.renderedFacts].join("\n") - : "## Records\n\n(none retrieved)", - ] - .filter(Boolean) - .join("\n\n---\n\n"); -} diff --git a/src/lib/agent/core/facts.ts b/src/lib/agent/core/facts.ts deleted file mode 100644 index c47e883d..00000000 --- a/src/lib/agent/core/facts.ts +++ /dev/null @@ -1,172 +0,0 @@ -/** - * Facts — the unit of grounded context. MIRRORED MODULE (see core/README.md). - * - * The problem this solves, concretely. Loki was asked who to contact and - * answered "Ilya Druzhnikov (UZH)". The stored record is: - * - * { displayName: "Ilya Druzhnikov", channels: { whatsapp: "+1650…" } } - * - * There is no org field, and the string "UZH" appears nowhere in the operator's - * data — it is the substring inside dr-UZH-nikov. A keyword match produced an - * affiliation out of a surname, and prose context gave the model no way to tell - * that "affiliation" was a field it had never been shown. - * - * The fix is representational, not a prompt instruction. A Fact is a RECORD with - * a DECLARED field set, and every declared field is rendered — including the ones - * with no value, which render as an explicit ``. A model that reads - * - * affiliation: - * - * is being told a specific negative, which is far harder to overwrite than the - * silence of a field that simply wasn't mentioned. Absence becomes evidence. - * - * Every fact also carries a short stable id ([F3]) so the answer can cite spans - * and `verify.ts` can check citations mechanically rather than by vibes. - * - * Pure: no DB, no network, no framework. Apps map their rows into Facts via - * their own adapters (FleetCrown: src/lib/agent/sources; OrangeCat: services/cat/sources). - */ - -/** A field that is declared for a record kind but has no stored value. */ -export const NOT_RECORDED = ""; - -/** - * One grounded record. `fields` must contain an entry for EVERY key in the - * kind's declared field list — `null` where nothing is stored. Builders should - * go through `makeFact`, which enforces that against the registry. - */ -export type Fact = { - /** Short citation handle, assigned by `assignFactIds` (F1, F2, …). */ - id: string; - /** Record kind — must be a key of the FACT_KINDS registry. */ - kind: string; - /** Human label for the record (a name, a title). Never invented. */ - subject: string; - /** Declared field → stored value, or null for "nothing stored". */ - fields: Record; - /** Where this came from, shown to the model: "people table", "goals table". */ - source: string; - /** - * Relevance score when the fact came from similarity search. Absent for facts - * fetched deterministically (a SQL filter) — those are not ranked, they are - * simply true, and the distinction matters to the reader. - */ - similarity?: number; -}; - -/** - * The declared field set per record kind — the SSOT for "what could be known - * about this kind of thing". Adding a field here makes it render as - * `` everywhere it is missing, which is the entire anti-invention - * mechanism: the model can only ever see fields we chose to declare. - * - * Deliberately includes fields we do NOT store (a person's `affiliation`, - * `role`, `employer`). That is not an oversight — those are exactly the - * attributes models invent, so naming them and marking them unrecorded is the - * point. Do not "clean up" this list by deleting the empty ones. - */ -export const FACT_KINDS: Record = { - person: ["name", "affiliation", "role", "how_we_met", "last_interaction", "notes", "channels"], - project: ["name", "status", "stack", "description", "latest_dev_log", "repo"], - goal: ["title", "project", "progress", "target_date", "last_updated"], - habit: ["title", "frequency", "current_streak", "last_checked"], - commitment: ["title", "due", "counterparty", "status"], - event: ["name", "type", "deadline", "url", "status"], - // Humans the operator delegates to, and the work handed to them. Separate - // from `person`/`commitment` because the questions are different: a crew - // member is asked what they are good FOR, an assignment is asked who has it - // and whether they said yes. - crew_member: ["name", "role", "skills", "engagement", "rate", "availability", "open_assignments"], - assignment: ["title", "assignee", "status", "due", "fee", "why"], - document: ["title", "source", "excerpt"], - pending_action: ["title", "type", "reasoning", "proposed_on", "id"], -}; - -/** Field list for a kind; unknown kinds fall back to whatever the fact carries. */ -export function declaredFields(kind: string, fallback: string[] = []): readonly string[] { - return FACT_KINDS[kind] ?? fallback; -} - -/** - * Build a Fact with every declared field present. Values not supplied become - * null (→ ``). Undeclared keys are DROPPED rather than passed - * through: if a field is worth showing the model it is worth declaring in - * FACT_KINDS, otherwise the registry stops describing what the model sees. - */ -export function makeFact(input: { - kind: string; - subject: string; - source: string; - values?: Record; - similarity?: number; -}): Fact { - const keys = declaredFields(input.kind, Object.keys(input.values ?? {})); - const fields: Record = {}; - for (const key of keys) { - const raw = input.values?.[key]; - const trimmed = typeof raw === "string" ? raw.trim() : raw; - fields[key] = trimmed ? String(trimmed) : null; - } - return { - id: "", - kind: input.kind, - subject: input.subject, - source: input.source, - fields, - ...(input.similarity !== undefined ? { similarity: input.similarity } : {}), - }; -} - -/** Stamp sequential citation ids. Call once, after assembling the final set. */ -export function assignFactIds(facts: Fact[]): Fact[] { - return facts.map((f, i) => ({ ...f, id: `F${i + 1}` })); -} - -/** Every citation handle in a fact set — the only legal citations in an answer. */ -export function factIds(facts: Fact[]): Set { - return new Set(facts.map((f) => f.id)); -} - -/** - * Render facts for the model. One block per record, every declared field on its - * own line, unrecorded fields stated explicitly. - * - * [F3] person — Elena Weber SINGA Switzerland (people table) - * name: Elena Weber SINGA Switzerland - * affiliation: - * role: - * channels: whatsapp +41774730093 - * - * The line-per-field shape matters for small models: a flat prose blob invites - * summarising (and summarising is where invention creeps in), whereas a field - * list invites lookup. Observed with 8B models — the same prompt over a blob - * hallucinates roles, over a field list it reports ``. - */ -export function renderFacts(facts: Fact[]): string { - if (facts.length === 0) return ""; - return facts - .map((f) => { - const head = `[${f.id}] ${f.kind} — ${f.subject} (${f.source})`; - const body = Object.entries(f.fields).map( - ([k, v]) => ` ${k}: ${v ?? NOT_RECORDED}`, - ); - return [head, ...body].join("\n"); - }) - .join("\n\n"); -} - -/** - * Which declared fields are unrecorded across the set, as - * `kind.field` keys. The contract block names these explicitly so the rule - * "do not state an affiliation" is anchored to a concrete gap in THIS turn's - * context rather than being a standing abstraction the model may ignore. - */ -export function unrecordedFields(facts: Fact[]): string[] { - const gaps = new Set(); - for (const f of facts) { - for (const [k, v] of Object.entries(f.fields)) { - if (v === null) gaps.add(`${f.kind}.${k}`); - } - } - return [...gaps].sort(); -} diff --git a/src/lib/agent/core/verify.ts b/src/lib/agent/core/verify.ts deleted file mode 100644 index 182c1751..00000000 --- a/src/lib/agent/core/verify.ts +++ /dev/null @@ -1,339 +0,0 @@ -/** - * Groundedness verifier — MIRRORED MODULE (see core/README.md). - * - * Runs on the generated answer and reports claims the fact set does not support. - * Deliberately deterministic: no second model call, no embedding round-trip, no - * added cost or latency. That is a requirement, not a shortcut — this must run - * on every turn including the free-tier ones, and a verifier that costs a - * frontier call is one that gets disabled exactly where it is needed most. - * - * The insight that makes a cheap check work: fabrication is overwhelmingly - * NOMINAL. Models invent organisations, titles, people, file paths, phone - * numbers and dates — tokens that are mechanically recognisable and that must, - * if genuine, have appeared in the retrieved records or in what the user said. - * Grammar and hedging are hard to check; proper nouns and digits are easy. - * - * Scored against the real failure this was built from, every fabricated claim - * is caught by the proper-noun or numeric rule: - * - * "Ilya Druzhnikov (UZH)" → UZH: novel acronym - * "Accelerator & Bridge Program Manager" → novel proper-noun run - * "University of Liechtenstein", "START Summit" → novel proper-noun runs - * "/opt/fleetcrown/runner/.env" → novel path - * - * while the true parts ("Elena Weber SINGA Switzerland", "+41774730093") appear - * verbatim in the records and pass clean. - */ -import { NOT_RECORDED, type Fact } from "./facts"; - -export type Violation = { - kind: "unknown-citation" | "novel-proper-noun" | "novel-number" | "novel-path" | "uncited-claim"; - /** The offending text. */ - text: string; - /** Why it is a problem, phrased for a repair prompt the model will read. */ - detail: string; -}; - -export type VerifyResult = { - ok: boolean; - violations: Violation[]; -}; - -/** - * Words that are capitalised for reasons other than being a proper noun, or - * that are part of this system's own vocabulary. Kept deliberately small — - * every entry is a hole in the check, so add only what demonstrably causes - * false positives, never to silence a true one. - */ -const COMMON = new Set( - [ - // Sentence/structural - "the", "a", "an", "and", "or", "but", "if", "then", "so", "because", "not", - "this", "that", "these", "those", "it", "its", "your", "you", "i", "we", - "there", "here", "what", "which", "who", "when", "where", "why", "how", - "no", "yes", "none", "nothing", "today", "tomorrow", "yesterday", "now", - "next", "last", "first", "one", "two", "three", "primary", "focus", "task", - "tasks", "outreach", "note", "notes", "summary", "status", "update", - // Days / months — real words, never evidence of a fabricated entity - "monday", "tuesday", "wednesday", "thursday", "friday", "saturday", "sunday", - "january", "february", "march", "april", "may", "june", "july", "august", - "september", "october", "november", "december", - // This system's own nouns - "loki", "cat", "fleetcrown", "orangecat", "not", "recorded", - ].map((w) => w.toLowerCase()), -); - -/** Normalise for containment tests: casefold, collapse punctuation and space. */ -function norm(s: string): string { - return s.toLowerCase().replace(/[^a-z0-9+]+/g, " ").replace(/\s+/g, " ").trim(); -} - -/** - * Everything the model was legitimately given this turn: record values, record - * subjects, and the user's own message (a name the user typed is fair to - * repeat). This is the corpus a claim must be traceable to. - */ -function buildEvidence(facts: Fact[], userMessage: string, extra: string[]): string { - const parts: string[] = [userMessage, ...extra]; - for (const f of facts) { - parts.push(f.subject, f.kind, f.source); - for (const v of Object.values(f.fields)) if (v) parts.push(v); - } - return norm(parts.join(" ")); -} - -/** - * Lowercase words that legitimately sit INSIDE a proper name and must not break - * it up: "University of Zurich", "Bank für Handel", "Institute for the Study of - * Complexity". Without these, the run splits at the connector and the check - * only ever sees the harmless halves ("University", "Zurich") while the actual - * fabricated entity slips through unnamed. - */ -const NAME_CONNECTORS = new Set(["of", "the", "for", "and", "de", "der", "des", "van", "von", "du", "da", "di", "für", "el", "al"]); - -/** - * Named-entity candidates: ALL-CAPS acronyms, capitalised words, and the - * multi-word runs they form (connectors allowed strictly between two - * capitalised tokens, never at an edge). - * - * Both the run AND its individual tokens are emitted, deliberately. The run - * catches composite inventions ("University of Zurich") that no single token - * reveals; the individual tokens catch an invented acronym sitting next to a - * real name ("Druzhnikov UZH"), where reporting only the run would name the - * real person in the violation and produce a repair prompt that deletes the - * true claim along with the false one. - * - * Sentence-initial single words are skipped — otherwise "Rotate the key" flags - * "Rotate". That costs a little recall at sentence starts and removes the - * dominant source of false positives; a fabricated name at a sentence start is - * still caught by its remaining tokens. - */ -function properNounRuns(text: string): string[] { - const out: string[] = []; - // Strip fenced and inline code — quoted identifiers are usually the user's - // own or a literal under discussion, not a claim about the world. - const prose = text.replace(/```[\s\S]*?```/g, " ").replace(/`[^`]*`/g, " "); - - for (const sentence of prose.split(/(?<=[.!?:\n])\s+/)) { - const tokens = sentence.match(/[A-Za-z][A-Za-z0-9&.'’-]*/g) ?? []; - let run: string[] = []; - - const flush = () => { - // Trim trailing connectors so "University of" never stands as a run. - while (run.length > 0 && NAME_CONNECTORS.has(run[run.length - 1].toLowerCase())) run.pop(); - if (run.length > 1) out.push(run.join(" ")); - run = []; - }; - - tokens.forEach((tok, i) => { - const bare = tok.replace(/[.'’-]+$/, ""); - const isAcronym = /^[A-Z]{2,}$/.test(bare); - const isCapitalised = /^[A-Z][a-z]/.test(bare); - const isConnector = NAME_CONNECTORS.has(bare.toLowerCase()); - - if (isAcronym || (isCapitalised && i > 0)) { - run.push(bare); - out.push(bare); // individually checkable - return; - } - // A connector only continues a run that has already started. - if (isConnector && run.length > 0) { - run.push(bare); - return; - } - flush(); - }); - flush(); - } - return out; -} - -/** Digit groups worth checking: phone numbers, years, percentages, counts ≥ 2 digits. */ -function numericClaims(text: string): string[] { - const prose = text.replace(/```[\s\S]*?```/g, " ").replace(/`[^`]*`/g, " "); - return (prose.match(/\+?\d[\d\s().-]{3,}\d|\b\d{2,}%?\b/g) ?? []).map((s) => s.trim()); -} - -/** - * File and path references — a favourite fabrication, and an unusually - * damaging one because naming a file implies the model READ it. - * - * Covers absolute paths (`/opt/fleetcrown/runner/.env`), relative paths - * (`data/contact-resolver.json`), and bare filenames with a data/config - * extension. The relative form matters: when challenged on the UZH claim, the - * model "corrected" itself by asserting what `data/contact-resolver.json` - * contained — a file it was never given. That reads as citing a source, which - * is precisely why an unverified correction is more corrosive than the - * original error: it spends the credibility the user was trying to restore. - */ -function pathClaims(text: string): string[] { - const patterns = [ - /(?:^|[\s("'`])(\/[A-Za-z0-9_.\-/]{4,})/g, // absolute - /(?:^|[\s("'`])([A-Za-z0-9_.-]+\/[A-Za-z0-9_.\-/]*[A-Za-z0-9_-]\.[a-z]{2,5})/g, // relative w/ extension - /(?:^|[\s("'`])([A-Za-z0-9_-]+\.(?:json|env|ya?ml|sql|toml|ini|conf|log))\b/g, // bare config filename - ]; - const out = new Set(); - for (const re of patterns) { - for (const m of text.matchAll(re)) if (m[1]) out.add(m[1]); - } - return [...out]; -} - -/** - * How strictly to treat unattested names — the one real difference between the - * two assistants that use this harness. - * - * `closed-world` (Loki): the assistant's entire job is reporting the operator's - * records, so ANY unattested proper noun is a fabrication. One operator, one - * data set, no legitimate outside knowledge in scope. - * - * `entity-attribution` (Cat): the assistant also answers general questions — - * how Lightning works, which payment rails exist in Switzerland — where naming - * Twint or Bitcoin is correct and required. Flagging those would make Cat - * useless. So the check narrows to what actually goes wrong: attributes - * attached to one of the USER'S OWN records. A sentence naming a known subject - * is checked; a sentence of general explanation is not. - * - * The narrower mode is genuinely weaker, and that is a real trade, not a - * loophole: Cat can still invent a fact about the wider world. It can no longer - * invent an employer for someone in your contacts, which is the failure that - * actually destroys trust in a personal assistant. - */ -export type VerifyMode = "closed-world" | "entity-attribution"; - -/** Does this sentence talk about one of the user's own records? */ -function mentionsSubject(sentence: string, subjects: string[]): boolean { - const s = norm(sentence); - return subjects.some((sub) => { - const n = norm(sub); - return n.length > 2 && s.includes(n); - }); -} - -/** - * Verify an answer against the facts it was supposed to come from. - * - * `extraEvidence` lets a caller admit sources outside the fact set — computed - * directive output, a tool result the model legitimately saw this turn. - * Anything not in facts, the user's message, or extraEvidence is unsupported - * by construction. - * - * `subjects` (entity-attribution mode) names the user's own records, so the - * check can tell "your contact Elena works at X" from "Lightning is instant". - */ -export function verifyAnswer(input: { - answer: string; - facts: Fact[]; - userMessage: string; - extraEvidence?: string[]; - mode?: VerifyMode; - subjects?: string[]; - /** - * Extra legal citation handles beyond the fact ids — the [D1] series for - * computed answers. Without these a model that correctly cites a computed - * result gets flagged for citing something "that does not exist", which - * would train the repair pass to delete true statements. - */ - extraCitationIds?: string[]; -}): VerifyResult { - const { answer, facts, userMessage } = input; - const mode: VerifyMode = input.mode ?? "closed-world"; - const subjects = input.subjects ?? facts.map((f) => f.subject); - const evidence = buildEvidence(facts, userMessage, input.extraEvidence ?? []); - const legalIds = new Set([ - ...facts.map((f) => f.id.toUpperCase()), - ...(input.extraCitationIds ?? []).map((id) => id.toUpperCase()), - ]); - const violations: Violation[] = []; - - /** - * In entity-attribution mode, only sentences about the user's own records are - * subject to the name check. Built once so the per-token loop stays cheap. - */ - const attributionScope = - mode === "entity-attribution" - ? answer - .split(/(?<=[.!?:\n])\s+/) - .filter((s) => mentionsSubject(s, subjects)) - .join(" ") - : answer; - - // 1. Citations must resolve. A citation to a record that does not exist is - // the strongest possible signal of fabrication — it invents its own proof. - for (const cite of answer.match(/\[[FD]\d+\]/g) ?? []) { - const id = cite.slice(1, -1).toUpperCase(); - if (!legalIds.has(id)) { - violations.push({ - kind: "unknown-citation", - text: cite, - detail: `${cite} is not a record in this turn's context. Cite only ids that were provided, or say there is no record.`, - }); - } - } - - // 2. Named entities must be traceable. This is the anti-"UZH" rule. - const seen = new Set(); - for (const run of properNounRuns(attributionScope)) { - const n = norm(run); - if (!n || seen.has(n)) continue; - seen.add(n); - // Single common words are noise; multi-word runs always checked. - const words = n.split(" "); - if (words.length === 1 && (COMMON.has(words[0]) || words[0].length < 2)) continue; - if (words.every((w) => COMMON.has(w))) continue; - if (evidence.includes(n)) continue; - // A multi-word run whose every word is individually attested is fine — - // it is a rephrasing, not a new entity. - if (words.length > 1 && words.every((w) => COMMON.has(w) || evidence.includes(w))) continue; - violations.push({ - kind: "novel-proper-noun", - text: run, - detail: `"${run}" does not appear in any record or in the operator's message. If it is an organisation, role, or place you associated with someone, the relevant field is ${NOT_RECORDED} — remove the claim.`, - }); - } - - // 3. Numbers must be traceable — invented phone numbers and dates read as - // authoritative precisely because they are specific. - for (const num of numericClaims(answer)) { - const n = norm(num); - if (!n || n.length < 2) continue; - if (evidence.includes(n)) continue; - // Compare digits-only too: "+41 77 473 00 93" vs stored "+41774730093". - const digits = num.replace(/\D/g, ""); - if (digits.length >= 4 && evidence.replace(/\D/g, "").includes(digits)) continue; - if (digits.length < 4) continue; // small counts ("3 tasks") are rhetorical - violations.push({ - kind: "novel-number", - text: num, - detail: `The number "${num}" is not in any record. Do not state contact details, dates, or metrics that were not provided.`, - }); - } - - // 4. Paths — "update the key in /opt/fleetcrown/runner/.env" was invented - // wholesale, and its specificity is what made it convincing. - for (const p of pathClaims(answer)) { - if (evidence.includes(norm(p))) continue; - violations.push({ - kind: "novel-path", - text: p, - detail: `The path "${p}" is not in any record. Do not state file locations you were not given.`, - }); - } - - return { ok: violations.length === 0, violations }; -} - -/** - * Turn violations into a repair instruction. One cheap retry with this appended - * fixes most turns, because the model is not being asked to know more — only to - * delete claims it cannot support. - */ -export function buildRepairPrompt(violations: Violation[], noBasisPhrase: string): string { - return [ - "Your previous answer contained claims not supported by the records. Rewrite it.", - "", - ...violations.map((v) => `- ${v.detail}`), - "", - `Remove every unsupported claim. Where removing one empties a requested item, write "${noBasisPhrase}" for that item instead of substituting something else. Keep everything that was supported, unchanged.`, - ].join("\n"); -} diff --git a/src/lib/agent/fact-budget.ts b/src/lib/agent/fact-budget.ts index 24daa8c5..0d8aa01b 100644 --- a/src/lib/agent/fact-budget.ts +++ b/src/lib/agent/fact-budget.ts @@ -1,4 +1,4 @@ -import type { Fact } from "@/lib/agent/core/facts"; +import type { Fact } from "ai-kit/grounding"; /** * Fact-budget policy for the tool loop — pure, and load-bearing. diff --git a/src/lib/agent/loop.ts b/src/lib/agent/loop.ts index 2c25b0da..822005c3 100644 --- a/src/lib/agent/loop.ts +++ b/src/lib/agent/loop.ts @@ -24,15 +24,15 @@ * observed failure mode, not defensive padding, and the loop always degrades to * "answer with what you have" rather than to an error. */ -import { assignFactIds, renderFacts, type Fact } from "@/lib/agent/core/facts"; +import { assignFactIds, renderFacts, type Fact } from "ai-kit/grounding"; import { trimFactsToBudget, mergeFactsWithCap, fitFactsToBudget, omissionNotice, } from "@/lib/agent/fact-budget"; -import { buildGroundedContext, buildContract, directiveId, NO_BASIS, type Directive } from "@/lib/agent/core/contract"; -import { verifyAnswer, buildRepairPrompt, type Violation } from "@/lib/agent/core/verify"; +import { buildGroundedContext, buildContract, directiveId, NO_BASIS, type Directive } from "ai-kit/grounding"; +import { verifyAnswer, buildRepairPrompt, type Violation } from "ai-kit/grounding"; import { callModelWithTools, type ChatMessage, type ToolCall } from "@/lib/agent/llm"; import { renderToolCatalog, toOpenAITools, toolNames, type ToolRegistry } from "@/lib/agent/tools/registry"; import { APP_NAME } from "@/config/brand"; diff --git a/src/lib/agent/sources.ts b/src/lib/agent/sources.ts index 0a0e31d4..c6a3055e 100644 --- a/src/lib/agent/sources.ts +++ b/src/lib/agent/sources.ts @@ -15,7 +15,7 @@ import { getPendingActions } from "@/db/queries/actions"; import { searchKnowledge, type KnowledgeHit } from "@/db/queries/knowledge-embeddings"; import { embeddingsEnabled } from "@/lib/rag/embeddings"; import { cleanDescription } from "@/lib/project-display"; -import { makeFact, type Fact } from "@/lib/agent/core/facts"; +import { makeFact, type Fact } from "ai-kit/grounding"; import { isChannelAttrKey, stripChannelPrefix } from "@/config/channels"; import { nameCandidates } from "@/lib/people-names"; import type { DevLogEntry, UserProject } from "@/db/schema/user-projects"; diff --git a/src/lib/agent/tools/handlers.ts b/src/lib/agent/tools/handlers.ts index a4bb4f25..3af321e2 100644 --- a/src/lib/agent/tools/handlers.ts +++ b/src/lib/agent/tools/handlers.ts @@ -22,7 +22,7 @@ import { createHumanTask, listOpenHumanTasks } from "@/db/queries/human-tasks"; import { HUMAN_TASK_STATUS_LABEL, TASK_ACTOR, formatFee } from "@/config/crew"; import { ACTION_TYPE, type ActionType } from "@/lib/constants/statuses"; import { askGatewayAgent, isGatewayConfigured } from "@/lib/openclaw-gateway"; -import { makeFact, type Fact } from "@/lib/agent/core/facts"; +import { makeFact, type Fact } from "ai-kit/grounding"; import { peopleFacts, projectFacts, documentFacts, pendingApprovalFacts, dateLabel } from "@/lib/agent/sources"; import { enrichReachPayload, reachFromPerson, resolvePersonToReach } from "@/lib/people-resolve"; import { defineTool } from "@/lib/agent/tools/registry"; diff --git a/src/lib/agent/tools/registry.ts b/src/lib/agent/tools/registry.ts index a19d5e46..b109ca32 100644 --- a/src/lib/agent/tools/registry.ts +++ b/src/lib/agent/tools/registry.ts @@ -27,7 +27,7 @@ * a refactor. */ import { z } from "zod"; -import type { Fact } from "@/lib/agent/core/facts"; +import type { Fact } from "ai-kit/grounding"; /** Read = lookup. Propose = enqueue a draft for the operator to approve. */ export type ToolKind = "read" | "propose"; diff --git a/src/lib/loki-core.ts b/src/lib/loki-core.ts index 3db34f8d..37664335 100644 --- a/src/lib/loki-core.ts +++ b/src/lib/loki-core.ts @@ -14,11 +14,11 @@ import { callGroqText, GROQ_FAST_MODEL } from "@/lib/groq"; import { getUserPreferences } from "@/db/queries/user-preferences"; import { buildGroundedTurn, directiveEvidence } from "@/lib/agent/context"; import { runLokiTurn } from "@/lib/agent/loop"; -import { verifyAnswer, buildRepairPrompt, type Violation } from "@/lib/agent/core/verify"; -import { NO_BASIS } from "@/lib/agent/core/contract"; +import { verifyAnswer, buildRepairPrompt, type Violation } from "ai-kit/grounding"; +import { NO_BASIS } from "ai-kit/grounding"; import { rateLimitMessage } from "@/lib/agent/groq-error"; import { checkAiBudget, recordAiSpend } from "@/lib/ai-budget/gate"; -import type { Fact } from "@/lib/agent/core/facts"; +import type { Fact } from "ai-kit/grounding"; import { APP_NAME } from "@/config/brand"; import { HTTP_TIMEOUT_LONG_MS } from "@/lib/constants/time";