Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -2,3 +2,4 @@ node_modules/
dist/
*.log
.DS_Store
node_modules
12 changes: 10 additions & 2 deletions package.json
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
{
"name": "ai-kit",
"version": "0.5.0",
"description": "One install for the AI layer of an app: which model to call, what to do when the vendor retires it, how to walk the fallback chain and know when none of it worked, how to read the three kinds of 429, a fair daily budget across users, and headless AI form filling.",
"version": "0.6.0",
"description": "One install for the AI layer of an app: which model to call, what to do when the vendor retires it, how to walk the fallback chain and know when none of it worked, how to read the three kinds of 429, a fair daily budget across users, headless AI form filling \u2014 and now the model registry (one SSOT for every callable id, with the paid/free boundary as a field) and the grounding harness (facts, contract, deterministic fabrication check).",
"license": "MIT",
"author": "Mao Nakamoto",
"homepage": "https://github.com/bitbaum/ai-kit#readme",
Expand Down Expand Up @@ -54,6 +54,14 @@
"./server": {
"types": "./dist/server.d.ts",
"default": "./dist/server.js"
},
"./registry": {
"types": "./dist/registry.d.ts",
"default": "./dist/registry.js"
},
"./grounding": {
"types": "./dist/grounding/index.d.ts",
"default": "./dist/grounding/index.js"
}
},
"scripts": {
Expand Down
176 changes: 176 additions & 0 deletions src/grounding/contract.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,176 @@
/**
* The grounding contract — the rules block that ships with every turn's facts.
* MIRRORED MODULE (see core/README.md).
*
* Why this is generated rather than a hand-written constant: a standing prose
* rule ("only use provided context") is a weak signal that models trade away
* under format pressure. The failure that motivated this harness was exactly
* that — a prompt demanding "1 focus, 3 tasks, 1 person, under 150 words, no
* hedging" got four confidently-formatted answers, three of them invented,
* against a context block that already said "if a question falls outside this
* context, say so rather than guessing".
*
* The lesson: the model did not disobey a rule it forgot. It obeyed the
* STRONGER of two conflicting instructions — fill five slots — because nothing
* made the empty slot expressible. So this block does three things a static
* prompt cannot:
*
* 1. Names the exact citation handles that exist this turn, so "cite a fact"
* is a closed-set choice rather than free text.
* 2. Names the exact fields that are unrecorded THIS TURN, so the prohibition
* is concrete ("you have no affiliation for any person here") instead of
* abstract.
* 3. Supplies the escape hatch verbatim, so refusing a slot is a cheaper
* token path than inventing one.
*/
import { NOT_RECORDED, unrecordedFields, type Fact } from "./facts.js";

/** The exact string the model must emit when a slot cannot be filled. */
export const NO_BASIS = "Not in your data.";

/**
* Build the contract for a specific fact set. Empty fact sets get the strictest
* form — with nothing retrieved, EVERY answer must be a refusal, and saying so
* plainly beats hoping the model notices the context block is empty.
*/
export function buildContract(facts: Fact[], directives: Directive[] = []): string {
const ids = [
...facts.map((f) => `[${f.id}]`),
...directives.map((_, i) => `[${directiveId(i)}]`),
].join(" ");
const gaps = unrecordedFields(facts);

const rules = [
"## Grounding contract — this overrides every formatting instruction below",
"",
"You are answering from a fixed set of records. They are the ONLY things you know about the operator.",
"",
facts.length === 0 && directives.length === 0
? `1. NO records were retrieved for this turn. You therefore cannot answer any question about the operator's projects, people, goals, habits, commitments or events. Reply "${NO_BASIS}" and say what you would need.`
: `1. Every claim about the operator MUST cite a record id. Legal citations this turn, and no others: ${ids}`,
`2. A field shown as \`${NOT_RECORDED}\` means you DO NOT KNOW it. Never supply a value for it — not from the record's own wording, not from a name that looks like a place or an organisation, not from general knowledge about a similarly-named person. A surname is not an employer.`,
`3. If any part of the request has no supporting record, answer that part with exactly "${NO_BASIS}" and continue with the parts you can support. A requested format NEVER obliges you to invent an item. Returning three of five requested items, each cited, is a correct and complete answer.`,
"4. Do not describe a person's role, employer, seniority, or history unless a record field states it. Do not infer an organisation from a name.",
"5. You have not browsed the web this turn. If asked to research someone, say you cannot and report only what the records hold.",
"6. If you are correcting an earlier answer, the correction is subject to every rule above — cite the record, or say the record does not exist.",
];

if (gaps.length > 0) {
rules.push(
"",
`Unrecorded in THIS turn's records — you have no value for any of these and must not state one: ${gaps.join(", ")}`,
);
}

return rules.join("\n");
}

/**
* The subset of the contract that needs no fact ids — for an assistant whose
* context is still prose (Cat) rather than typed records.
*
* Weaker than `buildContract` by construction: without ids there is nothing to
* cite, so rule 1 cannot exist and the verifier runs in entity-attribution
* mode. What survives is the part that stopped the worst failure — never state
* an attribute for someone in the user's data that their record does not carry,
* and never imply research you did not perform.
*
* This is a stepping stone, not the destination. It exists so a live product
* gets the protection now, without a same-day rewrite of its whole context
* layer; the destination is typed records here too.
*/
export function buildAssistantRules(opts: { subjectNoun: string }): string {
return [
"## Grounding rules — these override formatting instructions",
"",
`1. Everything you state about the user's own ${opts.subjectNoun} must come from the context above. Do not add an organisation, role, employer, history, or relationship that the context does not state.`,
"2. Do not infer an affiliation from a name. A word inside someone's name is not their employer or their city.",
"3. You have not browsed the web in this turn. If asked to research a person or company, say you cannot, and report only what the context holds.",
`4. If part of the request has no support in the context, answer that part with exactly "${NO_BASIS}" and continue with the parts you can support. A requested format never obliges you to invent an item.`,
"5. General knowledge (how Bitcoin, Lightning, or a payment method works) is fine to use and is not covered by rules 1–2. The restriction is on facts about THIS user and the people and organisations in their data.",
"6. A correction is a claim too. If you are correcting yourself, it must be supported by the context or stated as unknown.",
].join("\n");
}

/**
* A deterministic answer computed by the app, not the model.
*
* Some questions are not judgment calls at all. "Which goals are stuck at 0%
* for 30+ days", "what is due in the next 3 days", "which habit is at risk"
* are SQL predicates with exact answers, and asking a language model to derive
* them from injected prose is strictly worse than computing them: it can only
* introduce error. The model's job is to PHRASE the result, not to derive it.
*
* `answer` is empty when the query ran and found nothing — which is itself a
* real, citable answer ("nothing is due"), and crucially different from the
* query never having run.
*/
export type Directive = {
/** What was asked, in the app's words: "goals stuck 30+ days". */
question: string;
/** Computed result lines. Empty array = ran, found nothing. */
answer: string[];
/** How it was computed, shown to the model so it can be honest about method. */
method: string;
};

/**
* Citation handle for a computed answer, parallel to a Fact's [F1].
*
* Directives used to be uncitable, and the contract demands a citation for
* every claim — so a model reporting a computed result had nothing legal to
* point at and wrote "[no record id]" into the user's answer. That is the
* harness leaking its own plumbing onto the screen. Give computed answers real
* ids and the sentence cites [D1] like anything else.
*/
export function directiveId(index: number): string {
return `D${index + 1}`;
}

/**
* Render computed answers. These are stated as settled, because they are: the
* model must not re-derive, second-guess, or "improve" them, and an empty
* result must be reported as an empty result rather than backfilled from the
* fact set.
*/
export function renderDirectives(directives: Directive[]): string {
if (directives.length === 0) return "";
const blocks = directives.map((d, i) => {
const body =
d.answer.length > 0
? d.answer.map((a) => ` - ${a}`).join("\n")
: " (none — the query ran and matched nothing)";
return ` [${directiveId(i)}] ${d.question} [${d.method}]\n${body}`;
});
return [
"## Computed answers — already resolved, do not re-derive",
"These were computed directly from the database for this turn. They are exact.",
"Report them as given and cite their id, exactly as you would a record.",
"Where the result is empty, say so plainly — do not substitute a plausible item from the records.",
"",
...blocks,
].join("\n");
}

/**
* Assemble the full grounded context: contract, computed answers, then records.
*
* Order is deliberate and load-bearing. The contract comes FIRST so it frames
* everything read afterwards, and the records come LAST so they sit closest to
* the user's question — the position small models weight most heavily.
*/
export function buildGroundedContext(input: {
facts: Fact[];
directives?: Directive[];
renderedFacts: string;
}): string {
return [
buildContract(input.facts, input.directives ?? []),
renderDirectives(input.directives ?? []),
input.facts.length > 0
? ["## Records", "", input.renderedFacts].join("\n")
: "## Records\n\n(none retrieved)",
]
.filter(Boolean)
.join("\n\n---\n\n");
}
172 changes: 172 additions & 0 deletions src/grounding/facts.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,172 @@
/**
* Facts — the unit of grounded context. MIRRORED MODULE (see core/README.md).
*
* The problem this solves, concretely. Loki was asked who to contact and
* answered "Ilya Druzhnikov (UZH)". The stored record is:
*
* { displayName: "Ilya Druzhnikov", channels: { whatsapp: "+1650…" } }
*
* There is no org field, and the string "UZH" appears nowhere in the operator's
* data — it is the substring inside dr-UZH-nikov. A keyword match produced an
* affiliation out of a surname, and prose context gave the model no way to tell
* that "affiliation" was a field it had never been shown.
*
* The fix is representational, not a prompt instruction. A Fact is a RECORD with
* a DECLARED field set, and every declared field is rendered — including the ones
* with no value, which render as an explicit `<not recorded>`. A model that reads
*
* affiliation: <not recorded>
*
* is being told a specific negative, which is far harder to overwrite than the
* silence of a field that simply wasn't mentioned. Absence becomes evidence.
*
* Every fact also carries a short stable id ([F3]) so the answer can cite spans
* and `verify.ts` can check citations mechanically rather than by vibes.
*
* Pure: no DB, no network, no framework. Apps map their rows into Facts via
* their own adapters (FleetCrown: src/lib/agent/sources; OrangeCat: services/cat/sources).
*/

/** A field that is declared for a record kind but has no stored value. */
export const NOT_RECORDED = "<not recorded>";

/**
* One grounded record. `fields` must contain an entry for EVERY key in the
* kind's declared field list — `null` where nothing is stored. Builders should
* go through `makeFact`, which enforces that against the registry.
*/
export type Fact = {
/** Short citation handle, assigned by `assignFactIds` (F1, F2, …). */
id: string;
/** Record kind — must be a key of the FACT_KINDS registry. */
kind: string;
/** Human label for the record (a name, a title). Never invented. */
subject: string;
/** Declared field → stored value, or null for "nothing stored". */
fields: Record<string, string | null>;
/** Where this came from, shown to the model: "people table", "goals table". */
source: string;
/**
* Relevance score when the fact came from similarity search. Absent for facts
* fetched deterministically (a SQL filter) — those are not ranked, they are
* simply true, and the distinction matters to the reader.
*/
similarity?: number;
};

/**
* The declared field set per record kind — the SSOT for "what could be known
* about this kind of thing". Adding a field here makes it render as
* `<not recorded>` everywhere it is missing, which is the entire anti-invention
* mechanism: the model can only ever see fields we chose to declare.
*
* Deliberately includes fields we do NOT store (a person's `affiliation`,
* `role`, `employer`). That is not an oversight — those are exactly the
* attributes models invent, so naming them and marking them unrecorded is the
* point. Do not "clean up" this list by deleting the empty ones.
*/
export const FACT_KINDS: Record<string, readonly string[]> = {
person: ["name", "affiliation", "role", "how_we_met", "last_interaction", "notes", "channels"],
project: ["name", "status", "stack", "description", "latest_dev_log", "repo"],
goal: ["title", "project", "progress", "target_date", "last_updated"],
habit: ["title", "frequency", "current_streak", "last_checked"],
commitment: ["title", "due", "counterparty", "status"],
event: ["name", "type", "deadline", "url", "status"],
// Humans the operator delegates to, and the work handed to them. Separate
// from `person`/`commitment` because the questions are different: a crew
// member is asked what they are good FOR, an assignment is asked who has it
// and whether they said yes.
crew_member: ["name", "role", "skills", "engagement", "rate", "availability", "open_assignments"],
assignment: ["title", "assignee", "status", "due", "fee", "why"],
document: ["title", "source", "excerpt"],
pending_action: ["title", "type", "reasoning", "proposed_on", "id"],
};

/** Field list for a kind; unknown kinds fall back to whatever the fact carries. */
export function declaredFields(kind: string, fallback: string[] = []): readonly string[] {
return FACT_KINDS[kind] ?? fallback;
}

/**
* Build a Fact with every declared field present. Values not supplied become
* null (→ `<not recorded>`). Undeclared keys are DROPPED rather than passed
* through: if a field is worth showing the model it is worth declaring in
* FACT_KINDS, otherwise the registry stops describing what the model sees.
*/
export function makeFact(input: {
kind: string;
subject: string;
source: string;
values?: Record<string, string | null | undefined>;
similarity?: number;
}): Fact {
const keys = declaredFields(input.kind, Object.keys(input.values ?? {}));
const fields: Record<string, string | null> = {};
for (const key of keys) {
const raw = input.values?.[key];
const trimmed = typeof raw === "string" ? raw.trim() : raw;
fields[key] = trimmed ? String(trimmed) : null;
}
return {
id: "",
kind: input.kind,
subject: input.subject,
source: input.source,
fields,
...(input.similarity !== undefined ? { similarity: input.similarity } : {}),
};
}

/** Stamp sequential citation ids. Call once, after assembling the final set. */
export function assignFactIds(facts: Fact[]): Fact[] {
return facts.map((f, i) => ({ ...f, id: `F${i + 1}` }));
}

/** Every citation handle in a fact set — the only legal citations in an answer. */
export function factIds(facts: Fact[]): Set<string> {
return new Set(facts.map((f) => f.id));
}

/**
* Render facts for the model. One block per record, every declared field on its
* own line, unrecorded fields stated explicitly.
*
* [F3] person — Elena Weber SINGA Switzerland (people table)
* name: Elena Weber SINGA Switzerland
* affiliation: <not recorded>
* role: <not recorded>
* channels: whatsapp +41774730093
*
* The line-per-field shape matters for small models: a flat prose blob invites
* summarising (and summarising is where invention creeps in), whereas a field
* list invites lookup. Observed with 8B models — the same prompt over a blob
* hallucinates roles, over a field list it reports `<not recorded>`.
*/
export function renderFacts(facts: Fact[]): string {
if (facts.length === 0) return "";
return facts
.map((f) => {
const head = `[${f.id}] ${f.kind} — ${f.subject} (${f.source})`;
const body = Object.entries(f.fields).map(
([k, v]) => ` ${k}: ${v ?? NOT_RECORDED}`,
);
return [head, ...body].join("\n");
})
.join("\n\n");
}

/**
* Which declared fields are unrecorded across the set, as
* `kind.field` keys. The contract block names these explicitly so the rule
* "do not state an affiliation" is anchored to a concrete gap in THIS turn's
* context rather than being a standing abstraction the model may ignore.
*/
export function unrecordedFields(facts: Fact[]): string[] {
const gaps = new Set<string>();
for (const f of facts) {
for (const [k, v] of Object.entries(f.fields)) {
if (v === null) gaps.add(`${f.kind}.${k}`);
}
}
return [...gaps].sort();
}
Loading