- mobile-fast
- Mobile · Fast 4G · iPhone-class (2× CPU)
-
- n = 5
-
-
-
Metric
p50
p75
p90
p99
min
max
σ
-
-
Perf Score
-
100
-
100
-
100
-
100
-
100
-
100
-
0
-
-
LCP
-
407ms
-
407ms
-
408ms
-
408ms
-
406ms
-
408ms
-
1ms
-
-
CLS
-
0.000
-
0.000
-
0.000
-
0.000
-
0.000
-
0.000
-
0.000
-
-
TBT
-
0ms
-
0ms
-
0ms
-
0ms
-
0ms
-
0ms
-
0ms
-
-
FCP
-
407ms
-
407ms
-
408ms
-
408ms
-
406ms
-
408ms
-
1ms
-
-
TTFB
-
3ms
-
3ms
-
3ms
-
3ms
-
3ms
-
3ms
-
0ms
-
-
SI
-
407ms
-
407ms
-
408ms
-
408ms
-
406ms
-
408ms
-
1ms
-
-
-
-
-
-
-
- desktop
- Desktop · Cable (1× CPU)
-
- n = 5
-
-
-
Metric
p50
p75
p90
p99
min
max
σ
-
-
Perf Score
-
100
-
100
-
100
-
100
-
100
-
100
-
0
-
-
LCP
-
218ms
-
218ms
-
218ms
-
218ms
-
215ms
-
218ms
-
1ms
-
-
CLS
-
0.000
-
0.000
-
0.000
-
0.000
-
0.000
-
0.000
-
0.000
-
-
TBT
-
0ms
-
0ms
-
0ms
-
0ms
-
0ms
-
0ms
-
0ms
-
-
FCP
-
218ms
-
218ms
-
218ms
-
218ms
-
215ms
-
218ms
-
1ms
-
-
TTFB
-
3ms
-
3ms
-
3ms
-
3ms
-
2ms
-
3ms
-
0ms
-
-
SI
-
218ms
-
218ms
-
218ms
-
218ms
-
215ms
-
218ms
-
1ms
-
-
-
-
-
-
-
-
-
-
Trace insight
-
Derived diagnosis stored beside this swarm in local history
-
-
mobile-slow· builtin
-
LCP p75 1593ms across 5 runs · no failing audits captured
-
-
-
-
mobile-mid· builtin
-
LCP p75 811ms across 5 runs · no failing audits captured
-
-
-
-
mobile-fast· builtin
-
LCP p75 407ms across 5 runs · no failing audits captured
-
-
-
-
desktop· builtin
-
LCP p75 218ms across 5 runs · no failing audits captured
-
-
-
-
-
-
-
-
-
\ No newline at end of file
diff --git a/.fleet/evidence/landing-audit/scorecard.json b/.fleet/evidence/landing-audit/scorecard.json
deleted file mode 100644
index a3e8bccb..00000000
--- a/.fleet/evidence/landing-audit/scorecard.json
+++ /dev/null
@@ -1,97 +0,0 @@
-{
- "$schema": "fleet.landing-audit.scorecard.v1",
- "project": "codevetter",
- "priority": "P1",
- "candidate": "http://127.0.0.1:4405/",
- "productionChanged": false,
- "checkedAt": "2026-08-28",
- "purpose": {
- "score": 96,
- "minimum": 85,
- "p0": 0,
- "p1": 0
- },
- "design": {
- "score": 91,
- "minimum": 90,
- "p0": 0,
- "p1": 0
- },
- "seo": {
- "score": 100,
- "pages": 74,
- "passed": 1110,
- "failed": 0,
- "warnings": 0,
- "internalLinks": 111,
- "internalLinkIssues": 0,
- "externalLinks": 100,
- "externalRedirectOrAvailabilityIssues": 0,
- "ahrefs": {
- "projectId": 10205898,
- "productionBaselineHealth": 48,
- "productionBaselineErrors": 233,
- "productionBaselineWarnings": 95,
- "productionBaselineNotices": 178,
- "candidateRemediation": {
- "broken404Pages": 91,
- "canonicalRedirects": 49,
- "multipleH1Pages": 36,
- "pagesLinkingToBrokenPages": 30,
- "orphanPages": 2,
- "externalRedirects": 5
- },
- "status": "candidate-clean-production-recrawl-pending",
- "note": "Ahrefs remains the deployed-history baseline until an authorized deployment and fresh crawl. The Cloudflare Pages candidate resolves every sitemap page and internal link without redirects or errors."
- }
- },
- "geo": {
- "score": 100,
- "tier": "S",
- "surfaces": 25,
- "validSurfaces": 25,
- "note": "The local Worker served the full agent catalog and Markdown negotiation; the canonical public origin independently passes the S-tier index audit."
- },
- "performance": {
- "score": 99,
- "runs": 20,
- "successfulRuns": 20,
- "weightedLcpMs": 514,
- "weightedCls": 0,
- "weightedTbtMs": 0,
- "cwvLcpP75Ms": 813,
- "profiles": {
- "mobileSlowLcpP75Ms": 1593,
- "mobileMidLcpP75Ms": 811,
- "mobileFastLcpP75Ms": 407,
- "desktopLcpP75Ms": 218
- },
- "report": "performance.html"
- },
- "verification": {
- "astroBuild": "pass",
- "agentSurfaces": "25/25",
- "workerApiCatalog": "pass",
- "markdownNegotiation": "pass",
- "responsiveBrowser": "pass",
- "axeWcagAAndAA": {
- "desktopViolations": 0,
- "mobileViolations": 0
- },
- "footer": {
- "askAi": "pass",
- "projectStrip": "pass",
- "aboutAndContactInlinks": "pass"
- },
- "releaseTruth": "v1.11.0 Apple-silicon DMG and updater archive verified",
- "responsiveScreenshots": [
- "../../../artifacts/design/landing-audit/final-p1-390.png",
- "../../../artifacts/design/landing-audit/final-p1-768.png",
- "../../../artifacts/design/landing-audit/final-p1-1440.png"
- ]
- },
- "limitations": [
- "The real receipt is genuine self-dogfood evidence but is not output from the exact codevetter check command shown on the page.",
- "The current public production landing remains unchanged until an explicit deployment is authorized."
- ]
-}
diff --git a/.impeccable/critique/2026-07-27T05-06-55Z__scripts-run-structural-context-evaluation-mjs.md b/.impeccable/critique/2026-07-27T05-06-55Z__scripts-run-structural-context-evaluation-mjs.md
deleted file mode 100644
index c94c42c3..00000000
--- a/.impeccable/critique/2026-07-27T05-06-55Z__scripts-run-structural-context-evaluation-mjs.md
+++ /dev/null
@@ -1,103 +0,0 @@
----
-target: structural-context evaluation HTML report
-total_score: 33
-max_score: 40
-na_heuristics:
-p0_count: 0
-p1_count: 0
-timestamp: 2026-07-27T05-06-55Z
-slug: scripts-run-structural-context-evaluation-mjs
----
-Method: dual-agent (A: impeccable_assessment_a · B: impeccable_assessment_b)
-
-## Design Health Score
-
-| # | Heuristic | Score | Key finding |
-|---|---|---:|---|
-| 1 | Visibility of System Status | 4 | Qualification, pair counts, gates, source, and read-only state are explicit. |
-| 2 | Match System / Real World | 3 | A/A and discordance still assume evaluation fluency. |
-| 3 | User Control and Freedom | 2 | A static report has no filtering or bulk disclosure controls. |
-| 4 | Consistency and Standards | 4 | Evidence hierarchy, arm naming, and CodeVetter tokens are cohesive. |
-| 5 | Error Prevention | 4 | Claim boundaries and neutral diagnostic deltas prevent overstatement. |
-| 6 | Recognition Rather Than Recall | 4 | Mobile diagnostics now expose every comparison value in each metric card. |
-| 7 | Flexibility and Efficiency | 2 | Large experiments will eventually need anchors, filters, or condensed rows. |
-| 8 | Aesthetic and Minimalist Design | 4 | The report remains focused and qualification-first across all widths. |
-| 9 | Error Recovery | 3 | Invalid pairs explain concrete exclusion reasons; generation errors remain CLI-only. |
-| 10 | Help and Documentation | 3 | Inline caveats are strong; evaluation terms have no compact glossary. |
-| **Total** | | **33/40** | **Good, above the Fleet floor after polish.** |
-
-## Design Specificity Verdict
-
-The report is authored for CodeVetter rather than a generic analytics
-dashboard. Its sequence is the product's evidence model: claim boundary,
-paired executable outcome, changed checks and graph traces, qualification,
-activity diagnostics, and limitations. Amber remains the evidence accent and
-cyan is reserved for graph provenance.
-
-The CLI detector returned zero findings. The rendered detector found 15
-advisory issues before polish: eight small-text or line-length findings and
-seven cyan-palette findings. The cyan findings were false positives because the
-color has a stable graph-provenance meaning. The small text, touch target, copy
-measure, and mobile diagnostics issues were fixed.
-
-## Overall Impression
-
-The opening creates curiosity, then immediately constrains interpretation with
-an unqualified claim. The paired corridor is the visual peak. The closing
-authorized-claim block now restores that same boundary after the evidence
-detail, so a long read ends with the correct decision.
-
-## What's Working
-
-- Qualification appears before the favorable synthetic percentage.
-- Paired outcomes and hidden-check changes are readable without decorative
- metric cards.
-- Native details, semantic regions, a real data table, textual PASS/FAIL
- labels, visible focus, and high contrast support accessible inspection.
-
-## Priority Issues
-
-### [P2] Large-run navigation
-
-The schema permits much larger experiments than the two-pair sample. A future
-real corpus may need outcome filters, section anchors, or condensed tie rows.
-This does not block the bounded local report.
-
-### [P3] Evaluation terminology
-
-A/A, discordance, and coverage are correct but assume statistical fluency. A
-compact glossary may help less experienced product owners when real receipts
-arrive.
-
-### [P3] Fixed dark presentation
-
-The tokenized fixed-dark report is coherent with CodeVetter and includes print
-rules, but it does not offer an alternate light screen theme.
-
-## Persona Red Flags
-
-**Alex, power user:** The two-pair report is fast to scan, but dozens of pairs
-would require outcome filtering and condensed ties.
-
-**Sam, accessibility-dependent user:** Semantic structure, contrast, keyboard
-disclosures, 44px summary targets, and stacked mobile diagnostics now support
-the core reading path. A future large corpus needs skip links or section
-navigation.
-
-**Priya, technical product owner:** The synthetic and unqualified boundary is
-now tied to both the comparison corridor and the closing verdict. Activity
-deltas are neutral and explicitly mean less, not better.
-
-## Minor Observations
-
-- Long identities and source paths wrap safely.
-- A zero-valid-pair state withholds the percentage corridor.
-- Print semantic colors use darker values while retaining text labels.
-- Missing optional diagnostics remain missing rather than becoming zero.
-
-## Questions to Consider
-
-- At what corpus size should the evidence brief become a navigable
- investigation tool?
-- Should real-trial reports define a tiny inline glossary for A/A noise and
- qualification policy?
diff --git a/.impeccable/critique/2026-07-30T06-34-52Z__apps-desktop-src-components-sidebar-tsx.md b/.impeccable/critique/2026-07-30T06-34-52Z__apps-desktop-src-components-sidebar-tsx.md
deleted file mode 100644
index 4f4d8232..00000000
--- a/.impeccable/critique/2026-07-30T06-34-52Z__apps-desktop-src-components-sidebar-tsx.md
+++ /dev/null
@@ -1,102 +0,0 @@
----
-target: CodeVetter desktop sidebar
-total_score: 33
-max_score: 40
-na_heuristics:
-p0_count: 0
-p1_count: 0
-timestamp: 2026-07-30T06-34-52Z
-slug: apps-desktop-src-components-sidebar-tsx
----
-Method: dual-agent (A: sidebar_critique_a · B: sidebar_critique_b)
-
-## Design Health Score
-
-| # | Heuristic | Score | Key issue |
-|---|---|---:|---|
-| 1 | Visibility of system status | 3 | Active location is clear; the transient G chord remains intentionally quiet. |
-| 2 | Match system / real world | 3 | Product labels are established but assume some CodeVetter familiarity. |
-| 3 | User control and freedom | 4 | Navigation is reversible and the palette now restores focus to its trigger. |
-| 4 | Consistency and standards | 4 | Rows, grouping, focus, active state, and spacing follow the Evidence Bench system. |
-| 5 | Error prevention | 3 | Shortcut handling protects form controls; contenteditable remains a narrow edge case. |
-| 6 | Recognition rather than recall | 3 | Every destination is labeled; detailed descriptions remain in accessible tooltips. |
-| 7 | Flexibility and efficiency | 4 | Search, Cmd-K, and G chords provide strong expert acceleration. |
-| 8 | Aesthetic and minimalist design | 4 | The rail is calm, compact, and free of decorative feature noise. |
-| 9 | Error recovery | 3 | Search dismisses cleanly and restores focus; mistimed G chords remain silent. |
-| 10 | Help and documentation | 2 | Tooltips explain destinations, but the rail intentionally carries no dedicated help surface. |
-| **Total** | | **33/40** | **Good; no blocking or major issues remain.** |
-
-## Design Specificity Verdict
-
-The sidebar is clearly adapted to CodeVetter through its Context and
-Verification grouping, Evidence Workbench identity, warm verification accent,
-real product routes, resource utility, and keyboard model. Its basic rail
-composition is conventional, but the content and state grammar are not a
-generic mockup.
-
-The deterministic scan returned zero findings across `App.tsx`, `sidebar.tsx`,
-`ResourceChip.tsx`, and `command-palette.tsx`. Browser evidence confirmed AA
-contrast, one accessible active destination, no overflow at supported desktop
-sizes, a working Search trigger, and keyboard focus restoration. No reliable
-browser overlay was available because the exposed evaluation surface was
-read-only; live screenshots, computed styles, geometry, axe, and Playwright
-interaction checks were used instead.
-
-## Overall Impression
-
-The new rail feels like a quiet native instrument and carries the reference's
-search-first hierarchy without importing an unrelated cream visual system. The
-main opportunity was finishing keyboard and control-size details, both of which
-were corrected during the pass.
-
-## What's Working
-
-- The active state uses position, icon treatment, text, and `aria-current`, so
- it is legible without color alone.
-- Context, Verification, and bottom utilities produce a clear three-part
- information hierarchy.
-- Cmd-K, visible G chords, and direct search make the compact shell efficient
- for repeat users.
-
-## Priority Issues
-
-- **[P1, fixed] Command palette dialog naming:** Opening Search exposed a Radix
- accessibility error because the dialog had no screen-reader title. The
- palette now includes a visually hidden `DialogTitle`, and the interaction
- test asserts that opening and closing it emits no console error.
-- **[P2, fixed] Palette focus restoration:** Closing Search initially returned
- focus to the document body. The shell now remembers the invoking element and
- restores focus after Radix closes.
-- **[P2, fixed] Control sizing:** Root font sizing made Tailwind rem-based
- 40px controls render at 35px. Search and navigation rows now use explicit
- 40px dimensions and full 13–14px labels.
-- **[P3] Silent G-chord timeout:** A mistimed chord has no feedback. This is
- acceptable for a secondary expert accelerator, but could gain a tiny
- transient key hint if real usage shows failures.
-- **[P3] Destination descriptions rely on tooltips:** First-time users may
- need a little exploration to distinguish Work, Board, Review, and Testing.
- Existing product labels were preserved deliberately.
-
-## Persona Red Flags
-
-- **Power user:** Search and G chords are fast, but the 500ms G timeout may
- feel unforgiving until learned.
-- **First-timer:** The grouping helps, though the differences among Work,
- Board, Review, and Testing are learned through tooltips and page content.
-- **Keyboard or low-vision user:** The final build has a global amber focus
- ring, 40px controls, AA contrast, semantic groups, text labels, and focus
- restoration. No major barrier remains in the rail.
-
-## Minor Observations
-
-- The 224px rail stays proportionate at the configured 900px minimum window.
-- The warm ambient wash respects the single-accent rule.
-- The resource chip is absent in browser fallback because Tauri resource data
- is unavailable; it remains present in the desktop runtime.
-
-## Questions to Consider
-
-- Should future usage evidence show the current repository or verification run
- in this rail, or should project context stay inside the owning workspaces?
-- If users do not discover G chords, would one compact shortcuts hint be more
- useful than permanent suffixes?
diff --git a/.impeccable/critique/2026-08-01T17-55-01Z__apps-desktop-src-app-tsx.md b/.impeccable/critique/2026-08-01T17-55-01Z__apps-desktop-src-app-tsx.md
deleted file mode 100644
index 2a6b1ba4..00000000
--- a/.impeccable/critique/2026-08-01T17-55-01Z__apps-desktop-src-app-tsx.md
+++ /dev/null
@@ -1,171 +0,0 @@
----
-target: current CodeVetter desktop product and UI
-total_score: 23
-max_score: 40
-na_heuristics:
-p0_count: 1
-p1_count: 4
-timestamp: 2026-08-01T17-55-01Z
-slug: apps-desktop-src-app-tsx
----
-# CodeVetter product and desktop critique
-
-## Strategic verdict
-
-CodeVetter is impressive engineering but not yet a coherent product. The repository contains a substantial local verification stack: bounded execution, runtime receipts, structural and historical evidence, deterministic scoring, a qualified synthetic task corpus, CLI/MCP boundaries, and unusually honest failure states. But that core is buried beneath an older AI-review workbench, repository-intelligence suite, usage dashboard, agent workspace, board, and native agent presentation layer.
-
-The July pivot exists in product documentation and newer harness work. It does not yet exist as the user's product. The desktop's default object is still a dashboard or repository; it should be a verification case.
-
-The focused job should be:
-
-> Given a task and an agent-authored change, did it actually work? Show the executable evidence, state what remains unverified, and make the result reproducible.
-
-Comparative agent and context experiments are the second job, powered by the same receipts. Graph context is an experimental input, not the product.
-
-## Competition
-
-The tools initially identified are several different markets:
-
-- pgGraph and HydraDB are graph infrastructure. They are not meaningful product competitors.
-- CodeGraph, Graphify, and RepoWise are agent-readable context engines. RepoWise also spans human wiki, history, decisions, and code health, creating direct overlap with Repo Unpack.
-- DeepWiki is primarily human-readable generated documentation and grounded Q&A.
-- Sourcegraph is enterprise code search and multi-repository context.
-- CodeRabbit and Qodo compete with the legacy Review proposition and have much stronger pull-request distribution.
-- Harbor/Terminal-Bench and SWE-bench occupy coding-agent benchmark infrastructure.
-- Braintrust and LangSmith occupy general experiment, dataset, scoring, tracing, and comparison infrastructure.
-
-CodeVetter should not try to beat focused context providers at indexing, established review vendors at PR distribution, or general evaluation platforms at horizontal breadth. Its credible wedge is local, software-specific, execution-backed verification with hidden checks, immutable evidence identities, contamination detection, and reproducible comparisons.
-
-## Design Health Score
-
-| # | Heuristic | Score | Key issue |
-|---|---|---:|---|
-| 1 | Visibility of System Status | 3 | Strong local states, but no unified verification-run status across surfaces. |
-| 2 | Match System / Real World | 2 | Repo Unpack, T-Rex, warm verification, and Review with Claude obscure the core job. |
-| 3 | User Control and Freedom | 3 | Good cancellation, retry, persistence, and reversible actions; deeper exits and undo vary. |
-| 4 | Consistency and Standards | 2 | Coherent tokens, inconsistent page structures and navigation documentation. |
-| 5 | Error Prevention | 3 | Strong validation and confirmations, but advanced forms expose too many paths. |
-| 6 | Recognition Rather Than Recall | 2 | Users must remember how Repo, Review, Testing, and Work compose. |
-| 7 | Flexibility and Efficiency | 3 | Strong shortcuts, persistent state, history, and expert affordances. |
-| 8 | Aesthetic and Minimalist Design | 2 | Visually disciplined but functionally overloaded. |
-| 9 | Error Recovery | 2 | Several actionable errors, but no consistent guided recovery model. |
-| 10 | Help and Documentation | 1 | Onboarding teaches the outdated review product rather than verification evidence. |
-| **Total** | | **23/40** | **Acceptable craft; substantial product simplification required.** |
-
-## Design Specificity Verdict
-
-### Design assessment
-
-Visually authored, structurally unfocused. The dark ink and warm amber Evidence Bench language is coherent and appropriate. The app feels technically serious. But the shell presents several historical products as peers, so it reads as a consolidated suite rather than one verification instrument.
-
-### Deterministic scan
-
-The detector reported 10 `gray-on-color` findings: five in Home, three in AgentPanel, and two in QuickReview. Source inspection makes six definite false positives and the remaining four likely false positives because the backgrounds are mutually exclusive branches or very low-opacity tints over dark surfaces. The scan did not reveal a systemic mechanical design defect.
-
-This reinforces the main conclusion: the highest-impact UI problems are information architecture, terminology, and hierarchy—not Tailwind color cleanup.
-
-### Visual overlays
-
-No reliable visual overlay is available. Browser control reported no connected browser, so mutable injection and screenshots could not be performed. Five representative Vite routes returned HTTP 200, which confirms routing only, not rendered quality.
-
-## Overall Impression
-
-The strongest moments are the honest receipt and no-confidence states in Testing and Review. The weakest moment is the product entrance: onboarding teaches model selection and AI review, then the app opens on usage telemetry. A user must cross several legacy concepts before reaching the differentiated product.
-
-The biggest opportunity is not a redesign of each page. It is choosing one canonical object—`verification case`—and reorganizing everything around it.
-
-## What's Working
-
-- Honest semantic states such as partial coverage, passed with limits, and no confidence are unusually good.
-- Persistent routes, cancellation, retries, bounded output, and history show excellent operational care.
-- The ink/amber system, evidence typography, focus treatment, and written status labels are a solid craft foundation worth preserving.
-
-## Priority Issues
-
-### P0 — The visible product contradicts the stated product
-
-**Why it matters:** The repo says CLI/MCP verification is primary and desktop is a receipt viewer. The app leads with Usage, Repo Unpack, Work, Board, Review, and Testing. The landing page still sells a desktop AI reviewer and makes claims about vulnerability classes and offline behavior. Users cannot form a stable expectation.
-
-**Fix:** Pick the verification product explicitly. Rewrite landing, onboarding, navigation, and the default route around one verification case. Remove unsupported claims and demote unrelated surfaces.
-
-**Suggested command:** `$impeccable shape`
-
-### P1 — The shell contradicts the core loop
-
-**Why it matters:** Launching into usage telemetry makes administration feel more important than determining whether a change is correct. Work and Board are agent-control products placed inside Verification.
-
-**Fix:** Use a minimal shell such as Verify, Runs, Experiments, and Settings. Put repository context inside a case; move Usage, Work, Board, and Agent Island to Labs/Legacy or remove them from primary navigation.
-
-**Suggested command:** `$impeccable distill`
-
-### P1 — Review and Testing split one user question across two products
-
-**Why it matters:** A user asks whether a change is correct. Review emphasizes model findings; Testing owns the strongest executable receipts. The user must mentally merge them.
-
-**Fix:** Model a verification case with stages: target and intent, checks, findings, runtime evidence, verdict, limitations, and next action.
-
-**Suggested command:** `$impeccable shape`
-
-### P1 — Results bury the verdict beneath accumulated features
-
-**Why it matters:** Review's sidebar contains roughly a dozen evidence, graph, QA, export, and audience systems. Equal visual weight makes source-backed limitations and next actions hard to locate.
-
-**Fix:** Pin verdict, evidence strength, limitations, and next action. Move graphs, audience simulation, X-Ray, synthetic QA, and exports behind secondary disclosure.
-
-**Suggested command:** `$impeccable distill`
-
-### P1 — Onboarding installs the wrong mental model
-
-**Why it matters:** It teaches model selection, usage stats, and AI review instead of task completion and executable proof.
-
-**Fix:** First run should select a repository/change, run one bounded check, and teach how to read a receipt, failure, and limitation.
-
-**Suggested command:** `$impeccable onboard`
-
-### P2 — Dense evidence presentation strains accessibility
-
-**Why it matters:** Critical context is often 9–11px and muted; dense sidebars create long keyboard paths.
-
-**Fix:** Increase essential evidence metadata size and contrast, simplify result order, and confirm effective runtime contrast visually.
-
-**Suggested command:** `$impeccable audit`
-
-## Cognitive Load
-
-High: seven of eight checklist areas fail. Grouping is generally good, but single focus, chunking, hierarchy, one-thing-at-a-time flow, minimal choices, working-memory burden, and progressive disclosure do not.
-
-Decision points above four include:
-
-- six primary destinations plus Settings and command search;
-- up to eight Repo Unpack sections;
-- eleven Settings categories;
-- roughly a dozen Review result-side modules; and
-- seven setup concepts inside expanded Review context.
-
-## Emotional Journey
-
-The user expects verification, encounters usage administration, becomes uncertain about which surface owns the task, then finally reaches excellent evidence language in Testing. The product peaks late and ends without one calm closure: verified, failed, or no confidence, followed by the next safe action.
-
-## Persona Red Flags
-
-**Alex, power user:** Strong shortcuts and persistent state do not answer whether the same change belongs in Repo, Review, or Testing. A trustworthy evaluation in under a minute is unlikely.
-
-**Jordan, first-timer:** Usage telemetry and AI-review onboarding create the wrong model before they encounter Repo Unpack, T-Rex, warm verification, and scenario compilation.
-
-**Sam, keyboard/low-vision user:** Focus and reduced-motion support are positive, but tiny muted evidence text and the long Review sidebar journey reduce practical accessibility.
-
-## Minor Observations
-
-- Design and surface documentation describe a top rail while implementation uses a fixed left rail.
-- Board has a keyboard shortcut but is absent from the command palette.
-- Page-title structures differ substantially by route.
-- T-Rex is internal-history branding, not self-explanatory product language.
-- The sidebar subtitle Evidence workbench is good; the rest of the IA does not yet fulfill it.
-- The four largest page files total roughly 14,900 lines, mirroring feature and state accumulation in the user experience.
-
-## Questions to Consider
-
-- If Usage, Work, Board, Agent Island, and most Repo Unpack sections disappeared from primary navigation, would the actual verification product lose anything essential?
-- Why are Review and Testing separate when the user asks one question: is this change correct?
-- Does a panel change the verdict or explain its confidence? If not, why is it in the primary result view?
-- Is CodeVetter a daily verification tool, an evaluation research lab, or a broad agent workbench? It cannot lead with all three.
diff --git a/.impeccable/critique/2026-08-10T18-31-57Z__apps-desktop-src-pages-home-tsx.md b/.impeccable/critique/2026-08-10T18-31-57Z__apps-desktop-src-pages-home-tsx.md
deleted file mode 100644
index 6e31755f..00000000
--- a/.impeccable/critique/2026-08-10T18-31-57Z__apps-desktop-src-pages-home-tsx.md
+++ /dev/null
@@ -1,74 +0,0 @@
----
-target: Usage telemetry evidence tiers
-total_score: 36
-max_score: 40
-na_heuristics:
-p0_count: 0
-p1_count: 0
-timestamp: 2026-08-10T18-31-57Z
-slug: apps-desktop-src-pages-home-tsx
----
-## Design Health Score
-
-| # | Heuristic | Score | Key issue |
-|---|---|---:|---|
-| 1 | Visibility of system status | 4 | Verified, partial, stale, pending, and loading states are written explicitly. |
-| 2 | Match system / real world | 3 | API-equivalent remains specialist language, now explained as not subscription spend. |
-| 3 | User control and freedom | 4 | Reconcile and recovery settings are available at the diagnosis. |
-| 4 | Consistency and standards | 4 | One reconciliation verb now owns the refresh path. |
-| 5 | Error prevention | 4 | Legacy, ambiguous, stale, and unpriced data cannot masquerade as verified. |
-| 6 | Recognition rather than recall | 4 | Recovery settings are linked in context. |
-| 7 | Flexibility and efficiency | 3 | Aggregate categories are not yet drillable to individual sources. |
-| 8 | Aesthetic and minimalist design | 4 | Evidence hierarchy is compact and uses the incumbent workbench language. |
-| 9 | Error recovery | 3 | Recovery is complete, but source-level diagnostics remain aggregate. |
-| 10 | Help and documentation | 3 | Inline pricing and recovery explanations cover the main uncertainty model. |
-| **Total** | | **36/40** | **Excellent** |
-
-## Design Specificity Verdict
-
-The result is authored for CodeVetter's Evidence Bench. Accepted transcript observations,
-scanner revision, observation watermark, exact/ranged/unpriced pricing, and explicit legacy
-exclusion make the surface an evidence instrument rather than a generic analytics card.
-
-The deterministic detector returned five `gray-on-color` warnings in Home.tsx and none in
-Settings.tsx. All five are contextual false positives: the background is translucent over ink or
-the slate text classes are mutually exclusive with the cyan active state. Verified detector issue
-count: zero.
-
-## Overall Impression
-
-The trusted number leads, uncertainty is written rather than hidden, and recovery is attached to
-the diagnosis. The remaining opportunity is source/session drill-down, not another visual layer.
-
-## What's Working
-
-- Verified totals and legacy estimates are structurally separated.
-- Cost bounds explain unknown service tier and disclaim subscription spend.
-- Recovery is one bounded flow: import roots, then re-index and reconcile.
-
-## Priority Issues
-
-- **P2 — Aggregate diagnostics are not drillable.** Users can see affected counts but not the
- source identities. Add a source-detail disclosure after the read cutover is qualified.
-- **P3 — Narrow screenshots compress below the product contract.** The Tauri app enforces a 900px
- minimum; 390px is retained as evidence but is not a supported window state.
-
-## Persona Red Flags
-
-- **Alex:** source-level evidence is not yet inspectable from the aggregate.
-- **Sam:** the cost range is now explicitly API-equivalent and not subscription spend; written
- partial coverage does not rely on color.
-- **Riley:** import persistence failures are announced and the recovery action returns to a single
- reconciliation path.
-
-## Minor Observations
-
-- Legacy period estimates remain expanded for continuity; a later release may collapse them once
- users have migrated to verified reads.
-- The app's documented and configured minimum width is 900px, so mobile-shell adaptation is out of
- scope for this macOS desktop viewer.
-
-## Questions to Consider
-
-- Should the next qualified iteration expose the exact sessions behind each unresolved tier?
-- Once verified coverage stabilizes, should the legacy blended summary become collapsed by default?
diff --git a/.impeccable/critique/2026-08-15T20-33-20Z__apps-desktop-src-components-app-error-boundary-tsx.md b/.impeccable/critique/2026-08-15T20-33-20Z__apps-desktop-src-components-app-error-boundary-tsx.md
deleted file mode 100644
index 3c800eac..00000000
--- a/.impeccable/critique/2026-08-15T20-33-20Z__apps-desktop-src-components-app-error-boundary-tsx.md
+++ /dev/null
@@ -1,57 +0,0 @@
----
-target: apps/desktop/src/components/app-error-boundary.tsx
-total_score: 35
-maximum: 40
-p0: 0
-p1: 0
-p2: 1
-method: dual-agent
-timestamp: 2026-08-15T20-33-20Z
-slug: apps-desktop-src-components-app-error-boundary-tsx
----
-# CodeVetter crash recovery critique
-
-## Method
-
-Dual-agent review: a detector-blind visual/heuristic assessment plus an independent detector and responsive-browser evidence pass. The final state was then rechecked at 390, 768, and 1440 px after resolving the review findings.
-
-## Nielsen assessment — 35/40
-
-| Heuristic | Score | Final assessment |
-| --- | ---: | --- |
-| Visibility of system status | 3 | The interruption, local receipt, and copy status are explicit; repeated retry has no attempt counter. |
-| Match to the real world | 4 | Scope-aware language and plain recovery actions describe what happened and what each action does. |
-| User control and freedom | 3 | Retry, reload, and Usage escape cover the common exits; Usage remains a best-effort app route. |
-| Consistency and standards | 4 | Uses the established ink surface, amber action, semantic rose state, type, buttons, and focus treatment. |
-| Error prevention | 3 | The boundary contains the failure and avoids unsupported safety claims; it does not add a repeated-failure safe mode. |
-| Recognition over recall | 4 | Actions are visible and retry/reload behavior is stated directly. |
-| Flexibility and efficiency | 3 | Keyboard recovery and copyable diagnostics are available without exposing raw details by default. |
-| Aesthetic and minimalist design | 4 | The hierarchy stays focused: interruption, recovery, then local evidence. |
-| Error recognition and recovery | 4 | Scope, three recovery routes, incident identity, and technical evidence are all visible. |
-| Help and documentation | 3 | Technical details support reporting, but no dedicated troubleshooting route is present. |
-
-## Cognitive load — 8/8
-
-The surface has one focus, three clearly grouped recovery choices, a short behavioral explanation, and progressive disclosure for diagnostics. No decision point exceeds four choices.
-
-## Accessibility and responsive evidence
-
-- Focus moves to the recovery heading on mount; the next Tab reaches the primary recovery action.
-- The full-page alert was narrowed to the interruption announcement, leaving controls outside the live alert.
-- Muted metadata uses the higher-contrast zinc-400 token.
-- Axe reported no critical or serious violations in the focused Playwright check.
-- Document scroll width matched client width at 390, 768, and 1440 px.
-
-## Findings resolved
-
-- **P1 resolved:** removed the categorical claim that the repository was unmodified. The UI now states that repository state was not checked.
-- **P1 resolved:** application-shell failures now always expose a Return to Usage action in addition to retry and reload.
-- **P2 resolved:** recovery takes focus, metadata contrast was raised, and retry versus reload behavior is explained.
-
-## Remaining advisory item
-
-- **P2:** if the same render failure repeats, the surface does not yet count attempts or escalate to a dedicated safe mode. This is a future reliability enhancement, not a blocker for the bounded recovery layer.
-
-## Detector and integrity
-
-The advisory detector returned an empty result (`[]`) across the recovery component and entry point. No production dependency was added, raw error messages and stacks are not persisted, and repository/query data is excluded from the local incident receipt.
diff --git a/.impeccable/critique/2026-08-15T21-39-48Z__apps-desktop-src-pages-performance-tsx.md b/.impeccable/critique/2026-08-15T21-39-48Z__apps-desktop-src-pages-performance-tsx.md
deleted file mode 100644
index 3e1f0590..00000000
--- a/.impeccable/critique/2026-08-15T21-39-48Z__apps-desktop-src-pages-performance-tsx.md
+++ /dev/null
@@ -1,30 +0,0 @@
----
-timestamp: 2026-08-15T21-39-48Z
-slug: apps-desktop-src-pages-performance-tsx
----
-# Performance workbench critique
-
-Target: `apps/desktop/src/pages/Performance.tsx`
-
-## Outcome
-
-- Design heuristic score: 35/40 (good, near excellent).
-- Automated detector: 0 findings.
-- Responsive qualification: no horizontal overflow at 390, 768, or 1440 px.
-- Accessibility structure: one main landmark, labelled workload controls, labelled evidence region, and accessible form names.
-- Final severity: 0 P0, 0 P1.
-
-## Resolved during critique
-
-- Added a real same-scope paired-verification action and verdict-driven campaign states.
-- Invalidated stale evidence when scope fields or the selected repository change.
-- Cancelled and discarded late receipts from a prior repository generation.
-- Added truthful blocked, failed, and no-confidence recovery states.
-- Separated observed, inferred, and unverified evidence without truncating captured rows.
-- Raised low-contrast operational copy and removed empty machine-detail rows.
-
-## Evidence
-
-- `artifacts/design/product-surfaces-after-390.jpg`
-- `artifacts/design/product-surfaces-after-768.jpg`
-- `artifacts/design/product-surfaces-after-1440.jpg`
diff --git a/.impeccable/design.json b/.impeccable/design.json
deleted file mode 100644
index ad45e6cb..00000000
--- a/.impeccable/design.json
+++ /dev/null
@@ -1,212 +0,0 @@
-{
- "schemaVersion": 2,
- "generatedAt": "2026-07-29T00:00:00.000Z",
- "title": "Design System: CodeVetter",
- "extensions": {
- "colorMeta": {
- "canvas-ink": {
- "role": "neutral",
- "displayName": "Canvas Ink",
- "canonical": "#060708",
- "tonalRamp": [
- "#060708",
- "#0c0d0f",
- "#111316",
- "#17191d",
- "#35383e",
- "#6c7078",
- "#a1a1aa",
- "#f4f4f5"
- ]
- },
- "action-amber": {
- "role": "primary",
- "displayName": "Action Amber",
- "canonical": "#f3ad3d",
- "tonalRamp": [
- "#2a1b05",
- "#4b3008",
- "#71490d",
- "#9b6818",
- "#c88728",
- "#f3ad3d",
- "#ffc75e",
- "#fff0c7"
- ]
- },
- "failure-rose": {
- "role": "semantic",
- "displayName": "Failure Rose",
- "canonical": "#fb7185",
- "tonalRamp": [
- "#2e080e",
- "#54121d",
- "#7f2030",
- "#aa3448",
- "#d94f65",
- "#fb7185",
- "#fda4af",
- "#ffe4e6"
- ]
- },
- "verified-green": {
- "role": "semantic",
- "displayName": "Verified Green",
- "canonical": "#4ade80",
- "tonalRamp": [
- "#052e16",
- "#14532d",
- "#166534",
- "#15803d",
- "#22c55e",
- "#4ade80",
- "#86efac",
- "#dcfce7"
- ]
- }
- },
- "typographyMeta": {
- "title": {
- "displayName": "Workbench Title",
- "purpose": "Page, panel, and evidence-section headings."
- },
- "body": {
- "displayName": "Operating Body",
- "purpose": "Instructions, summaries, and supporting context."
- },
- "label": {
- "displayName": "Compact Label",
- "purpose": "Fields, controls, metrics, and metadata."
- },
- "evidence": {
- "displayName": "Evidence Mono",
- "purpose": "Paths, revisions, commands, and machine identities."
- }
- },
- "shadows": [
- {
- "name": "surface-ambient",
- "value": "0 28px 80px -52px rgba(0, 0, 0, 0.92)",
- "purpose": "Diffuse depth for major cards and overlays."
- },
- {
- "name": "action-warm",
- "value": "0 12px 30px -18px rgba(243, 173, 61, 0.9)",
- "purpose": "Restrained emphasis for primary action controls."
- }
- ],
- "motion": [
- {
- "name": "control-state",
- "value": "150ms ease",
- "purpose": "Color, border, shadow, and pressed-state transitions."
- },
- {
- "name": "content-enter",
- "value": "200ms ease-out",
- "purpose": "Short opacity and 4px translate entrance for newly available content."
- }
- ],
- "breakpoints": [
- {
- "name": "sm",
- "value": "640px"
- },
- {
- "name": "lg",
- "value": "1024px"
- },
- {
- "name": "desktop-window-min",
- "value": "900px"
- }
- ]
- },
- "components": [
- {
- "name": "Primary Button",
- "kind": "button",
- "refersTo": "button-primary",
- "description": "The single intentional action within a verification context.",
- "html": "",
- "css": ".ds-button-primary { height: 40px; padding: 8px 16px; border: 1px solid rgba(253,230,138,.2); border-radius: 10px; background: var(--cv-accent, #f3ad3d); color: #211609; font: 500 14px/1.25 -apple-system,BlinkMacSystemFont,\"SF Pro Text\",sans-serif; box-shadow: 0 12px 30px -18px rgba(243,173,61,.9), inset 0 1px 0 rgba(255,255,255,.3); transition: background-color 150ms ease, transform 150ms ease; } .ds-button-primary:hover { background: var(--cv-accent-strong, #ffc75e); } .ds-button-primary:focus-visible { outline: 2px solid rgba(243,173,61,.88); outline-offset: 2px; } .ds-button-primary:active { transform: translateY(1px); }"
- },
- {
- "name": "Outline Button",
- "kind": "button",
- "refersTo": "button-outline",
- "description": "A bounded secondary action that does not compete with execution.",
- "html": "",
- "css": ".ds-button-outline { height: 40px; padding: 8px 16px; border: 1px solid rgba(255,255,255,.11); border-radius: 10px; background: rgba(255,255,255,.035); color: #e4e4e7; font: 500 14px/1.25 -apple-system,BlinkMacSystemFont,\"SF Pro Text\",sans-serif; box-shadow: inset 0 1px 0 rgba(255,255,255,.04); transition: background-color 150ms ease, border-color 150ms ease; } .ds-button-outline:hover { border-color: rgba(255,255,255,.18); background: rgba(255,255,255,.075); color: #fff; } .ds-button-outline:focus-visible { outline: 2px solid rgba(243,173,61,.88); outline-offset: 2px; }"
- },
- {
- "name": "Evidence Input",
- "kind": "input",
- "refersTo": "input",
- "description": "A compact field for URLs, ranges, and verification parameters.",
- "html": "",
- "css": ".ds-input { width: 100%; height: 40px; padding: 8px 12px; border: 1px solid rgba(255,255,255,.1); border-radius: 10px; background: rgba(255,255,255,.035); color: #f4f4f5; font: 400 14px/1.5 -apple-system,BlinkMacSystemFont,\"SF Pro Text\",sans-serif; box-shadow: inset 0 1px 0 rgba(255,255,255,.025); transition: background-color 150ms ease, border-color 150ms ease, box-shadow 150ms ease; } .ds-input:hover { border-color: rgba(255,255,255,.15); } .ds-input:focus-visible { outline: 2px solid rgba(243,173,61,.15); outline-offset: 2px; border-color: rgba(252,211,77,.35); background: rgba(255,255,255,.05); }"
- },
- {
- "name": "Verification Card",
- "kind": "card",
- "refersTo": "card",
- "description": "The primary workbench plane for one verification mechanism.",
- "html": "
Test change in preview
Resolve exact source identity and return browser evidence.
",
- "css": ".ds-card { padding: 20px; border: 1px solid rgba(255,255,255,.075); border-radius: 12px; background: var(--cv-surface, #0c0d0f); color: #f4f4f5; box-shadow: 0 24px 70px -50px rgba(0,0,0,.95), inset 0 1px 0 rgba(255,255,255,.025); } .ds-card h3 { margin: 0; font: 600 18px/1.25 \"SF Pro Display\",-apple-system,sans-serif; letter-spacing: -.018em; } .ds-card p { margin: 6px 0 0; color: #a1a1aa; font: 400 14px/1.5 -apple-system,BlinkMacSystemFont,\"SF Pro Text\",sans-serif; }"
- },
- {
- "name": "Evidence Badge",
- "kind": "chip",
- "refersTo": "badge",
- "description": "A written status or scope qualifier paired with semantic color.",
- "html": "Passed with limits",
- "css": ".ds-badge { display: inline-flex; min-height: 24px; align-items: center; padding: 4px 10px; border: 1px solid rgba(252,211,77,.2); border-radius: 9999px; background: rgba(252,211,77,.1); color: #fde68a; font: 500 12px/1 -apple-system,BlinkMacSystemFont,\"SF Pro Text\",sans-serif; transition: background-color 150ms ease; } .ds-badge:hover { background: rgba(252,211,77,.16); } .ds-badge:focus-visible { outline: 2px solid rgba(243,173,61,.88); outline-offset: 2px; }"
- }
- ],
- "narrative": {
- "northStar": "The Evidence Bench",
- "overview": "CodeVetter feels like a precise local instrument: dark, quiet, dense enough for technical work, and candid about the strength of every claim. Warm amber marks the next intentional action. Semantic colors communicate verified, warning, or failure states only when the same meaning is also written in text or expressed with an icon. The interface should recede behind source identities, runtime results, and limitations.",
- "keyCharacteristics": [
- "Ink surfaces separated by restrained tonal steps and hairline borders.",
- "Compact native-feeling controls with generous focus treatment.",
- "Warm amber used sparingly for action, selection, and verification emphasis.",
- "Monospace reserved for paths, revisions, commands, and evidence identities.",
- "Every state remains understandable without color alone."
- ],
- "rules": [
- {
- "name": "The One Warm Voice Rule",
- "body": "Amber identifies intentional action or active verification context; it is not ambient decoration.",
- "section": "colors"
- },
- {
- "name": "The Written State Rule",
- "body": "Green, gold, rose, and blue may reinforce meaning, but a label or icon must communicate the same state.",
- "section": "colors"
- },
- {
- "name": "The Evidence Type Rule",
- "body": "Monospace signals data a user may compare, copy, or feed to another tool; prose and actions stay in the system sans.",
- "section": "typography"
- },
- {
- "name": "The Flat Evidence Rule",
- "body": "Evidence rows are stable nested planes; hover lift and decorative transform are reserved for actionable controls.",
- "section": "elevation"
- }
- ],
- "dos": [
- "Do lead with the action, exact identity, verdict, and limitation.",
- "Do reuse the established card, input, button, badge, and focus patterns.",
- "Do keep verification forms compact and preserve evidence below the action.",
- "Do provide loading, empty, error, limited, failed, and no-confidence states with plain-language labels."
- ],
- "donts": [
- "Don't present model opinion, topology, or a fixture as executable proof.",
- "Don't use amber across large decorative regions or for non-action accents.",
- "Don't communicate pass, warning, or failure through color alone.",
- "Don't add floating glass cards, hero typography, or agent theater to operating surfaces."
- ]
- }
-}
diff --git a/artifacts/design/after-1440.png b/artifacts/design/after-1440.png
deleted file mode 100644
index bba4650a..00000000
Binary files a/artifacts/design/after-1440.png and /dev/null differ
diff --git a/artifacts/design/after-390.png b/artifacts/design/after-390.png
deleted file mode 100644
index 7f51b7bb..00000000
Binary files a/artifacts/design/after-390.png and /dev/null differ
diff --git a/artifacts/design/after-768.png b/artifacts/design/after-768.png
deleted file mode 100644
index 943ae02f..00000000
Binary files a/artifacts/design/after-768.png and /dev/null differ
diff --git a/artifacts/design/content-cluster-after-1440.png b/artifacts/design/content-cluster-after-1440.png
deleted file mode 100644
index 1308002e..00000000
Binary files a/artifacts/design/content-cluster-after-1440.png and /dev/null differ
diff --git a/artifacts/design/content-cluster-after-390.png b/artifacts/design/content-cluster-after-390.png
deleted file mode 100644
index a09a037f..00000000
Binary files a/artifacts/design/content-cluster-after-390.png and /dev/null differ
diff --git a/artifacts/design/content-cluster-after-768.png b/artifacts/design/content-cluster-after-768.png
deleted file mode 100644
index 45c6c252..00000000
Binary files a/artifacts/design/content-cluster-after-768.png and /dev/null differ
diff --git a/artifacts/design/content-cluster-before.png b/artifacts/design/content-cluster-before.png
deleted file mode 100644
index 220f878f..00000000
Binary files a/artifacts/design/content-cluster-before.png and /dev/null differ
diff --git a/artifacts/design/crash-recovery-after-1440.png b/artifacts/design/crash-recovery-after-1440.png
deleted file mode 100644
index feb65e59..00000000
Binary files a/artifacts/design/crash-recovery-after-1440.png and /dev/null differ
diff --git a/artifacts/design/crash-recovery-after-390.png b/artifacts/design/crash-recovery-after-390.png
deleted file mode 100644
index a79edf58..00000000
Binary files a/artifacts/design/crash-recovery-after-390.png and /dev/null differ
diff --git a/artifacts/design/crash-recovery-after-768.png b/artifacts/design/crash-recovery-after-768.png
deleted file mode 100644
index 43ae1247..00000000
Binary files a/artifacts/design/crash-recovery-after-768.png and /dev/null differ
diff --git a/artifacts/design/crash-recovery-before-1440.png b/artifacts/design/crash-recovery-before-1440.png
deleted file mode 100644
index c8465e4c..00000000
Binary files a/artifacts/design/crash-recovery-before-1440.png and /dev/null differ
diff --git a/artifacts/design/crash-recovery-final-1440.jpg b/artifacts/design/crash-recovery-final-1440.jpg
deleted file mode 100644
index ff60facf..00000000
Binary files a/artifacts/design/crash-recovery-final-1440.jpg and /dev/null differ
diff --git a/artifacts/design/crash-recovery-final-390.jpg b/artifacts/design/crash-recovery-final-390.jpg
deleted file mode 100644
index 6725c920..00000000
Binary files a/artifacts/design/crash-recovery-final-390.jpg and /dev/null differ