Em dashes (—) and en dashes (–) had spread across comments, docs, tests, and a few UI strings, reading as AI-generated boilerplate rather than house style. Replaced each with punctuation matching its context: colon for explanatory clauses, comma for asides, plain hyphen for numeric/legal ranges (e.g. "21-23§"), "to"/"till" for date ranges, parentheses for paired-dash asides. messages/en.json and messages/sv.json were fixed by hand together to keep sv/en in sync. Left untouched where the dash is the functional subject rather than decorative punctuation: date-range-parser.ts's separator regex, charset-repair.ts's CP1252 byte-mapping table (and its test), the SIE encoding mojibake docs, generic-csv.ts's minus-sign normalizer, the agent system-prompt files that already instruct against em dashes, and a golden iXBRL test fixture compared byte-for-byte. Also fixes two bugs surfaced along the way: an off-by-one in ApiKeysPanel's scope-label split (a leftover from an earlier partial pass), and a charset-repair test that had lost the literal en-dash it exists to verify. Regenerated the agent atom seed migration (skills:generate) since 27 SKILL.md files changed. Added a CLAUDE.md rule against em/en dashes, with an explicit carve-out for the functional-dash cases above. Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
172 lines
6.3 KiB
TypeScript
172 lines
6.3 KiB
TypeScript
/**
|
|
* Atom-registry → MCP skill adapter.
|
|
*
|
|
* The in-app composer (lib/agent/composer/) writes to `agent_atom_registry`;
|
|
* each row points to a SKILL.md body on disk (under `.claude/skills/`). This
|
|
* loader hydrates those rows into `Skill` objects so the existing
|
|
* `gnubok_list_skills` / `gnubok_load_skill` tools can surface them to
|
|
* Claude.ai users with no further work (plan §13 MCP parity).
|
|
*
|
|
* Atom slugs match their registry id verbatim: "horizontal/swedish-vat",
|
|
* "vertical/konsult-it", "modifier/holding-ab". Workflow skills keep their
|
|
* flat slugs ("month-end-close") and never collide with atom slugs.
|
|
*
|
|
* Cached per-process. Reset between tests via `__resetAtomCache`.
|
|
*/
|
|
|
|
import { readFile } from 'node:fs/promises'
|
|
import { join } from 'node:path'
|
|
import type { SupabaseClient } from '@supabase/supabase-js'
|
|
import type { Skill, SkillTier } from './types'
|
|
|
|
interface AtomRegistryRow {
|
|
id: string
|
|
tier: 'horizontal' | 'vertical' | 'modifier'
|
|
title: string | null
|
|
description: string
|
|
sni_prefixes: string[] | null
|
|
body: string | null
|
|
body_path: string
|
|
}
|
|
|
|
let cache: Skill[] | null = null
|
|
|
|
/**
|
|
* SKILL.md frontmatter `description` fields are long, keyword-stuffed trigger
|
|
* lists authored for CLI skill-matching: not display copy (the project-accounting
|
|
* atom is ~1,100 chars). `gnubok_list_skills` and `gnubok_get_agent_briefing`
|
|
* surface them as one-line summaries, where the raw string gets truncated
|
|
* mid-sentence by the client. Trim to the first sentence, or a clean word
|
|
* boundary capped at `maxLen` with an ellipsis. The full text always stays
|
|
* available in the skill body (gnubok_load_skill). Idempotent for short input.
|
|
*/
|
|
export function toSummary(description: string, maxLen = 200): string {
|
|
const text = (description ?? '').trim().replace(/\s+/g, ' ')
|
|
if (text.length <= maxLen) return text
|
|
// Prefer the first sentence when it ends within the cap.
|
|
const firstStop = text.search(/[.!?](\s|$)/)
|
|
if (firstStop !== -1 && firstStop + 1 <= maxLen) return text.slice(0, firstStop + 1)
|
|
// Otherwise cut at the last word boundary before the cap and ellipsize so the
|
|
// truncation is ours (clean) rather than the client's (mid-word).
|
|
const slice = text.slice(0, maxLen)
|
|
const lastSpace = slice.lastIndexOf(' ')
|
|
return `${(lastSpace > 0 ? slice.slice(0, lastSpace) : slice).trimEnd()}…`
|
|
}
|
|
|
|
export async function loadAtomsAsSkills(supabase: SupabaseClient): Promise<Skill[]> {
|
|
if (cache) return cache
|
|
|
|
// `body` is read from the DB (seeded by scripts/generate-skill-bodies.ts): not
|
|
// from disk, so skills load identically on Vercel, Docker, and self-hosted.
|
|
// `mcp_exposed` is the curation kill-switch: only atoms flagged for the MCP
|
|
// surface reach Claude (swarm-* audit skills never become atoms in the first
|
|
// place; this guards against any future row that shouldn't be end-user-loadable).
|
|
const { data, error } = await supabase
|
|
.from('agent_atom_registry')
|
|
.select('id, tier, title, description, sni_prefixes, body, body_path')
|
|
.eq('is_active', true)
|
|
.eq('mcp_exposed', true)
|
|
.is('parent_atom_id', null) // list top-level skills only; references resolve via loadReferenceById
|
|
.order('id')
|
|
|
|
if (error) {
|
|
throw new Error(`Failed to load atom registry: ${error.message}`)
|
|
}
|
|
if (!data) {
|
|
cache = []
|
|
return cache
|
|
}
|
|
|
|
const rows = data as AtomRegistryRow[]
|
|
const out: Skill[] = []
|
|
|
|
for (const row of rows) {
|
|
let body = row.body ?? ''
|
|
if (!body) {
|
|
// Dev convenience: before `npm run skills:generate` has populated the DB,
|
|
// fall back to reading the SKILL.md from disk. In production the body is
|
|
// always seeded by the generated migration, so this path is never taken.
|
|
if (process.env.NODE_ENV !== 'production') {
|
|
try {
|
|
body = await readFile(join(process.cwd(), row.body_path), 'utf8')
|
|
} catch {
|
|
// fall through to the skip below
|
|
}
|
|
}
|
|
if (!body) {
|
|
console.warn(`[mcp-skills] atom ${row.id}: no body in DB and no on-disk fallback: skipping`)
|
|
continue
|
|
}
|
|
}
|
|
|
|
const sniRoot = row.sni_prefixes?.[0]?.split('.')[0]
|
|
const tags = [row.tier, ...(sniRoot ? [`sni-${sniRoot}`] : [])]
|
|
|
|
out.push({
|
|
slug: row.id,
|
|
name: row.title ?? row.id,
|
|
summary: toSummary(row.description),
|
|
tags,
|
|
body,
|
|
tier: row.tier as SkillTier,
|
|
})
|
|
}
|
|
|
|
cache = out
|
|
return cache
|
|
}
|
|
|
|
/**
|
|
* Resolve a single reference child (parent_atom_id IS NOT NULL) by exact id: * e.g. "horizontal/swedish-vat/vat-compliance-reference". References are
|
|
* deliberately excluded from `loadAtomsAsSkills` (the listed catalog), so this
|
|
* is the only path that surfaces them; gnubok_load_skill falls back here when a
|
|
* slug isn't a workflow or a top-level atom. Returns null for unknown ids,
|
|
* inactive rows, or rows the curation switch (mcp_exposed) has turned off.
|
|
*
|
|
* Not cached: references load rarely and on demand, so a per-id query is cheaper
|
|
* than holding ~90 reference bodies in the process between calls.
|
|
*/
|
|
export async function loadReferenceById(
|
|
supabase: SupabaseClient,
|
|
id: string,
|
|
): Promise<Skill | null> {
|
|
const { data, error } = await supabase
|
|
.from('agent_atom_registry')
|
|
.select('id, tier, title, description, sni_prefixes, body, body_path, is_active, mcp_exposed, parent_atom_id')
|
|
.eq('id', id)
|
|
.not('parent_atom_id', 'is', null)
|
|
.maybeSingle()
|
|
|
|
if (error) throw new Error(`Failed to load reference ${id}: ${error.message}`)
|
|
if (!data || data.is_active === false || data.mcp_exposed === false) return null
|
|
|
|
const row = data as AtomRegistryRow & { is_active: boolean; mcp_exposed: boolean }
|
|
let body = row.body ?? ''
|
|
if (!body) {
|
|
// Same dev fallback as loadAtomsAsSkills: before the seed migration runs,
|
|
// read the references/*.md straight off disk. Never taken in production.
|
|
if (process.env.NODE_ENV !== 'production') {
|
|
try {
|
|
body = await readFile(join(process.cwd(), row.body_path), 'utf8')
|
|
} catch {
|
|
return null
|
|
}
|
|
}
|
|
if (!body) return null
|
|
}
|
|
|
|
return {
|
|
slug: row.id,
|
|
name: row.title ?? row.id,
|
|
summary: toSummary(row.description),
|
|
tags: [row.tier, 'reference'],
|
|
body,
|
|
tier: row.tier as SkillTier,
|
|
}
|
|
}
|
|
|
|
/** Test-only: clear the module-level cache so the next call re-queries. */
|
|
export function __resetAtomCache(): void {
|
|
cache = null
|
|
}
|