Files
accounted/extensions/general/mcp-server/skills/atoms.ts
T
Jakob WennbergandClaude Sonnet 5 ec27228a8e style: remove em/en dashes repo-wide, add CLAUDE.md rule against them (#890)
Em dashes (—) and en dashes (–) had spread across comments, docs, tests,
and a few UI strings, reading as AI-generated boilerplate rather than
house style. Replaced each with punctuation matching its context: colon
for explanatory clauses, comma for asides, plain hyphen for numeric/legal
ranges (e.g. "21-23§"), "to"/"till" for date ranges, parentheses for
paired-dash asides. messages/en.json and messages/sv.json were fixed by
hand together to keep sv/en in sync.

Left untouched where the dash is the functional subject rather than
decorative punctuation: date-range-parser.ts's separator regex,
charset-repair.ts's CP1252 byte-mapping table (and its test), the SIE
encoding mojibake docs, generic-csv.ts's minus-sign normalizer, the
agent system-prompt files that already instruct against em dashes, and
a golden iXBRL test fixture compared byte-for-byte.

Also fixes two bugs surfaced along the way: an off-by-one in
ApiKeysPanel's scope-label split (a leftover from an earlier partial
pass), and a charset-repair test that had lost the literal en-dash it
exists to verify.

Regenerated the agent atom seed migration (skills:generate) since 27
SKILL.md files changed. Added a CLAUDE.md rule against em/en dashes,
with an explicit carve-out for the functional-dash cases above.

Co-authored-by: Claude Sonnet 5 <noreply@anthropic.com>
2026-07-04 15:58:06 +02:00

172 lines
6.3 KiB
TypeScript

/**
* Atom-registry → MCP skill adapter.
*
* The in-app composer (lib/agent/composer/) writes to `agent_atom_registry`;
* each row points to a SKILL.md body on disk (under `.claude/skills/`). This
* loader hydrates those rows into `Skill` objects so the existing
* `gnubok_list_skills` / `gnubok_load_skill` tools can surface them to
* Claude.ai users with no further work (plan §13 MCP parity).
*
* Atom slugs match their registry id verbatim: "horizontal/swedish-vat",
* "vertical/konsult-it", "modifier/holding-ab". Workflow skills keep their
* flat slugs ("month-end-close") and never collide with atom slugs.
*
* Cached per-process. Reset between tests via `__resetAtomCache`.
*/
import { readFile } from 'node:fs/promises'
import { join } from 'node:path'
import type { SupabaseClient } from '@supabase/supabase-js'
import type { Skill, SkillTier } from './types'
interface AtomRegistryRow {
id: string
tier: 'horizontal' | 'vertical' | 'modifier'
title: string | null
description: string
sni_prefixes: string[] | null
body: string | null
body_path: string
}
let cache: Skill[] | null = null
/**
* SKILL.md frontmatter `description` fields are long, keyword-stuffed trigger
* lists authored for CLI skill-matching: not display copy (the project-accounting
* atom is ~1,100 chars). `gnubok_list_skills` and `gnubok_get_agent_briefing`
* surface them as one-line summaries, where the raw string gets truncated
* mid-sentence by the client. Trim to the first sentence, or a clean word
* boundary capped at `maxLen` with an ellipsis. The full text always stays
* available in the skill body (gnubok_load_skill). Idempotent for short input.
*/
export function toSummary(description: string, maxLen = 200): string {
const text = (description ?? '').trim().replace(/\s+/g, ' ')
if (text.length <= maxLen) return text
// Prefer the first sentence when it ends within the cap.
const firstStop = text.search(/[.!?](\s|$)/)
if (firstStop !== -1 && firstStop + 1 <= maxLen) return text.slice(0, firstStop + 1)
// Otherwise cut at the last word boundary before the cap and ellipsize so the
// truncation is ours (clean) rather than the client's (mid-word).
const slice = text.slice(0, maxLen)
const lastSpace = slice.lastIndexOf(' ')
return `${(lastSpace > 0 ? slice.slice(0, lastSpace) : slice).trimEnd()}…`
}
export async function loadAtomsAsSkills(supabase: SupabaseClient): Promise<Skill[]> {
if (cache) return cache
// `body` is read from the DB (seeded by scripts/generate-skill-bodies.ts): not
// from disk, so skills load identically on Vercel, Docker, and self-hosted.
// `mcp_exposed` is the curation kill-switch: only atoms flagged for the MCP
// surface reach Claude (swarm-* audit skills never become atoms in the first
// place; this guards against any future row that shouldn't be end-user-loadable).
const { data, error } = await supabase
.from('agent_atom_registry')
.select('id, tier, title, description, sni_prefixes, body, body_path')
.eq('is_active', true)
.eq('mcp_exposed', true)
.is('parent_atom_id', null) // list top-level skills only; references resolve via loadReferenceById
.order('id')
if (error) {
throw new Error(`Failed to load atom registry: ${error.message}`)
}
if (!data) {
cache = []
return cache
}
const rows = data as AtomRegistryRow[]
const out: Skill[] = []
for (const row of rows) {
let body = row.body ?? ''
if (!body) {
// Dev convenience: before `npm run skills:generate` has populated the DB,
// fall back to reading the SKILL.md from disk. In production the body is
// always seeded by the generated migration, so this path is never taken.
if (process.env.NODE_ENV !== 'production') {
try {
body = await readFile(join(process.cwd(), row.body_path), 'utf8')
} catch {
// fall through to the skip below
}
}
if (!body) {
console.warn(`[mcp-skills] atom ${row.id}: no body in DB and no on-disk fallback: skipping`)
continue
}
}
const sniRoot = row.sni_prefixes?.[0]?.split('.')[0]
const tags = [row.tier, ...(sniRoot ? [`sni-${sniRoot}`] : [])]
out.push({
slug: row.id,
name: row.title ?? row.id,
summary: toSummary(row.description),
tags,
body,
tier: row.tier as SkillTier,
})
}
cache = out
return cache
}
/**
* Resolve a single reference child (parent_atom_id IS NOT NULL) by exact id: * e.g. "horizontal/swedish-vat/vat-compliance-reference". References are
* deliberately excluded from `loadAtomsAsSkills` (the listed catalog), so this
* is the only path that surfaces them; gnubok_load_skill falls back here when a
* slug isn't a workflow or a top-level atom. Returns null for unknown ids,
* inactive rows, or rows the curation switch (mcp_exposed) has turned off.
*
* Not cached: references load rarely and on demand, so a per-id query is cheaper
* than holding ~90 reference bodies in the process between calls.
*/
export async function loadReferenceById(
supabase: SupabaseClient,
id: string,
): Promise<Skill | null> {
const { data, error } = await supabase
.from('agent_atom_registry')
.select('id, tier, title, description, sni_prefixes, body, body_path, is_active, mcp_exposed, parent_atom_id')
.eq('id', id)
.not('parent_atom_id', 'is', null)
.maybeSingle()
if (error) throw new Error(`Failed to load reference ${id}: ${error.message}`)
if (!data || data.is_active === false || data.mcp_exposed === false) return null
const row = data as AtomRegistryRow & { is_active: boolean; mcp_exposed: boolean }
let body = row.body ?? ''
if (!body) {
// Same dev fallback as loadAtomsAsSkills: before the seed migration runs,
// read the references/*.md straight off disk. Never taken in production.
if (process.env.NODE_ENV !== 'production') {
try {
body = await readFile(join(process.cwd(), row.body_path), 'utf8')
} catch {
return null
}
}
if (!body) return null
}
return {
slug: row.id,
name: row.title ?? row.id,
summary: toSummary(row.description),
tags: [row.tier, 'reference'],
body,
tier: row.tier as SkillTier,
}
}
/** Test-only: clear the module-level cache so the next call re-queries. */
export function __resetAtomCache(): void {
cache = null
}