You can not select more than 25 topics Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
svelte-kit-vice/scripts/morfo-vocabulary-check.ts

170 lines
5.8 KiB

/**
* Morfo vocabulary consistency check.
*
* Walks every morfo's `data` declarations and reports any attr with
* `values: string[]` whose set is NOT one of the canonical vocabularies
* declared in `CANONICAL_VOCABULARIES` (exported from morfo/schema.ts).
*
* Rationale (see sema_pre.md §5 + study.md §10):
* Sema selectors rely on consistent state vocabularies across components.
* If Accordion uses `data-state: ['open', 'closed']` then Dialog and Drawer
* must use the same tokens — not `visible|hidden` or `expanded|collapsed`.
*
* Strategy: a whitelist of known vocabularies. Any enum that doesn't exactly
* match one of them is flagged as a WARNING, not a hard error. Promote to
* error only when the vocabulary registry is mature enough that divergence
* is always a bug.
*
* Exit codes:
* 0 — all enums match a canonical vocabulary (or are novel but acceptable).
* 1 — at least one enum diverges and should be investigated.
*/
import { readdirSync } from 'node:fs';
import { fileURLToPath, pathToFileURL } from 'node:url';
import { dirname, join } from 'node:path';
import type { Morfo, MorfoPart } from '../src/uix/morfo/types';
import { validateMorfo, CANONICAL_VOCABULARIES } from '../src/uix/morfo/schema';
const __dirname = dirname(fileURLToPath(import.meta.url));
const MORFOS_DIR = join(__dirname, '..', 'src', 'uix', 'morfo', 'components');
async function loadMorfos(): Promise<Morfo[]> {
const files = readdirSync(MORFOS_DIR).filter(
(f) => f.endsWith('.ts') && !f.endsWith('.test.ts')
);
const out: Morfo[] = [];
for (const f of files) {
const url = pathToFileURL(join(MORFOS_DIR, f)).href;
const mod = (await import(url)) as Record<string, unknown>;
for (const v of Object.values(mod)) {
if (typeof v === 'object' && v !== null && 'kebab' in v && 'parts' in v) {
out.push(validateMorfo(v));
}
}
}
return out;
}
function flatParts(parts: readonly MorfoPart[]): MorfoPart[] {
const out: MorfoPart[] = [];
for (const p of parts) {
out.push(p);
if (p.parts && p.parts.length > 0) out.push(...flatParts(p.parts));
}
return out;
}
type Classification =
| { kind: 'match'; vocabulary: string }
| { kind: 'subset'; vocabulary: string; extra: string[] }
| { kind: 'superset'; vocabulary: string; missing: string[] }
| { kind: 'unknown' };
function classify(values: readonly string[]): Classification {
const valueSet = new Set(values);
for (const [name, canonical] of Object.entries(CANONICAL_VOCABULARIES)) {
const canonicalSet = new Set(canonical);
if (
canonicalSet.size === valueSet.size &&
[...valueSet].every((v) => canonicalSet.has(v))
) {
return { kind: 'match', vocabulary: name };
}
// Enum extends a canonical (added values)
if ([...canonicalSet].every((v) => valueSet.has(v))) {
const extra = [...valueSet].filter((v) => !canonicalSet.has(v));
if (extra.length > 0 && extra.length <= 3) {
return { kind: 'subset', vocabulary: name, extra };
}
}
// Enum shrinks a canonical (removed values)
if ([...valueSet].every((v) => canonicalSet.has(v))) {
const missing = [...canonicalSet].filter((v) => !valueSet.has(v));
if (missing.length > 0 && missing.length <= 3) {
return { kind: 'superset', vocabulary: name, missing };
}
}
}
return { kind: 'unknown' };
}
// ── Main ────────────────────────────────────────────────────────────────────
const morfos = await loadMorfos();
console.error(`Scanning ${morfos.length} morfo${morfos.length === 1 ? '' : 's'} for vocabulary divergence...`);
type Finding = {
morfo: string;
part: string;
attr: string;
values: readonly string[];
classification: Classification;
};
/**
* Attrs whose value sets are per-component by design, not shared vocabulary.
* Skipped by the consistency check — Dialog's `saved|cancelled|…` is correctly
* different from Toast's `dismissed|auto-timeout|action`.
*/
const PER_COMPONENT_ATTRS = new Set(['data-last-action']);
const findings: Finding[] = [];
for (const morfo of morfos) {
for (const part of flatParts(morfo.parts)) {
for (const data of part.data) {
if (!data.values) continue;
if (PER_COMPONENT_ATTRS.has(data.attr)) continue;
const classification = classify(data.values);
if (classification.kind !== 'match') {
findings.push({
morfo: morfo.kebab,
part: part.kebab,
attr: data.attr,
values: data.values,
classification
});
}
}
}
}
if (findings.length === 0) {
console.log(`All enum values match canonical vocabularies.`);
process.exit(0);
}
console.log('');
console.log(`Found ${findings.length} divergent enum${findings.length === 1 ? '' : 's'}:`);
console.log('');
for (const f of findings) {
const loc = `${f.morfo}.${f.part}.${f.attr}`;
const values = `[${f.values.join(', ')}]`;
switch (f.classification.kind) {
case 'subset':
console.log(
`WARN ${loc} ${values} extends "${f.classification.vocabulary}" with: ${f.classification.extra.join(', ')}`
);
break;
case 'superset':
console.log(
`WARN ${loc} ${values} shrinks "${f.classification.vocabulary}" missing: ${f.classification.missing.join(', ')}`
);
break;
case 'unknown':
console.log(
`WARN ${loc} ${values} no canonical vocabulary matches. Consider if this should use one of: ${Object.keys(CANONICAL_VOCABULARIES).join(', ')}`
);
break;
}
}
console.log('');
console.log(
`These are WARNINGS — novel vocabularies may be legitimate. Review each above: if the value set SHOULD match a canonical vocabulary, align it. If the component introduces a NEW canonical vocabulary, add it to CANONICAL_VOCABULARIES in src/uix/morfo/schema.ts.`
);
process.exit(0); // WARN only — don't fail CI yet (until vocabulary registry is mature)

Powered by TurnKey Linux.