diff --git a/docs/architecture/10-interaction-and-feedback.md b/docs/architecture/10-interaction-and-feedback.md
index 15dc793..4d1b72a 100644
--- a/docs/architecture/10-interaction-and-feedback.md
+++ b/docs/architecture/10-interaction-and-feedback.md
@@ -180,6 +180,27 @@ contract; cite it, not the external source.)_
other shows the text visually with no live role. Two live regions would announce the same
message twice.
+**Resolved — feature-modal dismissal & initial focus.** A feature modal (Datasets, and
+later Settings/Chart Builder) is a **passive** `dialog-modal`: dismissed by the close
+button, Escape, or a backdrop click (a passive modal carries no in-flight transaction, so
+an outside click is a safe cancel — unlike the `alertdialog` confirm, where backdrop-dismiss
+is forbidden). `role="dialog"` + `aria-modal` + `aria-labelledby` the title; focus is trapped
+and **returns to the trigger** on close. Backdrop-dismiss stays correct even for the
+multi-view Datasets manager because an in-progress create/edit form is guarded separately by
+the discard prompt. **Initial focus depends on size** (APG dialog-modal): a large manager
+with semantic content (list + detail) focuses a **static title** (`tabindex="-1"`) so the
+content's start is perceived rather than skipped to the first control; a small form modal
+(Extract) focuses its **primary field**. _(Consulted via /council → WAI-ARIA APG
+`dialog-modal`. This bullet is the contract; cite it, not the APG file.)_
+
+**Resolved — an error names the right fix, not a boilerplate one.** Don't staple a generic
+remedy onto every failure. A missing dataset reference is **not** a JSON/spec syntax problem,
+so the preview gives it a tailored, fixable line — _"Dataset «X» not found. Create it from
+Datasets (⌘/Ctrl+K), or check the dataset name in your spec."_ — instead of the catch-all
+"check your JSON syntax" hint reserved for actual parse/Vega-Lite errors. The fix in the copy
+must match the actual cause (NN/g #9, GOV.UK error-message). The thrown
+`DatasetNotFoundError` carries `datasetName` so the surface can name it.
+
## 6. Motion & accessibility as default
Not features to add later — the baseline every surface is built on.
@@ -212,6 +233,7 @@ Not features to add later — the baseline every surface is built on.
- Treat loading/empty/error as three designed states for every data surface.
- Adopt the APG keyboard pattern for new widgets; route all global keys through the one
router.
+- Mark **optional** fields, not required ones (GOV.UK) — e.g. "Comment (optional)".
- Consult `/council` when this contract is silent — then record the answer back here.
**Don't**
@@ -224,3 +246,6 @@ Not features to add later — the baseline every surface is built on.
disclosure, the next step in the message.
- Don't invent a keyboard model, attach ad-hoc `window` listeners, or gate Escape behind
the typing check.
+- Don't ship a **dead disabled control** as a placeholder for an unbuilt feature — a
+ disabled button explains nothing and is skipped by assistive tech (GOV.UK, NN/g). Omit the
+ action until it works, then show it enabled (e.g. "Build Chart" appears with M4).
diff --git a/src/app/App.module.css b/src/app/App.module.css
index 21f80f6..1382370 100644
--- a/src/app/App.module.css
+++ b/src/app/App.module.css
@@ -61,6 +61,30 @@
flex: 1;
}
+/* Header text-button (Datasets, and future header actions). */
+.headerButton {
+ height: 32px;
+ padding: 0 var(--space-4);
+ border: var(--border-width) solid var(--border-strong);
+ border-radius: var(--radius);
+ background: transparent;
+ color: var(--text);
+ font: inherit;
+ font-size: 13px;
+ font-weight: 500;
+ cursor: pointer;
+ transition: background var(--dur-fast) var(--ease);
+}
+
+.headerButton:hover {
+ background: var(--layer-01);
+}
+
+.headerButton:focus-visible {
+ outline: 2px solid var(--focus);
+ outline-offset: 2px;
+}
+
.panes {
display: flex;
flex: 1 1 auto;
diff --git a/src/app/App.tsx b/src/app/App.tsx
index 21f6e8a..b27e509 100644
--- a/src/app/App.tsx
+++ b/src/app/App.tsx
@@ -1,10 +1,14 @@
+import { useEffect } from 'react';
import { ConfirmDialog } from './components/ConfirmDialog';
import { LivePreview } from './components/LivePreview';
+import { ModalShell } from './components/ModalShell';
import { ResizeHandle } from './components/ResizeHandle';
import { SnippetLibrary } from './components/SnippetLibrary';
import { SpecEditor } from './components/SpecEditor';
import { ThemeToggle } from './components/ThemeToggle';
import { Toaster } from './components/Toaster';
+import { openModal, setConfirm, toggleDatasets } from './modals/ModalCoordinator';
+import { confirm } from './stores/ConfirmStore';
import { usePanesStore } from './stores/PanesStore';
import styles from './App.module.css';
@@ -21,6 +25,25 @@ export function App() {
const libraryWidth = usePanesStore((s) => s.libraryWidth);
const previewWidth = usePanesStore((s) => s.previewWidth);
+ // Route the modal coordinator's discard prompt through the in-app confirm
+ // dialog (docs/architecture/03 → "The coordinator seam").
+ useEffect(() => {
+ setConfirm((message) => confirm({ title: 'Discard changes?', message, danger: true }));
+ }, []);
+
+ // Cmd/Ctrl+K toggles the Datasets manager (spec §05 → Opening). The full
+ // keyboard router (other shortcuts) lands in M6 (docs/architecture/04).
+ useEffect(() => {
+ const onKey = (e: KeyboardEvent) => {
+ if ((e.metaKey || e.ctrlKey) && (e.key === 'k' || e.key === 'K')) {
+ e.preventDefault();
+ toggleDatasets();
+ }
+ };
+ window.addEventListener('keydown', onKey);
+ return () => window.removeEventListener('keydown', onKey);
+ }, []);
+
return (
{/* Skip link (WCAG 2.4.1 / GOV.UK): the first focusable element, hidden
@@ -32,6 +55,15 @@ export function App() {
Astrolabe
v{__APP_VERSION__}
+
openModal('datasets')}
+ aria-keyshortcuts="Meta+K Control+K"
+ title="Datasets (⌘/Ctrl+K)"
+ >
+ Datasets
+
@@ -60,8 +92,12 @@ export function App() {
- {/* Global confirmation layer — sits above the (future) feature-modal
- shell so a discard-changes prompt can appear over an open modal. */}
+ {/* The one feature modal (Datasets / Extract / …), rendered from the
+ registry by the shared shell. At most one open at a time (spec §01C). */}
+
+
+ {/* Global confirmation layer — sits above the feature-modal shell so a
+ discard-changes prompt can appear over an open modal. */}
{/* Non-blocking notifications (failed saves, etc.) — top-right toasts,
diff --git a/src/app/components/DatasetsModal.module.css b/src/app/components/DatasetsModal.module.css
new file mode 100644
index 0000000..4f4bde3
--- /dev/null
+++ b/src/app/components/DatasetsModal.module.css
@@ -0,0 +1,437 @@
+/* Datasets manager — two-pane body inside the modal shell (spec §05 → Layout). */
+
+.manager {
+ display: grid;
+ grid-template-columns: 260px 1fr;
+ min-height: 0;
+ height: 100%;
+}
+
+/* --- List pane --- */
+.listPane {
+ display: flex;
+ flex-direction: column;
+ min-height: 0;
+ border-right: var(--border-width) solid var(--border);
+}
+
+.newButton {
+ flex: 0 0 auto;
+ margin: var(--space-4);
+ height: 40px;
+ padding: 0 var(--space-5);
+ border: var(--border-width) solid transparent;
+ border-radius: var(--radius);
+ background: var(--accent);
+ color: var(--accent-contrast);
+ font: inherit;
+ font-weight: 600;
+ cursor: pointer;
+ transition: background var(--dur-fast) var(--ease);
+}
+
+.newButton:hover {
+ background: var(--accent-hover);
+}
+
+.list {
+ list-style: none;
+ margin: 0;
+ padding: 0;
+ flex: 1 1 auto;
+ min-height: 0;
+ overflow: auto;
+ border-top: var(--border-width) solid var(--border);
+}
+
+.empty {
+ color: var(--text-secondary);
+ font-size: 13px;
+ line-height: 1.4;
+ padding: var(--space-5) var(--space-4);
+}
+
+.item {
+ display: flex;
+ align-items: center;
+ gap: var(--space-3);
+ padding: 0 var(--space-4);
+ border-left: 2px solid transparent;
+ transition: background var(--dur-fast) var(--ease);
+}
+
+.item + .item {
+ border-top: var(--border-width) solid var(--border);
+}
+
+.item:hover {
+ background: var(--layer-01);
+}
+
+.itemActive {
+ background: var(--layer-01);
+ border-left-color: var(--accent);
+}
+
+.itemMain {
+ flex: 1 1 auto;
+ min-width: 0;
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-1);
+ appearance: none;
+ border: none;
+ background: none;
+ padding: var(--space-3) 0;
+ font: inherit;
+ color: inherit;
+ text-align: left;
+ cursor: pointer;
+}
+
+.itemName {
+ font-size: 13px;
+ font-weight: 500;
+ white-space: nowrap;
+ overflow: hidden;
+ text-overflow: ellipsis;
+}
+
+.itemMeta {
+ font-size: 11px;
+ color: var(--text-secondary);
+ white-space: nowrap;
+ overflow: hidden;
+ text-overflow: ellipsis;
+}
+
+/* Usage badge — count of referencing snippets (spec §05 → List item). */
+.badge {
+ flex: 0 0 auto;
+ min-width: 18px;
+ height: 18px;
+ padding: 0 var(--space-2);
+ display: inline-flex;
+ align-items: center;
+ justify-content: center;
+ font-size: 11px;
+ font-weight: 600;
+ color: var(--text-secondary);
+ background: var(--layer-02);
+ border-radius: var(--radius);
+}
+
+/* --- Detail pane --- */
+.detailPane {
+ min-height: 0;
+ overflow: auto;
+}
+
+.detailEmpty {
+ display: flex;
+ align-items: center;
+ justify-content: center;
+ height: 100%;
+ color: var(--text-secondary);
+ font-size: 14px;
+ padding: var(--space-6);
+ text-align: center;
+}
+
+.detail,
+.form {
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-5);
+ padding: var(--space-5);
+}
+
+.detailHead {
+ display: flex;
+ align-items: flex-start;
+ justify-content: space-between;
+ gap: var(--space-4);
+ flex-wrap: wrap;
+}
+
+.detailName {
+ margin: 0;
+ font-size: 16px;
+ font-weight: 600;
+ color: var(--text);
+ word-break: break-word;
+}
+
+.detailActions,
+.formActions {
+ display: flex;
+ gap: var(--space-3);
+ flex-wrap: wrap;
+}
+
+.formActions {
+ justify-content: flex-end;
+}
+
+.action {
+ height: 32px;
+ padding: 0 var(--space-4);
+ border: var(--border-width) solid var(--border-strong);
+ border-radius: var(--radius);
+ background: transparent;
+ color: var(--text);
+ font: inherit;
+ font-size: 13px;
+ font-weight: 500;
+ cursor: pointer;
+ transition: background var(--dur-fast) var(--ease);
+}
+
+.action:hover:not(:disabled) {
+ background: var(--layer-01);
+}
+
+.action:disabled {
+ color: var(--text-placeholder);
+ border-color: var(--border);
+ cursor: not-allowed;
+}
+
+.action:focus-visible {
+ outline: 2px solid var(--focus);
+ outline-offset: 2px;
+}
+
+.primary {
+ background: var(--accent);
+ border-color: transparent;
+ color: var(--accent-contrast);
+ font-weight: 600;
+}
+
+.primary:hover:not(:disabled) {
+ background: var(--accent-hover);
+}
+
+.danger {
+ border-color: var(--border-strong);
+ color: var(--support-error);
+}
+
+.danger:hover:not(:disabled) {
+ background: var(--support-error);
+ color: var(--on-status);
+ border-color: transparent;
+}
+
+.comment {
+ margin: 0;
+ font-size: 13px;
+ line-height: 1.4;
+ color: var(--text-secondary);
+}
+
+.section {
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-3);
+}
+
+.sectionTitle {
+ margin: 0;
+ font-size: 11px;
+ font-weight: 600;
+ letter-spacing: 0.04em;
+ text-transform: uppercase;
+ color: var(--text-secondary);
+}
+
+.stats {
+ display: grid;
+ grid-template-columns: repeat(auto-fit, minmax(80px, 1fr));
+ gap: var(--space-4);
+ margin: 0;
+}
+
+.stats div {
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-1);
+}
+
+.stats dt {
+ font-size: 11px;
+ color: var(--text-secondary);
+}
+
+.stats dd {
+ margin: 0;
+ font-size: 14px;
+ font-weight: 500;
+ color: var(--text);
+}
+
+.columns {
+ list-style: none;
+ margin: 0;
+ padding: 0;
+ display: flex;
+ flex-direction: column;
+ border: var(--border-width) solid var(--border);
+}
+
+.column {
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--space-3);
+ padding: var(--space-2) var(--space-3);
+}
+
+.column + .column {
+ border-top: var(--border-width) solid var(--border);
+}
+
+.columnName {
+ font-size: 13px;
+ font-family: var(--font-mono);
+ overflow: hidden;
+ text-overflow: ellipsis;
+ white-space: nowrap;
+}
+
+.columnType {
+ flex: 0 0 auto;
+ font-size: 11px;
+ color: var(--text-secondary);
+ background: var(--layer-01);
+ padding: var(--space-1) var(--space-2);
+ border-radius: var(--radius);
+}
+
+.timestamps {
+ display: flex;
+ flex-wrap: wrap;
+ gap: var(--space-4);
+ font-size: 11px;
+ color: var(--text-secondary);
+}
+
+.preview {
+ margin: 0;
+ max-height: 220px;
+ overflow: auto;
+ padding: var(--space-3);
+ background: var(--layer-01);
+ border: var(--border-width) solid var(--border);
+ border-radius: var(--radius);
+ font-family: var(--font-mono);
+ font-size: 12px;
+ line-height: 1.5;
+ white-space: pre;
+}
+
+.muted {
+ margin: 0;
+ font-size: 13px;
+ color: var(--text-secondary);
+}
+
+.linked {
+ list-style: none;
+ margin: 0;
+ padding: 0;
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-2);
+}
+
+.linkButton {
+ appearance: none;
+ border: none;
+ background: none;
+ padding: 0;
+ font: inherit;
+ font-size: 13px;
+ color: var(--accent);
+ cursor: pointer;
+ text-align: left;
+}
+
+.linkButton:hover {
+ text-decoration: underline;
+}
+
+.linkButton:focus-visible {
+ outline: 2px solid var(--focus);
+ outline-offset: 2px;
+}
+
+/* --- Create / edit form --- */
+.field {
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-2);
+}
+
+.label {
+ font-size: 12px;
+ font-weight: 500;
+ color: var(--text-secondary);
+}
+
+.input,
+.textarea {
+ width: 100%;
+ padding: var(--space-3);
+ border: var(--border-width) solid var(--border-strong);
+ border-radius: var(--radius);
+ background: var(--bg);
+ color: var(--text);
+ font: inherit;
+ font-size: 13px;
+}
+
+.textarea {
+ font-family: var(--font-mono);
+ resize: vertical;
+ min-height: 160px;
+}
+
+.input:focus-visible,
+.textarea:focus-visible {
+ outline: 2px solid var(--focus);
+ outline-offset: -1px;
+}
+
+.input::placeholder,
+.textarea::placeholder {
+ color: var(--text-placeholder);
+}
+
+.detected {
+ display: flex;
+ align-items: center;
+ gap: var(--space-3);
+ margin-top: calc(-1 * var(--space-2));
+}
+
+.detectedBadge {
+ font-size: 11px;
+ font-weight: 600;
+ color: var(--text);
+ background: var(--layer-02);
+ padding: var(--space-1) var(--space-3);
+ border-radius: var(--radius);
+}
+
+.detectedHint {
+ font-size: 11px;
+ color: var(--text-secondary);
+}
+
+.formError {
+ margin: 0;
+ font-size: 13px;
+ color: var(--support-error);
+}
diff --git a/src/app/components/DatasetsModal.tsx b/src/app/components/DatasetsModal.tsx
new file mode 100644
index 0000000..5b87459
--- /dev/null
+++ b/src/app/components/DatasetsModal.tsx
@@ -0,0 +1,419 @@
+/**
+ * Datasets manager — the modal body (spec §05).
+ *
+ * A two-pane manager rendered inside the shared modal shell (App provides the
+ * backdrop, header, close, and focus trap). Left: a "New Dataset" action plus the
+ * dataset list, newest-modified first, each row carrying a source/rows/format/size
+ * meta line and a usage badge. Right: the selected dataset's detail, the
+ * create/edit form, or an empty prompt.
+ *
+ * State lives in DatasetStore; the bidirectional snippet↔dataset link is derived
+ * by scanning SnippetStore (docs/architecture/07 §4), so usage counts and Linked
+ * Snippets stay reactive without a stored back-pointer.
+ */
+
+import { useState } from 'react';
+import { useShallow } from 'zustand/react/shallow';
+import { datasetReference, type DataSource, type Dataset } from '@core/dataset';
+import { detectFormat, detectFormatFromUrl, type DataFormat } from '@core/format-detection';
+import { closeModal, resnapshot } from '../modals/ModalCoordinator';
+import { confirm } from '../stores/ConfirmStore';
+import { notify } from '../stores/NotificationStore';
+import { selectSelectedDataset, useDatasetStore, byModifiedDesc } from '../stores/DatasetStore';
+import { useSnippetStore } from '../stores/SnippetStore';
+import { SegmentedControl, type SegmentedOption } from './SegmentedControl';
+import styles from './DatasetsModal.module.css';
+
+/** Display label for a format (spec §05 → List item: JSON / CSV / TSV / TopoJSON). */
+function formatLabel(format: DataFormat): string {
+ return format === 'topojson' ? 'TopoJSON' : format.toUpperCase();
+}
+
+/** Human-readable byte size (B / KB / MB). */
+function humanBytes(bytes: number): string {
+ if (bytes < 1024) return `${bytes} B`;
+ const kb = bytes / 1024;
+ if (kb < 1024) return `${kb < 10 ? kb.toFixed(1) : Math.round(kb)} KB`;
+ const mb = kb / 1024;
+ return `${mb < 10 ? mb.toFixed(1) : Math.round(mb)} MB`;
+}
+
+/** Map of dataset name (lower-cased) → how many snippets reference it. */
+function usageByName(snippets: ReadonlyArray<{ datasetRefs: string[] }>): Map
{
+ const counts = new Map();
+ for (const s of snippets) {
+ for (const ref of s.datasetRefs) {
+ const key = ref.toLowerCase();
+ counts.set(key, (counts.get(key) ?? 0) + 1);
+ }
+ }
+ return counts;
+}
+
+const SOURCE_OPTIONS: ReadonlyArray> = [
+ { value: 'inline', label: 'Inline' },
+ { value: 'url', label: 'URL' },
+];
+
+export function DatasetsModal() {
+ const datasets = useDatasetStore(useShallow((s) => s.datasets));
+ const view = useDatasetStore((s) => s.view);
+ const selected = useDatasetStore(selectSelectedDataset);
+ const snippets = useSnippetStore(useShallow((s) => s.snippets));
+
+ const select = useDatasetStore((s) => s.select);
+ const startCreate = useDatasetStore((s) => s.startCreate);
+
+ const usage = usageByName(snippets);
+ const ordered = [...datasets].sort(byModifiedDesc);
+
+ const handleNew = () => {
+ startCreate();
+ resnapshot(); // baseline the discard check to the freshly-opened empty form
+ };
+
+ return (
+
+
+
+ + New Dataset
+
+
+ {ordered.length === 0 && (
+
+ No datasets yet — create one to reuse data across snippets.
+
+ )}
+ {ordered.map((d) => (
+ select(d.id)}
+ />
+ ))}
+
+
+
+
+ {view === 'new' || view === 'edit' ? (
+
+ ) : selected ? (
+
+ ) : (
+
Select a dataset or create a new one.
+ )}
+
+
+ );
+}
+
+function DatasetListItem({
+ dataset,
+ active,
+ usage,
+ onSelect,
+}: {
+ dataset: Dataset;
+ active: boolean;
+ usage: number;
+ onSelect: () => void;
+}) {
+ // Meta line: source ("URL" prefix), row count when known, format label, size.
+ const parts: string[] = [];
+ if (dataset.source === 'url') parts.push('URL');
+ if (dataset.rowCount !== null) parts.push(`${dataset.rowCount} rows`);
+ parts.push(formatLabel(dataset.format));
+ if (dataset.source !== 'url') parts.push(humanBytes(dataset.size));
+
+ return (
+
+
+ {dataset.name}
+ {parts.join(' · ')}
+
+ {usage > 0 && (
+
+ {usage}
+
+ )}
+
+ );
+}
+
+function DatasetDetail({
+ dataset,
+ snippets,
+}: {
+ dataset: Dataset;
+ snippets: ReadonlyArray<{ id: string; name: string; datasetRefs: string[] }>;
+}) {
+ const startEdit = useDatasetStore((s) => s.startEdit);
+ const remove = useDatasetStore((s) => s.remove);
+ const selectSnippet = useSnippetStore((s) => s.selectSnippet);
+ const [copied, setCopied] = useState(false);
+
+ const lower = dataset.name.toLowerCase();
+ const linked = snippets.filter((s) => s.datasetRefs.some((r) => r.toLowerCase() === lower));
+
+ const handleEdit = () => {
+ startEdit();
+ resnapshot();
+ };
+
+ const handleCopy = async () => {
+ const text = JSON.stringify(datasetReference(dataset.name), null, 2);
+ try {
+ await navigator.clipboard.writeText(text);
+ // Lightweight local feedback; the success-toast wiring is deferred to M6
+ // with the other success toasts (spec §05 → Actions).
+ setCopied(true);
+ setTimeout(() => setCopied(false), 1500);
+ } catch {
+ notify({
+ kind: 'error',
+ title: "Couldn't copy",
+ message: 'Your browser blocked clipboard access. Select and copy the reference manually.',
+ });
+ }
+ };
+
+ const handleDelete = async () => {
+ const ok = await confirm({
+ title: 'Delete dataset',
+ message: `Delete "${dataset.name}"? This cannot be undone.`,
+ confirmLabel: 'Delete',
+ danger: true,
+ });
+ // TODO (M6, spec §05): success toast on delete.
+ if (ok) remove(dataset.id);
+ };
+
+ return (
+
+
+
{dataset.name}
+
+ void handleCopy()}>
+ {copied ? 'Copied' : 'Copy Reference'}
+
+
+ Edit
+
+ {/* "Build Chart from dataset" (spec §05) lands enabled with the Chart
+ Builder in M4. Per council (GOV.UK / NN/g), we don't ship a dead
+ disabled control in the meantime — the action appears when it works. */}
+ void handleDelete()}
+ >
+ Delete
+
+
+
+
+ {dataset.comment &&
{dataset.comment}
}
+
+
+ Overview
+
+
+
Rows
+ {dataset.rowCount ?? 'N/A'}
+
+
+
Columns
+ {dataset.columnCount ?? 'N/A'}
+
+
+
Format
+ {formatLabel(dataset.format)}
+
+
+
Size
+ {dataset.source === 'url' ? 'N/A' : humanBytes(dataset.size)}
+
+
+
+ {dataset.columnTypes.length > 0 && (
+
+ {dataset.columnTypes.map((col) => (
+
+ {col.name}
+ {col.type}
+
+ ))}
+
+ )}
+
+
+ Created {new Date(dataset.created).toLocaleString()}
+ Modified {new Date(dataset.modified).toLocaleString()}
+
+
+
+
+ Preview
+ {previewText(dataset)}
+
+
+
+ Linked Snippets
+ {linked.length === 0 ? (
+ No snippets reference this dataset yet.
+ ) : (
+
+ {linked.map((s) => (
+
+ {
+ selectSnippet(s.id);
+ void closeModal();
+ }}
+ >
+ {s.name}
+
+
+ ))}
+
+ )}
+
+
+ );
+}
+
+/** A truncated rendering of the data: raw for csv/tsv/url, pretty JSON otherwise. */
+function previewText(dataset: Dataset): string {
+ const MAX = 2000;
+ let text: string;
+ if (dataset.source === 'url') {
+ text = String(dataset.data);
+ } else if (dataset.format === 'csv' || dataset.format === 'tsv') {
+ text = String(dataset.data);
+ } else {
+ try {
+ text = JSON.stringify(dataset.data, null, 2);
+ } catch {
+ text = String(dataset.data);
+ }
+ }
+ return text.length > MAX ? `${text.slice(0, MAX)}\n…` : text;
+}
+
+function DatasetFormView({ editing }: { editing: boolean }) {
+ const form = useDatasetStore((s) => s.form);
+ const formError = useDatasetStore((s) => s.formError);
+ const updateForm = useDatasetStore((s) => s.updateForm);
+ const cancelForm = useDatasetStore((s) => s.cancelForm);
+ const save = useDatasetStore((s) => s.save);
+
+ // Live format/source hint from the current input (spec §05 → Auto-detection).
+ const detected =
+ form.source === 'url'
+ ? { format: detectFormatFromUrl(form.input.trim()), confidence: 'url' as const }
+ : detectFormat(form.input);
+
+ const handleSave = () => {
+ if (save()) resnapshot(); // committed — re-baseline so a later close won't prompt
+ };
+
+ const handleCancel = () => {
+ cancelForm();
+ resnapshot();
+ };
+
+ return (
+
+
{editing ? 'Edit dataset' : 'New dataset'}
+
+
+ Name
+ updateForm({ name: e.target.value })}
+ placeholder="e.g. Sales 2024"
+ />
+
+
+
+ Source
+ updateForm({ source })}
+ />
+
+
+
+ {form.source === 'url' ? 'URL' : 'Data'}
+ {form.source === 'url' ? (
+ updateForm({ input: e.target.value })}
+ placeholder="https://example.com/data.csv"
+ />
+ ) : (
+
+
+ {form.input.trim() !== '' && (
+
+
+ {detected.format ? formatLabel(detected.format) : 'Unrecognized'}
+
+ {form.source !== 'url' && 'confidence' in detected && (
+ {detected.confidence} confidence
+ )}
+
+ )}
+
+
+ Comment (optional)
+ updateForm({ comment: e.target.value })}
+ placeholder="Notes about this dataset"
+ />
+
+
+ {formError && (
+
+ {formError}
+
+ )}
+
+
+
+ Cancel
+
+
+ {editing ? 'Save changes' : 'Create dataset'}
+
+
+
+ );
+}
diff --git a/src/app/components/ExtractModal.module.css b/src/app/components/ExtractModal.module.css
new file mode 100644
index 0000000..94592eb
--- /dev/null
+++ b/src/app/components/ExtractModal.module.css
@@ -0,0 +1,114 @@
+/* Extract-to-Dataset — a single-form modal body (spec §03F). */
+
+.extract {
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-5);
+ padding: var(--space-5);
+ min-width: 0;
+}
+
+.intro {
+ margin: 0;
+ font-size: 13px;
+ line-height: 1.4;
+ color: var(--text-secondary);
+}
+
+.muted {
+ margin: 0;
+ padding: var(--space-5);
+ font-size: 14px;
+ color: var(--text-secondary);
+}
+
+.field {
+ display: flex;
+ flex-direction: column;
+ gap: var(--space-2);
+}
+
+.label {
+ font-size: 12px;
+ font-weight: 500;
+ color: var(--text-secondary);
+}
+
+.input {
+ width: 100%;
+ padding: var(--space-3);
+ border: var(--border-width) solid var(--border-strong);
+ border-radius: var(--radius);
+ background: var(--bg);
+ color: var(--text);
+ font: inherit;
+ font-size: 13px;
+}
+
+.input:focus-visible {
+ outline: 2px solid var(--focus);
+ outline-offset: -1px;
+}
+
+.input::placeholder {
+ color: var(--text-placeholder);
+}
+
+.preview {
+ margin: 0;
+ max-height: 240px;
+ overflow: auto;
+ padding: var(--space-3);
+ background: var(--layer-01);
+ border: var(--border-width) solid var(--border);
+ border-radius: var(--radius);
+ font-family: var(--font-mono);
+ font-size: 12px;
+ line-height: 1.5;
+ white-space: pre;
+}
+
+.error {
+ margin: 0;
+ font-size: 13px;
+ color: var(--support-error);
+}
+
+.actions {
+ display: flex;
+ justify-content: flex-end;
+ gap: var(--space-3);
+}
+
+.action {
+ height: 36px;
+ padding: 0 var(--space-5);
+ border: var(--border-width) solid var(--border-strong);
+ border-radius: var(--radius);
+ background: transparent;
+ color: var(--text);
+ font: inherit;
+ font-weight: 500;
+ cursor: pointer;
+ transition: background var(--dur-fast) var(--ease);
+}
+
+.action:hover {
+ background: var(--layer-01);
+}
+
+.action:focus-visible {
+ outline: 2px solid var(--focus);
+ outline-offset: 2px;
+}
+
+.primary {
+ background: var(--accent);
+ border-color: transparent;
+ color: var(--accent-contrast);
+ font-weight: 600;
+}
+
+.primary:hover {
+ background: var(--accent-hover);
+}
diff --git a/src/app/components/ExtractModal.tsx b/src/app/components/ExtractModal.tsx
new file mode 100644
index 0000000..d876194
--- /dev/null
+++ b/src/app/components/ExtractModal.tsx
@@ -0,0 +1,90 @@
+/**
+ * Extract-to-Dataset — the modal body (spec §03F).
+ *
+ * Shows a read-only preview of the active snippet draft's inline data and asks
+ * for a dataset name. On confirm it saves the data as a new dataset and rewrites
+ * the draft to reference it by name (logic in ExtractStore), then force-closes
+ * (the commit is the user's confirmation, so no discard prompt). Cancel leaves
+ * the spec unchanged.
+ */
+
+import { useShallow } from 'zustand/react/shallow';
+import { closeModal } from '../modals/ModalCoordinator';
+import { useExtractStore } from '../stores/ExtractStore';
+import styles from './ExtractModal.module.css';
+
+const MAX_PREVIEW = 1500;
+
+function previewOf(values: unknown): string {
+ let text: string;
+ try {
+ text = typeof values === 'string' ? values : JSON.stringify(values, null, 2);
+ } catch {
+ text = String(values);
+ }
+ return text.length > MAX_PREVIEW ? `${text.slice(0, MAX_PREVIEW)}\n…` : text;
+}
+
+export function ExtractModal() {
+ const { name, source, error } = useExtractStore(
+ useShallow((s) => ({ name: s.name, source: s.source, error: s.error })),
+ );
+ const setName = useExtractStore((s) => s.setName);
+ const runConfirm = useExtractStore((s) => s.confirm);
+
+ const handleCreate = () => {
+ if (runConfirm()) void closeModal(true);
+ };
+
+ if (!source) {
+ return This snippet has no inline data to extract.
;
+ }
+
+ return (
+
+
+ Save this snippet’s inline data as a reusable dataset. The spec will be rewritten to
+ reference it by name.
+
+
+
+ Dataset name
+ setName(e.target.value)}
+ onKeyDown={(e) => {
+ if (e.key === 'Enter') handleCreate();
+ }}
+ placeholder="e.g. Sales 2024"
+ autoFocus
+ />
+
+
+
+
Data to extract ({source.format.toUpperCase()})
+
{previewOf(source.values)}
+
+
+ {error && (
+
+ {error}
+
+ )}
+
+
+ void closeModal()}>
+ Cancel
+
+
+ Create dataset
+
+
+
+ );
+}
diff --git a/src/app/components/LivePreview.tsx b/src/app/components/LivePreview.tsx
index 462de5a..2fffdf7 100644
--- a/src/app/components/LivePreview.tsx
+++ b/src/app/components/LivePreview.tsx
@@ -16,12 +16,14 @@
*/
import { useEffect, useRef } from 'react';
+import { useShallow } from 'zustand/react/shallow';
import type { VisualizationSpec } from 'vega-embed';
import type { FitMode } from '@core/rendering';
-import { prepareSpecForRender } from '@core/rendering';
+import { DatasetNotFoundError, prepareSpecForRender } from '@core/rendering';
import { chartConfigFor } from '@core/vega-themes';
import { renderSpec, type RenderHandle } from '../services/chart-renderer';
import { useAppStore } from '../stores/AppStore';
+import { useDatasetStore } from '../stores/DatasetStore';
import { usePreviewStore } from '../stores/PreviewStore';
import { selectShownText, useSnippetStore } from '../stores/SnippetStore';
import { SegmentedControl, type SegmentedOption } from './SegmentedControl';
@@ -73,6 +75,9 @@ export function LivePreview() {
const shownText = useSnippetStore(selectShownText);
const fitMode = useAppStore((s) => s.previewFitMode);
const uiTheme = useAppStore((s) => s.uiTheme);
+ // Datasets feed reference resolution (spec §04 step 1). Re-rendering on a
+ // dataset change keeps a referencing chart live as its data is edited.
+ const datasets = useDatasetStore(useShallow((s) => s.datasets));
const error = usePreviewStore((s) => s.error);
const setError = usePreviewStore((s) => s.setError);
@@ -107,7 +112,7 @@ export function LivePreview() {
const mine = ++generationRef.current;
try {
- const prepared = prepareSpecForRender(parsed, { fitMode });
+ const prepared = prepareSpecForRender(parsed, { fitMode, datasets });
const config = chartConfigFor(uiTheme);
handleRef.current?.destroy();
handleRef.current = null;
@@ -126,17 +131,27 @@ export function LivePreview() {
setError(null);
} catch (e) {
if (mine === generationRef.current) {
- setError(
- `Rendering error: ${(e as Error).message}. ` +
- `Check your JSON syntax and that the spec is valid Vega-Lite.`,
- );
+ // A missing dataset reference is not a JSON/spec problem, so it gets a
+ // tailored, fixable message instead of the generic syntax hint (council:
+ // GOV.UK error-message + NN/g #9 — name the problem, give the real fix).
+ if (e instanceof DatasetNotFoundError) {
+ setError(
+ `Dataset "${e.datasetName}" not found. Create it from Datasets ` +
+ `(⌘/Ctrl+K), or check the dataset name in your spec.`,
+ );
+ } else {
+ setError(
+ `Rendering error: ${(e as Error).message}. ` +
+ `Check your JSON syntax and that the spec is valid Vega-Lite.`,
+ );
+ }
}
}
})();
}, RENDER_DEBOUNCE_MS);
return () => clearTimeout(timer);
- }, [shownText, fitMode, uiTheme, setError]);
+ }, [shownText, fitMode, uiTheme, datasets, setError]);
// Re-fit the chart when its container resizes (e.g. a pane drag). Vega doesn't
// observe the element, so we do: one observer on the stable host node for the
diff --git a/src/app/components/ModalShell.module.css b/src/app/components/ModalShell.module.css
new file mode 100644
index 0000000..05963e5
--- /dev/null
+++ b/src/app/components/ModalShell.module.css
@@ -0,0 +1,95 @@
+/* Feature-modal shell (docs/architecture/03 → Layer 3). Sits below the confirm
+ layer (z 1000) so a discard prompt can appear above an open modal. */
+
+.backdrop {
+ position: fixed;
+ inset: 0;
+ z-index: 900;
+ display: flex;
+ align-items: center;
+ justify-content: center;
+ padding: var(--space-5);
+ background: rgb(0 0 0 / 0.5);
+}
+
+.modal {
+ display: flex;
+ flex-direction: column;
+ min-height: 0;
+ background: var(--layer-01);
+ border: var(--border-width) solid var(--border);
+ border-radius: var(--radius);
+ box-shadow: 0 2px 12px rgb(0 0 0 / 0.3);
+}
+
+/* Large: the two-pane managers (Datasets). Definite height so inner panes scroll. */
+.large {
+ width: min(960px, 94vw);
+ height: min(700px, 88vh);
+}
+
+/* Small: single-form modals (Extract). Grows with content up to a cap. */
+.small {
+ width: min(560px, 92vw);
+ max-height: 88vh;
+}
+
+.header {
+ flex: 0 0 auto;
+ display: flex;
+ align-items: center;
+ justify-content: space-between;
+ gap: var(--space-4);
+ height: var(--header-height);
+ padding: 0 var(--space-5);
+ border-bottom: var(--border-width) solid var(--border);
+}
+
+.title {
+ margin: 0;
+ font-size: 16px;
+ font-weight: 600;
+ color: var(--text);
+}
+
+/* The title is only a programmatic focus target (large-manager initial focus);
+ it shouldn't paint a ring the way a keyboard-reached control would. */
+.title:focus {
+ outline: none;
+}
+
+.close {
+ flex: 0 0 auto;
+ display: flex;
+ align-items: center;
+ justify-content: center;
+ width: 32px;
+ height: 32px;
+ border: none;
+ background: none;
+ color: var(--text-secondary);
+ font-size: 22px;
+ line-height: 1;
+ cursor: pointer;
+ border-radius: var(--radius);
+ transition: background var(--dur-fast) var(--ease);
+}
+
+.close:hover {
+ background: var(--layer-02);
+ color: var(--text);
+}
+
+.close:focus-visible {
+ outline: 2px solid var(--focus);
+ outline-offset: 2px;
+}
+
+/* The modal body. Background steps to --bg so the panes read as the work surface
+ against the --layer-01 chrome. Fills the remaining height; inner content scrolls. */
+.body {
+ flex: 1 1 auto;
+ min-height: 0;
+ overflow: auto;
+ background: var(--bg);
+}
diff --git a/src/app/components/ModalShell.tsx b/src/app/components/ModalShell.tsx
new file mode 100644
index 0000000..a6dab46
--- /dev/null
+++ b/src/app/components/ModalShell.tsx
@@ -0,0 +1,81 @@
+/**
+ * Modal shell (docs/architecture/03 → Layer 3).
+ *
+ * Renders exactly ONE modal — whichever `activeModal` names — inside a single
+ * reusable chrome: backdrop, header (title + close), and a focus trap. The
+ * modal's registered `component` fills the body. This is the only place a modal
+ * name maps to a view (` ` from the registry), so there is no
+ * `name === 'datasets' && ` chain anywhere.
+ *
+ * Dismissal is uniform for these passive feature modals: the close button,
+ * Escape, or a backdrop click — never a click inside the body (which stops
+ * propagation). Each modal owns its own action buttons; the shell stays generic.
+ */
+
+import { useAppStore } from '../stores/AppStore';
+import { getModalConfig, getModalTitle } from '../modals/modal-registry';
+import { closeModal } from '../modals/ModalCoordinator';
+import { useFocusTrap } from '../hooks/useFocusTrap';
+import styles from './ModalShell.module.css';
+
+export function ModalShell() {
+ const name = useAppStore((s) => s.activeModal);
+ const config = getModalConfig(name);
+
+ const isLarge = name === 'datasets';
+
+ // Move focus into the modal on open, return it to the trigger on close. For a
+ // large manager (list + detail), APG dialog-modal advises focusing a static
+ // element at the top — the title — so the content's start is perceived rather
+ // than jumping past it to the first control. Small form modals keep their
+ // first-field focus (Extract autofocuses its name input).
+ const modalRef = useFocusTrap(
+ name !== null,
+ isLarge ? '#modal-title' : undefined,
+ );
+
+ if (!config) return null;
+ const Body = config.component;
+
+ return (
+ void closeModal()}
+ onKeyDown={(e) => {
+ if (e.key === 'Escape') {
+ e.stopPropagation();
+ void closeModal();
+ }
+ }}
+ >
+
e.stopPropagation()}
+ >
+
+ {/* tabIndex -1 makes the title a programmatic focus target for the
+ large-manager initial focus (APG dialog-modal), without adding it to
+ the Tab order. */}
+
+ {getModalTitle(name)}
+
+ void closeModal()}
+ >
+ ×
+
+
+
+
+
+
+
+ );
+}
diff --git a/src/app/components/SnippetLibrary.module.css b/src/app/components/SnippetLibrary.module.css
index 769121a..fb0d01d 100644
--- a/src/app/components/SnippetLibrary.module.css
+++ b/src/app/components/SnippetLibrary.module.css
@@ -129,6 +129,18 @@
text-overflow: ellipsis;
}
+/* Linked-datasets indicator (spec §02): icon + count, meaning carried by the
+ icon + accessible label, not colour. Pushed to the row's trailing edge. */
+.datasets {
+ flex: 0 0 auto;
+ margin-left: auto;
+ display: inline-flex;
+ align-items: center;
+ gap: var(--space-1);
+ font-size: 11px;
+ color: var(--text-secondary);
+}
+
.delete {
flex: 0 0 auto;
align-self: center;
diff --git a/src/app/components/SnippetLibrary.tsx b/src/app/components/SnippetLibrary.tsx
index 5f297e7..c326cd7 100644
--- a/src/app/components/SnippetLibrary.tsx
+++ b/src/app/components/SnippetLibrary.tsx
@@ -14,6 +14,36 @@ import { confirm } from '../stores/ConfirmStore';
import { useSnippetStore } from '../stores/SnippetStore';
import styles from './SnippetLibrary.module.css';
+/** A small database glyph for the linked-datasets indicator (icons keep their own
+ * rounded geometry per the design tokens). Inherits `currentColor` so it themes. */
+function DatasetIcon() {
+ return (
+
+
+
+
+
+ );
+}
+
/** Compact relative date for the list (full date formatting lands in M5). */
function relativeDate(iso: string): string {
const then = new Date(iso);
@@ -97,6 +127,18 @@ export function SnippetLibrary() {
)}
{size ? `${date} · ${size}` : date}
+ {/* Linked datasets (spec §02): the snippet side of the
+ bidirectional link, maintained on publish via datasetRefs. */}
+ {s.datasetRefs.length > 0 && (
+
+
+ {s.datasetRefs.length}
+
+ )}
s.activeSnippetId !== null && hasInlineData(s.draftText),
+ );
const handlePublish = () => {
if (!useSnippetStore.getState().activeSnippetId) return;
@@ -85,6 +92,16 @@ function EditorToolbar() {
+ {canExtract && (
+ openModal('extract')}
+ title="Extract inline data into a reusable dataset"
+ >
+ Extract to Dataset
+
+ )}
= ['json', 'csv', 'tsv', 'topojson'];
+
+function asFormat(v: unknown): DataFormat {
+ return typeof v === 'string' && (FORMATS as string[]).includes(v) ? (v as DataFormat) : 'json';
+}
+
+function asSource(v: unknown): DataSource {
+ return v === 'url' ? 'url' : 'inline';
+}
+
+function asCount(v: unknown): number | null {
+ return typeof v === 'number' && Number.isFinite(v) ? v : null;
+}
+
+/** Upgrade a raw stored record to the current Dataset shape. */
+export function migrateDataset(raw: unknown): Dataset {
+ const r = { ...(raw as Record) };
+ return {
+ ...r,
+ id: typeof r.id === 'number' ? r.id : Number(r.id),
+ version: CURRENT_DATASET_VERSION,
+ name: typeof r.name === 'string' ? r.name : 'Untitled',
+ data: r.data,
+ format: asFormat(r.format),
+ source: asSource(r.source),
+ comment: typeof r.comment === 'string' ? r.comment : '',
+ rowCount: asCount(r.rowCount),
+ columnCount: asCount(r.columnCount),
+ columns: Array.isArray(r.columns) ? (r.columns as string[]) : [],
+ columnTypes: Array.isArray(r.columnTypes)
+ ? (r.columnTypes as Array<{ name: string; type: ColumnType }>)
+ : [],
+ size: typeof r.size === 'number' ? r.size : 0,
+ created: typeof r.created === 'string' ? r.created : new Date(0).toISOString(),
+ modified: typeof r.modified === 'string' ? r.modified : new Date(0).toISOString(),
+ };
+}
diff --git a/src/app/infrastructure/dataset-store.ts b/src/app/infrastructure/dataset-store.ts
new file mode 100644
index 0000000..332f6e9
--- /dev/null
+++ b/src/app/infrastructure/dataset-store.ts
@@ -0,0 +1,28 @@
+/**
+ * Dataset persistence adapter (docs/architecture/02, spec §09E).
+ *
+ * The typed seam between the dataset store and IndexedDB's high-capacity
+ * `datasets` object store (separate from snippets so large payloads live in a
+ * tier suited to them). Exposes plain async functions returning domain `Dataset`
+ * objects and migrates every record on read.
+ */
+
+import { CURRENT_DATASET_VERSION, type Dataset } from '@core/dataset';
+import { DATASETS_STORE, del, getAll, put } from './db';
+import { migrateDataset } from './dataset-migrations';
+
+/** Load every dataset, upgrading each record to the current shape. */
+export async function loadDatasets(): Promise {
+ const records = await getAll(DATASETS_STORE);
+ return records.map(migrateDataset);
+}
+
+/** Persist a dataset at the current schema version. Propagates failures. */
+export async function saveDataset(dataset: Dataset): Promise {
+ await put(DATASETS_STORE, { ...dataset, version: CURRENT_DATASET_VERSION });
+}
+
+/** Permanently remove a dataset by id. */
+export async function deleteDataset(id: number): Promise {
+ await del(DATASETS_STORE, id);
+}
diff --git a/src/app/modals/ModalCoordinator.ts b/src/app/modals/ModalCoordinator.ts
new file mode 100644
index 0000000..a61e8a8
--- /dev/null
+++ b/src/app/modals/ModalCoordinator.ts
@@ -0,0 +1,78 @@
+/**
+ * Modal coordinator (docs/architecture/03 → Layer 2).
+ *
+ * Owns the modal lifecycle: open (close any previous — at most one at a time),
+ * init transient state, snapshot for unsaved-change detection, and keep the URL
+ * in sync. Framework-light — pure functions over `useAppStore` — so it's unit
+ * testable without a DOM. The only coordinator-internal state is the change
+ * snapshot, a module-local variable (no component reads it).
+ */
+
+import { useAppStore } from '../stores/AppStore';
+import type { ModalName } from './types';
+import { getModalConfig } from './modal-registry';
+import { clearModalFromUrl, syncModalToUrl } from './UrlStateSync';
+
+/** Discard-confirmation seam — wired to ConfirmStore at startup (see App). */
+let confirmDiscard: (message: string) => Promise = () => Promise.resolve(true);
+export const setConfirm = (fn: typeof confirmDiscard): void => {
+ confirmDiscard = fn;
+};
+
+/** The getState() JSON captured at open (or re-baselined), compared on close. */
+let stateSnapshot: string | null = null;
+
+function snapshotOf(name: ModalName | null): string | null {
+ const get = getModalConfig(name)?.getState;
+ if (!get) return null;
+ const state = get();
+ return state === null ? null : JSON.stringify(state);
+}
+
+/** Open `name`, optionally with a sub-target (dataset id, source key). */
+export function openModal(name: ModalName, arg?: string): void {
+ useAppStore.getState().setActiveModal(name);
+ getModalConfig(name)?.init?.(arg);
+ stateSnapshot = snapshotOf(name);
+ syncModalToUrl(name, arg);
+}
+
+/**
+ * Re-baseline the change snapshot to the modal's current state. Call after a
+ * commit (the form was saved) or when a form view opens, so subsequent
+ * unsaved-change detection compares against the right baseline rather than the
+ * state at modal-open (a multi-view manager opens its form after open).
+ */
+export function resnapshot(): void {
+ stateSnapshot = snapshotOf(useAppStore.getState().activeModal);
+}
+
+/** True when the active modal's editable state differs from its baseline. */
+export function hasUnsavedChanges(): boolean {
+ const name = useAppStore.getState().activeModal;
+ if (!name) return false;
+ const current = snapshotOf(name);
+ if (current === null) return false; // nothing editable open ⇒ nothing to lose
+ return current !== stateSnapshot;
+}
+
+/** Close the active modal. Prompts on unsaved changes unless `force`. */
+export async function closeModal(force = false): Promise {
+ const name = useAppStore.getState().activeModal;
+ if (!name) return;
+
+ if (!force && hasUnsavedChanges()) {
+ const ok = await confirmDiscard('Discard your unsaved changes? This cannot be undone.');
+ if (!ok) return;
+ }
+
+ clearModalFromUrl(name);
+ useAppStore.getState().setActiveModal(null);
+ stateSnapshot = null;
+}
+
+/** Cmd/Ctrl+K toggle for the Datasets manager (spec §05 → Opening). */
+export function toggleDatasets(): void {
+ if (useAppStore.getState().activeModal === 'datasets') void closeModal();
+ else openModal('datasets');
+}
diff --git a/src/app/modals/UrlStateSync.ts b/src/app/modals/UrlStateSync.ts
new file mode 100644
index 0000000..3b8ab73
--- /dev/null
+++ b/src/app/modals/UrlStateSync.ts
@@ -0,0 +1,26 @@
+/**
+ * Modal ↔ URL hash sync (docs/architecture/03 → "URL & Keyboard Integration").
+ *
+ * The coordinator calls these so a navigable modal (Datasets, Settings, Chart
+ * Builder) becomes a shareable / back-navigable location, e.g.
+ * `#datasets/dataset-`. Full hash routing — view-state restore on load,
+ * Back/Forward — is milestone M6 (spec §01E, docs/architecture/04). Until then
+ * these are intentional no-ops so the coordinator's shape is final and M6 fills
+ * the bodies in without the call sites changing.
+ */
+
+import type { ModalName } from './types';
+import { getModalConfig } from './modal-registry';
+
+/** Reflect an open navigable modal (and optional sub-target) in the URL hash. */
+export function syncModalToUrl(name: ModalName, _arg?: string): void {
+ if (!getModalConfig(name)?.isUrlNavigable) return;
+ // TODO (M6, spec §01E): write `#${name}` / `#${name}/dataset-${arg}` to the
+ // hash via the routing layer (docs/architecture/04).
+}
+
+/** Return the hash to the underlying workspace when a navigable modal closes. */
+export function clearModalFromUrl(name: ModalName): void {
+ if (!getModalConfig(name)?.isUrlNavigable) return;
+ // TODO (M6, spec §01E): restore the pre-modal workspace hash.
+}
diff --git a/src/app/modals/modal-registry.ts b/src/app/modals/modal-registry.ts
new file mode 100644
index 0000000..7bd384d
--- /dev/null
+++ b/src/app/modals/modal-registry.ts
@@ -0,0 +1,73 @@
+/**
+ * Modal registry (docs/architecture/03 → Layer 1).
+ *
+ * One metadata entry per feature modal — the single source of truth the
+ * coordinator and shell read, so adding a modal is one entry plus its component
+ * rather than edits scattered across the shell, URL sync, and close logic.
+ *
+ * Divergence from the arch sketch's optional generic footer: each modal renders
+ * its OWN action row inside its body (the Datasets manager is multi-view, so a
+ * single shell-level Save/Cancel doesn't fit). The registry therefore omits
+ * `hasError`/`getError` (validity is the modal's own concern) and keeps only
+ * `getState` for unsaved-change detection on close. The registry is a partial
+ * map: modals land milestone by milestone (settings/about/donate → M5/M6,
+ * chartBuilder → M4), so only the implemented ones are registered here.
+ */
+
+import type { ComponentType } from 'react';
+import type { ActiveModal, ModalName } from './types';
+import { DatasetsModal } from '../components/DatasetsModal';
+import { ExtractModal } from '../components/ExtractModal';
+import { useDatasetStore } from '../stores/DatasetStore';
+import { useExtractStore } from '../stores/ExtractStore';
+
+export interface ModalConfig {
+ name: ModalName;
+ /** Header title (literal for now; an i18n key once strings are centralized). */
+ title: string;
+ /** The body rendered inside the shell. */
+ component: ComponentType;
+ /** Initialize transient state on open. `arg` carries an optional sub-target. */
+ init?: (arg?: string) => void;
+ /**
+ * Serializable snapshot of in-progress edits for unsaved-change detection.
+ * Return null when there is nothing to lose (browsing, or no open form) so
+ * closing doesn't prompt. Omit entirely for modals that apply immediately.
+ */
+ getState?: () => Record | null;
+ /** Whether the modal is reflected in the URL hash (navigable). */
+ isUrlNavigable?: boolean;
+}
+
+export const MODAL_REGISTRY: Partial> = {
+ // Navigable, multi-view manager. The snapshot captures only the open create/edit
+ // form, so browsing list↔detail never trips a false discard prompt.
+ datasets: {
+ name: 'datasets',
+ title: 'Datasets',
+ component: DatasetsModal,
+ isUrlNavigable: true,
+ init: (datasetId) => useDatasetStore.getState().select(datasetId ? Number(datasetId) : null),
+ getState: () => {
+ const s = useDatasetStore.getState();
+ return s.view === 'new' || s.view === 'edit' ? { form: s.form } : null;
+ },
+ },
+
+ // Opened from the snippet editor with the active draft's inline data to lift out.
+ extract: {
+ name: 'extract',
+ title: 'Extract to Dataset',
+ component: ExtractModal,
+ init: () => useExtractStore.getState().init(),
+ getState: () => ({ name: useExtractStore.getState().name }),
+ },
+};
+
+export const getModalConfig = (name: ActiveModal): ModalConfig | undefined =>
+ name ? MODAL_REGISTRY[name] : undefined;
+
+export const getModalTitle = (name: ActiveModal): string => getModalConfig(name)?.title ?? '';
+
+export const isUrlNavigable = (name: ActiveModal): boolean =>
+ getModalConfig(name)?.isUrlNavigable ?? false;
diff --git a/src/app/modals/types.ts b/src/app/modals/types.ts
new file mode 100644
index 0000000..f6438f0
--- /dev/null
+++ b/src/app/modals/types.ts
@@ -0,0 +1,19 @@
+/**
+ * Modal system — the closed set of named feature modals (docs/architecture/03).
+ *
+ * Modelled as a closed union so the registry, coordinator, and shell are
+ * exhaustively type-checked: a new modal that isn't handled everywhere fails to
+ * compile. Confirmation/alert dialogs are deliberately NOT in this set — they
+ * are a separate, lighter layer (see ConfirmStore) that may stack above a modal.
+ */
+
+export type ModalName =
+ | 'datasets' // Datasets manager (list / detail / new-dataset form)
+ | 'settings' // Appearance, editor, performance, formatting prefs (M5)
+ | 'about' // About & Help (M6)
+ | 'donate' // Donate (M6)
+ | 'chartBuilder' // Visual no-JSON chart composition for a dataset (M4)
+ | 'extract'; // Extract inline spec data into a new dataset (M3)
+
+/** The active modal, or `null` when none is open (at most one at a time, §01C). */
+export type ActiveModal = ModalName | null;
diff --git a/src/app/orchestration/dataset-persistence.ts b/src/app/orchestration/dataset-persistence.ts
new file mode 100644
index 0000000..fdb037d
--- /dev/null
+++ b/src/app/orchestration/dataset-persistence.ts
@@ -0,0 +1,54 @@
+/**
+ * Dataset persistence wiring (docs/architecture/01 §5, spec §09E).
+ *
+ * The dataset sibling of `persistence.ts`: a startup subscriber that diffs the
+ * `datasets` array against the previous snapshot and writes upserts/deletes
+ * through to the IndexedDB adapter. The store stays browser-free; failures
+ * surface as a toast rather than silent loss (spec §10). There is no debounced
+ * auto-save here — datasets change on explicit create/edit/delete, not per
+ * keystroke, so every change is a structural array edit.
+ */
+
+import { deleteDataset, saveDataset } from '../infrastructure/dataset-store';
+import { notify } from '../stores/NotificationStore';
+import { useDatasetStore } from '../stores/DatasetStore';
+
+type Unsubscribe = () => void;
+
+function datasetError(op: 'save' | 'delete', err: unknown) {
+ notify({
+ kind: 'error',
+ title: op === 'delete' ? "Couldn't delete the dataset" : "Couldn't save the dataset",
+ message:
+ 'A storage error stopped Astrolabe from completing the last dataset change, so it may not ' +
+ 'survive a reload. If this keeps happening, your browser may be blocking local storage.',
+ detail:
+ err instanceof Error ? `Dataset ${op} failed: ${err.name}: ${err.message}` : String(err),
+ });
+}
+
+/** Persist dataset upserts and deletions whenever the array changes. */
+function wireDatasetWriteThrough(): Unsubscribe {
+ let prevDatasets = useDatasetStore.getState().datasets;
+ return useDatasetStore.subscribe((s) => {
+ const next = s.datasets;
+ if (next === prevDatasets) return;
+ const prev = prevDatasets;
+ prevDatasets = next;
+
+ for (const old of prev) {
+ if (!next.some((n) => n.id === old.id)) {
+ deleteDataset(old.id).catch((err) => datasetError('delete', err));
+ }
+ }
+ for (const n of next) {
+ const old = prev.find((p) => p.id === n.id);
+ if (old !== n) saveDataset(n).catch((err) => datasetError('save', err));
+ }
+ });
+}
+
+/** Wire dataset persistence subscribers. Returns a teardown that detaches them. */
+export function wireDatasetPersistence(): Unsubscribe {
+ return wireDatasetWriteThrough();
+}
diff --git a/src/app/orchestration/startup.ts b/src/app/orchestration/startup.ts
index e6a301a..09ee396 100644
--- a/src/app/orchestration/startup.ts
+++ b/src/app/orchestration/startup.ts
@@ -8,11 +8,15 @@
*/
import { createSnippet, type Snippet } from '@core/snippet';
+import type { Dataset } from '@core/dataset';
import { loadSnippets, saveSnippet } from '../infrastructure/snippet-store';
+import { loadDatasets } from '../infrastructure/dataset-store';
import { storageErrorNotification } from '../services/storage-errors';
import { notify } from '../stores/NotificationStore';
import { useSnippetStore } from '../stores/SnippetStore';
+import { useDatasetStore } from '../stores/DatasetStore';
import { wirePersistence } from './persistence';
+import { wireDatasetPersistence } from './dataset-persistence';
let started = false;
@@ -47,9 +51,20 @@ export async function initApp(): Promise {
}
}
+ // Load the dataset library too (no first-run seed — datasets are user-created).
+ // A failure here is surfaced; the app stays usable with an empty dataset library.
+ let datasets: Dataset[] = [];
+ try {
+ datasets = await loadDatasets();
+ } catch (err) {
+ notify(storageErrorNotification('load', err));
+ }
+
useSnippetStore.getState().hydrate(snippets);
+ useDatasetStore.getState().hydrate(datasets);
// Wire persistence AFTER hydrate so write-through's baseline is the loaded set
- // — otherwise it would redundantly re-save every snippet on each startup.
+ // — otherwise it would redundantly re-save every record on each startup.
wirePersistence();
+ wireDatasetPersistence();
}
diff --git a/src/app/services/RelationshipService.test.ts b/src/app/services/RelationshipService.test.ts
new file mode 100644
index 0000000..c87e1e9
--- /dev/null
+++ b/src/app/services/RelationshipService.test.ts
@@ -0,0 +1,74 @@
+import { beforeEach, describe, expect, test } from 'vitest';
+import { createDataset } from '@core/dataset';
+import { createSnippet } from '@core/snippet';
+import { useDatasetStore } from '../stores/DatasetStore';
+import { useSnippetStore } from '../stores/SnippetStore';
+import {
+ datasetUsageCount,
+ findSnippetsReferencingDataset,
+ renameDatasetEverywhere,
+} from './RelationshipService';
+
+const T = new Date('2026-06-01T00:00:00Z');
+
+/** Seed a snippet whose published spec references `datasetName`. */
+function seedSnippet(id: string, datasetName: string) {
+ return createSnippet({
+ id,
+ spec: JSON.stringify({ data: { name: datasetName }, mark: 'bar' }),
+ now: T,
+ });
+}
+
+beforeEach(() => {
+ useDatasetStore.getState().reset();
+ useSnippetStore.getState().reset();
+});
+
+describe('reverse lookup', () => {
+ test('finds referencing snippets case-insensitively and counts them', () => {
+ useSnippetStore.getState().hydrate([seedSnippet('a', 'Sales'), seedSnippet('b', 'SALES')], 'a');
+ // datasetRefs are recomputed on publish; publish each active snippet to seed.
+ useSnippetStore.getState().publish(T);
+ useSnippetStore.getState().selectSnippet('b');
+ useSnippetStore.getState().publish(T);
+
+ expect(
+ findSnippetsReferencingDataset('sales')
+ .map((s) => s.id)
+ .sort(),
+ ).toEqual(['a', 'b']);
+ expect(datasetUsageCount('Sales')).toBe(2);
+ expect(datasetUsageCount('Other')).toBe(0);
+ });
+});
+
+describe('renameDatasetEverywhere', () => {
+ test('renames the dataset record and every referencing snippet', () => {
+ useDatasetStore
+ .getState()
+ .add(
+ createDataset({
+ name: 'Sales',
+ data: [{ a: 1 }],
+ format: 'json',
+ source: 'inline',
+ now: T,
+ }),
+ );
+ useSnippetStore.getState().hydrate([seedSnippet('a', 'Sales')], 'a');
+ useSnippetStore.getState().publish(T);
+
+ const updated = renameDatasetEverywhere('Sales', 'Revenue', new Date('2026-07-01T00:00:00Z'));
+
+ expect(updated).toBe(1);
+ expect(useDatasetStore.getState().datasets[0].name).toBe('Revenue');
+ const s = useSnippetStore.getState().snippets[0];
+ expect(s.datasetRefs).toEqual(['Revenue']);
+ expect(s.spec).toContain('"Revenue"');
+ });
+
+ test('is a no-op when old and new names are equal', () => {
+ expect(renameDatasetEverywhere('Sales', 'Sales')).toBe(0);
+ });
+});
diff --git a/src/app/services/RelationshipService.ts b/src/app/services/RelationshipService.ts
new file mode 100644
index 0000000..c8f6fbd
--- /dev/null
+++ b/src/app/services/RelationshipService.ts
@@ -0,0 +1,54 @@
+/**
+ * Snippet ↔ dataset relationships (docs/architecture/07 §4 + §6).
+ *
+ * The bidirectional link is name-based and has a single source of truth: a
+ * snippet's `datasetRefs` (recomputed from its published spec). The reverse
+ * direction — "which snippets use this dataset" — is therefore DERIVED by a scan,
+ * never stored, so it can't drift. Rename is the one graph operation: it renames
+ * the dataset record and propagates the new name into every referencing snippet's
+ * spec/draftSpec/refs.
+ *
+ * These functions read and mutate Zustand stores, so they live in the app layer
+ * (the pure rewrite/extraction helpers they build on live in `core/spec-refs`).
+ */
+
+import type { Snippet } from '@core/snippet';
+import { useDatasetStore } from '../stores/DatasetStore';
+import { useSnippetStore } from '../stores/SnippetStore';
+
+// TODO: nothing in production calls this service yet — DatasetsModal derives
+// usage/linked-snippets inline (it must read `snippets` reactively, not via the
+// getState snapshot these use), and DatasetStore.save propagates rename itself
+// (it updates other fields besides the name, so it can't delegate cleanly).
+// `renameDatasetEverywhere` also matches by exact name, unlike the rest of the
+// case-insensitive naming policy, and arch/07 documents it returning `{ updated }`
+// not a bare number. Either reconcile the duplication (extract a reactive-friendly
+// pure `snippetsReferencing(snippets, name)` helper both sides share) or fold this
+// service in when M4's "Build Chart from dataset" needs a non-reactive caller.
+
+/** Snippets whose `datasetRefs` include `name` (case-insensitive). */
+export function findSnippetsReferencingDataset(name: string): Snippet[] {
+ const lower = name.toLowerCase();
+ return useSnippetStore
+ .getState()
+ .snippets.filter((s) => s.datasetRefs.some((ref) => ref.toLowerCase() === lower));
+}
+
+/** Count for the dataset usage badge (spec §05 → List item). */
+export function datasetUsageCount(name: string): number {
+ return findSnippetsReferencingDataset(name).length;
+}
+
+/**
+ * Rename a dataset and propagate everywhere (docs/architecture/07 §6): renames
+ * the dataset record, then rewrites every referencing snippet's spec, draftSpec,
+ * and datasetRefs. The caller is responsible for collision policy on `newName`
+ * (the edit form rejects a taken name; programmatic paths pre-resolve a unique
+ * one). Returns the number of snippets updated.
+ */
+export function renameDatasetEverywhere(oldName: string, newName: string, now?: Date): number {
+ if (oldName === newName) return 0;
+ const dataset = useDatasetStore.getState().datasets.find((d) => d.name === oldName);
+ if (dataset) useDatasetStore.getState().update(dataset.id, { name: newName }, now);
+ return useSnippetStore.getState().renameDatasetRefs(oldName, newName, now);
+}
diff --git a/src/app/stores/AppStore.ts b/src/app/stores/AppStore.ts
index caf1687..7970a22 100644
--- a/src/app/stores/AppStore.ts
+++ b/src/app/stores/AppStore.ts
@@ -1,6 +1,7 @@
import { create } from 'zustand';
import type { FitMode } from '@core/rendering';
import type { UiTheme } from '@core/theme';
+import type { ModalName } from '../modals/types';
/**
* Centralized cross-cutting application state, as a Zustand store. Keep this
@@ -13,9 +14,9 @@ import type { UiTheme } from '@core/theme';
*/
export type { UiTheme };
-
-/** Which modal, if any, is currently open. At most one at a time (spec §01C). */
-export type ModalName = 'datasets' | 'settings' | 'about' | 'donate' | 'chartBuilder' | 'extract';
+// The modal union is defined once in the modal system (docs/architecture/03) and
+// re-exported here for the many callers that reach it through the app store.
+export type { ModalName };
export interface AppState {
/** Active UI theme; mirrored onto by a subscriber. */
diff --git a/src/app/stores/DatasetStore.test.ts b/src/app/stores/DatasetStore.test.ts
new file mode 100644
index 0000000..3642da9
--- /dev/null
+++ b/src/app/stores/DatasetStore.test.ts
@@ -0,0 +1,144 @@
+import { beforeEach, describe, expect, test } from 'vitest';
+import { createDataset } from '@core/dataset';
+import { createSnippet } from '@core/snippet';
+import { selectSelectedDataset, useDatasetStore, type DatasetForm } from './DatasetStore';
+import { useSnippetStore } from './SnippetStore';
+
+const store = () => useDatasetStore.getState();
+const T = new Date('2026-06-01T00:00:00Z');
+
+beforeEach(() => {
+ store().reset();
+ useSnippetStore.getState().reset();
+});
+
+describe('save — create', () => {
+ test('valid inline JSON is added, profiled, and selected', () => {
+ store().startCreate();
+ store().updateForm({ name: 'Sales', input: '[{"a":1,"b":2},{"a":3,"b":4}]' });
+
+ expect(store().save(T)).toBe(true);
+ const ds = selectSelectedDataset(store());
+ expect(ds?.name).toBe('Sales');
+ expect(ds?.format).toBe('json');
+ expect(ds?.rowCount).toBe(2);
+ expect(ds?.columns).toEqual(['a', 'b']);
+ expect(store().view).toBe('detail');
+ });
+
+ test('CSV input is stored as raw text and profiled', () => {
+ store().startCreate();
+ store().updateForm({ name: 'Regions', input: 'city,pop\nA,10\nB,20' });
+
+ expect(store().save(T)).toBe(true);
+ const ds = selectSelectedDataset(store());
+ expect(ds?.format).toBe('csv');
+ expect(ds?.data).toBe('city,pop\nA,10\nB,20');
+ expect(ds?.rowCount).toBe(2);
+ });
+
+ test('a valid URL infers format from the extension', () => {
+ store().startCreate();
+ store().updateForm({ name: 'Remote', source: 'url', input: 'https://example.com/data.csv' });
+
+ expect(store().save(T)).toBe(true);
+ const ds = selectSelectedDataset(store());
+ expect(ds?.source).toBe('url');
+ expect(ds?.format).toBe('csv');
+ expect(ds?.rowCount).toBeNull(); // URL datasets aren't profiled
+ });
+});
+
+describe('save — validation', () => {
+ const submit = (patch: Partial) => {
+ store().startCreate();
+ store().updateForm(patch);
+ return store().save(T);
+ };
+
+ test('blank name is rejected', () => {
+ expect(submit({ name: ' ', input: '[{"a":1}]' })).toBe(false);
+ expect(store().formError).toMatch(/name/i);
+ expect(store().datasets).toHaveLength(0);
+ });
+
+ test('duplicate name (case-insensitive) is rejected', () => {
+ store().add(
+ createDataset({ name: 'Sales', data: [{ a: 1 }], format: 'json', source: 'inline', now: T }),
+ );
+ expect(submit({ name: 'sales', input: '[{"a":1}]' })).toBe(false);
+ expect(store().formError).toMatch(/already exists/i);
+ });
+
+ test('empty and unrecognized inline data are rejected', () => {
+ expect(submit({ name: 'Empty', input: ' ' })).toBe(false);
+ expect(submit({ name: 'Junk', input: 'this is not data' })).toBe(false);
+ expect(store().formError).toMatch(/valid JSON/i);
+ });
+
+ test('a non-http URL is rejected', () => {
+ expect(submit({ name: 'Bad', source: 'url', input: 'ftp://x/y.csv' })).toBe(false);
+ expect(store().formError).toMatch(/url/i);
+ });
+});
+
+describe('save — edit', () => {
+ test('updating inline data re-profiles and advances modified', () => {
+ store().add(
+ createDataset({ name: 'D', data: [{ a: 1 }], format: 'json', source: 'inline', now: T }),
+ );
+ const id = store().selectedId!;
+ store().startEdit();
+ store().updateForm({ input: '[{"a":1,"b":2},{"a":3,"b":4}]' });
+
+ const later = new Date('2026-07-01T00:00:00Z');
+ expect(store().save(later)).toBe(true);
+ const ds = store().datasets.find((d) => d.id === id)!;
+ expect(ds.columnCount).toBe(2);
+ expect(ds.modified).toBe(later.toISOString());
+ });
+
+ test('renaming a referenced dataset propagates into referencing snippets', () => {
+ store().add(
+ createDataset({ name: 'Sales', data: [{ a: 1 }], format: 'json', source: 'inline', now: T }),
+ );
+ const snippet = createSnippet({
+ id: 's1',
+ spec: JSON.stringify({ data: { name: 'Sales' }, mark: 'bar' }),
+ now: T,
+ });
+ useSnippetStore.getState().hydrate([snippet], 's1');
+ useSnippetStore.getState().publish(T); // seed datasetRefs = ['Sales']
+
+ store().startEdit();
+ store().updateForm({ name: 'Revenue' });
+ expect(store().save(new Date('2026-08-01T00:00:00Z'))).toBe(true);
+
+ expect(store().datasets[0].name).toBe('Revenue');
+ const s = useSnippetStore.getState().snippets.find((x) => x.id === 's1')!;
+ expect(s.datasetRefs).toEqual(['Revenue']);
+ expect(s.spec).toContain('"Revenue"');
+ });
+});
+
+describe('remove & view transitions', () => {
+ test('removing the selected dataset clears the selection and returns to the list', () => {
+ store().add(
+ createDataset({ name: 'D', data: [{ a: 1 }], format: 'json', source: 'inline', now: T }),
+ );
+ const id = store().selectedId!;
+ store().remove(id);
+ expect(store().selectedId).toBeNull();
+ expect(store().view).toBe('list');
+ });
+
+ test('cancelForm returns to detail when a dataset is selected', () => {
+ store().add(
+ createDataset({ name: 'D', data: [{ a: 1 }], format: 'json', source: 'inline', now: T }),
+ );
+ store().startEdit();
+ expect(store().view).toBe('edit');
+ store().cancelForm();
+ expect(store().view).toBe('detail');
+ });
+});
diff --git a/src/app/stores/DatasetStore.ts b/src/app/stores/DatasetStore.ts
new file mode 100644
index 0000000..842b84e
--- /dev/null
+++ b/src/app/stores/DatasetStore.ts
@@ -0,0 +1,257 @@
+/**
+ * Dataset library state (spec §05, docs/architecture/01 + 07).
+ *
+ * Holds the durable dataset collection plus the Datasets-manager view state: which
+ * dataset is selected, which pane is showing (list browse, detail, the create
+ * form, or the edit form), and the in-progress create/edit form with its inline
+ * validation error.
+ *
+ * Two layers of action live here:
+ * - Low-level mutators (`add`/`update`/`remove`) are the single place the
+ * `datasets` array changes; the persistence subscriber writes them through to
+ * IndexedDB, and services (RelationshipService, future import) reuse them.
+ * - Form orchestration (`save`) validates the form (name required + unique,
+ * data detectable) and turns it into an `add`/`update`, keeping the modal
+ * component thin.
+ *
+ * Persistence is NOT done here — a startup subscriber observes this store and
+ * writes through to the IndexedDB adapter, so the store stays browser-free.
+ */
+
+import { create } from 'zustand';
+import { detectFormat, detectFormatFromUrl, type DataFormat } from '@core/format-detection';
+import { createDataset, computeDatasetProfile, type DataSource, type Dataset } from '@core/dataset';
+import { isNameTaken } from '@core/naming';
+import { useSnippetStore } from './SnippetStore';
+
+/** Which pane the Datasets manager is showing (spec §05 → Layout). */
+export type DatasetView = 'list' | 'detail' | 'new' | 'edit';
+
+/** The in-progress create/edit form. `input` is the paste area (inline) or URL field. */
+export interface DatasetForm {
+ name: string;
+ source: DataSource;
+ input: string;
+ comment: string;
+}
+
+const EMPTY_FORM: DatasetForm = { name: '', source: 'inline', input: '', comment: '' };
+
+export interface DatasetState {
+ datasets: Dataset[];
+ selectedId: number | null;
+ view: DatasetView;
+ form: DatasetForm;
+ /** Inline validation message for the create/edit form, or null when valid. */
+ formError: string | null;
+
+ /** Replace the library from storage. */
+ hydrate: (datasets: Dataset[]) => void;
+ /** Select a dataset (→ detail), or clear the selection (→ list). */
+ select: (id: number | null) => void;
+ /** Set the manager view directly (used by the modal's own navigation). */
+ setView: (view: DatasetView) => void;
+ /** Open the create form with an empty draft. */
+ startCreate: () => void;
+ /** Open the edit form pre-filled from the selected dataset. */
+ startEdit: () => void;
+ /** Patch the in-progress form (clears any stale error). */
+ updateForm: (patch: Partial) => void;
+ /** Leave the form, returning to the selected detail or the list. */
+ cancelForm: () => void;
+ /**
+ * Validate and commit the current form: creates a new dataset (view `new`) or
+ * updates the selected one (view `edit`), re-profiling its data and renaming
+ * referencing snippets when an edit changes the name. Returns whether it
+ * committed; on failure `formError` is set. `now` is injectable for tests.
+ */
+ save: (now?: Date) => boolean;
+
+ /** Low-level: add a fully-formed dataset and select it. */
+ add: (dataset: Dataset) => void;
+ /** Low-level: merge a patch into a dataset, advancing `modified`. */
+ update: (id: number, patch: Partial, now?: Date) => void;
+ /** Low-level: remove a dataset; clears the selection if it was selected. */
+ remove: (id: number) => void;
+
+ /** Reset to initial state (tests, future "new workspace"). */
+ reset: () => void;
+}
+
+/** Newest-modified first — the manager's default ordering (spec §05 → Layout). */
+export function byModifiedDesc(a: Dataset, b: Dataset): number {
+ return b.modified.localeCompare(a.modified);
+}
+
+/** Validate a form into a saveable shape, or return an error message. */
+function resolveForm(
+ form: DatasetForm,
+ datasets: Dataset[],
+ excludeId: number | undefined,
+): { name: string; data: unknown; format: DataFormat; source: DataSource } | { error: string } {
+ // Error copy follows the council resolution (docs/architecture/10 §error copy →
+ // GOV.UK error-message): action-oriented, specific, and says how to fix it.
+ const name = form.name.trim();
+ if (name === '') return { error: 'Enter a dataset name.' };
+ if (isNameTaken(name, datasets, excludeId)) {
+ return { error: `A dataset named "${name}" already exists. Choose a different name.` };
+ }
+
+ const input = form.input.trim();
+ if (input === '') {
+ return {
+ error: form.source === 'url' ? 'Enter a URL.' : 'Paste JSON, CSV, or TSV data to save.',
+ };
+ }
+
+ if (form.source === 'url') {
+ if (!/^https?:\/\//i.test(input)) {
+ return { error: 'Enter a URL starting with http:// or https://.' };
+ }
+ // Format is inferred from the extension; default to JSON when unknown (§05).
+ return { name, data: input, format: detectFormatFromUrl(input) ?? 'json', source: 'url' };
+ }
+
+ // Inline: the data must auto-detect to a known format (spec §05 → Auto-detection).
+ const { format } = detectFormat(form.input);
+ if (!format) {
+ return { error: 'Enter valid JSON, CSV, or TSV data.' };
+ }
+ // JSON/TopoJSON are stored parsed; CSV/TSV keep their raw text (data model §09B).
+ const data =
+ format === 'json' || format === 'topojson' ? (JSON.parse(form.input) as unknown) : form.input;
+ return { name, data, format, source: 'inline' };
+}
+
+export const useDatasetStore = create((set, get) => ({
+ datasets: [],
+ selectedId: null,
+ view: 'list',
+ form: EMPTY_FORM,
+ formError: null,
+
+ hydrate: (datasets) => set({ datasets }),
+
+ select: (id) =>
+ set({
+ selectedId: id,
+ view: id === null ? 'list' : 'detail',
+ form: EMPTY_FORM,
+ formError: null,
+ }),
+
+ setView: (view) => set({ view }),
+
+ startCreate: () => set({ view: 'new', form: EMPTY_FORM, formError: null }),
+
+ startEdit: () => {
+ const { datasets, selectedId } = get();
+ const ds = datasets.find((d) => d.id === selectedId);
+ if (!ds) return;
+ set({
+ view: 'edit',
+ formError: null,
+ form: {
+ name: ds.name,
+ source: ds.source,
+ comment: ds.comment,
+ // Re-render the stored payload as editable text: raw for csv/tsv/url,
+ // pretty-printed JSON for json/topojson.
+ input:
+ ds.source === 'url' || ds.format === 'csv' || ds.format === 'tsv'
+ ? String(ds.data)
+ : JSON.stringify(ds.data, null, 2),
+ },
+ });
+ },
+
+ updateForm: (patch) => set((s) => ({ form: { ...s.form, ...patch }, formError: null })),
+
+ cancelForm: () =>
+ set((s) => ({
+ view: s.selectedId === null ? 'list' : 'detail',
+ form: EMPTY_FORM,
+ formError: null,
+ })),
+
+ save: (now) => {
+ const { view, form, datasets, selectedId } = get();
+ const editing = view === 'edit';
+ const excludeId = editing ? (selectedId ?? undefined) : undefined;
+
+ const resolved = resolveForm(form, datasets, excludeId);
+ if ('error' in resolved) {
+ set({ formError: resolved.error });
+ return false;
+ }
+
+ if (editing && selectedId !== null) {
+ const existing = datasets.find((d) => d.id === selectedId);
+ if (!existing) return false;
+ const profile = computeDatasetProfile(resolved.data, resolved.format, resolved.source);
+ // Re-profile and update the record (including any new name), then propagate
+ // the rename across referencing snippets so each spec and its datasetRefs
+ // stay consistent (docs/architecture/07 §6). SnippetStore never imports this
+ // store, so the direct call is cycle-free.
+ get().update(
+ selectedId,
+ {
+ name: resolved.name,
+ data: resolved.data,
+ format: resolved.format,
+ source: resolved.source,
+ comment: form.comment,
+ ...profile,
+ },
+ now,
+ );
+ if (resolved.name !== existing.name) {
+ useSnippetStore.getState().renameDatasetRefs(existing.name, resolved.name, now);
+ }
+ set({ view: 'detail', form: EMPTY_FORM, formError: null });
+ return true;
+ }
+
+ const dataset = createDataset({
+ name: resolved.name,
+ data: resolved.data,
+ format: resolved.format,
+ source: resolved.source,
+ comment: form.comment,
+ now,
+ });
+ get().add(dataset);
+ set({ view: 'detail', form: EMPTY_FORM, formError: null });
+ return true;
+ },
+
+ // TODO: createDataset ids default to `Date.now()` and `add` does not reassign
+ // on collision (despite that file's comment). Interactive create is safe, but
+ // the documented non-interactive import path (naming.ts) creates many datasets
+ // in a tight loop where `Date.now()` repeats — duplicate numeric keys would
+ // collide in IndexedDB. Give the store a monotonic id source (or reassign here)
+ // before wiring import.
+ add: (dataset) => set((s) => ({ datasets: [dataset, ...s.datasets], selectedId: dataset.id })),
+
+ update: (id, patch, now) => {
+ const modified = patch.modified ?? (now ?? new Date()).toISOString();
+ set((s) => ({
+ datasets: s.datasets.map((d) => (d.id === id ? { ...d, ...patch, modified } : d)),
+ }));
+ },
+
+ remove: (id) =>
+ set((s) => ({
+ datasets: s.datasets.filter((d) => d.id !== id),
+ selectedId: s.selectedId === id ? null : s.selectedId,
+ view: s.selectedId === id ? 'list' : s.view,
+ form: s.selectedId === id ? EMPTY_FORM : s.form,
+ })),
+
+ reset: () =>
+ set({ datasets: [], selectedId: null, view: 'list', form: EMPTY_FORM, formError: null }),
+}));
+
+/** Selector: the selected dataset record, or null. Derive — never store. */
+export const selectSelectedDataset = (s: DatasetState): Dataset | null =>
+ s.datasets.find((d) => d.id === s.selectedId) ?? null;
diff --git a/src/app/stores/ExtractStore.ts b/src/app/stores/ExtractStore.ts
new file mode 100644
index 0000000..3909668
--- /dev/null
+++ b/src/app/stores/ExtractStore.ts
@@ -0,0 +1,133 @@
+/**
+ * Extract-inline-data → Dataset state (spec §03F).
+ *
+ * Backs the Extract modal: the reverse of a named reference. It captures the
+ * active snippet draft's inline data, takes a dataset name, and on confirm saves
+ * the data as a new dataset and rewrites the draft so the inline data is replaced
+ * by a by-name reference (`{ "data": { "name": … } }`).
+ *
+ * Scope (M3): the **top-level** `data` block of the draft spec — the common case
+ * for a single-view chart. Inline data nested inside layers/concats is left for a
+ * later pass; `hasInlineData` reflects exactly what `confirm` can lift, so the
+ * editor only offers the action when this store can act on it.
+ */
+
+import { create } from 'zustand';
+import type { DataFormat } from '@core/format-detection';
+import { createDataset } from '@core/dataset';
+import { isNameTaken } from '@core/naming';
+import { useDatasetStore } from './DatasetStore';
+import { useSnippetStore } from './SnippetStore';
+
+/** The inline `data` block of a parsed spec, if it carries `values`. */
+interface InlineData {
+ values: unknown;
+ format: DataFormat;
+}
+
+/**
+ * Read the top-level inline data from a draft spec's text, or null when there is
+ * none (no snippet, unparseable, or no `data.values`). The format comes from an
+ * explicit `data.format.type` when present (raw CSV/TSV strings), else JSON.
+ */
+export function readInlineData(draftText: string): InlineData | null {
+ let parsed: unknown;
+ try {
+ parsed = JSON.parse(draftText);
+ } catch {
+ return null;
+ }
+ if (!parsed || typeof parsed !== 'object') return null;
+ const data = (parsed as Record).data;
+ if (!data || typeof data !== 'object') return null;
+ const values = (data as Record).values;
+ if (values === undefined) return null;
+ const declared = (data as Record).format;
+ const type =
+ declared && typeof declared === 'object'
+ ? (declared as Record).type
+ : undefined;
+ const format: DataFormat =
+ type === 'csv' || type === 'tsv' || type === 'topojson' ? type : 'json';
+ return { values, format };
+}
+
+/** True when the active snippet's draft has top-level inline data to extract. */
+export function hasInlineData(draftText: string): boolean {
+ return readInlineData(draftText) !== null;
+}
+
+export interface ExtractState {
+ /** Proposed dataset name (required, unique). */
+ name: string;
+ /** The inline data captured at open, for the read-only preview. */
+ source: InlineData | null;
+ /** Inline validation message, or null. */
+ error: string | null;
+
+ /** Capture the active snippet's inline data and reset the form. */
+ init: () => void;
+ setName: (name: string) => void;
+ /**
+ * Validate, create the dataset, and rewrite the active draft to reference it by
+ * name. Returns whether it committed; on failure `error` is set. `now`
+ * injectable for tests.
+ */
+ confirm: (now?: Date) => boolean;
+ reset: () => void;
+}
+
+const INITIAL = { name: '', source: null as InlineData | null, error: null as string | null };
+
+export const useExtractStore = create((set, get) => ({
+ ...INITIAL,
+
+ init: () => {
+ const draft = useSnippetStore.getState().draftText;
+ set({ name: '', source: readInlineData(draft), error: null });
+ },
+
+ setName: (name) => set({ name, error: null }),
+
+ confirm: (now) => {
+ const name = get().name.trim();
+ if (name === '') {
+ set({ error: 'Enter a dataset name.' });
+ return false;
+ }
+ if (isNameTaken(name, useDatasetStore.getState().datasets)) {
+ set({ error: `A dataset named "${name}" already exists. Choose a different name.` });
+ return false;
+ }
+ const source = get().source;
+ if (!source) {
+ set({ error: 'No inline data to extract.' });
+ return false;
+ }
+
+ // JSON/TopoJSON store the parsed value; CSV/TSV keep raw text — but inline
+ // `values` is already the right runtime shape for each, so store it directly.
+ const dataset = createDataset({
+ name,
+ data: source.values,
+ format: source.format,
+ source: 'inline',
+ now,
+ });
+ useDatasetStore.getState().add(dataset);
+
+ // Rewrite the top-level data block to a by-name reference, preserving the rest
+ // of the spec and its pretty-printed text shape.
+ const draftText = useSnippetStore.getState().draftText;
+ const spec = JSON.parse(draftText) as Record;
+ spec.data = { name };
+ useSnippetStore.getState().replaceActiveDraft(JSON.stringify(spec, null, 2), now);
+
+ // TODO (M6, spec §03F): success toast "Dataset created" — deferred with the
+ // other success toasts (see SnippetStore publish/revert breadcrumbs).
+ set(INITIAL);
+ return true;
+ },
+
+ reset: () => set(INITIAL),
+}));
diff --git a/src/app/stores/SnippetStore.test.ts b/src/app/stores/SnippetStore.test.ts
index a841e47..0383882 100644
--- a/src/app/stores/SnippetStore.test.ts
+++ b/src/app/stores/SnippetStore.test.ts
@@ -224,3 +224,92 @@ describe('editorView + selectShownText (spec §03D)', () => {
expect(store().editorView).toBe('draft');
});
});
+
+describe('publish — datasetRefs recomputation', () => {
+ const refSpec = (name: string) => JSON.stringify({ data: { name }, mark: 'bar' });
+
+ test('publishing recomputes datasetRefs from the now-published spec', () => {
+ const a = createSnippet({ id: 'a', spec: '{}', now: new Date('2026-01-01T00:00:00Z') });
+ store().hydrate([a], 'a');
+ store().updateDraft(refSpec('Sales'));
+ store().publish(new Date('2026-02-01T00:00:00Z'));
+
+ expect(selectActiveSnippet(store())?.datasetRefs).toEqual(['Sales']);
+ });
+
+ test('refs drop when a published spec no longer references a dataset', () => {
+ const a = createSnippet({
+ id: 'a',
+ spec: refSpec('Sales'),
+ now: new Date('2026-01-01T00:00:00Z'),
+ });
+ store().hydrate([a], 'a');
+ // The factory leaves datasetRefs empty until a publish runs; publish to seed it.
+ store().publish(new Date('2026-01-02T00:00:00Z'));
+ expect(selectActiveSnippet(store())?.datasetRefs).toEqual(['Sales']);
+
+ store().updateDraft('{"mark":"bar"}');
+ store().publish(new Date('2026-02-01T00:00:00Z'));
+ expect(selectActiveSnippet(store())?.datasetRefs).toEqual([]);
+ });
+});
+
+describe('renameDatasetRefs', () => {
+ const refSpec = (name: string) => JSON.stringify({ data: { name }, mark: 'bar' }, null, 2);
+
+ test('rewrites spec, draftSpec, and refs of referencing snippets only', () => {
+ const a = createSnippet({
+ id: 'a',
+ spec: refSpec('Sales'),
+ now: new Date('2026-01-01T00:00:00Z'),
+ });
+ const b = createSnippet({
+ id: 'b',
+ spec: '{"mark":"line"}',
+ now: new Date('2026-01-02T00:00:00Z'),
+ });
+ store().hydrate([a, b], 'b');
+ store().publish(new Date('2026-01-03T00:00:00Z')); // seed a's refs would need a active; seed via select
+ store().selectSnippet('a');
+ store().publish(new Date('2026-01-04T00:00:00Z'));
+
+ const updated = store().renameDatasetRefs('Sales', 'Revenue', new Date('2026-02-01T00:00:00Z'));
+
+ expect(updated).toBe(1);
+ const renamed = store().snippets.find((s) => s.id === 'a')!;
+ expect(renamed.datasetRefs).toEqual(['Revenue']);
+ expect(renamed.spec).toContain('"Revenue"');
+ expect(renamed.draftSpec).toContain('"Revenue"');
+ // The non-referencing snippet is untouched.
+ expect(store().snippets.find((s) => s.id === 'b')!.spec).toBe('{"mark":"line"}');
+ });
+
+ test('refreshes the active draft buffer when the active snippet is rewritten', () => {
+ const a = createSnippet({
+ id: 'a',
+ spec: refSpec('Sales'),
+ now: new Date('2026-01-01T00:00:00Z'),
+ });
+ store().hydrate([a], 'a');
+ store().publish(new Date('2026-01-02T00:00:00Z'));
+ const epochBefore = store().bufferEpoch;
+
+ store().renameDatasetRefs('Sales', 'Revenue', new Date('2026-02-01T00:00:00Z'));
+
+ expect(store().draftText).toContain('"Revenue"');
+ expect(store().bufferEpoch).toBe(epochBefore + 1);
+ });
+
+ test('no-ops when the names are equal or nothing references the old name', () => {
+ const a = createSnippet({
+ id: 'a',
+ spec: refSpec('Sales'),
+ now: new Date('2026-01-01T00:00:00Z'),
+ });
+ store().hydrate([a], 'a');
+ store().publish(new Date('2026-01-02T00:00:00Z'));
+
+ expect(store().renameDatasetRefs('Sales', 'Sales')).toBe(0);
+ expect(store().renameDatasetRefs('Unknown', 'Other')).toBe(0);
+ });
+});
diff --git a/src/app/stores/SnippetStore.ts b/src/app/stores/SnippetStore.ts
index a6e4223..dd37d6b 100644
--- a/src/app/stores/SnippetStore.ts
+++ b/src/app/stores/SnippetStore.ts
@@ -19,6 +19,7 @@
import { create } from 'zustand';
import { createSnippet, type CreateSnippetOptions, type Snippet } from '@core/snippet';
+import { recomputeDatasetRefs, renameDatasetInSpec } from '@core/spec-refs';
/** Which version of the active snippet the editor is showing (spec §03D). */
export type EditorView = 'draft' | 'published';
@@ -48,6 +49,13 @@ export interface SnippetState {
removeSnippet: (id: string) => void;
/** Update the draft buffer only (no persistence; debounced commit follows). */
updateDraft: (text: string) => void;
+ /**
+ * Replace the active snippet's draft spec with new text and reload the editor
+ * on the draft view (bumps `bufferEpoch`). Used by programmatic rewrites such
+ * as Extract-to-Dataset (spec §03F), which substitutes inline data for a
+ * by-name reference. No-op when no snippet is active. `now` injectable.
+ */
+ replaceActiveDraft: (text: string, now?: Date) => void;
/**
* Persist the editor buffer into the active snippet's **draft**, if it parses
* as JSON. Returns whether it committed (a half-typed, unparseable buffer is
@@ -57,6 +65,14 @@ export interface SnippetState {
commitDraft: (now?: Date) => boolean;
/** Switch the editor between the draft and published views (spec §03D). */
setEditorView: (view: EditorView) => void;
+ /**
+ * Propagate a dataset rename across every referencing snippet: rewrite the
+ * named-data references in both `spec` and `draftSpec`, recompute `datasetRefs`,
+ * and refresh the live editor buffer if the active snippet was rewritten — so a
+ * user mid-edit doesn't see their draft silently break (docs/architecture/07
+ * §6). Returns the number of snippets changed. `now` injectable.
+ */
+ renameDatasetRefs: (oldName: string, newName: string, now?: Date) => number;
/**
* Promote the active snippet's current draft to its published version (spec
* §03D → Publish). Flushes the live buffer first, then makes `spec` identical
@@ -151,6 +167,20 @@ export const useSnippetStore = create((set, get) => ({
updateDraft: (draftText) => set({ draftText }),
+ replaceActiveDraft: (text, now) => {
+ const { activeSnippetId } = get();
+ if (!activeSnippetId) return;
+ const modified = (now ?? new Date()).toISOString();
+ set((s) => ({
+ snippets: s.snippets.map((x) =>
+ x.id === activeSnippetId ? { ...x, draftSpec: text, modified } : x,
+ ),
+ draftText: text,
+ editorView: 'draft',
+ bufferEpoch: s.bufferEpoch + 1,
+ }));
+ },
+
commitDraft: (now) => {
const { activeSnippetId, draftText, snippets } = get();
if (!activeSnippetId) return false;
@@ -178,6 +208,43 @@ export const useSnippetStore = create((set, get) => ({
setEditorView: (editorView) => set({ editorView }),
+ renameDatasetRefs: (oldName, newName, now) => {
+ if (oldName === newName) return 0;
+ const lower = oldName.toLowerCase();
+ const { snippets, activeSnippetId, editorView } = get();
+ let updated = 0;
+ let activeDraftAfter: string | null = null;
+
+ const next = snippets.map((s) => {
+ if (!s.datasetRefs.some((r) => r.toLowerCase() === lower)) return s;
+ updated++;
+ const spec = renameDatasetInSpec(s.spec, oldName, newName);
+ const draftSpec = renameDatasetInSpec(s.draftSpec, oldName, newName);
+ if (s.id === activeSnippetId) activeDraftAfter = draftSpec;
+ return {
+ ...s,
+ spec,
+ draftSpec,
+ datasetRefs: recomputeDatasetRefs(spec),
+ modified: (now ?? new Date()).toISOString(),
+ };
+ });
+
+ if (updated === 0) return 0;
+ set((st) => ({
+ snippets: next,
+ // If the active snippet's draft was rewritten, reload the editor buffer so
+ // the live (draft-view) buffer reflects the new name and the next auto-save
+ // doesn't clobber the rewrite with the stale old name.
+ ...(activeDraftAfter !== null && editorView === 'draft'
+ ? { draftText: activeDraftAfter, bufferEpoch: st.bufferEpoch + 1 }
+ : activeDraftAfter !== null
+ ? { bufferEpoch: st.bufferEpoch + 1 }
+ : {}),
+ }));
+ return updated;
+ },
+
publish: (now) => {
// Flush the live buffer into the draft first, so Publish promotes exactly
// what the user sees (an invalid buffer leaves the last valid draft in place).
@@ -189,8 +256,15 @@ export const useSnippetStore = create((set, get) => ({
set({
snippets: snippets.map((s) =>
s.id === activeSnippetId
- ? // M3: recompute datasetRefs from the now-published spec (spec §03D).
- { ...s, spec: s.draftSpec, modified }
+ ? // Promote the draft and recompute datasetRefs from the now-published
+ // spec, so the bidirectional snippet↔dataset link mirrors reality
+ // (spec §03D, docs/architecture/07 §3).
+ {
+ ...s,
+ spec: s.draftSpec,
+ datasetRefs: recomputeDatasetRefs(s.draftSpec),
+ modified,
+ }
: s,
),
});
diff --git a/src/core/dataset.test.ts b/src/core/dataset.test.ts
new file mode 100644
index 0000000..429c6e1
--- /dev/null
+++ b/src/core/dataset.test.ts
@@ -0,0 +1,180 @@
+import { describe, expect, test } from 'vitest';
+import {
+ computeDatasetProfile,
+ CURRENT_DATASET_VERSION,
+ createDataset,
+ datasetReference,
+ parseDelimited,
+} from './dataset';
+
+describe('datasetReference', () => {
+ test('produces the by-name reference object', () => {
+ expect(datasetReference('MyDataset')).toEqual({ data: { name: 'MyDataset' } });
+ });
+});
+
+describe('parseDelimited', () => {
+ test('parses CSV header + rows into objects', () => {
+ const rows = parseDelimited('a,b,c\n1,2,3\n4,5,6', 'csv');
+ expect(rows).toEqual([
+ { a: '1', b: '2', c: '3' },
+ { a: '4', b: '5', c: '6' },
+ ]);
+ });
+
+ test('parses TSV with the tab delimiter', () => {
+ const rows = parseDelimited('x\ty\n1\t2', 'tsv');
+ expect(rows).toEqual([{ x: '1', y: '2' }]);
+ });
+
+ test('drops a trailing newline (no phantom empty row) and skips blank lines', () => {
+ expect(parseDelimited('a,b\n1,2\n', 'csv')).toEqual([{ a: '1', b: '2' }]);
+ expect(parseDelimited('a,b\n1,2\n\n3,4', 'csv')).toEqual([
+ { a: '1', b: '2' },
+ { a: '3', b: '4' },
+ ]);
+ });
+
+ test('strips surrounding quotes from header and cells (RFC-4180)', () => {
+ // The quoted-TSV case from real exports: column keys must not keep the quotes.
+ const rows = parseDelimited('"email"\t"id"\n"a@b.com"\t1', 'tsv');
+ expect(rows).toEqual([{ email: 'a@b.com', id: '1' }]);
+ });
+
+ test('quoted fields keep embedded delimiters, newlines, and escaped quotes', () => {
+ const rows = parseDelimited('name,note\n"Doe, John","a ""quote""\nspans"', 'csv');
+ expect(rows).toEqual([{ name: 'Doe, John', note: 'a "quote"\nspans' }]);
+ });
+
+ test('does not trim unquoted whitespace (matches d3-dsv / Vega)', () => {
+ expect(parseDelimited('a, b\n1, 2', 'csv')).toEqual([{ a: '1', ' b': ' 2' }]);
+ });
+
+ test('ragged rows: missing cells are undefined, extra cells ignored', () => {
+ const rows = parseDelimited('a,b,c\n1\n4,5,6,7', 'csv');
+ expect(rows).toEqual([
+ { a: '1', b: undefined, c: undefined },
+ { a: '4', b: '5', c: '6' },
+ ]);
+ });
+
+ test('header-only or empty input yields no rows', () => {
+ expect(parseDelimited('a,b,c', 'csv')).toEqual([]);
+ expect(parseDelimited('', 'csv')).toEqual([]);
+ });
+});
+
+describe('computeDatasetProfile', () => {
+ test('inline JSON array-of-objects is profiled', () => {
+ const data = [
+ { a: 1, b: 'x' },
+ { a: 2, b: 'y' },
+ ];
+ const profile = computeDatasetProfile(data, 'json', 'inline');
+ expect(profile.rowCount).toBe(2);
+ expect(profile.columns).toEqual(['a', 'b']);
+ expect(profile.size).toBe(new TextEncoder().encode(JSON.stringify(data)).length);
+ });
+
+ test('inline JSON that is not an array is N/A but sized', () => {
+ const profile = computeDatasetProfile({ a: 1 }, 'json', 'inline');
+ expect(profile.rowCount).toBeNull();
+ expect(profile.size).toBeGreaterThan(0);
+ });
+
+ test('inline CSV is parsed and profiled; size is the raw text byte length', () => {
+ const text = 'a,b\n1,2';
+ const profile = computeDatasetProfile(text, 'csv', 'inline');
+ expect(profile.rowCount).toBe(1);
+ expect(profile.columns).toEqual(['a', 'b']);
+ expect(profile.size).toBe(new TextEncoder().encode(text).length);
+ });
+
+ test('inline TSV is parsed and profiled', () => {
+ const profile = computeDatasetProfile('a\tb\n1\t2', 'tsv', 'inline');
+ expect(profile.rowCount).toBe(1);
+ expect(profile.columns).toEqual(['a', 'b']);
+ });
+
+ test('inline TopoJSON is N/A (non-tabular) but sized', () => {
+ const topo = { type: 'Topology', objects: {} };
+ const profile = computeDatasetProfile(topo, 'topojson', 'inline');
+ expect(profile.rowCount).toBeNull();
+ expect(profile.columnCount).toBeNull();
+ expect(profile.size).toBe(new TextEncoder().encode(JSON.stringify(topo)).length);
+ });
+
+ test('URL is N/A; size is the byte length of the URL string', () => {
+ const url = 'https://example.com/data.csv';
+ const profile = computeDatasetProfile(url, 'csv', 'url');
+ expect(profile.rowCount).toBeNull();
+ expect(profile.columnCount).toBeNull();
+ expect(profile.size).toBe(new TextEncoder().encode(url).length);
+ });
+});
+
+describe('createDataset', () => {
+ test('stamps version, equal timestamps, and fills the profile for inline JSON', () => {
+ const now = new Date('2026-06-05T10:00:00.000Z');
+ const data = [
+ { city: 'Kyiv', pop: 2900000 },
+ { city: 'Lviv', pop: 720000 },
+ ];
+ const d = createDataset({
+ id: 99,
+ name: 'Cities',
+ data,
+ format: 'json',
+ source: 'inline',
+ now,
+ });
+
+ expect(d.id).toBe(99);
+ expect(d.version).toBe(CURRENT_DATASET_VERSION);
+ expect(d.name).toBe('Cities');
+ expect(d.created).toBe(now.toISOString());
+ expect(d.modified).toBe(d.created);
+ expect(d.comment).toBe('');
+ expect(d.rowCount).toBe(2);
+ expect(d.columnCount).toBe(2);
+ expect(d.columns).toEqual(['city', 'pop']);
+ expect(d.columnTypes).toEqual([
+ { name: 'city', type: 'string' },
+ { name: 'pop', type: 'number' },
+ ]);
+ expect(d.size).toBeGreaterThan(0);
+ });
+
+ test('fills the profile for inline CSV', () => {
+ const d = createDataset({
+ id: 1,
+ name: 'C',
+ data: 'a,b\n1,2\n3,4',
+ format: 'csv',
+ source: 'inline',
+ });
+ expect(d.rowCount).toBe(2);
+ expect(d.columns).toEqual(['a', 'b']);
+ });
+
+ test('URL dataset gets an N/A profile with a size', () => {
+ const d = createDataset({
+ id: 2,
+ name: 'Remote',
+ data: 'https://example.com/x.json',
+ format: 'json',
+ source: 'url',
+ comment: 'remote source',
+ });
+ expect(d.rowCount).toBeNull();
+ expect(d.columnCount).toBeNull();
+ expect(d.columns).toEqual([]);
+ expect(d.comment).toBe('remote source');
+ expect(d.size).toBeGreaterThan(0);
+ });
+
+ test('defaults the id when not injected', () => {
+ const d = createDataset({ name: 'X', data: [], format: 'json', source: 'inline' });
+ expect(typeof d.id).toBe('number');
+ });
+});
diff --git a/src/core/dataset.ts b/src/core/dataset.ts
new file mode 100644
index 0000000..995b1c9
--- /dev/null
+++ b/src/core/dataset.ts
@@ -0,0 +1,271 @@
+/**
+ * Dataset — a named, reusable data source snippets reference by name
+ * (spec §09B → Dataset; spec §05 → Datasets).
+ *
+ * Portable core: no browser APIs, no React. Defines the record shape, the current
+ * record schema version, the by-name reference object (spec §05 → Copy Reference),
+ * a simple delimited-text parser, the profiling orchestration, and a factory that
+ * stamps timestamps/version and fills the derived summary fields.
+ *
+ * A dataset has one of two **sources** — `inline` (data stored in the record) or
+ * `url` (only the link is stored, fetched on demand at render time) — and one of
+ * four **formats** (reused from format-detection: `json`/`csv`/`tsv`/`topojson`).
+ * The `data` field's shape follows source/format: a URL string for `url`; raw
+ * text for inline CSV/TSV; a parsed value for inline JSON/TopoJSON.
+ *
+ * Only tabular inline data (JSON array-of-objects, CSV, TSV) is profiled; URL and
+ * non-tabular data get an N/A profile but are still sized (see profile.ts).
+ */
+
+import type { DataFormat } from './format-detection';
+import { profileData, type DatasetProfile } from './profile';
+import type { ColumnType } from './type-inference';
+
+/** Current schema version for a Dataset record (read-time migration target). */
+export const CURRENT_DATASET_VERSION = 1;
+
+/** Where a dataset's data lives: embedded in the record, or fetched from a URL. */
+export type DataSource = 'inline' | 'url';
+
+export interface Dataset {
+ /** Unique numeric identifier. */
+ id: number;
+ /** Record schema version, for read-time migration. */
+ version: number;
+ /** Unique, human-readable name; the key snippets reference via `datasetRefs`. */
+ name: string;
+ /**
+ * The payload. For `source = url`: the URL string. For `source = inline`: the
+ * raw CSV/TSV text, or the parsed JSON/TopoJSON value.
+ */
+ data: unknown;
+ /** One of `json`, `csv`, `tsv`, `topojson`. */
+ format: DataFormat;
+ /** One of `inline` or `url`. */
+ source: DataSource;
+ /** Free-form user note about the dataset. */
+ comment: string;
+ /** Data rows, or `null` when N/A (URL / non-tabular). */
+ rowCount: number | null;
+ /** Columns, or `null` when N/A. */
+ columnCount: number | null;
+ /** Column names, in order. */
+ columns: string[];
+ /** Per-column inferred type. */
+ columnTypes: Array<{ name: string; type: ColumnType }>;
+ /** Approximate payload size in bytes. */
+ size: number;
+ /** ISO timestamp — when first added. */
+ created: string;
+ /** ISO timestamp — when last changed. */
+ modified: string;
+}
+
+/**
+ * The by-name reference object copied into a spec via "Copy Reference"
+ * (spec §05 → Actions): `{ "data": { "name": "MyDataset" } }`.
+ */
+export function datasetReference(name: string): { data: { name: string } } {
+ return { data: { name } };
+}
+
+/** UTF-8 byte length of a string (`TextEncoder` is a platform global, not DOM). */
+function byteLength(str: string): number {
+ return new TextEncoder().encode(str).length;
+}
+
+/**
+ * Tokenize delimited text into rows of string cells, honoring RFC-4180 quoting:
+ * a field wrapped in `"` may contain the delimiter, newlines, and escaped quotes
+ * (`""` → `"`); a quote is only special at the start of a field. This mirrors how
+ * d3-dsv (the parser Vega-Lite uses at render time) reads the same text, so the
+ * profile Astrolabe shows agrees with the field names the chart actually sees.
+ * Whitespace is NOT trimmed — like d3, the cell is taken verbatim (type inference
+ * trims internally when classifying, so numeric columns still detect).
+ */
+function tokenizeDelimited(text: string, delimiter: string): string[][] {
+ const rows: string[][] = [];
+ let row: string[] = [];
+ let field = '';
+ let inQuotes = false;
+ let started = false; // any character seen for the current record?
+ let i = text.charCodeAt(0) === 0xfeff ? 1 : 0; // skip a leading BOM
+ const n = text.length;
+
+ const endField = () => {
+ row.push(field);
+ field = '';
+ };
+ const endRow = () => {
+ endField();
+ rows.push(row);
+ row = [];
+ started = false;
+ };
+
+ while (i < n) {
+ const c = text[i];
+ if (inQuotes) {
+ if (c === '"') {
+ if (text[i + 1] === '"') {
+ field += '"';
+ i += 2;
+ continue;
+ }
+ inQuotes = false;
+ i++;
+ continue;
+ }
+ field += c;
+ i++;
+ continue;
+ }
+ if (c === '"' && field === '') {
+ inQuotes = true;
+ started = true;
+ i++;
+ continue;
+ }
+ if (c === delimiter) {
+ endField();
+ started = true;
+ i++;
+ continue;
+ }
+ if (c === '\n' || c === '\r') {
+ if (c === '\r' && text[i + 1] === '\n') i++;
+ endRow();
+ i++;
+ continue;
+ }
+ field += c;
+ started = true;
+ i++;
+ }
+ // Flush a final field/row only if the last record had content (no phantom row
+ // from a trailing newline).
+ if (started || field !== '' || row.length > 0) endRow();
+ return rows;
+}
+
+/**
+ * Parse delimited text into rows-of-objects (header row → keys), RFC-4180 quoting
+ * aware (see `tokenizeDelimited`). The first record is the header; each later
+ * record is zipped header→cell. A record with fewer cells than the header leaves
+ * the missing columns `undefined`; extra cells beyond the header are ignored.
+ * Fully-empty records (e.g. a blank line) are skipped.
+ */
+export function parseDelimited(
+ text: string,
+ format: 'csv' | 'tsv',
+): Array> {
+ const delimiter = format === 'tsv' ? '\t' : ',';
+ const records = tokenizeDelimited(text, delimiter);
+ if (records.length < 2) return [];
+
+ const header = records[0];
+ const rows: Array> = [];
+ for (let i = 1; i < records.length; i++) {
+ const cells = records[i];
+ if (cells.length === 1 && cells[0] === '') continue; // blank record
+ const row: Record = {};
+ for (let c = 0; c < header.length; c++) {
+ row[header[c]] = c < cells.length ? cells[c] : undefined;
+ }
+ rows.push(row);
+ }
+ return rows;
+}
+
+/** A JSON value is tabular when it is a non-empty array of plain objects. */
+function asObjectRows(value: unknown): Array> | null {
+ if (!Array.isArray(value) || value.length === 0) return null;
+ const allObjects = value.every((v) => v !== null && typeof v === 'object' && !Array.isArray(v));
+ return allObjects ? (value as Array>) : null;
+}
+
+/**
+ * Orchestrate profiling for a dataset payload: compute `size` (always), decide
+ * tabular vs N/A by source/format, and delegate to `profileData`.
+ *
+ * - `url` (any format) → N/A profile; size = byte length of the URL string.
+ * - inline `json` → rows when a non-empty array of objects, else N/A.
+ * - inline `topojson` → N/A (non-tabular).
+ * - inline `csv` / `tsv` → `parseDelimited` rows.
+ *
+ * `size` is the UTF-8 byte length of the raw string for csv/tsv/url, or of
+ * `JSON.stringify(data)` for json/topojson.
+ */
+export function computeDatasetProfile(
+ data: unknown,
+ format: DataFormat,
+ source: DataSource,
+): DatasetProfile {
+ if (source === 'url') {
+ const url = typeof data === 'string' ? data : (JSON.stringify(data) ?? '');
+ return profileData(null, byteLength(url));
+ }
+
+ switch (format) {
+ case 'csv':
+ case 'tsv': {
+ const text = typeof data === 'string' ? data : (JSON.stringify(data) ?? '');
+ return profileData(parseDelimited(text, format), byteLength(text));
+ }
+ case 'json': {
+ const size = byteLength(JSON.stringify(data) ?? '');
+ return profileData(asObjectRows(data), size);
+ }
+ case 'topojson':
+ default: {
+ const size = byteLength(JSON.stringify(data) ?? '');
+ return profileData(null, size);
+ }
+ }
+}
+
+export interface CreateDatasetOptions {
+ /** The dataset name (uniqueness is enforced upstream — see naming.ts). */
+ name: string;
+ /** The payload, shaped per source/format (see `Dataset.data`). */
+ data: unknown;
+ /** One of `json`, `csv`, `tsv`, `topojson`. */
+ format: DataFormat;
+ /** One of `inline` or `url`. */
+ source: DataSource;
+ /** Optional free-form note. */
+ comment?: string;
+ /** Clock injection for deterministic tests; defaults to the current time. */
+ now?: Date;
+ /** Id injection for deterministic tests; defaults to `Date.now()`. */
+ id?: number;
+}
+
+/**
+ * Create a Dataset: stamps version/timestamps and runs `computeDatasetProfile` to
+ * fill the derived summary fields. `id` defaults to `Date.now()` (numeric;
+ * collision-prone for batch creation — see the DatasetStore.add TODO) and is
+ * injectable for tests.
+ */
+export function createDataset(options: CreateDatasetOptions): Dataset {
+ const now = options.now ?? new Date();
+ const iso = now.toISOString();
+ const profile = computeDatasetProfile(options.data, options.format, options.source);
+
+ return {
+ id: options.id ?? Date.now(),
+ version: CURRENT_DATASET_VERSION,
+ name: options.name,
+ data: options.data,
+ format: options.format,
+ source: options.source,
+ comment: options.comment ?? '',
+ rowCount: profile.rowCount,
+ columnCount: profile.columnCount,
+ columns: profile.columns,
+ columnTypes: profile.columnTypes,
+ size: profile.size,
+ created: iso,
+ modified: iso,
+ };
+}
diff --git a/src/core/naming.test.ts b/src/core/naming.test.ts
new file mode 100644
index 0000000..abf6bc4
--- /dev/null
+++ b/src/core/naming.test.ts
@@ -0,0 +1,53 @@
+import { describe, expect, test } from 'vitest';
+import { isNameTaken, makeUniqueName } from './naming';
+
+describe('isNameTaken', () => {
+ const datasets = [
+ { id: 1, name: 'Sales' },
+ { id: 2, name: 'Regions' },
+ ];
+
+ test('matches case-insensitively', () => {
+ expect(isNameTaken('sales', datasets)).toBe(true);
+ expect(isNameTaken('SALES', datasets)).toBe(true);
+ expect(isNameTaken('Unknown', datasets)).toBe(false);
+ });
+
+ test('trims the desired name before comparing', () => {
+ expect(isNameTaken(' Sales ', datasets)).toBe(true);
+ });
+
+ test('excludeId lets a record ignore itself (e.g. a case-only rename)', () => {
+ expect(isNameTaken('Sales', datasets, 1)).toBe(false);
+ expect(isNameTaken('sales', datasets, 1)).toBe(false);
+ // Renaming to a name owned by a *different* record is still taken.
+ expect(isNameTaken('Regions', datasets, 1)).toBe(true);
+ });
+});
+
+describe('makeUniqueName', () => {
+ test('returns the trimmed desired name when free', () => {
+ expect(makeUniqueName('Fresh', ['Sales'])).toBe('Fresh');
+ expect(makeUniqueName(' Fresh ', ['Sales'])).toBe('Fresh');
+ });
+
+ test('suffixes collisions: Name -> Name 2 -> Name 3', () => {
+ expect(makeUniqueName('Name', ['Name'])).toBe('Name 2');
+ expect(makeUniqueName('Name', ['Name', 'Name 2'])).toBe('Name 3');
+ });
+
+ test('comparison is case-insensitive but casing is preserved', () => {
+ expect(makeUniqueName('Name', ['name'])).toBe('Name 2');
+ expect(makeUniqueName('MyData', ['mydata'])).toBe('MyData 2');
+ });
+
+ test('within-batch reservation is the caller responsibility: a single call does not reserve', () => {
+ // Two consecutive calls with the same existing set both yield "Name 2" —
+ // the caller must reserve each chosen name as it goes (arch 07 §5).
+ const existing = ['Name'];
+ expect(makeUniqueName('Name', existing)).toBe('Name 2');
+ expect(makeUniqueName('Name', existing)).toBe('Name 2');
+ // Reserving manually advances to the next free slot.
+ expect(makeUniqueName('Name', [...existing, 'Name 2'])).toBe('Name 3');
+ });
+});
diff --git a/src/core/naming.ts b/src/core/naming.ts
new file mode 100644
index 0000000..6e5f7e2
--- /dev/null
+++ b/src/core/naming.ts
@@ -0,0 +1,46 @@
+/**
+ * Name uniqueness for datasets (spec §05 → Naming & Uniqueness;
+ * docs/architecture/07 §2).
+ *
+ * Portable core: no browser APIs, no React, no store access — plain data in,
+ * plain data out. Dataset names are the key snippets reference via
+ * `datasetRefs` and the contract a Vega-Lite spec uses through
+ * `{ "data": { "name": "..." } }`, so they must be unique. Comparisons are
+ * **case-insensitive** throughout (`Sales` and `sales` collide) so a single
+ * display name maps to a single dataset regardless of how a reference is typed.
+ *
+ * - `isNameTaken` — reject duplicate create/rename in the UI. `excludeId` lets
+ * a rename ignore the record being renamed (renaming `Sales` to `Sales`, or a
+ * case-only edit, is not a self-collision).
+ * - `makeUniqueName` — for non-interactive paths (import, extract, build chart)
+ * where blocking the user is worse than a silent, reported rename: derive the
+ * next free `${base} ${n}` (n ≥ 2). We never parse meaning out of a name; a
+ * base already ending in a number still just gets a suffix (`Q1 2024 2`).
+ */
+
+/** Case-insensitive: is `desired` already used by another dataset (minus `excludeId`)? */
+export function isNameTaken(
+ desired: string,
+ datasets: ReadonlyArray<{ id: number; name: string }>,
+ excludeId?: number,
+): boolean {
+ const lower = desired.trim().toLowerCase();
+ return datasets.some((d) => d.id !== excludeId && d.name.toLowerCase() === lower);
+}
+
+/**
+ * Returns `desired` (trimmed) if free, else the first available `${desired} ${n}`
+ * (n ≥ 2). `existingNames` is the set of names already in the collection.
+ * Comparison is case-insensitive; the returned name preserves `desired`'s casing.
+ */
+export function makeUniqueName(desired: string, existingNames: Iterable): string {
+ const taken = new Set();
+ for (const n of existingNames) taken.add(n.toLowerCase());
+
+ const base = desired.trim();
+ if (!taken.has(base.toLowerCase())) return base;
+
+ let n = 2;
+ while (taken.has(`${base} ${n}`.toLowerCase())) n++;
+ return `${base} ${n}`;
+}
diff --git a/src/core/profile.test.ts b/src/core/profile.test.ts
new file mode 100644
index 0000000..b7b6a44
--- /dev/null
+++ b/src/core/profile.test.ts
@@ -0,0 +1,55 @@
+import { describe, expect, test } from 'vitest';
+import { profileData } from './profile';
+
+describe('profileData', () => {
+ test('profiles a JSON-array dataset fully', () => {
+ const rows = [
+ { city: 'Kyiv', pop: 2900000, capital: 'true' },
+ { city: 'Lviv', pop: 720000, capital: 'false' },
+ ];
+ const profile = profileData(rows, 123);
+ expect(profile.rowCount).toBe(2);
+ expect(profile.columnCount).toBe(3);
+ expect(profile.columns).toEqual(['city', 'pop', 'capital']);
+ expect(profile.columnTypes).toEqual([
+ { name: 'city', type: 'string' },
+ { name: 'pop', type: 'number' },
+ { name: 'capital', type: 'boolean' },
+ ]);
+ expect(profile.size).toBe(123);
+ });
+
+ test('null rows (URL) return the N/A profile but keep size', () => {
+ const profile = profileData(null, 42);
+ expect(profile.rowCount).toBeNull();
+ expect(profile.columnCount).toBeNull();
+ expect(profile.columns).toEqual([]);
+ expect(profile.columnTypes).toEqual([]);
+ expect(profile.size).toBe(42);
+ });
+
+ test('an empty array returns the N/A profile but keeps size', () => {
+ const profile = profileData([], 7);
+ expect(profile.rowCount).toBeNull();
+ expect(profile.columnCount).toBeNull();
+ expect(profile.columns).toEqual([]);
+ expect(profile.size).toBe(7);
+ });
+
+ test('column order follows first-seen key order across ragged rows', () => {
+ const rows = [{ a: 1 }, { b: 2, a: 3 }, { c: 4 }];
+ const profile = profileData(rows, 0);
+ expect(profile.columns).toEqual(['a', 'b', 'c']);
+ });
+
+ test('respects the sampling cap: rows beyond the head do not affect type', () => {
+ // First 200 rows numeric; row 201 is text. With head-only sampling the
+ // column still reads as number (an accepted, documented trade-off).
+ const rows: Array> = [];
+ for (let i = 0; i < 200; i++) rows.push({ v: i });
+ rows.push({ v: 'tail-text' });
+ const profile = profileData(rows, 0);
+ expect(profile.rowCount).toBe(201); // count reflects the whole payload
+ expect(profile.columnTypes).toEqual([{ name: 'v', type: 'number' }]);
+ });
+});
diff --git a/src/core/profile.ts b/src/core/profile.ts
new file mode 100644
index 0000000..d19fe36
--- /dev/null
+++ b/src/core/profile.ts
@@ -0,0 +1,99 @@
+/**
+ * Dataset profiling (spec §05 → Profiling; docs/architecture/06 §3–§4).
+ *
+ * Portable core: no browser APIs, no React. A **profile** is the set of derived
+ * summary fields stored on a dataset record so the UI can describe it without
+ * re-parsing the payload: row/column counts, column names (in order), a per-column
+ * inferred type, and an approximate byte size.
+ *
+ * `profileData` takes already-parsed rows-of-objects (the tabular form of a
+ * CSV/TSV/JSON-array payload) plus a precomputed `size`. Parsing the delimited
+ * text and deciding the payload shape happen *upstream* (see dataset.ts) so this
+ * function stays pure and trivially testable.
+ *
+ * - `null` rows (URL data) or an empty array (non-tabular) → the **N/A profile**
+ * (`rowCount`/`columnCount` null, empty columns), but `size` is still carried.
+ * - Columns are the union of keys across all rows, in **first-seen order**.
+ * - `rowCount`/`columnCount`/`size` reflect the **whole** payload; only type
+ * inference samples — capped at the first `SAMPLE_SIZE` rows for speed and
+ * determinism (arch 06 §4).
+ */
+
+import { inferColumnType, type ColumnType } from './type-inference';
+
+export interface ColumnTypeInfo {
+ /** The column name. */
+ name: string;
+ /** The inferred display type for the column. */
+ type: ColumnType;
+}
+
+export interface DatasetProfile {
+ /** Data rows, or `null` when N/A (URL / non-tabular). */
+ rowCount: number | null;
+ /** Columns, or `null` when N/A. */
+ columnCount: number | null;
+ /** Column names, in first-seen order. */
+ columns: string[];
+ /** Per-column inferred type. */
+ columnTypes: ColumnTypeInfo[];
+ /** Approximate payload size in bytes. */
+ size: number;
+}
+
+/** The N/A profile for URL / non-tabular data — still carries the byte size. */
+const naProfile = (size: number): DatasetProfile => ({
+ rowCount: null,
+ columnCount: null,
+ columns: [],
+ columnTypes: [],
+ size,
+});
+
+/**
+ * Cap on rows fed to type inference. Counts and size scan the whole payload; only
+ * the per-value type check is bounded, sampling the head for determinism (§4).
+ */
+const SAMPLE_SIZE = 200;
+
+const sampleRows = (rows: ReadonlyArray): ReadonlyArray =>
+ rows.length <= SAMPLE_SIZE ? rows : rows.slice(0, SAMPLE_SIZE);
+
+/**
+ * Profile a dataset payload. `rows` is the tabular form (CSV/TSV/JSON-array)
+ * already parsed to rows-of-objects, or `null` for URL / non-tabular data;
+ * `size` is the precomputed byte length of the stored payload.
+ */
+export function profileData(
+ rows: ReadonlyArray> | null,
+ size: number,
+): DatasetProfile {
+ if (!rows || rows.length === 0) return naProfile(size);
+
+ // Column order = first-seen order across all rows (handles ragged rows).
+ const columns: string[] = [];
+ const seen = new Set();
+ for (const row of rows) {
+ for (const key of Object.keys(row)) {
+ if (!seen.has(key)) {
+ seen.add(key);
+ columns.push(key);
+ }
+ }
+ }
+ if (columns.length === 0) return naProfile(size);
+
+ const sample = sampleRows(rows);
+ const columnTypes = columns.map((name) => ({
+ name,
+ type: inferColumnType(sample.map((r) => r[name])),
+ }));
+
+ return {
+ rowCount: rows.length,
+ columnCount: columns.length,
+ columns,
+ columnTypes,
+ size,
+ };
+}
diff --git a/src/core/rendering.test.ts b/src/core/rendering.test.ts
index 44dfbef..213c6d9 100644
--- a/src/core/rendering.test.ts
+++ b/src/core/rendering.test.ts
@@ -1,5 +1,10 @@
import { describe, expect, test } from 'vitest';
-import { escapeVegaField, prepareSpecForRender } from './rendering';
+import {
+ DatasetNotFoundError,
+ escapeVegaField,
+ prepareSpecForRender,
+ type ResolvableDataset,
+} from './rendering';
describe('prepareSpecForRender', () => {
test('returns a deep copy, never the same reference', () => {
@@ -95,6 +100,122 @@ describe('prepareSpecForRender — fit modes (spec §04 Rendering Contract step
});
});
+describe('prepareSpecForRender — dataset resolution (spec §04 Rendering Contract step 1)', () => {
+ const datasets: ResolvableDataset[] = [
+ { name: 'JsonDs', data: [{ a: 1 }], format: 'json', source: 'inline' },
+ { name: 'CsvDs', data: 'a,b\n1,2', format: 'csv', source: 'inline' },
+ { name: 'TsvDs', data: 'a\tb\n1\t2', format: 'tsv', source: 'inline' },
+ {
+ name: 'TopoDs',
+ data: { type: 'Topology', objects: {} },
+ format: 'topojson',
+ source: 'inline',
+ },
+ { name: 'UrlDs', data: 'https://x/y.csv', format: 'csv', source: 'url' },
+ ];
+
+ test('inline JSON → values inlined', () => {
+ const out = prepareSpecForRender({ data: { name: 'JsonDs' }, mark: 'bar' }, { datasets });
+ expect(out.data).toEqual({ values: [{ a: 1 }] });
+ });
+
+ test('inline CSV → raw text inlined, tagged csv', () => {
+ const out = prepareSpecForRender({ data: { name: 'CsvDs' } }, { datasets });
+ expect(out.data).toEqual({ values: 'a,b\n1,2', format: { type: 'csv' } });
+ });
+
+ test('inline TSV → raw text inlined, tagged tsv', () => {
+ const out = prepareSpecForRender({ data: { name: 'TsvDs' } }, { datasets });
+ expect(out.data).toEqual({ values: 'a\tb\n1\t2', format: { type: 'tsv' } });
+ });
+
+ test('inline TopoJSON → value inlined, tagged topojson (preserving feature)', () => {
+ const out = prepareSpecForRender(
+ { data: { name: 'TopoDs', format: { feature: 'counties' } } },
+ { datasets },
+ );
+ expect(out.data).toEqual({
+ values: { type: 'Topology', objects: {} },
+ format: { feature: 'counties', type: 'topojson' },
+ });
+ });
+
+ test('URL → url reference tagged with the dataset format', () => {
+ const out = prepareSpecForRender({ data: { name: 'UrlDs' } }, { datasets });
+ expect(out.data).toEqual({ url: 'https://x/y.csv', format: { type: 'csv' } });
+ });
+
+ test('resolves references in nested layer / concat / child sub-specs', () => {
+ const spec = {
+ layer: [{ data: { name: 'JsonDs' } }],
+ spec: { hconcat: [{ data: { name: 'CsvDs' } }] },
+ };
+ const out = prepareSpecForRender(spec, { datasets }) as unknown as {
+ layer: Array<{ data: unknown }>;
+ spec: { hconcat: Array<{ data: unknown }> };
+ };
+ expect(out.layer[0].data).toEqual({ values: [{ a: 1 }] });
+ expect(out.spec.hconcat[0].data).toEqual({ values: 'a,b\n1,2', format: { type: 'csv' } });
+ });
+
+ test('matches dataset names case-insensitively', () => {
+ const out = prepareSpecForRender({ data: { name: 'jsonds' } }, { datasets });
+ expect(out.data).toEqual({ values: [{ a: 1 }] });
+ });
+
+ test('throws DatasetNotFoundError naming the missing dataset', () => {
+ expect(() => prepareSpecForRender({ data: { name: 'Missing' } }, { datasets })).toThrow(
+ DatasetNotFoundError,
+ );
+ expect(() => prepareSpecForRender({ data: { name: 'Missing' } }, { datasets })).toThrow(
+ /Missing/,
+ );
+ });
+
+ test('a library reference with no datasets provided is still not-found', () => {
+ expect(() => prepareSpecForRender({ data: { name: 'Anything' } })).toThrow(
+ DatasetNotFoundError,
+ );
+ });
+
+ test('a spec with no named refs passes through untouched even with no datasets', () => {
+ const spec = { data: { values: [{ a: 1 }] }, mark: 'bar' };
+ expect(prepareSpecForRender(spec)).toEqual(spec);
+ });
+
+ test('a self-defined top-level datasets name is left untouched and does not throw', () => {
+ const spec = { datasets: { local: [{ a: 1 }] }, data: { name: 'local' }, mark: 'bar' };
+ const out = prepareSpecForRender(spec, { datasets });
+ expect(out.data).toEqual({ name: 'local' });
+ });
+
+ test('resolution runs before fit-mode: a ref + fit mode produce both transforms', () => {
+ const spec = { data: { name: 'JsonDs' }, mark: 'bar' };
+ const out = prepareSpecForRender(spec, { datasets, fitMode: 'full' }) as unknown as {
+ data: unknown;
+ width: string;
+ height: string;
+ };
+ expect(out.data).toEqual({ values: [{ a: 1 }] });
+ expect(out.width).toBe('container');
+ expect(out.height).toBe('container');
+ });
+
+ test('copy-not-mutate: the input spec is untouched during resolution', () => {
+ const spec = { data: { name: 'JsonDs' }, mark: 'bar' };
+ const before = structuredClone(spec);
+ prepareSpecForRender(spec, { datasets });
+ expect(spec).toEqual(before);
+ });
+
+ test('preserves pre-existing reference keys other than name', () => {
+ const out = prepareSpecForRender({ data: { name: 'JsonDs', foo: 1 } }, { datasets }) as {
+ data: Record;
+ };
+ expect(out.data).toEqual({ values: [{ a: 1 }], foo: 1 });
+ });
+});
+
describe('escapeVegaField', () => {
test('escapes dots and brackets that VL treats as accessors', () => {
expect(escapeVegaField('user.age')).toBe('user\\.age');
diff --git a/src/core/rendering.ts b/src/core/rendering.ts
index bd64a89..9bbb6f8 100644
--- a/src/core/rendering.ts
+++ b/src/core/rendering.ts
@@ -7,19 +7,59 @@
* deterministic steps, in order, **on a deep copy** so the user's stored spec is
* never mutated by rendering:
*
- * 1. Dataset reference resolution — arrives in M3 (no-op here).
+ * 1. Dataset reference resolution — implemented in M3 (see below).
* 2. Fit-mode sizing — implemented in M2.
*
+ * Step 1 (spec §04 → Rendering Contract): every named-data reference
+ * (`{ data: { name } }`) is replaced in-place with the referenced library
+ * dataset's actual contents, shaped by source and format. A name the spec defines
+ * for itself via a top-level `datasets` object is left untouched (Vega-Lite
+ * resolves it natively); an unknown library name throws `DatasetNotFoundError`.
+ * Resolution recurses into the same nested sub-specs as fit-mode, runs before
+ * sizing, and operates only on the copy.
+ *
* The copy-not-mutate invariant and the call site the renderer depends on are
- * fixed; M3 fills in step 1 without the preview pipeline changing shape.
+ * fixed.
*/
+import type { DataFormat } from './format-detection';
+import type { DataSource } from './dataset';
+
/** Preview sizing modes (spec §04 → Fit / Sizing Modes). `default` = Original. */
export type FitMode = 'default' | 'width' | 'height' | 'full';
+/**
+ * The minimal structural view of a dataset that reference resolution needs. The
+ * DatasetStore's records are structurally compatible, so passing them works
+ * without importing the full `Dataset` type (and keeps this free of any cycle).
+ */
+export interface ResolvableDataset {
+ /** The library name a spec references via `{ data: { name } }`. */
+ name: string;
+ /** The payload — see `Dataset.data` for the per-source/format shape. */
+ data: unknown;
+ /** One of `json`, `csv`, `tsv`, `topojson`. */
+ format: DataFormat;
+ /** One of `inline` or `url`. */
+ source: DataSource;
+}
+
+/** Thrown when a spec references a library dataset name that does not exist. */
+export class DatasetNotFoundError extends Error {
+ /** The missing dataset's name, so callers can build a tailored, fixable message. */
+ readonly datasetName: string;
+ constructor(name: string) {
+ super(`Dataset not found: "${name}"`);
+ this.name = 'DatasetNotFoundError';
+ this.datasetName = name;
+ }
+}
+
export interface PrepareOptions {
/** Active fit mode. Defaults to `'default'` (Original — spec sizing untouched). */
fitMode?: FitMode;
+ /** The dataset library used to resolve named-data references (step 1). */
+ datasets?: ReadonlyArray;
}
/** The container/sub-spec keys the rendering contract recurses into (spec §04). */
@@ -71,6 +111,83 @@ function applyFitMode(node: unknown, mode: FitMode): void {
if (isSpecNode(node.spec)) applyFitMode(node.spec, mode);
}
+/** The set of dataset names a spec defines for itself via top-level `datasets`. */
+function selfDefinedDatasetNames(spec: unknown): Set {
+ const names = new Set();
+ if (isSpecNode(spec)) {
+ const datasets = spec.datasets;
+ if (isSpecNode(datasets)) for (const key of Object.keys(datasets)) names.add(key);
+ }
+ return names;
+}
+
+/**
+ * Build the replacement `data` object for one resolved reference (spec §04 →
+ * Rendering Contract, step 1). `rest` is the reference's other keys (e.g. a
+ * `format` carrying a TopoJSON `feature`); the incoming `name` is dropped and any
+ * pre-existing `format` is merged so such keys survive.
+ */
+function resolvedData(dataset: ResolvableDataset, rest: SpecNode): SpecNode {
+ const restFormat = isSpecNode(rest.format) ? rest.format : {};
+ if (dataset.source === 'url') {
+ return {
+ ...rest,
+ url: typeof dataset.data === 'string' ? dataset.data : (JSON.stringify(dataset.data) ?? ''),
+ format: { ...restFormat, type: dataset.format },
+ };
+ }
+ switch (dataset.format) {
+ case 'json':
+ return { ...rest, values: dataset.data };
+ case 'topojson':
+ return { ...rest, values: dataset.data, format: { ...restFormat, type: 'topojson' } };
+ case 'csv':
+ case 'tsv':
+ return { ...rest, values: dataset.data, format: { ...restFormat, type: dataset.format } };
+ }
+}
+
+/**
+ * Replace every named-data reference in `node` with its library dataset's
+ * contents, recursing through the entire spec (arrays and objects) so refs
+ * anywhere are resolved — matching `extractDatasetRefs`. A self-defined name is
+ * left untouched; an unknown library name throws `DatasetNotFoundError`. Matching
+ * is case-insensitive, mirroring naming.ts. Mutates in place; the caller already
+ * works on a copy.
+ */
+function resolveDatasetRefs(
+ node: unknown,
+ byName: Map,
+ selfDefined: Set,
+): void {
+ if (Array.isArray(node)) {
+ for (const item of node) resolveDatasetRefs(item, byName, selfDefined);
+ return;
+ }
+ if (!isSpecNode(node)) return;
+
+ const data = node.data;
+ if (isSpecNode(data) && typeof data.name === 'string') {
+ const name = data.name;
+ if (!selfDefined.has(name)) {
+ const dataset = byName.get(name.toLowerCase());
+ if (!dataset) throw new DatasetNotFoundError(name);
+ const { name: _drop, ...rest } = data;
+ node.data = resolvedData(dataset, rest);
+ }
+ }
+
+ // TODO: this walks EVERY key, so it also descends into inlined data payloads
+ // (the just-resolved `values`, a spec's `datasets`/`data.values`). That's
+ // wasteful for large inline data on every debounced render, and a row with a
+ // field literally named `data` holding `{ name: "x" }` would be spuriously
+ // resolved or throw DatasetNotFoundError. extractDatasetRefs shares this broad
+ // walk. A scoped walk (recurse only into the known sub-spec/container keys +
+ // `transform[].lookup.from`, never into data payloads) would be safer and
+ // faster — change deliberately, with tests for where refs may legally appear.
+ for (const key of Object.keys(node)) resolveDatasetRefs(node[key], byName, selfDefined);
+}
+
/**
* Escape `.`/`[`/`]` so Vega-Lite treats a string as a literal field name rather
* than a nested-property accessor (docs/architecture/05 §4). Used wherever
@@ -88,7 +205,12 @@ export function escapeVegaField(name: string): string {
export function prepareSpecForRender(spec: T, options: PrepareOptions = {}): T {
const copy = structuredClone(spec);
- // 1. M3: resolveDatasetRefs(copy, datasets)
+ // 1. Dataset reference resolution — runs before sizing, on the same copy.
+ const datasets = options.datasets ?? [];
+ const byName = new Map();
+ for (const d of datasets) byName.set(d.name.toLowerCase(), d);
+ resolveDatasetRefs(copy, byName, selfDefinedDatasetNames(copy));
+
// 2. Fit-mode sizing.
applyFitMode(copy, options.fitMode ?? 'default');
diff --git a/src/core/spec-refs.test.ts b/src/core/spec-refs.test.ts
new file mode 100644
index 0000000..1f60bc1
--- /dev/null
+++ b/src/core/spec-refs.test.ts
@@ -0,0 +1,108 @@
+import { describe, expect, test } from 'vitest';
+import { extractDatasetRefs, recomputeDatasetRefs, renameDatasetInSpec } from './spec-refs';
+
+describe('extractDatasetRefs', () => {
+ test('collects a top-level named-data reference', () => {
+ expect(extractDatasetRefs({ data: { name: 'Sales' }, mark: 'bar' })).toEqual(['Sales']);
+ });
+
+ test('collects per-layer references', () => {
+ const spec = {
+ layer: [
+ { data: { name: 'A' }, mark: 'bar' },
+ { data: { name: 'B' }, mark: 'line' },
+ ],
+ };
+ expect(extractDatasetRefs(spec).sort()).toEqual(['A', 'B']);
+ });
+
+ test('collects references inside concat and a child spec (facet/repeat)', () => {
+ const spec = {
+ facet: { field: 'g' },
+ spec: { hconcat: [{ data: { name: 'C' } }, { vconcat: [{ data: { name: 'D' } }] }] },
+ };
+ expect(extractDatasetRefs(spec).sort()).toEqual(['C', 'D']);
+ });
+
+ test('accepts a JSON-text spec', () => {
+ const spec = JSON.stringify({ data: { name: 'FromString' } });
+ expect(extractDatasetRefs(spec)).toEqual(['FromString']);
+ });
+
+ test('unparseable string yields no refs', () => {
+ expect(extractDatasetRefs('{ not valid json')).toEqual([]);
+ });
+
+ test('inline-data and url-data references (no name) are ignored', () => {
+ expect(extractDatasetRefs({ data: { values: [{ a: 1 }] } })).toEqual([]);
+ expect(extractDatasetRefs({ data: { url: 'http://x/y.csv' } })).toEqual([]);
+ });
+
+ test('excludes names the spec defines for itself via top-level datasets', () => {
+ const spec = {
+ datasets: { foo: [{ a: 1 }] },
+ data: { name: 'foo' },
+ layer: [{ data: { name: 'Library' } }],
+ };
+ // "foo" is self-defined and not a library dependency; "Library" is.
+ expect(extractDatasetRefs(spec)).toEqual(['Library']);
+ });
+});
+
+describe('recomputeDatasetRefs', () => {
+ test('returns sorted, de-duped names', () => {
+ const spec = {
+ layer: [{ data: { name: 'B' } }, { data: { name: 'A' } }, { data: { name: 'B' } }],
+ };
+ expect(recomputeDatasetRefs(spec)).toEqual(['A', 'B']);
+ });
+});
+
+describe('renameDatasetInSpec', () => {
+ test('rewrites every occurrence of the old name', () => {
+ const spec = {
+ data: { name: 'Old' },
+ layer: [{ data: { name: 'Old' } }, { data: { name: 'Other' } }],
+ };
+ const out = renameDatasetInSpec(spec, 'Old', 'New');
+ expect(out.data.name).toBe('New');
+ expect(out.layer[0].data.name).toBe('New');
+ expect(out.layer[1].data.name).toBe('Other');
+ });
+
+ test('preserves other keys on the data object', () => {
+ const spec = { data: { name: 'Old', format: { type: 'csv' } } };
+ const out = renameDatasetInSpec(spec, 'Old', 'New');
+ expect(out.data).toEqual({ name: 'New', format: { type: 'csv' } });
+ });
+
+ test('does not mutate the input', () => {
+ const spec = { data: { name: 'Old' } };
+ const before = structuredClone(spec);
+ renameDatasetInSpec(spec, 'Old', 'New');
+ expect(spec).toEqual(before);
+ });
+
+ test('preserves object shape for an object spec', () => {
+ const out = renameDatasetInSpec({ data: { name: 'Old' } }, 'Old', 'New');
+ expect(typeof out).toBe('object');
+ });
+
+ test('preserves string shape for a string spec (returns pretty JSON text)', () => {
+ const out = renameDatasetInSpec(JSON.stringify({ data: { name: 'Old' } }), 'Old', 'New');
+ expect(typeof out).toBe('string');
+ expect(JSON.parse(out)).toEqual({ data: { name: 'New' } });
+ });
+
+ test('returns an unparseable string spec unchanged', () => {
+ const bad = '{ not json';
+ expect(renameDatasetInSpec(bad, 'Old', 'New')).toBe(bad);
+ });
+
+ test("does not rename a name defined by the spec's own top-level datasets", () => {
+ const spec = { datasets: { Old: [{ a: 1 }] }, data: { name: 'Old' } };
+ const out = renameDatasetInSpec(spec, 'Old', 'New');
+ expect(out.data.name).toBe('Old'); // self-defined — left untouched
+ expect(Object.keys(out.datasets)).toEqual(['Old']);
+ });
+});
diff --git a/src/core/spec-refs.ts b/src/core/spec-refs.ts
new file mode 100644
index 0000000..69b15ae
--- /dev/null
+++ b/src/core/spec-refs.ts
@@ -0,0 +1,128 @@
+/**
+ * Snippet → dataset reference extraction and rename propagation
+ * (spec §09F → Cross-entity relationships; docs/architecture/07 §3.1, §6).
+ *
+ * Portable core: no browser APIs, no React, no store access. A Vega-Lite spec
+ * references named data through `{ "data": { "name": "MyDataset" } }`, which can
+ * appear at the top level, per-layer, or inside `spec`/`facet`/concat children.
+ * Rather than enumerate the grammar, we walk the spec recursively and collect
+ * every `{ data: { name } }` we find — the single source of truth for "what does
+ * this spec reference", which the renderer's resolution must agree with.
+ *
+ * A spec may be stored as an **object** or as **JSON text** (see spec §09A); we
+ * normalize once at the boundary (unparseable text → no refs / unchanged spec) so
+ * the recursive walk never has to care.
+ *
+ * Refinement beyond the doc sketch: a spec may define its OWN inline named
+ * datasets via a top-level `datasets` object (e.g.
+ * `{ "datasets": { "foo": [...] }, "data": { "name": "foo" } }`). Names satisfied
+ * by the spec's own `datasets` are NOT library dependencies, so they are excluded
+ * from extraction and left untouched on rename — keeping extraction consistent
+ * with the renderer's resolution, which likewise must not treat self-defined
+ * names as library refs.
+ */
+
+type Json = unknown;
+
+/** Parse a string spec; an unparseable draft simply has no resolvable refs. */
+function safeParse(s: string): Json {
+ try {
+ return JSON.parse(s);
+ } catch {
+ return null;
+ }
+}
+
+/** The set of dataset names a spec defines for itself via top-level `datasets`. */
+function selfDefinedNames(spec: Json): Set {
+ const names = new Set();
+ if (spec && typeof spec === 'object' && !Array.isArray(spec)) {
+ const datasets = (spec as Record).datasets;
+ if (datasets && typeof datasets === 'object' && !Array.isArray(datasets)) {
+ for (const key of Object.keys(datasets)) names.add(key);
+ }
+ }
+ return names;
+}
+
+/**
+ * Collects every **library** dataset name referenced by `{ data: { name } }`
+ * anywhere in the spec, excluding names the spec defines for itself via a
+ * top-level `datasets` object. Accepts an object or JSON text.
+ */
+export function extractDatasetRefs(spec: Json): string[] {
+ const root = typeof spec === 'string' ? safeParse(spec) : spec;
+ const selfDefined = selfDefinedNames(root);
+ const names = new Set();
+
+ const walk = (node: Json): void => {
+ if (Array.isArray(node)) {
+ for (const item of node) walk(item);
+ return;
+ }
+ if (node && typeof node === 'object') {
+ const obj = node as Record;
+ const data = obj.data as Record | undefined;
+ if (data && typeof data === 'object' && typeof data.name === 'string') {
+ if (!selfDefined.has(data.name)) names.add(data.name);
+ }
+ // TODO: walks every key, so it also descends into data payloads
+ // (`data.values`, top-level `datasets`). A row with a field named `data`
+ // holding `{ name: "x" }` is falsely counted as a reference. Shared with
+ // rendering.ts resolveDatasetRefs — scope both to the keys where refs can
+ // legally appear, together and with tests. Benign for typical data.
+ for (const key of Object.keys(obj)) walk(obj[key]);
+ }
+ };
+
+ walk(root);
+ return [...names];
+}
+
+/** The list stored on `snippet.datasetRefs` — sorted + de-duped for stable diffs. */
+export function recomputeDatasetRefs(spec: Json): string[] {
+ return [...new Set(extractDatasetRefs(spec))].sort();
+}
+
+/**
+ * Returns a copy of `spec` with every `data.name === oldName` replaced by
+ * `newName`, recursing through arrays and objects. A name that is a top-level
+ * `datasets` key (self-defined) is left untouched. The input's stored shape is
+ * preserved: a string spec is parsed, rewritten, and re-serialized as pretty JSON
+ * text; an object spec returns an object. The generic `` reflects this shape
+ * preservation. The input is never mutated.
+ */
+export function renameDatasetInSpec(spec: T, oldName: string, newName: string): T {
+ const isString = typeof spec === 'string';
+ const root = isString ? safeParse(spec) : spec;
+ // Unparseable text → return it unchanged (nothing resolvable to rewrite).
+ if (isString && root === null) return spec;
+
+ const selfDefined = selfDefinedNames(root);
+
+ const rewrite = (node: Json): Json => {
+ if (Array.isArray(node)) return node.map(rewrite);
+ if (node && typeof node === 'object') {
+ const out: Record = {};
+ for (const [k, v] of Object.entries(node as Record)) {
+ if (
+ k === 'data' &&
+ v &&
+ typeof v === 'object' &&
+ !Array.isArray(v) &&
+ (v as Record).name === oldName &&
+ !selfDefined.has(oldName)
+ ) {
+ out[k] = { ...v, name: newName };
+ } else {
+ out[k] = rewrite(v);
+ }
+ }
+ return out;
+ }
+ return node;
+ };
+
+ const rewritten = rewrite(root);
+ return (isString ? JSON.stringify(rewritten, null, 2) : rewritten) as T;
+}
diff --git a/src/core/type-inference.test.ts b/src/core/type-inference.test.ts
new file mode 100644
index 0000000..7094dc8
--- /dev/null
+++ b/src/core/type-inference.test.ts
@@ -0,0 +1,58 @@
+import { describe, expect, test } from 'vitest';
+import { inferColumnType } from './type-inference';
+
+describe('inferColumnType', () => {
+ test('detects a clean number column', () => {
+ expect(inferColumnType([1, 2, 3])).toBe('number');
+ expect(inferColumnType(['1', '2.5', ' 3 '])).toBe('number');
+ });
+
+ test('detects a clean string column', () => {
+ expect(inferColumnType(['apple', 'banana', 'cherry'])).toBe('string');
+ });
+
+ test('detects a clean date column', () => {
+ expect(inferColumnType(['2024-01-01', '2025-12-31'])).toBe('date');
+ expect(inferColumnType(['2024/01/01', '12/31/2024'])).toBe('date');
+ });
+
+ test('detects a clean boolean column (any case)', () => {
+ expect(inferColumnType(['true', 'false'])).toBe('boolean');
+ expect(inferColumnType(['TRUE', 'False'])).toBe('boolean');
+ expect(inferColumnType([true, false])).toBe('boolean');
+ });
+
+ test('a mixed column falls to string', () => {
+ expect(inferColumnType([1, 2, 'three'])).toBe('string');
+ expect(inferColumnType(['2024-01-01', 'not-a-date'])).toBe('string');
+ });
+
+ test('ignores empty / whitespace cells before classifying', () => {
+ expect(inferColumnType([1, null, 2, undefined, ' ', 3])).toBe('number');
+ expect(inferColumnType(['true', '', 'false', ' '])).toBe('boolean');
+ });
+
+ test('an all-empty or zero-length column is string', () => {
+ expect(inferColumnType([])).toBe('string');
+ expect(inferColumnType([null, undefined, '', ' '])).toBe('string');
+ });
+
+ test('precedence: a true/false column is boolean, not string', () => {
+ expect(inferColumnType(['true', 'false'])).toBe('boolean');
+ });
+
+ test('precedence: a bare-year column is number, not date', () => {
+ expect(inferColumnType(['2024', '2025'])).toBe('number');
+ });
+
+ test('date shape guard rejects bare numbers and words even if Date.parse might accept them', () => {
+ expect(inferColumnType(['42'])).toBe('number');
+ expect(inferColumnType(['hello'])).toBe('string');
+ expect(inferColumnType(['March'])).toBe('string');
+ });
+
+ test('0 and 1 are number, never boolean', () => {
+ expect(inferColumnType([0, 1, 0, 1])).toBe('number');
+ expect(inferColumnType(['0', '1'])).toBe('number');
+ });
+});
diff --git a/src/core/type-inference.ts b/src/core/type-inference.ts
new file mode 100644
index 0000000..2651dc9
--- /dev/null
+++ b/src/core/type-inference.ts
@@ -0,0 +1,72 @@
+/**
+ * Column type inference (spec §05 → Profiling; docs/architecture/06 §2).
+ *
+ * Portable core: no browser APIs, no React — usable in Node and tests. Given the
+ * values of a single column, decide which of four display types it holds:
+ * `number`, `boolean`, `date`, or `string`. This is a **display hint** only —
+ * nothing downstream coerces values from it, and Vega-Lite does its own type
+ * handling at render time — so the rules favour being simple and predictable
+ * over being clever.
+ *
+ * The algorithm (arch 06 §2):
+ * 1. Drop empties (null/undefined/whitespace-only string) — empty cells carry
+ * no type signal.
+ * 2. An all-empty (or zero-length) column is `string` — no evidence otherwise.
+ * 3. Run the checks in precedence order **boolean → number → date → string**,
+ * narrowest evidence to widest; the first for which *every* present value
+ * matches wins. One stray value knocks the column down to the next candidate.
+ *
+ * Date detection guards with a shape regex *before* trusting `Date.parse`, which
+ * on some engines accepts `"42"` or `"March"` and would swallow number/string
+ * columns. `0`/`1` are numbers, never booleans.
+ */
+
+export type ColumnType = 'number' | 'string' | 'date' | 'boolean';
+
+/** Empty cells (null/undefined/whitespace-only string) carry no type signal. */
+const isEmpty = (v: unknown): boolean =>
+ v === null || v === undefined || (typeof v === 'string' && v.trim() === '');
+
+/** Native numbers pass when finite; strings must parse to a finite, non-NaN number. */
+const isNumeric = (v: unknown): boolean => {
+ if (typeof v === 'number') return Number.isFinite(v);
+ if (typeof v !== 'string') return false;
+ const t = v.trim();
+ if (t === '') return false; // reject blank so Number("") === 0 can't sneak through
+ const n = Number(t);
+ return !Number.isNaN(n) && Number.isFinite(n);
+};
+
+/** Native booleans pass; otherwise the trimmed, lower-cased string is exactly true/false. */
+const isBoolean = (v: unknown): boolean => {
+ if (typeof v === 'boolean') return true;
+ if (typeof v !== 'string') return false;
+ const t = v.trim().toLowerCase();
+ return t === 'true' || t === 'false';
+};
+
+/**
+ * Shape guard for dates: a leading `YYYY-MM-DD`/`YYYY/MM/DD`, or `M/D/YYYY`.
+ * Required *before* `Date.parse` — see the module note above.
+ */
+const DATE_SHAPE = /^\d{4}[-/]\d{2}[-/]\d{2}|^\d{1,2}\/\d{1,2}\/\d{4}/;
+const isDate = (v: unknown): boolean => {
+ if (typeof v !== 'string') return false;
+ const t = v.trim();
+ return DATE_SHAPE.test(t) && !Number.isNaN(Date.parse(t));
+};
+
+/**
+ * Infer one of four column types from a sample of column values. Empty cells are
+ * ignored; an all-empty (or zero-length) column is `string`. Precedence:
+ * boolean → number → date → string.
+ */
+export function inferColumnType(values: readonly unknown[]): ColumnType {
+ const present = values.filter((v) => !isEmpty(v));
+ if (present.length === 0) return 'string';
+
+ if (present.every(isBoolean)) return 'boolean';
+ if (present.every(isNumeric)) return 'number';
+ if (present.every(isDate)) return 'date';
+ return 'string';
+}