diff --git a/CHANGELOG.md b/CHANGELOG.md index bff53cf..a25c1bb 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,7 +7,28 @@ this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.htm ## [Unreleased] -Nothing yet. +### Added +- JSON Formatter: a Tree view alongside the text output. It shows the same + document the text does, with sorted keys and expanded embedded JSON, as a + collapsible tree. Only the rows on screen are rendered, so large documents + scroll smoothly. It opens a couple of levels deep, and has Expand all / + Collapse all and full keyboard navigation (arrow keys, Home/End, Enter). +- JSON Formatter: copy the selected tree node's path as an RFC 6901 pointer + (`/items/0/id`), a dot path (`items[0].id`) or a jq path (`.items[0].id`). +- JSON Formatter: embedded JSON strings can be expanded one field at a time. + A field nested inside another payload unlocks once its parent is expanded. +- JSON Formatter: auto-fixes for almost-JSON. It strips a JSONP wrapper, the + `)]}'` prefix, a byte-order mark and comments; converts single quotes; + quotes bare keys; and removes trailing commas. Apply them in place, or + review the change side by side in Diff first. +- JSON Formatter: lossless mode. When a document has numbers JavaScript + cannot hold exactly, like `12345678901234567890`, format and minify keep + them digit for digit instead of rounding them. This includes numbers inside + expanded embedded JSON. + +### Fixed +- JSON Formatter: with "Expand embedded JSON" on, a key named `__proto__` + was silently dropped from the output. It is now kept like any other key. ## [0.3.0] — 2026-09-25 diff --git a/package.json b/package.json index 3f47644..a7baf00 100644 --- a/package.json +++ b/package.json @@ -21,6 +21,7 @@ "@codemirror/merge": "6.12.2", "@codemirror/state": "6.7.5", "@codemirror/view": "6.43.12", + "@tanstack/react-virtual": "3.14.12", "change-case": "5.4.4", "clsx": "2.1.1", "cmdk": "1.1.1", diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 253770e..6123b54 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -17,6 +17,9 @@ importers: '@codemirror/view': specifier: 6.43.12 version: 6.43.12 + '@tanstack/react-virtual': + specifier: 3.14.12 + version: 3.14.12(react-dom@19.3.0(react@19.3.0))(react@19.3.0) change-case: specifier: 5.4.4 version: 5.4.4 @@ -1012,6 +1015,15 @@ packages: peerDependencies: vite: ^5.2.0 || ^6 || ^7 || ^8 + '@tanstack/react-virtual@3.14.12': + resolution: {integrity: sha512-EOJkBss/FRqzFrCpX8M96lIh+ZMc0jPt0LILhxWQ3tOHQ3XdVvuE0jceS2yZXpDtayNfpjeg2EnuYPm90CxIlA==} + peerDependencies: + react: ^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 + react-dom: ^16.8.0 || ^17.0.0 || ^18.0.0 || ^19.0.0 + + '@tanstack/virtual-core@3.17.10': + resolution: {integrity: sha512-eMorKWIi2nekElu+8DgSMJR4iR+sTDeh+W8VjKiMH79yUfEDJ3W9aL/96/wwbxNjyV6EfN39HKszSxjd90zAqQ==} + '@types/chai@5.2.3': resolution: {integrity: sha512-Mw558oeA9fFbv65/y4mHtXDs9bPnFMZAL/jxdPFUpOHHIXX91mcgEHbS5Lahr+pwZFR8A7GQleRWeI6cGFC2UA==} @@ -2418,6 +2430,14 @@ snapshots: tailwindcss: 4.3.3 vite: 8.3.0(@types/node@24.9.2)(esbuild@0.28.1)(jiti@2.7.0) + '@tanstack/react-virtual@3.14.12(react-dom@19.3.0(react@19.3.0))(react@19.3.0)': + dependencies: + '@tanstack/virtual-core': 3.17.10 + react: 19.3.0 + react-dom: 19.3.0(react@19.3.0) + + '@tanstack/virtual-core@3.17.10': {} + '@types/chai@5.2.3': dependencies: '@types/deep-eql': 4.0.2 diff --git a/src/lib/persist/handoff.ts b/src/lib/persist/handoff.ts new file mode 100644 index 0000000..2f8e059 --- /dev/null +++ b/src/lib/persist/handoff.ts @@ -0,0 +1,30 @@ +import { TOOLS_BY_SLUG, type ToolSlug } from '@/lib/registry' +import { readIdb, writeIdb } from './idb' +import { readLocal, writeLocal } from './store' + +/** + * Seed another tool's saved state before navigating to it -- e.g. the JSON + * tool opening its auto-fix preview in Diff. + * + * The patch is merged over whatever the target has stored, and `useToolState` + * merges that over the tool's own defaults on mount, so the caller only names + * the fields it means to set and never needs the target's module (importing it + * would pull that tool's chunk into the caller's). + * + * Resolves false for a `store: 'none'` tool (there is nowhere to put it) or a + * failed write, so the caller can stay put instead of opening an empty tool. + */ +export async function handOff(slug: ToolSlug, patch: Partial): Promise { + const tool = TOOLS_BY_SLUG[slug] + const version = tool.stateVersion + + if (tool.store.kind === 'idb') { + const current = (await readIdb(slug, version)) ?? {} + return writeIdb(slug, version, { ...current, ...patch }) + } + if (tool.store.kind === 'local') { + const current = readLocal(slug, version) ?? {} + return writeLocal(slug, version, { ...current, ...patch }) === 'ok' + } + return false +} diff --git a/src/tools/json/JsonTool.tsx b/src/tools/json/JsonTool.tsx index c8f291b..c76547f 100644 --- a/src/tools/json/JsonTool.tsx +++ b/src/tools/json/JsonTool.tsx @@ -1,19 +1,30 @@ -import { useDeferredValue, useMemo } from 'react' -import { AlertTriangle, Braces, CheckCircle2, Wand2 } from 'lucide-react' +import { useDeferredValue, useMemo, useState } from 'react' +import { useNavigate } from 'react-router' +import { AlertTriangle, Braces, CheckCircle2, GitCompare, Wand2 } from 'lucide-react' import { ToolFrame } from '@/components/layout/ToolFrame' import { TwoPane } from '@/components/layout/TwoPane' import { Badge } from '@/components/ui/Badge' -import { Button } from '@/components/ui/Button' +import { Button, Segmented } from '@/components/ui/Button' import { CodeArea } from '@/components/ui/CodeArea' import { CopyButton } from '@/components/ui/CopyButton' import { Select, Toggle } from '@/components/ui/Select' +import { handOff } from '@/lib/persist/handoff' import { useToolState } from '@/lib/persist/useToolState' import { useShortcuts } from '@/lib/keys/useShortcuts' import { useToolUsageTracker } from '@/lib/prefs' import { copyText } from '@/lib/util/clipboard' import { formatBytes, formatCount } from '@/lib/util/bytes' +import { cn } from '@/lib/util/cn' import { analyze, stripJsonc, type JsonIssue } from './core/errors' -import { expandEmbedded, findEmbeddedJson } from './core/embedded' +import { + embeddedFields, + expandEmbedded, + findEmbeddedJson, + type EmbeddedField, + type EmbeddedOptions, +} from './core/embedded' +import { runFixes } from './core/fixes' +import { losslessSupported, parseLossless } from './core/rawjson' import { escapeAsJsonString, findNumberIssues, @@ -21,10 +32,18 @@ import { minifyValue, ndjsonToArray, unescapeJsonString, + withSortedKeys, type Indent, } from './core/format' +import { TreeView } from './TreeView' type Mode = 'format' | 'minify' | 'escape' | 'unescape' +type View = 'text' | 'tree' + +const VIEWS = [ + { value: 'text', label: 'Text' }, + { value: 'tree', label: 'Tree' }, +] as const interface State { text: string @@ -32,7 +51,11 @@ interface State { indent: Indent sortKeys: boolean lenient: boolean + /** Expand every embedded JSON string. */ expandEmbedded: boolean + /** When not expanding all: the pointers picked one by one. */ + expandPointers: string[] + view: View } const INITIAL: State = { @@ -42,6 +65,8 @@ const INITIAL: State = { sortKeys: false, lenient: false, expandEmbedded: false, + expandPointers: [], + view: 'text', } function IssueCard({ issue }: { issue: JsonIssue }) { @@ -83,6 +108,84 @@ function IssueCard({ issue }: { issue: JsonIssue }) { ) } +const CHIP_LIMIT = 6 + +function EmbeddedPanel({ + fields, + isActive, + anyActive, + onToggle, + onAll, +}: { + fields: readonly EmbeddedField[] + isActive: (pointer: string) => boolean + anyActive: boolean + onToggle: (pointer: string) => void + onAll: (expand: boolean) => void +}) { + const [showAll, setShowAll] = useState(false) + const activeCount = fields.filter((f) => isActive(f.pointer)).length + const visible = showAll ? fields : fields.slice(0, CHIP_LIMIT) + + return ( +
+
+ + + {activeCount > 0 ? `Expanded ${formatCount(activeCount)} of ` : 'Found '} + + {formatCount(fields.length)} embedded JSON {fields.length === 1 ? 'string' : 'strings'} + + {fields.some((f) => f.parent !== null) && ' (some double-encoded)'} + + +
+
+ {visible.map((f) => { + const active = isActive(f.pointer) + // A field inside another payload only exists once that one is open. + const blocked = f.parent !== null && !isActive(f.parent) + return ( + + ) + })} + {fields.length > CHIP_LIMIT && ( + + )} +
+
+ ) +} + export default function JsonTool() { useToolUsageTracker('json') const { state, setState, reset } = useToolState('json', INITIAL) @@ -94,19 +197,68 @@ export default function JsonTool() { ) const numberIssues = useMemo(() => findNumberIssues(deferred.text), [deferred.text]) + const precisionLoss = numberIssues.filter((n) => n.kind === 'precision') + const reformatted = numberIssues.filter((n) => n.kind === 'reformatted') + + // Lossless mode: when the number gate trips, re-parse keeping those literals + // as their source text, so format/minify/sort round-trip them byte-exactly. + // Nothing in this tool does arithmetic on values, so nothing else changes. + const lossless = losslessSupported && numberIssues.length > 0 + const parsed = useMemo(() => { + if (analysis.value === undefined || !lossless) return analysis.value + try { + return parseLossless(analysis.flavor === 'jsonc' ? stripJsonc(deferred.text) : deferred.text) + } catch { + return analysis.value + } + }, [analysis, lossless, deferred.text]) + const embeddedOptions = useMemo( + () => (lossless ? { parse: parseLossless } : {}), + [lossless], + ) // Structured logs put their payload in a string, often double-encoded. Finding // those is the difference between reading this document here and copying a // field into another tab twice. - const embedded = useMemo( - () => (analysis.value === undefined ? [] : findEmbeddedJson(analysis.value)), - [analysis.value], + const fields = useMemo( + () => (parsed === undefined ? [] : embeddedFields(findEmbeddedJson(parsed, embeddedOptions))), + [parsed, embeddedOptions], ) - const precisionLoss = numberIssues.filter((n) => n.kind === 'precision') - const reformatted = numberIssues.filter((n) => n.kind === 'reformatted') + + // Expansion is a view over the parsed value; the source text is untouched, + // so toggling it off restores the original exactly. Sorting happens here too, + // so the text output and the tree show the same document. + const shown = useMemo(() => { + if (parsed === undefined) return undefined + const { expandEmbedded: all, expandPointers } = deferred + const value = all + ? expandEmbedded(parsed, embeddedOptions).value + : expandPointers.length > 0 + ? expandEmbedded(parsed, { ...embeddedOptions, only: new Set(expandPointers) }).value + : parsed + return deferred.sortKeys ? withSortedKeys(value) : value + }, [parsed, embeddedOptions, deferred]) + + const isFieldActive = (pointer: string) => + state.expandEmbedded || state.expandPointers.includes(pointer) + const anyFieldActive = fields.some((f) => isFieldActive(f.pointer)) + + function toggleField(pointer: string) { + setState((p) => { + const current = p.expandEmbedded ? fields.map((f) => f.pointer) : p.expandPointers + const next = current.includes(pointer) + ? current.filter((x) => x !== pointer) + : [...current, pointer] + return { ...p, expandEmbedded: false, expandPointers: next } + }) + } + + function setAllFields(expand: boolean) { + setState((p) => ({ ...p, expandEmbedded: expand, expandPointers: [] })) + } const output = useMemo(() => { - const { text, mode, indent, sortKeys } = deferred + const { text, mode, indent } = deferred if (text === '') return '' if (mode === 'escape') return escapeAsJsonString(text, { ascii: false }) @@ -114,18 +266,12 @@ export default function JsonTool() { const r = unescapeJsonString(text) return r.ok ? r.value : '' } - if (analysis.value === undefined) return '' - - // Expansion is a view over the parsed value; the source text is untouched, - // so toggling it off restores the original exactly. - const value = deferred.expandEmbedded - ? expandEmbedded(analysis.value).value - : analysis.value + if (shown === undefined) return '' return mode === 'minify' - ? minifyValue(value, { sortKeys }) - : formatValue(value, { indent, sortKeys }) - }, [analysis.value, deferred]) + ? minifyValue(shown, { sortKeys: false }) + : formatValue(shown, { indent, sortKeys: false }) + }, [shown, deferred]) const unescapeError = deferred.mode === 'unescape' && deferred.text !== '' @@ -137,6 +283,40 @@ export default function JsonTool() { const isTextMode = state.mode === 'escape' || state.mode === 'unescape' const valid = analysis.flavor === 'json' || analysis.flavor === 'jsonc' + const showTree = state.mode === 'format' && state.view === 'tree' && valid && shown !== undefined + + // Only worth running on a document that does not parse. Counts are exact: + // each fix sees the previous one's output. + const fixes = useMemo( + () => + analysis.flavor === 'invalid' || analysis.flavor === 'ndjson' ? runFixes(deferred.text) : null, + [analysis.flavor, deferred.text], + ) + const fixable = fixes !== null && fixes.applied.length > 0 + const fixesYieldJson = useMemo(() => { + if (!fixable) return false + try { + JSON.parse(fixes.text) + return true + } catch { + return false + } + }, [fixable, fixes]) + const lenientWouldHelp = + fixes?.applied.some((f) => f.id === 'comments' || f.id === 'trailingCommas') === true + + const navigate = useNavigate() + const [handOffFailed, setHandOffFailed] = useState(false) + + async function reviewInDiff() { + if (fixes === null) return + const ok = await handOff<{ left: string; right: string }>('diff', { + left: state.text, + right: fixes.text, + }) + setHandOffFailed(!ok) + if (ok) void navigate('/diff') + } function applyToInput(next: string) { setState((p) => ({ ...p, text: next })) @@ -193,6 +373,14 @@ export default function JsonTool() { )} + {state.mode === 'format' && ( + setState((p) => ({ ...p, view }))} + /> + )} + {!isTextMode && ( <> Allow comments / trailing commas - {embedded.length > 0 && ( - setState((p) => ({ ...p, expandEmbedded: v }))} - > + {fields.length > 0 && ( + Expand embedded JSON )} @@ -272,62 +457,73 @@ export default function JsonTool() { )} - {!valid && !isTextMode && analysis.issues.length > 0 && !state.lenient && ( -
- - Comments or trailing commas? - - - -
- )} - - {/* The whole point of the feature is that you did not know the - payload was in there. It has to announce itself. */} - {embedded.length > 0 && !isTextMode && ( + {fixable && !isTextMode && (
-
- +
+ - {state.expandEmbedded ? 'Expanded ' : 'Found '} - {formatCount(embedded.length)} embedded JSON{' '} - {embedded.length === 1 ? 'string' : 'strings'} - - {embedded.some((e) => e.depth > 1) && ' (some double-encoded)'} + {fixesYieldJson ? 'Fixable' : 'Partly fixable'}: + {' '} + + {fixes.applied.map((f) => f.label).join(' · ')} + + {!fixesYieldJson && ( + — some problems will remain + )} +
+
+ -
-
- {embedded.slice(0, 6).map((e) => ( - setState((p) => ({ ...p, lenient: true }))} > - {e.dotPath} - - ))} - {embedded.length > 6 && ( - - +{formatCount(embedded.length - 6)} more - + Allow comments / trailing commas + + )} + {handOffFailed && ( + Could not open Diff. )}
)} - {precisionLoss.length > 0 && ( + {/* The whole point of the feature is that you did not know the + payload was in there. It has to announce itself. */} + {fields.length > 0 && !isTextMode && ( + + )} + + {lossless && valid && ( +
+ Lossless mode.{' '} + {precisionLoss.length > 0 + ? `${formatCount(precisionLoss.length)} number${precisionLoss.length === 1 ? '' : 's'} (e.g. ${precisionLoss[0]!.literal}) cannot be held as a JavaScript number, so` + : 'To keep the output byte-exact,'}{' '} + numbers are passed through exactly as written — never rounded or reprinted. +
+ )} + + {!lossless && precisionLoss.length > 0 && (
{formatCount(precisionLoss.length)} number @@ -338,7 +534,7 @@ export default function JsonTool() {
)} - {precisionLoss.length === 0 && reformatted.length > 0 && ( + {!lossless && precisionLoss.length === 0 && reformatted.length > 0 && (
{formatCount(reformatted.length)} number {reformatted.length === 1 ? '' : 's'} keep their exact value but are printed @@ -356,6 +552,8 @@ export default function JsonTool() { ))}
+ ) : showTree ? ( + ) : ( )} diff --git a/src/tools/json/TreeView.tsx b/src/tools/json/TreeView.tsx new file mode 100644 index 0000000..3c1a40e --- /dev/null +++ b/src/tools/json/TreeView.tsx @@ -0,0 +1,283 @@ +import { useEffect, useMemo, useRef, useState, useSyncExternalStore, type KeyboardEvent } from 'react' +import { useVirtualizer } from '@tanstack/react-virtual' +import { ChevronRight } from 'lucide-react' +import { Button } from '@/components/ui/Button' +import { CopyButton } from '@/components/ui/CopyButton' +import { formatCount } from '@/lib/util/bytes' +import { cn } from '@/lib/util/cn' +import { + allContainerPointers, + countNodes, + defaultExpanded, + flattenTree, + pathOf, + toggle, + type FlatNode, +} from './core/tree' +import { toDotPath, toJsonPointer } from './core/embedded' +import { toJqPath } from './core/paths' + +// Matches the palette rows: a 44px touch target below md, the 22px code-table +// grid from DESIGN.md above it. +const NARROW = '(max-width: 47.99rem)' +const ROW_NARROW = 44 +const ROW_WIDE = 22 +const INDENT = 14 + +function subscribeNarrow(onChange: () => void) { + const mq = window.matchMedia(NARROW) + mq.addEventListener('change', onChange) + return () => mq.removeEventListener('change', onChange) +} + +function useRowHeight(): number { + const narrow = useSyncExternalStore( + subscribeNarrow, + () => window.matchMedia(NARROW).matches, + () => false, + ) + return narrow ? ROW_NARROW : ROW_WIDE +} + +interface Expansion { + source: unknown + expanded: ReadonlySet + /** + * Until the user toggles something, every new value opens with the default + * expansion. After that their choices are kept across edits -- pointers are + * stable as long as the keys are. + */ + touched: boolean + capped: boolean +} + +function RowValue({ row }: { row: FlatNode }) { + if (row.kind === 'object' || row.kind === 'array') { + const [open, close] = row.kind === 'object' ? ['{', '}'] : ['[', ']'] + const noun = row.kind === 'object' ? 'key' : 'item' + if (row.expanded) return {open} + return ( + + {open}{' '} + + {formatCount(row.childCount)} {noun} + {row.childCount === 1 ? '' : 's'} + {' '} + {close} + + ) + } + return ( + + {row.preview} + + ) +} + +export function TreeView({ value }: { value: unknown }) { + const rowHeight = useRowHeight() + + const [expansion, setExpansion] = useState(() => ({ + source: value, + expanded: defaultExpanded(value), + touched: false, + capped: false, + })) + + // Adjust state during render when the value changes, rather than in an + // effect, so the tree never paints one frame with the previous expansion. + let current = expansion + if (expansion.source !== value) { + current = { + source: value, + expanded: expansion.touched ? expansion.expanded : defaultExpanded(value), + touched: expansion.touched, + capped: false, + } + setExpansion(current) + } + + const rows = useMemo(() => flattenTree(value, current.expanded), [value, current.expanded]) + const total = useMemo(() => countNodes(value), [value]) + + const [selected, setSelected] = useState(null) + const selectedIndex = useMemo( + () => (selected === null ? -1 : rows.findIndex((r) => r.pointer === selected)), + [rows, selected], + ) + + // Copy targets for the selected row. Shown in the header rather than on the + // row itself, so they are reachable on touch and never hover-only. + const selectedPaths = useMemo(() => { + if (selectedIndex < 0) return null + const path = pathOf(rows, selectedIndex) + return { pointer: toJsonPointer(path), dot: toDotPath(path), jq: toJqPath(path) } + }, [rows, selectedIndex]) + + const scrollRef = useRef(null) + // The rule guards React Compiler memoisation, and this build does not run the + // compiler. TanStack Virtual is the documented exception it flags. + // oxlint-disable-next-line react/incompatible-library + const virtualizer = useVirtualizer({ + count: rows.length, + getScrollElement: () => scrollRef.current, + estimateSize: () => rowHeight, + overscan: 20, + }) + + // Rows are a fixed height, so a breakpoint change is the only re-measure. + useEffect(() => { + virtualizer.measure() + }, [rowHeight, virtualizer]) + + function setExpanded(next: ReadonlySet, capped = false) { + setExpansion({ source: value, expanded: next, touched: true, capped }) + } + + function toggleRow(row: FlatNode) { + if (row.childCount > 0) setExpanded(toggle(current.expanded, row.pointer)) + } + + function select(index: number) { + const row = rows[index] + if (row === undefined) return + setSelected(row.pointer) + virtualizer.scrollToIndex(index, { align: 'auto' }) + } + + function onKeyDown(e: KeyboardEvent) { + const i = selectedIndex + const row = rows[i] + switch (e.key) { + case 'ArrowDown': + select(i < 0 ? 0 : Math.min(i + 1, rows.length - 1)) + break + case 'ArrowUp': + select(i < 0 ? 0 : Math.max(i - 1, 0)) + break + case 'Home': + select(0) + break + case 'End': + select(rows.length - 1) + break + case 'ArrowRight': + if (row === undefined) select(0) + else if (row.childCount > 0 && !row.expanded) toggleRow(row) + else if (row.expanded) select(i + 1) + break + case 'ArrowLeft': + if (row === undefined) select(0) + else if (row.expanded) toggleRow(row) + else if (row.parent >= 0) select(row.parent) + break + case 'Enter': + case ' ': + if (row !== undefined) toggleRow(row) + break + default: + return + } + e.preventDefault() + } + + return ( +
+
+ + {formatCount(total)} {total === 1 ? 'node' : 'nodes'} + {current.capped && ` · expanded the first ${formatCount(rows.length)} rows`} + + + + +
+ +
+ {selectedPaths === null ? ( + Select a row to copy its path + ) : ( + <> + + {selectedPaths.dot} + + + + + + )} +
+ +
= 0 ? `json-tree-row-${selectedIndex}` : undefined} + onKeyDown={onKeyDown} + className="min-h-0 flex-1 overflow-auto bg-bg font-mono text-[13px] md:text-[12px] scroll-thin" + > +
+ {virtualizer.getVirtualItems().map((item) => { + const row = rows[item.index]! + const isSelected = item.index === selectedIndex + const isContainer = row.childCount > 0 + return ( +
60 ? row.preview : undefined} + onClick={() => { + setSelected(row.pointer) + toggleRow(row) + }} + className={cn( + 'absolute left-0 top-0 flex w-full cursor-default items-center gap-1 whitespace-nowrap border-l-2 pr-3', + isSelected ? 'border-accent bg-surface-2' : 'border-transparent', + )} + style={{ + height: rowHeight, + transform: `translateY(${item.start}px)`, + paddingLeft: 8 + row.depth * INDENT, + }} + > + + {isContainer && ( + + )} + + {row.key !== null && ( + + {typeof row.key === 'number' ? row.key : JSON.stringify(row.key)} + : + + )} + +
+ ) + })} +
+
+
+ ) +} diff --git a/src/tools/json/core/embedded.test.ts b/src/tools/json/core/embedded.test.ts index 3439231..dadfcc2 100644 --- a/src/tools/json/core/embedded.test.ts +++ b/src/tools/json/core/embedded.test.ts @@ -1,6 +1,7 @@ import fc from 'fast-check' import { describe, expect, it } from 'vitest' import { + embeddedFields, escapePointerToken, expandEmbedded, findEmbeddedJson, @@ -116,6 +117,15 @@ describe('findEmbeddedJson', () => { }) describe('expandEmbedded', () => { + it('keeps a "__proto__" key instead of assigning the prototype', () => { + // JSON.parse makes "__proto__" an own key; `out[key] = v` would instead + // set the prototype, silently dropping the key from the output. + const doc = JSON.parse('{"__proto__":{"x":1},"a":"{\\"b\\":1}"}') as unknown + const { value } = expandEmbedded(doc) + expect(JSON.stringify(value)).toBe('{"__proto__":{"x":1},"a":{"b":1}}') + expect(Object.getPrototypeOf(value)).toBe(Object.prototype) + }) + it('replaces the string with the parsed value', () => { const doc = { level: 'info', payload: '{"userId":42}' } const { value, expanded } = expandEmbedded(doc) @@ -208,3 +218,38 @@ describe('reversibility', () => { expect(expandEmbedded(once).value).toEqual(once) }) }) + +describe('embeddedFields', () => { + const doc = { + once: '{"a":1}', + nested: JSON.stringify({ deep: JSON.stringify({ z: 1 }) }), + } + const fields = embeddedFields(findEmbeddedJson(doc)) + + it('links a field revealed by another to that field', () => { + expect(fields.map((f) => [f.pointer, f.parent])).toEqual([ + ['/once', null], + ['/nested', null], + ['/nested/deep', '/nested'], + ]) + }) + + it('expanding one field leaves the others as raw strings', () => { + const { value } = expandEmbedded(doc, { only: new Set(['/once']) }) + expect(value).toEqual({ once: { a: 1 }, nested: doc.nested }) + }) + + it('a nested field expands only with its parent', () => { + const alone = expandEmbedded(doc, { only: new Set(['/nested/deep']) }).value + expect((alone as { nested: unknown }).nested).toBe(doc.nested) + const outer = expandEmbedded(doc, { only: new Set(['/nested']) }).value + expect((outer as { nested: unknown }).nested).toEqual({ deep: JSON.stringify({ z: 1 }) }) + const both = expandEmbedded(doc, { only: new Set(['/nested', '/nested/deep']) }).value + expect((both as { nested: unknown }).nested).toEqual({ deep: { z: 1 } }) + }) + + it('does not treat a sibling with a common prefix as a parent', () => { + const f = embeddedFields(findEmbeddedJson({ a: '{"x":1}', ab: '{"y":2}' })) + expect(f.map((x) => x.parent)).toEqual([null, null]) + }) +}) diff --git a/src/tools/json/core/embedded.ts b/src/tools/json/core/embedded.ts index 25b48ea..610cdbd 100644 --- a/src/tools/json/core/embedded.ts +++ b/src/tools/json/core/embedded.ts @@ -13,6 +13,8 @@ * get wrong. */ +import { isRawNumber } from './rawjson' + /** RFC 6901: "~" becomes "~0" and "/" becomes "~1", in that order. */ export function escapePointerToken(token: string): string { return token.replaceAll('~', '~0').replaceAll('/', '~1') @@ -28,7 +30,10 @@ export function toDotPath(path: readonly (string | number)[]): string { let out = '' for (const segment of path) { if (typeof segment === 'number') out += `[${segment}]` - else if (/^[A-Za-z_$][\w$]*$/.test(segment)) out += out === '' ? segment : `.${segment}` + // A bare leading `$` would read back as the root marker. + else if (/^[A-Za-z_$][\w$]*$/.test(segment) && !(out === '' && segment === '$')) { + out += out === '' ? segment : `.${segment}` + } else out += `[${JSON.stringify(segment)}]` } return out === '' ? '$' : out @@ -55,11 +60,17 @@ export interface EmbeddedOptions { maxDepth?: number /** Strings longer than this are skipped, to bound the cost of the scan. */ maxStringLength?: number + /** + * How to parse a candidate string. Lossless mode passes `parseLossless`, so a + * big number inside an embedded payload keeps its digits too. + */ + parse?: (text: string) => unknown } const DEFAULTS: Required = { maxDepth: 6, maxStringLength: 2_000_000, + parse: (text) => JSON.parse(text) as unknown, } /** @@ -72,6 +83,7 @@ const DEFAULTS: Required = { export function parseEmbedded( value: string, maxStringLength = DEFAULTS.maxStringLength, + parse = DEFAULTS.parse, ): { kind: 'object' | 'array'; parsed: unknown } | null { if (value.length > maxStringLength) return null @@ -85,7 +97,7 @@ export function parseEmbedded( let parsed: unknown try { - parsed = JSON.parse(trimmed) + parsed = parse(trimmed) } catch { return null } @@ -95,7 +107,7 @@ export function parseEmbedded( } function isPlainContainer(value: unknown): value is Record | unknown[] { - return value !== null && typeof value === 'object' + return value !== null && typeof value === 'object' && !isRawNumber(value) } /** @@ -103,14 +115,14 @@ function isPlainContainer(value: unknown): value is Record | un * after an outer layer is expanded. */ export function findEmbeddedJson(root: unknown, options: EmbeddedOptions = {}): Embedded[] { - const { maxDepth, maxStringLength } = { ...DEFAULTS, ...options } + const { maxDepth, maxStringLength, parse } = { ...DEFAULTS, ...options } const found: Embedded[] = [] const walk = (node: unknown, path: (string | number)[], depth: number): void => { if (depth > maxDepth) return if (typeof node === 'string') { - const hit = parseEmbedded(node, maxStringLength) + const hit = parseEmbedded(node, maxStringLength, parse) if (hit) { found.push({ pointer: toJsonPointer(path), @@ -158,7 +170,7 @@ export function expandEmbedded( root: unknown, options: EmbeddedOptions & { only?: ReadonlySet } = {}, ): ExpandResult { - const { maxDepth, maxStringLength } = { ...DEFAULTS, ...options } + const { maxDepth, maxStringLength, parse } = { ...DEFAULTS, ...options } const only = options.only const expanded: Embedded[] = [] @@ -169,7 +181,7 @@ export function expandEmbedded( const pointer = toJsonPointer(path) if (only !== undefined && !only.has(pointer)) return node - const hit = parseEmbedded(node, maxStringLength) + const hit = parseEmbedded(node, maxStringLength, parse) if (!hit) return node expanded.push({ @@ -190,11 +202,11 @@ export function expandEmbedded( } if (isPlainContainer(node)) { - const out: Record = {} - for (const [key, child] of Object.entries(node)) { - out[key] = transform(child, [...path, key], depth) - } - return out + // fromEntries defines own properties; `out[key] = v` would treat a + // "__proto__" key as the prototype setter and drop it from the output. + return Object.fromEntries( + Object.entries(node).map(([key, child]) => [key, transform(child, [...path, key], depth)]), + ) } return node @@ -202,3 +214,31 @@ export function expandEmbedded( return { value: transform(root, [], 1), expanded } } + +export interface EmbeddedField { + pointer: string + dotPath: string + rawLength: number + /** + * The embedded field this one lives inside, if any. It only exists once that + * field is expanded, so selecting it alone does nothing. + */ + parent: string | null +} + +/** + * Per-field toggles need to know which fields are nested in others: a field + * found inside an embedded payload only exists once that payload is expanded, + * so its toggle is meaningless until then. + */ +export function embeddedFields(found: readonly Embedded[]): EmbeddedField[] { + return found.map((field) => { + // Nearest enclosing field: the longest other pointer that prefixes this one. + let parent: string | null = null + for (const other of found) { + if (!field.pointer.startsWith(other.pointer + '/')) continue + if (parent === null || other.pointer.length > parent.length) parent = other.pointer + } + return { pointer: field.pointer, dotPath: field.dotPath, rawLength: field.rawLength, parent } + }) +} diff --git a/src/tools/json/core/fixes.test.ts b/src/tools/json/core/fixes.test.ts new file mode 100644 index 0000000..156c1b9 --- /dev/null +++ b/src/tools/json/core/fixes.test.ts @@ -0,0 +1,152 @@ +import fc from 'fast-check' +import { describe, expect, it } from 'vitest' +import { + applyEdits, + fixBom, + fixComments, + fixJsonp, + fixSingleQuotes, + fixTrailingCommas, + fixUnquotedKeys, + fixXssi, + runFixes, + type TextEdit, +} from './fixes' + +const apply = (text: string, find: (t: string) => TextEdit[]) => applyEdits(text, find(text)) + +describe('applyEdits', () => { + it('applies in offset order regardless of input order', () => { + expect( + applyEdits('abcdef', [ + { offset: 4, length: 1, content: 'E' }, + { offset: 0, length: 2, content: '' }, + ]), + ).toBe('cdEf') + }) + + it('refuses overlapping edits', () => { + expect(() => + applyEdits('abc', [ + { offset: 0, length: 2, content: '' }, + { offset: 1, length: 1, content: '' }, + ]), + ).toThrow() + }) +}) + +describe('individual fixes', () => { + it('removes a BOM', () => { + expect(apply('\uFEFF{"a":1}', fixBom)).toBe('{"a":1}') + expect(fixBom('{}')).toEqual([]) + }) + + it("strips the )]}' XSSI prefix, with or without a comma", () => { + expect(apply(")]}'\n{\"a\":1}", fixXssi)).toBe('{"a":1}') + expect(apply(")]}',\r\n[1]", fixXssi)).toBe('[1]') + expect(fixXssi('[")]}\'"]')).toEqual([]) + }) + + it('strips a JSONP wrapper', () => { + expect(apply('cb({"a":1});', fixJsonp)).toBe('{"a":1}') + expect(apply('/**/ jQuery.cb_1 ( [1,2] )\n', fixJsonp)).toBe('[1,2]') + expect(fixJsonp('cb(42)')).toEqual([]) + expect(fixJsonp('{"a":"cb("}')).toEqual([]) + }) + + it('removes comments but not comment-like text in strings', () => { + expect(apply('{"u":"http://x", // note\n"b":/* c */1}', fixComments)).toBe( + '{"u":"http://x", \n"b":1}', + ) + expect(fixComments('{"a":"/* no */"}')).toEqual([]) + }) + + it('takes a comment-only line with it', () => { + expect(apply('{\n // note\n "a": 1 // trailing\n}', fixComments)).toBe('{\n "a": 1 \n}') + expect(apply('[\r\n /* x */\r\n 1]', fixComments)).toBe('[\r\n 1]') + }) + + it('converts single-quoted strings, escaping and unescaping quotes', () => { + expect(apply("{'a': 'it\\'s \"x\"'}", fixSingleQuotes)).toBe('{"a": "it\'s \\"x\\""}') + expect(apply("['\\n']", fixSingleQuotes)).toBe('["\\n"]') + expect(fixSingleQuotes('{"a":"don\'t"}')).toEqual([]) + }) + + it('quotes bare keys only in key position', () => { + expect(apply('{a: 1, $b_2 : true, "c": x}', fixUnquotedKeys)).toBe( + '{"a": 1, "$b_2" : true, "c": x}', + ) + expect(apply('{\n // c\n a: 1}', fixUnquotedKeys)).toBe('{\n // c\n "a": 1}') + expect(fixUnquotedKeys('[true, null]')).toEqual([]) + }) + + it('removes trailing commas, looking past comments', () => { + expect(apply('{"a":[1,2,],}', fixTrailingCommas)).toBe('{"a":[1,2]}') + expect(apply('[1, // x\n]', fixTrailingCommas)).toBe('[1 // x\n]') + expect(fixTrailingCommas('{"a":","}')).toEqual([]) + }) +}) + +describe('runFixes', () => { + it('fixes a pasted JS object literal end to end', () => { + const src = `\uFEFFcallback({ + // user record + id: 7, + name: 'O\\'Brien', + tags: ['a', 'b',], + });` + const { text, applied } = runFixes(src) + expect(JSON.parse(text)).toEqual({ id: 7, name: "O'Brien", tags: ['a', 'b'] }) + expect(applied.map((a) => [a.id, a.count])).toEqual([ + ['bom', 1], + ['jsonp', 1], + ['comments', 1], + ['singleQuotes', 3], + ['unquotedKeys', 3], + ['trailingCommas', 2], + ]) + expect(applied.find((a) => a.id === 'unquotedKeys')?.label).toBe('Quote 3 bare keys') + }) + + it('handles the XSSI prefix before anything else reads it as a string', () => { + expect(runFixes(")]}'\n{a:1}").text).toBe('{"a":1}') + }) +}) + +/** Serialise like a sloppy JS literal: bare keys, single quotes, trailing commas, comments. */ +function sloppy(v: unknown): string { + if (typeof v === 'string') { + // Reuse JSON's escaping, then swap the quote style. + const body = JSON.stringify(v).slice(1, -1).replaceAll('\\"', '"').replaceAll("'", "\\'") + return `'${body}'` + } + if (Array.isArray(v)) return `[${v.map(sloppy).join(', ')}${v.length ? ',' : ''}]` + if (v !== null && typeof v === 'object') { + const entries = Object.entries(v).map(([k, x]) => + /^[A-Za-z_$][\w$]*$/.test(k) ? `${k}: ${sloppy(x)}` : `${sloppy(k)}: ${sloppy(x)}`, + ) + return `{ /* obj */ ${entries.join(',\n // item\n')}${entries.length ? ',' : ''} }` + } + return JSON.stringify(v) +} + +describe('properties', () => { + const value = fc.jsonValue({ maxDepth: 3 }).map((v) => JSON.parse(JSON.stringify(v)) as unknown) + + it('changes nothing in valid JSON', () => { + fc.assert( + fc.property(value, fc.constantFrom(0, 2), (v, indent) => { + const text = JSON.stringify(v, null, indent) + expect(runFixes(text)).toEqual({ text, applied: [] }) + }), + ) + }) + + it('repairs a sloppy literal to the same value', () => { + fc.assert( + fc.property(value, (v) => { + expect(JSON.parse(runFixes(sloppy(v)).text)).toEqual(v) + }), + ) + }) +}) diff --git a/src/tools/json/core/fixes.ts b/src/tools/json/core/fixes.ts new file mode 100644 index 0000000..e94cd63 --- /dev/null +++ b/src/tools/json/core/fixes.ts @@ -0,0 +1,288 @@ +/** + * Auto-fixes for "almost JSON": the JS object literals, JSONP responses and + * config files people paste expecting them to parse. + * + * Each fix is a pure function from text to `TextEdit[]`, so it can be tested on + * its own and previewed as a diff before anything touches the input. Fixes run + * as a pipeline -- each one sees the previous one's output -- which is what + * keeps their edits from ever overlapping, and what makes the counts exact. + */ + +export interface TextEdit { + offset: number + length: number + content: string +} + +export type FixId = + | 'bom' + | 'xssi' + | 'jsonp' + | 'comments' + | 'singleQuotes' + | 'unquotedKeys' + | 'trailingCommas' + +export function applyEdits(text: string, edits: readonly TextEdit[]): string { + const sorted = edits.toSorted((a, b) => a.offset - b.offset) + let out = '' + let at = 0 + for (const e of sorted) { + if (e.offset < at) throw new Error(`Overlapping edit at ${e.offset}`) + out += text.slice(at, e.offset) + e.content + at = e.offset + e.length + } + return out + text.slice(at) +} + +// --------------------------------------------------------------------------- +// Tokenizer: just enough structure to know what is inside a string. + +type TokenKind = 'ws' | 'comment' | 'dstring' | 'sstring' | 'punct' | 'word' | 'other' + +interface Token { + kind: TokenKind + offset: number + length: number + /** Strings and block comments: false when the input ended first. */ + closed: boolean +} + +const PUNCT = new Set(['{', '}', '[', ']', ':', ',']) +const WORD_END = /[\s{}[\]:,"'/()]/ + +function readString(text: string, start: number, quote: string): { end: number; closed: boolean } { + let i = start + 1 + while (i < text.length) { + const ch = text[i]! + if (ch === '\\') i += 2 + else if (ch === quote) return { end: i + 1, closed: true } + // A raw newline ends a runaway string, so one stray quote cannot swallow + // the rest of the document. + else if (ch === '\n') return { end: i, closed: false } + else i++ + } + return { end: text.length, closed: false } +} + +export function tokenize(text: string): Token[] { + const tokens: Token[] = [] + let i = 0 + const push = (kind: TokenKind, end: number, closed = true) => { + tokens.push({ kind, offset: i, length: end - i, closed }) + i = end + } + + while (i < text.length) { + const ch = text[i]! + if (/\s/.test(ch)) { + let j = i + 1 + while (j < text.length && /\s/.test(text[j]!)) j++ + push('ws', j) + } else if (ch === '/' && text[i + 1] === '/') { + const nl = text.indexOf('\n', i) + push('comment', nl === -1 ? text.length : nl) + } else if (ch === '/' && text[i + 1] === '*') { + const close = text.indexOf('*/', i + 2) + push('comment', close === -1 ? text.length : close + 2, close !== -1) + } else if (ch === '"' || ch === "'") { + const { end, closed } = readString(text, i, ch) + push(ch === '"' ? 'dstring' : 'sstring', end, closed) + } else if (PUNCT.has(ch)) { + push('punct', i + 1) + } else if (WORD_END.test(ch)) { + push('other', i + 1) + } else { + let j = i + 1 + while (j < text.length && !WORD_END.test(text[j]!)) j++ + push('word', j) + } + } + return tokens +} + +function isTrivia(t: Token): boolean { + return t.kind === 'ws' || t.kind === 'comment' +} + +function significantNeighbour(tokens: readonly Token[], from: number, step: 1 | -1): Token | undefined { + for (let i = from + step; i >= 0 && i < tokens.length; i += step) { + if (!isTrivia(tokens[i]!)) return tokens[i] + } + return undefined +} + +function tokenText(text: string, t: Token): string { + return text.slice(t.offset, t.offset + t.length) +} + +// --------------------------------------------------------------------------- +// The fixes. + +export function fixBom(text: string): TextEdit[] { + return text.startsWith('') ? [{ offset: 0, length: 1, content: '' }] : [] +} + +/** + * The anti-JSON-hijacking prefix Google and Angular APIs send: `)]}'` on its + * own line, sometimes followed by a comma. + */ +export function fixXssi(text: string): TextEdit[] { + const m = /^\s*\)\]\}'[ \t]*,?[ \t]*(\r?\n)?/.exec(text) + return m === null ? [] : [{ offset: 0, length: m[0].length, content: '' }] +} + +/** `callback({...});` -- including the `/**\/` some servers prepend. */ +export function fixJsonp(text: string): TextEdit[] { + const head = /^\s*(?:\/\*\*\/\s*)?[A-Za-z_$][\w$.]*\s*\(\s*/.exec(text) + if (head === null) return [] + const tail = /\s*\)\s*;?\s*$/.exec(text) + if (tail === null || tail.index < head[0].length) return [] + const inner = text.slice(head[0].length, tail.index) + if (!/^[{[]/.test(inner)) return [] + return [ + { offset: 0, length: head[0].length, content: '' }, + { offset: tail.index, length: tail[0].length, content: '' }, + ] +} + +/** + * A comment alone on its line takes the line with it, so the fixed text does + * not keep a whitespace-only line where each one was. + */ +export function fixComments(text: string): TextEdit[] { + const edits: TextEdit[] = [] + for (const t of tokenize(text)) { + if (t.kind !== 'comment') continue + const lineStart = text.lastIndexOf('\n', t.offset - 1) + 1 + const end = t.offset + t.length + // Sticky, so the match starts at `end` without copying the rest of the text. + const rest = /[ \t]*\r?\n/y + rest.lastIndex = end + const nl = rest.exec(text) + const alone = /^[ \t]*$/.test(text.slice(lineStart, t.offset)) && nl !== null + edits.push( + alone + ? { offset: lineStart, length: end + nl[0].length - lineStart, content: '' } + : { offset: t.offset, length: t.length, content: '' }, + ) + } + return edits +} + +/** Re-quote `'…'` as `"…"`: unescape `\'`, escape bare `"`. */ +export function fixSingleQuotes(text: string): TextEdit[] { + const edits: TextEdit[] = [] + for (const t of tokenize(text)) { + if (t.kind !== 'sstring' || !t.closed) continue + const body = text.slice(t.offset + 1, t.offset + t.length - 1) + let out = '' + for (let i = 0; i < body.length; i++) { + const ch = body[i]! + if (ch === '\\') { + const next = body[i + 1] ?? '' + out += next === "'" ? "'" : `\\${next}` + i++ + } else if (ch === '"') { + out += '\\"' + } else { + out += ch + } + } + edits.push({ offset: t.offset, length: t.length, content: `"${out}"` }) + } + return edits +} + +const KEY_WORD = /^[A-Za-z0-9_$]+$/ + +/** `{ foo: 1 }` → `{ "foo": 1 }`: a bare word in key position before a colon. */ +export function fixUnquotedKeys(text: string): TextEdit[] { + const tokens = tokenize(text) + const edits: TextEdit[] = [] + for (const [i, t] of tokens.entries()) { + if (t.kind !== 'word') continue + const word = tokenText(text, t) + if (!KEY_WORD.test(word)) continue + const prev = significantNeighbour(tokens, i, -1) + const next = significantNeighbour(tokens, i, 1) + if (prev?.kind !== 'punct' || next?.kind !== 'punct') continue + const before = tokenText(text, prev) + if ((before !== '{' && before !== ',') || tokenText(text, next) !== ':') continue + edits.push({ offset: t.offset, length: t.length, content: `"${word}"` }) + } + return edits +} + +export function fixTrailingCommas(text: string): TextEdit[] { + const tokens = tokenize(text) + const edits: TextEdit[] = [] + for (const [i, t] of tokens.entries()) { + if (t.kind !== 'punct' || tokenText(text, t) !== ',') continue + const next = significantNeighbour(tokens, i, 1) + if (next === undefined || next.kind !== 'punct') continue + const close = tokenText(text, next) + if (close === '}' || close === ']') edits.push({ offset: t.offset, length: 1, content: '' }) + } + return edits +} + +// --------------------------------------------------------------------------- +// The pipeline. + +interface FixDef { + id: FixId + find: (text: string) => TextEdit[] + label: (count: number) => string +} + +const plural = (n: number, one: string, many = `${one}s`) => `${n} ${n === 1 ? one : many}` + +/** + * Order matters: wrappers come off first so the tokenizer sees the document, + * comments go before trailing commas so `1, // note\n}` is caught, and quotes + * are normalised before keys so `'a': 1` is not also treated as a bare word. + */ +export const FIXES: readonly FixDef[] = [ + { id: 'bom', find: fixBom, label: () => 'Remove byte-order mark' }, + { id: 'xssi', find: fixXssi, label: () => "Strip )]}' prefix" }, + { id: 'jsonp', find: fixJsonp, label: () => 'Strip JSONP wrapper' }, + { id: 'comments', find: fixComments, label: (n) => `Remove ${plural(n, 'comment')}` }, + { + id: 'singleQuotes', + find: fixSingleQuotes, + label: (n) => `Convert ${plural(n, 'single-quoted string')}`, + }, + { id: 'unquotedKeys', find: fixUnquotedKeys, label: (n) => `Quote ${plural(n, 'bare key')}` }, + { + id: 'trailingCommas', + find: fixTrailingCommas, + label: (n) => `Remove ${plural(n, 'trailing comma')}`, + }, +] + +export interface FoundFix { + id: FixId + count: number + label: string +} + +export interface FixResult { + text: string + applied: FoundFix[] +} + +/** Run every fix in order; report the ones that changed something. */ +export function runFixes(text: string): FixResult { + let current = text + const applied: FoundFix[] = [] + for (const fix of FIXES) { + const edits = fix.find(current) + if (edits.length === 0) continue + // Wrapper fixes emit two edits for one change; count changes, not edits. + const count = fix.id === 'jsonp' ? 1 : edits.length + applied.push({ id: fix.id, count, label: fix.label(count) }) + current = applyEdits(current, edits) + } + return { text: current, applied } +} diff --git a/src/tools/json/core/format.ts b/src/tools/json/core/format.ts index 4245288..2c5c80f 100644 --- a/src/tools/json/core/format.ts +++ b/src/tools/json/core/format.ts @@ -1,3 +1,5 @@ +import { isRawNumber } from './rawjson' + export type Indent = 2 | 4 | 'tab' export interface FormatOptions { @@ -12,7 +14,7 @@ function indentValue(indent: Indent): string | number { /** Deep key sort. Never reorders arrays — their order is data. */ export function sortKeysDeep(value: unknown, cmp: (a: string, b: string) => number): unknown { if (Array.isArray(value)) return value.map((v) => sortKeysDeep(v, cmp)) - if (value === null || typeof value !== 'object') return value + if (value === null || typeof value !== 'object' || isRawNumber(value)) return value const entries = Object.entries(value as Record) return Object.fromEntries( @@ -22,6 +24,11 @@ export function sortKeysDeep(value: unknown, cmp: (a: string, b: string) => numb const keyCollator = new Intl.Collator(undefined, { numeric: true, sensitivity: 'variant' }).compare +/** Deep key sort with the same collator the formatter uses. */ +export function withSortedKeys(value: unknown): unknown { + return sortKeysDeep(value, keyCollator) +} + export function formatValue(value: unknown, o: FormatOptions): string { const prepared = o.sortKeys ? sortKeysDeep(value, keyCollator) : value return JSON.stringify(prepared, null, indentValue(o.indent)) diff --git a/src/tools/json/core/paths.test.ts b/src/tools/json/core/paths.test.ts new file mode 100644 index 0000000..357bd99 --- /dev/null +++ b/src/tools/json/core/paths.test.ts @@ -0,0 +1,92 @@ +import fc from 'fast-check' +import { describe, expect, it } from 'vitest' +import { toDotPath, toJsonPointer } from './embedded' +import { + MISSING, + parseDotPath, + parseJqPath, + parsePointer, + resolvePath, + toJqPath, +} from './paths' +import { allContainerPointers, flattenTree, pathOf } from './tree' + +describe('toJqPath', () => { + it('uses dots for identifiers and brackets for the rest', () => { + expect(toJqPath([])).toBe('.') + expect(toJqPath(['items', 0, 'id'])).toBe('.items[0].id') + expect(toJqPath([0, 'a'])).toBe('.[0].a') + expect(toJqPath(['a-b'])).toBe('.["a-b"]') + expect(toJqPath(['a', 'b c'])).toBe('.a["b c"]') + // `$` is a JS identifier character but not a jq one. + expect(toJqPath(['$ref'])).toBe('.["$ref"]') + }) +}) + +describe('toDotPath', () => { + it('does not confuse a "$" key with the root', () => { + expect(toDotPath(['$'])).not.toBe(toDotPath([])) + expect(parseDotPath(toDotPath(['$']))).toEqual(['$']) + }) +}) + +describe('parsers', () => { + it('reads each flavour back', () => { + expect(parsePointer('/a~1b/~0c/0')).toEqual(['a/b', '~c', '0']) + expect(parseDotPath('items[0]["a b"].c')).toEqual(['items', 0, 'a b', 'c']) + expect(parseJqPath('.items[0]["a\\"b"]')).toEqual(['items', 0, 'a"b']) + expect(parseJqPath('.[0]')).toEqual([0]) + }) + + it('rejects garbage', () => { + expect(() => parsePointer('a')).toThrow() + expect(() => parseJqPath('a')).toThrow() + expect(() => parseDotPath('a[x]')).toThrow() + }) +}) + +const doc = { + items: [{ id: 1, 'a b': { '~/': true } }], + '0': 'zero-key', + $: 'dollar', + '': 'empty', + 'é': ['ü'], +} + +describe('every row round-trips in all three flavours', () => { + const { expanded } = allContainerPointers(doc) + const rows = flattenTree(doc, expanded) + + for (const [index, row] of rows.entries()) { + it(row.pointer === '' ? '(root)' : row.pointer, () => { + const path = pathOf(rows, index) + const target = resolvePath(doc, path) + expect(target).not.toBe(MISSING) + expect(toJsonPointer(path)).toBe(row.pointer) + expect(resolvePath(doc, parsePointer(toJsonPointer(path)))).toBe(target) + expect(resolvePath(doc, parseDotPath(toDotPath(path)))).toBe(target) + expect(resolvePath(doc, parseJqPath(toJqPath(path)))).toBe(target) + }) + } +}) + +describe('round-trip properties', () => { + const segment = fc.oneof(fc.nat({ max: 50 }), fc.string(), fc.constantFrom('$', '0', '~1', 'a.b', '[x]')) + + it('dot and jq paths parse back to the exact segments', () => { + fc.assert( + fc.property(fc.array(segment, { maxLength: 6 }), (path) => { + expect(parseDotPath(toDotPath(path))).toEqual(path) + expect(parseJqPath(toJqPath(path))).toEqual(path) + }), + ) + }) + + it('pointers parse back to the stringified segments', () => { + fc.assert( + fc.property(fc.array(segment, { maxLength: 6 }), (path) => { + expect(parsePointer(toJsonPointer(path))).toEqual(path.map(String)) + }), + ) + }) +}) diff --git a/src/tools/json/core/paths.ts b/src/tools/json/core/paths.ts new file mode 100644 index 0000000..69650d4 --- /dev/null +++ b/src/tools/json/core/paths.ts @@ -0,0 +1,123 @@ +/** + * The three path flavours a tree row can be copied as, and their parsers. + * + * - RFC 6901 pointer (`/items/0/id`) -- what JSON Patch and JSON Schema use. + * - Dot/bracket path (`items[0].id`) -- what you paste into JS/TS code. + * - jq path (`.items[0].id`) -- what you paste into a terminal. + * + * The formatters for the first two live in `embedded.ts`. The parsers exist so + * that each flavour is proven to round-trip to the *same node*, which is the + * property that matters: a copied path that points somewhere else is worse + * than none. + */ + +import { isRawNumber } from './rawjson' + +export type PathSegment = string | number + +// jq's identifier grammar is narrower than JS's: no `$`. +const JQ_IDENT = /^[A-Za-z_][A-Za-z0-9_]*$/ + +/** jq path. Non-identifier keys use `.["key"]`, which every jq since 1.5 reads. */ +export function toJqPath(path: readonly PathSegment[]): string { + if (path.length === 0) return '.' + let out = '' + for (const segment of path) { + if (typeof segment === 'number') out += out === '' ? `.[${segment}]` : `[${segment}]` + else if (JQ_IDENT.test(segment)) out += `.${segment}` + else out += out === '' ? `.[${JSON.stringify(segment)}]` : `[${JSON.stringify(segment)}]` + } + return out +} + +/** RFC 6901 pointer to string tokens. Tokens stay strings: "0" may be a key. */ +export function parsePointer(pointer: string): string[] { + if (pointer === '') return [] + if (!pointer.startsWith('/')) throw new Error(`Not a JSON pointer: ${pointer}`) + return pointer + .slice(1) + .split('/') + .map((t) => t.replaceAll('~1', '/').replaceAll('~0', '~')) +} + +/** + * Shared reader for dot paths and jq paths, which differ only in the root + * marker and the identifier grammar. + */ +function parseAccessors(src: string, start: number, ident: RegExp): PathSegment[] { + const out: PathSegment[] = [] + let i = start + while (i < src.length) { + const ch = src[i] + if (ch === '.') { + i++ + if (src[i] === '[') continue + const m = ident.exec(src.slice(i)) + if (m === null) throw new Error(`Expected a key at ${i}`) + out.push(m[0]) + i += m[0].length + } else if (ch === '[') { + if (src[i + 1] === '"') { + // Find the closing quote, honouring escapes, then let JSON unescape. + let j = i + 2 + while (j < src.length && src[j] !== '"') j += src[j] === '\\' ? 2 : 1 + out.push(JSON.parse(src.slice(i + 1, j + 1)) as string) + i = j + 1 + } else { + const m = /^\d+/.exec(src.slice(i + 1)) + if (m === null) throw new Error(`Expected an index at ${i}`) + out.push(Number(m[0])) + i += 1 + m[0].length + } + if (src[i] !== ']') throw new Error(`Expected ] at ${i}`) + i++ + } else if (out.length === 0 && i === start) { + // A dot path's first key has no leading dot. + const m = ident.exec(src.slice(i)) + if (m === null) throw new Error(`Unexpected ${ch} at ${i}`) + out.push(m[0]) + i += m[0].length + } else { + throw new Error(`Unexpected ${ch} at ${i}`) + } + } + return out +} + +export function parseDotPath(path: string): PathSegment[] { + if (path === '$') return [] + return parseAccessors(path, 0, /^[A-Za-z_$][\w$]*/) +} + +export function parseJqPath(path: string): PathSegment[] { + if (path === '.') return [] + if (!path.startsWith('.')) throw new Error(`Not a jq path: ${path}`) + return parseAccessors(path, 0, /^[A-Za-z_][A-Za-z0-9_]*/) +} + +const MISSING = Symbol('missing') + +/** + * Follow a path. Array steps accept a numeric string (pointer tokens are + * strings); object steps accept a number (so `[0]` on `{"0": …}` resolves the + * way JS property access would). + */ +export function resolvePath(root: unknown, path: readonly PathSegment[]): unknown { + let node: unknown = root + for (const segment of path) { + if (Array.isArray(node)) { + const i = typeof segment === 'number' ? segment : /^(0|[1-9]\d*)$/.test(segment) ? Number(segment) : -1 + if (i < 0 || i >= node.length) return MISSING + node = node[i] + } else if (node !== null && typeof node === 'object' && !isRawNumber(node)) { + const key = String(segment) + if (!Object.hasOwn(node, key)) return MISSING + node = (node as Record)[key] + } else { + return MISSING + } + } + return node +} + +export { MISSING } diff --git a/src/tools/json/core/rawjson.test.ts b/src/tools/json/core/rawjson.test.ts new file mode 100644 index 0000000..30a00c3 --- /dev/null +++ b/src/tools/json/core/rawjson.test.ts @@ -0,0 +1,64 @@ +import { describe, expect, it } from 'vitest' +import { isRawNumber, losslessSupported, parseLossless, rawNumberText } from './rawjson' +import { formatValue, minifyValue, withSortedKeys } from './format' +import { expandEmbedded, findEmbeddedJson } from './embedded' +import { allContainerPointers, countNodes, flattenTree, kindOf } from './tree' + +const SRC = '{"z":12345678901234567890,"a":[1.0,-9223372036854775808,1e5,42,0.1],"s":"{\\"n\\":99999999999999999999}"}' + +describe('parseLossless', () => { + it('runs on this platform', () => { + expect(losslessSupported).toBe(true) + }) + + it('keeps lossy and reformatted literals exactly, and leaves the rest as numbers', () => { + const v = parseLossless(SRC) as { z: unknown; a: unknown[] } + expect(isRawNumber(v.z)).toBe(true) + expect(rawNumberText(v.z as never)).toBe('12345678901234567890') + expect(v.a.map((n) => (isRawNumber(n) ? rawNumberText(n) : n))).toEqual([ + '1.0', + '-9223372036854775808', + '1e5', + 42, + 0.1, + ]) + }) + + it('round-trips minify byte-exactly', () => { + expect(minifyValue(parseLossless(SRC), { sortKeys: false })).toBe(SRC) + }) + + it('formats and sorts keys without touching the digits', () => { + const out = formatValue(parseLossless(SRC), { indent: 2, sortKeys: true }) + expect(out).toContain('"z": 12345678901234567890') + expect(out.indexOf('"a"')).toBeLessThan(out.indexOf('"z"')) + expect(JSON.stringify(withSortedKeys(parseLossless('{"b":1e5,"a":1}')))).toBe('{"a":1,"b":1e5}') + }) + + it('throws where JSON.parse throws', () => { + expect(() => parseLossless('{"a":}')).toThrow() + }) +}) + +describe('walkers treat raw numbers as scalars', () => { + const v = parseLossless(SRC) + + it('tree', () => { + expect(kindOf((v as { z: unknown }).z)).toBe('number') + const { expanded } = allContainerPointers(v) + const rows = flattenTree(v, expanded) + expect(rows.find((r) => r.pointer === '/z')).toMatchObject({ + kind: 'number', + childCount: 0, + preview: '12345678901234567890', + }) + // 1 root + z + a + 5 items + s + expect(countNodes(v)).toBe(9) + }) + + it('embedded expansion', () => { + expect(findEmbeddedJson(v).map((e) => e.pointer)).toEqual(['/s']) + const out = expandEmbedded(v).value as { z: unknown } + expect(isRawNumber(out.z)).toBe(true) + }) +}) diff --git a/src/tools/json/core/rawjson.ts b/src/tools/json/core/rawjson.ts new file mode 100644 index 0000000..a6536df --- /dev/null +++ b/src/tools/json/core/rawjson.ts @@ -0,0 +1,62 @@ +/** + * Lossless numbers via the native JSON source-text-access API. + * + * `JSON.parse`'s reviver receives each primitive's original source text, and + * `JSON.rawJSON(text)` makes a value that `JSON.stringify` prints verbatim. So a + * number that a double cannot hold can be carried through format/minify/sort + * as its literal digits, with no parser dependency and nothing that needs + * `eval` (lossless-json would work too, but its LosslessNumber objects need + * the same special-casing in every walker, plus a dependency to vet). + * + * A raw value is a frozen, null-prototype object -- so every walker that + * recurses into objects must check `isRawNumber` first, or it will treat the + * number as a container with a single `rawJSON` key. + */ + +export interface RawNumber { + readonly rawJSON: string +} + +interface NativeRawJson { + rawJSON(text: string): RawNumber + isRawJSON(value: unknown): boolean +} + +interface ReviverContext { + source?: string +} + +type ContextReviver = (this: unknown, key: string, value: unknown, context?: ReviverContext) => unknown + +// TS's ES2023 lib predates the API; read it structurally rather than widen lib. +const native = JSON as unknown as Partial + +/** False on browsers without the API (Safari < 18.4); callers fall back. */ +export const losslessSupported = + typeof native.rawJSON === 'function' && typeof native.isRawJSON === 'function' + +export function isRawNumber(value: unknown): value is RawNumber { + return losslessSupported && value !== null && typeof value === 'object' && native.isRawJSON!(value) +} + +/** The literal a raw number was written as. */ +export function rawNumberText(value: RawNumber): string { + return value.rawJSON +} + +/** + * Parse strict JSON, keeping every number whose literal a double would print + * differently -- precision loss (`12345678901234567890`) and pure reformatting + * (`1.0`, `1e5`, `-9223372036854775808`) alike -- as its source text. + * + * Throws exactly when `JSON.parse` would. Must only be called when + * `losslessSupported`. + */ +export function parseLossless(text: string): unknown { + return JSON.parse(text, keepLiteral as (key: string, value: unknown) => unknown) +} + +const keepLiteral: ContextReviver = (_key, value, context) => { + if (typeof value !== 'number' || context?.source === undefined) return value + return String(value) === context.source ? value : native.rawJSON!(context.source) +} diff --git a/src/tools/json/core/tree.test.ts b/src/tools/json/core/tree.test.ts new file mode 100644 index 0000000..8bc42e3 --- /dev/null +++ b/src/tools/json/core/tree.test.ts @@ -0,0 +1,177 @@ +import fc from 'fast-check' +import { describe, expect, it } from 'vitest' +import { + PREVIEW_MAX, + allContainerPointers, + countNodes, + defaultExpanded, + flattenTree, + kindOf, + pathOf, + toggle, +} from './tree' + +const doc = { + a: 1, + 'x/y': { '~k': [true, null, 'z'] }, + list: [{ id: 1 }, { id: 2 }], +} + +describe('flattenTree', () => { + it('shows only the root when nothing is expanded', () => { + const rows = flattenTree(doc, new Set()) + expect(rows).toHaveLength(1) + expect(rows[0]).toMatchObject({ pointer: '', depth: 0, key: null, kind: 'object', childCount: 3 }) + expect(rows[0]!.expanded).toBe(false) + }) + + it('gives a scalar root a single row with its preview', () => { + expect(flattenTree(42, new Set(['']))).toEqual([ + expect.objectContaining({ kind: 'number', preview: '42', childCount: 0, expanded: false }), + ]) + expect(flattenTree('hi', new Set())[0]!.preview).toBe('"hi"') + expect(flattenTree(null, new Set())[0]!.kind).toBe('null') + }) + + it('escapes keys in pointers per RFC 6901', () => { + const { expanded } = allContainerPointers(doc) + const pointers = flattenTree(doc, expanded).map((r) => r.pointer) + expect(pointers).toContain('/x~1y') + expect(pointers).toContain('/x~1y/~0k/2') + }) + + it('emits rows in pre-order with correct depth and parent', () => { + const { expanded } = allContainerPointers(doc) + const rows = flattenTree(doc, expanded) + expect(rows.map((r) => r.pointer)).toEqual([ + '', + '/a', + '/x~1y', + '/x~1y/~0k', + '/x~1y/~0k/0', + '/x~1y/~0k/1', + '/x~1y/~0k/2', + '/list', + '/list/0', + '/list/0/id', + '/list/1', + '/list/1/id', + ]) + for (const [i, row] of rows.entries()) { + if (i === 0) expect(row.parent).toBe(-1) + else expect(rows[row.parent]!.depth).toBe(row.depth - 1) + } + expect(pathOf(rows, 4)).toEqual(['x/y', '~k', 0]) + expect(pathOf(rows, 0)).toEqual([]) + expect(rows[4]!.key).toBe(0) + }) + + it('never marks an empty container as expanded', () => { + const rows = flattenTree({ e: {}, f: [] }, new Set(['', '/e', '/f'])) + expect(rows.slice(1).map((r) => r.expanded)).toEqual([false, false]) + }) + + it('caps long string previews', () => { + const [row] = flattenTree('x'.repeat(PREVIEW_MAX * 3), new Set()) + expect(row!.preview.length).toBeLessThan(PREVIEW_MAX + 5) + expect(row!.preview.endsWith('…"')).toBe(true) + }) + + // Deep enough to overflow a recursive walk. Cost is quadratic in depth + // (each row's pointer is as long as the row is deep), hence the timeout. + it('survives nesting deeper than the call stack allows', { timeout: 20_000 }, () => { + let deep: unknown = 0 + for (let i = 0; i < 12_000; i++) deep = [deep] + const { expanded } = allContainerPointers(deep, Infinity) + expect(flattenTree(deep, expanded)).toHaveLength(12_001) + }) +}) + +const json = fc.jsonValue({ maxDepth: 4 }) + +describe('flattenTree properties', () => { + it('fully expanded, has one row per node with unique pointers', () => { + fc.assert( + fc.property(json, (value) => { + const { expanded } = allContainerPointers(value, Infinity) + const rows = flattenTree(value, expanded) + expect(rows).toHaveLength(countNodes(value)) + expect(new Set(rows.map((r) => r.pointer)).size).toBe(rows.length) + }), + ) + }) + + it('collapsing a node removes exactly its descendants', () => { + fc.assert( + fc.property(json, fc.nat(), (value, pick) => { + const { expanded } = allContainerPointers(value, Infinity) + const open = flattenTree(value, expanded) + const containers = open.filter((r) => r.expanded) + if (containers.length === 0) return + const target = containers[pick % containers.length]! + + const closed = flattenTree(value, toggle(expanded, target.pointer)) + const prefix = target.pointer + '/' + const expected = open.filter((r) => !r.pointer.startsWith(prefix)) + expect(closed.map((r) => r.pointer)).toEqual(expected.map((r) => r.pointer)) + }), + ) + }) +}) + +describe('defaultExpanded', () => { + it('opens the root and the first level of containers', () => { + const set = defaultExpanded(doc) + expect([...set].toSorted()).toEqual(['', '/list', '/x~1y']) + }) + + it('opens nothing for a scalar or empty root', () => { + expect(defaultExpanded(1).size).toBe(0) + expect(defaultExpanded({}).size).toBe(0) + }) + + it('stops before the visible rows exceed maxRows', () => { + const wide = Array.from({ length: 50 }, () => Array.from({ length: 30 }, (_, i) => i)) + const set = defaultExpanded(wide, { maxRows: 200 }) + const rows = flattenTree(wide, set) + expect(rows.length).toBeLessThanOrEqual(200) + expect(set.has('')).toBe(true) + }) + + it('always opens the root, even past maxRows', () => { + const big = Array.from({ length: 5000 }, (_, i) => i) + expect(defaultExpanded(big, { maxRows: 10 }).has('')).toBe(true) + }) +}) + +describe('allContainerPointers', () => { + it('reports when the row cap was hit', () => { + const big = Array.from({ length: 100 }, () => ({ a: 1, b: 2 })) + const capped = allContainerPointers(big, 150) + expect(capped.capped).toBe(true) + expect(flattenTree(big, capped.expanded).length).toBeLessThanOrEqual(150) + expect(allContainerPointers(big).capped).toBe(false) + }) +}) + +describe('scale', () => { + it('flattens 100k nodes, and a collapsed root costs one row', () => { + const big = Array.from({ length: 20_000 }, (_, i) => ({ id: i, name: `n${i}`, ok: true, v: null })) + expect(flattenTree(big, new Set())).toHaveLength(1) + const { expanded } = allContainerPointers(big) + expect(flattenTree(big, expanded)).toHaveLength(100_001) + }) +}) + +describe('kindOf', () => { + it('classifies every JSON type', () => { + expect([{}, [], '', 0, false, null].map(kindOf)).toEqual([ + 'object', + 'array', + 'string', + 'number', + 'boolean', + 'null', + ]) + }) +}) diff --git a/src/tools/json/core/tree.ts b/src/tools/json/core/tree.ts new file mode 100644 index 0000000..e03a1e3 --- /dev/null +++ b/src/tools/json/core/tree.ts @@ -0,0 +1,245 @@ +/** + * Flatten a parsed JSON value into the rows of a collapsible tree. + * + * The tree view virtualises over this list, so the list only ever contains the + * rows that are *reachable* -- a collapsed container contributes one row no + * matter how large it is. That keeps the cost proportional to what is on + * screen, not to the document: a 20 MB array collapsed at the root is 1 row. + * + * Rows are identified by their RFC 6901 pointer, which is stable across a + * re-parse as long as the keys are, so expansion state survives an edit. + */ + +import { escapePointerToken } from './embedded' +import { isRawNumber, rawNumberText } from './rawjson' + +export type NodeKind = 'object' | 'array' | 'string' | 'number' | 'boolean' | 'null' + +export interface FlatNode { + /** RFC 6901 pointer; '' for the root. */ + pointer: string + depth: number + /** Object key or array index; null for the root. */ + key: string | number | null + kind: NodeKind + /** Number of direct children; 0 for scalars. */ + childCount: number + expanded: boolean + /** Scalars rendered as JSON, capped at PREVIEW_MAX; '' for containers. */ + preview: string + /** Row index of the parent; -1 for the root. */ + parent: number +} + +/** + * A multi-MB string would otherwise be JSON.stringify'd for its row every time + * the list is rebuilt, and only ~100 characters of it fit on screen anyway. + */ +export const PREVIEW_MAX = 1000 + +export function kindOf(value: unknown): NodeKind { + if (value === null) return 'null' + if (Array.isArray(value)) return 'array' + if (isRawNumber(value)) return 'number' + switch (typeof value) { + case 'object': + return 'object' + case 'string': + return 'string' + case 'number': + return 'number' + case 'boolean': + return 'boolean' + default: + // JSON.parse never produces anything else; treat stray values as null + // rather than crashing the view. + return 'null' + } +} + +function childCount(value: unknown, kind: NodeKind): number { + if (kind === 'array') return (value as unknown[]).length + if (kind === 'object') return Object.keys(value as object).length + return 0 +} + +function preview(value: unknown, kind: NodeKind): string { + if (kind === 'object' || kind === 'array') return '' + if (kind === 'string') { + const s = value as string + if (s.length <= PREVIEW_MAX) return JSON.stringify(s) + // Drop the closing quote, mark the cut, and close it again. + return `${JSON.stringify(s.slice(0, PREVIEW_MAX)).slice(0, -1)}…"` + } + if (isRawNumber(value)) return rawNumberText(value) + return JSON.stringify(value) ?? 'null' +} + +/** Children in document order: indices for arrays, insertion order for objects. */ +function children(value: unknown, kind: NodeKind): [string | number, unknown][] { + if (kind === 'array') return (value as unknown[]).map((v, i) => [i, v]) + if (kind === 'object') return Object.entries(value as Record) + return [] +} + +function childPointer(parent: string, key: string | number): string { + return `${parent}/${escapePointerToken(String(key))}` +} + +interface Frame { + value: unknown + pointer: string + depth: number + key: string | number | null + parent: number +} + +/** + * Pre-order rows for every node reachable through `expanded` containers. + * + * Iterative, so a deeply nested document cannot overflow the call stack. + */ +export function flattenTree(root: unknown, expanded: ReadonlySet): FlatNode[] { + const rows: FlatNode[] = [] + const stack: Frame[] = [{ value: root, pointer: '', depth: 0, key: null, parent: -1 }] + + while (stack.length > 0) { + const frame = stack.pop()! + const kind = kindOf(frame.value) + const count = childCount(frame.value, kind) + const isOpen = count > 0 && expanded.has(frame.pointer) + const index = rows.length + + rows.push({ + pointer: frame.pointer, + depth: frame.depth, + key: frame.key, + kind, + childCount: count, + expanded: isOpen, + preview: preview(frame.value, kind), + parent: frame.parent, + }) + + if (!isOpen) continue + const entries = children(frame.value, kind) + // Reverse so the first child is popped first. + for (let i = entries.length - 1; i >= 0; i--) { + const [key, value] = entries[i]! + stack.push({ + value, + pointer: childPointer(frame.pointer, key), + depth: frame.depth + 1, + key, + parent: index, + }) + } + } + + return rows +} + +/** + * The key path of a row, rebuilt from its parent links. Rows do not carry one: + * copying a path per row costs O(depth) each, and only the selected row ever + * needs it. + */ +export function pathOf(rows: readonly FlatNode[], index: number): (string | number)[] { + const path: (string | number)[] = [] + for (let i = index; i > 0; i = rows[i]!.parent) path.push(rows[i]!.key!) + return path.toReversed() +} + +export interface ExpandOptions { + /** Containers shallower than this are expanded. 2 shows root + 2 levels. */ + maxDepth?: number + /** Stop expanding once the visible row count would exceed this. */ + maxRows?: number +} + +/** + * The expansion a freshly pasted document opens with: breadth-first, so a wide + * document shows its top level rather than drilling into its first element, + * and capped, so a 100k-element array does not open fully expanded. + * + * The root is always expanded -- a single collapsed row is never useful. + */ +export function defaultExpanded(root: unknown, options: ExpandOptions = {}): Set { + const { maxDepth = 2, maxRows = 1000 } = options + const out = new Set() + const rootKind = kindOf(root) + if (childCount(root, rootKind) === 0) return out + + out.add('') + let rows = 1 + childCount(root, rootKind) + let level: { value: unknown; pointer: string }[] = [{ value: root, pointer: '' }] + + for (let depth = 1; depth < maxDepth; depth++) { + const next: { value: unknown; pointer: string }[] = [] + for (const node of level) { + for (const [key, value] of children(node.value, kindOf(node.value))) { + const count = childCount(value, kindOf(value)) + if (count === 0) continue + if (rows + count > maxRows) return out + const pointer = childPointer(node.pointer, key) + out.add(pointer) + rows += count + next.push({ value, pointer }) + } + } + level = next + } + return out +} + +/** + * Every container pointer, for "Expand all" -- capped by the number of rows the + * result would produce, because expanding a million-node document fully would + * allocate a million rows. `capped` says whether the limit was hit, so the UI + * can say so instead of silently expanding only part of it. + */ +export function allContainerPointers( + root: unknown, + maxRows = 200_000, +): { expanded: Set; capped: boolean } { + const expanded = new Set() + let rows = 1 + const stack: { value: unknown; pointer: string }[] = [{ value: root, pointer: '' }] + + while (stack.length > 0) { + const { value, pointer } = stack.pop()! + const kind = kindOf(value) + const count = childCount(value, kind) + if (count === 0) continue + if (rows + count > maxRows) return { expanded, capped: true } + expanded.add(pointer) + rows += count + const entries = children(value, kind) + for (let i = entries.length - 1; i >= 0; i--) { + const [key, child] = entries[i]! + stack.push({ value: child, pointer: childPointer(pointer, key) }) + } + } + return { expanded, capped: false } +} + +export function toggle(expanded: ReadonlySet, pointer: string): Set { + const next = new Set(expanded) + if (next.has(pointer)) next.delete(pointer) + else next.add(pointer) + return next +} + +/** Total node count of a value, containers and scalars alike. */ +export function countNodes(root: unknown): number { + let n = 0 + const stack: unknown[] = [root] + while (stack.length > 0) { + const value = stack.pop() + n++ + const kind = kindOf(value) + if (kind === 'array') for (const v of value as unknown[]) stack.push(v) + else if (kind === 'object') for (const v of Object.values(value as object)) stack.push(v) + } + return n +}