diff --git a/.claude/settings.json b/.claude/settings.json index 55f0268bf..4541a7205 100644 --- a/.claude/settings.json +++ b/.claude/settings.json @@ -41,7 +41,17 @@ "hooks": [ { "type": "command", - "command": "in=$(cat); body=$(printf '%s' \"$in\" | sed 's/\"scratchpad_dir\":\"[^\"]*\"//g; s/\"transcript_path\":\"[^\"]*\"//g; s/\"cwd\":\"[^\"]*\"//g; s/\"description\":\"[^\"]*\"//g'); if printf '%s' \"$body\" | grep -qE '(^|[;&|(\")])[[:space:]]*(python[0-9.]*|pythonw|py)([[:space:]]|\")|sed[[:space:]]+-[a-zA-Z]*i|sed[[:space:]]+--in-place|>[[:space:]]*[^[:space:];&|]*\\.(cs|md|csproj|props|targets|sln|json|ps1|g4)|tee[[:space:]][^;&|]*\\.(cs|md|csproj|props|targets|sln|json|ps1|g4)'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Shell edits to source files (and Python) are disallowed: they bypass file checkpointing, so /rewind cannot undo them. Use Edit or Write instead. Writing logs to the scratchpad is fine.\"}}'; elif printf '%s' \"$body\" | grep -qE '(^|[;&(\")]|\\\\\\\\n)[[:space:]]*(grep|egrep|fgrep|rg|cat|head|tail|ls|dir|find|fd|fdfind|Get-Content|gc|Get-ChildItem|gci|Select-String|sls)([[:space:]]|\")'; then if ! printf '%s' \"$body\" | grep -qE '(scratchpad|\\.log|\\$log|/tmp/)'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Reading files through the shell costs a permission prompt and bypasses the dedicated tools. Use Grep for content, Glob for filenames, Read for file contents. Shell text tools stay allowed on scratchpad logs (paths containing scratchpad, .log or $log).\"}}'; elif ! printf '%s' \"$body\" | grep -qE '([;&`]|[$]\\()'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"allow\",\"permissionDecisionReason\":\"Pre-approved: a shell text tool reading a scratchpad/log path, as a single command with no chaining.\"}}'; fi; fi; true" + "command": "in=$(cat); body=$(printf '%s' \"$in\" | sed 's/\"scratchpad_dir\":\"[^\"]*\"//g; s/\"transcript_path\":\"[^\"]*\"//g; s/\"cwd\":\"[^\"]*\"//g; s/\"description\":\"[^\"]*\"//g'); if printf '%s' \"$body\" | grep -qE '(^|[;&|(\")])[[:space:]]*(python[0-9.]*|pythonw|py)([[:space:]]|\")|sed[[:space:]]+-[a-zA-Z]*i|sed[[:space:]]+--in-place|>[[:space:]]*[^[:space:];&|]*\\.(cs|md|csproj|props|targets|sln|json|ps1|g4)|tee[[:space:]][^;&|]*\\.(cs|md|csproj|props|targets|sln|json|ps1|g4)'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Shell edits to source files (and Python) are disallowed: they bypass file checkpointing, so /rewind cannot undo them. Use Edit or Write instead. Writing logs to the scratchpad is fine.\"}}'; elif printf '%s' \"$body\" | grep -qE '(^|[;&|(\")])[[:space:]]*git[[:space:]]+(-[^[:space:]]+[[:space:]]+([^-[:space:]][^[:space:]]*[[:space:]]+)?){0,3}(grep|cat-file)([[:space:]]|\"|$)'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"git grep and git cat-file read repository files through the shell - the same thing the read rule covers, just spelled to pass the git allow-list. Use Grep for content, Glob for filenames, Read for file contents. git diff, git log and git show stay allowed for reviewing history.\"}}'; elif printf '%s' \"$body\" | grep -qE '(^|[;&(\")]|\\\\\\\\n)[[:space:]]*(grep|egrep|fgrep|rg|cat|head|tail|ls|dir|find|fd|fdfind|Get-Content|gc|Get-ChildItem|gci|Select-String|sls)([[:space:]]|\")'; then if ! printf '%s' \"$body\" | grep -qE '(scratchpad|\\.log|\\$log|/tmp/)'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"Reading files through the shell costs a permission prompt and bypasses the dedicated tools. Use Grep for content, Glob for filenames, Read for file contents. Shell text tools stay allowed on scratchpad logs (paths containing scratchpad, .log or $log).\"}}'; elif ! printf '%s' \"$body\" | grep -qE '([;&`]|[$]\\()'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"allow\",\"permissionDecisionReason\":\"Pre-approved: a shell text tool reading a scratchpad/log path, as a single command with no chaining.\"}}'; fi; elif printf '%s' \"$body\" | grep -qE 'dotnet[[:space:]]+test' && printf '%s' \"$body\" | grep -qE 'LibRed\\.(Core|Engine)\\.AccessTests'; then if ! printf '%s' \"$body\" | grep -qE '(--filter|\"run_in_background\\\"?[[:space:]]*:[[:space:]]*true)'; then echo '{\"hookSpecificOutput\":{\"hookEventName\":\"PreToolUse\",\"permissionDecision\":\"deny\",\"permissionDecisionReason\":\"LibRed.Core.AccessTests takes about 7 minutes and LibRed.Engine.AccessTests about 1.5. Run it with --filter for the tests that cover the change, or pass run_in_background true and keep working while it runs. A whole suite in the foreground blocks the session for minutes at a time.\"}}'; fi; fi; true" + } + ] + } + ], + "Stop": [ + { + "hooks": [ + { + "type": "command", + "command": "in=$(cat); printf '%s' \"$in\" | grep -qE '\"stop_hook_active\"[[:space:]]*:[[:space:]]*true' && exit 0; cd \"$CLAUDE_PROJECT_DIR\" || exit 0; [ -z \"$(git status --porcelain -- src ':(glob,exclude)src/**/docs/**')\" ] && exit 0; echo '{\"decision\":\"block\",\"reason\":\"src/ has changes. Before finishing, go through your diff: for every method, overload, wrapper, helper or constant you added, name the existing code that already does that job. If one exists, change or split that code instead and remove yours. See Change the existing path in CLAUDE.md.\"}'" } ] } diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index 9be132d57..6ed0b0e93 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -4,4 +4,4 @@ - This project (EntityFrameworkCore.Jet) targets the Microsoft Access Jet/ACE database engine, not SQL Server. Generated SQL, type mappings, and literals must be Jet/ACE-compliant (e.g. decimal/currency instead of bigint, #...# date literals, TIMEVALUE()). Note: Access 2016+ (ACE) does have a native BIGINT (Large Number) type, which is a SIGNED 64-bit integer and cannot hold unsigned values like ulong.MaxValue; unsigned ulong/uint should map to decimal(20,0) to avoid overflow. Provider distinctions are OLE DB / ODBC via the ACE/Jet driver rather than SqlClient. Do not assume SQL Server/T-SQL semantics. - The Jet/Access SQL dialect does not support COALESCE or NZ. When rewriting queries for Jet, prefer IIF( IS NULL, , ) instead. - That rule is about the **Jet dialect**, not the whole repository. This repo also contains LibRed (`src/LibRed`), a from-scratch managed Jet/ACE engine, and its EF Core provider has two SQL modes. In `LibRedSqlMode.Compatible` the rule above applies, because the SQL is generated to also run against ACE. In `LibRedSqlMode.Extended` (the default) it does not: LibRed owns the engine that parses the SQL, so the generator deliberately emits standard SQL that ACE has no syntax for — COALESCE, CASE, NULLIF, CROSS/OUTER APPLY, window functions, OFFSET/FETCH paging and FULL OUTER JOIN. Do not "fix" those back into Jet workarounds; extended mode exists precisely to remove them. -- Deferred enhancement for EntityFrameworkCore.Jet: add ACE engine-version detection so that on Access 2016+ the provider can map to native BIGINT (signed long; unsigned ulong/uint still go to decimal(20,0)) and native DATETIME2, while falling back to decimal(20,0)/legacy datetime on older ACE versions. Not a priority right now. The gap is in the **provider's type mapping**, not in the engine: ACE itself accepts both types, and LibRed already reads, writes and indexes them, gating on the on-disk format version and raising a file's version byte rather than refusing the DDL. Caveat before treating this as a straightforward port: DATETIME2 through Jet/ACE is badly behaved, so it needs its own investigation rather than being mapped by analogy with BIGINT. The asymmetry between the two providers is deliberate, so do not assume either side behaves like the other. \ No newline at end of file +- Deferred enhancement for EntityFrameworkCore.Jet: add ACE engine-version detection so that the provider can map to native BIGINT on Access 2016+ (signed long; unsigned ulong/uint still go to decimal(20,0)) and native DATETIME2 on Access 2019+ — two different thresholds, not one — while falling back to decimal(20,0)/legacy datetime on older ACE versions. Not a priority right now. The gap is in the **provider's type mapping**, not in the engine: ACE itself accepts both types, and LibRed already reads, writes and indexes them, gating on the on-disk format version and raising a file's version byte rather than refusing the DDL. Caveat before treating this as a straightforward port: DATETIME2 through Jet/ACE is badly behaved, so it needs its own investigation rather than being mapped by analogy with BIGINT. The asymmetry between the two providers is deliberate, so do not assume either side behaves like the other. \ No newline at end of file diff --git a/.github/workflows/pull_request.yml b/.github/workflows/pull_request.yml index 5f70f03b4..f07d3d712 100644 --- a/.github/workflows/pull_request.yml +++ b/.github/workflows/pull_request.yml @@ -731,21 +731,35 @@ jobs: - name: 'Run Tests: LibRed.Core.AccessTests' if: env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.Core.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.Core.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m # The engine tests that cross-check against ACE. They belong here rather than in the cross-platform # LibRed job above, which runs on five platforms precisely to prove LibRed needs no ACE at all. - name: 'Run Tests: LibRed.Engine.AccessTests' if: always() && env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.Engine.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.Engine.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m - name: 'Run Tests: LibRed.Ado.Tests' if: always() && env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.Ado.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.Ado.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m - name: 'Run Tests: LibRed.EFCore.Tests' if: always() && env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.EFCore.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.EFCore.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m + # A run that dies partway reports only the tests that finished. The blame output names the test that + # was running (Sequence_*.xml) and holds any crash or hang dump, so keep it when a step fails. + - name: 'Upload Blame Results' + if: failure() + uses: actions/upload-artifact@v6 + with: + name: libredaccess-blame_ace_${{ matrix.aceVersion }} + path: | + test\LibRed.Core.AccessTests\TestResults\** + test\LibRed.Engine.AccessTests\TestResults\** + test\LibRed.Ado.Tests\TestResults\** + test\LibRed.EFCore.Tests\TestResults\** + if-no-files-found: ignore + retention-days: 7 LibRedFunctional: needs: diff --git a/.github/workflows/push.yml b/.github/workflows/push.yml index 8296ec1da..7ecc936b6 100644 --- a/.github/workflows/push.yml +++ b/.github/workflows/push.yml @@ -118,7 +118,7 @@ jobs: - 'test/EFCore.LibRed.FunctionalTests/**' - 'test/EFCore.LibRed.Extended.FunctionalTests/**' - 'src/EFCore.Jet.Common/**' - # Northwind.accdb lives here and every LibRed suite links to it as its fixture. + # Northwind.accdb lives here and every LibRed.* test project links to it as its fixture. - 'test/JetProviderExceptionTests/**' - name: 'Decide What To Run' id: Decide @@ -739,21 +739,35 @@ jobs: - name: 'Run Tests: LibRed.Core.AccessTests' if: env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.Core.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.Core.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m # The engine tests that cross-check against ACE. They belong here rather than in the cross-platform # LibRed job above, which runs on five platforms precisely to prove LibRed needs no ACE at all. - name: 'Run Tests: LibRed.Engine.AccessTests' if: always() && env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.Engine.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.Engine.AccessTests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m - name: 'Run Tests: LibRed.Ado.Tests' if: always() && env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.Ado.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.Ado.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m - name: 'Run Tests: LibRed.EFCore.Tests' if: always() && env.skipTests != 'true' shell: pwsh - run: dotnet test .\test\LibRed.EFCore.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-hang-timeout 5m + run: dotnet test .\test\LibRed.EFCore.Tests --configuration '${{ env.buildConfiguration }}' -p:FixedTestOrder=${{ env.deterministicTests }} --blame-crash --blame-hang-timeout 5m + # A run that dies partway reports only the tests that finished. The blame output names the test that + # was running (Sequence_*.xml) and holds any crash or hang dump, so keep it when a step fails. + - name: 'Upload Blame Results' + if: failure() + uses: actions/upload-artifact@v6 + with: + name: libredaccess-blame_ace_${{ matrix.aceVersion }} + path: | + test\LibRed.Core.AccessTests\TestResults\** + test\LibRed.Engine.AccessTests\TestResults\** + test\LibRed.Ado.Tests\TestResults\** + test\LibRed.EFCore.Tests\TestResults\** + if-no-files-found: ignore + retention-days: 7 LibRedFunctional: needs: @@ -888,12 +902,8 @@ jobs: $buildSha = '${{ github.sha }}'.SubString(0, 7); $pack = $officialBuild -or $ciBuildOnly - $pushToAzureArtifacts = $pack - $pushToMygetOrg = $pack $pushToNugetOrg = $pack -and $officialBuild - echo "pushToAzureArtifacts: $pushToAzureArtifacts" - echo "pushToMygetOrg: $pushToMygetOrg" echo "pushToNugetOrg: $pushToNugetOrg" echo "officialBuild: $officialBuild" @@ -944,24 +954,12 @@ jobs: } } - echo "pushToAzureArtifacts=$pushToAzureArtifacts" >> $env:GITHUB_ENV - echo "pushToMygetOrg=$pushToMygetOrg" >> $env:GITHUB_ENV echo "pushToNugetOrg=$pushToNugetOrg" >> $env:GITHUB_ENV - name: Upload Artifacts uses: actions/upload-artifact@v6 with: name: nupkgs path: nupkgs - - name: "NuGet Push - myget.org - Debug" - if: ${{ env.pushToMygetOrg == 'true' }} - working-directory: nupkgs - shell: pwsh - run: dotnet nuget push './Debug/withPdbs/**/*.nupkg' --api-key '${{ secrets.MYGETORG_CIRRUSRED_ALLPACKAGES_DEBUG_PUSHNEW }}' --source 'https://www.myget.org/F/cirrusred-debug/api/v3/index.json' --skip-duplicate - - name: "NuGet Push - myget.org - Release" - if: ${{ env.pushToMygetOrg == 'true' }} - working-directory: nupkgs - shell: pwsh - run: dotnet nuget push './Release/default/**/*.nupkg' --api-key '${{ secrets.MYGETORG_CIRRUSRED_ALLPACKAGES_PUSHNEW }}' --source 'https://www.myget.org/F/cirrusred/api/v3/index.json' --skip-duplicate - name: "NuGet Push - nuget.org - Release" if: ${{ env.pushToNugetOrg == 'true' }} working-directory: nupkgs diff --git a/AGENTS.md b/AGENTS.md index c78072b44..388a48190 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -7,7 +7,13 @@ It is kept in sync with `CLAUDE.md`; if you change guidance in one, check whethe EntityFrameworkCore.Jet is an EF Core provider for Microsoft Jet/ACE databases (Microsoft Access `.mdb`/`.accdb` files). The **Jet** provider runs **Windows only** and bridges EF Core to the Access database engine via either ODBC or OLE DB. Alongside it, **LibRed** (also in this repo, on `master`) is a from-scratch managed engine that reads/writes the file format directly and is **cross-platform** — see the LibRed section below. -Current version: `11.0.0-alpha.3` (`Version.props`) targeting EF Core 11 and `net11.0`; `global.json` pins the 11.0.100 RC1 SDK with `rollForward: latestFeature`. The test projects use **xunit v3**. +Current version: `11.0.0-alpha.4` (`Version.props`) targeting EF Core 11 and `net11.0`; `global.json` pins the 11.0.100 RC1 SDK with `rollForward: latestFeature`. The test projects use **xunit v3**, except `EFCore.Jet.Data.Tests` and `EFCore.Jet.IntegrationTests`, which use MSTest. + +`LibRed.Core`, `LibRed.Sql`, `LibRed.Engine` and `LibRed.Ado` are the exception: they multi-target +`$(LibRedTargetFrameworks)` = `net10.0;net11.0`, since none of them depends on an EF Core package. `LibRed.EFCore` +does, so it stays `net11.0`. Every test project is single-target `net11.0` (the Jet ones as `net11.0-windows7.0`, +via `$(JetTestWindowsOnlyTargetFramework)`) deliberately: the `net10.0` leg is compiled but never run, so tests stay +one run per suite and CI needs no second runtime. ### Which layer am I touching? @@ -46,14 +52,23 @@ To develop against a local EF Core build instead of NuGet packages, copy `Develo ## Tests -**Jet** tests require a real Microsoft Access driver installed (ODBC or OLE DB) and an actual `.accdb` file — no mocks. The connection string is configured via: +**Jet** tests require a real Microsoft Access driver installed (ODBC or OLE DB) and an actual `.accdb` file — no mocks. The EF suites' connection strings (the LibRed ones included, which need no driver) are configured via: - `test/EFCore.Jet.FunctionalTests/config.json` (OLE DB example present) -- `test/EFCore.Jet.Tests/config.json` (bare filename; picks up default provider) +- `test/EFCore.Jet.Tests/config.json` (ODBC `DBQ=` form) - `test/EFCore.LibRed.FunctionalTests/config.json` and `test/EFCore.LibRed.Extended.FunctionalTests/config.json` (LibRed connection, one per SQL mode) -- Or env var `EFCoreJet_DefaultConnection` +- Or env var `EFCoreJet_DefaultConnection` (the Jet suite) / `EFCoreLibRed_DefaultConnection` (both LibRed + suites) + +**LibRed** tests split in two: `LibRed.Core.Tests`, `LibRed.Engine.Tests`, `EFCore.LibRed.FunctionalTests` and `EFCore.LibRed.Extended.FunctionalTests` need **no driver at all** and CI runs them on Linux/Windows/macOS plus ARM64 legs — that matrix is what proves the cross-platform claim, so don't add an ACE dependency to them. `LibRed.Core.AccessTests` and `LibRed.Engine.AccessTests` deliberately cross-check LibRed's output against the real engine over OLE DB, so they need Windows + ACE. `LibRed.Ado.Tests` and `LibRed.EFCore.Tests` use no driver (no OLE DB, no DAO), but CI only runs them in the Windows ACE job for now. -**LibRed** tests split in two: `LibRed.Engine.Tests`, `EFCore.LibRed.FunctionalTests` and `EFCore.LibRed.Extended.FunctionalTests` need **no driver at all** and CI runs them on Linux/Windows/macOS plus ARM64 legs — that matrix is what proves the cross-platform claim, so don't add an ACE dependency to them. `LibRed.Core.Tests`, `LibRed.Engine.AccessTests`, `LibRed.Ado.Tests` and `LibRed.EFCore.Tests` deliberately cross-check LibRed's output against the real engine over OLE DB, so they need Windows + ACE. +> **The `*.AccessTests` split is by which engine a test needs, not by subject.** A file-format test belongs in +> `LibRed.Core.Tests` if LibRed alone can decide the answer, and in `LibRed.Core.AccessTests` if ACE has to be +> asked — the two halves share a namespace and their helpers (`test/LibRed.Shared/`), so moving a test between +> them is a file move and nothing else. The ACE half runs serially; the plain half runs **in parallel** and +> takes seconds, which is why adding an ACE dependency to it costs more than it looks. Note the dependency is +> not always visible in a name or a `using`: three DAO probes reach ACE through `Type.GetTypeFromProgID`, and +> were only caught because the plain project does not suppress `CA1416`. Leave that suppression off. **Run all tests** (requires x86 or x64 matching your driver bitness): @@ -84,7 +99,8 @@ is the most common way to waste minutes here. **When you do run a suite, capture the failing test *names* in the same run** — don't reduce the output to just the `Passed!/Failed!` count line and then re-run the whole suite to find which failed. Grep a pattern that catches both, e.g. `grep -iE "Passed!|Failed!|\[FAIL\]|error CS"` (xUnit prints `… [FAIL]` and `Failed ` lines as it goes), or tee the full output to a file and inspect it. -Tests run in **fixed order by default** (`FIXED_TEST_ORDER` compile constant, set unless `-p:FixedTestOrder=false`; see `test/Directory.Build.props`). All tests lock culture to `en-US` via a module initializer (`test/Shared/ModuleInitializer.cs`). +The three EF functional suites run in **fixed order by default** (`FIXED_TEST_ORDER` compile constant, set unless `-p:FixedTestOrder=false`; see `test/Directory.Build.props` — every project gets the constant, but only those suites and the empty +`EFCore.Jet.Tests` act on it) and lock culture to `en-US` via a module initializer (`test/Shared/ModuleInitializer.cs`, compiled into those suites only). The other test projects set culture themselves where it matters, and `LibRed.Core.Tests` runs its collections in parallel. Tests that require features Jet doesn't support are skipped with a reason on the test. @@ -94,13 +110,18 @@ Tests that require features Jet doesn't support are skipped with a reason on the `EFCore.Jet.FunctionalTests` is gated by a committed pass-list rather than by "everything must pass": `test/EFCore.Jet.FunctionalTests/GreenTests/ace___.txt` lists the tests that passed -previously for that matrix leg. CI merges the shards' `.trx` files and **fails if any listed test stops passing**; -newly-passing tests are appended and pushed back by the `auto_commit` workflow. So the meaningful question for a -change is "did anything that used to pass stop passing", not the raw failure count. - -CI splits the functional suite into three shards (query core / Northwind+GearsOfWar / non-query) and retries a shard -up to three times if the runner crashes. The two `EFCore.LibRed*.FunctionalTests` suites are `continue-on-error` -for now — they still run on every push, but their remaining failures don't block. +previously for that matrix leg. Only two legs have one today — `ace_2010_odbc_x86` and `ace_2010_oledb_x86` — and +the check runs only where the file exists, so the other six legs are not gated. CI merges the shards' `.trx` +files and **fails if any listed test stops passing**; after a successful pull-request run, newly-passing tests are +appended to an existing list and pushed back by the `auto_commit` workflow (it is triggered by +`pull_request.yml`, not by `push.yml`, and never creates a list). So the meaningful question for a change is "did +anything that used to pass stop passing", not the raw failure count. + +CI splits the functional suite into four shards (query core / Northwind+GearsOfWar / non-query without +CompiledModel / CompiledModel) and runs a shard up to three times if the runner crashes (on a leg with a +pass-list, a shard that crashes all three times fails it). The two `EFCore.LibRed*.FunctionalTests` suites are +`continue-on-error` for now — they run on every push that touches LibRed, but their remaining failures don't +block. ### Docker images @@ -133,20 +154,30 @@ test/ EFCore.Jet.Tests/ EMPTY — its test files are d [Windows + ACE] EFCore.Jet.IntegrationTests/ Integration scenario tests [Windows + ACE] JetProviderExceptionTests/ Exception-path tests; also hosts Northwind.accdb, - which every LibRed suite links to as its fixture [Windows + ACE] - LibRed.Core.Tests/ File-format read/write, cross-checked against ACE [Windows + ACE] + which every LibRed.* test project and LibRed.Benchmarks + link to as their fixture (the EF functional suites + build theirs from test/Northwind.sql) [Windows + ACE] + LibRed.Core.Tests/ File-format read/write through LibRed alone [cross-platform] + LibRed.Core.AccessTests/ File-format tests cross-checked against ACE [Windows + ACE] LibRed.Engine.Tests/ Planner/executor, no engine dependency [cross-platform] LibRed.Engine.AccessTests/ Engine tests that cross-check against ACE [Windows + ACE] - LibRed.Ado.Tests/ ADO.NET surface [Windows + ACE] + LibRed.Ado.Tests/ ADO.NET surface [no driver; CI's Windows ACE job] LibRed.EFCore.Tests/ LibRed EF Core provider: query round-trip, - database-first scaffolding [Windows + ACE] + database-first scaffolding [no driver; CI's Windows ACE job] + LibRed.Shared/ Helpers compiled into the Core/Engine test pairs and LibRed.Ado.Tests + (TemporaryDatabase, …; AceTestDatabase into the AccessTests only) EFCore.LibRed.FunctionalTests/ EF Core specification suite over LibRed, compatible SQL mode (Jet-dialect SQL) [cross-platform] EFCore.LibRed.Extended.FunctionalTests/ The same suite in extended SQL mode; its own baselines, because the SQL differs [cross-platform] - LibRed.Benchmarks/ BenchmarkDotNet harness (not a test project) - Shared/ ModuleInitializer.cs — locks culture to en-US + LibRed.Benchmarks/ BenchmarkDotNet harness over the engine directly — no EF, no + ADO. Suites for end-to-end queries, the parse/plan/execute + split, storage, writes, DDL, catalog, and an opt-in ACE + head-to-head. `-- --validate` checks the SQL corpus. + See its README.md [cross-platform] + Shared/ For the EF functional suites: ModuleInitializer.cs (locks culture to + en-US) and TestUtilities/ (test orderers, crash detection, conditions) tools/ sortkey-table/ Generates the Windows NLS sort-weight table LibRed's index keys use @@ -240,8 +271,8 @@ in `EFCore.Jet.Data`. ``` src/LibRed/ LibRed.Core/ File format: IO (PageChannel/PageBuffer), Formats (version offsets), - Pages, Catalog (MSysObjects → TableDef/ColumnDef/IndexDef), Storage - (Table/TableCursor/RowDecoder/UsageMap), Crypto; JetDatabase entry point + Pages, Catalog (MSysObjects → TableDefinition/ColumnDef/IndexDef), Storage + (Table/TableCursor/RowCodec/UsageMap), Crypto; JetDatabase entry point LibRed.Sql/ SQL front end: ANTLR grammar (AccessSql.g4), AST, parser, binder. NO Jet dependency — binds via the ISchemaProvider abstraction LibRed.Engine/ Logical Plan nodes, QueryPlanner, CatalogSchemaProvider (bridges the @@ -265,7 +296,10 @@ the names are `JetVersion` / `RequiredVersion` / `EnsureFormatAtLeast`: `DATETIME2` needs `Version17_2019` (0x06). **Different thresholds** — the natural assumption that ACE 16 added both at once is wrong, and it's measured in `docs/format/page-00-database.md`. - `AccessTypeMapper.MapType` **refuses** a type the open file is too old for, so a caller that can't upgrade - (read-only database) fails loudly instead of writing a column Access couldn't read. + (read-only database) fails loudly instead of writing a column Access couldn't read. That guard is on the SQL + path; the same rule is enforced again in Core over `JetDataType` (`JetDataTypeVersions.EnsureStorable`, called + from `TableDefinition` and `SchemaEditor`), because every `JetDatabase` method that defines a column takes a raw `ColumnSpec` and + would otherwise write the descriptor with nothing objecting. - `StatementExecutor.MapColumn` → `JetDatabase.EnsureFormatAtLeast` → `PageChannel.RaiseFormatVersion` **raises the file's version byte** rather than refusing the DDL, which is what ACE itself does. The raise goes through `WritePage`, so it joins the statement's transaction and a failed `CREATE`/`ALTER` takes it back down; @@ -292,7 +326,9 @@ Jet 4 / ACE format, **split one file per page type** (plus cross-cutting topics) the `appendix-structures.md` bare field-layout reference. It is the source of truth — keep it updated as the format understanding grows. (`docs/jet-ace-file-format.md` is now just a redirect stub to that folder.) Alongside it: `docs/functions.md` catalogs the supported VBA/Access function surface, -`docs/design/transactions.md` covers the page-level undo log, and `docs/mdbtools-spec-diff-todo.md` +`docs/design/transactions.md` covers the deferred-write page overlay, +`docs/design/index-key-checksum.md` records how the index-key truncation checksum was pinned down (and why the +rule it replaced looked verified for years), and `docs/mdbtools-spec-diff-todo.md` tracks where our spec and mdbtools' still disagree. > **Rule — spec sync on every `LibRed.Core` change.** Whenever you touch the actual on-disk @@ -312,9 +348,12 @@ parser: the lexer/parser are pre-generated and committed under `LibRed.Sql/Gramm `LibRed.Sql/Grammar/generate.ps1` after editing `AccessSql.g4`. > **LibRed is a single-writer engine.** It tolerates extra open handles (a `.accdb` is a shared-file database and -> EF's own test infra keeps a store connection open), but there is no lock file, no read isolation, and the -> transaction undo log is per-`PageChannel`. Two concurrent writers corrupt the file. See `src/LibRed/README.md` -> for the full statement of what is and isn't safe before designing anything concurrent on top of it. +> EF's own test infra keeps a store connection open), but there is no lock file and no cross-process page/record locking (its page locks are process-local), so +> two concurrent writers corrupt the file. Transactions use a **deferred-write overlay** per `PageChannel`, not +> an undo log: writes buffer until commit and publish under a lock, so a reader cannot see another handle's +> uncommitted pages and a rollback cannot discard another channel's committed work, and a stale writer fails +> commit with a write-conflict error. See `src/LibRed/README.md` for the full statement of what is and isn't safe +> before designing anything concurrent on top of it. ## Working in This Repo @@ -326,14 +365,27 @@ here, because it sends the work off in a direction that has to be unwound later. verify against the code rather than quoting the list; and a **memory or summary records what was true when it was written**, not what is true now. +**Change the existing path; don't add one beside it.** A fix reshapes the code that already does the job: split +a method at the point where its callers differ, change what it takes, move a check to where every caller passes +through. It does not add an overload, wrapper, helper, constant or parallel method next to it. Before writing +any new member, find the code that already does that job — by what it does, not by its name: a decode, a lookup, +a layout, a check, a value — and use or reshape it. A second copy is how this codebase got its bugs: two readers +of one field that disagreed, a merged path that lost a check only one copy had, hard-coded page numbers restating +a pointer, test helpers re-implementing Core and drifting from it. A genuinely new member is fine when nothing +existing does the job; say what you checked. (Claude Code backs this with a `Stop` hook in +`.claude/settings.json` that asks for the review while `src/` has uncommitted changes.) + **Do not edit source files through the shell.** No `sed -i`, no redirects or `tee` into `.cs`/`.md`/`.csproj`/ `.props`/`.json`/`.ps1`/`.g4`, and no Python scripts that rewrite files. Shell edits bypass the agent's file checkpointing, so the change cannot be rolled back. Use your editor/patch tooling instead. (Claude Code enforces -this with `PreToolUse` hooks in `.claude/settings.json`, which also block reading files via `cat`/`grep`/`ls` in -favour of its own file tools; the underlying convention applies to any agent working here.) +this with `PreToolUse` hooks in `.claude/settings.json`, which also block reading files via `cat`/`grep`/`ls` — +and via `git grep`/`git cat-file`, which read the same way but slip past the read-only-git allowance — in favour +of its own file tools; the underlying convention applies to any agent working here.) Writing scratch files and logs outside the repo is fine. `dotnet build`/`test`/`restore` and read-only `git` -commands are the expected shell usage. +commands are the expected shell usage. Run `LibRed.Core.AccessTests` (~7 min) and `LibRed.Engine.AccessTests` +(~1.5 min) with a `--filter` naming the tests that cover the change, or in the background — Claude Code's hooks +refuse either suite whole in the foreground. ## CI @@ -343,11 +395,14 @@ flags — `jet` and `libred` — and skips the jobs that don't apply: - **BuildAndTest** — the Jet matrix: ACE 2010/2016 × x64/x86 × ODBC/OLE DB on `windows-latest`. The x86 legs patch `IMAGE_FILE_LARGE_ADDRESS_AWARE` onto `dotnet.exe` and the test hosts, because ACE in a 2GB address space dies partway through the largest shard. -- **LibRed** — `LibRed.Engine.Tests` on Linux/Windows/macOS + ubuntu-arm/windows-arm, no ACE anywhere. -- **LibRedAccess** — the ACE cross-check suites on `windows-latest` with ACE 2016. +- **LibRed** — `LibRed.Engine.Tests` and `LibRed.Core.Tests` on Linux/Windows/macOS + ubuntu-arm/windows-arm, + no ACE anywhere. +- **LibRedAccess** — the ACE cross-check suites on `windows-latest` with ACE 2016, plus `LibRed.Ado.Tests` and + `LibRed.EFCore.Tests`, which need no driver but run only there for now. - **LibRedFunctional** — both `EFCore.LibRed.FunctionalTests` (compatible mode) and `EFCore.LibRed.Extended.FunctionalTests` (extended mode) on the five-platform matrix, `continue-on-error`. -- **NuGet** — packs and pushes to MyGet/NuGet for `master`, `*-servicing`, `*-wip` and release tags. +- **NuGet** — packs for `master`, `*-servicing`, `*-wip` and release tags, attaching the packages to the run + as the `nupkgs` artifact. It pushes to nuget.org on **release tags only**; there is no CI package feed. ## Versioning diff --git a/CLAUDE.md b/CLAUDE.md index e7c6234aa..23e37356a 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -6,7 +6,18 @@ This file provides guidance to Claude Code (claude.ai/code) when working with co EntityFrameworkCore.Jet is an EF Core provider for Microsoft Jet/ACE databases (Microsoft Access `.mdb`/`.accdb` files). The **Jet** provider runs **Windows only** and bridges EF Core to the Access database engine via either ODBC or OLE DB. Alongside it, **LibRed** (also in this repo, on `master`) is a from-scratch managed engine that reads/writes the file format directly and is **cross-platform** — see the LibRed section below. -Current version: `11.0.0-alpha.3` (`Version.props`) targeting EF Core 11 and `net11.0`; `global.json` pins the 11.0.100 RC1 SDK with `rollForward: latestFeature`. The test projects use **xunit v3**. +Current version: `11.0.0-alpha.4` (`Version.props`) targeting EF Core 11 and `net11.0`; `global.json` pins the 11.0.100 RC1 SDK with `rollForward: latestFeature`. The test projects use **xunit v3**, except `EFCore.Jet.Data.Tests` and `EFCore.Jet.IntegrationTests`, which use MSTest. + +**One exception to `net11.0`:** `LibRed.Core`, `LibRed.Sql`, `LibRed.Engine` and `LibRed.Ado` multi-target +`$(LibRedTargetFrameworks)` = `net10.0;net11.0`, because none of them depends on an EF Core package. `LibRed.EFCore` +is deliberately **not** in that list — EF Core 11 is `net11.0`-only — and neither is anything under `src/EFCore.Jet*`. + +**Every test project stays single-target `net11.0` (the Jet ones as `net11.0-windows7.0`, via +`$(JetTestWindowsOnlyTargetFramework)`), and that is deliberate — don't "fix" it.** The `net10.0` +leg is *compiled*, never run: multi-targeting the suites would double every run locally and on all five CI +platforms, and would need the .NET 10 runtime installed in CI, which the SDK `global.json` pins does not carry. +A solution build (`dotnet build EFCore.Jet.sln`) or a pack compiles the `net10.0` leg; `dotnet test` on a +single suite builds only the `net11.0` leg it runs. ### Which layer am I touching? @@ -45,14 +56,15 @@ To develop against a local EF Core build instead of NuGet packages, copy `Develo ## Tests -**Jet** tests require a real Microsoft Access driver installed (ODBC or OLE DB) and an actual `.accdb` file — no mocks. The connection string is configured via: +**Jet** tests require a real Microsoft Access driver installed (ODBC or OLE DB) and an actual `.accdb` file — no mocks. The EF suites' connection strings (the LibRed ones included, which need no driver) are configured via: - `test/EFCore.Jet.FunctionalTests/config.json` (OLE DB example present) -- `test/EFCore.Jet.Tests/config.json` (bare filename; picks up default provider) +- `test/EFCore.Jet.Tests/config.json` (ODBC `DBQ=` form) - `test/EFCore.LibRed.FunctionalTests/config.json` and `test/EFCore.LibRed.Extended.FunctionalTests/config.json` (LibRed connection, one per SQL mode) -- Or env var `EFCoreJet_DefaultConnection` +- Or env var `EFCoreJet_DefaultConnection` (the Jet suite) / `EFCoreLibRed_DefaultConnection` (both LibRed + suites) -**LibRed** tests split in two: `LibRed.Core.Tests`, `LibRed.Engine.Tests`, `EFCore.LibRed.FunctionalTests` and `EFCore.LibRed.Extended.FunctionalTests` need **no driver at all** and CI runs them on Linux/Windows/macOS plus ARM64 legs — that matrix is what proves the cross-platform claim, so don't add an ACE dependency to them. `LibRed.Core.AccessTests`, `LibRed.Engine.AccessTests`, `LibRed.Ado.Tests` and `LibRed.EFCore.Tests` deliberately cross-check LibRed's output against the real engine over OLE DB, so they need Windows + ACE. +**LibRed** tests split in two: `LibRed.Core.Tests`, `LibRed.Engine.Tests`, `EFCore.LibRed.FunctionalTests` and `EFCore.LibRed.Extended.FunctionalTests` need **no driver at all** and CI runs them on Linux/Windows/macOS plus ARM64 legs — that matrix is what proves the cross-platform claim, so don't add an ACE dependency to them. `LibRed.Core.AccessTests` and `LibRed.Engine.AccessTests` deliberately cross-check LibRed's output against the real engine over OLE DB, so they need Windows + ACE. `LibRed.Ado.Tests` and `LibRed.EFCore.Tests` use no driver (no OLE DB, no DAO), but CI only runs them in the Windows ACE job for now. > **The `*.AccessTests` split is by which engine a test needs, not by subject.** A file-format test belongs in > `LibRed.Core.Tests` if LibRed alone can decide the answer, and in `LibRed.Core.AccessTests` if ACE has to be @@ -91,7 +103,8 @@ is the most common way to waste minutes here. **When you do run a suite, capture the failing test *names* in the same run** — don't reduce the output to just the `Passed!/Failed!` count line and then re-run the whole suite to find which failed. Grep a pattern that catches both, e.g. `grep -iE "Passed!|Failed!|\[FAIL\]|error CS"` (xUnit prints `… [FAIL]` and `Failed ` lines as it goes), or tee the full output to a file and inspect it. -Tests run in **fixed order by default** (`FIXED_TEST_ORDER` compile constant, set unless `-p:FixedTestOrder=false`; see `test/Directory.Build.props`). All tests lock culture to `en-US` via a module initializer (`test/Shared/ModuleInitializer.cs`). +The three EF functional suites run in **fixed order by default** (`FIXED_TEST_ORDER` compile constant, set unless `-p:FixedTestOrder=false`; see `test/Directory.Build.props` — every project gets the constant, but only those suites and the empty +`EFCore.Jet.Tests` act on it) and lock culture to `en-US` via a module initializer (`test/Shared/ModuleInitializer.cs`, compiled into those suites only). The other test projects set culture themselves where it matters, and `LibRed.Core.Tests` runs its collections in parallel. Tests that require features Jet doesn't support are skipped with a reason on the test. @@ -101,13 +114,18 @@ Tests that require features Jet doesn't support are skipped with a reason on the `EFCore.Jet.FunctionalTests` is gated by a committed pass-list rather than by "everything must pass": `test/EFCore.Jet.FunctionalTests/GreenTests/ace___.txt` lists the tests that passed -previously for that matrix leg. CI merges the shards' `.trx` files and **fails if any listed test stops passing**; -newly-passing tests are appended and pushed back by the `auto_commit` workflow. So the meaningful question for a +previously for that matrix leg. Only two legs have one today — `ace_2010_odbc_x86` and `ace_2010_oledb_x86` — and +the check runs only where the file exists, so the other six legs are not gated. CI merges the shards' `.trx` +files and **fails if any listed test stops passing**; after a successful pull-request run, newly-passing tests are +appended to an existing list and pushed back by the `auto_commit` workflow (it is triggered by +`pull_request.yml`, not by `push.yml`, and never creates a list). So the meaningful question for a change is "did anything that used to pass stop passing", not the raw failure count. -CI splits the functional suite into three shards (query core / Northwind+GearsOfWar / non-query) and retries a shard -up to three times if the runner crashes. The two `EFCore.LibRed*.FunctionalTests` suites are `continue-on-error` -for now — they still run on every push, but their remaining failures don't block. +CI splits the functional suite into four shards (query core / Northwind+GearsOfWar / non-query without +CompiledModel / CompiledModel) and runs a shard up to three times if the runner crashes (on a leg with a +pass-list, a shard that crashes all three times fails it). The two `EFCore.LibRed*.FunctionalTests` suites are +`continue-on-error` for now — they run on every push that touches LibRed, but their remaining failures don't +block. ### Docker images @@ -140,14 +158,18 @@ test/ EFCore.Jet.Tests/ EMPTY — its test files are d [Windows + ACE] EFCore.Jet.IntegrationTests/ Integration scenario tests [Windows + ACE] JetProviderExceptionTests/ Exception-path tests; also hosts Northwind.accdb, - which every LibRed suite links to as its fixture [Windows + ACE] + which every LibRed.* test project and LibRed.Benchmarks + link to as their fixture (the EF functional suites + build theirs from test/Northwind.sql) [Windows + ACE] LibRed.Core.Tests/ File-format read/write through LibRed alone [cross-platform] LibRed.Core.AccessTests/ File-format tests cross-checked against ACE [Windows + ACE] LibRed.Engine.Tests/ Planner/executor, no engine dependency [cross-platform] LibRed.Engine.AccessTests/ Engine tests that cross-check against ACE [Windows + ACE] - LibRed.Ado.Tests/ ADO.NET surface [Windows + ACE] + LibRed.Ado.Tests/ ADO.NET surface [no driver; CI's Windows ACE job] LibRed.EFCore.Tests/ LibRed EF Core provider: query round-trip, - database-first scaffolding [Windows + ACE] + database-first scaffolding [no driver; CI's Windows ACE job] + LibRed.Shared/ Helpers compiled into the Core/Engine test pairs and LibRed.Ado.Tests + (TemporaryDatabase, …; AceTestDatabase into the AccessTests only) EFCore.LibRed.FunctionalTests/ EF Core specification suite over LibRed, compatible SQL mode (Jet-dialect SQL) [cross-platform] EFCore.LibRed.Extended.FunctionalTests/ @@ -158,7 +180,8 @@ test/ split, storage, writes, DDL, catalog, and an opt-in ACE head-to-head. `-- --validate` checks the SQL corpus. See its README.md [cross-platform] - Shared/ ModuleInitializer.cs — locks culture to en-US + Shared/ For the EF functional suites: ModuleInitializer.cs (locks culture to + en-US) and TestUtilities/ (test orderers, crash detection, conditions) tools/ sortkey-table/ Generates the Windows NLS sort-weight table LibRed's index keys use @@ -252,8 +275,8 @@ in `EFCore.Jet.Data`. ``` src/LibRed/ LibRed.Core/ File format: IO (PageChannel/PageBuffer), Formats (version offsets), - Pages, Catalog (MSysObjects → TableDef/ColumnDef/IndexDef), Storage - (Table/TableCursor/RowDecoder/UsageMap), Crypto; JetDatabase entry point + Pages, Catalog (MSysObjects → TableDefinition/ColumnDef/IndexDef), Storage + (Table/TableCursor/RowCodec/UsageMap), Crypto; JetDatabase entry point LibRed.Sql/ SQL front end: ANTLR grammar (AccessSql.g4), AST, parser, binder. NO Jet dependency — binds via the ISchemaProvider abstraction LibRed.Engine/ Logical Plan nodes, QueryPlanner, CatalogSchemaProvider (bridges the @@ -279,7 +302,7 @@ the names are `JetVersion` / `RequiredVersion` / `EnsureFormatAtLeast`: - `AccessTypeMapper.MapType` **refuses** a type the open file is too old for, so a caller that can't upgrade (read-only database) fails loudly instead of writing a column Access couldn't read. That guard is on the SQL path; the same rule is enforced again in Core over `JetDataType` (`JetDataTypeVersions.EnsureStorable`, called - from `TdefBuilder` and `TableCreator`), because every `JetDatabase` DDL method takes a raw `ColumnSpec` and + from `TableDefinition` and `SchemaEditor`), because every `JetDatabase` method that defines a column takes a raw `ColumnSpec` and would otherwise write the descriptor with nothing objecting. - `StatementExecutor.MapColumn` → `JetDatabase.EnsureFormatAtLeast` → `PageChannel.RaiseFormatVersion` **raises the file's version byte** rather than refusing the DDL, which is what ACE itself does. The raise goes @@ -329,7 +352,7 @@ parser: the lexer/parser are pre-generated and committed under `LibRed.Sql/Gramm `LibRed.Sql/Grammar/generate.ps1` after editing `AccessSql.g4`. > **LibRed is a single-writer engine.** It tolerates extra open handles (a `.accdb` is a shared-file database and -> EF's own test infra keeps a store connection open), but there is no lock file and no page/record locking, so +> EF's own test infra keeps a store connection open), but there is no lock file and no cross-process page/record locking (its page locks are process-local), so > two concurrent writers corrupt the file. Transactions use a **deferred-write overlay** per `PageChannel`, not > an undo log: writes buffer until commit and publish under a lock, so a reader cannot see another handle's > uncommitted pages and a rollback cannot discard another channel's committed work — both hazards the older @@ -346,16 +369,34 @@ here, because it sends the work off in a direction that has to be unwound later. verify against the code rather than quoting the list; and a **memory or summary records what was true when it was written**, not what is true now. -`.claude/settings.json` installs `PreToolUse` hooks that **deny** two things in `Bash`/`PowerShell`: +**Change the existing path; don't add one beside it.** A fix reshapes the code that already does the job: split +a method at the point where its callers differ, change what it takes, move a check to where every caller passes +through. It does not add an overload, wrapper, helper, constant or parallel method next to it. Before writing +any new member, find the code that already does that job — by what it does, not by its name: a decode, a lookup, +a layout, a check, a value — and use or reshape it. A second copy is how this codebase got its bugs: two readers +of one field that disagreed, a merged path that lost a check only one copy had, hard-coded page numbers restating +a pointer, test helpers re-implementing Core and drifting from it. A genuinely new member is fine when nothing +existing does the job; say what you checked. + +`.claude/settings.json` installs `PreToolUse` hooks that **deny** three things in `Bash`/`PowerShell`: - **Reading files through the shell** (`cat`, `head`, `grep`, `ls`, `find`, `Get-Content`, `Select-String`, …) — - use `Read`, `Grep`, `Glob` instead. Shell text tools stay allowed on paths containing `scratchpad`, `.log` or `/tmp/`. + use `Read`, `Grep`, `Glob` instead. Shell text tools stay allowed on paths containing `scratchpad`, `.log`, `$log` or `/tmp/`. + `git grep` and `git cat-file` are denied with them: read-only `git` is otherwise pre-allowed, which made them a + way to spell the same read and skip the prompt. `git diff`/`log`/`show` stay allowed for reviewing history. - **Editing source files through the shell** (`sed -i`, redirects/`tee` into `.cs`/`.md`/`.csproj`/`.props`/`.json`/ `.ps1`/`.g4`, and any `python` invocation) — these bypass file checkpointing, so `/rewind` cannot undo them. Use `Edit`/`Write`. +- **`LibRed.Core.AccessTests` or `LibRed.Engine.AccessTests` in the foreground with no `--filter`** — they are + ~7 and ~1.5 minutes, and waiting on them is dead session time. Add a `--filter` naming the tests that cover + the change, or pass `run_in_background` and carry on working while they run. The other suites are unaffected. -`dotnet build`/`test`/`restore` and read-only `git` commands are pre-allowed, so don't work around the hooks — the -denial message is telling you which tool to use, not that the action is forbidden. +`dotnet build`/`test`/`restore` and, in Bash, `git status`/`diff`/`log`/`show`/`branch`/`add`/`commit`/`push` +are pre-allowed, and PowerShell is allowed wholesale — so don't work around the hooks; the denial message is +telling you which tool to use, not that the action is forbidden. A `PostToolUse` hook also rewrites every file +the `Write` tool writes to CRLF line endings. A `Stop` hook blocks the end of a turn once while `src/` (its +`docs` folders aside) has uncommitted changes, asking for the review in "Change the existing path" above; the +next stop goes through. ## CI @@ -365,11 +406,14 @@ flags — `jet` and `libred` — and skips the jobs that don't apply: - **BuildAndTest** — the Jet matrix: ACE 2010/2016 × x64/x86 × ODBC/OLE DB on `windows-latest`. The x86 legs patch `IMAGE_FILE_LARGE_ADDRESS_AWARE` onto `dotnet.exe` and the test hosts, because ACE in a 2GB address space dies partway through the largest shard. -- **LibRed** — `LibRed.Engine.Tests` on Linux/Windows/macOS + ubuntu-arm/windows-arm, no ACE anywhere. -- **LibRedAccess** — the ACE cross-check suites on `windows-latest` with ACE 2016. +- **LibRed** — `LibRed.Engine.Tests` and `LibRed.Core.Tests` on Linux/Windows/macOS + ubuntu-arm/windows-arm, + no ACE anywhere. +- **LibRedAccess** — the ACE cross-check suites on `windows-latest` with ACE 2016, plus `LibRed.Ado.Tests` and + `LibRed.EFCore.Tests`, which need no driver but run only there for now. - **LibRedFunctional** — both `EFCore.LibRed.FunctionalTests` (compatible mode) and `EFCore.LibRed.Extended.FunctionalTests` (extended mode) on the five-platform matrix, `continue-on-error`. -- **NuGet** — packs and pushes to MyGet/NuGet for `master`, `*-servicing`, `*-wip` and release tags. +- **NuGet** — packs for `master`, `*-servicing`, `*-wip` and release tags, attaching the packages to the run + as the `nupkgs` artifact. It pushes to nuget.org on **release tags only**; there is no CI package feed. ## Versioning diff --git a/Directory.Build.props b/Directory.Build.props index a8c271f58..42bc930e3 100644 --- a/Directory.Build.props +++ b/Directory.Build.props @@ -30,6 +30,10 @@ $(EfCoreTargetFramework) net11.0 $(JetTestTargetFramework)-windows7.0 + + net10.0;$(JetTargetFramework) diff --git a/NuGet.Config b/NuGet.Config index aa6e59c7d..246161763 100644 --- a/NuGet.Config +++ b/NuGet.Config @@ -10,11 +10,6 @@ feed the three suites that inherit EF's test bases fail to restore with NU1101. --> - - \ No newline at end of file diff --git a/Version.props b/Version.props index f7528b9b9..6b69c475e 100644 --- a/Version.props +++ b/Version.props @@ -17,7 +17,7 @@ --> 11.0.0 alpha - 3 + 4 + + @@ -33,6 +36,9 @@ + + diff --git a/src/LibRed/LibRed.Core/Pages/DataPage.cs b/src/LibRed/LibRed.Core/Pages/DataPage.cs index 94143e3c6..ae75667ad 100644 --- a/src/LibRed/LibRed.Core/Pages/DataPage.cs +++ b/src/LibRed/LibRed.Core/Pages/DataPage.cs @@ -1,15 +1,10 @@ using LibRed.Formats; using LibRed.IO; +using LibRed.Storage; +using System.Buffers.Binary; namespace LibRed.Pages; -/// One entry in a data page's row slot directory. -/// Byte offset of the row record within the page. -/// Length of the row record in bytes. -/// The row is marked deleted. -/// The slot points at an overflow/lookup record rather than inline data. -public readonly record struct RowSlot(int Offset, int Length, bool IsDeleted, bool HasOverflow); - /// /// A data page holding the rows of a single table. Rows are addressed by a slot /// directory near the front of the page and packed from the page end backward. @@ -17,12 +12,22 @@ namespace LibRed.Pages; /// public sealed class DataPage : Page { + /// One entry in a data page's row slot directory. + /// Byte offset of the row record within the page. + /// Length of the row record in bytes. + /// The row is marked deleted. + /// The slot points at an overflow/lookup record rather than inline data. + public readonly record struct RowSlot(int Offset, int Length, bool IsDeleted, bool HasOverflow); + - private readonly List _rows = []; + private readonly List _rows = []; private PageBuffer _buffer; public override PageType Type => PageType.DataPage; + /// Marks this data page released, preserving its row directory and data. + internal static void MarkReleased(Span page) => PageHeader.WriteType(page, PageType.ReleasedDataPage); + /// The TDEF page of the table that owns this data page (0 for long-value pages). public int OwningTablePage { get; private set; } @@ -32,39 +37,188 @@ public sealed class DataPage : Page public int FreeSpace { get; private set; } public int RowCount { get; private set; } - public IReadOnlyList Rows => _rows; + public IReadOnlyList Rows => _rows; - public override void Read(PageBuffer buffer, JetFormatBase format) + internal override void Read(PageBuffer buffer, JetFormatBase format) { ValidateHeader(buffer, format, out int rowCount, out int directoryEnd); _buffer = buffer; PageNumber = buffer.PageNumber; - uint owner = buffer.ReadUInt32(format.DataOwnerOffset); - IsLongValuePage = owner == LongValueFormat.LvalMarker; + uint owner = ReadOwner(buffer.Span, format); + IsLongValuePage = owner == JetFormatBase.LongValuePageMarker; OwningTablePage = IsLongValuePage ? 0 : (int)owner; - FreeSpace = buffer.ReadUInt16(format.DataFreeSpaceOffset); + FreeSpace = ReadFreeSpace(buffer.Span, format); RowCount = rowCount; _rows.Clear(); int prevEnd = buffer.Length; for (int i = 0; i < RowCount; i++) { - int raw = buffer.ReadUInt16(format.DataRowDirectoryOffset + i * 2); - int offset = raw & RowPointer.OffsetMask; - bool deleted = (raw & RowPointer.DeletedFlag) != 0; - bool overflow = (raw & RowPointer.OverflowFlag) != 0; + (int offset, RowSlotFlags flags) = ReadSlot(buffer.Span, format, i); // Rows are packed from the page end backward, so a slot runs from its own // offset up to where the previous slot's row began. ValidateSlot(buffer, i, offset, prevEnd, directoryEnd); - int length = prevEnd - offset; - _rows.Add(new RowSlot(offset, length, deleted, overflow)); + _rows.Add(Slot(offset, prevEnd - offset, flags)); prevEnd = offset; } } + private static DataPage.RowSlot Slot(int offset, int length, RowSlotFlags flags) => + new(offset, length, flags.HasFlag(RowSlotFlags.Deleted), flags.HasFlag(RowSlotFlags.Overflow)); + + /// A slot's , the inverse of how reads them. + internal static RowSlotFlags Flags(DataPage.RowSlot slot) => + (slot.IsDeleted ? RowSlotFlags.Deleted : RowSlotFlags.None) + | (slot.HasOverflow ? RowSlotFlags.Overflow : RowSlotFlags.None); + + /// The page's owner field: the TDEF page of the table it belongs to, + /// on a long-value page, and 0 on a usage-map page, which belongs to no table. + internal static uint ReadOwner(ReadOnlySpan page, JetFormatBase format) => + BinaryPrimitives.ReadUInt32LittleEndian(page.Slice(format.DataOwnerOffset, sizeof(uint))); + + /// A fresh data page belonging to (as reads it), with no + /// rows: a zero count, and the whole page free. Every data page this engine creates starts here. + internal static byte[] NewPage(JetFormatBase format, uint owner) + { + var page = new byte[format.PageSize]; + PageHeader.WriteType(page, PageType.DataPage); + BinaryPrimitives.WriteUInt32LittleEndian(page.AsSpan(format.DataOwnerOffset, sizeof(uint)), owner); + LayRows(page, format, [], []); + return page; + } + + /// The page's row count — the number of entries in its slot directory. + internal static int ReadRowCount(ReadOnlySpan page, JetFormatBase format) => + BinaryPrimitives.ReadUInt16LittleEndian(page.Slice(format.DataRowCountOffset, sizeof(ushort))); + + /// The page's declared free space: the bytes between the end of the slot directory and the lowest row. + internal static int ReadFreeSpace(ReadOnlySpan page, JetFormatBase format) => + BinaryPrimitives.ReadUInt16LittleEndian(page.Slice(format.DataFreeSpaceOffset, sizeof(ushort))); + + /// Where a slot directory of entries ends — the lowest offset a row may take. + internal static int DirectoryEnd(JetFormatBase format, int rowCount) => + format.DataRowDirectoryOffset + rowCount * format.DataRowDirectoryEntrySize; + + private static void WriteHeader(Span page, JetFormatBase format, int rowCount, int freeSpace) + { + BinaryPrimitives.WriteUInt16LittleEndian(page.Slice(format.DataRowCountOffset, sizeof(ushort)), (ushort)rowCount); + BinaryPrimitives.WriteUInt16LittleEndian(page.Slice(format.DataFreeSpaceOffset, sizeof(ushort)), (ushort)freeSpace); + } + + /// The slot directory entry of : the row's offset, and the + /// in the bits above it. Unvalidated — the caller bounds the row and the offset. + internal static (int Offset, RowSlotFlags Flags) ReadSlot(ReadOnlySpan page, JetFormatBase format, int row) + { + int raw = BinaryPrimitives.ReadUInt16LittleEndian( + page.Slice(format.DataRowDirectoryOffset + row * format.DataRowDirectoryEntrySize, format.DataRowDirectoryEntrySize)); + return (raw & format.DataRowOffsetMask, (RowSlotFlags)(raw & ~format.DataRowOffsetMask)); + } + + /// Writes the slot directory entry of — the inverse of . + internal static void WriteSlot(Span page, JetFormatBase format, int row, int offset, RowSlotFlags flags) => + BinaryPrimitives.WriteUInt16LittleEndian( + page.Slice(format.DataRowDirectoryOffset + row * format.DataRowDirectoryEntrySize, format.DataRowDirectoryEntrySize), + (ushort)((int)flags | (offset & format.DataRowOffsetMask))); + + /// The offset of the lowest row on the page — where the next row packed from the end goes below — + /// or the page size when there are none. + internal static int LowestRowOffset(ReadOnlySpan page, JetFormatBase format) + { + int lowest = format.PageSize; + for (int i = ReadRowCount(page, format) - 1; i >= 0; i--) + lowest = Math.Min(lowest, ReadSlot(page, format, i).Offset); + return lowest; + } + + /// + /// Lays on from its end backward in slot order — row 0 + /// nearest the end — each slot carrying its own , and sets the row count and free space + /// to match. A zero-length record is a tombstone on the previous row's start. Every write that rearranges a + /// page's rows comes through here; what precedes the directory and what lies between it and the rows is left as + /// it stands. Returns false, writing nothing, when the records and their directory do not fit the page. + /// + internal static bool LayRows(Span page, JetFormatBase format, IReadOnlyList records, + IReadOnlyList flags) + { + int directoryEnd = DirectoryEnd(format, records.Count); + if (format.PageSize - records.Sum(r => r.Length) < directoryEnd) return false; + + int offset = format.PageSize; + for (int i = 0; i < records.Count; i++) + { + offset -= records[i].Length; + records[i].CopyTo(page[offset..]); + WriteSlot(page, format, i, offset, flags[i]); + } + WriteHeader(page, format, records.Count, offset - directoryEnd); + return true; + } + + /// + /// Appends below the page's lowest row as a new slot carrying , + /// leaving every existing row where it is, and updates the row count and free space — the declared free space + /// less the record and its slot, so bytes the page already counted are not recounted. is + /// the new slot — the old row count. Returns false, writing nothing, when the record and its slot do not fit + /// between the directory and the lowest row. + /// + internal static bool TryAppendRow(Span page, JetFormatBase format, ReadOnlySpan record, + RowSlotFlags flags, out int row) + { + row = ReadRowCount(page, format); + int offset = LowestRowOffset(page, format) - record.Length; + if (offset < DirectoryEnd(format, row + 1)) return false; + + record.CopyTo(page[offset..]); + WriteSlot(page, format, row, offset, flags); + WriteHeader(page, format, row + 1, + ReadFreeSpace(page, format) - record.Length - format.DataRowDirectoryEntrySize); + return true; + } + + /// + /// Takes a row's bytes off its page the way ACE does: the rows stored below it slide up to close the + /// gap, their slot offsets follow, and the emptied slot becomes a zero-length tombstone whose offset is + /// the row's FORMER END, flagged deleted + overflow. The freed bytes go back to the page's free-space + /// count, so a delete-heavy table stops growing where ACE's would not. + /// + /// Slot indices never move, which is what keeps index entries and row ids valid — only offsets + /// change. Pointing the tombstone at the former end rather than at the page end is what keeps the + /// directory non-increasing, which relies on to derive each row's length from the + /// previous slot. Verified against ACE for a first, middle and last row: deleting the first of three + /// 19-byte rows gives D000 0FED 0FDA, the middle 0FED CFED 0FDA, the last + /// 0FED 0FDA CFDA, with free space rising by 19 in each case. + /// + /// + internal static void ReclaimRow(Span page, JetFormatBase format, int row) + { + int rowCount = ReadRowCount(page, format); + + // Slot offsets are non-increasing with slot index, so row i occupies [offset(i), offset(i-1)) and + // every row stored below this one is simply a LATER slot. Working by slot index rather than by + // comparing offsets is what keeps zero-length tombstones correct: one sitting at exactly this row's + // offset has to move up with the rows after it, and an offset comparison leaves it behind — where it + // then absorbs this row's length and starves the next live row down to zero. + int start = ReadSlot(page, format, row).Offset; + int end = row == 0 ? format.PageSize : ReadSlot(page, format, row - 1).Offset; + int length = end - start; + + int lowest = rowCount > 0 ? ReadSlot(page, format, rowCount - 1).Offset : format.PageSize; + if (length > 0 && start > lowest) + page[lowest..start].CopyTo(page[(lowest + length)..]); + + for (int i = row + 1; i < rowCount; i++) + { + (int offset, RowSlotFlags flags) = ReadSlot(page, format, i); + WriteSlot(page, format, i, offset + length, flags); + } + + WriteSlot(page, format, row, end, RowSlotFlags.Deleted | RowSlotFlags.Overflow); + WriteHeader(page, format, rowCount, ReadFreeSpace(page, format) + length); + } + /// Returns the raw bytes of the row at in the slot directory. /// The index is outside the slot directory. Callers pass a row /// number read out of the file (a usage-map pointer, a long-value descriptor), so out of range means @@ -75,7 +229,7 @@ public ReadOnlySpan GetRow(int index) throw new InvalidDataException( $"Row {index} is outside this page's slot directory ({_rows.Count} rows)."); - RowSlot slot = _rows[index]; + DataPage.RowSlot slot = _rows[index]; return _buffer.Slice(slot.Offset, slot.Length); } @@ -84,7 +238,7 @@ public ReadOnlySpan GetRow(int index) /// where parsing every slot on the page just to take one row was the dominant cost. Returns false when /// is past the page's row count. public static bool TryReadRow(PageBuffer buffer, JetFormatBase format, int index, - out RowSlot slot, out ReadOnlySpan bytes) + out DataPage.RowSlot slot, out ReadOnlySpan bytes) { ValidateHeader(buffer, format, out int rowCount, out int directoryEnd); if (index < 0 || index >= rowCount) @@ -94,14 +248,12 @@ public static bool TryReadRow(PageBuffer buffer, JetFormatBase format, int index return false; } - int dir = format.DataRowDirectoryOffset; - int raw = buffer.ReadUInt16(dir + index * 2); - int offset = raw & RowPointer.OffsetMask; + (int offset, RowSlotFlags flags) = ReadSlot(buffer.Span, format, index); // Rows pack from the page end backward, so this slot runs up to where the previous slot began // (or the page end for slot 0) — read just those two directory entries instead of walking all of them. - int prevEnd = index == 0 ? buffer.Length : buffer.ReadUInt16(dir + (index - 1) * 2) & RowPointer.OffsetMask; + int prevEnd = index == 0 ? buffer.Length : ReadSlot(buffer.Span, format, index - 1).Offset; ValidateSlot(buffer, index, offset, prevEnd, directoryEnd); - slot = new RowSlot(offset, prevEnd - offset, (raw & RowPointer.DeletedFlag) != 0, (raw & RowPointer.OverflowFlag) != 0); + slot = Slot(offset, prevEnd - offset, flags); bytes = buffer.Slice(offset, prevEnd - offset); return true; } @@ -111,23 +263,29 @@ private static void ValidateHeader(PageBuffer buffer, JetFormatBase format, out if (buffer.Length != format.PageSize) throw new InvalidDataException( $"Data page {buffer.PageNumber} has {buffer.Length} bytes; expected {format.PageSize}."); - if (buffer.ReadByte(0) != (byte)PageType.DataPage) + if (PageHeader.ReadType(buffer.Span) != PageType.DataPage) throw new InvalidDataException( - $"Page {buffer.PageNumber} is type 0x{buffer.ReadByte(0):X2}, not a data page (0x01)."); + $"Page {buffer.PageNumber} is type 0x{(ushort)PageHeader.ReadType(buffer.Span):X4}, not a data page (0x0101)."); - rowCount = buffer.ReadUInt16(format.DataRowCountOffset); - long end = (long)format.DataRowDirectoryOffset + rowCount * 2L; - if (end > buffer.Length) + rowCount = ReadRowCount(buffer.Span, format); + // An index addresses a row by a one-byte slot number, so a page can hold at most 255 and ACE stops + // there. Reading further is not tolerance, it is reading rows no index can name — including any this + // engine wrote past the cap, which is exactly the bug a strict reader is here to surface. + if (rowCount > format.MaxRowsPerPage) + throw new InvalidDataException( + $"Data page {buffer.PageNumber} declares {rowCount} rows; a page holds at most " + + $"{format.MaxRowsPerPage}, the most a one-byte slot number can address."); + directoryEnd = DirectoryEnd(format, rowCount); // the count is capped above, so this cannot overflow + if (directoryEnd > buffer.Length) throw new InvalidDataException( $"Data page {buffer.PageNumber} declares {rowCount} rows, placing its slot directory past the page."); - directoryEnd = (int)end; } private static void ValidateSlot(PageBuffer buffer, int index, int offset, int previousEnd, int directoryEnd) => ValidateSlot(buffer.PageNumber, buffer.Length, index, offset, previousEnd, directoryEnd); /// The slot-directory invariant, over raw values so the write paths can share it. Offsets are - /// masked out of 13 bits and so can reach 8191 on a 4096-byte page; rows are packed from the page end + /// masked out of and so can exceed the page; rows are packed from the page end /// backward, so they must also never increase. RowInserter's in-place repackers re-derived this /// arithmetic without the checks, which on a corrupt directory produced an out-of-range exception from a /// half-repacked page rather than a diagnosis. @@ -142,4 +300,63 @@ internal static void ValidateSlot( throw new InvalidDataException( $"Data page {pageNumber} row slot {index} has offset {offset}, above the previous row boundary {previousEnd}."); } + + /// A validated, page-backed relocated-row target. The bytes remain zero-copy for index seeks. + internal readonly record struct RelocatedRow(PageBuffer Buffer, DataPage.RowSlot Slot, int RowNumber) + { + public ReadOnlySpan Bytes => Buffer.Slice(Slot.Offset, Slot.Length); + } + + + + /// + /// Validates and follows the forward pointer at the START of a live overflow row slot. + /// + /// + /// The slot is normally exactly 4 bytes: ACE's DML and both trim it down to the + /// pointer when a row is relocated. Measured over 317 relocations with no exception, across ACE x64, the + /// ACE 2010 x86 runtime, and LibRed's own writer, under growing and shrinking text, repeated re-relocation, + /// page fragmentation by interleaved deletes, and an OLE column going from NULL to a value. + /// + /// Real files nevertheless contain longer ones. Northwind's MSysAccessStorage has live overflow slots + /// of 45-63 bytes, and their contents are the row as it was BEFORE it moved, with only the leading 4 bytes + /// replaced by the pointer: every field lands where the row format puts it once those 4 bytes are discounted, + /// the keys match the row it forwards to, and the remnant's null bitmap differs from its target's in exactly + /// the OLE column's bit — the value whose arrival grew the row and forced the move. The slot simply kept the + /// old row's width. + /// + /// What wrote them is NOT known: no write path reproduces the shape, including the OLE-column transition the + /// bytes themselves record. So this reads the leading pointer and ignores whatever follows, rather than + /// asserting a width. The checks that matter are unchanged and do the real work — the target must be in the + /// file, owned by the same table, and a nonempty hidden inline row. + /// + internal static RelocatedRow ResolveRelocation(PageChannel channel, int owningTablePage, + DataPage.RowSlot sourceSlot, ReadOnlySpan sourceBytes) + { + if (sourceSlot.IsDeleted || !sourceSlot.HasOverflow) + throw new InvalidDataException("A relocation source must be a live overflow row slot."); + if (sourceBytes.Length < PageBuffer.RecordPointerSize) + throw new InvalidDataException( + $"A relocation source must begin with a {PageBuffer.RecordPointerSize}-byte pointer; found {sourceBytes.Length} bytes."); + + (int rowNumber, int pageNumber) = PageBuffer.ReadRecordPointer(sourceBytes, 0); + if (pageNumber <= 0 || pageNumber >= channel.PageCount) + throw new InvalidDataException( + $"Relocation pointer targets page {pageNumber}, outside the file's 1..{channel.PageCount - 1} range."); + + PageBuffer targetBuffer = channel.ReadPageShared(pageNumber); + uint owner = DataPage.ReadOwner(targetBuffer.Span, channel.Format); + if (owner != (uint)owningTablePage) + throw new InvalidDataException( + $"Relocation target page {pageNumber} belongs to TDEF {owner}, not TDEF {owningTablePage}."); + if (!DataPage.TryReadRow(targetBuffer, channel.Format, rowNumber, out DataPage.RowSlot targetSlot, out _)) + throw new InvalidDataException( + $"Relocation pointer targets missing row {rowNumber} on page {pageNumber}."); + if (!targetSlot.IsDeleted || targetSlot.HasOverflow || targetSlot.Length == 0) + throw new InvalidDataException( + $"Relocation target {pageNumber}:{rowNumber} is not a nonempty hidden inline row."); + + return new RelocatedRow(targetBuffer, targetSlot, rowNumber); + } + } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Pages/DatabaseDefinitionPage.cs b/src/LibRed/LibRed.Core/Pages/DatabaseDefinitionPage.cs index d2723ef56..b71bd28a4 100644 --- a/src/LibRed/LibRed.Core/Pages/DatabaseDefinitionPage.cs +++ b/src/LibRed/LibRed.Core/Pages/DatabaseDefinitionPage.cs @@ -1,5 +1,8 @@ using LibRed.Catalog; using LibRed.IO; +using LibRed.Formats; +using LibRed.Storage; +using System.Text; using System.Buffers.Binary; namespace LibRed.Pages; @@ -44,76 +47,247 @@ public sealed class DatabaseDefinitionPage : Page new((CollatingOrder)DefaultCollationLcid, DefaultCollationVersion, DefaultCollationSortId); /// Page number of the MSysObjects TDEF (the catalog root), read from the bootstrap - /// pointer at . 2 in every observed file. + /// pointer at . 2 where ACE creates it, but any page the pointer names. public int CatalogRootPage { get; internal set; } + /// Page number of the MSysACEs TDEF, from the bootstrap pointer at + /// . + public int AcesRootPage { get; internal set; } + + /// Page number of the MSysQueries TDEF, from the bootstrap pointer at + /// . + public int QueriesRootPage { get; internal set; } + + /// Page number of the MSysRelationships TDEF, from the bootstrap pointer at + /// . + public int RelationshipsRootPage { get; internal set; } + + /// Page number of the MSysAccounts TDEF, from the bootstrap pointer at + /// ; zero unless this is a workgroup file. + public int AccountsRootPage { get; internal set; } + + /// Page number of the MSysGroups TDEF, from the bootstrap pointer at + /// ; zero unless this is a workgroup file. + public int GroupsRootPage { get; internal set; } + /// Where the global free-pages usage map lives, from the [row:1][page:3] pointer at - /// . Page 1 row 0 in every file ACE writes. + /// . Page 1 row 0 where ACE creates it. public (int Row, int Page) FreePagesMap { get; internal set; } /// Where the global released-pages usage map lives, from the [row:1][page:3] pointer at - /// . Page 1 row 1 in every file ACE writes. + /// . Page 1 row 1 where ACE creates it. public (int Row, int Page) ReleasedPagesMap { get; internal set; } public DateTime DatabaseCreationDate { get; internal set; } - public override void Read(PageBuffer buffer, Formats.JetFormatBase format) + internal override void Read(PageBuffer buffer, Formats.JetFormatBase format) { PageNumber = buffer.PageNumber; FormatIdentifier = Formats.JetFormatBase.ReadFormatIdentifier(buffer.Span); - JetVersion = buffer.ReadByte(Formats.JetFormatBase.VersionOffset); + JetVersion = Formats.JetFormatBase.ReadVersionByte(buffer.Span); - // The header from 0x18 is XOR-obfuscated with the fixed 128-byte mask; de-obfuscate the + // The header from 0x18 is XOR-obfuscated with a fixed RC4 keystream; de-obfuscate the // whole region once, then read the fields out of the clear copy. - Span clear = stackalloc byte[Formats.JetFormatBase.PageZeroHeaderMask.Length]; - Demask(buffer.Span, clear); - int b = Formats.JetFormatBase.PageZeroHeaderMaskStart; - - CodePage = BinaryPrimitives.ReadUInt16LittleEndian(clear.Slice(Formats.JetFormatBase.CodePageOffset - b, 2)); - DatabaseKey = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(Formats.JetFormatBase.DatabaseKeyOffset - b, 4)); - DefaultCollationLcid = BinaryPrimitives.ReadUInt16LittleEndian(clear.Slice(Formats.JetFormatBase.CollationSortOrderOffset - b, 2)); - DefaultCollationSortId = clear[Formats.JetFormatBase.CollationSortIdOffset - b]; - DefaultCollationVersion = clear[Formats.JetFormatBase.CollationVersionOffset - b]; - CatalogRootPage = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(Formats.JetFormatBase.CatalogRootPointerOffset - b, 4)); - FreePagesMap = ReadMapPointer(buffer.Span, Formats.JetFormatBase.FreePagesMapPointerOffset); - ReleasedPagesMap = ReadMapPointer(buffer.Span, Formats.JetFormatBase.ReleasedPagesMapPointerOffset); + Span clear = stackalloc byte[format.PageZeroHeaderMaskLength]; + int b = format.PageZeroHeaderMaskStart; + ReadMasked(buffer.Span, b, clear, format); + + CodePage = BinaryPrimitives.ReadUInt16LittleEndian(clear.Slice(format.CodePageOffset - b, 2)); + DatabaseKey = ReadDatabaseKey(buffer.Span, format); + DefaultCollationLcid = BinaryPrimitives.ReadUInt16LittleEndian(clear.Slice(format.CollationSortOrderOffset - b, 2)); + DefaultCollationSortId = clear[format.CollationSortIdOffset - b]; + DefaultCollationVersion = clear[format.CollationVersionOffset - b]; + CatalogRootPage = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(format.CatalogRootPointerOffset - b, 4)); + AcesRootPage = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(format.AcesRootPointerOffset - b, 4)); + QueriesRootPage = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(format.QueriesRootPointerOffset - b, 4)); + RelationshipsRootPage = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(format.RelationshipsRootPointerOffset - b, 4)); + AccountsRootPage = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(format.AccountsRootPointerOffset - b, 4)); + GroupsRootPage = BinaryPrimitives.ReadInt32LittleEndian(clear.Slice(format.GroupsRootPointerOffset - b, 4)); + FreePagesMap = ReadMapPointer(buffer.Span, format.FreePagesMapPointerOffset, format); + ReleasedPagesMap = ReadMapPointer(buffer.Span, format.ReleasedPagesMapPointerOffset, format); // An OLE Automation date, so it is decoded by the OA function rather than by hand: the two disagree // below the epoch, where OA keeps the time fraction positive (-1.25 is 1899-12-29 06:00, not // 1899-12-28 18:00). And the value comes straight off page 0, so a NaN, an infinity or anything past // DateTime.MaxValue is corruption in the very first thing an open does — reported as such, rather than // escaping as ArgumentOutOfRangeException from inside AddDays. - double days = BinaryPrimitives.ReadDoubleLittleEndian(clear.Slice(Formats.JetFormatBase.CreationDateOffset - b, 8)); - if (!double.IsFinite(days) || days <= MinOleAutomationDate || days >= MaxOleAutomationDate) - throw new InvalidDataException( - $"Page 0's creation date ({days}) is not a valid OLE Automation date."); - DatabaseCreationDate = DateTime.FromOADate(days); + double days = BinaryPrimitives.ReadDoubleLittleEndian(clear.Slice(format.CreationDateOffset - b, sizeof(double))); + DatabaseCreationDate = Storage.Types.JetTypeCodec.TryFromOaDate(days, out DateTime created) + ? created + : throw new InvalidDataException($"Page 0's creation date ({days}) is not a valid OLE Automation date."); + } + + /// Decodes one of page 0's global usage-map pointers: a record pointer under the header mask. + internal static (int Row, int Page) ReadMapPointer(ReadOnlySpan page, int offset, Formats.JetFormatBase format) + { + Span pointer = stackalloc byte[PageBuffer.RecordPointerSize]; + ReadMasked(page, offset, pointer, format); + return PageBuffer.ReadRecordPointer(pointer, 0); + } + + /// Reads 's length in bytes of the masked header from page offset + /// , removing the fixed header mask + /// (). + internal static void ReadMasked(ReadOnlySpan page0, int offset, Span clear, Formats.JetFormatBase format) + { + ReadOnlySpan mask = format.PageZeroHeaderMask[(offset - format.PageZeroHeaderMaskStart)..]; + for (int i = 0; i < clear.Length; i++) + clear[i] = (byte)(page0[offset + i] ^ mask[i]); + } + + /// Writes into the masked header at page offset + /// under the fixed header mask — the inverse of . + internal static void WriteMasked(Span page0, int offset, ReadOnlySpan clear, Formats.JetFormatBase format) + { + ReadOnlySpan mask = format.PageZeroHeaderMask[(offset - format.PageZeroHeaderMaskStart)..]; + for (int i = 0; i < clear.Length; i++) + page0[offset + i] = (byte)(clear[i] ^ mask[i]); + } + + /// The database (encryption) key at ; nonzero + /// when the pages are encrypted. + internal static int ReadDatabaseKey(ReadOnlySpan page0, Formats.JetFormatBase format) + { + Span key = stackalloc byte[sizeof(int)]; + ReadMasked(page0, format.DatabaseKeyOffset, key, format); + return BinaryPrimitives.ReadInt32LittleEndian(key); } - /// Decodes one of page 0's global usage-map pointers: a masked little-endian word whose low byte is - /// the record's row and whose upper three bytes are its page. - internal static (int Row, int Page) ReadMapPointer(ReadOnlySpan page, int offset) + /// The declared length of the EncryptionInfo descriptor at + /// , from the cleartext field at + /// ; 0 when there is none. Unbounded — it comes out of + /// the file, and each caller checks it against what it is about to touch. + internal static int ReadEncryptionInfoLength(ReadOnlySpan page0, Formats.JetFormatBase format) => + BinaryPrimitives.ReadUInt16LittleEndian(page0.Slice(format.EncryptionInfoLengthOffset, sizeof(ushort))); + + /// Writes the database key — the inverse of . + internal static void WriteDatabaseKey(Span page0, int key, Formats.JetFormatBase format) { - ReadOnlySpan mask = Formats.JetFormatBase.PageZeroHeaderMask; - int start = Formats.JetFormatBase.PageZeroHeaderMaskStart; - uint value = 0; - for (int i = 0; i < 4; i++) - value |= (uint)(page[offset + i] ^ mask[offset - start + i]) << (8 * i); - return ((int)(value & 0xFF), (int)(value >> 8)); + Span clear = stackalloc byte[sizeof(int)]; + BinaryPrimitives.WriteInt32LittleEndian(clear, key); + WriteMasked(page0, format.DatabaseKeyOffset, clear, format); } - /// XOR-de-obfuscates the page-0 header region into , whose length - /// equals the mask length; clear[i] corresponds to page offset - /// + i. - internal static void Demask(ReadOnlySpan page, Span clear) + /// Writes the password field: zero-padded to the field's size, XOR'd with its + /// creation-date mask (, so the creation date must already be in place), under + /// the header mask. An .mdb's value is its Jet password, UTF-16LE — empty when it has none; an + /// .accdb's is the low byte of its database key, repeated across the field (zero when unencrypted). + internal static void WritePassword(Span page0, ReadOnlySpan value, Formats.JetFormatBase format) { - ReadOnlySpan mask = Formats.JetFormatBase.PageZeroHeaderMask; - int start = Formats.JetFormatBase.PageZeroHeaderMaskStart; - for (int i = 0; i < mask.Length; i++) - clear[i] = (byte)(page[start + i] ^ mask[i]); + Span creationDate = stackalloc byte[sizeof(double)]; + ReadMasked(page0, format.CreationDateOffset, creationDate, format); + + Span field = stackalloc byte[format.PasswordSize]; + field.Clear(); + value.CopyTo(field); + XorPasswordDateMask(field, creationDate); + WriteMasked(page0, format.PasswordOffset, field, format); + } + + /// XORs the password field's own mask over , in either direction: the + /// integer part of the creation date (, the clear 8-byte double), as a + /// 4-byte little-endian value, cycled across the field. It lies under the header mask as well. + internal static void XorPasswordDateMask(Span field, ReadOnlySpan creationDate) + { + Span dateMask = stackalloc byte[sizeof(int)]; + BinaryPrimitives.WriteInt32LittleEndian(dateMask, (int)BinaryPrimitives.ReadDoubleLittleEndian(creationDate)); + for (int i = 0; i < field.Length; i++) + field[i] ^= dateMask[i % dateMask.Length]; + } + // The engine build LibRed stamps at page-0 EngineBuildOffset. A file keeps whatever its creator wrote, so this + // is the creator's choice, not part of the format: a Jet 4 .mdb gets Jet 4.0.9801, what the current msjet40.dll + // (x86 only) stamps on anything created through the Jet OLE DB 4.0 provider, and an ACCDB gets ACE 12.0.4518 + // (Office 2007 RTM), which every ACE version stamps. + private const int Jet4EngineBuild = 0x2649; + private const int AceEngineBuild = 0x11A6; + + /// The minor byte ACE writes when it CREATES a database of : 0x01 + /// for the 2010 format, 0x00 for every other. Like the build, the creator's choice rather than the + /// format's: a version raise writes 0x00 whatever the target, so a 2007 file raised to 2010 does not + /// carry the 0x01 a created one does. + private static byte CreatedMinorVersion(byte version) => + (byte)(version == (byte)Formats.JetVersion.Version14_2010 ? 0x01 : 0x00); + + /// + /// Synthesises page 0 (the database definition page) — the exact inverse of + /// . Byte-for-byte identical to a real empty file's + /// page 0 for the same parameters (verified against Access-created files). + /// + /// Format version byte (e.g. 0x02 = ACE 12 / Access 2007). + /// true for the ACCDB identifier, false for the MDB (Jet) identifier. + /// ANSI code page of the collation's language (1252 for en-US, 0 for a language + /// with none) — see . + /// The database's default collation — its LCID and sort-order version + /// (1033 / version 0 is General Legacy en-US). + /// Creation timestamp as an OLE-automation date (days since 1899-12-30), passed + /// as the raw double so the exact millisecond-precise bit pattern is preserved — the file's SIDs are masked + /// with a keystream folded from those bits and the rest of the header (page-00 §2.3). + /// The page holding the global usage maps: free pages in row 0, released pages in + /// row 1. + /// The MSysObjects TDEF page. + /// The MSysACEs TDEF page. + /// The MSysQueries TDEF page. + /// The MSysRelationships TDEF page. + /// The MSysAccounts TDEF page of a workgroup file; zero for a database, + /// which has no such table. + /// The MSysGroups TDEF page of a workgroup file; zero for a database. + public static byte[] Build( + byte version, bool isAccdb, int codePage, Collation collation, double creationDays, + int globalMapPage, int objectsPage, int acesPage, int queriesPage, int relationshipsPage, + int accountsPage, int groupsPage) + { + JetFormatBase format = JetFormatBase.FromVersionByte(version); + var page = new byte[format.PageSize]; + + // --- Pre-mask region (0x00..0x17, cleartext) --- + PageHeader.WriteType(page, PageType.DatabaseDefinition); + string id = isAccdb ? JetFormatBase.AceIdentifier : JetFormatBase.JetIdentifier; + Encoding.ASCII.GetBytes(id).CopyTo(page, JetFormatBase.FormatIdentifierOffset); // 0x04, 15 bytes; 0x13 stays NUL + format.WriteVersion(page, version, CreatedMinorVersion(version)); + + // --- Masked header (0x18..0x97): build the clear image, then XOR the fixed mask over it. --- + int b = format.PageZeroHeaderMaskStart; + Span clear = stackalloc byte[format.PageZeroHeaderMaskLength]; + + // [row][page] pointers to the global usage maps — free pages (row 0), released pages (row 1). + PageBuffer.WriteRecordPointer(clear, format.FreePagesMapPointerOffset - b, row: 0, globalMapPage); + PageBuffer.WriteRecordPointer(clear, format.ReleasedPagesMapPointerOffset - b, row: 1, globalMapPage); + // System-table bootstrap pointers = MSysObjects/ACEs/Queries/Relationships/Accounts/Groups pages. + BinaryPrimitives.WriteInt32LittleEndian(clear[(format.CatalogRootPointerOffset - b)..], objectsPage); + BinaryPrimitives.WriteInt32LittleEndian(clear[(format.AcesRootPointerOffset - b)..], acesPage); + BinaryPrimitives.WriteInt32LittleEndian(clear[(format.QueriesRootPointerOffset - b)..], queriesPage); + BinaryPrimitives.WriteInt32LittleEndian(clear[(format.RelationshipsRootPointerOffset - b)..], relationshipsPage); + BinaryPrimitives.WriteInt32LittleEndian(clear[(format.AccountsRootPointerOffset - b)..], accountsPage); + BinaryPrimitives.WriteInt32LittleEndian(clear[(format.GroupsRootPointerOffset - b)..], groupsPage); + BinaryPrimitives.WriteUInt16LittleEndian(clear[(format.CodePageOffset - b)..], (ushort)codePage); // 0x3C + // 0x3E database key = 0 (unencrypted); leave clear zero. + // 0x72..0x79 creation date (OLE double). + BinaryPrimitives.WriteDoubleLittleEndian(clear.Slice(format.CreationDateOffset - b, sizeof(double)), creationDays); + // The creating engine's build number. + BinaryPrimitives.WriteInt32LittleEndian(clear[(format.EngineBuildOffset - b)..], + format.IsAccdb ? AceEngineBuild : Jet4EngineBuild); + // 0x6E..0x71 collating sort order: LANGID, sort id at 0x70, version at 0x71 — a 32-bit LCID carrying + // the sort-order version in its unused top byte. Mirrors a column descriptor's 0x0B..0x0E. + BinaryPrimitives.WriteUInt16LittleEndian(clear[(format.CollationSortOrderOffset - b)..], (ushort)collation.Order); + clear[format.CollationSortIdOffset - b] = collation.SortId; + clear[format.CollationVersionOffset - b] = collation.Version; + + WriteMasked(page, b, clear, format); + // 0x42..0x69 password, empty: masked by its creation date as any password is, so even the empty one is + // not zero on disk. + WritePassword(page, [], format); + + // --- Post-mask tail (cleartext) --- + BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(format.PageZeroConstantOffset), format.PageZeroConstant); + Encoding.ASCII.GetBytes(JetFormatBase.Jet40EngineVersion).CopyTo(page, JetFormatBase.EngineVersionOffset); + + // User commit-byte table: 2 bytes per user to the end of the header page — Jet 3.x's 0x600 commit region + // relocated to the end of the 4 KB page (see the Jet locking white paper). Each pair is a per-user + // commit/lock status. A fresh file must seed every slot to the neutral idle value 00 01 — NOT 00 00, + // which Jet reads as "mid-write to disk"; with no matching .ldb user lock that reads as a + // suspect/corrupt database and forces a repair before Access will open it. + for (int i = format.CommitByteTableOffset; i < page.Length; i++) page[i] = (byte)(i & 1); + + return page; } - // The range DateTime.FromOADate accepts, checked before the call so a corrupt page 0 reports as corruption - // rather than as an argument error. Its own documented bounds: 0100-01-01 through 9999-12-31. - private const double MinOleAutomationDate = -657435.0; - private const double MaxOleAutomationDate = 2958466.0; } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Pages/Page.cs b/src/LibRed/LibRed.Core/Pages/Page.cs index 654c36fed..59b7f6dc3 100644 --- a/src/LibRed/LibRed.Core/Pages/Page.cs +++ b/src/LibRed/LibRed.Core/Pages/Page.cs @@ -16,5 +16,5 @@ public abstract class Page public abstract PageType Type { get; } /// Decodes this page's fields from the supplied buffer using version-specific offsets. - public abstract void Read(PageBuffer buffer, JetFormatBase format); + internal abstract void Read(PageBuffer buffer, JetFormatBase format); } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Pages/PageType.cs b/src/LibRed/LibRed.Core/Pages/PageType.cs index 4cd0a1e7a..0d3b85ae0 100644 --- a/src/LibRed/LibRed.Core/Pages/PageType.cs +++ b/src/LibRed/LibRed.Core/Pages/PageType.cs @@ -1,30 +1,48 @@ +using System.Buffers.Binary; + namespace LibRed.Pages; /// -/// The page-type marker stored in byte 0 of every page. Values match the on-disk -/// Jet/ACE encoding. +/// The page type: the little-endian 16-bit word at offset 0 of every page. Every type Jet/ACE writes has high +/// byte 0x01, page 0's included — the byte once read as a constant "flags" field is half of the type, and +/// ACE tests the whole word: it reads a table's owned page as rows only when the word is one of the row-bearing +/// types, and skips a page whose low byte is right but whose high byte is not (verified against ACE, every +/// combination of low byte 00–09 and high byte). See docs/format/README.md. /// -public enum PageType : byte +public enum PageType : ushort { - DatabaseDefinition = 0x00, - DataPage = 0x01, - TableDefinition = 0x02, - IntermediateIndexPage = 0x03, - LeafIndexPage = 0x04, - PageUsageBitmap = 0x05, + DatabaseDefinition = 0x0100, + DataPage = 0x0101, + TableDefinition = 0x0102, + IntermediateIndexPage = 0x0103, + LeafIndexPage = 0x0104, + PageUsageBitmap = 0x0105, /// A table-definition page that has been released by DROP TABLE. Access marks it by - /// setting this type byte and changing nothing else — the old definition stays on the page — and it + /// setting this type and changing nothing else — the old definition stays on the page — and it /// leaves the data, long-value and usage-map holder pages it frees at their original types. Measured: /// exactly one byte of the 4,096 differs across an ACE drop. Compact reclaims the page. /// See docs/format/page-08-released-tdef.md. - ReleasedTableDefinition = 0x08, + ReleasedTableDefinition = 0x0108, + + /// A data page released because its last live row was deleted. It covers both an ordinary + /// table's data page emptied by DELETE and a packed long-value page whose last tenant was deleted — + /// structurally the same event, and the same one-byte stamp. Every row slot is left a 0-length + /// deleted+overflow tombstone and the page is given back. A chained long value owns its pages + /// outright and they go back at instead. Nothing needs to handle it on read: + /// allocation selects on the free map, not on this type. + /// See docs/format/page-09-released-data.md. + ReleasedDataPage = 0x0109, +} + +/// Reads and writes a page's — the whole word at offset 0, never its low byte +/// alone. +public static class PageHeader +{ + /// The page's type, as stored. + public static PageType ReadType(ReadOnlySpan page) => (PageType)BinaryPrimitives.ReadUInt16LittleEndian(page); - /// A long-value page released because the last value sharing it was deleted. Several small - /// (single-page form) values pack onto one LVAL page; each delete retires its row to a 0-length - /// deleted+overflow tombstone, and when none are left the page is stamped with this type and freed. - /// A chained value owns its pages outright and they go back at instead, - /// which is why only the packed form produces this. Nothing needs to handle it on read: allocation - /// selects on the free map, not on this byte. See docs/format/page-09-released-long-value.md. - ReleasedLongValuePage = 0x09, -} \ No newline at end of file + /// Stamps at offset 0. + public static void WriteType(Span page, PageType type) => + BinaryPrimitives.WriteUInt16LittleEndian(page, (ushort)type); +} diff --git a/src/LibRed/LibRed.Core/Pages/TableDefinitionPage.cs b/src/LibRed/LibRed.Core/Pages/TableDefinitionPage.cs deleted file mode 100644 index 5743bfd1d..000000000 --- a/src/LibRed/LibRed.Core/Pages/TableDefinitionPage.cs +++ /dev/null @@ -1,421 +0,0 @@ -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.IO; -using System.Text; - -namespace LibRed.Pages; - -/// -/// A table definition (TDEF) page: row count, table type, and the column descriptors -/// and names. Verified against the Jet 4 / ACE layout. May be continued across pages -/// for wide tables (see ). -/// -public sealed class TableDefinitionPage : Page -{ - private const int MaxColumnsPerTable = 255; - private const int MaxIndexesPerTable = 32; - private const int MaxNameBytes = JetName.MaxLength * 2; - private static readonly Encoding StrictUnicode = new UnicodeEncoding( - bigEndian: false, byteOrderMark: false, throwOnInvalidBytes: true); - private readonly List _columns = []; - - public override PageType Type => PageType.TableDefinition; - - public int NextDefinitionPage { get; private set; } - public int RowCount { get; private set; } - - /// The complex-type AutoNumber high-water (header 0x1C) — the next id for a complex - /// (multi-value/attachment) column. Read and carried for faithful round-trip; 0 for every table without - /// such a column (LibRed neither creates nor consumes complex columns). - public int ComplexAutoNumber { get; private set; } - - public TableType TableType { get; private set; } - public int VariableColumnCount { get; private set; } - public int ColumnCount { get; private set; } - public int LogicalIndexCount { get; private set; } - public int IndexCount { get; private set; } - - public IReadOnlyList Columns => _columns; - - private readonly List _indexes = []; - public IReadOnlyList Indexes => _indexes; - - private readonly List _logicalIndexes = []; - /// Every logical index, in the order the TDEF lists them — including the relationship names that - /// share a real index with a named one, which keeps only one of. - public IReadOnlyList LogicalIndexes => _logicalIndexes; - - private readonly Dictionary _longValueOwnedMaps = []; - /// Per long-value (memo/OLE) column id → its owned-pages usage-map pointer (record row + - /// page), from the §3.3.2 list after the index names. Used to record a newly allocated LVAL page. - public IReadOnlyDictionary LongValueOwnedMaps => _longValueOwnedMaps; - - private readonly Dictionary _longValueFreeMaps = []; - /// Per long-value column id → its free-pages usage-map pointer (LVAL pages with spare room). - public IReadOnlyDictionary LongValueFreeMaps => _longValueFreeMaps; - - // Index structures (Jet 4 / ACE) follow the column names, in this order: - // IndexCount (0x33) data blocks : 52 bytes each — columns, flags, root page - // LogicalIndexCount (0x2F) info blocks : 28 bytes each — links a name to a data block - // LogicalIndexCount names : 2-byte length + UTF-16 - // A logical index may be a relationship (FK) sharing a data block with a real index. - // Index-block layout and flag values are shared with the writers via IndexBlockFormat / IndexFlags. - - /// - /// Reads a table definition starting at , transparently - /// stitching continuation pages (wide tables whose definition spans multiple pages) - /// into one contiguous buffer before parsing. - /// - public void Read(PageChannel channel, int page) - { - (PageBuffer buffer, _) = TdefChainReader.Read(channel, page); - Read(buffer, channel.Format); - } - - public override void Read(PageBuffer buffer, JetFormatBase format) - { - if (buffer.Length < format.TdefRealIndexBlockOffset) - throw new InvalidDataException( - $"TDEF buffer is {buffer.Length} bytes; the fixed header requires {format.TdefRealIndexBlockOffset}."); - int declaredLength = buffer.ReadInt32(format.TdefLengthOffset); - if (declaredLength < format.TdefRealIndexBlockOffset || declaredLength > buffer.Length) - throw new InvalidDataException( - $"TDEF declares length {declaredLength}, outside the available {buffer.Length}-byte buffer."); - if (declaredLength != buffer.Length) - buffer = new PageBuffer(buffer.Data[..declaredLength], buffer.PageNumber); - - PageNumber = buffer.PageNumber; - - NextDefinitionPage = buffer.ReadInt32(format.TdefNextPageOffset); - RowCount = buffer.ReadInt32(format.TdefRowCountOffset); - ComplexAutoNumber = buffer.ReadInt32(format.TdefComplexAutoNumberOffset); - TableType = (TableType)buffer.ReadByte(format.TdefTableTypeOffset); - VariableColumnCount = buffer.ReadUInt16(format.TdefVariableColumnsOffset); - ColumnCount = buffer.ReadUInt16(format.TdefColumnCountOffset); - LogicalIndexCount = buffer.ReadInt32(format.TdefLogicalIndexCountOffset); - IndexCount = buffer.ReadInt32(format.TdefIndexCountOffset); - - if (ColumnCount > MaxColumnsPerTable) - throw new InvalidDataException($"TDEF declares {ColumnCount} columns; Jet/ACE permits at most {MaxColumnsPerTable}."); - if (VariableColumnCount > MaxColumnsPerTable) - throw new InvalidDataException( - $"TDEF declares a variable-column high-water of {VariableColumnCount}; Jet/ACE permits at most {MaxColumnsPerTable}."); - if (IndexCount is < 0 or > MaxIndexesPerTable) - throw new InvalidDataException($"TDEF declares {IndexCount} real indexes; Jet/ACE permits 0 through {MaxIndexesPerTable}."); - // Capped at 32 exactly as IndexCount is, and this is the check that matters: a table gains a logical - // block per INCOMING relationship without gaining a data block, so it overruns here while 0x33 stays - // legal. Previously only the sign was checked, which let a file written past the limit read back as - // sound - the one shape where LibRed produces a database Access reports as an unrecognized format - // while seeing nothing wrong with it itself. - if (LogicalIndexCount is < 0 or > MaxIndexesPerTable) - throw new InvalidDataException( - $"TDEF declares {LogicalIndexCount} logical indexes; Jet/ACE permits 0 through {MaxIndexesPerTable}."); - - // The column descriptors follow a per-index block sized by the REAL index count at - // 0x33 (IndexCount) — NOT the logical count at 0x2F (LogicalIndexCount). The two are - // equal for MSysObjects but differ for user tables (e.g. logical=2, real=1). - // The buffer here may already be a stitched multi-page definition (see Read(channel, page)). - int columnBlock = CheckedRegionEnd( - format.TdefRealIndexBlockOffset, IndexCount, format.RealIndexEntrySize, buffer.Span.Length, "index statistics"); - _ = CheckedRegionEnd( - columnBlock, ColumnCount, format.ColumnDescriptorSize, buffer.Span.Length, "column descriptors"); - int afterNames = ReadColumns(buffer, format, columnBlock); - ReadIndexes(buffer, format, afterNames); - } - - /// - /// Parses the index structures following the column names into s - /// (one per index-data block): columns + sort order, unique/primary flags, root page, - /// and the index name (resolved from the logical-index info blocks). - /// - private void ReadIndexes(PageBuffer buffer, JetFormatBase format, int blockStart) - { - _indexes.Clear(); - var byColumnId = _columns.ToDictionary(c => c.ColumnId); - int infoStart = CheckedRegionEnd( - blockStart, IndexCount, IndexBlockFormat.DataBlockSize, buffer.Span.Length, "index-data blocks"); - _ = CheckedRegionEnd( - infoStart, LogicalIndexCount, IndexBlockFormat.InfoBlockSize, buffer.Span.Length, "logical-index blocks"); - - // 1. Index-data blocks (one IndexDef each): columns, unique flag, root page. - for (int i = 0; i < IndexCount; i++) - { - int block = blockStart + i * IndexBlockFormat.DataBlockSize; - - // Per-index statistics live in the 12-byte block at TdefRealIndexBlockOffset: - // [+0] total entries (= row count), [+4] unique entry count (cumulative, never - // decremented by Access), [+8] reserved. - int statsBlock = format.TdefRealIndexBlockOffset + i * format.RealIndexEntrySize; - int uniqueEntryCount = buffer.ReadInt32(statsBlock + 4); - - var columns = new List<(ColumnDef Column, bool Ascending)>(); - for (int slot = 0; slot < IndexBlockFormat.MaxColumns; slot++) - { - int entry = block + IndexBlockFormat.ColumnsOffset + slot * IndexBlockFormat.ColumnSlotSize; - short columnId = buffer.ReadInt16(entry); - if (columnId == IndexBlockFormat.ColumnUnused) continue; - if (byColumnId.TryGetValue(columnId, out ColumnDef? column)) - columns.Add((column, (buffer.ReadByte(entry + 2) & IndexBlockFormat.ColumnAscending) != 0)); - } - - _indexes.Add(new IndexDef - { - Name = string.Empty, - Columns = columns, - IsUnique = (buffer.ReadUInt16(block + IndexBlockFormat.FlagsOffset) & IndexFlags.Unique) != 0, - IgnoreNulls = (buffer.ReadUInt16(block + IndexBlockFormat.FlagsOffset) & IndexFlags.IgnoreNulls) != 0, - Required = (buffer.ReadUInt16(block + IndexBlockFormat.FlagsOffset) & IndexFlags.Required) != 0, - IsPrimaryKey = false, - UniqueEntryCount = uniqueEntryCount, - RootPage = buffer.ReadInt32(block + IndexBlockFormat.RootPageOffset), - RealIndexOrdinal = i, - }); - } - - int afterIndexNames = ResolveIndexNames(buffer, infoStart); - ReadLongValueMaps(buffer, afterIndexNames); - } - - /// Bounds one variable-length TDEF region against the assembled definition, in long so a - /// file-sourced count cannot overflow the multiply back into range. Shared with IndexWriter, which - /// walks the very same regions to reach an index-data block: an unchecked walk there lands the write on - /// the wrong block rather than throwing. - internal static int CheckedRegionEnd( - int start, int count, int itemSize, int bufferLength, string section) - { - long end = (long)start + (long)count * itemSize; - if (start < 0 || count < 0 || end < start || end > bufferLength) - throw new InvalidDataException( - $"TDEF {section} extend past the assembled definition ({start} + {count} * {itemSize} > {bufferLength})."); - return (int)end; - } - - /// Parses the §3.3.2 long-value column usage-map list (after the index names): one 10-byte - /// entry {col_num:2, used_ptr:4, free_ptr:4} per memo/OLE column, terminated by col_num 0xFFFF. Each - /// pointer is a 1-byte record row + 3-byte page. Captures the owned- (used-pages) map pointer. - private void ReadLongValueMaps(PageBuffer buffer, int pos) - { - _longValueOwnedMaps.Clear(); - _longValueFreeMaps.Clear(); - var seen = new HashSet(); - while (true) - { - EnsureAvailable(buffer, pos, 2, "long-value map terminator"); - int colNum = buffer.ReadUInt16(pos); - if (colNum == 0xFFFF) - { - pos += 2; - if (pos != buffer.Length) - throw new InvalidDataException( - $"TDEF has {buffer.Length - pos} trailing bytes after the long-value map terminator."); - return; - } - - EnsureAvailable(buffer, pos, 10, "long-value map entry"); - ColumnDef? column = _columns.FirstOrDefault(c => c.ColumnId == colNum); - if (column is null) - throw new InvalidDataException($"TDEF long-value map references unknown column id {colNum}."); - // A calculated column also gets a map, whatever its declared type: its cached result is stored - // as an envelope in the variable section and spills to an LVAL page when it outgrows the row. - // ACE writes one for a calculated Memo declared as Text, which this guard used to reject — - // and because the catalog loads every TDEF, that made the whole database unopenable. - if (column.Type is not (JetDataType.Memo or JetDataType.Ole) && !column.IsCalculated) - throw new InvalidDataException( - $"TDEF long-value map references non-long-value column '{column.Name}' ({column.Type})."); - if (!seen.Add(colNum)) - throw new InvalidDataException($"TDEF contains duplicate long-value map entries for column id {colNum}."); - - column.HasLongValueMap = true; - _longValueOwnedMaps[colNum] = (buffer.ReadByte(pos + 2), buffer.ReadInt24(pos + 3)); - _longValueFreeMaps[colNum] = (buffer.ReadByte(pos + 6), buffer.ReadInt24(pos + 7)); - pos += 10; - } - } - - /// - /// Reads the logical-index info blocks and their names, then attaches each name (and the - /// primary-key flag) to the index-data block it references. A data block may be referenced - /// by several logical indexes (e.g. a relationship plus the real index); the real index's - /// name wins over a foreign-key relationship's. - /// - private int ResolveIndexNames(PageBuffer buffer, int infoStart) - { - int logicalCount = LogicalIndexCount; // 0x2F — the logical-index (slot) count - var info = new (int DataNumber, bool IsRelationship, byte Type, byte FkType)[logicalCount]; - for (int i = 0; i < logicalCount; i++) - { - int block = infoStart + i * IndexBlockFormat.InfoBlockSize; - info[i] = ( - buffer.ReadInt32(block + IndexBlockFormat.InfoDataNumberOffset), - buffer.ReadInt32(block + IndexBlockFormat.InfoFkTablePageOffset) != 0, - buffer.ReadByte(block + IndexBlockFormat.InfoTypeOffset), - buffer.ReadByte(block + IndexBlockFormat.InfoFkTypeOffset)); - } - - int namePos = infoStart + logicalCount * IndexBlockFormat.InfoBlockSize; - var priority = new int[_indexes.Count]; - _logicalIndexes.Clear(); - for (int i = 0; i < logicalCount; i++) - { - (string name, namePos) = ReadName(buffer, namePos, $"logical index {i}"); - - (int dataNumber, bool isRelationship, byte type, byte fkType) = info[i]; - if (dataNumber < 0 || dataNumber >= _indexes.Count) continue; - - _logicalIndexes.Add(new LogicalIndexDef(name, dataNumber, isRelationship, - !isRelationship && type == IndexBlockFormat.TypePrimary, fkType)); - - // Prefer a real index name over a relationship's; prefer the primary among real ones. - int p = isRelationship ? 1 : type == IndexBlockFormat.TypePrimary ? 3 : 2; - if (p > priority[dataNumber]) - { - priority[dataNumber] = p; - _indexes[dataNumber] = _indexes[dataNumber] with - { - Name = name, - IsPrimaryKey = !isRelationship && type == IndexBlockFormat.TypePrimary, - }; - } - } - - return namePos; - } - - private int ReadColumns(PageBuffer buffer, JetFormatBase format, int columnBlock) - { - _columns.Clear(); - - // Pass 1: fixed-size column descriptors. - var descriptors = new (JetDataType Type, int ColumnId, byte Flags, byte ExtFlags, int FixedOffset, int Length, byte Precision, byte Scale, int VariableIndex, Collation Collation)[ColumnCount]; - var columnIds = new HashSet(); - for (int i = 0; i < ColumnCount; i++) - { - int entry = columnBlock + i * format.ColumnDescriptorSize; - var type = (JetDataType)buffer.ReadByte(entry + format.ColumnTypeOffset); - int columnId = buffer.ReadUInt16(entry + format.ColumnNumberOffset); - if (!Enum.IsDefined(type)) - throw new InvalidDataException($"TDEF column {i} has unknown type code 0x{(byte)type:X2}."); - if (columnId >= MaxColumnsPerTable) - throw new InvalidDataException( - $"TDEF column {i} has id {columnId}; valid ids are 0 through {MaxColumnsPerTable - 1}."); - if (!columnIds.Add(columnId)) - throw new InvalidDataException($"TDEF contains duplicate column id {columnId}."); - - // Bytes 0x0B/0x0C are precision/scale for a Decimal/Numeric column and the text-collation LCID - // for everything else; 0x0D is the collation's sort-order version. Read whichever applies. - bool numeric = type == JetDataType.FixedPoint; - descriptors[i] = ( - type, - columnId, - buffer.ReadByte(entry + format.ColumnFlagsOffset), - buffer.ReadByte(entry + format.ColumnExtendedFlagsOffset), - buffer.ReadUInt16(entry + format.ColumnFixedOffsetOffset), - buffer.ReadUInt16(entry + format.ColumnLengthOffset), - numeric ? buffer.ReadByte(entry + format.ColumnPrecisionOffset) : (byte)0, - numeric ? buffer.ReadByte(entry + format.ColumnScaleOffset) : (byte)0, - // The variable-table index is **stored** in the descriptor (0x07), not derived. Reading it - // (rather than ranking column ids) is what lets a table with a **dropped column** decode: - // ACE's DROP COLUMN removes a descriptor but does NOT renumber the survivors or rewrite - // rows, so a survivor keeps its original variable index even though ranking would shift it. - buffer.ReadUInt16(entry + format.ColumnVariableIndexOffset), - numeric ? Collation.GeneralLegacy - // 0x0B..0x0E are one 32-bit LCID with the sort-order version in the top byte: LANGID, - // then the sort id at 0x0D (non-zero only for a Windows alternate sort order, e.g. - // Hungarian Technical), then the version at 0x0E (0 = legacy table, 1 = Access-2010). - : new Collation((CollatingOrder)buffer.ReadUInt16(entry + format.ColumnLocaleOffset), - buffer.ReadByte(entry + format.ColumnCollationVersionOffset), - buffer.ReadByte(entry + format.ColumnCollationSortIdOffset))); - } - - // Pass 2: column names, in the same order, immediately after the descriptor block. - // Each name is a 2-byte (little-endian) byte length followed by UTF-16LE text. - int namePos = columnBlock + ColumnCount * format.ColumnDescriptorSize; - for (int i = 0; i < ColumnCount; i++) - { - (string name, namePos) = ReadName(buffer, namePos, $"column {i}"); - - var d = descriptors[i]; - bool isFixed = (d.Flags & JetFormatBase.ColumnFlagFixedLength) != 0; - if ((!isFixed && d.VariableIndex >= VariableColumnCount) - || (isFixed && d.VariableIndex > VariableColumnCount)) - throw new InvalidDataException( - $"TDEF column '{name}' has variable-table index {d.VariableIndex}, " + - $"outside high-water {VariableColumnCount}."); - _columns.Add(new ColumnDef - { - Name = name, - Type = d.Type, - Index = i, - ColumnId = d.ColumnId, - Length = d.Length, - FixedOffset = d.FixedOffset, - VariableIndex = isFixed ? -1 : d.VariableIndex, - // Byte 7 is stored on fixed columns too (the running count of preceding variable columns); keep - // it so a faithful rebuild re-emits the exact value instead of clobbering fixed columns to 0. - VariableTableIndex = d.VariableIndex, - IsFixedLength = isFixed, - IsAutoNumber = (d.Flags & JetFormatBase.ColumnFlagAutoNumber) != 0, - // Every documented flag bit is modelled (0x0F: updatable/GUID-autonumber/hyperlink; 0x10: - // compressed-Unicode / calculated) so it round-trips explicitly, not via RawDescriptor. - IsUpdatable = (d.Flags & JetFormatBase.ColumnFlagUpdatable) != 0, - IsGuidAutoNumber = (d.Flags & JetFormatBase.ColumnFlagGuidAutoNumber) != 0, - IsHyperlink = (d.Flags & JetFormatBase.ColumnFlagHyperlink) != 0, - SupportsCompressedUnicode = (d.ExtFlags & JetFormatBase.ColumnExtFlagCompressedUnicode) != 0, - IsCalculated = (d.ExtFlags & JetFormatBase.ColumnExtFlagCalculated) != 0, - Precision = d.Precision, - Scale = d.Scale, - Collation = d.Collation, - // Keep the original 25 bytes so a rewrite preserves fields we don't model (faithful round-trip). - RawDescriptor = buffer.Slice(columnBlock + i * format.ColumnDescriptorSize, format.ColumnDescriptorSize).ToArray(), - }); - } - - // AutoNumber seed/increment from the TDEF header: 0x18 = increment, 0x14 = last-assigned value. On a - // freshly created table the last value is Seed-Increment, so Seed = last + increment (matching what a - // no-insert scaffold reports). - // - // At most one column draws on THAT pair — but it is not the only counter a table has. A complex column - // carries the very same 0x04 flag and is allocated from 0x1C (ComplexAutoNumber), so a table can hold an - // ordinary counter and any number of complex columns all reading IsAutoNumber (complex1.accdb's Table1 - // has five). Applying the header pair to those too reports a seed and increment that describe a - // different counter, so skip them: their high-water is the table's ComplexAutoNumber. - int increment = buffer.ReadInt32(format.TdefAutoNumberIncrementOffset); - if (increment == 0) increment = 1; - int lastAuto = buffer.ReadInt32(format.TdefLastAutoNumberOffset); - foreach (ColumnDef column in _columns) - if (column.IsAutoNumber && column.Type != JetDataType.Complex) - { - column.Increment = increment; - column.Seed = lastAuto + increment; - } - - return namePos; - } - - private static (string Name, int Next) ReadName(PageBuffer buffer, int pos, string kind) - { - EnsureAvailable(buffer, pos, 2, $"{kind} name length"); - int byteLength = buffer.ReadUInt16(pos); - pos += 2; - if (byteLength == 0 || byteLength > MaxNameBytes || (byteLength & 1) != 0) - throw new InvalidDataException( - $"TDEF {kind} name has invalid UTF-16 byte length {byteLength}; expected an even value from 2 through {MaxNameBytes}."); - EnsureAvailable(buffer, pos, byteLength, $"{kind} name"); - try - { - return (StrictUnicode.GetString(buffer.Slice(pos, byteLength)), pos + byteLength); - } - catch (DecoderFallbackException ex) - { - throw new InvalidDataException($"TDEF {kind} name is not valid UTF-16LE.", ex); - } - } - - private static void EnsureAvailable(PageBuffer buffer, int pos, int length, string section) - { - long end = (long)pos + length; - if (pos < 0 || length < 0 || end > buffer.Length) - throw new InvalidDataException( - $"TDEF {section} extends past the declared definition ({pos} + {length} > {buffer.Length})."); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Pages/TdefChainReader.cs b/src/LibRed/LibRed.Core/Pages/TdefChainReader.cs deleted file mode 100644 index 3e089d844..000000000 --- a/src/LibRed/LibRed.Core/Pages/TdefChainReader.cs +++ /dev/null @@ -1,90 +0,0 @@ -using LibRed.Formats; -using LibRed.IO; - -namespace LibRed.Pages; - -/// Reads one TDEF chain into its absolute logical coordinate space. -internal static class TdefChainReader -{ - // A valid Jet/ACE table is constrained to 255 columns, 32 real indexes, and 64-character names. - // One MiB is deliberately generous while still preventing a hostile 32-bit length from driving - // process-scale allocation. This is an implementation safety budget, not an on-disk field width. - internal const int MaxDefinitionLength = 1024 * 1024; - - internal static (PageBuffer Buffer, IReadOnlyList ContinuationPages) Read( - PageChannel channel, int firstPage) - { - JetFormatBase format = channel.Format; - ValidatePageNumber(channel, firstPage, "TDEF root"); - PageBuffer first = channel.ReadPage(firstPage); - if (first.ReadByte(0) != (byte)PageType.TableDefinition || first.ReadByte(1) != 0x01) - throw new InvalidDataException( - $"TDEF root page {firstPage} has header " + - $"[{first.ReadByte(0):X2} {first.ReadByte(1):X2}], expected [02 01]."); - - int definitionLength = first.ReadInt32(format.TdefLengthOffset); - if (definitionLength < format.TdefRealIndexBlockOffset || definitionLength > MaxDefinitionLength) - throw new InvalidDataException( - $"TDEF page {firstPage} declares length {definitionLength}; supported validated range is " + - $"{format.TdefRealIndexBlockOffset} through {MaxDefinitionLength} bytes."); - - // The chain holds the definition AND its 8-byte trailing reserve, which follows the last definition byte - // and spills onto a page of its own when it does not fit — so a continuation page can carry no definition - // bytes at all (verified vs ACE: a 4,090-byte definition has a continuation holding two reserve bytes). - int pageSize = format.PageSize; - int bodySize = pageSize - JetFormatBase.TdefContinuationHeaderSize; - int stored = definitionLength + JetFormatBase.TdefContinuationHeaderSize; - int continuationCount = stored <= pageSize - ? 0 - : (stored - pageSize + bodySize - 1) / bodySize; - - int next = first.ReadInt32(format.TdefNextPageOffset); - var continuationPages = new List(continuationCount); - var continuationBuffers = new List(continuationCount); - var visited = new HashSet { firstPage }; - - for (int i = 0; i < continuationCount; i++) - { - if (next == 0) - throw new InvalidDataException( - $"TDEF page {firstPage} ends after {i} continuation pages but its declared length requires {continuationCount}."); - ValidatePageNumber(channel, next, "TDEF continuation"); - if (!visited.Add(next)) - throw new InvalidDataException($"TDEF page {firstPage} contains a continuation cycle at page {next}."); - - PageBuffer continuation = channel.ReadPage(next); - if (continuation.ReadByte(0) != (byte)PageType.TableDefinition || continuation.ReadByte(1) != 0x01) - throw new InvalidDataException( - $"TDEF continuation page {next} has header " + - $"[{continuation.ReadByte(0):X2} {continuation.ReadByte(1):X2}], expected [02 01]."); - - continuationPages.Add(next); - continuationBuffers.Add(continuation); - next = continuation.ReadInt32(format.TdefNextPageOffset); - } - - if (next != 0) - throw new InvalidDataException( - $"TDEF page {firstPage} has more continuation pages than its declared length permits."); - - var assembled = new byte[definitionLength]; - int written = Math.Min(pageSize, definitionLength); - first.Span[..written].CopyTo(assembled); - foreach (PageBuffer continuation in continuationBuffers) - { - int take = Math.Min(bodySize, definitionLength - written); - continuation.Span.Slice(JetFormatBase.TdefContinuationHeaderSize, take) - .CopyTo(assembled.AsSpan(written)); - written += take; - } - - return (new PageBuffer(assembled, firstPage), continuationPages); - } - - private static void ValidatePageNumber(PageChannel channel, int pageNumber, string role) - { - if (pageNumber <= 0 || pageNumber >= channel.PageCount) - throw new InvalidDataException( - $"{role} page {pageNumber} is outside the database's {channel.PageCount} pages."); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChinesePronunciation.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChinesePronunciation.bin new file mode 100644 index 000000000..c6d191fa6 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChinesePronunciation.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChinesePronunciationLegacy.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChinesePronunciationLegacy.bin new file mode 100644 index 000000000..2232f7d9e Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChinesePronunciationLegacy.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChineseStrokeCount.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseStrokeCount.bin new file mode 100644 index 000000000..b9b5992a0 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseStrokeCount.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChineseStrokeCountLegacy.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseStrokeCountLegacy.bin new file mode 100644 index 000000000..64b44bf34 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseStrokeCountLegacy.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradBopomofo.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradBopomofo.bin new file mode 100644 index 000000000..1b508c1e1 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradBopomofo.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradBopomofoLegacy.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradBopomofoLegacy.bin new file mode 100644 index 000000000..7cedaa329 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradBopomofoLegacy.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradStrokeCount.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradStrokeCount.bin new file mode 100644 index 000000000..844333630 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradStrokeCount.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradStrokeCountLegacy.bin b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradStrokeCountLegacy.bin new file mode 100644 index 000000000..59c01b301 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/ChineseTradStrokeCountLegacy.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/Japanese.bin b/src/LibRed/LibRed.Core/Resources/Cjk/Japanese.bin new file mode 100644 index 000000000..7a0399f7d Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/Japanese.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/JapaneseLegacy.bin b/src/LibRed/LibRed.Core/Resources/Cjk/JapaneseLegacy.bin new file mode 100644 index 000000000..3d8b71884 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/JapaneseLegacy.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/JapaneseRadicalStrokeCount.bin b/src/LibRed/LibRed.Core/Resources/Cjk/JapaneseRadicalStrokeCount.bin new file mode 100644 index 000000000..cc01d1ea7 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/JapaneseRadicalStrokeCount.bin differ diff --git a/src/LibRed/LibRed.Core/Resources/Cjk/Korean.bin b/src/LibRed/LibRed.Core/Resources/Cjk/Korean.bin new file mode 100644 index 000000000..25cca70a9 Binary files /dev/null and b/src/LibRed/LibRed.Core/Resources/Cjk/Korean.bin differ diff --git a/src/LibRed/LibRed.Core/Storage/BitmapBits.cs b/src/LibRed/LibRed.Core/Storage/BitmapBits.cs new file mode 100644 index 000000000..f37ce2933 --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/BitmapBits.cs @@ -0,0 +1,79 @@ +using System.Numerics; + +namespace LibRed.Storage; + +/// +/// The bits of the format's bitmaps, which all number their bits the same way — bit i of byte n is +/// bit n×8 + i, least significant first: a usage map's (bit k marking page basePage + k, +/// page-05 §9), a row's null bitmap (bit k for column id k, page-01 §5) and an index page's +/// entry-position mask (bit k ending an entry at offset k, page-03-04 §10.2). Every read and write of +/// a bit in any of them comes through here; the caller passes the bitmap without whatever header carried it. +/// +internal static class BitmapBits +{ + /// The bytes a bitmap of bits takes. + public static int ByteCount(int bits) => (bits + 7) >> 3; + + /// Whether is set. + public static bool Get(ReadOnlySpan bitmap, int bit) => (bitmap[bit >> 3] & (1 << (bit & 7))) != 0; + + /// Sets or clears . + public static void Set(Span bitmap, int bit, bool value) + { + byte mask = (byte)(1 << (bit & 7)); + if (value) bitmap[bit >> 3] |= mask; + else bitmap[bit >> 3] &= (byte)~mask; + } + + /// The lowest set bit at or above , or -1 when there is none. + public static int NextSetBit(ReadOnlySpan bitmap, int from) + { + for (int i = from >> 3; i < bitmap.Length; i++) + { + int bits = i == from >> 3 ? bitmap[i] & (0xFF << (from & 7)) : bitmap[i]; + if (bits != 0) return i * 8 + BitOperations.TrailingZeroCount(bits); + } + return -1; + } + + /// The highest set bit, or -1 when there is none. + public static int LastSetBit(ReadOnlySpan bitmap) + { + for (int i = bitmap.Length - 1; i >= 0; i--) + if (bitmap[i] != 0) return i * 8 + 31 - BitOperations.LeadingZeroCount((uint)bitmap[i]); + return -1; + } + + /// + /// Appends the pages a usage-map marks to . + /// + /// Collects the page numbers, in ascending order. + /// The map's bits, without whatever header carried them. + /// The page bit 0 stands for. + /// + /// The file's page count, to reject a bit naming a page outside it — or null to keep every bit. + /// Which of those is right is a property of the caller, not an oversight. A map read feeds + /// its numbers straight into page reads, so one outside the file is corruption and says so. A map being + /// rewritten must keep bits it cannot currently represent: a free map's window slides with the + /// append tail and can sit above a page it still has to record, so the record widens rather than dropping + /// them (page-05 §9). Rejecting there is not theoretical — it failed an ordinary DROP TABLE on + /// complex1.accdb, whose MSysObjects free map starts at page 2288 while its catalog rows + /// live at page 17. + /// + /// Names the map in the rejection message. + public static void AppendPages( + List pages, ReadOnlySpan bitmap, int basePage, int? rejectBeyond, string what) + { + // The bound is read once by the caller and passed in: this loop runs per set bit on every insert, and + // PageChannel.PageCount used to be a file-length syscall, which made a per-bit test cost a + // non-transactional insert ~1.9x. Nothing here writes, so the count cannot move under it. + for (int bit = NextSetBit(bitmap, 0); bit >= 0; bit = NextSetBit(bitmap, bit + 1)) + { + long page = (long)basePage + bit; + if (rejectBeyond is { } pageCount && (page <= 1 || page >= pageCount)) + throw new InvalidDataException( + $"{what} names page {page}, outside the file's 2..{pageCount - 1} range."); + pages.Add((int)page); + } + } +} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedEvaluator.cs b/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedEvaluator.cs index cfc181bad..078ca3b15 100644 --- a/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedEvaluator.cs +++ b/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedEvaluator.cs @@ -121,8 +121,8 @@ internal static class CalculatedEvaluator // itself, so refusing is matching it, not falling short of it. if (name.EndsWith('$')) throw new CalculatedExpressionException( - $"'{name}' cannot be evaluated in a calculated column. Access offers the '$' name variants and " - + $"accepts them at design time, but ACE fails every insert into such a table; use '{name[..^1]}'."); + $"'{name}' cannot be evaluated in a calculated column: the '$' variants are not supported there; " + + $"use '{name[..^1]}'."); // IIf and Choose SHORT-CIRCUIT: only the selected branch is evaluated, matching both the Access // expression service and LibRed.Engine's own evaluator. (Access's VBA-side IIf famously does NOT @@ -175,9 +175,8 @@ internal static class CalculatedEvaluator // needed to get rid of it (page-02e-calculated-columns). if (args.Length > 0 && args[0] is null && name.Equals("CDbl", StringComparison.OrdinalIgnoreCase)) throw new CalculatedExpressionException( - "CDbl cannot convert Null — the VBA conversions raise on Null rather than propagating it, so " - + "this row's cached value would be an error ACE could not read back. Give the column a " - + "value, or guard the expression, e.g. IIf(IsNull([x]), 0, CDbl([x]))."); + "CDbl cannot convert Null in a calculated column. Give the column a value, or guard the " + + "expression, e.g. IIf(IsNull([x]), 0, CDbl([x]))."); // Everything else propagates Null through any argument. if (args.Any(a => a is null)) return null; diff --git a/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedExpression.cs b/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedExpression.cs index 1783273f4..1cdeab357 100644 --- a/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedExpression.cs +++ b/src/LibRed/LibRed.Core/Storage/Calculated/CalculatedExpression.cs @@ -99,8 +99,8 @@ void Check(CalcNode node) // insert into the table -- so a column using one can never be populated (§3.4a). case CalcCall f when f.Name.EndsWith('$'): throw new CalculatedExpressionException( - $"The expression {f.Name} cannot be used in a calculated column. Access accepts the '$' " - + $"name variants at design time but cannot populate such a column; use '{f.Name[..^1]}'."); + $"The expression {f.Name} cannot be used in a calculated column: the '$' variants are not " + + $"supported there; use '{f.Name[..^1]}'."); case CalcCall f when !Accepted.Contains(f.Name): throw new CalculatedExpressionException( $"The expression {f.Name} cannot be used in a calculated column."); diff --git a/src/LibRed/LibRed.Core/Storage/CalculatedValue.cs b/src/LibRed/LibRed.Core/Storage/CalculatedValue.cs index 0de1a0c49..eccaf045a 100644 --- a/src/LibRed/LibRed.Core/Storage/CalculatedValue.cs +++ b/src/LibRed/LibRed.Core/Storage/CalculatedValue.cs @@ -1,4 +1,5 @@ using LibRed.Catalog; +using LibRed.Formats; using System.Buffers.Binary; namespace LibRed.Storage; @@ -21,22 +22,23 @@ namespace LibRed.Storage; /// internal static class CalculatedValue { - /// Header bytes before the length: a 4-byte status followed by 12 reserved bytes that are zero - /// in every file measured. - private const int HeaderSize = 16; + /// The status field, at the envelope's start. Zero is the success code; anything else is a VBA + /// runtime error number, and the expression produced no value at all. + internal const int StatusSize = 4; - /// The status field's size. Zero is the success code; anything else is a VBA runtime error - /// number, and the expression produced no value at all. - private const int StatusSize = 4; + /// Header bytes before the length: the status, then 12 reserved bytes that are zero in every file + /// measured. + internal const int HeaderSize = 16; - /// Little-endian payload length. - private const int LengthSize = 4; + /// The little-endian payload length, at ; the payload follows it. + internal const int LengthSize = 4; /// Zero padding after the payload. - private const int PaddingSize = 3; + internal const int PaddingSize = 3; /// The smallest envelope: header, length and padding with an empty payload. - public const int MinimumSize = HeaderSize + LengthSize + PaddingSize; + internal const int MinimumSize = HeaderSize + LengthSize + PaddingSize; + /// Evaluates a calculated column's expression against one row and coerces the result to the /// type its payload is stored in. @@ -83,10 +85,10 @@ public static IReadOnlySet ReferencedIndexes(ColumnDef column, IReadOnlyLis /// Builds the envelope for — the inverse of /// . A Null result is a zero-length payload, not an absent one: the column's /// null-bitmap bit stays set either way (§3.4a). - public static byte[] Encode(ColumnDef column, object? value) + public static byte[] Encode(ColumnDef column, object? value, JetFormatBase format) { JetDataType type = StoredType(column, 0); - byte[] payload = value is null ? [] : EncodePayload(column, type, value); + byte[] payload = value is null ? [] : EncodePayload(column, type, value, format); var envelope = new byte[MinimumSize + payload.Length]; BinaryPrimitives.WriteUInt32LittleEndian(envelope.AsSpan(HeaderSize, LengthSize), (uint)payload.Length); @@ -110,7 +112,7 @@ public static byte[] Encode(ColumnDef column, object? value) /// envelope back under the 64-byte inline limit and so decides whether the value reaches a page at all. /// /// - private static byte[] EncodePayload(ColumnDef column, JetDataType type, object value) + private static byte[] EncodePayload(ColumnDef column, JetDataType type, object value, JetFormatBase format) { var culture = System.Globalization.CultureInfo.InvariantCulture; switch (type) @@ -125,7 +127,7 @@ private static byte[] EncodePayload(ColumnDef column, JetDataType type, object v return Types.JetTypeCodec.TryCompressText(column, text, requireCapableFlag: false) ?? utf16; } default: - return Types.JetTypeCodec.Encode(column, type, value); + return Types.JetTypeCodec.Encode(column, type, value, format); } } diff --git a/src/LibRed/LibRed.Core/Storage/CopiedPrimary.cs b/src/LibRed/LibRed.Core/Storage/CopiedPrimary.cs new file mode 100644 index 000000000..fc9863132 --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/CopiedPrimary.cs @@ -0,0 +1,56 @@ +namespace LibRed.Storage; + +/// +/// The primary weight a character contributed, kept in case the next character copies it — a shadda doubling +/// it, or an iteration mark repeating it. Both encoders record one for every character, and almost every weight is +/// one or two bytes, so those are held inline and recording allocates nothing; a longer weight keeps an array. +/// +internal readonly struct CopiedPrimary +{ + private readonly byte[]? _bytes; + private readonly byte _first, _second; + + /// A weight's bytes, which the caller does not change afterwards. + public CopiedPrimary(byte[] bytes) + { + _bytes = bytes; + Length = bytes.Length; + } + + public CopiedPrimary(byte only) + { + _first = only; + Length = 1; + } + + public CopiedPrimary(byte first, byte second) + { + (_first, _second) = (first, second); + Length = 2; + } + + /// A weight as it stands, copied only when it is longer than two bytes. + public static CopiedPrimary Of(ReadOnlySpan weight) => weight.Length switch + { + 1 => new CopiedPrimary(weight[0]), + 2 => new CopiedPrimary(weight[0], weight[1]), + _ => new CopiedPrimary(weight.ToArray()), + }; + + /// How many bytes the weight has; zero when nothing was contributed. + public int Length { get; } + + public void AppendTo(List primaries) + { + if (_bytes is not null) + { + primaries.AddRange(_bytes); + return; + } + if (Length > 0) primaries.Add(_first); + if (Length > 1) primaries.Add(_second); + } + + /// The bytes as an array — for the copying itself, which is rare, rather than the recording. + public byte[] ToArray() => _bytes ?? (Length switch { 0 => [], 1 => [_first], _ => [_first, _second] }); +} diff --git a/src/LibRed/LibRed.Core/Storage/DatabaseCreator.cs b/src/LibRed/LibRed.Core/Storage/DatabaseCreator.cs deleted file mode 100644 index 1515e39aa..000000000 --- a/src/LibRed/LibRed.Core/Storage/DatabaseCreator.cs +++ /dev/null @@ -1,531 +0,0 @@ -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.Pages; -using System.Buffers.Binary; -using System.Globalization; -using System.Text; - -namespace LibRed.Storage; - -/// -/// Builds the pages of a brand-new, empty Jet/ACE database from scratch — the native, cross-platform -/// replacement for the DAO/ADOX file creator. Currently synthesises page 0 (the database definition -/// page); the system catalog follows. -/// -public static class DatabaseCreator -{ - private static readonly DateTime OleEpoch = new(1899, 12, 30); - - /// - /// Synthesises page 0 (the database definition page) — the exact inverse of - /// . Byte-for-byte identical to a real empty file's - /// page 0 for the same parameters (verified against Access-created files). - /// - /// Format version byte (e.g. 0x02 = ACE 12 / Access 2007). - /// true for the ACCDB identifier, false for the MDB (Jet) identifier. - /// ANSI code page (1252 for en-US). - /// The database's default collation — its LCID and sort-order version - /// (1033 / version 0 is General Legacy en-US). - /// Creation timestamp as an OLE-automation date (days since 1899-12-30), passed - /// as the raw double so the exact millisecond-precise bit pattern is preserved — the page-0 SID mask is bound - /// to those exact bits (see ). - public static byte[] BuildDefinitionPage( - byte version, bool isAccdb, int codePage, Collation collation, double creationDays) - { - var page = new byte[4096]; - - // --- Pre-mask region (0x00..0x17, cleartext) --- - page[0x00] = 0x00; // page type - page[0x01] = 0x01; // observed constant 01 00 00 - string id = isAccdb ? JetFormatBase.AceIdentifier : JetFormatBase.JetIdentifier; - Encoding.ASCII.GetBytes(id).CopyTo(page, JetFormatBase.FormatIdentifierOffset); // 0x04, 15 bytes; 0x13 stays NUL - page[JetFormatBase.VersionOffset] = version; // 0x14 - page[JetFormatBase.MinorVersionOffset] = JetFormatBase.CreatedMinorVersion(version); // 0x15 - - // --- Masked header (0x18..0x97): build the clear image, then XOR the fixed mask over it. --- - int b = JetFormatBase.PageZeroHeaderMaskStart; - Span clear = stackalloc byte[JetFormatBase.PageZeroHeaderMask.Length]; - - // 0x18/0x1C: [row][page] pointers to the global usage maps on page 1 — free pages (row 0), released pages (row 1). - BinaryPrimitives.WriteInt32LittleEndian(clear[(JetFormatBase.FreePagesMapPointerOffset - b)..], 0x00000100); // free map: page 1, row 0 - BinaryPrimitives.WriteInt32LittleEndian(clear[(JetFormatBase.ReleasedPagesMapPointerOffset - b)..], 0x00000101); // released map: page 1, row 1 - // 0x20..0x2C: system-catalog bootstrap pointers = MSysObjects/ACEs/Queries/Relationships pages. - BinaryPrimitives.WriteInt32LittleEndian(clear[(0x20 - b)..], 2); - BinaryPrimitives.WriteInt32LittleEndian(clear[(0x24 - b)..], 3); - BinaryPrimitives.WriteInt32LittleEndian(clear[(0x28 - b)..], 4); - BinaryPrimitives.WriteInt32LittleEndian(clear[(0x2C - b)..], 5); - BinaryPrimitives.WriteUInt16LittleEndian(clear[(JetFormatBase.CodePageOffset - b)..], (ushort)codePage); // 0x3C - // 0x3E database key = 0 (unencrypted); leave clear zero. - // 0x42..0x69 password (empty): on disk the field is additionally masked by a 4-byte value derived - // from the creation date, so the clear image (pre-header-mask) of an empty password is that value - // repeated — reproduce it exactly. - double days = creationDays; - Span dateMask = stackalloc byte[4]; - BinaryPrimitives.WriteInt32LittleEndian(dateMask, (int)days); - for (int i = 0; i < 40; i++) - clear[JetFormatBase.PasswordOffset - b + i] = dateMask[i % 4]; - // 0x6A fixed sentinel constant. - BinaryPrimitives.WriteInt32LittleEndian(clear[(0x6A - b)..], 0x000011A6); - // 0x6E..0x71 collating sort order: LANGID, sort id at 0x70, version at 0x71 — a 32-bit LCID carrying - // the sort-order version in its unused top byte. Mirrors a column descriptor's 0x0B..0x0E. - BinaryPrimitives.WriteUInt16LittleEndian(clear[(JetFormatBase.CollationSortOrderOffset - b)..], (ushort)collation.Order); - clear[JetFormatBase.CollationSortIdOffset - b] = collation.SortId; - clear[JetFormatBase.CollationVersionOffset - b] = collation.Version; - // 0x72..0x79 creation date (OLE double). - BinaryPrimitives.WriteDoubleLittleEndian(clear[(JetFormatBase.CreationDateOffset - b)..], days); - - ReadOnlySpan mask = JetFormatBase.PageZeroHeaderMask; - for (int i = 0; i < mask.Length; i++) - page[b + i] = (byte)(clear[i] ^ mask[i]); - - // --- Post-mask tail (cleartext) --- - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(0x98), 0x00000654); // fixed constant - Encoding.ASCII.GetBytes("4.0").CopyTo(page, 0x9C); // engine version string - - // User commit-byte table (0xE00–0xFFF): 256 users × 2 bytes at the end of the header page — Jet 3.x's - // 0x600 commit region relocated to the end of the 4 KB page (see the Jet locking white paper). Each - // pair is a per-user commit/lock status. A fresh file must seed every slot to the neutral idle value - // 00 01 — NOT 00 00, which Jet reads as "mid-write to disk"; with no matching .ldb user lock that - // reads as a suspect/corrupt database and forces a repair before Access will open it. - for (int i = 0xE00; i < 0x1000; i++) page[i] = (byte)(i & 1); - - return page; - } - - // System-column flag bits: Sys = 0x10 (system-catalog column), SysSid = 0x10|0x20 (also a security id). - private const byte Sys = 0x10, SysSid = 0x30; - - // MSysObjects — the system catalog. Declared in the real physical (alphabetical) descriptor order with - // explicit canonical ColumnIds and system flags, exactly as an Access-created file stores it. - private static readonly ColumnSpec[] MSysObjectsColumns = - [ - new("Connect", JetDataType.Memo, 0, false, ColumnId: 9, SystemFlags: Sys), - new("Database", JetDataType.Memo, 0, false, ColumnId: 8, SystemFlags: Sys), - new("DateCreate", JetDataType.DateTime, 8, true, ColumnId: 4, SystemFlags: Sys), - new("DateUpdate", JetDataType.DateTime, 8, true, ColumnId: 5, SystemFlags: Sys), - new("Flags", JetDataType.Int32, 4, true, ColumnId: 7, SystemFlags: Sys), - new("ForeignName", JetDataType.Text, 510, false, ColumnId: 10, SystemFlags: Sys), - new("Id", JetDataType.Int32, 4, true, ColumnId: 0, SystemFlags: Sys), - new("Lv", JetDataType.Ole, 0, false, ColumnId: 13, SystemFlags: Sys), - new("LvExtra", JetDataType.Ole, 0, false, ColumnId: 16, SystemFlags: Sys), - new("LvModule", JetDataType.Ole, 0, false, ColumnId: 15, SystemFlags: Sys), - new("LvProp", JetDataType.Ole, 0, false, ColumnId: 14, SystemFlags: Sys), - new("Name", JetDataType.Text, 510, false, ColumnId: 2, SystemFlags: Sys), - new("Owner", JetDataType.Binary, 510, false, ColumnId: 6, SystemFlags: SysSid), - new("ParentId", JetDataType.Int32, 4, true, ColumnId: 1, SystemFlags: Sys), - new("RmtInfoLong", JetDataType.Ole, 0, false, ColumnId: 12, SystemFlags: Sys), - new("RmtInfoShort", JetDataType.Binary, 510, false, ColumnId: 11, SystemFlags: Sys), - new("Type", JetDataType.Int16, 2, true, ColumnId: 3, SystemFlags: Sys), - ]; - - // MSysACEs — per-object access-control rows. Real physical order + ColumnIds (ObjectId at fixed offset 0). - private static readonly ColumnSpec[] MSysAcesColumns = - [ - new("ACM", JetDataType.Int32, 4, true, ColumnId: 2, SystemFlags: Sys), - new("FInheritable", JetDataType.Boolean, 1, true, ColumnId: 3, SystemFlags: Sys), - new("ObjectId", JetDataType.Int32, 4, true, ColumnId: 0, SystemFlags: Sys), - new("SID", JetDataType.Binary, 510, false, ColumnId: 1, SystemFlags: SysSid), - ]; - - // MSysQueries — stored query/view definitions (empty in a fresh database). - private static readonly ColumnSpec[] MSysQueriesColumns = - [ - new("Attribute", JetDataType.Byte, 1, true, ColumnId: 1, SystemFlags: Sys), - new("Expression", JetDataType.Memo, 0, false, ColumnId: 5, SystemFlags: Sys), - new("Flag", JetDataType.Int16, 2, true, ColumnId: 6, SystemFlags: Sys), - new("LvExtra", JetDataType.Int32, 4, true, ColumnId: 7, SystemFlags: Sys), - new("Name1", JetDataType.Text, 510, false, ColumnId: 3, SystemFlags: Sys), - new("Name2", JetDataType.Text, 510, false, ColumnId: 4, SystemFlags: Sys), - new("ObjectId", JetDataType.Int32, 4, true, ColumnId: 0, SystemFlags: Sys), - new("Order", JetDataType.Binary, 510, false, ColumnId: 2, SystemFlags: Sys), - ]; - - // MSysRelationships — relationship (foreign key) definitions (empty in a fresh database). - private static readonly ColumnSpec[] MSysRelationshipsColumns = - [ - new("ccolumn", JetDataType.Int32, 4, true, ColumnId: 2, SystemFlags: Sys), - new("grbit", JetDataType.Int32, 4, true, ColumnId: 1, SystemFlags: Sys), - new("icolumn", JetDataType.Int32, 4, true, ColumnId: 3, SystemFlags: Sys), - new("szColumn", JetDataType.Text, 510, false, ColumnId: 5, SystemFlags: Sys), - new("szObject", JetDataType.Text, 510, false, ColumnId: 4, SystemFlags: Sys), - new("szReferencedColumn", JetDataType.Text, 510, false, ColumnId: 7, SystemFlags: Sys), - new("szReferencedObject", JetDataType.Text, 510, false, ColumnId: 6, SystemFlags: Sys), - new("szRelationship", JetDataType.Text, 510, false, ColumnId: 0, SystemFlags: Sys), - ]; - - - // ---- Complex-column system tables (ACE 12 / Access 2007 and later) -------------------------------------- - // - // Complex columns are Access's multi-value and attachment columns. Their registry is MSysComplexColumns, - // and each supported element type gets a flat storage table. Jet 4 (.mdb) has none of this — the feature - // arrived with ACE 12 — so these are only created from version byte 0x02 up. - // - // MSysComplexColumns is not optional even for a database that never uses a complex column: ACE consults it - // on every CREATE TABLE, and without it DDL through the OLE DB provider fails with "Cannot find table or - // constraint" (isolated in AceDdlOnLibRedDatabaseProbeTest by dropping exactly this table from a working - // DAO-created database). ACE never writes to it — it only has to resolve. - // - // Column ids are the ones the real engine assigns (creation order, which is not the alphabetical order the - // descriptors are stored in), so a byte-comparison against a DAO-created file lines up. - - private static readonly ColumnSpec[] MSysComplexColumnsColumns = - [ - new("ColumnName", JetDataType.Text, 510, false, ColumnId: 0, SystemFlags: Sys), - new("ComplexID", JetDataType.Int32, 4, true, IsAutoNumber: true, ColumnId: 4, SystemFlags: Sys), - new("ComplexTypeObjectID", JetDataType.Int32, 4, true, ColumnId: 1, SystemFlags: Sys), - new("ConceptualTableID", JetDataType.Int32, 4, true, ColumnId: 3, SystemFlags: Sys), - new("FlatTableID", JetDataType.Int32, 4, true, ColumnId: 2, SystemFlags: Sys), - ]; - - /// The flat storage tables, in the order the engine creates them. Each holds a single - /// Value column of its element type; Attachment is the exception, carrying the file metadata. - private static readonly (string Name, ColumnSpec[] Columns)[] MSysComplexTypeTables = - [ - ("MSysComplexType_UnsignedByte", [new("Value", JetDataType.Byte, 1, true, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_Short", [new("Value", JetDataType.Int16, 2, true, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_Long", [new("Value", JetDataType.Int32, 4, true, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_IEEESingle", [new("Value", JetDataType.Single, 4, true, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_IEEEDouble", [new("Value", JetDataType.Double, 8, true, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_GUID", [new("Value", JetDataType.Guid, 16, true, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_Decimal", [new("Value", JetDataType.FixedPoint, 9, false, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_Text", [new("Value", JetDataType.Text, 510, false, ColumnId: 0, SystemFlags: Sys)]), - ("MSysComplexType_Attachment", - [ - new("FileData", JetDataType.Ole, 0, false, ColumnId: 3, SystemFlags: Sys), - new("FileFlags", JetDataType.Int32, 4, true, ColumnId: 5, SystemFlags: Sys), - new("FileName", JetDataType.Text, 510, false, ColumnId: 1, SystemFlags: Sys), - new("FileTimeStamp", JetDataType.DateTime, 8, true, ColumnId: 4, SystemFlags: Sys), - new("FileType", JetDataType.Text, 510, false, ColumnId: 2, SystemFlags: Sys), - new("FileURL", JetDataType.Memo, 0, false, ColumnId: 0, SystemFlags: Sys), - ]), - ]; - - /// MSysObjects.Flags for the complex tables, as the real engine writes them: the registry carries - /// the plain system flag, the flat storage tables an extra 0x00030000. - private const int ComplexStorageFlags = unchecked((int)0x80030000); - - private const int SystemFlag = unchecked((int)0x80000000); - - // Per-file SID cluster. A database's on-disk 2-byte SIDs are the DEFAULT WORKGROUP's account SIDs XOR'd with - // a per-file 2-byte mask. The account SIDs were read verbatim from a real System.mdw (the file Access opens - // first to authenticate): admin(user)=03-01, Users(group)=02-01, Engine=02-03, Creator=02-04; the Admins - // group alone has a long per-workgroup SID (which Access materialises as a 98-byte SID on first open, so we - // don't emit it). Object ownership uses the "user" form (byte0 0x03) of Engine/Creator, matching real DAO - // files. Verified against WideTable with mask 24-CC: Users 02-01^24-CC = 26-CD, admin 03-01^24-CC = 27-CD, - // Engine-as-user 03-03^24-CC = 27-CF (system-object owner), Creator-as-user 03-04^24-CC = 27-C8. - // - // The mask is bound to the millisecond-precise creation date: Access derives both from a shared PRNG state - // at create time, so there is NO closed-form date->mask function (144 files, no checksum/PRNG fit) and the - // two MUST travel together. We bake one verified, self-consistent (creation-date, mask) pair — the from- - // scratch analogue of the account-SID constants — giving an Access-openable file with no template/graft. - // TODO: per-file-random dates (and custom/secured workgroups) need the date<->mask coupling cracked. - internal const long SeedCreationDateBits = 0x40E68F1E8943D217L; // 2026-06-27 22:54:07.716 (WideTable) — pairs with SidMask - private static readonly byte[] SidMask = [0x24, 0xCC]; // WideTable's per-file mask (pairs with SeedCreationDateBits) - private static byte[] Masked(byte b0, byte b1) => [(byte)(b0 ^ SidMask[0]), (byte)(b1 ^ SidMask[1])]; - internal static readonly byte[] SidUsers = Masked(0x02, 0x01); // Users group — read grantee / owner of user tables - internal static readonly byte[] SidAdmin = Masked(0x03, 0x01); // admin user — full grantee - private static readonly byte[] SidEngine = Masked(0x03, 0x03); // Engine (user form) — owner of system tables + DAO containers - private static readonly byte[] SidCreator = Masked(0x03, 0x04); // Creator (user form) — inheritable container grant - - /// - /// Creates a new, empty database at from scratch — no DAO/ADOX. Hand-builds the - /// bootstrap for the given — Jet 4 (0x01, the Access 2000 / - /// 2002-2003 .mdb) or any ACCDB version — (page 0, the page-1 free map, and the - /// MSysObjects/MSysACEs TDEFs with their - /// usage maps + self-registering catalog rows), then the file is a normal LibRed database: further tables - /// are added through the ordinary writers. The file opens in the Access GUI: the 0xE00 user - /// commit-byte table is seeded here, and Access adds the system tables it wants (MSysAccessStorage, the - /// navigation-pane objects) itself — hand-creating those was tried and made things worse. - /// - /// Where the new file is written. - /// The format version byte to create it at. - /// - /// The database's default text collating order, written to page 0 and inherited by every column created - /// in it. Defaults to General-Legacy (LCID 1033, version 0), which is what the engine writes; pass - /// for the order Access 2010+ offers as "General". - /// - /// Any order accepts can be created — 405 configurations - /// (399 at version 0, and the six orders that have a version-1 table), the - /// two General orders and every locale in JetLocaleTailoring, each verified by having ACE build an - /// index in the created file and agree on the keys (CreatedDatabaseCollationAccessTests). It - /// cannot be otherwise: the system-table indexes are built here, in this order, so creating a database - /// REQUIRES encoding its collation. That is why a new locale is unavailable to this method until it is - /// implemented, and why measuring one for the first time needs DAO to author the file. - /// - /// - public static void CreateEmpty(string path, byte version = 0x02, Collation? collation = null) - { - Collation sortOrder = collation ?? Collation.GeneralLegacy; - - // ACE 15 (0x04) can be read but never created: ACE REFUSES a file carrying the byte, an empty one - // included, and restamping 0x14 to 0x03 opens the identical bytes - // (Ace_refuses_the_0x04_version_byte_and_nothing_else_about_the_file). FromVersionByte still maps 0x04 - // onto the 0x03 layout, so reading one stays supported; only writing it is refused. - if (version == (byte)JetVersion.Version15_2013) - throw new NotSupportedException( - $"Cannot create a database at {nameof(JetVersion.Version15_2013)} (version byte 0x04): the " - + "Access engine refuses to open a file stamped with it. Access 2013 writes the Access 2010 " - + $"format, so use {nameof(JetVersion.Version14_2010)} for a 2013-era database. Existing 0x04 " - + "files can still be opened for reading."); - - JetFormatBase format = JetFormatBase.FromVersionByte(version); - - // Jet 4 (0x01, the Access 2000 / 2002-2003 `.mdb`) and the ACCDB versions can both be created; the - // identifier follows the version byte in BuildDefinitionPage, so the pair is always consistent and the - // file reopens. Jet 3 is rejected by FromVersionByte already — DAO cannot create one either - // ("Could not find installable ISAM"), so there is nothing to compare against. - // - // A Jet 4 file needs no other difference: DAO's own dbVersion40 database contains exactly the same - // four core system tables this method builds. Access adds MSysAccessStorage, the navigation-pane - // tables and the MSysDb properties (AccessVersion, Build, ProjVer, …) when it first opens the file, - // for a Jet 4 file just as for an ACCDB — measured by diffing a DAO-created Access 2000 database - // before and after Access opened it. - // - // It adds AccessVersion **09.50** and MSysAccessStorage even to a Jet 4 file. The legacy - // MSysAccessObjects store (08.50, and the only place the unmodelled 0x11 type appears) comes from - // Access *creating* the database itself in the Access 2000 generation — a new file from its own New - // dialog. So the generation is chosen at creation by Access, and a file created by anything else gets - // the modern one when Access first opens it. Nothing a creator can or should reproduce: DAO does not, - // and neither does this. - // The four core system tables live at the exact pages the page-0 bootstrap pointers name (2/3/4/5); - // their usage maps follow at 6..9. Access uses those pointers to find the catalog. - const int objPage = 2, acesPage = 3, queriesPage = 4, relPage = 5; - - // The system tables take the database collation too: Access writes v1 descriptors on MSys* in a - // General (v1) database, so anything else would be a mixed-collation file it never produces. - var (objTdef, objMap) = BuildSystemTable(format, MSysObjectsColumns, usageMapPage: 6, sortOrder); - var (acesTdef, acesMap) = BuildSystemTable(format, MSysAcesColumns, usageMapPage: 7, sortOrder); - var (queriesTdef, queriesMap) = BuildSystemTable(format, MSysQueriesColumns, usageMapPage: 8, sortOrder); - var (relTdef, relMap) = BuildSystemTable(format, MSysRelationshipsColumns, usageMapPage: 9, sortOrder); - const int seedPages = 10; // page 0, page 1, 4 core TDEFs (2..5), 4 usage maps (6..9) - byte[][] seed = - [ - BuildDefinitionPage(version, format.IsAccdb, 1252, sortOrder, - BitConverter.Int64BitsToDouble(SeedCreationDateBits)), - BuildFreeMapPage(format, seedPages), // page 1: global free-pages map - objTdef, acesTdef, queriesTdef, relTdef, // pages 2..5: core TDEFs - objMap, acesMap, queriesMap, relMap, // pages 6..9: their usage maps - ]; - // Creation is an explicit create-new operation. FileMode.Create would silently truncate an existing - // database before any of the format bootstrap work could validate or fail. - using (var fs = new FileStream(path, FileMode.CreateNew, FileAccess.Write, FileShare.None)) - foreach (byte[] p in seed) fs.Write(p, 0, format.PageSize); - - // Reopen through the normal stack and self-register the four core tables, so the catalog then finds - // them and every other table can go through TableCreator. - using var db = JetDatabase.Open(path, readOnly: false); - // The DAO catalog hierarchy Access navigates: a root (0x0F000000) parents the three containers - // (Tables/Databases/Relationships); tables live under Tables, MSysDb under Databases. Access looks - // objects up by (ParentId, Name), so these parents must be exact. - const int root = 0x0F000000, tablesC = 0x0F000001, databasesC = 0x0F000002, relationshipsC = 0x0F000003; - - Table msysObjects = db.OpenTableAt(objPage, "MSysObjects"); - Table msysAces = db.OpenTableAt(acesPage, "MSysACEs"); - void Ace(int objectId, byte[] sid, int acm, bool inherit = false) - { - var v = new object?[msysAces.Definition.Columns.Count]; - v[msysAces.Definition.FindColumn("ObjectId")!.Index] = objectId; - v[msysAces.Definition.FindColumn("SID")!.Index] = sid; - v[msysAces.Definition.FindColumn("ACM")!.Index] = acm; - v[msysAces.Definition.FindColumn("FInheritable")!.Index] = inherit; - msysAces.Insert(v); - } - - // Catalog rows in the real stored order: DAO containers + MSysDb first, then the system tables, then - // the DAO "SingleRecord" pseudo-object. Access's catalog bootstrap walks MSysObjects in this order. - InsertCatalogRow(msysObjects, tablesC, "Tables", type: 3, SystemFlag, parentId: root, owner: SidEngine); - InsertCatalogRow(msysObjects, databasesC, "Databases", type: 3, SystemFlag, parentId: root, owner: SidEngine); - InsertCatalogRow(msysObjects, relationshipsC, "Relationships", type: 3, SystemFlag, parentId: root, owner: SidEngine); - InsertCatalogRow(msysObjects, 0x10000000, "MSysDb", type: 2, SystemFlag, parentId: databasesC, owner: SidUsers); - InsertCatalogRow(msysObjects, objPage, "MSysObjects", type: 1, SystemFlag, parentId: tablesC, owner: SidEngine); - InsertCatalogRow(msysObjects, acesPage, "MSysACEs", type: 1, SystemFlag, parentId: tablesC, owner: SidEngine); - InsertCatalogRow(msysObjects, queriesPage, "MSysQueries", type: 1, SystemFlag, parentId: tablesC, owner: SidEngine); - InsertCatalogRow(msysObjects, relPage, "MSysRelationships", type: 1, SystemFlag, parentId: tablesC, owner: SidEngine); - InsertCatalogRow(msysObjects, unchecked((int)0x80000000), "SingleRecord", type: 9, flags: 0x10000000, parentId: relationshipsC); - - // Access-control rows (verified per-object-class masks) in the same object order as the catalog rows. - Ace(tablesC, SidCreator, 0x0F00FE, inherit: true); Ace(tablesC, SidUsers, 0x060001); Ace(tablesC, SidAdmin, 0x0FFEFF, inherit: true); - Ace(databasesC, SidUsers, 0x060000); - Ace(relationshipsC, SidCreator, 0x0F00FE, inherit: true); Ace(relationshipsC, SidUsers, 0x060001); Ace(relationshipsC, SidAdmin, 0x0FFFFF, inherit: true); - Ace(0x10000000, SidUsers, 0x06000E); Ace(0x10000000, SidAdmin, 0x00000E); // MSysDb - Ace(objPage, SidUsers, 0x060000); Ace(objPage, SidAdmin, 0x000014); // MSysObjects - Ace(acesPage, SidUsers, 0x060000); // MSysACEs (Users only) - Ace(queriesPage, SidUsers, 0x060000); Ace(queriesPage, SidAdmin, 0x000014); // MSysQueries - Ace(relPage, SidUsers, 0x0E0000); Ace(relPage, SidAdmin, 0x000014); // MSysRelationships - - // The system tables carry the indexes Access uses to navigate the catalog. ParentIdName is the first - // real index, Id (the PK) second — matching real files. - db.CreateIndex("MSysObjects", "ParentIdName", [("ParentId", false), ("Name", false)], isUnique: true); - db.CreateIndex("MSysObjects", "Id", [("Id", false)], isUnique: true, isPrimary: true); - db.CreateIndex("MSysACEs", "ObjectId", [("ObjectId", false)], disallowNull: true); - db.CreateIndex("MSysQueries", "ObjectIdAttribute", [("ObjectId", false), ("Attribute", false), ("Order", false)], isUnique: true, isPrimary: true); - db.CreateIndex("MSysRelationships", "szRelationship", [("szRelationship", false)]); - db.CreateIndex("MSysRelationships", "szObject", [("szObject", false)]); - db.CreateIndex("MSysRelationships", "szReferencedObject", [("szReferencedObject", false)]); - - // Complex-column system tables — ACE 12 and later only (see CreateComplexSystemTables). - if (version >= 0x02) CreateComplexSystemTables(db); - - // Note: MSysAccessStorage and the MSysNavPane* tables are deliberately NOT created here. Verified across - // ~135 pure-DAO reference files: none of them carry those tables — Access creates them (plus the nav-pane - // long SID) itself on first open. Emitting them ourselves both diverged from real DAO output and produced - // a table Access's compact rejected ("-1206 Unrecognized database format"). A faithful native file mirrors - // DAO: core catalog only, and Access augments on first open. - } - - /// - /// Creates MSysComplexColumns and the MSysComplexType_* storage tables — the complex-column - /// (multi-value / attachment) infrastructure Access 2007 / ACE 12 introduced. Jet 4 has none of it, - /// so the caller gates this on version byte 0x02 or later. - /// - /// The registry table is required even in a database that never uses a complex column: ACE consults - /// it on every CREATE TABLE, and a database without it rejects DDL through the OLE DB provider with - /// "Cannot find table or constraint". It stays empty — ACE reads it, never writes it (both facts isolated - /// in AceDdlOnLibRedDatabaseProbeTest). The storage tables are created for completeness so a - /// complex column added later has somewhere to live. - /// - /// These go through the ordinary writers, so each gets its TDEF, usage map, catalog row and index - /// roots the same way a user table does; the rows are then corrected to the system flags and owner the - /// real engine writes. Page numbers therefore follow LibRed's own allocation rather than matching a - /// DAO-created file position for position — DAO's numbering is a consequence of how it lays out the core - /// four tables' usage maps and index roots, which LibRed does differently. - /// - private static void CreateComplexSystemTables(JetDatabase db) - { - db.CreateTable("MSysComplexColumns", MSysComplexColumnsColumns); - // Index names, order and flags as the engine writes them: the ComplexID primary key first, then the - // two non-unique lookups the engine uses to find a table's complex columns. - db.CreateIndex("MSysComplexColumns", "IdxID", [("ComplexID", false)], - isUnique: true, isPrimary: true, disallowNull: true, ignoreNulls: true); - db.CreateIndex("MSysComplexColumns", "IdxConceptualTableID", [("ConceptualTableID", false)], - disallowNull: true, ignoreNulls: true); - db.CreateIndex("MSysComplexColumns", "IdxFlatTableID", [("FlatTableID", false)], - disallowNull: true, ignoreNulls: true); - MarkAsSystemTable(db, "MSysComplexColumns", SystemFlag); - - foreach ((string name, ColumnSpec[] columns) in MSysComplexTypeTables) - { - db.CreateTable(name, columns); - MarkAsSystemTable(db, name, ComplexStorageFlags); - } - } - - /// Turns a table the ordinary writers just created into a system object: the MSysObjects row gets - /// the engine's flags and owner, and the TDEF's table-type byte becomes 'S'. Creating it as a user table - /// first and correcting it reuses all the allocation, usage-map and index machinery. - private static void MarkAsSystemTable(JetDatabase db, string name, int flags) - { - TableDef definition = db.Catalog.FindTable(name) - ?? throw new InvalidOperationException($"'{name}' was not found after creating it."); - - Table msysObjects = db.OpenTable("MSysObjects"); - TableDef objectsDef = msysObjects.Definition; - int idIndex = objectsDef.FindColumn("Id")!.Index; - int flagsIndex = objectsDef.FindColumn("Flags")!.Index; - int ownerIndex = objectsDef.FindColumn("Owner")!.Index; - - foreach ((RowId rowId, object?[] values) in msysObjects.Rows().WithIds()) - { - if (values[idIndex] is not { } id - || Convert.ToInt32(id, CultureInfo.InvariantCulture) != definition.DefinitionPage) continue; - values[flagsIndex] = flags; - values[ownerIndex] = SidEngine; - msysObjects.Update(rowId, values, new HashSet { flagsIndex, ownerIndex }); - break; - } - - // TDEF table type: 'N' user -> 'S' system. - IO.PageChannel channel = msysObjects.Channel; - byte[] tdef = channel.ReadPageShared(definition.DefinitionPage).Span.ToArray(); - tdef[channel.Format.TdefTableTypeOffset] = (byte)TableType.System; - channel.WritePage(definition.DefinitionPage, tdef); - db.Catalog.Invalidate(); - } - - private static void InsertCatalogRow(Table msysObjects, int id, string name, short type, int flags, int parentId = 0, byte[]? owner = null) - { - var values = new object?[msysObjects.Definition.Columns.Count]; - void Set(string col, object? v) => values[msysObjects.Definition.FindColumn(col)!.Index] = v; - Set("Id", id); - Set("ParentId", parentId); - Set("Name", name); - Set("Type", type); // 1 = table, 2 = database (MSysDb), 3 = DAO container, 9 = SingleRecord - Set("Flags", flags); - if (owner != null) Set("Owner", owner); - Set("DateCreate", DateTime.Now); - Set("DateUpdate", DateTime.Now); - msysObjects.Insert(values); - } - - /// Builds the page-1 global free-pages map: a data page with two 69-byte inline maps. Row 0 is - /// the free map — pages < are used (bit 0); pages from there to the map's - /// 512-page reach are marked free (bit 1), pre-declaring space beyond the file end so an allocator - /// (LibRed's or Access's) grabs a "free" page and grows the file. Row 1 is an all-zeros companion, as real - /// files carry. - private static byte[] BuildFreeMapPage(JetFormatBase format, int usedPages) - { - const int mapLen = 1 + 4 + 64; // inline map: type + start page + 64-byte bitmap (512 pages) - var page = new byte[format.PageSize]; - page[0] = (byte)PageType.DataPage; - page[1] = 0x01; - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(format.DataOwnerOffset, 4), 1); // global-map owner (observed) - - int row0 = format.PageSize - mapLen; // free map (highest offset — the one the allocator reads) - int row1 = row0 - mapLen; // all-zeros companion - for (int p = usedPages; p < 64 * 8; p++) // mark pages >= usedPages free - page[row0 + 5 + (p >> 3)] |= (byte)(1 << (p & 7)); - - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset, 2), (ushort)row0); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + 2, 2), (ushort)row1); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), 2); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), (ushort)(row1 - format.DataRowDirectoryOffset - 4)); - return page; - } - - /// Builds a system table's TDEF page and its owned/free usage-map page. Long-value (Memo/Ole) - /// columns each get an owned+free map on that page (rows 2 onward, after the two data-page maps), so a - /// row that stores a long value — e.g. an MSysObjects catalog row carrying an LvProp blob — - /// has somewhere to record its LVAL page. - private static (byte[] Tdef, byte[] UsageMap) BuildSystemTable( - JetFormatBase format, IReadOnlyList columns, int usageMapPage, Collation collation) - { - var longValueCols = columns.Select((c, pos) => (c, id: c.ColumnId ?? pos)) - .Where(x => x.c.Type is JetDataType.Memo or JetDataType.Ole).ToList(); - var longValueSpecs = new List(longValueCols.Count); - for (int j = 0; j < longValueCols.Count; j++) - longValueSpecs.Add(new LongValueColumnSpec(longValueCols[j].id, UsedRow: 2 + 2 * j, FreeRow: 3 + 2 * j, MapPage: usageMapPage)); - - byte[] tdef = TdefBuilder.Build(format, TableType.System, columns, longValueColumns: longValueSpecs, - collation: collation).Page; - tdef[format.TdefOwnedPagesOffset] = 0; WriteInt24(tdef, format.TdefOwnedPagesOffset + 1, usageMapPage); - tdef[format.TdefFreePagesOffset] = 1; WriteInt24(tdef, format.TdefFreePagesOffset + 1, usageMapPage); - var tdefPage = new byte[format.PageSize]; - Array.Copy(tdef, tdefPage, format.PageSize); // these system TDEFs fit one page - - byte[] usageMap = BuildUsageMapPage(format, 2 + longValueCols.Count * 2); - return (tdefPage, usageMap); - } - - private static byte[] BuildUsageMapPage(JetFormatBase format, int mapCount) - { - const int mapLen = 1 + 4 + 64; // inline map record: type + start page + 64-byte bitmap - var page = new byte[format.PageSize]; - page[0] = (byte)PageType.DataPage; - page[1] = 0x01; - int offset = format.PageSize; - for (int row = 0; row < mapCount; row++) - { - offset -= mapLen; - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + row * 2, 2), (ushort)offset); - } - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), (ushort)mapCount); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(offset - format.DataRowDirectoryOffset - mapCount * 2)); - return page; - } - - private static void WriteInt24(byte[] b, int o, int v) - { - b[o] = (byte)v; b[o + 1] = (byte)(v >> 8); b[o + 2] = (byte)(v >> 16); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/IndexCursor.cs b/src/LibRed/LibRed.Core/Storage/IndexCursor.cs index 127a6a5fd..9a72afa6a 100644 --- a/src/LibRed/LibRed.Core/Storage/IndexCursor.cs +++ b/src/LibRed/LibRed.Core/Storage/IndexCursor.cs @@ -4,16 +4,13 @@ namespace LibRed.Storage; -/// An index entry: the decoded key values (in index column order) and the row they point at. -public readonly record struct IndexEntry(object?[] Key, RowId Row); - /// /// Walks an index B-tree and yields the row pointers in index (key) order. /// /// -/// Each index page has an entry-position bitmask at whose set +/// Each index page has an entry-position bitmask at whose set /// bits give the end offsets of successive entries within the entry-data region that begins -/// at . A leaf entry ends with a 4-byte big-endian row pointer +/// at . A leaf entry ends with a 4-byte big-endian row pointer /// (page in the high 24 bits, row in the low 8); a node entry instead ends with the 4-byte /// child page number. Key bytes are not decoded here — only the trailing pointers are read — /// so the order-preserving key encoding is not needed to enumerate rows in order. @@ -29,8 +26,8 @@ public sealed class IndexCursor(PageChannel channel, int rootPage) /// Yields each entry with its decoded key (per ) in index order. /// Key columns that use Jet's lossy text/binary collation decode as null. /// - public IEnumerable Entries(IReadOnlyList<(ColumnDef Column, bool Ascending)> columns) => - WalkRaw().Select(e => new IndexEntry(IndexKeyDecoder.Decode(columns, e.Key), e.Row)); + public IEnumerable<(object?[] Key, RowId Row)> Entries(IReadOnlyList<(ColumnDef Column, bool Ascending)> columns) => + WalkRaw().Select(e => (Key: IndexKeyCodec.Decode(columns, e.Key), Row: e.Row)); /// /// Yields each entry's full (decompressed) key bytes and row pointer, without decoding — used @@ -43,19 +40,32 @@ public IEnumerable Entries(IReadOnlyList<(ColumnDef Column, bool Asc var pending = new Stack(); var visited = new HashSet(); int? owner = null; + byte[]? previous = null; pending.Push(_rootPage); while (pending.Count > 0) { int pageNumber = pending.Pop(); if (!visited.Add(pageNumber)) throw new InvalidDataException($"Index traversal contains a repeated/cyclic page {pageNumber}."); - CheckedIndexPage page = IndexPageReader.Read(_channel, pageNumber, owner); + IndexTree.CheckedPage page = IndexTree.Read(_channel, pageNumber, owner); owner ??= page.Owner; if (page.Type == PageType.LeafIndexPage) { - foreach ((byte[] key, int pointer) in IndexPageReader.DecodeEntries(page)) - yield return (key, new RowId(pointer >> 8, pointer & 0xFF)); + foreach ((byte[] key, int pointer) in IndexTree.DecodeEntries(page)) + { + // The one invariant a B-tree exists for: walked in order, stored keys never go backwards. + // Nothing else checks it, and a walk that silently accepts a descending step is a walk that + // cannot tell a sound tree from one this engine mis-split — the keys come back, the seeks + // that binary-search the same pages do not. Equal keys are ordinary (a non-unique index), + // and a descending index inverts its bytes, so the stored order ascends either way. + if (previous is not null && previous.AsSpan().SequenceCompareTo(key) > 0) + throw new InvalidDataException( + $"Index page {pageNumber} entry {Convert.ToHexString(key)} sorts before the entry " + + $"before it, {Convert.ToHexString(previous)}."); + previous = key; + yield return (key, RowId.FromPacked(pointer)); + } continue; } pending.Push(page.Tail); @@ -64,8 +74,8 @@ public IEnumerable Entries(IReadOnlyList<(ColumnDef Column, bool Asc // prefix can cover the first bytes of the trailer, and on a node whose 0x18 is nonzero — which ACE // tolerates — that offset lands on the wrong bytes, or inside the entry bitmask for a stored entry // under 4 bytes, still in bounds. Either way it yields a wrong child page with no exception, and - // bypasses the child-page validation IndexPageReader.Read performs on the reconstructed entries. - var children = IndexPageReader.DecodeEntries(page).Select(e => e.Trailer).ToList(); + // bypasses the child-page validation IndexTree.Read performs on the reconstructed entries. + List children = IndexTree.Trailers(page); for (int i = children.Count - 1; i >= 0; i--) pending.Push(children[i]); } diff --git a/src/LibRed/LibRed.Core/Storage/IndexKeyCodec.cs b/src/LibRed/LibRed.Core/Storage/IndexKeyCodec.cs new file mode 100644 index 000000000..9d14bde4a --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/IndexKeyCodec.cs @@ -0,0 +1,690 @@ +using EntityFrameworkCore.Jet.Data; +using LibRed.Catalog; +using LibRed.Storage.Types; +using System.Buffers.Binary; +using System.Globalization; +using System.Runtime.InteropServices; + +namespace LibRed.Storage; + +/// +/// Reads and writes Jet/ACE index keys, owning their layout, reversible value transforms, +/// collation framing, length limit and truncation checksum. +/// +/// +/// Encoding reads values by ; decoding returns values in index-column order. +/// Text, Binary and DATETIME2 cannot be decoded reliably from a stored key. Decoding stops at the first +/// such column or incomplete value, leaving it and subsequent columns null. The existing text collations +/// supply weights; this codec owns the key structure around them. +/// +public static class IndexKeyCodec +{ + // --- Limits --- + + /// + /// The longest index entry ACE stores verbatim. Measured: an entry of exactly 510 bytes comes back + /// byte-for-byte, and one that would be 511 comes back as 510 — the weights cut short and the last two + /// bytes replaced by a value that varies with the string (…0E0602 for one 254-character value, + /// …0EDE2A for the 255-character one). That is a truncated key plus a checksum, which is why two + /// long values never collide. It caps the whole entry rather than each column: two 200-character text + /// columns are about 404 bytes of key each and ACE stores their combined entry truncated. + /// + internal const int MaxKeyBytes = 510; + + /// The checksum that ends a truncated key, big-endian. + internal const int ChecksumSize = sizeof(ushort); + + /// The bytes of an over-long key ACE keeps ahead of its checksum. + internal const int KeptKeyBytes = MaxKeyBytes - ChecksumSize; + + /// Access indexes only the first 255 characters of a Memo (Long Text) value (verified vs ACE). + internal const int MemoKeyMaxChars = 255; + + // --- Column prefixes and fixed-width values --- + + /// The prefix of a present value in a column of the given direction. + internal static byte Start(bool ascending) => (byte)(ascending ? IndexKeyPrefix.AscStart : IndexKeyPrefix.DescStart); + + /// The prefix of a null in a column of the given direction. + internal static byte Null(bool ascending) => (byte)(ascending ? IndexKeyPrefix.AscNull : IndexKeyPrefix.DescNull); + + /// The byte after a Yes/No column's start prefix: 00 for true and FF for false, so true + /// sorts first, both inverted in a descending column (verified against ACE: 7F 00 / 7F FF, 80 FF / 80 00). + internal static byte Boolean(bool value, bool ascending) + { + byte b = value ? (byte)0x00 : (byte)0xFF; + return ascending ? b : (byte)~b; + } + + /// + /// The key width of a fixed-width column type, or -1 where the key is not fixed-width — TEXT, Binary, + /// GUID and DATETIME2 all encode to a variable, and for the first three lossy, form + /// (page-03-04 §10.4). Here for the same reason as the prefixes above: the encoder and the decoder each had + /// their own copy of this table and they drifted, the decoder never learning the widths the encoder + /// writes for and . + /// + internal static int FixedKeySize(JetDataType type) => type switch + { + JetDataType.Byte => 1, + JetDataType.Int16 => 2, + JetDataType.Int32 => 4, + // A complex (multi-value / attachment) column's key is its Int32 complex id, encoded exactly as an + // Int32 — verified against ACE over 43 entries across 9 such indexes in two files, covering + // attachment, Text and Long element types, with no difference in any byte. + JetDataType.Complex => 4, + JetDataType.Single => 4, + JetDataType.Double or JetDataType.DateTime => 8, + // Int64/BIGINT keys like Currency — both are an int64, sign bit flipped, big-endian. Its VARIABLE + // storage does not change that: this dispatch is on the type, not on where the row keeps it. + JetDataType.Currency or JetDataType.Int64 => 8, + JetDataType.FixedPoint => FixedPointMagnitudeOffset + FixedPointMagnitudeSize, + _ => -1, + }; + + /// The sign byte of a non-negative FixedPoint (Decimal/Numeric) key. A negative key is the bitwise + /// complement of the whole positive form, so its sign byte reads 00. + internal const byte FixedPointNonNegative = 0xFF; + + /// A FixedPoint key's magnitude — |value| × 10^scale as one big-endian 128-bit integer — after the sign + /// byte. + internal const int FixedPointMagnitudeOffset = 1; + internal const int FixedPointMagnitudeSize = 16; + + // --- Chunked values (Binary, GUID, DATETIME2) --- + + /// The data bytes in one chunk of a chunked key, the last zero-padded to the full width. + internal const int ChunkSize = 8; + + /// The control byte after a chunk that more follow. Constant in a descending column too, where the + /// final chunk's control byte — its real-byte count — is inverted. + internal const byte ChunkContinues = 0x09; + + // --- Text --- + + /// Appended after a descending text key's inverted bytes (verified against ACE). + internal const byte DescendingTextEnd = 0x00; + + // The bytes inside a text key's body, shared by every collation (General v0, General v1 and the locale + // tailorings over them). [MS-UCODEREF] frames the key as primaries SEP diacritics SEP case SEP extra SEP + // specials TERM; Access leaves the case section empty. + + /// Separates the sections of a text key — primaries, diacritics, case, extra. + internal const byte SectionSeparator = 0x01; + + /// Ends a text key's body. + internal const byte EndKey = 0x00; + + /// The top byte of an inline (positional) record's 16-bit position field. + internal const byte InlineStart = 0x80; + + /// The byte inside the extra section a kana fills: between its small-form codes and its mark codes, + /// inside its closing run, and after that run when inline records follow. What it denotes is not + /// established; where it sits is measured. + internal const byte KanaRunSeparator = 0xFF; + + /// The secondary weight of a character with no accent. + internal const byte DefaultSecondary = 0x02; + + /// The primary ACE gives a mark with nothing to act on — a shadda or an iteration mark with no weight + /// before it — and the unweighted characters of both tables. + internal static ReadOnlySpan UnweightedPrimary => [0xFF, 0xFF]; + + /// What a Han character's four-byte primary starts with under General (v1), before its own alphabetic + /// and diacritic weights (verified against ACE: U+4E00 is 7F FD FF 3C 6A 01 00). + internal static ReadOnlySpan HanPrimaryMarker => [0xFD, 0xFF]; + + /// + /// [MS-UCODEREF] PUNCTUATION, the script member of a word-sort ignorable: it carries no primary weight but + /// is recorded positionally so co-op stays beside coop. The apostrophe and hyphen live here (their + /// 0x80/0x82 inline codes are simply their Alphabetic Weights), which is why exactly those two are + /// special — it is the platform's rule, not an Access one. Both versions write it in the inline record. + /// + internal const byte WordSortScriptMember = 6; + /// + /// The per-column prefix byte of an order-preserving index key. Each non-boolean column is prefixed by a + /// start byte (present value) or null byte, with distinct values for ascending vs descending columns so that + /// a lexicographic byte compare matches the index's logical order. + /// + private enum IndexKeyPrefix : byte + { + /// Ascending column, null value. + AscNull = 0x00, + + /// Ascending column, present value. + AscStart = 0x7F, + + /// Descending column, present value. + DescStart = 0x80, + + /// Descending column, null value. + DescNull = 0xFF, + } + + public static byte[] Encode(IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, object?[] values) => + Encode(columns, values, enforceLengthLimit: true); + + /// + /// The key LibRed would build if ACE had no length limit — the input the truncation works ON. + /// + /// + /// Only the research that is trying to identify ACE's two-byte checksum wants this: recovering the + /// function means pairing what ACE stored against the full key it was derived from, and the ordinary + /// entry point refuses exactly those values. Not a way around the limit — a key this returns is longer + /// than ACE would store and must never be written to a file. + /// + internal static byte[] EncodeWithoutLengthLimit( + IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, object?[] values) => + Encode(columns, values, enforceLengthLimit: false); + + // The lists a key is built in, this thread's, cleared rather than allocated — as the collation encoders keep + // theirs: an index-nested-loop join encodes a key for every outer row, and two growing lists per key were most + // of what the seek allocated. Safe because nothing here re-enters Encode. + [ThreadStatic] private static List? t_buffer; + [ThreadStatic] private static List? t_textKey; + + private static byte[] Encode( + IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, object?[] values, bool enforceLengthLimit) + { + List buffer = t_buffer ??= []; + buffer.Clear(); + + for (int i = 0; i < columns.Count; i++) + { + (ColumnDef column, bool ascending) = columns[i]; + object? value = values[column.Index]; + + if (column.Type == JetDataType.Boolean) + { + // The start flag, then the value's byte. The value is read with the row's truthiness — -1 and 7 + // are true — or the key disagrees with the row it indexes. A Yes/No column cannot be null, and a + // null is keyed as the false the row stores for it. + buffer.Add(Start(ascending)); + buffer.Add(Boolean(RowCodec.IsTruthy(value), ascending)); + continue; + } + + if (value is null) + { + buffer.Add(Null(ascending)); + continue; + } + + // Text uses Jet's collation: start flag then the collation key body (weights, inline + // ignorable codes, terminator). Descending inverts every byte of that ascending key + // and appends a 0x00 (verified against ACE). + // + // A Memo (Long Text) column IS indexable in Access, and its key is the *same* collation key + // over only the value's first 255 characters — verified vs ACE: a 256- or 300-character memo + // produces byte-for-byte the key of its 255-character prefix. + if (column.Type is JetDataType.Text or JetDataType.Memo) + { + // Weights are implemented for the two General orders plus the locale tailorings in + // JetLocaleTailoring. Refuse anything else up front rather than emit wrong bytes with the + // English table — a wrong key does not fail, it silently disagrees with ACE's. The collation + // is read per-column from the descriptor (0x0B–0x0E). + if (!column.Collation.IsIndexKeyEncodable) + throw new NotSupportedException( + $"Index key encoding for column '{column.Name}' uses collation {column.Collation.Order} " + + $"version {column.Collation.Version}" + + (column.Collation.SortId == 0 ? "" : $" sort id {column.Collation.SortId}") + + ", which is not implemented yet."); + + string text = (string)value; + if (column.Type == JetDataType.Memo && text.Length > MemoKeyMaxChars) + text = text[..MemoKeyMaxChars]; + + List ascendingKey = t_textKey ??= []; + ascendingKey.Clear(); + ascendingKey.Add(Start(ascending: true)); + LocaleTailoring? tailoring = JetLocaleTailoring.For(column.Collation); + bool encoded = column.Collation.Version == Collation.GeneralVersion + ? JetTextCollationV1.TryEncode(text, ascendingKey, tailoring) + : JetTextCollation.TryEncode(text, ascendingKey, tailoring); + if (!encoded) + throw new NotSupportedException( + $"Text index key '{text}' contains a character with no weight in the {column.Collation.Order} " + + $"v{column.Collation.Version} collation table."); + + if (ascending) + { + buffer.AddRange(ascendingKey); + } + else + { + foreach (byte b in ascendingKey) buffer.Add((byte)~b); + buffer.Add(DescendingTextEnd); + } + continue; + } + + // GUID key (verified against ACE): the 16 GUID bytes in canonical *string* order (NOT the + // mixed-endian .ToByteArray layout), chunked exactly as a 16-byte Binary value is — two full chunks, + // the first followed by the continuation marker and the second by its count, 8. Descending inverts + // every byte except the continuation marker — verified against ACE. + if (column.Type == JetDataType.Guid) + { + Guid guid = value switch + { + Guid g => g, + byte[] b when b.Length == 16 => new Guid(b), + string text when JetTypeCodec.TryParseGuid(text, out Guid parsed) => parsed, + _ => throw new NotSupportedException($"Cannot encode GUID index key from {value.GetType().Name}."), + }; + EncodeBinaryChunked(buffer, Convert.FromHexString(guid.ToString("N")), ascending); + continue; + } + + // Binary key (verified against ACE's EverythingIsBytes fixture): see EncodeBinaryChunked. The old fixed + // 4-byte MSysQueries.Order case is the single-chunk form (7F <4B> 00 00 00 00 04). + if (column.Type == JetDataType.Binary) + { + EncodeBinaryChunked(buffer, (byte[])value, ascending); + continue; + } + + // DATETIME2 keys the whole 42-byte stored value through that same chunking, rather than folding + // it to a number the way DateTime folds to its OA double — verified against ACE, which stores + // 7F <8B> 09 … over exactly the bytes on the page. It works because the encoding + // is already order-preserving: both numeric fields are zero-padded to 19 digits, so byte order is + // chronological order. Note the value's 42nd byte is a NUL (see JetTypeCodec) and lands in the key. + if (column.Type == JetDataType.DateTimeExtended) + { + EncodeBinaryChunked( + buffer, + JetTypeCodec.EncodeExtendedDateTime(Convert.ToDateTime(value, CultureInfo.InvariantCulture)), + ascending); + continue; + } + + int size = FixedKeySize(column.Type); + if (size <= 0) + throw new NotSupportedException( + $"Index key encoding for {column.Type} (binary collation) is not supported yet."); + + buffer.Add(Start(ascending)); + byte[] raw = EncodeFixed(column, value, size); + if (!ascending) + Complement(raw); + buffer.AddRange(raw); + } + + // Past MaxKeyBytes ACE keeps the leading KeptKeyBytes and replaces the rest with a checksum over what it + // dropped, which is why two long values sharing a prefix still sort apart. The limit is on the WHOLE + // entry, not per column (see MaxKeyBytes). + if (!enforceLengthLimit || buffer.Count <= MaxKeyBytes) return [.. buffer]; + + // A discarded word-sort record used to be refused here, on the reasoning that the record sits in the + // part ACE dropped and so what it held is unobservable — and that if ACE recomputed its position when + // truncating, LibRed's reconstruction would be feeding the checksum the wrong bytes. Both halves are + // now measured and neither holds: ACE does NOT recompute the position (the record's position byte + // tracks where the mark actually sat), and the checksum over LibRed's reconstructed key reproduces + // ACE's exactly — 16 of 16 over two mark characters at eight positions each. The record was never + // unobservable; it just could not be checked until the checksum's own arithmetic was pinned down. + // See docs/design/index-key-checksum.md. + byte[] truncated = new byte[MaxKeyBytes]; + buffer.CopyTo(0, truncated, 0, KeptKeyBytes); + BinaryPrimitives.WriteUInt16BigEndian(truncated.AsSpan(KeptKeyBytes), + ComputeChecksum(CollectionsMarshal.AsSpan(buffer)[KeptKeyBytes..])); + return truncated; + } + + /// + /// Appends Jet's order-preserving chunked key: the start flag, then the data in + /// -byte chunks — real bytes left-aligned, the last zero-padded — each + /// followed by a control byte: when another chunk follows, otherwise + /// the real-byte count of this final chunk (1..8). Descending inverts every byte except the continuation + /// markers, which stay constant so the structure stays parseable. Binary, GUID and DATETIME2 keys all take this + /// form (verified against ACE). An empty value is the start flag alone. + /// + private static void EncodeBinaryChunked(List buffer, byte[] data, bool ascending) + { + buffer.Add(Start(ascending)); + if (data.Length == 0) + return; + + int offset = 0; + do + { + int n = Math.Min(ChunkSize, data.Length - offset); + for (int j = 0; j < ChunkSize; j++) + { + byte b = j < n ? data[offset + j] : (byte)0; + buffer.Add(ascending ? b : (byte)~b); + } + offset += n; + + if (offset < data.Length) + buffer.Add(ChunkContinues); + else + buffer.Add(ascending ? (byte)n : (byte)~n); + } + while (offset < data.Length); + } + + /// A fixed-width column's value as its -byte ascending key + /// (). + private static byte[] EncodeFixed(ColumnDef column, object value, int size) + { + var c = CultureInfo.InvariantCulture; + switch (column.Type) + { + case JetDataType.Byte: + return [Convert.ToByte(value, c)]; + case JetDataType.Int16: + return EncodeInteger(Convert.ToInt16(value, c), size); + case JetDataType.Int32: + case JetDataType.Complex: // the complex id, keyed as the Int32 it is + return EncodeInteger(Convert.ToInt32(value, c), size); + case JetDataType.Currency: + return EncodeInteger(JetTypeCodec.CurrencyToScaled(value, c), size); + case JetDataType.Int64: // BIGINT — verified against ACE across 0, ±1, ±42 and both extremes + return EncodeInteger(Convert.ToInt64(value, c), size); + case JetDataType.Single: + return EncodeFloatBits(BitConverter.SingleToInt32Bits(Convert.ToSingle(value, c)), size); + case JetDataType.Double: + return EncodeFloatBits(BitConverter.DoubleToInt64Bits(Convert.ToDouble(value, c)), size); + case JetDataType.DateTime: + return EncodeFloatBits(BitConverter.DoubleToInt64Bits(JetTypeCodec.ToOaDate(column, value, c)), size); + case JetDataType.FixedPoint: + return EncodeFixedPoint(JetDecimalConverter.ToDecimal(value, c), column.Scale, size); + default: + throw new NotSupportedException($"Index key type {column.Type} is not encodable."); + } + } + + /// + /// Encodes a FixedPoint (Numeric/Decimal) index key — a sign byte followed by the value's 16-byte + /// big-endian **unscaled magnitude** (|value| × 10^scale, the same integer the row codec stores). + /// A non-negative value uses sign 0xFF; a negative value is the **bitwise complement of the + /// whole 17-byte positive form** (sign becomes 0x00, magnitude is one's-complemented), so byte + /// order equals numeric order: negatives (sign 0x00) precede non-negatives (0xFF), and complementing + /// makes a larger magnitude sort earlier among negatives. A negative zero — what a value too small for + /// the scale truncates to — keeps its sign, as ACE keys it (7F 00 FF…FF), so the sign is taken with + /// : < 0 is false for -0.0000m. + /// Verified byte-for-byte against ACE (see DecimalKeyEncodingTests). + /// + private static byte[] EncodeFixedPoint(decimal value, byte scale, int size) + { + decimal factor = 1m; + for (int i = 0; i < scale; i++) factor *= 10m; + // Truncated toward zero, matching ACE and — necessarily — JetTypeCodec.EncodeNumeric: quantise a key + // differently from its row and the value is indexed under a number the row does not contain. + decimal magnitude = decimal.Truncate(Math.Abs(value) * factor); + int[] bits = decimal.GetBits(magnitude); // [lo, mid, hi, flags]; magnitude has scale 0 + + var key = new byte[size]; + key[0] = FixedPointNonNegative; + // The 96-bit magnitude as the 128-bit field holds it, so its top 32 bits are always 0. + var unscaled = ((UInt128)(uint)bits[2] << 64) | ((UInt128)(uint)bits[1] << 32) | (uint)bits[0]; + BinaryPrimitives.WriteUInt128BigEndian( + key.AsSpan(FixedPointMagnitudeOffset, FixedPointMagnitudeSize), unscaled); + + if (decimal.IsNegative(value)) + Complement(key); + return key; + } + + /// Complements a key payload for descending order or a negative magnitude. + private static void Complement(Span bytes) + { + for (int i = 0; i < bytes.Length; i++) bytes[i] = (byte)~bytes[i]; + } + + /// Big-endian with the sign bit flipped, so signed values sort lexicographically. + private static byte[] EncodeInteger(long value, int size) + { + var raw = new byte[size]; + for (int i = size - 1; i >= 0; i--) + { + raw[i] = (byte)(value & 0xFF); + value >>= 8; + } + raw[0] ^= 0x80; + return raw; + } + + /// + /// IEEE bits big-endian with the order-preserving transform: positive numbers flip the high + /// bit, negative numbers invert every byte (so negatives sort below positives, descending). + /// + private static byte[] EncodeFloatBits(long bits, int size) + { + var raw = new byte[size]; + for (int i = size - 1; i >= 0; i--) + { + raw[i] = (byte)(bits & 0xFF); + bits >>= 8; + } + + if (raw[0] < 0x80) // sign bit clear → non-negative value + raw[0] ^= 0x80; + else + Complement(raw); + + return raw; + } + public static object?[] Decode(IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, ReadOnlySpan key) + { + var values = new object?[columns.Count]; + int pos = 0; + + for (int i = 0; i < columns.Count; i++) + { + (ColumnDef column, bool ascending) = columns[i]; + if (pos >= key.Length) break; + + byte flag = key[pos++]; + if (flag == Null(ascending)) + { + values[i] = null; + continue; + } + // Otherwise flag is the start flag. + + if (column.Type == JetDataType.Boolean) + { + if (pos >= key.Length) break; + values[i] = key[pos++] == Boolean(true, ascending); + continue; + } + + // GUID key: the 16 bytes of the GUID's canonical string order, chunked (see IndexKeyCodec). + if (column.Type == JetDataType.Guid) + { + if (ReadChunked(key, ref pos, ascending) is not { Length: 16 } guid) break; + values[i] = new Guid(Convert.ToHexString(guid)); + continue; + } + + int size = FixedKeySize(column.Type); + if (size <= 0 || pos + size > key.Length) + break; // text/binary/unsupported (lossy) — cannot reliably continue + + Span raw = key.Slice(pos, size).ToArray(); + pos += size; + values[i] = DecodeFixed(column, raw, ascending); + } + + return values; + } + + /// Reads a chunked value — the inverse of — from just after + /// its start flag, advancing past it. Null when the key ends inside it or a final chunk + /// claims more than a chunk holds. + private static byte[]? ReadChunked(ReadOnlySpan key, ref int pos, bool ascending) + { + var data = new List(); + while (true) + { + if (pos + ChunkSize + 1 > key.Length) return null; + ReadOnlySpan chunk = key.Slice(pos, ChunkSize); + byte control = key[pos + ChunkSize]; + pos += ChunkSize + 1; + + bool more = control == ChunkContinues; + int real = more ? ChunkSize : ascending ? control : (byte)~control; + if (real > ChunkSize) return null; + foreach (byte b in chunk[..real]) data.Add(ascending ? b : (byte)~b); + if (!more) return [.. data]; + } + } + + private static object DecodeFixed(ColumnDef column, Span raw, bool ascending) + { + switch (column.Type) + { + case JetDataType.Byte: + if (!ascending) raw[0] = (byte)~raw[0]; + return raw[0]; + + case JetDataType.Int16: + return (short)DecodeInteger(raw, ascending); + case JetDataType.Int32: + case JetDataType.Complex: // the complex id, keyed as the Int32 it is + return (int)DecodeInteger(raw, ascending); + case JetDataType.Currency: + return JetTypeCodec.CurrencyFromScaled(DecodeInteger(raw, ascending)); + case JetDataType.Int64: + return DecodeInteger(raw, ascending); + + case JetDataType.Single: + return BitConverter.Int32BitsToSingle((int)DecodeFloatBits(raw, ascending)); + case JetDataType.Double: + return BitConverter.Int64BitsToDouble(DecodeFloatBits(raw, ascending)); + case JetDataType.DateTime: + double serial = BitConverter.Int64BitsToDouble(DecodeFloatBits(raw, ascending)); + return JetTypeCodec.TryFromOaDate(serial, out DateTime date) + ? date + : throw new InvalidDataException($"An index key on '{column.Name}' holds {serial}, which is not a date."); + + case JetDataType.FixedPoint: + return DecodeFixedPoint(raw, ascending, column.Scale); + + default: + throw new NotSupportedException($"Index key type {column.Type} is not decodable."); + } + } + + /// Reverses : a negative key is the complement of the positive + /// form, whose sign byte is and whose 16-byte big-endian + /// magnitude is the value times 10^scale. A negative zero keeps its sign, as the key does. + private static decimal DecodeFixedPoint(Span raw, bool ascending, byte scale) + { + if (!ascending) + Complement(raw); + bool negative = raw[0] != FixedPointNonNegative; + if (negative) + Complement(raw); + + UInt128 unscaled = BinaryPrimitives.ReadUInt128BigEndian( + raw.Slice(FixedPointMagnitudeOffset, FixedPointMagnitudeSize)); + if (unscaled >> 96 != 0) + throw new OverflowException("A Decimal index key's magnitude exceeds the 96 bits System.Decimal holds."); + return new decimal((int)(uint)unscaled, (int)(uint)(unscaled >> 32), (int)(uint)(unscaled >> 64), negative, scale); + } + + /// Reverses the integer key transform (descending = bytes inverted; sign bit flipped; big-endian). + private static long DecodeInteger(Span raw, bool ascending) + { + if (!ascending) + Complement(raw); + raw[0] ^= 0x80; + + long value = 0; + bool negative = (raw[0] & 0x80) != 0; + if (negative) value = -1; // sign-extend + foreach (byte b in raw) value = (value << 8) | b; + return value; + } + + /// Reverses the floating-point key transform, returning the raw IEEE bits big-endian. + private static long DecodeFloatBits(Span raw, bool ascending) + { + if (ascending) + { + if ((raw[0] & 0x80) != 0) raw[0] ^= 0x80; // was positive: undo first-bit flip + else Complement(raw); // was negative: undo full invert + } + else + { + if ((raw[0] & 0x80) == 0) // was positive + { + Complement(raw); + raw[0] ^= 0x80; + } + // was negative: stored as-is + } + + long bits = 0; + foreach (byte b in raw) bits = (bits << 8) | b; + return bits; + } + /// + /// The step's action on each bit. The upper eight are a plain right shift by eight, which makes the + /// operator the familiar (x >> 8) ^ T(x & 0xFF) of a table-driven CRC; the lower eight are the + /// table itself, measured from ACE. + /// + private static ReadOnlySpan ChecksumStepBits => + [ + 0x0580, 0x0F80, 0x1B80, 0x3380, 0x6380, 0xC380, 0x8381, 0x0383, + 0x0001, 0x0002, 0x0004, 0x0008, 0x0010, 0x0020, 0x0040, 0x0080, + ]; + + private static readonly ushort[] ChecksumTable = BuildChecksumTable(); + + private static ushort[] BuildChecksumTable() + { + var table = new ushort[256]; + for (int value = 0; value < 256; value++) + { + ushort result = 0; + for (int bit = 0; bit < 8; bit++) if ((value & (1 << bit)) != 0) result ^= ChecksumStepBits[bit]; + table[value] = result; + } + return table; + } + + /// + /// The checksum over the bytes ACE dropped — everything from on. + /// + /// + /// A key of at most bytes is stored as built. Past that ACE keeps + /// the first and replaces the rest with this value, computed over + /// the bytes it dropped — which is why two long values that share a + /// 508-byte prefix still sort apart instead of colliding. + /// + /// Recovered by measurement, not documentation. Three tails differing in one byte showed the function is + /// affine over GF(2) (L(0xA3) ^ L(0x13) = L(0xB0) exactly), and it proved shift-invariant across 173 + /// observations, so a byte at distance d from the end contributes S^(d-1) of itself. Sweeping all + /// 65,536 polynomials in the usual framings found nothing, because the usual framing is wrong: the standard + /// reflected update is crc = (crc >> 8) ^ T[(crc ^ b) & 0xFF], passing the byte THROUGH the table, + /// while ACE computes crc = (crc >> 8) ^ T[crc & 0xFF] ^ b and injects it raw. The step operator + /// was then solved directly by Gaussian elimination over the measured contributions, and predicts all 657 of + /// them. There is no initial value and no final XOR. + /// + /// + /// It holds where the dropped bytes contain a word-sort record too, which was long assumed + /// uncheckable: the record sits in the part ACE discarded, so what it held looked unobservable, and ACE might + /// have recomputed its position when truncating. Neither is so — the position byte tracks where the mark + /// actually sat, and the checksum over the reconstructed key matches ACE's for both mark characters at + /// positions spread through the value. See . + /// + /// + /// The last byte does not go through a full step. Every byte before it is folded in the usual way, and + /// then the final one is XORed into the high half — it never gets its own shift or table lookup. + /// This was first written as "the terminator is excluded", which is the same thing whenever that + /// byte is 0x00: XOR-ing zero changes nothing. A key ending in text always ends in its 0x00 + /// terminator, and every measurement behind the original rule used one, so the two readings could not be + /// told apart. They diverge the moment the key's last column is numeric — (TEXT, TEXT, LONG) put + /// a data byte there and the keys parted company from ACE's, silently. Re-measured over LONG, CURRENCY + /// and DOUBLE tails across 24 keys: this form matches ACE on every one, and still matches on the all-text + /// keys the old form was derived from. See docs/design/index-key-checksum.md. + /// + /// + private static ushort ComputeChecksum(ReadOnlySpan discarded) + { + ushort crc = 0; + foreach (byte b in discarded[..^1]) crc = (ushort)((crc >> 8) ^ ChecksumTable[crc & 0xFF] ^ b); + return (ushort)(crc ^ (discarded[^1] << 8)); + } +} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/IndexKeyDecoder.cs b/src/LibRed/LibRed.Core/Storage/IndexKeyDecoder.cs deleted file mode 100644 index b05af2575..000000000 --- a/src/LibRed/LibRed.Core/Storage/IndexKeyDecoder.cs +++ /dev/null @@ -1,148 +0,0 @@ -using LibRed.Catalog; -using LibRed.Formats; - -namespace LibRed.Storage; - -/// -/// Decodes the order-preserving key bytes of an index entry back into column values. -/// -/// -/// Each non-boolean column is prefixed by a flag byte (0x7F start / 0x00 null for ascending; -/// 0x80 / 0xFF for descending). Fixed/numeric types use a reversible transform (sign-bit flip -/// + big-endian for integers; an IEEE transform for floating point). TEXT/Binary/GUID keys use -/// Jet's collation encoding, which is lossy and not reversible — decoding stops at the first -/// such column (its value and any following columns are returned as null). -/// -public static class IndexKeyDecoder -{ - private const byte AscBooleanTrue = 0x00; // ascending: true sorts before false - - public static object?[] Decode(IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, ReadOnlySpan key) - { - var values = new object?[columns.Count]; - int pos = 0; - - for (int i = 0; i < columns.Count; i++) - { - (ColumnDef column, bool ascending) = columns[i]; - if (pos >= key.Length) break; - - // Booleans carry no flag byte — the value IS the byte. - if (column.Type == JetDataType.Boolean) - { - byte b = key[pos++]; - values[i] = ascending ? b == AscBooleanTrue : b != AscBooleanTrue; - continue; - } - - byte flag = key[pos++]; - if (flag == (ascending ? IndexKeyFlags.AscNull : IndexKeyFlags.DescNull)) - { - values[i] = null; - continue; - } - // Otherwise flag is the start flag (0x7F / 0x80). - - // GUID key: 8 bytes, a 0x09 marker, 8 bytes, a terminator — the 16 bytes are the GUID's - // canonical string order (see IndexKeyEncoder). Descending inverts every data byte (the 0x09 - // marker stays constant), so undo that here. - if (column.Type == JetDataType.Guid) - { - if (pos + 18 > key.Length) break; - var s = new byte[16]; - for (int j = 0; j < 8; j++) s[j] = ascending ? key[pos + j] : (byte)~key[pos + j]; - for (int j = 0; j < 8; j++) s[8 + j] = ascending ? key[pos + 9 + j] : (byte)~key[pos + 9 + j]; - pos += 18; - values[i] = new Guid(Convert.ToHexString(s)); - continue; - } - - int size = FixedKeySize(column.Type); - if (size <= 0 || pos + size > key.Length) - break; // text/binary/unsupported (lossy) — cannot reliably continue - - Span raw = key.Slice(pos, size).ToArray(); - pos += size; - values[i] = DecodeFixed(column.Type, raw, ascending); - } - - return values; - } - - private static int FixedKeySize(JetDataType type) => type switch - { - JetDataType.Byte => 1, - JetDataType.Int16 => 2, - JetDataType.Int32 => 4, - JetDataType.Single => 4, - JetDataType.Double or JetDataType.DateTime => 8, - JetDataType.Currency or JetDataType.Int64 => 8, - _ => -1, - }; - - private static object DecodeFixed(JetDataType type, Span raw, bool ascending) - { - switch (type) - { - case JetDataType.Byte: - if (!ascending) raw[0] = (byte)~raw[0]; - return raw[0]; - - case JetDataType.Int16: - return (short)DecodeInteger(raw, ascending); - case JetDataType.Int32: - return (int)DecodeInteger(raw, ascending); - case JetDataType.Currency: - return DecodeInteger(raw, ascending) / 10000m; - case JetDataType.Int64: - return DecodeInteger(raw, ascending); - - case JetDataType.Single: - return BitConverter.Int32BitsToSingle((int)DecodeFloatBits(raw, ascending)); - case JetDataType.Double: - return BitConverter.Int64BitsToDouble(DecodeFloatBits(raw, ascending)); - case JetDataType.DateTime: - return DateTime.FromOADate(BitConverter.Int64BitsToDouble(DecodeFloatBits(raw, ascending))); - - default: - throw new NotSupportedException($"Index key type {type} is not decodable."); - } - } - - /// Reverses the integer key transform (descending = bytes inverted; sign bit flipped; big-endian). - private static long DecodeInteger(Span raw, bool ascending) - { - if (!ascending) - for (int i = 0; i < raw.Length; i++) raw[i] = (byte)~raw[i]; - raw[0] ^= 0x80; - - long value = 0; - bool negative = (raw[0] & 0x80) != 0; - if (negative) value = -1; // sign-extend - foreach (byte b in raw) value = (value << 8) | b; - return value; - } - - /// Reverses the floating-point key transform, returning the raw IEEE bits big-endian. - private static long DecodeFloatBits(Span raw, bool ascending) - { - if (ascending) - { - if ((raw[0] & 0x80) != 0) raw[0] ^= 0x80; // was positive: undo first-bit flip - else for (int i = 0; i < raw.Length; i++) raw[i] = (byte)~raw[i]; // was negative: undo full invert - } - else - { - if ((raw[0] & 0x80) == 0) // was positive - { - for (int i = 0; i < raw.Length; i++) raw[i] = (byte)~raw[i]; - raw[0] ^= 0x80; - } - // was negative: stored as-is - } - - long bits = 0; - foreach (byte b in raw) bits = (bits << 8) | b; - return bits; - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/IndexKeyEncoder.cs b/src/LibRed/LibRed.Core/Storage/IndexKeyEncoder.cs deleted file mode 100644 index 94d2f919e..000000000 --- a/src/LibRed/LibRed.Core/Storage/IndexKeyEncoder.cs +++ /dev/null @@ -1,369 +0,0 @@ -using EntityFrameworkCore.Jet.Data; -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.Storage.Types; -using System.Buffers.Binary; -using System.Globalization; -using System.Runtime.InteropServices; - -namespace LibRed.Storage; - -/// -/// Encodes index column values into Jet's order-preserving key bytes — the inverse of -/// . Lexicographic comparison of the produced bytes matches the -/// index's logical order, so a freshly encoded key can be slotted into a leaf by byte compare. -/// -/// -/// Each non-boolean column is prefixed by a flag byte (0x7F start / 0x00 null ascending; -/// 0x80 / 0xFF descending). Fixed/numeric types use the reversible transform (sign-bit flip + -/// big-endian for integers; an IEEE transform for floating point); descending inverts the bytes. -/// GUID keys are encoded byte-faithfully (string-order halves split by 0x09, terminated by 0x08). -/// Text uses Jet's collation; general Binary keys — and DATETIME2, whose stored form is already -/// order-preserving — use the same 0x09-chunked layout for any length. -/// -public static class IndexKeyEncoder -{ - /// Access indexes only the first 255 characters of a Memo (Long Text) value (verified vs ACE). - private const int MemoKeyMaxChars = 255; - - /// - /// The longest index entry ACE stores verbatim. Measured: an entry of exactly 510 bytes comes back - /// byte-for-byte, and one that would be 511 comes back as 510 — the weights cut short and the last two - /// bytes replaced by a value that varies with the string (…0E0602 for one 254-character value, - /// …0EDE2A for the 255-character one). That is a truncated key plus a checksum, which is why two - /// long values never collide, and it is why LibRed cannot simply cut its own key to match. - /// - /// It caps the whole entry rather than each column: two 200-character text columns are about 404 bytes - /// of key each and ACE stores their combined entry hashed at 510. - /// - /// - private const int MaxIndexKeyBytes = 510; - - public static byte[] Encode(IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, object?[] values) => - Encode(columns, values, enforceLengthLimit: true); - - /// - /// The key LibRed would build if ACE had no length limit — the input the truncation works ON. - /// - /// - /// Only the research that is trying to identify ACE's two-byte checksum wants this: recovering the - /// function means pairing what ACE stored against the full key it was derived from, and the ordinary - /// entry point refuses exactly those values. Not a way around the limit — a key this returns is longer - /// than ACE would store and must never be written to a file. - /// - internal static byte[] EncodeWithoutLengthLimit( - IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, object?[] values) => - Encode(columns, values, enforceLengthLimit: false); - - private static byte[] Encode( - IReadOnlyList<(ColumnDef Column, bool Ascending)> columns, object?[] values, bool enforceLengthLimit) - { - var buffer = new List(); - - for (int i = 0; i < columns.Count; i++) - { - (ColumnDef column, bool ascending) = columns[i]; - object? value = values[column.Index]; - - if (column.Type == JetDataType.Boolean) - { - // No flag byte: ascending true sorts before false (0x00 < 0xFF); descending mirrors. - bool b = value is true; - buffer.Add((byte)((b ^ !ascending) ? 0x00 : 0xFF)); - continue; - } - - if (value is null) - { - buffer.Add(ascending ? IndexKeyFlags.AscNull : IndexKeyFlags.DescNull); - continue; - } - - // Text uses Jet's collation: start flag then the collation key body (weights, inline - // ignorable codes, terminator). Descending inverts every byte of that ascending key - // and appends a 0x00 (verified against ACE). - // - // A Memo (Long Text) column IS indexable in Access, and its key is the *same* collation key - // over only the value's first 255 characters — verified vs ACE: a 256- or 300-character memo - // produces byte-for-byte the key of its 255-character prefix. - if (column.Type is JetDataType.Text or JetDataType.Memo) - { - // Weights are implemented for the two General orders plus the locale tailorings in - // JetLocaleTailoring. Refuse anything else up front rather than emit wrong bytes with the - // English table — a wrong key does not fail, it silently disagrees with ACE's. The collation - // is read per-column from the descriptor (0x0B–0x0E). - if (!column.Collation.IsIndexKeyEncodable) - throw new NotSupportedException( - $"Index key encoding for column '{column.Name}' uses collation {column.Collation.Order} " + - $"version {column.Collation.Version}" + - (column.Collation.SortId == 0 ? "" : $" sort id {column.Collation.SortId}") + - ", which is not implemented yet."); - - string text = (string)value; - if (column.Type == JetDataType.Memo && text.Length > MemoKeyMaxChars) - text = text[..MemoKeyMaxChars]; - - var ascendingKey = new List { IndexKeyFlags.AscStart }; - LocaleTailoring? tailoring = JetLocaleTailoring.For(column.Collation); - // The word-sort flag is discarded: it existed only to refuse a key whose record the truncation - // would drop, and that refusal is gone — see the note at the truncation below. - bool encoded = column.Collation.Version == Collation.GeneralVersion - ? JetTextCollationV1.TryEncode(text, ascendingKey, tailoring, out _) - : JetTextCollation.TryEncode(text, ascendingKey, tailoring, out _); - if (!encoded) - throw new NotSupportedException( - $"Text index key '{text}' contains a character with no weight in the {column.Collation.Order} " + - $"v{column.Collation.Version} collation table."); - - if (ascending) - { - buffer.AddRange(ascendingKey); - } - else - { - foreach (byte b in ascendingKey) buffer.Add((byte)~b); - buffer.Add(0x00); - } - continue; - } - - // GUID key (verified against ACE): the start flag, then the 16 GUID bytes in canonical *string* - // order (NOT the mixed-endian .ToByteArray layout), split into two 8-byte halves by a constant - // 0x09 marker, and terminated by 0x08. Fixed 19-byte key; data bytes equal to 0x08/0x09 need no - // escaping because every field is at a fixed offset. Descending inverts every byte EXCEPT the - // 0x09 field marker (which stays constant so the structure is parseable, and doesn't affect - // ordering since it's equal in every key) — verified against ACE. - if (column.Type == JetDataType.Guid) - { - Guid guid = value switch - { - Guid g => g, - byte[] b when b.Length == 16 => new Guid(b), - _ => throw new NotSupportedException($"Cannot encode GUID index key from {value.GetType().Name}."), - }; - byte[] s = Convert.FromHexString(guid.ToString("N")); // 16 bytes, canonical string order - - if (ascending) - { - buffer.Add(IndexKeyFlags.AscStart); // 0x7F - buffer.AddRange(s.AsSpan(0, 8)); - buffer.Add(0x09); - buffer.AddRange(s.AsSpan(8, 8)); - buffer.Add(0x08); - } - else - { - buffer.Add(IndexKeyFlags.DescStart); // 0x80 = ~0x7F - for (int j = 0; j < 8; j++) buffer.Add((byte)~s[j]); - buffer.Add(0x09); // field marker kept as-is - for (int j = 8; j < 16; j++) buffer.Add((byte)~s[j]); - buffer.Add(unchecked((byte)~0x08)); // 0xF7 - } - continue; - } - - // Binary key (verified against ACE's EverythingIsBytes fixture): the start flag, then the raw - // bytes in **8-byte chunks**. Each chunk is 8 bytes (real bytes left-aligned, zero-padded on the - // right) followed by a control byte: 0x09 when another chunk follows (a full 8-byte chunk with - // more to come), otherwise the real-byte count of this final chunk (1..8; 0x08 for a full final - // chunk, 0x00 for empty data). This is the same chunking as the GUID key (a 16-byte value → two - // chunks: 8, 0x09, 8, 0x08); the old fixed 4-byte MSysQueries.Order case is the single-chunk form - // (7F <4B> 00 00 00 00 04). Descending inverts every byte EXCEPT the 0x09 continuation markers - // (which stay constant so structure is parseable and they're equal across keys) — mirrors GUID. - if (column.Type == JetDataType.Binary) - { - EncodeBinaryChunked(buffer, (byte[])value, ascending); - continue; - } - - // DATETIME2 keys the whole 42-byte stored value through that same chunking, rather than folding - // it to a number the way DateTime folds to its OA double — verified against ACE, which stores - // 7F <8B> 09 … over exactly the bytes on the page. It works because the encoding - // is already order-preserving: both numeric fields are zero-padded to 19 digits, so byte order is - // chronological order. Note the value's 42nd byte is a NUL (see JetTypeCodec) and lands in the key. - if (column.Type == JetDataType.DateTimeExtended) - { - EncodeBinaryChunked( - buffer, - JetTypeCodec.EncodeExtendedDateTime(Convert.ToDateTime(value, CultureInfo.InvariantCulture)), - ascending); - continue; - } - - int size = FixedKeySize(column.Type); - if (size <= 0) - throw new NotSupportedException( - $"Index key encoding for {column.Type} (binary collation) is not supported yet."); - - buffer.Add(ascending ? IndexKeyFlags.AscStart : IndexKeyFlags.DescStart); - byte[] raw = EncodeFixed(column, value); - if (!ascending) - for (int j = 0; j < raw.Length; j++) raw[j] = (byte)~raw[j]; - buffer.AddRange(raw); - } - - // Past 510 bytes ACE keeps the first 508 and replaces the rest with a checksum over what it dropped, - // which is why two long values sharing a prefix still sort apart. The limit is on the WHOLE entry, - // not per column: two 200-character text columns weigh about 404 bytes each, comfortably under the - // cap individually, and ACE stores their combined entry truncated. - if (!enforceLengthLimit || buffer.Count <= MaxIndexKeyBytes) return [.. buffer]; - - // A discarded word-sort record used to be refused here, on the reasoning that the record sits in the - // part ACE dropped and so what it held is unobservable — and that if ACE recomputed its position when - // truncating, LibRed's reconstruction would be feeding the checksum the wrong bytes. Both halves are - // now measured and neither holds: ACE does NOT recompute the position (the record's position byte - // tracks where the mark actually sat), and the checksum over LibRed's reconstructed key reproduces - // ACE's exactly — 16 of 16 over two mark characters at eight positions each. The record was never - // unobservable; it just could not be checked until the checksum's own arithmetic was pinned down. - // See docs/design/index-key-checksum.md. - byte[] truncated = new byte[MaxIndexKeyBytes]; - buffer.CopyTo(0, truncated, 0, JetIndexKeyChecksum.KeptBytes); - ushort checksum = JetIndexKeyChecksum.Compute(CollectionsMarshal.AsSpan(buffer)[JetIndexKeyChecksum.KeptBytes..]); - truncated[JetIndexKeyChecksum.KeptBytes] = (byte)(checksum >> 8); - truncated[JetIndexKeyChecksum.KeptBytes + 1] = (byte)checksum; - return truncated; - } - - /// - /// Appends Jet's order-preserving binary index key: start flag, then 8-byte chunks each followed by - /// a control byte (0x09 = "full chunk, more follow"; 1..8 = real-byte count of the final chunk). - /// Descending inverts every byte except the 0x09 continuation markers (verified against ACE). - /// - private static void EncodeBinaryChunked(List buffer, byte[] data, bool ascending) - { - buffer.Add(ascending ? IndexKeyFlags.AscStart : IndexKeyFlags.DescStart); - // ACE represents an empty Binary value by the start flag alone. - if (data.Length == 0) - return; - - int offset = 0; - do - { - int n = Math.Min(8, data.Length - offset); - for (int j = 0; j < 8; j++) - { - byte b = j < n ? data[offset + j] : (byte)0; - buffer.Add(ascending ? b : (byte)~b); - } - offset += n; - - if (offset < data.Length) - buffer.Add(0x09); // continuation marker (constant either way) - else - buffer.Add(ascending ? (byte)n : (byte)~n); // terminator = final-chunk length - } - while (offset < data.Length); - } - - private static int FixedKeySize(JetDataType type) => type switch - { - JetDataType.Byte => 1, - JetDataType.Int16 => 2, - JetDataType.Int32 => 4, - // A complex (multi-value / attachment) column's key is its Int32 complex id, encoded exactly as an - // Int32 — verified against ACE over 43 entries across 9 such indexes in two files, covering - // attachment, Text and Long element types, with no difference in any byte. - JetDataType.Complex => 4, - JetDataType.Single => 4, - JetDataType.Double or JetDataType.DateTime => 8, - // Int64/BIGINT keys like Currency — both are an int64, sign bit flipped, big-endian. Its VARIABLE - // storage does not change that: this dispatch is on the type, not on where the row keeps it. - JetDataType.Currency or JetDataType.Int64 => 8, - JetDataType.FixedPoint => 17, // sign byte + 16-byte big-endian magnitude - _ => -1, - }; - - private static byte[] EncodeFixed(ColumnDef column, object value) - { - var c = CultureInfo.InvariantCulture; - switch (column.Type) - { - case JetDataType.Byte: - return [Convert.ToByte(value, c)]; - case JetDataType.Int16: - return EncodeInteger(Convert.ToInt16(value, c), 2); - case JetDataType.Int32: - case JetDataType.Complex: // the complex id, keyed as the Int32 it is - return EncodeInteger(Convert.ToInt32(value, c), 4); - case JetDataType.Currency: - return EncodeInteger((long)decimal.Round(JetDecimalConverter.ToDecimal(value, c) * 10000m), 8); - case JetDataType.Int64: // BIGINT — verified against ACE across 0, ±1, ±42 and both extremes - return EncodeInteger(Convert.ToInt64(value, c), 8); - case JetDataType.Single: - return EncodeFloatBits(BitConverter.SingleToInt32Bits(Convert.ToSingle(value, c)), 4); - case JetDataType.Double: - return EncodeFloatBits(BitConverter.DoubleToInt64Bits(Convert.ToDouble(value, c)), 8); - case JetDataType.DateTime: - return EncodeFloatBits(BitConverter.DoubleToInt64Bits(Convert.ToDateTime(value, c).ToOADate()), 8); - case JetDataType.FixedPoint: - return EncodeFixedPoint(JetDecimalConverter.ToDecimal(value, c), column.Scale); - default: - throw new NotSupportedException($"Index key type {column.Type} is not encodable."); - } - } - - /// - /// Encodes a FixedPoint (Numeric/Decimal) index key — a sign byte followed by the value's 16-byte - /// big-endian **unscaled magnitude** (|value| × 10^scale, the same integer the row codec stores). - /// A non-negative value uses sign 0xFF; a negative value is the **bitwise complement of the - /// whole 17-byte positive form** (sign becomes 0x00, magnitude is one's-complemented), so byte - /// order equals numeric order: negatives (sign 0x00) precede non-negatives (0xFF), and complementing - /// makes a larger magnitude sort earlier among negatives. Zero encodes as positive. - /// Verified byte-for-byte against ACE (see DecimalKeyEncodingTests). - /// - private static byte[] EncodeFixedPoint(decimal value, byte scale) - { - decimal factor = 1m; - for (int i = 0; i < scale; i++) factor *= 10m; - // Truncated toward zero, matching ACE and — necessarily — JetTypeCodec.EncodeNumeric: quantise a key - // differently from its row and the value is indexed under a number the row does not contain. - decimal magnitude = decimal.Truncate(Math.Abs(value) * factor); - int[] bits = decimal.GetBits(magnitude); // [lo, mid, hi, flags]; magnitude has scale 0 - - var key = new byte[17]; - key[0] = 0xFF; // non-negative marker - // 16-byte big-endian magnitude: the top 32-bit word is always 0 for a System.Decimal, then hi/mid/lo. - BinaryPrimitives.WriteUInt32BigEndian(key.AsSpan(1, 4), 0); - BinaryPrimitives.WriteUInt32BigEndian(key.AsSpan(5, 4), (uint)bits[2]); - BinaryPrimitives.WriteUInt32BigEndian(key.AsSpan(9, 4), (uint)bits[1]); - BinaryPrimitives.WriteUInt32BigEndian(key.AsSpan(13, 4), (uint)bits[0]); - - if (value < 0) - for (int i = 0; i < key.Length; i++) key[i] = (byte)~key[i]; - return key; - } - - /// Big-endian with the sign bit flipped, so signed values sort lexicographically. - private static byte[] EncodeInteger(long value, int size) - { - var raw = new byte[size]; - for (int i = size - 1; i >= 0; i--) - { - raw[i] = (byte)(value & 0xFF); - value >>= 8; - } - raw[0] ^= 0x80; - return raw; - } - - /// - /// IEEE bits big-endian with the order-preserving transform: positive numbers flip the high - /// bit, negative numbers invert every byte (so negatives sort below positives, descending). - /// - private static byte[] EncodeFloatBits(long bits, int size) - { - var raw = new byte[size]; - for (int i = size - 1; i >= 0; i--) - { - raw[i] = (byte)(bits & 0xFF); - bits >>= 8; - } - - if (raw[0] < 0x80) // sign bit clear → non-negative value - raw[0] ^= 0x80; - else - for (int i = 0; i < size; i++) raw[i] = (byte)~raw[i]; - - return raw; - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/IndexPageReader.cs b/src/LibRed/LibRed.Core/Storage/IndexPageReader.cs deleted file mode 100644 index 0a55e747a..000000000 --- a/src/LibRed/LibRed.Core/Storage/IndexPageReader.cs +++ /dev/null @@ -1,140 +0,0 @@ -using LibRed.IO; -using LibRed.Pages; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -internal sealed record CheckedIndexPage( - PageBuffer Buffer, PageType Type, int Owner, int Previous, int Next, int Tail, - int CompressedByteCount, IReadOnlyList<(int Start, int End)> EntryRanges); - -/// Checks the common header, owner, pointers, and entry boundaries of an index page. -internal static class IndexPageReader -{ - internal const int OwnerOffset = 0x04; - internal const int PrevPageOffset = 0x0C; - internal const int NextPageOffset = 0x10; - internal const int ChildTailOffset = 0x14; - internal const int CompressedByteCountOffset = 0x18; - internal const int EntryMaskOffset = 0x1B; - internal const int EntryDataOffset = 0x1E0; - - public static CheckedIndexPage Read(PageChannel channel, int pageNumber, int? expectedOwner) - { - ValidatePageNumber(channel, pageNumber, "index page"); - PageBuffer buffer = channel.ReadPageShared(pageNumber); - var type = (PageType)buffer.ReadByte(0); - if (type is not (PageType.LeafIndexPage or PageType.IntermediateIndexPage)) - throw new InvalidDataException( - $"Page {pageNumber} is type 0x{(byte)type:X2}, not an index page (0x03/0x04)."); - - int owner = buffer.ReadInt32(OwnerOffset); - if (expectedOwner is not null && owner != expectedOwner) - throw new InvalidDataException( - $"Index page {pageNumber} belongs to TDEF {owner}, not TDEF {expectedOwner}."); - - int previous = buffer.ReadInt32(PrevPageOffset); - int next = buffer.ReadInt32(NextPageOffset); - int tail = buffer.ReadInt32(ChildTailOffset); - if (type == PageType.LeafIndexPage) - { - ValidateOptionalPageNumber(channel, previous, "previous leaf"); - ValidateOptionalPageNumber(channel, next, "next leaf"); - } - else - { - ValidatePageNumber(channel, tail, "node child-tail"); - } - - // The shared prefix is measured across the WHOLE entry, trailer included — not just the key. Where - // many rows share a key the trailer's leading bytes are common too (consecutive rows on one data - // page), so ACE compresses those away and the stored remainder can be as little as two bytes. Size - // limits therefore apply to the reconstructed entry, never to what is stored. - int compressed = buffer.ReadUInt16(CompressedByteCountOffset); - - var ranges = new List<(int Start, int End)>(); - int start = 0; - for (int i = EntryMaskOffset; i < EntryDataOffset; i++) - { - byte mask = buffer.ReadByte(i); - for (int bit = 0; bit < 8; bit++) - { - if ((mask & (1 << bit)) == 0) continue; - int end = (i - EntryMaskOffset) * 8 + bit; - if (EntryDataOffset + end > buffer.Length) - throw new InvalidDataException( - $"Index page {pageNumber} entry [{start}, {end}) runs past the end of the page."); - // The first entry is stored whole; every later one is the prefix plus what is stored. - int length = ranges.Count == 0 ? end - start : compressed + (end - start); - if (length < 4) - throw new InvalidDataException( - $"Index page {pageNumber} entry [{start}, {end}) reconstructs to {length} bytes, " + - "too few for its 4-byte trailer."); - ranges.Add((start, end)); - start = end; - } - } - - if (ranges.Count == 0 && compressed != 0) - throw new InvalidDataException($"Empty index page {pageNumber} declares a compressed prefix."); - if (ranges.Count > 0 && compressed > ranges[0].End - ranges[0].Start) - throw new InvalidDataException( - $"Index page {pageNumber} compressed prefix {compressed} exceeds its first entry."); - - var page = new CheckedIndexPage(buffer, type, owner, previous, next, tail, compressed, ranges); - - // Node children have to be read from the RECONSTRUCTED entry, for the same reason. - if (type == PageType.IntermediateIndexPage) - foreach ((_, int child) in DecodeEntries(page)) - ValidatePageNumber(channel, child, "node child"); - - return page; - } - - public static int ReadInt32BigEndian(PageBuffer page, int offset) => - BinaryPrimitives.ReadInt32BigEndian(page.Slice(offset, 4)); - - /// Decodes a checked page's entries in order, decompressing each entry's shared prefix: the first - /// entry is stored whole and its leading CompressedByteCount bytes are the prefix reapplied to every - /// following entry. Yields the full key bytes and the 4-byte big-endian trailer (a leaf entry's row pointer - /// or a node entry's child page). Shared by the cursor's leaf enumeration and the writer's parse so the - /// prefix rule lives in exactly one place. - /// - /// The prefix covers the entry whole, so it can reach into the trailer — with many equal keys the - /// rows are consecutive on one data page and share the trailer's leading bytes too. Both the key and the - /// trailer are therefore taken from the reconstructed entry, never from the stored bytes. - /// - public static IEnumerable<(byte[] Key, int Trailer)> DecodeEntries(CheckedIndexPage page) - { - byte[] prefix = []; - bool first = true; - foreach ((int start, int end) in page.EntryRanges) - { - ReadOnlySpan stored = page.Buffer.Slice(EntryDataOffset + start, end - start); - byte[] entry = first ? stored.ToArray() : Concat(prefix, stored); - if (first) { prefix = entry[..page.CompressedByteCount]; first = false; } - int trailer = BinaryPrimitives.ReadInt32BigEndian(entry.AsSpan(entry.Length - 4)); - yield return (entry[..^4], trailer); - } - } - - private static byte[] Concat(ReadOnlySpan a, ReadOnlySpan b) - { - var result = new byte[a.Length + b.Length]; - a.CopyTo(result); - b.CopyTo(result.AsSpan(a.Length)); - return result; - } - - private static void ValidateOptionalPageNumber(PageChannel channel, int pageNumber, string kind) - { - if (pageNumber != 0) ValidatePageNumber(channel, pageNumber, kind); - } - - private static void ValidatePageNumber(PageChannel channel, int pageNumber, string kind) - { - if (pageNumber <= 0 || pageNumber >= channel.PageCount) - throw new InvalidDataException( - $"Index {kind} pointer {pageNumber} is outside the file's 1..{channel.PageCount - 1} range."); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/IndexTree.cs b/src/LibRed/LibRed.Core/Storage/IndexTree.cs new file mode 100644 index 000000000..d113ddb71 --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/IndexTree.cs @@ -0,0 +1,1255 @@ +using LibRed.Catalog; +using LibRed.Formats; +using LibRed.IO; +using LibRed.Pages; +using System.Buffers.Binary; + +namespace LibRed.Storage; + +/// +/// Maintains an index B-tree on row insert: descends from the root to the target leaf, inserts the key, +/// and — when a page overflows — splits it, promoting a separator into the parent and propagating +/// splits up the tree (the tree grows a level when the root itself splits, the root keeping its page). +/// Leaf pages keep their doubly-linked prev/next chain; pages are written with prefix +/// compression. A key column whose collating order LibRed cannot encode still throws — see +/// for which orders it can. +/// +/// +/// A page entry is [key bytes][4-byte big-endian trailer]: on a leaf the trailer is the row +/// pointer (page<<8 | row) and the key is the column key; on a node the trailer is the child +/// page and the key is a full leaf key (column key ++ row pointer) used as the separator = the maximum +/// key of that child. See §10. +/// +internal sealed class IndexTree(PageChannel channel, TableDefinition table) +{ + private readonly PageChannel _channel = channel; + private readonly TableDefinition _table = table; + private readonly PageAllocator _allocator = channel.Allocator; + private readonly UsageMap _usageMaps = new(channel); + + private readonly record struct Entry(byte[] Key, int Trailer); + + public void AddEntry(IndexDef index, object?[] values, RowId rowId) + { + byte[] key = IndexKeyCodec.Encode(index.Columns, values); + int pointer = rowId.Packed; + byte[] fullKey = WithTrailer(key, pointer); // key ++ 4-byte pointer (what node separators store) + + var path = Descend(index.RootPage, fullKey); // [root, …, leaf] page numbers + InsertIntoLeaf(index, path, key, pointer); + } + + /// + /// Whether the index already contains an entry with this key (ignoring the row pointer) — used to enforce + /// a UNIQUE/PRIMARY index on insert. Descends to the leaf the key belongs in (with the smallest pointer, + /// so we land at/just-before any equal-key entry) and scans forward while keys could still match. The + /// caller skips null keys (Jet allows multiple nulls in a unique index — verified vs ACE). + /// + public bool KeyExists(IndexDef index, object?[] values, int? excludePointer = null) + { + byte[] key = IndexKeyCodec.Encode(index.Columns, values); + int leaf = Descend(index.RootPage, WithTrailer(key, 0), path: null); + int steps = 0; + while (leaf != 0) + { + if (!InLeafChain(ref steps)) + throw new InvalidDataException($"Index leaf chain contains a cycle at page {leaf}."); + ParsedIndexPage page = ReadIndexPage(leaf); + if (page.Type != PageType.LeafIndexPage) + throw new InvalidDataException($"Index leaf chain points to non-leaf page {leaf}."); + foreach (Entry e in page.Entries) + { + int cmp = CompareBytes(e.Key, key); + if (cmp > 0) return false; // sorted past where the key would be — it's absent + // Same key held by a *different* row (for an UPDATE, the row's own entry is excluded). + if (cmp == 0 && e.Trailer != excludePointer) return true; + } + leaf = page.Next; // all keys here sort below it — may continue on the next leaf + } + return false; + } + + /// + /// Seeks the index for the rows whose key equals (an equality lookup): descends + /// the B-tree to the leaf where the key belongs, then walks the leaf chain yielding matching row ids until + /// a larger key is reached. O(log n) descent + O(matches), versus a full table scan. + /// + /// + /// The key encoding is order-preserving but lossy for text/binary collation, so distinct values can + /// share a key — the seek is an access path that may over-return; the caller re-applies the real predicate. + /// + public IEnumerable Seek(IndexDef index, object?[] values) + { + byte[] key = IndexKeyCodec.Encode(index.Columns, values); + int leaf = Descend(index.RootPage, WithTrailer(key, 0), path: null); + int steps = 0; + while (leaf != 0) + { + if (!InLeafChain(ref steps)) + throw new InvalidDataException($"Index leaf chain contains a cycle at page {leaf}."); + ParsedIndexPage page = ReadIndexPage(leaf); + if (page.Type != PageType.LeafIndexPage) + throw new InvalidDataException($"Index leaf chain points to non-leaf page {leaf}."); + foreach (Entry e in page.Entries) + { + int cmp = CompareBytes(e.Key, key); + if (cmp > 0) yield break; // sorted past the key — no more matches + if (cmp == 0) yield return RowId.FromPacked(e.Trailer); + } + leaf = page.Next; // matches may continue on the next leaf + } + } + + /// + /// Seeks the index for the rows whose key lies in the range [, ] + /// (either bound null = open): descends to the bound that sorts first in the index and walks the leaf chain, + /// yielding row ids up to the other bound. The key encoding is order-preserving so this returns the range in + /// index order — ascending by value on an ASC index, descending on a DESC one. Like it may + /// over-return at the boundaries (lossy keys / strict-vs-inclusive) — the caller re-applies the real predicate. + /// + public IEnumerable SeekRange(IndexDef index, object?[]? low, object?[]? high) + { + byte[]? lowKey = low is null ? null : IndexKeyCodec.Encode(index.Columns, low); + byte[]? highKey = high is null ? null : IndexKeyCodec.Encode(index.Columns, high); + + // A DESC index inverts its key bytes, so the low VALUE is the byte-greater key and the tree is walked + // from the high bound down to it. The bounds are stated in values; here on they are byte bounds, so + // swap them and the walk below reads the same way for either direction. + if (!index.Columns[0].Ascending) + (lowKey, highKey) = (highKey, lowKey); + + int leaf = Descend(index.RootPage, WithTrailer(lowKey ?? [], 0), path: null); + int steps = 0; + while (leaf != 0) + { + if (!InLeafChain(ref steps)) + throw new InvalidDataException($"Index leaf chain contains a cycle at page {leaf}."); + ParsedIndexPage page = ReadIndexPage(leaf); + if (page.Type != PageType.LeafIndexPage) + throw new InvalidDataException($"Index leaf chain points to non-leaf page {leaf}."); + foreach (Entry e in page.Entries) + { + if (lowKey is not null && CompareBytes(e.Key, lowKey) < 0) continue; // before the low bound + if (highKey is not null && CompareBytes(e.Key, highKey) > 0) yield break; // past the high bound + yield return RowId.FromPacked(e.Trailer); + } + leaf = page.Next; + } + } + + /// + /// Moves a row's entry when its key changes: removes the old-key entry and inserts the new-key one (the + /// row id is unchanged — Access rewrites rows in place). Honours WITH IGNORE NULL on each side (a row with + /// a null key is simply absent from the index). Used by UPDATE of an indexed column. + /// + public void MoveEntry(IndexDef index, object?[] oldValues, object?[] newValues, RowId rowId) + { + if (!(index.IgnoreNulls && HasNullKey(index, oldValues))) RemoveEntry(index, oldValues, rowId); + if (!(index.IgnoreNulls && HasNullKey(index, newValues))) AddEntry(index, newValues, rowId); + } + + /// Removes a row's entry when the row is deleted — a no-op for a WITH IGNORE NULL index whose + /// key the row was absent from (null key), otherwise . + public void DeleteEntry(IndexDef index, object?[] values, RowId rowId) + { + if (index.IgnoreNulls && HasNullKey(index, values)) return; + RemoveEntry(index, values, rowId); + } + + /// + /// Removes a row's entry from the index. Descends to the entry's leaf, drops it, and rewrites the leaf. + /// No rebalancing of an underfull leaf, and a stale separator (if the removed entry was a leaf's maximum) + /// stays a valid upper bound, so later descents still route correctly — matching Access's lazy delete. A + /// leaf the delete leaves empty is the exception: Access takes it out of the tree, and so does this + /// (see ). + /// + public void RemoveEntry(IndexDef index, object?[] values, RowId rowId) + { + byte[] key = IndexKeyCodec.Encode(index.Columns, values); + int pointer = rowId.Packed; + + List path = Descend(index.RootPage, WithTrailer(key, pointer)); + int leafPage = path[^1]; + CheckedPage page = ReadMutationPage(leafPage, PageType.LeafIndexPage); + (List entries, _) = Parse(page); + + int idx = entries.FindIndex(e => e.Trailer == pointer && CompareBytes(e.Key, key) == 0); + if (idx < 0) + throw new InvalidOperationException( + $"Index '{index.Name}': entry for row {rowId.Page}:{rowId.Row} was not found on leaf {leafPage}."); + entries.RemoveAt(idx); + + if (entries.Count == 0 && UnlinkEmptyLeaf(index, path, page)) return; + + // Removing only shrinks the page, so Build never overflows. The page keeps the prefix length it was + // already stored at: ACE re-compresses a leaf only when it must (see InsertIntoLeaf), and a delete + // never must. Letting Build pick the largest prefix now available instead repacks entries ACE left + // alone — measured as a 4-byte-shorter live region on every leaf a cascading delete touched. + // Dropping an entry can only keep or widen what the rest share, so the stored length stays valid. + WriteOrThrow(leafPage, + Build(PageType.LeafIndexPage, page.Previous, page.Next, tail: 0, level: 0, entries, + page.CompressedByteCount)); + } + + /// + /// Takes a leaf whose last entry has just been removed out of the tree: past it in the leaf chain, its + /// separator gone from the parent node, and the page itself back to the allocator. Returns false when the + /// leaf has to stay, and the caller writes it back empty instead. + /// + /// + /// This is what ACE does. Measured on a two-leaf tree whose low leaf was emptied by a range delete: + /// the surviving leaf comes back with prev = 0, the node keeps only its child-tail pointer with no + /// separator entries left, and the emptied page is no longer reachable from the root. + /// Two shapes keep their empty leaf. A leaf that is the root has nowhere to go — an index with + /// no rows is exactly one empty leaf. And a leaf that is its parent's only remaining child cannot be + /// unlinked without leaving the parent pointing at nothing, a node shape ACE has not been observed to + /// write; an empty leaf is a valid one, so the tree keeps it rather than inventing that. + /// + private bool UnlinkEmptyLeaf(IndexDef index, List path, CheckedPage leaf) + { + if (path.Count < 2) return false; + + int leafPage = path[^1]; + int parentPage = path[^2]; + CheckedPage parent = ReadMutationPage(parentPage, PageType.IntermediateIndexPage); + (List entries, int tail) = Parse(parent); + + int slot = entries.FindIndex(e => e.Trailer == leafPage); + if (slot >= 0) + { + entries.RemoveAt(slot); + } + else if (tail == leafPage) + { + // The tail has no key bound of its own, so the last separator's child takes its place and that + // separator's key — an upper bound on the leaf now leaving — goes with it. + if (entries.Count == 0) return false; + tail = entries[^1].Trailer; + entries.RemoveAt(entries.Count - 1); + } + else + { + throw new InvalidDataException( + $"Index '{index.Name}': node {parentPage} does not point at leaf {leafPage}."); + } + + if (leaf.Previous != 0) + SetSiblingLink(leaf.Previous, _channel.Format.IndexNextPageOffset, leaf.Next, PageType.LeafIndexPage); + if (leaf.Next != 0) + SetSiblingLink(leaf.Next, _channel.Format.IndexPrevPageOffset, leaf.Previous, PageType.LeafIndexPage); + + // Dropping a separator only shrinks the node, so Build never overflows, and the node keeps its prefix + // as a leaf does on a delete (see RemoveEntry). A leaf's parent is one level above the leaves by + // definition. + WriteOrThrow(parentPage, Build(PageType.IntermediateIndexPage, parent.Previous, parent.Next, tail, level: 1, + entries, parent.CompressedByteCount)); + + // The page leaves the index the way AllocateIndexPage brought it in: its bit out of the index's own + // pages map, then released — held until this handle closes, the route ACE takes for a freed page. + _usageMaps.SetBit(index.UsageMap.Row, index.UsageMap.Page, leafPage, set: false); + _allocator.Release(leafPage); + return true; + } + + internal static bool HasNullKey(IndexDef index, object?[] values) => + index.Columns.Any(c => values[c.Column.Index] is null or DBNull); + + /// Descends to the leaf that should hold the key, recording the path from the root. + private List Descend(int rootPage, byte[] fullKey) + { + var path = new List(); + Descend(rootPage, fullKey, path); + return path; + } + + /// Counts one more leaf of a chain walk, and says whether the walk can still be a chain: it cannot + /// visit more leaves than the file has pages, so one that tries is going round a loop. Counting catches the + /// loop as surely as a set of the leaves visited did, without the set a seek allocated every time. + private bool InLeafChain(ref int steps) => ++steps <= _channel.PageCount; + + /// The deepest a real B-tree can be. A node holds at least two children, and a file of at most 2 GB + /// has at most 2^20 pages even at 2 KB each, so no tree is 21 levels deep; this leaves room to spare. + private const int MaxDepth = 64; + + /// Descends to the leaf that should hold the key, returning it, and recording the path from the + /// root in when there is one — a seek needs only the leaf. + /// A descent that goes on past is circling through a corrupt node. Counting + /// levels catches that as surely as remembering every page visited did, without a set allocated per seek — + /// which an index-nested-loop join does once for every outer row. + private int Descend(int rootPage, byte[] fullKey, List? path) + { + int pageNumber = rootPage; + for (int depth = 0; ; depth++) + { + if (depth > MaxDepth) + throw new InvalidDataException($"Index descent contains a cycle at page {pageNumber}."); + path?.Add(pageNumber); + ParsedIndexPage page = ReadIndexPage(pageNumber); + if (page.Type == PageType.LeafIndexPage) return pageNumber; + if (page.Type != PageType.IntermediateIndexPage) + throw new InvalidDataException($"Index descent reached non-index page {pageNumber}."); + + int child = page.Tail; + foreach (Entry e in page.Entries) + if (CompareBytes(e.Key, fullKey) >= 0) { child = e.Trailer; break; } + pageNumber = child; + } + } + + /// An index page decoded: its type, entries, node child-tail, and the leaf header fields a + /// rewrite of the page has to carry forward (previous/next links and the stored prefix length). + private sealed record ParsedIndexPage( + PageType Type, int Owner, List Entries, int Tail, int Next, int Previous, int Compressed); + + /// Reads an index page as decoded entries, served from the channel's parsed-page cache on a repeat + /// visit — a B-tree descent re-reads its root/internal pages on every seek, so caching the decode (not just + /// the bytes) removes both the page copy and the entry decode. A cached parse is dropped whenever the bytes + /// change (any channel) or the page is evicted, so a hit is always consistent with the bytes. + private ParsedIndexPage ReadIndexPage(int pageNumber) + { + if (_channel.TryGetParsedPage(pageNumber, out object? cached) && cached is ParsedIndexPage hit) + { + if (hit.Owner != _table.DefinitionPage) + throw new InvalidDataException( + $"Index page {pageNumber} belongs to TDEF {hit.Owner}, not TDEF {_table.DefinitionPage}."); + return hit; + } + + CheckedPage page = Read(_channel, pageNumber, _table.DefinitionPage); + (List entries, int tail) = Parse(page); + var parsed = new ParsedIndexPage( + page.Type, page.Owner, entries, tail, page.Next, page.Previous, page.CompressedByteCount); + _channel.SetParsedPage(pageNumber, parsed); + return parsed; + } + + /// + /// The decoded page a read-modify-write is about to rewrite, with an owned entry list the caller may + /// mutate. Serves the parse the descent just made rather than reading and decoding the page a second time. + /// + /// + /// Every insert descends to its leaf and then rewrites it, and the two steps each parsed the page — + /// decoding a fresh byte[] for every entry on it, twice. This reuses the first parse. + /// The list is copied before it is handed over, because the cached parse is shared with every + /// other reader of the file and 's contract forbids mutating it. The + /// copy is shallow, which is the point: is an immutable struct holding a reference to + /// its key, so copying the list shares the key arrays and allocates one array of structs instead of one + /// array per entry. Keys are never written through, only read by . + /// This does not weaken the revalidation the write paths perform (§10.2). A cached parse exists only + /// while the bytes behind it are unchanged — any write, from any channel, drops it, and a page buffered in + /// an open transaction's overlay is never served — so a hit carries the same guarantee a re-read would, and + /// the type and owner recorded in it are checked exactly as before. + /// + private ParsedIndexPage ReadMutablePage(int pageNumber, PageType expectedType) + { + ParsedIndexPage parsed = ReadIndexPage(pageNumber); + if (parsed.Type != expectedType) + throw new InvalidDataException( + $"Index mutation expected page {pageNumber} to be {expectedType}, but found {parsed.Type}."); + + // Room for the one entry InsertIntoLeaf inserts: copied at its exact size, the list doubled its array on that + // insert — about 19 KB per index per row on a full leaf, a fifth of everything an insert allocated. + var entries = new List(parsed.Entries.Count + 1); + entries.AddRange(parsed.Entries); + return parsed with { Entries = entries }; + } + + private void InsertIntoLeaf(IndexDef index, List path, byte[] key, int pointer) + { + int leafPage = path[^1]; + ParsedIndexPage page = ReadMutablePage(leafPage, PageType.LeafIndexPage); + List entries = page.Entries; + + // Insert in key order (key then pointer tiebreaker) — the full leaf key is key ++ pointer. Compared + // without materialising each entry's concatenation: this scan runs over every entry on the page for + // every row inserted, so building one throwaway array per comparison was the write path's largest + // single allocator. + byte[] fullKey = WithTrailer(key, pointer); + int pos = 0; + while (pos < entries.Count && CompareWithTrailer(entries[pos].Key, entries[pos].Trailer, fullKey) < 0) pos++; + entries.Insert(pos, new Entry(key, pointer)); + + // ACE compresses a leaf only when it has to, and splits only when compressing is not enough. A page + // starts uncompressed and stores whole keys; when the next entry will not fit, the shared prefix is + // computed and the page rewritten IN PLACE, which typically frees most of it; filling then continues + // at the shorter size; and only when the compressed page fills does it split. Watching a sequential + // load shows the cycle twice — a leaf reaching 400 entries at 9 bytes each with 16 bytes left, then + // reading 410 entries at 6 bytes with 1153 free, then splitting at 602 (see §10.3). + // + // Rebuilding at the largest available prefix on every write instead would be smaller, but it is not + // what ACE writes, and the tail page of a sequential load is the visible difference. + int share = Share(entries); + int keep = Math.Min(page.Compressed, share); // the new key may not share the old prefix + + if (Build(PageType.LeafIndexPage, page.Previous, page.Next, tail: 0, level: 0, entries, keep) is { } asIs) + { + _channel.WritePage(leafPage, KeepTail(leafPage, asIs), AsBuilt(page, entries, keep)); + return; + } + + if (share > keep + && Build(PageType.LeafIndexPage, page.Previous, page.Next, tail: 0, level: 0, entries, share) + is { } compressed) + { + _channel.WritePage(leafPage, KeepTail(leafPage, compressed), AsBuilt(page, entries, share)); + return; + } + + // Where to cut. Splitting down the middle is right when keys arrive all over the range, because the + // lower half's free space is room for the next key near it. When the new entry is the page's MAXIMUM + // it is waste instead: nothing sorts below a maximum, so half the page is stranded for ever. ACE + // splits at the right edge in that case — the page stays full and the new entry starts a fresh one — + // which is why a sequentially loaded index of ACE's packs its leaves to capacity and LibRed's used to + // settle near half (see docs/format/page-03-04-index-btree.md §10.5 for the measured comparison). + // + // AutoNumber and identity keys are ascending by construction, so this is the ordinary case. The + // condition cannot fire on a random insert, which is why the general behaviour is unchanged. + // + // Compressing did not make room, but ACE still compresses first: the page's old entries are rewritten + // in place at the prefix they share, and the split is made over that. It shows past each half's live + // end, where the compressed entries stand. The left half then stays at that prefix — unless the new + // entry became its first, when ACE writes it whole. (A new first entry that does not split the page + // keeps the prefix: verified both ways.) + var old = new List(entries); + old.RemoveAt(pos); + int oldShare = Share(old); + int stored = Math.Max(page.Compressed, oldShare); + if (oldShare > page.Compressed) + WriteOrThrow(leafPage, Build(PageType.LeafIndexPage, page.Previous, page.Next, 0, 0, old, oldShare)); + + // Everywhere else ACE cuts that compressed page at its byte midpoint — every old entry that STARTS before + // it stays left, so the entry straddling it does too — and the new entry joins whichever half its key + // falls in. A key landing below the cut therefore leaves one MORE entry behind than a key landing on or + // above it. With equal entries that is the old entries halved, the odd one left: a 17-entry leaf keeps + // 10 for a key at any position from 1 to 8 and 9 from 9 to 16, and a 602-entry one keeps 302 for a key at + // 1 or 50 and 301 at 301, 302 or 400. It is bytes, not entries, when keys differ in length: a 91-entry + // leaf of 38- to 40-byte entries keeps 45, not 46. A new FIRST entry is the exception: then the entries + // including it are halved by count, rounding up — 9 of 18, 302 of 603 (§10.5). + int splitAt; + if (pos == entries.Count - 1) splitAt = entries.Count - 1; + else if (pos == 0) splitAt = (entries.Count + 1) / 2; + else + { + long total = -(long)stored * (old.Count - 1); + int trailerSize = _channel.Format.IndexEntryTrailerSize; + foreach (Entry e in old) total += e.Key.Length + trailerSize; + int oldLeft = 0; + for (long start = 0; oldLeft < old.Count && start * 2 < total; oldLeft++) + start += old[oldLeft].Key.Length + trailerSize - (oldLeft == 0 ? 0 : stored); + splitAt = pos < oldLeft ? oldLeft + 1 : oldLeft; + } + + SplitAndPropagate(index, path, path.Count - 1, entries, PageType.LeafIndexPage, + page.Previous, page.Next, splitAt, leftPrefix: pos == 0 ? 0 : stored, newFirst: pos == 0); + } + + /// + /// Splits the (leaf or node) page at into two, writes both, then promotes a + /// separator into the parent — splitting parents in turn, or turning the root into the node over both halves. + /// + /// The index whose tree is being split. + /// The pages from the root down to the one being split, one per level. + /// Which entry of is the page to split. + /// That page's entries, in key order, including the one just inserted. + /// Leaf or node — what the two halves are written as. + /// The split page's left sibling, for the leaf chain. + /// Its right sibling. + /// How many entries stay on the left page; negative for the default half. A leaf split + /// always sets it (see InsertIntoLeaf); a node split sets it for its right-edge case (see InsertSeparator), + /// where the entry at is the one promoted. + /// The prefix a leaf's left half is written at; null for the largest available, + /// which is what the right half and a node's halves are written at. + /// Whether the entry just inserted is a leaf's first. + private void SplitAndPropagate(IndexDef index, List path, int level, List entries, + PageType type, int prev, int next, int splitAt = -1, int? leftPrefix = null, bool newFirst = false) + { + // The root never moves: when it splits, both halves go to new pages, left first, and the root page is + // rewritten as the node over them — so the index-data block's root pointer stays as it is. The left half + // is the root's page carried over, so what lies past its live end is what the root held there. + int leftPage = level == 0 ? AllocateIndexPage(index) : path[level]; + int rightPage = AllocateIndexPage(index); + int nodeLevel = path.Count - 1 - level; // height above the leaves of the page being split + + byte[] promoted; + if (type == PageType.LeafIndexPage) + { + // The left page always fits: at worst it is the page as it stood before the insert that + // overflowed it, and that fitted. + int mid = splitAt < 0 ? entries.Count / 2 : splitAt; + var left = entries.GetRange(0, mid); + var right = entries.GetRange(mid, entries.Count - mid); + promoted = WithTrailer(left[^1].Key, left[^1].Trailer); // left's max full key + + // Right first, so both halves read the page as it stood: the right half takes its dead bytes from it + // when the new entry became the left half's first, a new page's zeros otherwise (verified both ways). + WriteOrThrow(rightPage, Build(type, leftPage, next, tail: 0, nodeLevel, right), + tailFrom: newFirst ? path[level] : null); + WriteOrThrow(leftPage, Build(type, prev, rightPage, tail: 0, nodeLevel, left, leftPrefix), + tailFrom: path[level]); + } + else + { + // Node split: the middle entry's key is promoted; its child becomes the left node's tail. + int mid = splitAt < 0 ? entries.Count / 2 : splitAt; + Entry middle = entries[mid]; + var left = entries.GetRange(0, mid); + var right = entries.GetRange(mid + 1, entries.Count - mid - 1); + promoted = middle.Key; + int oldTail = _splitTail; + + byte[] leftBytes = KeepTail(path[level], Build(type, prev, rightPage, tail: middle.Trailer, nodeLevel, left) + ?? throw new NotSupportedException("An index node still overflows after a split.")); + LeaveMiddleBehind(leftBytes, prev, rightPage, nodeLevel, left, middle); + _channel.WritePage(leftPage, leftBytes); + WriteOrThrow(rightPage, Build(type, leftPage, next, tail: oldTail, nodeLevel, right)); + } + if (next != 0) SetSiblingLink(next, _channel.Format.IndexPrevPageOffset, rightPage, type); // the old next page's back-link + + if (level == 0) + { + // The root split: the root becomes the node [promoted -> left] with the right page as its tail. Its + // bytes past the new live end stay as they were, as on any rewrite of a page in place. + WriteOrThrow(path[0], Build(PageType.IntermediateIndexPage, 0, 0, tail: rightPage, nodeLevel + 1, + [new Entry(promoted, leftPage)])); + return; + } + + InsertSeparator(index, path, level - 1, leftPage, promoted, rightPage); + } + + /// Inserts a promoted separator into the parent node; repoints the old child to the new right + /// page and splits the parent if it overflows. + private void InsertSeparator(IndexDef index, List path, int level, int oldChild, byte[] promoted, int newRight) + { + int parentPage = path[level]; + CheckedPage page = ReadMutationPage(parentPage, PageType.IntermediateIndexPage); + (List entries, int tail) = Parse(page); + + // The old child's pointer is repointed first, so the page ACE compresses in place when the separator + // does not fit already carries it (see below). + int slot = entries.FindIndex(e => e.Trailer == oldChild); + if (slot >= 0) entries[slot] = entries[slot] with { Trailer = newRight }; + else tail = newRight; // oldChild was the tail + var old = new List(entries); + entries.Insert(slot >= 0 ? slot : entries.Count, new Entry(promoted, oldChild)); + + // A node fills, compresses and splits exactly as a leaf does (see InsertIntoLeaf). + int parentLevel = path.Count - 1 - level; + int share = Share(entries); + int keep = Math.Min(page.CompressedByteCount, share); + if (Build(PageType.IntermediateIndexPage, page.Previous, page.Next, tail, parentLevel, entries, keep) + is { } asIs) + { + _channel.WritePage(parentPage, KeepTail(parentPage, asIs)); + return; + } + + if (share > keep + && Build(PageType.IntermediateIndexPage, page.Previous, page.Next, tail, parentLevel, entries, share) + is { } compressed) + { + _channel.WritePage(parentPage, KeepTail(parentPage, compressed)); + return; + } + + // As a leaf does, the full node is compressed in place before it splits (see InsertIntoLeaf). + int oldShare = Share(old); + if (oldShare > page.CompressedByteCount) + WriteOrThrow(parentPage, + Build(PageType.IntermediateIndexPage, page.Previous, page.Next, tail, parentLevel, old, oldShare)); + + // A node has a right-edge split too: when the new separator is its last entry — the child that split was + // the tail, as on every split of an ascending load — the left node keeps every old entry but the last, + // that one is promoted, and the new separator starts the right node alone (verified vs ACE: a node of 16 + // separators taking a 17th is cut 15, promote 1, 1). Anywhere else it is cut in the middle. + _splitTail = tail; + SplitAndPropagate(index, path, level, entries, PageType.IntermediateIndexPage, page.Previous, page.Next, + splitAt: slot < 0 ? entries.Count - 2 : -1); + } + + private int _splitTail; // carries a node's tail into SplitAndPropagate + + /// + /// The parse of a leaf has just made from at + /// , with 's links: what decoding the written page gives, so the + /// write can hand it to the channel rather than have the next insert into the leaf decode every entry again. + /// + /// Consecutive inserts mostly land on one leaf, and each write dropped the parse the next one needed: + /// re-decoding the leaf — hundreds of entries, an array each — was two-fifths of what an insert allocated. + /// Every field is what Build writes: the leaf's own owner, no child tail, and a prefix of 0 for a lone entry. + /// The entry list becomes the shared parse, so the caller must not touch it again. + /// CachedParseMatchesPage is the check that the two agree. + private ParsedIndexPage AsBuilt(ParsedIndexPage read, List entries, int prefix) => + read with + { + Type = PageType.LeafIndexPage, + Owner = _table.DefinitionPage, + Entries = entries, + Tail = 0, + Compressed = entries.Count <= 1 ? 0 : prefix, + }; + + /// Whether the parse cached for is exactly what decoding the page's bytes + /// gives now — the invariant relies on — or null when no parse of one of this table's + /// index pages is cached there. For tests. + internal bool? CachedParseMatchesPage(int pageNumber) + { + if (!_channel.TryGetParsedPage(pageNumber, out object? cached) || cached is not ParsedIndexPage hit + || hit.Owner != _table.DefinitionPage) + return null; + CheckedPage page = Read(_channel, pageNumber, _table.DefinitionPage); + (List entries, int tail) = Parse(page); + return hit.Type == page.Type && hit.Owner == page.Owner && hit.Tail == tail + && hit.Next == page.Next && hit.Previous == page.Previous && hit.Compressed == page.CompressedByteCount + && hit.Entries.Count == entries.Count + && hit.Entries.Zip(entries).All(p => p.First.Trailer == p.Second.Trailer + && p.First.Key.AsSpan().SequenceEqual(p.Second.Key)); + } + + /// Parses a checked page's entries, decompressing their shared prefix. + private static (List Entries, int Tail) Parse(CheckedPage page) + { + var entries = new List(page.EntryRanges.Count); + foreach ((byte[] key, int trailer) in DecodeEntries(page)) + entries.Add(new Entry(key, trailer)); + return (entries, page.Tail); + } + + /// Revalidates a page immediately before mutation, closing the gap between B-tree descent and + /// the final read-modify-write operation. + private CheckedPage ReadMutationPage(int pageNumber, PageType expectedType) + { + CheckedPage page = Read(_channel, pageNumber, _table.DefinitionPage); + if (page.Type != expectedType) + throw new InvalidDataException( + $"Index mutation expected page {pageNumber} to be {expectedType}, but found {page.Type}."); + return page; + } + + /// Builds a page from entries; null if they overflow the page. Leaf and node pages alike are + /// prefix-compressed at the length the caller gives, and a node carries its height above the leaves at + /// , matching what Access writes. (An isolation test showed a node's height is not + /// required — Access reads a node with 0x1A=0 just fine; it is kept purely for byte-faithfulness. The + /// one hard requirement is a leaf's 0x1A=0 and the leaf-chain offsets at 0x0C/0x10.) + /// Leaf or node. + /// The page's left sibling at the same level. + /// Its right sibling. + /// The page's trailing pointer — a node's rightmost child. + /// A node's height above the leaves; 0 on a leaf. + /// The entries to write, in key order. + /// The shared-prefix length to store the entries at. Null computes the largest + /// available, which is what a split writes. It must not exceed what the entries actually share. + private byte[]? Build(PageType type, int prev, int next, int tail, int level, List entries, + int? prefix = null) => + BuildPage(_channel.Format, _table.DefinitionPage, type, prev, next, tail, level, entries, prefix); + + /// An empty leaf — a new index's root, owned by the table at . + internal static byte[] EmptyLeaf(JetFormatBase format, int owner) => + BuildPage(format, owner, PageType.LeafIndexPage, prev: 0, next: 0, tail: 0, level: 0, [])!; + + /// for a page owned by the table at . + private static byte[]? BuildPage(JetFormatBase format, int owner, PageType type, int prev, int next, int tail, + int level, List entries, int? prefix = null) + { + int pageSize = format.PageSize; + var page = new byte[pageSize]; + PageHeader.WriteType(page, type); + page[format.IndexLevelOffset] = (byte)level; // 0 on a leaf; the node's height above the leaves otherwise + WriteInt32Le(page, format.IndexOwnerOffset, owner); + WriteInt32Le(page, format.IndexPrevPageOffset, prev); + WriteInt32Le(page, format.IndexNextPageOffset, next); + WriteInt32Le(page, format.IndexChildTailOffset, tail); + + // A single entry has no common-prefix compression — ACE writes 0 here (the whole key with itself would + // otherwise "compress" to its full length, which ACE does not do for one entry). Otherwise the caller + // chooses, for a node as for a leaf: see InsertIntoLeaf for when a page is compressed at all. + int compress = entries.Count <= 1 ? 0 : prefix ?? Share(entries); + BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.IndexCompressedByteCountOffset, 2), (ushort)compress); + + int pos = format.IndexEntryDataOffset; + bool first = true; + Span trailer = stackalloc byte[format.IndexEntryTrailerSize]; + // Bit k of the entry mask ends an entry at offset k of the entry-data region. + Span entryMask = page.AsSpan(format.IndexEntryMaskOffset, format.IndexEntryDataOffset - format.IndexEntryMaskOffset); + foreach (Entry e in entries) + { + // The prefix covers the entry WHOLE — key ++ trailer — so where many rows share a key it reaches + // past the key into the row pointer, and what is stored is the tail of that concatenation. ACE + // writes leaves this way and DecodeEntries reads them back the same way; taking + // the tail of the key alone throws on exactly those pages. + int skip = first ? 0 : compress; + first = false; + int keySkip = Math.Min(skip, e.Key.Length); + int trailerSkip = skip - keySkip; + int len = e.Key.Length - keySkip + trailer.Length - trailerSkip; + if (pos + len > pageSize) return null; // overflow + + BinaryPrimitives.WriteInt32BigEndian(trailer, e.Trailer); + e.Key.AsSpan(keySkip).CopyTo(page.AsSpan(pos)); + trailer[trailerSkip..].CopyTo(page.AsSpan(pos + e.Key.Length - keySkip)); + pos += len; + + BitmapBits.Set(entryMask, pos - format.IndexEntryDataOffset, true); + } + + BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.IndexFreeSpaceOffset, 2), (ushort)(pageSize - pos)); + return page; + } + + /// Repoints the index-data block's B-tree root — when a new table's foreign-key index is given its + /// root (a split never moves one) — through . + /// + /// A wide table's definition spans continuation pages, and the data blocks sit past the column names — + /// well beyond the first page for a 255-column table. The block is therefore found in the stitched + /// definition (the absolute coordinate space the descriptors use), and only its bytes are written back, + /// mapped to whichever pages actually hold them. Nothing changes length, so no re-split is needed. + /// + internal void UpdateIndexRoot(IndexDef index, int newRoot) + { + (byte[] tdef, IReadOnlyList continuations, int block) = LocateIndexBlock(index); + Span data = tdef.AsSpan(block, _channel.Format.IndexDataBlockSize); + TableDefinition.WriteIndexTree(data, _channel.Format, newRoot, index.UsageMap.Row, index.UsageMap.Page); + _table.WriteIntoDefinition(_channel, continuations, block, data); + // The root pointer is part of the definition every other handle caches: until they reload it they + // descend from the old root — by now only a page inside the tree — and an insert through one would + // split that page as if it were the root and write it over this one. + _channel.MarkSchemaChanged(); + } + + /// Walks the stitched definition (stats → column descriptors → column names → data blocks) to + /// the index's data block, returning the buffer, its continuation pages, and the block's absolute + /// offset. A wide table's blocks sit past the column names, well beyond the first page. + private (byte[] Definition, IReadOnlyList Continuations, int BlockOffset) LocateIndexBlock(IndexDef index) + { + (byte[] tdef, IReadOnlyList continuations) = ReadDefinition(); + JetFormatBase format = _channel.Format; + TableDefinition.Regions regions = TableDefinition.Regions.Of(tdef, format); + + // Bounded like every region before it, because this offset decides where UpdateIndexRoot writes: an + // ordinal past the blocks would otherwise repoint whatever follows them. + int block = TableDefinition.CheckedRegionEnd( + regions.DataBlocks, index.RealIndexOrdinal + 1, format.IndexDataBlockSize, tdef.Length, + "index-data blocks") + - format.IndexDataBlockSize; + return (tdef, continuations, block); + } + + /// Allocates a fresh B-tree page for and records it in the index's own + /// pages usage map, exactly as Access does — the map then covers every page of the index's B-tree, not + /// just the root. (Reads use the B-tree's own links, so this is for byte-faithfulness and to feed Access's + /// own maintenance, not for LibRed's own traversal.) + private int AllocateIndexPage(IndexDef index) + { + int page = _allocator.Allocate(); + _usageMaps.SetBit(index.UsageMap.Row, index.UsageMap.Page, page, set: true); + return page; + } + + /// Reads the table definition, stitching continuation pages into one contiguous buffer, and + /// returns the continuation page numbers in chain order. + private (byte[] Definition, IReadOnlyList ContinuationPages) ReadDefinition() + { + (PageBuffer buffer, IReadOnlyList continuations) = + TableDefinition.ReadChain(_channel, _table.DefinitionPage); + return (buffer.Span.ToArray(), continuations); + } + + /// Patches one end of a neighbouring page's sibling link — + /// or — without disturbing its entries. A split repairs the back-link of the page + /// it pushed right, leaf or node; unlinking an emptied leaf repairs both of its neighbours. + private void SetSiblingLink(int pageNumber, int offset, int target, PageType type) + { + CheckedPage checkedPage = Read(_channel, pageNumber, _table.DefinitionPage); + if (checkedPage.Type != type) + throw new InvalidDataException( + $"Sibling pointer targets page {pageNumber}, a {checkedPage.Type} where a {type} was expected."); + byte[] page = checkedPage.Buffer.Span.ToArray(); + WriteInt32Le(page, offset, target); + _channel.WritePage(pageNumber, page); + } + + /// The page to write. + /// The built page; null when its entries overflowed. + /// The page whose dead bytes the write carries over — this page itself, unless the + /// content is moving here from another (a split root's left half). + private void WriteOrThrow(int pageNumber, byte[]? page, int? tailFrom = null) => + _channel.WritePage(pageNumber, KeepTail(tailFrom ?? pageNumber, page ?? throw new NotSupportedException( + "An index page still overflows after a split (a key wider than half a page)."))); + + /// + /// Carries the destination's bytes past the new live end into a freshly built page, because that is what + /// Access leaves behind. + /// + /// + /// works in a zeroed buffer and fills it only as far as the last entry, so + /// everything beyond is written back as zeros. ACE instead edits a page in place: it moves the free-space + /// pointer and lets the bytes the entries used to occupy stand. Measured on an insert that splits three + /// levels of [Order Details]: on every index page ACE rewrote (both leaves and nodes) the region + /// past the new live end still held the original bytes, and on the one page it took fresh from the + /// allocator that region was zero. So the rule is zero-fill on allocation, never clear again. + /// Only a rewrite of this index's own page qualifies — a page just recycled from somewhere else + /// still carries the previous owner's type and owner id, and ACE zero-fills that one, which is what + /// building in a clean buffer already does. + /// The bytes kept are dead: they sit past the free-space boundary, so no reader reaches them. + /// Keeping them is for byte-faithfulness with ACE, not for meaning. + /// + private byte[] KeepTail(int pageNumber, byte[] built) => Merge(built, ImageOf(pageNumber)); + + /// What a rewrite of keeps past its live end: the page's bytes when it + /// is already one of this table's index pages, zeros otherwise (see ). + private byte[] ImageOf(int pageNumber) + { + if (pageNumber >= _channel.PageCount) return new byte[_channel.PageSize]; + + // A leaf turning node — the root, when it splits — still counts as the same page rewritten. + ReadOnlySpan existing = _channel.ReadPage(pageNumber).Span; + if (PageHeader.ReadType(existing) is not (PageType.LeafIndexPage or PageType.IntermediateIndexPage) + || ReadOwner(existing, _channel.Format) != _table.DefinitionPage) + return new byte[_channel.PageSize]; + return existing.ToArray(); + } + + /// Carries 's bytes past 's live end into it. + private byte[] Merge(byte[] built, byte[] image) + { + int liveEnd = LiveEnd(built); + image.AsSpan(liveEnd).CopyTo(built.AsSpan(liveEnd)); + return built; + } + + /// Stores a split node's promoted middle entry just past its left half's live end, at the page's + /// prefix: ACE writes it into the page and then ends the page before it. + private void LeaveMiddleBehind(byte[] leftBytes, int prev, int right, int level, List left, Entry middle) + { + int prefix = ReadCompressedByteCount(leftBytes, _channel.Format); + byte[] withMiddle = Build(PageType.IntermediateIndexPage, prev, right, middle.Trailer, level, [.. left, middle], prefix)!; + int leftEnd = LiveEnd(leftBytes); + withMiddle.AsSpan(leftEnd, LiveEnd(withMiddle) - leftEnd).CopyTo(leftBytes.AsSpan(leftEnd)); + } + + /// Where a built page's entries end: everything past it is free space. + private int LiveEnd(byte[] page) => + _channel.PageSize - ReadFreeSpace(page, _channel.Format); + + // The trailer is carried as an int and written big-endian (WriteInt32Be), so the helpers below that build or + // compare key ++ trailer without a page work in sizeof(int); the page layout uses the format's + // IndexEntryTrailerSize. + private static byte[] WithTrailer(byte[] key, int trailer) + { + var result = new byte[key.Length + sizeof(int)]; + key.CopyTo(result, 0); + WriteInt32Be(result, key.Length, trailer); + return result; + } + + /// + /// Fills an empty index from , leaving exactly the pages that inserting them + /// one at a time in key order would leave, but building a page only when its layout changes rather than + /// rewriting it for every entry. + /// + /// The empty index to fill. + /// (key, row pointer) pairs with a NullKey marker; any order. Sorted here. + /// Enforce uniqueness — adjacent equal keys after the sort, null keys exempt + /// (Jet's uniqueness is over the non-null keys only). + /// + /// This is what ACE's CREATE INDEX writes: its tree is the incremental one for sorted input, + /// byte for byte — the root keeps its page, leaves fill uncompressed, are compressed in place when full and + /// split at the right edge, and the nodes above fill, compress and split the same way, all allocated in the + /// order those splits happen (verified vs ACE). Sorted input only ever reaches the right edge of the tree, + /// so the one page per level being filled there — the spine — is all that is held; every page to its left is + /// finished, and written, when it is split off. + /// Whether an entry fits is decided by arithmetic, not by calling : Build + /// allocates a page-sized array, so probing with it would allocate two pages per entry and lose exactly + /// what this exists to save. mirrors Build's layout — entry data starts at + /// 0x1E0, the first entry stores its whole key, the rest drop the shared prefix, and each carries a + /// 4-byte trailer (see page-01/page-03-04 §10.2–10.3). A page is built only when the prefix it is stored at + /// changes, when it splits, and at the end: the incremental path's writes between those only extend the + /// same layout, so they leave nothing past the live end that a later write keeps. + /// + internal void BulkBuild(IndexDef index, List<(byte[] Key, int Pointer, bool NullKey)> entries, bool rejectDuplicates) + { + if (entries.Count == 0) + return; // the caller's freshly created empty root is already the correct tree + + entries.Sort((a, b) => CompareEntries(a.Key, a.Pointer, b.Key, b.Pointer)); + + if (rejectDuplicates) + for (int i = 1; i < entries.Count; i++) + if (!entries[i].NullKey && !entries[i - 1].NullKey + && CompareBytes(entries[i].Key, entries[i - 1].Key) == 0) + throw new InvalidOperationException( + $"Cannot create unique index '{index.Name}' on '{_table.Name}': duplicate key values exist."); + + // The empty leaf the caller created is the root, and stays the root however tall the tree grows. + var spine = new List { new(index.RootPage, previous: 0, ImageOf(index.RootPage)) }; + foreach ((byte[] key, int pointer, _) in entries) + Append(index, spine, level: 0, new Entry(key, pointer)); + + for (int level = 0; level < spine.Count; level++) + { + SpinePage page = spine[level]; + _channel.WritePage(page.Number, Merge(BuildSpine(page, level, next: 0, page.Entries, page.Stored), page.Image)); + } + } + + /// The page being filled at one level of a : its entries, the prefix they + /// are stored at, and what the incremental path's writes would have left past the live end. + private sealed class SpinePage(int number, int previous, byte[] image) + { + public int Number { get; } = number; + public int Previous { get; } = previous; + public byte[] Image { get; set; } = image; + public List Entries { get; } = []; + public long KeyBytes { get; set; } + public int Stored { get; set; } // the prefix the page is stored at (0x18), as the last write left it + public int Tail { get; set; } // a node's rightmost child + } + + /// Appends an entry to the page at of the spine as an insert of the page's + /// new maximum would: at the prefix the page is stored at, compressed in place when that is what makes room, + /// and split at the right edge when nothing does (see , ). + private void Append(IndexDef index, List spine, int level, Entry entry) + { + SpinePage page = spine[level]; + page.Entries.Add(entry); + page.KeyBytes += entry.Key.Length; + + int share = Share(page.Entries); + int keep = Math.Min(page.Stored, share); + if (LeafFits(page.Entries.Count, page.KeyBytes, keep)) { StoreAt(page, level, keep); return; } + if (share > keep && LeafFits(page.Entries.Count, page.KeyBytes, share)) { StoreAt(page, level, share); return; } + + // Full: the page as it stood, compressed in place when its entries share more than it is stored at, and + // then split with the new entry starting the next page. + page.Entries.RemoveAt(page.Entries.Count - 1); + page.KeyBytes -= entry.Key.Length; + Materialize(page, level, page.Entries, page.Stored); + int oldShare = Share(page.Entries); + if (oldShare > page.Stored) + { + page.Stored = oldShare; + Materialize(page, level, page.Entries, oldShare); + } + SplitSpine(index, spine, level, entry); + } + + /// Moves a spine page to . When that changes its layout, the write before + /// the entry just appended is what stands past the new live end, so it is made concrete first. + private void StoreAt(SpinePage page, int level, int prefix) + { + if (prefix == page.Stored) return; + Materialize(page, level, page.Entries.GetRange(0, page.Entries.Count - 1), page.Stored); + page.Stored = prefix; + } + + /// What writing at over the page leaves. + private void Materialize(SpinePage page, int level, List entries, int prefix) => + page.Image = Merge(BuildSpine(page, level, next: 0, entries, prefix), page.Image); + + private byte[] BuildSpine(SpinePage page, int level, int next, List entries, int prefix) => + Build(level == 0 ? PageType.LeafIndexPage : PageType.IntermediateIndexPage, page.Previous, next, page.Tail, + level, entries, prefix) + ?? throw new NotSupportedException("An index page overflows (a key wider than half a page)."); + + /// The right-edge split of the spine page at , as + /// makes it: the finished left half is written, the new entry starts the + /// right half, which takes the page's place on the spine, and the separator goes up a level — or, when the + /// page is the root, both halves move out and the root becomes the node over them. + private void SplitSpine(IndexDef index, List spine, int level, Entry entry) + { + SpinePage page = spine[level]; + bool root = level == spine.Count - 1; + int left = root ? AllocateIndexPage(index) : page.Number; + int right = AllocateIndexPage(index); + + byte[] leftBytes, promoted; + int rightTail; + if (level == 0) + { + // Every old entry stays, at the prefix the page is stored at; a copy of the last is promoted. + Entry last = page.Entries[^1]; + promoted = WithTrailer(last.Key, last.Trailer); + leftBytes = Merge(Build(PageType.LeafIndexPage, page.Previous, right, tail: 0, level: 0, page.Entries, + page.Stored)!, page.Image); + rightTail = 0; + } + else + { + // A node's right-edge split: its last entry is promoted, that entry's child becoming the left node's + // tail, and the new separator starts the right node under the old tail. + Entry middle = page.Entries[^1]; + List kept = page.Entries.GetRange(0, page.Entries.Count - 1); + promoted = middle.Key; + leftBytes = Merge(Build(PageType.IntermediateIndexPage, page.Previous, right, middle.Trailer, level, kept)!, + page.Image); + LeaveMiddleBehind(leftBytes, page.Previous, right, level, kept, middle); + rightTail = page.Tail; + } + _channel.WritePage(left, leftBytes); + + var next = new SpinePage(right, previous: left, ImageOf(right)) { Tail = rightTail }; + next.Entries.Add(entry); + next.KeyBytes = entry.Key.Length; + spine[level] = next; + + if (root) + { + // The root keeps its page and becomes the node over both halves; its image stays past the live end. + var top = new SpinePage(page.Number, previous: 0, page.Image) { Tail = right }; + top.Entries.Add(new Entry(promoted, left)); + top.KeyBytes = promoted.Length; + spine.Add(top); + return; + } + + spine[level + 1].Tail = right; + Append(index, spine, level + 1, new Entry(promoted, left)); + } + + /// The prefix every entry shares — the first and last, since they are in key order. None for a + /// single entry, which ACE stores whole. + private static int Share(List entries) => + entries.Count <= 1 ? 0 : CommonPrefixLength(entries[0].Key, entries[^1].Key); + + /// Whether entries totalling of key data fit + /// one leaf at prefix — Build's layout, without building anything. + private bool LeafFits(int count, long keyBytes, int prefix) => + _channel.Format.IndexEntryDataOffset + keyBytes + (long)_channel.Format.IndexEntryTrailerSize * count + - (long)prefix * (count - 1) <= _channel.PageSize; + + /// Orders two entries as their stored key ++ trailer bytes compare, without building either. + internal static int CompareEntries(byte[] aKey, int aTrailer, byte[] bKey, int bTrailer) + { + int shared = Math.Min(aKey.Length, bKey.Length); + for (int i = 0; i < shared; i++) + if (aKey[i] != bKey[i]) return aKey[i] - bKey[i]; + + // The keys agree as far as the shorter runs, so the shorter one's trailer meets the longer one's + // remaining key bytes — the same crossing CompareWithTrailer handles, and why this cannot be + // "compare keys, then compare trailers". + return aKey.Length == bKey.Length + ? CompareTrailers(aTrailer, bTrailer) + : aKey.Length < bKey.Length + ? -CompareWithTrailer(bKey, bTrailer, WithTrailer(aKey, aTrailer)) + : CompareWithTrailer(aKey, aTrailer, WithTrailer(bKey, bTrailer)); + } + + private static int CompareTrailers(int a, int b) + { + for (int shift = 24; shift >= 0; shift -= 8) + { + int x = (byte)(a >> shift), y = (byte)(b >> shift); + if (x != y) return x - y; + } + + return 0; + } + + /// + /// Compares ++ its 4-byte big-endian against + /// , byte for byte, without building the concatenation. + /// + /// + /// Exactly CompareBytes(WithTrailer(key, trailer), other), and it has to be: comparing the keys + /// alone and then breaking the tie on the trailer is NOT the same relation. + /// compares the shared prefix and only then falls back to length, so when one key is a prefix of another + /// the real comparison runs on into the trailer bytes — precisely the case a naive rewrite gets wrong, and + /// it would misplace an entry rather than fail. + /// + internal static int CompareWithTrailer(byte[] key, int trailer, ReadOnlySpan other) + { + int total = key.Length + sizeof(int); + int n = Math.Min(total, other.Length); + for (int i = 0; i < n; i++) + { + // Past the key, read the trailer's bytes most-significant first, as WriteInt32Be lays them out. + int mine = i < key.Length ? key[i] : (byte)(trailer >> (8 * (sizeof(int) - 1 - (i - key.Length)))); + if (mine != other[i]) return mine - other[i]; + } + + return total - other.Length; + } + + private static int CommonPrefixLength(byte[] a, byte[] b) + { + int n = Math.Min(a.Length, b.Length), i = 0; + while (i < n && a[i] == b[i]) i++; + return i; + } + + private static int CompareBytes(ReadOnlySpan a, ReadOnlySpan b) + { + int n = Math.Min(a.Length, b.Length); + for (int i = 0; i < n; i++) + if (a[i] != b[i]) return a[i] - b[i]; + return a.Length - b.Length; + } + + private static void WriteInt32Le(byte[] page, int offset, int value) => BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(offset, 4), value); + private static void WriteInt32Be(byte[] page, int offset, int value) => BinaryPrimitives.WriteInt32BigEndian(page.AsSpan(offset, 4), value); + + /// A checked index page. Each of is a stored entry's offsets within the page. + internal sealed record CheckedPage( + PageBuffer Buffer, JetFormatBase Format, PageType Type, int Owner, int Previous, int Next, int Tail, + int CompressedByteCount, IReadOnlyList<(int Start, int End)> EntryRanges); + + + + /// An index page's owner field: the TDEF page of the table whose index it belongs to. Unchecked — the + /// caller has established that is an index page. + internal static int ReadOwner(ReadOnlySpan page, JetFormatBase format) => + BinaryPrimitives.ReadInt32LittleEndian(page.Slice(format.IndexOwnerOffset, sizeof(int))); + + /// An index page's free-space field: the bytes past the end of its entries. + internal static int ReadFreeSpace(ReadOnlySpan page, JetFormatBase format) => + BinaryPrimitives.ReadUInt16LittleEndian(page.Slice(format.IndexFreeSpaceOffset, sizeof(ushort))); + + /// An index page's compressed-byte count: the prefix every entry after the first shares with the first. + internal static int ReadCompressedByteCount(ReadOnlySpan page, JetFormatBase format) => + BinaryPrimitives.ReadUInt16LittleEndian(page.Slice(format.IndexCompressedByteCountOffset, sizeof(ushort))); + + /// An index page's previous and next sibling pointers, 0 at either end of a level. + internal static (int Previous, int Next) ReadSiblings(ReadOnlySpan page, JetFormatBase format) => + (BinaryPrimitives.ReadInt32LittleEndian(page.Slice(format.IndexPrevPageOffset, sizeof(int))), + BinaryPrimitives.ReadInt32LittleEndian(page.Slice(format.IndexNextPageOffset, sizeof(int)))); + + internal static CheckedPage Read(PageChannel channel, int pageNumber, int? expectedOwner) + { + ValidatePageNumber(channel, pageNumber, "index page"); + PageBuffer buffer = channel.ReadPageShared(pageNumber); + var type = PageHeader.ReadType(buffer.Span); + if (type is not (PageType.LeafIndexPage or PageType.IntermediateIndexPage)) + throw new InvalidDataException( + $"Page {pageNumber} is type 0x{(ushort)type:X4}, not an index page (0x0103/0x0104)."); + + JetFormatBase format = channel.Format; + int owner = ReadOwner(buffer.Span, format); + if (expectedOwner is not null && owner != expectedOwner) + throw new InvalidDataException( + $"Index page {pageNumber} belongs to TDEF {owner}, not TDEF {expectedOwner}."); + + (int previous, int next) = ReadSiblings(buffer.Span, format); + int tail = buffer.ReadInt32(format.IndexChildTailOffset); + if (type == PageType.LeafIndexPage) + { + ValidateOptionalPageNumber(channel, previous, "previous leaf"); + ValidateOptionalPageNumber(channel, next, "next leaf"); + } + else + { + ValidatePageNumber(channel, tail, "node child-tail"); + } + + // The shared prefix is measured across the WHOLE entry, trailer included — not just the key. Where + // many rows share a key the trailer's leading bytes are common too (consecutive rows on one data + // page), so ACE compresses those away and the stored remainder can be as little as two bytes. Size + // limits therefore apply to the reconstructed entry, never to what is stored. + int compressed = ReadCompressedByteCount(buffer.Span, format); + + // Bit k of the entry mask ends an entry at offset k of the entry-data region. + int dataOffset = format.IndexEntryDataOffset; + ReadOnlySpan entryMask = buffer.Slice(format.IndexEntryMaskOffset, dataOffset - format.IndexEntryMaskOffset); + var ranges = new List<(int Start, int End)>(); + int start = dataOffset; + for (int bit = BitmapBits.NextSetBit(entryMask, 0); bit >= 0; bit = BitmapBits.NextSetBit(entryMask, bit + 1)) + { + int end = dataOffset + bit; + if (end > buffer.Length) + throw new InvalidDataException( + $"Index page {pageNumber} entry [{start}, {end}) runs past the end of the page."); + // The first entry is stored whole; every later one is the prefix plus what is stored. + int length = ranges.Count == 0 ? end - start : compressed + (end - start); + if (length < format.IndexEntryTrailerSize) + throw new InvalidDataException( + $"Index page {pageNumber} entry [{start}, {end}) reconstructs to {length} bytes, " + + $"too few for its {format.IndexEntryTrailerSize}-byte trailer."); + ranges.Add((start, end)); + start = end; + } + + if (ranges.Count == 0 && compressed != 0) + throw new InvalidDataException($"Empty index page {pageNumber} declares a compressed prefix."); + if (ranges.Count > 0 && compressed > ranges[0].End - ranges[0].Start) + throw new InvalidDataException( + $"Index page {pageNumber} compressed prefix {compressed} exceeds its first entry."); + + var page = new CheckedPage(buffer, format, type, owner, previous, next, tail, compressed, ranges); + + // Node children have to be read from the RECONSTRUCTED entry, for the same reason. + if (type == PageType.IntermediateIndexPage) + foreach (int child in Trailers(page)) + ValidatePageNumber(channel, child, "node child"); + + return page; + } + + /// Decodes a checked page's entries in order, decompressing each entry's shared prefix: the first + /// entry is stored whole and its leading CompressedByteCount bytes are the prefix reapplied to every + /// following entry. Yields the full key bytes and the big-endian trailer (a leaf entry's row pointer + /// or a node entry's child page). Shared by the cursor's leaf enumeration and the writer's parse so the + /// prefix rule lives in exactly one place. + /// + /// The prefix covers the entry whole, so it can reach into the trailer — with many equal keys the + /// rows are consecutive on one data page and share the trailer's leading bytes too. Both the key and the + /// trailer are therefore taken from the reconstructed entry, never from the stored bytes. + /// + internal static IEnumerable<(byte[] Key, int Trailer)> DecodeEntries(CheckedPage page) + { + byte[] prefix = []; + bool first = true; + foreach ((int start, int end) in page.EntryRanges) + { + ReadOnlySpan stored = page.Buffer.Slice(start, end - start); + ReadOnlySpan lead = first ? [] : prefix; + // The key is the reconstructed entry less its trailer, copied straight from the prefix and the stored + // bytes into the one array returned — when the prefix reaches into the trailer, all of it is prefix. + var key = new byte[lead.Length + stored.Length - page.Format.IndexEntryTrailerSize]; + int fromLead = Math.Min(lead.Length, key.Length); + lead[..fromLead].CopyTo(key); + stored[..(key.Length - fromLead)].CopyTo(key.AsSpan(fromLead)); + int trailer = Trailer(lead, stored, page.Format); + if (first) { prefix = stored[..page.CompressedByteCount].ToArray(); first = false; } + yield return (key, trailer); + } + } + + /// A checked page's entry trailers in order — a node's child pages, a leaf's row pointers — read from the + /// reconstructed entries as reads them, without building the keys. + internal static List Trailers(CheckedPage page) + { + var trailers = new List(page.EntryRanges.Count); + ReadOnlySpan lead = []; + for (int i = 0; i < page.EntryRanges.Count; i++) + { + (int start, int end) = page.EntryRanges[i]; + ReadOnlySpan stored = page.Buffer.Slice(start, end - start); + trailers.Add(Trailer(lead, stored, page.Format)); + if (i == 0) lead = stored[..page.CompressedByteCount]; + } + return trailers; + } + + /// The big-endian trailer of the entry stored as after + /// — the shared prefix it omits, empty for the first entry — read from the entry as + /// reconstructed, since the prefix can reach into it. + private static int Trailer(ReadOnlySpan lead, ReadOnlySpan stored, JetFormatBase format) + { + int length = lead.Length + stored.Length, trailer = 0; + for (int i = length - format.IndexEntryTrailerSize; i < length; i++) + trailer = (trailer << 8) | (i < lead.Length ? lead[i] : stored[i - lead.Length]); + return trailer; + } + + private static void ValidateOptionalPageNumber(PageChannel channel, int pageNumber, string kind) + { + if (pageNumber != 0) ValidatePageNumber(channel, pageNumber, kind); + } + + private static void ValidatePageNumber(PageChannel channel, int pageNumber, string kind) + { + if (pageNumber <= 0 || pageNumber >= channel.PageCount) + throw new InvalidDataException( + $"Index {kind} pointer {pageNumber} is outside the file's 1..{channel.PageCount - 1} range."); + } + +} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/IndexWriter.cs b/src/LibRed/LibRed.Core/Storage/IndexWriter.cs deleted file mode 100644 index ca26d679d..000000000 --- a/src/LibRed/LibRed.Core/Storage/IndexWriter.cs +++ /dev/null @@ -1,882 +0,0 @@ -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.IO; -using LibRed.Pages; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -/// -/// Maintains an index B-tree on row insert: descends from the root to the target leaf, inserts the key, -/// and — when a page overflows — splits it, promoting a separator into the parent and propagating -/// splits up the tree (growing a new root when the root itself splits, and repointing the index-data -/// block's root). Leaf pages keep their doubly-linked prev/next chain; pages are written with prefix -/// compression. Indexes whose key columns need unsupported (text/binary) collation still throw. -/// -/// -/// A page entry is [key bytes][4-byte big-endian trailer]: on a leaf the trailer is the row -/// pointer (page<<8 | row) and the key is the column key; on a node the trailer is the child -/// page and the key is a full leaf key (column key ++ row pointer) used as the separator = the maximum -/// key of that child. See §10. -/// -public sealed class IndexWriter(PageChannel channel, TableDef table) -{ - private const int FreeSpaceOffset = 0x02; - private const int OwnerOffset = 0x04; - private const int PrevPageOffset = 0x0C; // leaf page: previous (lower-key) leaf - private const int NextPageOffset = 0x10; // leaf page: next (higher-key) leaf — Access walks this for COUNT/scan - private const int ChildTailOffset = 0x14; - private const int CompressedByteCountOffset = 0x18; - private const int LevelOffset = 0x1A; // 0 on a leaf, its height above the leaves on a node - private const int EntryMaskOffset = 0x1B; - private const int EntryDataOffset = 0x1E0; - - private readonly PageChannel _channel = channel; - private readonly TableDef _table = table; - private readonly PageAllocator _allocator = new(channel); - private readonly UsageMapWriter _usageMaps = new(channel); - - private readonly record struct Entry(byte[] Key, int Trailer); - - public void AddEntry(IndexDef index, object?[] values, RowId rowId) - { - byte[] key = IndexKeyEncoder.Encode(index.Columns, values); - int pointer = (rowId.Page << 8) | rowId.Row; - byte[] fullKey = WithTrailer(key, pointer); // key ++ 4-byte pointer (what node separators store) - - var path = Descend(index.RootPage, fullKey); // [root, …, leaf] page numbers - InsertIntoLeaf(index, path, key, pointer); - } - - /// - /// Whether the index already contains an entry with this key (ignoring the row pointer) — used to enforce - /// a UNIQUE/PRIMARY index on insert. Descends to the leaf the key belongs in (with the smallest pointer, - /// so we land at/just-before any equal-key entry) and scans forward while keys could still match. The - /// caller skips null keys (Jet allows multiple nulls in a unique index — verified vs ACE). - /// - public bool KeyExists(IndexDef index, object?[] values, int? excludePointer = null) - { - byte[] key = IndexKeyEncoder.Encode(index.Columns, values); - int leaf = Descend(index.RootPage, WithTrailer(key, 0))[^1]; - var visitedLeaves = new HashSet(); - while (leaf != 0) - { - if (!visitedLeaves.Add(leaf)) - throw new InvalidDataException($"Index leaf chain contains a cycle at page {leaf}."); - ParsedIndexPage page = ReadIndexPage(leaf); - if (page.Type != PageType.LeafIndexPage) - throw new InvalidDataException($"Index leaf chain points to non-leaf page {leaf}."); - foreach (Entry e in page.Entries) - { - int cmp = CompareBytes(e.Key, key); - if (cmp > 0) return false; // sorted past where the key would be — it's absent - // Same key held by a *different* row (for an UPDATE, the row's own entry is excluded). - if (cmp == 0 && e.Trailer != excludePointer) return true; - } - leaf = page.Next; // all keys here sort below it — may continue on the next leaf - } - return false; - } - - /// - /// Seeks the index for the rows whose key equals (an equality lookup): descends - /// the B-tree to the leaf where the key belongs, then walks the leaf chain yielding matching row ids until - /// a larger key is reached. O(log n) descent + O(matches), versus a full table scan. - /// - /// - /// The key encoding is order-preserving but lossy for text/binary collation, so distinct values can - /// share a key — the seek is an access path that may over-return; the caller re-applies the real predicate. - /// - public IEnumerable Seek(IndexDef index, object?[] values) - { - byte[] key = IndexKeyEncoder.Encode(index.Columns, values); - int leaf = Descend(index.RootPage, WithTrailer(key, 0))[^1]; - var visitedLeaves = new HashSet(); - while (leaf != 0) - { - if (!visitedLeaves.Add(leaf)) - throw new InvalidDataException($"Index leaf chain contains a cycle at page {leaf}."); - ParsedIndexPage page = ReadIndexPage(leaf); - if (page.Type != PageType.LeafIndexPage) - throw new InvalidDataException($"Index leaf chain points to non-leaf page {leaf}."); - foreach (Entry e in page.Entries) - { - int cmp = CompareBytes(e.Key, key); - if (cmp > 0) yield break; // sorted past the key — no more matches - if (cmp == 0) yield return new RowId(e.Trailer >> 8, e.Trailer & 0xFF); - } - leaf = page.Next; // matches may continue on the next leaf - } - } - - /// - /// Seeks the index for the rows whose key lies in the range [, ] - /// (either bound null = open): descends to the low bound's leaf and walks the leaf chain, yielding row ids - /// while the key does not exceed the high bound. The key encoding is order-preserving so this returns the - /// range in order. Like it may over-return at the boundaries (lossy keys / strict-vs- - /// inclusive) — the caller re-applies the real predicate. - /// - public IEnumerable SeekRange(IndexDef index, object?[]? low, object?[]? high) - { - byte[]? lowKey = low is null ? null : IndexKeyEncoder.Encode(index.Columns, low); - byte[]? highKey = high is null ? null : IndexKeyEncoder.Encode(index.Columns, high); - - int leaf = Descend(index.RootPage, WithTrailer(lowKey ?? [], 0))[^1]; - var visitedLeaves = new HashSet(); - while (leaf != 0) - { - if (!visitedLeaves.Add(leaf)) - throw new InvalidDataException($"Index leaf chain contains a cycle at page {leaf}."); - ParsedIndexPage page = ReadIndexPage(leaf); - if (page.Type != PageType.LeafIndexPage) - throw new InvalidDataException($"Index leaf chain points to non-leaf page {leaf}."); - foreach (Entry e in page.Entries) - { - if (lowKey is not null && CompareBytes(e.Key, lowKey) < 0) continue; // before the low bound - if (highKey is not null && CompareBytes(e.Key, highKey) > 0) yield break; // past the high bound - yield return new RowId(e.Trailer >> 8, e.Trailer & 0xFF); - } - leaf = page.Next; - } - } - - /// - /// Moves a row's entry when its key changes: removes the old-key entry and inserts the new-key one (the - /// row id is unchanged — Access rewrites rows in place). Honours WITH IGNORE NULL on each side (a row with - /// a null key is simply absent from the index). Used by UPDATE of an indexed column. - /// - public void MoveEntry(IndexDef index, object?[] oldValues, object?[] newValues, RowId rowId) - { - if (!(index.IgnoreNulls && HasNullKey(index, oldValues))) RemoveEntry(index, oldValues, rowId); - if (!(index.IgnoreNulls && HasNullKey(index, newValues))) AddEntry(index, newValues, rowId); - } - - /// Removes a row's entry when the row is deleted — a no-op for a WITH IGNORE NULL index whose - /// key the row was absent from (null key), otherwise . - public void DeleteEntry(IndexDef index, object?[] values, RowId rowId) - { - if (index.IgnoreNulls && HasNullKey(index, values)) return; - RemoveEntry(index, values, rowId); - } - - /// - /// Removes a row's entry from the index. Descends to the entry's leaf, drops it, and rewrites the leaf. - /// No rebalancing: an underfull or empty leaf is fine, and a stale separator (if the removed entry was a - /// leaf's maximum) stays a valid upper bound, so later descents still route correctly — matching Access's - /// lazy delete. - /// - public void RemoveEntry(IndexDef index, object?[] values, RowId rowId) - { - byte[] key = IndexKeyEncoder.Encode(index.Columns, values); - int pointer = (rowId.Page << 8) | rowId.Row; - - List path = Descend(index.RootPage, WithTrailer(key, pointer)); - int leafPage = path[^1]; - CheckedIndexPage page = ReadMutationPage(leafPage, PageType.LeafIndexPage); - (List entries, _) = Parse(page); - - int idx = entries.FindIndex(e => e.Trailer == pointer && CompareBytes(e.Key, key) == 0); - if (idx < 0) - throw new InvalidOperationException( - $"Index '{index.Name}': entry for row {rowId.Page}:{rowId.Row} was not found on leaf {leafPage}."); - entries.RemoveAt(idx); - - // Removing only shrinks the page, so Build never overflows. The page keeps the prefix length it was - // already stored at: ACE re-compresses a leaf only when it must (see InsertIntoLeaf), and a delete - // never must. Letting Build pick the largest prefix now available instead repacks entries ACE left - // alone — measured as a 4-byte-shorter live region on every leaf a cascading delete touched. - // Dropping an entry can only keep or widen what the rest share, so the stored length stays valid. - WriteOrThrow(leafPage, - Build(PageType.LeafIndexPage, page.Previous, page.Next, tail: 0, level: 0, entries, - page.CompressedByteCount)); - } - - private static bool HasNullKey(IndexDef index, object?[] values) => - index.Columns.Any(c => values[c.Column.Index] is null or DBNull); - - /// Descends to the leaf that should hold the key, recording the path from the root. - private List Descend(int rootPage, byte[] fullKey) - { - var path = new List(); - var visited = new HashSet(); - int pageNumber = rootPage; - while (true) - { - if (!visited.Add(pageNumber)) - throw new InvalidDataException($"Index descent contains a cycle at page {pageNumber}."); - path.Add(pageNumber); - ParsedIndexPage page = ReadIndexPage(pageNumber); - if (page.Type == PageType.LeafIndexPage) return path; - if (page.Type != PageType.IntermediateIndexPage) - throw new InvalidDataException($"Index descent reached non-index page {pageNumber}."); - - int child = page.Tail; - foreach (Entry e in page.Entries) - if (CompareBytes(e.Key, fullKey) >= 0) { child = e.Trailer; break; } - pageNumber = child; - } - } - - /// An index page decoded: its type, entries, node child-tail, and the leaf header fields a - /// rewrite of the page has to carry forward (previous/next links and the stored prefix length). - private sealed record ParsedIndexPage( - PageType Type, int Owner, List Entries, int Tail, int Next, int Previous, int Compressed); - - /// Reads an index page as decoded entries, served from the channel's parsed-page cache on a repeat - /// visit — a B-tree descent re-reads its root/internal pages on every seek, so caching the decode (not just - /// the bytes) removes both the page copy and the entry decode. A cached parse is dropped whenever the bytes - /// change (any channel) or the page is evicted, so a hit is always consistent with the bytes. - private ParsedIndexPage ReadIndexPage(int pageNumber) - { - if (_channel.TryGetParsedPage(pageNumber, out object? cached) && cached is ParsedIndexPage hit) - { - if (hit.Owner != _table.DefinitionPage) - throw new InvalidDataException( - $"Index page {pageNumber} belongs to TDEF {hit.Owner}, not TDEF {_table.DefinitionPage}."); - return hit; - } - - CheckedIndexPage page = IndexPageReader.Read(_channel, pageNumber, _table.DefinitionPage); - (List entries, int tail) = Parse(page); - var parsed = new ParsedIndexPage( - page.Type, page.Owner, entries, tail, page.Next, page.Previous, page.CompressedByteCount); - _channel.SetParsedPage(pageNumber, parsed); - return parsed; - } - - /// - /// The decoded page a read-modify-write is about to rewrite, with an owned entry list the caller may - /// mutate. Serves the parse the descent just made rather than reading and decoding the page a second time. - /// - /// - /// Every insert descends to its leaf and then rewrites it, and the two steps each parsed the page — - /// decoding a fresh byte[] for every entry on it, twice. This reuses the first parse. - /// The list is copied before it is handed over, because the cached parse is shared with every - /// other reader of the file and 's contract forbids mutating it. The - /// copy is shallow, which is the point: is an immutable struct holding a reference to - /// its key, so copying the list shares the key arrays and allocates one array of structs instead of one - /// array per entry. Keys are never written through, only read by . - /// This does not weaken the revalidation the write paths perform (§10.2). A cached parse exists only - /// while the bytes behind it are unchanged — any write, from any channel, drops it, and a page buffered in - /// an open transaction's overlay is never served — so a hit carries the same guarantee a re-read would, and - /// the type and owner recorded in it are checked exactly as before. - /// - private ParsedIndexPage ReadMutablePage(int pageNumber, PageType expectedType) - { - ParsedIndexPage parsed = ReadIndexPage(pageNumber); - if (parsed.Type != expectedType) - throw new InvalidDataException( - $"Index mutation expected page {pageNumber} to be {expectedType}, but found {parsed.Type}."); - - return parsed with { Entries = [.. parsed.Entries] }; - } - - private void InsertIntoLeaf(IndexDef index, List path, byte[] key, int pointer) - { - int leafPage = path[^1]; - ParsedIndexPage page = ReadMutablePage(leafPage, PageType.LeafIndexPage); - List entries = page.Entries; - - // Insert in key order (key then pointer tiebreaker) — the full leaf key is key ++ pointer. Compared - // without materialising each entry's concatenation: this scan runs over every entry on the page for - // every row inserted, so building one throwaway array per comparison was the write path's largest - // single allocator. - byte[] fullKey = WithTrailer(key, pointer); - int pos = 0; - while (pos < entries.Count && CompareWithTrailer(entries[pos].Key, entries[pos].Trailer, fullKey) < 0) pos++; - entries.Insert(pos, new Entry(key, pointer)); - - // ACE compresses a leaf only when it has to, and splits only when compressing is not enough. A page - // starts uncompressed and stores whole keys; when the next entry will not fit, the shared prefix is - // computed and the page rewritten IN PLACE, which typically frees most of it; filling then continues - // at the shorter size; and only when the compressed page fills does it split. Watching a sequential - // load shows the cycle twice — a leaf reaching 400 entries at 9 bytes each with 16 bytes left, then - // reading 410 entries at 6 bytes with 1153 free, then splitting at 602 (see §10.3). - // - // Rebuilding at the largest available prefix on every write instead would be smaller, but it is not - // what ACE writes, and the tail page of a sequential load is the visible difference. - int share = entries.Count <= 1 ? 0 : CommonPrefixLength(entries[0].Key, entries[^1].Key); - int keep = Math.Min(page.Compressed, share); // the new key may not share the old prefix - - if (Build(PageType.LeafIndexPage, page.Previous, page.Next, tail: 0, level: 0, entries, keep) is { } asIs) - { - _channel.WritePage(leafPage, KeepTail(leafPage, asIs)); - return; - } - - if (share > keep - && Build(PageType.LeafIndexPage, page.Previous, page.Next, tail: 0, level: 0, entries, share) - is { } compressed) - { - _channel.WritePage(leafPage, KeepTail(leafPage, compressed)); - return; - } - - // Where to cut. Splitting down the middle is right when keys arrive all over the range, because the - // lower half's free space is room for the next key near it. When the new entry is the page's MAXIMUM - // it is waste instead: nothing sorts below a maximum, so half the page is stranded for ever. ACE - // splits at the right edge in that case — the page stays full and the new entry starts a fresh one — - // which is why a sequentially loaded index of ACE's packs its leaves to capacity and LibRed's used to - // settle near half (see docs/format/page-03-04-index-btree.md §10.5 for the measured comparison). - // - // AutoNumber and identity keys are ascending by construction, so this is the ordinary case. The - // condition cannot fire on a random insert, which is why the general behaviour is unchanged. - // - // Everywhere else ACE halves the entries the page held BEFORE this insert, and the new entry then - // joins whichever side its key falls in — so a key landing below the midpoint leaves one MORE entry - // behind than a key landing on or above it. Halving the post-insert list instead always hands the odd - // entry to the right page, which agrees with ACE only for the upper half. Measured on a 602-entry - // leaf split by one further key (§10.5): ACE keeps 302 entries for a key at position 1 or 50, and 301 - // for one at 301, 302 or 400 — so the midpoint itself goes right. - int mid = (entries.Count - 1) / 2; // midpoint of the pre-insert entries - int splitAt = pos == entries.Count - 1 - ? entries.Count - 1 - : Math.Max(1, pos < mid ? mid + 1 : mid); // never leave the left page empty - SplitAndPropagate(index, path, path.Count - 1, entries, PageType.LeafIndexPage, - page.Previous, page.Next, splitAt); - } - - /// - /// Splits the (leaf or node) page at into two, writes both, then promotes a - /// separator into the parent — splitting parents in turn, or growing a new root at the top. - /// - /// The index whose tree is being split. - /// The pages from the root down to the one being split, one per level. - /// Which entry of is the page to split. - /// That page's entries, in key order, including the one just inserted. - /// Leaf or node — what the two halves are written as. - /// The split page's left sibling, for the leaf chain. - /// Its right sibling. - /// How many entries stay on the left page; negative for the default half. Only a - /// leaf split sets it, to keep a page full when the new entry is its maximum (see InsertIntoLeaf). - private void SplitAndPropagate(IndexDef index, List path, int level, List entries, - PageType type, int prev, int next, int splitAt = -1) - { - int leftPage = path[level]; - int rightPage = AllocateIndexPage(index); - int nodeLevel = path.Count - 1 - level; // height above the leaves of the page being split - - byte[] promoted; - if (type == PageType.LeafIndexPage) - { - // The left page always fits: at worst it is the page as it stood before the insert that - // overflowed it, and that fitted. - int mid = splitAt < 0 ? entries.Count / 2 : splitAt; - var left = entries.GetRange(0, mid); - var right = entries.GetRange(mid, entries.Count - mid); - promoted = WithTrailer(left[^1].Key, left[^1].Trailer); // left's max full key - - WriteOrThrow(leftPage, Build(type, prev, rightPage, tail: 0, nodeLevel, left)); - WriteOrThrow(rightPage, Build(type, leftPage, next, tail: 0, nodeLevel, right)); - if (next != 0) SetPrev(next, rightPage); // fix the old next leaf's back-link - } - else - { - // Node split: the middle entry's key is promoted; its child becomes the left node's tail. - int mid = entries.Count / 2; - Entry middle = entries[mid]; - var left = entries.GetRange(0, mid); - var right = entries.GetRange(mid + 1, entries.Count - mid - 1); - promoted = middle.Key; - int oldTail = _splitTail; - - WriteOrThrow(leftPage, Build(type, 0, 0, tail: middle.Trailer, nodeLevel, left)); - WriteOrThrow(rightPage, Build(type, 0, 0, tail: oldTail, nodeLevel, right)); - } - - if (level == 0) - { - // The root split: build a new root node [promoted -> old root] with the new page as its tail. - int newRoot = AllocateIndexPage(index); - WriteOrThrow(newRoot, Build(PageType.IntermediateIndexPage, 0, 0, tail: rightPage, nodeLevel + 1, - [new Entry(promoted, leftPage)])); - UpdateIndexRoot(index, newRoot); - index.RootPage = newRoot; // keep the in-memory def in step for the next insert - return; - } - - InsertSeparator(index, path, level - 1, leftPage, promoted, rightPage); - } - - /// Inserts a promoted separator into the parent node; repoints the old child to the new right - /// page and splits the parent if it overflows. - private void InsertSeparator(IndexDef index, List path, int level, int oldChild, byte[] promoted, int newRight) - { - int parentPage = path[level]; - CheckedIndexPage page = ReadMutationPage(parentPage, PageType.IntermediateIndexPage); - (List entries, int tail) = Parse(page); - - int slot = entries.FindIndex(e => e.Trailer == oldChild); - if (slot >= 0) - { - entries[slot] = entries[slot] with { Trailer = newRight }; - entries.Insert(slot, new Entry(promoted, oldChild)); - } - else // oldChild was the tail - { - tail = newRight; - entries.Add(new Entry(promoted, oldChild)); - } - - int parentLevel = path.Count - 1 - level; - if (Build(PageType.IntermediateIndexPage, 0, 0, tail, parentLevel, entries) is { } built) - { - _channel.WritePage(parentPage, KeepTail(parentPage, built)); - return; - } - - _splitTail = tail; - SplitAndPropagate(index, path, level, entries, PageType.IntermediateIndexPage, 0, 0); - } - - private int _splitTail; // carries a node's tail into SplitAndPropagate - - /// Parses a checked page's entries, decompressing their shared prefix. - private static (List Entries, int Tail) Parse(CheckedIndexPage page) - { - var entries = new List(page.EntryRanges.Count); - foreach ((byte[] key, int trailer) in IndexPageReader.DecodeEntries(page)) - entries.Add(new Entry(key, trailer)); - return (entries, page.Tail); - } - - /// Revalidates a page immediately before mutation, closing the gap between B-tree descent and - /// the final read-modify-write operation. - private CheckedIndexPage ReadMutationPage(int pageNumber, PageType expectedType) - { - CheckedIndexPage page = IndexPageReader.Read(_channel, pageNumber, _table.DefinitionPage); - if (page.Type != expectedType) - throw new InvalidDataException( - $"Index mutation expected page {pageNumber} to be {expectedType}, but found {page.Type}."); - return page; - } - - /// Builds a page from entries; null if they overflow the page. Leaf pages are prefix-compressed; - /// node pages are stored uncompressed and carry their height above the leaves at , - /// both matching what Access writes. (An isolation test showed neither is strictly required — Access reads - /// a node with 0x1A=0 and compressed just fine; they are kept purely for byte-faithfulness. The one - /// hard requirement is a leaf's 0x1A=0 and the leaf-chain offsets at 0x0C/0x10.) - /// Leaf or node. - /// The page's left sibling, written into the leaf chain. - /// Its right sibling. - /// The page's trailing pointer — a node's rightmost child. - /// A node's height above the leaves; 0 on a leaf. - /// The entries to write, in key order. - /// The shared-prefix length to store the entries at. Null computes the largest - /// available, which is what a split writes. It must not exceed what the entries actually share. - private byte[]? Build(PageType type, int prev, int next, int tail, int level, List entries, - int? prefix = null) - { - bool isLeaf = type == PageType.LeafIndexPage; - int pageSize = _channel.PageSize; - var page = new byte[pageSize]; - page[0] = (byte)type; - page[1] = 0x01; // page flags (observed constant) - page[LevelOffset] = (byte)level; // 0 on a leaf; the node's height above the leaves otherwise - WriteInt32Le(page, OwnerOffset, _table.DefinitionPage); - WriteInt32Le(page, PrevPageOffset, prev); - WriteInt32Le(page, NextPageOffset, next); - WriteInt32Le(page, ChildTailOffset, tail); - - // A single entry (or a node) has no common-prefix compression — ACE writes 0 here (the whole key with - // itself would otherwise "compress" to its full length, which ACE does not do for one entry). - // Otherwise the caller chooses: see InsertIntoLeaf for when a page is compressed at all. - int compress = !isLeaf || entries.Count <= 1 - ? 0 - : prefix ?? CommonPrefixLength(entries[0].Key, entries[^1].Key); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(CompressedByteCountOffset, 2), (ushort)compress); - - int pos = EntryDataOffset; - bool first = true; - foreach (Entry e in entries) - { - ReadOnlySpan stored = first ? e.Key : e.Key.AsSpan(compress); - first = false; - int len = stored.Length + 4; - if (pos + len > pageSize) return null; // overflow - - stored.CopyTo(page.AsSpan(pos)); - WriteInt32Be(page, pos + stored.Length, e.Trailer); - pos += len; - - int end = pos - EntryDataOffset; - page[EntryMaskOffset + (end >> 3)] |= (byte)(1 << (end & 7)); - } - - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(FreeSpaceOffset, 2), (ushort)(pageSize - pos)); - return page; - } - - /// Repoints the index-data block's B-tree root (0x26) after the root grows a level. Walks - /// stats → column descriptors → column names → data blocks to the index's block. - /// - /// A wide table's definition spans continuation pages, and the data blocks sit past the column names — - /// well beyond the first page for a 255-column table. The walk therefore runs over the stitched - /// definition (the absolute coordinate space the descriptors use), and only the 4 root bytes are written - /// back, mapped to whichever page actually holds them. Nothing changes length, so no re-split is needed. - /// - private void UpdateIndexRoot(IndexDef index, int newRoot) - { - (_, IReadOnlyList continuations, int block) = LocateIndexBlock(index); - WriteInt32IntoDefinition(continuations, block + IndexBlockFormat.RootPageOffset, newRoot); - } - - /// The (row, page) pointer to the index's own pages usage map, read from its data block. - private (int MapRow, int MapPage) IndexUsageMapPointer(IndexDef index) - { - (byte[] tdef, _, int block) = LocateIndexBlock(index); - int pointer = block + IndexBlockFormat.UsageMapRowOffset; // 1-byte row + 3-byte page - int row = tdef[pointer]; - int mapPage = tdef[pointer + 1] | tdef[pointer + 2] << 8 | tdef[pointer + 3] << 16; - return (row, mapPage); - } - - /// Walks the stitched definition (stats → column descriptors → column names → data blocks) to - /// the index's 52-byte data block, returning the buffer, its continuation pages, and the block's absolute - /// offset. A wide table's blocks sit past the column names, well beyond the first page. - /// - /// Every region is bounded, because this walk decides where writes. The - /// counts and name lengths all come out of the file, and unchecked they can carry pos past the - /// buffer — or, when the multiply overflows, back inside it at the wrong place, which would repoint some - /// other index's B-tree root with no error at all. The read side bounds the identical regions. - /// - private (byte[] Definition, IReadOnlyList Continuations, int BlockOffset) LocateIndexBlock(IndexDef index) - { - JetFormatBase format = _channel.Format; - (byte[] tdef, IReadOnlyList continuations) = ReadDefinition(); - int dataCount = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(format.TdefIndexCountOffset, 4)); - int colCount = BinaryPrimitives.ReadUInt16LittleEndian(tdef.AsSpan(format.TdefColumnCountOffset, 2)); - - int columnBlock = TableDefinitionPage.CheckedRegionEnd( - format.TdefRealIndexBlockOffset, dataCount, format.RealIndexEntrySize, tdef.Length, "index statistics"); - int pos = TableDefinitionPage.CheckedRegionEnd( - columnBlock, colCount, format.ColumnDescriptorSize, tdef.Length, "column descriptors"); - for (int i = 0; i < colCount; i++) - { - int nameEnd = TableDefinitionPage.CheckedRegionEnd(pos, 1, 2, tdef.Length, "a column-name length"); - pos = TableDefinitionPage.CheckedRegionEnd( - nameEnd, 1, BinaryPrimitives.ReadUInt16LittleEndian(tdef.AsSpan(pos, 2)), tdef.Length, "a column name"); - } - - int block = TableDefinitionPage.CheckedRegionEnd( - pos, index.RealIndexOrdinal + 1, IndexBlockFormat.DataBlockSize, tdef.Length, "index-data blocks") - - IndexBlockFormat.DataBlockSize; - return (tdef, continuations, block); - } - - /// Allocates a fresh B-tree page for and records it in the index's own - /// pages usage map, exactly as Access does — the map then covers every page of the index's B-tree, not - /// just the root. (Reads use the B-tree's own links, so this is for byte-faithfulness and to feed Access's - /// own maintenance, not for LibRed's own traversal.) - private int AllocateIndexPage(IndexDef index) - { - int page = _allocator.Allocate(); - (int mapRow, int mapPage) = IndexUsageMapPointer(index); - _usageMaps.SetBit(mapRow, mapPage, page, set: true); - return page; - } - - /// Reads the table definition, stitching continuation pages into one contiguous buffer, and - /// returns the continuation page numbers in chain order. - private (byte[] Definition, IReadOnlyList ContinuationPages) ReadDefinition() - { - (PageBuffer buffer, IReadOnlyList continuations) = - TdefChainReader.Read(_channel, _table.DefinitionPage); - return (buffer.Span.ToArray(), continuations); - } - - /// Maps an absolute definition offset to the page holding it and the offset within that page. - private (int Page, int Offset) MapDefinitionOffset(IReadOnlyList continuations, int offset) - { - int pageSize = _channel.Format.PageSize; - if (offset < pageSize) return (_table.DefinitionPage, offset); - - int body = pageSize - JetFormatBase.TdefContinuationHeaderSize; - int relative = offset - pageSize; - int index = relative / body; - if (index >= continuations.Count) - throw new InvalidOperationException( - $"Definition offset {offset} lies past the end of table '{_table.Name}'s definition chain."); - return (continuations[index], JetFormatBase.TdefContinuationHeaderSize + relative % body); - } - - /// Writes 4 little-endian bytes at an absolute definition offset, splitting the write when the - /// field straddles a continuation-page boundary. - private void WriteInt32IntoDefinition(IReadOnlyList continuations, int offset, int value) - { - Span bytes = stackalloc byte[4]; - BinaryPrimitives.WriteInt32LittleEndian(bytes, value); - - for (int i = 0; i < 4;) - { - int pageNumber = MapDefinitionOffset(continuations, offset + i).Page; - byte[] page = _channel.ReadPage(pageNumber).Span.ToArray(); - - int j = i; - for (; j < 4; j++) - { - (int target, int within) = MapDefinitionOffset(continuations, offset + j); - if (target != pageNumber) break; - page[within] = bytes[j]; - } - - _channel.WritePage(pageNumber, page); - i = j; - } - } - - private void SetPrev(int pageNumber, int prev) - { - CheckedIndexPage checkedPage = IndexPageReader.Read(_channel, pageNumber, _table.DefinitionPage); - if (checkedPage.Type != PageType.LeafIndexPage) - throw new InvalidDataException($"Leaf next-pointer targets non-leaf page {pageNumber}."); - byte[] page = checkedPage.Buffer.Span.ToArray(); - WriteInt32Le(page, PrevPageOffset, prev); - _channel.WritePage(pageNumber, page); - } - - private void WriteOrThrow(int pageNumber, byte[]? page) => - _channel.WritePage(pageNumber, KeepTail(pageNumber, page ?? throw new NotSupportedException( - "An index page still overflows after a split (a key wider than half a page)."))); - - /// - /// Carries the destination's bytes past the new live end into a freshly built page, because that is what - /// Access leaves behind. - /// - /// - /// works in a zeroed buffer and fills it only as far as the last entry, so - /// everything beyond is written back as zeros. ACE instead edits a page in place: it moves the free-space - /// pointer and lets the bytes the entries used to occupy stand. Measured on an insert that splits three - /// levels of [Order Details]: on every index page ACE rewrote (both leaves and nodes) the region - /// past the new live end still held the original bytes, and on the one page it took fresh from the - /// allocator that region was zero. So the rule is zero-fill on allocation, never clear again. - /// Only a rewrite of this index's own page qualifies — a page just recycled from somewhere else - /// still carries the previous owner's type and owner id, and ACE zero-fills that one, which is what - /// building in a clean buffer already does. - /// The bytes kept are dead: they sit past the free-space boundary, so no reader reaches them. - /// Keeping them is for byte-faithfulness with ACE, not for meaning. - /// - private byte[] KeepTail(int pageNumber, byte[] built) - { - if (pageNumber >= _channel.PageCount) return built; - - ReadOnlySpan existing = _channel.ReadPage(pageNumber).Span; - if (existing[0] != built[0] - || BinaryPrimitives.ReadInt32LittleEndian(existing[OwnerOffset..]) != _table.DefinitionPage) - return built; - - int liveEnd = _channel.PageSize - - BinaryPrimitives.ReadUInt16LittleEndian(built.AsSpan(FreeSpaceOffset, 2)); - existing[liveEnd..].CopyTo(built.AsSpan(liveEnd)); - return built; - } - - /// Width of the row/child pointer an entry carries after its key. - private const int TrailerSize = 4; - - private static byte[] WithTrailer(byte[] key, int trailer) - { - var result = new byte[key.Length + TrailerSize]; - key.CopyTo(result, 0); - WriteInt32Be(result, key.Length, trailer); - return result; - } - - /// - /// Fills an empty index from by writing each page once, instead of - /// inserting the entries one at a time and rewriting a whole leaf per entry. - /// - /// The empty index to fill. - /// (key, row pointer) pairs with a NullKey marker; any order. Sorted here. - /// Enforce uniqueness — adjacent equal keys after the sort, null keys exempt - /// (Jet's uniqueness is over the non-null keys only). - /// - /// This is not a different B-tree: it is the SAME fill performs, driven - /// over pre-sorted input so each page is finished before the next begins. A sequential load already took - /// the right-edge split — a new entry that is the page's maximum leaves the full page alone and starts a - /// fresh one — so feeding sorted entries reproduces the leaf partitioning ACE produces; the difference is - /// only that a page is built once rather than rebuilt per entry. - /// Whether an entry fits is decided by arithmetic, not by calling : Build - /// allocates a page-sized array, so probing with it would allocate two pages per entry and lose exactly - /// what this exists to save. mirrors Build's layout — entry data starts at - /// 0x1E0, the first entry stores its whole key, the rest drop the shared prefix, and each carries a - /// 4-byte trailer (see page-01/page-03-04 §10.2–10.3). - /// Node levels are then built bottom-up, splitting at the middle and promoting the middle entry as - /// does, so interior pages keep the shape the incremental path gives them. - /// - internal void BulkBuild(IndexDef index, List<(byte[] Key, int Pointer, bool NullKey)> entries, bool rejectDuplicates) - { - if (entries.Count == 0) - return; // the caller's freshly created empty root is already the correct tree - - entries.Sort((a, b) => CompareEntries(a.Key, a.Pointer, b.Key, b.Pointer)); - - if (rejectDuplicates) - for (int i = 1; i < entries.Count; i++) - if (!entries[i].NullKey && !entries[i - 1].NullKey - && CompareBytes(entries[i].Key, entries[i - 1].Key) == 0) - throw new InvalidOperationException( - $"Cannot create unique index '{index.Name}' on '{_table.Name}': duplicate key values exist."); - - // --- leaves --------------------------------------------------------------------------------------- - var separators = new List<(byte[] Key, int Child)>(); - int page = index.RootPage; // the empty leaf the caller created; it becomes the first leaf - int previous = 0; - var current = new List(); - long keyBytes = 0; // running sum of the current page's key lengths - int compressed = 0; // the prefix the page would be written at, as filling stands - - foreach ((byte[] key, int pointer, _) in entries) - { - current.Add(new Entry(key, pointer)); - keyBytes += key.Length; - - int share = current.Count <= 1 ? 0 : CommonPrefixLength(current[0].Key, current[^1].Key); - if (LeafFits(current.Count, keyBytes, Math.Min(compressed, share))) - continue; - if (share > compressed && LeafFits(current.Count, keyBytes, share)) - { - compressed = share; // the compress-in-place step, without writing the page yet - continue; - } - - // Full. The entry that did not fit starts the next leaf, as the right-edge split does. - current.RemoveAt(current.Count - 1); - keyBytes -= key.Length; - int next = AllocateIndexPage(index); - WriteOrThrow(page, Build(PageType.LeafIndexPage, previous, next, tail: 0, level: 0, current)); - separators.Add((WithTrailer(current[^1].Key, current[^1].Trailer), page)); - - previous = page; - page = next; - current = [new Entry(key, pointer)]; - keyBytes = key.Length; - compressed = 0; - } - - WriteOrThrow(page, Build(PageType.LeafIndexPage, previous, next: 0, tail: 0, level: 0, current)); - - // --- node levels, until one page covers the level -------------------------------------------------- - int level = 1; - while (separators.Count > 0) - (separators, page) = BuildNodeLevel(index, separators, tailChild: page, level++); - - if (page != index.RootPage) - { - UpdateIndexRoot(index, page); - index.RootPage = page; - } - } - - /// Whether entries totalling of key data fit - /// one leaf at prefix — Build's layout, without building anything. - private bool LeafFits(int count, long keyBytes, int prefix) => - EntryDataOffset + keyBytes + (long)TrailerSize * count - (long)prefix * (count - 1) <= _channel.PageSize; - - /// Writes one level of interior nodes over , returning the separators - /// promoted to the level above and the page that is this level's rightmost (tail) child. - private (List<(byte[] Key, int Child)> Promoted, int Tail) BuildNodeLevel( - IndexDef index, List<(byte[] Key, int Child)> children, int tailChild, int level) - { - var promoted = new List<(byte[] Key, int Child)>(); - var current = new List(); - long keyBytes = 0; - int page = AllocateIndexPage(index); - - foreach ((byte[] key, int child) in children) - { - current.Add(new Entry(key, child)); - keyBytes += key.Length; - if (EntryDataOffset + keyBytes + (long)TrailerSize * current.Count <= _channel.PageSize) - continue; - - // Overflow: split at the middle and promote the middle entry, whose child becomes the left node's - // tail — the division InsertSeparator makes. Nodes are stored uncompressed. - int mid = current.Count / 2; - Entry middle = current[mid]; - List left = current.GetRange(0, mid); - List right = current.GetRange(mid + 1, current.Count - mid - 1); - - WriteOrThrow(page, Build(PageType.IntermediateIndexPage, 0, 0, middle.Trailer, level, left)); - promoted.Add((middle.Key, page)); - - page = AllocateIndexPage(index); - current = right; - keyBytes = right.Sum(e => (long)e.Key.Length); - } - - WriteOrThrow(page, Build(PageType.IntermediateIndexPage, 0, 0, tailChild, level, current)); - return (promoted, page); - } - - /// Orders two entries as their stored key ++ trailer bytes compare, without building either. - internal static int CompareEntries(byte[] aKey, int aTrailer, byte[] bKey, int bTrailer) - { - int shared = Math.Min(aKey.Length, bKey.Length); - for (int i = 0; i < shared; i++) - if (aKey[i] != bKey[i]) return aKey[i] - bKey[i]; - - // The keys agree as far as the shorter runs, so the shorter one's trailer meets the longer one's - // remaining key bytes — the same crossing CompareWithTrailer handles, and why this cannot be - // "compare keys, then compare trailers". - return aKey.Length == bKey.Length - ? CompareTrailers(aTrailer, bTrailer) - : aKey.Length < bKey.Length - ? -CompareWithTrailer(bKey, bTrailer, WithTrailer(aKey, aTrailer)) - : CompareWithTrailer(aKey, aTrailer, WithTrailer(bKey, bTrailer)); - } - - private static int CompareTrailers(int a, int b) - { - for (int shift = 24; shift >= 0; shift -= 8) - { - int x = (byte)(a >> shift), y = (byte)(b >> shift); - if (x != y) return x - y; - } - - return 0; - } - - /// - /// Compares ++ its 4-byte big-endian against - /// , byte for byte, without building the concatenation. - /// - /// - /// Exactly CompareBytes(WithTrailer(key, trailer), other), and it has to be: comparing the keys - /// alone and then breaking the tie on the trailer is NOT the same relation. - /// compares the shared prefix and only then falls back to length, so when one key is a prefix of another - /// the real comparison runs on into the trailer bytes — precisely the case a naive rewrite gets wrong, and - /// it would misplace an entry rather than fail. - /// - internal static int CompareWithTrailer(byte[] key, int trailer, ReadOnlySpan other) - { - int total = key.Length + TrailerSize; - int n = Math.Min(total, other.Length); - for (int i = 0; i < n; i++) - { - // Past the key, read the trailer's bytes most-significant first, as WriteInt32Be lays them out. - int mine = i < key.Length ? key[i] : (byte)(trailer >> (8 * (TrailerSize - 1 - (i - key.Length)))); - if (mine != other[i]) return mine - other[i]; - } - - return total - other.Length; - } - - private static int CommonPrefixLength(byte[] a, byte[] b) - { - int n = Math.Min(a.Length, b.Length), i = 0; - while (i < n && a[i] == b[i]) i++; - return i; - } - - private static int CompareBytes(ReadOnlySpan a, ReadOnlySpan b) - { - int n = Math.Min(a.Length, b.Length); - for (int i = 0; i < n; i++) - if (a[i] != b[i]) return a[i] - b[i]; - return a.Length - b.Length; - } - - private static void WriteInt32Le(byte[] page, int offset, int value) => BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(offset, 4), value); - private static void WriteInt32Be(byte[] page, int offset, int value) => BinaryPrimitives.WriteInt32BigEndian(page.AsSpan(offset, 4), value); -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/JetCjkSortOrders.cs b/src/LibRed/LibRed.Core/Storage/JetCjkSortOrders.cs new file mode 100644 index 000000000..eaa051aae --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/JetCjkSortOrders.cs @@ -0,0 +1,120 @@ +using LibRed.Catalog; + +namespace LibRed.Storage; + +/// +/// The Chinese, Japanese and Korean sort orders: each is General of the same version plus a table of +/// per-character weights, measured from ACE and embedded one resource per order. +/// +/// +/// +/// Measured character by character over the whole BMP against Access-authored fixtures, every character these +/// orders weigh differently from General gets a single weight in its place — a replacement primary, and in +/// version 1 a secondary with it. The ideographs are what move: into pronunciation, stroke-count, Bopomofo or +/// radical order, 7,000 to 28,000 characters an order, with a handful of symbols besides (the Japanese orders +/// weigh \ as ¥ and give ― the unweighted FF FF). An iteration mark copies the +/// tailored weight before it. +/// +/// +/// Korean is the one order that is not only a table: it sorts Hangul first, moving the lead byte of every +/// weight General contributes (see ), and its table holds the hanja — +/// weighed by their Hangul reading, the primary a syllable's and the secondary telling hanja apart — with the +/// \ weighed as ₩. +/// +/// +/// Not every order Access offers is here. "Japanese - Unicode" (0x00010411) and "Korean - Unicode" +/// (0x00010412) are created by Access but refused by ACE on open, because Windows no longer supports +/// those alternate sorts — so there is no engine to measure them against, and they stay refused. +/// +/// +internal static class JetCjkSortOrders +{ + /// + /// Suppresses the tables, leaving only the Korean lead-byte move, so the encoder is General of each order's + /// version. Only the generator sets this, for the reason + /// gives: it records where the encoder disagrees with ACE, and must not measure an encoder that already + /// agrees. + /// + internal static bool Suppressed { get; set; } + + /// Each order's collation and the name of its resource, which is the name of the fixture it was + /// measured from. + internal static readonly (Collation Collation, string Name)[] Orders = + [ + (new(CollatingOrder.ChineseSimplified, 1), "ChinesePronunciation"), + (new(CollatingOrder.ChineseSimplified, 0), "ChinesePronunciationLegacy"), + (new(CollatingOrder.ChineseSimplified, 1, SortId: 2), "ChineseStrokeCount"), + (new(CollatingOrder.ChineseSimplified, 0, SortId: 2), "ChineseStrokeCountLegacy"), + (new(CollatingOrder.ChineseTraditional, 1, SortId: 3), "ChineseTradBopomofo"), + (new(CollatingOrder.ChineseTraditional, 0, SortId: 3), "ChineseTradBopomofoLegacy"), + (new(CollatingOrder.ChineseTraditional, 1), "ChineseTradStrokeCount"), + (new(CollatingOrder.ChineseTraditional, 0), "ChineseTradStrokeCountLegacy"), + (new(CollatingOrder.Japanese, 1), "Japanese"), + (new(CollatingOrder.Japanese, 0), "JapaneseLegacy"), + (new(CollatingOrder.Japanese, 1, SortId: 4), "JapaneseRadicalStrokeCount"), + (new(CollatingOrder.Korean, 0), "Korean"), + ]; + + private static readonly Dictionary> Tables = + Orders.ToDictionary(o => o.Collation, o => new Lazy(() => Load(o.Collation, o.Name))); + + private static readonly byte[] KoreanLeads = KoreanLeadBytes(); + + private static readonly Lazy Suppressable = new(() => new LocaleTailoring( + new Dictionary(), leadBytes: KoreanLeads)); + + /// The order's tailoring, or null when the collation is not a CJK order. + public static LocaleTailoring? For(Collation collation) + { + if (!Tables.TryGetValue(collation, out Lazy? table)) return null; + if (!Suppressed) return table.Value; + return collation.Order == CollatingOrder.Korean + ? Suppressable.Value + : new LocaleTailoring(new Dictionary()); + } + + /// + /// Korean's move of General's lead bytes: 81–F2 down 0x37 and 4A–80 + /// up 0x72, everything else in place. Measured over every weight General gives the BMP, with no + /// exception in either band. + /// + private static byte[] KoreanLeadBytes() + { + var leads = new byte[256]; + for (int b = 0; b < 256; b++) + leads[b] = (byte)(b switch + { + >= 0x4A and <= 0x80 => b + 0x72, + >= 0x81 and <= 0xF2 => b - 0x37, + _ => b, + }); + return leads; + } + + private static LocaleTailoring Load(Collation collation, string name) + { + using Stream stream = typeof(JetCjkSortOrders).Assembly + .GetManifestResourceStream($"LibRed.Resources.Cjk.{name}.bin") + ?? throw new InvalidOperationException($"The {name} sort-order resource is missing from the assembly."); + + var reader = new BinaryReader(stream); + int count = reader.ReadInt32(); + byte[] deltas = JetTextCollationV1Overrides.ReadStream(reader); + byte[] primaryLengths = JetTextCollationV1Overrides.ReadStream(reader); + byte[] primaryBytes = JetTextCollationV1Overrides.ReadStream(reader); + byte[] secondaries = JetTextCollationV1Overrides.ReadStream(reader); + + var entries = new Dictionary(count); + int codePoint = 0, cursor = 0, primary = 0; + for (int i = 0; i < count; i++) + { + codePoint += JetTextCollationV1Overrides.ReadVarInt(deltas, ref cursor); + entries[((char)codePoint).ToString()] = + new TailoredWeight(primaryBytes[primary..(primary + primaryLengths[i])], secondaries[i]); + primary += primaryLengths[i]; + } + return new LocaleTailoring(entries, + leadBytes: collation.Order == CollatingOrder.Korean ? KoreanLeads : null, + weighsDecompositions: false); + } +} diff --git a/src/LibRed/LibRed.Core/Storage/JetCodePages.cs b/src/LibRed/LibRed.Core/Storage/JetCodePages.cs new file mode 100644 index 000000000..37d43ba9f --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/JetCodePages.cs @@ -0,0 +1,96 @@ +using System.Collections.Frozen; +using LibRed.Catalog; + +namespace LibRed.Storage; + +/// +/// The ANSI code page ACE writes at page 0 0x3C for a database created in a given collating order. +/// +/// +/// +/// ACE stamps the code page of the order's language: 1250 for Czech, 936 for Chinese Simplified, and +/// 0 for a language Windows gives no ANSI code page (Georgian, Hindi, Armenian …). It depends on the +/// LANGID alone — every sort id of a language gets the same one (Chinese Pronunciation and Stroke Count are +/// both 936, Georgian and Georgian Modern both 0), and so does every version. DAO takes a CP= in its +/// locale string and ignores it: CP=932 on an en-US database still writes 1252. +/// +/// +/// Measured, not derived: DAO created a database in every LCID LibRed can create in, and the value read back +/// is what is listed here (CodePageProbeTest); the four orders DAO refuses — they exist only at +/// version 1 — come from Access-authored files. The runtime's own TextInfo.ANSICodePage happened to +/// agree on every one, but it comes from ICU on Linux, and a table cannot drift. +/// +/// +internal static class JetCodePages +{ + private static readonly FrozenDictionary ByLangId = Build( + [ + (0, + [ + 0x002B, 0x0030, 0x0031, 0x0033, 0x0037, 0x0039, 0x003A, 0x003D, 0x003F, 0x0045, 0x0046, 0x0047, + 0x0048, 0x0049, 0x004A, 0x004B, 0x004C, 0x004D, 0x004E, 0x004F, 0x0051, 0x0053, 0x0054, 0x0055, + 0x0057, 0x0058, 0x005A, 0x005B, 0x005C, 0x005E, 0x0060, 0x0061, 0x0063, 0x0065, 0x0072, 0x0073, + 0x0077, 0x0078, 0x0081, 0x042B, 0x0430, 0x0431, 0x0433, 0x0437, 0x0439, 0x043A, 0x043D, 0x043F, + 0x0445, 0x0446, 0x0447, 0x0448, 0x0449, 0x044A, 0x044B, 0x044C, 0x044D, 0x044E, 0x044F, 0x0451, + 0x0453, 0x0454, 0x0455, 0x0457, 0x0459, 0x045A, 0x045B, 0x045C, 0x045D, 0x045E, 0x0460, 0x0461, + 0x0463, 0x0465, 0x0472, 0x0473, 0x0477, 0x0478, 0x0481, 0x0845, 0x0849, 0x0850, 0x0860, 0x0861, + 0x0873, 0x0C50, 0x0C51, 0x105F, 0x785F, 0x7C50, + ]), + (874, [0x001E, 0x041E]), + (932, [0x0411]), + (936, [0x0804]), + (949, [0x0412]), + (950, [0x0404]), + (1250, + [ + 0x0005, 0x000E, 0x0015, 0x0018, 0x001A, 0x001B, 0x001C, 0x0024, 0x0042, 0x0405, 0x040E, 0x0415, + 0x0418, 0x041A, 0x041B, 0x041C, 0x0424, 0x0442, 0x0818, 0x081A, 0x101A, 0x141A, 0x181A, 0x241A, + 0x2C1A, 0x681A, 0x701A, 0x781A, 0x7C1A, + ]), + (1251, + [ + 0x0002, 0x0019, 0x0022, 0x0023, 0x0028, 0x002F, 0x0040, 0x0044, 0x0050, 0x006D, 0x0085, 0x0402, + 0x0419, 0x0422, 0x0423, 0x0428, 0x042F, 0x0440, 0x0444, 0x0450, 0x046D, 0x0485, 0x0819, 0x082C, + 0x0843, 0x0C1A, 0x1C1A, 0x201A, 0x281A, 0x301A, 0x641A, 0x6C1A, 0x742C, 0x7843, + ]), + (1252, + [ + 0x0003, 0x0006, 0x0007, 0x0009, 0x000A, 0x000B, 0x000C, 0x000F, 0x0010, 0x0013, 0x0014, 0x0016, + 0x0017, 0x001D, 0x0021, 0x002D, 0x002E, 0x0032, 0x0034, 0x0035, 0x0036, 0x0038, 0x003B, 0x003C, + 0x003E, 0x0041, 0x0052, 0x0056, 0x005D, 0x005F, 0x0062, 0x0064, 0x0066, 0x0067, 0x0068, 0x0069, + 0x006A, 0x006C, 0x006E, 0x006F, 0x0070, 0x0071, 0x0074, 0x0075, 0x0076, 0x0079, 0x007A, 0x007C, + 0x007E, 0x0082, 0x0083, 0x0084, 0x0086, 0x0087, 0x0088, 0x0091, 0x0403, 0x0406, 0x0407, 0x0409, + 0x040A, 0x040B, 0x040C, 0x040F, 0x0410, 0x0413, 0x0414, 0x0416, 0x0417, 0x041D, 0x0421, 0x042D, + 0x042E, 0x0432, 0x0434, 0x0435, 0x0436, 0x0438, 0x043B, 0x043E, 0x0441, 0x0452, 0x0456, 0x0462, + 0x0464, 0x0466, 0x0468, 0x0469, 0x046A, 0x046B, 0x046C, 0x046E, 0x046F, 0x0470, 0x0474, 0x0475, + 0x0479, 0x047A, 0x047C, 0x047E, 0x0482, 0x0483, 0x0484, 0x0486, 0x0487, 0x0488, 0x0491, 0x0807, + 0x0809, 0x080A, 0x080C, 0x0810, 0x0813, 0x0814, 0x0816, 0x081D, 0x082E, 0x0832, 0x083B, 0x083C, + 0x083E, 0x085D, 0x0867, 0x0C07, 0x0C09, 0x0C0A, 0x0C0C, 0x0C3B, 0x1007, 0x1009, 0x100A, 0x100C, + 0x103B, 0x1407, 0x1409, 0x140A, 0x140C, 0x143B, 0x1809, 0x180A, 0x180C, 0x183B, 0x1C09, 0x1C0A, + 0x1C0C, 0x1C3B, 0x2009, 0x200A, 0x200C, 0x203B, 0x2409, 0x240A, 0x240C, 0x243B, 0x2809, 0x280A, + 0x280C, 0x2C09, 0x2C0A, 0x2C0C, 0x3009, 0x300A, 0x300C, 0x3409, 0x340A, 0x340C, 0x3809, 0x380A, + 0x380C, 0x3C09, 0x3C0A, 0x3C0C, 0x4009, 0x400A, 0x4409, 0x440A, 0x4809, 0x480A, 0x4C0A, 0x500A, + 0x540A, 0x580A, 0x5C0A, 0x703B, 0x743B, 0x7814, 0x783B, 0x7C14, 0x7C2E, 0x7C3B, 0x7C5D, 0x7C67, + ]), + (1253, [0x0008, 0x0408]), + (1254, [0x001F, 0x002C, 0x0043, 0x041F, 0x042C, 0x0443, 0x782C, 0x7C43]), + (1255, [0x000D, 0x040D]), + (1256, + [ + 0x0001, 0x0020, 0x0029, 0x0059, 0x0080, 0x0401, 0x0420, 0x0429, 0x045F, 0x0480, 0x048C, 0x0492, + 0x0801, 0x0820, 0x0846, 0x0859, 0x0C01, 0x1001, 0x1401, 0x1801, 0x1C01, 0x2001, 0x2401, 0x2801, + 0x2C01, 0x3001, 0x3401, 0x3801, 0x3C01, 0x4001, 0x7C46, 0x7C59, + ]), + (1257, [0x0025, 0x0026, 0x0027, 0x0425, 0x0426, 0x0427]), + (1258, [0x002A, 0x042A]), + ]); + + /// The code page ACE writes for a database in , or null when it has + /// not been measured for that order's language. + public static int? For(Collation collation) => + ByLangId.TryGetValue(collation.Lcid & 0xFFFF, out int codePage) ? codePage : null; + + private static FrozenDictionary Build((int CodePage, int[] LangIds)[] groups) => + groups.SelectMany(g => g.LangIds.Select(langId => (langId, g.CodePage))) + .ToFrozenDictionary(e => e.langId, e => e.CodePage); +} diff --git a/src/LibRed/LibRed.Core/Storage/JetIndexKeyChecksum.cs b/src/LibRed/LibRed.Core/Storage/JetIndexKeyChecksum.cs deleted file mode 100644 index 6f6d3802d..000000000 --- a/src/LibRed/LibRed.Core/Storage/JetIndexKeyChecksum.cs +++ /dev/null @@ -1,77 +0,0 @@ -namespace LibRed.Storage; - -/// -/// The two bytes ACE puts at the end of an index entry too long to store whole. -/// -/// -/// An entry of at most 510 bytes is stored as built. Past that ACE keeps the first 508 bytes and replaces the -/// rest with this value, computed over the bytes it dropped — which is why two long values that share a -/// 508-byte prefix still sort apart instead of colliding. -/// -/// Recovered by measurement, not documentation. Three tails differing in one byte showed the function is -/// affine over GF(2) (L(0xA3) ^ L(0x13) = L(0xB0) exactly), and it proved shift-invariant across 173 -/// observations, so a byte at distance d from the end contributes S^(d-1) of itself. Sweeping all -/// 65,536 polynomials in the usual framings found nothing, because the usual framing is wrong: the standard -/// reflected update is crc = (crc >> 8) ^ T[(crc ^ b) & 0xFF], passing the byte THROUGH the table, -/// while ACE computes crc = (crc >> 8) ^ T[crc & 0xFF] ^ b and injects it raw. The step operator -/// was then solved directly by Gaussian elimination over the measured contributions, and predicts all 657 of -/// them. There is no initial value and no final XOR. -/// -/// -/// Not verified where the dropped bytes contain a word-sort record. Those cannot be checked even in -/// principle: the record sits in the part ACE discarded, so what it contained is unobservable, and if ACE -/// recomputes its position when truncating then the input differs from anything reconstructable here. The -/// caller refuses those rather than guess — see . -/// -/// -internal static class JetIndexKeyChecksum -{ - /// Where the kept bytes end and the checksum begins. - public const int KeptBytes = 508; - - /// - /// The step's action on each bit. The upper eight are a plain right shift by eight, which makes the - /// operator the familiar (x >> 8) ^ T(x & 0xFF) of a table-driven CRC; the lower eight are the - /// table itself, measured from ACE. - /// - private static ReadOnlySpan StepBits => - [ - 0x0580, 0x0F80, 0x1B80, 0x3380, 0x6380, 0xC380, 0x8381, 0x0383, - 0x0001, 0x0002, 0x0004, 0x0008, 0x0010, 0x0020, 0x0040, 0x0080, - ]; - - private static readonly ushort[] Table = BuildTable(); - - private static ushort[] BuildTable() - { - var table = new ushort[256]; - for (int value = 0; value < 256; value++) - { - ushort result = 0; - for (int bit = 0; bit < 8; bit++) if ((value & (1 << bit)) != 0) result ^= StepBits[bit]; - table[value] = result; - } - return table; - } - - /// - /// The checksum over the bytes ACE dropped — everything from on. - /// - /// - /// The last byte does not go through a full step. Every byte before it is folded in the usual way, and - /// then the final one is XORed into the high half — it never gets its own shift or table lookup. - /// This was first written as "the terminator is excluded", which is the same thing whenever that - /// byte is 0x00: XOR-ing zero changes nothing. A key ending in text always ends in its 0x00 - /// terminator, and every measurement behind the original rule used one, so the two readings could not be - /// told apart. They diverge the moment the key's last column is numeric — (TEXT, TEXT, LONG) put - /// a data byte there and the keys parted company from ACE's, silently. Re-measured over LONG, CURRENCY - /// and DOUBLE tails across 24 keys: this form matches ACE on every one, and still matches on the all-text - /// keys the old form was derived from. See docs/design/index-key-checksum.md. - /// - public static ushort Compute(ReadOnlySpan discarded) - { - ushort crc = 0; - foreach (byte b in discarded[..^1]) crc = (ushort)((crc >> 8) ^ Table[crc & 0xFF] ^ b); - return (ushort)(crc ^ (discarded[^1] << 8)); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/JetKanaSection.cs b/src/LibRed/LibRed.Core/Storage/JetKanaSection.cs index f961dfa0a..c93306f26 100644 --- a/src/LibRed/LibRed.Core/Storage/JetKanaSection.cs +++ b/src/LibRed/LibRed.Core/Storage/JetKanaSection.cs @@ -16,46 +16,109 @@ internal static class JetKanaSection /// The page byte every kana primary starts with: a kana weighs 7F <sound>. public const byte KanaPage = 0x7F; - /// Closes the kana section, after the FF that introduces the prolonged-mark flags. + /// Closes the kana section, after the mark codes. /// Constant across hiragana, katakana, halfwidth, small and voiced forms in every string measured, so it /// is emitted literally; what it denotes is not established. - private static ReadOnlySpan Tail => [0x02, 0x80, 0xFF, 0x80]; + private static ReadOnlySpan Tail => [0x02, 0x80, IndexKeyCodec.KanaRunSeparator, 0x80]; + + /// The mark code of a kana that is a letter in its own right. + public const byte Letter = 0b01; + + /// The mark code of a prolonged sound mark ー lengthening the kana before it. + public const byte Prolonged = 0b11; + + /// The mark code of an iteration mark (ゝ, ヽ, 々 …) repeating the kana before + /// it — かゝ closes FF 98, where かー closes FF 9C and かあ a bare FF. + public const byte Repeat = 0b10; + + /// + /// Whether is an iteration mark — [MS-UCODEREF]'s PW_REPEAT — and the + /// secondary it adds to the weight it repeats. + /// + /// + /// An iteration mark weighs as a copy of the weight the character before it contributed, with its OWN + /// secondary: 人々 is 9FD4 9FD4 with secondaries 02 05, and がゝ is が twice + /// with secondaries 03 02 — the mark does not inherit the voicing, ゞ adds its own. The long + /// vowel mark is one too wherever it has no kana to lengthen: 人ー and aー double what came + /// before. Measured against ACE under both versions, which agree except that 〻 and ꀕ are + /// repeat marks only in version 1 — version 0 ignores both. + /// + /// The character. + /// Whether the order is version 1. + /// The secondary the mark adds. + public static bool TryGetIterationMark(char character, bool version1, out byte secondary) + { + secondary = character switch + { + (char)0x3005 => 0x05, // 々 + (char)0x309D or (char)0x30FD or (char)0x3031 => 0x02, // ゝ ヽ 〱 + _ when IsProlongedSoundMark(character) => 0x02, // ー ー, with no kana to lengthen + (char)0x309E or (char)0x30FE or (char)0x3032 => 0x03, // ゞ ヾ 〲 — voiced + (char)0x303B when version1 => 0x05, // 〻 + (char)0xA015 when version1 => 0x07, // ꀕ + _ => 0, + }; + return secondary != 0; + } + + /// Whether is a prolonged sound mark — ー or halfwidth ー. Right + /// after a kana it lengthens that kana's vowel, taking the vowel's primary and the kana's small flag and marking + /// itself ; with no kana before it, it is an iteration mark. + public static bool IsProlongedSoundMark(char character) => character is (char)0x30FC or (char)0xFF70; + + /// Whether is a halfwidth voicing mark, and the secondary it gives the kana + /// right before it: ゙ voices it (03), ゚ semi-voices it (04). The marks are combining — + /// ACE folds them into that kana rather than weighing them — and with no kana before them they are weighed alone. + public static bool TryGetHalfwidthVoicing(char character, out byte secondary) + { + secondary = character switch + { + (char)0xFF9E => 0x03, + (char)0xFF9F => 0x04, + _ => 0, + }; + return secondary != 0; + } /// - /// Appends 01 01, the packed small/normal flags, the prolonged-mark flags and the closing constant. + /// Appends 01 01, the packed small/normal flags, the mark codes and the closing constant. /// Emitted whenever the string holds any kana at all, even if every one of them is a normal form. /// - public static void Append(List output, List small, List prolonged) + /// The key being built. + /// Per kana, whether it is a small form. + /// Per kana, , or . + public static void Append(List output, List small, List marks) { - output.Add(0x01); - output.Add(0x01); - AddFlags(output, small, marked: 0b10, unmarked: 0b11); - output.Add(0xFF); - AddFlags(output, prolonged, marked: 0b11, unmarked: 0b01); + output.Add(IndexKeyCodec.SectionSeparator); // ends the diacritics + output.Add(IndexKeyCodec.SectionSeparator); // ends the empty case section + AddCodes(output, small.Count, i => small[i] ? (byte)0b10 : (byte)0b11, unmarked: 0b11); + output.Add(IndexKeyCodec.KanaRunSeparator); + AddCodes(output, marks.Count, i => marks[i], unmarked: Letter); output.AddRange(Tail); } /// - /// Packs one flag per kana, three to a byte, most significant first, under a 10 marker in - /// the top two bits: 11 normal, 10 small, 00 padding. So one small kana is - /// A0, "normal small" is B8, and four kana take two bytes, the second repeating the marker. - /// Verified against ACE over all 30 combinations up to four kana. + /// Packs one two-bit code per kana, three to a byte, most significant first, under a 10 marker + /// in the top two bits, 00 padding the last byte. The small flags are 11 normal and 10 + /// small, so one small kana is A0, "normal small" is B8, and four kana take two bytes, the + /// second repeating the marker — verified against ACE over all 30 combinations up to four kana. The mark + /// codes pack the same way. /// - /// Nothing is emitted at all when no flag is set, which is why a lone normal kana closes straight into - /// the tail. + /// Codes are written only up to the last one that is not , so nothing at all is + /// emitted when every kana is unmarked, which is why a lone normal kana closes straight into the tail. /// /// - private static void AddFlags(List output, List flags, int marked, int unmarked) + private static void AddCodes(List output, int count, Func code, byte unmarked) { - int last = flags.LastIndexOf(true); + int last = count - 1; + while (last >= 0 && code(last) == unmarked) last--; for (int start = 0; start <= last; start += 3) { int packed = 0x80; for (int slot = 0; slot < 3; slot++) { int index = start + slot; - int code = index > last ? 0b00 : flags[index] ? marked : unmarked; - packed |= code << (4 - 2 * slot); + packed |= (index > last ? 0b00 : code(index)) << (4 - 2 * slot); } output.Add((byte)packed); } diff --git a/src/LibRed/LibRed.Core/Storage/JetLocaleTailoring.cs b/src/LibRed/LibRed.Core/Storage/JetLocaleTailoring.cs index ce74dc9c6..a33ec1505 100644 --- a/src/LibRed/LibRed.Core/Storage/JetLocaleTailoring.cs +++ b/src/LibRed/LibRed.Core/Storage/JetLocaleTailoring.cs @@ -1,5 +1,6 @@ using LibRed.Catalog; using System.Collections.Frozen; +using static LibRed.Storage.IndexKeyCodec; namespace LibRed.Storage; @@ -30,14 +31,59 @@ internal sealed class LocaleTailoring public LocaleTailoring( IReadOnlyDictionary entries, bool doublesDigraphs = false, - bool reverseDiacritics = false) + bool reverseDiacritics = false, + byte[]? leadBytes = null, + bool weighsDecompositions = true) { Entries = entries; DoublesDigraphs = doublesDigraphs; ReverseDiacritics = reverseDiacritics; + LeadBytes = leadBytes; + WeighsDecompositions = weighsDecompositions; MaxLength = entries.Count == 0 ? 0 : entries.Keys.Max(k => k.Length); + + // Looked up by span, never by a string made for the purpose: every character of every key asks, and a + // string or two per question was most of what encoding a key allocated. Every table is an ordinal + // Dictionary, which takes the lookup as it is; anything else is copied into one. + Dictionary table = + entries is Dictionary dictionary + && dictionary.TryGetAlternateLookup>(out _) + ? dictionary + : new Dictionary(entries, StringComparer.Ordinal); + _bySpan = table.GetAlternateLookup>(); } + private readonly Dictionary.AlternateLookup> _bySpan; + + /// + /// Whether the characters a General decomposition produces take this tailoring's weights. + /// + /// + /// A Latin locale's letters reach into decompositions: Croatian's DŽ expands to D + Ž, + /// and it is Croatian's Ž that ACE stores. A CJK table does not. Version 1's base table + /// expands the CJK radicals and compatibility ideographs to the unified ideograph they stand for — + /// ⼀ U+2F00 to 一, U+FA30 to U+4FAE — and ACE weighs them as General does, + /// never with the order's weight for that ideograph: 315 to 421 characters in each version-1 CJK order, + /// measured against ACE. The tables weigh the characters they list and nothing reached through a + /// decomposition. + /// + public bool WeighsDecompositions { get; } + + /// + /// A replacement for the lead byte of every weight the General table contributes, indexed by that byte — + /// or null for an order that keeps General's lead bytes, which is every order but one. + /// + /// + /// Korean reorders whole scripts rather than letters: Hangul sorts first, so General's lead bytes + /// 81–F2 move down 0x37 and 4A–80 (Latin, Greek, Cyrillic, the other + /// scripts, kana, Bopomofo) move up 0x72 to make room — A is BC, 가 is + /// 4A 03. It applies to each WEIGHT, not to each byte: the second byte of a two-byte weight keeps its + /// value (ᄀ 81 02 becomes 4A 02), and an expansion has every weight moved (IJ + /// 59 5B becomes CB CD). A tailored entry already states its final bytes and is not moved, + /// nor is a copy an iteration mark makes of a weight already moved. + /// + public byte[]? LeadBytes { get; } + public IReadOnlyDictionary Entries { get; } public bool DoublesDigraphs { get; } @@ -103,16 +149,13 @@ public bool TryMatch( /// text never contained. /// public bool TryMatchSingle(char character, out TailoredWeight weight) => - Entries.TryGetValue(character.ToString(), out weight) || - Entries.TryGetValue(character.ToString().ToUpperInvariant(), out weight); + TryGet(new ReadOnlySpan(in character), out weight); private bool TryLongest(ReadOnlySpan text, int start, out TailoredWeight weight, out int consumed) { for (int length = Math.Min(MaxLength, text.Length - start); length >= 1; length--) { - ReadOnlySpan candidate = text.Slice(start, length); - if (Entries.TryGetValue(candidate.ToString(), out weight) || - Entries.TryGetValue(candidate.ToString().ToUpperInvariant(), out weight)) + if (TryGet(text.Slice(start, length), out weight)) { consumed = length; return true; @@ -122,6 +165,16 @@ private bool TryLongest(ReadOnlySpan text, int start, out TailoredWeight w consumed = 0; return false; } + + /// The entry for as written, else for its invariant uppercase — the order + /// the class remarks give, which lets a locale disagree with invariant casing. + private bool TryGet(ReadOnlySpan key, out TailoredWeight weight) + { + if (_bySpan.TryGetValue(key, out weight)) return true; + Span upper = stackalloc char[key.Length]; + key.ToUpperInvariant(upper); + return _bySpan.TryGetValue(upper, out weight); + } } /// @@ -134,17 +187,16 @@ private bool TryLongest(ReadOnlySpan text, int start, out TailoredWeight w /// Every weight here was measured from ACE: an indexed text column built by ACE inside a database carrying /// the order, with the stored index keys read back (ContractionProbeTest, /// LocaleFixtureCollationProbeTest) and then asserted byte-for-byte against this encoder over the -/// whole of printable ASCII, Latin-1 and Latin Extended-A (LocaleCollationAccessTests). +/// whole of printable ASCII, Latin-1 and Latin Extended-A (CreatedDatabaseCollationAccessTests). /// internal static class JetLocaleTailoring { - private const byte DefaultSecondary = 0x02; - /// The tailoring for a collation, or null when it has none — either because it is General /// itself, or because LibRed cannot express it. An empty tailoring is meaningful and not the same /// as null: it records that the order was measured to be indistinguishable from General. public static LocaleTailoring? For(Collation collation) => Tailorings.GetValueOrDefault(collation) + ?? JetCjkSortOrders.For(collation) ?? (collation is { Version: 0, SortId: 0 } && GeneralV0.Contains(collation.Order) ? None : null); /// Shared by every order in : no entries, no reversal, nothing to do. diff --git a/src/LibRed/LibRed.Core/Storage/JetTextCollation.cs b/src/LibRed/LibRed.Core/Storage/JetTextCollation.cs index abbebc075..43e3f907f 100644 --- a/src/LibRed/LibRed.Core/Storage/JetTextCollation.cs +++ b/src/LibRed/LibRed.Core/Storage/JetTextCollation.cs @@ -1,3 +1,5 @@ +using static LibRed.Storage.IndexKeyCodec; + namespace LibRed.Storage; /// @@ -14,7 +16,7 @@ namespace LibRed.Storage; /// base letter's primary weight and record the accent in a secondary section (see below); a handful of /// other characters are still reported unencodable. /// -/// The sort-order version is the column descriptor's 0x0D field (spec §3.4). Access 2010+ +/// The sort-order version is the column descriptor's 0x0E field (spec §3.4). Access 2010+ /// introduced a different default "General" order (1033, version 1) with other key bytes; this class /// does not implement it. A version-1 column/index therefore needs a separate weight table — /// see the spec §10.4 note. @@ -22,11 +24,6 @@ namespace LibRed.Storage; /// internal static class JetTextCollation { - private const byte EndPrimary = 0x01; - private const byte EndKey = 0x00; - private const byte InlineStart = 0x80; - private const byte InlineMid = 0x06; - /// ARABIC SHADDA — the gemination mark. Not an ignorable like the harakat around it: it /// doubles the preceding weight. See the rule in TryEncode. private const char Shadda = (char)0x0651; @@ -70,7 +67,6 @@ internal static class JetTextCollation [(char)0x0650] = 0xA5, // kasra [(char)0x0652] = 0xA6, // sukun }; - private const byte DefaultSecondary = 0x02; // a character with no accent // Secondary (diacritic) weight per Unicode combining mark — depends only on the accent, not the base // letter (verified against ACE: acute weighs 0x0E on a/e/i/o/u/y alike, etc.). @@ -250,49 +246,161 @@ void Add(int ligature, params int[] components) return ligatures; } + /// The word-sort inline code ACE gives a control character other than NUL and tab through carriage + /// return: the 60 of them — U+0001–0008, U+000E–001F, DEL and U+0080–009F — take consecutive codes + /// 0x03–0x3D in code-point order (measured against ACE, every one). + private static bool TryGetControlInlineCode(char c, out byte code) + { + int? value = c switch + { + >= '\u0001' and <= '\u0008' => c + 0x02, // 0x03–0x0A + >= '\u000E' and <= '\u001F' => c - 0x03, // 0x0B–0x1C + '\u007F' => 0x1D, + >= '\u0080' and <= '\u009F' => c - 0x62, // 0x1E–0x3D + _ => null, + }; + code = (byte)(value ?? 0); + return value is not null; + } + + /// + /// Ends a text key the way both sort-order versions end it, from the weights a version has worked out: the + /// primaries, the diacritic section, the kana section, the inline records, and . + /// [MS-UCODEREF] frames the key as primaries SEP diacritics SEP case SEP extra SEP specials TERM, and Access + /// emits that frame leaving the case section EMPTY — which is why case and width fold, since the Case Weight is + /// where width lives. + /// + /// The key being built. + /// The primary weight bytes. + /// One diacritic weight per primary weight. + /// A French-style order, which writes the diacritic section backwards. + /// Per kana, whether it is a small form. + /// Per kana, its mark code. + /// The inline records: each one's position in primary weights, script member and weight. + internal static void AppendSections(List output, List primaries, List secondaries, + bool reverseDiacritics, List kana, List marks, + List<(int Position, byte ScriptMember, byte Weight)> inline) + { + output.AddRange(primaries); + output.Add(SectionSeparator); + + // Diacritic section: only emitted when a character carries a non-default weight; it lists the weight of + // every primary weight from the first up to and including the last accented one. + // + // A French-style order writes it BACKWARDS, so accents are weighed from the end of the word and coté + // sorts before côte. The trimming mirrors too — the run of defaults comes off the LEFT, because that is + // the end that becomes trailing once reversed. [MS-UCODEREF] calls the flag IsReverseDW and states both + // halves; verified against ACE, where côté is [02 12 02 0E] trimmed to [12 02 0E] and stored 0E 02 12. + if (reverseDiacritics) + { + int firstAccent = secondaries.FindIndex(w => w != DefaultSecondary); + if (firstAccent >= 0) + for (int i = secondaries.Count - 1; i >= firstAccent; i--) output.Add(secondaries[i]); + } + else + { + int lastAccent = secondaries.FindLastIndex(w => w != DefaultSecondary); + for (int i = 0; i <= lastAccent; i++) output.Add(secondaries[i]); + } + + if (kana.Count > 0) JetKanaSection.Append(output, kana, marks); + + // The inline records follow the remaining separators: the end of the diacritics, an empty case section + // and an empty extra section — 01 01 01. A kana section fills the extra section, and the run is then + // FF 01 (measured from "あ-" and "-あ"; what the FF denotes is not established). + if (inline.Count > 0) + { + if (kana.Count > 0) + { + output.Add(KanaRunSeparator); + output.Add(SectionSeparator); + } + else + { + output.Add(SectionSeparator); + output.Add(SectionSeparator); + output.Add(SectionSeparator); + } + foreach ((int position, byte scriptMember, byte weight) in inline) + { + // [MS-UCODEREF] SpecialWeightType is (Position: 16-bit, ScriptMember, PrimaryWeight), and its + // Position is emitted big-endian — "Byte1 = Position >> 8, Byte2 = Position & 0xff" — so this is + // ONE sixteen-bit field with bit 15 set, not a 0x80 marker followed by a byte. Both readings give + // the same bytes below 0x100 and only the field reading survives past it, which is why treating + // 0x80 as a marker looked right for every short value and silently produced a wrong key for + // anything longer. Measured against ACE: a hyphen at character 250 is 83 EF. + int position16 = InlineStart << 8 | (0x07 + 4 * position); + output.Add((byte)(position16 >> 8)); + output.Add((byte)position16); + output.Add(scriptMember); + output.Add(weight); + } + } + output.Add(EndKey); + } + + /// The lists a key is built in — by and by version 1's encoder alike — kept per + /// thread and reused. A text comparison encodes both sides every time, and building five lists per string made + /// a GROUP BY over a text column spend most of its time collecting them. Safe because neither encoder re-enters + /// either, and both are done with the lists once has written the key. + internal sealed class EncodeScratch + { + [ThreadStatic] private static EncodeScratch? t_current; + + /// This thread's lists, cleared. + public static EncodeScratch ForThisThread() + { + EncodeScratch scratch = t_current ??= new EncodeScratch(); + scratch.Clear(); + return scratch; + } + + public List Primaries { get; } = []; + public List Secondaries { get; } = []; + public List<(int Position, byte ScriptMember, byte Weight)> Inline { get; } = []; + public List Kana { get; } = []; + public List KanaMarks { get; } = []; + + private void Clear() + { + Primaries.Clear(); + Secondaries.Clear(); + Inline.Clear(); + Kana.Clear(); + KanaMarks.Clear(); + } + } + /// - /// Appends the order-preserving collation key body for (everything - /// after the start flag: primary weights, end-of-primary marker, any ignorable-char inline - /// codes, and the terminator). Trailing spaces are dropped. Returns false if any character is - /// not yet supported. + /// Appends the order-preserving collation key body for (everything after the start + /// flag: primary weights, the section separators, any ignorable-char inline codes, and the terminator). + /// Trailing spaces are dropped. Returns false if any character is not yet supported. /// /// The text to encode. /// The key body is appended to this. /// Per-character overrides for a locale order other than General; null for /// General itself. See . - public static bool TryEncode(string value, List output, LocaleTailoring? tailoring = null) => - TryEncode(value, output, tailoring, out _); - - /// The text to encode. - /// The key body is appended to this. - /// Per-character overrides for a locale order other than General; null for General. - /// - /// Whether the key carries an inline word-sort section. The caller needs this to decide whether an - /// over-long entry may be truncated: the checksum that replaces the dropped bytes is unverified when - /// those bytes hold such a record, because the record is precisely what cannot be observed. - /// - /// - public static bool TryEncode( - string value, List output, LocaleTailoring? tailoring, out bool hasWordSortRecord) + public static bool TryEncode(string value, List output, LocaleTailoring? tailoring = null) { - hasWordSortRecord = false; ReadOnlySpan s = value.AsSpan().TrimEnd(' '); + EncodeScratch scratch = EncodeScratch.ForThisThread(); + // Build the primary weight bytes and a parallel secondary weight per byte (0x02 = no accent). - var primaries = new List(); - var secondaries = new List(); + List primaries = scratch.Primaries; + List secondaries = scratch.Secondaries; // Apostrophe/hyphen carry no primary weight; they record (position, code) for the inline section, // where position is the count of primary **WEIGHTS** emitted before them — not bytes. A two-byte // weight counts once: ACE puts the hyphen of "£-" at 0x0B (0x07 + 4x1) though £ is 34 A7, while // "ß-" is 0x0F because ß expands to two one-byte weights. Latin-only strings cannot tell the two // rules apart, which is why this read as "bytes" for so long. secondaries.Count is the weight count, // since every weight contributes exactly one secondary slot. - var inline = new List<(int Position, byte Code)>(); + List<(int Position, byte ScriptMember, byte Weight)> inline = scratch.Inline; // One entry per kana in the string: true for a small form. Emitted as a section of its own. - var kana = new List(); - // True where that kana is a prolonged sound mark. Packed like the small flags but with its own - // codes, into a section of its own. - var prolonged = new List(); + List kana = scratch.Kana; + // What each kana is — a letter, a prolonged sound mark or an iteration mark. Packed like the small + // flags but with its own codes, into a section of its own. + List marks = scratch.KanaMarks; // Index of the weight the last kana produced, so a following halfwidth voicing mark can reach it, // together with the vowel and small flag a following prolonged mark inherits. int kanaWeight = -1; @@ -303,12 +411,36 @@ public static bool TryEncode( // trailing record carries, since that is this character's whole contribution. Empty at the start of // the string and after a weight that folded into its predecessor — in both cases there is nothing to // double. - byte[] lastWeight = []; + CopiedPrimary lastWeight = default; + bool lastWeightIsKana = false; + + // What an iteration mark copies: the one weight — or the one inline record — the character before it + // contributed, and whether that was a kana. A character contributing two weights (ß, æ, a ligature) or a + // weight and an inline record leaves nothing to copy, and so does a mark that found nothing: ß々 and + // 々々 keep FF FF. A character contributing NOTHING is transparent — a combining accent folding into the + // letter before it, or an astral character, which version 0 ignores: e + U+0301 + 々 and a𐀀々 both copy + // the letter. Recomputed from the counts at the start of each character, except after a mark or a + // combining voicing, which leave it as it was — かゝゝ repeats か twice, and かーゝ repeats か, not the vowel. + CopiedPrimary? repeatable = null; + bool repeatableIsKana = false; + bool keepRepeatable = true; + int weightsAtCharacter = 0, recordsAtCharacter = 0; // Indexed rather than foreach, because a tailoring entry can consume several characters: a // contraction is a digraph weighing as one letter (Czech "ch", Hungarian "gy", Danish "aa"). for (int position = 0; position < s.Length; position++) { + int weights = secondaries.Count - weightsAtCharacter, records = inline.Count - recordsAtCharacter; + if (!keepRepeatable && (weights > 0 || records > 0)) + { + bool single = weights == 1 && records == 0 || weights == 0 && records == 1; + repeatable = single ? lastWeight : null; + repeatableIsKana = single && lastWeightIsKana; + } + keepRepeatable = false; + weightsAtCharacter = secondaries.Count; + recordsAtCharacter = inline.Count; + char c = s[position]; char u = char.ToUpperInvariant(c); @@ -318,13 +450,30 @@ public static bool TryEncode( // Neither half contributes, so both are skipped and an astral character vanishes from the key. if (char.IsSurrogate(c)) continue; + // The control characters, DEL and U+FEFF are in neither table; ACE keys all 66 (page-03-04 §10.4). + // NUL and U+FEFF vanish from the key altogether, as an astral character does. + if (c is '\0' or '') continue; + + // Tab, line feed, vertical tab, form feed and carriage return each weigh a two-byte primary. + if (c is >= '\t' and <= '\r') + { + AddWeight([0x08, (byte)(c - '\t' + 0x03)], DefaultSecondary); + continue; + } + // The 20 hand-verified ignorables first, then the 40 measured ones — 60 across the BMP, every dash - // and quotation form, the Arabic harakat, and the CJK and fullwidth punctuation. - if (Ignorables.TryGetValue(c, out byte code) || - JetTextCollationTableV0.TryGetInlineCode(c, out code)) + // and quotation form, the Arabic harakat, and the CJK and fullwidth punctuation — then the 60 + // remaining controls, which are word-sort ignorables too. + // A locale that weighs one as a character of its own wins: the Japanese orders give U+2015 the + // unweighted FF FF where General records it here. + if ((Ignorables.TryGetValue(c, out byte code) || + JetTextCollationTableV0.TryGetInlineCode(c, out code) || + TryGetControlInlineCode(c, out code)) && + tailoring?.TryMatchSingle(c, out _) != true) { - inline.Add((secondaries.Count, code)); - lastWeight = [InlineMid, code]; + inline.Add((secondaries.Count, WordSortScriptMember, code)); + lastWeight = new CopiedPrimary(WordSortScriptMember, code); + lastWeightIsKana = false; continue; } @@ -335,12 +484,13 @@ public static bool TryEncode( // primary — がー is 7F 0A then 7F 02, "ga" lengthened by "a" — and inherits its small flag, // while marking itself in a second packed section. With no kana ahead of it there is nothing to // lengthen, and it stays the ordinary FF FF primary the table already holds. - if (c is (char)0x30FC or (char)0xFF70 && kanaVowel != 0 && kanaWeight == secondaries.Count - 1) + if (JetKanaSection.IsProlongedSoundMark(c) && kanaVowel != 0 && kanaWeight == secondaries.Count - 1) { AddWeight([JetKanaSection.KanaPage, kanaVowel], DefaultSecondary); kana.Add(kanaSmall); - prolonged.Add(true); + marks.Add(JetKanaSection.Prolonged); kanaWeight = secondaries.Count - 1; + keepRepeatable = true; continue; } @@ -348,8 +498,9 @@ public static bool TryEncode( out byte vowel)) { AddWeight([JetKanaSection.KanaPage, sound], voicing); + lastWeightIsKana = true; kana.Add(small); - prolonged.Add(false); + marks.Add(JetKanaSection.Letter); kanaWeight = secondaries.Count - 1; kanaVowel = vowel; kanaSmall = small; @@ -362,9 +513,39 @@ public static bool TryEncode( // With no kana immediately before it there is nothing to voice, and the mark falls through to be // weighed on its own — ACE stores it as ignorable. Both halves of the guard matter: for a lone // mark kanaWeight and Count-1 are each -1, which would otherwise pass and index the list at -1. - if (c is (char)0xFF9E or (char)0xFF9F && kanaWeight >= 0 && kanaWeight == secondaries.Count - 1) + if (JetKanaSection.TryGetHalfwidthVoicing(c, out byte voiced) && kanaWeight >= 0 + && kanaWeight == secondaries.Count - 1) + { + secondaries[kanaWeight] = voiced; + keepRepeatable = true; + continue; + } + + // An iteration mark weighs as a copy of what the character before it contributed, with its own + // secondary (see JetKanaSection.TryGetIterationMark). Copying a kana makes it a kana too, recorded + // in the kana section as a repeat and inheriting the small flag — ゃゝ is small twice — and it + // stands where a halfwidth voicing mark or a long vowel mark can reach it: カヾ voices the copy, + // かゝー lengthens it. With nothing to copy it is the unweighted FF FF — and, like a shadda with nothing + // to double, it takes NO secondary slot: 々が is secondaries 03, not 02 03, which the Korean hanja + // (whose secondaries are never the default) are what showed. It leaves nothing to copy after it. + // A locale that weighs the mark as a character of its own wins, as its tailoring wins everywhere: + // Japanese radical/stroke order gives 々 a radical weight, and 人々 is then two different weights. + if (JetKanaSection.TryGetIterationMark(c, version1: false, out byte markSecondary) && + tailoring?.TryMatchSingle(c, out _) != true) { - secondaries[kanaWeight] = c == (char)0xFF9E ? (byte)0x03 : (byte)0x04; + keepRepeatable = true; + if (repeatable is null) + { + primaries.AddRange(UnweightedPrimary); + continue; + } + AddWeight(repeatable.Value.ToArray(), markSecondary, final: true); + if (repeatableIsKana) + { + kana.Add(kanaSmall); + marks.Add(JetKanaSection.Repeat); + kanaWeight = secondaries.Count - 1; + } continue; } @@ -386,9 +567,9 @@ public static bool TryEncode( if (c == Shadda) { if (lastWeight.Length > 0) - AddWeight(lastWeight, DefaultSecondary); + AddWeight(lastWeight.ToArray(), DefaultSecondary, final: true); else - primaries.AddRange((byte[])[0xFF, 0xFF]); + primaries.AddRange(UnweightedPrimary); continue; } @@ -397,7 +578,7 @@ public static bool TryEncode( tailoring.TryMatch(s, position, out TailoredWeight tailored, out int consumed, out bool repeat)) { for (int emit = repeat ? 2 : 1; emit > 0; emit--) - AddWeight(tailored.Primaries, tailored.Secondary); + AddWeight(tailored.Primaries, tailored.Secondary, final: true); position += consumed - 1; } // A ligature character weighs as its decomposition, one component at a time — ACE stores DŽ @@ -415,66 +596,7 @@ public static bool TryEncode( return false; // not handled yet } - output.AddRange(primaries); - output.Add(EndPrimary); - - // Secondary (diacritic) section: only emitted when a character carries a non-default weight; it lists - // the secondary weight of every byte from the first up to and including the last accented one. - // - // A French-style order writes it BACKWARDS, so accents are weighed from the end of the word and coté - // sorts before côte. The trimming mirrors too — the run of defaults comes off the LEFT, because that - // is the end that becomes trailing once reversed. [MS-UCODEREF] calls the flag IsReverseDW and states - // both halves; verified against ACE, where côté is [02 12 02 0E] trimmed to [12 02 0E] and stored - // 0E 02 12. - if (tailoring?.ReverseDiacritics == true) - { - int firstAccent = secondaries.FindIndex(w => w != DefaultSecondary); - if (firstAccent >= 0) - for (int i = secondaries.Count - 1; i >= firstAccent; i--) - output.Add(secondaries[i]); - } - else - { - int lastAccent = secondaries.FindLastIndex(w => w != DefaultSecondary); - for (int i = 0; i <= lastAccent; i++) - output.Add(secondaries[i]); - } - - hasWordSortRecord = inline.Count > 0; - - if (kana.Count > 0) JetKanaSection.Append(output, kana, prolonged); - - // Apostrophe/hyphen inline (tertiary) section. Its introducer depends on whether a kana section came - // first: 01 01 01 on its own, but FF 01 after one — measured from "あ-" and "-あ". - if (inline.Count > 0) - { - if (kana.Count > 0) - { - output.Add(0xFF); - output.Add(0x01); - } - else - { - output.Add(0x01); - output.Add(0x01); - output.Add(0x01); - } - foreach (var (position, code) in inline) - { - // [MS-UCODEREF] SpecialWeightType is (Position: 16 bit integer, ScriptMember, PrimaryWeight), - // and its Position is emitted big-endian — "Byte1 = Position >> 8, Byte2 = Position & 0xff" — - // so this is ONE sixteen-bit field with bit 15 set, not a 0x80 marker followed by a byte. - // Both readings give the same bytes below 0x100 and only the field reading survives past it, - // which is why treating 0x80 as a marker looked right for every short value and silently - // produced a wrong key for anything longer. Measured: a hyphen at character 250 is 83 EF. - int position16 = InlineStart << 8 | (0x07 + 4 * position); - output.Add((byte)(position16 >> 8)); - output.Add((byte)position16); - output.Add(InlineMid); - output.Add(code); - } - } - output.Add(EndKey); + AppendSections(output, primaries, secondaries, tailoring?.ReverseDiacritics == true, kana, marks, inline); return true; // One character's weights, with no contraction and no tailoring: the path a ligature's components @@ -484,10 +606,8 @@ bool WeighCharacter(char character) char upper = char.ToUpperInvariant(character); // The locale still applies to a single character — only the contraction matcher is bypassed. // A ligature's components take the locale's letters: Slovenian's DŽ is D plus SLOVENIAN's ž. - if (tailoring is not null && - (tailoring.Entries.TryGetValue(character.ToString(), out TailoredWeight tailoredOne) || - tailoring.Entries.TryGetValue(upper.ToString(), out tailoredOne))) - AddWeight(tailoredOne.Primaries, tailoredOne.Secondary); + if (tailoring is not null && tailoring.TryMatchSingle(character, out TailoredWeight tailoredOne)) + AddWeight(tailoredOne.Primaries, tailoredOne.Secondary, final: true); else if (ExtraLetters.TryGetValue(character, out TailoredWeight own) || ExtraLetters.TryGetValue(upper, out own)) AddWeight(own.Primaries, own.Secondary); @@ -504,10 +624,14 @@ bool WeighCharacter(char character) // single-character measurement cannot tell one two-byte weight from two one-byte ones. ß is two // weights (S+S); the table records it as the two bytes 6B 6B and would make it one, which is // invisible until something counts weights — an accent after it, or an inline record's position. - else if (TryAddExplicit(upper, Add)) + // Each expanded letter (ß=SS, Þ=TH, Æ=AE) is its own weight; an atomic accent (Ø, Ð) is its base + // letter with a secondary. + else if (Expansions.TryGetValue(upper, out string? expansion)) { - // handled by the expansion / atomic-accent tables + foreach (char letter in expansion) Add(Letters[letter - 'A']); } + else if (AtomicAccents.TryGetValue(upper, out (char Base, byte Secondary) atomic)) + Add(Letters[atomic.Base - 'A'], atomic.Secondary); // The measured table for the rest of the BMP — Greek, Cyrillic, Hebrew, Arabic, the Latin // extensions, punctuation, CJK and the rest. It covers nothing the hand-verified Latin-1 and // Latin Extended-A tables above do, so it cannot override anything already proven — but it DOES @@ -518,30 +642,44 @@ bool WeighCharacter(char character) // Locales share it. A locale CAN reweigh a character here, but measuring all 21 against General // showed the departures are tiny — most add one or two entries across the whole range, Croatian // eleven — and every one is listed in its tailoring, which is consulted first. - // `LocaleCollationAccessTests` asserts the whole range for every locale, so a missed departure + // `CreatedDatabaseCollationAccessTests` asserts the whole range for the locale cases, so a missed departure // fails rather than writing a silently wrong key. else if (JetTextCollationTableV0.TryGet(character, out TailoredWeight? block)) { if (block is { } weight) AddWeight(weight.Primaries, weight.Secondary); } - else if (!TryAddAccented(upper, Add)) + else if (TryDecomposeAccented(upper, out char baseLetter, out byte accent)) + Add(Letters[baseLetter - 'A'], accent); + else return false; return true; } void Add(byte primary, byte secondary = DefaultSecondary) { + if (tailoring?.LeadBytes is { } leads) primary = leads[primary]; primaries.Add(primary); secondaries.Add(secondary); - lastWeight = [primary]; + lastWeight = new CopiedPrimary(primary); + lastWeightIsKana = false; } // A primary WEIGHT may be one or two bytes, and the secondary section has one entry per weight — // not per byte. Measured against ACE: Norwegian "ö" is 7F 79 06 01 13 00, two primary bytes and a // single secondary. The inline apostrophe/hyphen section counts weights too, so both sections index // the same way; `secondaries.Count` is the weight count for both. - void AddWeight(ReadOnlySpan weight, byte secondary) + // + // `final` marks bytes that are already what the key holds — a tailored entry, or a copy of a weight + // already added — which an order moving General's lead bytes (Korean) must not move a second time. + void AddWeight(ReadOnlySpan weight, byte secondary, bool final = false) { + if (!final && !weight.IsEmpty && tailoring?.LeadBytes is { } leads && leads[weight[0]] != weight[0]) + { + byte[] moved = weight.ToArray(); + moved[0] = leads[moved[0]]; + weight = moved; + } + // A weight with NO primary carries only its secondary, and where something precedes it that // secondary FOLDS into the one before rather than taking a slot of its own. ACE encodes ไก่ as // two weights with secondaries 03 06 — the tone mark's 03 added to the ก's 03 — not as three @@ -559,33 +697,19 @@ void AddWeight(ReadOnlySpan weight, byte secondary) foreach (byte b in weight) primaries.Add(b); secondaries.Add(secondary); - lastWeight = weight.ToArray(); - } - } - - /// The explicit, hand-verified half: a multi-letter expansion (ß=SS, Þ=TH, Æ=AE) or an atomic - /// accent (Ø, Ð). Each expanded letter is its own weight — the part no single-character - /// measurement can capture, since a key cannot show whether two bytes are one weight or two — so this is - /// consulted ahead of the measured table. - private static bool TryAddExplicit(char u, Action add) - { - if (Expansions.TryGetValue(u, out string? expansion)) - { - foreach (char letter in expansion) add(Letters[letter - 'A'], DefaultSecondary); - return true; + lastWeight = CopiedPrimary.Of(weight); + lastWeightIsKana = false; } - if (AtomicAccents.TryGetValue(u, out (char Base, byte Secondary) atomic)) - { - add(Letters[atomic.Base - 'A'], atomic.Secondary); - return true; - } - return false; } /// The derived half: a Unicode canonical decomposition into a base A–Z letter plus one combining /// mark we hold a weight for. Guesswork beside a measurement, so it runs last of all. - private static bool TryAddAccented(char u, Action add) + /// It answers rather than adding, as the explicit tables above it do inline: handing the encoder's + /// own Add in as a delegate made its captured state a heap object, allocated for every key encoded. + private static bool TryDecomposeAccented(char u, out char letter, out byte weight) { + letter = default; + weight = 0; // Normalize throws on anything that is not a well-formed scalar — an unpaired surrogate, or a // NONCHARACTER (U+FDD0..U+FDEF and any code point ending FFFE/FFFF). Those are legal in a .NET // string, so refuse them rather than letting an ArgumentException escape a Try- method. @@ -594,9 +718,9 @@ private static bool TryAddAccented(char u, Action add) // Canonical decomposition: a base A–Z letter followed by one combining diacritic we know. string nfd = u.ToString().Normalize(System.Text.NormalizationForm.FormD); - if (nfd.Length == 2 && nfd[0] is >= 'A' and <= 'Z' && DiacriticWeights.TryGetValue(nfd[1], out byte weight)) + if (nfd.Length == 2 && nfd[0] is >= 'A' and <= 'Z' && DiacriticWeights.TryGetValue(nfd[1], out weight)) { - add(Letters[nfd[0] - 'A'], weight); + letter = nfd[0]; return true; } return false; diff --git a/src/LibRed/LibRed.Core/Storage/JetTextCollationTableV0.cs b/src/LibRed/LibRed.Core/Storage/JetTextCollationTableV0.cs index ce0076627..e9c2da0aa 100644 --- a/src/LibRed/LibRed.Core/Storage/JetTextCollationTableV0.cs +++ b/src/LibRed/LibRed.Core/Storage/JetTextCollationTableV0.cs @@ -29,13 +29,20 @@ internal static class JetTextCollationTableV0 public static bool TryGet(char c, out TailoredWeight? weight) { Table table = Loaded.Value; - int index = Array.BinarySearch(table.CodePoints, c); + int index = table.Slots[c] - 1; if (index < 0) { weight = null; return false; } if (table.Lengths[index] == IgnorableLength) { weight = null; return true; } - int start = table.PrimaryOffsets[index]; - int length = table.Lengths[index]; - weight = new TailoredWeight(table.Primaries[start..(start + length)], table.Secondaries[index]); + // Each character's primary bytes are sliced out once and kept: a text comparison weighs every character + // of both sides, and a fresh slice each time was an allocation per character. A race only builds the + // same slice twice. + byte[]? primaries = table.PrimarySlices[index]; + if (primaries is null) + { + int start = table.PrimaryOffsets[index]; + table.PrimarySlices[index] = primaries = table.Primaries[start..(start + table.Lengths[index])]; + } + weight = new TailoredWeight(primaries, table.Secondaries[index]); return true; } @@ -46,7 +53,7 @@ public static bool TryGet(char c, out TailoredWeight? weight) public static bool TryGetInlineCode(char c, out byte code) { Table table = Loaded.Value; - int index = Array.BinarySearch(table.InlineCodePoints, c); + int index = table.InlineSlots[c] - 1; code = index < 0 ? (byte)0 : table.InlineCodes[index]; return index >= 0; } @@ -66,7 +73,7 @@ public static bool TryGetInlineCode(char c, out byte code) public static bool TryGetKana(char c, out byte sound, out byte secondary, out bool small, out byte vowel) { Table table = Loaded.Value; - int index = Array.BinarySearch(table.KanaCodePoints, c); + int index = table.KanaSlots[c] - 1; if (index < 0) { sound = 0; secondary = 0; small = false; vowel = 0; return false; } sound = table.KanaSounds[index]; secondary = table.KanaSecondaries[index]; @@ -75,11 +82,25 @@ public static bool TryGetKana(char c, out byte sound, out byte secondary, out bo return true; } + /// Each of the three sets is reached through a slot per BMP code point — its entry's index plus one, + /// zero for none — rather than a binary search: every character of every text compared is looked up in all + /// three. 128 KB apiece, where the searches were most of a comparison's time. private sealed record Table( - char[] CodePoints, byte[] Lengths, int[] PrimaryOffsets, byte[] Primaries, byte[] Secondaries, - char[] InlineCodePoints, byte[] InlineCodes, - char[] KanaCodePoints, byte[] KanaSounds, byte[] KanaSecondaries, byte[] KanaSmall, - byte[] KanaVowels); + ushort[] Slots, byte[] Lengths, int[] PrimaryOffsets, byte[] Primaries, byte[] Secondaries, + ushort[] InlineSlots, byte[] InlineCodes, + ushort[] KanaSlots, byte[] KanaSounds, byte[] KanaSecondaries, byte[] KanaSmall, + byte[] KanaVowels) + { + public byte[]?[] PrimarySlices { get; } = new byte[Lengths.Length][]; + } + + private static ushort[] SlotsOf(char[] codePoints) + { + var slots = new ushort[char.MaxValue + 1]; + for (int i = 0; i < codePoints.Length; i++) + slots[codePoints[i]] = checked((ushort)(i + 1)); + return slots; + } // Lazy so the cost is paid only by a database that actually reaches beyond the hand-written tables. private static readonly Lazy Loaded = new(Load); @@ -135,8 +156,8 @@ private static Table Load() kanaCodePoints[i] = (char)codePoint; } - return new Table(codePoints, lengths, offsets, primaries, secondaries, inlineCodePoints, inlineCodes, - kanaCodePoints, kanaSounds, kanaSecondaries, kanaSmall, kanaVowels); + return new Table(SlotsOf(codePoints), lengths, offsets, primaries, secondaries, SlotsOf(inlineCodePoints), inlineCodes, + SlotsOf(kanaCodePoints), kanaSounds, kanaSecondaries, kanaSmall, kanaVowels); } private static byte[] ReadStream(BinaryReader reader) diff --git a/src/LibRed/LibRed.Core/Storage/JetTextCollationV1.cs b/src/LibRed/LibRed.Core/Storage/JetTextCollationV1.cs index 8a6baa53c..7fdf7f67a 100644 --- a/src/LibRed/LibRed.Core/Storage/JetTextCollationV1.cs +++ b/src/LibRed/LibRed.Core/Storage/JetTextCollationV1.cs @@ -2,6 +2,7 @@ using System.Diagnostics.CodeAnalysis; using System.IO.Compression; using System.Text; +using static LibRed.Storage.IndexKeyCodec; namespace LibRed.Storage; @@ -20,11 +21,6 @@ namespace LibRed.Storage; /// internal static class JetTextCollationV1 { - private const byte EndPrimary = 0x01; - private const byte EndKey = 0x00; - private const byte InlineStart = 0x80; - private const byte DefaultSecondary = 0x02; - // The script members below were derived by measuring ACE, and [MS-UCODEREF] "GetWindowsSortKey // Pseudocode" names every one of them. Its constants are UNSORTABLE 0, NONSPACE_MARK 1, EXPANSION 2, // EASTASIA_SPECIAL 3, JAMO_SPECIAL 4, EXTENSION_A 5, PUNCTUATION 6, SYMBOL_1..6 7-12, DIGIT 13, LATIN 14. @@ -32,11 +28,7 @@ internal static class JetTextCollationV1 // rather than being weighed the ordinary way — which is exactly the set of classes needing bespoke // handling here, arrived at one measurement at a time. - /// [MS-UCODEREF] PUNCTUATION. Characters that carry no primary weight but are recorded - /// positionally so co-op stays beside coop. The apostrophe and hyphen live here (their - /// 0x80/0x82 inline codes are simply their Alphabetic Weights), which is why exactly those - /// two are special — it is the platform's rule, not an Access one. - private const byte WordSortScriptMember = 6; + // PUNCTUATION 6 is IndexKeyCodec.WordSortScriptMember, shared with version 0. /// [MS-UCODEREF] EXTENSION_A. The CJK ideographs, their extensions, the compatibility /// forms and the Kangxi radicals. ACE gives every one of them a four-byte primary FD FF AW DW and @@ -71,44 +63,65 @@ internal static class JetTextCollationV1 /// false if any character has no weight in the table (the caller reports it rather than emitting a key /// that would sort wrongly). /// - public static bool TryEncode(string value, List output, LocaleTailoring? tailoring = null) => - TryEncode(value, output, tailoring, out _); - /// The text to encode. /// The key body is appended to this. /// Per-character overrides for a version-1 locale order other than General; null /// for General itself. The same mechanism as version 0 uses, and the same six devices — the entries just /// carry a two-byte (Script Member, Alphabetic Weight) primary instead of v0's single byte. - /// - /// Whether the key carries an inline word-sort section. The caller needs this to decide whether an - /// over-long entry may be truncated: the checksum that replaces the dropped bytes is unverified when - /// those bytes hold such a record, because the record is precisely what cannot be observed. - /// - public static bool TryEncode( - string value, List output, LocaleTailoring? tailoring, out bool hasWordSortRecord) + public static bool TryEncode(string value, List output, LocaleTailoring? tailoring = null) { - hasWordSortRecord = false; WeightTable table = Table.Value; ReadOnlySpan text = value.AsSpan().TrimEnd(' '); - var primaries = new List(); - var secondaries = new List(); + JetTextCollation.EncodeScratch scratch = JetTextCollation.EncodeScratch.ForThisThread(); + + List primaries = scratch.Primaries; + List secondaries = scratch.Secondaries; // Position is counted in primary *weights*, not bytes. In v0 the two coincide (one byte per weight); // here a weight is two bytes, and ACE still counts weights — verified against ACE (`O'Brien` puts the // apostrophe at 0x0B = 0x07 + 4x1 in both orders, though v1 has emitted twice as many bytes by then). - var inline = new List<(int Position, byte ScriptMember, byte AlphabeticWeight)>(); - - // The kana small/normal and prolonged-mark flags, and the running state the prolonged mark needs. - var kana = new List(); - var prolonged = new List(); + // A Han character is the exception: its FD FF marker counts as a weight of its own, so 人- puts the + // hyphen at 0x0F and 人人- at 0x17, though each Han character takes one secondary slot. A one-byte + // primary still counts once (a Lao vowel then a hyphen is 0x0B), so this is not a byte count halved. + List<(int Position, byte ScriptMember, byte Weight)> inline = scratch.Inline; + int hanMarkers = 0; + + // The kana small/normal flags and mark codes, and the running state the prolonged mark needs. + List kana = scratch.Kana; + List marks = scratch.KanaMarks; int kanaWeight = -1; byte kanaVowel = 0; bool kanaSmall = false; + // What an iteration mark copies — the rule version 0 follows (see JetTextCollation), except in WHAT is + // copied: the table's (script member, alphabetic weight) pair, not the bytes the weight was encoded as. + // The two differ exactly where the encoding rearranges the table: a Han character encodes as FD FF AW DW + // and is copied as 05 AW (人々 is FDFF3D26 053D), a jamo encodes as AW DW and is copied as 04 AW (ᄀ々 is + // C002 04C0). Everywhere else the pair IS the primary, so a kana, a Hangul syllable, a measured + // override or a tailored letter is copied as it was written. + CopiedPrimary lastUnit = new([]); + bool lastUnitIsKana = false; + CopiedPrimary? repeatable = null; + bool repeatableIsKana = false; + bool keepRepeatable = true; + int weightsAtCharacter = 0, recordsAtCharacter = 0; + // Indexed rather than foreach, because a tailoring entry can consume several characters: a // contraction is a digraph weighing as one letter (Croatian "dž", "lj", "nj"). for (int position = 0; position < text.Length; position++) { + // A character contributing nothing is transparent, as in version 0: e + U+0301 + 々 copies the e. + int weights = secondaries.Count - weightsAtCharacter, records = inline.Count - recordsAtCharacter; + if (!keepRepeatable && (weights > 0 || records > 0)) + { + bool single = weights == 1 && records == 0 || weights == 0 && records == 1; + repeatable = single ? lastUnit : null; + repeatableIsKana = single && lastUnitIsKana; + } + keepRepeatable = false; + weightsAtCharacter = secondaries.Count; + recordsAtCharacter = inline.Count; + char character = text[position]; // A locale tailoring overrides everything below it. The mechanism is v0's exactly — the entries @@ -123,6 +136,8 @@ public static bool TryEncode( primaries.AddRange(tailored.Primaries); secondaries.Add(tailored.Secondary); } + lastUnit = new(tailored.Primaries); + lastUnitIsKana = false; position += consumed - 1; continue; } @@ -148,15 +163,16 @@ public static bool TryEncode( // The prolonged sound mark lengthens the preceding kana's VOWEL, so it takes that vowel's primary // and inherits its small flag, while marking itself in a second packed section. With no kana ahead // of it there is nothing to lengthen, and it falls through to the ordinary table weight. - if (character is (char)0x30FC or (char)0xFF70 && kanaVowel != 0 && + if (JetKanaSection.IsProlongedSoundMark(character) && kanaVowel != 0 && kanaWeight == secondaries.Count - 1) { primaries.Add(JetKanaSection.KanaPage); primaries.Add(kanaVowel); secondaries.Add(DefaultSecondary); kana.Add(kanaSmall); - prolonged.Add(true); + marks.Add(JetKanaSection.Prolonged); kanaWeight = secondaries.Count - 1; + keepRepeatable = true; continue; } @@ -165,13 +181,36 @@ public static bool TryEncode( // single-character sweep cannot catch it, and it has to be carried over from v0 deliberately. // Both halves of the guard matter: for a lone mark kanaWeight and Count-1 are each -1, which // would otherwise pass and index the list at -1. - if (character is (char)0xFF9E or (char)0xFF9F && kanaWeight >= 0 && + if (JetKanaSection.TryGetHalfwidthVoicing(character, out byte voiced) && kanaWeight >= 0 && kanaWeight == secondaries.Count - 1) { - secondaries[kanaWeight] = character == (char)0xFF9E ? (byte)0x03 : (byte)0x04; + secondaries[kanaWeight] = voiced; + keepRepeatable = true; continue; } + // An iteration mark: a copy of what came before, with the mark's own secondary — the same rule as + // version 0, with 〻 and ꀕ added (JetKanaSection.TryGetIterationMark). A tailoring that weighs the + // mark as a character of its own has already taken it above. With nothing to copy it is weighed + // alone below, as the FF FF the measured overrides hold — with no secondary slot, which is what the + // overrides carry — and leaves nothing to copy after it. + if (JetKanaSection.TryGetIterationMark(character, version1: true, out byte markSecondary)) + { + keepRepeatable = true; + if (repeatable is { } copy) + { + copy.AppendTo(primaries); + secondaries.Add(markSecondary); + if (repeatableIsKana) + { + kana.Add(kanaSmall); + marks.Add(JetKanaSection.Repeat); + kanaWeight = secondaries.Count - 1; + } + continue; + } + } + if (JetTextCollationTableV0.TryGetKana( character, out byte sound, out byte voicing, out bool small, out byte vowel)) { @@ -188,8 +227,10 @@ public static bool TryEncode( primaries.Add(JetKanaSection.KanaPage); primaries.Add(sound); secondaries.Add(voicing); + lastUnit = new(JetKanaSection.KanaPage, sound); + lastUnitIsKana = true; kana.Add(small); - prolonged.Add(false); + marks.Add(JetKanaSection.Letter); kanaWeight = secondaries.Count - 1; kanaVowel = vowel; kanaSmall = small; @@ -216,6 +257,8 @@ public static bool TryEncode( { foreach (byte b in measuredPrimaries) primaries.Add(b); foreach (byte b in measuredSecondaries) secondaries.Add(b); + lastUnit = new(measuredPrimaries.ToArray()); + lastUnitIsKana = false; continue; } @@ -241,8 +284,10 @@ public static bool TryEncode( primaries.Add(JetKanaSection.KanaPage); primaries.Add(tableSound); secondaries.Add(tableVoicing); + lastUnit = new(JetKanaSection.KanaPage, tableSound); + lastUnitIsKana = true; kana.Add(isSmall); - prolonged.Add(false); + marks.Add(JetKanaSection.Letter); kanaWeight = secondaries.Count - 1; kanaVowel = 0; kanaSmall = isSmall; @@ -260,11 +305,13 @@ public static bool TryEncode( // expanding a ligature could trip a digraph entry that the original text never contained. foreach (char expanded in sequence) { - if (tailoring is not null && + if (tailoring is { WeighsDecompositions: true } && tailoring.TryMatchSingle(expanded, out TailoredWeight component)) { primaries.AddRange(component.Primaries); secondaries.Add(component.Secondary); + lastUnit = new(component.Primaries); + lastUnitIsKana = false; continue; } if (!Append(expanded)) return false; @@ -285,12 +332,18 @@ bool Append(char character) // instead records inline as 0x83 — a real difference between the two weight tables). if (scriptMember == 0 && alphabetic == 0) return true; + // Whatever this character contributes, the pair is what an iteration mark after it would copy. + if (alphabetic != 0) + { + lastUnit = new(scriptMember, alphabetic); + lastUnitIsKana = false; + } + if (scriptMember == WordSortScriptMember) { - // secondaries.Count is the weight count: every weight contributes exactly one secondary slot, - // whereas primaries.Count/2 assumed each weight is two bytes — which the four-byte Han - // primary above breaks. - inline.Add((secondaries.Count, scriptMember, alphabetic)); + // secondaries.Count is the weight count — every weight contributes exactly one secondary slot — + // plus one for each Han marker (see the declaration of `inline`). + inline.Add((secondaries.Count + hanMarkers, scriptMember, alphabetic)); return true; } @@ -300,11 +353,11 @@ bool Append(char character) // U+4E00 ACE 7F FD FF 3C 6A 01 00, where the NLS entry is SM 05, AW 3C, DW 6A. if (scriptMember == HanScriptMember) { - primaries.Add(0xFD); - primaries.Add(0xFF); + primaries.AddRange(HanPrimaryMarker); primaries.Add(alphabetic); primaries.Add(diacritic); secondaries.Add(DefaultSecondary); + hanMarkers++; return true; } @@ -341,67 +394,12 @@ bool Append(char character) return true; } - output.AddRange(primaries); - output.Add(EndPrimary); - - // Secondary section: emitted up to and including the last character carrying a non-default accent — - // or BACKWARDS from the last to the first accented one for a French-style order, where the trimming - // mirrors too because the leading defaults become the trailing ones once reversed. Identical to v0's - // rule; ReverseDiacritics lives on the shared LocaleTailoring, and this encoder used to ignore it, - // so a v1 tailoring that set the flag was accepted by IsIndexKeyEncodable and then silently produced - // a forward section. Latent — the one order that sets it (French) is v0 — but a silently wrong key is - // the exact failure this subsystem exists to prevent, so the two encoders agree rather than differ. - if (tailoring?.ReverseDiacritics == true) - { - int firstAccent = secondaries.FindIndex(weight => weight != DefaultSecondary); - if (firstAccent >= 0) - for (int i = secondaries.Count - 1; i >= firstAccent; i--) output.Add(secondaries[i]); - } - else - { - int lastAccent = secondaries.FindLastIndex(weight => weight != DefaultSecondary); - for (int i = 0; i <= lastAccent; i++) output.Add(secondaries[i]); - } - - hasWordSortRecord = inline.Count > 0; - - if (kana.Count > 0) JetKanaSection.Append(output, kana, prolonged); - - // Those three 0x01s are not an "introducer" but three SECTION SEPARATORS. [MS-UCODEREF] gives the key - // as primaries SEP diacritics SEP case SEP extra SEP specials TERM, and Access emits the same frame - // while leaving the case section EMPTY — which is why case and width fold, since the Case Weight is - // where width lives. So the run is: end of diacritics, an empty case section, an empty extra section. - // A kana section fills that extra section, and the run shortens accordingly. - if (inline.Count > 0) - { - if (kana.Count > 0) - { - output.Add(0xFF); - output.Add(EndPrimary); - } - else - { - output.Add(EndPrimary); - output.Add(EndPrimary); - output.Add(EndPrimary); - } - foreach ((int position, byte scriptMember, byte alphabetic) in inline) - { - // [MS-UCODEREF] SpecialWeightType is (Position: 16-bit, ScriptMember, PrimaryWeight), and its - // Position is emitted big-endian — "Byte1 = Position >> 8, Byte2 = Position & 0xff" — so this - // is ONE sixteen-bit field with bit 15 set, not a 0x80 marker followed by a byte. Both - // readings give the same bytes below 0x100 and only the field reading survives past it, which - // is why treating 0x80 as a marker looked right for every short value and silently produced a - // wrong key for anything longer. Measured against ACE: a hyphen at character 250 is 83 EF. - int position16 = InlineStart << 8 | (0x07 + 4 * position); - output.Add((byte)(position16 >> 8)); - output.Add((byte)position16); - output.Add(scriptMember); - output.Add(alphabetic); - } - } - - output.Add(EndKey); + // ReverseDiacritics lives on the shared LocaleTailoring, and this encoder used to ignore it, so a v1 + // tailoring that set the flag was accepted by IsIndexKeyEncodable and then silently produced a forward + // section. Latent — the one order that sets it (French) is v0 — but the two versions now write the + // trailing sections through one method, so they cannot differ there again. + JetTextCollation.AppendSections(output, primaries, secondaries, tailoring?.ReverseDiacritics == true, + kana, marks, inline); return true; } @@ -477,15 +475,26 @@ private static int ReadVarInt(byte[] data, ref int offset) } } - /// Code points sorted ascending with their weights in parallel arrays — a binary search over - /// ~58k entries, rather than a dictionary, to keep the table near 300 KB resident instead of several MB. + /// Weights in parallel arrays, reached through a slot per BMP code point — the entry's index plus + /// one, zero for none. Arrays rather than a dictionary keep the table near 400 KB resident instead of several + /// MB; the slots, 128 KB of it, replaced a binary search that every character of every comparison paid. private sealed class WeightTable( ushort[] codePoints, byte[] scriptMembers, byte[] alphabetics, byte[] diacritics, Dictionary expansions) { + private readonly ushort[] _slots = Slots(codePoints); + + private static ushort[] Slots(ushort[] codePoints) + { + var slots = new ushort[char.MaxValue + 1]; + for (int i = 0; i < codePoints.Length; i++) + slots[codePoints[i]] = checked((ushort)(i + 1)); + return slots; + } + public bool TryGetWeight(char character, out byte scriptMember, out byte alphabetic, out byte diacritic) { - int index = Array.BinarySearch(codePoints, (ushort)character); + int index = _slots[character] - 1; if (index < 0) { scriptMember = alphabetic = diacritic = 0; diff --git a/src/LibRed/LibRed.Core/Storage/JetTextCollationV1Overrides.cs b/src/LibRed/LibRed.Core/Storage/JetTextCollationV1Overrides.cs index 6e055e462..d1db52806 100644 --- a/src/LibRed/LibRed.Core/Storage/JetTextCollationV1Overrides.cs +++ b/src/LibRed/LibRed.Core/Storage/JetTextCollationV1Overrides.cs @@ -44,6 +44,7 @@ public static bool TryGet(char c, out ReadOnlySpan primaries, out ReadOnly primaries = secondaries = default; if (Suppressed) return false; Table table = Loaded.Value; + if (!IsSet(table.OverrideBits, c)) return false; int index = Array.BinarySearch(table.CodePoints, c); if (index < 0) return false; primaries = table.PrimaryBytes.AsSpan(table.PrimaryOffsets[index], table.PrimaryLengths[index]); @@ -59,21 +60,21 @@ public static bool TryGet(char c, out ReadOnlySpan primaries, out ReadOnly /// them. They are stored as runs rather than weights, since the only fact worth keeping is membership: /// 5,029 characters collapse into a few hundred ranges. /// - public static bool IsIgnorable(char c) - { - if (Suppressed) return false; - int[] starts = Loaded.Value.IgnorableStarts; - int index = Array.BinarySearch(starts, (int)c); - if (index >= 0) return true; - index = ~index - 1; - return index >= 0 && c < starts[index] + Loaded.Value.IgnorableLengths[index]; - } + public static bool IsIgnorable(char c) => !Suppressed && IsSet(Loaded.Value.IgnorableBits, c); + /// Both questions are asked of every character a version-1 key is built from, and the answer is + /// almost always no: 501 overrides and about 5,000 ignorables in 65,536 code points. A bit per code point — + /// 8 KB a set — answers in one read, where a binary search over each was a seventh of building a key. The + /// override's weights are still found by searching, but only for a character that has some. private sealed record Table( char[] CodePoints, byte[] PrimaryLengths, int[] PrimaryOffsets, byte[] PrimaryBytes, byte[] SecondaryLengths, int[] SecondaryOffsets, byte[] SecondaryBytes, - int[] IgnorableStarts, int[] IgnorableLengths); + ulong[] OverrideBits, ulong[] IgnorableBits); + + private static bool IsSet(ulong[] bits, char c) => (bits[c >> 6] & (1UL << (c & 63))) != 0; + + private static void Set(ulong[] bits, int c) => bits[c >> 6] |= 1UL << (c & 63); private static readonly Lazy
Loaded = new(Load); @@ -97,34 +98,38 @@ private static Table Load() var codePoints = new char[count]; var primaryOffsets = new int[count]; var secondaryOffsets = new int[count]; + var overrideBits = new ulong[(char.MaxValue + 1) / 64]; int codePoint = 0, cursor = 0, primary = 0, secondary = 0; for (int i = 0; i < count; i++) { codePoint += ReadVarInt(deltas, ref cursor); codePoints[i] = (char)codePoint; + Set(overrideBits, codePoint); primaryOffsets[i] = primary; secondaryOffsets[i] = secondary; primary += primaryLengths[i]; secondary += secondaryLengths[i]; } - var starts = new int[rangeCount]; - var lengths = new int[rangeCount]; + var ignorableBits = new ulong[(char.MaxValue + 1) / 64]; int start = 0, startCursor = 0, lengthCursor = 0; for (int i = 0; i < rangeCount; i++) { start += ReadVarInt(rangeStarts, ref startCursor); - starts[i] = start; - lengths[i] = ReadVarInt(rangeLengths, ref lengthCursor); + int length = ReadVarInt(rangeLengths, ref lengthCursor); + for (int c = start; c < start + length; c++) + Set(ignorableBits, c); } return new Table( codePoints, primaryLengths, primaryOffsets, primaryBytes, secondaryLengths, secondaryOffsets, secondaryBytes, - starts, lengths); + overrideBits, ignorableBits); } - private static byte[] ReadStream(BinaryReader reader) + /// Reads one length-prefixed zlib stream — the unit every generated collation resource is built + /// from. + internal static byte[] ReadStream(BinaryReader reader) { byte[] compressed = reader.ReadBytes(reader.ReadInt32()); var output = new MemoryStream(); @@ -133,7 +138,9 @@ private static byte[] ReadStream(BinaryReader reader) return output.ToArray(); } - private static int ReadVarInt(byte[] source, ref int offset) + /// Reads a little-endian base-128 integer, seven bits a byte, the high bit set on all but the last. + /// + internal static int ReadVarInt(byte[] source, ref int offset) { int value = 0, shift = 0; while (true) diff --git a/src/LibRed/LibRed.Core/Storage/JetTextComparer.cs b/src/LibRed/LibRed.Core/Storage/JetTextComparer.cs index 37b729f2a..1319cc750 100644 --- a/src/LibRed/LibRed.Core/Storage/JetTextComparer.cs +++ b/src/LibRed/LibRed.Core/Storage/JetTextComparer.cs @@ -1,22 +1,116 @@ +using System.Collections.Concurrent; using System.Runtime.InteropServices; +using LibRed.Catalog; namespace LibRed.Storage; /// -/// Text compared in Jet's General sort order — the order ACE's text index keys hold, and the one it compares text in -/// (case folded, accents significant, ß as ss, hyphens and apostrophes weighed only after the letters, -/// trailing spaces ignored). See . +/// Text compared as ACE compares it: in one collation, by the keys its index encoder writes for that collation +/// (case folded, accents significant, hyphens and apostrophes weighed only after the letters, trailing spaces +/// ignored, and each locale's own letters where it has them — see ). /// -public static class JetTextComparer +/// +/// A query compares text in the database's collation — page 0's, never a column's (page-02b §3.4, +/// verified against ACE) — so an engine uses the one comparer gives for +/// . +/// Built on LibRed's own weight tables alone, which cover the whole Basic Multilingual Plane, so the answer +/// is the same on every platform: nothing here consults the runtime's culture data, whose collation differs +/// between ICU and Windows NLS and between their versions. An order the index encoder refuses is refused here +/// too, at the first comparison, rather than answered in some other order. +/// +public sealed class JetTextComparer : IEqualityComparer { - /// The sign of against , or null when either holds a character - /// the order does not cover. - public static int? Compare(string a, string b) + private static readonly ConcurrentDictionary Comparers = new(); + + /// The comparer for . + public static JetTextComparer For(Collation collation) => Comparers.GetOrAdd(collation, c => new JetTextComparer(c)); + + private readonly LocaleTailoring? _tailoring; + private readonly bool _version1; + + private JetTextComparer(Collation collation) { - var left = new List(a.Length + 4); - var right = new List(b.Length + 4); - if (!JetTextCollation.TryEncode(a, left) || !JetTextCollation.TryEncode(b, right)) - return null; + Collation = collation; + _tailoring = collation.IsIndexKeyEncodable ? JetLocaleTailoring.For(collation) : null; + _version1 = collation.Version == Collation.GeneralVersion; + } + + /// The collation this compares in. + public Collation Collation { get; } + + /// Whether this is a General order with nothing tailored — no letter of its own and no reversed + /// accents — so plain ASCII compares as it does in General (see PlainTextCollationTests). A locale + /// that tailors nothing counts: its keys are General's. + public bool IsUntailoredGeneral => + Collation.IsIndexKeyEncodable && _tailoring is null or { Entries.Count: 0, ReverseDiacritics: false }; + + // Both sides are encoded into this thread's buffers: a comparison is made per row, or per pair in a sort, + // and fresh lists each time were most of what a text GROUP BY allocated. + [ThreadStatic] private static List? t_left; + [ThreadStatic] private static List? t_right; + + /// The key sorts by. Two keys compare, byte for byte, as + /// compares their texts, so a sort can build each value's key once. + public byte[] Key(string text) + { + List key = t_left ??= []; + Encode(text, key); + return [.. key]; + } + + /// The sign of against . + public int Compare(string a, string b) + { + List left = t_left ??= [], right = t_right ??= []; + Encode(a, left); + Encode(b, right); return Math.Sign(CollectionsMarshal.AsSpan(left).SequenceCompareTo(CollectionsMarshal.AsSpan(right))); } + + /// The sign of against the text whose is + /// — for a side compared with many values, whose key is made once. + public int Compare(string a, byte[] bKey) + { + List left = t_left ??= []; + Encode(a, left); + return Math.Sign(CollectionsMarshal.AsSpan(left).SequenceCompareTo(bKey)); + } + + /// Whether the two texts are one value in this collation. + /// Texts identical once trailing spaces are dropped are equal in any order, so they are answered + /// without encoding either. + public bool Equals(string? x, string? y) + { + if (x is null || y is null) return x is null && y is null; + return x.AsSpan().TrimEnd(' ').SequenceEqual(y.AsSpan().TrimEnd(' ')) || Compare(x, y) == 0; + } + + /// A hash that agrees with : taken over the key. + public int GetHashCode(string text) + { + List key = t_left ??= []; + Encode(text, key); + var hash = new HashCode(); + hash.AddBytes(CollectionsMarshal.AsSpan(key)); + return hash.ToHashCode(); + } + + private void Encode(string text, List output) + { + if (!Collation.IsIndexKeyEncodable) + throw new NotSupportedException( + $"Text is compared in the database's collation, {Collation.Order} version {Collation.Version}" + + (Collation.SortId == 0 ? "" : $" sort id {Collation.SortId}") + ", which is not implemented yet."); + + output.Clear(); + bool encoded = _version1 + ? JetTextCollationV1.TryEncode(text, output, _tailoring) + : JetTextCollation.TryEncode(text, output, _tailoring); + + // Both tables cover the whole BMP, and each order handles surrogates (CollationBmpCoverageTests), so + // this cannot be reached by any string; refusing is still the answer if it ever is. + if (!encoded) + throw new NotSupportedException( + $"'{text}' holds a character the {Collation.Order} v{Collation.Version} order has no weight for."); + } } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/LongValueReader.cs b/src/LibRed/LibRed.Core/Storage/LongValueReader.cs deleted file mode 100644 index d383df7d6..000000000 --- a/src/LibRed/LibRed.Core/Storage/LongValueReader.cs +++ /dev/null @@ -1,153 +0,0 @@ -using LibRed.Formats; -using LibRed.IO; -using LibRed.Pages; - -namespace LibRed.Storage; - -/// -/// Resolves a long value (Memo / OLE) from its 12-byte in-row descriptor to the full -/// byte payload, following LVAL pages as needed. -/// -/// -/// Descriptor layout: bytes 0-3 = length with storage flags in the high two bits, bytes 4-7 = a -/// row+page pointer to the first LVAL chunk, bytes 8-11 the chain stamp (chained form only, checked against -/// the first chain page — see ). Flags: -/// 0x80 = inline (payload follows the descriptor); 0x40 = single LVAL page (the row is -/// the whole payload); otherwise the payload is chained across LVAL pages, each row -/// beginning with a 4-byte pointer to the next chunk. -/// -public sealed class LongValueReader(PageChannel channel) -{ - private readonly PageChannel _channel = channel; - - public byte[] Resolve(ReadOnlySpan descriptor) => ResolveWithPages(descriptor, out _); - - internal byte[] ResolveWithPages(ReadOnlySpan descriptor, out IReadOnlyList pages) - { - if (descriptor.Length < LongValueFormat.DescriptorSize) - throw new InvalidDataException( - $"Long-value descriptor has {descriptor.Length} bytes; expected at least {LongValueFormat.DescriptorSize}."); - - int length = System.Buffers.Binary.BinaryPrimitives.ReadInt32LittleEndian(descriptor) & LongValueFormat.LengthMask; - byte flags = (byte)(descriptor[3] & LongValueFormat.FlagMask); - if (flags is not (LongValueFormat.FlagInline or LongValueFormat.FlagSinglePage or LongValueFormat.FlagChained)) - throw new InvalidDataException($"Long-value descriptor has unsupported flags 0x{flags:X2}."); - - if ((flags & LongValueFormat.FlagInline) != 0) - { - if (length > descriptor.Length - 12) - throw new InvalidDataException( - $"Inline long value declares {length} bytes but only {descriptor.Length - 12} are present."); - pages = []; - return descriptor.Slice(12, length).ToArray(); - } - - int row = descriptor[4]; - int page = descriptor[5] | (descriptor[6] << 8) | (descriptor[7] << 16); - - if ((flags & LongValueFormat.FlagSinglePage) != 0) - { - byte[] value = ReadLvalRow(page, row); - if (value.Length != length) - throw new InvalidDataException( - $"Single-page long value declares {length} bytes but row {page}:{row} has {value.Length}."); - pages = [page]; - return value; - } - - VerifyChainStamp(descriptor, page); - return ReadChain(page, row, length, out pages); - } - - /// - /// A chained descriptor and the first page of its chain carry the same 4-byte stamp, minted when - /// the chain was written. Disagreement means the page is no longer the one this descriptor was written - /// against — the chain was rewritten, or its pages were freed and reused under another value — so the - /// bytes behind the pointer belong to something else and must not be returned as this value. - /// - /// - /// ACE enforces this and reports it as "you and another user are attempting to change the same data at - /// the same time"; the diagnosis is the point, even if the wording is about the cause rather than what - /// was found. LibRed is single-writer, so it cannot produce the interleaving ACE guards against, but it - /// can be handed a file another engine wrote and must not read a stale chain as though it were live. - /// Only the entry page is checked, because every chunk after it is reached from a page already validated. - /// - private void VerifyChainStamp(ReadOnlySpan descriptor, int page) - { - if (page <= 0 || page >= _channel.PageCount) return; // ReadChain reports the bad pointer itself - - uint declared = System.Buffers.Binary.BinaryPrimitives.ReadUInt32LittleEndian( - descriptor.Slice(LongValueFormat.ChainStampOffset, 4)); - uint stored = System.Buffers.Binary.BinaryPrimitives.ReadUInt32LittleEndian( - _channel.ReadPageShared(page).Span.Slice(_channel.Format.DataChainStampOffset, 4)); - if (declared != stored) - throw new InvalidDataException( - $"Long-value chain at page {page} carries stamp 0x{stored:X8} but its descriptor declares " - + $"0x{declared:X8}; the chain is not the one this row was written against."); - } - - private byte[] ReadChain(int page, int row, int length, out IReadOnlyList pages) - { - var result = new byte[length]; - int written = 0; - var visited = new HashSet<(int Page, int Row)>(); - var chainPages = new List(); - - while (written < length) - { - if (page == 0) - throw new InvalidDataException( - $"Long-value chain ended after {written} of {length} declared bytes."); - if (!visited.Add((page, row))) - throw new InvalidDataException($"Long-value chain contains a cycle at row {page}:{row}."); - chainPages.Add(page); - - byte[] chunk = ReadLvalRow(page, row); - if (chunk.Length < 4) - throw new InvalidDataException( - $"Long-value chain row {page}:{row} has {chunk.Length} bytes; at least 4 are required for its next pointer."); - - // Each chained chunk starts with a 4-byte pointer (row + 3-byte page) to the next. - int nextRow = chunk[0]; - int nextPage = chunk[1] | (chunk[2] << 8) | (chunk[3] << 16); - - int copy = chunk.Length - 4; - if (copy == 0) - throw new InvalidDataException($"Long-value chain row {page}:{row} makes no payload progress."); - if (copy > length - written) - throw new InvalidDataException( - $"Long-value chain row {page}:{row} exceeds the declared length by {copy - (length - written)} bytes."); - Array.Copy(chunk, 4, result, written, copy); - written += copy; - - if (written == length && nextPage != 0) - throw new InvalidDataException( - $"Long-value chain continues to page {nextPage} after its declared {length} bytes."); - page = nextPage; - row = nextRow; - } - - pages = chainPages; - return result; - } - - private byte[] ReadLvalRow(int page, int row) - { - if (page <= 0 || page >= _channel.PageCount) - throw new InvalidDataException( - $"Long-value page pointer {page} is outside the file's 1..{_channel.PageCount - 1} range."); - - var lval = new DataPage(); - lval.Read(_channel.ReadPage(page), _channel.Format); - if (!lval.IsLongValuePage) - throw new InvalidDataException($"Long-value pointer {page}:{row} targets a non-LVAL data page."); - if (row < 0 || row >= lval.RowCount) - throw new InvalidDataException( - $"Long-value row pointer {page}:{row} is outside the page's 0..{lval.RowCount - 1} range."); - RowSlot slot = lval.Rows[row]; - if (slot.IsDeleted || slot.HasOverflow) - throw new InvalidDataException( - $"Long-value pointer {page}:{row} targets a deleted or overflow row slot."); - return lval.GetRow(row).ToArray(); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/LongValueStore.cs b/src/LibRed/LibRed.Core/Storage/LongValueStore.cs new file mode 100644 index 000000000..e1811c2de --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/LongValueStore.cs @@ -0,0 +1,333 @@ +using LibRed.Formats; +using LibRed.IO; +using LibRed.Pages; +using System.Buffers.Binary; + +namespace LibRed.Storage; + +/// +/// Reads and writes a long value (Memo / OLE), owning its in-row descriptor and full +/// byte payload, following LVAL pages as needed. +/// +/// +/// The descriptor () carries the length and its , a +/// record pointer to the first LVAL chunk, and the chain stamp (chained form only, checked against the first chain +/// page — see ). An inline payload follows the descriptor; a single-page value is the +/// whole of its row; a chained one runs across LVAL pages, each row beginning with a record pointer to the next +/// chunk. +/// +public sealed class LongValueStore(PageChannel channel) +{ + /// + /// A pre-built long-value (memo/OLE) in-row descriptor, written verbatim by the row encoder instead of + /// inlining. Produced by when a value is stored on an LVAL page. + /// + internal sealed record DescriptorValue(byte[] Bytes); + + private readonly PageChannel _channel = channel; + private readonly PageAllocator _allocator = channel.Allocator; + + public byte[] Resolve(ReadOnlySpan descriptor) => ResolveWithPages(descriptor, out _); + + internal byte[] ResolveWithPages(ReadOnlySpan descriptor, out IReadOnlyList pages) + { + JetFormatBase format = _channel.Format; + (int length, StorageKind storage, int row, int page, uint stamp) = Read(descriptor, format); + + if (storage == StorageKind.Inline) + { + int size = format.LongValueDescriptorSize; + if (length > descriptor.Length - size) + throw new InvalidDataException( + $"Inline long value declares {length} bytes but only {descriptor.Length - size} are present."); + pages = []; + return descriptor.Slice(size, length).ToArray(); + } + + if (storage == StorageKind.SinglePage) + { + byte[] value = ReadLvalRow(page, row); + if (value.Length != length) + throw new InvalidDataException( + $"Single-page long value declares {length} bytes but row {page}:{row} has {value.Length}."); + pages = [page]; + return value; + } + + VerifyChainStamp(stamp, page); + return ReadChain(page, row, length, out pages); + } + + /// + /// A chained descriptor and the first page of its chain carry the same 4-byte stamp, minted when + /// the chain was written. Disagreement means the page is no longer the one this descriptor was written + /// against — the chain was rewritten, or its pages were freed and reused under another value — so the + /// bytes behind the pointer belong to something else and must not be returned as this value. + /// + /// + /// ACE enforces this and reports it as "you and another user are attempting to change the same data at + /// the same time"; the diagnosis is the point, even if the wording is about the cause rather than what + /// was found. LibRed is single-writer, so it cannot produce the interleaving ACE guards against, but it + /// can be handed a file another engine wrote and must not read a stale chain as though it were live. + /// Only the entry page is checked, because every chunk after it is reached from a page already validated. + /// + private void VerifyChainStamp(uint declared, int page) + { + if (page <= 0 || page >= _channel.PageCount) return; // ReadChain reports the bad pointer itself + + uint stored = BinaryPrimitives.ReadUInt32LittleEndian( + _channel.ReadPageShared(page).Span.Slice(_channel.Format.DataChainStampOffset, sizeof(uint))); + if (declared != stored) + throw new InvalidDataException( + $"Long-value chain at page {page} carries stamp 0x{stored:X8} but its descriptor declares " + + $"0x{declared:X8}; the chain is not the one this row was written against."); + } + + private byte[] ReadChain(int page, int row, int length, out IReadOnlyList pages) + { + var result = new byte[length]; + int written = 0; + var visited = new HashSet<(int Page, int Row)>(); + var chainPages = new List(); + + while (written < length) + { + if (page == 0) + throw new InvalidDataException( + $"Long-value chain ended after {written} of {length} declared bytes."); + if (!visited.Add((page, row))) + throw new InvalidDataException($"Long-value chain contains a cycle at row {page}:{row}."); + chainPages.Add(page); + + byte[] chunk = ReadLvalRow(page, row); + if (chunk.Length < PageBuffer.RecordPointerSize) + throw new InvalidDataException( + $"Long-value chain row {page}:{row} has {chunk.Length} bytes; at least " + + $"{PageBuffer.RecordPointerSize} are required for its next pointer."); + + // Each chained chunk starts with a record pointer to the next. + (int nextRow, int nextPage) = PageBuffer.ReadRecordPointer(chunk, 0); + + int copy = chunk.Length - PageBuffer.RecordPointerSize; + if (copy == 0) + throw new InvalidDataException($"Long-value chain row {page}:{row} makes no payload progress."); + if (copy > length - written) + throw new InvalidDataException( + $"Long-value chain row {page}:{row} exceeds the declared length by {copy - (length - written)} bytes."); + Array.Copy(chunk, PageBuffer.RecordPointerSize, result, written, copy); + written += copy; + + if (written == length && nextPage != 0) + throw new InvalidDataException( + $"Long-value chain continues to page {nextPage} after its declared {length} bytes."); + page = nextPage; + row = nextRow; + } + + pages = chainPages; + return result; + } + + private byte[] ReadLvalRow(int page, int row) + { + if (page <= 0 || page >= _channel.PageCount) + throw new InvalidDataException( + $"Long-value page pointer {page} is outside the file's 1..{_channel.PageCount - 1} range."); + + var lval = new DataPage(); + lval.Read(_channel.ReadPage(page), _channel.Format); + if (!lval.IsLongValuePage) + throw new InvalidDataException($"Long-value pointer {page}:{row} targets a non-LVAL data page."); + if (row < 0 || row >= lval.RowCount) + throw new InvalidDataException( + $"Long-value row pointer {page}:{row} is outside the page's 0..{lval.RowCount - 1} range."); + DataPage.RowSlot slot = lval.Rows[row]; + if (slot.IsDeleted || slot.HasOverflow) + throw new InvalidDataException( + $"Long-value pointer {page}:{row} targets a deleted or overflow row slot."); + return lval.GetRow(row).ToArray(); + } + + /// Writes across one or more LVAL pages, returning its descriptor + /// and the pages used (all owned; the last is also free, having spare room). + internal (byte[] Descriptor, IReadOnlyList OwnedPages, int FreePage) Write(byte[] payload) + { + JetFormatBase format = _channel.Format; + ValidateLength(payload.Length, format); + if (payload.Length <= format.LongValueMaxSinglePage) + { + int page = _allocator.Allocate(); + WriteChunkPage(page, payload); // a single-page row is the payload itself (no next pointer) + return ( + Descriptor(format, payload.Length, StorageKind.SinglePage, page), [page], page); + } + + // Chained: split into chunks that each fit a row after the record pointer to the next. + int maxChunkData = format.LongValueMaxRowSize - PageBuffer.RecordPointerSize; + int chunkCount = (payload.Length + maxChunkData - 1) / maxChunkData; + var pages = new int[chunkCount]; + for (int i = 0; i < chunkCount; i++) pages[i] = _allocator.Allocate(); + + // The chain stamp goes in two places and must match in both: here on the first chunk page and in the + // descriptor below. It is a version tag on the CHAIN, minted per write of it, so a reader arriving + // through a descriptor can tell that the pages it is about to follow are the ones that descriptor was + // written against and not a later value's. Only the entry page carries it — every chunk after that is + // reached from a page already validated. ACE uses GetTickCount(); the value is arbitrary and only the + // agreement is checked, so matching its choice keeps our pages the shape Access produces. + uint stamp = (uint)Environment.TickCount; + + for (int i = 0; i < chunkCount; i++) + { + int start = i * maxChunkData; + int len = Math.Min(maxChunkData, payload.Length - start); + int nextPage = i + 1 < chunkCount ? pages[i + 1] : 0; + + var row = new byte[PageBuffer.RecordPointerSize + len]; + PageBuffer.WriteRecordPointer(row, 0, row: 0, nextPage); // always row 0 — one chunk per page + payload.AsSpan(start, len).CopyTo(row.AsSpan(PageBuffer.RecordPointerSize)); + WriteChunkPage(pages[i], row, i == 0 ? stamp : 0); + } + + return ( + Descriptor(format, payload.Length, StorageKind.Chained, pages[0], stamp: stamp), + pages, FreePage: 0); + } + + /// Allocates a fresh LVAL page, writes as its row 0, and returns the + /// page number — the caller records it in the column's usage maps. is as + /// for . + internal int WriteNewPage(byte[] row, byte[]? uncompressed = null) + { + int page = _allocator.Allocate(); + WriteChunkPage(page, row, uncompressed: uncompressed); + return page; + } + + /// Appends to an existing LVAL page if it has room, returning the new + /// row index and the page's remaining free space (null if it does not fit). Lets several small long + /// values share one page, the way Access packs them. + /// + /// is the value before compression, when is its + /// compressed form. ACE places such a value as though it were uncompressed — the page must have room for the + /// uncompressed bytes — writes those bytes where they would go, then the compressed row over their upper end, + /// so the rest of the uncompressed image stays behind in the page's free space. Both are measured + /// (long-values.md). + /// + internal (int Row, int RemainingFree)? TryAppend(int pageNumber, byte[] row, byte[]? uncompressed = null) + { + JetFormatBase format = _channel.Format; + if (pageNumber <= 0 || pageNumber >= _channel.PageCount) + throw new InvalidDataException($"LVAL append page {pageNumber} is outside the physical file."); + if (row.Length > format.LongValueMaxRowSize) + throw new ArgumentOutOfRangeException(nameof(row), + $"An LVAL row cannot exceed {format.LongValueMaxRowSize} bytes."); + + PageBuffer buffer = _channel.ReadPage(pageNumber); + var parsed = new DataPage(); + parsed.Read(buffer, format); + if (!parsed.IsLongValuePage) + throw new InvalidDataException($"LVAL append target {pageNumber} is not owned by the LVAL store."); + + int rowCount = parsed.RowCount; + // A long-value descriptor addresses its row with a ONE-BYTE field (the row byte of its record pointer), + // exactly as an index entry addresses a data row — so the same 256-slot ceiling applies, and passing it + // would alias one value's descriptor onto another's row with no error. Today it is unreachable: a + // payload of 64 bytes or less inlines and never arrives here, and a page leaves the free-pages map once + // it has no room left for a MinLvalValue-byte value and its slot (RowInserter), which caps a page at 104 rows + // even for the smallest thing + // that can reach it (a 33-character memo compressed to 35 bytes — compression is applied AFTER the + // inline test, so the floor is lower than the 65-byte inline limit suggests). That margin is emergent, + // not stated: it moves if the inline limit or the free-map threshold changes. Refusing the page here + // costs nothing and makes the ceiling structural — the caller allocates a fresh page, as it does when + // the page is out of room. + if (rowCount >= format.MaxRowsPerPage) return null; + + int slotSize = format.DataRowDirectoryEntrySize; + byte[] page = buffer.Span.ToArray(); + int lowest = DataPage.LowestRowOffset(page, format); + int physicalFree = lowest - DataPage.DirectoryEnd(format, rowCount); + if (parsed.FreeSpace != physicalFree) + throw new InvalidDataException( + $"LVAL page {pageNumber} declares {parsed.FreeSpace} free bytes but its row geometry has {physicalFree}."); + // Row data + its directory entry; a compressed value needs room for its uncompressed bytes. + if (physicalFree < Math.Max(row.Length, uncompressed?.Length ?? 0) + slotSize) return null; + + uncompressed?.CopyTo(page.AsSpan(lowest - uncompressed.Length)); + DataPage.TryAppendRow(page, format, row, RowSlotFlags.None, out int newRow); // room proven above + _channel.WritePage(pageNumber, page); + return (newRow, DataPage.ReadFreeSpace(page, format)); + } + + /// Writes one row () to a fresh LVAL data page, packed from the page end. + /// is the chain stamp for the first page of a chain, and zero everywhere else — + /// which is what ACE writes on a single-page value and on every chunk after the first. + /// is as for . + private void WriteChunkPage(int pageNumber, byte[] row, uint stamp = 0, byte[]? uncompressed = null) + { + JetFormatBase format = _channel.Format; + byte[] page = DataPage.NewPage(format, JetFormatBase.LongValuePageMarker); + BinaryPrimitives.WriteUInt32LittleEndian(page.AsSpan(format.DataChainStampOffset, 4), stamp); + + uncompressed?.CopyTo(page.AsSpan(format.PageSize - uncompressed.Length)); + DataPage.LayRows(page, format, [row], [RowSlotFlags.None]); // one row on an empty page always fits + _channel.WritePage(pageNumber, page); + } + + + internal static void ValidateLength(int length, JetFormatBase format) + { + if ((uint)length > (uint)format.LongValueLengthMask) + throw new ArgumentOutOfRangeException(nameof(length), "The long-value length would overwrite its storage flags."); + } + + /// A descriptor: the length with its bits, the record pointer to the value's + /// row or first chunk, and the chain stamp. The pointer is zero on an inline value, and the stamp is non-zero + /// only on a chained one, where it repeats the first chain page's own. + internal static byte[] Descriptor(JetFormatBase format, int length, StorageKind storage, + int page = 0, int row = 0, uint stamp = 0) + { + ValidateLength(length, format); + var d = new byte[format.LongValueDescriptorSize]; + BinaryPrimitives.WriteUInt32LittleEndian(d, (uint)length | (uint)storage); + PageBuffer.WriteRecordPointer(d, format.LongValueDescriptorPointerOffset, row, page); + BinaryPrimitives.WriteUInt32LittleEndian(d.AsSpan(format.LongValueDescriptorChainStampOffset, sizeof(uint)), stamp); + return d; + } + + /// A descriptor's length, storage, record pointer and chain stamp, after proving it has all its bytes + /// and a storage form that exists. Byte 3 carries the length's top byte AND the storage bits — the length runs + /// to , and a 16 MB value puts 0x01 there — so the two are + /// masked apart. + internal static (int Length, StorageKind Storage, int Row, int Page, uint Stamp) Read( + ReadOnlySpan descriptor, JetFormatBase format) + { + if (descriptor.Length < format.LongValueDescriptorSize) + throw new InvalidDataException( + $"Long-value descriptor has {descriptor.Length} bytes; expected at least {format.LongValueDescriptorSize}."); + + uint head = BinaryPrimitives.ReadUInt32LittleEndian(descriptor); + var storage = (StorageKind)(head & ~(uint)format.LongValueLengthMask); + if (storage is not (StorageKind.Inline or StorageKind.SinglePage or StorageKind.Chained)) + throw new InvalidDataException($"Long-value descriptor has unsupported flags 0x{(uint)storage >> 24:X2}."); + (int row, int page) = PageBuffer.ReadRecordPointer(descriptor, format.LongValueDescriptorPointerOffset); + uint stamp = BinaryPrimitives.ReadUInt32LittleEndian( + descriptor.Slice(format.LongValueDescriptorChainStampOffset, sizeof(uint))); + return ((int)(head & (uint)format.LongValueLengthMask), storage, row, page, stamp); + } + + /// How a long value is stored: the bits of its descriptor's first 4 bytes above + /// . + internal enum StorageKind : uint + { + /// Across several LVAL pages, one chunk per page, each beginning with a record pointer to the + /// next. + Chained = 0x00000000, + + /// A single LVAL page row: the row is the whole payload. + SinglePage = 0x40000000, + + /// In the row, after the descriptor: no LVAL page. + Inline = 0x80000000, + } + +} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/LongValueWriter.cs b/src/LibRed/LibRed.Core/Storage/LongValueWriter.cs deleted file mode 100644 index 8c57857b4..000000000 --- a/src/LibRed/LibRed.Core/Storage/LongValueWriter.cs +++ /dev/null @@ -1,196 +0,0 @@ -using LibRed.Formats; -using LibRed.IO; -using LibRed.Pages; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -/// -/// A pre-built long-value (memo/OLE) in-row descriptor, written verbatim by the row encoder instead of -/// inlining. Produced by when a value is stored on an LVAL page. -/// -public sealed record LongValueDescriptor(byte[] Bytes); - -/// -/// The result of writing a long value to LVAL page(s): the 12-byte in-row descriptor, the pages it now -/// occupies (to record in the column's owned-pages map), and the one page that still has spare room, to be -/// recorded in the free-pages map — or 0 when no page may be shared. -/// -/// -/// A chained value reports 0. Its last chunk usually leaves room, but a chained value owns its pages -/// outright: freeing the value frees every page in the chain, so a small value packed onto the tail would be -/// destroyed under a live descriptor when the chain's owner is updated or deleted. The spare room in a chain -/// tail is the price of the chain. -/// -public sealed record LongValueResult(byte[] Descriptor, IReadOnlyList OwnedPages, int FreePage); - -/// -/// Writes a long value (memo/OLE) to LVAL page(s) and returns its 12-byte in-row reference descriptor. -/// A value up to one page is a single page (flag 0x40): the row is the payload. A -/// larger value is chained (flag 0x00) across several pages: each page holds one chunk row -/// that begins with a 4-byte [row:1][page:3] pointer to the next chunk (zero on the last). -/// -/// -/// An LVAL page is a data page (type 0x01) whose owner field is the ASCII marker "LVAL"; a chunk is -/// stored as its row 0. Access's property loader (and general long-value reads) require this page form — -/// an inline value is not recognised for object properties (LvProp). -/// -public sealed class LongValueWriter(PageChannel channel) -{ - - // Access caps an LVAL page row at MAX_LONG_VALUE_ROW_SIZE (Jackcess) — 4076 on Jet4 (Jet3 = 2032), - // 4 bytes short of the page's usable space. A single-page value up to this fits in one row; a chained - // chunk row is this size (a 4-byte next-pointer + up to 4072 data bytes). Verified against ACE's own - // chained OLE values (Northwind Employee photos: 4076, 4076, 2606-byte chunk rows). - private const int MaxLvalRowSize = 4076; - private const int MaxChunkData = MaxLvalRowSize - 4; - - private readonly PageChannel _channel = channel; - private readonly PageAllocator _allocator = new(channel); - - /// Writes across one or more LVAL pages, returning its descriptor - /// and the pages used (all owned; the last is also free, having spare room). - public LongValueResult Write(byte[] payload) - { - LongValueFormat.ValidateLength(payload.Length); - if (payload.Length <= LongValueFormat.MaxSinglePageValue) - { - int page = _allocator.Allocate(); - WriteChunkPage(page, payload); // a single-page row is the payload itself (no next pointer) - return new LongValueResult(Descriptor(payload.Length, LongValueFormat.FlagSinglePage, page), [page], page); - } - - // Chained: split into chunks that each fit on a page after a 4-byte next-pointer. - int chunkCount = (payload.Length + MaxChunkData - 1) / MaxChunkData; - var pages = new int[chunkCount]; - for (int i = 0; i < chunkCount; i++) pages[i] = _allocator.Allocate(); - - // The chain stamp goes in two places and must match in both: here on the first chunk page and in the - // descriptor below. It is a version tag on the CHAIN, minted per write of it, so a reader arriving - // through a descriptor can tell that the pages it is about to follow are the ones that descriptor was - // written against and not a later value's. Only the entry page carries it — every chunk after that is - // reached from a page already validated. ACE uses GetTickCount(); the value is arbitrary and only the - // agreement is checked, so matching its choice keeps our pages the shape Access produces. - uint stamp = (uint)Environment.TickCount; - - for (int i = 0; i < chunkCount; i++) - { - int start = i * MaxChunkData; - int len = Math.Min(MaxChunkData, payload.Length - start); - int nextPage = i + 1 < chunkCount ? pages[i + 1] : 0; - - var row = new byte[4 + len]; - row[0] = 0; // next row (always row 0 — one chunk per page) - row[1] = (byte)nextPage; - row[2] = (byte)(nextPage >> 8); - row[3] = (byte)(nextPage >> 16); - payload.AsSpan(start, len).CopyTo(row.AsSpan(4)); - WriteChunkPage(pages[i], row, i == 0 ? stamp : 0); - } - - return new LongValueResult( - Descriptor(payload.Length, LongValueFormat.FlagChained, pages[0], stamp: stamp), pages, FreePage: 0); - } - - /// Allocates a fresh LVAL page, writes as its row 0, and returns the - /// page number — the caller records it in the column's usage maps. - public int WriteNewPage(byte[] row) - { - int page = _allocator.Allocate(); - WriteChunkPage(page, row); - return page; - } - - /// Appends to an existing LVAL page if it has room, returning the new - /// row index and the page's remaining free space (null if it does not fit). Lets several small long - /// values share one page, the way Access packs them. - public (int Row, int RemainingFree)? TryAppend(int pageNumber, byte[] row) - { - JetFormatBase format = _channel.Format; - if (pageNumber <= 0 || pageNumber >= _channel.PageCount) - throw new InvalidDataException($"LVAL append page {pageNumber} is outside the physical file."); - if (row.Length > MaxLvalRowSize) - throw new ArgumentOutOfRangeException(nameof(row), - $"An LVAL row cannot exceed {MaxLvalRowSize} bytes."); - - PageBuffer buffer = _channel.ReadPage(pageNumber); - var parsed = new DataPage(); - parsed.Read(buffer, format); - if (!parsed.IsLongValuePage) - throw new InvalidDataException($"LVAL append target {pageNumber} is not owned by the LVAL store."); - - int rowCount = parsed.RowCount; - // A long-value descriptor addresses its row with a ONE-BYTE field (Descriptor writes `d[4] = (byte)row`), - // exactly as an index entry addresses a data row — so the same 256-slot ceiling applies, and passing it - // would alias one value's descriptor onto another's row with no error. Today it is unreachable: a - // payload of 64 bytes or less inlines and never arrives here, and a page leaves the free-pages map once - // it has under MinLvalRow bytes left, which caps a page at roughly 108 rows even for the smallest thing - // that can reach it (a 33-character memo compressed to 35 bytes — compression is applied AFTER the - // inline test, so the floor is lower than the 65-byte inline limit suggests). That margin is emergent, - // not stated: it moves if the inline limit or the free-map threshold changes. Refusing the page here - // costs nothing and makes the ceiling structural — the caller allocates a fresh page, as it does when - // the page is out of room. - if (rowCount >= RowPointer.MaxRowsPerPage) return null; - - int lowest = parsed.Rows.Count == 0 ? format.PageSize : parsed.Rows.Min(r => r.Offset); - int directoryEnd = format.DataRowDirectoryOffset + rowCount * 2; - int physicalFree = lowest - directoryEnd; - if (parsed.FreeSpace != physicalFree) - throw new InvalidDataException( - $"LVAL page {pageNumber} declares {parsed.FreeSpace} free bytes but its row geometry has {physicalFree}."); - if (physicalFree < row.Length + 2) return null; // row data + its 2-byte directory entry - - byte[] page = buffer.Span.ToArray(); - - int offset = lowest - row.Length; - row.CopyTo(page.AsSpan(offset)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + rowCount * 2, 2), (ushort)offset); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), (ushort)(rowCount + 1)); - int remaining = physicalFree - row.Length - 2; - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), (ushort)remaining); - _channel.WritePage(pageNumber, page); - return (rowCount, remaining); - } - - /// The single-page (0x40) descriptor for a value stored at (, - /// ) — used when a value is packed onto an existing page at a non-zero row. - public static byte[] SinglePageDescriptor(int length, int page, int row) => - Descriptor(length, LongValueFormat.FlagSinglePage, page, row); - - /// Writes one row () to a fresh LVAL data page, packed from the page end. - /// is the chain stamp for the first page of a chain, and zero everywhere else — - /// which is what ACE writes on a single-page value and on every chunk after the first. - private void WriteChunkPage(int pageNumber, byte[] row, uint stamp = 0) - { - JetFormatBase format = _channel.Format; - var page = new byte[format.PageSize]; - page[0] = (byte)PageType.DataPage; - page[1] = 0x01; // page flags (observed constant) - BinaryPrimitives.WriteUInt32LittleEndian(page.AsSpan(format.DataOwnerOffset, 4), LongValueFormat.LvalMarker); - BinaryPrimitives.WriteUInt32LittleEndian(page.AsSpan(format.DataChainStampOffset, 4), stamp); - - int offset = format.PageSize - row.Length; - row.CopyTo(page.AsSpan(offset)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset, 2), (ushort)offset); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), 1); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(offset - format.DataRowDirectoryOffset - 2)); - _channel.WritePage(pageNumber, page); - } - - /// Builds the 12-byte descriptor: 4-byte length with storage flags, row/page, chain stamp. The - /// stamp is non-zero only on the chained form, where it repeats the first chain page's own — ACE leaves it - /// zero on the inline and single-page forms, and checks it on neither. - private static byte[] Descriptor(int length, byte flag, int firstPage, int row = 0, uint stamp = 0) - { - LongValueFormat.ValidateLength(length); - var d = new byte[12]; - BinaryPrimitives.WriteUInt32LittleEndian(d, (uint)length | ((uint)flag << 24)); - d[4] = (byte)row; - d[5] = (byte)firstPage; - d[6] = (byte)(firstPage >> 8); - d[7] = (byte)(firstPage >> 16); - BinaryPrimitives.WriteUInt32LittleEndian(d.AsSpan(LongValueFormat.ChainStampOffset, 4), stamp); - return d; - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/PageAllocator.cs b/src/LibRed/LibRed.Core/Storage/PageAllocator.cs index 01102bb65..5b8ce5b39 100644 --- a/src/LibRed/LibRed.Core/Storage/PageAllocator.cs +++ b/src/LibRed/LibRed.Core/Storage/PageAllocator.cs @@ -1,7 +1,6 @@ +using LibRed.Formats; using LibRed.IO; using LibRed.Pages; -using System.Buffers.Binary; -using System.Numerics; namespace LibRed.Storage; @@ -14,58 +13,58 @@ namespace LibRed.Storage; /// /// Pages set in the **global released-pages map** (named at 0x1C) are never allocated, as ACE never /// allocates them: they were released by a session that has not yet merged them back into the free map. Both -/// maps are located only through their page-0 pointers, row included — page 1 rows 0 and 1 in every file ACE -/// writes, but ACE follows the pointers wherever they lead (docs/format/page-05-usage-maps.md §9.1). +/// maps are located only through their page-0 pointers, row included — page 1 rows 0 and 1 where ACE creates +/// them, but ACE follows the pointers wherever they lead (docs/format/page-05-usage-maps.md §9.1). /// Freed pages take one of two routes, as ACE's do: makes a page reusable at once, and /// holds it until runs when the handle closes. /// -public sealed class PageAllocator(PageChannel channel) +internal sealed class PageAllocator { - private const byte InlineMapType = 0x00; - private const byte ReferenceMapType = 0x01; - /// Bytes preceding the bitmap on a dedicated usage-bitmap page (type 0x05). - private const int BitmapPageHeaderSize = 4; + private readonly PageChannel _channel; - /// A reference map is a fixed 69-byte record: the type byte + 17 bitmap-page pointers (17 being - /// exactly enough to span Jet's 2 GB ceiling). See the usage-maps spec (§9). - private const int ReferenceMapSlots = 17; + internal PageAllocator(PageChannel channel) => _channel = channel; - private readonly PageChannel _channel = channel; + /// + /// Allocates a fresh page by growing the file by one page, returning its number. Jet also + /// recycles freed pages via usage maps; appending at the end is always valid since the page + /// count is simply the file length divided by the page size. + /// + internal int Append() + { + if (_channel.IsReadOnly) + throw new InvalidOperationException("This channel was opened read-only."); + + int pageNumber = _channel.PageCount; + _channel.WritePage(pageNumber, new byte[_channel.PageSize]); + return pageNumber; + } /// One of the two global map records, as read through its page-0 pointer. - private sealed record MapRecord(string Name, int PageNumber, int Row, byte[] Page, RowSlot Slot) + private sealed record MapRecord(string Name, int PageNumber, int Row, byte[] Page, DataPage.RowSlot Slot) { public ReadOnlySpan Record => Page.AsSpan(Slot.Offset, Slot.Length); - public byte Type => Page[Slot.Offset]; + public UsageMapType Type => UsageMap.RecordType(Record); } public int Allocate() { (MapRecord free, MapRecord released) = ReadGlobalMaps(); var releasedPages = new ReleasedPages(this, released); - if (free.Type == ReferenceMapType) + if (free.Type == UsageMapType.Reference) return AllocateFromReferenceMap(free, released, releasedPages); - byte[] page = free.Page; - int mapOffset = free.Slot.Offset; - int startPage = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(mapOffset + 1, 4)); - int bitmapStart = mapOffset + 5; - int bitmapEnd = mapOffset + free.Slot.Length; - for (int i = bitmapStart; i < bitmapEnd; i++) + JetFormatBase format = _channel.Format; + int startPage = UsageMap.StartPage(free.Record, format); + Span bitmap = UsageMap.InlineBits(free.Page.AsSpan(free.Slot.Offset, free.Slot.Length), format); + for (int bit = BitmapBits.NextSetBit(bitmap, 0); bit >= 0; bit = BitmapBits.NextSetBit(bitmap, bit + 1)) { - int bits = page[i]; - while (bits != 0) - { - int bit = BitOperations.TrailingZeroCount(bits); - bits &= bits - 1; - int allocated = startPage + (i - bitmapStart) * 8 + bit; - if (releasedPages.Contains(allocated)) continue; // released, not yet reusable - ValidateReusablePage(allocated, "inline free bit", free, released, AppendBoundary(releasedPages)); - EnsurePhysicalAllocation(allocated, releasedPages); - page[i] &= (byte)~(1 << bit); // no longer free - _channel.WritePage(free.PageNumber, page); - return allocated; - } + int allocated = startPage + bit; + if (releasedPages.Contains(allocated)) continue; // released, not yet reusable + ValidateReusablePage(allocated, "inline free bit", free, released, AppendBoundary(releasedPages)); + EnsurePhysicalAllocation(allocated, releasedPages); + BitmapBits.Set(bitmap, bit, false); // no longer free + _channel.WritePage(free.PageNumber, free.Page); + return allocated; } // An unrepresented page is not safely recorded as used. Grow the global map before appending. @@ -80,21 +79,28 @@ public void Free(int page) (MapRecord free, MapRecord released) = ReadGlobalMaps(); ValidateReusablePage(page, "page being freed", free, released, _channel.PageCount - 1); - if (free.Type == ReferenceMapType) + if (free.Type == UsageMapType.Reference) { FreeInReferenceMap(free, released, page); return; } - byte[] p = free.Page; - int mapOffset = free.Slot.Offset; - int startPage = BinaryPrimitives.ReadInt32LittleEndian(p.AsSpan(mapOffset + 1, 4)); + JetFormatBase format = _channel.Format; + int startPage = UsageMap.StartPage(free.Record, format); + Span bitmap = UsageMap.InlineBits(free.Page.AsSpan(free.Slot.Offset, free.Slot.Length), format); int bit = page - startPage; - int byteIndex = mapOffset + 5 + bit / 8; - if (bit < 0 || byteIndex >= mapOffset + free.Slot.Length) return; // outside the inline window + // A page the map has no bit for cannot be recorded as free, and dropping it here is how a page is lost + // for good — nothing else remembers it. It does not arise in a well-formed file: allocation extends the + // map to the file's frontier, and ACE's own map covers its whole file (measured: a 1,761-page file + // carries a 229-byte record from page 0, and records pages freed above the original 512-page window). + // So this is a malformed or foreign map, and it says so rather than quietly leaking the page. + if (bit < 0 || bit >= bitmap.Length * 8) + throw new InvalidDataException( + $"Cannot record page {page} as free: the global free-pages map covers pages {startPage} through " + + $"{startPage + bitmap.Length * 8 - 1}, so the page has no bit in it."); - p[byteIndex] |= (byte)(1 << (bit % 8)); - _channel.WritePage(free.PageNumber, p); + BitmapBits.Set(bitmap, bit, true); + _channel.WritePage(free.PageNumber, free.Page); } /// @@ -122,7 +128,9 @@ public void ReturnReleasedPages() SortedSet pages = [.. _channel.PagesReleasedAtClose]; if (pages.Count == 0 && !_channel.HasPublishedWrites) return; (_, MapRecord released) = ReadGlobalMaps(); - pages.UnionWith(ReleasedMapPages(released)); + // Every bit kept, none bounded by the file: a released page past its end was never materialized, and is + // skipped below. + pages.UnionWith(UsageMap.PagesInRecord(_channel, released.Record, rejectBeyond: null, "The global released-pages map")); if (pages.Count == 0) return; bool ownTransaction = !_channel.InTransaction; @@ -150,49 +158,23 @@ public void ReturnReleasedPages() } } - /// The pages set in the global released-pages map, inline or reference form. - private List ReleasedMapPages(MapRecord released) - { - ReadOnlySpan record = released.Record; - var pages = new List(); - if (released.Type == InlineMapType) - { - int start = BinaryPrimitives.ReadInt32LittleEndian(record.Slice(1, 4)); - for (int i = 5; i < record.Length; i++) - for (int bits = record[i]; bits != 0; bits &= bits - 1) - pages.Add(start + (i - 5) * 8 + BitOperations.TrailingZeroCount(bits)); - return pages; - } - - int pagesPerBitmap = (_channel.PageSize - BitmapPageHeaderSize) * 8; - for (int slot = 0; slot < ReferenceMapSlots; slot++) - { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(record.Slice(1 + slot * 4, 4)); - if (bitmapPage == 0) continue; - ReadOnlySpan bitmap = _channel.ReadPage(bitmapPage).Span; - for (int i = BitmapPageHeaderSize; i < bitmap.Length; i++) - for (int bits = bitmap[i]; bits != 0; bits &= bits - 1) - pages.Add(slot * pagesPerBitmap + (i - BitmapPageHeaderSize) * 8 + BitOperations.TrailingZeroCount(bits)); - } - return pages; - } - /// /// Sizes the released-pages map for the way ACE's close does, before they are merged. /// An inline record that already covers them is left alone. Otherwise it is lengthened, keeping its start - /// page, to the shortest that covers the highest page — the 5-byte header, then the bitmap in whole 4-byte - /// words — as long as its holder keeps 4 bytes free. When that is too long, the window moves instead: the - /// start becomes the lowest page released rounded down to a byte, and the record is sized from there. When - /// even that is too long, the record is grown at its old start to cover the highest released page it can - /// reach, the released pages it covers are marked in it, and it is converted to reference form: a bitmap page is allocated for each range - /// holding a released page, in range order, and a 69-byte reference record takes its place, the longer - /// record's bytes staying on the page below it. A map already in reference form gains a bitmap page for each - /// range holding a released page that it has none for. + /// page, to the shortest that covers the highest page — the header, then the bitmap in whole growth steps — + /// as long as its holder keeps bytes free. When that is too + /// long, the window moves instead: the start becomes the lowest page released rounded down to a byte, and the + /// record is sized from there. When even that is too long, the record is grown at its old start to cover the + /// highest released page it can reach, the released pages it covers are marked in it, and it is converted to + /// reference form: a bitmap page is allocated for each range holding a released page, in range order, and a + /// reference record takes its place, the longer record's bytes staying on the page below it. A map already in + /// reference form gains a bitmap page for each range holding a released page that it has none for. /// private void SizeReleasedMap(SortedSet pages) { + JetFormatBase format = _channel.Format; (_, MapRecord released) = ReadGlobalMaps(); - if (released.Type == ReferenceMapType) + if (released.Type == UsageMapType.Reference) { byte[] existing = released.Record.ToArray(); if (!AddReleasedBitmapPages(existing, pages)) return; @@ -201,50 +183,47 @@ private void SizeReleasedMap(SortedSet pages) return; } - int start = BinaryPrimitives.ReadInt32LittleEndian(released.Record.Slice(1, 4)); + int headerSize = format.UsageMapInlineHeaderSize; + int start = UsageMap.StartPage(released.Record, format); int lowest = pages.Min, highest = pages.Max; - int Covering(int from) => 5 + ((highest - from) / 8 + 1 + 3) / 4 * 4; - if (lowest >= start && highest < start + (released.Slot.Length - 5) * 8) return; + int Covering(int from) => headerSize + UsageMap.InlineBitmapBytes(format, highest - from + 1); + if (lowest >= start && highest < start + (released.Slot.Length - headerSize) * 8) return; var holder = new DataPage(); - holder.Read(new PageBuffer(released.Page, released.PageNumber), _channel.Format); + holder.Read(new PageBuffer(released.Page, released.PageNumber), format); int others = 0; for (int row = 0; row < holder.RowCount; row++) if (row != released.Row) others += holder.Rows[row].Length; - int room = _channel.PageSize - (_channel.Format.DataRowDirectoryOffset + holder.RowCount * 2) - others - 4; - int longest = Math.Max(released.Slot.Length, 5 + (room - 5) / 4 * 4); + int room = format.PageSize - DataPage.DirectoryEnd(format, holder.RowCount) - others - format.UsageMapHolderReserve; + int growth = format.UsageMapInlineGrowthSize; + int longest = Math.Max(released.Slot.Length, headerSize + (room - headerSize) / growth * growth); if (lowest >= start && Covering(start) <= longest) { - var grown = new byte[Covering(start)]; - released.Record[..5].CopyTo(grown); - LayMapRecord(released, grown); + LayMapRecord(released, UsageMap.NewInlineRecord(format, start, Covering(start) - headerSize)); return; } int moved = lowest / 8 * 8; if (Covering(moved) <= longest) { - var window = new byte[Math.Max(Covering(moved), released.Slot.Length)]; - BinaryPrimitives.WriteInt32LittleEndian(window.AsSpan(1, 4), moved); - LayMapRecord(released, window); + LayMapRecord(released, + UsageMap.NewInlineRecord(format, moved, Math.Max(Covering(moved), released.Slot.Length) - headerSize)); return; } // Grown only as far as the highest released page it can still cover — the whole of its longest length only // when released pages reach that far. - int reach = start + (longest - 5) * 8 - 1; + int reach = start + (longest - headerSize) * 8 - 1; SortedSet reachable = pages.GetViewBetween(Math.Min(start, reach), reach); int covered = reachable.Count == 0 ? released.Slot.Length - : Math.Max(released.Slot.Length, 5 + ((reachable.Max - start) / 8 + 1 + 3) / 4 * 4); - var record = new byte[covered]; - released.Record[..5].CopyTo(record); - foreach (int page in pages.GetViewBetween(start, start + (record.Length - 5) * 8 - 1)) - record[5 + (page - start) / 8] |= (byte)(1 << ((page - start) % 8)); + : Math.Max(released.Slot.Length, headerSize + UsageMap.InlineBitmapBytes(format, reachable.Max - start + 1)); + byte[] record = UsageMap.NewInlineRecord(format, start, covered - headerSize); + foreach (int page in pages.GetViewBetween(start, start + (covered - headerSize) * 8 - 1)) + BitmapBits.Set(UsageMap.InlineBits(record, format), page - start, true); LayMapRecord(released, record); - var reference = new byte[1 + ReferenceMapSlots * 4]; - reference[0] = ReferenceMapType; + byte[] reference = UsageMap.NewReferenceRecord(format); AddReleasedBitmapPages(reference, pages); (_, released) = ReadGlobalMaps(); LayMapRecord(released, reference); @@ -255,61 +234,45 @@ private void SizeReleasedMap(SortedSet pages) /// was added. private bool AddReleasedBitmapPages(byte[] reference, SortedSet pages) { - int pagesPerBitmap = (_channel.PageSize - BitmapPageHeaderSize) * 8; + JetFormatBase format = _channel.Format; + int pagesPerBitmap = format.UsageMapPagesPerBitmapPage; bool added = false; foreach (int slot in pages.Select(p => p / pagesPerBitmap).Distinct()) { - if (slot >= ReferenceMapSlots) + if (slot >= format.UsageMapReferenceSlots) throw new InvalidDataException($"Released page {pages.Max} lies past the global map's bitmap slots."); - if (BinaryPrimitives.ReadInt32LittleEndian(reference.AsSpan(1 + slot * 4)) != 0) continue; + if (UsageMap.ReferencePointer(reference, slot, format) != 0) continue; int bitmapPage = Allocate(); - var bitmap = new byte[_channel.PageSize]; - bitmap[0] = (byte)PageType.PageUsageBitmap; - bitmap[1] = 1; - _channel.WritePage(bitmapPage, bitmap); - BinaryPrimitives.WriteInt32LittleEndian(reference.AsSpan(1 + slot * 4), bitmapPage); + _channel.WritePage(bitmapPage, UsageMap.NewBitmapPage(format)); + UsageMap.WriteReferencePointer(reference, slot, format, bitmapPage); added = true; } return added; } - /// Replaces a global map record, repacking its holder's records from the page end, and lays the result - /// over the page as it stands: bytes a moved record vacates are not cleared, as ACE leaves them. + /// Replaces a global map record, repacking its holder's records from the page end. private void LayMapRecord(MapRecord map, byte[] record) { + JetFormatBase format = _channel.Format; byte[] page = _channel.ReadPage(map.PageNumber).Span.ToArray(); var holder = new DataPage(); - holder.Read(new PageBuffer(page, map.PageNumber), _channel.Format); - byte[] repacked = UsageMapWriter.ReplaceMapRecord(page, holder, _channel.Format, map.Row, record, out _) + holder.Read(new PageBuffer(page, map.PageNumber), format); + byte[] repacked = UsageMap.ReplaceMapRecord(page, holder, format, map.Row, record, out _) ?? throw new InvalidDataException($"Global {map.Name} map cannot fit its holder page."); - - var laid = new DataPage(); - laid.Read(new PageBuffer(repacked, map.PageNumber), _channel.Format); - repacked.AsSpan(0, _channel.Format.DataRowDirectoryOffset + laid.RowCount * 2).CopyTo(page); - foreach (RowSlot slot in laid.Rows) - repacked.AsSpan(slot.Offset, slot.Length).CopyTo(page.AsSpan(slot.Offset)); - _channel.WritePage(map.PageNumber, page); + _channel.WritePage(map.PageNumber, repacked); } /// Clears every bit of the global released-pages map: in place for an inline record, and on each /// bitmap page, header kept, for a reference record. private void ClearReleasedMap(MapRecord released) { - if (released.Type == ReferenceMapType) + if (released.Type == UsageMapType.Reference) { - ReadOnlySpan map = released.Record; - for (int slot = 0; slot < ReferenceMapSlots; slot++) - { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1 + slot * 4, 4)); - if (bitmapPage == 0) continue; - byte[] bitmap = _channel.ReadPage(bitmapPage).Span.ToArray(); - bitmap.AsSpan(BitmapPageHeaderSize).Clear(); - _channel.WritePage(bitmapPage, bitmap); - } + new UsageMap(_channel).ClearBitmapPages(released.Record); return; } - released.Page.AsSpan(released.Slot.Offset + 5, released.Slot.Length - 5).Clear(); + UsageMap.InlineBits(released.Page.AsSpan(released.Slot.Offset, released.Slot.Length), _channel.Format).Clear(); _channel.WritePage(released.PageNumber, released.Page); } @@ -321,49 +284,45 @@ private void ClearReleasedMap(MapRecord released) public void ValidateGlobalMaps() => ReadGlobalMaps(); /// Allocates from a reference-type global free map (huge databases): the record is a list of - /// pointers to dedicated bitmap pages (type 0x05), pointer k covering the page range starting at - /// k × (pageSize − 4) × 8. A **set bit is a free page** (the global map's sense, opposite of a + /// pointers to dedicated bitmap pages (type 0x0105), pointer k covering the page range starting at + /// k × UsageMapPagesPerBitmapPage. A **set bit is a free page** (the global map's sense, opposite of a /// per-table owned map). Finds the first free page, clears its bit on the bitmap page, and returns it; /// grows the file when no bitmap records a free page. private int AllocateFromReferenceMap(MapRecord free, MapRecord released, ReleasedPages releasedPages) { - var format = _channel.Format; + JetFormatBase format = _channel.Format; ReadOnlySpan map = free.Record; - int pagesPerBitmap = (format.PageSize - BitmapPageHeaderSize) * 8; + int pagesPerBitmap = format.UsageMapPagesPerBitmapPage; HashSet bitmapPages = ReferenceBitmapPages(free, released); - for (int slot = 0; slot < ReferenceMapSlots; slot++) + for (int slot = 0; slot < format.UsageMapReferenceSlots; slot++) { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1 + slot * 4, 4)); + int bitmapPage = UsageMap.ReferencePointer(map, slot, format); if (bitmapPage == 0) continue; // no bitmap page allocated for this range - byte[] bitmap = ValidateBitmapPage(bitmapPage, free, released); - for (int i = BitmapPageHeaderSize; i < format.PageSize; i++) + byte[] page = ValidateBitmapPage(bitmapPage, free, released); + Span bitmap = UsageMap.BitmapPageBits(page, format); + for (int bit = BitmapBits.NextSetBit(bitmap, 0); bit >= 0; bit = BitmapBits.NextSetBit(bitmap, bit + 1)) { - int bits = bitmap[i]; - while (bits != 0) - { - int bit = BitOperations.TrailingZeroCount(bits); - bits &= bits - 1; - int allocated = slot * pagesPerBitmap + (i - BitmapPageHeaderSize) * 8 + bit; - if (releasedPages.Contains(allocated)) continue; // released, not yet reusable - ValidateReusablePage(allocated, $"reference-map slot {slot} free bit", free, released, - AppendBoundary(releasedPages)); - if (bitmapPages.Contains(allocated)) - throw new InvalidDataException($"Global free map marks bitmap page {allocated} itself as free."); - EnsurePhysicalAllocation(allocated, releasedPages); - bitmap[i] &= (byte)~(1 << bit); // no longer free - _channel.WritePage(bitmapPage, bitmap); - return allocated; - } + int allocated = slot * pagesPerBitmap + bit; + if (releasedPages.Contains(allocated)) continue; // released, not yet reusable + ValidateReusablePage(allocated, $"reference-map slot {slot} free bit", free, released, + AppendBoundary(releasedPages)); + if (bitmapPages.Contains(allocated)) + throw new InvalidDataException($"Global free map marks bitmap page {allocated} itself as free."); + EnsurePhysicalAllocation(allocated, releasedPages); + BitmapBits.Set(bitmap, bit, false); // no longer free + _channel.WritePage(bitmapPage, page); + return allocated; } } return GrowAndAllocate(); } - /// Extends allocation metadata before the physical file: four-byte inline growth, four spare - /// bytes left on the holder page before promoting to reference form, and a reference bitmap allocated + /// Extends allocation metadata before the physical file: inline growth in + /// steps, + /// spare bytes left on the holder page before promoting to reference form, and a reference bitmap allocated /// before the first data page in its range. All three measured against ACE and asserted by /// GlobalMapGrowthTests. private int GrowAndAllocate() @@ -385,30 +344,32 @@ private int GrowAndAllocate() private int GrowAndAllocateCore() { + JetFormatBase format = _channel.Format; (MapRecord free, MapRecord released) = ReadGlobalMaps(); var releasedPages = new ReleasedPages(this, released); byte[] record = free.Record.ToArray(); int frontier = _channel.PageCount; - if (record[0] == InlineMapType) + if (free.Type == UsageMapType.Inline) { - int start = BinaryPrimitives.ReadInt32LittleEndian(record.AsSpan(1)); + int headerSize = format.UsageMapInlineHeaderSize; + int start = UsageMap.StartPage(record, format); if (start != 0) throw new NotSupportedException("Cannot grow a global inline map with a nonzero start page."); - int bitmapBytes = ((frontier / 8 + 1 + 3) / 4) * 4; - if (bitmapBytes <= record.Length - 5) + int bitmapBytes = UsageMap.InlineBitmapBytes(format, frontier + 1); + if (bitmapBytes <= record.Length - headerSize) return AppendUnreleasedPage(releasedPages); // already represented as used - var grown = new byte[5 + bitmapBytes]; + var grown = new byte[headerSize + bitmapBytes]; // Preserve existing free bits; newly covered physical pages are already used. Only future // pages start free. The requested frontier is cleared by the ordinary allocation path. record.CopyTo(grown, 0); for (int bit = frontier; bit < bitmapBytes * 8; bit++) - grown[5 + bit / 8] |= (byte)(1 << (bit % 8)); + BitmapBits.Set(UsageMap.InlineBits(grown, format), bit, true); var holder = new DataPage(); - holder.Read(_channel.ReadPage(free.PageNumber), _channel.Format); - byte[]? rewritten = UsageMapWriter.ReplaceMapRecord(free.Page, holder, _channel.Format, free.Row, grown, out _); + holder.Read(_channel.ReadPage(free.PageNumber), format); + byte[]? rewritten = UsageMap.ReplaceMapRecord(free.Page, holder, format, free.Row, grown, out _); if (rewritten is not null && - BinaryPrimitives.ReadUInt16LittleEndian(rewritten.AsSpan(_channel.Format.DataFreeSpaceOffset)) >= 4) + DataPage.ReadFreeSpace(rewritten, format) >= format.UsageMapHolderReserve) { _channel.WritePage(free.PageNumber, rewritten); return Allocate(); @@ -416,36 +377,33 @@ private int GrowAndAllocateCore() // Inline exhausted: every existing page is used (Allocate already searched all free bits). // Reserve the bitmap pages first so their own bits are clear in the finished map. - record = new byte[1 + ReferenceMapSlots * 4]; - record[0] = ReferenceMapType; - int span = (_channel.PageSize - BitmapPageHeaderSize) * 8; + record = UsageMap.NewReferenceRecord(format); + int span = format.UsageMapPagesPerBitmapPage; for (int range = 0; range <= _channel.PageCount / span; range++) { - if (range >= ReferenceMapSlots) + if (range >= format.UsageMapReferenceSlots) throw new NotSupportedException("Global allocation map has no remaining bitmap slots."); - int bitmap = AppendUnreleasedPage(releasedPages); - BinaryPrimitives.WriteInt32LittleEndian(record.AsSpan(1 + range * 4), bitmap); + UsageMap.WriteReferencePointer(record, range, format, AppendUnreleasedPage(releasedPages)); } - for (int range = 0; range < ReferenceMapSlots; range++) + for (int range = 0; range < format.UsageMapReferenceSlots; range++) { - int bitmap = BinaryPrimitives.ReadInt32LittleEndian(record.AsSpan(1 + range * 4)); + int bitmap = UsageMap.ReferencePointer(record, range, format); if (bitmap != 0) WriteNewGlobalBitmap(bitmap, range); } - WriteGlobalRecord(free, record); + LayMapRecord(free, record); return Allocate(); } - int pagesPerBitmap = (_channel.PageSize - BitmapPageHeaderSize) * 8; - int rangeIndex = frontier / pagesPerBitmap; - if (rangeIndex >= ReferenceMapSlots) + int rangeIndex = frontier / format.UsageMapPagesPerBitmapPage; + if (rangeIndex >= format.UsageMapReferenceSlots) throw new NotSupportedException("Global allocation map has no remaining bitmap slots."); - if (BinaryPrimitives.ReadInt32LittleEndian(record.AsSpan(1 + rangeIndex * 4)) != 0) + if (UsageMap.ReferencePointer(record, rangeIndex, format) != 0) return AppendUnreleasedPage(releasedPages); // represented range, bit already clear int newBitmap = AppendUnreleasedPage(releasedPages); - BinaryPrimitives.WriteInt32LittleEndian(record.AsSpan(1 + rangeIndex * 4), newBitmap); + UsageMap.WriteReferencePointer(record, rangeIndex, format, newBitmap); WriteNewGlobalBitmap(newBitmap, rangeIndex); - WriteGlobalRecord(free, record); + LayMapRecord(free, record); return Allocate(); } @@ -454,51 +412,40 @@ private int GrowAndAllocateCore() private int AppendUnreleasedPage(ReleasedPages releasedPages) { while (releasedPages.Contains(_channel.PageCount)) - _channel.AllocatePage(); - return _channel.AllocatePage(); + Append(); + return Append(); } private void WriteNewGlobalBitmap(int number, int range) { - var bitmap = new byte[_channel.PageSize]; - bitmap[0] = (byte)PageType.PageUsageBitmap; - bitmap[1] = 1; - int span = (_channel.PageSize - BitmapPageHeaderSize) * 8; + JetFormatBase format = _channel.Format; + byte[] bitmap = UsageMap.NewBitmapPage(format); + int span = format.UsageMapPagesPerBitmapPage; int firstFree = Math.Clamp(_channel.PageCount - range * span, 0, span); for (int bit = firstFree; bit < span; bit++) - bitmap[BitmapPageHeaderSize + bit / 8] |= (byte)(1 << (bit % 8)); + BitmapBits.Set(UsageMap.BitmapPageBits(bitmap, format), bit, true); _channel.WritePage(number, bitmap); } - private void WriteGlobalRecord(MapRecord free, byte[] record) - { - byte[] page = _channel.ReadPage(free.PageNumber).Span.ToArray(); - var holder = new DataPage(); - holder.Read(_channel.ReadPage(free.PageNumber), _channel.Format); - byte[] rewritten = UsageMapWriter.ReplaceMapRecord(page, holder, _channel.Format, free.Row, record, out _) - ?? throw new InvalidDataException("Global allocation map cannot fit its holder page."); - _channel.WritePage(free.PageNumber, rewritten); - } - /// Returns a page to a reference-type global free map by setting its bit on the bitmap page for /// its range. If that range has no bitmap page (e.g. a page grown past the map's coverage), the page is /// left unrecorded — it simply won't be reused, matching the pre-existing inline-window behaviour. private void FreeInReferenceMap(MapRecord free, MapRecord released, int page) { - var format = _channel.Format; + JetFormatBase format = _channel.Format; ReadOnlySpan map = free.Record; - int pagesPerBitmap = (format.PageSize - BitmapPageHeaderSize) * 8; + int pagesPerBitmap = format.UsageMapPagesPerBitmapPage; int slot = page / pagesPerBitmap; - if (slot < 0 || slot >= ReferenceMapSlots) return; // beyond the map's ~2 GB reach + if (slot < 0 || slot >= format.UsageMapReferenceSlots) return; // beyond the map's ~2 GB reach - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1 + slot * 4, 4)); + int bitmapPage = UsageMap.ReferencePointer(map, slot, format); if (bitmapPage == 0) return; // range has no bitmap page — nothing to record into int bitInRange = page - slot * pagesPerBitmap; byte[] bitmap = ValidateBitmapPage(bitmapPage, free, released); if (page == bitmapPage) throw new InvalidDataException($"Usage-map bitmap page {page} cannot be marked globally free."); - bitmap[BitmapPageHeaderSize + bitInRange / 8] |= (byte)(1 << (bitInRange % 8)); + BitmapBits.Set(UsageMap.BitmapPageBits(bitmap, format), bitInRange, true); _channel.WritePage(bitmapPage, bitmap); } @@ -507,10 +454,11 @@ private void FreeInReferenceMap(MapRecord free, MapRecord released, int page) private (MapRecord Free, MapRecord Released) ReadGlobalMaps() { ReadOnlySpan page0 = _channel.ReadPage(0).Span; + JetFormatBase format = _channel.Format; (int Row, int Page) freePointer = - DatabaseDefinitionPage.ReadMapPointer(page0, Formats.JetFormatBase.FreePagesMapPointerOffset); + DatabaseDefinitionPage.ReadMapPointer(page0, format.FreePagesMapPointerOffset, format); (int Row, int Page) releasedPointer = - DatabaseDefinitionPage.ReadMapPointer(page0, Formats.JetFormatBase.ReleasedPagesMapPointerOffset); + DatabaseDefinitionPage.ReadMapPointer(page0, format.ReleasedPagesMapPointerOffset, format); if (freePointer == releasedPointer) throw new InvalidDataException( $"Page 0 names the same record (page {freePointer.Page}, row {freePointer.Row}) for the global " + @@ -524,54 +472,26 @@ private void FreeInReferenceMap(MapRecord free, MapRecord released, int page) private MapRecord ReadMapRecord(string name, (int Row, int Page) pointer) { - if (pointer.Page <= 0 || pointer.Page >= _channel.PageCount) - throw new InvalidDataException( - $"Page 0's global {name} map pointer names page {pointer.Page}, outside the file's pages 1..{_channel.PageCount - 1}."); - PageBuffer buffer = _channel.ReadPage(pointer.Page); - if (buffer.Span[0] != (byte)PageType.DataPage) - throw new InvalidDataException( - $"Page 0's global {name} map pointer names page {pointer.Page}, which is not a data page."); - var data = new DataPage(); - data.Read(buffer, _channel.Format); - if (pointer.Row >= data.RowCount) - throw new InvalidDataException( - $"Page 0's global {name} map pointer names row {pointer.Row} of page {pointer.Page}, which has {data.RowCount} rows."); - RowSlot slot = data.Rows[pointer.Row]; - if (slot.IsDeleted || slot.HasOverflow || slot.Length == 0) - throw new InvalidDataException( - $"Global {name} map (page {pointer.Page}, row {pointer.Row}) is deleted, overflowed, or empty."); - - var record = new MapRecord(name, pointer.Page, pointer.Row, buffer.Span.ToArray(), slot); - if (record.Type == InlineMapType) - { - if (slot.Length < 5) - throw new InvalidDataException($"Global inline {name} map is shorter than its 5-byte header."); - } - else if (record.Type == ReferenceMapType) - { - if (slot.Length != 1 + ReferenceMapSlots * 4) - throw new InvalidDataException( - $"Global reference {name} map must be exactly {1 + ReferenceMapSlots * 4} bytes; got {slot.Length}."); - } - else - { - throw new InvalidDataException($"Global {name} map has unknown type 0x{record.Type:X2}."); - } - return record; + // The global maps' holder belongs to no table either, but its owner field reads 1 rather than 0 (observed), + // so it takes only the common checks. The page is copied: Allocate and Free write the record in place. + (PageBuffer page, _, DataPage.RowSlot slot) = UsageMap.ReadRecord( + _channel, pointer.Row, pointer.Page, $"Page 0's global {name} map pointer"); + return new MapRecord(name, pointer.Page, pointer.Row, page.Span.ToArray(), slot); } /// The bitmap pages the two reference-form maps own, each validated; a page may belong to only /// one slot of one map. private HashSet ReferenceBitmapPages(MapRecord free, MapRecord released) { + JetFormatBase format = _channel.Format; var pages = new HashSet(); foreach (MapRecord map in new[] { free, released }) { - if (map.Type != ReferenceMapType) continue; + if (map.Type != UsageMapType.Reference) continue; ReadOnlySpan record = map.Record; - for (int slot = 0; slot < ReferenceMapSlots; slot++) + for (int slot = 0; slot < format.UsageMapReferenceSlots; slot++) { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(record.Slice(1 + slot * 4, 4)); + int bitmapPage = UsageMap.ReferencePointer(record, slot, format); if (bitmapPage == 0) continue; ValidateBitmapPage(bitmapPage, free, released); if (!pages.Add(bitmapPage)) @@ -584,10 +504,7 @@ private HashSet ReferenceBitmapPages(MapRecord free, MapRecord released) private byte[] ValidateBitmapPage(int pageNumber, MapRecord free, MapRecord released) { ValidateReusablePage(pageNumber, "usage-map bitmap pointer", free, released, _channel.PageCount - 1); - byte[] page = _channel.ReadPage(pageNumber).Span.ToArray(); - if (page[0] != (byte)PageType.PageUsageBitmap || page[1] != 0x01 || page[2] != 0 || page[3] != 0) - throw new InvalidDataException($"Global map pointer {pageNumber} does not target a valid bitmap page."); - return page; + return UsageMap.ReadBitmapPage(_channel, pageNumber).Span.ToArray(); } /// Page 0 and the pages holding the two global maps are never allocatable; nor is a page past @@ -616,10 +533,10 @@ private void EnsurePhysicalAllocation(int page, ReleasedPages releasedPages) if (!releasedPages.Contains(_channel.PageCount)) throw new InvalidDataException( $"Global free map selected page {page} past a gap at page {_channel.PageCount} that is neither free nor released."); - _channel.AllocatePage(); + Append(); } if (page < _channel.PageCount) return; - int allocated = _channel.AllocatePage(); + int allocated = Append(); if (allocated != page) throw new InvalidDataException( $"Global free map selected append page {page}, but contiguous allocation produced page {allocated}."); @@ -641,27 +558,26 @@ public ReleasedPages(PageAllocator owner, MapRecord map) public bool Contains(int page) { if (page < 0) return false; + JetFormatBase format = _owner._channel.Format; ReadOnlySpan record = _map.Record; - if (_map.Type == InlineMapType) + if (_map.Type == UsageMapType.Inline) { - int start = BinaryPrimitives.ReadInt32LittleEndian(record.Slice(1, 4)); - long bit = (long)page - start; - if (bit < 0 || bit / 8 >= record.Length - 5) return false; - return (record[5 + (int)(bit / 8)] & (1 << (int)(bit % 8))) != 0; + ReadOnlySpan bits = record[format.UsageMapInlineHeaderSize..]; + long bit = (long)page - UsageMap.StartPage(record, format); + return bit >= 0 && bit < bits.Length * 8L && BitmapBits.Get(bits, (int)bit); } - int pagesPerBitmap = (_owner._channel.PageSize - BitmapPageHeaderSize) * 8; + int pagesPerBitmap = format.UsageMapPagesPerBitmapPage; int slot = page / pagesPerBitmap; - if (slot >= ReferenceMapSlots) return false; + if (slot >= format.UsageMapReferenceSlots) return false; if (!_bitmaps.TryGetValue(slot, out byte[]? bitmap)) { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(record.Slice(1 + slot * 4, 4)); - bitmap = bitmapPage == 0 ? null : _owner._channel.ReadPage(bitmapPage).Span.ToArray(); + int bitmapPage = UsageMap.ReferencePointer(record, slot, format); + bitmap = bitmapPage == 0 ? null : UsageMap.ReadBitmapPage(_owner._channel, bitmapPage).Span.ToArray(); _bitmaps[slot] = bitmap; } - if (bitmap is null) return false; - int inRange = page - slot * pagesPerBitmap; - return (bitmap[BitmapPageHeaderSize + inRange / 8] & (1 << (inRange % 8))) != 0; + return bitmap is not null + && BitmapBits.Get(UsageMap.BitmapPageBits(bitmap, format), page - slot * pagesPerBitmap); } } } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/RowCodec.cs b/src/LibRed/LibRed.Core/Storage/RowCodec.cs new file mode 100644 index 000000000..0f2eb49aa --- /dev/null +++ b/src/LibRed/LibRed.Core/Storage/RowCodec.cs @@ -0,0 +1,668 @@ +using LibRed.Catalog; +using LibRed.Formats; +using LibRed.Storage.Types; +using System.Buffers.Binary; + +namespace LibRed.Storage; + +/// +/// Encodes and decodes CLR values and owns the Jet 4 / ACE inline row layout and declaration limits. +/// The null bitmap marks present (non-null) columns; a Boolean column has no data and its +/// bit carries the value. Variable columns are laid out in ascending VariableIndex order with +/// an end-first offset table. A memo/OLE column's value is written as an *inline* long-value +/// (12-byte descriptor + payload, §8) when it is small enough; anything larger is stored on LVAL +/// pages by before the row reaches here, so only the descriptor is encoded. +/// A calculated column is the exception, because its value is derived here rather than supplied: an +/// oversized result is spilled through the spillCalculated callback the writer passes in. +/// +/// +/// Decoding a long value needs , and says so. A memo or OLE column +/// holds a 12-byte descriptor rather than its data, so without page access the decode can only hand back the +/// pointer — a byte[] where the caller expects a string, which is an error nowhere and arrives at an +/// index key encoder as one. Rather than return it, refuses. The reader stays optional +/// because a row with no long value in it needs no pages, which is what lets the row codec be tested without +/// a file; it is only the value that cannot be faked. +/// The two operations that want the stored descriptors rather than the values — +/// LongValueDescriptors and CalculatedSlots — are static. They need no reader, so +/// they are not reached through an instance that might lack one, and no caller has to decide what an +/// instance "mode" means. Each takes either the row alone or a the caller has already +/// parsed, so a caller wanting both off one row derives the trailer arithmetic once. +/// +public sealed class RowCodec(IReadOnlyList columns, JetFormatBase format, + int? fixedDataLength = null, int? variableColumnCount = null, + Func? spillCalculated = null, int? columnIdHighWater = null, + LongValueStore? longValues = null, bool[]? decode = null) +{ + private readonly IReadOnlyList _columns = columns; + + // The row's field sizes, the long-value descriptor sizes and the inline limit all come from here. + private readonly JetFormatBase _format = format; + + // How an oversized calculated result reaches an LVAL page, returning the in-row descriptor. A calculated + // column is the one value the encoder DERIVES rather than receives, so it cannot have been materialised + // before the row arrived here the way a plain memo is; the writer that owns the pages supplies this + // instead. Null for a standalone encode, which has no channel to write to and so must refuse. + private readonly Func? _spillCalculated = spillCalculated; + + // How many variable slots a row carries. The TDEF's 0x2B when the caller has it — a high-water that + // never decrements — else the tight maximum over the live columns, which is the same number until the + // LAST variable column is dropped and is all a standalone encode can know. + private readonly int? _variableColumnCount = variableColumnCount; + + // How many column ids the row's leading count and null bitmap span: the TDEF's 0x29 high-water when the + // caller has it, which is what ACE writes even after the highest-id column has been dropped — else the + // highest live id + 1, all a standalone encode can know. + private readonly int? _columnIdHighWater = columnIdHighWater; + + // Fixed (non-boolean) columns occupy a contiguous region; its length is defined by the + // table definition. Default to the tight max so a standalone encode round-trips; INSERT + // passes the TDEF's actual fixed-row size so the on-disk layout matches Access. + private readonly int _fixedDataLength = fixedDataLength ?? ComputeFixedDataLength(columns); + + public byte[] Encode(object?[] values) => Encode(values, null); + + /// Encodes a row, recomputing its calculated columns. + /// The row's values, one per column in table order. + /// Envelopes to carry over verbatim, keyed by + /// . An UPDATE that touches nothing a calculated column reads must leave + /// the cached value exactly as it was: ACE recomputes only when a referenced column is written, so + /// recomputing unconditionally would write bytes ACE would not have (§3.4a). Null on INSERT, where every + /// calculated column is computed fresh. + /// The row as its columns hold it, for the calculated columns to read, when + /// already carries long values as their on-disk descriptors. A memo's text is + /// what an expression reads, not the descriptor that points at it. Null when the two are the same. + public byte[] Encode(object?[] values, IReadOnlyDictionary? preservedCalculated, + object?[]? logicalValues = null) + { + if (values.Length != _columns.Count) + throw new ArgumentException($"Expected {_columns.Count} values, got {values.Length}.", nameof(values)); + + // The leading count and the null-bitmap width span every column id the table has ever handed out — the + // TDEF's 0x29 high-water — NOT the live column count, and not the highest live id either. The three + // coincide while ids are contiguous (fresh table / ADD COLUMN) and diverge once one is dead: a burned + // type-change id, or a DROP COLUMN gap. Measured vs ACE: after dropping the only other column of a + // two-column table, ACE still writes count 2 (spec §5). AssembleRow derives the bitmap width from this. + int maxColumnId = Math.Max( + (_columnIdHighWater ?? 0) - 1, + _columns.Count == 0 ? -1 : _columns.Max(c => c.ColumnId)); + + // And the variable section is addressed the same way: by VariableIndex, NOT by position among the + // live variable columns. DROP COLUMN leaves a hole in the index space — the TDEF's 0x2B count is a + // high-water mark that never decrements, and a column added later takes the next index above it — so + // packing the chunks densely puts every column after the hole one slot too low. The decoder reads + // VarChunk(column.VariableIndex) and so does ACE, which is what makes it silent: the row is written + // and read back happily by nothing at all. + // The trailer survives the loss of the last variable column, too: ACE keeps writing it, with numVar at + // the 0x2B high-water, once a table has ever had one (measured — a table whose only TEXT column was + // dropped still gets `06 00 06 00 | 01 00`). Only a table that never had one has no trailer at all. + var varCols = _columns.Where(c => !c.IsFixedLength).ToList(); + int numVar = Math.Max( + _variableColumnCount ?? 0, + varCols.Count == 0 ? 0 : varCols.Max(c => c.VariableIndex) + 1); + + // Encode each region's payload first so we can size the row exactly. + var fixedRegion = new byte[_fixedDataLength]; + foreach (ColumnDef column in _columns) + { + if (column.Type == JetDataType.Boolean || !column.IsFixedLength) continue; + object? v = values[column.Index]; + if (v is null) continue; // null fixed value: leave its slot zeroed, clear the bit below + byte[] encoded = JetTypeCodec.Encode(column, v, _format); + if (encoded.Length != column.Length) + throw new InvalidOperationException($"Column '{column.Name}' encoded to {encoded.Length} bytes, expected {column.Length}."); + encoded.CopyTo(fixedRegion.AsSpan(column.FixedOffset)); + } + + var varChunks = new byte[numVar][]; + Array.Fill(varChunks, []); // a dropped column's slot stays present, and empty + foreach (ColumnDef column in varCols) + { + if (column.IsCalculated) + { + varChunks[column.VariableIndex] = EncodeCalculated(column, logicalValues ?? values, preservedCalculated); + continue; + } + object? v = values[column.Index]; + varChunks[column.VariableIndex] = v is null ? [] : JetTypeCodec.Encode(column, v, _format); + } + + return AssembleRow(_format, maxColumnId, fixedRegion, varChunks, _columns, values); + } + + /// The slot for a calculated column: the envelope holding the result, wrapped in a long-value + /// descriptor when the column owns a long-value map (which is how ACE stores a calculated Memo). + private byte[] EncodeCalculated(ColumnDef column, object?[] values, + IReadOnlyDictionary? preservedCalculated) + { + if (preservedCalculated is not null && preservedCalculated.TryGetValue(column.Index, out byte[]? kept)) + return kept; + + byte[] envelope = CalculatedValue.Encode(column, CalculatedValue.Evaluate(column, _columns, values), _format); + if (!column.HasLongValueMap) return envelope; + + // A memo-backed result inlines while it fits and spills to an LVAL page once it does not — the same + // 64-byte boundary a plain memo uses, and for the same reason: Access reads an inlined long value + // back but refuses one that should have been on a page. What goes on the page is the WHOLE envelope, + // header and padding included, because that is what the in-row slot would otherwise have held. + if (envelope.Length <= _format.LongValueMaxInline) + return JetTypeCodec.EncodeInlineLongValue(envelope, _format); + + return _spillCalculated is not null + ? _spillCalculated(column, envelope) + : throw new NotSupportedException( + $"Calculated column '{column.Name}' produced a {envelope.Length}-byte value, too large to " + + "store inline. Encoding it needs a writer that can allocate long-value pages, which a " + + "standalone RowEncoder has no channel for — insert or update through RowInserter."); + } + + /// Rejects a variable TEXT/BINARY value longer than its column's declared width, as ACE does + /// (measured in ColumnLengthAccessTests); without it LibRed wrote rows Access will not read back. + /// Memo/OLE are exempt — they encode to a long-value descriptor whose size is unrelated to + /// . Fixed columns are checked by + /// before the codec pads them, since padding to width would otherwise hide an over-long value. + private static void EnsureFitsDeclaredLength(ColumnDef column, byte[] encoded) + { + if (column.Type is not (JetDataType.Text or JetDataType.Binary or JetDataType.BigBinary)) return; + if (column.Length <= 0) return; + + // In the column's own units: TEXT declares characters, BINARY bytes. A TEXT value is counted by + // decoding it, because WITH COMPRESSION stores a Latin-1 value one byte per character — counted in + // bytes, a TEXT(5) would take 8 characters, where ACE refuses the sixth. Neither encoding spends + // fewer than one byte per character, so a value within the declared count of BYTES is within the + // declared count of characters and needs no decoding — which is every ordinary value. + bool text = column.Type == JetDataType.Text; + int declared = text ? column.Length / 2 : column.Length; + if (encoded.Length <= declared) return; + + int actual = text ? JetTypeCodec.DecodeText(encoded).Length : encoded.Length; + if (actual <= declared) return; + throw new InvalidOperationException( + $"The field '{column.Name}' is too small to accept the amount of data you attempted to add: " + + $"{actual} {(text ? "characters" : "bytes")} into a column declared to hold {declared}."); + } + + /// + /// Smallest fixed region ACE writes in a row that has no variable trailer. It is a floor, not an + /// alignment: measured against ACE, a region of 0 bytes (a table of only Booleans, which occupy none) + /// is padded to 2 and 1 byte (a lone BYTE column) to 2, while 3 stays 3 — a three-BYTE table's + /// row is 6 bytes, odd region and all. A row that carries a variable trailer is exempt: ACE leaves a + /// TEXT-only table's fixed region at 0. + /// + /// + /// Matching it is not cosmetic. Without the pad an all-Boolean table of eight columns or fewer encodes to a + /// 3-byte record, and ACE misreads that record — every Boolean in it comes back False, whichever + /// engine created the table. Measured both ways round: ACE's own table filled by LibRed read False, and + /// LibRed's table filled by ACE read True, which is what pins the fault to the record rather than the TDEF. + /// The cliff is at 4 bytes — a 16-Boolean row (2-byte bitmap, so 4 bytes) reads back correctly — but ACE's + /// own writer never emits a record under 5, so the short form is simply a shape its reader has never met. + /// The TDEF's fixed-row length keeps the true, unpadded value: ACE stores 1 for a BYTE table while + /// writing 5-byte rows into it, so this rounding happens at row-write time and nowhere else. + /// + private const int MinFixedRegion = 2; + + /// Assembles the on-disk row bytes from a prepared fixed region and the ordered variable chunks: + /// [count][fixed][var data][var-offset table][numVar] (the variable section is omitted entirely when + /// there are none) then [null bitmap]. The count and bitmap width are maxColumnId + 1; a + /// column's bit is set when present (Boolean = its truthy value), and a dead id's (a gap below the max, from a + /// burned/dropped id) is taken from — the row's bitmap before an ALTER COLUMN + /// re-lay — or left clear for a new row — all verified vs ACE (§5). Shared by Encode and the ALTER + /// COLUMN row re-lay so the two can never drift. + /// + /// The declared-width check runs HERE rather than in Encode. It used to sit above this call, + /// which meant the ALTER COLUMN re-lay — the other caller — never got it, and a narrowing retype could + /// write rows Access refuses. A guard that both paths must pass through belongs on the shared path. + /// + internal static byte[] AssembleRow(JetFormatBase format, int maxColumnId, ReadOnlySpan fixedRegion, + IReadOnlyList varChunks, IReadOnlyList columns, object?[] values, + ReadOnlySpan priorBitmap = default) + { + // A calculated column is exempt from the declared-width check: its length field is a constant ACE + // writes (39 for a value type, 509 for any Text, whatever size was asked for), not a limit — a + // Text(5) calculated column stores a far longer result and ACE reads it back in full (§3.4a). + foreach (ColumnDef column in columns) + if (!column.IsFixedLength && !column.IsCalculated + && column.VariableIndex >= 0 && column.VariableIndex < varChunks.Count) + EnsureFitsDeclaredLength(column, varChunks[column.VariableIndex]); + + int countSize = format.RowColumnCountSize; + int offsetSize = format.RowVariableOffsetSize, numVarSize = format.RowVariableCountSize; + int count = maxColumnId + 1; + int nullBitmapSize = BitmapBits.ByteCount(count); + int numVar = varChunks.Count; + int varDataLength = 0; + for (int j = 0; j < numVar; j++) varDataLength += varChunks[j].Length; + int varSectionLen = numVar > 0 ? varDataLength + (numVar + 1) * offsetSize + numVarSize : 0; + + // ACE pads an all-fixed row's fixed region out to MinFixedRegion; a row with a variable trailer is + // left alone. The pad is zero bytes between the fixed values and the null bitmap. + int fixedLen = numVar > 0 ? fixedRegion.Length : Math.Max(fixedRegion.Length, MinFixedRegion); + + var row = new byte[countSize + fixedLen + varSectionLen + nullBitmapSize]; + BinaryPrimitives.WriteUInt16LittleEndian(row.AsSpan(0, countSize), (ushort)count); + fixedRegion.CopyTo(row.AsSpan(countSize)); + + if (numVar > 0) + { + int varDataStart = countSize + fixedLen; + int pos = varDataStart; + for (int j = 0; j < numVar; j++) { varChunks[j].CopyTo(row.AsSpan(pos)); pos += varChunks[j].Length; } + + // End-first offset table: entry[numVar] = var-data start, entry[numVar-j-1] = end of var col j. + int tableStart = pos; + BinaryPrimitives.WriteUInt16LittleEndian(row.AsSpan(tableStart + numVar * offsetSize, offsetSize), (ushort)varDataStart); + int running = varDataStart; + for (int j = 0; j < numVar; j++) + { + running += varChunks[j].Length; + BinaryPrimitives.WriteUInt16LittleEndian( + row.AsSpan(tableStart + (numVar - j - 1) * offsetSize, offsetSize), (ushort)running); + } + BinaryPrimitives.WriteUInt16LittleEndian( + row.AsSpan(tableStart + (numVar + 1) * offsetSize, numVarSize), (ushort)numVar); + } + + // The null bitmap ends the row whatever precedes it. + Span nullBitmap = row.AsSpan(row.Length - nullBitmapSize); + var liveIds = new HashSet(); + foreach (ColumnDef column in columns) + { + liveIds.Add(column.ColumnId); + // A calculated column is always present, even when its expression evaluated to Null — ACE marks + // the bit and stores a zero-length payload, so here the bit means "has an envelope" and says + // nothing about the value (§3.4a). It is also why a calculated Boolean cannot use the bit as its + // value the way a real one does. + bool present = column.IsCalculated + || (column.Type == JetDataType.Boolean ? IsTruthy(values[column.Index]) : values[column.Index] is not null); + if (present) BitmapBits.Set(nullBitmap, column.ColumnId, true); + } + // A dead id's bit depends on which route wrote the row, and both are measured against ACE: the ALTER + // COLUMN re-lay carries the old row's bit forward — set where the retyped column held a value, clear + // where it was NULL — while a row INSERTED afterwards leaves it clear: the same statement pair gives + // ACE 0x0F for a re-laid row with a value and 0x0D for the next insert. + for (int id = 0; id <= maxColumnId && id < priorBitmap.Length * 8; id++) + if (!liveIds.Contains(id) && BitmapBits.Get(priorBitmap, id)) + BitmapBits.Set(nullBitmap, id, true); + return row; + } + + /// Access truthiness for a Boolean (bit) value being stored: a bool is itself, any non-zero + /// number is true, 0 / null is false. The index key uses the same rule, so a key always agrees with its row. + internal static bool IsTruthy(object? value) => value switch + { + null => false, + bool b => b, + _ => Convert.ToBoolean(value, System.Globalization.CultureInfo.InvariantCulture), + }; + + private static int ComputeFixedDataLength(IReadOnlyList columns) + { + int length = 0; + foreach (ColumnDef c in columns) + if (c.IsFixedLength && c.Type != JetDataType.Boolean) + length = Math.Max(length, c.FixedOffset + c.Length); + return length; + } + + private readonly bool[]? _decode = decode; + + private readonly LongValueStore? _longValues = longValues; + + // Decode runs once per row, so it walks an array rather than enumerating the interface (a boxed + // enumerator per row), and answers Layout.HasVariableSection from the lowest variable column id + // instead of re-scanning the columns: a row has a variable section exactly when its stored count + // reaches past that id. + private readonly ColumnDef[] _columnArray = [.. columns]; + private readonly int _lowestVariableId = LowestVariableId(columns); + + private static readonly object BoxedTrue = true; + private static readonly object BoxedFalse = false; + + private static int LowestVariableId(IReadOnlyList columns) + { + int lowest = int.MaxValue; + foreach (ColumnDef column in columns) + if (!column.IsFixedLength && column.ColumnId < lowest) + lowest = column.ColumnId; + return lowest; + } + + /// Decodes the row into one value per column (aligned to ). + public object?[] Decode(ReadOnlySpan row) + { + var values = new object?[_columnArray.Length]; + + // The null-bitmap width comes from the row's own leading count (= max id + 1), NOT the live column + // count — they differ once ids have a gap (a burned type-change id / DROP COLUMN gap). Reading the + // stored count is exactly how ACE sizes it, and is robust to any id scheme (spec §5). A real inline + // row is at least: column count + an empty var table (1 entry) + var count + null bitmap; anything + // shorter is an overflow/lookup pointer slot the caller should have skipped. + bool hasVar = _lowestVariableId < Layout.ReadColumnCount(row, _format); + Layout layout = Layout.Parse(row, _format, hasVar); + + foreach (ColumnDef column in _columnArray) + { + if (_decode is not null && !_decode[column.Index]) + continue; + + bool present = layout.IsPresent(column.ColumnId); + + // Jet stores Boolean (YesNo) columns with no fixed/variable data: the value + // IS the null-bitmap bit (set = true). Booleans are never null. + // + // A CALCULATED Boolean is the exception, and a silent one: it carries a real variable-length + // slot, and its bitmap bit means "present" like every other calculated column, so it is set + // for a stored False too. Reading the bit here would report True for every row. + if (column.Type == JetDataType.Boolean && !column.IsCalculated) + { + values[column.Index] = present ? BoxedTrue : BoxedFalse; + continue; + } + + if (!present) + { + values[column.Index] = null; + continue; + } + + ReadOnlySpan raw = column.IsFixedLength + ? FixedSlice(row, layout, column) + : layout.VarChunk(column.VariableIndex); + + // A calculated column stores ACE's cached result wrapped in an envelope, routed through a + // long-value descriptor when the column owns a long-value map (which is how a calculated Memo + // arrives — ACE declares it Text, so the Memo branch below would never fire for it). + if (column.IsCalculated) + { + ReadOnlySpan envelope = column.HasLongValueMap + ? Reader(column).Resolve(raw) + : raw; + values[column.Index] = CalculatedValue.Decode(column, envelope); + continue; + } + + // Memo / OLE columns store a long-value descriptor, not the data itself. + if (column.Type is JetDataType.Memo or JetDataType.Ole) + { + byte[] data = Reader(column).Resolve(raw); + values[column.Index] = column.Type == JetDataType.Memo + ? JetTypeCodec.DecodeText(data) + : data; + continue; + } + + values[column.Index] = JetTypeCodec.Decode(column, raw); + } + + return values; + } + + /// Returns the raw in-row long-value descriptor bytes for each present memo/OLE column (keyed by + /// ), WITHOUT resolving the value. Used by UPDATE/DELETE to preserve an + /// unchanged column's descriptor verbatim (avoiding a needless re-materialise) and to free a replaced or + /// deleted value's LVAL pages. + public static Dictionary LongValueDescriptors( + IReadOnlyList columns, JetFormatBase format, ReadOnlySpan row) => + LongValueDescriptors(columns, ParseLayout(columns, format, row), row); + + /// + /// Takes a layout the caller has already parsed — an UPDATE wants this and + /// off one row, and + /// exists so that arithmetic is done once. + internal static Dictionary LongValueDescriptors( + IReadOnlyList columns, Layout layout, ReadOnlySpan row) + { + var result = new Dictionary(); + + // Keyed on owning a long-value map rather than on the declared type: a calculated Memo is declared + // Text and still stores a descriptor, so a type test alone walks past its pages and orphans them. + foreach (ColumnDef column in columns) + if ((column.Type is JetDataType.Memo or JetDataType.Ole || column.HasLongValueMap) + && layout.IsPresent(column.ColumnId)) + result[column.Index] = layout.VarChunk(column.VariableIndex).ToArray(); + + return result; + } + + /// Returns each calculated column's stored slot verbatim (keyed by ) + /// — the envelope, or the long-value descriptor wrapping it — WITHOUT decoding or resolving it. An UPDATE + /// that touches nothing the expression reads writes these back unchanged, which is what ACE does. + public static Dictionary CalculatedSlots( + IReadOnlyList columns, JetFormatBase format, ReadOnlySpan row) => + CalculatedSlots(columns, ParseLayout(columns, format, row), row); + + /// + /// Takes a layout the caller has already parsed; see the + /// overload. + internal static Dictionary CalculatedSlots( + IReadOnlyList columns, Layout layout, ReadOnlySpan row) + { + var result = new Dictionary(); + + foreach (ColumnDef column in columns) + if (column.IsCalculated && !column.IsFixedLength + && layout.IsPresent(column.ColumnId) + && column.VariableIndex >= 0 && column.VariableIndex < layout.NumVar) + result[column.Index] = layout.VarChunk(column.VariableIndex).ToArray(); + + return result; + } + + private ReadOnlySpan FixedSlice(ReadOnlySpan row, Layout layout, ColumnDef column) + { + int start = _format.RowColumnCountSize + column.FixedOffset; + long end = (long)start + column.Length; + if (column.FixedOffset < 0 || column.Length < 0 + || end > _format.RowColumnCountSize + (long)layout.FixedRegionLength) + throw new InvalidDataException( + $"Row fixed value for column '{column.Name}' extends outside the fixed-data region."); + return row.Slice(start, column.Length); + } + + /// The long-value reader, or a refusal naming the column that needed it. Silently returning the + /// descriptor instead is the failure this exists to prevent: it is a byte[] that looks like a + /// value, and it travels — into an index key, into a comparison, into another row — before anything + /// notices. + private LongValueStore Reader(ColumnDef column) => + _longValues ?? throw new InvalidOperationException( + $"Column '{column.Name}' stores its value on long-value pages, so decoding it needs a " + + "LongValueReader; this RowDecoder was constructed without one. " + + $"Use {nameof(LongValueDescriptors)} to read the stored descriptors instead."); + + internal static Layout ParseLayout(IReadOnlyList columns, JetFormatBase format, ReadOnlySpan row) => + Layout.Parse(row, format, Layout.HasVariableSection(row, columns, format)); + + + /// The widest a single non-Memo/OLE column may be declared: 255 Text characters, or 510 bytes + /// of Binary. + public const int MaxFieldBytes = 510; + + /// The widest a BigBinary column may be declared: BIGBINARY(4001) gives ACE's "Size of + /// field is too long". + public const int MaxBigBinaryBytes = 4000; + + /// Throws if a column is declared wider than ACE stores. Memo and OLE are exempt — their data + /// lives on long-value pages and the in-row descriptor is a fixed size. + public static void ValidateFieldWidth(string columnName, JetDataType type, int lengthBytes) + { + if (type is JetDataType.Memo or JetDataType.Ole) return; + if (type == JetDataType.BigBinary) + { + if (lengthBytes > MaxBigBinaryBytes) + throw new NotSupportedException( + $"Column '{columnName}' is declared {lengthBytes} bytes wide; a BigBinary column holds at most " + + $"{MaxBigBinaryBytes}."); + return; + } + if (lengthBytes > MaxFieldBytes) + throw new NotSupportedException( + $"Column '{columnName}' is declared {lengthBytes} bytes wide; a column holds at most {MaxFieldBytes} " + + $"bytes ({MaxFieldBytes / 2} Text characters). Use Memo or OLE for anything longer."); + } + + /// The largest record the declaration can produce: the row header, the fixed region, the + /// variable section's overhead and the null bitmap. Mirrors 's layout + /// except that a table with no variable columns still gets ACE's 4-byte allowance for the section. + /// Sum of the fixed columns' widths, excluding Boolean (which has no data). + /// Number of variable-length columns, Memo and OLE included. + /// The column-id high-water plus one — what sizes the null bitmap. + /// The file's format, for the row's field sizes. + /// The variable section is budgeted even with no variable columns — one offset entry and the count, + /// which is ACE's allowance for that case. + public static int WidestRecord(int fixedBytes, int variableColumns, int columnCount, JetFormatBase format) => + format.RowColumnCountSize + + fixedBytes + + (variableColumns + 1) * format.RowVariableOffsetSize + format.RowVariableCountSize + + BitmapBits.ByteCount(columnCount); + + /// Throws if the declaration's widest possible record is one ACE would refuse the file for. + /// may be null where the caller does not know it (the table is still being + /// built), in which case the message just says "the table". + public static void ValidateRecordFits( + string? tableName, int fixedBytes, int variableColumns, int columnCount, JetFormatBase format) + { + int widest = WidestRecord(fixedBytes, variableColumns, columnCount, format); + if (widest > format.MaxRecordSize) + throw new NotSupportedException( + $"{(tableName is null ? "The table" : $"Table '{tableName}'")} declares {fixedBytes} bytes of " + + $"fixed-length columns over {columnCount} column ids, so its widest record would be {widest} " + + $"bytes; a record holds at most {format.MaxRecordSize}. Make the wide columns variable-length, " + + "or move them to Memo/OLE, which live on their own pages."); + } + + /// + /// Parses the structural trailer of an inline row record once (spec §5), so the several call sites that + /// need to locate a row's regions don't each re-derive the offset arithmetic. Layout: + /// + /// [count] [fixed data] [var data] [varOffsetTable: numVar+1 entries] [numVar] [nullBitmap] + /// + /// The field sizes are the format's (, + /// , ). The leading + /// count is maxColumnId + 1 and sets the null-bitmap width, one bit per column id. A table with NO variable + /// columns omits the whole variable section (offset table + numVar) — such a row can't self-describe that, so the + /// caller passes hasVar from the schema. + /// + internal readonly ref struct Layout + { + private readonly ReadOnlySpan _row; + private readonly int _offsetSize; + + /// The leading column count (= max column id + 1). + public int ColumnCount { get; } + /// Null-bitmap width in bytes, from the leading count. + public int NullBitmapSize { get; } + /// Number of variable columns stored (0 when the table has no variable section). + public int NumVar { get; } + /// Offset of the variable-offset table, or -1 when there is no variable section. + public int VarTableStart { get; } + /// Length of the fixed-data region (bytes between the leading count and the variable data). + public int FixedRegionLength { get; } + + private Layout(ReadOnlySpan row, JetFormatBase format, bool hasVar) + { + int countSize = format.RowColumnCountSize; + _row = row; + _offsetSize = format.RowVariableOffsetSize; + ColumnCount = ReadColumnCount(row, format); + NullBitmapSize = BitmapBits.ByteCount(ColumnCount); + int minimum = MinimumLength(format, ColumnCount, hasVar); + if (row.Length < minimum) + throw new InvalidDataException( + $"Row is {row.Length} bytes, shorter than the {minimum} its {NullBitmapSize}-byte null bitmap" + + (hasVar ? " and variable-column trailer need." : " needs.")); + + if (!hasVar) + { + NumVar = 0; + VarTableStart = -1; + FixedRegionLength = row.Length - countSize - NullBitmapSize; + return; + } + + int numVarSize = format.RowVariableCountSize; + int numVarPos = row.Length - NullBitmapSize - numVarSize; + NumVar = BinaryPrimitives.ReadUInt16LittleEndian(row.Slice(numVarPos, numVarSize)); + long tableStart = (long)numVarPos - ((long)NumVar + 1) * _offsetSize; + if (tableStart < countSize || tableStart > numVarPos) + throw new InvalidDataException( + $"Row declares {NumVar} variable slots, placing its offset table outside the row."); + VarTableStart = (int)tableStart; + + int previous = VarOffset(0); + if (previous < countSize || previous > VarTableStart) + throw new InvalidDataException( + $"Row variable-data end {previous} is outside the data region ending at {VarTableStart}."); + for (int entry = 1; entry <= NumVar; entry++) + { + int current = VarOffset(entry); + if (current < countSize || current > previous) + throw new InvalidDataException( + $"Row variable offset {entry} ({current}) is outside or above its preceding boundary {previous}."); + previous = current; + } + // The last offset-table entry is the variable-data start (= count field + fixed region). + FixedRegionLength = previous - countSize; + } + + /// Parses ; is whether the row carries a variable + /// section (). + public static Layout Parse(ReadOnlySpan row, JetFormatBase format, bool hasVar) => new(row, format, hasVar); + + /// The fewest bytes a row of column ids can take: the count, the null + /// bitmap, and — when it has a variable section — the offset table's one mandatory entry (the variable-data + /// start) and the variable-column count. Fixed data and variable data may both be empty. + public static int MinimumLength(JetFormatBase format, int columnCount, bool hasVar) => + format.RowColumnCountSize + BitmapBits.ByteCount(columnCount) + + (hasVar ? format.RowVariableOffsetSize + format.RowVariableCountSize : 0); + + /// The row's leading column count: the highest column id the table had handed out when the row was + /// written, plus one. + public static int ReadColumnCount(ReadOnlySpan row, JetFormatBase format) + { + int countSize = format.RowColumnCountSize; + if (row.Length < countSize) + throw new InvalidDataException($"Row is too short to contain its {countSize}-byte column count."); + return BinaryPrimitives.ReadUInt16LittleEndian(row[..countSize]); + } + + /// Whether this row carries a variable section — the argument every + /// caller needs, derived once here rather than at each call site. + /// + /// It is not "does the schema have a variable column": a row written before the table's first variable + /// ADD COLUMN has no trailer even though the current schema does, because ADD COLUMN is metadata-only. + /// The row's own stored count is what dates it — a column id at or above the count did not exist when + /// the row was written. Get this wrong and the parse reads the last bytes of FIXED data as numVar and an + /// offset table, which the bounds checks above usually catch, but not always. + /// + public static bool HasVariableSection(ReadOnlySpan row, IReadOnlyList columns, JetFormatBase format) + { + int storedCount = ReadColumnCount(row, format); + foreach (ColumnDef column in columns) + if (!column.IsFixedLength && column.ColumnId < storedCount) return true; + return false; + } + + /// The row's null bitmap: one bit per column id below , set when the column + /// holds a value — or, for a Boolean, when it is true. + public ReadOnlySpan NullBitmap => _row[^NullBitmapSize..]; + + /// Whether column 's null-bitmap bit is set. A column id the row predates — + /// at or above its stored count — has no bit, and reads as clear. + public bool IsPresent(int columnId) => + columnId >= 0 && columnId < ColumnCount && BitmapBits.Get(NullBitmap, columnId); + + /// The raw bytes of variable column (end-first offset table). + public ReadOnlySpan VarChunk(int variableIndex) + { + if (variableIndex < 0 || variableIndex >= NumVar) + throw new InvalidDataException( + $"Row has {NumVar} variable slots but column metadata requests slot {variableIndex}."); + int start = VarOffset(NumVar - variableIndex); + int end = VarOffset(NumVar - variableIndex - 1); + return _row[start..end]; + } + + private int VarOffset(int entry) => + BinaryPrimitives.ReadUInt16LittleEndian(_row.Slice(VarTableStart + entry * _offsetSize, _offsetSize)); + } + +} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/RowDecoder.cs b/src/LibRed/LibRed.Core/Storage/RowDecoder.cs deleted file mode 100644 index 0014a39c1..000000000 --- a/src/LibRed/LibRed.Core/Storage/RowDecoder.cs +++ /dev/null @@ -1,154 +0,0 @@ -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.Storage.Types; - -namespace LibRed.Storage; - -/// -/// Decodes a single raw row record into CLR values. Implements the Jet 4 / ACE row -/// layout (verified against real data): -/// -/// [colCount:2] [fixed data] [var data] [varOffsetTable:(numVar+1)x2] [numVar:2] [nullBitmap] -/// -/// The null bitmap is indexed by column id (bit set = value present). Variable columns -/// are addressed via the trailing offset table in ascending column-id order. -/// Jet 4 / ACE uses 2-byte variable offsets at any row size — there is no Jet 3-style -/// jump table (1-byte offsets), so rows larger than 256 bytes decode the same way. -/// -public sealed class RowDecoder(IReadOnlyList columns, JetFormatBase format, LongValueReader? longValues = null) -{ - private readonly IReadOnlyList _columns = columns; - private readonly JetFormatBase _format = format; - private readonly LongValueReader? _longValues = longValues; - - /// Decodes the row into one value per column (aligned to ). - public object?[] Decode(ReadOnlySpan row) - { - var values = new object?[_columns.Count]; - - // The null-bitmap width comes from the row's own leading count (= max id + 1), NOT the live column - // count — they differ once ids have a gap (a burned type-change id / DROP COLUMN gap). Reading the - // stored count is exactly how ACE sizes it, and is robust to any id scheme (spec §5). A real inline - // row is at least: column count + an empty var table (1 entry) + var count + null bitmap; anything - // shorter is an overflow/lookup pointer slot the caller should have skipped. - RowLayout layout = ParseLayout(row); - int nullBitmapSize = layout.NullBitmapSize; - - ReadOnlySpan nullBitmap = row[^nullBitmapSize..]; - - foreach (ColumnDef column in _columns) - { - bool present = IsPresent(nullBitmap, layout.ColumnCount, column.ColumnId); - - // Jet stores Boolean (YesNo) columns with no fixed/variable data: the value - // IS the null-bitmap bit (set = true). Booleans are never null. - // - // A CALCULATED Boolean is the exception, and a silent one: it carries a real variable-length - // slot, and its bitmap bit means "present" like every other calculated column, so it is set - // for a stored False too. Reading the bit here would report True for every row. - if (column.Type == JetDataType.Boolean && !column.IsCalculated) - { - values[column.Index] = present; - continue; - } - - if (!present) - { - values[column.Index] = null; - continue; - } - - ReadOnlySpan raw = column.IsFixedLength - ? FixedSlice(row, layout, column) - : layout.VarChunk(column.VariableIndex); - - // A calculated column stores ACE's cached result wrapped in an envelope, routed through a - // long-value descriptor when the column owns a long-value map (which is how a calculated Memo - // arrives — ACE declares it Text, so the Memo branch below would never fire for it). - if (column.IsCalculated) - { - ReadOnlySpan envelope = column.HasLongValueMap && _longValues is not null - ? _longValues.Resolve(raw) - : raw; - values[column.Index] = CalculatedValue.Decode(column, envelope); - continue; - } - - // Memo / OLE columns store a long-value descriptor, not the data itself. - if (_longValues is not null && column.Type is JetDataType.Memo or JetDataType.Ole) - { - byte[] data = _longValues.Resolve(raw); - values[column.Index] = column.Type == JetDataType.Memo - ? JetTypeCodec.DecodeText(data) - : data; - continue; - } - - values[column.Index] = JetTypeCodec.Decode(column, raw); - } - - return values; - } - - /// Returns the raw in-row long-value descriptor bytes for each present memo/OLE column (keyed by - /// ), WITHOUT resolving the value. Used by UPDATE/DELETE to preserve an - /// unchanged column's descriptor verbatim (avoiding a needless re-materialise) and to free a replaced or - /// deleted value's LVAL pages. - public Dictionary LongValueRaw(ReadOnlySpan row) - { - var result = new Dictionary(); - RowLayout layout = ParseLayout(row); - int nullBitmapSize = layout.NullBitmapSize; - - ReadOnlySpan nullBitmap = row[^nullBitmapSize..]; - - // Keyed on owning a long-value map rather than on the declared type: a calculated Memo is declared - // Text and still stores a descriptor, so a type test alone walks past its pages and orphans them. - foreach (ColumnDef column in _columns) - if ((column.Type is JetDataType.Memo or JetDataType.Ole || column.HasLongValueMap) - && IsPresent(nullBitmap, layout.ColumnCount, column.ColumnId)) - result[column.Index] = layout.VarChunk(column.VariableIndex).ToArray(); - - return result; - } - - /// Returns each calculated column's stored slot verbatim (keyed by ) - /// — the envelope, or the long-value descriptor wrapping it — WITHOUT decoding or resolving it. An UPDATE - /// that touches nothing the expression reads writes these back unchanged, which is what ACE does. - public Dictionary CalculatedRaw(ReadOnlySpan row) - { - var result = new Dictionary(); - RowLayout layout = ParseLayout(row); - ReadOnlySpan nullBitmap = row[^layout.NullBitmapSize..]; - - foreach (ColumnDef column in _columns) - if (column.IsCalculated && !column.IsFixedLength - && IsPresent(nullBitmap, layout.ColumnCount, column.ColumnId) - && column.VariableIndex >= 0 && column.VariableIndex < layout.NumVar) - result[column.Index] = layout.VarChunk(column.VariableIndex).ToArray(); - - return result; - } - - private ReadOnlySpan FixedSlice(ReadOnlySpan row, RowLayout layout, ColumnDef column) - { - int start = _format.RowColumnCountSize + column.FixedOffset; - long end = (long)start + column.Length; - if (column.FixedOffset < 0 || column.Length < 0 - || end > _format.RowColumnCountSize + (long)layout.FixedRegionLength) - throw new InvalidDataException( - $"Row fixed value for column '{column.Name}' extends outside the fixed-data region."); - return row.Slice(start, column.Length); - } - - private RowLayout ParseLayout(ReadOnlySpan row) - { - if (row.Length < _format.RowColumnCountSize) - throw new InvalidDataException("Row is too short to be an inline record."); - return RowLayout.Parse(row, _format.RowColumnCountSize, RowLayout.HasVariableSection(row, _columns)); - } - - private static bool IsPresent(ReadOnlySpan nullBitmap, int storedColumnCount, int columnId) => - columnId >= 0 && columnId < storedColumnCount - && (nullBitmap[columnId >> 3] & (1 << (columnId & 7))) != 0; -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/RowEncoder.cs b/src/LibRed/LibRed.Core/Storage/RowEncoder.cs deleted file mode 100644 index ed6dfcc0a..000000000 --- a/src/LibRed/LibRed.Core/Storage/RowEncoder.cs +++ /dev/null @@ -1,273 +0,0 @@ -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.Storage.Types; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -/// -/// Encodes a row of CLR values into the Jet 4 / ACE inline row layout — the inverse of -/// : -/// -/// [colCount:2] [fixed data] [var data] [varOffsetTable:(numVar+1)x2] [numVar:2] [nullBitmap] -/// -/// The null bitmap marks present (non-null) columns; a Boolean column has no data and its -/// bit carries the value. Variable columns are laid out in ascending VariableIndex order with -/// an end-first offset table. A memo/OLE column's value is written as an *inline* long-value -/// (12-byte descriptor + payload, §8) when it is small enough; anything larger is stored on LVAL -/// pages by before the row reaches here, so only the descriptor is encoded. -/// A calculated column is the exception, because its value is derived here rather than supplied: an -/// oversized result is spilled through the spillCalculated callback the writer passes in. -/// -public sealed class RowEncoder(IReadOnlyList columns, JetFormatBase format, - int? fixedDataLength = null, int? variableColumnCount = null, - Func? spillCalculated = null) -{ - private readonly IReadOnlyList _columns = columns; - - // Nothing in the row layout varies by format across the Jet 4 family, so this is unread today. It stays - // because Jet 3 rows differ (1-byte column counts, no variable-length offset widening), and the encoder - // will have to branch on it when that format is implemented. -#pragma warning disable CA1823, IDE0052 // Deliberately unread; see above. - private readonly JetFormatBase _format = format; -#pragma warning restore CA1823, IDE0052 - - // How an oversized calculated result reaches an LVAL page, returning the in-row descriptor. A calculated - // column is the one value the encoder DERIVES rather than receives, so it cannot have been materialised - // before the row arrived here the way a plain memo is; the writer that owns the pages supplies this - // instead. Null for a standalone encode, which has no channel to write to and so must refuse. - private readonly Func? _spillCalculated = spillCalculated; - - // How many variable slots a row carries. The TDEF's 0x2B when the caller has it — a high-water that - // never decrements — else the tight maximum over the live columns, which is the same number until the - // LAST variable column is dropped and is all a standalone encode can know. - private readonly int? _variableColumnCount = variableColumnCount; - - // Fixed (non-boolean) columns occupy a contiguous region; its length is defined by the - // table definition. Default to the tight max so a standalone encode round-trips; INSERT - // passes the TDEF's actual fixed-row size so the on-disk layout matches Access. - private readonly int _fixedDataLength = fixedDataLength ?? ComputeFixedDataLength(columns); - - public byte[] Encode(object?[] values) => Encode(values, null); - - /// Encodes a row, recomputing its calculated columns. - /// The row's values, one per column in table order. - /// Envelopes to carry over verbatim, keyed by - /// . An UPDATE that touches nothing a calculated column reads must leave - /// the cached value exactly as it was: ACE recomputes only when a referenced column is written, so - /// recomputing unconditionally would write bytes ACE would not have (§3.4a). Null on INSERT, where every - /// calculated column is computed fresh. - /// The row as its columns hold it, for the calculated columns to read, when - /// already carries long values as their on-disk descriptors. A memo's text is - /// what an expression reads, not the descriptor that points at it. Null when the two are the same. - public byte[] Encode(object?[] values, IReadOnlyDictionary? preservedCalculated, - object?[]? logicalValues = null) - { - if (values.Length != _columns.Count) - throw new ArgumentException($"Expected {_columns.Count} values, got {values.Length}.", nameof(values)); - - // The leading count and the null-bitmap width are driven by the highest column id + 1, NOT the live - // column count — the two coincide only while ids are contiguous (fresh table / ADD COLUMN), and diverge - // once ids have a gap (a burned type-change id, or a DROP COLUMN gap). Verified vs ACE (spec §5). - // AssembleRow derives both from this. - int maxColumnId = _columns.Count == 0 ? -1 : _columns.Max(c => c.ColumnId); - - // And the variable section is addressed the same way: by VariableIndex, NOT by position among the - // live variable columns. DROP COLUMN leaves a hole in the index space — the TDEF's 0x2B count is a - // high-water mark that never decrements, and a column added later takes the next index above it — so - // packing the chunks densely puts every column after the hole one slot too low. The decoder reads - // VarChunk(column.VariableIndex) and so does ACE, which is what makes it silent: the row is written - // and read back happily by nothing at all. - var varCols = _columns.Where(c => !c.IsFixedLength).ToList(); - int numVar = varCols.Count == 0 ? 0 - : Math.Max(_variableColumnCount ?? 0, varCols.Max(c => c.VariableIndex) + 1); - - // Encode each region's payload first so we can size the row exactly. - var fixedRegion = new byte[_fixedDataLength]; - foreach (ColumnDef column in _columns) - { - if (column.Type == JetDataType.Boolean || !column.IsFixedLength) continue; - object? v = values[column.Index]; - if (v is null) continue; // null fixed value: leave its slot zeroed, clear the bit below - byte[] encoded = JetTypeCodec.Encode(column, v); - if (encoded.Length != column.Length) - throw new InvalidOperationException($"Column '{column.Name}' encoded to {encoded.Length} bytes, expected {column.Length}."); - encoded.CopyTo(fixedRegion.AsSpan(column.FixedOffset)); - } - - var varChunks = new byte[numVar][]; - Array.Fill(varChunks, []); // a dropped column's slot stays present, and empty - foreach (ColumnDef column in varCols) - { - if (column.IsCalculated) - { - varChunks[column.VariableIndex] = EncodeCalculated(column, logicalValues ?? values, preservedCalculated); - continue; - } - object? v = values[column.Index]; - varChunks[column.VariableIndex] = v is null ? [] : JetTypeCodec.Encode(column, v); - } - - return AssembleRow(maxColumnId, fixedRegion, varChunks, _columns, values); - } - - /// The slot for a calculated column: the envelope holding the result, wrapped in a long-value - /// descriptor when the column owns a long-value map (which is how ACE stores a calculated Memo). - private byte[] EncodeCalculated(ColumnDef column, object?[] values, - IReadOnlyDictionary? preservedCalculated) - { - if (preservedCalculated is not null && preservedCalculated.TryGetValue(column.Index, out byte[]? kept)) - return kept; - - byte[] envelope = CalculatedValue.Encode(column, CalculatedValue.Evaluate(column, _columns, values)); - if (!column.HasLongValueMap) return envelope; - - // A memo-backed result inlines while it fits and spills to an LVAL page once it does not — the same - // 64-byte boundary a plain memo uses, and for the same reason: Access reads an inlined long value - // back but refuses one that should have been on a page. What goes on the page is the WHOLE envelope, - // header and padding included, because that is what the in-row slot would otherwise have held. - if (envelope.Length <= LongValueFormat.MaxInlineValue) - return JetTypeCodec.EncodeInlineLongValue(envelope); - - return _spillCalculated is not null - ? _spillCalculated(column, envelope) - : throw new NotSupportedException( - $"Calculated column '{column.Name}' produced a {envelope.Length}-byte value, too large to " - + "store inline. Encoding it needs a writer that can allocate long-value pages, which a " - + "standalone RowEncoder has no channel for — insert or update through RowInserter."); - } - - /// Rejects a variable TEXT/BINARY value longer than its column's declared width, as ACE does - /// (measured in ColumnLengthAccessTests); without it LibRed wrote rows Access will not read back. - /// Memo/OLE are exempt — they encode to a long-value descriptor whose size is unrelated to - /// . Fixed columns are checked by - /// before the codec pads them, since padding to width would otherwise hide an over-long value. - private static void EnsureFitsDeclaredLength(ColumnDef column, byte[] encoded) - { - if (column.Type is not (JetDataType.Text or JetDataType.Binary)) return; - if (column.Length <= 0 || encoded.Length <= column.Length) return; - - // Report in the column's own units: TEXT declares characters and stores UTF-16, BINARY declares bytes. - bool text = column.Type == JetDataType.Text; - int declared = text ? column.Length / 2 : column.Length; - int actual = text ? encoded.Length / 2 : encoded.Length; - throw new InvalidOperationException( - $"The field '{column.Name}' is too small to accept the amount of data you attempted to add: " - + $"{actual} {(text ? "characters" : "bytes")} into a column declared to hold {declared}."); - } - - /// - /// Smallest fixed region ACE writes in a row that has no variable trailer. It is a floor, not an - /// alignment: measured against ACE, a region of 0 bytes (a table of only Booleans, which occupy none) - /// is padded to 2 and 1 byte (a lone BYTE column) to 2, while 3 stays 3 — a three-BYTE table's - /// row is 6 bytes, odd region and all. A row that carries a variable trailer is exempt: ACE leaves a - /// TEXT-only table's fixed region at 0. - /// - /// - /// Matching it is not cosmetic. Without the pad an all-Boolean table of eight columns or fewer encodes to a - /// 3-byte record, and ACE misreads that record — every Boolean in it comes back False, whichever - /// engine created the table. Measured both ways round: ACE's own table filled by LibRed read False, and - /// LibRed's table filled by ACE read True, which is what pins the fault to the record rather than the TDEF. - /// The cliff is at 4 bytes — a 16-Boolean row (2-byte bitmap, so 4 bytes) reads back correctly — but ACE's - /// own writer never emits a record under 5, so the short form is simply a shape its reader has never met. - /// The TDEF's fixed-row length keeps the true, unpadded value: ACE stores 1 for a BYTE table while - /// writing 5-byte rows into it, so this rounding happens at row-write time and nowhere else. - /// - private const int MinFixedRegion = 2; - - /// Assembles the on-disk row bytes from a prepared fixed region and the ordered variable chunks: - /// [count][fixed][var data][var-offset table][numVar] (the variable section is omitted entirely when - /// there are none) then [null bitmap]. The count and bitmap width are maxColumnId + 1; a - /// column's bit is set when present (Boolean = its truthy value), and dead ids (gaps below the max, from a - /// burned/dropped id) are set present too — all verified vs ACE (§5). Shared by Encode and - /// the ALTER COLUMN row re-lay so the two can never drift. - /// - /// The declared-width check runs HERE rather than in Encode. It used to sit above this call, - /// which meant the ALTER COLUMN re-lay — the other caller — never got it, and a narrowing retype could - /// write rows Access refuses. A guard that both paths must pass through belongs on the shared path. - /// - internal static byte[] AssembleRow(int maxColumnId, ReadOnlySpan fixedRegion, - IReadOnlyList varChunks, IReadOnlyList columns, object?[] values) - { - // A calculated column is exempt from the declared-width check: its length field is a constant ACE - // writes (39 for a value type, 509 for any Text, whatever size was asked for), not a limit — a - // Text(5) calculated column stores a far longer result and ACE reads it back in full (§3.4a). - foreach (ColumnDef column in columns) - if (!column.IsFixedLength && !column.IsCalculated - && column.VariableIndex >= 0 && column.VariableIndex < varChunks.Count) - EnsureFitsDeclaredLength(column, varChunks[column.VariableIndex]); - - const int countSize = 2; - int count = maxColumnId + 1; - int nullBitmapSize = (count + 7) / 8; - int numVar = varChunks.Count; - int varDataLength = 0; - for (int j = 0; j < numVar; j++) varDataLength += varChunks[j].Length; - int varSectionLen = numVar > 0 ? varDataLength + (numVar + 1) * 2 + 2 : 0; - - // ACE pads an all-fixed row's fixed region out to MinFixedRegion; a row with a variable trailer is - // left alone. The pad is zero bytes between the fixed values and the null bitmap. - int fixedLen = numVar > 0 ? fixedRegion.Length : Math.Max(fixedRegion.Length, MinFixedRegion); - - var row = new byte[countSize + fixedLen + varSectionLen + nullBitmapSize]; - BinaryPrimitives.WriteUInt16LittleEndian(row, (ushort)count); - fixedRegion.CopyTo(row.AsSpan(countSize)); - - int bitmapPos; - if (numVar > 0) - { - int varDataStart = countSize + fixedLen; - int pos = varDataStart; - for (int j = 0; j < numVar; j++) { varChunks[j].CopyTo(row.AsSpan(pos)); pos += varChunks[j].Length; } - - // End-first offset table: entry[numVar] = var-data start, entry[numVar-j-1] = end of var col j. - int tableStart = pos; - BinaryPrimitives.WriteUInt16LittleEndian(row.AsSpan(tableStart + numVar * 2, 2), (ushort)varDataStart); - int running = varDataStart; - for (int j = 0; j < numVar; j++) - { - running += varChunks[j].Length; - BinaryPrimitives.WriteUInt16LittleEndian(row.AsSpan(tableStart + (numVar - j - 1) * 2, 2), (ushort)running); - } - int numVarPos = tableStart + (numVar + 1) * 2; - BinaryPrimitives.WriteUInt16LittleEndian(row.AsSpan(numVarPos, 2), (ushort)numVar); - bitmapPos = numVarPos + 2; - } - else bitmapPos = countSize + fixedLen; - - var liveIds = new HashSet(); - foreach (ColumnDef column in columns) - { - liveIds.Add(column.ColumnId); - // A calculated column is always present, even when its expression evaluated to Null — ACE marks - // the bit and stores a zero-length payload, so here the bit means "has an envelope" and says - // nothing about the value (§3.4a). It is also why a calculated Boolean cannot use the bit as its - // value the way a real one does. - bool present = column.IsCalculated - || (column.Type == JetDataType.Boolean ? IsTruthy(values[column.Index]) : values[column.Index] is not null); - if (present) row[bitmapPos + (column.ColumnId >> 3)] |= (byte)(1 << (column.ColumnId & 7)); - } - for (int id = 0; id <= maxColumnId; id++) // dead ids read present in ACE - if (!liveIds.Contains(id)) - row[bitmapPos + (id >> 3)] |= (byte)(1 << (id & 7)); - return row; - } - - /// Access truthiness for a Boolean (bit) value being stored: a bool is itself, any non-zero - /// number is true, 0 / null is false. - private static bool IsTruthy(object? value) => value switch - { - null => false, - bool b => b, - _ => Convert.ToBoolean(value, System.Globalization.CultureInfo.InvariantCulture), - }; - - private static int ComputeFixedDataLength(IReadOnlyList columns) - { - int length = 0; - foreach (ColumnDef c in columns) - if (c.IsFixedLength && c.Type != JetDataType.Boolean) - length = Math.Max(length, c.FixedOffset + c.Length); - return length; - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/RowId.cs b/src/LibRed/LibRed.Core/Storage/RowId.cs index 36613a7a0..d9e7acea5 100644 --- a/src/LibRed/LibRed.Core/Storage/RowId.cs +++ b/src/LibRed/LibRed.Core/Storage/RowId.cs @@ -1,4 +1,12 @@ namespace LibRed.Storage; /// A pointer to a row: the data page it lives on and its slot index within that page. -public readonly record struct RowId(int Page, int Row); \ No newline at end of file +public readonly record struct RowId(int Page, int Row) +{ + /// The row as one integer, page in the upper three bytes and row in the lowest: the value an index + /// entry's trailer carries (big-endian), and a record pointer (little-endian). + public int Packed => (Page << 8) | (Row & 0xFF); + + /// The inverse of . + public static RowId FromPacked(int packed) => new((int)((uint)packed >> 8), packed & 0xFF); +} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/RowInserter.cs b/src/LibRed/LibRed.Core/Storage/RowInserter.cs index 55bb79bdd..5faab4eb9 100644 --- a/src/LibRed/LibRed.Core/Storage/RowInserter.cs +++ b/src/LibRed/LibRed.Core/Storage/RowInserter.cs @@ -3,7 +3,6 @@ using LibRed.IO; using LibRed.Pages; using System.Buffers; -using System.Buffers.Binary; using System.Globalization; using System.Text; @@ -14,11 +13,11 @@ namespace LibRed.Storage; /// stores any long values, maintains every index B-tree, and enforces unique keys. The page's slot /// directory grows forward while row data is packed from the page end backward (see ). /// -public sealed class RowInserter(PageChannel channel, TableDef table) +internal sealed class RowInserter(PageChannel channel, TableDefinition table) { private readonly PageChannel _channel = channel; - private readonly UsageMapWriter _usageMaps = new(channel); - private readonly TableDef _table = table; + private readonly UsageMap _usageMaps = new(channel); + private readonly TableDefinition _table = table; private bool HasCalculatedColumns => _table.Columns.Any(c => c.IsCalculated); @@ -41,81 +40,98 @@ public void Insert(object?[] values, bool updateIndexes) // as-is (Jet, unlike SQL Server, permits explicit AutoNumber values); either way the row's // final id drives both the row encoding and the high-water update below. bool[]? generatedAutoNumbers = AssignAutoNumbers(format, values); - if (updateIndexes) EnforceUniqueIndexes(values); // reject a duplicate before writing anything + if (updateIndexes) + { + EnforceNonNullKeys(values); // after AssignAutoNumbers, so a counter key has its value + EnforceUniqueIndexes(values); // reject a duplicate before writing anything + } - // Index keys are encoded from the *logical* values. MaterializeLongValues replaces a memo/OLE value - // with its on-disk LongValueDescriptor, and a Memo column IS indexable (its key is the collation key - // of the first 255 characters), so snapshot the values first and key the index off that snapshot. - // A calculated column reads the same logical values, so the snapshot serves it too. + // From here the row exists in two forms, and they are two ARRAYS. `values` stays as the caller gave + // it — the logical row — and `storage` is the copy long values are materialised into, where a memo + // becomes the 12-byte descriptor naming its pages. Everything that needs the value reads `values` + // (index keys — a Memo is indexable, on the collation key of its first 255 characters — and + // calculated expressions); only the encoder reads `storage`. See ToStorageValues. bool calculated = HasCalculatedColumns; - object?[] keyValues = updateIndexes || calculated ? (object?[])values.Clone() : values; - MaterializeLongValues(values); + object?[] storage = ToStorageValues(values); // Encode first: the fixed-region length is pinned by any existing row (to match Access), // or derived from the columns for a just-created empty table. - var encoder = new RowEncoder(_table.Columns, format, InferFixedDataLength(format), - _table.VariableColumnCount, SpillCalculated); - byte[] record = encoder.Encode(values, null, calculated ? keyValues : null); + var encoder = new RowCodec(_table.Columns, format, InferFixedDataLength(format), + _table.VariableColumnCount, SpillCalculated, _table.ColumnIdHighWater); + byte[] record = encoder.Encode(storage, null, calculated ? values : null); EnsureRecordFits(format, record); - // Then find an owned page with room for the record plus its 2-byte slot entry. - (int pageNumber, byte[] page) = FindPageWithRoom(format, record.Length + 2); - - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2)); - int freeSpace = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2)); - - // Rows are packed from the page end backward, with strictly decreasing slot offsets, - // so the new row goes just below the current lowest row start. - int lowestOffset = LowestRowOffset(page, format, rowCount); - int newOffset = lowestOffset - record.Length; - record.CopyTo(page.AsSpan(newOffset)); - - // Append the slot, bump the row count, shrink free space by row + slot-entry bytes. - // The new row's slot index is the old row count, giving its row id on this page. - BinaryPrimitives.WriteUInt16LittleEndian( - page.AsSpan(format.DataRowDirectoryOffset + rowCount * 2, 2), (ushort)(newOffset & RowPointer.OffsetMask)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), (ushort)(rowCount + 1)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), (ushort)(freeSpace - record.Length - 2)); - + // Then find an owned page with room for the record plus its slot entry. + (int pageNumber, byte[] page) = FindPageWithRoom(format, record.Length + format.DataRowDirectoryEntrySize); + int rowCount = AppendRow(format, pageNumber, page, record, RowSlotFlags.None); _channel.WritePage(pageNumber, page); // Asked before the row's own entries go in, so an index "already has" a key only through another row. - HashSet newKeys = updateIndexes ? IndexesGainingANewKey(keyValues) : []; + HashSet newKeys = updateIndexes ? IndexesGainingANewKey(values) : []; UpdateTdefCounters(format, values, generatedAutoNumbers, newKeys); if (updateIndexes) - UpdateIndexes(keyValues, new RowId(pageNumber, rowCount)); + UpdateIndexes(values, new RowId(pageNumber, rowCount)); } /// /// Rewrites an existing row in place at its slot (page + row index preserved, matching Access, so /// index rowid pointers stay valid). Any changed memo/OLE value is re-materialized onto LVAL pages; /// the page is repacked to absorb a size change (slot order = physical order, as Access keeps it). - /// Throws if the row no longer fits its page (relocation not implemented yet); index-key maintenance for - /// a changed indexed column is the caller's responsibility. Does not touch the old LVAL pages (freeing - /// them is a follow-up). + /// A row that no longer fits its page is relocated — the slot becomes a 4-byte forward pointer and + /// the row moves to a page with room, as Access does — and one that outgrows its new page moves again. + /// Index-key maintenance for a changed indexed column is the caller's responsibility. /// public void Update(RowId id, object?[] values, IReadOnlySet changedColumns) + { + // An update is several writes — the old long values are freed, the new ones materialised onto pages, + // then the row itself — and the row is the last of them, so it is the last thing that can fail (a + // record that has grown past 4060, a page that will not take it). Under the SQL engine every + // statement already runs in one, but a caller using the Core API directly had no such cover: the + // failure left the old memo's pages freed and the new value's written, with the row still naming the + // old one. The order is deliberate and unchanged — ACE frees an update's pages at once, and the new + // value may land on them — so what this adds is only the undo. + bool ownTransaction = !_channel.InTransaction; + if (ownTransaction) _channel.BeginTransaction(); + try + { + UpdateCore(id, values, changedColumns); + if (ownTransaction) _channel.CommitTransaction(flush: false); + } + catch when (ownTransaction) + { + _channel.RollbackTransaction(); + throw; + } + } + + private void UpdateCore(RowId id, object?[] values, IReadOnlySet changedColumns) { JetFormatBase format = _channel.Format; RejectExplicitCalculatedValues(values, changedColumns); + EnforceNonNullKeys(values, changedColumns); + EnforceUniqueIndexes(values, id.Packed, changedColumns); - // The row as its columns hold it, for a recomputed calculated column to read: the loop below swaps an - // unchanged memo's text for its on-disk descriptor, and MaterializeLongValues the changed ones. - object?[]? logicalValues = HasCalculatedColumns ? (object?[])values.Clone() : null; + // Two arrays, as on the insert path: `values` stays the logical row the caller gave, and `storage` is + // where the long values become descriptors. The caller keeps the values it passed — an index entry + // this update moves has to be keyed off them, and a descriptor is not a key. + object?[] storage = (object?[])values.Clone(); // Long-value (memo/OLE) columns: keep an unchanged column's on-disk descriptor verbatim (so it is not - // needlessly re-materialised onto fresh LVAL pages), and free a changed column's old chained pages. + // needlessly re-materialised onto fresh LVAL pages), and free a changed column's old chained pages — + // after the new values are written, as ACE does: the UPDATE that frees them never reuses them. byte[] oldRow = ReadRowBytes(id); - var decoder = new RowDecoder(_table.Columns, format); - var oldDescriptors = decoder.LongValueRaw(oldRow); - var oldCalculated = decoder.CalculatedRaw(oldRow); + // One layout for both: the trailer arithmetic is the same for either selection off this row. + RowCodec.Layout oldLayout = RowCodec.ParseLayout(_table.Columns, format, oldRow); + var oldDescriptors = RowCodec.LongValueDescriptors(_table.Columns, oldLayout, oldRow); + var oldCalculated = RowCodec.CalculatedSlots(_table.Columns, oldLayout, oldRow); + var replaced = new List<(ColumnDef Column, byte[] Descriptor)>(); foreach (ColumnDef column in _table.Columns) { if (column.Type is not (JetDataType.Memo or JetDataType.Ole)) continue; if (!oldDescriptors.TryGetValue(column.Index, out byte[]? oldDescriptor)) continue; // old value was null - if (changedColumns.Contains(column.Index)) FreeLongValue(column, oldDescriptor, releaseAtClose: false); - else values[column.Index] = new LongValueDescriptor(oldDescriptor); + if (changedColumns.Contains(column.Index)) replaced.Add((column, oldDescriptor)); + else storage[column.Index] = new LongValueStore.DescriptorValue(oldDescriptor); } // A calculated column is recomputed only when the UPDATE writes a column its expression READS — @@ -141,19 +157,21 @@ public void Update(RowId id, object?[] values, IReadOnlySet changedColumns) if (!column.IsCalculated || !column.HasLongValueMap) continue; if (preservedCalculated?.ContainsKey(column.Index) == true) continue; if (oldCalculated.TryGetValue(column.Index, out byte[]? stale) - && stale.Length >= LongValueFormat.DescriptorSize) + && stale.Length >= format.LongValueDescriptorSize) FreeLongValue(column, stale, releaseAtClose: false); } - MaterializeLongValues(values); + MaterializeLongValues(storage); + foreach ((ColumnDef column, byte[] descriptor) in replaced) + FreeLongValue(column, descriptor, releaseAtClose: false); byte[] srcPage = _channel.ReadPageShared(id.Page).Span.ToArray(); // Use the same guarded inference as Insert (Math.Max with the column-derived length): the raw per-row // parse returns a negative length for an all-fixed-column table (no variable columns — e.g. Northwind // Order Details), which without the guard would overflow `new byte[len]`. - var encoder = new RowEncoder(_table.Columns, format, InferFixedDataLength(format), - _table.VariableColumnCount, SpillCalculated); - byte[] record = encoder.Encode(values, preservedCalculated, logicalValues); + var encoder = new RowCodec(_table.Columns, format, InferFixedDataLength(format), + _table.VariableColumnCount, SpillCalculated, _table.ColumnIdHighWater); + byte[] record = encoder.Encode(storage, preservedCalculated, HasCalculatedColumns ? values : null); // Here as well as on the insert path, and before the in-place rewrite rather than beside the // page-search: a row that grows past the cap but still fits its current page is rewritten where it @@ -161,17 +179,22 @@ public void Update(RowId id, object?[] values, IReadOnlySet changedColumns) // unreadable row an INSERT was stopped from producing. EnsureRecordFits(format, record); - int raw = BinaryPrimitives.ReadUInt16LittleEndian(srcPage.AsSpan(format.DataRowDirectoryOffset + id.Row * 2, 2)); - if ((raw & RowPointer.OverflowFlag) != 0) + if (DataPage.ReadSlot(srcPage, format, id.Row).Flags.HasFlag(RowSlotFlags.Overflow)) { - // This slot is a 4-byte forward pointer to the real (relocated) row; rewrite it on its target page - // (which keeps its hidden "deleted" flag). If it grows past that page too, we'd need to re-relocate. + // This slot is a forward record pointer to the real (relocated) row; rewrite it on its target page + // (which keeps its hidden "deleted" flag). DataPage.TryReadRow(new PageBuffer(srcPage, id.Page), format, id.Row, - out RowSlot sourceSlot, out ReadOnlySpan sourceBytes); - RelocatedRow target = RowRelocationReader.Resolve( + out DataPage.RowSlot sourceSlot, out ReadOnlySpan sourceBytes); + DataPage.RelocatedRow target = DataPage.ResolveRelocation( _channel, _table.DefinitionPage, sourceSlot, sourceBytes); - if (!TryRewriteRowInPlace(target.Buffer.PageNumber, target.RowNumber, record)) - throw new NotSupportedException("Re-relocating an already-relocated row that grew again is not supported yet."); + if (TryRewriteRowInPlace(target.Buffer.PageNumber, target.RowNumber, record)) return; + + // It has outgrown the page it was moved to as well, so it moves again. The old hidden row is + // reclaimed first — through the same path a delete uses, and while the source slot still points at + // it — and the pointer is then re-aimed, exactly as the first move wrote it. Nothing else changes: + // the row keeps its id, so every index entry still names it. + ReclaimRelocationTarget(format, target); + Relocate(format, id, record); return; } @@ -181,16 +204,24 @@ public void Update(RowId id, object?[] values, IReadOnlySet changedColumns) // It no longer fits: relocate the row to another page as a hidden ("deleted") record, and turn this // slot into a 4-byte forward pointer (row id preserved, so index entries stay valid) — Access's own // overflow mechanism, verified against ACE. + Relocate(format, id, record); + } + + /// Writes as a hidden row on a page with room and turns the slot at + /// into the forward record pointer that names it. A pointer always fits the slot the + /// row is vacating, so this cannot fail for want of space. + private void Relocate(JetFormatBase format, RowId id, byte[] record) + { (int targetPage, int targetRow) = WriteHiddenRow(format, record); - var pointerBytes = new byte[4]; - BinaryPrimitives.WriteInt32LittleEndian(pointerBytes, (targetPage << 8) | targetRow); - TryRewriteRowInPlace(id.Page, id.Row, pointerBytes, addFlags: RowPointer.OverflowFlag); // 4 bytes always fits + var pointerBytes = new byte[PageBuffer.RecordPointerSize]; + PageBuffer.WriteRecordPointer(pointerBytes, 0, targetRow, targetPage); + TryRewriteRowInPlace(id.Page, id.Row, pointerBytes, addFlags: RowSlotFlags.Overflow); } /// Rewrites the row at (page, slot) in place, repacking the page from the end in slot order so /// every row id is preserved and each slot keeps its flags (plus on the /// target). Returns false without writing if the row no longer fits the page. - private bool TryRewriteRowInPlace(int pageNumber, int slot, byte[] record, int addFlags = 0) + private bool TryRewriteRowInPlace(int pageNumber, int slot, byte[] record, RowSlotFlags addFlags = RowSlotFlags.None) { JetFormatBase format = _channel.Format; // Pool the transient working page: read into a rented buffer, repack in place, write back, return it — @@ -199,17 +230,15 @@ private bool TryRewriteRowInPlace(int pageNumber, int slot, byte[] record, int a try { _channel.ReadPage(pageNumber, page); - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2)); + int rowCount = DataPage.ReadRowCount(page, format); var rows = new byte[rowCount][]; - var rawDir = new int[rowCount]; - int directoryEnd = format.DataRowDirectoryOffset + rowCount * 2; + var flags = new RowSlotFlags[rowCount]; + int directoryEnd = DataPage.DirectoryEnd(format, rowCount); int prevEnd = format.PageSize; for (int i = 0; i < rowCount; i++) { - int raw = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + i * 2, 2)); - rawDir[i] = raw; - int offset = raw & RowPointer.OffsetMask; + (int offset, flags[i]) = DataPage.ReadSlot(page, format, i); // Same invariant DataPage enforces on the read side — this repacker rewrites the page, so a // bad directory entry must be reported before the first CopyTo rather than mid-repack. DataPage.ValidateSlot(pageNumber, format.PageSize, i, offset, prevEnd, directoryEnd); @@ -217,21 +246,9 @@ private bool TryRewriteRowInPlace(int pageNumber, int slot, byte[] record, int a prevEnd = offset; } rows[slot] = record; - rawDir[slot] |= addFlags; - - int total = rows.Sum(r => r.Length); - if (total > format.PageSize - format.DataRowDirectoryOffset - rowCount * 2) return false; + flags[slot] |= addFlags; - int off = format.PageSize; - for (int i = 0; i < rowCount; i++) - { - off -= rows[i].Length; - rows[i].CopyTo(page.AsSpan(off)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + i * 2, 2), - (ushort)((rawDir[i] & ~RowPointer.OffsetMask) | (off & RowPointer.OffsetMask))); - } - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(off - format.DataRowDirectoryOffset - rowCount * 2)); + if (!DataPage.LayRows(page.AsSpan(0, format.PageSize), format, rows, flags)) return false; _channel.WritePage(pageNumber, page.AsSpan(0, format.PageSize)); return true; } @@ -242,32 +259,34 @@ private bool TryRewriteRowInPlace(int pageNumber, int slot, byte[] record, int a /// "deleted" so scans skip it there — it's only reached via the forward pointer). Returns its location. private (int Page, int Row) WriteHiddenRow(JetFormatBase format, byte[] record) { - (int pageNumber, byte[] page) = FindPageWithRoom(format, record.Length + 2); - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2)); - int freeSpace = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2)); - int newOffset = LowestRowOffset(page, format, rowCount) - record.Length; - record.CopyTo(page.AsSpan(newOffset)); - - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + rowCount * 2, 2), - (ushort)((newOffset & RowPointer.OffsetMask) | RowPointer.DeletedFlag)); // hidden target - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), (ushort)(rowCount + 1)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), (ushort)(freeSpace - record.Length - 2)); + (int pageNumber, byte[] page) = FindPageWithRoom(format, record.Length + format.DataRowDirectoryEntrySize); + int rowCount = AppendRow(format, pageNumber, page, record, RowSlotFlags.Deleted); // hidden target _channel.WritePage(pageNumber, page); return (pageNumber, rowCount); } + /// Appends to a page chose, returning its row. + /// That page declared the room, so a page whose rows leave none is corrupt. + private static int AppendRow(JetFormatBase format, int pageNumber, byte[] page, byte[] record, RowSlotFlags flags) => + DataPage.TryAppendRow(page, format, record, flags, out int row) + ? row + : throw new InvalidDataException( + $"Data page {pageNumber} declares room for a {record.Length}-byte row, but its rows leave none."); + /// /// Deletes the row at and reclaims its space, then decrements the TDEF row count - /// (0x10). The caller removes the row's index entries first. (The row's LVAL pages, if any, are freed - /// above; nothing else about them is reclaimed yet.) + /// (0x10). The caller removes the row's index entries first. A relocated row's hidden target is reclaimed + /// too, and its LVAL pages are freed — held until the connection closes, as ACE holds a delete's. A page + /// whose last live row this was is released outright (). /// public void Delete(RowId id) { JetFormatBase format = _channel.Format; + bool released; // Free the deleted row's chained long-value pages — held until close, as ACE holds them. byte[] rowBytes = ReadRowBytes(id); - var oldDescriptors = new RowDecoder(_table.Columns, format).LongValueRaw(rowBytes); + var oldDescriptors = RowCodec.LongValueDescriptors(_table.Columns, format, rowBytes); // Which real indexes this row was the LAST holder of a key for. Asked before the row goes, and with // the row itself excluded, so the answer is "does another row still carry this key" either way — @@ -282,8 +301,7 @@ public void Delete(RowId id) try { _channel.ReadPage(id.Page, page); - int dir = format.DataRowDirectoryOffset + id.Row * 2; - ushort entry = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(dir, 2)); + RowSlotFlags entry = DataPage.ReadSlot(page, format, id.Row).Flags; // A LIVE slot carrying the overflow flag is a 4-byte forward pointer to a relocated row rather // than the row itself, so the row it forwards to has to go first — otherwise deleting a relocated @@ -294,20 +312,31 @@ public void Delete(RowId id) // Deleted must be excluded, not just overflow: ReclaimRow's own tombstone is deleted+overflow // (0xC000) and zero-length, so testing the overflow bit alone treats a tombstone as a relocation // source and reads four bytes of the NEIGHBOURING row as a page/row pointer. - if ((entry & (RowPointer.DeletedFlag | RowPointer.OverflowFlag)) == RowPointer.OverflowFlag) - ReclaimRelocationTarget(format, page, id.Row); + if ((entry & (RowSlotFlags.Deleted | RowSlotFlags.Overflow)) == RowSlotFlags.Overflow) + { + DataPage.TryReadRow(new PageBuffer(page.AsMemory(0, format.PageSize), id.Page), format, id.Row, + out DataPage.RowSlot sourceSlot, out ReadOnlySpan sourceBytes); + ReclaimRelocationTarget(format, DataPage.ResolveRelocation( + _channel, _table.DefinitionPage, sourceSlot, sourceBytes)); + } - ReclaimRow(format, page, id.Row); - _channel.WritePage(id.Page, page.AsSpan(0, format.PageSize)); + DataPage.ReclaimRow(page.AsSpan(0, format.PageSize), format, id.Row); + released = WriteReclaimedPage(format, page, id.Page); } finally { ArrayPool.Shared.Return(page); } + // The page has room again, so its bit goes back into the table's free-pages map — the mirror of the + // insert path clearing it when the page filled (AllocateDataPage). Without this the space a delete + // frees in a full page is space no later insert ever looks at, and ACE sets the bit here too + // (measured: deleting a row from the first of many pages, ACE writes the map holder and LibRed did not). + // A released page is the exception: it has already left both of the table's maps. + if (!released) UpdateUsageBit(format.TdefFreePagesOffset, id.Page, set: true); + byte[] tdef = ArrayPool.Shared.Rent(format.PageSize); try { _channel.ReadPage(_table.DefinitionPage, tdef); - int rowCount = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(format.TdefRowCountOffset, 4)); - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(format.TdefRowCountOffset, 4), rowCount - 1); + TableDefinition.WriteRowCount(tdef, format, TableDefinition.ReadRowCount(tdef, format) - 1); // ACE's index entry counts on delete, measured against a DAO delete of the same row // (ComplexDeleteByteParityProbeTest): the TOTAL count (+0) drops by one on every real index, and @@ -322,14 +351,10 @@ public void Delete(RowId id) // ACE is not maintaining, and it stops touching the pair. Without this the totals go negative. foreach (IndexDef index in _table.Indexes.GroupBy(i => i.RealIndexOrdinal).Select(g => g.First())) { - int at = format.TdefRealIndexBlockOffset + index.RealIndexOrdinal * format.RealIndexEntrySize; - int total = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(at, 4)); + (int total, int unique) = TableDefinition.ReadIndexCounts(tdef, format, index.RealIndexOrdinal); if (total == 0) continue; - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(at, 4), total - 1); - - if (!lastKeyLost.Contains(index.RealIndexOrdinal)) continue; - int unique = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(at + 4, 4)); - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(at + 4, 4), unique - 1); + TableDefinition.WriteIndexCounts(tdef, format, index.RealIndexOrdinal, total - 1, + lastKeyLost.Contains(index.RealIndexOrdinal) ? unique - 1 : unique); } _channel.WritePage(_table.DefinitionPage, tdef.AsSpan(0, format.PageSize)); @@ -337,6 +362,25 @@ public void Delete(RowId id) finally { ArrayPool.Shared.Return(tdef); } } + /// Counts an UPDATE's key change in 's statistics as ACE does (verified): the + /// total drops by one, as on a delete, and the unique count is then held to no more than the total. Whether the + /// old key had another holder does not matter, and the new key advances nothing. A total already reading 0 is + /// left alone, as a delete leaves it. + public void CountKeyMoved(IndexDef index) + { + JetFormatBase format = _channel.Format; + byte[] tdef = ArrayPool.Shared.Rent(format.PageSize); + try + { + _channel.ReadPage(_table.DefinitionPage, tdef); + (int total, int unique) = TableDefinition.ReadIndexCounts(tdef, format, index.RealIndexOrdinal); + if (total == 0) return; + TableDefinition.WriteIndexCounts(tdef, format, index.RealIndexOrdinal, total - 1, Math.Min(unique, total - 1)); + _channel.WritePage(_table.DefinitionPage, tdef.AsSpan(0, format.PageSize)); + } + finally { ArrayPool.Shared.Return(tdef); } + } + /// /// The real-index ordinals for which the row at holds the only copy of its key, so /// removing the row also removes a distinct key. An index the row is excluded from (IgnoreNulls with a @@ -347,104 +391,108 @@ private HashSet IndexesLosingTheirLastKey(JetFormatBase format, byte[] rowB var lost = new HashSet(); if (_table.Indexes.Count == 0) return lost; - object?[] values = new RowDecoder(_table.Columns, format).Decode(rowBytes); - var writer = new IndexWriter(_channel, _table); + // With a long-value reader, because these values are handed to the index key encoder: a Memo column IS + // indexable, and decoded without the reader it comes back as its 12-byte on-disk descriptor, which the + // encoder's text path casts to string and dies on. A decode that feeds a key needs the value the + // column holds, not the pointer to it. + object?[] values = new RowCodec(_table.Columns, format, longValues: new LongValueStore(_channel)).Decode(rowBytes); + var writer = new IndexTree(_channel, _table); foreach (IndexDef index in _table.Indexes.GroupBy(i => i.RealIndexOrdinal).Select(g => g.First())) { if (index.RootPage <= 0) continue; if (index.IgnoreNulls && HasNullKey(index, values)) continue; - if (!writer.KeyExists(index, values, (id.Page << 8) | id.Row)) + if (!writer.KeyExists(index, values, id.Packed)) lost.Add(index.RealIndexOrdinal); } return lost; } - /// Reclaims the row a relocation pointer forwards to, on whatever page it lives. The pointer - /// is the first four bytes of the source row: page = pointer >> 8, row = pointer & 0xFF. - /// The target is flagged deleted so scans skip it, which is why it is read straight off the directory - /// rather than through — but it carries that reader's checks, because - /// this one WRITES: an unvalidated pointer runs ReclaimRow over a live row of an unrelated page, sliding - /// its neighbours and stamping a tombstone. Anything that fails a check is left alone rather than - /// throwing, so a delete still removes the row the caller asked about. - private void ReclaimRelocationTarget(JetFormatBase format, byte[] sourcePage, int row) + /// Reclaims the row a relocation pointer forwards to, on whatever page it lives. The caller resolves + /// through , whose checks matter more here than on a + /// read, because this WRITES: an unvalidated pointer runs ReclaimRow over a live row of an unrelated page, + /// sliding its neighbours and stamping a tombstone. + /// A failed check is corruption, and says so. It used to return quietly so that a delete + /// still removed the row the caller asked about — but the caller only reaches here because the slot is a + /// live relocation pointer, so a pointer that does not resolve means the file already disagrees with + /// itself, and carrying on strands the moved row silently. That is also how a bug in this engine's own + /// relocation writer would look: nothing raised, a page quietly leaked. + private void ReclaimRelocationTarget(JetFormatBase format, DataPage.RelocatedRow target) { - int offset = BinaryPrimitives.ReadUInt16LittleEndian( - sourcePage.AsSpan(format.DataRowDirectoryOffset + row * 2, 2)) & RowPointer.OffsetMask; - int end = row == 0 - ? format.PageSize - : BinaryPrimitives.ReadUInt16LittleEndian( - sourcePage.AsSpan(format.DataRowDirectoryOffset + (row - 1) * 2, 2)) & RowPointer.OffsetMask; - if (offset < format.DataRowDirectoryOffset || end > format.PageSize || end - offset < 4) return; - - int pointer = BinaryPrimitives.ReadInt32LittleEndian(sourcePage.AsSpan(offset, 4)); - int targetPage = pointer >> 8, targetRow = pointer & 0xFF; - if (targetPage <= 0 || targetPage >= _channel.PageCount) return; - byte[] page = ArrayPool.Shared.Rent(format.PageSize); try { - _channel.ReadPage(targetPage, page); - if (BinaryPrimitives.ReadUInt32LittleEndian(page.AsSpan(format.DataOwnerOffset, 4)) - != (uint)_table.DefinitionPage) return; - if (targetRow >= BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2))) return; - - // The target is a hidden inline row: deleted, not itself a relocation source, and nonempty. - ushort slot = BinaryPrimitives.ReadUInt16LittleEndian( - page.AsSpan(format.DataRowDirectoryOffset + targetRow * 2, 2)); - if ((slot & (RowPointer.DeletedFlag | RowPointer.OverflowFlag)) != RowPointer.DeletedFlag) return; - - ReclaimRow(format, page, targetRow); - _channel.WritePage(targetPage, page.AsSpan(0, format.PageSize)); + target.Buffer.Span.CopyTo(page); + DataPage.ReclaimRow(page.AsSpan(0, format.PageSize), format, target.RowNumber); + // Written back, never released, even when that was the page's last row — ACE keeps a page emptied + // this way as an ordinary data page (measured: deleting a relocated row leaves the page it had + // moved to at type 0x01 with a bare 4080 free and its one slot tombstoned, still in the table's + // maps). The release belongs to deleting a LIVE row (see WriteReclaimedPage); the row reclaimed + // here was already hidden, so emptying its page is a space operation rather than a deletion. + _channel.WritePage(target.Buffer.PageNumber, page.AsSpan(0, format.PageSize)); } finally { ArrayPool.Shared.Return(page); } } /// - /// Takes a row's bytes off its page the way ACE does: the rows stored below it slide up to close the - /// gap, their slot offsets follow, and the emptied slot becomes a zero-length tombstone whose offset is - /// the row's FORMER END, flagged deleted + overflow. The freed bytes go back to the page's free-space - /// count, so a delete-heavy table stops growing where ACE's would not. - /// - /// Slot indices never move, which is what keeps index entries and row ids valid — only offsets - /// change. Pointing the tombstone at the former end rather than at the page end is what keeps the - /// directory non-increasing, which relies on to derive each row's length - /// from the previous slot. Verified against ACE for a first, middle and last row: deleting the first of - /// three 19-byte rows gives D000 0FED 0FDA, the middle 0FED CFED 0FDA, the last - /// 0FED 0FDA CFDA, with free space rising by 19 in each case. - /// + /// Writes back the page a DELETE has just reclaimed a row from, and gives the page away when that row was + /// the last live one on it. Returns whether it was released. Only the delete path: a page emptied by + /// reclaiming a hidden relocation target is kept, as ACE keeps it (see ). + /// + /// + /// A page emptied by DELETE is stamped (0x0109), taken + /// out of the table's owned and free maps, and released — held until this handle closes, the route ACE + /// takes for a delete's pages. Nothing else about the page changes: the row count stands and every slot + /// stays the 0-length deleted+overflow tombstone left, which is already what ACE + /// writes, so the stamp is the only differing byte. + /// Measured against ACE over a 1,200-row table with 699 rows deleted from the low end: 18 data pages + /// emptied, and on each one ACE changed exactly the type's low byte, cleared the page from both of the table's + /// maps, and set its bit in the global map. This is the same event as an emptied packed long-value page + /// () and carries the same stamp — 0x0109 is not confined to + /// long-value pages, and these carry the table's own owner at 0x04 rather than an LVAL + /// signature. + /// "Emptied" is every slot deleted and zero-length, not merely every slot deleted: a hidden + /// relocation target is a live row carrying the deleted flag, so a page holding one is not empty. + /// The table's first data page is the exception and is kept whatever happens to it — see + /// . + /// + private bool WriteReclaimedPage(JetFormatBase format, byte[] page, int pageNumber) + { + bool release = IsEmptied(format, page) && !IsFirstDataPage(pageNumber); + if (release) DataPage.MarkReleased(page); + _channel.WritePage(pageNumber, page.AsSpan(0, format.PageSize)); + if (!release) return false; + + UpdateUsageBit(format.TdefOwnedPagesOffset, pageNumber, set: false); + UpdateUsageBit(format.TdefFreePagesOffset, pageNumber, set: false); + _channel.Allocator.Release(pageNumber); + return true; + } + + /// + /// The first page of the table's owned-pages map — the one data page ACE never gives back, however empty + /// it gets. Measured: deleting every row of a 35-page table releases 34 of them and leaves this one at + /// 0x01, still in both of the table's maps, with all its slots tombstoned like the rest. It is not + /// "keep one page": it stays even while later pages still hold live rows. /// - internal static void ReclaimRow(JetFormatBase format, byte[] page, int row) + /// Read as the lowest page in the owned map, which in every file measured is also the page the + /// table was created with — the two readings are not distinguished here. + private bool IsFirstDataPage(int pageNumber) => + new UsageMap(_channel, _table).MinDataPage() == pageNumber; + + /// True when no live row is left on the page: every slot a zero-length deleted tombstone. Slot + /// offsets are non-increasing, so a slot is zero-length exactly when it repeats the previous one's offset + /// (the page size standing in for the first). + private static bool IsEmptied(JetFormatBase format, byte[] page) { - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2)); - int Offset(int i) => BinaryPrimitives.ReadUInt16LittleEndian( - page.AsSpan(format.DataRowDirectoryOffset + i * 2, 2)) & RowPointer.OffsetMask; - - // Slot offsets are non-increasing with slot index, so row i occupies [offset(i), offset(i-1)) and - // every row stored below this one is simply a LATER slot. Working by slot index rather than by - // comparing offsets is what keeps zero-length tombstones correct: one sitting at exactly this row's - // offset has to move up with the rows after it, and an offset comparison leaves it behind — where it - // then absorbs this row's length and starves the next live row down to zero. - int start = Offset(row); - int end = row == 0 ? format.PageSize : Offset(row - 1); - int length = end - start; - - int lowest = rowCount > 0 ? Offset(rowCount - 1) : format.PageSize; - if (length > 0 && start > lowest) - page.AsSpan(lowest, start - lowest).CopyTo(page.AsSpan(lowest + length)); - - for (int i = row + 1; i < rowCount; i++) + int rowCount = DataPage.ReadRowCount(page, format); + int previous = format.PageSize; + for (int i = 0; i < rowCount; i++) { - int at = format.DataRowDirectoryOffset + i * 2; - ushort entry = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(at, 2)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(at, 2), - (ushort)((entry & ~RowPointer.OffsetMask) | ((entry & RowPointer.OffsetMask) + length))); + (int offset, RowSlotFlags flags) = DataPage.ReadSlot(page, format, i); + if (!flags.HasFlag(RowSlotFlags.Deleted) || offset != previous) return false; + previous = offset; } - - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + row * 2, 2), - (ushort)(end | RowPointer.DeletedFlag | RowPointer.OverflowFlag)); - - int free = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), (ushort)(free + length)); + return true; } /// The full inline bytes of the row at (following an overflow pointer). @@ -460,26 +508,26 @@ public void RewriteRowRaw(RowId id, byte[] record) // The third entry point that writes a row, and the one most likely to cross the cap: its caller is // the in-place ALTER re-lay, which by design makes rows LONGER — the retyped column is appended and // its old slot left as dead space. The record arrives pre-built (BuildRelaidRecord → AssembleRow), - // so it never passes through RowEncoder.Encode or either guarded entry point. + // so it never passes through RowCodec.Encode or either guarded entry point. EnsureRecordFits(format, record); byte[] srcPage = _channel.ReadPageShared(id.Page).Span.ToArray(); - int raw = BinaryPrimitives.ReadUInt16LittleEndian(srcPage.AsSpan(format.DataRowDirectoryOffset + id.Row * 2, 2)); - if ((raw & RowPointer.OverflowFlag) != 0) + if (DataPage.ReadSlot(srcPage, format, id.Row).Flags.HasFlag(RowSlotFlags.Overflow)) { DataPage.TryReadRow(new PageBuffer(srcPage, id.Page), format, id.Row, - out RowSlot sourceSlot, out ReadOnlySpan sourceBytes); - RelocatedRow target = RowRelocationReader.Resolve( + out DataPage.RowSlot sourceSlot, out ReadOnlySpan sourceBytes); + DataPage.RelocatedRow target = DataPage.ResolveRelocation( _channel, _table.DefinitionPage, sourceSlot, sourceBytes); - if (!TryRewriteRowInPlace(target.Buffer.PageNumber, target.RowNumber, record)) - throw new NotSupportedException("Re-relocating an already-relocated row that grew again is not supported yet."); + if (TryRewriteRowInPlace(target.Buffer.PageNumber, target.RowNumber, record)) return; + + // Outgrown its new page as well, so it moves again — the old hidden row reclaimed while the + // pointer still names it, then the pointer re-aimed. See Update, which does the same. + ReclaimRelocationTarget(format, target); + Relocate(format, id, record); return; } if (TryRewriteRowInPlace(id.Page, id.Row, record)) return; - (int targetPage, int targetRow) = WriteHiddenRow(format, record); - var pointerBytes = new byte[4]; - BinaryPrimitives.WriteInt32LittleEndian(pointerBytes, (targetPage << 8) | targetRow); - TryRewriteRowInPlace(id.Page, id.Row, pointerBytes, addFlags: RowPointer.OverflowFlag); + Relocate(format, id, record); } /// The full inline bytes of the row at , following an overflow-forward @@ -488,11 +536,11 @@ private byte[] ReadRowBytes(RowId id) { JetFormatBase format = _channel.Format; PageBuffer page = _channel.ReadPageShared(id.Page); - if (!DataPage.TryReadRow(page, format, id.Row, out RowSlot slot, out ReadOnlySpan bytes)) + if (!DataPage.TryReadRow(page, format, id.Row, out DataPage.RowSlot slot, out ReadOnlySpan bytes)) throw new InvalidDataException($"Row {id.Page}:{id.Row} does not exist."); if (!slot.HasOverflow) return bytes.ToArray(); - return RowRelocationReader.Resolve(_channel, _table.DefinitionPage, slot, bytes).Bytes.ToArray(); + return DataPage.ResolveRelocation(_channel, _table.DefinitionPage, slot, bytes).Bytes.ToArray(); } /// @@ -508,38 +556,30 @@ private byte[] ReadRowBytes(RowId id) /// private void FreeLongValue(ColumnDef column, byte[] descriptor, bool releaseAtClose) { - // The descriptor comes off the row's variable chunk, so its width is whatever the offset table said. - // LongValueReader requires the full 12 bytes before reading any field; reclaiming has to agree, or a - // short chunk indexes past the end and escapes Delete/Update as IndexOutOfRangeException. - if (descriptor.Length < LongValueFormat.DescriptorSize) - throw new InvalidDataException( - $"Long-value descriptor has {descriptor.Length} bytes; expected at least {LongValueFormat.DescriptorSize}."); - - // Byte 3 carries the length's top byte AND the flags, so it must be masked before comparison — the - // length runs to 0x3FFFFFFF, and a 16 MB value puts 0x01 there. Masked exactly as LongValueReader does. - byte flags = (byte)(descriptor[3] & LongValueFormat.FlagMask); - if (flags is not (LongValueFormat.FlagInline or LongValueFormat.FlagSinglePage or LongValueFormat.FlagChained)) - throw new InvalidDataException($"Long-value descriptor has unknown flags 0x{flags:X2}."); - // Inline (0x80) keeps its payload in the row, so there is nothing to give back. - if (flags == LongValueFormat.FlagInline) return; - - TableDefinitionPage definition = ReadDefinition(); + // The descriptor comes off the row's variable chunk, so its width is whatever the offset table said, and + // it is read — and checked — exactly as LongValueStore reads it: a short chunk would otherwise index past + // the end and escape Delete/Update as IndexOutOfRangeException. + (_, LongValueStore.StorageKind storage, int valueRow, int valuePage, _) = LongValueStore.Read(descriptor, _channel.Format); + // An inline value keeps its payload in the row, so there is nothing to give back. + if (storage == LongValueStore.StorageKind.Inline) return; + + TableDefinition definition = LongValueMaps; definition.LongValueOwnedMaps.TryGetValue(column.ColumnId, out (int Row, int Page) owned); definition.LongValueFreeMaps.TryGetValue(column.ColumnId, out (int Row, int Page) free); - // A single-page (0x40) value shares its page with other values, so the page goes back only once the - // last of them is gone — until then just its own row is retired. - if (flags == LongValueFormat.FlagSinglePage) + // A single-page value shares its page with other values, so the page goes back only once the last of + // them is gone — until then just its own row is retired. + if (storage == LongValueStore.StorageKind.SinglePage) { - ReleasePackedValue(column, descriptor[5] | (descriptor[6] << 8) | (descriptor[7] << 16), - descriptor[4], owned, free, releaseAtClose); + ReleasePackedValue(column, valuePage, valueRow, owned, free, releaseAtClose); return; } - var allocator = new PageAllocator(_channel); - var reader = new LongValueReader(_channel); + var allocator = _channel.Allocator; + var reader = new LongValueStore(_channel); _ = reader.ResolveWithPages(descriptor, out IReadOnlyList pages); - HashSet ownedPages = MapPages(owned.Row, owned.Page).ToHashSet(); - _ = MapPages(free.Row, free.Page); // validate both mutation targets before the first free + var maps = new UsageMap(_channel, _table); + HashSet ownedPages = maps.PagesInMap(owned.Row, owned.Page).ToHashSet(); + if (free.Page != 0) _ = maps.PagesInMap(free.Row, free.Page); // validate both mutation targets before the first free foreach (int page in pages) if (!ownedPages.Contains(page)) throw new InvalidDataException( @@ -553,7 +593,7 @@ private void FreeLongValue(ColumnDef column, byte[] descriptor, bool releaseAtCl if (releaseAtClose) allocator.Release(page); else allocator.Free(page); _usageMaps.SetBit(owned.Row, owned.Page, page, set: false); - _usageMaps.SetBit(free.Row, free.Page, page, set: false); + if (free.Page != 0) _usageMaps.SetBit(free.Row, free.Page, page, set: false); } } @@ -561,7 +601,7 @@ private void FreeLongValue(ColumnDef column, byte[] descriptor, bool releaseAtCl /// Retires one value from a shared (single-page form) long-value page: its row becomes a 0-length /// deleted+overflow tombstone and the page is re-laid, the surviving records packing from the page end /// in slot order so the freed space is reclaimed. When nothing live is left the page is given back — its - /// type byte set to , its bit cleared from the column's owned + /// type set to , its bit cleared from the column's owned /// and free maps, and the page returned to the global allocator. /// /// @@ -571,10 +611,10 @@ private void FreeLongValue(ColumnDef column, byte[] descriptor, bool releaseAtCl /// 4 gone 0x01 n=5 free=3272 [4096DO,4096DO,4096DO,4096DO,3296] the survivor slid to the top /// all 0x09 n=5 free=4072 [4096DO x5] /// - /// This is where page type 0x09 comes from — an emptied packed long-value page, which the spec had - /// recorded as a released page of unidentified origin (page-09). Chained values are unaffected: they own - /// their pages outright and are freed below, leaving them at 0x01, which is why no experiment that - /// used a memo large enough to chain ever produced one. + /// This is one of the two routes to page type 0x09; the other is an ordinary data page emptied by + /// DELETE (see ). Chained values take neither: they own their pages outright and are + /// freed below, leaving them at 0x01, which is why no experiment that used a memo large enough to + /// chain ever produced one. /// private void ReleasePackedValue(ColumnDef column, int pageNumber, int row, (int Row, int Page) owned, (int Row, int Page) free, bool releaseAtClose) @@ -593,60 +633,64 @@ private void ReleasePackedValue(ColumnDef column, int pageNumber, int row, $"Long-value row pointer {pageNumber}:{row} is outside the page's 0..{holder.RowCount - 1} range."); if (holder.Rows[row].IsDeleted) return; // already retired; freeing twice must not double-count - int dir = format.DataRowDirectoryOffset; + // Every dead row — this one, and any already retired — becomes a 0-length deleted + overflow tombstone, as + // ACE writes it. var records = new byte[holder.RowCount][]; - var flags = new ushort[holder.RowCount]; + var flags = new RowSlotFlags[holder.RowCount]; for (int i = 0; i < holder.RowCount; i++) { bool dead = i == row || holder.Rows[i].IsDeleted; records[i] = dead ? [] : page.AsSpan(holder.Rows[i].Offset, holder.Rows[i].Length).ToArray(); - flags[i] = (ushort)((dead ? RowPointer.DeletedFlag : 0) - | (dead || holder.Rows[i].HasOverflow ? RowPointer.OverflowFlag : 0)); - } - - int offset = format.PageSize; - for (int i = 0; i < holder.RowCount; i++) - { - offset -= records[i].Length; // a 0-length tombstone lands on the page end, as ACE writes it - records[i].CopyTo(page.AsSpan(offset)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(dir + i * 2, 2), - (ushort)(flags[i] | (offset & RowPointer.OffsetMask))); + flags[i] = dead ? RowSlotFlags.Deleted | RowSlotFlags.Overflow : DataPage.Flags(holder.Rows[i]); } - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(offset - (dir + holder.RowCount * 2))); + DataPage.LayRows(page, format, records, flags); // shrinking, so it always fits bool emptied = records.All(r => r.Length == 0); - if (emptied) page[0] = (byte)PageType.ReleasedLongValuePage; + if (emptied) DataPage.MarkReleased(page); _channel.WritePage(pageNumber, page); // A page that survives has room again, so it goes back into the column's free-pages map — the same - // map TryAppend consults when looking for somewhere to pack the next small value. + // map TryAppend consults when looking for somewhere to pack the next small value. A column with no + // free-pages map packs nothing, so there it simply stays owned. if (!emptied) { - _usageMaps.SetBit(free.Row, free.Page, pageNumber, set: true); + if (free.Page != 0) _usageMaps.SetBit(free.Row, free.Page, pageNumber, set: true); return; } _usageMaps.SetBit(owned.Row, owned.Page, pageNumber, set: false); - _usageMaps.SetBit(free.Row, free.Page, pageNumber, set: false); - if (releaseAtClose) new PageAllocator(_channel).Release(pageNumber); - else new PageAllocator(_channel).Free(pageNumber); + if (free.Page != 0) _usageMaps.SetBit(free.Row, free.Page, pageNumber, set: false); + if (releaseAtClose) _channel.Allocator.Release(pageNumber); + else _channel.Allocator.Free(pageNumber); + } + + /// Rejects the insert if a primary-key or WITH DISALLOW NULL index would get a Null in any of its + /// columns — ACE refuses it on every write, not only when the index is built (verified). + private void EnforceNonNullKeys(object?[] values, IReadOnlySet? changedColumns = null) + { + foreach (IndexDef index in _table.Indexes.Where(i => (i.IsPrimaryKey || i.Required) && i.RootPage > 0)) + { + if (changedColumns is not null && !index.Columns.Any(c => changedColumns.Contains(c.Column.Index))) continue; + if (index.Columns.FirstOrDefault(c => values[c.Column.Index] is null or DBNull).Column is { } nullColumn) + throw new InvalidOperationException( + $"Index or primary key cannot contain a Null value: column '{nullColumn.Name}' of " + + $"'{_table.Name}' is Null, and index '{index.Name}' does not allow it."); + } } /// Rejects the insert if a UNIQUE or PRIMARY index would gain a duplicate key. A row with a /// null in any of a unique index's columns is skipped — Jet treats nulls as distinct, so a unique index /// allows multiple nulls (verified vs ACE). Runs before the row is written so nothing is half-inserted. - private void EnforceUniqueIndexes(object?[] values) + private void EnforceUniqueIndexes(object?[] values, int? excludePointer = null, IReadOnlySet? changedColumns = null) { - IndexWriter? writer = null; - foreach (IndexDef index in _table.Indexes - .Where(i => i.IsUnique && i.RootPage > 0) - .GroupBy(i => i.RootPage).Select(g => g.First())) + IndexTree? writer = null; + foreach (IndexDef index in _table.RealIndexes.Where(i => i.IsUnique)) { + if (changedColumns is not null && !index.Columns.Any(c => changedColumns.Contains(c.Column.Index))) continue; if (HasNullKey(index, values)) continue; // nulls are distinct — multiple allowed - writer ??= new IndexWriter(_channel, _table); - if (writer.KeyExists(index, values)) + writer ??= new IndexTree(_channel, _table); + if (writer.KeyExists(index, values, excludePointer)) throw new ConstraintViolationException( - $"Cannot insert into '{_table.Name}': a row with the same {(index.IsPrimaryKey ? "primary key" : "unique key")} " + + $"Cannot {(excludePointer is null ? "insert into" : "update")} '{_table.Name}': a row with the same {(index.IsPrimaryKey ? "primary key" : "unique key")} " + $"already exists (index '{index.Name}').", index.Name, index.IsPrimaryKey); @@ -657,11 +701,8 @@ private void EnforceUniqueIndexes(object?[] values) /// indexes share a real index's data) so indexed lookups — and Access — find it. private void UpdateIndexes(object?[] values, RowId rowId) { - var writer = new IndexWriter(_channel, _table); - foreach (IndexDef index in _table.Indexes - .Where(i => i.RootPage > 0) - .GroupBy(i => i.RootPage) - .Select(g => g.First())) + var writer = new IndexTree(_channel, _table); + foreach (IndexDef index in _table.RealIndexes) { // WITH IGNORE NULL: a row with a null in any indexed column is not added to this index. if (index.IgnoreNulls && HasNullKey(index, values)) continue; @@ -678,13 +719,11 @@ private void UpdateIndexes(object?[] values, RowId rowId) private HashSet IndexesGainingANewKey(object?[] values) { var gaining = new HashSet(); - IndexWriter? writer = null; - foreach (IndexDef index in _table.Indexes - .Where(i => !i.IsUnique && i.RootPage > 0) - .GroupBy(i => i.RootPage).Select(g => g.First())) + IndexTree? writer = null; + foreach (IndexDef index in _table.RealIndexes.Where(i => !i.IsUnique)) { if (index.IgnoreNulls && HasNullKey(index, values)) continue; - writer ??= new IndexWriter(_channel, _table); + writer ??= new IndexTree(_channel, _table); if (!writer.KeyExists(index, values)) gaining.Add(index.RootPage); } return gaining; @@ -709,8 +748,8 @@ private static void EnsureRecordFits(JetFormatBase format, byte[] record) { if (record.Length > format.MaxRecordSize) throw new InvalidOperationException( - $"Record is too large: {record.Length} bytes, and Jet/ACE stores at most {format.MaxRecordSize} " - + "excluding long values. Move the large columns to Memo/OLE, which live on their own pages."); + $"Record is too large: {record.Length} bytes, and a record holds at most {format.MaxRecordSize} " + + "bytes excluding long values. Move the large columns to Memo/OLE, which live on their own pages."); } private (int PageNumber, byte[] Page) FindPageWithRoom(JetFormatBase format, int needed) @@ -718,12 +757,12 @@ private static void EnsureRecordFits(JetFormatBase format, byte[] record) foreach (int pageNumber in new UsageMap(_channel, _table).FreeDataPages()) { byte[] page = _channel.ReadPageShared(pageNumber).Span.ToArray(); - int freeSpace = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2)); + int freeSpace = DataPage.ReadFreeSpace(page, format); // Space is not the only limit: an index addresses a row by a one-byte slot number, so a page that - // already holds RowPointer.MaxRowsPerPage rows has no addressable slot left however much room it - // has. Narrow rows hit this long before they fill the page. - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2)); - if (freeSpace >= needed && rowCount < RowPointer.MaxRowsPerPage) + // already holds MaxRowsPerPage rows has no addressable slot left however much room it has. Narrow rows + // hit this long before they fill the page. + int rowCount = DataPage.ReadRowCount(page, format); + if (freeSpace >= needed && rowCount < format.MaxRowsPerPage) return (pageNumber, page); } @@ -745,15 +784,9 @@ private static void EnsureRecordFits(JetFormatBase format, byte[] record) // the last of six equally-full pages stays in the free map.) int previousTail = new UsageMap(_channel, _table).MaxDataPage(); - int pageNumber = new PageAllocator(_channel).Allocate(); + int pageNumber = _channel.Allocator.Allocate(); - var page = new byte[format.PageSize]; - page[0] = (byte)PageType.DataPage; - page[1] = 0x01; // page flags (observed constant) - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(format.DataOwnerOffset, 4), _table.DefinitionPage); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), 0); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(format.PageSize - format.DataRowDirectoryOffset)); + byte[] page = DataPage.NewPage(format, (uint)_table.DefinitionPage); _channel.WritePage(pageNumber, page); if (previousTail >= 0) @@ -768,14 +801,17 @@ private static void EnsureRecordFits(JetFormatBase format, byte[] record) /// referenced by the TDEF pointer at (row byte + 3-byte page). private void UpdateUsageBit(int tdefPointerOffset, int targetPage, bool set) { - PageBuffer tdef = _channel.ReadPageShared(_table.DefinitionPage); - _usageMaps.SetBit(tdef.ReadByte(tdefPointerOffset), tdef.ReadInt24(tdefPointerOffset + 1), targetPage, set, + (int row, int page) = _channel.ReadPageShared(_table.DefinitionPage).ReadRecordPointer(tdefPointerOffset); + _usageMaps.SetBit(row, page, targetPage, set, movableWindow: tdefPointerOffset == _channel.Format.TdefFreePagesOffset); } - /// Pins the fixed-region length to an existing row anywhere in the table (so the layout - /// matches Access), or returns null for an empty table (the encoder then derives it). + /// Pins the fixed-region length to an existing row (so the layout matches Access), or returns + /// null for the encoder to derive it from the columns. + /// The region may never shrink below what existing rows carry (page-02a §3.1), and a retired + /// column id — dropped, or burned by a retype — leaves its bytes as a hole nothing points at, so the live + /// descriptors can under-count where the region ends. Only then is a row worth reading. private int? InferFixedDataLength(JetFormatBase format) { // The current fixed-region end from the column descriptors — this includes a just-added fixed column, @@ -788,24 +824,16 @@ private void UpdateUsageBit(int tdefPointerOffset, int targetPage, bool set) if (_table.Columns.All(c => c.IsFixedLength)) return null; + // 0x29 counts ids ever handed out and never decrements, so equality means none was retired. + if (_table.Columns.Count == _table.ColumnIdHighWater) return null; + foreach (int pageNumber in new UsageMap(_channel, _table).DataPages()) { byte[] page = _channel.ReadPageShared(pageNumber).Span.ToArray(); if (InferFixedDataLength(page, format) is { } pinned) return Math.Max(pinned, derived); // ADD COLUMN of a fixed column grows the region past old rows } - return null; // empty table → RowEncoder derives the same length from the columns - } - - private static int LowestRowOffset(byte[] page, JetFormatBase format, int rowCount) - { - int lowest = format.PageSize; - for (int i = 0; i < rowCount; i++) - { - int raw = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + i * 2, 2)); - lowest = Math.Min(lowest, raw & RowPointer.OffsetMask); - } - return lowest; + return null; // empty table → RowCodec derives the same length from the columns } /// @@ -815,7 +843,7 @@ private static int LowestRowOffset(byte[] page, JetFormatBase format, int rowCou /// private int? InferFixedDataLength(byte[] page, JetFormatBase format) { - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2)); + int rowCount = DataPage.ReadRowCount(page, format); // Take the largest sane fixed length across the page's inline rows rather than trusting the first one: // some rows decode to a nonsensical (e.g. zero) variable-data offset, which would yield a negative @@ -824,17 +852,26 @@ private static int LowestRowOffset(byte[] page, JetFormatBase format, int rowCou int prevEnd = format.PageSize; for (int i = 0; i < rowCount; i++) { - int raw = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + i * 2, 2)); - int offset = raw & RowPointer.OffsetMask; + (int offset, RowSlotFlags flags) = DataPage.ReadSlot(page, format, i); int length = prevEnd - offset; prevEnd = offset; - if ((raw & (RowPointer.DeletedFlag | RowPointer.OverflowFlag)) != 0) continue; - if (length < format.RowColumnCountSize + 2) continue; + if ((flags & (RowSlotFlags.Deleted | RowSlotFlags.Overflow)) != RowSlotFlags.None) continue; + if (length < format.RowColumnCountSize) continue; ReadOnlySpan row = page.AsSpan(offset, length); - bool rowHasVar = RowLayout.HasVariableSection(row, _table.Columns); - RowLayout layout = RowLayout.Parse(row, format.RowColumnCountSize, rowHasVar); + bool rowHasVar = RowCodec.Layout.HasVariableSection(row, _table.Columns, format); + + // A row written while a since-dropped variable column existed still carries its trailer, but the + // live columns no longer say so, and reading it as all-fixed measures the trailer as fixed data — + // which would then pad every new row out to it, up to a false "Record is too large". The table's + // 0x2B high-water is the tell: it outlives the drop, so a row that reads as all-fixed in a table + // that has ever had a variable column cannot be trusted to pin the length. The schema does it. + if (!rowHasVar && _table.VariableColumnCount > 0) continue; + // Too short for even an empty row of its own column count: not an inline record, so it cannot pin + // the length. + if (length < RowCodec.Layout.MinimumLength(format, RowCodec.Layout.ReadColumnCount(row, format), rowHasVar)) continue; + RowCodec.Layout layout = RowCodec.Layout.Parse(row, format, rowHasVar); int varDataStart = layout.FixedRegionLength + format.RowColumnCountSize; // The variable-data start is the end of the fixed region; it must lie after the column-count field @@ -854,43 +891,88 @@ private static int LowestRowOffset(byte[] page, JetFormatBase format, int rowCou /// whose value exceeds the inline limit, write it to an LVAL page and substitute the 12-byte reference /// descriptor (short values, and pre-built descriptors from other callers, are left as-is to inline). /// + /// + /// The row as it is stored: a copy of the logical values in which every memo/OLE value too large to + /// inline has become the descriptor naming its pages. + /// + /// + /// The copy is the point. A row has two representations — the values a caller holds, and the descriptors + /// the record carries — and they are indistinguishable at the type level, since both live in an + /// object?[]. Materialising in place made the caller's array silently change meaning halfway + /// through a write, and everything that then read it got the wrong one: a delete keyed an index off a + /// byte[] descriptor, an update moved an index entry using a descriptor as a key, a caller was left + /// holding pages that had been freed. Two arrays makes the distinction one the code can state. + /// + private object?[] ToStorageValues(object?[] values) + { + object?[] storage = (object?[])values.Clone(); + MaterializeLongValues(storage); + return storage; + } + + /// private void MaterializeLongValues(object?[] values) { - const int maxInline = LongValueFormat.MaxInlineValue; - LongValueWriter? writer = null; - TableDefinitionPage? definition = null; + // ACE writes a row's chained values first and its single-page values after, each in column order + // (measured), so a single-page value in an earlier column takes the page after the chains, not before. + LongValueStore? writer = null; + int maxSinglePage = _channel.Format.LongValueMaxSinglePage; + foreach (bool chained in new[] { true, false }) + foreach (ColumnDef column in _table.Columns) + if (column.Type is JetDataType.Memo or JetDataType.Ole && Chains(values[column.Index]) == chained) + values[column.Index] = MaterializeLongValue(column, values[column.Index], ref writer); + + // The storage form follows the uncompressed length, as MaterializeLongValue decides it. + bool Chains(object? value) => value switch + { + string s => Encoding.Unicode.GetByteCount(s) > maxSinglePage, + byte[] b => b.Length > maxSinglePage, + _ => false, + }; + } + + /// The storage form of one Memo/OLE value: a value over 64 bytes is written to an LVAL page and + /// becomes its ; anything else — a short value, null, a descriptor already + /// built — comes back as it was, for the row codec to inline. For a caller re-laying a single column. + internal object? MaterializeLongValue(ColumnDef column, object? value) + { + LongValueStore? writer = null; + return MaterializeLongValue(column, value, ref writer); + } - foreach (ColumnDef column in _table.Columns) + private object? MaterializeLongValue(ColumnDef column, object? value, ref LongValueStore? writer) + { + byte[]? payload = value switch { - if (column.Type is not (JetDataType.Memo or JetDataType.Ole)) continue; - byte[]? payload = values[column.Index] switch - { - string s => Encoding.Unicode.GetBytes(s), // memo: UTF-16LE - byte[] b => b, // OLE: raw bytes - _ => null, // null, or an already-built LongValueDescriptor - }; - if (payload is null || payload.Length <= maxInline) continue; - - // Compress before storing, but AFTER the inline test above and using the uncompressed length for - // the single-page test below: ACE decides the storage form on the uncompressed size and applies - // compression to whatever form results, never to a chained value (LongTextStorageAccessTests). - if (payload.Length <= LongValueFormat.MaxSinglePageValue - && values[column.Index] is string text - && Types.JetTypeCodec.TryCompressText(column, text) is { } compressed) - payload = compressed; - - writer ??= new LongValueWriter(_channel); - definition ??= ReadDefinition(); - definition.LongValueOwnedMaps.TryGetValue(column.ColumnId, out (int Row, int Page) owned); - definition.LongValueFreeMaps.TryGetValue(column.ColumnId, out (int Row, int Page) free); - values[column.Index] = new LongValueDescriptor(StoreLongValue(writer, payload, owned, free)); - } + string s => Encoding.Unicode.GetBytes(s), // memo: UTF-16LE + byte[] b => b, // OLE: raw bytes + _ => null, // null, or an already-built LongValueStore.DescriptorValue + }; + JetFormatBase format = _channel.Format; + if (payload is null || payload.Length <= format.LongValueMaxInline) return value; + + // Compress before storing, but AFTER the inline test above and using the uncompressed length for + // the single-page test below: ACE decides the storage form on the uncompressed size and applies + // compression to whatever form results, never to a chained value (LongTextStorageAccessTests). + // ACE places a compressed value by its uncompressed bytes and leaves them behind (LongValueStore.TryAppend). + byte[]? uncompressed = null; + if (payload.Length <= format.LongValueMaxSinglePage + && value is string text + && Types.JetTypeCodec.TryCompressText(column, text) is { } compressed) + (uncompressed, payload) = (payload, compressed); + + writer ??= new LongValueStore(_channel); + TableDefinition definition = LongValueMaps; + definition.LongValueOwnedMaps.TryGetValue(column.ColumnId, out (int Row, int Page) owned); + definition.LongValueFreeMaps.TryGetValue(column.ColumnId, out (int Row, int Page) free); + return new LongValueStore.DescriptorValue(StoreLongValue(writer, payload, owned, free, uncompressed)); } - // A page is dropped from the free-pages map once it cannot hold the smallest long value (a 65-byte - // payload — anything up to 64 inlines — plus its 2-byte row-directory entry). The largest such row is - // 4076 bytes on a Jet 4 page (Jackcess MAX_LONG_VALUE_ROW_SIZE), which nothing here needs to name. - private const int MinLvalRow = 65 + 2; + // A page is dropped from the free-pages map once it cannot hold a 256-byte value and its row-directory entry + // (verified vs ACE, memo and OLE alike: a page left with 257 bytes free leaves the map, one with 258 stays). + // Not the smallest long value a page could still take — anything over the inline limit is one — but ACE's + // own cut-off, below which it stops offering the page. + private const int MinLvalValue = 256; /// Rejects a caller-supplied value for a calculated column, as ACE does — it refuses both an /// INSERT naming one and an UPDATE setting one, with "Cannot update 'x'; field not updateable." @@ -915,22 +997,26 @@ private void RejectExplicitCalculatedValues(object?[] values, IReadOnlySet? } /// Writes an oversized calculated result to an LVAL page and returns its in-row descriptor. - /// Handed to , which derives the value and so is the only place that knows how + /// Handed to , which derives the value and so is the only place that knows how /// big it turned out; the pages and their usage maps stay this class's business. private byte[] SpillCalculated(ColumnDef column, byte[] envelope) - => StorePackedLongValue(column.ColumnId, envelope); - - /// Stores on an LVAL page for long-value column - /// — packing onto a free page as usual — and returns the in-row descriptor. - /// For a caller that must use a page regardless of size (the MSysObjects LvProp property blob, - /// which Access reads only from a page, never inline). Call before - /// so the row carries the returned descriptor as a . - public byte[] StorePackedLongValue(int columnId, byte[] payload) + => StorePackedLongValue(column.ColumnId, envelope).Bytes; + + /// Stores for long-value column and returns + /// the in-row descriptor: inline up to 64 bytes, as any long value is, else on an LVAL page — packing onto a + /// free page as usual. The MSysObjects LvProp property blob takes the same rule (verified: ACE inlines + /// a 63-byte blob and puts a 65-byte one on a page). Call before and put + /// the descriptor in the row's slot for the column. + public LongValueStore.DescriptorValue StorePackedLongValue(int columnId, byte[] payload) { - TableDefinitionPage definition = ReadDefinition(); + JetFormatBase format = _channel.Format; + if (payload.Length <= format.LongValueMaxInline) + return new LongValueStore.DescriptorValue(Types.JetTypeCodec.EncodeInlineLongValue(payload, format)); + + TableDefinition definition = ReadDefinition(); definition.LongValueOwnedMaps.TryGetValue(columnId, out (int Row, int Page) owned); definition.LongValueFreeMaps.TryGetValue(columnId, out (int Row, int Page) free); - return StoreLongValue(new LongValueWriter(_channel), payload, owned, free); + return new LongValueStore.DescriptorValue(StoreLongValue(new LongValueStore(_channel), payload, owned, free)); } /// @@ -939,19 +1025,26 @@ public byte[] StorePackedLongValue(int columnId, byte[] payload) /// free-pages map with room), the way Access shares a page across many small values; only when none has /// room is a fresh page allocated (owned + free). A value larger than one page is chained across /// dedicated pages. + /// A column with an owned-pages map and no free-pages map (a free pointer to page 0) never shares + /// a page: each value takes a fresh one, recorded in the owned map alone. Access writes MSysNameMap.NameMap + /// and MSysAccessXML.LValue that way, and packs nothing into them — measured across every such file, + /// one LVAL page per stored value. /// - private byte[] StoreLongValue(LongValueWriter writer, byte[] payload, (int Row, int Page) owned, (int Row, int Page) free) + private byte[] StoreLongValue(LongValueStore writer, byte[] payload, (int Row, int Page) owned, (int Row, int Page) free, + byte[]? uncompressed = null) { - if (owned.Page == 0 || free.Page == 0) - throw new InvalidDataException("Long-value column has no complete owned/free usage-map pointers."); - _ = MapPages(owned.Row, owned.Page); // validate both map targets before allocating or writing LVAL pages - IReadOnlyList freePages = MapPages(free.Row, free.Page); + if (owned.Page == 0) + throw new InvalidDataException("Long-value column has no owned-pages usage-map pointer."); + var maps = new UsageMap(_channel, _table); + _ = maps.PagesInMap(owned.Row, owned.Page); // validate both map targets before allocating or writing LVAL pages + IReadOnlyList freePages = free.Page == 0 ? [] : [.. maps.PagesInMap(free.Row, free.Page)]; - if (payload.Length > LongValueFormat.MaxSinglePageValue) + JetFormatBase format = _channel.Format; + if (payload.Length > format.LongValueMaxSinglePage) { - LongValueResult chained = writer.Write(payload); + (byte[] Descriptor, IReadOnlyList OwnedPages, int FreePage) chained = writer.Write(payload); foreach (int page in chained.OwnedPages) _usageMaps.SetBit(owned.Row, owned.Page, page, set: true); - if (chained.FreePage != 0) + if (chained.FreePage != 0 && free.Page != 0) _usageMaps.SetBit(free.Row, free.Page, chained.FreePage, set: true, movableWindow: true); return chained.Descriptor; } @@ -959,85 +1052,39 @@ private byte[] StoreLongValue(LongValueWriter writer, byte[] payload, (int Row, // Pack onto the first free page that has room for the value plus its directory entry. if (free.Page != 0) foreach (int page in freePages) - if (writer.TryAppend(page, payload) is (int row, int remaining)) + if (writer.TryAppend(page, payload, uncompressed) is (int row, int remaining)) { - if (remaining < MinLvalRow) _usageMaps.SetBit(free.Row, free.Page, page, set: false); // now full - return LongValueWriter.SinglePageDescriptor(payload.Length, page, row); + if (remaining < MinLvalValue + format.DataRowDirectoryEntrySize) + _usageMaps.SetBit(free.Row, free.Page, page, set: false); // now full + return LongValueStore.Descriptor(format, payload.Length, LongValueStore.StorageKind.SinglePage, page, row); } // No free page had room: a fresh page (owned, and free — it still has spare room). - int newPage = writer.WriteNewPage(payload); + int newPage = writer.WriteNewPage(payload, uncompressed); _usageMaps.SetBit(owned.Row, owned.Page, newPage, set: true); - _usageMaps.SetBit(free.Row, free.Page, newPage, set: true, movableWindow: true); - return LongValueWriter.SinglePageDescriptor(payload.Length, newPage, 0); - } - - /// Reads every page marked in a validated inline or reference usage map. - private List MapPages(int mapRow, int mapPage) - { - if (mapPage <= 1 || mapPage >= _channel.PageCount) - throw new InvalidDataException($"Long-value usage-map page {mapPage} is outside the physical file."); - var holder = new DataPage(); - holder.Read(_channel.ReadPageShared(mapPage), _channel.Format); - if (holder.IsLongValuePage || holder.OwningTablePage != 0) - throw new InvalidDataException( - $"Long-value usage-map pointer {mapPage}:{mapRow} does not target an owner-zero usage-map data page."); - if (mapRow < 0 || mapRow >= holder.RowCount) - throw new InvalidDataException($"Long-value usage-map row {mapPage}:{mapRow} does not exist."); - RowSlot slot = holder.Rows[mapRow]; - if (slot.IsDeleted || slot.HasOverflow || slot.Length == 0) - throw new InvalidDataException($"Long-value usage-map row {mapPage}:{mapRow} is deleted, overflowed, or empty."); - ReadOnlySpan map = holder.GetRow(mapRow); - var result = new List(); - - if (map[0] == 0x00) - { - if (map.Length < 5) - throw new InvalidDataException("Inline long-value usage map is shorter than its 5-byte header."); - int startPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1, 4)); - AppendMapBits(result, map[5..], startPage); - return result; - } - - if (map[0] != 0x01 || map.Length != 69) - throw new InvalidDataException( - $"Long-value usage map has invalid type/length 0x{map[0]:X2}/{map.Length}."); - int pagesPerBitmap = (_channel.PageSize - 4) * 8; - var bitmapPages = new HashSet(); - for (int i = 0; i < 17; i++) - { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1 + i * 4, 4)); - if (bitmapPage == 0) continue; - if (bitmapPage <= 1 || bitmapPage >= _channel.PageCount || !bitmapPages.Add(bitmapPage)) - throw new InvalidDataException($"Long-value usage map has invalid bitmap-page pointer {bitmapPage}."); - ReadOnlySpan bitmap = _channel.ReadPageShared(bitmapPage).Span; - if (bitmap[0] != (byte)PageType.PageUsageBitmap || bitmap[1] != 0x01 || bitmap[2] != 0 || bitmap[3] != 0) - throw new InvalidDataException($"Long-value usage-map pointer {bitmapPage} is not a bitmap page."); - AppendMapBits(result, bitmap[4..], i * pagesPerBitmap); - } - return result; + if (free.Page != 0) + _usageMaps.SetBit(free.Row, free.Page, newPage, set: true, movableWindow: true); + return LongValueStore.Descriptor(format, payload.Length, LongValueStore.StorageKind.SinglePage, newPage); } - private void AppendMapBits(List result, ReadOnlySpan bitmap, int startPage) - { - // Bound read once, as UsageMap.AppendSetBits does: the loop runs per bit on a hot path. Nothing here writes. - int pageCount = _channel.PageCount; + /// + /// The table's definition, read once and kept — for the long-value map pointers only, which are what + /// every caller of this wants and which no row write can move. + /// + /// + /// Freeing or storing a long value writes the map holder page and the global map; + /// never writes the definition page, and an inline-to-reference conversion + /// rewrites the record inside the same map row. So the §3.3.2 pointers are fixed for the life of this + /// inserter, where the row count and AutoNumber high-water on the same page are not — read those from the + /// page, never from here. + /// + private TableDefinition LongValueMaps => _longValueMaps ??= ReadDefinition(); - for (int i = 0; i < bitmap.Length; i++) - for (int bit = 0; bit < 8; bit++) - if ((bitmap[i] & (1 << bit)) != 0) - { - int page = startPage + i * 8 + bit; - if (page <= 1 || page >= pageCount) - throw new InvalidDataException( - $"Long-value usage map names page {page}, outside the physical reusable range."); - result.Add(page); - } - } + private TableDefinition? _longValueMaps; - private TableDefinitionPage ReadDefinition() + private TableDefinition ReadDefinition() { - var definition = new TableDefinitionPage(); + var definition = new TableDefinition(); definition.Read(_channel, _table.DefinitionPage); return definition; } @@ -1059,7 +1106,7 @@ private TableDefinitionPage ReadDefinition() if (!needed) return null; ReadOnlySpan tdef = _channel.ReadPageShared(_table.DefinitionPage).Span; - int highWater = BinaryPrimitives.ReadInt32LittleEndian(tdef.Slice(format.TdefLastAutoNumberOffset, 4)); + int highWater = TableDefinition.ReadLastAutoNumber(tdef, format); // A complex (multi-value / attachment) column is an AutoNumber too, but it draws on its own counter // at 0x1C and there is ONE id per row shared by every complex column of the table — the counter is a @@ -1104,8 +1151,7 @@ private void AssignComplexId(JetFormatBase format, object?[] values, ReadOnlySpa .ToList(); if (complex.Count == 0) return; - int next = unchecked(BinaryPrimitives.ReadInt32LittleEndian( - tdef.Slice(format.TdefComplexAutoNumberOffset, 4)) + 1); + int next = unchecked(TableDefinition.ReadComplexAutoNumber(tdef, format) + 1); foreach (ColumnDef column in complex) values[column.Index] = next; } @@ -1132,7 +1178,7 @@ private static int RandomAutoNumber() /// (the row brings a key it does not hold yet). Access advances it on insert only, /// once per real index, and never decrements it. The sibling **total-entry count** /// (`+0`) is deliberately left untouched **on insert**: Access does not raise it there — it is written - /// when the index is built (to its entries, in TableCreator's back-fill) or the file compacted + /// when the index is built (to its entries, in SchemaEditor's back-fill) or the file compacted /// (verified: an index created on an empty table reads total `0` through any number of inserts). /// Delete is not the mirror of this — ACE drops the total by one on every real index when a row /// goes, so the figure drifts downward across insert/delete cycles in ACE as much as here. Measured, and @@ -1143,8 +1189,7 @@ private void UpdateTdefCounters(JetFormatBase format, object?[] values, bool[]? { byte[] tdef = _channel.ReadPageShared(_table.DefinitionPage).Span.ToArray(); - int count = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(format.TdefRowCountOffset, 4)); - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(format.TdefRowCountOffset, 4), count + 1); + TableDefinition.WriteRowCount(tdef, format, TableDefinition.ReadRowCount(tdef, format) + 1); // The record's complex id rides its own counter at 0x1C, and every complex column of the table holds // the one id, so the high-water moves once. Deleting a row never rolls it back (verified vs ACE — @@ -1153,11 +1198,8 @@ private void UpdateTdefCounters(JetFormatBase format, object?[] values, bool[]? && values[complexColumn.Index] is { } complexValue) { int assignedId = Convert.ToInt32(complexValue, CultureInfo.InvariantCulture); - int complexHighWater = BinaryPrimitives.ReadInt32LittleEndian( - tdef.AsSpan(format.TdefComplexAutoNumberOffset, 4)); - if (assignedId > complexHighWater) - BinaryPrimitives.WriteInt32LittleEndian( - tdef.AsSpan(format.TdefComplexAutoNumberOffset, 4), assignedId); + if (assignedId > TableDefinition.ReadComplexAutoNumber(tdef, format)) + TableDefinition.WriteComplexAutoNumber(tdef, format, assignedId); } foreach (ColumnDef column in _table.Columns) @@ -1171,7 +1213,7 @@ private void UpdateTdefCounters(JetFormatBase format, object?[] values, bool[]? // don't form a monotone sequence) and diverge from Access's on-disk state. if (column.IsRandomAutoNumber) continue; int assigned = Convert.ToInt32(value, CultureInfo.InvariantCulture); - int highWater = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(format.TdefLastAutoNumberOffset, 4)); + int highWater = TableDefinition.ReadLastAutoNumber(tdef, format); // An id this insert *generated* always becomes the new high-water: it came from 0x14 + increment, // so it is by construction the next value in the sequence. That includes the wrap past // int.MaxValue (or int.MinValue for a descending counter), where the new id compares as going @@ -1191,12 +1233,12 @@ private void UpdateTdefCounters(JetFormatBase format, object?[] values, bool[]? || (column.Increment >= 0 ? assigned > highWater : assigned < highWater); int newHighWater = advances ? assigned : highWater; if (advances) - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(format.TdefLastAutoNumberOffset, 4), assigned); + TableDefinition.WriteLastAutoNumber(tdef, format, assigned); // Keep the cached catalog seed in sync with the on-disk 0x14. The insert path itself reads the - // high-water from disk (AssignAutoNumbers), so this isn't needed for assigning ids — but RewriteColumn - // (an ALTER on a table that has an AutoNumber) reconstructs the counter from the cached - // ColumnDef.Seed; if left stale, a rebuild after the high rows were deleted resets the counter to its - // create-time value (verified: next id dropped to 1 instead of continuing past 6). Seed = next id. + // high-water from disk (AssignAutoNumbers), so this isn't needed for assigning ids — but anything that + // reconstructs the counter from the cached ColumnDef.Seed would, left stale, reset it to its + // create-time value after the high rows were deleted (seen once: next id dropped to 1 instead of + // continuing past 6). Seed = next id. column.Seed = unchecked(newHighWater + column.Increment); } @@ -1205,9 +1247,8 @@ private void UpdateTdefCounters(JetFormatBase format, object?[] values, bool[]? { if (!index.IsUnique && !newKeys.Contains(index.RootPage)) continue; if (index.IgnoreNulls && HasNullKey(index, values)) continue; // row was excluded from the index - int statsUnique = format.TdefRealIndexBlockOffset + index.RealIndexOrdinal * format.RealIndexEntrySize + 4; - int unique = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(statsUnique, 4)); - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(statsUnique, 4), unique + 1); + (int total, int unique) = TableDefinition.ReadIndexCounts(tdef, format, index.RealIndexOrdinal); + TableDefinition.WriteIndexCounts(tdef, format, index.RealIndexOrdinal, total, unique + 1); } _channel.WritePage(_table.DefinitionPage, tdef); diff --git a/src/LibRed/LibRed.Core/Storage/RowLayout.cs b/src/LibRed/LibRed.Core/Storage/RowLayout.cs deleted file mode 100644 index 419855eaf..000000000 --- a/src/LibRed/LibRed.Core/Storage/RowLayout.cs +++ /dev/null @@ -1,112 +0,0 @@ -using LibRed.Catalog; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -/// -/// Parses the structural trailer of an inline row record once (spec §5), so the several call sites that -/// need to locate a row's regions don't each re-derive the offset arithmetic. Layout: -/// -/// [count:2] [fixed data] [var data] [varOffsetTable:(numVar+1)x2] [numVar:2] [nullBitmap] -/// -/// The leading count is maxColumnId + 1 and drives the null-bitmap width. A table with NO variable -/// columns omits the whole variable section (offset table + numVar) — such a row can't self-describe that, -/// so the caller passes hasVar from the schema. -/// -internal readonly ref struct RowLayout -{ - private readonly ReadOnlySpan _row; - - /// The leading column count (= max column id + 1). - public int ColumnCount { get; } - /// Null-bitmap width in bytes, from the leading count. - public int NullBitmapSize { get; } - /// Number of variable columns stored (0 when the table has no variable section). - public int NumVar { get; } - /// Offset of the variable-offset table, or -1 when there is no variable section. - public int VarTableStart { get; } - /// Length of the fixed-data region (bytes between the leading count and the variable data). - public int FixedRegionLength { get; } - - private RowLayout(ReadOnlySpan row, int countSize, bool hasVar) - { - if (countSize != 2 || row.Length < countSize) - throw new InvalidDataException("Row is too short to contain its 2-byte column count."); - - _row = row; - ColumnCount = BinaryPrimitives.ReadUInt16LittleEndian(row[..countSize]); - NullBitmapSize = (ColumnCount + 7) / 8; - if (row.Length < countSize + NullBitmapSize) - throw new InvalidDataException( - $"Row is {row.Length} bytes, too short for its {NullBitmapSize}-byte null bitmap."); - - if (!hasVar) - { - NumVar = 0; - VarTableStart = -1; - FixedRegionLength = row.Length - countSize - NullBitmapSize; - return; - } - - int numVarPos = row.Length - NullBitmapSize - 2; - if (numVarPos < countSize + 2) - throw new InvalidDataException("Row is too short to contain a variable-column trailer."); - NumVar = BinaryPrimitives.ReadUInt16LittleEndian(row.Slice(numVarPos, 2)); - long tableStart = (long)numVarPos - ((long)NumVar + 1) * 2; - if (tableStart < countSize || tableStart > numVarPos) - throw new InvalidDataException( - $"Row declares {NumVar} variable slots, placing its offset table outside the row."); - VarTableStart = (int)tableStart; - - int previous = VarOffset(0); - if (previous < countSize || previous > VarTableStart) - throw new InvalidDataException( - $"Row variable-data end {previous} is outside the data region ending at {VarTableStart}."); - for (int entry = 1; entry <= NumVar; entry++) - { - int current = VarOffset(entry); - if (current < countSize || current > previous) - throw new InvalidDataException( - $"Row variable offset {entry} ({current}) is outside or above its preceding boundary {previous}."); - previous = current; - } - // The last offset-table entry is the variable-data start (= count field + fixed region). - FixedRegionLength = previous - countSize; - } - - /// Parses ; is whether the schema has any variable column. - public static RowLayout Parse(ReadOnlySpan row, int countSize, bool hasVar) => new(row, countSize, hasVar); - - /// Whether this row carries a variable section — the argument every - /// caller needs, derived once here rather than at each call site. - /// - /// It is not "does the schema have a variable column": a row written before the table's first variable - /// ADD COLUMN has no trailer even though the current schema does, because ADD COLUMN is metadata-only. - /// The row's own stored count is what dates it — a column id at or above the count did not exist when - /// the row was written. Get this wrong and the parse reads the last bytes of FIXED data as numVar and an - /// offset table, which the bounds checks above usually catch, but not always. - /// - public static bool HasVariableSection(ReadOnlySpan row, IReadOnlyList columns) - { - if (row.Length < 2) - throw new InvalidDataException("Row is too short to contain its 2-byte column count."); - int storedCount = BinaryPrimitives.ReadUInt16LittleEndian(row[..2]); - foreach (ColumnDef column in columns) - if (!column.IsFixedLength && column.ColumnId < storedCount) return true; - return false; - } - - /// The raw bytes of variable column (end-first offset table). - public ReadOnlySpan VarChunk(int variableIndex) - { - if (variableIndex < 0 || variableIndex >= NumVar) - throw new InvalidDataException( - $"Row has {NumVar} variable slots but column metadata requests slot {variableIndex}."); - int start = VarOffset(NumVar - variableIndex); - int end = VarOffset(NumVar - variableIndex - 1); - return _row[start..end]; - } - - private int VarOffset(int entry) => - BinaryPrimitives.ReadUInt16LittleEndian(_row.Slice(VarTableStart + entry * 2, 2)); -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/RowRelocationReader.cs b/src/LibRed/LibRed.Core/Storage/RowRelocationReader.cs deleted file mode 100644 index 6ba62986f..000000000 --- a/src/LibRed/LibRed.Core/Storage/RowRelocationReader.cs +++ /dev/null @@ -1,66 +0,0 @@ -using LibRed.IO; -using LibRed.Pages; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -/// A validated, page-backed relocated-row target. The bytes remain zero-copy for index seeks. -internal readonly record struct RelocatedRow(PageBuffer Buffer, RowSlot Slot, int RowNumber) -{ - public ReadOnlySpan Bytes => Buffer.Slice(Slot.Offset, Slot.Length); -} - -/// -/// Validates and follows the forward pointer at the START of a live overflow row slot. -/// -/// -/// The slot is normally exactly 4 bytes: ACE's DML and both trim it down to the -/// pointer when a row is relocated. Measured over 317 relocations with no exception, across ACE x64, the -/// ACE 2010 x86 runtime, and LibRed's own writer, under growing and shrinking text, repeated re-relocation, -/// page fragmentation by interleaved deletes, and an OLE column going from NULL to a value. -/// -/// Real files nevertheless contain longer ones. Northwind's MSysAccessStorage has live overflow slots -/// of 45-63 bytes, and their contents are the row as it was BEFORE it moved, with only the leading 4 bytes -/// replaced by the pointer: every field lands where the row format puts it once those 4 bytes are discounted, -/// the keys match the row it forwards to, and the remnant's null bitmap differs from its target's in exactly -/// the OLE column's bit — the value whose arrival grew the row and forced the move. The slot simply kept the -/// old row's width. -/// -/// What wrote them is NOT known: no write path reproduces the shape, including the OLE-column transition the -/// bytes themselves record. So this reads the leading pointer and ignores whatever follows, rather than -/// asserting a width. The checks that matter are unchanged and do the real work — the target must be in the -/// file, owned by the same table, and a nonempty hidden inline row. -/// -internal static class RowRelocationReader -{ - public static RelocatedRow Resolve(PageChannel channel, int owningTablePage, - RowSlot sourceSlot, ReadOnlySpan sourceBytes) - { - if (sourceSlot.IsDeleted || !sourceSlot.HasOverflow) - throw new InvalidDataException("A relocation source must be a live overflow row slot."); - if (sourceBytes.Length < 4) - throw new InvalidDataException( - $"A relocation source must begin with a 4-byte pointer; found {sourceBytes.Length} bytes."); - - int pointer = BinaryPrimitives.ReadInt32LittleEndian(sourceBytes[..4]); - int pageNumber = pointer >> 8; - int rowNumber = pointer & 0xFF; - if (pageNumber <= 0 || pageNumber >= channel.PageCount) - throw new InvalidDataException( - $"Relocation pointer targets page {pageNumber}, outside the file's 1..{channel.PageCount - 1} range."); - - PageBuffer targetBuffer = channel.ReadPageShared(pageNumber); - uint owner = targetBuffer.ReadUInt32(channel.Format.DataOwnerOffset); - if (owner != (uint)owningTablePage) - throw new InvalidDataException( - $"Relocation target page {pageNumber} belongs to TDEF {owner}, not TDEF {owningTablePage}."); - if (!DataPage.TryReadRow(targetBuffer, channel.Format, rowNumber, out RowSlot targetSlot, out _)) - throw new InvalidDataException( - $"Relocation pointer targets missing row {rowNumber} on page {pageNumber}."); - if (!targetSlot.IsDeleted || targetSlot.HasOverflow || targetSlot.Length == 0) - throw new InvalidDataException( - $"Relocation target {pageNumber}:{rowNumber} is not a nonempty hidden inline row."); - - return new RelocatedRow(targetBuffer, targetSlot, rowNumber); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/Table.cs b/src/LibRed/LibRed.Core/Storage/Table.cs index 307e38073..81ef48173 100644 --- a/src/LibRed/LibRed.Core/Storage/Table.cs +++ b/src/LibRed/LibRed.Core/Storage/Table.cs @@ -1,40 +1,112 @@ using LibRed.Catalog; using LibRed.IO; +using LibRed.Pages; namespace LibRed.Storage; /// -/// An opened table: pairs a with the means to read its rows. +/// An opened table: pairs a with the means to read its rows. /// The primary entry point for scanning data out of the storage layer. /// public sealed class Table { - public Table(PageChannel channel, TableDef definition) + internal Table(PageChannel channel, TableDefinition definition, JetCatalog catalog) { Channel = channel; + _catalog = catalog; Definition = definition; UsageMap = new UsageMap(channel, definition); } - public PageChannel Channel { get; } - public TableDef Definition { get; } - public UsageMap UsageMap { get; } + internal PageChannel Channel { get; } + private readonly JetCatalog _catalog; + public TableDefinition Definition { get; } + internal UsageMap UsageMap { get; } public string Name => Definition.Name; /// Returns a forward-only cursor over all rows in the table. - public TableCursor Rows() => new(this); + /// Which columns to decode, by , or null for all; a column + /// left out reads as null. For a reader that never looks at it — see . + public TableCursor Rows(bool[]? decode = null) => new(this, decode); + + /// A decode mask for and the seeks that reads only + /// (by ): for a reader that looks at nothing else, such as a key comparison, + /// which would otherwise decode every other column of every row it passes over for nothing. + public bool[] DecodeOnly(IEnumerable columns) + { + var mask = new bool[Definition.Columns.Count]; + foreach (int column in columns) mask[column] = true; + return mask; + } + + /// The rows whose satisfy , each read in full. + /// The search decodes only the key; a match is read again whole, because a caller looking a row up by key + /// is usually about to rewrite or delete it, which takes every value. Lazy, as is. + public IEnumerable<(RowId Id, object?[] Values)> RowsWhere(IEnumerable keyColumns, Func match) + { + foreach ((RowId id, object?[] key) in Rows(DecodeOnly(keyColumns)).WithIds()) + if (match(key)) + yield return (id, GetRow(id) + ?? throw new InvalidDataException($"Row {id.Page}:{id.Row} of '{Name}' was read a moment ago and is gone.")); + } + + /// The rows whose hold (position for position), + /// each read in full — seeked through an index on exactly those columns when the table has one, and found + /// by 's scan when it has none. decides every row either way: + /// an index key is lossy (text folds case and trailing spaces), so the seek only narrows. + /// The key has to be in each column's own kind already, because an index answers only in its + /// column's kind; a caller that cannot promise that uses . A null in the key scans, + /// since how an index keys a null is not the question a key comparison asks. + public IEnumerable<(RowId Id, object?[] Values)> RowsWithKey( + int[] columns, object?[] key, Func match) + { + IndexDef? index = key.Any(k => k is null) ? null : Definition.Indexes.FirstOrDefault(i => + i.RootPage > 0 && i.Columns.Count == columns.Length && i.Columns.All(c => columns.Contains(c.Column.Index))); + if (index is null) + return RowsWhere(columns, match); + + // A seek key is addressed by column ordinal, not by position in the index. + var seekKey = new object?[Definition.Columns.Count]; + for (int i = 0; i < columns.Length; i++) + seekKey[columns[i]] = key[i]; + return SeekRowsWithIds(index, seekKey).Where(r => match(r.Values)); + } /// A row decoder over this table's columns — reuse one across a seek/scan rather than allocating - /// per row (each carries a shared ). - private RowDecoder NewDecoder() => new(Definition.Columns, Channel.Format, new LongValueReader(Channel)); + /// per row (each carries a shared ). + private RowCodec NewDecoder(bool[]? decode = null) => + new(Definition.Columns, Channel.Format, longValues: new LongValueStore(Channel), decode: decode); + + /// The decoder a seek reads its rows with: the last one made, while it was made for the same column + /// mask (the same array — a caller works one out and passes it to every seek), else a new one. + /// An index-nested-loop join seeks once per outer row, and building a decoder each time — its column + /// array copied, a long-value reader made — was a sixth of what such a join allocated. A decoder holds nothing + /// that changes as it decodes, so one serves every seek, interleaved or not. The mask and its decoder are held + /// as one object, so no reader can pair one with the other's partner. + private RowCodec SeekDecoder(bool[]? decode) + { + if (_seekDecoder is { } held && ReferenceEquals(held.Mask, decode)) return held.Decoder; + RowCodec decoder = NewDecoder(decode); + _seekDecoder = new MaskedDecoder(decode, decoder); + return decoder; + } + + private sealed record MaskedDecoder(bool[]? Mask, RowCodec Decoder); + + private MaskedDecoder? _seekDecoder; + + /// The index reader every seek goes through, made once: seeking changes nothing about it. + private IndexTree IndexReader => _indexReader ??= new IndexTree(Channel, Definition); + + private IndexTree? _indexReader; /// Decodes the row at (following an overflow forward-pointer to a /// relocated row), or if the slot is empty/deleted. Used by an index seek, which /// yields row ids. public object?[]? GetRow(RowId id) => GetRow(id, NewDecoder()); - private object?[]? GetRow(RowId id, RowDecoder decoder) + private object?[]? GetRow(RowId id, RowCodec decoder) { if (id.Page <= 0 || id.Page >= Channel.PageCount) throw new InvalidDataException( @@ -43,13 +115,18 @@ public Table(PageChannel channel, TableDef definition) // Read just the one wanted slot straight from the page directory (O(1)), over the shared cache buffer // without copying the 4 KB page out — the bytes are consumed immediately by Decode. Both were the // seek's per-row hot cost. - if (!Pages.DataPage.TryReadRow(Channel.ReadPageShared(id.Page), Channel.Format, id.Row, out Pages.RowSlot slot, out ReadOnlySpan bytes)) + PageBuffer page = Channel.ReadPageShared(id.Page); + if (!Pages.DataPage.TryReadRow(page, Channel.Format, id.Row, out DataPage.RowSlot slot, out ReadOnlySpan bytes)) return null; + uint owner = DataPage.ReadOwner(page.Span, Channel.Format); + if (owner != (uint)Definition.DefinitionPage) + throw new InvalidDataException($"Row page {id.Page} belongs to TDEF {owner}, not TDEF {Definition.DefinitionPage}."); + if (slot.IsDeleted) return null; if (slot.HasOverflow) { - RelocatedRow target = RowRelocationReader.Resolve( + DataPage.RelocatedRow target = DataPage.ResolveRelocation( Channel, Definition.DefinitionPage, slot, bytes); return decoder.Decode(target.Bytes); } @@ -58,32 +135,34 @@ public Table(PageChannel channel, TableDef definition) /// Yields the rows whose key equals — an index /// seek (equality) instead of a full scan. May over-return (lossy text/binary keys); the caller re-checks - /// the predicate. - public IEnumerable SeekRows(IndexDef index, object?[] values) + /// the predicate. is as for . + public IEnumerable SeekRows(IndexDef index, object?[] values, bool[]? decode = null) { - var decoder = NewDecoder(); - foreach (RowId id in new IndexWriter(Channel, Definition).Seek(index, values)) + RowCodec decoder = SeekDecoder(decode); + foreach (RowId id in IndexReader.Seek(index, values)) if (GetRow(id, decoder) is { } row) yield return row; } /// Like but yields each matching row together with its — - /// for an UPDATE/DELETE join that must know which physical row to rewrite/remove, not just its values. - public IEnumerable<(RowId Id, object?[] Values)> SeekRowsWithIds(IndexDef index, object?[] values) + /// for an UPDATE/DELETE join that must know which physical row to rewrite/remove, not just its values. + /// is as for : a row that is then written has to be read again + /// in full. + public IEnumerable<(RowId Id, object?[] Values)> SeekRowsWithIds(IndexDef index, object?[] values, bool[]? decode = null) { - var decoder = NewDecoder(); - foreach (RowId id in new IndexWriter(Channel, Definition).Seek(index, values)) + RowCodec decoder = SeekDecoder(decode); + foreach (RowId id in IndexReader.Seek(index, values)) if (GetRow(id, decoder) is { } row) yield return (id, row); } /// Yields the rows whose key lies in [, /// ] (either bound null = open) — an index range scan. May over-return at the - /// boundaries; the caller re-checks the predicate. - public IEnumerable SeekRangeRows(IndexDef index, object?[]? low, object?[]? high) + /// boundaries; the caller re-checks the predicate. is as for . + public IEnumerable SeekRangeRows(IndexDef index, object?[]? low, object?[]? high, bool[]? decode = null) { - var decoder = NewDecoder(); - foreach (RowId id in new IndexWriter(Channel, Definition).SeekRange(index, low, high)) + RowCodec decoder = SeekDecoder(decode); + foreach (RowId id in IndexReader.SeekRange(index, low, high)) if (GetRow(id, decoder) is { } row) yield return row; } @@ -94,28 +173,86 @@ public Table(PageChannel channel, TableDef definition) /// Rewrites the row at in place with new values (row id preserved). /// are the columns that actually changed — an unchanged memo/OLE column /// keeps its stored descriptor (no re-materialise), a changed one has its old LVAL pages reclaimed. - public void Update(RowId id, object?[] values, IReadOnlySet changedColumns) => - new RowInserter(Channel, Definition).Update(id, values, changedColumns); + public void Update(RowId id, object?[] values, IReadOnlySet changedColumns) + { + Write(() => + { + object?[] original = GetRow(id) ?? throw new InvalidOperationException($"Row '{id}' does not exist."); + var changed = new HashSet(changedColumns); + for (int i = 0; i < values.Length; i++) + if (original[i] is byte[] oldBytes && values[i] is byte[] newBytes + ? !oldBytes.AsSpan().SequenceEqual(newBytes) : !Equals(original[i], values[i])) + changed.Add(i); + + new RowInserter(Channel, Definition).Update(id, values, changed); + foreach (IndexDef index in Definition.RealIndexes + .Where(i => i.Columns.Any(c => changed.Contains(c.Column.Index)))) + MoveIndexEntry(index, original, values, id); + }); + } /// Rewrites the row treating every column as changed (materialises all long values). public void Update(RowId id, object?[] values) => Update(id, values, new HashSet(System.Linq.Enumerable.Range(0, values.Length))); /// Moves a row's entry in one index when its key changes (remove old key, add new; row id - /// unchanged). Used by UPDATE of an indexed column. - public void MoveIndexEntry(IndexDef index, object?[] oldValues, object?[] newValues, RowId id) => - new IndexWriter(Channel, Definition).MoveEntry(index, oldValues, newValues, id); + /// unchanged), and counts the move in the index's statistics as ACE does. Used by UPDATE of an indexed + /// column. + internal void MoveIndexEntry(IndexDef index, object?[] oldValues, object?[] newValues, RowId id) + { + new IndexTree(Channel, Definition).MoveEntry(index, oldValues, newValues, id); + // An IGNORE NULL index the row was absent from never counted it, so has nothing to count out. + if (!(index.IgnoreNulls && IndexTree.HasNullKey(index, oldValues))) + new RowInserter(Channel, Definition).CountKeyMoved(index); + } /// Whether ' key already exists in for a row /// other than — used to enforce a UNIQUE/PRIMARY index on UPDATE. public bool HasDuplicateKey(IndexDef index, object?[] values, RowId excludeRow) => - new IndexWriter(Channel, Definition).KeyExists(index, values, (excludeRow.Page << 8) | excludeRow.Row); + new IndexTree(Channel, Definition).KeyExists(index, values, excludeRow.Packed); /// Soft-deletes the row at (row bytes kept, slot flagged; TDEF row count - /// decremented). The caller removes its index entries first via . - public void Delete(RowId id) => new RowInserter(Channel, Definition).Delete(id); + /// decremented), removing every index entry before reclaiming the row. + public void Delete(RowId id) + { + Write(() => + { + object?[] values = GetRow(id) ?? throw new InvalidOperationException($"Row '{id}' does not exist."); + foreach (ComplexColumn complex in _catalog.ComplexColumns) + { + if (complex.OwnerTable.DefinitionPage != Definition.DefinitionPage) continue; + if (values[complex.OwnerTable.RequireColumn(complex.ColumnName).Index] is not { } raw) continue; + int recordId = Convert.ToInt32(raw, System.Globalization.CultureInfo.InvariantCulture); + var flat = new Table(Channel, complex.FlatTable, _catalog); + int linkColumn = complex.OwnerLink.Index; + foreach ((RowId flatId, _) in flat.RowsWithKey([linkColumn], [recordId], row => + row[linkColumn] is { } link + && Convert.ToInt32(link, System.Globalization.CultureInfo.InvariantCulture) == recordId).ToList()) + flat.Delete(flatId); + } + foreach (IndexDef index in Definition.RealIndexes) + RemoveIndexEntry(index, values, id); + new RowInserter(Channel, Definition).Delete(id); + }); + } + + private void Write(Action action) + { + bool ownTransaction = !Channel.InTransaction; + if (ownTransaction) Channel.BeginTransaction(); + try + { + action(); + if (ownTransaction) Channel.CommitTransaction(flush: false); + } + catch + { + if (ownTransaction) Channel.RollbackTransaction(); + throw; + } + } /// Removes a deleted row's entry from one index. - public void RemoveIndexEntry(IndexDef index, object?[] values, RowId id) => - new IndexWriter(Channel, Definition).DeleteEntry(index, values, id); + internal void RemoveIndexEntry(IndexDef index, object?[] values, RowId id) => + new IndexTree(Channel, Definition).DeleteEntry(index, values, id); } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/TableCreator.cs b/src/LibRed/LibRed.Core/Storage/TableCreator.cs deleted file mode 100644 index 278f8827f..000000000 --- a/src/LibRed/LibRed.Core/Storage/TableCreator.cs +++ /dev/null @@ -1,3514 +0,0 @@ -using EntityFrameworkCore.Jet.Data; -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.IO; -using LibRed.Pages; -using System.Buffers.Binary; -using System.Globalization; -using System.Text; -using MapRetirement = ((int Row, int Page) Map, System.Collections.Generic.IReadOnlyList<(int Row, int Page)> Clear, System.Collections.Generic.IReadOnlyList Pages); - -namespace LibRed.Storage; - -/// -/// Creates and alters tables in an existing database. allocates and writes the TDEF -/// page, its indexes' B-tree roots and an owned-pages usage map, then records the table in MSysObjects so -/// the catalog finds it. The rest of the class is the incremental DDL EF issues one statement at a time — -/// ADD/DROP/ALTER/RENAME for columns, indexes, relationships and CHECK constraints — each a surgical edit -/// of the existing TDEF rather than a rebuild, so unmodelled descriptor bytes survive. -/// -public sealed class TableCreator(PageChannel channel, JetCatalog catalog, Collation? collation = null) -{ - private readonly PageChannel _channel = channel; - private readonly JetCatalog _catalog = catalog; - private readonly PageAllocator _allocator = new(channel); - - // The database's default collating order, written into new non-numeric columns. Defaults to General - // legacy for callers that don't create columns (most alter operations). - private readonly Collation _collation = collation ?? Collation.GeneralLegacy; - - /// - /// Jet/ACE caps a table at 32 indexes and the cap applies to BOTH TDEF counts — the index-data blocks at - /// 0x33 and the logical index-info blocks at 0x2F. Microsoft states it against the logical - /// one: "Number of indexes in a table: 32, including indexes created internally to maintain table - /// relationships, single-field and composite indexes." - /// - /// - /// - /// checks both when a table is created with all its indexes at once, but EF does - /// not work that way: it creates the table, then adds indexes and relationships one statement at a time, - /// and each of those comes through a surgical insert here instead. Nothing checked those paths, so the - /// counts simply walked past 32 — the fields are Int32, so nothing overflowed. - /// - /// - /// The logical count is the one that binds, because a data block must be named by a logical block, so - /// 0x33 ≤ 0x2F always holds. A table many others reference gains a logical block per incoming - /// relationship and no data block, so it overruns on 0x2F while 0x33 still looks healthy — - /// which is exactly what the read-side check inspected. Building EF Core's - /// ComplexNavigationsSharedType model, Level1 reached 46 logical against 31 data, and the - /// resulting file was unreadable by Access ("Unrecognized database format", the table missing entirely) - /// while LibRed read it back without complaint. Measured in IndexCountLimitAccessTests; see - /// docs/format/page-02d-constraints.md. - /// - /// - private const int MaxIndexesPerTable = 32; - - /// Rejects an incremental index or relationship that would take either TDEF count past the - /// Jet/ACE limit, naming both counts so the failure says which one bound. - private static void EnsureIndexCapacity(string tableName, string what, int dataCount, int logicalCount) - { - if (dataCount > MaxIndexesPerTable || logicalCount > MaxIndexesPerTable) - { - throw new NotSupportedException( - $"Cannot add {what} to '{tableName}': it would leave the table with {dataCount} index-data blocks and " + - $"{logicalCount} logical index blocks. Jet/ACE allows at most {MaxIndexesPerTable} of each, counting " + - "those backing primary keys, unique constraints and relationships - and a table referenced by many " + - "others accumulates a logical block per incoming relationship without gaining a data block."); - } - } - - - public void Create( - string name, - IReadOnlyList columns, - IReadOnlyList? primaryKey = null, - IReadOnlyList? relationships = null, - IReadOnlyList? uniqueConstraints = null, - IReadOnlyList<(string Column, string DefaultSql)>? columnDefaults = null, - IReadOnlyList<(string Name, string Expression)>? checkConstraints = null, - string? primaryKeyName = null) - { - relationships ??= []; - uniqueConstraints ??= []; - columnDefaults ??= []; - checkConstraints ??= []; - - // Reject names ACE can't use before writing anything: > 64 chars corrupts the whole file for ACE, and - // the characters . ! ` [ ] make the name unreferenceable in ACE SQL (both verified vs ACE). Applies only - // to caller-supplied names — LibRed's own hidden .rN relationship-index names are generated later. - JetName.Validate(name, "table name"); - foreach (ColumnSpec c in columns) - JetName.Validate(c.Name, "column name"); - // Constraint names go to disk too (PK/unique → index names, FK → MSysRelationships, CHECK → LvProp) and - // carry the same 64-char + forbidden-char limits — verified: a 100-char FK name overruns into adjacent - // data and a 100-char index name breaks ACE's index enumeration. Only validate caller-supplied names. - if (primaryKeyName is not null) JetName.Validate(primaryKeyName, "primary key name"); - foreach (RelationshipSpec r in relationships) JetName.Validate(r.Name, "foreign key name"); - var relationshipNames = new HashSet(StringComparer.OrdinalIgnoreCase); - foreach (RelationshipSpec r in relationships) - if (!relationshipNames.Add(r.Name)) throw RelationshipNameTaken(r.Name); - else EnsureRelationshipNameFree(r.Name); - foreach (UniqueIndexSpec u in uniqueConstraints) JetName.Validate(u.Name, "unique constraint name"); - foreach ((string checkName, _) in checkConstraints) JetName.Validate(checkName, "check constraint name"); - - // A table name is unique (case-insensitively) across the database; reject a duplicate rather - // than writing a second MSysObjects row that shadows the existing table. - if (_catalog.FindTable(name) is not null) - throw new SchemaObjectExistsException($"Table '{name}' already exists.", name); - - // Jet/ACE caps a table at 255 columns. The count/id fields are 2 bytes wide so we could physically - // write more, but Access would refuse to open the table — fail early with a clear message instead. - if (columns.Count > MaxColumnsPerTable) - throw new InvalidOperationException( - $"Table '{name}' has {columns.Count} columns; Jet/ACE tables are limited to {MaxColumnsPerTable}."); - - // Before the foreign keys' type match, as ACE checks it: an OLE column referencing a LONG key gets this. - RejectOleIndexColumns( - (primaryKey ?? []).Concat(uniqueConstraints.SelectMany(u => u.Columns)) - .Concat(relationships.SelectMany(r => r.Columns.Select(c => c.Column))), - n => columns.FirstOrDefault(c => string.Equals(c.Name, n, StringComparison.OrdinalIgnoreCase))?.Type); - - relationships = relationships.Select(fk => ResolvePrimaryKeyReference(fk, creatingTable: name)).ToList(); - foreach (RelationshipSpec fk in relationships) - EnsureSameDataTypes(fk, ColumnOf(columns), string.Equals(fk.ReferencedTable, name, StringComparison.OrdinalIgnoreCase) - ? ColumnOf(columns) - : ColumnOf(_catalog.FindTable(fk.ReferencedTable))); - - JetFormatBase format = _channel.Format; - - // Allocate the pages the table needs through the global free-pages map (so Access accounts - // for them). Like Access, a fresh table has NO data page — the first is allocated lazily on - // the first insert — so its usage maps start empty. - int tdefPage = _allocator.Allocate(); - int usageMapPage = _allocator.Allocate(); - - // Key the long-value maps by the column's *id*, not its position. The two coincide on an ordinary - // CREATE TABLE, but a spec can carry an explicit id — the faithful-rebuild path does, and ids are - // never reused after a DROP COLUMN — and the TDEF's long-value map is read back by id, so using the - // position there silently points a Memo/OLE column's usage maps at the wrong column. - // A calculated column with a Memo RESULT needs the maps too, and its declared type does not say so: - // ACE declares such a column Text with length 0 and reaches the value through a long-value - // descriptor, so keying off Type alone leaves it without maps and its result nowhere to go (§3.4a). - var longValueCols = columns.Select((c, i) => (Column: c, Id: c.ColumnId ?? i)) - .Where(x => x.Column.Type is JetDataType.Memo or JetDataType.Ole - || x.Column.CalculatedResultType is JetDataType.Memo) - .ToList(); - - // The table's data-block indexes: the primary key (unique), then a unique index per UNIQUE - // constraint, then one non-unique index per foreign key over its child columns — Access enforces - // a relationship through an index on the FK columns. Each carries the relationship (if any) it backs. - var indexPlans = new List<(string Name, IReadOnlyList Columns, bool IsPk, bool IsUnique, RelationshipSpec? Fk)>(); - if (primaryKey is { Count: > 0 }) - // Name the PK index after the CONSTRAINT if one was given (ACE does the same, and the scaffolder - // round-trips it). If unnamed, LibRed picks the stable "PrimaryKey" (the DAO/Access-UI convention) - // — an engine choice, since ACE-via-SQL instead generates a random "Index_" with no fixed - // value to reproduce, and nothing downstream depends on the exact name. - indexPlans.Add((primaryKeyName ?? "PrimaryKey", primaryKey, true, true, null)); - foreach (UniqueIndexSpec unique in uniqueConstraints) - indexPlans.Add((unique.Name, unique.Columns, false, true, null)); - foreach (RelationshipSpec fk in relationships) - indexPlans.Add((fk.Name, fk.Columns.Select(c => c.Column).ToList(), false, false, fk)); - - // Every index this CREATE would build, including the ones arriving as a PRIMARY KEY or UNIQUE - // constraint rather than as an index — the inline `col type PRIMARY KEY` form is refused earlier, at - // the SQL layer, but the table-level CONSTRAINT form reaches here. - foreach (var plan in indexPlans) - RejectCalculatedIndexColumns(plan.Name, plan.Columns, - n => columns.FirstOrDefault( - c => c.CalculatedExpression is not null - && string.Equals(c.Name, n, StringComparison.OrdinalIgnoreCase))?.Name); - - // Usage-map layout (verified vs ACE): the primary page holds row 0 = table owned, row 1 = table - // free, then one row per index, then two rows (owned + free) per long-value (memo/OLE) column — as - // many *whole* columns as fit (a page holds ~57 inline records). Once the primary page is full, each - // remaining long-value column gets its OWN usage-map page (owned = row 0, free = row 1). All maps - // start empty. This keeps a wide table's per-column maps from overflowing a single page. - // How many 69-byte inline map records (plus their 2-byte directory slot) fit on one page. - int mapsPerPage = (format.PageSize - format.DataRowDirectoryOffset) / (UsageMapRecordLength + 2); - int primaryRecords = 2 + indexPlans.Count; // data owned/free + one per index - int colsOnPrimary = Math.Clamp((mapsPerPage - primaryRecords) / 2, 0, longValueCols.Count); - WriteUsageMaps(format, usageMapPage, mapCount: primaryRecords + colsOnPrimary * 2); - - // §3.3.2 entries: a long-value column's maps are on the primary page (if it fit) or a dedicated page. - var longValueSpecs = new List(longValueCols.Count); - for (int j = 0; j < longValueCols.Count; j++) - { - int colId = longValueCols[j].Id; - if (j < colsOnPrimary) - longValueSpecs.Add(new LongValueColumnSpec( - colId, UsedRow: primaryRecords + 2 * j, FreeRow: primaryRecords + 2 * j + 1, MapPage: usageMapPage)); - else - { - int columnMapPage = _allocator.Allocate(); - WriteUsageMaps(format, columnMapPage, mapCount: 2); // owned = row 0, free = row 1 - longValueSpecs.Add(new LongValueColumnSpec(colId, UsedRow: 0, FreeRow: 1, MapPage: columnMapPage)); - } - } - - // Each index is an empty leaf root, populated as rows are inserted. Its usage map is on the primary - // page right after the two data-page maps (row 2 + i). - var indexes = new List(indexPlans.Count); - for (int i = 0; i < indexPlans.Count; i++) - { - var plan = indexPlans[i]; - int rootPage = _allocator.Allocate(); - WriteEmptyLeafIndexPage(format, rootPage, owner: tdefPage); - // Record the root in the index's own pages usage map — Access does this at CREATE, before any - // row exists (verified: a freshly created empty index has exactly its root bit set). As the tree - // grows, IndexWriter adds each page it allocates, so the map covers the whole B-tree. - new UsageMapWriter(_channel).SetBit(2 + i, usageMapPage, rootPage, set: true); - indexes.Add(new IndexSpec(plan.Name, plan.Columns, plan.IsPk, plan.IsUnique, - rootPage, UsageMapRow: 2 + i, UsageMapPage: usageMapPage)); - } - - // Build the child's logical index-info blocks. A plain index (PK) maps 1:1 to its data block; - // a foreign key's data block instead carries the *outgoing* relationship block (§3.6), linked - // to an *incoming* block. The two ends cross-reference by index_num. For a cross-table FK the - // incoming block is added to the parent's TDEF; for a self-reference it lives in this same TDEF. - var childLogical = new List(indexPlans.Count); - var incoming = new List(); - var parentAdds = new Dictionary(); - // Incoming blocks that this table hosts for its own self-references are numbered after the - // data-block logical indexes (verified vs ACE: a self-ref adds one such block at num = data count). - int selfIncomingNum = indexPlans.Count; - for (int i = 0; i < indexPlans.Count; i++) - { - var plan = indexPlans[i]; - if (plan.Fk is null) - { - childLogical.Add(new TdefBuilder.LogicalIndexSpec( - Number: i, DataOrdinal: i, FkType: 0, FkNumber: 0xFFFFFFFF, FkTablePage: 0, - UpdateAction: IndexBlockFormat.PlainAction, DeleteAction: IndexBlockFormat.PlainAction, - Type: plan.IsPk ? IndexBlockFormat.TypePrimary : IndexBlockFormat.TypeSecondary, Name: plan.Name)); - continue; - } - - RelationshipSpec fk = plan.Fk; - if (fk.UpdateSetNull) throw UpdateSetNullNotImplemented(); - byte upd = fk.CascadeUpdate ? CascadeAction : NoCascadeAction; - byte del = fk.CascadeDelete ? CascadeAction : fk.DeleteSetNull ? SetNullAction : NoCascadeAction; - byte outgoingType = fk.NoIndex ? FkTypeOutgoingNoIndex : FkTypeOutgoing; - - // A self-referencing FK: the table is not in the catalog yet (we are creating it), so resolve - // the referenced index within the plans we are building and host both ends here. - if (string.Equals(fk.ReferencedTable, name, StringComparison.OrdinalIgnoreCase)) - { - int refOrdinal = SelfReferencedOrdinal(indexPlans, fk); - int inNum = selfIncomingNum++; - // Outgoing block (this table's child side) — NO INDEX flags it 0x03 instead of 0x02. - childLogical.Add(new TdefBuilder.LogicalIndexSpec( - Number: i, DataOrdinal: i, FkType: outgoingType, FkNumber: (uint)inNum, - FkTablePage: tdefPage, UpdateAction: upd, DeleteAction: del, - Type: IndexBlockFormat.TypeForeign, Name: fk.Name)); - // Incoming block (this table's parent side), hidden ".r" name unique within the table. - childLogical.Add(new TdefBuilder.LogicalIndexSpec( - Number: inNum, DataOrdinal: refOrdinal, FkType: FkTypeIncoming, FkNumber: (uint)i, - FkTablePage: tdefPage, UpdateAction: upd, DeleteAction: del, - Type: IndexBlockFormat.TypeForeign, Name: NextHiddenRelationshipName(childLogical.Select(l => l.Name).ToList()))); - continue; - } - - (int parentPage, int refOrd, int parentNextNum) = ResolveParent(fk, tdefPage); - int parentNum = parentNextNum + parentAdds.GetValueOrDefault(parentPage); - parentAdds[parentPage] = parentAdds.GetValueOrDefault(parentPage) + 1; - - childLogical.Add(new TdefBuilder.LogicalIndexSpec( - Number: i, DataOrdinal: i, FkType: outgoingType, - FkNumber: (uint)parentNum, FkTablePage: parentPage, UpdateAction: upd, DeleteAction: del, - Type: IndexBlockFormat.TypeForeign, Name: fk.Name)); - incoming.Add(new IncomingRelationship(parentPage, parentNum, refOrd, - ChildBlockNumber: (uint)i, ChildPage: tdefPage, upd, del)); - } - - // Access stores logical blocks sorted by name (with their names in the same order). - childLogical.Sort((a, b) => string.CompareOrdinal(a.Name, b.Name)); - - // Build the definition and point it at the usage maps: owned-pages = row 0, free-pages = - // row 1, both on the usage-map page. - byte[] tdef = TdefBuilder.Build(format, TableType.User, columns, indexes, longValueSpecs, childLogical, _collation).Page; - tdef[format.TdefOwnedPagesOffset] = 0; // owned map record row - WriteInt24(tdef, format.TdefOwnedPagesOffset + 1, usageMapPage); - tdef[format.TdefFreePagesOffset] = 1; // free map record row - WriteInt24(tdef, format.TdefFreePagesOffset + 1, usageMapPage); - // A wide table's definition can exceed one page; write it split across continuation pages if needed. - int defEnd = BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(format.TdefLengthOffset, 4)); - WriteDefinition(tdefPage, tdef[..defEnd], [], rewrite: false); - - // Per-column extended properties, in column order with DefaultValue before Required (matching ACE): - // a DEFAULT is a memo property; a NOT NULL column carries a boolean Required property, and a nullable - // one none. An AutoNumber follows the same rule — ACE writes Required for COUNTER NOT NULL and not for - // a bare COUNTER (verified by reading its property blob back). - var columnProps = new List(); - foreach (ColumnSpec col in columns) - { - var def = columnDefaults.FirstOrDefault(d => string.Equals(d.Column, col.Name, StringComparison.OrdinalIgnoreCase)); - if (def.DefaultSql is not null) - columnProps.Add(new PropertyBlob.Property(col.Name, PropertyBlob.DefaultValueProperty, def.DefaultSql)); - if (!col.IsNullable) - columnProps.Add(PropertyBlob.Bool(col.Name, PropertyBlob.RequiredProperty, true)); - columnProps.AddRange(CalculatedProperties(col)); - } - - AddCatalogRow(name, tdefPage, columnProps, checkConstraints); - AddPermissionRows(tdefPage); - foreach (RelationshipSpec fk in relationships) - AddRelationshipRows(name, fk); - foreach (IncomingRelationship inc in incoming) - AddIncomingRelationshipBlock(inc); - } - - /// The property-blob entries that make a column calculated (§3.4a), or nothing for an ordinary - /// one. All three pieces have to agree or Access reads the payload wrongly: Expression is the text - /// the engine evaluates, ResultType is the authority on the payload's encoding, and the three - /// FCMin*Ver strings declare the Access floor a calculated column forces. The version properties - /// are not DDL properties, unlike the first two — matching what ACE writes. - private static IEnumerable CalculatedProperties(ColumnSpec column) - { - if (column.CalculatedExpression is not { } expression) yield break; - - JetDataType resultType = column.CalculatedResultType ?? column.Type; - yield return new PropertyBlob.Property( - column.Name, PropertyBlob.ExpressionProperty, expression, JetDataType.Memo); - // A Byte property is one raw byte, so the value has to be supplied as RawValue -- a Value string - // would be written as UTF-16 text and read back as a nonsense type code. - yield return new PropertyBlob.Property( - column.Name, PropertyBlob.ResultTypeProperty, - ((byte)resultType).ToString(System.Globalization.CultureInfo.InvariantCulture), - JetDataType.Byte, RawValue: [(byte)resultType]); - foreach (string version in CalculatedVersionProperties) - yield return new PropertyBlob.Property( - column.Name, version, CalculatedMinimumVersion, JetDataType.Text) - { IsDdl = false }; - } - - private static readonly string[] CalculatedVersionProperties = - ["FCMinReadVer", "FCMinWriteVer", "FCMinDesignVer"]; - - /// Access 2010 (ACE 14) is the floor a calculated column declares, which is also the on-disk - /// version byte it needs — see . - private const string CalculatedMinimumVersion = "14.0.0000.0000"; - - // Index-info block field values (§3.6), verified against ACE-created relationships. - // (PlainAction and the index-type bytes are shared with the reader via IndexBlockFormat.) - private const byte NoCascadeAction = 0x00; // relationship without ON UPDATE/DELETE CASCADE - private const byte CascadeAction = 0x01; // relationship with cascade - private const byte SetNullAction = 0x02; // ON DELETE SET NULL (verified vs ACE, index-info block +0x16) - - /// ON UPDATE SET NULL pathway: the docs list it, but the ACE OLE DB provider rejects it via SQL, - /// so its on-disk storage (the grbit flag + the index-info +0x15 action byte) is unverified. Rather than - /// guess the bytes, fail loudly until a UI/DAO-created sample can be probed. - private static NotImplementedException UpdateSetNullNotImplemented() => new( - "ON UPDATE SET NULL is not implemented: its Jet storage bytes are unverified (the ACE OLE DB provider " + - "rejects the DDL, so they could not be probed). Only ON UPDATE {NO ACTION | CASCADE} are supported."); - private const byte FkTypeIncoming = 0x01; // this table is the parent/referenced end - private const byte FkTypeOutgoing = 0x02; // this table is the child/referencing end (indexed) - private const byte FkTypeOutgoingNoIndex = 0x03; // child/referencing end declared FOREIGN KEY NO INDEX - - /// An incoming-relationship logical block to add to a parent table's TDEF. - private readonly record struct IncomingRelationship( - int ParentPage, int Number, int ReferencedOrdinal, uint ChildBlockNumber, int ChildPage, - byte UpdateAction, byte DeleteAction); - - /// - /// ACE's "same data types" rule for a relationship, measured over every pairing of the column types: each child - /// column must have its parent column's storage type, whatever either one's length — TEXT(5), - /// TEXT(20) and CHAR(10) all pair with one another, as do DECIMALs of any precision and - /// scale and BINARY with VARBINARY. An AutoNumber is a Long on either side. Checked before - /// anything is written. A column that is not found is left to the check that reports it. - /// - private static void EnsureSameDataTypes(RelationshipSpec fk, - Func childColumn, Func parentColumn) - { - foreach ((string column, string referenced) in fk.Columns) - { - if (childColumn(column) is not { } child || parentColumn(referenced) is not { } parent) continue; - if (child != parent) - throw new InvalidOperationException( - "Relationship must be on the same number of fields with the same data types. " + - $"'{column}' ({child}) cannot reference '{fk.ReferencedTable}.{referenced}' ({parent})."); - } - } - - private static Func ColumnOf(IReadOnlyList columns) => - name => columns.FirstOrDefault(c => string.Equals(c.Name, name, StringComparison.OrdinalIgnoreCase))?.Type; - - private static Func ColumnOf(TableDef? table) => name => table?.FindColumn(name)?.Type; - - /// The data-block ordinal of the index over a self-reference's referenced columns, found - /// among the indexes being created for this table (the table is not in the catalog yet). - private static int SelfReferencedOrdinal( - List<(string Name, IReadOnlyList Columns, bool IsPk, bool IsUnique, RelationshipSpec? Fk)> indexPlans, - RelationshipSpec fk) - { - var refColumns = fk.Columns.Select(c => c.ReferencedColumn).ToList(); - for (int j = 0; j < indexPlans.Count; j++) - if (indexPlans[j].Columns.SequenceEqual(refColumns, StringComparer.OrdinalIgnoreCase)) - return j; - throw new InvalidOperationException( - $"Self-referencing foreign key '{fk.Name}' references ({string.Join(", ", refColumns)}), which is not a key or index of '{fk.ReferencedTable}'."); - } - - /// - /// Resolves a cross-table relationship's parent: its TDEF page, the data-block ordinal of the parent - /// index over the referenced columns (normally the PK), and the parent's current logical-index count - /// (used to number the incoming block we will add). Self-references are handled by the caller before - /// this is reached (the table is not yet in the catalog). - /// - /// The next free logical index_num on a TDEF: max + 1 over its info blocks, never - /// the block COUNT. Dropping a relationship removes a block without renumbering the survivors' index_num - /// (only their data ordinals move), so after any DROP CONSTRAINT the count is below the max and the next - /// incoming block would collide with a live one — leaving two blocks claiming one number, which is what - /// the child's cross-link at +0x0D names. The child side has always used max + 1; this is the parent - /// side agreeing with it. - private int NextLogicalIndexNumber(int tdefPage) - { - JetFormatBase format = _channel.Format; - (LibRed.IO.PageBuffer buf, _) = ReadDefinition(tdefPage); - int dataCount = buf.ReadInt32(format.TdefIndexCountOffset); - int logicalCount = buf.ReadInt32(format.TdefLogicalIndexCountOffset); - int colCount = buf.ReadUInt16(format.TdefColumnCountOffset); - - int pos = format.TdefRealIndexBlockOffset + dataCount * format.RealIndexEntrySize - + colCount * format.ColumnDescriptorSize; - for (int i = 0; i < colCount; i++) pos += 2 + buf.ReadUInt16(pos); - int infoStart = pos + dataCount * IndexBlockFormat.DataBlockSize; - - int maxNum = -1; - for (int i = 0; i < logicalCount; i++) - maxNum = Math.Max(maxNum, buf.ReadInt32( - infoStart + i * IndexBlockFormat.InfoBlockSize + IndexBlockFormat.InfoNumberOffset)); - return maxNum + 1; - } - - /// A relationship's parent table, or ACE's error when it does not exist. - private TableDef ReferencedTableOf(RelationshipSpec fk) => - _catalog.FindTable(fk.ReferencedTable) - ?? throw new InvalidOperationException( - $"Cannot find table or constraint: the referenced table '{fk.ReferencedTable}' does not exist."); - - /// - /// Resolves REFERENCES table with no column list () - /// to the parent's primary key, pairing the child columns with the key's in order whatever either side's - /// columns are named — as ACE does. Any other relationship comes back unchanged. - /// - /// - /// ACE refuses it when the parent has no primary key (a unique index does not stand in for one) and when the - /// columns differ in number (verified). A table referencing itself in its own CREATE TABLE - /// () has no key in the catalog yet; the SQL parser pairs such a reference - /// with a key the statement declares earlier, so one still unpaired here has no key to reference. - /// - private RelationshipSpec ResolvePrimaryKeyReference(RelationshipSpec fk, string? creatingTable) - { - if (!fk.ReferencesPrimaryKey) return fk; - bool selfInCreate = creatingTable is not null - && string.Equals(fk.ReferencedTable, creatingTable, StringComparison.OrdinalIgnoreCase); - IndexDef primaryKey = (selfInCreate ? null : ReferencedTableOf(fk).Indexes.FirstOrDefault(i => i.IsPrimaryKey)) - ?? throw new InvalidOperationException( - $"Cannot create relationship. Referenced table '{fk.ReferencedTable}' does not have a primary key."); - return fk with - { - Columns = RelationshipSpec.PairColumns(fk.ReferencedTable, - fk.Columns.Select(c => c.Column).ToList(), primaryKey.Columns.Select(c => c.Column.Name).ToList()), - ReferencesPrimaryKey = false, - }; - } - - private (int Page, int ReferencedOrdinal, int NextIndexNumber) ResolveParent(RelationshipSpec fk, int childPage) - { - TableDef parent = ReferencedTableOf(fk); - if (parent.DefinitionPage == childPage) - throw new InvalidOperationException($"Self-referencing foreign key '{fk.Name}' should have been handled inline."); - - var ptdef = new Pages.TableDefinitionPage(); - ptdef.Read(_channel, parent.DefinitionPage); - var refColumns = fk.Columns.Select(c => c.ReferencedColumn).ToList(); - IndexDef refIndex = FindParentKeyIndex(ptdef.Indexes, refColumns, fk.ReferencedTable); - return (parent.DefinitionPage, refIndex.RealIndexOrdinal, NextLogicalIndexNumber(parent.DefinitionPage)); - } - - /// ACE refuses a relationship name another relationship already has (verified); a table or query may - /// share it. Checked before anything is written. - private void EnsureRelationshipNameFree(string name) - { - if (_catalog.Relationships.Any(r => string.Equals(r.Name, name, StringComparison.OrdinalIgnoreCase))) - throw RelationshipNameTaken(name); - } - - private static SchemaObjectExistsException RelationshipNameTaken(string name) => - new($"There is already a relationship named '{name}' in the current database.", name); - - /// - /// Writes the MSysRelationships rows for one relationship — one row per column pair, with - /// ccolumn = the pair count, icolumn = the 0-based pair index, and grbit - /// encoding enforce/cascade (verified against Access: an enforced no-cascade FK stores grbit 0) — and the - /// relationship's own MSysObjects object, which ACE records for every relationship. - /// - private void AddRelationshipRows(string childTable, RelationshipSpec fk) - { - TableDef msys = _catalog.FindTable("MSysRelationships") - ?? throw new InvalidOperationException("MSysRelationships catalog table was not found."); - new ViewCreator(_channel, _catalog).CreateRelationshipObject(fk.Name); - - int grbit = 0; - if (!fk.IsEnforced) grbit |= RelationshipFlags.DontEnforce; - if (fk.CascadeUpdate) grbit |= RelationshipFlags.UpdateCascade; - if (fk.CascadeDelete) grbit |= RelationshipFlags.DeleteCascade; - if (fk.DeleteSetNull) grbit |= RelationshipFlags.DeleteSetNull; - - for (int i = 0; i < fk.Columns.Count; i++) - { - var (column, referencedColumn) = fk.Columns[i]; - var values = new object?[msys.Columns.Count]; - SetByName(msys, values, "szRelationship", fk.Name); - SetByName(msys, values, "szObject", childTable); - SetByName(msys, values, "szColumn", column); - SetByName(msys, values, "szReferencedObject", fk.ReferencedTable); - SetByName(msys, values, "szReferencedColumn", referencedColumn); - SetByName(msys, values, "ccolumn", fk.Columns.Count); - SetByName(msys, values, "icolumn", i); - SetByName(msys, values, "grbit", grbit); - new RowInserter(_channel, msys).Insert(values, updateIndexes: true); - } - } - - - /// - /// Adds an index to an existing table for CREATE INDEX. Surgically inserts a statistics block, an - /// index-data block and a logical index-info block into the TDEF (preserving the existing columns, - /// indexes, relationship linkage and long-value entries byte-for-byte), grows the usage-map page by one - /// row and writes a B-tree root, back-filled from the table's existing rows. - /// - public void AddIndex(string tableName, string indexName, IReadOnlyList<(string Column, bool Descending)> columns, - bool isUnique, bool isPrimary, bool disallowNull, bool ignoreNulls) - { - JetName.Validate(indexName, "index name"); - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found."); - // ACE rejects a duplicate index name, and so must LibRed: every lookup downstream (DROP INDEX, the - // back-fill, DROP CONSTRAINT) finds an index by name with First/FirstOrDefault, so two blocks sharing - // one name make those operations pick an arbitrary block — a DROP that removes the wrong one. - if (table.Indexes.Any(i => string.Equals(i.Name, indexName, StringComparison.OrdinalIgnoreCase))) - throw new InvalidOperationException( - $"Table '{table.Name}' already has an index named '{indexName}'."); - RejectCalculatedIndexColumns(indexName, columns.Select(c => c.Column), - n => table.Columns.FirstOrDefault( - c => c.IsCalculated && string.Equals(c.Name, n, StringComparison.OrdinalIgnoreCase))?.Name); - RejectOleIndexColumns(columns.Select(c => c.Column), n => table.FindColumn(n)?.Type); - var slots = ResolveSlots(table, columns.Select(c => (c.Column, Ascending: !c.Descending))); - // A table has one primary key, and ACE refuses a second (verified). - if (isPrimary && table.Indexes.Any(i => i.IsPrimaryKey)) - throw new InvalidOperationException($"Primary key already exists on table '{table.Name}'."); - InsertIndex(table, indexName, slots, - unique: isUnique || isPrimary, required: isPrimary || disallowNull, ignoreNulls, - (num, ord) => BuildPlainInfoBlock(num, ord, isPrimary)); - } - - /// Refuses an index over a calculated column, on whichever route asked for it. - /// This matches Access and diverges only from ACE's SQL layer, which is the one that gets it wrong. - /// Access's designer does not offer a calculated column in the Indexes dialog at all and will not - /// let it be the primary key; the storage engine agrees, refusing every INSERT into a table whose - /// calculated column is indexed — "Operation is not supported for this type of object." Only - /// ACE via SQL accepts CREATE INDEX (and even CREATE UNIQUE INDEX) on one, and what - /// it produces is a table into which no row can ever be written. Measured for Int16 and Int32 results, - /// with the index created both before and after rows exist (page-02e-calculated-columns). - private static void RejectCalculatedIndexColumns( - string indexName, IEnumerable columnNames, Func findCalculated) - { - foreach (string name in columnNames) - if (findCalculated(name) is { } calculated) - throw new NotSupportedException( - $"Index '{indexName}' cannot include calculated column '{calculated}'. Access does not " - + "offer one for indexing, and an index over it makes the table refuse every insert."); - } - - /// Refuses an index over an OLE column — a key, a unique constraint, a relationship's — before anything - /// is written, as ACE does on every route (verified: CREATE INDEX, PRIMARY KEY and UNIQUE both in CREATE TABLE and - /// added, a foreign key in either place, and ALTER COLUMN of an indexed column to OLE). An OLE value has no index - /// key, and without this the definition is accepted on an empty table and every later insert fails. - private static void RejectOleIndexColumns(IEnumerable columnNames, Func typeOf) - { - foreach (string name in columnNames) - if (typeOf(name) == JetDataType.Ole) - throw new InvalidOperationException($"Invalid field definition '{name}' in definition of index or relationship."); - } - - /// Resolves index column names to (columnId, ascending) slots against a table. - private static List<(int Id, bool Ascending)> ResolveSlots( - TableDef table, IEnumerable<(string Column, bool Ascending)> columns) - { - var byName = table.Columns.ToDictionary(c => c.Name, c => c.ColumnId, StringComparer.OrdinalIgnoreCase); - return columns.Select(c => byName.TryGetValue(c.Column, out int id) ? (Id: id, c.Ascending) - : throw new InvalidOperationException($"Column '{c.Column}' does not exist in '{table.Name}'.")).ToList(); - } - - /// Surgically inserts one data index and its logical info block into an existing table's TDEF, - /// name-sorted. gets (block number, data-block ordinal) and - /// returns the 28-byte info block — a plain index or an outgoing-FK block. Returns the new block number. - private int InsertIndex(TableDef table, string indexName, List<(int Id, bool Ascending)> slots, - bool unique, bool required, bool ignoreNulls, Func buildInfo) - { - JetFormatBase format = _channel.Format; - - // The index-data block holds exactly IndexBlockFormat.MaxColumns slots, with no count and no - // continuation, so a wider index cannot be represented. TdefBuilder rejects this when a table is - // created with its indexes; this is the incremental path, where BuildIndexDataBlock would otherwise - // write the first ten and mark the rest unused — silently storing a different index from the one - // asked for, which ACE reads without complaint. ACE refuses instead: "Cannot have more than 10 - // fields in an index." - if (slots.Count > IndexBlockFormat.MaxColumns) - throw new NotSupportedException( - $"Cannot create index '{indexName}' on '{table.Name}' over {slots.Count} columns: " - + $"Jet/ACE stores at most {IndexBlockFormat.MaxColumns} fields in an index."); - - // Read the whole definition (stitching any existing continuation pages) so the surgical insert - // works in absolute coordinates; the old continuation pages are reused when we write it back. - (LibRed.IO.PageBuffer buf, IReadOnlyList existingContinuations) = ReadDefinition(table.DefinitionPage); - int existingRowCount = buf.ReadInt32(format.TdefRowCountOffset); - - // A unique index over rows that already exist has to be rejected if those rows aren't unique, and a - // required one if any row leaves a key column NULL — ACE refuses the DDL for both. Done *here*, before a - // single byte of the TDEF moves, rather than during the back-fill: this path is not transactional, so a - // failure discovered mid-back-fill would leave the index committed to the TDEF and half-populated. - // Scanning first means a rejected CREATE INDEX leaves the file exactly as it was. - if ((unique || required) && existingRowCount != 0) - EnsureExistingRowsFitIndex(table, indexName, slots, unique, required); - - int dataCount = buf.ReadInt32(format.TdefIndexCountOffset); - int logicalCount = buf.ReadInt32(format.TdefLogicalIndexCountOffset); - int colCount = buf.ReadUInt16(format.TdefColumnCountOffset); - - // A plain index or an outgoing FK adds one of each. Checked before a byte moves, like the - // duplicate-key scan above, so a rejection leaves the file exactly as it was. - EnsureIndexCapacity(table.Name, $"index '{indexName}'", dataCount + 1, logicalCount + 1); - - // Walk the TDEF regions: stats -> column descriptors -> column names -> data blocks -> info blocks. - int afterStats = format.TdefRealIndexBlockOffset + dataCount * format.RealIndexEntrySize; - int pos = afterStats + colCount * format.ColumnDescriptorSize; - for (int i = 0; i < colCount; i++) pos += 2 + buf.ReadUInt16(pos); - int afterColumns = pos; // start of the data blocks - int afterDataBlocks = afterColumns + dataCount * IndexBlockFormat.DataBlockSize; - int infoStart = afterDataBlocks; - - // Existing logical blocks and names, plus the max index_num, so the new block gets a fresh number. - int namePos = infoStart + logicalCount * IndexBlockFormat.InfoBlockSize; - var blocks = new List(logicalCount + 1); - int maxNum = -1; - for (int i = 0; i < logicalCount; i++) - { - byte[] block = buf.Slice(infoStart + i * IndexBlockFormat.InfoBlockSize, IndexBlockFormat.InfoBlockSize).ToArray(); - maxNum = Math.Max(maxNum, System.Buffers.Binary.BinaryPrimitives.ReadInt32LittleEndian(block.AsSpan(4, 4))); - blocks.Add(block); - } - var names = new List(logicalCount + 1); - var nameBytes = new List(logicalCount + 1); - for (int i = 0; i < logicalCount; i++) - { - int len = buf.ReadUInt16(namePos); - nameBytes.Add(buf.Slice(namePos, 2 + len).ToArray()); - names.Add(System.Text.Encoding.Unicode.GetString(buf.Slice(namePos + 2, len))); - namePos += 2 + len; - } - - int defEnd = buf.ReadInt32(format.TdefLengthOffset); - byte[] lvalRegion = buf.Slice(namePos, defEnd - namePos).ToArray(); // §3.3.2 list + 0xFFFF terminator - int lvalCount = (lvalRegion.Length - 2) / 10; // 10 bytes per entry, then 0xFFFF - - // Allocate the new index's root (empty leaf) and its usage-map row (appended after existing rows). - int rootPage = _allocator.Allocate(); - WriteEmptyLeafIndexPage(format, rootPage, owner: table.DefinitionPage); - int usageMapPage = ReadInt24(buf, format.TdefOwnedPagesOffset + 1); - // Where the new index's usage map goes. The primary page holds the table's two maps, one row per - // index, then two rows per long-value column — but only for the columns that FIT; create spills the - // rest onto dedicated pages. So on a wide memo table the row this formula names is past the end. - // - // ACE's answer is not to squeeze it in but to spill, exactly as it does for the columns: measured on - // a 40-memo table, CREATE INDEX leaves the full 57-row primary page alone and puts the new index's - // map at row 0 of a page of its own (WideMemoUsageMapProbeTests). Match that. - var primaryMap = new DataPage(); - primaryMap.Read(_channel.ReadPage(usageMapPage), format); - int primaryFree = BinaryPrimitives.ReadUInt16LittleEndian( - _channel.ReadPage(usageMapPage).Span.Slice(format.DataFreeSpaceOffset, 2)); - - // Clamped, because after a DROP INDEX left an orphaned row the formula lands below the row count and - // that slot is meant to be reused rather than a duplicate appended. - int newIndexUsageRow = Math.Min(2 + lvalCount * 2 + dataCount, primaryMap.Rows.Count); - bool ownPage = newIndexUsageRow == primaryMap.Rows.Count - && primaryFree < UsageMapRecordLength + 2; - if (ownPage) - { - usageMapPage = _allocator.Allocate(); - WriteUsageMaps(format, usageMapPage, mapCount: 1); - newIndexUsageRow = 0; - } - // Append the new index's (empty) usage-map row, preserving every existing record. The data maps are - // empty on an empty table but the *existing indexes'* maps already carry their root bits (set at - // creation), so we must not rewrite the page from scratch even when the table has no rows. A page of - // its own already has the row, written empty above. - if (!ownPage) AppendEmptyUsageMapRow(format, usageMapPage, newIndexUsageRow); - - // Record this index's own root, as Access does at CREATE INDEX (the empty root is the index's sole - // page until it splits, after which IndexWriter adds each page it allocates). - new UsageMapWriter(_channel).SetBit(newIndexUsageRow, usageMapPage, rootPage, set: true); - - // Assemble the new definition: header + existing stats, a new stats block, columns + names + - // existing data blocks, the new data block, then the logical blocks (new one inserted, name-sorted) - // and their names, and finally the unchanged long-value region. - byte[] newData = BuildIndexDataBlock(slots, rootPage, newIndexUsageRow, usageMapPage, unique, required, ignoreNulls); - byte[] newInfo = buildInfo(maxNum + 1, dataCount); - - int k = names.Count(n => string.CompareOrdinal(n, indexName) < 0); // name-sorted insert position - blocks.Insert(k, newInfo); - nameBytes.Insert(k, EncodeName(indexName)); - - int newDefEnd = infoStart + IndexBlockFormat.DataBlockSize // one new data block shifts info start - + blocks.Count * IndexBlockFormat.InfoBlockSize + nameBytes.Sum(n => n.Length) + lvalRegion.Length - + format.RealIndexEntrySize; // one new stats block at the front - - // Build the full definition buffer (may exceed one page — split across continuation pages below). - var def = new byte[newDefEnd]; - var src = buf.Span; - int w = 0; - void Append(ReadOnlySpan s) { s.CopyTo(def.AsSpan(w)); w += s.Length; } - - Append(src[..afterStats]); // header + existing stats blocks - Append(new byte[format.RealIndexEntrySize]); // new (zero) stats block - Append(src[afterStats..afterDataBlocks]); // columns + names + existing data blocks - Append(newData); // new index-data block - foreach (byte[] b in blocks) Append(b); // logical blocks (new inserted, sorted) - foreach (byte[] n in nameBytes) Append(n); // their names, same order - Append(lvalRegion); // §3.3.2 list + terminator (unchanged) - - // Bump the two index counts and the definition length in the header. - System.Buffers.Binary.BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefIndexCountOffset, 4), dataCount + 1); - System.Buffers.Binary.BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefLogicalIndexCountOffset, 4), logicalCount + 1); - System.Buffers.Binary.BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefLengthOffset, 4), newDefEnd); - - WriteDefinition(table.DefinitionPage, def, existingContinuations, rewrite: true); - _catalog.Invalidate(); - - // Back-fill the new (empty) index B-tree with an entry per existing row, so the index is complete. - if (existingRowCount != 0) - BackfillIndex(table.Name, indexName, ignoreNulls, validateUnique: false); - return maxNum + 1; - } - - /// - /// Throws if the table's existing rows cannot go into a would-be index: a duplicate key for a unique one, or - /// a NULL in any key column for a required one — a primary key or WITH DISALLOW NULL, which ACE refuses with - /// "Index or primary key cannot contain a Null value" (verified; an ADD COLUMN … PRIMARY KEY on a table that - /// already holds rows is the usual way to get there). Purely a read, touching nothing on disk, so it is safe - /// to call before the index exists. For uniqueness, rows with a null in any key column are exempt — Jet's - /// uniqueness is over the non-null keys only, so several rows may be null (verified vs ACE; see the - /// constraints page); a WITH IGNORE NULL index leaves them out of the B-tree altogether, so either way they - /// cannot collide. - /// - /// - /// Comparison is on the encoded key, not the raw values, which is deliberate: the encoding is what the - /// B-tree stores and it is collation-lossy, so "ABC" and "abc" share a key. That is precisely Access's - /// uniqueness domain, and it keeps this agreeing with the insert- and update-time checks, which compare - /// the same way via . - /// - private void EnsureExistingRowsFitIndex( - TableDef table, string indexName, IReadOnlyList<(int Id, bool Ascending)> slots, bool unique, bool required) - { - var keyColumns = slots - .Select(s => (Column: table.Columns.First(c => c.ColumnId == s.Id), s.Ascending)) - .ToArray(); - - var seen = new HashSet(); - foreach (object?[] values in new Table(_channel, table).Rows()) - { - if (keyColumns.Any(k => values[k.Column.Index] is null)) - { - if (required) - throw new InvalidOperationException( - $"Index or primary key cannot contain a Null value: a row of '{table.Name}' has no value for index '{indexName}'."); - continue; - } - if (unique && !seen.Add(Convert.ToHexString(IndexKeyEncoder.Encode(keyColumns, values)))) - throw new InvalidOperationException( - $"Cannot create unique index '{indexName}' on '{table.Name}': duplicate key values exist."); - } - } - - /// Populates a freshly added index over a table's existing rows: encodes every live row's key, - /// then hands the lot to , which sorts them and writes each B-tree page - /// once. Rows with a null in an IGNORE NULL index's key are skipped, matching the per-insert path. - /// - /// This used to insert the rows one at a time, which re-read, re-parsed and rebuilt a whole leaf page per - /// row: an index over 10,000 rows cost ~440 ms and allocated over a gigabyte, nearly all of it leaves - /// replaced by the next entry. Sorting first lets each leaf be filled and written once. - /// Uniqueness moves with it: on sorted keys a duplicate is an adjacent pair, so the per-row - /// descent is gone. The comparison is still on the encoded key, which - /// is Access's uniqueness domain (see ). - /// A built index's statistics are set as ACE sets them (verified, for CREATE INDEX, a foreign key's - /// backing index and an ALTER COLUMN's rebuild): the total entry count to the entries it now holds and the - /// unique entry count to its distinct keys — both from the rows present, not from any earlier history. - /// - private void BackfillIndex(string tableName, string indexName, bool ignoreNulls, bool validateUnique) - { - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found after adding the index."); - IndexDef index = table.Indexes.First(ix => string.Equals(ix.Name, indexName, StringComparison.OrdinalIgnoreCase)); - List<(byte[] Key, int Pointer, bool NullKey)> entries = IndexEntries(table, index, ignoreNulls); - - new IndexWriter(_channel, table).BulkBuild(index, entries, validateUnique && index.IsUnique); - SetBuiltStatistics(table, index, entries); - } - - /// The entries an index over the table's current rows holds: every live row's encoded key and row - /// pointer, less the rows an IGNORE NULL index leaves out. - private List<(byte[] Key, int Pointer, bool NullKey)> IndexEntries(TableDef table, IndexDef index, bool ignoreNulls) - { - var keyColumnIds = index.Columns.Select(c => c.Column.Index).ToArray(); - var entries = new List<(byte[] Key, int Pointer, bool NullKey)>(); - foreach ((RowId id, object?[] values) in new Table(_channel, table).Rows().WithIds()) - { - bool hasNullKey = keyColumnIds.Any(i => values[i] is null); - if (ignoreNulls && hasNullKey) continue; - entries.Add((IndexKeyEncoder.Encode(index.Columns, values), (id.Page << 8) | id.Row, hasNullKey)); - } - return entries; - } - - /// Sets a just-built index's statistics block as ACE sets it: total = the entries it holds, unique = - /// its distinct keys among them. - private void SetBuiltStatistics(TableDef table, IndexDef index, List<(byte[] Key, int Pointer, bool NullKey)> entries) => - WriteIndexStatistics(table, index, entries.Count, entries.Select(e => Convert.ToHexString(e.Key)).Distinct().Count()); - - /// Every logical index's statistics — its real index's total and unique entry counts — by index name. - /// The blocks sit on the definition's first page. - private Dictionary IndexStatistics(TableDef table) - { - byte[] tdef = _channel.ReadPage(table.DefinitionPage).Span.ToArray(); - var statistics = new Dictionary(StringComparer.OrdinalIgnoreCase); - foreach (IndexDef index in table.Indexes) - { - int at = _channel.Format.TdefRealIndexBlockOffset + index.RealIndexOrdinal * _channel.Format.RealIndexEntrySize; - statistics[index.Name] = (BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(at, 4)), - BinaryPrimitives.ReadInt32LittleEndian(tdef.AsSpan(at + 4, 4))); - } - return statistics; - } - - private void WriteIndexStatistics(TableDef table, IndexDef index, int total, int unique) - { - byte[] tdef = _channel.ReadPage(table.DefinitionPage).Span.ToArray(); - int at = _channel.Format.TdefRealIndexBlockOffset + index.RealIndexOrdinal * _channel.Format.RealIndexEntrySize; - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(at, 4), total); - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(at + 4, 4), unique); - _channel.WritePage(table.DefinitionPage, tdef); - } - - /// Appends one empty inline usage-map record (row ) to an existing - /// usage-map page, preserving every existing record. The new index tracks no pages here (IndexWriter - /// navigates the B-tree structurally), so an empty bitmap is correct. - private void AppendEmptyUsageMapRow(JetFormatBase format, int pageNumber, int newRow) - { - const int MapLength = 1 + 4 + 64; // inline type + start page + 64-byte bitmap (matches WriteUsageMaps) - byte[] page = _channel.ReadPage(pageNumber).Span.ToArray(); - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2)); - - // Reuse a row slot left orphaned by a prior index drop — DropConstraint/DropIndex leave the dropped - // index's usage-map row in place (matching ACE), and a re-added index takes the same slot number. Clear - // its bitmap in place rather than appending a duplicate. (Needed when a parent-side ALTER COLUMN rewrite - // drops and re-adds a child's foreign key.) - if (newRow < rowCount) - { - int existing = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + newRow * 2, 2)); - Array.Clear(page, existing, MapLength); - _channel.WritePage(pageNumber, page); - return; - } - if (rowCount != newRow) - throw new InvalidOperationException( - $"Usage-map page has {rowCount} rows; expected {newRow} before appending the new index's map."); - - int minOffset = format.PageSize; - for (int r = 0; r < rowCount; r++) - minOffset = Math.Min(minOffset, BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + r * 2, 2))); - - int newOffset = minOffset - MapLength; - // A record written below the directory would overwrite the slots themselves — which reads back later - // as a slot with offset 0 "outside the row heap", a long way from the cause. The caller is expected - // to spill onto a fresh page rather than get here; this is the backstop that keeps a mistake loud. - if (newOffset < format.DataRowDirectoryOffset + (rowCount + 1) * 2) - throw new InvalidOperationException( - $"Usage-map page {pageNumber} has no room for another record: {rowCount} rows already reach " - + $"offset {minOffset}. The new map belongs on a page of its own."); - Array.Clear(page, newOffset, MapLength); // inline type 0x00, start page 0, zero bitmap - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + newRow * 2, 2), (ushort)newOffset); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), (ushort)(rowCount + 1)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(newOffset - format.DataRowDirectoryOffset - (rowCount + 1) * 2)); - _channel.WritePage(pageNumber, page); - } - - /// - /// Adds a foreign key to an existing child table: a backing non-unique index over the child - /// columns carrying an outgoing-relationship block, an incoming block on the parent's TDEF, and the - /// MSysRelationships rows. The child index and parent block are written the same way inline-FK creation - /// does (verified byte-faithful vs ACE). A self-reference hosts both ends in the one table; FOREIGN KEY - /// NO INDEX is not yet handled. - /// - public void AddForeignKey(string childTable, RelationshipSpec fk) - { - JetName.Validate(fk.Name, "foreign key name"); - EnsureRelationshipNameFree(fk.Name); - TableDef child = _catalog.FindTable(childTable) - ?? throw new InvalidOperationException($"Table '{childTable}' was not found."); - if (fk.NoIndex) - throw new NotSupportedException("ALTER TABLE ADD FOREIGN KEY … NO INDEX is not supported yet."); - if (fk.UpdateSetNull) throw UpdateSetNullNotImplemented(); - - RejectOleIndexColumns(fk.Columns.Select(c => c.Column), n => child.FindColumn(n)?.Type); - fk = ResolvePrimaryKeyReference(fk, creatingTable: null); - EnsureSameDataTypes(fk, ColumnOf(child), ColumnOf(_catalog.FindTable(fk.ReferencedTable))); - - byte upd = fk.CascadeUpdate ? CascadeAction : NoCascadeAction; - byte del = fk.CascadeDelete ? CascadeAction : fk.DeleteSetNull ? SetNullAction : NoCascadeAction; - var slots = ResolveSlots(child, fk.Columns.Select(c => (c.Column, Ascending: true))); - - // A self-reference (child == parent) hosts both ends in the same TDEF: the outgoing block links to - // an incoming block whose number is one past the outgoing block's (mirrors inline self-ref creation). - if (string.Equals(fk.ReferencedTable, childTable, StringComparison.OrdinalIgnoreCase)) - { - int selfRefOrdinal = ReferencedOrdinalIn(child, fk); - int outNum = InsertIndex(child, fk.Name, slots, - unique: false, required: false, ignoreNulls: false, - (num, ord) => BuildOutgoingInfoBlock(num, ord, FkTypeOutgoing, num + 1, child.DefinitionPage, upd, del)); - AddIncomingRelationshipBlock(new IncomingRelationship( - child.DefinitionPage, outNum + 1, selfRefOrdinal, (uint)outNum, child.DefinitionPage, upd, del)); - AddRelationshipRows(childTable, fk); - _catalog.Invalidate(); - return; - } - - (int parentPage, int refOrdinal, int parentNum) = ResolveParent(fk, child.DefinitionPage); - int childBlockNum = InsertIndex(child, fk.Name, slots, - unique: false, required: false, ignoreNulls: false, - (num, ord) => BuildOutgoingInfoBlock(num, ord, FkTypeOutgoing, parentNum, parentPage, upd, del)); - - AddIncomingRelationshipBlock(new IncomingRelationship( - parentPage, parentNum, refOrdinal, (uint)childBlockNum, child.DefinitionPage, upd, del)); - AddRelationshipRows(childTable, fk); - _catalog.Invalidate(); - } - - /// - /// Drops a named FOREIGN KEY constraint, byte-faithfully with ACE: removes the child's backing index - /// (its stats + index-data + outgoing info blocks + name) and the parent's incoming info block from the - /// two TDEFs, frees the index's B-tree root page back to the global free map, and soft-deletes the - /// relationship's MSysRelationships rows (the usage-map page is left untouched — ACE leaves the - /// orphan map row). A self-reference hosts both ends in one TDEF. Returns false if no such relationship - /// exists on . - /// - public bool DropConstraint(string childTable, string name) - { - ForeignKey? rel = _catalog.Relationships.FirstOrDefault(r => - string.Equals(r.Name, name, StringComparison.OrdinalIgnoreCase) && - string.Equals(r.Table, childTable, StringComparison.OrdinalIgnoreCase)); - if (rel is null) return false; - - TableDef child = _catalog.FindTable(childTable) - ?? throw new InvalidOperationException($"Table '{childTable}' was not found."); - IndexDef? fkIndex = child.Indexes.FirstOrDefault(i => string.Equals(i.Name, name, StringComparison.OrdinalIgnoreCase)); - bool selfRef = string.Equals(rel.ReferencedTable, childTable, StringComparison.OrdinalIgnoreCase); - - if (fkIndex is not null) - { - TdefParts childParts = ParseTdef(child.DefinitionPage); - int outgoing = childParts.Logical.FindIndex(b => NameOf(b.Name).Equals(name, StringComparison.OrdinalIgnoreCase)); - int childBlockNum = BinaryPrimitives.ReadInt32LittleEndian(childParts.Logical[outgoing].Info.AsSpan(0x04, 4)); - - // Remove the FK index (data ordinal) + its outgoing info block from the child, plus — for a - // self-reference — the incoming block, which also lives here. - RemoveTdefBlocks(childParts, removeDataOrdinal: fkIndex.RealIndexOrdinal, removeLogical: b => - NameOf(b.Name).Equals(name, StringComparison.OrdinalIgnoreCase) || - (selfRef && IsIncomingBlockFor(b.Info, childBlockNum, child.DefinitionPage))); - WriteTdef(child.DefinitionPage, childParts); - - if (!selfRef) - { - TableDef parent = _catalog.FindTable(rel.ReferencedTable) - ?? throw new InvalidOperationException($"Table '{rel.ReferencedTable}' was not found."); - TdefParts parentParts = ParseTdef(parent.DefinitionPage); - RemoveTdefBlocks(parentParts, removeDataOrdinal: null, removeLogical: b => - IsIncomingBlockFor(b.Info, childBlockNum, child.DefinitionPage)); - WriteTdef(parent.DefinitionPage, parentParts); - } - - new PageAllocator(_channel).Release(fkIndex.RootPage); - } - - SoftDeleteRelationshipRows(name); - // Its MSysObjects object and permission rows go with it, as ACE removes them (verified). A relationship - // written before LibRed recorded the object has none. - if (FindObjectId(name, CatalogFormat.ObjectTypeRelationship) is { } objectId) - { - DeleteCatalogRows("MSysObjects", "Id", objectId); - DeleteCatalogRows("MSysACEs", "ObjectId", objectId); - } - _catalog.Invalidate(); - return true; - } - - /// - /// Drops a table — DROP TABLE table. Removes the object's MSysObjects row and its - /// MSysACEs permission rows (soft-delete, as ACE does), and frees the table's pages back to the - /// global free map when the database closes (verified vs ACE): every page of its indexes, its data and - /// long-value pages, the TDEF page and its continuation pages, a reference-form map's bitmap pages, and any - /// usage-map holder left with no live record. Returns false if the table doesn't exist. - /// - /// A table that is the child (referencing) side of relationships can be dropped directly: ACE - /// lets you drop the referencing table while the parent stays, so each such relationship is removed first - /// (via ). But a table still referenced as a parent by a surviving - /// child cannot be dropped — drop the referencing table (or the relationship) first. EF drops FKs before - /// tables, but database-first scaffolding cleanup drops child tables directly, which must work. - /// - public bool DropTable(string tableName) - { - TableDef? table = _catalog.FindTable(tableName); - if (table is null) return false; - - // Remove the relationships this table owns as the child (referencing) side. Materialize first — - // DropConstraint rewrites TDEFs and invalidates the catalog on each call. - foreach (ForeignKey rel in _catalog.Relationships - .Where(r => string.Equals(r.Table, tableName, StringComparison.OrdinalIgnoreCase)) - .ToList()) - DropConstraint(tableName, rel.Name); - - // A table still referenced by a surviving child (as the parent) cannot be dropped. - if (_catalog.Relationships.Any(r => - string.Equals(r.ReferencedTable, tableName, StringComparison.OrdinalIgnoreCase))) - throw new InvalidOperationException( - $"Cannot drop table '{tableName}': it is referenced by a relationship — drop the referencing table first."); - - // Re-fetch: DropConstraint above rewrote this table's TDEF (removed FK indexes) and invalidated the catalog. - table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found."); - - int tdefPage = table.DefinitionPage; - var allocator = new PageAllocator(_channel); - var maps = new UsageMap(_channel, table); - - // Collect before freeing anything: the pointers are read out of the TDEF, which this method frees. - var owned = new HashSet(); - // Through the chain reader: a wide table's definition spans continuation pages, and parsing only the - // first one throws on the declared length. (The map POINTERS below are at fixed offsets inside the - // first page, so those are read from it directly, as UsageMap does.) - var definition = new TableDefinitionPage(); - definition.Read(_channel, tdefPage); - - // The map RECORDS live as rows on owner-zero data pages, and each is retired in turn — its bits cleared - // first where ACE clears them, then its row reclaimed, which slides every record below it up the page. - // The order is ACE's and it shows on disk: a slide leaves a copy of the records it moved in the space - // they vacated, so retiring the same records in another order leaves different bytes behind. Measured - // by whole-file diff against ACE drops: each long-value column's owned then free map, bits cleared; - // then each index's owned map in index order, bits cleared; then the table's own owned map, bits - // cleared, and its free map, whose bits stay. - var retire = new List(); - - foreach (ColumnDef column in table.Columns) - QueueLongValueMaps(definition, column, maps, owned, retire); - - // Each real index keeps its B-tree pages in its own owned map, whose (row, page) pointer sits in its data - // block. Freeing only the root strands every other page of a multi-level index, and leaving the map's - // record live keeps its holder from ever being freed. - foreach (byte[] block in ParseTdef(tdefPage).DataBlocks) - { - int at = IndexBlockFormat.UsageMapRowOffset; - (int Row, int Page) map = (block[at], block[at + 1] | block[at + 2] << 8 | block[at + 3] << 16); - if (map.Page == 0) continue; - List pages = maps.PagesInMap(map.Row, map.Page).ToList(); - owned.UnionWith(pages); - retire.Add((map, [map], pages)); - } - foreach (IndexDef index in table.Indexes.Where(i => i.RootPage > 0).GroupBy(i => i.RootPage).Select(g => g.First())) - owned.Add(index.RootPage); - List dataPages = maps.DataPages().ToList(); - owned.UnionWith(dataPages); - owned.Add(tdefPage); - // A wide table's definition continues on further pages. ACE frees them too, and leaves every byte of - // them alone — only the first page is marked released. - owned.UnionWith(TdefChainReader.Read(_channel, tdefPage).ContinuationPages); - - PageBuffer tdef = _channel.ReadPage(tdefPage); - (int Row, int Page) dataOwned = (tdef.ReadByte(_channel.Format.TdefOwnedPagesOffset), - tdef.ReadInt24(_channel.Format.TdefOwnedPagesOffset + 1)); - retire.Add((dataOwned, [dataOwned], dataPages)); - retire.Add(((tdef.ReadByte(_channel.Format.TdefFreePagesOffset), - tdef.ReadInt24(_channel.Format.TdefFreePagesOffset + 1)), [], [])); - - RetireMapRecords(retire, maps, owned); - - // Access marks the released definition page itself: its type byte becomes 0x08 and nothing else on - // the page changes, so the old definition is still sitting there when Compact comes to reclaim it. - // Measured across an ACE DROP TABLE: exactly one byte of the 4,096 differs. Only the TDEF is marked — - // the data, long-value and map-holder pages ACE frees keep their 0x01. - byte[] released = _channel.ReadPage(tdefPage).Span.ToArray(); - released[0] = (byte)PageType.ReleasedTableDefinition; - _channel.WritePage(tdefPage, released); - - foreach (int page in owned) - allocator.Release(page); // reusable only after this handle closes, as ACE holds them - - DeleteCatalogRows("MSysObjects", "Id", tdefPage); - DeleteCatalogRows("MSysACEs", "ObjectId", tdefPage); - _catalog.Invalidate(); - return true; - } - - /// - /// Queues a long-value column's two usage-map records for , owned then free, and - /// adds the pages its owned map records to . A Memo/OLE (or calculated long) column owns - /// its LVAL pages through a PER-COLUMN usage map, whose pointer sits in the TDEF keyed by column id. Those pages - /// are not in the table's data-page map, so freeing only the data pages leaves every long value stranded — and - /// for a memo-heavy table that is nearly the whole table. Measured against ACE: dropping a 60-row memo table - /// returned 123 pages through ACE and 2 through LibRed, the missing 121 being LVAL pages. Nothing is queued for - /// a column with no long-value maps. - /// - private static void QueueLongValueMaps( - TableDefinitionPage definition, ColumnDef column, UsageMap maps, HashSet owned, List retire) - { - definition.LongValueOwnedMaps.TryGetValue(column.ColumnId, out (int Row, int Page) map); - definition.LongValueFreeMaps.TryGetValue(column.ColumnId, out (int Row, int Page) columnFree); - if (map.Page != 0) - { - // Clear each page's bit on the way out, exactly as releasing a single value does: the record's - // bitmap bytes are zeroed before its row is retired. Except a page still in the free map, whose bit - // stays in both records — measured by whole-file diff of ACE drops: of an OLE column owning a chain, - // two full single-value pages and its current append page, every bit went but the append page's. - List pages = maps.PagesInMap(map.Row, map.Page).ToList(); - owned.UnionWith(pages); - HashSet stillFree = columnFree.Page != 0 ? maps.PagesInMap(columnFree.Row, columnFree.Page).ToHashSet() : []; - retire.Add((map, columnFree.Page != 0 ? [map, columnFree] : [map], pages.Where(p => !stillFree.Contains(p)).ToList())); - } - if (columnFree.Page != 0) retire.Add((columnFree, [], [])); - } - - /// - /// Takes usage-map records off their pages the way ACE does, in the order given — tombstone the slot, slide the - /// rows below it up, return the bytes to the page's free count — rather than leaving dead maps behind. On a - /// shared holder that is the whole fix: the page survives and must not keep records for a map that no longer - /// exists. Each record's Clear maps first have its Pages cleared. Adds to - /// the pages to release: a reference-form record's bitmap pages, and every holder left with no live row. - /// - private void RetireMapRecords(IEnumerable retire, UsageMap maps, HashSet owned) - { - var usageMaps = new UsageMapWriter(_channel); - var holders = new List(); - var retired = new HashSet<(int Row, int Page)>(); - foreach (((int Row, int Page) map, IReadOnlyList<(int Row, int Page)> clear, IReadOnlyList pages) in retire) - { - if (map.Page <= 1 || map.Page >= _channel.PageCount || !retired.Add(map)) continue; - foreach ((int Row, int Page) cleared in clear) - foreach (int page in pages) - usageMaps.SetBit(cleared.Row, cleared.Page, page, set: false); - - // A reference-form record keeps its bitmap on dedicated pages. ACE zeroes each one's bitmap — even - // for the table's own owned map, whose bits an inline record keeps — leaves its header, and frees it. - foreach (int bitmapPage in maps.BitmapPagesOf(map.Row, map.Page)) - { - byte[] bitmap = _channel.ReadPage(bitmapPage).Span.ToArray(); - bitmap.AsSpan(4).Clear(); - _channel.WritePage(bitmapPage, bitmap); - owned.Add(bitmapPage); - } - - byte[] holderBytes = _channel.ReadPage(map.Page).Span.ToArray(); - RowInserter.ReclaimRow(_channel.Format, holderBytes, map.Row); - _channel.WritePage(map.Page, holderBytes); - if (!holders.Contains(map.Page)) holders.Add(map.Page); - } - - // ACE frees a holder once the dropped records were the only thing on it — measured: for a one-memo-column - // table ACE returned the long-value map's holder. A holder can carry records for several columns or tables - // as separate rows, so releasing one that still serves another map would hand away a live page: corruption - // rather than a leak. Hence the holder goes only when no live row is left on it. - foreach (int holderPage in holders) - { - var holder = new DataPage(); - holder.Read(_channel.ReadPage(holderPage), _channel.Format); - bool live = false; - for (int row = 0; row < holder.RowCount && !live; row++) - live = !holder.Rows[row].IsDeleted && holder.Rows[row].Length > 0; - if (!live) owned.Add(holderPage); - } - } - - /// - /// Drops a view or stored procedure — DROP VIEW name / DROP PROCEDURE name. Both are a - /// type-5 MSysObjects object; ACE's two statements are interchangeable (verified: DROP VIEW works - /// on a procedure and vice versa), so this handles either. The inverse of ViewCreator: deletes the - /// object's MSysObjects row, its MSysQueries rows, and its two MSysACEs permission rows (index entries - /// removed, not just soft-deleted). No pages to free (a query owns none — its MSysQueries rows live on the - /// shared MSysQueries pages). Returns false if no such query object exists. - /// - public bool DropQueryObject(string name) - { - if (FindObjectId(name, StoredQueryFormat.ObjectTypeQuery) is not { } objId) return false; - - DeleteCatalogRows("MSysObjects", "Id", objId); - DeleteCatalogRows("MSysQueries", "ObjectId", objId); - DeleteCatalogRows("MSysACEs", "ObjectId", objId); - _catalog.Invalidate(); - return true; - } - - /// The MSysObjects id of the object of named , - /// or null when there is none. - private int? FindObjectId(string name, short type) - { - TableDef mo = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIdx = (mo.FindColumn("Id") ?? throw new InvalidOperationException("MSysObjects is missing 'Id'.")).Index; - int nameIdx = (mo.FindColumn("Name") ?? throw new InvalidOperationException("MSysObjects is missing 'Name'.")).Index; - int typeIdx = (mo.FindColumn("Type") ?? throw new InvalidOperationException("MSysObjects is missing 'Type'.")).Index; - - foreach (object?[] values in new Table(_channel, mo).Rows()) - if (string.Equals(values[nameIdx] as string, name, StringComparison.OrdinalIgnoreCase) - && Convert.ToInt16(values[typeIdx] ?? (short)0, CultureInfo.InvariantCulture) == type) - return Convert.ToInt32(values[idIdx], CultureInfo.InvariantCulture); - return null; - } - - - /// - /// Renames a table — ALTER TABLE … RENAME TO. Measured against ACE (see RenameFanOutProbeTest): - /// only two things move — the object's MSysObjects.Name, and the by-name table references in - /// MSysRelationships. Everything else is deliberately left alone, matching ACE exactly: - /// - /// indexes (including the PK) keep their own names and need no fixup — they reference the table by id; - /// the relationship keeps its own name and its enforcement; - /// stored queries/views are left dangling — ACE does not rewrite MSysQueries (Name - /// AutoCorrect is an Access application feature), and "helpfully" fixing them would diverge from Jet. - /// - /// Returns false if no such table exists; throws if the new name is already taken. - /// - public bool RenameTable(string oldName, string newName) - { - TableDef? table = _catalog.FindTable(oldName); - if (table is null) return false; - // The same names Create refuses. A rename reaches the identical bytes by a different route, so - // validating only on the way in left it open: renaming a COLUMN to over 64 characters makes the - // whole database unreadable to ACE ("Unrecognized database format"), which is exactly what the - // create-side check exists to prevent (RenameNameValidationAccessTests). - JetName.Validate(newName, "table name"); - // The table being renamed is not a collision with itself: renaming to the same name is a no-op that ACE - // allows (and EF's schema "move" degrades to exactly that on a schema-less engine), as is a case-only - // change. Both verified — RenameFanOutProbeTest. - if (ObjectNameExists(newName, exceptObjectId: table.DefinitionPage)) - throw new SchemaObjectExistsException( - $"ALTER TABLE '{oldName}' RENAME TO '{newName}': a table or query named '{newName}' already exists.", - newName); - - RenameCatalogObject(table.DefinitionPage, newName); - RepointRelationshipTables(oldName, newName); - return true; - } - - /// Sets the Name of the MSysObjects row whose Id is this table's TDEF page. - private void RenameCatalogObject(int tdefPage, string newName) - { - TableDef def = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIndex = ColumnIndexOf(def, "Id"); - int nameIndex = ColumnIndexOf(def, "Name"); - var table = new Table(_channel, def); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds() - .Where(r => r.Values[idIndex] is not null - && Convert.ToInt32(r.Values[idIndex], CultureInfo.InvariantCulture) == tdefPage) - .ToList()) - { - SetCatalogValues(table, def, id, values, (nameIndex, newName)); - } - } - - /// Repoints every relationship that names on either side. Both the - /// child (szObject) and parent (szReferencedObject) are stored by name, and a self-reference - /// names the table twice — hence updating both columns in one pass over each row. - private void RepointRelationshipTables(string oldName, string newName) - { - TableDef? def = _catalog.FindTable("MSysRelationships"); - if (def is null) return; // a database with no relationships has no catalog table to fix up - - int childIndex = ColumnIndexOf(def, "szObject"); - int parentIndex = ColumnIndexOf(def, "szReferencedObject"); - var table = new Table(_channel, def); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds().ToList()) - { - var updates = new List<(int Column, object? Value)>(2); - if (NameMatches(values[childIndex], oldName)) updates.Add((childIndex, newName)); - if (NameMatches(values[parentIndex], oldName)) updates.Add((parentIndex, newName)); - if (updates.Count > 0) - SetCatalogValues(table, def, id, values, updates.ToArray()); - } - } - - /// - /// Renames a column — ALTER TABLE … RENAME COLUMN … TO. Measured against ACE (see - /// RenameFanOutProbeTest): the name in the TDEF's column region moves, MSysRelationships' - /// by-name column references are repointed, and the column's LvProp property block is re-owned so it - /// keeps its DEFAULT. Nothing else moves — indexes reference columns by id, so they keep their own - /// names and need no fixup, and stored queries are left dangling exactly as ACE leaves them. - /// Returns false if no such column exists; throws if the new name is already used on the table. - /// - public bool RenameColumn(string tableName, string oldName, string newName) - { - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found."); - ColumnDef? col = table.Columns.FirstOrDefault(c => string.Equals(c.Name, oldName, StringComparison.OrdinalIgnoreCase)); - if (col is null) return false; - JetName.Validate(newName, "column name"); // see RenameTable: unchecked, this corrupts the file - if (table.Columns.Any(c => string.Equals(c.Name, newName, StringComparison.OrdinalIgnoreCase))) - throw new InvalidOperationException( - $"ALTER TABLE '{tableName}' RENAME COLUMN '{oldName}' TO '{newName}': the table already has a column named '{newName}'."); - - TdefParts parts = ParseTdef(table.DefinitionPage); // stitches continuation pages for a multi-page TDEF - RenameColumnInParts(parts, table.Columns.Count, col.Index, newName, _channel.Format); - WriteTdef(table.DefinitionPage, parts); - RenameColumnProperties(table.DefinitionPage, oldName, newName); - RepointRelationshipColumns(tableName, oldName, newName); - _catalog.Invalidate(); - return true; - } - - /// Replaces the name entry of the column at in the column region. - /// The descriptors are fixed-size and untouched; only the variable-length name pool is rebuilt (a - /// different-length name shifts every following entry), and the column count is unchanged. - private static void RenameColumnInParts( - TdefParts parts, int colCount, int renameIndex, string newName, JetFormatBase format) - { - int descSize = format.ColumnDescriptorSize; - ReadOnlySpan cols = parts.Columns; - - var descriptors = new List(colCount); - for (int i = 0; i < colCount; i++) - descriptors.Add(cols.Slice(i * descSize, descSize).ToArray()); - - int np = colCount * descSize; - var names = new List(colCount); - for (int i = 0; i < colCount; i++) - { - int len = BinaryPrimitives.ReadUInt16LittleEndian(cols.Slice(np, 2)); - names.Add(cols.Slice(np, 2 + len).ToArray()); - np += 2 + len; - } - - byte[] nameBytes = System.Text.Encoding.Unicode.GetBytes(newName); - byte[] entry = new byte[2 + nameBytes.Length]; - BinaryPrimitives.WriteUInt16LittleEndian(entry.AsSpan(0, 2), (ushort)nameBytes.Length); - nameBytes.CopyTo(entry, 2); - names[renameIndex] = entry; - - var blob = new List(parts.Columns.Length); - foreach (byte[] d in descriptors) blob.AddRange(d); - foreach (byte[] n in names) blob.AddRange(n); - parts.Columns = [.. blob]; - } - - /// Re-owns the renamed column's extended-property block in its table's MSysObjects.LvProp - /// blob, so its DefaultValue/Required/validation survive the rename (ACE does this — verified). No-op when - /// the column had no properties. - private void RenameColumnProperties(int tdefPage, string oldName, string newName) - { - TableDef msys = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIndex = ColumnIndexOf(msys, "Id"); - ColumnDef lvProp = msys.FindColumn("LvProp") - ?? throw new InvalidOperationException("MSysObjects is missing 'LvProp'."); - var table = new Table(_channel, msys); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds()) - { - if (values[idIndex] is null || Convert.ToInt32(values[idIndex], CultureInfo.InvariantCulture) != tdefPage) continue; - if (values[lvProp.Index] is not byte[] { Length: > 0 } blob) return; - - byte[] renamed = RewriteCalculatedReferences( - PropertyBlob.RenameOwner(blob, oldName, newName), oldName, newName); - // Nothing owned by, or referring to, this column — leave the blob exactly as it was. - if (renamed.AsSpan().SequenceEqual(blob)) return; - - byte[] descriptor = new RowInserter(_channel, msys).StorePackedLongValue(lvProp.ColumnId, renamed); - values[lvProp.Index] = new LongValueDescriptor(descriptor); - table.Update(id, values, new HashSet { lvProp.Index }); - return; - } - } - - /// The calculated columns of whose expression reads - /// . A malformed expression counts as reading nothing rather than throwing: - /// refusing to drop is a safeguard, and it must not turn into a refusal to drop anything at all because - /// some other column's expression cannot be parsed. - private static List CalculatedColumnsReading(TableDef table, ColumnDef column) - { - var dependents = new List(); - foreach (ColumnDef candidate in table.Columns) - { - if (!candidate.IsCalculated || candidate.CalculatedExpression is null) continue; - try - { - if (CalculatedValue.ReferencedIndexes(candidate, table.Columns).Contains(column.Index)) - dependents.Add(candidate.Name); - } - catch (Calculated.CalculatedExpressionException) { /* unparseable: reads nothing we can prove */ } - } - return dependents; - } - - /// Repoints every calculated Expression in the blob that READS the renamed column. - /// moves the renamed column's own properties; this is about the - /// OTHER columns that mention it, which nothing else would fix. Measured: without it a rename leaves - /// [Qty]*2 pointing at a column that no longer exists, and ACE fails every read of the calculated - /// column — a table broken by an operation that named a different column entirely. - private static byte[] RewriteCalculatedReferences(byte[] blob, string oldName, string newName) - { - var properties = PropertyBlob.Read(blob).ToList(); - bool changed = false; - for (int i = 0; i < properties.Count; i++) - { - if (properties[i].Name != PropertyBlob.ExpressionProperty) continue; - string rewritten = Calculated.CalculatedExpression.RenameColumnReference( - properties[i].Value, oldName, newName); - if (rewritten == properties[i].Value) continue; - // Drop RawValue so the new text is encoded rather than the original bytes replayed. - properties[i] = properties[i] with { Value = rewritten, RawValue = null }; - changed = true; - } - return changed - ? PropertyBlob.Write(properties, blob.Length >= 4 ? blob.AsSpan(0, 4) : default) - : blob; - } - - /// Repoints every relationship that names this column, on whichever side owns it. Unlike a table - /// name, a column name is only unique within its table, so each side is matched on its table name too. - private void RepointRelationshipColumns(string tableName, string oldName, string newName) - { - TableDef? def = _catalog.FindTable("MSysRelationships"); - if (def is null) return; - - int childTable = ColumnIndexOf(def, "szObject"); - int childColumn = ColumnIndexOf(def, "szColumn"); - int parentTable = ColumnIndexOf(def, "szReferencedObject"); - int parentColumn = ColumnIndexOf(def, "szReferencedColumn"); - var table = new Table(_channel, def); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds().ToList()) - { - var updates = new List<(int Column, object? Value)>(2); - if (NameMatches(values[childTable], tableName) && NameMatches(values[childColumn], oldName)) - updates.Add((childColumn, newName)); - if (NameMatches(values[parentTable], tableName) && NameMatches(values[parentColumn], oldName)) - updates.Add((parentColumn, newName)); - if (updates.Count > 0) - SetCatalogValues(table, def, id, values, updates.ToArray()); - } - } - - /// MSysObjects.Type for a table object (queries use ). - private const short ObjectTypeTable = 1; - - /// - /// Whether any table or saved query already uses this name. Access keeps tables and queries in a - /// single namespace: ACE rejects renaming a table onto either (verified — RenameFanOutProbeTest), - /// even though they live in different MSysObjects containers, so the unique (ParentId, Name) index - /// would not catch a table/query collision on its own. Scanned straight from MSysObjects rather than - /// the catalog's reconstructed Views/ActionQueries, which omit queries LibRed can't rebuild. - /// - private bool ObjectNameExists(string name, int exceptObjectId) - { - TableDef mo = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIndex = ColumnIndexOf(mo, "Id"); - int nameIndex = ColumnIndexOf(mo, "Name"); - int typeIndex = ColumnIndexOf(mo, "Type"); - - foreach (object?[] values in new Table(_channel, mo).Rows()) - { - if (!NameMatches(values[nameIndex], name)) continue; - // Skip the object being renamed — it can't collide with itself (same-name and case-only renames). - if (values[idIndex] is not null - && Convert.ToInt32(values[idIndex], CultureInfo.InvariantCulture) == exceptObjectId) continue; - short type = Convert.ToInt16(values[typeIndex] ?? (short)0, CultureInfo.InvariantCulture); - if (type is ObjectTypeTable or StoredQueryFormat.ObjectTypeQuery) return true; - } - - return false; - } - - private static bool NameMatches(object? value, string name) => - value is string s && string.Equals(s, name, StringComparison.OrdinalIgnoreCase); - - private static int ColumnIndexOf(TableDef def, string column) => - (def.FindColumn(column) ?? throw new InvalidOperationException($"'{def.Name}' is missing '{column}'.")).Index; - - /// Rewrites some columns of one catalog row, keeping any index whose key covers a changed column in - /// step — MSysObjects is uniquely indexed on (ParentId, Name), so a rename has to move that entry rather - /// than just overwrite the value. - private static void SetCatalogValues( - Table table, TableDef def, RowId id, object?[] values, params (int Column, object? Value)[] updates) - { - var newValues = (object?[])values.Clone(); - var changed = new HashSet(); - foreach ((int column, object? value) in updates) - { - newValues[column] = value; - changed.Add(column); - } - - foreach (IndexDef index in def.Indexes.Where(i => i.RootPage > 0).GroupBy(i => i.RootPage).Select(g => g.First())) - if (index.Columns.Any(c => changed.Contains(c.Column.Index))) - table.MoveIndexEntry(index, values, newValues, id); - - table.Update(id, newValues, changed); - } - - /// Deletes every row of whose equals - /// (the object id) — used to remove a dropped table's MSysObjects and MSysACEs - /// rows. A full delete: its index entries are removed (not just the slot soft-deleted) so, e.g., the - /// MSysObjects ParentIdName unique index doesn't retain a stale entry that would then reject - /// re-creating a same-named table. - private void DeleteCatalogRows(string catalogTable, string keyColumn, int keyValue) - { - TableDef t = _catalog.FindTable(catalogTable) - ?? throw new InvalidOperationException($"{catalogTable} catalog table was not found."); - int idx = (t.FindColumn(keyColumn) ?? throw new InvalidOperationException($"{catalogTable} is missing '{keyColumn}'.")).Index; - var table = new Table(_channel, t); - - var rows = table.Rows().WithIds() - .Where(r => r.Values[idx] is not null - && Convert.ToInt32(r.Values[idx], CultureInfo.InvariantCulture) == keyValue) - .ToList(); - foreach ((RowId id, object?[] values) in rows) - { - foreach (IndexDef index in t.Indexes.Where(i => i.RootPage > 0).GroupBy(i => i.RootPage).Select(g => g.First())) - table.RemoveIndexEntry(index, values, id); - table.Delete(id); - } - } - - /// - /// Drops a secondary/unique/primary index — DROP INDEX index ON table. Byte-faithful with ACE - /// (probed): remove the index's 12-byte stats block, 52-byte index-data block, and its 28-byte logical - /// info block + name from the TDEF (decrementing counts and the data-ordinal ref of any block past it), - /// and free its B-tree root page back to the global free map — the same index-removal path as DROP - /// CONSTRAINT, minus the relationship linkage. A secondary index lives only in the TDEF (no MSys row). - /// Returns false if no such index exists. Throws if the index backs a relationship (ACE rejects that — - /// drop the relationship first) or the TDEF is multi-page. (PK and unique indexes ARE droppable.) - /// - public bool DropIndex(string tableName, string indexName) - { - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found."); - IndexDef? index = table.Indexes.FirstOrDefault(i => string.Equals(i.Name, indexName, StringComparison.OrdinalIgnoreCase)); - if (index is null) return false; - - if (IndexParticipatesInRelationship(table, index)) - throw new InvalidOperationException( - $"Cannot drop index '{indexName}': it is used in a relationship — drop the relationship first."); - - TdefParts parts = ParseTdef(table.DefinitionPage); // stitches continuation pages for a multi-page TDEF - RemoveTdefBlocks(parts, removeDataOrdinal: index.RealIndexOrdinal, - removeLogical: b => NameOf(b.Name).Equals(indexName, StringComparison.OrdinalIgnoreCase)); - WriteTdef(table.DefinitionPage, parts); - new PageAllocator(_channel).Release(index.RootPage); - _catalog.Invalidate(); - return true; - } - - /// True if the index IS a relationship's enforcement index — the one specific index ACE refuses - /// to drop while the relationship exists. This is NOT "any index over the relationship's columns": a - /// redundant same-columns secondary index is droppable, and EF relies on that (it creates an explicit - /// index, adds the FK, then drops the now-redundant explicit index). ACE-verified (ZzProbe): with a - /// relationship on POrd.CustomerId → PCust.Id, dropping a coincident IX_POrd_CustomerId / IX_PCust_Id - /// succeeds, but dropping the FK's own child index (named after the relationship) or the referenced PK - /// fails with "used in a relationship". - /// Two indexes are protected: on the child, the FK's backing index — named after the relationship, - /// as both ACE and create it; on the parent, the referenced key — the - /// unique/primary index over the referenced columns. - private bool IndexParticipatesInRelationship(TableDef table, IndexDef index) - { - var cols = index.Columns.Select(c => c.Column.Name).ToList(); - bool SameCols(IEnumerable other) => - other.OrderBy(x => x, StringComparer.OrdinalIgnoreCase) - .SequenceEqual(cols.OrderBy(x => x, StringComparer.OrdinalIgnoreCase), StringComparer.OrdinalIgnoreCase); - - // The parent side matches a unique/primary index only, which is exactly the set FindParentKeyIndex - // will accept as a parent key — ACE refuses a relationship over anything else, so no file this engine - // writes can have one. - return _catalog.Relationships.Any(r => - (string.Equals(r.Table, table.Name, StringComparison.OrdinalIgnoreCase) - && string.Equals(index.Name, r.Name, StringComparison.OrdinalIgnoreCase)) || - (string.Equals(r.ReferencedTable, table.Name, StringComparison.OrdinalIgnoreCase) - && (index.IsUnique || index.IsPrimaryKey) - && SameCols(r.Columns.Select(c => c.ReferencedColumn)))); - } - - /// - /// Adds a column — ALTER TABLE t ADD COLUMN c type. A metadata TDEF edit (probed vs ACE, the - /// inverse of DROP COLUMN): appends the column's 25-byte descriptor + name, gives it the next column id - /// from the 0x29 max-columns high-water (which keeps counting even past dropped ids), appends its - /// fixed offset (end of the fixed region) or variable index (current variable count), and bumps - /// ColumnCount (0x2D), the 0x29 high-water, and — for a variable column — VariableColumnCount (0x2B). - /// Existing rows are not rewritten; they read the new column as NULL via the null bitmap. Fully correct - /// on an empty table (new inserts include it); on a populated table the column is visible and old rows - /// read NULL. A memo/OLE column additionally gets its §3.3.2 usage-map entry (two empty maps appended to - /// the table's usage-map page — or, if that page is full, a dedicated map page, the fallback CREATE TABLE - /// uses on a wide table). Returns false if the column already exists. Multi-page TDEFs are handled. - /// - public bool AddColumn(string tableName, ColumnSpec spec, string? defaultValue = null) - { - JetName.Validate(spec.Name, "column name"); - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found."); - if (table.Columns.Any(c => string.Equals(c.Name, spec.Name, StringComparison.OrdinalIgnoreCase))) - return false; - if (table.Columns.Count >= MaxColumnsPerTable) - throw new NotSupportedException($"Table '{tableName}' already has {MaxColumnsPerTable} columns (Jet/ACE limit)."); - JetFormatBase format = _channel.Format; - // As on the create path, a calculated column with a Memo RESULT needs the long-value maps even though - // its declared type is Text — the value reaches a page through a descriptor either way (§3.4a). - bool isLongValue = spec.Type is JetDataType.Memo or JetDataType.Ole - || spec.CalculatedResultType is JetDataType.Memo; - TdefParts parts = ParseTdef(table.DefinitionPage); // stitches continuation pages for a multi-page TDEF - - int maxCols = BinaryPrimitives.ReadUInt16LittleEndian(parts.Header.AsSpan(format.TdefMaxColumnsOffset, 2)); - int varCount = BinaryPrimitives.ReadUInt16LittleEndian(parts.Header.AsSpan(format.TdefVariableColumnsOffset, 2)); - int colCount = BinaryPrimitives.ReadUInt16LittleEndian(parts.Header.AsSpan(format.TdefColumnCountOffset, 2)); - - // The 0x29 column-id high-water never decrements on DROP COLUMN, so once 255 ids have been handed out - // no further column can be added — even if the *live* count is lower (the guard above) — until the - // database is compacted (which renumbers and reclaims dropped ids). ACE enforces exactly this: create - // 255 columns, drop some, ADD COLUMN → "Too many fields defined." Mirror it rather than write a 256th - // id ACE can't represent. Verified vs ACE. - if (maxCols >= MaxColumnsPerTable) - throw new NotSupportedException( - $"Cannot add column '{spec.Name}' to '{tableName}': too many fields defined — {MaxColumnsPerTable} column ids " + - "have been used over this table's lifetime (Jet/ACE caps the id high-water; a compact is required to reclaim dropped ids)."); - - // The width limits Create enforces through TdefBuilder apply just as much to a column added later: - // the new column's own width, and what it does to the widest record the table can now hold. - RecordLayout.ValidateFieldWidth(spec.Name, spec.Type, spec.Length); - JetDataTypeVersions.EnsureStorable(spec.Type, _channel.Format.Version, spec.Name); - RecordLayout.ValidateRecordFits(tableName, - FixedBytes(table) + (spec.IsFixedLength && spec.Type != JetDataType.Boolean ? spec.Length : 0), - varCount + (spec.IsFixedLength ? 0 : 1), - maxCols + 1, - format); - - // Same single-counter rule the CREATE path and the promote path enforce; ADD COLUMN had neither. The - // rule is about the header's seed/increment pair, so complex columns — flagged 0x04 but allocated from - // 0x1C — are neither the existing counter nor a conflicting one. - if (spec.IsAutoNumber && spec.Type != JetDataType.Complex - && table.Columns.Any(c => c.IsAutoNumber && c.Type != JetDataType.Complex)) - throw new NotSupportedException( - $"Cannot add AutoNumber column '{spec.Name}': table '{table.Name}' already has one " - + "(Jet allows a single column to draw on the table's seed/increment counter)."); - - var newColumn = new ColumnDef - { - Name = spec.Name, - Type = spec.Type, - Index = colCount, - ColumnId = maxCols, // next id from the high-water (dropped ids are never reused) - Length = spec.Length, - // Boolean is fixed but occupies no data — the bit IS the value — so it must not advance the fixed - // offset. Every other computation of this quantity excludes it; this one did not, putting an added - // column one byte past where ACE puts it on a table whose only fixed columns are Booleans. - FixedOffset = spec.IsFixedLength - ? table.Columns.Where(c => c.IsFixedLength && c.Type != JetDataType.Boolean) - .Select(c => c.FixedOffset + c.Length).DefaultIfEmpty(0).Max() - : 0, - VariableIndex = spec.IsFixedLength ? -1 : varCount, - // Descriptor 0x07. A VARIABLE column carries its own variable index, which is the 0x2B high-water - // (measured vs ACE in VariableColumnHighWaterAccessTests — NOT the count of live variable columns, - // which is lower once one has been dropped), so leave it unset and let TdefBuilder use VariableIndex. - // A FIXED column carries the count of variable columns with a smaller id, the way the create path - // computes it; unset, the legacy fallback writes 0, the one value the spec says it must not be. - VariableTableIndex = spec.IsFixedLength - ? table.Columns.Count(c => !c.IsFixedLength && c.ColumnId < maxCols) - : -1, - IsFixedLength = spec.IsFixedLength, - IsAutoNumber = spec.IsAutoNumber, - Precision = spec.Precision, - Scale = spec.Scale, - IsNullable = spec.IsNullable, - Collation = spec.Type == JetDataType.FixedPoint ? Collation.GeneralLegacy : _collation, - // Without this the descriptor gets no 0xC0 and the column reads back as an ordinary one: its - // Expression and ResultType properties are written, nothing looks at them, and every row stores - // NULL where the computed value should be. - IsCalculated = spec.CalculatedExpression is not null, - CalculatedExpression = spec.CalculatedExpression, - CalculatedResultType = spec.CalculatedResultType, - }; - - AppendColumnToParts(parts, colCount, TdefBuilder.BuildColumnDescriptor(newColumn, format), spec.Name, format); - - BinaryPrimitives.WriteUInt16LittleEndian(parts.Header.AsSpan(format.TdefColumnCountOffset, 2), (ushort)(colCount + 1)); - BinaryPrimitives.WriteUInt16LittleEndian(parts.Header.AsSpan(format.TdefMaxColumnsOffset, 2), (ushort)(maxCols + 1)); - if (!spec.IsFixedLength) - BinaryPrimitives.WriteUInt16LittleEndian(parts.Header.AsSpan(format.TdefVariableColumnsOffset, 2), (ushort)(varCount + 1)); - - // A memo/OLE column needs a §3.3.2 usage-map entry (its owned + free page maps). ACE appends the two - // maps to the table's existing usage-map page right after the data/index maps (verified), and adds the - // 10-byte entry before the list's 0xFFFF terminator. - int lvMapPage = 0, lvUsedRow = 0, lvFreeRow = 0; - bool lvDedicated = false; - if (isLongValue) - { - int o = format.TdefOwnedPagesOffset + 1; - int primaryPage = parts.Header[o] | (parts.Header[o + 1] << 8) | (parts.Header[o + 2] << 16); - byte[] primaryBytes = _channel.ReadPage(primaryPage).Span.ToArray(); - int primaryFree = BinaryPrimitives.ReadUInt16LittleEndian(primaryBytes.AsSpan(format.DataFreeSpaceOffset, 2)); - - if (primaryFree >= 2 * (UsageMapRecordLength + 2)) - { - // Room on the table's usage-map page — append the two maps there (as ACE does). - lvMapPage = primaryPage; - lvUsedRow = BinaryPrimitives.ReadUInt16LittleEndian(primaryBytes.AsSpan(format.DataRowCountOffset, 2)); - lvFreeRow = lvUsedRow + 1; - } - else - { - // Full — give the column its own usage-map page (owned = row 0, free = row 1), the same - // fallback CREATE TABLE uses once its primary map page fills on a wide table. - lvMapPage = _allocator.Allocate(); - lvUsedRow = 0; lvFreeRow = 1; lvDedicated = true; - } - AddLongValueMapEntry(parts, maxCols, lvUsedRow, lvFreeRow, lvMapPage); - } - - WriteTdef(table.DefinitionPage, parts); - - if (isLongValue) - { - if (lvDedicated) - WriteUsageMaps(format, lvMapPage, mapCount: 2); // owned = row 0, free = row 1, both empty - else - { - AppendEmptyUsageMapRow(format, lvMapPage, lvUsedRow); - AppendEmptyUsageMapRow(format, lvMapPage, lvFreeRow); - } - } - - // NOT NULL / DEFAULT go in the table's LvProp blob (DefaultValue before Required, matching ACE), the - // same properties CREATE TABLE writes — appended to the existing blob without disturbing other columns'. - var props = new List(); - if (defaultValue is not null) props.Add(new PropertyBlob.Property(spec.Name, PropertyBlob.DefaultValueProperty, defaultValue)); - if (!spec.IsNullable) props.Add(PropertyBlob.Bool(spec.Name, PropertyBlob.RequiredProperty, true)); - props.AddRange(CalculatedProperties(spec)); - if (props.Count > 0) SetColumnProperties(table.DefinitionPage, spec.Name, props); - - _catalog.Invalidate(); - if (spec.IsAutoNumber) NumberExistingRows(tableName, spec); - return true; - } - - /// - /// Gives the rows already in a table values in an AutoNumber column just added to it, as ACE does rather than - /// leaving them NULL, and sets where the counter carries on (all verified). - /// - /// - /// The existing rows are numbered 1, 2, 3 … in table order whatever the column's seed and increment. A counter - /// with the default seed 1 and increment 1 — however it was spelled — then continues after them: two rows take - /// 1 and 2, the next insert 3. Any other counter restarts at its own seed, even where that repeats a value the - /// rows were given: COUNTER(2, 1) over two rows goes on 2, 3, 4, and COUNTER(1, 5) 1, 6, 11. A - /// primary key over the new column has a value in every row either way. - /// - private void NumberExistingRows(string tableName, ColumnSpec spec) - { - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found after adding column '{spec.Name}'."); - ColumnDef column = table.FindColumn(spec.Name)!; - int increment = spec.Increment == 0 ? 1 : spec.Increment; - - var rows = new Table(_channel, table).Rows().WithIds().ToList(); - if (rows.Count > 0) - { - var writer = new RowInserter(_channel, table); - var changed = new HashSet { column.Index }; - int number = 0; - foreach ((RowId id, object?[] values) in rows) - { - values[column.Index] = ++number; - writer.Update(id, values, changed); - } - } - - ReseedCounter(table, column, spec.Seed == 1 && increment == 1 ? rows.Count + 1 : spec.Seed, increment); - } - - /// Inserts a long-value (memo/OLE) column's 10-byte §3.3.2 usage-map entry - /// ({col_num:2}{used row+page:4}{free row+page:4}) just before the list's 0xFFFF terminator. - /// The new column has the highest id, so appending keeps the list in ascending column order. - private static void AddLongValueMapEntry(TdefParts parts, int columnId, int usedRow, int freeRow, int mapPage) - { - byte[] lval = parts.Lval; - int at = lval.Length - 2; // before the terminator - - var entry = new byte[10]; - BinaryPrimitives.WriteUInt16LittleEndian(entry, (ushort)columnId); - entry[2] = (byte)usedRow; WriteInt24(entry, 3, mapPage); - entry[6] = (byte)freeRow; WriteInt24(entry, 7, mapPage); - - var result = new byte[lval.Length + 10]; - Array.Copy(lval, 0, result, 0, at); - entry.CopyTo(result, at); - Array.Copy(lval, at, result, at + 10, 2); // the 0xFFFF terminator - parts.Lval = result; - } - - /// Removes a long-value column's 10-byte §3.3.2 usage-map entry from the list, keeping the other - /// entries and the 0xFFFF terminator. A no-op for a column without one. - private static void RemoveLongValueMapEntry(TdefParts parts, int columnId) - { - byte[] lval = parts.Lval; - for (int at = 0; at + 2 < lval.Length; at += 10) - { - if (BinaryPrimitives.ReadUInt16LittleEndian(lval.AsSpan(at, 2)) != columnId) continue; - var result = new byte[lval.Length - 10]; - Array.Copy(lval, 0, result, 0, at); - Array.Copy(lval, at + 10, result, at, lval.Length - at - 10); - parts.Lval = result; - return; - } - } - - /// Sets (replaces) a column's DefaultValue in the table's MSysObjects.LvProp blob — - /// ALTER TABLE … ALTER COLUMN … DEFAULT. Reads all properties, drops any existing DefaultValue for the - /// column, adds the new one, and rewrites the blob (preserving every other property). - public void SetColumnDefault(string tableName, string columnName, string defaultSql) - => MutateLvPropForColumn(tableName, columnName, props => - { - props.RemoveAll(p => string.Equals(p.Owner, columnName, StringComparison.OrdinalIgnoreCase) - && p.Name == PropertyBlob.DefaultValueProperty); - props.Add(new PropertyBlob.Property(columnName, PropertyBlob.DefaultValueProperty, defaultSql)); - }); - - /// Removes a column's DefaultValue from the table's MSysObjects.LvProp blob — - /// ALTER TABLE … ALTER COLUMN … DROP DEFAULT. Drops only that property, so the column's type and its - /// Required (NOT NULL) property survive — ACE-verified. A no-op if the column had no default. - public void DropColumnDefault(string tableName, string columnName) - => MutateLvPropForColumn(tableName, columnName, props => - props.RemoveAll(p => string.Equals(p.Owner, columnName, StringComparison.OrdinalIgnoreCase) - && p.Name == PropertyBlob.DefaultValueProperty)); - - /// Sets or clears a column's Required (NOT NULL) property in the table's - /// MSysObjects.LvProp blob — ALTER TABLE … ALTER COLUMN … NOT NULL / NULL. A required column carries - /// a boolean Required property; a nullable one simply has none, so this drops any existing one and - /// re-adds it only when (matching the CREATE-side write, and read back into - /// ). ACE-verified: ACE writes the same property for - /// ALTER COLUMN … NOT NULL and enforces it. - public void SetColumnRequired(string tableName, string columnName, bool required) - => MutateLvPropForColumn(tableName, columnName, props => - { - props.RemoveAll(p => string.Equals(p.Owner, columnName, StringComparison.OrdinalIgnoreCase) - && p.Name == PropertyBlob.RequiredProperty); - if (required) props.Add(PropertyBlob.Bool(columnName, PropertyBlob.RequiredProperty, true)); - }); - - /// Reads the table's MSysObjects.LvProp property blob, applies , - /// and rewrites it — the shared read-modify-write behind ALTER COLUMN … SET/DROP DEFAULT. - private void MutateLvPropForColumn(string tableName, string columnName, Action> mutate) - { - TableDef target = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' does not exist."); - int tdefPage = target.DefinitionPage; - - TableDef msys = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIdx = (msys.FindColumn("Id") ?? throw new InvalidOperationException("MSysObjects is missing 'Id'.")).Index; - ColumnDef lvProp = msys.FindColumn("LvProp") ?? throw new InvalidOperationException("MSysObjects is missing 'LvProp'."); - var table = new Table(_channel, msys); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds()) - { - if (values[idIdx] is null || Convert.ToInt32(values[idIdx], CultureInfo.InvariantCulture) != tdefPage) continue; - byte[] blob = values[lvProp.Index] as byte[] ?? []; - var props = PropertyBlob.Read(blob).ToList(); - mutate(props); - byte[] updated = PropertyBlob.Write(props, blob.Length >= 4 ? blob.AsSpan(0, 4) : default); - byte[] descriptor = new RowInserter(_channel, msys).StorePackedLongValue(lvProp.ColumnId, updated); - values[lvProp.Index] = new LongValueDescriptor(descriptor); - table.Update(id, values, new HashSet { lvProp.Index }); - return; - } - throw new InvalidOperationException($"MSysObjects row for table '{tableName}' (page {tdefPage}) was not found."); - } - - /// Appends a column's extended properties (DefaultValue/Required) to its table's - /// MSysObjects.LvProp blob and re-stores it — the add-side counterpart of - /// . - private void SetColumnProperties(int tdefPage, string columnName, IReadOnlyList props) - { - TableDef msys = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIdx = (msys.FindColumn("Id") ?? throw new InvalidOperationException("MSysObjects is missing 'Id'.")).Index; - ColumnDef lvProp = msys.FindColumn("LvProp") ?? throw new InvalidOperationException("MSysObjects is missing 'LvProp'."); - var table = new Table(_channel, msys); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds()) - { - if (values[idIdx] is null || Convert.ToInt32(values[idIdx], CultureInfo.InvariantCulture) != tdefPage) continue; - byte[] blob = values[lvProp.Index] as byte[] ?? []; - byte[] updated = PropertyBlob.AddColumnProperties(blob, columnName, props); - byte[] descriptor = new RowInserter(_channel, msys).StorePackedLongValue(lvProp.ColumnId, updated); - values[lvProp.Index] = new LongValueDescriptor(descriptor); - table.Update(id, values, new HashSet { lvProp.Index }); - return; - } - } - - /// Adds a table-level CHECK to the table's MSysObjects.LvProp blob — ALTER TABLE ADD - /// CONSTRAINT … CHECK. Merges with any existing checks: reads the current CheckConstraints property, - /// appends the new (name, expression), and rewrites the single empty-owner table block (RemoveOwner + re-add), - /// keeping the name pool and every column block intact. The check is enforced by the engine from the - /// re-loaded TableDef.CheckConstraints. - public void AddCheckConstraint(string tableName, string checkName, string expression) - { - JetName.Validate(checkName, "check constraint name"); - TableDef target = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' does not exist."); - int tdefPage = target.DefinitionPage; - - TableDef msys = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIdx = (msys.FindColumn("Id") ?? throw new InvalidOperationException("MSysObjects is missing 'Id'.")).Index; - ColumnDef lvProp = msys.FindColumn("LvProp") ?? throw new InvalidOperationException("MSysObjects is missing 'LvProp'."); - var table = new Table(_channel, msys); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds()) - { - if (values[idIdx] is null || Convert.ToInt32(values[idIdx], CultureInfo.InvariantCulture) != tdefPage) continue; - byte[] blob = values[lvProp.Index] as byte[] ?? []; - - var checks = PropertyBlob.ReadCheckConstraints(blob).ToList(); - checks.Add((checkName, expression)); - - // Replace only the CheckConstraints entry. Dropping the whole table-owned block and re-adding one - // property takes every OTHER table-level property with it — ValidationRule / ValidationText above - // all, which LibRed reads and reports but does not re-emit, so an Access-authored table validation - // rule vanished on the first CHECK anyone added. The column-property paths already do it this way. - byte[] updated = ReplaceTableProperty(blob, PropertyBlob.CheckConstraintsProperty, - checks.Count > 0 ? PropertyBlob.WriteCheckList(checks) : null); - - byte[] descriptor = new RowInserter(_channel, msys).StorePackedLongValue(lvProp.ColumnId, updated); - values[lvProp.Index] = new LongValueDescriptor(descriptor); - table.Update(id, values, new HashSet { lvProp.Index }); - return; - } - throw new InvalidOperationException($"MSysObjects row for table '{tableName}' (page {tdefPage}) was not found."); - } - - /// Drops a named table-level CHECK — ALTER TABLE … DROP CONSTRAINT. Removes the matching entry from - /// the CheckConstraints list in the table's MSysObjects.LvProp blob (the inverse of - /// ): if any remain, rewrites the list; if it was the last one, drops the - /// whole table-level property block. ACE-verified: after the drop ACE stops enforcing the check. Returns - /// false if no CHECK of that name exists (so the caller can try other constraint kinds). - public bool DropCheckConstraint(string tableName, string checkName) - { - TableDef target = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' does not exist."); - int tdefPage = target.DefinitionPage; - - TableDef msys = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIdx = (msys.FindColumn("Id") ?? throw new InvalidOperationException("MSysObjects is missing 'Id'.")).Index; - ColumnDef lvProp = msys.FindColumn("LvProp") ?? throw new InvalidOperationException("MSysObjects is missing 'LvProp'."); - var table = new Table(_channel, msys); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds()) - { - if (values[idIdx] is null || Convert.ToInt32(values[idIdx], CultureInfo.InvariantCulture) != tdefPage) continue; - byte[] blob = values[lvProp.Index] as byte[] ?? []; - - var checks = PropertyBlob.ReadCheckConstraints(blob).ToList(); - if (checks.RemoveAll(c => string.Equals(c.Name, checkName, StringComparison.OrdinalIgnoreCase)) == 0) - return false; // no CHECK of that name — let the caller try FK/PK/unique - - // Rewrite the list, or remove the entry when that was the last check — leaving every other - // table-level property (ValidationRule, ValidationText, …) untouched. See AddCheckConstraint. - byte[] updated = ReplaceTableProperty(blob, PropertyBlob.CheckConstraintsProperty, - checks.Count > 0 ? PropertyBlob.WriteCheckList(checks) : null); - - byte[] descriptor = new RowInserter(_channel, msys).StorePackedLongValue(lvProp.ColumnId, updated); - values[lvProp.Index] = new LongValueDescriptor(descriptor); - table.Update(id, values, new HashSet { lvProp.Index }); - return true; - } - throw new InvalidOperationException($"MSysObjects row for table '{tableName}' (page {tdefPage}) was not found."); - } - - /// Changes a column's declared type — ALTER TABLE … ALTER COLUMN. A **variable text/binary - /// column's max length** is a descriptor-length edit at ColumnLengthOffset: variable columns store - /// each row's actual length, so widening rewrites no rows, and narrowing only scans them to check they - /// still fit. Every other change — numeric type, a fixed column's size, fixed↔variable — is a full column - /// rewrite, handled by (byte-faithful with ACE) or, for a Memo/OLE - /// target, by . Nothing here throws NotSupported for a storage-type change. - public void AlterColumn(string tableName, string columnName, ColumnSpec newSpec) - { - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' does not exist."); - ColumnDef col = table.FindColumn(columnName) - ?? throw new InvalidOperationException($"Column '{columnName}' does not exist in '{tableName}'."); - EnsureColumnIsNotInRelationship(table, col); - - // Widening a column reaches the same bytes as declaring it wide in the first place, so the limits - // Create enforces apply here too. This sits ahead of the identity and counter short-circuits below - // deliberately: re-declaring an already-oversized column should report the problem, not wave it on. - RecordLayout.ValidateFieldWidth(newSpec.Name, newSpec.Type, newSpec.Length); - JetDataTypeVersions.EnsureStorable(newSpec.Type, _channel.Format.Version, newSpec.Name); - // A pre-check, so an oversized re-declaration reports rather than being waved on by the short-circuits - // below; AlterColumnTypeInPlace re-runs it against the measured fixed region, which is authoritative. - // The variable-slot count comes from the stored 0x2B high-water — it never decrements, so deriving it - // from the live columns under-counts on a table that has dropped or retyped one. - RecordLayout.ValidateRecordFits(tableName, - FixedBytes(table) - - (col.IsFixedLength && col.Type != JetDataType.Boolean ? col.Length : 0) - + (newSpec.IsFixedLength && newSpec.Type != JetDataType.Boolean ? newSpec.Length : 0), - table.VariableColumnCount + (col.IsFixedLength && !newSpec.IsFixedLength ? 1 : 0), - // A type change burns a fresh column id, so the bitmap can widen by one. - HighWater(table) + (col.Type == newSpec.Type ? 0 : 1), - _channel.Format); - - // A pure reseed of an existing counter — ALTER COLUMN c COUNTER(seed, increment) where c is already an - // AutoNumber of the same storage type — changes only the next id, not the data or layout. It's an - // in-place TDEF header edit (0x14/0x18), exactly what ACE does; RewriteColumn would needlessly rebuild - // the whole table. (Changing the numeric type still rebuilds.) - if (col.IsAutoNumber && newSpec.IsAutoNumber && col.Type == newSpec.Type) - { - ReseedCounter(table, col, newSpec.Seed, newSpec.Increment); - return; - } - - // Promote a plain Int32 column to an AutoNumber — a counter is stored identically (both a 4-byte Int32); - // the only differences are the column's 0x04 flag and the header's seed/increment. So it's a metadata - // edit, not a rebuild. (ACE/SQL Server reject this; PostgreSQL/MySQL and LibRed allow it — see spec.) - if (!col.IsAutoNumber && newSpec.IsAutoNumber && col.Type == newSpec.Type) - { - PromoteColumnToCounter(table, col, newSpec.Seed, newSpec.Increment); - return; - } - - // Demote a counter back to a plain Int32 — the reverse, and likewise a metadata edit: clear the 0x04 - // flag and reset the header to a non-AutoNumber table's state (0x14 = 0, 0x18 = 1). ACE *allows* this - // (unlike promotion), so LibRed matches; existing values are kept and the column stops auto-assigning. - if (col.IsAutoNumber && !newSpec.IsAutoNumber && col.Type == newSpec.Type) - { - DemoteCounterToInt(table, col); - return; - } - - // ACE identity ALTER succeeds at exhausted ids for these measured scalar/short-value types, so a - // re-declaration that changes nothing must not burn one. Memo/OLE still consume an id even for an - // identical declaration, and so fall through. - // - // Nullability is deliberately NOT compared: no ALTER path carries it. The SQL layer always builds - // the spec with NotNull false and applies Required separately afterwards, and RewriteColumn discards - // newSpec.IsNullable outright in favour of the target's. Comparing it here would make the check fail - // for every NOT NULL column, so the same statement would burn an id — and throw at 255 — purely - // because the column was required. - if (col.Type is JetDataType.Boolean or JetDataType.Byte or JetDataType.Int16 or JetDataType.Int32 - or JetDataType.Single or JetDataType.Double or JetDataType.Currency or JetDataType.DateTime - or JetDataType.Guid or JetDataType.Text or JetDataType.Binary or JetDataType.FixedPoint - && col.Type == newSpec.Type - && col.Length == newSpec.Length && col.IsFixedLength == newSpec.IsFixedLength - && (col.Type != JetDataType.FixedPoint || (col.Precision == newSpec.Precision && col.Scale == newSpec.Scale))) - return; - - bool variableLengthChange = - !col.IsFixedLength && !newSpec.IsFixedLength && col.Type == newSpec.Type && - newSpec.Type is JetDataType.Text or JetDataType.Binary; - // A variable text/binary length change is a cheap in-place descriptor edit (below). A storage-type change - // (numeric type, fixed size, fixed↔variable) is a full column rewrite: the byte-faithful in-place edit - // where it applies (all-fixed non-indexed target), else the logical rebuild (AlterColumnTypeInPlace picks). - if (!variableLengthChange) - { - AlterColumnTypeInPlace(tableName, columnName, newSpec); - return; - } - - // Widening needs no row work — a variable column stores each row's actual length. NARROWING does: the - // invariant "no stored value exceeds its column's declared width" is enforced on every insert and - // update, so the statement that changes the declaration has to hold it too. Without this the ALTER - // succeeds and leaves behind exactly the rows Access will not read back that the insert-time check - // exists to prevent. Scanned before the TDEF is touched, so a refusal changes nothing on disk. - if (newSpec.Length < col.Length) - EnsureExistingValuesFit(table, col, newSpec.Length); - - JetFormatBase format = _channel.Format; - TdefParts parts = ParseTdef(table.DefinitionPage); - byte[] cols = parts.Columns; - int descSize = format.ColumnDescriptorSize; - for (int i = 0; i < table.Columns.Count; i++) - { - int entry = i * descSize; - int colId = BinaryPrimitives.ReadUInt16LittleEndian(cols.AsSpan(entry + format.ColumnNumberOffset, 2)); - if (colId != col.ColumnId) continue; - BinaryPrimitives.WriteUInt16LittleEndian(cols.AsSpan(entry + format.ColumnLengthOffset, 2), (ushort)newSpec.Length); - WriteTdef(table.DefinitionPage, parts); - return; - } - throw new InvalidOperationException($"Descriptor for column '{columnName}' (id {col.ColumnId}) was not found."); - } - - /// Sets or removes one property in the blob's table-owned (empty-owner) block, leaving every - /// other property — table-level and column-level — exactly as it was. null - /// removes the entry. Read-modify-write over the parsed property list, so unmodelled properties survive - /// on their RawValue passthrough. - private static byte[] ReplaceTableProperty(byte[] blob, string name, string? value) - { - var props = PropertyBlob.Read(blob).ToList(); - props.RemoveAll(p => p.Owner.Length == 0 && p.Name == name); - if (value is not null) - props.Add(new PropertyBlob.Property("", name, value)); - return PropertyBlob.Write(props, blob.Length >= 4 ? blob.AsSpan(0, 4) : default); - } - - /// Refuses a narrowing ALTER when a stored value would no longer fit, reporting the same way the - /// insert-time width check does. Text declares characters and stores UTF-16, hence the halving. - private void EnsureExistingValuesFit(TableDef table, ColumnDef column, int newLength) - { - bool text = column.Type == JetDataType.Text; - foreach (object?[] values in new Table(_channel, table).Rows()) - { - int stored = values[column.Index] switch - { - string s => Encoding.Unicode.GetByteCount(s), - byte[] b => b.Length, - _ => 0, - }; - if (stored <= newLength) continue; - throw new InvalidOperationException( - $"The field '{column.Name}' cannot be narrowed to {(text ? newLength / 2 : newLength)} " - + $"{(text ? "characters" : "bytes")}: the table holds a value of " - + $"{(text ? stored / 2 : stored)}."); - } - } - - /// The table's fixed-data region, counted the way counts it on create: - /// Boolean is fixed but occupies no data, so it contributes nothing. - private static int FixedBytes(TableDef table) => - table.Columns.Where(c => c.IsFixedLength && c.Type != JetDataType.Boolean).Sum(c => c.Length); - - /// The TDEF's `0x29` column-id high-water — the number of ids handed out over the table's - /// lifetime, which is what sizes a record's null bitmap (dropped ids keep their bit). - private int HighWater(TableDef table) => - ReadDefinition(table.DefinitionPage).Buffer.ReadUInt16(_channel.Format.TdefMaxColumnsOffset); - - /// Whether the column is either end of a relationship — the child's FK column or the parent's - /// referenced key. ACE refuses to alter or drop such a column; the two callers differ only in the message - /// they raise, so the rule itself lives here. - private bool ColumnIsInRelationship(TableDef table, ColumnDef column) - { - const StringComparison oic = StringComparison.OrdinalIgnoreCase; - return _catalog.Relationships.Any(r => - (string.Equals(r.Table, table.Name, oic) && - r.Columns.Any(c => string.Equals(c.Column, column.Name, oic))) || - (string.Equals(r.ReferencedTable, table.Name, oic) && - r.Columns.Any(c => string.Equals(c.ReferencedColumn, column.Name, oic)))); - } - - /// ACE rejects every type/length alteration of a relationship column, on either the - /// referencing or referenced side. Keep this check ahead of all specialized ALTER paths so an - /// in-place descriptor edit cannot bypass the same rule enforced by a logical table rebuild. - private void EnsureColumnIsNotInRelationship(TableDef table, ColumnDef column) - { - if (ColumnIsInRelationship(table, column)) - throw new InvalidOperationException( - $"Cannot change field '{column.Name}'. It is part of one or more relationships."); - } - - /// Reseeds an existing AutoNumber column in place — ALTER COLUMN c COUNTER(seed, increment). Writes - /// the TDEF header's last-value (0x14 = seed − increment, so the next assigned id is seed) and - /// increment (0x18); no data or descriptor changes. ACE rejects reseeding a counter that participates - /// in a relationship ("Cannot change field 'X'. It is part of one or more relationships." — verified); match - /// that. - private void ReseedCounter(TableDef table, ColumnDef col, int seed, int increment) - { - EnsureColumnIsNotInRelationship(table, col); - - if (increment == 0) increment = 1; - JetFormatBase format = _channel.Format; - byte[] tdef = _channel.ReadPage(table.DefinitionPage).Span.ToArray(); - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(format.TdefLastAutoNumberOffset, 4), seed - increment); - BinaryPrimitives.WriteInt32LittleEndian(tdef.AsSpan(format.TdefAutoNumberIncrementOffset, 4), increment); - _channel.WritePage(table.DefinitionPage, tdef); - _catalog.Invalidate(); - } - - /// Promotes a plain Int32 column to an AutoNumber in place — ALTER COLUMN c COUNTER(seed, increment) - /// where c is a plain integer. A counter is stored identically to a Long Integer, so this only sets the - /// column descriptor's 0x04 AutoNumber flag and the header's seed/increment (0x14/0x18); - /// existing values are untouched. Only one column may draw on that pair, so a second is rejected — complex - /// columns are flagged 0x04 too but allocate from 0x1C, so they do not count as the existing - /// one; and (like the reseed path) a column in a relationship is rejected, matching ACE. - private void PromoteColumnToCounter(TableDef table, ColumnDef col, int seed, int increment) - { - if (table.Columns.Any(c => c.IsAutoNumber && c.Type != JetDataType.Complex && c.ColumnId != col.ColumnId)) - throw new InvalidOperationException( - $"Cannot make '{col.Name}' an AutoNumber: table '{table.Name}' already has one " - + "(Jet allows a single column to draw on the table's seed/increment counter)."); - EnsureColumnIsNotInRelationship(table, col); - - if (increment == 0) increment = 1; - JetFormatBase format = _channel.Format; - TdefParts parts = ParseTdef(table.DefinitionPage); - int descSize = format.ColumnDescriptorSize; - for (int i = 0; i < table.Columns.Count; i++) - { - int entry = i * descSize; - if (BinaryPrimitives.ReadUInt16LittleEndian(parts.Columns.AsSpan(entry + format.ColumnNumberOffset, 2)) != col.ColumnId) continue; - parts.Columns[entry + format.ColumnFlagsOffset] |= JetFormatBase.ColumnFlagAutoNumber; - break; - } - BinaryPrimitives.WriteInt32LittleEndian(parts.Header.AsSpan(format.TdefLastAutoNumberOffset, 4), seed - increment); - BinaryPrimitives.WriteInt32LittleEndian(parts.Header.AsSpan(format.TdefAutoNumberIncrementOffset, 4), increment); - WriteTdef(table.DefinitionPage, parts); - _catalog.Invalidate(); - - // COUNTER(seed, increment) is a *sequential* counter. A surviving GenUniqueID() default would instead - // make it a "Random" AutoNumber (IsRandomAutoNumber) — assigning random ids and ignoring the seed — so - // clear it to honour the requested sequence. Other (literal) defaults are inert on a counter (the insert - // path skips defaults for AutoNumber columns) and are left as-is. - if (col.DefaultValue?.Trim().Equals("GenUniqueID()", StringComparison.OrdinalIgnoreCase) == true) - DropColumnDefault(table.Name, col.Name); - } - - /// Demotes an AutoNumber column back to a plain Int32 in place — ALTER COLUMN c LONG where c is a - /// counter. Clears the descriptor's 0x04 flag and resets the header to a non-AutoNumber table's state - /// (0x14 = 0, 0x18 = 1); existing values are kept, the column just stops auto-assigning. ACE - /// permits this (unlike int→counter promotion), so no divergence. - private void DemoteCounterToInt(TableDef table, ColumnDef col) - { - JetFormatBase format = _channel.Format; - TdefParts parts = ParseTdef(table.DefinitionPage); - int descSize = format.ColumnDescriptorSize; - for (int i = 0; i < table.Columns.Count; i++) - { - int entry = i * descSize; - if (BinaryPrimitives.ReadUInt16LittleEndian(parts.Columns.AsSpan(entry + format.ColumnNumberOffset, 2)) != col.ColumnId) continue; - parts.Columns[entry + format.ColumnFlagsOffset] &= unchecked((byte)~JetFormatBase.ColumnFlagAutoNumber); - break; - } - BinaryPrimitives.WriteInt32LittleEndian(parts.Header.AsSpan(format.TdefLastAutoNumberOffset, 4), 0); - BinaryPrimitives.WriteInt32LittleEndian(parts.Header.AsSpan(format.TdefAutoNumberIncrementOffset, 4), 1); - WriteTdef(table.DefinitionPage, parts); - _catalog.Invalidate(); - } - - /// Changes a column's storage type, matching ACE's column-modify semantics (verified): the column - /// keeps its position but is internally a new column — it gets a fresh id burned from the 0x29 - /// high-water, while every other column keeps its id and its original descriptor bytes (so fields - /// LibRed doesn't model are preserved, per the faithful round-trip rule). All target values are converted - /// in memory first (an unconvertible value fails before anything is written), then the rebuild — drop, - /// recreate with the new type, re-insert, recreate secondary indexes, re-add relationships — runs inside a - /// page-level transaction that rolls back atomically on any later failure. Column order, the primary - /// key, unique/secondary indexes, CHECK constraints, defaults, and AutoNumber values are preserved. Rejects a - /// table whose target column is in a relationship (drop the FK first). This is a logical rebuild (a fresh TDEF - /// page, not ACE's byte-exact in-place edit), but the resulting column layout — position, burned id, and the - /// untouched columns' bytes — matches what ACE produces. - private void RewriteColumn(string tableName, string columnName, ColumnSpec newColumnSpec) - { - TableDef def = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' does not exist."); - ColumnDef target = def.FindColumn(columnName) - ?? throw new InvalidOperationException($"Column '{columnName}' does not exist in '{tableName}'."); - - EnsureColumnIsNotInRelationship(def, target); - - // Read straight off the definition page: 0x29 is in the header, so this needs none of the block - // slicing (or continuation-page stitching) a full ParseTdef would do for one 16-bit field. - int priorHighWater = _channel.ReadPage(def.DefinitionPage).ReadUInt16(_channel.Format.TdefMaxColumnsOffset); - if (priorHighWater >= MaxColumnsPerTable) - throw new NotSupportedException($"Cannot alter '{columnName}': too many fields defined — {MaxColumnsPerTable} column ids have been used."); - - // The rebuild drops + recreates the table, so every relationship it touches is captured and restored - // afterwards. OUTGOING FKs (this table is the child) are cascaded away by DropTable and re-added; their - // backing indexes are recreated with them, so they're excluded from `secondary`. INCOMING FKs (this table - // is the referenced parent) are dropped up front so the parent can be dropped, then re-added — this is - // what makes a parent-side rewrite work. Self-references count as outgoing only. - const StringComparison oic = StringComparison.OrdinalIgnoreCase; - var foreignKeys = _catalog.ForeignKeysOf(tableName).ToList(); - var incoming = _catalog.Relationships - .Where(r => string.Equals(r.ReferencedTable, tableName, oic) && !string.Equals(r.Table, tableName, oic)) - .ToList(); - var fkColumnSets = foreignKeys.Select(fk => fk.Columns.Select(c => c.Column).ToArray()).ToList(); - - // 1. Materialise all rows (values indexed by column position) before dropping the table. - var rows = new Table(_channel, def).Rows().Select(r => (object?[])r.Clone()).ToList(); - - // Every index's statistics, here and on each table referencing this one: the rebuild re-inserts the rows and - // re-creates the indexes, which would recount them all, where ACE's ALTER leaves every index alone but the - // ones over the column it changes (verified, for Memo/OLE retypes as for the rest). - var statistics = new[] { tableName }.Concat(incoming.Select(r => r.Table)).Distinct(StringComparer.OrdinalIgnoreCase) - .ToDictionary(t => t, t => IndexStatistics(_catalog.FindTable(t)!), StringComparer.OrdinalIgnoreCase); - - // 2. Reconstruct the schema — column order preserved, the target re-typed. Every OTHER column keeps its - // original descriptor bytes (RawDescriptor passthrough), so fields LibRed doesn't model survive the - // rewrite (the faithful round-trip rule); the target builds fresh (RawDescriptor null). Column ids stay - // contiguous by position — NOT burned like ACE — because the row codec's null bitmap is currently - // keyed by column id, which only agrees with ACE's (position-keyed) bitmap when id == position. Burning - // the id needs the codec switched to position-keying first (verified vs an ACE-modified file). TODO. - int targetIndex = target.Index; - var specs = def.Columns.Select(c => c.Index == targetIndex - ? newColumnSpec with { IsNullable = target.IsNullable, RawDescriptor = null } - : new ColumnSpec(c.Name, c.Type, c.Length, c.IsFixedLength, c.IsAutoNumber, c.Precision, c.Scale, - c.IsNullable, c.Seed, c.Increment, RawDescriptor: c.RawDescriptor, - CalculatedExpression: c.CalculatedExpression, CalculatedResultType: c.CalculatedResultType)).ToList(); - - IndexDef? pk = def.Indexes.FirstOrDefault(i => i.IsPrimaryKey); - IReadOnlyList? primaryKey = pk?.Columns.Select(c => c.Column.Name).ToList(); - // Secondary indexes to recreate — excluding the primary key and any FK-backing index (re-added with its FK). - var secondary = def.Indexes - .Where(i => !i.IsPrimaryKey) - .Where(i => !fkColumnSets.Any(cols => - cols.SequenceEqual(i.Columns.Select(c => c.Column.Name), StringComparer.OrdinalIgnoreCase))) - .ToList(); - var checks = def.CheckConstraints.ToList(); - var defaults = def.Columns.Where(c => c.DefaultValue is not null) - .Select(c => (Column: c.Name, DefaultSql: c.DefaultValue!)).ToList(); - - // 3. Pre-check: convert every target value in memory BEFORE touching disk. An unconvertible value - // (e.g. non-numeric text → INT) throws here, with nothing written — the caller sees a clean failure. - foreach (object?[] row in rows) - row[targetIndex] = ConvertValue(row[targetIndex], newColumnSpec.Type, newColumnSpec.Name); - - // 4. Apply the rebuild atomically: wrap it in a page-level transaction so any failure that slips past the - // pre-check (a unique-index collision after narrowing, NOT NULL, an I/O or allocation error) rolls the - // whole operation back and leaves the table byte-unchanged — never a half-converted table. - bool ownTransaction = !_channel.InTransaction; - if (ownTransaction) _channel.BeginTransaction(); - try - { - // Drop incoming relationships (so the parent becomes unreferenced) → drop → recreate (PK only) → - // re-insert → recreate secondary indexes → re-add outgoing then incoming relationships. - foreach (ForeignKey r in incoming) { DropConstraint(r.Table, r.Name); _catalog.Invalidate(); } - DropTable(tableName); - _catalog.Invalidate(); - Create(tableName, specs, primaryKey, relationships: null, uniqueConstraints: null, - columnDefaults: defaults, checkConstraints: checks, primaryKeyName: pk?.Name); - _catalog.Invalidate(); - - // The logical rebuild still lays live columns out contiguously, but must not reset - // lifetime id consumption. ACE consumes one even for identity Memo/OLE ALTER. - int rebuiltPage = _catalog.FindTable(tableName)!.DefinitionPage; - TdefParts rebuilt = ParseTdef(rebuiltPage); - BinaryPrimitives.WriteUInt16LittleEndian( - rebuilt.Header.AsSpan(_channel.Format.TdefMaxColumnsOffset, 2), (ushort)(priorHighWater + 1)); - WriteTdef(rebuiltPage, rebuilt); - _catalog.Invalidate(); - - var dest = new Table(_channel, _catalog.FindTable(tableName)!); - int[] calculatedColumns = dest.Definition.Columns - .Where(c => c.IsCalculated) - .Select(c => c.Index) - .ToArray(); - foreach (object?[] row in rows) - { - // A calculated value is a cache, not caller-supplied data. Recompute it from its expression - // while rebuilding, as an ordinary insert does; reinserting the old cache is rejected. - foreach (int index in calculatedColumns) row[index] = null; - dest.Insert(row); - } - - // Restore each index AS IT WAS. IgnoreNulls and Required are read off the 0x2E flags word and are - // right here on the IndexDef; hard-coding them false made an ALTER COLUMN on an unrelated column - // silently turn a WITH IGNORE NULL index into a plain one — changing which rows are in the B-tree - // — and stop ACE enforcing DISALLOW NULL. - foreach (IndexDef ix in secondary) - { - AddIndex(tableName, ix.Name, ix.Columns.Select(c => (c.Column.Name, !c.Ascending)).ToList(), - ix.IsUnique, isPrimary: false, disallowNull: ix.Required, ignoreNulls: ix.IgnoreNulls); - _catalog.Invalidate(); - } - - // Re-add the outgoing foreign keys (recreates their backing index + linkage + MSysRelationships rows). - foreach (ForeignKey fk in foreignKeys) - { - AddForeignKey(tableName, new RelationshipSpec(fk.Name, fk.ReferencedTable, fk.Columns.ToList(), - fk.IsEnforced, fk.CascadeUpdate, fk.CascadeDelete, NoIndex: false, - DeleteSetNull: fk.DeleteSetNull, UpdateSetNull: fk.UpdateSetNull)); - _catalog.Invalidate(); - } - - // Re-add the incoming relationships — each child's FK back to the rebuilt parent. - foreach (ForeignKey r in incoming) - { - AddForeignKey(r.Table, new RelationshipSpec(r.Name, r.ReferencedTable, r.Columns.ToList(), - r.IsEnforced, r.CascadeUpdate, r.CascadeDelete, NoIndex: false, - DeleteSetNull: r.DeleteSetNull, UpdateSetNull: r.UpdateSetNull)); - _catalog.Invalidate(); - } - - // Put the statistics back as ACE leaves them: every index as it was, except one over the changed column, - // which ACE rebuilds — counted from the rows it now holds, the primary key too (which the re-insert - // counted as inserts). - foreach ((string name, Dictionary before) in statistics) - { - TableDef table = _catalog.FindTable(name)!; - foreach (IndexDef index in table.Indexes) - { - if (string.Equals(name, tableName, oic) - && index.Columns.Any(c => string.Equals(c.Column.Name, columnName, oic))) - SetBuiltStatistics(table, index, IndexEntries(table, index, index.IgnoreNulls)); - else if (before.TryGetValue(index.Name, out (int Total, int Unique) kept)) - WriteIndexStatistics(table, index, kept.Total, kept.Unique); - } - } - - if (ownTransaction) _channel.CommitTransaction(); - } - catch when (ownTransaction) - { - _channel.RollbackTransaction(); - _catalog.Invalidate(); // the in-memory catalog cache is stale after the pages are restored - throw; - } - } - - /// Applies ACE's in-place column retype to the target descriptor within - /// (no page write — the caller writes the TDEF once): the target becomes a NEW column with a fresh id from the - /// 0x29 high-water and its fixed data appended to the END of the current fixed region (its old slot left - /// as dead space — ACE does not compact); 0x29 bumps, and 0x2B too for a variable retype. Only - /// the target descriptor changes; every other descriptor stays byte-identical. Returns the burned new id. - private static int EditTargetDescriptor(TdefParts parts, ColumnDef target, ColumnSpec newSpec, int fixedEnd, - Collation collation, JetFormatBase format) - { - int maxCols = BinaryPrimitives.ReadUInt16LittleEndian(parts.Header.AsSpan(format.TdefMaxColumnsOffset, 2)); - // ACE-only probe: with 254 columns one retype succeeds, the next fails; with 255 - // columns the first retype fails. A same-type ALTER does not reach this id-burning path. - if (maxCols >= MaxColumnsPerTable) - throw new NotSupportedException( - $"Cannot change the type of '{target.Name}': too many fields defined — {MaxColumnsPerTable} column ids have been used."); - int varCount = BinaryPrimitives.ReadUInt16LittleEndian(parts.Header.AsSpan(format.TdefVariableColumnsOffset, 2)); - - Span d = parts.Columns.AsSpan(target.Index * format.ColumnDescriptorSize, format.ColumnDescriptorSize); - d[format.ColumnTypeOffset] = (byte)newSpec.Type; - BinaryPrimitives.WriteUInt16LittleEndian(d[format.ColumnNumberOffset..], (ushort)maxCols); // +0x05 id burned - // The target's var-index (+0x07) becomes the old variable-column count — the next var slot — for BOTH a - // fixed and a variable retype (verified vs ACE); a variable retype also bumps the 0x2B var-column count. - BinaryPrimitives.WriteUInt16LittleEndian(d[format.ColumnVariableIndexOffset..], (ushort)varCount); - // The duplicate id at +0x09 is deliberately left unchanged — verified ACE does not update it. - byte flags = d[format.ColumnFlagsOffset]; - flags = newSpec.IsFixedLength ? (byte)(flags | JetFormatBase.ColumnFlagFixedLength) - : (byte)(flags & ~JetFormatBase.ColumnFlagFixedLength); - flags = newSpec.IsAutoNumber ? (byte)(flags | JetFormatBase.ColumnFlagAutoNumber) - : (byte)(flags & ~JetFormatBase.ColumnFlagAutoNumber); - d[format.ColumnFlagsOffset] = flags; - BinaryPrimitives.WriteUInt16LittleEndian(d[format.ColumnFixedOffsetOffset..], (ushort)(newSpec.IsFixedLength ? fixedEnd : 0)); // +0x15 - BinaryPrimitives.WriteUInt16LittleEndian(d[format.ColumnLengthOffset..], (ushort)newSpec.Length); // +0x17 - // 0x0B–0x0E is a union keyed by type, so the WHOLE union is rewritten, not just the decimal arm. - // Writing precision/scale on the way in but nothing on the way out left a former DECIMAL(12,3) with - // 0x0C 0x03 in its LANGID bytes, which reads back as collating order 0x030C on a text column. - TdefBuilder.WriteLocaleUnion(d, newSpec.Type, newSpec.Precision, newSpec.Scale, collation, format); - - BinaryPrimitives.WriteUInt16LittleEndian(parts.Header.AsSpan(format.TdefMaxColumnsOffset, 2), (ushort)(maxCols + 1)); // 0x29++ - if (!newSpec.IsFixedLength) - BinaryPrimitives.WriteUInt16LittleEndian(parts.Header.AsSpan(format.TdefVariableColumnsOffset, 2), (ushort)(varCount + 1)); // 0x2B++ - return maxCols; - } - - /// Full in-place column type change, byte-for-byte like ACE for fixed and variable columns and - /// targets, fixed↔variable, and indexed targets; a Memo/OLE source or target falls back to - /// . Edits the TDEF in place () and re-lays every row — the target's - /// OLD fixed slot is kept as dead space, its converted value appended at the new offset, count + null bitmap - /// updated. Converts values in memory first (throws on bad data before any write); runs in a transaction. - public void AlterColumnTypeInPlace(string tableName, string columnName, ColumnSpec newSpec) - { - TableDef oldDef = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' does not exist."); - ColumnDef oldTarget = oldDef.FindColumn(columnName) - ?? throw new InvalidOperationException($"Column '{columnName}' does not exist in '{tableName}'."); - EnsureColumnIsNotInRelationship(oldDef, oldTarget); - if (newSpec.Type == JetDataType.Ole && oldDef.Indexes.Any(i => i.Columns.Any(c => c.Column.ColumnId == oldTarget.ColumnId))) - RejectOleIndexColumns([oldTarget.Name], _ => JetDataType.Ole); - - // Also reached directly, not only through AlterColumn, so it carries the width limits itself. - // The record-fits check needs the true fixed-region end, so it runs once that is measured, below. - RecordLayout.ValidateFieldWidth(newSpec.Name, newSpec.Type, newSpec.Length); - JetDataTypeVersions.EnsureStorable(newSpec.Type, _channel.Format.Version, newSpec.Name); - - // A long-value (Memo/OLE) target — or converting one away — needs long-value column mechanics (a §3.3.2 - // usage-map entry, LVAL pages, freeing the old value). That's out of scope for the byte-faithful in-place - // edit; the logical rebuild (Create handles long-value columns) does it correctly, if not byte-exactly. - if (newSpec.Type is JetDataType.Memo or JetDataType.Ole || oldTarget.Type is JetDataType.Memo or JetDataType.Ole) - { - RewriteColumn(tableName, columnName, newSpec); - return; - } - - // Indexes that include the target column must be rebuilt (their keys change type) — captured now. - var affectedIndexes = oldDef.Indexes - .Where(i => i.Columns.Any(col => col.Column.Index == oldTarget.Index)) - .Select(i => i.Name).ToList(); - int oldTargetId = oldTarget.ColumnId; - - // 1. Materialize (id + raw bytes + values) before touching disk; conversion throws here on bad data. - var reader = new RowInserter(_channel, oldDef); - var rows = new Table(_channel, oldDef).Rows().WithIds() - .Select(r => (r.Id, Raw: reader.ReadRow(r.Id), Values: (object?[])r.Values.Clone())) - .ToList(); - foreach (var r in rows) - r.Values[oldTarget.Index] = ConvertValue(r.Values[oldTarget.Index], newSpec.Type, newSpec.Name); - - // The fixed-region length is authoritative from the existing rows (their var-data-start), NOT the live - // column descriptors — those diverge once a high-offset column has been retyped to variable and left a - // dead fixed slot at the end. Take the MAX over every row, not row 0: ADD COLUMN of a fixed column is - // metadata-only, so a table legitimately holds short rows written before it alongside full-width ones. - // Sizing the whole re-lay from whichever row happened to be first either truncates the long rows' fixed - // tails or drags the short rows' variable data up into their fixed region. The schema floor covers an - // empty table, and rows shorter than the result are zero-filled by BuildRelaidRecord. - int oldFixedLen = oldDef.Columns.Where(c => c.IsFixedLength && c.Type != JetDataType.Boolean) - .Select(c => c.FixedOffset + c.Length).DefaultIfEmpty(0).Max(); - foreach (var r in rows) - oldFixedLen = Math.Max(oldFixedLen, FixedRegionLength(r.Raw, RowLayout.HasVariableSection(r.Raw, oldDef.Columns))); - - // Now the widest-record check, against what this path actually produces. Both counts come from stored - // state, not the live column list: the re-lay KEEPS the old target's fixed slot as dead space rather - // than reclaiming it (so nothing is subtracted), and 0x2B is a high-water that never decrements (so a - // variable→fixed retype leaves it where it is). Deriving either from the live columns under-counts on - // any table that has dropped or retyped a column, passing a declaration that then overflows 4060 — - // and per RecordLayout's own remarks, Access cannot open a database containing such a table at all. - RecordLayout.ValidateRecordFits(tableName, - oldFixedLen + (newSpec.IsFixedLength && newSpec.Type != JetDataType.Boolean ? newSpec.Length : 0), - oldDef.VariableColumnCount + (oldTarget.IsFixedLength && !newSpec.IsFixedLength ? 1 : 0), - HighWater(oldDef) + 1, // the type change burns a fresh id - _channel.Format); - - bool ownTx = !_channel.InTransaction; - if (ownTx) _channel.BeginTransaction(); - try - { - JetFormatBase format = _channel.Format; - - // 2. One TDEF edit for the whole modify: patch only the target descriptor (bump 0x29 / 0x2B — its - // appended fixed offset is the row's true fixed-region end incl. dead slots) AND re-point every - // index over the target, all into the SAME parts, then write the TDEF a single time. Each index - // re-point needs the fresh root allocated + owned-map recycled first (page work off the TDEF). - TdefParts parts = ParseTdef(oldDef.DefinitionPage); - int newTargetId = EditTargetDescriptor(parts, oldTarget, newSpec, oldFixedLen, _collation, format); - - var pending = new List<(string Name, int OldRoot, int NewRoot, bool IgnoreNulls)>(); - foreach (string ixName in affectedIndexes) - { - IndexDef index = oldDef.Indexes.First(i => string.Equals(i.Name, ixName, StringComparison.OrdinalIgnoreCase)); - int newRoot = PrepareIndexRebuild(parts, oldDef, index, oldTargetId, newTargetId); - pending.Add((ixName, index.RootPage, newRoot, index.IgnoreNulls)); - } - - WriteTdef(oldDef.DefinitionPage, parts); - _catalog.Invalidate(); - - TableDef newDef = _catalog.FindTable(tableName)!; - ColumnDef newTarget = newDef.FindColumn(columnName)!; - int newMaxId = newDef.Columns.Max(c => c.ColumnId); - int newFixedLen = newTarget.IsFixedLength ? oldFixedLen + newTarget.Length : oldFixedLen; - - // Slots per row comes from the TDEF high-water, exactly as RowEncoder derives it — never from a - // row's own stored numVar. A row written before a variable ADD COLUMN carries fewer slots than the - // table has, and appending the retyped column onto such a row lands it at the wrong index while its - // descriptor names the high-water one, which is the "A column Id is incorrect" file ACE rejects. - var newVarCols = newDef.Columns.Where(c => !c.IsFixedLength).ToList(); - int newVarCount = newVarCols.Count == 0 ? 0 - : Math.Max(newDef.VariableColumnCount, newVarCols.Max(c => c.VariableIndex) + 1); - var writer = new RowInserter(_channel, newDef); - - // 3. Re-lay each row: old fixed region + old var chunks verbatim (incl. the dead old slot), target - // appended (a new fixed slot, or a new variable chunk); count/var-table/null-bitmap rebuilt. - foreach (var r in rows) - writer.RewriteRowRaw(r.Id, BuildRelaidRecord( - r.Raw, oldFixedLen, oldDef.Columns, newTarget, r.Values, newDef.Columns, newMaxId, newFixedLen, newVarCount)); - - // 4. Finish each index rebuild: backfill the fresh B-tree with new-type keys, then free the old root - // (last, so the new root got the appended page rather than reusing this one) — as ACE does. - foreach (var p in pending) - { - BackfillIndex(tableName, p.Name, p.IgnoreNulls, validateUnique: true); - _allocator.Release(p.OldRoot); - } - - if (ownTx) _channel.CommitTransaction(); - } - catch when (ownTx) { _channel.RollbackTransaction(); _catalog.Invalidate(); throw; } - _catalog.Invalidate(); - } - - /// Builds the re-laid row record, matching ACE's in-place modify byte-for-byte: the OLD fixed - /// region and OLD variable chunks are kept verbatim (the dead old-target slot/chunk keeps its stale bytes), - /// the converted target is appended (a new fixed slot if it is fixed, else a new variable chunk), and the - /// leading count (= max id + 1), variable-offset table + numVar (omitted if none), and null bitmap - /// (dead-id bits set present) are rebuilt. - private static byte[] BuildRelaidRecord(byte[] oldRow, int oldFixedLen, IReadOnlyList oldCols, - ColumnDef newTarget, object?[] values, IReadOnlyList newCols, int newMaxId, int newFixedLen, - int newVarCount) - { - object? tv = values[newTarget.Index]; - byte[] targetBytes = tv is null - ? (newTarget.IsFixedLength ? new byte[newTarget.Length] : []) - : Types.JetTypeCodec.Encode(newTarget, tv); - - // Whether THIS row has a variable trailer, not whether the schema does — see RowLayout.HasVariableSection. - bool hasVar = RowLayout.HasVariableSection(oldRow, oldCols); - - // Fixed region: old fixed bytes verbatim (incl. a dead fixed slot); append the target if it is fixed. - // A row predating a fixed ADD COLUMN is shorter than the region; copy what it has and leave the rest - // zeroed, which is what its null bitmap already says those columns are. - var newFixed = new byte[newFixedLen]; - int rowFixedLen = Math.Min(oldFixedLen, RowLayout.Parse(oldRow, 2, hasVar).FixedRegionLength); - Array.Copy(oldRow, 2, newFixed, 0, rowFixedLen); - if (newTarget.IsFixedLength && tv is not null) - Array.Copy(targetBytes, 0, newFixed, newTarget.FixedOffset, newTarget.Length); - - // Variable chunks: old chunks verbatim (incl. a dead variable chunk), padded out to the table's slot - // count so the target lands on the index its descriptor names, then the target placed at that index. - List chunks = ExtractVarChunks(oldRow, hasVar); - while (chunks.Count < newVarCount) chunks.Add([]); - if (!newTarget.IsFixedLength) chunks[newTarget.VariableIndex] = targetBytes; - - // Assemble via the shared row layout (count + var table + null bitmap identical to a fresh encode). - return RowEncoder.AssembleRow(newMaxId, newFixed, chunks, newCols, values); - } - - /// The length of a row's fixed-data region (bytes between the leading count and the variable data), - /// read from the row itself — its variable-offset table's last entry is the variable-data start (= 2 + fixed - /// length), or for an all-fixed row it's the whole row minus the count field and null bitmap. This is - /// authoritative over the live column descriptors, which omit dead fixed slots left by prior retypes. - private static int FixedRegionLength(byte[] row, bool hasVar) => - RowLayout.Parse(row, 2, hasVar).FixedRegionLength; - - /// Extracts a row's variable-column chunks (in variable-index order) verbatim, using the row's own - /// stored numVar. (from the schema) says whether a variable section exists at all — - /// an all-fixed table omits it entirely, so its "numVar" bytes would otherwise be misread from fixed data. - private static List ExtractVarChunks(byte[] row, bool hasVar) - { - RowLayout layout = RowLayout.Parse(row, 2, hasVar); - var chunks = new List(layout.NumVar); - for (int j = 0; j < layout.NumVar; j++) - chunks.Add(layout.VarChunk(j).ToArray()); - return chunks; - } - - /// Prepares one index rebuild over a just-modified column, matching ACE's reconstruction: allocate a - /// fresh empty root leaf (appended — the old root is left orphaned) and extend/recycle the owned usage map to - /// track it, then re-point the index-data block within to the new root with the - /// target's burned column id and the new usage-map row (bumping the stats block). The caller writes the TDEF - /// once, then backfills the fresh B-tree and frees the old root. Returns the new root page. - private int PrepareIndexRebuild(TdefParts parts, TableDef table, IndexDef index, int oldTargetId, int newTargetId) - { - JetFormatBase format = _channel.Format; - - // A fresh empty root leaf, appended; the old root is freed by the caller afterwards (ACE reuses it on the - // next alloc). This and the owned-map recycle touch pages OFF the TDEF, so they happen before the single - // TDEF write; only the index-data block + stats mutations below go into the shared parts. - int newRoot = _allocator.Allocate(); - WriteEmptyLeafIndexPage(format, newRoot, owner: table.DefinitionPage); - - int usageMapPage = parts.Header[format.TdefOwnedPagesOffset + 1] - | (parts.Header[format.TdefOwnedPagesOffset + 2] << 8) | (parts.Header[format.TdefOwnedPagesOffset + 3] << 16); - - // Recycle the index's owned-map row (ACE soft-deletes the old row and reuses its space for a new row - // tracking the new root), reading the current row number from the (as-yet-unwritten) data block. - Span block = parts.DataBlocks[index.RealIndexOrdinal]; - int oldUsageRow = block[IndexBlockFormat.UsageMapRowOffset]; - int newRow = RecycleOwnedMapRow(format, usageMapPage, oldUsageRow, newRoot); - - // Re-point the index-data block: the target's burned id in its column slot, the new root, the new - // usage-map row. Its statistics are set by the backfill that follows, from the rows it then holds. - for (int slot = 0; slot < IndexBlockFormat.MaxColumns; slot++) - { - int at = IndexBlockFormat.ColumnsOffset + slot * IndexBlockFormat.ColumnSlotSize; - if (BinaryPrimitives.ReadInt16LittleEndian(block.Slice(at, 2)) == oldTargetId) - BinaryPrimitives.WriteInt16LittleEndian(block.Slice(at, 2), (short)newTargetId); - } - block[IndexBlockFormat.UsageMapRowOffset] = (byte)newRow; - BinaryPrimitives.WriteInt32LittleEndian(block.Slice(IndexBlockFormat.RootPageOffset, 4), newRoot); - return newRoot; - } - - /// - /// Recycles an index's owned-pages usage-map row the way ACE does on a rebuild, in the two writes whose - /// combined result is observable on disk: (1) append a fresh row at the bottom of the holder page - /// and set the new root's bit — those bytes are then abandoned and stay as a stale copy; (2) lay - /// the page out again with the old row's record reclaimed: its slot becomes a 0-length - /// deleted+overflow tombstone at the preceding record's offset, every later row keeps its number while its - /// record slides up, and the fresh map takes the position freed at the end of the live region under the - /// appended row number. Returns that number — the only pointer the caller re-points, because no other - /// row's number changes and each one's data travels with it. - /// - /// - /// Both halves are load-bearing and each was missed once. The stale copy decides a whole-file byte diff - /// against ACE on a single byte (the new root's bit, at offset 49 of the abandoned record) and is - /// invisible to free-space accounting, since it lies below the lowest live record inside the region free - /// space already covers. The compaction is invisible to slot offsets alone and shows up only when records - /// are identified by content — ACE's own page, before → after, with a long-value column's maps below the - /// index's: - /// - /// before row2 @3889 pages=[353] row3 @3820 pages=[] row4 @3751 pages=[] - /// after row2 @3958 TOMBSTONE row3 @3889 pages=[] row4 @3820 pages=[] row5 @3751 pages=[355] - /// stale: @3731 = 0x08 (the abandoned append, at 3682) - /// - /// The long-value maps slid up a record width and kept rows 3 and 4; only the index's pointer moved, to - /// the appended row 5. Writing step (1)'s record into the old row's slot instead — which produces the same - /// bytes whenever the recycled row happens to be the LAST one, the only case an ACE-built schema gives — - /// points a slot back up the page as soon as it is not, and no reader can walk that: a row's extent runs - /// to where the previous slot begins. AlterColumnTypeInPlace scans the table through this map - /// immediately afterwards, so the ALTER failed outright on any table that had gained a Memo or OLE - /// column after its index. - /// - private int RecycleOwnedMapRow(JetFormatBase format, int usageMapPage, int oldRow, int newRoot) - { - int dir = format.DataRowDirectoryOffset; - int rowCount = BinaryPrimitives.ReadUInt16LittleEndian( - _channel.ReadPage(usageMapPage).Span.Slice(format.DataRowCountOffset, 2)); - int newRow = rowCount; - if (oldRow < 0 || oldRow >= rowCount) - throw new InvalidDataException( - $"Usage-map row {usageMapPage}:{oldRow} does not exist; the page has {rowCount} rows."); - - // (1) ACE's first write, kept verbatim: the appended row is where the new root's bit is set, and the - // bytes it leaves behind are part of the file ACE produces. - AppendEmptyUsageMapRow(format, usageMapPage, newRow); - new UsageMapWriter(_channel).SetBit(newRow, usageMapPage, newRoot, set: true); - - // (2) Re-lay the live records. Starting from the page as it stands keeps everything this does not - // write — the abandoned append included — exactly where ACE leaves it. - byte[] page = _channel.ReadPage(usageMapPage).Span.ToArray(); - var holder = new DataPage(); - holder.Read(_channel.ReadPage(usageMapPage), format); - - var records = new byte[rowCount + 1][]; - var flags = new ushort[rowCount + 1]; - for (int i = 0; i <= rowCount; i++) - { - records[i] = i == oldRow ? [] : page.AsSpan(holder.Rows[i].Offset, holder.Rows[i].Length).ToArray(); - flags[i] = (ushort)((i == oldRow || holder.Rows[i].IsDeleted ? RowPointer.DeletedFlag : 0) - | (i == oldRow || holder.Rows[i].HasOverflow ? RowPointer.OverflowFlag : 0)); - } - - int directoryEnd = dir + (rowCount + 1) * 2; - int offset = format.PageSize; - for (int i = 0; i <= rowCount; i++) - { - offset -= records[i].Length; // a 0-length tombstone lands on the previous start - if (offset < directoryEnd) - throw new InvalidOperationException( - $"Usage-map page {usageMapPage} has no room to recycle row {oldRow}: {rowCount} rows already. " - + "The new map belongs on a page of its own."); - records[i].CopyTo(page.AsSpan(offset)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(dir + i * 2, 2), - (ushort)(flags[i] | (offset & RowPointer.OffsetMask))); - } - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(offset - directoryEnd)); - _channel.WritePage(usageMapPage, page); - return newRow; - } - - /// Converts a stored value to the CLR type for a new column type (ALTER COLUMN). NULL stays NULL; - /// an unconvertible value throws (as ACE's rewrite would). - /// Throwing is the intent; the type has to be actionable. The bare Convert.To* calls leaked - /// //, none - /// naming the column and none distinguishable from a bug in the rewrite. - private static object? ConvertValue(object? value, JetDataType type, string columnName) - { - if (value is null) return null; - try - { - return ConvertCore(value, type); - } - catch (Exception ex) when (ex is InvalidCastException or FormatException or OverflowException - or ArgumentException) - { - throw new InvalidOperationException( - $"Column '{columnName}' cannot be changed to {type}: the existing value " - + $"'{Describe(value)}' ({value.GetType().Name}) cannot be converted to it.", ex); - } - } - - /// Bounded rendering for an error message, so a memo does not paste thousands of characters into - /// one. - private static string Describe(object value) => value switch - { - byte[] bytes => $"{bytes.Length} bytes", - string { Length: > 40 } text => $"{text[..40]}…", - _ => value.ToString() ?? "", - }; - - private static object? ConvertCore(object value, JetDataType type) - { - var inv = System.Globalization.CultureInfo.InvariantCulture; - return type switch - { - JetDataType.Boolean => value is bool b ? b : Convert.ToBoolean(value, inv), - JetDataType.Byte => Convert.ToByte(value, inv), - JetDataType.Int16 => Convert.ToInt16(value, inv), - JetDataType.Int32 => Convert.ToInt32(value, inv), - JetDataType.Int64 => Convert.ToInt64(value, inv), - JetDataType.Single => Convert.ToSingle(value, inv), - JetDataType.Double => Convert.ToDouble(value, inv), - JetDataType.Currency or JetDataType.FixedPoint => JetDecimalConverter.ToDecimal(value, inv), - JetDataType.DateTime => value is DateTime d ? d : Convert.ToDateTime(value, inv), - JetDataType.Text or JetDataType.Memo => Convert.ToString(value, inv), - JetDataType.Guid => value is Guid g ? g : Guid.Parse(value.ToString()!), - JetDataType.Binary or JetDataType.Ole => value as byte[] ?? System.Text.Encoding.Unicode.GetBytes(value.ToString()!), - _ => value, - }; - } - - - /// Appends the new column's descriptor (after the existing descriptors) and its name (after the - /// existing names) to the column region. - private static void AppendColumnToParts(TdefParts parts, int colCount, byte[] descriptor, string name, JetFormatBase format) - { - int namesStart = colCount * format.ColumnDescriptorSize; - ReadOnlySpan cols = parts.Columns; - - byte[] nameBytes = System.Text.Encoding.Unicode.GetBytes(name); - var blob = new List(parts.Columns.Length + descriptor.Length + 2 + nameBytes.Length); - blob.AddRange(cols[..namesStart].ToArray()); // existing descriptors - blob.AddRange(descriptor); // new descriptor - blob.AddRange(cols[namesStart..].ToArray()); // existing names - blob.Add((byte)nameBytes.Length); blob.Add((byte)(nameBytes.Length >> 8)); - blob.AddRange(nameBytes); // new name - parts.Columns = [.. blob]; - } - - /// - /// Drops a column byte-faithfully with ACE (probed): a **metadata-only TDEF edit** — removes the - /// column's 25-byte descriptor and its name, and decrements the live ColumnCount (0x2D). It does - /// **not** renumber the surviving columns, recompute their fixed offsets/variable indexes, decrement the - /// VariableColumnCount (0x2B stays a high-water mark), or rewrite existing rows — survivors keep - /// their stored variable index (§3.4) so old rows still decode (the dropped column's data becomes dead - /// bytes). Returns false if the column doesn't exist. Multi-page TDEFs are handled. Throws for a column - /// that backs an index/key (drop that first). - /// A memo/OLE column also owns long-value pages through its own usage maps, and ACE retires those as - /// DROP TABLE does (measured by whole-file diff): its §3.3.2 entry leaves the TDEF, its owned and free map - /// records are retired from their holder, and its owned pages go back to the global free map at close. The - /// pages themselves are left as they were, and the other long-value columns keep their entries and records. - /// - public bool DropColumn(string tableName, string columnName) - { - TableDef table = _catalog.FindTable(tableName) - ?? throw new InvalidOperationException($"Table '{tableName}' was not found."); - ColumnDef? col = table.Columns.FirstOrDefault(c => string.Equals(c.Name, columnName, StringComparison.OrdinalIgnoreCase)); - if (col is null) return false; - - // A DELIBERATE divergence from ACE. ACE accepts dropping a column a calculated expression reads and - // simply leaves that column unevaluatable — measured: every subsequent read of it fails, and there is - // no way back short of recreating the column, because ACE offers no route to edit an expression at - // all. Refusing keeps the file readable, and the caller who really wants it gone can drop the - // calculated column first. - if (CalculatedColumnsReading(table, col) is { Count: > 0 } dependents) - throw new InvalidOperationException( - $"Column '{col.Name}' cannot be dropped because " - + $"{string.Join(", ", dependents.Select(d => $"'{d}'"))} " - + (dependents.Count == 1 - ? "is a calculated column that reads it. Drop it first." - : "are calculated columns that read it. Drop those first.")); - - // ACE rejects dropping a column that participates in a relationship (as the child FK column or the - // referenced parent key) — even a NO INDEX FK with no backing index — with "It is part of one or more - // relationships"; you must drop the relationship first. This is correct, permanent behaviour (not a - // gap), so mirror it. Verified vs ACE. - if (ColumnIsInRelationship(table, col)) - throw new InvalidOperationException( - $"Cannot drop column '{columnName}': it is part of one or more relationships — drop the relationship first."); - - // ACE likewise rejects dropping an indexed/keyed column ("part of an index or is needed by the - // system"); the index must be dropped first. Also correct, permanent behaviour. Verified vs ACE. - if (table.Indexes.Any(ix => ix.Columns.Any(c => c.Column.ColumnId == col.ColumnId))) - throw new InvalidOperationException( - $"Cannot drop column '{columnName}': it is part of an index or key — drop the index/constraint first."); - - // A long-value column's maps, read from the TDEF before it loses them. - var definition = new TableDefinitionPage(); - definition.Read(_channel, table.DefinitionPage); - var maps = new UsageMap(_channel, table); - var owned = new HashSet(); - var retire = new List(); - QueueLongValueMaps(definition, col, maps, owned, retire); - - TdefParts parts = ParseTdef(table.DefinitionPage); // stitches continuation pages for a multi-page TDEF - RemoveColumnFromParts(parts, table.Columns.Count, col.Index, _channel.Format); - RemoveLongValueMapEntry(parts, col.ColumnId); - WriteTdef(table.DefinitionPage, parts); - - RetireMapRecords(retire, maps, owned); - var allocator = new PageAllocator(_channel); - foreach (int page in owned) - allocator.Release(page); // reusable only after this handle closes, as ACE holds them - - RemoveColumnProperties(table.DefinitionPage, columnName); // drop its DefaultValue/Required from LvProp (ACE does) - _catalog.Invalidate(); - return true; - } - - /// Removes a dropped column's extended-property block (DefaultValue, Required, …) from its - /// table's MSysObjects.LvProp blob — what ACE does on DROP COLUMN (verified). Surgically removes - /// just that column's block (keeps the name pool + other columns' blocks), re-stores the smaller blob on - /// an LvProp page and updates the row. No-op when the column had no properties. - private void RemoveColumnProperties(int tdefPage, string columnName) - { - TableDef msys = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIdx = (msys.FindColumn("Id") ?? throw new InvalidOperationException("MSysObjects is missing 'Id'.")).Index; - ColumnDef lvProp = msys.FindColumn("LvProp") ?? throw new InvalidOperationException("MSysObjects is missing 'LvProp'."); - var table = new Table(_channel, msys); - - foreach ((RowId id, object?[] values) in table.Rows().WithIds()) - { - if (values[idIdx] is null || Convert.ToInt32(values[idIdx], CultureInfo.InvariantCulture) != tdefPage) continue; - if (values[lvProp.Index] is not byte[] { Length: > 0 } blob) return; - - byte[] cleaned = PropertyBlob.RemoveOwner(blob, columnName); - if (cleaned.Length == blob.Length) return; // the column had no property block — nothing to remove - - byte[] descriptor = new RowInserter(_channel, msys).StorePackedLongValue(lvProp.ColumnId, cleaned); - values[lvProp.Index] = new LongValueDescriptor(descriptor); - table.Update(id, values, new HashSet { lvProp.Index }); - return; - } - } - - /// Removes the descriptor + name of the column at from the - /// column region and decrements the header's live ColumnCount (0x2D). VariableColumnCount (0x2B) is - /// deliberately left unchanged — ACE keeps it as a high-water mark (verified). - private static void RemoveColumnFromParts(TdefParts parts, int colCount, int removeIndex, JetFormatBase format) - { - int descSize = format.ColumnDescriptorSize; - ReadOnlySpan cols = parts.Columns; - - var descriptors = new List(colCount); - for (int i = 0; i < colCount; i++) - descriptors.Add(cols.Slice(i * descSize, descSize).ToArray()); - - int np = colCount * descSize; - var names = new List(colCount); - for (int i = 0; i < colCount; i++) - { - int len = BinaryPrimitives.ReadUInt16LittleEndian(cols.Slice(np, 2)); - names.Add(cols.Slice(np, 2 + len).ToArray()); - np += 2 + len; - } - - descriptors.RemoveAt(removeIndex); - names.RemoveAt(removeIndex); - - var blob = new List(parts.Columns.Length); - foreach (byte[] d in descriptors) blob.AddRange(d); - foreach (byte[] n in names) blob.AddRange(n); - parts.Columns = [.. blob]; - - BinaryPrimitives.WriteUInt16LittleEndian(parts.Header.AsSpan(format.TdefColumnCountOffset, 2), (ushort)(colCount - 1)); - } - - /// True if is the incoming relationship block that cross-links to the - /// child's outgoing block number on the child's TDEF page (info block layout: +0x0C fk_type, - /// +0x0D child block number, +0x11 child page). - private static bool IsIncomingBlockFor(byte[] info, int childBlockNum, int childPage) => - info[IndexBlockFormat.InfoFkTypeOffset] == FkTypeIncoming && - (int)BinaryPrimitives.ReadUInt32LittleEndian(info.AsSpan(IndexBlockFormat.InfoFkNumberOffset, 4)) == childBlockNum && - BinaryPrimitives.ReadInt32LittleEndian(info.AsSpan(IndexBlockFormat.InfoFkTablePageOffset, 4)) == childPage; - - /// Soft-deletes every MSysRelationships row for the named relationship. - private void SoftDeleteRelationshipRows(string name) - { - TableDef msys = _catalog.FindTable("MSysRelationships") - ?? throw new InvalidOperationException("MSysRelationships catalog table was not found."); - int nameIdx = (msys.FindColumn("szRelationship") - ?? throw new InvalidOperationException("MSysRelationships is missing the 'szRelationship' column.")).Index; - - var rows = new List(); - foreach ((RowId id, object?[] values) in new Table(_channel, msys).Rows().WithIds()) - if (string.Equals(values[nameIdx] as string, name, StringComparison.OrdinalIgnoreCase)) - rows.Add(id); - foreach (RowId id in rows) SoftDeleteRow(id); - } - - /// The name text of a TDEF name entry (2-byte UTF-16 length, then the chars). - private static string NameOf(byte[] nameEntry) => - System.Text.Encoding.Unicode.GetString(nameEntry, 2, BinaryPrimitives.ReadUInt16LittleEndian(nameEntry.AsSpan(0, 2))); - - /// The parsed regions of a table definition, for surgical block removal. A multi-page definition - /// is stitched into one buffer by ; carries its extra - /// pages so can reuse them. - private sealed class TdefParts - { - public required byte[] Header; // [0, TdefRealIndexBlockOffset) - public required List Stats; // one 12-byte stats block per data index - public required byte[] Columns; // column descriptors + names region - public required List DataBlocks; // one 52-byte index-data block per data index - public required List<(byte[] Info, byte[] Name)> Logical; // 28-byte info block + its name, name-sorted - public required byte[] Lval; // §3.3.2 list + terminator - public IReadOnlyList Continuations = []; // continuation-page numbers (multi-page TDEF) - } - - private TdefParts ParseTdef(int tdefPage) - { - JetFormatBase format = _channel.Format; - // Stitch any continuation pages into one contiguous buffer (offsets are absolute from page 1), so the - // surgery below works the same for single- and multi-page definitions. - (LibRed.IO.PageBuffer buf, IReadOnlyList continuations) = ReadDefinition(tdefPage); - - int dataCount = buf.ReadInt32(format.TdefIndexCountOffset); - int logicalCount = buf.ReadInt32(format.TdefLogicalIndexCountOffset); - int colCount = buf.ReadUInt16(format.TdefColumnCountOffset); - - int statsStart = format.TdefRealIndexBlockOffset; - int afterStats = statsStart + dataCount * format.RealIndexEntrySize; - int pos = afterStats + colCount * format.ColumnDescriptorSize; - for (int i = 0; i < colCount; i++) pos += 2 + buf.ReadUInt16(pos); - int afterColumns = pos; - int infoStart = afterColumns + dataCount * IndexBlockFormat.DataBlockSize; - int namePos = infoStart + logicalCount * IndexBlockFormat.InfoBlockSize; - int defEnd = buf.ReadInt32(format.TdefLengthOffset); - - var stats = new List(dataCount); - for (int i = 0; i < dataCount; i++) stats.Add(buf.Slice(statsStart + i * format.RealIndexEntrySize, format.RealIndexEntrySize).ToArray()); - var dataBlocks = new List(dataCount); - for (int i = 0; i < dataCount; i++) dataBlocks.Add(buf.Slice(afterColumns + i * IndexBlockFormat.DataBlockSize, IndexBlockFormat.DataBlockSize).ToArray()); - - var logical = new List<(byte[], byte[])>(logicalCount); - int np = namePos; - for (int i = 0; i < logicalCount; i++) - { - byte[] info = buf.Slice(infoStart + i * IndexBlockFormat.InfoBlockSize, IndexBlockFormat.InfoBlockSize).ToArray(); - int len = buf.ReadUInt16(np); - byte[] nm = buf.Slice(np, 2 + len).ToArray(); - np += 2 + len; - logical.Add((info, nm)); - } - - return new TdefParts - { - Header = buf.Slice(0, statsStart).ToArray(), - Stats = stats, - Columns = buf.Slice(afterStats, afterColumns - afterStats).ToArray(), - DataBlocks = dataBlocks, - Logical = logical, - Lval = buf.Slice(np, defEnd - np).ToArray(), - Continuations = continuations, - }; - } - - /// Removes a data index (its stats + data block at , - /// decrementing the data-ordinal reference (+0x08) of every remaining info block that pointed past it) - /// and every logical block matching (with its name). - private static void RemoveTdefBlocks(TdefParts parts, int? removeDataOrdinal, Func<(byte[] Info, byte[] Name), bool> removeLogical) - { - if (removeDataOrdinal is int ord) - { - parts.Stats.RemoveAt(ord); - parts.DataBlocks.RemoveAt(ord); - foreach ((byte[] info, _) in parts.Logical) - { - int num2 = BinaryPrimitives.ReadInt32LittleEndian(info.AsSpan(0x08, 4)); - if (num2 > ord) BinaryPrimitives.WriteInt32LittleEndian(info.AsSpan(0x08, 4), num2 - 1); - } - } - parts.Logical.RemoveAll(b => removeLogical(b)); - } - - private void WriteTdef(int tdefPage, TdefParts parts) - { - JetFormatBase format = _channel.Format; - var body = new List(format.PageSize); - body.AddRange(parts.Header); - foreach (byte[] s in parts.Stats) body.AddRange(s); - body.AddRange(parts.Columns); - foreach (byte[] d in parts.DataBlocks) body.AddRange(d); - foreach ((byte[] info, _) in parts.Logical) body.AddRange(info); - foreach ((_, byte[] nm) in parts.Logical) body.AddRange(nm); - body.AddRange(parts.Lval); - byte[] def = [.. body]; - int defEnd = def.Length; - - BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefIndexCountOffset, 4), parts.DataBlocks.Count); - BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefLogicalIndexCountOffset, 4), parts.Logical.Count); - BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefLengthOffset, 4), defEnd); - - // Write across the first page and continuation pages as needed (fresh ones, the old released) — handles a - // definition that shrinks to one page, stays multi-page, or grows past a page (e.g. ADD COLUMN). - WriteDefinition(tdefPage, def, parts.Continuations, rewrite: true); - } - - /// Marks a row deleted by setting the deleted flag (0x8000) on its slot-directory entry — a - /// Jet soft delete: the row bytes stay but scans (and Access) skip it. - private void SoftDeleteRow(RowId id) - { - JetFormatBase format = _channel.Format; - byte[] page = _channel.ReadPage(id.Page).Span.ToArray(); - int dirOffset = format.DataRowDirectoryOffset + id.Row * 2; - ushort entry = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(dirOffset, 2)); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(dirOffset, 2), (ushort)(entry | 0x8000)); - _channel.WritePage(id.Page, page); - } - - /// The data-block ordinal of a table's own index over the FK's referenced columns (for a - /// self-reference — normally the primary key). - private static int ReferencedOrdinalIn(TableDef table, RelationshipSpec fk) => - FindParentKeyIndex(table.Indexes, fk.Columns.Select(c => c.ReferencedColumn).ToList(), table.Name) - .RealIndexOrdinal; - - /// The parent-side key index of a relationship: an index over exactly the referenced columns - /// that is unique or primary. - /// - /// The uniqueness requirement is ACE's, measured: over a plain non-unique index ACE refuses the - /// relationship with "No unique index found for the referenced field of the primary table", while the - /// same shape over a PRIMARY KEY succeeds. LibRed used to accept any index over the columns, which wrote - /// a relationship ACE would not have created — and one whose backing index could then be dropped, since - /// the DROP INDEX guard was looking for a unique one. - /// - private static IndexDef FindParentKeyIndex( - IReadOnlyList candidates, IReadOnlyList refColumns, string parentTable) - { - bool MatchesColumns(IndexDef ix) => - ix.Columns.Select(c => c.Column.Name).SequenceEqual(refColumns, StringComparer.OrdinalIgnoreCase); - - IndexDef? match = candidates.FirstOrDefault(ix => MatchesColumns(ix) && (ix.IsUnique || ix.IsPrimaryKey)); - if (match is not null) return match; - - // Distinguish "no index at all" from "an index, but not a unique one" — ACE's own message names the - // second case, and it is the one a caller can fix by declaring the key unique. - throw new InvalidOperationException(candidates.Any(MatchesColumns) - ? $"No unique index found for the referenced field of the primary table: '{parentTable}' " - + $"({string.Join(", ", refColumns)}) is indexed, but not uniquely." - : $"Referenced table '{parentTable}' has no index over ({string.Join(", ", refColumns)})."); - } - - /// The child (outgoing) end of a relationship: index_num2 = the child's own FK data block, - /// Fk_type = outgoing, Fk_number/Fk_table = the parent's incoming block. Mirrors the inline-FK block - /// TdefBuilder writes at creation time. - private static byte[] BuildOutgoingInfoBlock(int number, int dataOrdinal, byte fkType, int fkNumber, int fkTablePage, byte upd, byte del) - { - var b = new byte[IndexBlockFormat.InfoBlockSize]; - BinaryPrimitives.WriteUInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoMarkerOffset, 4), JetFormatBase.TdefRecordMarker); - BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoNumberOffset, 4), number); - BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoDataNumberOffset, 4), dataOrdinal); - b[IndexBlockFormat.InfoFkTypeOffset] = fkType; - BinaryPrimitives.WriteUInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoFkNumberOffset, 4), (uint)fkNumber); - BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoFkTablePageOffset, 4), fkTablePage); - b[IndexBlockFormat.InfoUpdateActionOffset] = upd; - b[IndexBlockFormat.InfoDeleteActionOffset] = del; - b[IndexBlockFormat.InfoTypeOffset] = IndexBlockFormat.TypeForeign; - return b; - } - - - /// Reads a table definition, stitching continuation pages into one contiguous buffer (in - /// the absolute coordinate space the descriptors use), and returns the continuation page numbers. - private (LibRed.IO.PageBuffer Buffer, IReadOnlyList ContinuationPages) ReadDefinition(int firstPage) - => TdefChainReader.Read(_channel, firstPage); - - /// - /// Writes a definition buffer across the first page and, if it overflows, continuation pages (each - /// [0x02][0x01][free:2][next:4] then data). The first page carries the whole definition in its - /// coordinate space; each continuation contributes -offset data. - /// Rewriting an existing definition () is done as ACE does it (verified by - /// whole-file diff, growing and shrinking): the first page is rewritten in place — alone, only the 8-byte - /// reserve past the new end is zeroed and older bytes beyond it are left — and continuation data always goes - /// to freshly allocated pages, while are released untouched. - /// - private void WriteDefinition(int firstPage, byte[] def, IReadOnlyList oldContinuations, bool rewrite) - { - JetFormatBase format = _channel.Format; - int ps = format.PageSize; - int nextOffset = format.TdefNextPageOffset; - - foreach (int old in oldContinuations) - _allocator.Release(old); // reusable only after this handle closes, as ACE holds them - - if (def.Length + JetFormatBase.TdefContinuationHeaderSize <= ps) - { - byte[] only = rewrite ? _channel.ReadPage(firstPage).Span.ToArray() : new byte[ps]; - def.CopyTo(only, 0); - only.AsSpan(def.Length, JetFormatBase.TdefContinuationHeaderSize).Clear(); // the reserve - BinaryPrimitives.WriteInt32LittleEndian(only.AsSpan(nextOffset, 4), 0); - BinaryPrimitives.WriteUInt16LittleEndian(only.AsSpan(format.TdefFreeSpaceOffset, 2), (ushort)(ps - def.Length - JetFormatBase.TdefContinuationHeaderSize)); - _channel.WritePage(firstPage, only); - return; - } - - // The chain holds the definition and then its 8-byte trailing reserve, as ACE lays it out (verified): every - // page is filled before the next begins, and the reserve follows the last definition byte, spilling onto a - // page of its own when it does not fit — so a continuation can hold reserve bytes and no definition. A - // 4,090-byte definition fills page 1 with 4,090 bytes and six of the reserve, and its continuation holds the - // other two, free 4,086. Each page's free space is what it has left once both are placed. - int maxMiddle = ps - JetFormatBase.TdefContinuationHeaderSize; - int stored = def.Length + JetFormatBase.TdefContinuationHeaderSize; - var chunks = new List<(int Offset, int Length, int Free)>(); - for (int consumed = ps; consumed < stored;) - { - int placed = Math.Min(maxMiddle, stored - consumed); - chunks.Add((consumed, Math.Clamp(def.Length - consumed, 0, placed), maxMiddle - placed)); - consumed += placed; - } - - // ACE allocates the last continuation first (verified: from free pages 354.. a two-page continuation became - // first → 355 → 354, and from the end of a file first → n+1 → n). - var pageNumbers = new int[chunks.Count]; - for (int i = chunks.Count - 1; i >= 0; i--) - pageNumbers[i] = _allocator.Allocate(); - - var page1 = new byte[ps]; - Array.Copy(def, 0, page1, 0, Math.Min(ps, def.Length)); // page 1 is completely full in a multi-page definition - BinaryPrimitives.WriteInt32LittleEndian(page1.AsSpan(nextOffset, 4), pageNumbers[0]); - BinaryPrimitives.WriteUInt16LittleEndian(page1.AsSpan(format.TdefFreeSpaceOffset, 2), 0); - _channel.WritePage(firstPage, page1); - - for (int i = 0; i < chunks.Count; i++) - { - var (offset, length, free) = chunks[i]; - var page = new byte[ps]; - page[0] = (byte)PageType.TableDefinition; - page[1] = 0x01; - if (length > 0) // a page holding only the reserve starts past the definition's end - Array.Copy(def, offset, page, JetFormatBase.TdefContinuationHeaderSize, length); - int next = i + 1 < pageNumbers.Length ? pageNumbers[i + 1] : 0; - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(nextOffset, 4), next); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.TdefFreeSpaceOffset, 2), (ushort)free); - _channel.WritePage(pageNumbers[i], page); - } - } - - private static byte[] BuildIndexDataBlock(List<(int Id, bool Ascending)> columns, int rootPage, int usageRow, int usagePage, bool unique, bool required, bool ignoreNulls) - { - var b = new byte[IndexBlockFormat.DataBlockSize]; - System.Buffers.Binary.BinaryPrimitives.WriteUInt32LittleEndian(b.AsSpan(0, 4), IndexBlockFormat.DataMarker); - for (int slot = 0; slot < IndexBlockFormat.MaxColumns; slot++) - { - int entry = IndexBlockFormat.ColumnsOffset + slot * IndexBlockFormat.ColumnSlotSize; - if (slot < columns.Count) - { - System.Buffers.Binary.BinaryPrimitives.WriteInt16LittleEndian(b.AsSpan(entry, 2), (short)columns[slot].Id); - b[entry + 2] = columns[slot].Ascending ? IndexBlockFormat.ColumnAscending : (byte)0x00; // 0x00 = descending - } - else System.Buffers.Binary.BinaryPrimitives.WriteInt16LittleEndian(b.AsSpan(entry, 2), IndexBlockFormat.ColumnUnused); - } - b[IndexBlockFormat.UsageMapRowOffset] = (byte)usageRow; - b[IndexBlockFormat.UsageMapRowOffset + 1] = (byte)usagePage; - b[IndexBlockFormat.UsageMapRowOffset + 2] = (byte)(usagePage >> 8); - b[IndexBlockFormat.UsageMapRowOffset + 3] = (byte)(usagePage >> 16); - System.Buffers.Binary.BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.RootPageOffset, 4), rootPage); - ushort flags = IndexFlags.AlwaysSet; - if (unique) flags |= IndexFlags.Unique; - if (ignoreNulls) flags |= IndexFlags.IgnoreNulls; - if (required) flags |= IndexFlags.Required; - System.Buffers.Binary.BinaryPrimitives.WriteUInt16LittleEndian(b.AsSpan(IndexBlockFormat.FlagsOffset, 2), flags); - return b; - } - - private static byte[] BuildPlainInfoBlock(int number, int dataOrdinal, bool isPrimary) - { - var b = new byte[IndexBlockFormat.InfoBlockSize]; - System.Buffers.Binary.BinaryPrimitives.WriteUInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoMarkerOffset, 4), JetFormatBase.TdefRecordMarker); - System.Buffers.Binary.BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoNumberOffset, 4), number); - System.Buffers.Binary.BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoDataNumberOffset, 4), dataOrdinal); - System.Buffers.Binary.BinaryPrimitives.WriteUInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoFkNumberOffset, 4), IndexBlockFormat.NoForeignKey); // no foreign key - b[IndexBlockFormat.InfoUpdateActionOffset] = IndexBlockFormat.PlainAction; - b[IndexBlockFormat.InfoDeleteActionOffset] = IndexBlockFormat.PlainAction; - b[IndexBlockFormat.InfoTypeOffset] = isPrimary ? IndexBlockFormat.TypePrimary : IndexBlockFormat.TypeSecondary; - return b; - } - - private static int ReadInt24(LibRed.IO.PageBuffer buf, int offset) => - buf.ReadByte(offset) | (buf.ReadByte(offset + 1) << 8) | (buf.ReadByte(offset + 2) << 16); - - /// - /// Adds an incoming-relationship logical index-info block (§3.6) to a parent table's already-written - /// TDEF: it reuses the parent's referenced-key data block (no new data block), links back to the - /// child's outgoing block, and grows the logical-index list by one (kept name-sorted). The definition is - /// rewritten through , so it may span or spill onto continuation pages. - /// - private void AddIncomingRelationshipBlock(IncomingRelationship inc) - { - JetFormatBase format = _channel.Format; - (LibRed.IO.PageBuffer buf, IReadOnlyList existingContinuations) = ReadDefinition(inc.ParentPage); - - int dataCount = buf.ReadInt32(format.TdefIndexCountOffset); // 0x33 real data blocks - int logicalCount = buf.ReadInt32(format.TdefLogicalIndexCountOffset); // 0x2F logical blocks - int colCount = buf.ReadUInt16(format.TdefColumnCountOffset); - - // An incoming relationship adds a logical block and no data block, so this is the path a referenced - // table overruns: 0x33 stays where it was while 0x2F climbs with every table that points here. - EnsureIndexCapacity( - _catalog.Tables.FirstOrDefault(t => t.DefinitionPage == inc.ParentPage)?.Name ?? $"page {inc.ParentPage}", - "an incoming relationship", dataCount, logicalCount + 1); - - // Walk to the logical index-info blocks: stats + column descriptors -> column names -> data blocks. - int pos = format.TdefRealIndexBlockOffset + dataCount * format.RealIndexEntrySize - + colCount * format.ColumnDescriptorSize; - for (int i = 0; i < colCount; i++) pos += 2 + buf.ReadUInt16(pos); - int infoStart = pos + dataCount * IndexBlockFormat.DataBlockSize; - - var blocks = new List(logicalCount + 1); - for (int i = 0; i < logicalCount; i++) - blocks.Add(buf.Slice(infoStart + i * IndexBlockFormat.InfoBlockSize, IndexBlockFormat.InfoBlockSize).ToArray()); - - int namePos = infoStart + logicalCount * IndexBlockFormat.InfoBlockSize; - var names = new List(logicalCount + 1); - var nameBytes = new List(logicalCount + 1); - for (int i = 0; i < logicalCount; i++) - { - int len = buf.ReadUInt16(namePos); - nameBytes.Add(buf.Slice(namePos, 2 + len).ToArray()); - names.Add(System.Text.Encoding.Unicode.GetString(buf.Slice(namePos + 2, len))); - namePos += 2 + len; - } - - int defEnd = buf.ReadInt32(format.TdefLengthOffset); - byte[] lvalRegion = buf.Slice(namePos, defEnd - namePos).ToArray(); // §3.3.2 list + 0xFFFF terminator - - string newName = NextHiddenRelationshipName(names); - int k = names.Count(n => string.CompareOrdinal(n, newName) < 0); // name-sorted insert position - blocks.Insert(k, BuildIncomingInfoBlock(inc)); - nameBytes.Insert(k, EncodeName(newName)); - - int newDefEnd = infoStart + blocks.Count * IndexBlockFormat.InfoBlockSize + nameBytes.Sum(n => n.Length) + lvalRegion.Length; - - var def = new byte[newDefEnd]; - buf.Span[..infoStart].CopyTo(def); - int w = infoStart; - foreach (byte[] b in blocks) { b.CopyTo(def.AsSpan(w)); w += b.Length; } - foreach (byte[] n in nameBytes) { n.CopyTo(def.AsSpan(w)); w += n.Length; } - lvalRegion.CopyTo(def.AsSpan(w)); - - BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefLogicalIndexCountOffset, 4), logicalCount + 1); - BinaryPrimitives.WriteInt32LittleEndian(def.AsSpan(format.TdefLengthOffset, 4), newDefEnd); - WriteDefinition(inc.ParentPage, def, existingContinuations, rewrite: true); - } - - private static byte[] BuildIncomingInfoBlock(IncomingRelationship inc) - { - var b = new byte[IndexBlockFormat.InfoBlockSize]; - BinaryPrimitives.WriteUInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoMarkerOffset, 4), JetFormatBase.TdefRecordMarker); - BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoNumberOffset, 4), inc.Number); // index_num - BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoDataNumberOffset, 4), inc.ReferencedOrdinal); // index_num2 -> referenced-key data block - b[IndexBlockFormat.InfoFkTypeOffset] = FkTypeIncoming; - BinaryPrimitives.WriteUInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoFkNumberOffset, 4), inc.ChildBlockNumber); // cross-link to child block - BinaryPrimitives.WriteInt32LittleEndian(b.AsSpan(IndexBlockFormat.InfoFkTablePageOffset, 4), inc.ChildPage); - b[IndexBlockFormat.InfoUpdateActionOffset] = inc.UpdateAction; - b[IndexBlockFormat.InfoDeleteActionOffset] = inc.DeleteAction; - b[IndexBlockFormat.InfoTypeOffset] = IndexBlockFormat.TypeForeign; - return b; - } - - private static byte[] EncodeName(string name) - { - byte[] chars = System.Text.Encoding.Unicode.GetBytes(name); - var entry = new byte[2 + chars.Length]; - BinaryPrimitives.WriteUInt16LittleEndian(entry.AsSpan(0, 2), (ushort)chars.Length); - chars.CopyTo(entry.AsSpan(2)); - return entry; - } - - /// The hidden name Access gives an incoming relationship index: ".r" + a letter unique - /// among the table's index names (Access starts at 'B'). - private static string NextHiddenRelationshipName(IReadOnlyCollection existing) - { - for (char c = 'B'; c <= 'Z'; c++) - { - string candidate = $".r{c}"; - if (!existing.Contains(candidate)) return candidate; - } - return $".r{Guid.NewGuid():N}"[..8]; - } - - /// Writes an empty B-tree leaf (no entries) to serve as a fresh index root. - private void WriteEmptyLeafIndexPage(JetFormatBase format, int pageNumber, int owner) - { - const int EntryDataOffset = 0x1E0; - const int OwnerOffset = 0x04; - - var page = new byte[format.PageSize]; - page[0] = (byte)PageType.LeafIndexPage; - page[1] = 0x01; // page flags (observed constant) - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(OwnerOffset, 4), owner); - // No entries: empty mask, no prefix compression, free space is the whole entry region. - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(format.PageSize - EntryDataOffset)); - _channel.WritePage(pageNumber, page); - } - - /// - /// Writes a data page of empty inline usage-map records — like - /// Access does for a fresh table that has no data page yet. Each record is - /// [0x00][startPage = 0][all-zero bitmap]: row 0 = table owned-pages, row 1 = table - /// free-pages, and (with an index) row 2 = the index's owned-pages. The first insert allocates a - /// data page and sets the corresponding bit. - /// - private void WriteUsageMaps(JetFormatBase format, int pageNumber, int mapCount) - { - // An empty inline usage map: type byte + start page (0) + a bitmap of all-zero bytes. Access - // writes a full-width bitmap; match its record length so the page layout matches byte-for-byte. - const int MapLength = UsageMapRecordLength; - - var page = new byte[format.PageSize]; - page[0] = (byte)PageType.DataPage; - page[1] = 0x01; // page flags (observed constant) - // Owner of a usage-map page is 0 (it belongs to no table). - - int offset = format.PageSize; - for (int row = 0; row < mapCount; row++) - { - offset -= MapLength; - // page[offset] already 0x00 (inline type), start page already 0, bitmap already zero. - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowDirectoryOffset + row * 2, 2), (ushort)offset); - } - - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataRowCountOffset, 2), (ushort)mapCount); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.DataFreeSpaceOffset, 2), - (ushort)(offset - format.DataRowDirectoryOffset - mapCount * 2)); - _channel.WritePage(pageNumber, page); - } - - // Jet/ACE hard limit on columns in a table. - private const int MaxColumnsPerTable = 255; - - // An inline usage-map record: type byte + 4-byte start page + a 64-byte all-zero bitmap = 69 bytes. - private const int UsageMapRecordLength = 1 + 4 + 64; - - // A user table's owner + grantee SIDs, matching the cluster DatabaseCreator seeds for the system objects - // (Users owns user objects; Users + Admin get the user-table grants). - private static readonly byte[] DefaultOwner = DatabaseCreator.SidUsers; // Users/Engine (per-file masked) - private static readonly byte[] AdminSid = DatabaseCreator.SidAdmin; // Admin user (per-file masked) - - /// - /// Adds the MSysObjects row describing the new table so Access (and the catalog) see it: Id = - /// TDEF page, ParentId = Tables container, Type = table, Name, Flags, Owner, and create/update - /// dates. Any column DEFAULT values are written into the extended-properties blob (LvProp, an OLE - /// long value) as DefaultValue properties. MSysObjects' own indexes (Id, and the composite - /// ParentId+Name used for name resolution) are maintained so Access can open the table by name. - /// - private void AddCatalogRow(string name, int tdefPage, - IReadOnlyList columnProps, - IReadOnlyList<(string Name, string Expression)> checkConstraints) - { - TableDef msysObjects = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - - DateTime now = DateTime.Now; - var values = new object?[msysObjects.Columns.Count]; - SetByName(msysObjects, values, "Id", tdefPage); - SetByName(msysObjects, values, "ParentId", CatalogFormat.ObjectContainerParentId); - SetByName(msysObjects, values, "Type", (short)1); // table object - SetByName(msysObjects, values, "Name", name); - SetByName(msysObjects, values, "Flags", 0); - SetByName(msysObjects, values, "Owner", DefaultOwner); - SetByName(msysObjects, values, "DateCreate", now); - SetByName(msysObjects, values, "DateUpdate", now); - - // Per-column properties (DefaultValue / Required) and CHECK constraints (a table property) both - // live in the object's extended-properties (LvProp) blob. - var props = columnProps.ToList(); - if (checkConstraints.Count > 0) - props.Add(new PropertyBlob.Property("", PropertyBlob.CheckConstraintsProperty, - PropertyBlob.WriteCheckList(checkConstraints))); - - var inserter = new RowInserter(_channel, msysObjects); - if (props.Count > 0) - { - // Access reads object properties only from an LVAL-page long value, not an inline one, so - // store the blob on a page (packed onto a shared LvProp page like Access) and keep the descriptor. - int lvPropColumn = (msysObjects.FindColumn("LvProp") - ?? throw new InvalidOperationException("MSysObjects is missing the 'LvProp' column.")).ColumnId; - byte[] reference = inserter.StorePackedLongValue(lvPropColumn, PropertyBlob.Write(props)); - SetByName(msysObjects, values, "LvProp", new LongValueDescriptor(reference)); - } - - inserter.Insert(values, updateIndexes: true); - } - - // A new user table's per-object-class masks (verified against DAO-created files): Users get read/write - // data (0x0F00FE), Admin gets full-access-minus-ownership (0x0FFEFF). - private const int UserTableUsersMask = 0x0F00FE; - private const int UserTableAdminMask = 0x0FFEFF; - - /// - /// Adds the two MSysACEs permission rows Access writes for a new user table (Users + Admin, with the - /// user-table masks), maintaining the table's ObjectId index so Access's security check sees them. - /// - private void AddPermissionRows(int objectId) - { - TableDef msysAces = _catalog.FindTable("MSysACEs") - ?? throw new InvalidOperationException("MSysACEs catalog table was not found."); - - foreach ((byte[] sid, int acm) in new[] { (DefaultOwner, UserTableUsersMask), (AdminSid, UserTableAdminMask) }) - { - var values = new object?[msysAces.Columns.Count]; - SetByName(msysAces, values, "ACM", acm); - SetByName(msysAces, values, "FInheritable", false); - SetByName(msysAces, values, "ObjectId", objectId); - SetByName(msysAces, values, "SID", sid); - new RowInserter(_channel, msysAces).Insert(values, updateIndexes: true); - } - } - - private static void SetByName(TableDef table, object?[] values, string column, object value) - { - ColumnDef def = table.FindColumn(column) - ?? throw new InvalidOperationException($"MSysObjects is missing the '{column}' column."); - values[def.Index] = value; - } - - private static void WriteInt24(byte[] buffer, int offset, int value) - { - buffer[offset] = (byte)value; - buffer[offset + 1] = (byte)(value >> 8); - buffer[offset + 2] = (byte)(value >> 16); - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/TableCursor.cs b/src/LibRed/LibRed.Core/Storage/TableCursor.cs index eb881e78a..bdbc8fd1c 100644 --- a/src/LibRed/LibRed.Core/Storage/TableCursor.cs +++ b/src/LibRed/LibRed.Core/Storage/TableCursor.cs @@ -7,9 +7,12 @@ namespace LibRed.Storage; /// A forward-only cursor over the rows of a table. Walks the table's data pages, /// decodes each inline row, and yields one value array per row. /// -public sealed class TableCursor(Table table) : IEnumerable +/// The table to read. +/// Which columns to decode, or null for all; see . +public sealed class TableCursor(Table table, bool[]? decode = null) : IEnumerable { private readonly Table _table = table; + private readonly bool[]? _decode = decode; public IEnumerator GetEnumerator() { @@ -21,10 +24,11 @@ public sealed class TableCursor(Table table) : IEnumerable /// reference the row (e.g. back-filling an index over existing data). public IEnumerable<(RowId Id, object?[] Values)> WithIds() { - var decoder = new RowDecoder( + var decoder = new RowCodec( _table.Definition.Columns, _table.Channel.Format, - new LongValueReader(_table.Channel)); + longValues: new LongValueStore(_table.Channel), + decode: _decode); foreach (int pageNumber in _table.UsageMap.DataPages()) { @@ -37,7 +41,7 @@ public sealed class TableCursor(Table table) : IEnumerable for (int i = 0; i < page.RowCount; i++) { - RowSlot slot = page.Rows[i]; + DataPage.RowSlot slot = page.Rows[i]; if (slot.IsDeleted) continue; // deleted, or a relocated row's hidden target (reached via its pointer) // A relocated row: the slot holds a 4-byte pointer (page<<8 | row) to the real row, which lives @@ -45,7 +49,7 @@ public sealed class TableCursor(Table table) : IEnumerable // index entries keep pointing here. Follow the pointer and decode the target's bytes. if (slot.HasOverflow) { - RelocatedRow target = RowRelocationReader.Resolve( + DataPage.RelocatedRow target = DataPage.ResolveRelocation( _table.Channel, _table.Definition.DefinitionPage, slot, page.GetRow(i)); yield return (new RowId(pageNumber, i), decoder.Decode(target.Bytes)); continue; diff --git a/src/LibRed/LibRed.Core/Storage/Types/JetTypeCodec.cs b/src/LibRed/LibRed.Core/Storage/Types/JetTypeCodec.cs index 972498602..a9a32b00a 100644 --- a/src/LibRed/LibRed.Core/Storage/Types/JetTypeCodec.cs +++ b/src/LibRed/LibRed.Core/Storage/Types/JetTypeCodec.cs @@ -7,10 +7,11 @@ namespace LibRed.Storage.Types; /// -/// Decodes individual column values from their on-disk byte representation. Centralises +/// Decodes and encodes individual column values in their on-disk byte representation. Centralises /// the per-type quirks: the 1899-12-30 OLE date epoch, Jet CURRENCY (scaled int64), -/// GUID byte order, and UTF-16LE text. Long values (memo/OLE) that live on LVAL pages -/// are not resolved here yet. +/// GUID byte order, and UTF-16LE text. A memo/OLE value that lives on LVAL pages is handled as its +/// descriptor here — resolving the pages needs page access this codec deliberately does not have, and is +/// RowCodec's job through its LongValueStore. /// public static class JetTypeCodec { @@ -31,7 +32,7 @@ public static class JetTypeCodec JetDataType.Int32 or JetDataType.Single or JetDataType.Complex => 4, JetDataType.Int64 or JetDataType.Double or JetDataType.DateTime or JetDataType.Currency => 8, JetDataType.Guid => 16, - JetDataType.FixedPoint => 17, + JetDataType.FixedPoint => NumericLength, JetDataType.DateTimeExtended => ExtendedDateTimeLength, _ => -1, }; @@ -61,22 +62,26 @@ public static class JetTypeCodec case JetDataType.Double: return BinaryPrimitives.ReadDoubleLittleEndian(value); case JetDataType.DateTime: - return DateTime.FromOADate(BinaryPrimitives.ReadDoubleLittleEndian(value)); + double serial = BinaryPrimitives.ReadDoubleLittleEndian(value); + return TryFromOaDate(serial, out DateTime date) + ? date + : throw new InvalidDataException($"Column '{column.Name}' holds {serial}, which is not a date."); case JetDataType.DateTimeExtended: // ACE 17 DATETIME2 return DecodeExtendedDateTime(value); case JetDataType.Currency: - return BinaryPrimitives.ReadInt64LittleEndian(value) / 10000m; + return CurrencyFromScaled(BinaryPrimitives.ReadInt64LittleEndian(value)); case JetDataType.Guid: return new Guid(value[..16]); case JetDataType.Text: return DecodeText(value); case JetDataType.Binary: + case JetDataType.BigBinary: return value.ToArray(); case JetDataType.FixedPoint: return DecodeNumeric(value, column.Scale); // Long values live on LVAL pages, so the inline bytes are only a descriptor. Resolving them needs - // page access this codec deliberately does not have: RowDecoder holds the LongValueReader and + // page access this codec deliberately does not have: RowCodec holds the LongValueStore and // substitutes the real value, and hands the raw bytes here only when it has none. case JetDataType.Memo: case JetDataType.Ole: @@ -128,7 +133,7 @@ private static DateTime DecodeExtendedDateTime(ReadOnlySpan value) /// /// The padding byte is 0x00, not a space: verified by reading the row bytes ACE itself wrote /// (… 3A 37 00). It matters beyond byte-faithfulness — the whole 42 bytes go into the index key - /// verbatim (see IndexKeyEncoder), so a space there would put every key we wrote out of step with + /// verbatim (see IndexKeyCodec), so a space there would put every key we wrote out of step with /// ACE's and make its seeks miss our rows. /// The precision is always 7. ACE's DDL accepts no other form: DATETIME2(7) and every other /// parenthesised spelling is a syntax error, so a Date/Time Extended column can only be declared bare, and @@ -153,18 +158,25 @@ internal static byte[] EncodeExtendedDateTime(DateTime value) /// The fixed on-disk width of a DATETIME2 (Date/Time Extended) value. internal const int ExtendedDateTimeLength = 42; + // A Decimal/Numeric value's fixed layout: a sign byte, then a 128-bit magnitude as four 32-bit little-endian + // words in big-endian word order — top, high, middle, low. + private const int NumericLength = 17; + private const byte NumericNegative = 0x80; // in the sign byte, at 0 + private const int NumericTopWord = 1, NumericHighWord = 5, NumericMidWord = 9, NumericLowWord = 13; + /// - /// Decodes a Jet Decimal/Numeric value (17 bytes): a sign byte (0x80 = negative) followed - /// by a 128-bit magnitude stored as four 32-bit little-endian words in big-endian word - /// order (the low word last). The value is the magnitude divided by 10^scale. + /// Decodes a Jet Decimal/Numeric value ( bytes): a sign byte + /// ( = negative) followed by a 128-bit magnitude stored as four 32-bit + /// little-endian words in big-endian word order (the low word last). The value is the magnitude divided by + /// 10^scale. /// private static decimal DecodeNumeric(ReadOnlySpan value, byte scale) { - bool negative = (value[0] & 0x80) != 0; - uint lo = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(13, 4)); - uint mid = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(9, 4)); - uint hi = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(5, 4)); - uint top = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(1, 4)); + bool negative = (value[0] & NumericNegative) != 0; + uint lo = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(NumericLowWord, sizeof(uint))); + uint mid = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(NumericMidWord, sizeof(uint))); + uint hi = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(NumericHighWord, sizeof(uint))); + uint top = BinaryPrimitives.ReadUInt32LittleEndian(value.Slice(NumericTopWord, sizeof(uint))); if (top != 0) throw new OverflowException("Numeric value exceeds the range of System.Decimal."); @@ -194,22 +206,83 @@ private static decimal DecodeNumeric(ReadOnlySpan value, byte scale) public static string DecodeText(ReadOnlySpan value) { if (value.Length < 2 || value[0] != 0xFF || value[1] != 0xFE) - return Encoding.Unicode.GetString(value); - - var text = new StringBuilder(value.Length - 2); + return DecodeUtf16(value); + + // With no switch byte the whole value stays in 1-byte mode, where each byte is the character of the + // same number — which is Latin-1 exactly. It is the common case by far, and one vectorised call. + ReadOnlySpan body = value[2..]; + if (!body.Contains((byte)0x00)) + return Encoding.Latin1.GetString(body); + + // Every character takes at least one byte, so the body's length bounds the decoded length. + char[]? rented = null; + Span chars = body.Length <= 256 + ? stackalloc char[256] + : (rented = System.Buffers.ArrayPool.Shared.Rent(body.Length)); + int count = 0; bool oneByte = true; - for (int i = 2; i < value.Length;) + for (int i = 0; i < body.Length;) { - if (value[i] == 0x00) { oneByte = !oneByte; i++; continue; } - if (oneByte) { text.Append((char)value[i]); i++; } + if (body[i] == 0x00) { oneByte = !oneByte; i++; continue; } + if (oneByte) { chars[count++] = (char)body[i]; i++; } else { - if (i + 1 >= value.Length) break; // a truncated trailing pair: take what is whole - text.Append((char)(value[i] | (value[i + 1] << 8))); + if (i + 1 >= body.Length) break; // a truncated trailing pair: take what is whole + chars[count++] = (char)(body[i] | (body[i + 1] << 8)); i += 2; } } - return text.ToString(); + + string text = new(chars[..count]); + if (rented is not null) + System.Buffers.ArrayPool.Shared.Return(rented); + return text; + } + + /// A CURRENCY's stored int64 (the value × 10,000) as the decimal raw / 10000m gives — the same + /// value and the same scale, which shows in its text. + /// Decimal division returns the smallest scale that holds the quotient exactly, so it amounts to + /// dropping the four places' trailing zeros. Doing that on the integer skips a full decimal divide per value, + /// which was a scan's single largest cost after text. CurrencyDecodeTests holds the two equal, bits + /// and all. + internal static decimal CurrencyFromScaled(long raw) + { + byte scale = CurrencyScale; + while (scale > 0 && raw % 10 == 0) + { + raw /= 10; + scale--; + } + + // |long.MinValue| does not fit a long, so the magnitude is taken in unsigned arithmetic. + ulong magnitude = raw < 0 ? (ulong)(-(raw + 1)) + 1 : (ulong)raw; + return new decimal((int)(uint)magnitude, (int)(uint)(magnitude >> 32), 0, raw < 0, scale); + } + + /// A value as a CURRENCY stores it — the inverse of : the decimal + /// (through , never the runtime's conversion) × 10,000, rounded. + internal static long CurrencyToScaled(object value, IFormatProvider c) => + (long)decimal.Round(JetDecimalConverter.ToDecimal(value, c) * CurrencyFactor); + + /// A CURRENCY's fixed decimal places — OLE Automation's CY, an int64 scaled by 10,000. + private const byte CurrencyScale = 4; + private const decimal CurrencyFactor = 10000m; + + /// Plain UTF-16LE text, as decodes it. + /// Without a surrogate code unit, and at an even length, UTF-16LE bytes ARE the string's chars, so + /// they are copied rather than decoded — the decoder's validation pass, counting and then converting, was + /// most of a text column's cost. Anything it could treat differently (a lone surrogate it replaces, an odd + /// trailing byte) still goes through it. + private static string DecodeUtf16(ReadOnlySpan value) + { + if (BitConverter.IsLittleEndian && (value.Length & 1) == 0) + { + ReadOnlySpan chars = System.Runtime.InteropServices.MemoryMarshal.Cast(value); + if (!chars.ContainsAnyInRange('\uD800', '\uDFFF')) + return new string(chars); + } + + return Encoding.Unicode.GetString(value); } /// @@ -221,19 +294,19 @@ public static string DecodeText(ReadOnlySpan value) /// one reaches here as the pre-built descriptor of a value /// has already put on LVAL pages. /// - public static byte[] Encode(ColumnDef column, object value) => Encode(column, column.Type, value); + public static byte[] Encode(ColumnDef column, object value, JetFormatBase format) => Encode(column, column.Type, value, format); /// Encodes as rather than the column's declared /// type — the write-side counterpart of the /// overload, and needed for the same reason: a calculated column's payload is encoded in its /// ResultType, not in the promoted type its descriptor carries. - internal static byte[] Encode(ColumnDef column, JetDataType type, object value) + internal static byte[] Encode(ColumnDef column, JetDataType type, object value, JetFormatBase format) { var c = System.Globalization.CultureInfo.InvariantCulture; // A long value already written to an LVAL page arrives as its pre-built 12-byte reference // descriptor, which is written verbatim (memo/OLE columns only). - if (value is LibRed.Storage.LongValueDescriptor descriptor) + if (value is LibRed.Storage.LongValueStore.DescriptorValue descriptor) return descriptor.Bytes; // Jet represents a boolean as -1 (true) / 0 (false). EF maps a CLR bool onto a numeric @@ -261,33 +334,35 @@ internal static byte[] Encode(ColumnDef column, JetDataType type, object value) case JetDataType.Double: return Bytes(8, b => BinaryPrimitives.WriteDoubleLittleEndian(b, Convert.ToDouble(value, c))); case JetDataType.DateTime: - return Bytes(8, b => BinaryPrimitives.WriteDoubleLittleEndian(b, ToOaDate(value, c))); + return Bytes(8, b => BinaryPrimitives.WriteDoubleLittleEndian(b, ToOaDate(column, value, c))); case JetDataType.DateTimeExtended: // ACE 17 DATETIME2 return EncodeExtendedDateTime(Convert.ToDateTime(value, c)); case JetDataType.Currency: - return Bytes(8, b => BinaryPrimitives.WriteInt64LittleEndian(b, (long)decimal.Round(JetDecimalConverter.ToDecimal(value, c) * 10000m))); + return Bytes(8, b => BinaryPrimitives.WriteInt64LittleEndian(b, CurrencyToScaled(value, c))); case JetDataType.Guid: // Coerced, not cast: every other type here accepts what the caller has (AsText, AsBinary, ToOaDate, - // Convert.To*), and TableCreator.ConvertValue already parses a string GUID on the ALTER path. A hard + // Convert.To*), and SchemaEditor.ConvertValue already parses a string GUID on the ALTER path. A hard // cast turned a string reaching a GUID column into an InvalidCastException with no column named. return (value switch { Guid g => g, byte[] b when b.Length == 16 => new Guid(b), - string s when Guid.TryParse(s, out Guid parsed) => parsed, + string s when TryParseGuid(s, out Guid parsed) => parsed, _ => throw new NotSupportedException( $"Cannot store {value.GetType().Name} in GUID column '{column.Name}'."), }).ToByteArray(); case JetDataType.Text: return EncodeText(column, AsText(value, c)); case JetDataType.Binary: + case JetDataType.BigBinary: return EncodeBinary(column, AsBinary(column, value)); case JetDataType.FixedPoint: return EncodeNumeric(column, JetDecimalConverter.ToDecimal(value, c)); // Long values (memo/OLE): store the payload inline after the 12-byte descriptor (memo - // text as UTF-16LE, OLE as raw bytes). LongValueReader reads this back via the inline - // flag. Chained LVAL pages for values too large to inline are not written yet. + // text as UTF-16LE, OLE as raw bytes). LongValueStore reads this back via the inline flag. Only a + // value the caller has already decided to inline reaches here; anything larger is written to LVAL + // pages, single or chained, by LongValueStore before the row is encoded. case JetDataType.Memo: { // An inline memo compresses whether or not the column was declared WITH COMPRESSION — the @@ -295,10 +370,10 @@ string s when Guid.TryParse(s, out Guid parsed) => parsed, // the caller has already decided to inline reach here, so no storage-form test is needed. string memo = AsText(value, c); return EncodeInlineLongValue( - TryCompressText(column, memo, requireCapableFlag: false) ?? Encoding.Unicode.GetBytes(memo)); + TryCompressText(column, memo, requireCapableFlag: false) ?? Encoding.Unicode.GetBytes(memo), format); } case JetDataType.Ole: - return EncodeInlineLongValue(AsBinary(column, value)); + return EncodeInlineLongValue(AsBinary(column, value), format); default: throw new NotSupportedException($"Encoding {column.Type} is not supported yet."); @@ -394,7 +469,7 @@ private static byte[] EncodeText(ColumnDef column, string value) /// The padding exists for the short case — ACE stores fixed text space-padded to the full width — but it /// truncates the long case just as silently, so CHAR(3) accepted 'abcdef' and stored 'abc' while the /// variable column of the same width raised. Same message and units as the variable-width check in - /// RowEncoder, because to a caller it is the same mistake. + /// RowCodec, because to a caller it is the same mistake. internal static void EnsureFitsFixedWidth(ColumnDef column, int encodedLength, int declaredUnits) { if (encodedLength <= column.Length) return; @@ -458,43 +533,64 @@ private static byte[] Bytes(int length, Action> write) }; /// The OLE-automation epoch (1899-12-30), which is also Jet's zero date and the base for - /// storing a / as a date offset. - private static readonly DateTime OleEpoch = new(1899, 12, 30); + /// storing a / as a date offset — on the way in here, on the way + /// out of a reader, and for a parameter bound to one. + public static readonly DateTime OleEpoch = new(1899, 12, 30); /// /// Converts a date/time-ish CLR value to the OLE-automation double stored in a Jet DateTime /// column. Jet has no dedicated TimeSpan/DateOnly/TimeOnly type, so — like EFCore.Jet — a /// and are stored as an offset from the epoch, and a /// as that date at midnight. + /// ACE stores nothing before 100-01-01 (serial -657434) and refuses a date or serial below it + /// (verified). .NET's ToOADate would instead throw for such a date without naming the column — or, + /// for , return 0.0 and store the epoch — so the floor is checked here. The + /// index key encodes through this too, so a key can never name a date its row could not hold. /// - private static double ToOaDate(object value, IFormatProvider c) => value switch + internal static double ToOaDate(ColumnDef column, object value, IFormatProvider c) { - DateTime dt => dt.ToOADate(), - TimeSpan ts => (OleEpoch + ts).ToOADate(), - DateOnly d => d.ToDateTime(TimeOnly.MinValue).ToOADate(), - TimeOnly t => (OleEpoch + t.ToTimeSpan()).ToOADate(), - _ => Convert.ToDateTime(value, c).ToOADate(), - }; + DateTime date = value switch + { + DateTime dt => dt, + TimeSpan ts => OleEpoch + ts, + DateOnly d => d.ToDateTime(TimeOnly.MinValue), + TimeOnly t => OleEpoch + t.ToTimeSpan(), + _ => Convert.ToDateTime(value, c), + }; + if (date < MinOaDate) + throw new InvalidOperationException( + $"Date {date:yyyy-MM-dd} is out of range for column '{column.Name}': " + + "the earliest date is 0100-01-01."); + return date.ToOADate(); + } - /// - /// Builds an inline long-value (memo/OLE) in-row value: a 12-byte descriptor - /// (length combined with the 0x80 inline flag, then 8 unused bytes) followed by the payload. - /// This is the exact shape reads back for an inline - /// value. - /// - internal static byte[] EncodeInlineLongValue(ReadOnlySpan payload) + /// The earliest date ACE stores: 0100-01-01, serial -657434. + private static readonly DateTime MinOaDate = new(100, 1, 1); + + /// The date at an OLE Automation serial read from a file, or false where + /// would refuse it — not finite, or outside its range — so a caller reports a bad value as it chooses rather than + /// as an argument error from inside the conversion. + internal static bool TryFromOaDate(double serial, out DateTime date) { - LongValueFormat.ValidateLength(payload.Length); - var result = new byte[12 + payload.Length]; - BinaryPrimitives.WriteUInt32LittleEndian(result, (uint)payload.Length | 0x80000000u); - payload.CopyTo(result.AsSpan(12)); - return result; + // FromOADate's own bounds, both exclusive: 0100-01-01 through 9999-12-31. + const double MinSerial = -657435.0, MaxSerial = 2958466.0; + bool valid = double.IsFinite(serial) && serial > MinSerial && serial < MaxSerial; + date = valid ? DateTime.FromOADate(serial) : default; + return valid; } - /// Inverse of : 17 bytes, sign + 128-bit magnitude (top word 0). + /// + /// Builds an inline long-value (memo/OLE) in-row value: the descriptor + /// (, no pointer, no stamp) followed by the payload. This is the exact + /// shape reads back for an inline value. + /// + internal static byte[] EncodeInlineLongValue(ReadOnlySpan payload, JetFormatBase format) => + [.. LongValueStore.Descriptor(format, payload.Length, LongValueStore.StorageKind.Inline), .. payload]; + /// The largest precision Jet/ACE accepts on a NUMERIC/DECIMAL column. internal const byte MaxNumericPrecision = 28; + /// Inverse of . private static byte[] EncodeNumeric(ColumnDef column, decimal value) { byte scale = column.Scale; @@ -508,7 +604,7 @@ private static byte[] EncodeNumeric(ColumnDef column, decimal value) throw new InvalidOperationException( $"Value {value} does not fit column '{column.Name}', declared " + $"DECIMAL({column.Precision},{column.Scale}): it holds at most {column.Precision - scale} " - + $"digits before the decimal point. Access refuses such a value rather than storing it."); + + "digits before the decimal point."); decimal factor = 1m; for (int i = 0; i < scale; i++) factor *= 10m; @@ -516,16 +612,27 @@ private static byte[] EncodeNumeric(ColumnDef column, decimal value) // Truncate toward zero: ACE coerces excess scale rather than refusing it, and truncation matches it in // every measured case (1.23456 → 1.2345, 1.99999 → 1.9999, -1.23455 → -1.2345). Was decimal.Round(…, 0) // — ToEven — which differed silently, each engine reading its own answer back happily. - // IndexKeyEncoder.EncodeFixedPoint must quantise identically or keys stop matching their rows. + // IndexKeyCodec.EncodeFixedPoint must quantise identically or keys stop matching their rows. decimal magnitude = decimal.Truncate(Math.Abs(value) * factor); int[] bits = decimal.GetBits(magnitude); // [lo, mid, hi, flags]; magnitude has scale 0 - var result = new byte[17]; - result[0] = (byte)(value < 0 ? 0x80 : 0x00); - // bytes[1..5) top word = 0; hi at 5, mid at 9, lo at 13 (see DecodeNumeric). - BinaryPrimitives.WriteUInt32LittleEndian(result.AsSpan(5, 4), (uint)bits[2]); - BinaryPrimitives.WriteUInt32LittleEndian(result.AsSpan(9, 4), (uint)bits[1]); - BinaryPrimitives.WriteUInt32LittleEndian(result.AsSpan(13, 4), (uint)bits[0]); + var result = new byte[NumericLength]; + // IsNegative, not < 0: a negative zero read back from ACE's row (-0.0000m) has to be written back as one. + result[0] = decimal.IsNegative(value) ? NumericNegative : (byte)0x00; + // The top word stays 0: a System.Decimal is 96 bits. + BinaryPrimitives.WriteUInt32LittleEndian(result.AsSpan(NumericHighWord, sizeof(uint)), (uint)bits[2]); + BinaryPrimitives.WriteUInt32LittleEndian(result.AsSpan(NumericMidWord, sizeof(uint)), (uint)bits[1]); + BinaryPrimitives.WriteUInt32LittleEndian(result.AsSpan(NumericLowWord, sizeof(uint)), (uint)bits[0]); return result; } + + /// Text read as a GUID: the forms reads, and ACE's + /// {guid {…}} (verified vs ACE: it inserts '{guid {…}}' into a GUID column). The key encoder reads a GUID + /// the same way, so a row and its index key always come from one parse. + internal static bool TryParseGuid(string text, out Guid guid) + { + string s = text.Trim(); + if (s.StartsWith("{guid", StringComparison.OrdinalIgnoreCase) && s.EndsWith('}')) s = s[5..^1].Trim(); + return Guid.TryParse(s, out guid); + } } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/UsageMap.cs b/src/LibRed/LibRed.Core/Storage/UsageMap.cs index 147cb0566..305848100 100644 --- a/src/LibRed/LibRed.Core/Storage/UsageMap.cs +++ b/src/LibRed/LibRed.Core/Storage/UsageMap.cs @@ -11,23 +11,34 @@ namespace LibRed.Storage; /// /// /// The TDEF holds a pointer (row + page) to the usage-map record. An inline map -/// (type 0x00) stores a start page and a bitmap where bit i marks page (startPage + i) -/// as owned. A reference map (type 0x01, for very large tables) instead stores a list of -/// pointers to dedicated bitmap pages (type 0x05); pointer k's bitmap covers the page -/// range starting at k * (pageSize - 4) * 8. +/// () stores a start page and a bitmap where bit i marks page (startPage + i) +/// as owned. A reference map (, for very large tables) instead stores a list +/// of pointers to dedicated bitmap pages (page type 0x0105); pointer k's bitmap covers the page range starting +/// at k * . /// -public sealed class UsageMap(PageChannel channel, TableDef table) +internal sealed class UsageMap(PageChannel channel, TableDefinition? table = null) { - private const byte MapTypeInline = 0x00; - private const byte MapTypeReference = 0x01; - private const int ReferenceMapSlots = 17; - private const int ReferenceMapRecordSize = 1 + ReferenceMapSlots * 4; - - /// Bytes preceding the bitmap on a dedicated usage-bitmap page (type 0x05). - private const int BitmapPageHeaderSize = 4; + /// Builds the global map page: two full-width inline maps. Row 0 is the free map — pages < + /// are used (bit 0); pages from there to the map's reach are marked free + /// (bit 1), pre-declaring space beyond the file end so an allocator (LibRed's or Access's) grabs a "free" page + /// and grows the file. Row 1 is the released map, empty, as real files carry. + internal static byte[] BuildGlobalMapPage(JetFormatBase format, int usedPages) + { + // The holder's owner field reads 1 on the global map page (observed), where a table's usage-map page has 0. + byte[] page = UsageMap.NewMapPage(format, mapCount: 2, owner: 1); + + Span freeMap = UsageMap.InlineBits( + page.AsSpan(DataPage.ReadSlot(page, format, 0).Offset, format.UsageMapInlineRecordSize), format); // row 0's + for (int p = usedPages; p < format.UsageMapInlineBitmapSize * 8; p++) + BitmapBits.Set(freeMap, p, true); + return page; + } private readonly PageChannel _channel = channel; - private readonly TableDef _table = table; + private readonly TableDefinition? _table = table; + + private int DefinitionPage => _table?.DefinitionPage + ?? throw new InvalidOperationException("A table definition is required to enumerate its data pages."); /// Yields the page numbers of every data page owned by the table, in ascending order. public IEnumerable DataPages() => PagesAt(_channel.Format.TdefOwnedPagesOffset); @@ -37,215 +48,504 @@ public sealed class UsageMap(PageChannel channel, TableDef table) /// being appended to, so it is the map to consult when looking for somewhere to put a new row. public IEnumerable FreeDataPages() => PagesAt(_channel.Format.TdefFreePagesOffset); - /// The pages recorded by the usage map at an explicit : pointer, rather than one of the TDEF's two fixed-offset maps. A long-value column's - /// owned and free maps are reached this way — their pointers sit in the TDEF keyed by column id, so the - /// pages holding a table's Memo/OLE content are invisible to . - public IEnumerable PagesInMap(int mapRow, int mapPage) => ReadMapAt(mapRow, mapPage); + /// The highest-numbered data page the table owns, or -1 when it owns none. + public int MaxDataPage() => EdgeDataPage(fromEnd: true); + + /// The lowest-numbered data page the table owns, or -1 when it owns none. + public int MinDataPage() => EdgeDataPage(fromEnd: false); - /// The dedicated bitmap pages (type 0x05) a reference-form map record at the pointer names, each - /// validated; none for an inline record. - public IReadOnlyList BitmapPagesOf(int mapRow, int mapPage) + /// The owned-pages map's last () or first page. + /// + /// Scans the bitmap from one end rather than enumerating and taking the extreme: + /// callers ask this on every page allocation and on every delete that empties a page, and materializing + /// every owned page each time would make a bulk load quadratic. Cost here is bounded by the bitmap size, + /// not the table's page count. Both ends share this walk so only the scan direction differs. + /// + private int EdgeDataPage(bool fromEnd) { - if (mapPage <= 1 || mapPage >= _channel.PageCount) - throw new InvalidDataException( - $"Usage-map pointer names page {mapPage}, outside the file's 2..{_channel.PageCount - 1} range."); - var holder = new DataPage(); - holder.Read(_channel.ReadPage(mapPage), _channel.Format); - if (mapRow < 0 || mapRow >= holder.RowCount) - throw new InvalidDataException($"Usage-map row {mapPage}:{mapRow} does not exist."); - ReadOnlySpan map = holder.GetRow(mapRow); - if (map.Length == 0 || map[0] != MapTypeReference) return []; + JetFormatBase format = _channel.Format; + PageBuffer tdef = _channel.ReadPage(DefinitionPage); + (int row, int page) = tdef.ReadRecordPointer(format.TdefOwnedPagesOffset); + ReadOnlySpan record = ReadRecordAt(row, page); - ValidateReferenceRecord(map); - var pages = new List(); - for (int e = 0; e < ReferenceMapSlots; e++) + if (RecordType(record) == UsageMapType.Inline) { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1 + e * 4, 4)); + int bit = EdgeSetBit(record[format.UsageMapInlineHeaderSize..], fromEnd); + return bit < 0 ? -1 : StartPage(record, format) + bit; + } + + for (int k = 0; k < format.UsageMapReferenceSlots; k++) + { + int e = fromEnd ? format.UsageMapReferenceSlots - 1 - k : k; + int bitmapPage = ReferencePointer(record, e, format); if (bitmapPage == 0) continue; - _ = ReadBitmapPage(bitmapPage); - pages.Add(bitmapPage); + + int found = EdgeSetBit(BitmapPageBits(ReadBitmapPage(_channel, bitmapPage).Span, format), fromEnd); + if (found >= 0) return e * format.UsageMapPagesPerBitmapPage + found; } - return pages; + return -1; } - /// The highest-numbered data page the table owns, or -1 when it owns none. - /// - /// Scans the bitmap backwards rather than enumerating and taking the maximum: - /// callers ask this on every page allocation, and materializing every owned page each time would make a - /// bulk load quadratic. Cost here is bounded by the bitmap size, not the table's page count. - /// - public int MaxDataPage() - { - byte[] record = ReadMapRecord(_channel.Format.TdefOwnedPagesOffset); + private static int EdgeSetBit(ReadOnlySpan bitmap, bool fromEnd) => + fromEnd ? BitmapBits.LastSetBit(bitmap) : BitmapBits.NextSetBit(bitmap, 0); - if (record.Length == 0) - throw new InvalidDataException("A usage-map record cannot be empty."); + /// Reads the usage map whose (row, page) pointer sits at in + /// the TDEF. Both maps share the same pointer shape and record format. + private IEnumerable PagesAt(int pointerOffset) + { + (int row, int page) = _channel.ReadPage(DefinitionPage).ReadRecordPointer(pointerOffset); + return PagesInMap(row, page); + } - if (record[0] == MapTypeInline) + /// The pages recorded by the usage map at a : + /// pointer. The TDEF's own two maps come through here (, ), + /// and so does every other map reached by an explicit pointer — a long-value column's owned and free maps, + /// whose pointers sit in the TDEF keyed by column id, and an index's own map. + public IEnumerable PagesInMap(int mapRow, int mapPage) => + PagesInRecord(_channel, ReadRecordAt(mapRow, mapPage), _channel.PageCount, "A usage map"); + + /// The pages a map marks, inline or reference form. Pointer k's bitmap covers + /// the page range starting at k * UsageMapPagesPerBitmapPage; a zero pointer means the range has no pages in the + /// map, and a bitmap page belongs to one range only. A page at or past is + /// corruption; null keeps every bit, for a map that may name pages past the file's end (see + /// ). names the map in the messages. + internal static List PagesInRecord(PageChannel channel, ReadOnlySpan record, int? rejectBeyond, string what) + { + JetFormatBase format = channel.Format; + var pages = new List(); + if (RecordType(record) == UsageMapType.Inline) { - if (record.Length < 5) - throw new InvalidDataException("An inline usage-map record must contain its 5-byte header."); - int startPage = BinaryPrimitives.ReadInt32LittleEndian(record.AsSpan(1, 4)); - int bit = HighestSetBit(record.AsSpan(5)); - return bit < 0 ? -1 : startPage + bit; + BitmapBits.AppendPages(pages, record[format.UsageMapInlineHeaderSize..], StartPage(record, format), + rejectBeyond, what); + return pages; } - if (record[0] != MapTypeReference) - throw new NotSupportedException($"Unknown usage map type 0x{record[0]:X2}."); - - ValidateReferenceRecord(record); - - int pagesPerBitmap = (_channel.PageSize - BitmapPageHeaderSize) * 8; - for (int e = ReferenceMapSlots - 1; e >= 0; e--) + var bitmapPages = new HashSet(); + for (int e = 0; e < format.UsageMapReferenceSlots; e++) { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(record.AsSpan(1 + e * 4, 4)); + int bitmapPage = ReferencePointer(record, e, format); if (bitmapPage == 0) continue; - - int bit = HighestSetBit(ReadBitmapPage(bitmapPage)); - if (bit >= 0) return e * pagesPerBitmap + bit; + if (!bitmapPages.Add(bitmapPage)) + throw new InvalidDataException($"{what} repeats bitmap page {bitmapPage}."); + BitmapBits.AppendPages(pages, BitmapPageBits(ReadBitmapPage(channel, bitmapPage).Span, format), + e * format.UsageMapPagesPerBitmapPage, rejectBeyond, what); } + return pages; + } - return -1; + /// The usage-map record a (row, page) pointer out of a TDEF names: , and on top + /// of it an owner-zero holder — a table's, an index's and a long-value column's maps all live on usage-map + /// pages that belong to no table. + /// Both halves of the pointer come out of the TDEF, so both are corruption when wrong. Unchecked, + /// the page number reached the channel as an out-of-range read and the row number reached GetRow as an + /// index. + internal ReadOnlySpan ReadRecordAt(int mapRow, int mapPage) + { + (PageBuffer page, _, DataPage.RowSlot slot) = ReadRecord(_channel, mapRow, mapPage, "Usage-map pointer"); + if (DataPage.ReadOwner(page.Span, _channel.Format) != 0) + throw new InvalidDataException( + $"Usage-map pointer {mapPage}:{mapRow} does not target an owner-zero usage-map data page."); + return page.Slice(slot.Offset, slot.Length); } - /// Index of the highest set bit in , or -1 if it is all zeros. - private static int HighestSetBit(ReadOnlySpan bitmap) + /// + /// The usage-map record a (row, page) pointer names, after proving the pointer names one: a data page inside the + /// file, a live, non-empty record on it, and a record of a known type and a valid length. The one place a map + /// record is located — the TDEF's maps here, page 0's global maps in , and every map + /// writes; names the pointer in the messages. The holder + /// comes back parsed as well, for a writer that repacks it. + /// + internal static (PageBuffer Page, DataPage Holder, DataPage.RowSlot Slot) ReadRecord( + PageChannel channel, int mapRow, int mapPage, string what) { - for (int i = bitmap.Length - 1; i >= 0; i--) + JetFormatBase format = channel.Format; + if (mapPage <= 0 || mapPage >= channel.PageCount) + throw new InvalidDataException( + $"{what} names page {mapPage}, outside the file's 1..{channel.PageCount - 1} range."); + + PageBuffer page = channel.ReadPageShared(mapPage); + if (PageHeader.ReadType(page.Span) != PageType.DataPage) + throw new InvalidDataException($"{what} names page {mapPage}, which is not a data page."); + var holder = new DataPage(); + holder.Read(page, format); + if (mapRow < 0 || mapRow >= holder.RowCount) + throw new InvalidDataException($"{what} names row {mapPage}:{mapRow}, which does not exist."); + DataPage.RowSlot slot = holder.Rows[mapRow]; + if (slot.IsDeleted || slot.HasOverflow || slot.Length == 0) + throw new InvalidDataException($"{what} names row {mapPage}:{mapRow}, which is deleted, overflowed, or empty."); + + ReadOnlySpan record = page.Slice(slot.Offset, slot.Length); + switch (RecordType(record)) { - if (bitmap[i] == 0) continue; - for (int bit = 7; bit >= 0; bit--) - if ((bitmap[i] & (1 << bit)) != 0) - return i * 8 + bit; + case UsageMapType.Inline: + if (record.Length < format.UsageMapInlineHeaderSize) + throw new InvalidDataException( + $"An inline usage-map record must contain its {format.UsageMapInlineHeaderSize}-byte header."); + break; + case UsageMapType.Reference: + if (record.Length != format.UsageMapReferenceRecordSize) + throw new InvalidDataException( + $"A reference usage-map record must be exactly {format.UsageMapReferenceRecordSize} bytes; got {record.Length}."); + break; + default: + throw new InvalidDataException($"{what} names a usage map of unknown type 0x{record[0]:X2}."); } - return -1; + return (page, holder, slot); } - /// The raw usage-map record whose (row, page) pointer sits at - /// in the TDEF. - private byte[] ReadMapRecord(int pointerOffset) - { - JetFormatBase format = _channel.Format; - PageBuffer tdef = _channel.ReadPage(_table.DefinitionPage); + /// A record's , its first byte. + internal static UsageMapType RecordType(ReadOnlySpan record) => (UsageMapType)record[0]; - var holder = new DataPage(); - holder.Read(_channel.ReadPage(tdef.ReadInt24(pointerOffset + 1)), format); - return holder.GetRow(tdef.ReadByte(pointerOffset)).ToArray(); + /// A new inline record starting at with bytes + /// of bitmap, every bit clear. + internal static byte[] NewInlineRecord(JetFormatBase format, int startPage, int bitmapBytes) + { + var record = new byte[format.UsageMapInlineHeaderSize + bitmapBytes]; + record[0] = (byte)UsageMapType.Inline; + BinaryPrimitives.WriteInt32LittleEndian(record.AsSpan(format.UsageMapStartPageOffset, sizeof(int)), startPage); + return record; } - /// Reads the usage map whose (row, page) pointer sits at in - /// the TDEF. Both maps share the same pointer shape and record format. - private List PagesAt(int pointerOffset) + /// A new reference record with no bitmap pages. + internal static byte[] NewReferenceRecord(JetFormatBase format) { - PageBuffer tdef = _channel.ReadPage(_table.DefinitionPage); - return ReadMapAt(tdef.ReadByte(pointerOffset), tdef.ReadInt24(pointerOffset + 1)); + var record = new byte[format.UsageMapReferenceRecordSize]; + record[0] = (byte)UsageMapType.Reference; + return record; } - /// Reads the usage-map record at a (row, page) pointer. Shared by the TDEF's own two maps and by - /// the per-column long-value maps, which differ only in where the pointer is stored. - private List ReadMapAt(int mapRow, int mapPage) + /// An inline record's bitmap — what follows its header. + internal static Span InlineBits(Span record, JetFormatBase format) => + record[format.UsageMapInlineHeaderSize..]; + + /// An inline record's start page. + internal static int StartPage(ReadOnlySpan record, JetFormatBase format) => + BinaryPrimitives.ReadInt32LittleEndian(record.Slice(format.UsageMapStartPageOffset, sizeof(int))); + + /// A reference record's pointer for range ; 0 when the range has no bitmap page. + internal static int ReferencePointer(ReadOnlySpan record, int slot, JetFormatBase format) => + BinaryPrimitives.ReadInt32LittleEndian(record.Slice(ReferencePointerOffset(slot, format), sizeof(int))); + + /// Writes a reference record's pointer for range . + internal static void WriteReferencePointer(Span record, int slot, JetFormatBase format, int bitmapPage) => + BinaryPrimitives.WriteInt32LittleEndian(record.Slice(ReferencePointerOffset(slot, format), sizeof(int)), bitmapPage); + + private static int ReferencePointerOffset(int slot, JetFormatBase format) => + format.UsageMapReferencePointersOffset + slot * sizeof(int); + + /// Whether carries a dedicated bitmap page's complete + /// [05 01 00 00] header — the check every reader and writer makes before trusting a pointer to one. + private static bool IsBitmapPage(ReadOnlySpan page, JetFormatBase format) => + PageHeader.ReadType(page) == PageType.PageUsageBitmap + && !page[sizeof(ushort)..format.UsageMapBitmapPageHeaderSize].ContainsAnyExcept((byte)0); // past the type: zero + + /// + /// The dedicated bitmap page (type 0x0105) a reference record's pointer names, after proving it is one: inside the + /// file and carrying the complete bitmap-page header. The pointer comes out of the file, so this is the one check + /// every reader and writer makes before trusting it — the spec states it as mandatory "before any bitmap is + /// expanded into page numbers", and a writer that skipped it OR'd a bit into an ordinary data, TDEF or index page. + /// The bytes are the channel's shared copy; a writer copies them before changing any. + /// + internal static PageBuffer ReadBitmapPage(PageChannel channel, int pageNumber) { - // Both halves of the pointer come out of the TDEF, so both are corruption when wrong. Unchecked, the - // page number reached the channel as an out-of-range read and the row number reached GetRow as an - // index; the long-value map's equivalent pointer is validated the same way in RowInserter.MapPages. - if (mapPage <= 1 || mapPage >= _channel.PageCount) + if (pageNumber <= 0 || pageNumber >= channel.PageCount) + throw new InvalidDataException( + $"Usage map names bitmap page {pageNumber}, outside the file's 1..{channel.PageCount - 1} range."); + PageBuffer page = channel.ReadPageShared(pageNumber); + if (!IsBitmapPage(page.Span, channel.Format)) throw new InvalidDataException( - $"Usage-map pointer names page {mapPage}, outside the file's 2..{_channel.PageCount - 1} range."); + $"Page {pageNumber} is named as a usage bitmap but does not carry the [05 01 00 00] header."); + return page; + } - var holder = new DataPage(); - holder.Read(_channel.ReadPage(mapPage), _channel.Format); - if (mapRow < 0 || mapRow >= holder.RowCount) - throw new InvalidDataException($"Usage-map row {mapPage}:{mapRow} does not exist."); - ReadOnlySpan map = holder.GetRow(mapRow); + /// A bitmap page's bits — what follows its header. + internal static ReadOnlySpan BitmapPageBits(ReadOnlySpan page, JetFormatBase format) => + page[format.UsageMapBitmapPageHeaderSize..]; - if (map.Length == 0) - throw new InvalidDataException("A usage-map record cannot be empty."); + /// A bitmap page's bits, writable — over a page copy the caller will write back. + internal static Span BitmapPageBits(byte[] page, JetFormatBase format) => + page.AsSpan(format.UsageMapBitmapPageHeaderSize); - return map[0] switch - { - MapTypeInline => ReadInlineMap(map), - MapTypeReference => ReadReferenceMap(map), - byte t => throw new NotSupportedException($"Unknown usage map type 0x{t:X2}."), - }; + /// A fresh, empty dedicated bitmap page (type 0x0105). + internal static byte[] NewBitmapPage(JetFormatBase format) + { + var bitmap = new byte[format.PageSize]; + PageHeader.WriteType(bitmap, PageType.PageUsageBitmap); + return bitmap; } - private List ReadInlineMap(ReadOnlySpan map) + /// A usage-map data page holding empty full-width inline records — + /// what Access writes for a fresh table that has no data page yet. Each record is [0x00][startPage = 0] + /// [all-zero bitmap], row 0 nearest the page end. A table's map page belongs to no table (owner 0); the + /// global maps' holder carries 1. + internal static byte[] NewMapPage(JetFormatBase format, int mapCount, uint owner = 0) { - if (map.Length < 5) - throw new InvalidDataException("An inline usage-map record must contain its 5-byte header."); - int startPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1, 4)); - var pages = new List(); - AppendSetBits(pages, map[5..], startPage); - return pages; + byte[] page = DataPage.NewPage(format, owner); + // An all-zero record is an empty inline map: inline type, start page 0, zero bitmap. + var records = Enumerable.Range(0, mapCount).Select(_ => new byte[format.UsageMapInlineRecordSize]).ToArray(); + DataPage.LayRows(page, format, records, new RowSlotFlags[mapCount]); + return page; } - private List ReadReferenceMap(ReadOnlySpan map) + /// How many empty full-width inline records, each with its directory slot, fit in + /// of a usage-map page. + internal static int RecordsFitting(JetFormatBase format, int freeBytes) => + freeBytes / (format.UsageMapInlineRecordSize + format.DataRowDirectoryEntrySize); + + /// Sets or clears the bit for in the usage map at record + /// on . + /// + /// Handles both map types. An inline map is grown in place while its record still fits the page; once it + /// cannot, the map is converted to a reference map — exactly Access's own threshold. + /// + /// marks a map whose set bits stay clustered near the append tail — a + /// free-pages map. Rather than growing a bitmap from startPage = 0 all the way out to the tail, + /// such a map slides a fixed full-width window, as Access does, so its record stays full-width forever. + /// An owned-pages map cannot do this: it must retain every page it has ever taken. + /// + /// + public void SetBit(int mapRow, int mapPage, int targetPage, bool set, bool movableWindow = false) { - ValidateReferenceRecord(map); - int pagesPerBitmap = (_channel.PageSize - BitmapPageHeaderSize) * 8; - var pages = new List(); + JetFormatBase format = _channel.Format; + // The pointer comes out of the file, so the record is located as every reader locates it: a live, non-empty + // record of a known type and a valid length, or corruption said as such. This writer used to index the slot + // list bare and fall through on an unknown type byte, mangling the record in place. + (PageBuffer shared, DataPage holder, DataPage.RowSlot slot) = ReadRecord(_channel, mapRow, mapPage, "Usage-map pointer"); + byte[] page = shared.Span.ToArray(); + int mapOffset = slot.Offset; + + if (RecordType(page.AsSpan(mapOffset)) == UsageMapType.Reference) + { + SetReferenceBit(page, mapPage, mapOffset, targetPage, set); + return; + } + + int headerSize = format.UsageMapInlineHeaderSize; + int startPage = StartPage(page.AsSpan(mapOffset), format); + int bitmapBits = (slot.Length - headerSize) * 8; + int bitIndex = targetPage - startPage; - // The record is a list of 4-byte pointers to bitmap pages; pointer k's bitmap - // covers the page range starting at k * pagesPerBitmap. A zero pointer means the - // range has no owned pages. - for (int e = 0; e < ReferenceMapSlots; e++) + // The inline bitmap covers pages [startPage, startPage + bitmapBits). Clearing a bit outside that + // window is a no-op — it is already 0. + if (bitIndex < 0 || bitIndex >= bitmapBits) { - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(map.Slice(1 + e * 4, 4)); - if (bitmapPage == 0) continue; + if (!set) return; + + // A free-pages map slides its window onto the target instead of growing (and can therefore also + // move *backwards*, which a grown map could never do). + if (movableWindow && TryRepositionWindow(page, holder, format, mapPage, mapRow, targetPage)) + return; - int rangeBase = e * pagesPerBitmap; - ReadOnlySpan bitmap = ReadBitmapPage(bitmapPage); - AppendSetBits(pages, bitmap, rangeBase); + if (bitIndex < 0) + throw new NotSupportedException( + $"Page {targetPage} is below the usage map's start page {startPage}; this map's window cannot move."); } - return pages; + // When Access needs to mark a page beyond the window it grows the bitmap record in place (still + // inline, same startPage), extending it in UsageMapInlineGrowthSize steps. + if (bitIndex >= bitmapBits) + { + int neededBitmapBytes = InlineBitmapBytes(format, bitIndex + 1); + byte[] grownRecord = new byte[headerSize + neededBitmapBytes]; // extra bitmap bytes stay zero + page.AsSpan(mapOffset, slot.Length).CopyTo(grownRecord); + + byte[]? grown = ReplaceMapRecord(page, holder, format, mapRow, grownRecord, out mapOffset); + if (grown is null) + { + // The record can no longer grow within its page: switch to a reference map and retry there. + ConvertInlineToReference(mapPage, mapRow); + (PageBuffer converted, _, DataPage.RowSlot reference) = ReadRecord(_channel, mapRow, mapPage, "Usage-map pointer"); + SetReferenceBit(converted.Span.ToArray(), mapPage, reference.Offset, targetPage, set); + return; + } + + page = grown; + } + + BitmapBits.Set(InlineBits(page.AsSpan(mapOffset), format), bitIndex, set); + _channel.WritePage(mapPage, page); } - private static void ValidateReferenceRecord(ReadOnlySpan map) + /// + /// Slides an inline map's window onto : a full-width bitmap starting at the + /// window boundary below the target, carrying over every bit already set and adding the target's. + /// Returns — leaving the map untouched — when some page already marked would + /// fall outside the new window, since moving would silently forget it; the caller then grows instead. + /// + private bool TryRepositionWindow(byte[] page, DataPage holder, JetFormatBase format, int mapPage, int mapRow, int targetPage) { - if (map.Length != ReferenceMapRecordSize) - throw new InvalidDataException( - $"A reference usage-map record must be exactly {ReferenceMapRecordSize} bytes; got {map.Length}."); + DataPage.RowSlot slot = holder.Rows[mapRow]; + int headerSize = format.UsageMapInlineHeaderSize; + int startPage = StartPage(page.AsSpan(slot.Offset), format); + ReadOnlySpan bitmap = page.AsSpan(slot.Offset + headerSize, slot.Length - headerSize); + + int windowPages = format.UsageMapInlineBitmapSize * 8; + int newStart = targetPage / windowPages * windowPages; + int newEnd = newStart + windowPages; + + var marked = new List(); + BitmapBits.AppendPages(marked, bitmap, startPage, rejectBeyond: null, "A free-pages map"); + bool fitsWindow = marked.TrueForAll(p => p >= newStart && p < newEnd); + + // A window that has already moved above the target cannot slide back down without dropping the pages + // it still advertises, so it widens instead: the start drops to the lowest page it must cover (rounded + // down to a byte, as the released map's move does) and the record is sized to reach the highest. That + // keeps it inline and keeps every bit. Marking a page a table's free map cannot represent is not a + // corruption — the bit only advertises room — but silently dropping it is what leaves reusable space + // invisible, and throwing outright failed an ordinary DROP TABLE on a real file (complex1.accdb, whose + // MSysObjects free map sits at page 2288 while its catalog rows live at page 17). + int bitmapBytes = format.UsageMapInlineBitmapSize; + if (!fitsWindow) + { + int lowest = Math.Min(targetPage, marked.Count == 0 ? targetPage : marked.Min()); + int highest = Math.Max(targetPage, marked.Count == 0 ? targetPage : marked.Max()); + newStart = lowest / 8 * 8; + bitmapBytes = InlineBitmapBytes(format, highest - newStart + 1); + } + + byte[] record = NewInlineRecord(format, newStart, bitmapBytes); + marked.Add(targetPage); + foreach (int markedPage in marked) + BitmapBits.Set(InlineBits(record, format), markedPage - newStart, true); + + byte[]? rewritten = ReplaceMapRecord(page, holder, format, mapRow, record, out _); + if (rewritten is null) return false; // shrinking or same size, so effectively unreachable + + _channel.WritePage(mapPage, rewritten); + return true; } - private ReadOnlySpan ReadBitmapPage(int pageNumber) + /// Sets or clears 's bit in a reference map: pointer slot + /// targetPage / UsageMapPagesPerBitmapPage names the bitmap page holding it. A slot's bitmap page is + /// allocated lazily, only when a bit in its range is first set. + private void SetReferenceBit(byte[] page, int mapPage, int mapOffset, int targetPage, bool set) { - if (pageNumber <= 0 || pageNumber >= _channel.PageCount) - throw new InvalidDataException($"Usage-map bitmap page {pageNumber} is outside the database."); + JetFormatBase format = _channel.Format; + int pagesPerBitmap = format.UsageMapPagesPerBitmapPage; + int slot = targetPage / pagesPerBitmap; + if (slot >= format.UsageMapReferenceSlots) + throw new NotSupportedException( + $"Page {targetPage} lies past the {format.UsageMapReferenceSlots} bitmap slots a usage map can address (Jet's 2 GB file limit)."); + + Span record = page.AsSpan(mapOffset); + int bitmapPage = ReferencePointer(record, slot, format); + if (bitmapPage == 0) + { + if (!set) return; // the bit is already clear — no need to materialize the bitmap page + bitmapPage = AllocateBitmapPage(); + WriteReferencePointer(record, slot, format, bitmapPage); + _channel.WritePage(mapPage, page); + } - ReadOnlySpan page = _channel.ReadPage(pageNumber).Span; - if (page[0] != (byte)PageType.PageUsageBitmap || page[1] != 0x01 || page[2] != 0 || page[3] != 0) - throw new InvalidDataException($"Usage-map pointer {pageNumber} does not reference a valid bitmap page."); - return page[BitmapPageHeaderSize..]; + // The pointer comes out of the file, so the page is proven to be a bitmap page before a bit goes into it. + byte[] bitmap = ReadBitmapPage(_channel, bitmapPage).Span.ToArray(); + BitmapBits.Set(BitmapPageBits(bitmap, format), targetPage % pagesPerBitmap, set); + _channel.WritePage(bitmapPage, bitmap); } - /// Expands a bitmap's set bits into page numbers, rejecting any that cannot exist in this file. - /// The base page comes out of the map record, so an unchecked expansion turns corrupt bytes into ownership - /// data — and these lists feed straight into page reads (TableCursor, FindPageWithRoom), where a bad number - /// would surface as an out-of-range or end-of-stream error instead. The long-value map's twin of this loop - /// in RowInserter already range-checks; this is the same check. - private void AppendSetBits(List pages, ReadOnlySpan bitmap, int basePage) + /// Zeroes the bitmap of every dedicated bitmap page a reference-form names, + /// leaving each page's header, and returns them; nothing for an inline record. ACE does this both when it + /// clears the global released-pages map at close and when it retires a dropped object's map — even the table's + /// own owned map, whose bits an inline record keeps. + internal List ClearBitmapPages(ReadOnlySpan record) { - // Read the bound ONCE: this loop runs per set bit on every insert. PageChannel.PageCount used to be a - // file-length syscall outside a transaction, which made a per-bit test cost a non-transactional insert - // ~1.9x; it is a cached field now, but one read is still all the loop needs. Nothing in the loop writes, - // so the count cannot move under it. - int pageCount = _channel.PageCount; + JetFormatBase format = _channel.Format; + var pages = new List(); + if (RecordType(record) != UsageMapType.Reference) return pages; + for (int slot = 0; slot < format.UsageMapReferenceSlots; slot++) + { + int bitmapPage = ReferencePointer(record, slot, format); + if (bitmapPage == 0) continue; + byte[] bitmap = ReadBitmapPage(_channel, bitmapPage).Span.ToArray(); + BitmapPageBits(bitmap, format).Clear(); + _channel.WritePage(bitmapPage, bitmap); + pages.Add(bitmapPage); + } + return pages; + } + + /// Allocates and initialises an empty dedicated usage-bitmap page (type 0x0105). + private int AllocateBitmapPage() + { + int pageNumber = _channel.Allocator.Allocate(); + _channel.WritePage(pageNumber, NewBitmapPage(_channel.Format)); + return pageNumber; + } + + /// + /// Rewrites an inline usage map as a reference map: every page the inline bitmap marked is re-marked in a + /// dedicated bitmap page, and the record shrinks to the fixed-size pointer table. Bits are grouped by slot + /// so each bitmap page is written once rather than once per page. + /// + private void ConvertInlineToReference(int mapPage, int mapRow) + { + JetFormatBase format = _channel.Format; + (PageBuffer shared, _, DataPage.RowSlot slot) = ReadRecord(_channel, mapRow, mapPage, "Usage-map pointer"); + byte[] page = shared.Span.ToArray(); - for (int i = 0; i < bitmap.Length; i++) + int headerSize = format.UsageMapInlineHeaderSize; + int startPage = StartPage(page.AsSpan(slot.Offset), format); + ReadOnlySpan bitmap = page.AsSpan(slot.Offset + headerSize, slot.Length - headerSize); + + // Every bit kept, as above: this is re-expressing a map the file already holds, not reading one. + var marked = new List(); + BitmapBits.AppendPages(marked, bitmap, startPage, rejectBeyond: null, "An inline usage map"); + + byte[] record = NewReferenceRecord(format); + + int pagesPerBitmap = format.UsageMapPagesPerBitmapPage; + foreach (IGrouping group in marked.GroupBy(p => p / pagesPerBitmap)) { - byte b = bitmap[i]; - if (b == 0) continue; - for (int bit = 0; bit < 8; bit++) - { - if ((b & (1 << bit)) == 0) continue; - long page = (long)basePage + i * 8 + bit; - if (page <= 1 || page >= pageCount) - throw new InvalidDataException( - $"A usage map names page {page}, outside the file's 2..{pageCount - 1} range."); - pages.Add((int)page); - } + if (group.Key >= format.UsageMapReferenceSlots) + throw new NotSupportedException( + $"Page {group.First()} lies past the {format.UsageMapReferenceSlots} bitmap slots a usage map can address (Jet's 2 GB file limit)."); + + int bitmapPage = AllocateBitmapPage(); + byte[] bits = _channel.ReadPage(bitmapPage).Span.ToArray(); + foreach (int ownedPage in group) + BitmapBits.Set(BitmapPageBits(bits, format), ownedPage % pagesPerBitmap, true); + _channel.WritePage(bitmapPage, bits); + WriteReferencePointer(record, group.Key, format, bitmapPage); } + + // Re-read: allocating bitmap pages above may have grown the file, though not this page. + (PageBuffer fresh, DataPage current, _) = ReadRecord(_channel, mapRow, mapPage, "Usage-map pointer"); + byte[] rewritten = ReplaceMapRecord(fresh.Span.ToArray(), current, format, mapRow, record, out _) + ?? throw new InvalidOperationException("The reference-map record does not fit its usage-map page."); + _channel.WritePage(mapPage, rewritten); } + + /// The bitmap bytes an inline record needs to cover pages from its start page: + /// whole bytes, rounded up to steps. + internal static int InlineBitmapBytes(JetFormatBase format, int pages) + { + int unit = format.UsageMapInlineGrowthSize; + return (BitmapBits.ByteCount(pages) + unit - 1) / unit * unit; + } + + /// Replaces the usage-map record at with , + /// repacking every record on the page from the end backward — the way Access enlarges a table's owned/free + /// bitmap once it spans past the current window. Records keep their directory order (row 0 nearest the + /// page end), and are laid over the page as it stands: bytes a moved record vacates are not cleared, as ACE + /// leaves them. Returns the rewritten page and the record's new offset, or if the + /// record no longer fits the page. + internal static byte[]? ReplaceMapRecord(byte[] page, DataPage holder, JetFormatBase format, int mapRow, byte[] newRecord, out int newOffset) + { + int rowCount = holder.Rows.Count; + var records = new byte[rowCount][]; + for (int i = 0; i < rowCount; i++) + records[i] = page.AsSpan(holder.Rows[i].Offset, holder.Rows[i].Length).ToArray(); + records[mapRow] = newRecord; + + // Carry each slot's flags across, not just its offset. A usage-map page can hold the deleted + overflow + // tombstone that the index-rebuild recycle deliberately leaves behind (RecycleOwnedMapRow, reproduced + // byte-for-byte from ACE); rebuilding the entry from the offset alone cleared those two bits and turned + // that tombstone back into a live zero-length record. + newOffset = -1; + byte[] result = [.. page]; + if (!DataPage.LayRows(result, format, records, [.. holder.Rows.Select(DataPage.Flags)])) return null; + newOffset = DataPage.ReadSlot(result, format, mapRow).Offset; + return result; + } + } \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/UsageMapWriter.cs b/src/LibRed/LibRed.Core/Storage/UsageMapWriter.cs deleted file mode 100644 index 8a43d847d..000000000 --- a/src/LibRed/LibRed.Core/Storage/UsageMapWriter.cs +++ /dev/null @@ -1,342 +0,0 @@ -using LibRed.Formats; -using LibRed.IO; -using LibRed.Pages; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -/// -/// Writes a usage map — the bitmap recording which pages belong to a table, an index, or a long-value -/// column. The counterpart to , which reads them. -/// -/// -/// A map is a single record on a usage-map page, addressed by a (row, page) pointer in the TDEF. It starts -/// as an inline map (type 0x00: a start page plus a bitmap) and, once its record can no longer -/// grow within its page, is rewritten as a reference map (type 0x01: pointers to dedicated -/// bitmap pages). See §9 of the format spec. -/// -public sealed class UsageMapWriter(PageChannel channel) -{ - // Access grows an inline usage-map bitmap in 32-bit (4-byte) steps. Verified against owned-map record - // lengths on a 255-column ACE table whose data pages start at 353: 8,000 rows → 1053, 12,000 → 1553, - // 30,000 → 3801, i.e. 5 + roundUp(ceil((353 + rows) / 8), 4) exactly. (A 32-byte chunk would give 1056 / - // 1568 / 3808.) Getting this right matters beyond tidiness: overshooting the record length spends the - // usage-map page's remaining room and converts the map to reference type earlier than Access would. - private const int UsageMapChunkBytes = 4; - - private const byte InlineMapType = 0x00; - private const byte ReferenceMapType = 0x01; - private const int InlineMapHeaderSize = 5; // type byte + 4-byte start page - private const int BitmapPageHeaderSize = 4; // type + flags + 2 unused, then the bitmap - - // A reference map's record is a type byte followed by 17 four-byte bitmap-page pointers (69 bytes). - // Seventeen is not arbitrary: each bitmap page covers (pageSize - 4) * 8 = 32,736 pages, so 17 slots - // span ~2.28 GB — just past Jet's 2 GB file ceiling. Verified against an ACE-built 134 MB table. - private const int ReferenceMapSlots = 17; - private const int ReferenceMapRecordSize = 1 + ReferenceMapSlots * 4; - - // A movable inline window is exactly 512 pages (a 64-byte bitmap) aligned to a 512-page boundary. - // Verified against ACE free-pages maps: the sole set bit at page 852/1227/1852/2852 sat in a 64-byte - // record whose startPage was 512/1024/1536/2560 — i.e. floor(page / 512) * 512. - private const int InlineWindowBitmapBytes = 64; - private const int InlineWindowPages = InlineWindowBitmapBytes * 8; - - private readonly PageChannel _channel = channel; - - /// Sets or clears the bit for in the usage map at record - /// on . - /// - /// Handles both map types. An inline map (0x00) is grown in place while its record still fits the page; - /// once it cannot, the map is converted to a reference map (0x01) — exactly Access's own threshold. - /// - /// marks a map whose set bits stay clustered near the append tail — a - /// free-pages map. Rather than growing a bitmap from startPage = 0 all the way out to the tail, - /// such a map slides a fixed 512-page window, as Access does, so its record stays 69 bytes forever. - /// An owned-pages map cannot do this: it must retain every page it has ever taken. - /// - /// - public void SetBit(int mapRow, int mapPage, int targetPage, bool set, bool movableWindow = false) - { - JetFormatBase format = _channel.Format; - byte[] page = _channel.ReadPage(mapPage).Span.ToArray(); - var holder = new DataPage(); - holder.Read(_channel.ReadPage(mapPage), format); - - // mapRow comes from a TDEF pointer, i.e. out of the file. The readers of this same pointer all - // bounds-check it and report corruption; this writer indexed the slot list bare, so a malformed - // pointer escaped as ArgumentOutOfRangeException from a write path instead. - if (mapRow < 0 || mapRow >= holder.RowCount) - throw new InvalidDataException($"Usage-map row {mapPage}:{mapRow} does not exist."); - int mapOffset = holder.Rows[mapRow].Offset; - - if (page[mapOffset] == ReferenceMapType) - { - SetReferenceBit(page, mapPage, mapOffset, targetPage, set); - return; - } - - // Anything that is neither type byte is corruption, not an inline map. Falling through treated it as - // inline and read the next four bytes as a start page — mangling the record in place, so the eventual - // complaint came from a reader at a different call site. Both readers reject an unknown type. - if (page[mapOffset] != InlineMapType) - throw new InvalidDataException( - $"Usage map at {mapPage}:{mapRow} has unknown type 0x{page[mapOffset]:X2}."); - if (holder.Rows[mapRow].Length < InlineMapHeaderSize) - throw new InvalidDataException( - $"Inline usage map at {mapPage}:{mapRow} is {holder.Rows[mapRow].Length} bytes, too short for its header."); - - int startPage = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(mapOffset + 1, 4)); - int bitmapBits = (holder.Rows[mapRow].Length - InlineMapHeaderSize) * 8; - int bitIndex = targetPage - startPage; - - // The inline bitmap covers pages [startPage, startPage + bitmapBits). Clearing a bit outside that - // window is a no-op — it is already 0. - if (bitIndex < 0 || bitIndex >= bitmapBits) - { - if (!set) return; - - // A free-pages map slides its window onto the target instead of growing (and can therefore also - // move *backwards*, which a grown map could never do). - if (movableWindow && TryRepositionWindow(page, holder, format, mapPage, mapRow, targetPage)) - return; - - if (bitIndex < 0) - throw new NotSupportedException( - $"Page {targetPage} is below the usage map's start page {startPage}; this map's window cannot move."); - } - - // When Access needs to mark a page beyond the window it grows the bitmap record in place (still - // type 0x00, same startPage), extending it in 4-byte steps. - if (bitIndex >= bitmapBits) - { - int neededBitmapBytes = RoundUpTo(bitIndex / 8 + 1, UsageMapChunkBytes); - byte[] grownRecord = new byte[InlineMapHeaderSize + neededBitmapBytes]; // extra bitmap bytes stay zero - page.AsSpan(mapOffset, holder.Rows[mapRow].Length).CopyTo(grownRecord); - - byte[]? grown = ReplaceMapRecord(page, holder, format, mapRow, grownRecord, out mapOffset); - if (grown is null) - { - // The record can no longer grow within its page: switch to a reference map and retry there. - ConvertInlineToReference(mapPage, mapRow); - page = _channel.ReadPage(mapPage).Span.ToArray(); - var converted = new DataPage(); - converted.Read(_channel.ReadPage(mapPage), format); - SetReferenceBit(page, mapPage, converted.Rows[mapRow].Offset, targetPage, set); - return; - } - - page = grown; - } - - int byteIndex = mapOffset + InlineMapHeaderSize + bitIndex / 8; - byte mask = (byte)(1 << (bitIndex % 8)); - if (set) page[byteIndex] |= mask; - else page[byteIndex] &= (byte)~mask; - _channel.WritePage(mapPage, page); - } - - /// - /// Slides an inline map's window onto : a 64-byte bitmap starting at - /// floor(targetPage / 512) * 512, carrying over every bit already set and adding the target's. - /// Returns — leaving the map untouched — when some page already marked would - /// fall outside the new window, since moving would silently forget it; the caller then grows instead. - /// - private bool TryRepositionWindow(byte[] page, DataPage holder, JetFormatBase format, int mapPage, int mapRow, int targetPage) - { - RowSlot slot = holder.Rows[mapRow]; - int startPage = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(slot.Offset + 1, 4)); - ReadOnlySpan bitmap = page.AsSpan(slot.Offset + InlineMapHeaderSize, slot.Length - InlineMapHeaderSize); - - int newStart = targetPage / InlineWindowPages * InlineWindowPages; - int newEnd = newStart + InlineWindowPages; - - var marked = new List(); - for (int i = 0; i < bitmap.Length; i++) - { - if (bitmap[i] == 0) continue; - for (int bit = 0; bit < 8; bit++) - { - if ((bitmap[i] & (1 << bit)) == 0) continue; - int markedPage = startPage + i * 8 + bit; - if (markedPage < newStart || markedPage >= newEnd) return false; - marked.Add(markedPage); - } - } - - var record = new byte[InlineMapHeaderSize + InlineWindowBitmapBytes]; - record[0] = InlineMapType; - BinaryPrimitives.WriteInt32LittleEndian(record.AsSpan(1, 4), newStart); - marked.Add(targetPage); - foreach (int markedPage in marked) - { - int bit = markedPage - newStart; - record[InlineMapHeaderSize + bit / 8] |= (byte)(1 << (bit % 8)); - } - - byte[]? rewritten = ReplaceMapRecord(page, holder, format, mapRow, record, out _); - if (rewritten is null) return false; // shrinking or same size, so effectively unreachable - - _channel.WritePage(mapPage, rewritten); - return true; - } - - /// Number of pages one dedicated bitmap page (type 0x05) covers. - private int PagesPerBitmapPage => (_channel.Format.PageSize - BitmapPageHeaderSize) * 8; - - /// Sets or clears 's bit in a reference map: pointer slot - /// targetPage / PagesPerBitmapPage names the bitmap page holding it. A slot's bitmap page is - /// allocated lazily, only when a bit in its range is first set. - private void SetReferenceBit(byte[] page, int mapPage, int mapOffset, int targetPage, bool set) - { - int slot = targetPage / PagesPerBitmapPage; - if (slot >= ReferenceMapSlots) - throw new NotSupportedException( - $"Page {targetPage} lies past the {ReferenceMapSlots} bitmap slots a usage map can address (Jet's 2 GB file limit)."); - - int pointerOffset = mapOffset + 1 + slot * 4; - int bitmapPage = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(pointerOffset, 4)); - if (bitmapPage == 0) - { - if (!set) return; // the bit is already clear — no need to materialize the bitmap page - bitmapPage = AllocateBitmapPage(); - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(pointerOffset, 4), bitmapPage); - _channel.WritePage(mapPage, page); - } - - SetBitmapPageBit(bitmapPage, targetPage % PagesPerBitmapPage, set); - } - - /// Flips one bit on a dedicated bitmap page, after proving the page IS one. - /// - /// The pointer comes out of the map record, i.e. out of the file. All three readers of this same pointer - /// verify the page is in range and carries the complete [05 01 00 00] bitmap header before trusting - /// it — the spec states that validation as mandatory "before any bitmap is expanded into page numbers". - /// The writer skipped both, so a stale or corrupt in-range pointer had it OR a bit into an ordinary data, - /// TDEF or index page and write it back. The readers then reject that same pointer, meaning the damage - /// landed on a different page from the one eventually diagnosed. - /// - private void SetBitmapPageBit(int bitmapPage, int bit, bool set) - { - if (bitmapPage <= 1 || bitmapPage >= _channel.PageCount) - throw new InvalidDataException( - $"Usage map names bitmap page {bitmapPage}, outside the file's 2..{_channel.PageCount - 1} range."); - - byte[] bitmap = _channel.ReadPage(bitmapPage).Span.ToArray(); - if (bitmap[0] != (byte)PageType.PageUsageBitmap || bitmap[1] != 0x01 || bitmap[2] != 0 || bitmap[3] != 0) - throw new InvalidDataException( - $"Page {bitmapPage} is named as a usage bitmap but does not carry the [05 01 00 00] header."); - - int byteIndex = BitmapPageHeaderSize + bit / 8; - byte mask = (byte)(1 << (bit % 8)); - if (set) bitmap[byteIndex] |= mask; - else bitmap[byteIndex] &= (byte)~mask; - _channel.WritePage(bitmapPage, bitmap); - } - - /// Allocates and initialises an empty dedicated usage-bitmap page (type 0x05). - private int AllocateBitmapPage() - { - int pageNumber = new PageAllocator(_channel).Allocate(); - var bitmap = new byte[_channel.Format.PageSize]; - bitmap[0] = (byte)PageType.PageUsageBitmap; - bitmap[1] = 0x01; // page flags (observed constant, as on data pages) - _channel.WritePage(pageNumber, bitmap); - return pageNumber; - } - - /// - /// Rewrites an inline (0x00) usage map as a reference (0x01) map: every page the inline bitmap marked is - /// re-marked in a dedicated bitmap page, and the record shrinks to the fixed 69-byte pointer table. Bits - /// are grouped by slot so each bitmap page is written once rather than once per page. - /// - private void ConvertInlineToReference(int mapPage, int mapRow) - { - JetFormatBase format = _channel.Format; - byte[] page = _channel.ReadPage(mapPage).Span.ToArray(); - var holder = new DataPage(); - holder.Read(_channel.ReadPage(mapPage), format); - RowSlot slot = holder.Rows[mapRow]; - - int startPage = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(slot.Offset + 1, 4)); - ReadOnlySpan bitmap = page.AsSpan(slot.Offset + InlineMapHeaderSize, slot.Length - InlineMapHeaderSize); - - var marked = new List(); - for (int i = 0; i < bitmap.Length; i++) - for (int bit = 0; bit < 8; bit++) - if ((bitmap[i] & (1 << bit)) != 0) - marked.Add(startPage + i * 8 + bit); - - var record = new byte[ReferenceMapRecordSize]; - record[0] = ReferenceMapType; - - foreach (IGrouping group in marked.GroupBy(p => p / PagesPerBitmapPage)) - { - if (group.Key >= ReferenceMapSlots) - throw new NotSupportedException( - $"Page {group.First()} lies past the {ReferenceMapSlots} bitmap slots a usage map can address (Jet's 2 GB file limit)."); - - int bitmapPage = AllocateBitmapPage(); - byte[] bits = _channel.ReadPage(bitmapPage).Span.ToArray(); - foreach (int ownedPage in group) - { - int bit = ownedPage % PagesPerBitmapPage; - bits[BitmapPageHeaderSize + bit / 8] |= (byte)(1 << (bit % 8)); - } - _channel.WritePage(bitmapPage, bits); - BinaryPrimitives.WriteInt32LittleEndian(record.AsSpan(1 + group.Key * 4, 4), bitmapPage); - } - - // Re-read: allocating bitmap pages above may have grown the file, though not this page. - byte[] fresh = _channel.ReadPage(mapPage).Span.ToArray(); - var current = new DataPage(); - current.Read(_channel.ReadPage(mapPage), format); - byte[] rewritten = ReplaceMapRecord(fresh, current, format, mapRow, record, out _) - ?? throw new InvalidOperationException("The reference-map record does not fit its usage-map page."); - _channel.WritePage(mapPage, rewritten); - } - - private static int RoundUpTo(int value, int unit) => (value + unit - 1) / unit * unit; - - /// Replaces the usage-map record at with , - /// repacking every record on the page from the end backward — the way Access enlarges a table's owned/free - /// bitmap once it spans past the current window. Records keep their directory order (row 0 nearest the - /// page end). Returns the rewritten page and the record's new offset, or if the - /// record no longer fits the page. - internal static byte[]? ReplaceMapRecord(byte[] page, DataPage holder, JetFormatBase format, int mapRow, byte[] newRecord, out int newOffset) - { - int rowCount = holder.Rows.Count; - var records = new byte[rowCount][]; - for (int i = 0; i < rowCount; i++) - records[i] = page.AsSpan(holder.Rows[i].Offset, holder.Rows[i].Length).ToArray(); - records[mapRow] = newRecord; - - // Check the fit *before* laying anything out: the records are packed from the page end backward, so - // an oversized record would otherwise run past offset 0 mid-copy rather than reporting "doesn't fit". - newOffset = -1; - int directoryEnd = format.DataRowDirectoryOffset + rowCount * 2; - if (format.PageSize - records.Sum(r => r.Length) < directoryEnd) return null; - - var result = new byte[format.PageSize]; - Array.Copy(page, result, format.DataRowDirectoryOffset); // preserve type/flags/owner/header - int offset = format.PageSize; - for (int i = 0; i < rowCount; i++) - { - offset -= records[i].Length; - Array.Copy(records[i], 0, result, offset, records[i].Length); - // Carry each slot's flags across, not just its offset. A usage-map page can hold the deleted + - // overflow tombstone that the index-rebuild recycle deliberately leaves behind (RecycleOwnedMapRow, - // reproduced byte-for-byte from ACE); rebuilding the entry from the offset alone cleared those two - // bits and turned that tombstone back into a live zero-length record. The data-page repacker in - // RowInserter preserves them explicitly for the same reason. - ushort slotFlags = (ushort)((holder.Rows[i].IsDeleted ? RowPointer.DeletedFlag : 0) - | (holder.Rows[i].HasOverflow ? RowPointer.OverflowFlag : 0)); - BinaryPrimitives.WriteUInt16LittleEndian(result.AsSpan(format.DataRowDirectoryOffset + i * 2, 2), - (ushort)(slotFlags | (offset & RowPointer.OffsetMask))); - if (i == mapRow) newOffset = offset; - } - - BinaryPrimitives.WriteUInt16LittleEndian(result.AsSpan(format.DataRowCountOffset, 2), (ushort)rowCount); - BinaryPrimitives.WriteUInt16LittleEndian(result.AsSpan(format.DataFreeSpaceOffset, 2), (ushort)(offset - directoryEnd)); - return result; - } -} \ No newline at end of file diff --git a/src/LibRed/LibRed.Core/Storage/ViewCreator.cs b/src/LibRed/LibRed.Core/Storage/ViewCreator.cs deleted file mode 100644 index f5895e351..000000000 --- a/src/LibRed/LibRed.Core/Storage/ViewCreator.cs +++ /dev/null @@ -1,327 +0,0 @@ -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.IO; -using System.Buffers.Binary; - -namespace LibRed.Storage; - -/// -/// Creates a view (a stored SELECT query) the way Access does — an MSysObjects row of type 5 with -/// a negative synthetic id, plus the query decomposed into MSysQueries rows (one per column / table -/// / join / where, bracketed by a type row and an end row). Verified byte-faithful against ACE for the -/// "simple SELECT" views a view is allowed to contain. -/// -public sealed class ViewCreator(PageChannel channel, JetCatalog catalog) -{ - // MSysObjects.Flags for a stored query: 0x10000000 plus the kind, and the kind byte is DAO's own - // QueryDef.Type value (verified vs ACE for all six: crosstab 16, delete 32, update 48, append 64, - // make-table 80, data-definition 96) — not the Flag the MSysQueries action row carries, which numbers - // the kinds differently. - private const int ViewFlags = 0x10000000; // a SELECT query / view (DAO type 0) - private const int DeleteFlags = 0x10000020; - private const int UpdateFlags = 0x10000030; - private const int AppendFlags = 0x10000040; // an INSERT (append) query - private const int MakeTableFlags = 0x10000050; - private const int DataDefinitionFlags = 0x10000060; // a CREATE/DROP TABLE (data-definition) query - private static readonly byte[] DefaultOwner = [0x69, 0x0C]; - private static readonly byte[] AdminSid = [0x68, 0x0C]; - - // MSysACEs permission rows a QUERY/VIEW object gets — distinct from a table's (owner and admin both get - // full 0xFFEFF on a table). Verified against every Northwind view: owner (0x690C) = 0xF00FE, admin/users - // (0x680C) = 0xFFEFF. Without these, Access opens the file but warns about permissions on the query. - private const int QueryOwnerAcm = 0xF00FE; // 983294 - private const int QueryAdminAcm = 0xFFEFF; // 1048319 - - // A relationship object's MSysACEs rows (verified vs ACE): owner 0xF00FE as a query's, admin 0xFFFFF. - private const int RelationshipOwnerAcm = 0xF00FE; // 983294 - private const int RelationshipAdminAcm = 0xFFFFF; // 1048575 - - - private readonly PageChannel _channel = channel; - private readonly JetCatalog _catalog = catalog; - - public void Create(string name, ViewSpec spec) - { - int objectId = AllocateQueryObject(name, ViewFlags); - AddQueryRows(objectId, spec); - } - - /// Persists a stored action query (a non-SELECT CREATE PROCEDURE body) byte-faithfully. - public void CreateAction(string name, ActionQuerySpec spec) - { - int flags = spec.Kind switch - { - ActionQueryKind.DataDefinition => DataDefinitionFlags, - ActionQueryKind.Append => AppendFlags, - ActionQueryKind.Update => UpdateFlags, - ActionQueryKind.Delete => DeleteFlags, - ActionQueryKind.MakeTable => MakeTableFlags, - _ => throw new NotSupportedException($"Action query kind {spec.Kind} is not stored yet."), - }; - int objectId = AllocateQueryObject(name, flags); - AddActionRows(objectId, spec); - } - - /// - /// Records a relationship as ACE does (verified): a type-8 MSysObjects object in the Relationships - /// container, named after it, flags 0, with the next high-bit id — the sequence queries draw from, so the two - /// interleave, and a dropped one's id is taken again — and its two MSysACEs rows. Refuses a name - /// another relationship has, as ACE does; a table or query may share it. - /// - public void CreateRelationshipObject(string name) => - AllocateObject(name, CatalogFormat.ObjectTypeRelationship, CatalogFormat.RelationshipContainerParentId, flags: 0, - RelationshipOwnerAcm, RelationshipAdminAcm); - - private int AllocateQueryObject(string name, int flags) => - AllocateObject(name, StoredQueryFormat.ObjectTypeQuery, CatalogFormat.ObjectContainerParentId, flags, - QueryOwnerAcm, QueryAdminAcm); - - /// Reserves the next free high-bit object id, checks the name is free, and writes - /// the MSysObjects row and its two MSysACEs rows. For a query the distinguish view / - /// append / data-definition. - private int AllocateObject(string name, short type, int parentId, int flags, int ownerAcm, int adminAcm) - { - TableDef msysObjects = _catalog.FindTable("MSysObjects") - ?? throw new InvalidOperationException("MSysObjects catalog table was not found."); - int idIndex = ColumnIndex(msysObjects, "Id"); - int nameIndex = ColumnIndex(msysObjects, "Name"); - int parentIndex = ColumnIndex(msysObjects, "ParentId"); - - // A query's name must be unique among all objects (it also cannot equal an existing table name); a - // relationship's only among the relationships, as ACE has it. Find the next free negative id (they - // increment from 0x80000000) in the same scan. - bool relationship = parentId == CatalogFormat.RelationshipContainerParentId; - int nextId = unchecked((int)0x80000000); - foreach (object?[] row in new Table(_channel, msysObjects).Rows()) - { - if ((!relationship || row[parentIndex] is int parent && parent == parentId) - && string.Equals(row[nameIndex] as string, name, StringComparison.OrdinalIgnoreCase)) - throw new SchemaObjectExistsException(relationship - ? $"There is already a relationship named '{name}' in the current database." - : $"An object named '{name}' already exists.", name); - if (row[idIndex] is int id && id < 0 && id >= nextId) nextId = id + 1; - } - - AddObjectRow(msysObjects, name, nextId, type, parentId, flags); - AddPermissionRows(nextId, ownerAcm, adminAcm); - return nextId; - } - - /// - /// Adds the two MSysACEs permission rows Access writes for a new query/view or relationship object — owner - /// (0x690C) and admin/users (0x680C) — maintaining the ObjectId index so Access's security check finds them. - /// Without these Access warns about permissions when opening a query. - /// - private void AddPermissionRows(int objectId, int ownerAcm, int adminAcm) - { - TableDef msysAces = _catalog.FindTable("MSysACEs") - ?? throw new InvalidOperationException("MSysACEs catalog table was not found."); - - foreach ((byte[] sid, int acm) in new[] { (DefaultOwner, ownerAcm), (AdminSid, adminAcm) }) - { - var values = new object?[msysAces.Columns.Count]; - SetByName(msysAces, values, "ACM", acm); - SetByName(msysAces, values, "FInheritable", false); - SetByName(msysAces, values, "ObjectId", objectId); - SetByName(msysAces, values, "SID", sid); - new RowInserter(_channel, msysAces).Insert(values, updateIndexes: true); - } - } - - private void AddObjectRow(TableDef msysObjects, string name, int objectId, short type, int parentId, int flags) - { - DateTime now = DateTime.Now; - var values = new object?[msysObjects.Columns.Count]; - SetByName(msysObjects, values, "Id", objectId); - SetByName(msysObjects, values, "ParentId", parentId); - SetByName(msysObjects, values, "Type", type); - SetByName(msysObjects, values, "Name", name); - SetByName(msysObjects, values, "Flags", flags); - SetByName(msysObjects, values, "Owner", DefaultOwner); - SetByName(msysObjects, values, "DateCreate", now); - SetByName(msysObjects, values, "DateUpdate", now); - new RowInserter(_channel, msysObjects).Insert(values, updateIndexes: true); - } - - private void AddActionRows(int objectId, ActionQuerySpec spec) - { - TableDef mq = _catalog.FindTable("MSysQueries") - ?? throw new InvalidOperationException("MSysQueries catalog table was not found."); - - Row(mq, objectId, StoredQueryFormat.AttrType, order: 1, flag: StoredQueryFormat.QueryTypeSelect); - Row(mq, objectId, StoredQueryFormat.AttrEnd, order: 1); - AddParameterRows(mq, objectId, spec.Parameters); - - if (spec.Kind == ActionQueryKind.DataDefinition) - { - // The whole DDL statement is stored verbatim in one row; Access records it with a leading space. - // Nothing is decomposed: a data-definition query has no sources, columns or predicate. - Row(mq, objectId, StoredQueryFormat.AttrOperation, order: 1, flag: StoredQueryFormat.ActionDdl, - expression: " " + spec.DdlSql); - return; - } - - // The action row carries the kind, and the target table for the two kinds that write into one. - short kind = spec.Kind switch - { - ActionQueryKind.Append => StoredQueryFormat.ActionAppend, - ActionQueryKind.Update => StoredQueryFormat.ActionUpdate, - ActionQueryKind.Delete => StoredQueryFormat.ActionDelete, - ActionQueryKind.MakeTable => StoredQueryFormat.ActionMakeTable, - _ => throw new NotSupportedException($"Action query kind {spec.Kind} is not stored yet."), - }; - Row(mq, objectId, StoredQueryFormat.AttrOperation, order: 1, flag: kind, - name1: spec.Kind is ActionQueryKind.Append or ActionQueryKind.MakeTable ? spec.TargetTable : null); - - // Sources first: Access processes the rows in order, and a derived-table source defines an alias the - // column expressions reference (the same reason the view path writes tables before columns). - AddSourceRows(mq, objectId, spec.Body); - - var values = spec.Values ?? []; - switch (spec.Kind) - { - case ActionQueryKind.Append: - // Name2 = target column, Expression = the value; the 0x8000 flag marks an INSERT … VALUES, - // where a column fed by the query's own source carries Flag 0. - for (int i = 0; i < values.Count; i++) - Row(mq, objectId, StoredQueryFormat.AttrColumn, order: i + 1, - flag: spec.Body is null ? StoredQueryFormat.AppendValueFlag : (short)0, - expression: values[i].ValueExpression, name2: values[i].Column); - break; - - case ActionQueryKind.Update: - // One row per SET assignment, stored exactly as an append's columns are. - for (int i = 0; i < values.Count; i++) - Row(mq, objectId, StoredQueryFormat.AttrColumn, order: i + 1, flag: 0, - expression: values[i].ValueExpression, name2: values[i].Column); - break; - - case ActionQueryKind.Delete: - // `DELETE t.* FROM …` keeps that target verbatim in a single column row; `DELETE FROM …` - // stores no column row at all, and Access renders it back as `DELETE * FROM …`. - if (spec.DeleteTarget is { } target) - Row(mq, objectId, StoredQueryFormat.AttrColumn, order: 1, flag: 0, expression: target); - break; - - case ActionQueryKind.MakeTable: - // An ordinary output list: Expression = the column, Name1 = its alias. - var columns = spec.Body?.Columns ?? []; - for (int i = 0; i < columns.Count; i++) - Row(mq, objectId, StoredQueryFormat.AttrColumn, order: i + 1, flag: 0, - expression: columns[i].Expression, name1: columns[i].Alias); - break; - } - - AddJoinAndWhereRows(mq, objectId, spec.Body); - } - - /// The 0x02 parameter rows, in declaration order — written identically for a view and for - /// an action query. - private void AddParameterRows(TableDef mq, int objectId, IReadOnlyList? parameters) - { - for (int i = 0; i < (parameters?.Count ?? 0); i++) - { - ViewParameterSpec p = parameters![i]; - Row(mq, objectId, StoredQueryFormat.AttrParameter, order: i + 1, - flag: p.TypeCode, name1: p.Name, - lvExtra: StoredQueryFormat.PackParameterFacets((JetDataType)p.TypeCode, p.Size, p.Scale)); - } - } - - /// The 0x05 FROM rows: a named table in Name1 (alias in Name2), or a derived table whose - /// subquery SQL goes in Expression with Name1 empty. - private void AddSourceRows(TableDef mq, int objectId, ViewSpec? body) - { - var tables = body?.Tables ?? []; - for (int i = 0; i < tables.Count; i++) - { - ViewTableSpec t = tables[i]; - if (t.SubquerySql is { } sub) - Row(mq, objectId, StoredQueryFormat.AttrTable, order: i + 1, expression: sub, name2: t.Alias); - else - Row(mq, objectId, StoredQueryFormat.AttrTable, order: i + 1, name1: t.Table, name2: t.Alias); - } - } - - /// The 0x07 join rows (condition, kind, and the two tables the condition names) and the - /// single 0x08 WHERE row. - private void AddJoinAndWhereRows(TableDef mq, int objectId, ViewSpec? body) - { - var joins = body?.Joins ?? []; - for (int i = 0; i < joins.Count; i++) - { - ViewJoinSpec j = joins[i]; - Row(mq, objectId, StoredQueryFormat.AttrJoin, order: i + 1, flag: (short)j.Kind, - expression: j.Condition, name1: j.LeftAlias, name2: j.RightAlias); - } - if (body?.Where is { } where) - Row(mq, objectId, StoredQueryFormat.AttrWhere, order: 1, expression: where); - } - - private void AddQueryRows(int objectId, ViewSpec spec) - { - TableDef mq = _catalog.FindTable("MSysQueries") - ?? throw new InvalidOperationException("MSysQueries catalog table was not found."); - - // ACE's row order (verified against every Northwind view): type, end, distinct, TABLES, COLUMNS, - // joins, where. Tables must precede columns — a derived-table source defines an alias that the - // column expressions reference, and Access processes the rows in order, so columns-before-tables - // makes it fail to run the view (it opens, but SELECT-from-view errors). Order fields are - // per-attribute counters, independent of this insertion order. - Row(mq, objectId, StoredQueryFormat.AttrType, order: 1, flag: StoredQueryFormat.QueryTypeSelect); - Row(mq, objectId, StoredQueryFormat.AttrEnd, order: 1); - // Declared parameters (CREATE PROCEDURE) come right after the End row, before the tables. - AddParameterRows(mq, objectId, spec.Parameters); - // DISTINCT and TOP are both StoredQueryFormat.AttrOption (0x03) rows, distinguished by their Flag bits; a TOP row also - // carries the count in Name1. The bits are cumulative, so Access can put both on one row -- writing - // them separately is equally valid and keeps the two spec fields independent here. - // Give them distinct Order values so the composite PK stays unique. - int flagOrder = 1; - if (spec.Distinct) - Row(mq, objectId, StoredQueryFormat.AttrOption, order: flagOrder++, flag: StoredQueryFormat.FlagDistinct); - if (spec.Top is { } top) - Row(mq, objectId, StoredQueryFormat.AttrOption, order: flagOrder++, flag: StoredQueryFormat.FlagTop, name1: top.ToString(System.Globalization.CultureInfo.InvariantCulture)); - AddSourceRows(mq, objectId, spec); - for (int i = 0; i < spec.Columns.Count; i++) - Row(mq, objectId, StoredQueryFormat.AttrColumn, order: i + 1, flag: 0, - expression: spec.Columns[i].Expression, name1: spec.Columns[i].Alias); - AddJoinAndWhereRows(mq, objectId, spec); - for (int i = 0; i < (spec.GroupBy?.Count ?? 0); i++) - Row(mq, objectId, StoredQueryFormat.AttrGroupBy, order: i + 1, flag: 0, expression: spec.GroupBy![i]); - // The group filter carries no flag of its own, exactly as the WHERE row doesn't. - if (spec.Having is { } having) - Row(mq, objectId, StoredQueryFormat.AttrHaving, order: 1, expression: having); - for (int i = 0; i < (spec.OrderBy?.Count ?? 0); i++) - Row(mq, objectId, StoredQueryFormat.AttrOrderBy, order: i + 1, expression: spec.OrderBy![i].Expression, - name1: spec.OrderBy[i].Descending ? "d" : null); - } - - private void Row(TableDef mq, int objectId, byte attribute, int order, - short? flag = null, string? expression = null, string? name1 = null, string? name2 = null, - int? lvExtra = null) - { - var values = new object?[mq.Columns.Count]; - SetByName(mq, values, "ObjectId", objectId); - SetByName(mq, values, "Attribute", attribute); - var orderBytes = new byte[4]; - BinaryPrimitives.WriteInt32BigEndian(orderBytes, order); // 4-byte big-endian per-attribute counter - SetByName(mq, values, "Order", orderBytes); - if (flag is { } f) SetByName(mq, values, "Flag", f); - if (expression is not null) SetByName(mq, values, "Expression", expression); - if (name1 is not null) SetByName(mq, values, "Name1", name1); - if (name2 is not null) SetByName(mq, values, "Name2", name2); - // A declared parameter's length; ACE writes it here and renders the PARAMETERS clause from it. - if (lvExtra is { } extra) SetByName(mq, values, "LvExtra", extra); - new RowInserter(_channel, mq).Insert(values, updateIndexes: true); - } - - private static void SetByName(TableDef table, object?[] values, string column, object value) - { - ColumnDef def = table.FindColumn(column) - ?? throw new InvalidOperationException($"'{table.Name}' is missing the '{column}' column."); - values[def.Index] = value; - } - - private static int ColumnIndex(TableDef table, string column) => - (table.FindColumn(column) ?? throw new InvalidOperationException($"'{table.Name}' is missing the '{column}' column.")).Index; -} \ No newline at end of file diff --git a/src/LibRed/LibRed.EFCore/Extensions/LibRedDbContextOptionsBuilderExtensions.cs b/src/LibRed/LibRed.EFCore/Extensions/LibRedDbContextOptionsBuilderExtensions.cs index 6ffafb1a6..afe2aaf8c 100644 --- a/src/LibRed/LibRed.EFCore/Extensions/LibRedDbContextOptionsBuilderExtensions.cs +++ b/src/LibRed/LibRed.EFCore/Extensions/LibRedDbContextOptionsBuilderExtensions.cs @@ -175,11 +175,6 @@ public static DbContextOptionsBuilder UseLibRed( ArgumentNullException.ThrowIfNull(optionsBuilder); ArgumentNullException.ThrowIfNull(connection); - if (connection is not LibRedConnection) - { - throw new ArgumentException($"The {nameof(connection)} parameter must be of type {nameof(LibRedConnection)}."); - } - var extension = (LibRedOptionsExtension)GetOrCreateExtension(optionsBuilder) .WithConnection(connection, contextOwnsConnection); ((IDbContextOptionsBuilderInfrastructure)optionsBuilder).AddOrUpdateExtension(extension); @@ -219,11 +214,6 @@ public static DbContextOptionsBuilder UseLibRed( ArgumentNullException.ThrowIfNull(optionsBuilder); ArgumentNullException.ThrowIfNull(connection); - if (connection is not LibRedConnection) - { - throw new ArgumentException($"The {nameof(connection)} parameter must be of type {nameof(LibRedConnection)}."); - } - var extension = ((LibRedOptionsExtension)GetOrCreateExtension(optionsBuilder) .WithConnection(connection, contextOwnsConnection)) .WithSqlMode(sqlMode); diff --git a/src/LibRed/LibRed.EFCore/Migrations/LibRedMigrationsSqlGenerator.cs b/src/LibRed/LibRed.EFCore/Migrations/LibRedMigrationsSqlGenerator.cs index 0d5ed70db..db435402c 100644 --- a/src/LibRed/LibRed.EFCore/Migrations/LibRedMigrationsSqlGenerator.cs +++ b/src/LibRed/LibRed.EFCore/Migrations/LibRedMigrationsSqlGenerator.cs @@ -31,32 +31,8 @@ public class LibRedMigrationsSqlGenerator( MigrationsSqlGeneratorDependencies dependencies, ICommandBatchPreparer commandBatchPreparer) : MigrationsSqlGenerator(dependencies) { - private IReadOnlyList _operations = null!; private readonly ICommandBatchPreparer _commandBatchPreparer = commandBatchPreparer; - /// - /// Generates commands from a list of operations. - /// - /// The operations. - /// The target model which may be if the operations exist without a model. - /// The options to use when generating commands. - /// The list of commands to be executed or scripted. - public override IReadOnlyList Generate( - IReadOnlyList operations, - IModel? model = null, - MigrationsSqlGenerationOptions options = MigrationsSqlGenerationOptions.Default) - { - _operations = operations; - try - { - return base.Generate(operations, model, options); - } - finally - { - _operations = null!; - } - } - /// /// /// Builds commands for the given by making calls on the given @@ -184,7 +160,6 @@ protected override void Generate( Dependencies.MigrationsLogger.ColumnOrderIgnoredWarning(operation); } - IEnumerable? indexesToRebuild = null; var column = model?.GetRelationalModel().FindTable(operation.Table, operation.Schema) ?.Columns.FirstOrDefault(c => c.Name == operation.Name); @@ -225,12 +200,11 @@ protected override void Generate( addColumnOperation.AddAnnotations(operation.GetAnnotations()); // TODO: Use a column rebuild instead - indexesToRebuild = GetIndexesToRebuild(column, operation).ToList(); - DropIndexes(indexesToRebuild, builder); + // No index to drop first and recreate after: Jet cannot index a calculated column in any usable + // way (ACE accepts CREATE INDEX on one, then refuses every INSERT into the table). Generate(dropColumnOperation, model, builder, terminate: false); builder.AppendLine(Dependencies.SqlGenerationHelper.StatementTerminator); Generate(addColumnOperation, model, builder); - CreateIndexes(indexesToRebuild, builder); builder.EndCommand(); return; @@ -266,12 +240,8 @@ protected override void Generate( || operation is { IsNullable: false, OldColumn.IsNullable: true }; } - if (narrowed) - { - indexesToRebuild = GetIndexesToRebuild(column, operation).ToList(); - DropIndexes(indexesToRebuild, builder); - } - + // No DROP INDEX / CREATE INDEX around the ALTER: Jet's ALTER COLUMN rebuilds the indexes over the column + // itself, keeping each one's name, columns, order and flags, the primary key included. var newAnnotations = operation.GetAnnotations().Where(a => a.Name != JetAnnotationNames.Identity); var oldAnnotations = operation.OldColumn.GetAnnotations().Where(a => a.Name != JetAnnotationNames.Identity); @@ -393,11 +363,6 @@ protected override void Generate( builder.AppendLine(Dependencies.SqlGenerationHelper.StatementTerminator); } - if (narrowed) - { - CreateIndexes(indexesToRebuild!, builder); - } - builder.EndCommand(); } @@ -999,87 +964,6 @@ protected virtual void DropDefaultConstraint( .AppendLine(Dependencies.SqlGenerationHelper.StatementTerminator); } - /// - /// Gets the list of indexes that need to be rebuilt when the given column is changing. - /// - /// The column. - /// The operation which may require a rebuild. - /// The list of indexes affected. - protected virtual IEnumerable GetIndexesToRebuild( - IColumn? column, - MigrationOperation currentOperation) - { - if (column == null) - { - yield break; - } - - var table = column.Table; - var createIndexOperations = _operations.SkipWhile(o => o != currentOperation) - .Skip(1) - .OfType() - .ToList(); - foreach (var index in table.Indexes) - { - var indexName = index.Name; - if (createIndexOperations.Any(o => o.Name == indexName)) - { - continue; - } - - if (index.Columns.Any(c => c == column)) - { - yield return index; - } - else if (index[JetAnnotationNames.Include] is IReadOnlyList includeColumns - && includeColumns.Contains(column.Name)) - { - yield return index; - } - } - } - - /// - /// Generates SQL to drop the given indexes. - /// - /// The indexes to drop. - /// The command builder to use to build the commands. - protected virtual void DropIndexes( - IEnumerable indexes, - MigrationCommandListBuilder builder) - { - foreach (var index in indexes) - { - var table = index.Table; - var operation = new DropIndexOperation - { - Schema = table.Schema, - Table = table.Name, - Name = index.Name - }; - operation.AddAnnotations(index.GetAnnotations()); - - Generate(operation, table.Model.Model, builder, terminate: false); - builder.AppendLine(Dependencies.SqlGenerationHelper.StatementTerminator); - } - } - - /// - /// Generates SQL to create the given indexes. - /// - /// The indexes to create. - /// The command builder to use to build the commands. - protected virtual void CreateIndexes( - IEnumerable indexes, - MigrationCommandListBuilder builder) - { - foreach (var index in indexes) - { - Generate(CreateIndexOperation.CreateFrom(index), index.Table.Model.Model, builder, terminate: false); - builder.AppendLine(Dependencies.SqlGenerationHelper.StatementTerminator); - } - } - private static bool IsIdentity(ColumnOperation operation) => operation[JetAnnotationNames.Identity] != null || operation[JetAnnotationNames.ValueGenerationStrategy] as JetValueGenerationStrategy? diff --git a/src/LibRed/LibRed.EFCore/Query/Sql/Internal/LibRedQuerySqlGenerator.cs b/src/LibRed/LibRed.EFCore/Query/Sql/Internal/LibRedQuerySqlGenerator.cs index a60daf821..5ee3a563d 100644 --- a/src/LibRed/LibRed.EFCore/Query/Sql/Internal/LibRedQuerySqlGenerator.cs +++ b/src/LibRed/LibRed.EFCore/Query/Sql/Internal/LibRedQuerySqlGenerator.cs @@ -46,6 +46,35 @@ protected override bool TryGenerateWithoutWrappingSelect(SelectExpression select => selectExpression.Tables is not [ValuesExpression] && base.TryGenerateWithoutWrappingSelect(selectExpression); + /// + /// A VALUES table as the standard writes it, its columns named by a column list after the alias: + /// (VALUES (0, CLNG(1)), (1, 2)) AS `v`(`_ord`, `Value`). EF's base names them on a leading SELECT + /// instead, with the other rows after UNION ALL VALUES, for databases that have no column list; + /// LibRed's engine has one. + /// + protected override Expression VisitValues(ValuesExpression valuesExpression) + { + base.VisitValues(valuesExpression); + + Sql.Append("("); + GenerateList(valuesExpression.ColumnNames, name => Sql.Append(_sqlGenerationHelper.DelimitIdentifier(name))); + Sql.Append(")"); + + return valuesExpression; + } + + /// + protected override void GenerateValues(ValuesExpression valuesExpression) + { + if (valuesExpression.RowValues is not { Count: > 0 } rowValues) + { + throw new InvalidOperationException(RelationalStrings.EmptyCollectionNotSupportedAsInlineQueryRoot); + } + + Sql.Append("VALUES "); + GenerateList(rowValues, row => Visit(row)); + } + private void GenerateList( IReadOnlyList items, Action generationAction, @@ -251,6 +280,17 @@ SqlFunctionExpression WrapConvert(SqlExpression inner) => return convertExpression; } + // .NET converts a char to a number by its code point, so (uint)'1' is 49. Passing the operand + // through leaves a one-character string, which Jet then coerces by parsing it: 1, not 49. + if (typeMapping.ClrType.IsInteger() && typeMapping.ClrType != typeof(char) && convertExpression.Operand.Type == typeof(char)) + { + // Widen ASCW's signed Int16 before masking: ACE otherwise sign-extends a 16-bit left operand. + Sql.Append("(CLNG(ASCW("); + Visit(convertExpression.Operand); + Sql.Append(")) BAND 65535)"); + return convertExpression; + } + //Just pass the operand in the default case //If we have a type mapping on the operand, then it seems to work fine //Jet appears to be fairly flexible when types aren't specifically mentioned @@ -439,4 +479,4 @@ protected override void CheckComposableSqlTrimmed(ReadOnlySpan sql) } } } -} \ No newline at end of file +} diff --git a/src/LibRed/LibRed.EFCore/Scaffolding/Internal/LibRedDatabaseModelFactory.cs b/src/LibRed/LibRed.EFCore/Scaffolding/Internal/LibRedDatabaseModelFactory.cs index e40671b7f..bf8c11c39 100644 --- a/src/LibRed/LibRed.EFCore/Scaffolding/Internal/LibRedDatabaseModelFactory.cs +++ b/src/LibRed/LibRed.EFCore/Scaffolding/Internal/LibRedDatabaseModelFactory.cs @@ -72,7 +72,7 @@ private List GetTables(JetDatabase database, DatabaseModel databa { var tables = new List(); - foreach (TableDef definition in database.Catalog.UserTables) + foreach (TableDefinition definition in database.Catalog.UserTables) { _logger.TableFound(definition.Name); @@ -95,7 +95,7 @@ private void GetColumns(JetDatabase database, IReadOnlyList table { foreach (DatabaseTable table in tables) { - TableDef definition = database.Catalog.FindTable(table.Name)!; + TableDefinition definition = database.Catalog.FindTable(table.Name)!; for (int ordinal = 0; ordinal < definition.Columns.Count; ordinal++) { @@ -163,7 +163,7 @@ private void GetIndexes(JetDatabase database, IReadOnlyList table { foreach (DatabaseTable table in tables) { - TableDef definition = database.Catalog.FindTable(table.Name)!; + TableDefinition definition = database.Catalog.FindTable(table.Name)!; foreach (IndexDef index in definition.Indexes) { @@ -235,8 +235,8 @@ private void GetRelations(JetDatabase database, IReadOnlyList tab DatabaseTable? referencingTable = Find(relation.Table); if (referencingTable is null) continue; - // Jet supports ON DELETE NO ACTION / CASCADE / SET NULL (read from MSysRelationships.grbit + - // the index-info action byte). EF's scaffolding models OnDelete only (no OnUpdate). + // Jet supports ON DELETE NO ACTION / CASCADE / SET NULL (read from the relationship's index-info action + // byte, which is what ACE acts on). EF's scaffolding models OnDelete only (no OnUpdate). ReferentialAction onDeleteAction = relation.CascadeDelete ? ReferentialAction.Cascade : relation.DeleteSetNull ? ReferentialAction.SetNull : ReferentialAction.NoAction; diff --git a/src/LibRed/LibRed.EFCore/Storage/Internal/LibRedDatabaseCreator.cs b/src/LibRed/LibRed.EFCore/Storage/Internal/LibRedDatabaseCreator.cs index 1e24bab65..9ebd07bef 100644 --- a/src/LibRed/LibRed.EFCore/Storage/Internal/LibRedDatabaseCreator.cs +++ b/src/LibRed/LibRed.EFCore/Storage/Internal/LibRedDatabaseCreator.cs @@ -20,7 +20,7 @@ public class LibRedDatabaseCreator( // connection used to run a CREATE/DROP DATABASE migration command. LibRedRelationalConnection // doesn't support that (there's nothing to connect to before the file exists), so these bypass // it entirely and go straight through LibRedConnection's own bootstrap (see its CreateDatabase - // remarks: DatabaseCreator.CreateEmpty synthesises the file from scratch - no DAO/ADOX). + // remarks: JetDatabase.Create synthesises the file from scratch - no DAO/ADOX). public override void Create() => LibRedConnection.CreateDatabase(relationalConnection.DbConnection.ConnectionString); diff --git a/src/LibRed/LibRed.Engine/Execution/AccessTypeMapper.cs b/src/LibRed/LibRed.Engine/Execution/AccessTypeMapper.cs index 45dfd46ac..c4a2d505b 100644 --- a/src/LibRed/LibRed.Engine/Execution/AccessTypeMapper.cs +++ b/src/LibRed/LibRed.Engine/Execution/AccessTypeMapper.cs @@ -1,6 +1,7 @@ using LibRed.Catalog; using LibRed.Formats; using LibRed.Sql.Ast; +using LibRed.Storage; namespace LibRed.Engine.Execution; @@ -33,9 +34,8 @@ public static ColumnSpec ToColumnSpec(ColumnDefinition column, JetVersion versio // because each would otherwise produce a column that looks declared one way and behaves another. if (column.PrimaryKey) throw new NotSupportedException( - $"Column '{column.Name}' cannot be both calculated and a key: ACE accepts an index on a " - + "calculated column and then refuses every insert into the table, so such a table can never " - + "hold a row."); + $"Column '{column.Name}' cannot be both calculated and a key: a table with an index on a " + + "calculated column cannot hold any rows."); if (column.Default is not null) throw new NotSupportedException( $"Column '{column.Name}' cannot have a DEFAULT as well as a calculated expression — its value " @@ -153,7 +153,7 @@ private static ColumnSpec MapType(ColumnDefinition column, JetVersion version) // VARIABLE region (verified: every GUID column ACE's DDL creates reads back fixed=False, at 1, // 2, 10, 250 and 252 columns alike — it is not a fallback for wide tables, and SELECT INTO // agrees). ACE's own system tables are the exception: MSysComplexType_GUID.Value is fixed, and - // DatabaseCreator reproduces that. ACE reads either layout back correctly, so this is about + // JetDatabase reproduces that. ACE reads either layout back correctly, so this is about // matching what ACE writes; it also stops a GUID column spending fixed-record budget ACE does // not spend, which made a 252-GUID table ACE creates happily exceed the declared record cap. "GUID" or "UNIQUEIDENTIFIER" @@ -181,6 +181,10 @@ private static ColumnSpec MapType(ColumnDefinition column, JetVersion version) => Binary(column, isFixed: true), "VARBINARY" or "BINARY VARYING" or "BIT VARYING" => Binary(column, isFixed: false), + // Up to 4000 bytes, stored inline like VARBINARY under its own type code, and bare BIGBINARY takes + // the maximum (verified vs ACE). There is no fixed-length form in DDL. + "BIGBINARY" + => Binary(column, isFixed: false, JetDataType.BigBinary, RowCodec.MaxBigBinaryBytes), // Long-value columns: variable-length with no fixed byte length. The in-row value is a // 12-byte long-value descriptor; short values are stored inline after it. @@ -230,22 +234,23 @@ private static ColumnSpec Text(ColumnDefinition column, bool isFixed) if (characters > MaxTextCharacters) throw new InvalidOperationException( $"Size of field '{column.Name}' is too long: a char/varchar column holds at most {MaxTextCharacters} " + - $"characters in Jet/ACE (got {characters}). Use LONGTEXT/MEMO for longer text."); + $"characters (got {characters}). Use LONGTEXT/MEMO for longer text."); return new(column.Name, JetDataType.Text, characters * 2, IsFixedLength: isFixed); } // A binary column: length is in bytes (not char-doubled). A size-less binary/varbinary takes the // MAXIMUM (510 bytes) — verified vs ACE, which defaults a bare BINARY/VARBINARY to a 510-byte field. - private static ColumnSpec Binary(ColumnDefinition column, bool isFixed) + private static ColumnSpec Binary(ColumnDefinition column, bool isFixed, + JetDataType type = JetDataType.Binary, int maxBytes = MaxBinaryBytes) { - int bytes = column.Size ?? MaxBinaryBytes; + int bytes = column.Size ?? maxBytes; if (bytes <= 0) throw new InvalidOperationException( $"Size of field '{column.Name}' must be positive (got {bytes})."); - if (bytes > MaxBinaryBytes) + if (bytes > maxBytes) throw new InvalidOperationException( - $"Size of field '{column.Name}' is too long: a binary/varbinary column holds at most {MaxBinaryBytes} " + - $"bytes in Jet/ACE (got {bytes}). Use LONGBINARY/OLE for longer data."); - return new(column.Name, JetDataType.Binary, bytes, IsFixedLength: isFixed); + $"Size of field '{column.Name}' is too long: a {column.TypeName.ToLowerInvariant()} column holds at most " + + $"{maxBytes} bytes (got {bytes}). Use LONGBINARY/OLE for longer data."); + return new(column.Name, type, bytes, IsFixedLength: isFixed); } } \ No newline at end of file diff --git a/src/LibRed/LibRed.Engine/Execution/EvalScope.cs b/src/LibRed/LibRed.Engine/Execution/EvalScope.cs index 9dd19cfaa..621d99082 100644 --- a/src/LibRed/LibRed.Engine/Execution/EvalScope.cs +++ b/src/LibRed/LibRed.Engine/Execution/EvalScope.cs @@ -16,7 +16,7 @@ internal sealed class EvalScope( // the schema and outer link are fixed for the scope's lifetime. private object?[] row = row; - /// Points this scope at a new row of the same schema (see 's callers for why + /// Points this scope at a new row of the same schema (see 's callers for why /// reuse is worthwhile). Returns the scope for fluent use. public EvalScope Rebind(object?[] newRow) { @@ -24,6 +24,18 @@ public EvalScope Rebind(object?[] newRow) return this; } + // The precomputed aggregates, rebindable with the row: a grouped query evaluates every group through one + // scope, each group bringing its key row and its own aggregate values. + private IReadOnlyDictionary? aggregates = aggregates; + + /// Points this scope at another group: its key row and its aggregates' values. + public EvalScope Rebind(object?[] newRow, IReadOnlyDictionary newAggregates) + { + row = newRow; + aggregates = newAggregates; + return this; + } + /// Resolves a precomputed aggregate (by reference), walking out to enclosing scopes so an outer /// aggregate referenced inside a correlated subquery (e.g. … WHERE x = MAX(o.Col) …) is found. public bool TryResolveAggregate(FunctionCall call, out object? value) @@ -34,19 +46,26 @@ public bool TryResolveAggregate(FunctionCall call, out object? value) return false; } + // Where each reference resolved, by reference. Only the row moves between rows (see Rebind), so where a + // reference lands does not, and neither does whether it is ambiguous. + private Dictionary? _resolved; + public bool TryResolve(ColumnReference reference, out object? value) { - int found = -1; - for (int i = 0; i < schema.Count; i++) + // Resolve each reference once rather than re-scanning every column — two case-insensitive compares + // apiece, and no early exit, since the same pass is proving the reference unambiguous — for every row. + // A scope with no schema of its own resolves nothing here and goes straight out, so it memoises + // nothing. An ambiguous reference throws rather than caching, and would throw again either way. + int found; + if (schema.Count == 0) { - bool nameMatch = string.Equals(schema[i].Name, reference.Column, StringComparison.OrdinalIgnoreCase); - bool qualifierMatch = reference.Table is null - || string.Equals(schema[i].Qualifier, reference.Table, StringComparison.OrdinalIgnoreCase); - if (!nameMatch || !qualifierMatch) continue; - - if (found >= 0) - throw new InvalidOperationException($"Column reference '{Describe(reference)}' is ambiguous."); - found = i; + found = -1; + } + else + { + _resolved ??= []; + if (!_resolved.TryGetValue(reference, out found)) + _resolved[reference] = found = Locate(schema, reference); } if (found >= 0) @@ -62,6 +81,29 @@ public bool TryResolve(ColumnReference reference, out object? value) return false; } + /// The index of the one column of that names, + /// or -1 for none. Scans to the end whether or not it has matched, because finding a second match is what + /// makes the reference ambiguous — which throws, or with false is -1 too, + /// for a caller planning ahead that must leave the error to the row that would raise it. + internal static int Locate(IReadOnlyList schema, ColumnReference reference, bool throwIfAmbiguous = true) + { + int found = -1; + for (int i = 0; i < schema.Count; i++) + { + bool nameMatch = string.Equals(schema[i].Name, reference.Column, StringComparison.OrdinalIgnoreCase); + bool qualifierMatch = reference.Table is null + || string.Equals(schema[i].Qualifier, reference.Table, StringComparison.OrdinalIgnoreCase); + if (!nameMatch || !qualifierMatch) continue; + + if (found >= 0) + return throwIfAmbiguous + ? throw new InvalidOperationException($"Column reference '{Describe(reference)}' is ambiguous.") + : -1; + found = i; + } + return found; + } + internal static string Describe(ColumnReference r) => r.Table is null ? r.Column : $"{r.Table}.{r.Column}"; /// The current row's value at a 1-based column position. @@ -98,6 +140,10 @@ public IReadOnlyList AllColumns() /// Executes a subquery, correlating it to an enclosing query's scope. internal interface IScalarSubqueryRunner { + /// How text compares in this database: its page-0 collation, which every query-time text + /// comparison uses whatever its operands (see ). + LibRed.Storage.JetTextComparer TextComparer { get; } + /// The subquery's single value, correlated to . object? ExecuteScalar(SqlStatement query, EvalScope outerScope); diff --git a/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.Format.cs b/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.Format.cs index 9dded3d03..f3dc6983c 100644 --- a/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.Format.cs +++ b/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.Format.cs @@ -77,8 +77,9 @@ private static string FormatText(object? value, string format, DayOfWeek first, case "short time": return FormatDate(value, DateTimeFormats[4], first, rule); } - List sections = FormatSections(format); - return FormatKindOf(format) switch + (FormatKind kind, IReadOnlyList sections) = Shapes.GetOrAdd( + format, static f => (FormatKindOf(f), FormatSections(f))); + return kind switch { FormatKind.Text => FormatTextSections(value, sections), FormatKind.Date => FormatDate(value, sections[0], first, rule), @@ -86,6 +87,13 @@ private static string FormatText(object? value, string format, DayOfWeek first, }; } + /// A format string's split sections and its kind, both of which depend on the string alone. The + /// format is a literal in nearly every query, so without this each row re-scans it twice — once to split on + /// the semicolons outside quotes and escapes, once to classify it. The same memo-by-text + /// CalculatedExpression.ParseCached uses for calculated columns, and for the same reason. + private static readonly System.Collections.Concurrent.ConcurrentDictionary< + string, (FormatKind Kind, IReadOnlyList Sections)> Shapes = new(StringComparer.Ordinal); + /// A value as Format writes it without a format: as CStr does, but a date with its year padded and /// rounded to the second. private static string GeneralText(object value) => @@ -179,7 +187,7 @@ private static FormatKind FormatKindOf(string format) /// change nothing). Characters the placeholders do not take follow the format's output (left to right, the /// leftmost are dropped instead). Empty text and Null use the second section, or are empty. /// - private static string FormatTextSections(object? value, List sections) + private static string FormatTextSections(object? value, IReadOnlyList sections) { string text = value is null ? "" : GeneralText(value); if (text.Length == 0) @@ -394,7 +402,7 @@ private static int Meridiem(string format, int index) /// section uses the first with a minus sign. A value that the section rounds to zero is a zero, and a zero uses the /// third section, or the first when there is none or it is empty. An empty first section writes nothing. /// - private static string FormatNumberSections(object? value, List sections) + private static string FormatNumberSections(object? value, IReadOnlyList sections) { string Section(int i) => i < sections.Count ? sections[i] : ""; if (value is null) diff --git a/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.cs b/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.cs index 4f82b69cc..f4f7e2078 100644 --- a/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.cs +++ b/src/LibRed/LibRed.Engine/Execution/ExpressionEvaluator.cs @@ -1,4 +1,6 @@ using EntityFrameworkCore.Jet.Data; +using LibRed.Catalog; +using LibRed.Engine.Planning; using LibRed.Sql.Ast; using LibRed.Storage; using System.Globalization; @@ -43,6 +45,16 @@ public ExpressionEvaluator Rebind(object?[] row) return this; } + /// Rebinds to another group of a grouped query: its key row and its aggregates' values. + public ExpressionEvaluator Rebind(object?[] row, IReadOnlyDictionary aggregates) + { + scope.Rebind(row, aggregates); + return this; + } + + /// How text compares here: the database's collation, for every comparison whatever its operands. + private JetTextComparer Text => subqueries.TextComparer; + public object? Evaluate(Expression expression) => expression switch { LiteralExpression l => l.Value, @@ -128,7 +140,7 @@ private static bool TryNiladicFunction(ColumnReference c, out object? value) foreach (object? item in items) { if (item is null) hasNull = true; - else if (Compare(val, item) == 0) { found = true; break; } + else if (Compare(val, item, Text) == 0) { found = true; break; } } } } @@ -152,7 +164,7 @@ private static bool TryNiladicFunction(ColumnReference c, out object? value) foreach (Expression itemExpr in inl.Items) { if (Evaluate(itemExpr) is not { } item) hasNull = true; - else if (CompareAsKinds(val, item) == 0) { found = true; break; } + else if (CompareAsKinds(inl.Value, val, itemExpr, item, Text) == 0) { found = true; break; } } return !found && hasNull ? null : found != inl.Negated; } @@ -165,7 +177,7 @@ private static bool TryNiladicFunction(ColumnReference c, out object? value) object? val = Evaluate(be.Value), low = Evaluate(be.Low), high = Evaluate(be.High); if (val is null || low is null || high is null) return null; - int toLow = CompareAsKinds(val, low), toHigh = CompareAsKinds(val, high); + int toLow = CompareAsKinds(be.Value, val, be.Low, low, Text), toHigh = CompareAsKinds(be.Value, val, be.High, high, Text); bool inside = (toLow >= 0 && toHigh <= 0) || (toLow <= 0 && toHigh >= 0); return inside != be.Negated; } @@ -213,6 +225,7 @@ private static bool TryNiladicFunction(ColumnReference c, out object? value) "SWITCH" => Switch(f), "NULLIF" => NullIf(f), "COALESCE" => Coalesce(f), + "NZ" => Nz(f), "GREATEST" => Extreme(f, greatest: true), "LEAST" => Extreme(f, greatest: false), "DATEPART" => DatePart(f), @@ -224,7 +237,8 @@ private static bool TryNiladicFunction(ColumnReference c, out object? value) // does: text as a number, a date as its serial, True as -1 — so CByte(True) overflows. CInt/CLng/CByte // round half to even, as Convert.ToInt16/Int32/Byte do, and a value past the type is an overflow. ACE // raises "Invalid use of Null" for a Null argument; LibRed returns Null. CVar passes its argument - // through (LibRed has no Variant type; ACE hands the value back as text). + // through: a Variant keeps its own type while an expression uses it, and the executor writes it out as + // text where ACE does (QueryExecutor.Variance). "CCUR" => DecimalArgument(f, ToCurrency), "CBOOL" => Convert1(f, v => VbaBool(v)), "CBYTE" => Convert1(f, v => Convert.ToByte(ConversionNumber(v), CultureInfo.InvariantCulture)), @@ -324,7 +338,9 @@ private static bool TryNiladicFunction(ColumnReference c, out object? value) // More VBA/Access built-ins (verified vs ACE via the function-whitelist sweep). All NULL-propagating // via Convert1 unless noted; positions are 1-based. // Asc and Chr work in the system ANSI code page (Chr takes 0-255; Chr(128) is '€', Asc('Ā') is 65 by - // best fit); AscW and ChrW in UTF-16 code units, AscW signed and ChrW taking -32768 to 65535. + // best fit); AscW and ChrW in UTF-16 code units, AscW signed and ChrW taking -32768 to 65535. The + // machine's code page, NOT the database's: in a 1251, 1253 or 932 database ACE on a 1252 machine still + // gives Chr(192) = 'À', and Asc of a Cyrillic or Greek letter is 63. "ASC" => Convert1(f, v => (int)Ansi.GetBytes(FirstCharacter(v))[0]), "CHR" => Convert1(f, v => AnsiCharacter(InRange(AsLong(v), 0, 255)).ToString()), "SPACE" => Convert1(f, v => new string(' ', Count(v))), @@ -432,6 +448,7 @@ internal static void ValidateArity(string name, int count) // COALESCE(expression [, ...n]). SQL Server insists on two, but one is harmless and the standard's // own grammar allows it, so only an empty list is rejected. "COALESCE" => (1, int.MaxValue), + "NZ" => (1, 2), // GREATEST/LEAST(expression [, ...n]), as SQL Server and PostgreSQL take them: one argument or more. "GREATEST" or "LEAST" => (1, int.MaxValue), @@ -514,7 +531,7 @@ _ when RunningAggregate.Supports(name) => RunningAggregate.IsPair(name) ? (2, 2) if (left is null) return null; object? right = Evaluate(f.Arguments[1]); - return right is not null && Compare(left, right) == 0 ? null : left; + return right is not null && Compare(left, right, Text) == 0 ? null : left; } /// @@ -542,6 +559,20 @@ _ when RunningAggregate.Supports(name) => RunningAggregate.IsPair(name) ? (2, 2) return null; } + /// + /// Access's Nz(value [, valueIfNull]): value, or when it is Null valueIfNull, or VBA's + /// when there is none. ACE's expression service has no Nz — it is the Access application's, + /// so it runs in queries opened in Access but not over OLE DB — and it is here for the queries written in Access. + /// + /// + /// In Access the result is a Variant, and LibRed makes it one (QueryExecutor.VarianceOf), with what that brings + /// (verified vs Access): it is written out as text — Nz(K, 0) is "0" and Nz(Null, 5) is + /// "5" — and sorts and groups as its text, so ORDER BY Nz(K, 0) puts 10 before 2; as an operand it + /// keeps its own value, so Nz(K, 0) + 1 adds and Nz(K, 0) > 2 compares as a number. + /// + private object? Nz(FunctionCall f) => + Evaluate(f.Arguments[0]) ?? (f.Arguments.Count == 2 ? Evaluate(f.Arguments[1]) : VbaEmpty.Value); + /// /// GREATEST(a, b, …) and LEAST(a, b, …) — the largest or smallest of the arguments, compared /// as the < and > operators compare. Access/ACE has neither, so like COALESCE they are @@ -561,7 +592,7 @@ _ when RunningAggregate.Supports(name) => RunningAggregate.IsPair(name) ? (2, 2) object? value = Evaluate(argument); if (value is null) continue; - if (result is null || (greatest ? Compare(value, result) > 0 : Compare(value, result) < 0)) + if (result is null || (greatest ? Compare(value, result, Text) > 0 : Compare(value, result, Text) < 0)) result = value; } @@ -671,9 +702,9 @@ private static bool IsLocaleId(int lcid) /// a position and length, or (-1, 0). A textual match compares in the database sort order, so 'SS' finds 'ß' and /// the matched length can differ from 's. /// - private static (int Index, int Length) FindText(string text, string find, int start, bool binary) + private static (int Index, int Length) FindText(string text, string find, int start, bool binary, JetTextComparer order) { - if (binary || IsPlainText(text) && IsPlainText(find)) + if (binary || order.IsUntailoredGeneral && IsPlainText(text) && IsPlainText(find)) { int index = text.IndexOf(find, start, binary ? StringComparison.Ordinal : StringComparison.OrdinalIgnoreCase); return (index, find.Length); @@ -683,17 +714,19 @@ private static (int Index, int Length) FindText(string text, string find, int st { for (int length = shortest; length <= Math.Min(text.Length - i, find.Length * 2); length++) { - if (CompareText(text.Substring(i, length), find) == 0) + if (order.Compare(text.Substring(i, length), find) == 0) return (i, length); } } return (-1, 0); } - /// Text whose database order is plain case-insensitive order: ASCII with no hyphen or apostrophe, - /// which the order weighs apart, and no trailing space, which it ignores. + /// Text an untailored General order compares as plain case-insensitive text: printable ASCII with no + /// hyphen or apostrophe, which the order weighs apart, and no trailing space, which it ignores (held for both + /// versions by PlainTextCollationTests). A tailored order gives such letters weights of its own, so the + /// caller asks first. private static bool IsPlainText(string text) => - text.All(c => c < 0x80 && c is not ('-' or '\'')) && !text.EndsWith(' '); + text.All(c => c is >= ' ' and <= '~' and not ('-' or '\'')) && !text.EndsWith(' '); /// /// Access String(count, character): the character repeated. A text gives its first character (an empty @@ -723,7 +756,7 @@ private static bool IsPlainText(string text) => string left = ConcatText(a), right = ConcatText(b); if (binary) return Math.Sign(string.CompareOrdinal(left, right)); - int order = CompareText(left, right); + int order = Text.Compare(left, right); return order != 0 ? order : Math.Sign(TrailingSpaces(left) - TrailingSpaces(right)); static int TrailingSpaces(string s) => s.Length - s.TrimEnd(' ').Length; @@ -758,7 +791,7 @@ private static bool IsPlainText(string text) => string window = s1[..start]; // search within Left(string1, start) if (s2.Length == 0) return start; // empty needle → the effective start position int last = -1; - for ((int index, int _) = FindText(window, s2, 0, binary); index >= 0; (index, _) = FindText(window, s2, index + 1, binary)) + for ((int index, int _) = FindText(window, s2, 0, binary, Text); index >= 0; (index, _) = FindText(window, s2, index + 1, binary, Text)) last = index; return last + 1; } @@ -895,8 +928,22 @@ decimal when IsCurrency(expression) => 6, // vbCurrency }; /// Whether an expression is a Currency: a Currency column, CCur, or arithmetic that keeps one. - private bool IsCurrency(Expression expression) => - NumberTypeOf(expression, scope.AllColumns(), _ => null).Class == NumberClass.Currency; + /// Decided by the expression and the scope's columns, never the row, so it is worked out once per node + /// for this evaluator — which is reused across rows. Working it out walks the schema, and did so for every row + /// of every Currency-typed arithmetic result. + private bool IsCurrency(Expression expression) + { + // IDE0028's only fix here is `[]`, which would drop the comparer and key the nodes structurally. +#pragma warning disable IDE0028 + _currency ??= new Dictionary(ReferenceEqualityComparer.Instance); +#pragma warning restore IDE0028 + if (!_currency.TryGetValue(expression, out bool currency)) + _currency[expression] = currency = + NumberTypeOf(expression, scope.AllColumns(), _ => null).Class == NumberClass.Currency; + return currency; + } + + private Dictionary? _currency; /// /// Access StrConv(string, conversion, [LCID]) (verified vs ACE). 0 leaves the text as it is. 1, 2 and 3 @@ -1542,7 +1589,7 @@ private static DateTime DateValueArgument(object v) => if (s1.Length == 0) return 0; if (s2.Length == 0) return start; if (start > s1.Length) return 0; - return FindText(s1, s2, start - 1, binary).Index + 1; + return FindText(s1, s2, start - 1, binary, Text).Index + 1; } /// Access REPLACE(string, find, replace[, start[, count[, compare]]]) — the text from start on, with @@ -1571,7 +1618,7 @@ private static DateTime DateValueArgument(object v) => int pos = 0, replaced = 0; while (true) { - (int j, int length) = count >= 0 && replaced >= count ? (-1, 0) : FindText(s, find, pos, binary); + (int j, int length) = count >= 0 && replaced >= count ? (-1, 0) : FindText(s, find, pos, binary, Text); if (j < 0) { sb.Append(s.AsSpan(pos)); break; } sb.Append(s, pos, j - pos).Append(repl); pos = j + length; @@ -1964,10 +2011,20 @@ private static DateTime InDateRange(Func compute) UnaryOperator.BitNot => v is null ? null : BitNot(v), UnaryOperator.IsNull => v is null, UnaryOperator.IsNotNull => v is not null, + UnaryOperator.IsTrue => AsBool(v) is true, + UnaryOperator.IsNotTrue => AsBool(v) is not true, + UnaryOperator.IsFalse => AsBool(v) is false, + UnaryOperator.IsNotFalse => AsBool(v) is not false, _ => throw new NotSupportedException($"Unary operator {u.Operator}."), }; } + // A comparison's result as an object without boxing a new bool for every row it is asked of: a filter or a + // join's residual ON compares once per row, and each answer was an allocation. + private static readonly object BoxedTrue = true, BoxedFalse = false; + + private static object Boxed(bool value) => value ? BoxedTrue : BoxedFalse; + private object? EvaluateBinary(BinaryExpression b) { // AND/OR use Kleene three-valued logic, and short-circuit: `false AND x` is false and `true OR x` @@ -2021,6 +2078,16 @@ private static DateTime InDateRange(Func compute) ? null : (left is null ? "" : ConcatText(left)) + (right is null ? "" : ConcatText(right)); + // IS [NOT] DISTINCT FROM is '=' with Null taken as a value, so it is never Null: two Nulls are not + // distinct, a Null and a value are. + if (b.Operator is BinaryOperator.IsDistinctFrom or BinaryOperator.IsNotDistinctFrom) + { + bool distinct = left is null || right is null + ? (left is null) != (right is null) + : CompareAsKinds(b.Left, left, b.Right, right, Text) != 0; + return distinct == (b.Operator == BinaryOperator.IsDistinctFrom); + } + // The arithmetic operators other than '+' read text as a number even when the other side is Null, so text // that is not a number, a GUID or a binary value is a type mismatch before Null propagates (verified vs // ACE: 'abc' * NULL fails, '1' * NULL is Null). @@ -2048,7 +2115,7 @@ private static DateTime InDateRange(Func compute) return null; if (b.Operator is BinaryOperator.Equal or BinaryOperator.NotEqual && TruthTest(b, left, right) is bool truth) - return b.Operator == BinaryOperator.Equal ? truth : !truth; + return Boxed(b.Operator == BinaryOperator.Equal ? truth : !truth); // A result that is a Currency (NumberTypeOf) is one at every step, not only in the result column: its four // places and its range apply to it where it is worked out, as CCur applies them (verified vs ACE: Currency @@ -2067,12 +2134,12 @@ private static DateTime InDateRange(Func compute) return b.Operator switch { - BinaryOperator.Equal => CompareAsKinds(left, right) == 0, - BinaryOperator.NotEqual => CompareAsKinds(left, right) != 0, - BinaryOperator.LessThan => CompareAsKinds(left, right) < 0, - BinaryOperator.LessThanOrEqual => CompareAsKinds(left, right) <= 0, - BinaryOperator.GreaterThan => CompareAsKinds(left, right) > 0, - BinaryOperator.GreaterThanOrEqual => CompareAsKinds(left, right) >= 0, + BinaryOperator.Equal => Boxed(CompareOperands(b, left, right) == 0), + BinaryOperator.NotEqual => Boxed(CompareOperands(b, left, right) != 0), + BinaryOperator.LessThan => Boxed(CompareOperands(b, left, right) < 0), + BinaryOperator.LessThanOrEqual => Boxed(CompareOperands(b, left, right) <= 0), + BinaryOperator.GreaterThan => Boxed(CompareOperands(b, left, right) > 0), + BinaryOperator.GreaterThanOrEqual => Boxed(CompareOperands(b, left, right) >= 0), // LIKE reads any other value as the text CStr gives it (verified vs ACE: TRUE LIKE '-1' is True). A binary // value becomes text too, so LIKE is case-insensitive over a binary column even though '=' on the same // column is byte-wise: `B LIKE 'A%'` matches both 0x4100 ('A') and 0x6100 ('a'). @@ -2272,11 +2339,72 @@ private static (object Left, object Right) Comparable(object left, object right) return (leftText ? TextAsNumber((string)left) : Serial(left), rightText ? TextAsNumber((string)right) : Serial(right)); } - /// The order of two values once has brought them to a common kind. - private static int CompareAsKinds(object left, object right) + /// The order of a comparison operator's two operands. Two texts compare by collation key, as + /// would compare them, with two shortcuts that change no answer: texts identical + /// once trailing spaces go are equal in any order, and the key of a literal or parameter side is made once for + /// this evaluator — which is reused across rows — rather than for every row it is compared with. + private int CompareOperands(BinaryExpression b, object left, object right) { + if (left is string l && right is string r) + { + if (l.AsSpan().TrimEnd(' ').SequenceEqual(r.AsSpan().TrimEnd(' '))) + return 0; + if (b.Right is LiteralExpression or ParameterExpression) + return Text.Compare(l, ConstantKey(b.Right, r)); + if (b.Left is LiteralExpression or ParameterExpression) + return -Text.Compare(r, ConstantKey(b.Left, l)); + } + + return CompareAsKinds(b.Left, left, b.Right, right, Text); + } + + // Collation keys of the literal and parameter operands met so far, by node: a statement's constants do not + // change between the rows its evaluator is rebound to. + private Dictionary? _constantKeys; + + private byte[] ConstantKey(Expression constant, string text) + { + // IDE0028's only fix here is `[]`, which would drop the comparer and key the nodes structurally. +#pragma warning disable IDE0028 + _constantKeys ??= new Dictionary(ReferenceEqualityComparer.Instance); +#pragma warning restore IDE0028 + if (!_constantKeys.TryGetValue(constant, out byte[]? key)) + _constantKeys[constant] = key = Text.Key(text); + return key; + } + + /// The order of two values once has brought them to a common kind. A parameter + /// compared with text takes the text's type (verified vs ACE: a numeric parameter against a text column compares + /// as text, so [S] > ? with 100 counts 'abc' and '11'). + private static int CompareAsKinds( + Expression leftOperand, object left, Expression rightOperand, object right, JetTextComparer text) + { + if (leftOperand is ParameterExpression && right is string && left is not string) left = ConcatText(left); + if (rightOperand is ParameterExpression && left is string && right is not string) right = ConcatText(right); (object l, object r) = Comparable(left, right); - return Compare(l, r); + return Compare(l, r, text); + } + + /// The key an index seek must use for column = value to select exactly the rows the + /// comparison selects, or false when there is no such key and the caller has to scan instead. + /// + /// An index answers only in its column's own kind — its keys are encoded and ordered as that type — so a + /// comparison that happens in a different kind has no key range to seek. S = 1 on text compares as + /// a number (see ), matching ' 1 ', '1.0' and '+1' as well as + /// '1', which are scattered through the index rather than adjacent in it. The one cross-kind case + /// that IS seekable is a parameter against text, which converts to text before + /// comparing; the seek converts it the same way and so asks the index the same question. + /// + internal static bool TryGetSeekKey(ColumnDef column, Expression operand, object? value, out object? key) + { + key = value; + if (value is null) return true; + + IndexSelection.TypeKind? columnKind = IndexSelection.Classify(column.Type); + if (columnKind == IndexSelection.TypeKind.Text && value is not string + && (operand is ParameterExpression || value is char)) + key = ConcatText(value); + return IndexSelection.KindOf(key) == columnKind; } private static object Serial(object value) => value is DateTime d ? d.ToOADate() : value; @@ -2467,9 +2595,9 @@ private static NumberType Sum(NumberType left, NumberType right, bool add) /// /// A value converted to the type its result column declares, when it has another: a number to a wider number - /// (a Boolean as -1 or 0, a Double into a Decimal the OLE Automation way), anything to text as & writes - /// it, and anything to binary as its bytes (). Null, or no , leaves - /// the value as it is. + /// (a Boolean as -1 or 0, a Double into a Decimal the OLE Automation way), a date to a number as its serial and a + /// number to a date as CDate reads it, anything to text as & writes it, and anything to binary as its + /// bytes (). Null, or no , leaves the value as it is. /// internal static object? AsColumnType(object? value, Type? type, bool currency) { @@ -2477,8 +2605,9 @@ private static NumberType Sum(NumberType left, NumberType right, bool add) return value; if (type == typeof(string)) return ConcatText(value); if (type == typeof(byte[])) return ColumnBytes(value, currency); - if (type == typeof(decimal)) return Dec(value); - if (type == typeof(double)) return Dbl(value); + if (type == typeof(decimal)) return ArithmeticDecimal(value); + if (type == typeof(double)) return Oa(value); + if (type == typeof(DateTime)) return ToDate(value); return Convert.ChangeType(Numeric(value), type, CultureInfo.InvariantCulture); } @@ -2586,6 +2715,7 @@ private static object Add(object left, object right) => /// is one character of text. private static object? NumericOperand(object? v) => v switch { + VbaEmpty => (short)0, string s => TextAsNumber(s), char c => TextAsNumber(c.ToString()), Guid or byte[] => throw new InvalidCastException("Type mismatch: a GUID or binary value is not a number."), @@ -2601,6 +2731,7 @@ private static object Add(object left, object right) => internal static string ConcatText(object v) => v switch { string s => s, + VbaEmpty => "", bool b => b ? "-1" : "0", double d => FloatingText(d, 15), float f => FloatingText(f, 7), @@ -2817,7 +2948,12 @@ v is bool b ? b : Dbl(ConversionNumber(v)) != 0; // Jet's boolean convention (true = -1, false = 0) so a bool matches the numeric column it is stored in. - private static object Numeric(object v) => v is bool b ? (b ? -1 : 0) : v; + private static object Numeric(object v) => v switch + { + bool b => b ? -1 : 0, + VbaEmpty => (short)0, + _ => v, + }; private static decimal Dec(object v) => JetDecimalConverter.ToDecimal(Numeric(v), CultureInfo.InvariantCulture); private static double Dbl(object v) => Convert.ToDouble(Numeric(v), CultureInfo.InvariantCulture); // Narrow to single precision (the cast yields ±Infinity for an out-of-range double rather than throwing). @@ -2828,8 +2964,12 @@ v is bool b ? b // For date arithmetic: a DateTime becomes its OLE Automation serial; a number is taken verbatim (as days). private static double Oa(object v) => v is DateTime d ? d.ToOADate() : Dbl(v); - private static int Compare(object left, object right) + private static int Compare(object left, object right, JetTextComparer text) { + // Empty is "" beside text and 0 beside anything else, as VBA compares it. + if (left is VbaEmpty) left = right is string ? "" : (short)0; + if (right is VbaEmpty) right = left is string ? "" : (short)0; + if (IsNumeric(left) && IsNumeric(right)) { // A single-precision operand (a Single column value, a CSNG result, a SUM of singles) compares in @@ -2851,13 +2991,13 @@ private static int Compare(object left, object right) // Binary (byte[]) columns: structural, length-sensitive byte compare — lexicographic then by // length, so a shorter value sorts before a longer one sharing its prefix (Jet's binary order, - // matching IndexKeyEncoder). Without this, byte[] falls through to ToString() ("System.Byte[]" + // matching IndexKeyCodec). Without this, byte[] falls through to ToString() ("System.Byte[]" // for every array) and all binaries compare *equal* — so `WHERE binKey = @p` matches every row. if (left is byte[] lb && right is byte[] rb) return CompareBytes(lb, rb); if (left is string || right is string) - return CompareText(left.ToString()!, right.ToString()!); + return text.Compare(left.ToString()!, right.ToString()!); // Dates compare by their OLE Automation serial rather than chronologically. Below the epoch // (1899-12-30) the day count is negative while the time fraction stays positive, so 1899-12-29 06:00 is @@ -2865,7 +3005,7 @@ private static int Compare(object left, object right) // serial and therefore puts later pre-epoch times first (verified in // LibRed.Core.Tests.AcePreEpochDateProbeTest: `06:00 < 18:00` is False, ORDER BY gives 1,3,2,4,5,6). // - // Matching it is not only about ACE parity: IndexKeyEncoder writes this same serial as the index key, + // Matching it is not only about ACE parity: IndexKeyCodec writes this same serial as the index key, // and that encoding cannot change because ACE writes those keys too. Comparing chronologically here // while the index compares by serial made an index seek and a table scan return DIFFERENT rows for a // pre-epoch range (see PreEpochDateOrderingTests). From the epoch onward the two orders are identical, @@ -2884,22 +3024,22 @@ private static int Compare(object left, object right) if (left is IComparable c && left.GetType() == right.GetType()) return c.CompareTo(right); - return CompareText(left.ToString()!, right.ToString()!); + return text.Compare(left.ToString()!, right.ToString()!); } /// Whether two non-null values are equal under the same coercions as = (used by the hash /// join to re-check a bucket candidate). Only meaningful within one type kind — see . - public static bool KeyEqual(object a, object b) => Compare(a, b) == 0; + public static bool KeyEqual(object a, object b, JetTextComparer text) => + a is string sa && b is string sb ? text.Equals(sa, sb) : Compare(a, b, text) == 0; /// A hash for a non-null join key that agrees with within a type kind: values - /// the evaluator treats as equal hash the same (numeric via double, text via Access's case-insensitive/ - /// trailing-space-trimmed collation, binary structurally). The planner only builds a hash join over - /// same-kind key columns, so this is total over the keys it actually sees. - public static int KeyHash(object v) => v switch + /// the evaluator treats as equal hash the same (numeric via double, text by its collation key, binary + /// structurally). The planner only builds a hash join over same-kind key columns, so this is total over the + /// keys it actually sees. + public static int KeyHash(object v, JetTextComparer text) => v switch { byte[] b => BinaryHash(b), - string s => System.Globalization.CultureInfo.InvariantCulture.CompareInfo - .GetHashCode(s.TrimEnd(' '), System.Globalization.CompareOptions.IgnoreCase), + string s => text.GetHashCode(s), _ when IsNumeric(v) => Dbl(v).GetHashCode(), _ => v.GetHashCode(), }; @@ -2920,27 +3060,35 @@ private static int CompareBytes(byte[] a, byte[] b) return a.Length.CompareTo(b.Length); } - /// Access text comparison, in the database sort order (): case-insensitive, - /// trailing spaces ignored, an accented letter beside its base letter but not equal to it (verified vs ACE: - /// 'é' < 'f', 'café' ≠ 'cafe'), 'ß' = 'ss', and a hyphen weighed after the letters. A - /// character that order does not cover compares case-insensitively. - // The linguistic comparison is the point: ordinal (CA1309) would put 'é' after 'z' and make 'ß' ≠ 'ss', - // neither of which is what ACE does. -#pragma warning disable CA1309 - private static int CompareText(string a, string b) => - JetTextComparer.Compare(a, b) - ?? Math.Sign(string.Compare(a.TrimEnd(' '), b.TrimEnd(' '), StringComparison.InvariantCultureIgnoreCase)); -#pragma warning restore CA1309 - - /// Orders two values for SORT (nulls first), using the same coercion as comparisons. - public static int CompareForSort(object? a, object? b) => (a, b) switch + /// Orders two values for SORT (nulls first), using the same coercion as comparisons, and text in + /// 's collation. + public static int CompareForSort(object? a, object? b, JetTextComparer text) => (a, b) switch { (null, null) => 0, (null, _) => -1, (_, null) => 1, - _ => Compare(a, b), + // Two sort keys compare their collation keys, which is what comparing the texts would encode again; + // one against anything else unwraps to its text and compares as text always does. + (CollatedText x, CollatedText y) => Math.Sign(x.Key.AsSpan().SequenceCompareTo(y.Key)), + (CollatedText x, _) => CompareForSort(x.Text, b, text), + (_, CollatedText y) => CompareForSort(a, y.Text, text), + _ => Compare(a, b, text), }; + /// A value as a sort key: text carries its collation key, made once, so sorting n rows encodes n + /// strings rather than two per comparison; anything else is itself. Only for values that are compared with + /// in the same collation and never returned. + internal static object? SortKey(object? value, JetTextComparer text) => + value is string s ? new CollatedText(s, text.Key(s)) : value; + + /// A text sort key and its collation key (). + internal sealed class CollatedText(string text, byte[] key) + { + public string Text { get; } = text; + public byte[] Key { get; } = key; + public override string ToString() => Text; + } + // Booleans count as numeric for comparison: EF maps CLR bool to a numeric (smallint) column, and // a boolean predicate (e.g. IS NOT NULL) must compare equal to that stored value. The comparison // coercions (Dec/Dbl) use Jet's convention (false = 0, true = -1) so a bool matches the numeric value diff --git a/src/LibRed/LibRed.Engine/Execution/HoistedInSet.cs b/src/LibRed/LibRed.Engine/Execution/HoistedInSet.cs index 30defc6b3..ae672fb16 100644 --- a/src/LibRed/LibRed.Engine/Execution/HoistedInSet.cs +++ b/src/LibRed/LibRed.Engine/Execution/HoistedInSet.cs @@ -13,8 +13,8 @@ namespace LibRed.Engine.Execution; /// within one type kind, which is the same constraint the hash join lives under: 5 = '5' and /// 5 = 5.0, but '5' ≠ '5.0', so no single hash can agree with = across kinds. Numeric and /// text are taken because is defined to agree with -/// for exactly those (numeric via double, text via Access's -/// case-insensitive, trailing-space-trimmed collation). Everything else — mixed kinds in the body, or a probe +/// for exactly those (numeric via double, text via the database's +/// collation key). Everything else — mixed kinds in the body, or a probe /// of a different kind from the body — declines, and the caller scans the list as it always did. Declining /// costs nothing but the old behaviour; a wrong hash would silently drop matching rows. /// Dates deliberately do not qualify. The evaluator compares two DateTimes by their OLE Automation @@ -32,8 +32,6 @@ private enum Kind Text, } - private static readonly IEqualityComparer Comparer = new EvaluatorEquality(); - private readonly HashSet _values; private readonly Kind _kind; @@ -49,12 +47,13 @@ private HoistedInSet(HashSet values, Kind kind, bool hasNull) /// Builds a set over , or null when they cannot be hashed consistently /// (a kind outside , or more than one kind among them). An empty or all-null body also - /// returns null: there is nothing to accelerate, and the caller's scan of it is already trivial. - public static HoistedInSet? TryBuild(IReadOnlyList values) + /// returns null: there is nothing to accelerate, and the caller's scan of it is already trivial. Text is + /// compared in 's collation, as = compares it. + public static HoistedInSet? TryBuild(IReadOnlyList values, LibRed.Storage.JetTextComparer text) { Kind? kind = null; bool hasNull = false; - var set = new HashSet(Comparer); + var set = new HashSet(new EvaluatorEquality(text)); foreach (object? value in values) { @@ -87,11 +86,11 @@ _ when ExpressionEvaluator.IsNumeric(value) => Kind.Numeric, }; /// Equality and hashing delegated to the evaluator, so the set agrees with = exactly. - private sealed class EvaluatorEquality : IEqualityComparer + private sealed class EvaluatorEquality(LibRed.Storage.JetTextComparer text) : IEqualityComparer { public new bool Equals(object? a, object? b) => - a is not null && b is not null && ExpressionEvaluator.KeyEqual(a, b); + a is not null && b is not null && ExpressionEvaluator.KeyEqual(a, b, text); - public int GetHashCode(object value) => ExpressionEvaluator.KeyHash(value); + public int GetHashCode(object value) => ExpressionEvaluator.KeyHash(value, text); } } \ No newline at end of file diff --git a/src/LibRed/LibRed.Engine/Execution/LikeMatcher.cs b/src/LibRed/LibRed.Engine/Execution/LikeMatcher.cs index 3c6518ab3..d29031a54 100644 --- a/src/LibRed/LibRed.Engine/Execution/LikeMatcher.cs +++ b/src/LibRed/LibRed.Engine/Execution/LikeMatcher.cs @@ -20,9 +20,17 @@ namespace LibRed.Engine.Execution; /// [] matches nothing at all, so []] is a plain ] and [[]] the text []. The /// Access documentation gives ^ as the ANSI-92 negation, but ACE does not treat it so: [^ae] lists /// ^, a and e, and ! negates as it does in ANSI-89. -/// Case is ignored and accents are not. ß counts as ss and æ as ae, in the pattern, -/// in a bracket list and in the value, so 'aßb' LIKE 'a[s]sb' is True; _ still takes the whole -/// character. +/// Case is ignored and accents are not, by ACE's own table rather than any runtime's casing: it folds 973 +/// case pairs and leaves out what a runtime would add — the micro sign, long s, final sigma, the Greek symbol +/// variants, the titlecase digraphs and every letter Unicode added later. Four letters are spelt out: ß as +/// ss, æ as ae, œ as oe and þ as th, capitals likewise — in the +/// pattern, in a bracket list and in the value, so 'aßb' LIKE 'a[s]sb' is True; _ still takes the whole +/// character. The table is measured (LikeFoldTableGeneratorTest) and embedded, so LIKE answers the same on every +/// platform. +/// None of it depends on the database's collation: the same pairs match in every one of the 405 non-CJK orders +/// LibRed can create, Turkish included (verified vs ACE; the CJK orders are not yet screened). Width, kana, superscripts, hyphens and apostrophes, which the +/// collation folds or ignores, all count here, and so do trailing spaces: 'a ' LIKE 'a' is False where +/// 'a ' = 'a' is True. /// A bracket that is never closed, or a range written backwards, is an invalid pattern. It is only reported /// when the match reaches it with a character left to test: '' LIKE '[' and 'z' LIKE 'x[z-a]' are /// both False. @@ -35,14 +43,55 @@ namespace LibRed.Engine.Execution; /// internal static class LikeMatcher { - /// The two characters a character counts as (ß as ss, æ and Æ as ae), or null for itself. - private static string? Expansion(char c) => c switch + /// What each character folds to, and the characters spelt out as several — ACE's own table. + private static readonly (char[] Fold, string?[] Expansions) Table = LoadTable(); + + /// Laid out as the sort-key tables are: the two counts, then each section deflated on its own — the + /// case pairs as (character, what it folds to), the expansions as (character, letter count, letters). Both are + /// held as arrays indexed by character — the expansions only as far as the highest one — because every + /// character of every value tested is looked up in each. + private static (char[] Fold, string?[] Expansions) LoadTable() { - 'ß' => "ss", - 'æ' => "ae", - 'Æ' => "AE", - _ => null, - }; + var fold = new char[char.MaxValue + 1]; + for (int c = 0; c <= char.MaxValue; c++) fold[c] = (char)c; + var spelt = new Dictionary(); + + using Stream stream = typeof(LikeMatcher).Assembly.GetManifestResourceStream("LibRed.Engine.Resources.LikeFold.bin") + ?? throw new InvalidOperationException("The LIKE fold table resource is missing from the assembly."); + var reader = new BinaryReader(stream); + int caseCount = reader.ReadInt32(), expansionCount = reader.ReadInt32(); + var cases = new BinaryReader(new MemoryStream(Inflate(reader))); + var expansionStream = new BinaryReader(new MemoryStream(Inflate(reader))); + + for (int i = 0; i < caseCount; i++) + fold[cases.ReadUInt16()] = (char)cases.ReadUInt16(); + for (int i = 0; i < expansionCount; i++) + { + char c = (char)expansionStream.ReadUInt16(); + var letters = new char[expansionStream.ReadByte()]; + for (int k = 0; k < letters.Length; k++) letters[k] = (char)expansionStream.ReadUInt16(); + spelt[c] = new string(letters); + } + + var expansions = new string?[spelt.Count == 0 ? 0 : spelt.Keys.Max() + 1]; + foreach ((char c, string letters) in spelt) expansions[c] = letters; + return (fold, expansions); + + static byte[] Inflate(BinaryReader reader) + { + byte[] compressed = reader.ReadBytes(reader.ReadInt32()); + var output = new MemoryStream(); + using (var inflate = new System.IO.Compression.ZLibStream(new MemoryStream(compressed), System.IO.Compression.CompressionMode.Decompress)) + inflate.CopyTo(output); + return output.ToArray(); + } + } + + /// The letters a character is spelt out as, already folded (ß as SS), or null for itself. + private static string? Expansion(char c) => c < Table.Expansions.Length ? Table.Expansions[c] : null; + + /// A character folded by ACE's case table. + private static char FoldChar(char c) => Table.Fold[c]; public static bool IsMatch(string value, string pattern) { @@ -128,19 +177,35 @@ private static int[] CharacterEnds(string value, int foldedLength) private static bool Invalid(int first, string text) => first < text.Length ? throw new ArgumentException("Invalid pattern string.") : false; - /// Text as the match compares it: expansions spelt out, then upper case. Upper-casing keeps the length, - /// so positions in the folded text line up with one another. + /// Text as the match compares it: expansions spelt out, every character folded. Folding keeps the + /// length, so positions in the folded text line up with one another. private static string Fold(string text) { + // Almost no text holds a character that is spelt out, and without one the folded text is the same length: + // written straight into the result, a table read per character. + bool spelt = false; + foreach (char c in text) + { + if (Expansion(c) is not null) { spelt = true; break; } + } + if (!spelt) + { + return string.Create(text.Length, text, static (span, source) => + { + char[] fold = Table.Fold; + for (int i = 0; i < span.Length; i++) span[i] = fold[source[i]]; + }); + } + var folded = new System.Text.StringBuilder(text.Length); foreach (char c in text) { if (Expansion(c) is { } expansion) folded.Append(expansion); else - folded.Append(c); + folded.Append(FoldChar(c)); } - return folded.ToString().ToUpperInvariant(); + return folded.ToString(); } private sealed class BracketList(bool negated, List members, List<(char Low, char High)> ranges) @@ -160,7 +225,7 @@ private sealed class BracketList(bool negated, List members, List<(char { if (body[i] > body[i + 2]) return null; - ranges.Add((char.ToUpperInvariant(body[i]), char.ToUpperInvariant(body[i + 2]))); + ranges.Add((FoldChar(body[i]), FoldChar(body[i + 2]))); i += 2; } else diff --git a/src/LibRed/LibRed.Engine/Execution/ListAgg.cs b/src/LibRed/LibRed.Engine/Execution/ListAgg.cs index 6323ee105..ad3bd5fe7 100644 --- a/src/LibRed/LibRed.Engine/Execution/ListAgg.cs +++ b/src/LibRed/LibRed.Engine/Execution/ListAgg.cs @@ -13,17 +13,17 @@ internal static class ListAgg /// The list over — each a value and its WITHIN GROUP key values — or Null when no value /// is present. Each value is written as & writes it. Rows whose keys tie keep their order. Under /// a value repeated — equal as GROUP BY takes values to be equal — is listed once, - /// where it first comes. + /// where it first comes. Text is ordered and compared in 's collation. /// public static string? Of( IEnumerable<(object? Value, object?[] Keys)> rows, string separator, IReadOnlyList directions, - bool distinct) + bool distinct, LibRed.Storage.JetTextComparer text) { var comparer = Comparer.Create((a, b) => { for (int k = 0; k < directions.Count; k++) { - int c = ExpressionEvaluator.CompareForSort(a[k], b[k]); + int c = ExpressionEvaluator.CompareForSort(a[k], b[k], text); if (c != 0) return directions[k] == SortDirection.Descending ? -c : c; } @@ -33,10 +33,10 @@ internal static class ListAgg if (distinct) { var seen = new HashSet(); - values = values.Where(v => seen.Add(new QueryExecutor.GroupKey([v]))); + values = values.Where(v => seen.Add(new QueryExecutor.GroupKey([v], text))); } - var text = values.Select(ExpressionEvaluator.ConcatText).ToList(); - return text.Count == 0 ? null : string.Join(separator, text); + var listed = values.Select(ExpressionEvaluator.ConcatText).ToList(); + return listed.Count == 0 ? null : string.Join(separator, listed); } } \ No newline at end of file diff --git a/src/LibRed/LibRed.Engine/Execution/ParameterBag.cs b/src/LibRed/LibRed.Engine/Execution/ParameterBag.cs index e6304bef6..f3347867c 100644 --- a/src/LibRed/LibRed.Engine/Execution/ParameterBag.cs +++ b/src/LibRed/LibRed.Engine/Execution/ParameterBag.cs @@ -1,3 +1,5 @@ +using LibRed.Storage.Types; + namespace LibRed.Engine.Execution; /// @@ -13,9 +15,6 @@ namespace LibRed.Engine.Execution; /// internal sealed class ParameterBag { - /// The OLE epoch: Jet stores a time as the epoch plus the time of day. - private static readonly DateTime OleEpoch = new(1899, 12, 30); - // IDE0028's only fix here is `[]`, which would silently drop the comparer and make parameter // lookup case-sensitive. #pragma warning disable IDE0028 @@ -36,8 +35,8 @@ public ParameterBag(IReadOnlyDictionary? values) if (_values.TryGetValue(Normalize(name), out object? value)) return value switch { - TimeSpan span => OleEpoch + span, - TimeOnly time => OleEpoch + time.ToTimeSpan(), + TimeSpan span => JetTypeCodec.OleEpoch + span, + TimeOnly time => JetTypeCodec.OleEpoch + time.ToTimeSpan(), _ => value, }; throw new InvalidOperationException($"No value was supplied for parameter '{name}'."); diff --git a/src/LibRed/LibRed.Engine/Execution/Percentile.cs b/src/LibRed/LibRed.Engine/Execution/Percentile.cs index f4215dde5..9226e72a6 100644 --- a/src/LibRed/LibRed.Engine/Execution/Percentile.cs +++ b/src/LibRed/LibRed.Engine/Execution/Percentile.cs @@ -29,7 +29,9 @@ internal static class Percentile /// either side of it; text cannot be interpolated. The fraction is read as the Decimal it was written as, so /// that 0.7 of 10 values is the 7th rather than the 8th, as a Double's 7.000000000000001 would make it. /// - public static object? Of(string name, IEnumerable values, object? fraction, SortDirection direction) + public static object? Of( + string name, IEnumerable values, object? fraction, SortDirection direction, + LibRed.Storage.JetTextComparer text) { if (fraction is null) return null; @@ -38,7 +40,7 @@ internal static class Percentile throw new ArgumentException($"Invalid procedure call: a percentile fraction must be from 0 to 1, not {p}."); // A stable sort, so values that compare equal but differ — 'a' and 'A' — keep their order. - var comparer = Comparer.Create(ExpressionEvaluator.CompareForSort); + var comparer = Comparer.Create((a, b) => ExpressionEvaluator.CompareForSort(a, b, text)); var present = values.Where(v => v is not null); List sorted = direction == SortDirection.Descending ? [.. present.OrderByDescending(v => v, comparer)] diff --git a/src/LibRed/LibRed.Engine/Execution/QueryExecutor.cs b/src/LibRed/LibRed.Engine/Execution/QueryExecutor.cs index a9a0d9737..2e05f27ea 100644 --- a/src/LibRed/LibRed.Engine/Execution/QueryExecutor.cs +++ b/src/LibRed/LibRed.Engine/Execution/QueryExecutor.cs @@ -22,9 +22,13 @@ internal enum ColumnOrigin { Expression, Aggregate, SetOperation } /// caller describing a query (the schema rowsets, for a view's columns) can report the declared type and its /// length rather than only the CLR type. Null for anything computed. /// What computes the column, where does not stand behind it. +/// Marks a column of Variants (CVar and what keeps one): its values keep their own types +/// until a result, a scalar subquery or a set operation writes them out as text; is that +/// text. internal readonly record struct OutputColumn( string? Qualifier, string Name, Type? ClrType = null, bool Currency = false, int? Scale = null, - bool Null = false, LibRed.Catalog.ColumnDef? Source = null, ColumnOrigin Origin = ColumnOrigin.Expression) + bool Null = false, LibRed.Catalog.ColumnDef? Source = null, ColumnOrigin Origin = ColumnOrigin.Expression, + bool Variant = false) { /// The output of a stored column. public static OutputColumn Of(string? qualifier, LibRed.Catalog.ColumnDef column) => @@ -71,6 +75,14 @@ public sealed class QueryExecutor : IScalarSubqueryRunner private readonly JetDatabase _database; private readonly ParameterBag _parameters; private readonly SessionState? _session; + private LibRed.Storage.JetTextComparer? _textComparer; + + /// How text compares in this database: in its page-0 collation, for every comparison a query makes + /// (page-02b §3.4) — never a column's own, and never the runtime's culture. + internal LibRed.Storage.JetTextComparer TextComparer => + _textComparer ??= LibRed.Storage.JetTextComparer.For(_database.Collation); + + LibRed.Storage.JetTextComparer IScalarSubqueryRunner.TextComparer => TextComparer; // Every cache below is keyed by reference identity, so IDE0028 is suppressed across the block: its only // fix is a collection expression, which would drop ReferenceEqualityComparer and key these on the AST @@ -87,6 +99,9 @@ public sealed class QueryExecutor : IScalarSubqueryRunner // for every outer row of a correlated subquery / nested-loop inner. private readonly Dictionary _projectionSchemas = new(ReferenceEqualityComparer.Instance); + // Each scalar subquery's column, keyed by AST node: described once, however often DeclaredType asks. + private readonly Dictionary _scalarSubqueryColumns = new(ReferenceEqualityComparer.Instance); + // Decorrelated EXISTS subqueries, keyed by AST node. A present-but-null value records "analysed, not // decorrelatable", so an unsound-to-rewrite subquery isn't re-analysed on every outer row. private readonly Dictionary _semiJoins = new(ReferenceEqualityComparer.Instance); @@ -139,6 +154,7 @@ public QueryExecutor( bool describing = false) { _database = database; + _parameterValues = parameters; _parameters = new ParameterBag(parameters); _session = session; _describing = describing; @@ -146,9 +162,19 @@ public QueryExecutor( private readonly bool _describing; + /// The parameter values as passed, for the describing executor a scalar subquery is typed with. + private readonly IReadOnlyDictionary? _parameterValues; + public ResultSet ExecuteQuery(PlanNode plan) { var (columns, rows) = Execute(plan, null); + // A result writes its Variants out as text, as ACE's expression service does (verified vs ACE). + if (columns.Any(c => c.Variant)) + { + List written = [.. columns.Select(c => c with { Variant = false })]; + rows = ToColumnTypes(rows, columns, written); + columns = written; + } return new ResultSet( columns.Select(c => c.Name).ToList(), rows, @@ -177,6 +203,61 @@ public ResultSet ExecuteQuery(PlanNode plan) /// so describing a query costs only the planning. Used to report a view's columns in schema metadata. internal IReadOnlyList DescribeQuery(PlanNode plan) => Execute(plan, null).Columns; + /// + /// What a join passes on of its [left.., right..] rows: every column, or — where + /// says which names the query reads — only the columns with those names, the + /// output built that narrow from the start rather than cut down afterwards. + /// + /// A join's output row is a new array for every match, so its width is paid once per joined row. + /// Built whole, a grouped join over a ten-column table carried eight columns no expression ever looked at, in + /// every row. The kept positions of each side are worked out once, here. + private sealed class JoinShape + { + private readonly int[]? _left, _right; + private readonly int _leftWidth, _rightWidth; + + /// The columns of the whole row, left side first. + /// How many of them are the left side's. + /// The names the query reads, or null to pass on every column. + public JoinShape(IReadOnlyList joined, int leftWidth, IReadOnlySet? keep) + { + _leftWidth = leftWidth; + _rightWidth = joined.Count - leftWidth; + Columns = joined; + if (keep is null || joined.All(c => keep.Contains(c.Name))) return; + + var left = new List(); + var right = new List(); + var columns = new List(); + for (int i = 0; i < joined.Count; i++) + { + if (!keep.Contains(joined[i].Name)) continue; + columns.Add(joined[i]); + if (i < leftWidth) left.Add(i); + else right.Add(i - leftWidth); + } + (_left, _right, Columns) = ([.. left], [.. right], columns); + } + + /// The columns the join passes on, in the order its rows hold them. + public IReadOnlyList Columns { get; } + + /// One output row from a row of each side; a null side stands for a row of nulls, an outer join's + /// padding. + public object?[] Combine(object?[]? left, object?[]? right) + { + if (_left is null) + return [.. left ?? new object?[_leftWidth], .. right ?? new object?[_rightWidth]]; + + var row = new object?[_left.Length + _right!.Length]; + if (left is not null) + for (int i = 0; i < _left.Length; i++) row[i] = left[_left[i]]; + if (right is not null) + for (int i = 0; i < _right.Length; i++) row[_left.Length + i] = right[_right[i]]; + return row; + } + } + /// The columns a join publishes. Joining something already collapsed — an aggregate, a DISTINCT, /// a union — makes the whole join non-updatable, as Access counts updatability, so a stored column reached /// through one no longer stands for a row anybody can write back to. @@ -211,6 +292,15 @@ public ResultSet ExecuteSystemVariableSelect(SystemVariableSelectStatement state } object? IScalarSubqueryRunner.ExecuteScalar(SqlStatement query, EvalScope outerScope) + { + object? value = ScalarValue(query, outerScope); + // A scalar subquery writes a Variant out as text (verified vs ACE: one added to itself concatenates). + return value is not null && ScalarSubqueryColumn(query) is { Variant: true } + ? ExpressionEvaluator.ConcatText(value) + : value; + } + + private object? ScalarValue(SqlStatement query, EvalScope outerScope) { if (_hoistedScalar.TryGetValue(query, out object? hoisted)) return hoisted; @@ -348,8 +438,8 @@ private bool TryHoist(SqlStatement query, EvalScope outerScope, Func Keys, HashSet NullTailKeys) BuildSemiJoinKeys( SelectStatement keyQuery, int keyWidth, IReadOnlyList nullSafe, bool trackNullTail = false) { - var keys = new HashSet(HashKeyComparer.Instance); - var nullTail = new HashSet(HashKeyComparer.Instance); + var keys = new HashSet(new HashKeyComparer(TextComparer)); + var nullTail = new HashSet(new HashKeyComparer(TextComparer)); var (_, rows) = Execute(SubqueryPlan(keyQuery, new EvalScope([], [], null)), null); foreach (object?[] row in rows) { @@ -397,7 +487,7 @@ private bool TryHoist(SqlStatement query, EvalScope outerScope, Func BuildGroupedAggregate( SelectStatement keyQuery, int keyWidth, IReadOnlyList nullSafe) { - var values = new Dictionary(HashKeyComparer.Instance); + var values = new Dictionary(new HashKeyComparer(TextComparer)); var (_, rows) = Execute(SubqueryPlan(keyQuery, new EvalScope([], [], null)), null); foreach (object?[] row in rows) { @@ -433,7 +523,7 @@ private bool TryHoist(SqlStatement query, EvalScope outerScope, Func(SqlStatement query, EvalScope outerScope, Func OutputColumn.Of(alias, c)).ToList(); - return (columns, _describing ? [] : table.Rows()); + return (columns, _describing ? [] : table.Rows(ColumnPruning.Mask(table.Definition, scan.Decode))); } case IndexSeekNode seek: @@ -568,10 +659,20 @@ private PlanNode SubqueryPlan(SqlStatement query, EvalScope outerScope) // row); a single-table seek's key is a constant/parameter. var evaluator = new ExpressionEvaluator(new EvalScope([], [], outer), this, parameters: _parameters, session: _session); var keyValues = new object?[table.Definition.Columns.Count]; - for (int i = 0; i < seek.Keys.Count; i++) - keyValues[seek.Index.Columns[i].Column.Index] = evaluator.Evaluate(seek.Keys[i]); + bool seekable = true; + for (int i = 0; i < seek.Keys.Count && seekable; i++) + { + var keyColumn = seek.Index.Columns[i].Column; + seekable = ExpressionEvaluator.TryGetSeekKey( + keyColumn, seek.Keys[i], evaluator.Evaluate(seek.Keys[i]), out keyValues[keyColumn.Index]); + } - return (columns, table.SeekRows(seek.Index, keyValues)); + // A value of another kind is not a key this index can be searched by, so the rows are found + // by scanning instead. The FilterNode this seek was planned under is kept whatever happens + // (see IndexSelection), so it re-checks every row either way and the answer is the same one + // the table would give with no index on it at all. + bool[]? decode = ColumnPruning.Mask(table.Definition, seek.Decode); + return (columns, seekable ? table.SeekRows(seek.Index, keyValues, decode) : table.Rows(decode)); } case IndexRangeSeekNode range: @@ -590,20 +691,32 @@ private PlanNode SubqueryPlan(SqlStatement query, EvalScope outerScope) v[col] = evaluator.Evaluate(e); return v; } - return (columns, table.SeekRangeRows(range.Index, Bound(range.Low), Bound(range.High))); + return (columns, table.SeekRangeRows(range.Index, Bound(range.Low), Bound(range.High), + ColumnPruning.Mask(table.Definition, range.Decode))); } case DerivedTableNode derived: { var (inner, rows) = Execute(derived.Input, outer); - var columns = inner.Select(c => c with { Qualifier = derived.Alias }).ToList(); - return (columns, rows); + return (DerivedColumns(inner, derived.Alias, derived.Columns), rows); } case FilterNode filter: { var (columns, rows) = Execute(filter.Input, outer); - return (columns, rows.Where(row => Eval(columns, row, outer).IsTrue(filter.Predicate))); + return (columns, FilteredRows()); + + // One scope/evaluator per enumeration, rebound per row, as in ExecuteJoin. Created inside the + // iterator so two enumerations of the result cannot move each other's row. + IEnumerable FilteredRows() + { + ExpressionEvaluator eval = Eval(columns, [], outer); + foreach (object?[] row in rows) + { + if (eval.Rebind(row).IsTrue(filter.Predicate)) + yield return row; + } + } } // A lateral join re-runs its right side per left row, so it cannot go through ExecuteJoin (which @@ -631,7 +744,7 @@ private PlanNode SubqueryPlan(SqlStatement query, EvalScope outerScope) int? bound = sort.Limit is { } lim && !_describing ? Convert.ToInt32(Eval([], [], outer).Evaluate(lim), System.Globalization.CultureInfo.InvariantCulture) : null; - return (columns, SortRows(sort.Keys, columns, outer, rows, bound)); + return (columns, SortRows(sort.Keys, columns, outer, rows, bound, _describing ? null : sort.Ties)); } case ProjectNode project: @@ -643,16 +756,30 @@ private PlanNode SubqueryPlan(SqlStatement query, EvalScope outerScope) ProjectionSchema schema = ProjectionSchemaFor(project, columns); var plan = schema.Plan; - var projected = rows.Select(row => + return (schema.Columns, ProjectedRows()); + + // One scope/evaluator per enumeration, rebound per row, as in FilteredRows above. + IEnumerable ProjectedRows() { - var eval = Eval(columns, row, outer); - return plan.Select(p => p.InputIndex >= 0 - ? row[p.InputIndex] - : ExpressionEvaluator.ToResultPlaces( - ExpressionEvaluator.AsColumnType(eval.Evaluate(p.Expr!), p.ConvertTo, currency: false), p.Type)).ToArray(); - }); - - return (schema.Columns, projected); + ExpressionEvaluator eval = Eval(columns, [], outer); + foreach (object?[] row in rows) + { + eval.Rebind(row); + var values = new object?[plan.Count]; + for (int i = 0; i < plan.Count; i++) + { + var p = plan[i]; + values[i] = p.InputIndex >= 0 + ? row[p.InputIndex] + : ExpressionEvaluator.ToResultPlaces( + ExpressionEvaluator.AsColumnType( + p.Slot >= 0 ? row[p.Slot] : eval.Evaluate(p.Expr!), p.ConvertTo, currency: false), + p.Type); + } + + yield return values; + } + } } case SetOperationNode setOp: @@ -751,14 +878,15 @@ private PlanNode SubqueryPlan(SqlStatement query, EvalScope outerScope) /// A set operation's column: the left query's, typed as ACE types it (verified vs ACE). A bare NULL takes /// the other query's type. Numbers widen on , a Boolean counting as an Integer /// (-1); a GUID or binary value with anything else makes a binary column; any other mix — text, or a date with a - /// number or a Boolean — makes a text column. An unknown type on either side leaves the column untyped. + /// number or a Boolean — makes a text column. An unknown type on either side leaves the column untyped. A column + /// of Variants counts as text, and the set operation writes its values out as text (verified vs ACE). /// private static OutputColumn SetOperationColumn(OutputColumn left, OutputColumn right) { // Whatever the arms hold, the combined column is no longer any one stored column, so it carries no // source: a caller describing the query sees a computed column, which is what it is. if (right.Null) - return left with { Source = null, Origin = ColumnOrigin.SetOperation }; + return left with { Source = null, Origin = ColumnOrigin.SetOperation, Variant = false }; if (left.Null) return left with { @@ -768,6 +896,7 @@ private static OutputColumn SetOperationColumn(OutputColumn left, OutputColumn r Null = false, Source = null, Origin = ColumnOrigin.SetOperation, + Variant = false, }; // A Decimal column is Currency unless one side is a Decimal of its own, and keeps a scale both sides share. @@ -787,18 +916,20 @@ private static OutputColumn SetOperationColumn(OutputColumn left, OutputColumn r : leftDecimal ? left.Scale : rightDecimal ? right.Scale : null, Source = null, Origin = ColumnOrigin.SetOperation, + Variant = false, }; - - static Type AsInteger(Type type) => type == typeof(bool) ? typeof(short) : type; } + /// A Boolean as the Integer (-1 or 0) it counts as beside a number; any other type as it is. + private static Type AsInteger(Type type) => type == typeof(bool) ? typeof(short) : type; + /// The rows with each value converted to its output column's type, where the query's own column had - /// another. + /// another — or held Variants, whose values are still their own types until written out. private static IEnumerable ToColumnTypes( IEnumerable rows, IReadOnlyList from, List to) { int[] changed = Enumerable.Range(0, to.Count) - .Where(i => to[i].ClrType is { } type && !from[i].Null && from[i].ClrType != type) + .Where(i => to[i].ClrType is { } type && !from[i].Null && (from[i].ClrType != type || from[i].Variant)) .ToArray(); if (changed.Length == 0) return rows; @@ -811,7 +942,7 @@ private static OutputColumn SetOperationColumn(OutputColumn left, OutputColumn r }); } - private static IEnumerable ExecuteSetOp(SetOperator op, IEnumerable left, IEnumerable right) + private IEnumerable ExecuteSetOp(SetOperator op, IEnumerable left, IEnumerable right) { switch (op) { @@ -821,13 +952,13 @@ private static OutputColumn SetOperationColumn(OutputColumn left, OutputColumn r return Distinct(left.Concat(right)); case SetOperator.Intersect: { - var keep = new HashSet(right.Select(r => new GroupKey(r))); - return Distinct(left).Where(r => keep.Contains(new GroupKey(r))); + var keep = new HashSet(right.Select(r => new GroupKey(r, TextComparer))); + return Distinct(left).Where(r => keep.Contains(new GroupKey(r, TextComparer))); } case SetOperator.Except: { - var remove = new HashSet(right.Select(r => new GroupKey(r))); - return Distinct(left).Where(r => !remove.Contains(new GroupKey(r))); + var remove = new HashSet(right.Select(r => new GroupKey(r, TextComparer))); + return Distinct(left).Where(r => !remove.Contains(new GroupKey(r, TextComparer))); } default: throw new NotSupportedException($"Set operator {op} is not supported."); @@ -835,24 +966,24 @@ private static OutputColumn SetOperationColumn(OutputColumn left, OutputColumn r } /// Yields rows with duplicates removed by structural (value-wise) equality. - private static IEnumerable Distinct(IEnumerable rows) + private IEnumerable Distinct(IEnumerable rows) { var seen = new HashSet(); foreach (object?[] row in rows) - if (seen.Add(new GroupKey(row))) + if (seen.Add(new GroupKey(row, TextComparer))) yield return row; } /// Yields the first row for each distinct combination of the values at /// (the columns of the DISTINCTROW contributing tables), preserving order. - private static IEnumerable DistinctByIndexes(IEnumerable rows, int[] indexes) + private IEnumerable DistinctByIndexes(IEnumerable rows, int[] indexes) { var seen = new HashSet(); foreach (object?[] row in rows) { var key = new object?[indexes.Length]; for (int i = 0; i < indexes.Length; i++) key[i] = row[indexes[i]]; - if (seen.Add(new GroupKey(key))) + if (seen.Add(new GroupKey(key, TextComparer))) yield return row; } } @@ -863,8 +994,11 @@ private static OutputColumn SetOperationColumn(OutputColumn left, OutputColumn r /// A ProjectNode's flattened output: the per-item plan (source input index, or an expression to /// evaluate, with the type its values are converted to, if any) and the resulting output columns. /// Structural — the same for every outer row. + /// Slot is where an item that is just a column of the input finds its value, so a row reads + /// it rather than resolving the name again; it still takes the item's conversions, which a passed-through + /// InputIndex column does not. -1 sends the item through the evaluator. private sealed record ProjectionSchema( - List<(OutputColumn Column, int InputIndex, Expression? Expr, NumberType Type, Type? ConvertTo)> Plan, + List<(OutputColumn Column, int InputIndex, Expression? Expr, NumberType Type, Type? ConvertTo, int Slot)> Plan, List Columns); /// Builds — or reuses — a ProjectNode's schema. Flattens the projection, expanding a qualified star @@ -875,14 +1009,14 @@ private ProjectionSchema ProjectionSchemaFor(ProjectNode project, IReadOnlyList< if (_projectionSchemas.TryGetValue(project, out ProjectionSchema? cached)) return cached; - var plan = new List<(OutputColumn Column, int InputIndex, Expression? Expr, NumberType Type, Type? ConvertTo)>(); + var plan = new List<(OutputColumn Column, int InputIndex, Expression? Expr, NumberType Type, Type? ConvertTo, int Slot)>(); foreach (SelectItem item in project.Projection) { if (item.Value is QualifiedStarExpression star) { for (int ci = 0; ci < columns.Count; ci++) if (string.Equals(columns[ci].Qualifier, star.Table, StringComparison.OrdinalIgnoreCase)) - plan.Add((columns[ci], ci, null, default, null)); + plan.Add((columns[ci], ci, null, default, null, -1)); } else { @@ -894,11 +1028,15 @@ private ProjectionSchema ProjectionSchemaFor(ProjectNode project, IReadOnlyList< OutputColumn? referenced = item.Value is ColumnReference reference ? OutputColumn.Find(columns, reference) : null; + (Type? convertTo, bool variant) = ItemConversion(item.Value, columns); plan.Add(( OutputColumn.Computed(name, DeclaredType(item.Value, columns), type, item.Value, referenced?.Source) with - { Origin = referenced?.Origin ?? ColumnOrigin.Expression }, - -1, item.Value, type, ChoiceConversion(item.Value, columns))); + { Origin = referenced?.Origin ?? ColumnOrigin.Expression, Variant = variant }, + -1, item.Value, type, convertTo, + // The evaluator resolves this scope's columns before any outer one, so a name that is one + // column here is that column on every row. None, or more than one, stays with the evaluator. + item.Value is ColumnReference own ? EvalScope.Locate(columns, own, throwIfAmbiguous: false) : -1)); } } @@ -909,6 +1047,10 @@ private ProjectionSchema ProjectionSchemaFor(ProjectNode project, IReadOnlyList< private Type? DeclaredType(Expression expression, IReadOnlyList columns) { + // A Variant or a Mixed value is text wherever it is written out. + if (VarianceOf(expression, columns) != Variance.None) + return typeof(string); + switch (expression) { case LiteralExpression { Value: { } value }: @@ -929,20 +1071,30 @@ private ProjectionSchema ProjectionSchemaFor(ProjectNode project, IReadOnlyList< ? typeof(int) : _session?.LastIdentity?.GetType(); case ExistsExpression or InSubqueryExpression or InListExpression or BetweenExpression: return typeof(bool); + case ScalarSubquery subquery: + return ScalarSubqueryColumn(subquery.Query)?.ClrType; case UnaryExpression unary: return unary.Operator is UnaryOperator.Not or UnaryOperator.IsNull or UnaryOperator.IsNotNull - ? typeof(bool) : DeclaredUnaryType(unary.Operator, DeclaredType(unary.Operand, columns)); + or UnaryOperator.IsTrue or UnaryOperator.IsNotTrue or UnaryOperator.IsFalse or UnaryOperator.IsNotFalse + ? typeof(bool) : DeclaredUnaryType(unary.Operator, OperandType(unary.Operand, columns)); case BinaryExpression binary: if (binary.Operator is BinaryOperator.Equal or BinaryOperator.NotEqual or BinaryOperator.LessThan or BinaryOperator.LessThanOrEqual or BinaryOperator.GreaterThan or BinaryOperator.GreaterThanOrEqual or BinaryOperator.And or BinaryOperator.Or or BinaryOperator.Xor or BinaryOperator.Eqv - or BinaryOperator.Imp or BinaryOperator.Like or BinaryOperator.In) + or BinaryOperator.Imp or BinaryOperator.Like or BinaryOperator.In + or BinaryOperator.IsDistinctFrom or BinaryOperator.IsNotDistinctFrom) return typeof(bool); if (binary.Operator == BinaryOperator.Concat) return typeof(string); - Type? left = DeclaredType(binary.Left, columns); - Type? right = DeclaredType(binary.Right, columns); + // A Mixed value beside text may concatenate, so the two add as text (verified vs ACE: '8' and '8' are + // '88', and 2 and '9' are '11'). + if (binary.Operator == BinaryOperator.Add + && (VarianceOf(binary.Left, columns) == Variance.Mixed && IsPlainText(binary.Right, columns) + || VarianceOf(binary.Right, columns) == Variance.Mixed && IsPlainText(binary.Left, columns))) + return typeof(string); + Type? left = OperandType(binary.Left, columns); + Type? right = OperandType(binary.Right, columns); // A date plus or less a span is a date — ExpressionEvaluator.IsSpan, which the evaluator moves it by. if (left == typeof(DateTime) && IsSpan(binary.Right) && binary.Operator is BinaryOperator.Add or BinaryOperator.Subtract @@ -958,14 +1110,148 @@ or BinaryOperator.And or BinaryOperator.Or or BinaryOperator.Xor or BinaryOperat } } + /// + /// What ACE's expression service makes of a value beyond its type (verified vs ACE, over OLE DB, for IIF, SWITCH + /// and CHOOSE; CASE, COALESCE, GREATEST and LEAST, which ACE does not have, take IIF's rule). + /// A Variant — CVar(x), Access's Nz (verified vs Access), a Variant plus a Variant, a + /// Mixed value or text, a negated one, a column or scalar subquery holding them, and a choice whose values all are — keeps its own type through an + /// expression and through a derived table, so X + X over CVar(B) AS X still adds. A result, a + /// scalar subquery and a set operation write it out as text, as CStr writes it. + /// A Mixed value is a choice whose values disagree in kind — text beside anything else, or a Variant + /// beside anything that is not one — or a negated one. It is text wherever it is written out, a derived table + /// included, so X + X over IIF(…, T, 2) AS X concatenates. + /// Both sort, group and take Min, Max, First and Last as their text, so 10 sorts before 3. As the operand of + /// anything else — arithmetic, a function, an aggregate, another choice — each counts as a Double. + /// + private enum Variance { None, Variant, Mixed } + + private Variance VarianceOf(Expression expression, IReadOnlyList columns) + { + switch (expression) + { + // CVar(Null) is left untyped, as a bare Null is. ACE makes it a Variant, and a union with an arm of them a + // text column; EFCore.Jet writes it for every projected Null, and LibRed keeps such a union typed from its + // other arm. + case FunctionCall { Arguments: [var argument] } function + when function.Name.TrimEnd('$').Equals("CVAR", StringComparison.OrdinalIgnoreCase): + return argument is LiteralExpression { Value: null } ? Variance.None : Variance.Variant; + // Access's Nz always gives a Variant, whatever it is given (verified vs Access): Nz(K, 0) writes "0". + case FunctionCall function when function.Name.TrimEnd('$').Equals("NZ", StringComparison.OrdinalIgnoreCase): + return Variance.Variant; + case ColumnReference reference: + return OutputColumn.Find(columns, reference) is { Variant: true } ? Variance.Variant : Variance.None; + case ScalarSubquery subquery: + return ScalarSubqueryColumn(subquery.Query) is { Variant: true } ? Variance.Variant : Variance.None; + case UnaryExpression { Operator: UnaryOperator.Negate } negation: + return VarianceOf(negation.Operand, columns); + // Only + keeps a Variant, and only beside something that may concatenate; the other operators, and + + // beside a number, make a number. + case BinaryExpression { Operator: BinaryOperator.Add } add: + { + Variance left = VarianceOf(add.Left, columns), right = VarianceOf(add.Right, columns); + if (left != Variance.Variant && right != Variance.Variant) + return Variance.None; + (Variance other, Expression otherSide) = left == Variance.Variant ? (right, add.Right) : (left, add.Left); + return other != Variance.None || DeclaredType(otherSide, columns) == typeof(string) + ? Variance.Variant : Variance.None; + } + case CaseExpression @case: + return ChoiceVariance(CaseResults(@case), columns); + case FunctionCall function when ChoiceArms(function) is { } arms: + return ChoiceVariance(arms, columns); + default: + return Variance.None; + } + } + + /// A choice's variance: a Variant when every value is one, Mixed when only some are, or when text stands + /// beside another known kind. A bare Null is no value, and a Mixed value counts as the Double it is as an operand + /// — so a choice between one and a number is a number. + private Variance ChoiceVariance(IEnumerable arms, IReadOnlyList columns) + { + List values = [.. arms.Where(a => a is not LiteralExpression { Value: null })]; + if (values.Count == 0) + return Variance.None; + List variances = [.. values.Select(a => VarianceOf(a, columns))]; + if (variances.TrueForAll(v => v == Variance.Variant)) + return Variance.Variant; + if (variances.Contains(Variance.Variant)) + return Variance.Mixed; + List types = [.. values.Select((a, i) => variances[i] == Variance.Mixed ? typeof(double) : DeclaredType(a, columns))]; + return types.TrueForAll(t => t is not null) && types.Contains(typeof(string)) && types.Exists(t => t != typeof(string)) + ? Variance.Mixed : Variance.None; + } + + /// The values a choice picks among, or null for a function that is not one. + private static IEnumerable? ChoiceArms(FunctionCall function) => + function.Name.TrimEnd('$').ToUpperInvariant() switch + { + "IIF" when function.Arguments.Count == 3 => function.Arguments.Skip(1), + "SWITCH" => function.Arguments.Where((_, i) => i % 2 == 1), + "CHOOSE" => function.Arguments.Skip(1), + "COALESCE" or "GREATEST" or "LEAST" => function.Arguments, + _ => null, + }; + + /// The type an expression counts as where it is an operand: a Variant or a Mixed value as a Double. + private Type? OperandType(Expression expression, IReadOnlyList columns) => + VarianceOf(expression, columns) != Variance.None ? typeof(double) : DeclaredType(expression, columns); + + /// Whether is text that is neither a Variant nor a Mixed value. + private bool IsPlainText(Expression expression, IReadOnlyList columns) => + VarianceOf(expression, columns) == Variance.None && DeclaredType(expression, columns) == typeof(string); + + /// Whether a choice, a Variant or a Mixed value appears anywhere in — each + /// of which can hold a value of another type than the one it declares. + private bool HoldsChoiceOrVariance(Expression expression, IReadOnlyList columns) => + expression is CaseExpression + || expression is FunctionCall function && ChoiceArms(function) is not null + || VarianceOf(expression, columns) != Variance.None + || (expression.Operands()?.Any(o => HoldsChoiceOrVariance(o, columns)) ?? false); + + /// + /// What a projected value is converted to, and whether its column holds Variants. A Variant is left as it is and + /// written out later. Anything else holding a choice, a Variant or a Mixed value becomes the type it declares, + /// which its values need not already have — a Mixed value becomes text, a Variant date plus 1 a Double, and + /// IIF(…, D, 2) + 1 a date where the choice picks the 2. A choice with an arm of unknown type declares + /// nothing, so nothing is converted. + /// + private (Type? ConvertTo, bool Variant) ItemConversion(Expression expression, IReadOnlyList columns) => + VarianceOf(expression, columns) == Variance.Variant + ? (null, true) + : (HoldsChoiceOrVariance(expression, columns) ? DeclaredType(expression, columns) : null, false); + + /// Which of compare as text: a Variant or a Mixed value sorts, groups and takes + /// Min, Max, First and Last as its text. + private bool[] ComparedAsText(IEnumerable keys, IReadOnlyList columns) => + [.. keys.Select(k => VarianceOf(k, columns) != Variance.None)]; + + /// A value as it compares: its text, where . + private static object? Compared(object? value, bool asText) => + asText && value is not null ? ExpressionEvaluator.ConcatText(value) : value; + + /// A scalar subquery's one column, found by describing its plan: the type it declares, and whether it + /// holds Variants. Describing reads no row and evaluates nothing, so a correlated subquery never needs the outer + /// row it cannot have here; a column only that row could type is left untyped. Its values already come back in + /// this type — each is the subquery's own projected value, and a Variant is written out as text. + private OutputColumn? ScalarSubqueryColumn(SqlStatement query) + { + if (!_scalarSubqueryColumns.TryGetValue(query, out OutputColumn? column)) + _scalarSubqueryColumns[query] = column = + new QueryExecutor(_database, _parameterValues, _session, describing: true) + .DescribeQuery(QueryPlanner.PlanStatement(query)) is [var first, ..] ? first : null; + return column; + } + /// /// The declared type of a CASE — the standard's "highest precedence type from the set of types in - /// result_expressions and the optional else_result_expression". A branch whose own type is unknown - /// contributes nothing rather than poisoning the answer, which is what makes a bare NULL arm - /// harmless: a NULL literal has no type and the standard ignores it for precedence too. Numeric branches - /// widen on , so THEN 1 ELSE 2.5 declares Double. A genuine mix - /// (a string branch and a numeric one) declares nothing rather than guessing, leaving the column untyped - /// exactly as it was before CASE was understood at all. + /// result_expressions and the optional else_result_expression". A bare NULL arm contributes nothing + /// rather than poisoning the answer: a NULL literal has no type and the standard ignores it for precedence + /// too. Any other arm whose type is unknown leaves the whole choice unknown, since it may hold anything — + /// letting the known arms declare alone made IIF(x IS NULL, 0, x) over an untyped Decimal + /// x declare Integer, and the value was then converted to it. Numeric branches + /// widen on , so THEN 1 ELSE 2.5 declares Double. Text beside another kind + /// is Mixed () and declares text, as IIF does in ACE. /// private Type? DeclaredCaseType(CaseExpression @case, IReadOnlyList columns) => UnifiedType(CaseResults(@case), columns); @@ -991,9 +1277,27 @@ private static IEnumerable CaseResults(CaseExpression c) foreach (Expression alternative in alternatives) { - Type? branchType = IsWrittenDecimal(alternative) ? typeof(decimal) : DeclaredType(alternative, columns); + Type? branchType = IsWrittenDecimal(alternative) ? typeof(decimal) : OperandType(alternative, columns); if (branchType is null) - continue; + { + if (alternative is LiteralExpression { Value: null }) + continue; + return null; + } + // A date beside a number is a date, the number read as a serial; a Boolean beside one counts as the + // Integer -1 or 0 (verified vs ACE: IIF(…, D, 2) is 1900-01-01 where it picks the 2). + if (result is not null && result != branchType) + { + if (result == typeof(DateTime) && IsNumeric(AsInteger(branchType)) + || branchType == typeof(DateTime) && IsNumeric(AsInteger(result))) + { + result = typeof(DateTime); + currency = false; + continue; + } + result = AsInteger(result); + branchType = AsInteger(branchType); + } bool branchCurrency = branchType == typeof(decimal) && ExpressionEvaluator.NumberTypeOf(alternative, columns, e => DeclaredType(e, columns)).Class == NumberClass.Currency; @@ -1023,30 +1327,6 @@ private static bool IsWrittenDecimal(Expression expression) return expression is LiteralExpression { Written: decimal }; } - /// - /// The type an expression that picks one of several alternatives () converts its value - /// to, so the value has the type the column declares; null when nothing is converted. Only when every - /// alternative's type is known is the declared type sure to hold each of them. - /// - private Type? ChoiceConversion(Expression expression, IReadOnlyList columns) - { - IEnumerable? alternatives = expression switch - { - CaseExpression @case => CaseResults(@case), - FunctionCall function => function.Name.TrimEnd('$').ToUpperInvariant() switch - { - "IIF" when function.Arguments.Count == 3 => function.Arguments.Skip(1), - "COALESCE" or "GREATEST" or "LEAST" => function.Arguments, - _ => null, - }, - _ => null, - }; - if (alternatives is null - || alternatives.Any(a => a is not LiteralExpression { Value: null } && DeclaredType(a, columns) is null)) - return null; - return DeclaredType(expression, columns); - } - /// /// The type the values of two numeric types share (verified vs ACE, as it types a UNION): the wider whole /// number of the two; a Single with a Byte or an Integer, and a Double for a Single with anything wider; a @@ -1103,14 +1383,23 @@ var pair when RunningAggregate.IsPair(pair) => typeof(double), : typeof(int), "AVG" => argument == typeof(decimal) ? typeof(decimal) : typeof(double), "VAR" or "VARP" or "STDEV" or "STDEVP" or "STDDEV" or "STDDEVP" => typeof(double), - _ => argument, // MIN, MAX, FIRST and LAST keep the argument's type + // MIN, MAX, FIRST and LAST keep the argument's type. Everything else needs a case above: a new + // aggregate that reached a `_ => argument` default would be declared as whatever it was fed, which is + // right for these four and wrong for any statistic. Null says "not typed" instead, and + // AggregateSurfaceTests holds every name RunningAggregate computes to having one. + "MIN" or "MAX" or "FIRST" or "LAST" => argument, + _ => null, }; private Type? DeclaredFunctionType(FunctionCall function, IReadOnlyList columns) { string name = function.Name.TrimEnd('$').ToUpperInvariant(); Type? argument = function.Arguments.Count == 0 ? null - : DeclaredType(function.Arguments[function.WithinGroup is null ? 0 : ^1], columns); + : OperandType(function.Arguments[function.WithinGroup is null ? 0 : ^1], columns); + // Min, Max, First and Last take a Variant's or a Mixed value's text, and give it (verified vs ACE). + if (name is "MIN" or "MAX" or "FIRST" or "LAST" && function.Arguments.Count == 1 + && VarianceOf(function.Arguments[0], columns) != Variance.None) + return typeof(string); if (QueryPlanner.IsAggregate(name)) return AggregateResultType(name, argument); return name switch @@ -1145,6 +1434,8 @@ var pair when RunningAggregate.IsPair(pair) => typeof(double), // IIF chooses between two values as CASE does, so it takes CASE's rule rather than ACE's own (which // makes every whole number a Long and lets Currency beat Double). "IIF" when function.Arguments.Count == 3 => UnifiedType(function.Arguments.Skip(1), columns), + // SWITCH and CHOOSE pick one of their values as IIF does, and take its rule. + "SWITCH" or "CHOOSE" => UnifiedType(ChoiceArms(function)!, columns), // The standard makes COALESCE shorthand for a CASE over its arguments, so it takes the same rule: // the highest-precedence type among them. Unified the same way, which also means a bare NULL // argument contributes no type rather than erasing the others. @@ -1339,8 +1630,10 @@ private static HashSet ContributingQualifiers( string innerAlias = seek.Alias ?? seek.Table; int innerWidth = innerTable.Definition.Columns.Count; var seekColumns = innerTable.Definition.Columns.Select(c => OutputColumn.Of(innerAlias, c)).ToList(); - var joinColumns = leftColumns.Concat(seekColumns).ToList(); + var seekShape = new JoinShape(leftColumns.Concat(seekColumns).ToList(), leftColumns.Count, join.Keep); + IReadOnlyList joinColumns = seekShape.Columns; int[] keyCols = seek.Index.Columns.Select(c => c.Column.Index).ToArray(); + bool[]? decode = ColumnPruning.Mask(innerTable.Definition, seek.Decode); IEnumerable SeekRows() { @@ -1356,13 +1649,21 @@ private static HashSet ContributingQualifiers( foreach (object?[] left in leftRows) { keyScope.Rebind(left); - for (int i = 0; i < seek.Keys.Count; i++) - keyValues[keyCols[i]] = keyEval.Evaluate(seek.Keys[i]); + // As the single-table seek above: a key of another kind cannot be looked up in this index, + // and the join's ON is kept whole as the residual, so scanning the inner table for that + // outer row gives the same rows the seek would have had to find. + bool seekable = true; + for (int i = 0; i < seek.Keys.Count && seekable; i++) + seekable = ExpressionEvaluator.TryGetSeekKey( + seek.Index.Columns[i].Column, seek.Keys[i], keyEval.Evaluate(seek.Keys[i]), + out keyValues[keyCols[i]]); bool matched = false; - foreach (object?[] right in innerTable.SeekRows(seek.Index, keyValues)) + foreach (object?[] right in seekable + ? innerTable.SeekRows(seek.Index, keyValues, decode) + : innerTable.Rows(decode)) { - object?[] combined = [.. left, .. right]; + object?[] combined = seekShape.Combine(left, right); if (on is null || onEval.Rebind(combined).IsTrue(on)) { matched = true; @@ -1370,7 +1671,7 @@ private static HashSet ContributingQualifiers( } } if (leftOuter && !matched) - yield return [.. left, .. new object?[innerWidth]]; + yield return seekShape.Combine(left, null); } } @@ -1379,7 +1680,8 @@ private static HashSet ContributingQualifiers( var (rightColumns, rightRowsEnum) = Execute(join.Right, outer); - var columns = JoinedColumns(leftColumns, rightColumns); + var shape = new JoinShape(JoinedColumns(leftColumns, rightColumns), leftColumns.Count, join.Keep); + IReadOnlyList columns = shape.Columns; var rightRows = rightRowsEnum.ToList(); // re-iterated per left row if (on is null && join.Kind != JoinKind.Cross) throw new NotSupportedException("Joins require an ON condition."); @@ -1420,7 +1722,7 @@ private static HashSet ContributingQualifiers( bool matched = false; for (int r = 0; r < rightRows.Count; r++) { - object?[] combined = [.. left, .. rightRows[r]]; + object?[] combined = shape.Combine(left, rightRows[r]); if (rowOn is null || onEval.Rebind(combined).IsTrue(rowOn)) { matched = true; @@ -1430,13 +1732,13 @@ private static HashSet ContributingQualifiers( } if (leftOuter && !matched) - yield return [.. left, .. new object?[rightColumns.Count]]; + yield return shape.Combine(left, null); } // Right-preserving tail: every right row no left row matched, null-padded on the left. for (int r = 0; rightMatched is not null && r < rightMatched.Length; r++) if (!rightMatched[r]) - yield return [.. new object?[leftColumns.Count], .. rightRows[r]]; + yield return shape.Combine(null, rightRows[r]); } return (columns, Rows()); @@ -1446,8 +1748,8 @@ private static HashSet ContributingQualifiers( { var (leftColumns, leftRows) = Execute(join.Left, outer); var (rightColumns, rightRowsEnum) = Execute(join.Right, outer); - var joinColumns = JoinedColumns(leftColumns, rightColumns); - int leftWidth = leftColumns.Count, rightWidth = rightColumns.Count; + var shape = new JoinShape(JoinedColumns(leftColumns, rightColumns), leftColumns.Count, join.Keep); + IReadOnlyList joinColumns = shape.Columns; Expression on = join.On; // INNER/LEFT build the right side and probe with the left; RIGHT builds the left and probes with the @@ -1469,7 +1771,7 @@ private static HashSet ContributingQualifiers( { // Build phase: hash the build side by its keys. A row with any null key can never satisfy an // equi-join (SQL null = null is not true), so it is dropped from the table. - var table = new Dictionary>(HashKeyComparer.Instance); + var table = new Dictionary>(new HashKeyComparer(TextComparer)); var buildScope = new EvalScope(buildColumns, [], outer); var buildEval = new ExpressionEvaluator(buildScope, this, parameters: _parameters, session: _session); // A null-key build row can never match, so it is normally dropped outright. Under FULL the build @@ -1509,7 +1811,7 @@ private static HashSet ContributingQualifiers( { object?[] left = buildRight ? p : b; object?[] right = buildRight ? b : p; - object?[] combined = [.. left, .. right]; + object?[] combined = shape.Combine(left, right); if (onEval.Rebind(combined).IsTrue(on)) { matched = true; @@ -1520,8 +1822,8 @@ private static HashSet ContributingQualifiers( } if (preserveProbe && !matched) yield return buildRight - ? [.. p, .. new object?[rightWidth]] // probe is the left side; right is null - : [.. new object?[leftWidth], .. p]; // probe is the right side; left is null + ? shape.Combine(p, null) // probe is the left side; right is null + : shape.Combine(null, p); // probe is the right side; left is null } // FULL only: the build side is preserved as well, so every build row the probe never matched is @@ -1530,8 +1832,8 @@ private static HashSet ContributingQualifiers( if (preserveBuild) { object?[] Unmatched(object?[] b) => buildRight - ? [.. new object?[leftWidth], .. b] // build is the right side; left is null - : [.. b, .. new object?[rightWidth]]; // build is the left side; right is null + ? shape.Combine(null, b) // build is the right side; left is null + : shape.Combine(b, null); // build is the left side; right is null foreach (List bucket in table.Values) foreach (object?[] b in bucket) @@ -1569,10 +1871,8 @@ private sealed class RowIdentityComparer : IEqualityComparer /// Hash/equality over a composite join key that mirrors the evaluator's = within a type kind /// (the planner only builds a hash join over same-kind key columns). Key elements are never null. - private sealed class HashKeyComparer : IEqualityComparer + private sealed class HashKeyComparer(LibRed.Storage.JetTextComparer text) : IEqualityComparer { - public static readonly HashKeyComparer Instance = new(); - // A null element equals only a null element. KeyEqual/KeyHash are documented for non-null keys, and for a // plain `=` correlation no null ever reaches here (such rows are dropped from the build and short-circuit // on probe). A null-safe correlation — EF's `a = b OR (a IS NULL AND b IS NULL)` — does hash nulls, so the @@ -1589,7 +1889,7 @@ public bool Equals(object?[]? a, object?[]? b) continue; } - if (!ExpressionEvaluator.KeyEqual(a[i]!, b[i]!)) return false; + if (!ExpressionEvaluator.KeyEqual(a[i]!, b[i]!, text)) return false; } return true; } @@ -1597,7 +1897,7 @@ public bool Equals(object?[]? a, object?[]? b) public int GetHashCode(object?[] a) { var h = new HashCode(); - foreach (object? v in a) h.Add(v is null ? 0 : ExpressionEvaluator.KeyHash(v)); + foreach (object? v in a) h.Add(v is null ? 0 : ExpressionEvaluator.KeyHash(v, text)); return h.ToHashCode(); } } @@ -1638,19 +1938,22 @@ public int GetHashCode(object?[] a) IReadOnlyList columns, EvalScope? outer, IEnumerable rows, - int? bound = null) + int? bound = null, + TieCut? ties = null) { // Rows paired with their evaluated keys and their input position. The position makes the ordering TOTAL, // which is what lets the bounded path below be stable without relying on a stable algorithm. var decorated = new List<(object?[] Row, object?[] Keys, int Index)>(); var index = 0; + bool[] byText = ComparedAsText(keys.Select(k => k.Value), columns); + ExpressionEvaluator eval = Eval(columns, [], outer); // one evaluator, rebound per row foreach (object?[] row in rows) { - ExpressionEvaluator eval = Eval(columns, row, outer); + eval.Rebind(row); var rowKeys = new object?[keys.Count]; for (var i = 0; i < keys.Count; i++) { - rowKeys[i] = eval.Evaluate(keys[i].Value); + rowKeys[i] = ExpressionEvaluator.SortKey(Compared(eval.Evaluate(keys[i].Value), byText[i]), TextComparer); } decorated.Add((row, rowKeys, index++)); @@ -1658,7 +1961,7 @@ public int GetHashCode(object?[] a) { // Keep only the best `max` so far. Doing this in batches (rather than per row) amortises the sort // over the rows it discards, so the list never grows past 2·max however large the input is. - Trim(decorated, keys, max); + Trim(decorated, keys, max, TextComparer); } } @@ -1672,33 +1975,69 @@ public int GetHashCode(object?[] a) { decorated.RemoveRange(take, decorated.Count - take); } + if (ties is not null) + decorated = CutWithTies(decorated, x => x.Keys, keys, ties, outer); return decorated.Select(x => x.Row).ToList(); int Compare((object?[] Row, object?[] Keys, int Index) a, (object?[] Row, object?[] Keys, int Index) b) { - int c = CompareEvaluatedKeys(keys, a.Keys, b.Keys); + int c = CompareEvaluatedKeys(keys, a.Keys, b.Keys, TextComparer); // Ties fall back to input order, reproducing a stable sort's result exactly. return c != 0 ? c : a.Index.CompareTo(b.Index); } - static void Trim(List<(object?[] Row, object?[] Keys, int Index)> list, IReadOnlyList keys, int max) + static void Trim( + List<(object?[] Row, object?[] Keys, int Index)> list, IReadOnlyList keys, int max, + LibRed.Storage.JetTextComparer text) { list.Sort((a, b) => { - int c = CompareEvaluatedKeys(keys, a.Keys, b.Keys); + int c = CompareEvaluatedKeys(keys, a.Keys, b.Keys, text); return c != 0 ? c : a.Index.CompareTo(b.Index); }); list.RemoveRange(max, list.Count - max); } } - /// Compares two rows' already-evaluated ORDER BY key values, honouring each key's direction. - private static int CompareEvaluatedKeys(IReadOnlyList keys, object?[] a, object?[] b) + /// What a WITH TIES cut keeps of : the offset skipped, the count taken — as TOP n + /// PERCENT takes one, ceil(rows × n / 100), when it is a percentage — and then every further row whose ORDER BY + /// keys equal the last kept row's (ACE's own TOP keeps them all, verified). + private List CutWithTies( + List sorted, Func keysOf, IReadOnlyList keys, TieCut cut, EvalScope? outer) + { + // As in LimitNode: the counts are literal/parameter/arithmetic, so an empty row scope suffices. + ExpressionEvaluator counts = Eval([], [], outer); + int Count(Expression e) => Convert.ToInt32(counts.Evaluate(e), System.Globalization.CultureInfo.InvariantCulture); + int n = Count(cut.Count); + int take = cut.Percent ? (int)(((long)sorted.Count * n + 99) / 100) : n; + int skip = cut.Offset is { } offset ? Math.Max(0, Count(offset)) : 0; + if (take <= 0 || skip >= sorted.Count) + return []; + int end = (int)Math.Min(sorted.Count, (long)skip + take); + while (end < sorted.Count && CompareEvaluatedKeys(keys, keysOf(sorted[end - 1]), keysOf(sorted[end]), TextComparer) == 0) + end++; + return sorted.GetRange(skip, end - skip); + } + + /// A derived table's columns: under its alias, and named by its column list in order where it has one + /// (AS t(a, b)), which must name every column. + internal static List DerivedColumns(IReadOnlyList inner, string? alias, IReadOnlyList? names) + { + if (names is not null && names.Count != inner.Count) + throw new InvalidOperationException( + $"The derived table '{alias}' has {inner.Count} column(s), but its column list names {names.Count}."); + return [.. inner.Select((c, i) => c with { Qualifier = alias, Name = names?[i] ?? c.Name })]; + } + + /// Compares two rows' already-evaluated ORDER BY key values, honouring each key's direction, with + /// text in 's collation. + private static int CompareEvaluatedKeys( + IReadOnlyList keys, object?[] a, object?[] b, LibRed.Storage.JetTextComparer text) { for (var i = 0; i < keys.Count; i++) { - int c = ExpressionEvaluator.CompareForSort(a[i], b[i]); + int c = ExpressionEvaluator.CompareForSort(a[i], b[i], text); if (keys[i].Direction == SortDirection.Descending) { c = -c; @@ -1826,7 +2165,7 @@ private void ComputeWindow( for (int k = 0; k < key.Length; k++) key[k] = eval.Evaluate(fn.Over.PartitionBy[k]); - var groupKey = new GroupKey(key); + var groupKey = new GroupKey(key, TextComparer); if (!partitions.TryGetValue(groupKey, out List? members)) partitions[groupKey] = members = []; members.Add(i); @@ -1862,7 +2201,7 @@ private void ComputeWindow( if (fn.Over.OrderBy.Count > 0) members.Sort((a, b) => { - int c = CompareEvaluatedKeys(fn.Over.OrderBy, sortKeys[a], sortKeys[b]); + int c = CompareEvaluatedKeys(fn.Over.OrderBy, sortKeys[a], sortKeys[b], TextComparer); return c != 0 ? c : a.CompareTo(b); }); @@ -1872,7 +2211,7 @@ private void ComputeWindow( { // With no ORDER BY every row of the partition is a peer of every other. bool samePeer = fn.Over.OrderBy.Count == 0 - || CompareEvaluatedKeys(fn.Over.OrderBy, sortKeys[members[i - 1]], sortKeys[members[i]]) == 0; + || CompareEvaluatedKeys(fn.Over.OrderBy, sortKeys[members[i - 1]], sortKeys[members[i]], TextComparer) == 0; peerStart[i] = samePeer ? peerStart[i - 1] : i; peerOrdinal[i] = samePeer ? peerOrdinal[i - 1] : peerOrdinal[i - 1] + 1; } @@ -1886,7 +2225,7 @@ private void ComputeWindow( var output = new object?[members.Count]; def.Evaluate( - new WindowPartition(peerStart, peerOrdinal, members.Select(m => arguments[m]).ToList(), call, frameInput, + new WindowPartition(peerStart, peerOrdinal, members.Select(m => arguments[m]).ToList(), TextComparer, call, frameInput, included is null ? null : members.Select(m => included[m]).ToList()), output); @@ -1934,6 +2273,7 @@ private static void CheckFrame(WindowFunction fn, WindowFrame frame) var outTypes = node.Projection .Select(item => ExpressionEvaluator.NumberTypeOf(item.Value, columns, e => DeclaredType(e, columns))).ToList(); + var conversions = node.Projection.Select(item => ItemConversion(item.Value, columns)).ToList(); var outColumns = node.Projection .Select((item, i) => OutputColumn.Computed( item.Alias ?? (item.Value is ColumnReference c ? c.Column : $"Expr{i + 1}"), @@ -1951,9 +2291,14 @@ private static void CheckFrame(WindowFunction fn, WindowFrame frame) Origin = Aggregates(item.Value).Any() || item.Value is ColumnReference ? ColumnOrigin.Aggregate : ColumnOrigin.Expression, + Variant = conversions[i].Variant, }) .ToList(); - var conversions = node.Projection.Select(item => ChoiceConversion(item.Value, columns)).ToList(); + + // Describing evaluates nothing. An aggregate is the one node that has a row over no input — COUNT(*) is 0 — + // so letting it compute one hands the nodes above a row to evaluate, and a correlated subquery being + // described has no outer row for them to read. + if (_describing) return (outColumns, []); // A bare `SELECT COUNT(*)` wants the number of rows, not the rows. Everything below materialises the // whole input first — which for this shape is the entire cost, and pure waste: holding every decoded row @@ -1962,7 +2307,6 @@ private static void CheckFrame(WindowFunction fn, WindowFrame frame) if (IsBareCountStar(node)) return (outColumns, [[CountRows(inRowsEnum)]]); - var inRows = inRowsEnum.ToList(); // Aggregates can appear in the projection, HAVING (e.g. HAVING COUNT(*) > 30) and ORDER BY // (e.g. ORDER BY COUNT(*)); precompute all of them per group so each instance resolves. // A window's arguments and keys may hold aggregates too — RANK() OVER (ORDER BY SUM(x)). @@ -1970,19 +2314,78 @@ private static void CheckFrame(WindowFunction fn, WindowFrame frame) .Concat(node.Having is { } h ? Aggregates(h) : []) .Concat(node.OrderBy.SelectMany(k => Aggregates(k.Value))) .Concat(windows.SelectMany(w => w.Function.Expressions().SelectMany(Aggregates))) + .Select(call => new AggregateSpec(this, call, inColumns)) .ToList(); - // The groups HAVING keeps, each with the row that resolves its keys and its aggregates' values. - var groups = new List<(object?[] KeyRow, Dictionary Values, ExpressionEvaluator Eval)>(); - foreach (List group in GroupRows(inRows, node.GroupBy, inColumns, outer)) + GroupState NewGroupState(object?[]? firstRow) { - var values = new Dictionary(ReferenceComparer.Instance); - foreach (FunctionCall call in aggregateCalls) + var calls = new GroupAggregate?[aggregateCalls.Count]; + for (int i = 0; i < calls.Length; i++) { // An aggregate collected from a nested subquery may belong to that subquery (its argument - // references the subquery's own columns, not this group's) — it can't be computed here, so - // skip it; the subquery computes it itself. A genuine outer aggregate resolves fine. - try { values[call] = ComputeAggregate(call, group, inColumns, outer); } + // references the subquery's own columns, not this group's) — it can't be computed here, so it is + // left out; the subquery computes it itself. A genuine outer aggregate resolves fine. + try { calls[i] = new GroupAggregate(this, aggregateCalls[i]); } + catch (InvalidOperationException) { } + } + return new GroupState { KeyRow = firstRow, Calls = calls }; + } + + // One pass over the input, each row fed to its group's aggregates as it arrives; no row is held beyond + // its group's first and, for LAST, its latest. Holding every row for the length of the grouping was most + // of what a GROUP BY allocated, and it kept all of them alive into the collector's older generations. + // An empty input still has its one group when there is no GROUP BY: COUNT(*) over nothing is 0. + ExpressionEvaluator rowEval = Eval(inColumns, [], outer); // one evaluator, rebound per row + var groupStates = new List(); + GroupState? single = node.GroupBy.Count == 0 ? NewGroupState(null) : null; + if (single is not null) groupStates.Add(single); + var byKey = new Dictionary(); + bool[] keyByText = ComparedAsText(node.GroupBy, inColumns); + foreach (object?[] row in inRowsEnum) + { + rowEval.Rebind(row); + GroupState? state = single; + if (state is null) + { + // Text is keyed here, once — as GroupKey would key it, so it holds these values as they are — and + // the group keeps them for ordering the output, where they used to be evaluated and keyed again. + var keyValues = new object?[node.GroupBy.Count]; + for (int i = 0; i < keyValues.Length; i++) + keyValues[i] = ExpressionEvaluator.SortKey( + Compared(rowEval.Evaluate(node.GroupBy[i]), keyByText[i]), TextComparer); + var key = new GroupKey(keyValues, TextComparer); + if (!byKey.TryGetValue(key, out state)) + { + byKey[key] = state = NewGroupState(row); + state.Keys = key.Values; + groupStates.Add(state); // first-appearance order, which a stable ORDER BY keeps between ties + } + } + state.KeyRow ??= row; + + GroupAggregate?[] calls = state.Calls; + for (int i = 0; i < calls.Length; i++) + { + if (calls[i] is not { } aggregate) continue; + try { aggregate.Add(row, rowEval); } + catch (InvalidOperationException) { calls[i] = null; } + } + } + + // The groups HAVING keeps, each with the row that resolves its keys, its aggregates' values, and its + // grouping key as compared. Every group is evaluated through one scope, rebound to it: HAVING, the windows + // and the projection each take the groups one at a time, and a scope per group made every group resolve its + // column references afresh. + var groups = new List<(object?[] KeyRow, Dictionary Values, object?[] Keys)>(); + ExpressionEvaluator groupEval = Eval(inColumns, [], outer); + foreach (GroupState state in groupStates) + { + var values = new Dictionary(ReferenceComparer.Instance); + for (int i = 0; i < aggregateCalls.Count; i++) + { + // A call that failed on one of this group's rows is left out, as one that fails here is. + if (state.Calls[i] is not { } aggregate) continue; + try { values[aggregateCalls[i].Call] = aggregate.Result(rowEval); } catch (InvalidOperationException) { } } @@ -1990,59 +2393,68 @@ private static void CheckFrame(WindowFunction fn, WindowFrame frame) // calls resolve from the precomputed map (threaded via the scope so correlated subqueries can // reach an outer aggregate). An empty group only happens for an aggregate with no GROUP BY over // zero rows (e.g. COUNT(*) -> 0); there are no key columns to resolve, so a null row suffices. - object?[] keyRow = group.Count > 0 ? group[0] : new object?[inColumns.Count]; - var eval = new ExpressionEvaluator(new EvalScope(inColumns, keyRow, outer, values), this, _parameters, _session); + object?[] keyRow = state.KeyRow ?? new object?[inColumns.Count]; // HAVING filters whole groups after aggregation. - if (node.Having is not null && !eval.IsTrue(node.Having)) + if (node.Having is not null && !groupEval.Rebind(keyRow, values).IsTrue(node.Having)) continue; - groups.Add((keyRow, values, eval)); + groups.Add((keyRow, values, state.Keys)); } // The windows see the groups as their rows, as the standard orders it: after HAVING, before the projection. - var windowValues = new object?[groups.Count][]; - for (int i = 0; i < groups.Count; i++) + var windowValues = new object?[windows.Count > 0 ? groups.Count : 0][]; + for (int i = 0; i < windowValues.Length; i++) windowValues[i] = new object?[windows.Count]; for (int slot = 0; slot < windows.Count; slot++) - ComputeWindow(windows[slot].Function, groups.Count, i => groups[i].Eval, inColumns, windowValues, slot, windowTypes[slot]); - - // Each output row carries its ORDER BY key values AND its grouping-key values, evaluated in the same - // group scope as the projection, to sort the groups afterward: by ORDER BY if present, otherwise — - // matching Access, which returns GROUP BY results ascending by the grouping columns — by the group key - // (this also makes a TOP-1-over-a-GROUP-BY deterministic, as Access/SQL Server do). - var outRows = new List<(object?[] Row, object?[] SortKeys, object?[] GroupKeys)>(); + ComputeWindow(windows[slot].Function, groups.Count, i => groupEval.Rebind(groups[i].KeyRow, groups[i].Values), + inColumns, windowValues, slot, windowTypes[slot]); + + // Each output row carries its ORDER BY key values AND its grouping-key values, to sort the groups + // afterward: by ORDER BY if present, otherwise — matching Access, which returns GROUP BY results ascending + // by the grouping columns — by the group key (this also makes a TOP-1-over-a-GROUP-BY deterministic, as + // Access/SQL Server do). The grouping key is the one the group was found by, already keyed. + var outRows = new List<(object?[] Row, object?[] SortKeys, object?[] GroupKeys)>(groups.Count); + bool[] sortByText = ComparedAsText(node.OrderBy.Select(k => k.Value), columns); + ExpressionEvaluator? windowEval = windows.Count > 0 ? Eval(columns, [], outer) : null; for (int g = 0; g < groups.Count; g++) { - var (keyRow, values, eval) = groups[g]; - if (windows.Count > 0) - eval = new ExpressionEvaluator( - new EvalScope(columns, [.. keyRow, .. windowValues[g]], outer, values), this, _parameters, _session); - - object?[] row = node.Projection - .Select((item, i) => ExpressionEvaluator.ToResultPlaces( - ExpressionEvaluator.AsColumnType(eval.Evaluate(item.Value), conversions[i], currency: false), outTypes[i])) - .ToArray(); - object?[] sortKeys = node.OrderBy.Select(k => eval.Evaluate(k.Value)).ToArray(); - object?[] groupKeys = node.GroupBy.Select(k => eval.Evaluate(k)).ToArray(); - outRows.Add((row, sortKeys, groupKeys)); + var (keyRow, values, keys) = groups[g]; + ExpressionEvaluator eval = windowEval is null + ? groupEval.Rebind(keyRow, values) + : windowEval.Rebind([.. keyRow, .. windowValues[g]], values); + + var row = new object?[node.Projection.Count]; + for (int i = 0; i < row.Length; i++) + row[i] = ExpressionEvaluator.ToResultPlaces( + ExpressionEvaluator.AsColumnType(eval.Evaluate(node.Projection[i].Value), conversions[i].ConvertTo, currency: false), + outTypes[i]); + // Only compared, never returned — so text keys carry their collation keys (see SortKey). + var sortKeys = node.OrderBy.Count == 0 ? [] : new object?[node.OrderBy.Count]; + for (int i = 0; i < sortKeys.Length; i++) + sortKeys[i] = ExpressionEvaluator.SortKey(Compared(eval.Evaluate(node.OrderBy[i].Value), sortByText[i]), TextComparer); + outRows.Add((row, sortKeys, keys)); } if (node.OrderBy.Count > 0) // Stable (see SortNode): groups with equal ORDER BY keys keep their input (first-appearance) order. outRows = outRows.OrderBy(x => x, Comparer<(object?[] Row, object?[] SortKeys, object?[] GroupKeys)>.Create( - (a, b) => CompareEvaluatedKeys(node.OrderBy, a.SortKeys, b.SortKeys))).ToList(); + (a, b) => CompareEvaluatedKeys(node.OrderBy, a.SortKeys, b.SortKeys, TextComparer))).ToList(); else if (node.GroupBy.Count > 0) // No explicit ORDER BY: Access orders GROUP BY output ascending by the grouping columns. outRows.Sort((a, b) => { for (int i = 0; i < node.GroupBy.Count; i++) { - int c = ExpressionEvaluator.CompareForSort(a.GroupKeys[i], b.GroupKeys[i]); + int c = ExpressionEvaluator.CompareForSort(a.GroupKeys[i], b.GroupKeys[i], TextComparer); if (c != 0) return c; } return 0; }); + // WITH TIES is cut here, where the groups' ORDER BY keys still are (it always has an ORDER BY). + if (node.Ties is { } ties) + outRows = CutWithTies(outRows, x => x.SortKeys, node.OrderBy, ties, outer); + return (outColumns, outRows.Select(x => x.Row)); } @@ -2064,7 +2476,7 @@ private static bool IsBareCountStar(AggregateNode node) => && string.Equals(call.Name, "COUNT", StringComparison.OrdinalIgnoreCase); /// Counts a row sequence without retaining it. COUNT is an Access Long Integer, so the count is an - /// — the same type returns, which EF reads with GetInt32. + /// — the same type gives, which EF reads with GetInt32. private static int CountRows(IEnumerable rows) { int count = 0; @@ -2073,114 +2485,186 @@ private static int CountRows(IEnumerable rows) return count; } - private List> GroupRows(List rows, IReadOnlyList keys, IReadOnlyList columns, EvalScope? outer) + /// One group while the input streams past: the row that resolves its keys, and each aggregate call's + /// accumulator — null for a call that cannot be computed over this group. + private sealed class GroupState { - if (keys.Count == 0) - return [rows]; // a single group over all rows (even if empty) - - var order = new List(); - var groups = new Dictionary>(); - foreach (object?[] row in rows) - { - var eval = Eval(columns, row, outer); - var key = new GroupKey(keys.Select(k => eval.Evaluate(k)).ToArray()); - if (!groups.TryGetValue(key, out var list)) - { - groups[key] = list = []; - order.Add(key); - } - list.Add(row); - } - return order.Select(k => groups[k]).ToList(); + public object?[]? KeyRow; + public required GroupAggregate?[] Calls; + /// The grouping key's values as compared, which order the groups when there is no ORDER BY; + /// empty without a GROUP BY. + public object?[] Keys = []; } - /// The distinct set of scalar values, using the same value-equality () as - /// SELECT DISTINCT / GROUP BY, so COUNT(DISTINCT col) dedupes exactly as ACE groups. Order preserved. - private static List DistinctValues(IEnumerable values) + /// + /// What an aggregate call decides from its input's columns alone — its name, whether it reads a Variant's text, + /// whether its argument is a Currency — made once per execution and read by every group. + /// + /// Each of these walks the column list. Asked per group they cost a multi-aggregate GROUP BY over a + /// thousand groups more than a quarter of its time, for answers that never change between groups. Currency is + /// decided on first use rather than up front, so it is only ever asked where the computation reaches it, as + /// before. + private sealed class AggregateSpec(QueryExecutor executor, FunctionCall call, IReadOnlyList columns) { - var seen = new HashSet(); - var result = new List(); - foreach (object? v in values) - if (seen.Add(new GroupKey([v]))) - result.Add(v); - return result; + private bool? _currency; + private bool? _byText; + + public FunctionCall Call => call; + public string Name { get; } = call.Name.ToUpperInvariant(); + + /// Min, Max, First and Last take a Variant's or a Mixed value's text (verified vs ACE: Max of 3, 10 + /// and 25 held as Variants is "3"). + public bool ByText => _byText ??= Name is "MIN" or "MAX" or "FIRST" or "LAST" + && executor.VarianceOf(call.Arguments[0], columns) != Variance.None; + + public bool Currency => _currency ??= executor.IsCurrency(call.Arguments[0], columns); } - private object? ComputeAggregate(FunctionCall call, List group, IReadOnlyList columns, EvalScope? outer) + /// + /// One aggregate call over one group, fed the group's rows as they stream past () and read + /// once they have all arrived (). It holds what the answer needs and no more: a count, a + /// running sum, the distinct values seen, the listed values — never the rows, except the one or two that + /// First, Last and a percentile's fraction are evaluated on. + /// + /// Every rule of the old per-group computation carries over unchanged, and so does its order: FILTER + /// narrows before anything else looks at a row, COUNT(*) included; First and Last are evaluated on their row + /// only when read, so an argument that would fail on some other row still does not; values reach the running + /// aggregate in scan order, which SUM's type (its first value's) depends on. + private sealed class GroupAggregate { - string name = call.Name.ToUpperInvariant(); - ExpressionEvaluator.ValidateArity(name, call.Arguments.Count); - Expression? arg = call.Arguments.Count > 0 ? call.Arguments[0] : null; - - // FILTER (WHERE …) narrows the group before anything else looks at it — COUNT(*) included. - if (call.Filter is { } filter) - group = group.Where(r => Eval(columns, r, outer).IsTrue(filter)).ToList(); - - // COUNT(*) counts rows; DISTINCT is meaningless there (and EF never emits it). - if (name == "COUNT" && arg is StarExpression or null) - return group.Count; - - // FIRST/LAST return the argument's value from the first/last row of the group in scan order — NOT - // null-filtered (verified vs ACE: First over a leading NULL row returns NULL). - if (name == "FIRST") - return group.Count == 0 ? null : Eval(columns, group[0], outer).Evaluate(arg!); - if (name == "LAST") - return group.Count == 0 ? null : Eval(columns, group[^1], outer).Evaluate(arg!); - - // A list aggregate, ordered or not: STRING_AGG may go without its WITHIN GROUP, in which case the - // values list in the order the rows arrive (no keys, no directions — the sort is stable). - if (FunctionCall.IsListAggregate(name)) - { - if (group.Count == 0) - return null; - IReadOnlyList order = call.WithinGroup ?? []; - IReadOnlyList keys = call.WithinGroupKeys; - return ListAgg.Of( - group.Select(r => - { - ExpressionEvaluator e = Eval(columns, r, outer); - return (e.Evaluate(call.Arguments[0]), keys.Select(k => e.Evaluate(k)).ToArray()); - }), - call.Arguments.Count - keys.Count == 2 ? (string)((LiteralExpression)call.Arguments[1]).Value! : "", - order, - call.Distinct); - } - - if (call.WithinGroup is { } directions) + private enum Kind { CountRows, First, Last, List, Percentile, Pair, Running } + + private readonly QueryExecutor _executor; + private readonly FunctionCall _call; + private readonly string _name; + private readonly Kind _kind; + private readonly bool _byText; + + private int _count; + private object?[]? _first; + private object?[]? _last; + private readonly List<(object? Value, object?[] Keys)>? _listed; + private readonly List? _values; + private readonly RunningAggregate? _running; + private readonly HashSet? _seen; + + public GroupAggregate(QueryExecutor executor, AggregateSpec spec) { - if (group.Count == 0) - return null; - // The fraction is the group's, so any row gives it; the standard makes it a constant. - return Percentile.Of(name, - group.Select(r => Eval(columns, r, outer).Evaluate(call.Arguments[1])), - Eval(columns, group[0], outer).Evaluate(call.Arguments[0]), - directions[0]); + _executor = executor; + _call = spec.Call; + _name = spec.Name; + ExpressionEvaluator.ValidateArity(_name, _call.Arguments.Count); + Expression? arg = _call.Arguments.Count > 0 ? _call.Arguments[0] : null; + + // COUNT(*) counts rows; DISTINCT is meaningless there (and EF never emits it). + if (_name == "COUNT" && arg is StarExpression or null) + { + _kind = Kind.CountRows; + return; + } + + _byText = spec.ByText; + if (_name == "FIRST") _kind = Kind.First; + else if (_name == "LAST") _kind = Kind.Last; + else if (FunctionCall.IsListAggregate(_name)) + { + _kind = Kind.List; + _listed = []; + } + else if (_call.WithinGroup is not null) + { + _kind = Kind.Percentile; + _values = []; + } + else if (RunningAggregate.IsPair(_name)) + { + // A binary set function reads a pair from each row; the standard gives it no DISTINCT. + if (_call.Distinct) + throw new NotSupportedException($"{_call.Name} takes no DISTINCT."); + _kind = Kind.Pair; + _running = new RunningAggregate(_name, countRows: false, currency: false, executor.TextComparer); + } + else + { + _kind = Kind.Running; + _running = new RunningAggregate(_name, countRows: false, currency: spec.Currency, executor.TextComparer); + // COUNT(DISTINCT)/SUM(DISTINCT)/… aggregate the distinct set of the argument's values, deduped on + // the same value-equality as SELECT DISTINCT and GROUP BY, so they dedupe exactly as ACE groups. + // MIN/MAX are unaffected by dedup, but applying it uniformly keeps the one code path. + if (_call.Distinct) _seen = []; + } } - // A binary set function reads a pair from each row; the standard gives it no DISTINCT. - if (RunningAggregate.IsPair(name)) + /// Feeds one of the group's rows, with already bound to it. + public void Add(object?[] row, ExpressionEvaluator eval) { - if (call.Distinct) - throw new NotSupportedException($"{call.Name} takes no DISTINCT."); - var pair = new RunningAggregate(name, countRows: false, currency: false); - foreach (object?[] row in group) + // FILTER (WHERE …) narrows the group before anything else looks at it — COUNT(*) included. + if (_call.Filter is { } filter && !eval.IsTrue(filter)) + return; + + switch (_kind) { - ExpressionEvaluator rowEval = Eval(columns, row, outer); - pair.AddPair(rowEval.Evaluate(call.Arguments[0]), rowEval.Evaluate(call.Arguments[1])); + case Kind.CountRows: + _count++; + break; + // FIRST/LAST return the argument's value from the first/last row of the group in scan order — NOT + // null-filtered (verified vs ACE: First over a leading NULL row returns NULL). + case Kind.First: + _first ??= row; + break; + case Kind.Last: + _last = row; + break; + // A list aggregate, ordered or not: STRING_AGG may go without its WITHIN GROUP, in which case the + // values list in the order the rows arrive (no keys, no directions — the sort is stable). + case Kind.List: + { + IReadOnlyList keys = _call.WithinGroupKeys; + object? value = eval.Evaluate(_call.Arguments[0]); + var keyValues = new object?[keys.Count]; + for (int i = 0; i < keyValues.Length; i++) keyValues[i] = eval.Evaluate(keys[i]); + _listed!.Add((value, keyValues)); + break; + } + case Kind.Percentile: + _first ??= row; + _values!.Add(eval.Evaluate(_call.Arguments[1])); + break; + case Kind.Pair: + _running!.AddPair(eval.Evaluate(_call.Arguments[0]), eval.Evaluate(_call.Arguments[1])); + break; + default: + { + object? value = Compared(eval.Evaluate(_call.Arguments[0]), _byText); + if (_seen is not null && (value is null || !_seen.Add(new GroupKey([value], _executor.TextComparer)))) + return; + _running!.Add(value); + break; + } } - return pair.Result; } - IEnumerable values = group.Select(r => Eval(columns, r, outer).Evaluate(arg!)); - // COUNT(DISTINCT)/SUM(DISTINCT)/… aggregate the distinct set of the argument's values. MIN/MAX are - // unaffected by dedup, but applying it uniformly keeps the one code path. - if (call.Distinct) - values = DistinctValues(values.Where(v => v is not null)); - - var aggregate = new RunningAggregate(name, countRows: false, currency: IsCurrency(arg!, columns)); - foreach (object? value in values) - aggregate.Add(value); - return aggregate.Result; + /// The aggregate over every row fed. is rebound to a held row where one + /// is evaluated. + public object? Result(ExpressionEvaluator eval) => _kind switch + { + Kind.CountRows => _count, + Kind.First => _first is null ? null : Compared(eval.Rebind(_first).Evaluate(_call.Arguments[0]), _byText), + Kind.Last => _last is null ? null : Compared(eval.Rebind(_last).Evaluate(_call.Arguments[0]), _byText), + Kind.List => _listed!.Count == 0 ? null : ListAgg.Of( + _listed, + _call.Arguments.Count - _call.WithinGroupKeys.Count == 2 ? (string)((LiteralExpression)_call.Arguments[1]).Value! : "", + _call.WithinGroup ?? [], + _call.Distinct, + _executor.TextComparer), + // The fraction is the group's, so any row gives it; the standard makes it a constant. + Kind.Percentile => _values!.Count == 0 ? null : Percentile.Of(_name, + _values, + eval.Rebind(_first!).Evaluate(_call.Arguments[0]), + _call.WithinGroup![0], + _executor.TextComparer), + _ => _running!.Result, + }; } /// Whether an aggregate's argument is a Currency, which the statistical aggregates square exactly. @@ -2235,34 +2719,72 @@ private static IEnumerable Aggregates(Expression e) _ => [], }; - /// Groups by structural equality of the key value tuple. - // DISTINCT / GROUP BY / INTERSECT / EXCEPT key. String keys use Access text semantics — case-insensitive - // and trailing-space-insensitive — so 'London' and 'LONDON ' group together as Access does. - internal sealed class GroupKey(object?[] values) : IEquatable + /// The DISTINCT / GROUP BY / INTERSECT / EXCEPT key: a value tuple compared on exactly the terms + /// the evaluator compares values on, and hashed on its matching . + /// + /// + /// Two keys are the same key when = would call them equal: text is one value when the database's + /// collation says so (case and trailing spaces fold, and so does whatever else that order folds — ß + /// and ss), and a number folds across its CLR types, so a LONG 1 and a DOUBLE 1.0 arriving under one + /// key are one group. Both are measured — ACE returns a single group for either — and a column alone never + /// mixes numeric types, so the second only shows up through an expression, e.g. an IIF whose arms + /// are typed differently. + /// + internal sealed class GroupKey(object?[] values, LibRed.Storage.JetTextComparer text) : IEquatable { - private readonly object?[] _values = values; + // Text is held as its collation key, made once: every Equals and hash then works on bytes, and agrees + // with the text comparison exactly because it is that comparison's own key. + private readonly object?[] _values = Keyed(values, text); + private readonly LibRed.Storage.JetTextComparer _text = text; + + /// The values with each text replaced by its sort key. Anything else is its own key, so values + /// holding no text are kept as passed rather than copied — a copy per key cost a numeric DISTINCT a + /// quarter of its time. + private static object?[] Keyed(object?[] values, LibRed.Storage.JetTextComparer text) + { + if (Array.FindIndex(values, v => v is string) < 0) + return values; + var keyed = new object?[values.Length]; + for (int i = 0; i < values.Length; i++) + keyed[i] = ExpressionEvaluator.SortKey(values[i], text); + return keyed; + } + + public bool Equals(GroupKey? other) + { + if (other is null || _values.Length != other._values.Length) return false; + for (int i = 0; i < _values.Length; i++) + if (!KeyEquals(_values[i], other._values[i], _text)) return false; + return true; + } - public bool Equals(GroupKey? other) => - other is not null && _values.Length == other._values.Length - && _values.Zip(other._values).All(p => KeyEquals(p.First, p.Second)); + /// The values as compared — text as its collation key — which also order the groups. + public object?[] Values => _values; public override bool Equals(object? obj) => Equals(obj as GroupKey); public override int GetHashCode() { var hash = new HashCode(); foreach (object? v in _values) - hash.Add(v is string s ? StringComparer.InvariantCultureIgnoreCase.GetHashCode(s.TrimEnd(' ')) : v?.GetHashCode() ?? 0); + { + if (v is ExpressionEvaluator.CollatedText collated) + hash.AddBytes(collated.Key); + else + hash.Add(v is null ? 0 : ExpressionEvaluator.KeyHash(v, _text)); + } return hash.ToHashCode(); } - // Grouping has to agree with ExpressionEvaluator's text comparison, which is linguistic on purpose; - // ordinal (CA1309) would split groups ACE puts together, and would disagree with GetHashCode above. -#pragma warning disable CA1309 - private static bool KeyEquals(object? a, object? b) => - a is string sa && b is string sb - ? string.Equals(sa.TrimEnd(' '), sb.TrimEnd(' '), StringComparison.InvariantCultureIgnoreCase) - : Equals(a, b); -#pragma warning restore CA1309 + // Folds within a kind, because GetHashCode partitions by kind: calling a number and its text spelling + // one key would bucket them apart and split the group anyway. + private static bool KeyEquals(object? a, object? b, LibRed.Storage.JetTextComparer text) => (a, b) switch + { + (null, null) => true, + (null, _) or (_, null) => false, + (ExpressionEvaluator.CollatedText x, ExpressionEvaluator.CollatedText y) => x.Key.AsSpan().SequenceEqual(y.Key), + (ExpressionEvaluator.CollatedText, _) or (_, ExpressionEvaluator.CollatedText) => false, + _ => ExpressionEvaluator.KeyEqual(a, b, text), + }; } private sealed class ReferenceComparer : IEqualityComparer diff --git a/src/LibRed/LibRed.Engine/Execution/RunningAggregate.cs b/src/LibRed/LibRed.Engine/Execution/RunningAggregate.cs index c676c06f8..12ad939e5 100644 --- a/src/LibRed/LibRed.Engine/Execution/RunningAggregate.cs +++ b/src/LibRed/LibRed.Engine/Execution/RunningAggregate.cs @@ -72,18 +72,29 @@ internal sealed class RunningAggregate /// Whether this is COUNT(*), which counts rows rather than values. /// Whether the argument is a Currency, which the statistical aggregates square /// exactly. - public RunningAggregate(string name, bool countRows, bool currency) + /// How MIN and MAX order text: the database's collation. + public RunningAggregate(string name, bool countRows, bool currency, LibRed.Storage.JetTextComparer text) { _name = Supports(name) ? Canonical(name) : throw new NotSupportedException($"Aggregate {name} is not supported."); _countRows = countRows; _currency = currency; + _text = text; } + private readonly LibRed.Storage.JetTextComparer _text; + /// Whether (upper case) is an aggregate this computes. - public static bool Supports(string name) => - Canonical(name) is "COUNT" or "SUM" or "AVG" or "MIN" or "MAX" or "VAR" or "VARP" or "STDEV" or "STDEVP" - or "STDDEV" or "STDDEVP" - || IsPair(name); + public static bool Supports(string name) => ScalarNames.Contains(Canonical(name)) || IsPair(name); + + /// The aggregates this computes, under every name they answer to. One list, which + /// derives from, so a name cannot be added to the computation and missed by the + /// test that binds this set to the planner and to the declared result types. + public static IEnumerable SupportedNames => ScalarNames.Concat(PairNames).Concat(StandardNames.Keys); + + private static readonly HashSet ScalarNames = + [ + "COUNT", "SUM", "AVG", "MIN", "MAX", "VAR", "VARP", "STDEV", "STDEVP", "STDDEV", "STDDEVP", + ]; /// Whether (upper case) is a binary set function, fed by . public static bool IsPair(string name) => PairNames.Contains(name); @@ -131,8 +142,8 @@ public void Add(object? value) { if (_extreme is null or string { Length: 0 } || (_name == "MAX" - ? ExpressionEvaluator.CompareForSort(value, _extreme) > 0 - : ExpressionEvaluator.CompareForSort(value, _extreme) < 0)) + ? ExpressionEvaluator.CompareForSort(value, _extreme, _text) > 0 + : ExpressionEvaluator.CompareForSort(value, _extreme, _text) < 0)) _extreme = value; return; } diff --git a/src/LibRed/LibRed.Engine/Execution/StatementExecutor.cs b/src/LibRed/LibRed.Engine/Execution/StatementExecutor.cs index 88cece22c..6a9c2b1b6 100644 --- a/src/LibRed/LibRed.Engine/Execution/StatementExecutor.cs +++ b/src/LibRed/LibRed.Engine/Execution/StatementExecutor.cs @@ -19,6 +19,14 @@ internal sealed class StatementExecutor(JetDatabase database, IReadOnlyDictionar // For evaluating VALUES expressions (literals, parameters, and any scalar subqueries). private readonly QueryExecutor _scalarRunner = new(database, parameters, session); + // What the constraint checks look up, resolved once per statement rather than once per row: CHECK expressions + // parsed, each table's enforced relationships from either end, and each relationship's identity. + private readonly Dictionary _checkExpressions = []; + private readonly Dictionary _foreignKeysOf = new([], StringComparer.OrdinalIgnoreCase); + private readonly Dictionary _referencesTo = new([], StringComparer.OrdinalIgnoreCase); + private readonly Dictionary _identities = []; + private readonly Dictionary _ends = []; + public int Execute(SqlStatement statement) => statement switch { CreateTableStatement create => ExecuteCreateTable(create), @@ -69,9 +77,19 @@ private int ExecuteCreateTable(CreateTableStatement statement) var relationships = statement.ForeignKeys.Select(fk => ToRelationshipSpec(statement.Table, fk)).ToList(); + // An unnamed constraint's generated name has to fit the 64 characters a name may have, however long the + // table's own name is. + string Generated(string prefix, int i) + { + string suffix = $"_{i}"; + int room = JetName.MaxLength - prefix.Length - 1 - suffix.Length; + return $"{prefix}_{(statement.Table.Length > room ? statement.Table[..room] : statement.Table)}{suffix}"; + } + var uniques = statement.UniqueConstraints.Select((u, i) => new UniqueIndexSpec( - Name: u.Name ?? $"UQ_{statement.Table}_{i}", - Columns: u.Columns)).ToList(); + Name: u.Name ?? Generated("UQ", i), + Columns: u.Columns, + DeclaredAfterColumns: u.DeclaredAfterColumns)).ToList(); var defaults = statement.Columns .Where(c => c.Default is not null) @@ -79,11 +97,11 @@ private int ExecuteCreateTable(CreateTableStatement statement) .ToList(); var checks = statement.CheckConstraints - .Select((ck, i) => (Name: ck.Name ?? $"CK_{statement.Table}_{i}", ck.Expression)) + .Select((ck, i) => (Name: ck.Name ?? Generated("CK", i), ck.Expression)) .ToList(); _database.CreateTable(statement.Table, columns, primaryKey, relationships, uniques, defaults, checks, - statement.PrimaryKeyName); + statement.PrimaryKeyName, statement.PrimaryKeyDeclaredAfterColumns); return 0; } @@ -117,16 +135,20 @@ private static void EnforceRequired(string table, IReadOnlyList colum /// row. A CHECK is violated only when its expression is explicitly FALSE; NULL/unknown passes (SQL /// three-valued CHECK semantics). The expression may reference the row's own columns and use (uncorrelated) /// subqueries. Matches Access's validation-rule enforcement. - private void EnforceCheckConstraints(TableDef definition, object?[] values) + private void EnforceCheckConstraints(TableDefinition definition, object?[] values) { if (definition.CheckConstraints.Count == 0) return; var schema = definition.Columns.Select(c => OutputColumn.Of(definition.Name, c)).ToList(); var evaluator = new ExpressionEvaluator(new EvalScope(schema, values, null), _scalarRunner, _parameters, _session); foreach (var (name, expression) in definition.CheckConstraints) - if (evaluator.Evaluate(_parser.ParseExpression(expression)) is false) + { + if (!_checkExpressions.TryGetValue(expression, out Expression? parsed)) + _checkExpressions[expression] = parsed = _parser.ParseExpression(expression); + if (evaluator.Evaluate(parsed) is false) throw new InvalidOperationException( $"One or more values are prohibited by the validation rule '{name}' set for '{definition.Name}'. " + "Enter a value that the expression for this field can accept."); + } } /// The relationship a parsed foreign key creates on , for CREATE TABLE and @@ -144,7 +166,8 @@ private void EnforceCheckConstraints(TableDef definition, object?[] values) NoIndex: fk.NoIndex, DeleteSetNull: fk.OnDelete == ReferentialAction.SetNull, UpdateSetNull: fk.OnUpdate == ReferentialAction.SetNull, - ReferencesPrimaryKey: fk.ReferencedColumns.Count == 0); + ReferencesPrimaryKey: fk.ReferencedColumns.Count == 0, + DeclaredAfterColumns: fk.DeclaredAfterColumns); /// Access-style fallback name when the constraint is unnamed: "childparent". private static string DefaultRelationshipName(string childTable, ForeignKeyConstraint fk) => @@ -155,105 +178,235 @@ private static string DefaultRelationshipName(string childTable, ForeignKeyConst /// ACE's **MATCH FULL** rule (verified vs ACE, and unlike SQL Server's MATCH SIMPLE): a composite FK is /// skipped only when **every** column is null; a **partial** null (some null, some not) can never match a /// parent key and is rejected. Only enforced relationships (grbit without "don't enforce") are checked. + /// says the row is an UPDATE's rather than an INSERT's, for the message. /// - private void EnforceReferentialIntegrity(string childTable, Table table, object?[] values) + private void EnforceReferentialIntegrity(string childTable, Table table, object?[] values, bool update = false) { - foreach (ForeignKey fk in _database.Catalog.ForeignKeysOf(childTable)) + foreach (ForeignKey fk in ForeignKeysOf(childTable)) { - if (!fk.IsEnforced) continue; - - var target = new object?[fk.Columns.Count]; - int nullCount = 0; - for (int i = 0; i < fk.Columns.Count; i++) - { - ColumnDef col = table.Definition.FindColumn(fk.Columns[i].Column) - ?? throw new InvalidOperationException($"Column '{fk.Columns[i].Column}' does not exist in '{childTable}'."); - object? v = values[col.Index]; - if (v is null) nullCount++; - else target[i] = v; - } - if (nullCount == fk.Columns.Count) continue; // all FK columns null → the FK is not applied to this row - - // MATCH FULL: a partial null can never reference a full parent key, so it's a violation — treated - // like a missing parent (ACE gives the same "a related record is required" error). - bool partialNull = nullCount > 0; + RelationshipEnds ends = EndsOf(fk); + // All FK columns null → the FK is not applied to this row. + if (ForeignKeyValue(values, ends.ChildColumns, out bool partialNull) is not { } target) continue; // A **self-referencing** FK: a fully-specified row that points at its own key (the root of a // required self-ref, e.g. Inverse1Id = Id) satisfies the FK even though it isn't on disk yet — the // parent scan runs before the insert. Access allows this (EF's ComplexNavigations seed relies on it). if (!partialNull && string.Equals(fk.ReferencedTable, childTable, StringComparison.OrdinalIgnoreCase) - && RowSatisfiesOwnKey(fk, table, values, target)) + && KeyEquals(values, ends.ParentColumns, target)) continue; - if (partialNull || !ParentRowExists(fk, target)) + if (partialNull || !ParentRowExists(ends, target)) throw new InvalidOperationException( - $"INSERT into '{childTable}' violates foreign key '{fk.Name}': no matching row in '{fk.ReferencedTable}'."); + $"{(update ? "UPDATE of" : "INSERT into")} '{childTable}' violates foreign key '{fk.Name}': " + + $"no matching row in '{fk.ReferencedTable}'."); + + // Inside a transaction, finding the parent now is not enough: reading it writes no page, so the + // commit's page-conflict check cannot see the dependency, and a concurrent transaction is free to + // delete that parent and commit — each having checked the other's precondition, leaving a child row + // referencing nothing. The condition is re-checked when this transaction commits, once per + // relationship and key however many rows relied on it. + DependOnRelationship(fk, target, + $"Transaction conflict on foreign key '{fk.Name}': the row in '{fk.ReferencedTable}' that " + + $"'{childTable}' was checked against no longer exists."); } } - /// For a self-referencing FK, whether the row's own referenced-column values equal the FK - /// target — i.e. the row points at itself (or at its own composite key), which satisfies the FK. - private static bool RowSatisfiesOwnKey(ForeignKey fk, Table table, object?[] values, object?[] target) + /// + /// Makes the transaction's commit re-check that no row of 's child table holds + /// without a parent row holding it too. Both ends of a relationship need it, because + /// each end's check reads a row the other end's transaction can change and reading writes no page: a child + /// write found its parent, or a parent delete or key change found no children, and a concurrent transaction + /// can commit the opposite write in between. The condition is the same from either end, so it is held once per + /// relationship and key however many rows relied on it. + /// + private void DependOnRelationship(ForeignKey fk, object?[] key, string conflict) + { + var dependency = new ForeignKeyDependency( + IdentityOf(fk) ?? throw new InvalidOperationException( + $"Relationship '{fk.Name}' names a table or column that does not exist."), + key); + _database.DependOn(() => ForeignKeyDependencyHolds(dependency), conflict, key: dependency); + } + + /// + /// Whether the parent a transaction's write relied on is still there, or no longer needed. It is needed only + /// while the relationship survives and a child row still holds the key: later statements may delete the child, + /// move its key, cascade it, or drop the relationship. The relationship is found by name and is the same one + /// only while it joins the same tables over the same columns, whatever they are called by now. The parent is + /// looked for first, since it is almost always still there. + /// + private bool ForeignKeyDependencyHolds(ForeignKeyDependency dependency) + { + if (CurrentRelationship(dependency.Relationship) is not { } relationship) return true; + RelationshipEnds ends = Ends(relationship); // as the catalog holds them now, not when the write was made + return ParentRowExists(ends, dependency.Key) + || !RowsHoldingKey(ends.Child, ends.ChildColumns, dependency.Key).Any(); + } + + /// The enforced relationship identifies, as the catalog holds it now; null + /// once it has been dropped. + private ForeignKey? CurrentRelationship(RelationshipIdentity identity) => + _database.Catalog.Relationships.FirstOrDefault(r => r.IsEnforced + && string.Equals(r.Name, identity.Name, StringComparison.OrdinalIgnoreCase) + && identity.Equals(Identify(r))); + + /// Whether every row of the relationship's child table has its parent, on the MATCH FULL terms + /// applies to one row: a key that is entirely null references + /// nothing and needs nothing, and a partly null one can never match. + private bool ExistingRowsHaveParents(ForeignKey fk) + { + RelationshipEnds ends = Ends(fk); + foreach (object?[] row in ends.Child.Rows(ends.Child.DecodeOnly(ends.ChildColumns))) + if (ForeignKeyValue(row, ends.ChildColumns, out bool partialNull) is { } key + && (partialNull || !ParentRowExists(ends, key))) + return false; + return true; + } + + /// The foreign key a row holds, on ACE's **MATCH FULL** terms (verified vs ACE, and unlike SQL + /// Server's MATCH SIMPLE): null when every column is null — the row references nothing and the relationship is + /// not applied to it — and when only some are, a key that can never match a full + /// parent key and so breaks the relationship (ACE gives the same "a related record is required" error). + private static object?[]? ForeignKeyValue(object?[] row, int[] columns, out bool partialNull) + { + object?[] key = [.. columns.Select(c => row[c])]; + int nulls = key.Count(v => v is null); + partialNull = nulls > 0 && nulls < key.Length; + return nulls == key.Length ? null : key; + } + + /// A relationship's two tables, and the positions its columns occupy in each one's rows. + private sealed record RelationshipEnds(Table Child, int[] ChildColumns, Table Parent, int[] ParentColumns); + + /// The relationship's two ends as the catalog holds them now. + private RelationshipEnds Ends(ForeignKey fk) + { + Table child = _database.OpenTable(fk.Table); + Table parent = _database.OpenTable(fk.ReferencedTable); + return new RelationshipEnds( + child, [.. fk.Columns.Select(c => child.Definition.RequireColumn(c.Column).Index)], + parent, [.. fk.Columns.Select(c => parent.Definition.RequireColumn(c.ReferencedColumn).Index)]); + } + + /// , once per relationship per statement — what every row the statement writes + /// checks against. A commit-time check resolves them afresh, since later statements may have changed them. + private RelationshipEnds EndsOf(ForeignKey fk) { + if (!_ends.TryGetValue(fk, out RelationshipEnds? ends)) + _ends[fk] = ends = Ends(fk); + return ends; + } + + /// A relationship as the file identifies it rather than by its names: the definition pages of the + /// tables it joins and the ids of its columns, which renaming a table or a column leaves alone. Null when one + /// of them no longer exists. + private RelationshipIdentity? Identify(ForeignKey fk) + { + if (_database.Catalog.FindTable(fk.Table) is not { } child + || _database.Catalog.FindTable(fk.ReferencedTable) is not { } parent) + return null; + var childColumns = new int[fk.Columns.Count]; + var parentColumns = new int[fk.Columns.Count]; for (int i = 0; i < fk.Columns.Count; i++) { - ColumnDef refCol = table.Definition.FindColumn(fk.Columns[i].ReferencedColumn) - ?? throw new InvalidOperationException($"Column '{fk.Columns[i].ReferencedColumn}' does not exist in '{table.Name}'."); - if (ExpressionEvaluator.CompareForSort(values[refCol.Index], target[i]) != 0) return false; + if (child.FindColumn(fk.Columns[i].Column) is not { } childColumn + || parent.FindColumn(fk.Columns[i].ReferencedColumn) is not { } parentColumn) + return null; + childColumns[i] = childColumn.ColumnId; + parentColumns[i] = parentColumn.ColumnId; } - return true; + return new RelationshipIdentity(fk.Name, child.DefinitionPage, parent.DefinitionPage, childColumns, parentColumns); } - /// Scans the parent table for a row whose referenced columns equal the child key values. - private bool ParentRowExists(ForeignKey fk, object?[] target) + /// A relationship's name, and what it joins as finds it: equal only for the same + /// relationship, so one dropped and added again under its name over other columns is another. + private sealed record RelationshipIdentity( + string Name, int ChildTable, int ParentTable, int[] ChildColumns, int[] ParentColumns) { - Table parent = _database.OpenTable(fk.ReferencedTable); - int[] parentCols = fk.Columns.Select(c => - (parent.Definition.FindColumn(c.ReferencedColumn) - ?? throw new InvalidOperationException($"Column '{c.ReferencedColumn}' does not exist in '{fk.ReferencedTable}'.")).Index) - .ToArray(); + public bool Equals(RelationshipIdentity? other) => other is not null + && string.Equals(Name, other.Name, StringComparison.OrdinalIgnoreCase) + && ChildTable == other.ChildTable && ParentTable == other.ParentTable + && ChildColumns.AsSpan().SequenceEqual(other.ChildColumns) + && ParentColumns.AsSpan().SequenceEqual(other.ParentColumns); + + public override int GetHashCode() => + HashCode.Combine(StringComparer.OrdinalIgnoreCase.GetHashCode(Name), ChildTable, ParentTable); + } + + /// A parent a transaction's write relied on: the relationship, and the key it was checked for. Equal + /// values are one dependency, which is what lets the channel hold it once. + private sealed record ForeignKeyDependency(RelationshipIdentity Relationship, object?[] Key) + { + public bool Equals(ForeignKeyDependency? other) => other is not null + && Relationship.Equals(other.Relationship) + && System.Collections.StructuralComparisons.StructuralEqualityComparer.Equals(Key, other.Key); + + public override int GetHashCode() => HashCode.Combine( + Relationship, System.Collections.StructuralComparisons.StructuralEqualityComparer.GetHashCode(Key)); + } + + /// Whether a parent row holds the key a child row references. + private bool ParentRowExists(RelationshipEnds ends, object?[] key) => + RowsHoldingKey(ends.Parent, ends.ParentColumns, key).Any(); - foreach (object?[] row in parent.Rows()) + /// The rows of whose hold + /// on 's terms, read in full: seeked through an index on those columns when the key + /// can be sought (every value in its column's kind, as a query's column = value seek requires), else + /// scanned. A relationship's parent side always has one — its referenced columns are a primary or unique + /// key — so the check every child INSERT makes, which scanned the whole parent table, is a seek. + private IEnumerable<(RowId Id, object?[] Values)> RowsHoldingKey(Table table, int[] columns, object?[] key) + { + bool Holds(object?[] values) => KeyEquals(values, columns, key); + + var seekKey = new object?[key.Length]; + for (int i = 0; i < key.Length; i++) { - bool match = true; - for (int i = 0; i < parentCols.Length; i++) - if (ExpressionEvaluator.CompareForSort(row[parentCols[i]], target[i]) != 0) { match = false; break; } - if (match) return true; + ColumnDef column = table.Definition.Columns.First(c => c.Index == columns[i]); + if (!ExpressionEvaluator.TryGetSeekKey(column, new LiteralExpression(key[i]), key[i], out seekKey[i])) + return table.RowsWhere(columns, Holds); } - return false; + + return table.RowsWithKey(columns, seekKey, Holds); } /// The enforced relationships for which is the referenced /// (parent) side — i.e. those whose child rows a delete/key-update of a parent row must handle. - private IEnumerable ChildRelationshipsOf(string parentTable) => - _database.Catalog.Relationships.Where(r => - r.IsEnforced && string.Equals(r.ReferencedTable, parentTable, StringComparison.OrdinalIgnoreCase)); + private ForeignKey[] ChildRelationshipsOf(string parentTable) + { + if (!_referencesTo.TryGetValue(parentTable, out ForeignKey[]? found)) + _referencesTo[parentTable] = found = [.. _database.Catalog.Relationships.Where(r => + r.IsEnforced && string.Equals(r.ReferencedTable, parentTable, StringComparison.OrdinalIgnoreCase))]; + return found; + } - /// The referenced-column values from a parent row (the key children point at). - private object?[] ReferencedKey(ForeignKey fk, object?[] parentValues) + /// The enforced relationships holds the foreign key of. + private ForeignKey[] ForeignKeysOf(string childTable) { - Table parent = _database.OpenTable(fk.ReferencedTable); - return fk.Columns.Select(c => parentValues[parent.Definition.FindColumn(c.ReferencedColumn)!.Index]).ToArray(); + if (!_foreignKeysOf.TryGetValue(childTable, out ForeignKey[]? found)) + _foreignKeysOf[childTable] = found = [.. _database.Catalog.ForeignKeysOf(childTable).Where(f => f.IsEnforced)]; + return found; } - /// Child rows whose FK columns equal (a null FK column never matches). - private List<(RowId Id, object?[] Values)> FindChildRows(ForeignKey fk, object?[] key) + /// , once per relationship per statement. + private RelationshipIdentity? IdentityOf(ForeignKey fk) { - Table child = _database.OpenTable(fk.Table); - int[] childCols = fk.Columns.Select(c => child.Definition.FindColumn(c.Column)!.Index).ToArray(); - var result = new List<(RowId, object?[])>(); - foreach ((RowId id, object?[] values) in child.Rows().WithIds()) - { - bool match = true; - for (int i = 0; i < childCols.Length; i++) - if (values[childCols[i]] is null || ExpressionEvaluator.CompareForSort(values[childCols[i]], key[i]) != 0) - { match = false; break; } - if (match) result.Add((id, values)); - } - return result; + if (!_identities.TryGetValue(fk, out RelationshipIdentity? identity)) + _identities[fk] = identity = Identify(fk); + return identity; } + /// ACE's refusal to delete a parent row, or change its key, while a child row still holds it. + private static InvalidOperationException RelatedRecords(ForeignKey fk) => + new($"The record cannot be deleted or changed because table '{fk.Table}' includes related records."); + + /// Child rows whose FK columns equal . Both callers skip a key with a null + /// in it — nothing references one — so a null FK column simply fails the comparison. + private List<(RowId Id, object?[] Values)> FindChildRows(RelationshipEnds ends, object?[] key) => + // Seeked where the child has an index on its key (ACE gives an enforced relationship one), else found by + // a scan of the key; a match is read whole either way, because the cascade rewrites or deletes it. + [.. RowsHoldingKey(ends.Child, ends.ChildColumns, key)]; + /// /// Deletes the given root rows and everything ON DELETE CASCADE reaches from them, applying SET NULL and /// NO ACTION as it goes. An explicit worklist replaces the former recursion: each row is scheduled at most @@ -268,7 +421,10 @@ private void CascadeDelete(Table rootTable, IEnumerable<(RowId Id, object?[] Val var order = new List<(Table Table, RowId Id, object?[] Values)>(); var stack = new Stack<(Table Table, RowId Id, object?[] Values, bool ChildrenExpanded)>(); - foreach (var (id, values) in roots) + // Pushed last first, so the roots come off the stack — and are deleted — in the order they were found, as + // ACE deletes them. The order shows in the file: a delete slides the rows below it up and leaves the old + // bytes in the space it frees, so deleting the same rows the other way round leaves different bytes. + foreach (var (id, values) in roots.Reverse()) if (scheduled.Add((rootTable.Name, id))) stack.Push((rootTable, id, values, false)); @@ -285,47 +441,7 @@ private void CascadeDelete(Table rootTable, IEnumerable<(RowId Id, object?[] Val } foreach (var (table, id, values) in order) - { - DeleteComplexValues(table, values); - foreach (IndexDef index in table.Definition.Indexes.Where(i => i.RootPage > 0) - .GroupBy(i => i.RootPage).Select(g => g.First())) - table.RemoveIndexEntry(index, values, id); table.Delete(id); - } - } - - /// - /// Removes the values every complex (multi-value / attachment) column of holds - /// for the row being deleted — the flat-table rows carrying that record's complex id. - /// - /// - /// Measured against ACE (ComplexDeleteCascadeProbeTest): deleting a record removes its values from - /// every complex column of the table — a record holding three attachments in one column and two in - /// another loses all five — and rolls no counter back. The owner's 0x1C and each flat - /// table's 0x14 keep their values, so the next row still takes the following id and the freed - /// value ids are never reissued. Leaving the rows behind would orphan values no record points at. - /// - private void DeleteComplexValues(Table table, object?[] values) - { - foreach (ComplexColumn complex in _database.Catalog.ComplexColumns) - { - if (!string.Equals(complex.OwnerTable.Name, table.Name, StringComparison.OrdinalIgnoreCase)) continue; - if (values[complex.OwnerTable.FindColumn(complex.ColumnName)!.Index] is not { } raw) continue; - int recordId = Convert.ToInt32(raw, System.Globalization.CultureInfo.InvariantCulture); - - Table flat = _database.OpenTable(complex.FlatTable.Name); - foreach ((RowId flatId, object?[] flatValues) in flat.Rows().WithIds().ToList()) - { - if (flatValues[complex.OwnerLink.Index] is not { } link - || Convert.ToInt32(link, System.Globalization.CultureInfo.InvariantCulture) != recordId) - continue; - - foreach (IndexDef index in flat.Definition.Indexes.Where(i => i.RootPage > 0) - .GroupBy(i => i.RootPage).Select(g => g.First())) - flat.RemoveIndexEntry(index, flatValues, flatId); - flat.Delete(flatId); - } - } } /// Applies each enforced relationship's ON DELETE action to a parent row about to be deleted: @@ -338,93 +454,158 @@ private void EnqueueCascadeChildren(string parentTable, object?[] parentValues, { foreach (ForeignKey fk in ChildRelationshipsOf(parentTable)) { - object?[] key = ReferencedKey(fk, parentValues); + RelationshipEnds ends = EndsOf(fk); + object?[] key = [.. ends.ParentColumns.Select(c => parentValues[c])]; if (key.Any(k => k is null)) continue; // a null parent key is referenced by nobody - var children = FindChildRows(fk, key); + DependOnRelationship(fk, key, + $"Transaction conflict on foreign key '{fk.Name}': a row in '{fk.Table}' was added for the row " + + $"of '{fk.ReferencedTable}' this transaction deleted."); + var children = FindChildRows(ends, key); if (children.Count == 0) continue; if (fk.CascadeDelete) { - Table child = _database.OpenTable(fk.Table); foreach (var (cid, cvals) in children) - if (scheduled.Add((child.Name, cid))) - stack.Push((child, cid, cvals, false)); + if (scheduled.Add((ends.Child.Name, cid))) + stack.Push((ends.Child, cid, cvals, false)); } else if (fk.DeleteSetNull) - foreach (var (cid, cvals) in children) SetChildKey(fk, cid, cvals, newKey: null); + foreach (var (cid, cvals) in children) + { + // A child this same statement is already deleting is left alone. Nulling it would rewrite a + // row whose values the delete is still holding — captured when it was scheduled — so the + // delete would then look its index entries up under a key the index no longer has + // ("entry not found"). Reachable on a self-referencing table, where the parent and the + // child are rows of one table and one DELETE can name both. Skipping costs nothing: the row + // and its entries go in a moment either way, and nobody can observe the intermediate null. + if (scheduled.Contains((ends.Child.Name, cid))) continue; + SetChildKey(ends, cid, cvals, newKey: null); + } else - throw new InvalidOperationException( - $"The record cannot be deleted or changed because table '{fk.Table}' includes related records."); + throw RelatedRecords(fk); } } /// Applies each enforced relationship's ON UPDATE action when a parent row's referenced key /// changes: CASCADE rewrites the children's FK to the new key; NO ACTION rejects if any child exists. /// (Jet has no ON UPDATE SET NULL.) - private void CascadeParentKeyUpdate(string parentTable, object?[] oldValues, object?[] newValues) + /// stillToWrite carries the rows a running UPDATE has yet to write: a cascade onto one of + /// them changes the values that write will carry instead of writing the row here (see SetChildKey). + private void CascadeParentKeyUpdate(string parentTable, object?[] oldValues, object?[] newValues, + Dictionary<(int Tdef, RowId Id), object?[]>? stillToWrite = null) { foreach (ForeignKey fk in ChildRelationshipsOf(parentTable)) { - object?[] oldKey = ReferencedKey(fk, oldValues); - object?[] newKey = ReferencedKey(fk, newValues); - if (oldKey.Any(k => k is null) || KeyEquals(oldKey, newKey)) continue; - var children = FindChildRows(fk, oldKey); + RelationshipEnds ends = EndsOf(fk); + object?[] oldKey = [.. ends.ParentColumns.Select(c => oldValues[c])]; + object?[] newKey = [.. ends.ParentColumns.Select(c => newValues[c])]; + if (oldKey.Any(k => k is null) || KeyEquals(oldKey, columns: null, newKey)) continue; + DependOnRelationship(fk, oldKey, + $"Transaction conflict on foreign key '{fk.Name}': a row in '{fk.Table}' was added for the key of " + + $"'{fk.ReferencedTable}' this transaction changed."); + var children = FindChildRows(ends, oldKey); if (children.Count == 0) continue; if (fk.CascadeUpdate) - foreach (var (cid, cvals) in children) SetChildKey(fk, cid, cvals, newKey); + foreach (var (cid, cvals) in children) SetChildKey(ends, cid, cvals, newKey, stillToWrite); else - throw new InvalidOperationException( - $"The record cannot be deleted or changed because table '{fk.Table}' includes related records."); + throw RelatedRecords(fk); } } - private static bool KeyEquals(object?[] a, object?[] b) + /// Whether holds — through + /// when the values are a whole row the key sits in columns of, position for + /// position when they are a key already. + /// Compared on the evaluator's terms rather than the CLR's, so referential integrity reaches the + /// same rows a query predicate over the same key would: a LONG 1 and a DOUBLE 1.0 are one key, and text is + /// compared in the database's collation. + private bool KeyEquals(object?[] values, int[]? columns, object?[] key) { - for (int i = 0; i < a.Length; i++) - if (ExpressionEvaluator.CompareForSort(a[i], b[i]) != 0) return false; + for (int i = 0; i < key.Length; i++) + if (ExpressionEvaluator.CompareForSort(values[columns is null ? i : columns[i]], key[i], _scalarRunner.TextComparer) != 0) + return false; return true; } - /// Rewrites a child row's FK columns to (or NULL for SET NULL), - /// maintaining any index over them. - private void SetChildKey(ForeignKey fk, RowId childId, object?[] childValues, object?[]? newKey) + /// Rewrites a child row's FK columns to (or NULL for SET NULL). + private void SetChildKey(RelationshipEnds ends, RowId childId, object?[] childValues, object?[]? newKey, + Dictionary<(int Tdef, RowId Id), object?[]>? stillToWrite = null) { - Table child = _database.OpenTable(fk.Table); var newValues = (object?[])childValues.Clone(); + for (int i = 0; i < ends.ChildColumns.Length; i++) + newValues[ends.ChildColumns[i]] = newKey?[i]; + + // A child row the running statement has yet to write itself: hand it the new key and let its own write + // carry it, so the row it writes and the index entries it moves both hold the cascaded value. Writing + // here as well would be undone by that write — it was built before this cascade ran — while the index + // entry moved here would keep the new key, which is a row disagreeing with its own index. + if (stillToWrite?.TryGetValue((ends.Child.Definition.DefinitionPage, childId), out object?[]? waiting) == true) + { + foreach (int idx in ends.ChildColumns) + if (!Equals(newValues[idx], childValues[idx])) waiting[idx] = newValues[idx]; + return; + } + + // A cascade rewrites a row, so it owes the row everything an UPDATE of it does — and the same code does it. + UpdateRow(ends.Child, childId, childValues, newValues, stillToWrite); + } + + /// + /// Rewrites one row whose values now hold — an UPDATE's own row, or a child a cascade + /// reaches — after checking it against every rule an updated row is held to, then applies the ON UPDATE rules + /// of any relationship whose key it changed. The one place a row is rewritten: a cascaded row is checked + /// exactly as a directly updated one, and a cascade that changes a key others reference carries on down. + /// + private void UpdateRow(Table table, RowId id, object?[] original, object?[] values, + Dictionary<(int Tdef, RowId Id), object?[]>? stillToWrite) + { var changed = new HashSet(); - for (int i = 0; i < fk.Columns.Count; i++) + for (int i = 0; i < values.Length; i++) + if (!Equals(original[i], values[i])) changed.Add(i); + if (changed.Count == 0) { - int idx = child.Definition.FindColumn(fk.Columns[i].Column)!.Index; - object? nv = newKey?[i]; - if (!Equals(nv, childValues[idx])) { newValues[idx] = nv; changed.Add(idx); } + stillToWrite?.Remove((table.Definition.DefinitionPage, id)); + return; // unchanged after all } - if (changed.Count == 0) return; - - // A cascade rewrites a row, so it owes the row the same invariants an UPDATE does. It used to apply - // none of them: ON DELETE SET NULL would write NULL into a column carrying the Required property, and - // ON UPDATE CASCADE could drive two children onto the same unique key — states the UPDATE path a few - // lines below explicitly refuses, reached by a statement that never names the child table. - // Table.Update carries no enforcement of its own (unlike Insert), so there is no backstop under this. - EnforceRequired(child.Name, child.Definition.Columns, newValues); - - foreach (IndexDef index in child.Definition.Indexes - .Where(i => i.IsUnique && i.RootPage > 0 && i.Columns.Any(c => changed.Contains(c.Column.Index))) - .GroupBy(i => i.RootPage).Select(g => g.First())) - if (!index.Columns.Any(c => newValues[c.Column.Index] is null) && child.HasDuplicateKey(index, newValues, childId)) + + // UPDATE must preserve the same Required/NOT NULL invariant as INSERT. Check the complete + // post-assignment row before any referential action, row rewrite, or index mutation occurs. + EnforceRequired(table.Name, table.Definition.Columns, values); + + // A changed primary-key or WITH DISALLOW NULL column must not become Null, the insert rule. + foreach (IndexDef index in table.Definition.Indexes + .Where(i => (i.IsPrimaryKey || i.Required) && i.Columns.Any(c => changed.Contains(c.Column.Index)))) + if (index.Columns.FirstOrDefault(c => values[c.Column.Index] is null).Column is { } nullColumn) + throw new InvalidOperationException( + $"Index or primary key cannot contain a Null value: column '{nullColumn.Name}' of " + + $"'{table.Name}' is Null, and index '{index.Name}' does not allow it."); + + // Child side: a changed FK column must still reference an existing parent (like an insert). + if (ForeignKeysOf(table.Name).Any(f => EndsOf(f).ChildColumns.Any(changed.Contains))) + EnforceReferentialIntegrity(table.Name, table, values, update: true); + + // A changed UNIQUE/PRIMARY key must not collide with another row (null keys are distinct — a + // unique index permits multiple nulls, so they're skipped, matching the insert rule). + foreach (IndexDef index in table.Definition.RealIndexes + .Where(i => i.IsUnique && i.Columns.Any(c => changed.Contains(c.Column.Index)))) + if (!index.Columns.Any(c => values[c.Column.Index] is null) && table.HasDuplicateKey(index, values, id)) throw new ConstraintViolationException( - $"Cannot cascade to '{child.Name}': a row with the same " + + $"Cannot update '{table.Name}': a row with the same " + $"{(index.IsPrimaryKey ? "primary key" : "unique key")} already exists (index '{index.Name}').", index.Name, index.IsPrimaryKey); - EnforceCheckConstraints(child.Definition, newValues); + // The updated row must still satisfy every CHECK constraint (evaluated against the full new row). + EnforceCheckConstraints(table.Definition, values); + + table.Update(id, values, changed); + stillToWrite?.Remove((table.Definition.DefinitionPage, id)); - child.Update(childId, newValues, changed); - foreach (IndexDef index in child.Definition.Indexes - .Where(i => i.RootPage > 0 && i.Columns.Any(c => changed.Contains(c.Column.Index))) - .GroupBy(i => i.RootPage).Select(g => g.First())) - child.MoveIndexEntry(index, childValues, newValues, childId); + // Parent side, once the row holds its new key: a changed referenced-key column triggers each relationship's + // ON UPDATE action (CASCADE rewrites the children, NO ACTION rejects if any exist). After the write, so a + // cascaded child's own foreign-key check finds the new key — and a row referencing itself is found as its + // own child and rewritten like any other. + CascadeParentKeyUpdate(table.Name, original, values, stillToWrite); } private int ExecuteCreateIndex(CreateIndexStatement statement) @@ -442,7 +623,8 @@ private int ExecuteCreateIndex(CreateIndexStatement statement) private int ExecuteCreateView(CreateViewStatement statement) { - _database.CreateView(statement.Name, BuildViewSpec(statement.Definition)); + // CREATE VIEW drops a leading space from the name, where every other route refuses one (verified vs ACE). + _database.CreateView(statement.Name.TrimStart(' '), BuildViewSpec(statement.Definition)); return 0; } @@ -468,7 +650,7 @@ private int ExecuteCreateProcedure(CreateProcedureStatement statement) ? null : parameters .Select(p => new ViewParameterSpec( - p.Name, + p.Stored ?? p.Name, // The declared size decides the type code as well as being stored: Text(50) is a Text // parameter (code 10) where a bare Text is a memo (12), which is what ACE records. (byte)AccessTypeMapper.ToColumnSpec( @@ -704,7 +886,30 @@ private int DropColumnDefault(string table, AlterColumnDropDefaultAction drop) private int AddForeignKey(string table, ForeignKeyConstraint fk) { - _database.AddForeignKey(table, ToRelationshipSpec(table, fk)); + RelationshipSpec relationship = ToRelationshipSpec(table, fk); + _database.AddForeignKey(table, relationship); + + // ACE refuses a relationship the existing rows already break — an orphan, or a partly null composite key + // (verified; the ON DELETE action makes no difference). Checked against the relationship as the catalog + // now holds it, a REFERENCES with no column list resolved to the parent's primary key; the statement's own + // rollback takes the relationship back out when the check fails. + ForeignKey added = _database.Catalog.Relationships.Single( + r => string.Equals(r.Name, relationship.Name, StringComparison.OrdinalIgnoreCase)); + RelationshipIdentity identity = Identify(added)!; + bool Holds() => CurrentRelationship(identity) is not { } current || ExistingRowsHaveParents(current); + if (!Holds()) + throw new InvalidOperationException( + "Cannot create relationships to enforce referential integrity. Existing data in table " + + $"'{added.Table}' violates referential integrity rules in table '{added.ReferencedTable}'."); + + // ACE then holds both tables exclusively until the transaction ends, so nothing can delete a parent the + // check found (verified: another connection's DELETE is refused). LibRed takes no table locks, so the + // same check is made again when the transaction commits, for as long as the relationship survives. + _database.DependOn( + Holds, + $"Transaction conflict on foreign key '{added.Name}': a row in '{added.ReferencedTable}' that the " + + $"existing rows of '{added.Table}' were checked against no longer exists.", + key: identity); return 0; } @@ -732,7 +937,8 @@ private int ExecuteCreateActionProcedure(CreateActionProcedureStatement statemen Values: statement.AppendColumns?.Select(c => new AppendColumnSpec(c.Column, c.ValueExpression)).ToList(), Body: statement.Body is { } body ? BuildViewSpec(body) : null, Parameters: BuildParameterSpecs(statement.Parameters), - DeleteTarget: statement.DeleteTarget); + DeleteTarget: statement.DeleteTarget, + OwnerAccess: statement.OwnerAccess); _database.CreateActionQuery(statement.Name, spec); return 0; @@ -750,7 +956,9 @@ private int ExecuteCreateActionProcedure(CreateActionProcedureStatement statemen d.Having, Parameters: null, OrderBy: d.OrderBy.Select(o => new ViewOrderBySpec(o.Expression, o.Descending)).ToList(), - Top: d.Top); + Top: d.Top, + TopPercent: d.TopPercent, + OwnerAccess: d.OwnerAccess); /// /// A make-table query: SELECT … INTO newtable FROM source. @@ -765,15 +973,15 @@ private int ExecuteCreateActionProcedure(CreateActionProcedureStatement statemen /// private int ExecuteSelectInto(SelectStatement statement) { + // A taken name — a table's, a query's or a linked table's — is refused by CreateTable, before anything is + // written; the read below has no side effect to undo. string target = statement.Into!; - if (_database.Catalog.Tables.Any(t => string.Equals(t.Name, target, StringComparison.OrdinalIgnoreCase))) - throw new SchemaObjectExistsException($"Table '{target}' already exists.", target); // Run the query first — with INTO stripped, or planning would recurse back into this method — and // materialise it. The rows have to exist before the table does: the source may read a table this // statement is about to change, and the row count is not known until the read completes. - ResultSet source = _scalarRunner.ExecuteQuery(Planning.IndexSelection.Apply( - Planning.QueryPlanner.PlanSelect(statement with { Into = null }), _database.Catalog)); + ResultSet source = _scalarRunner.ExecuteQuery(Planning.ColumnPruning.Apply(Planning.IndexSelection.Apply( + Planning.QueryPlanner.PlanSelect(statement with { Into = null }), _database.Catalog))); var rows = source.Rows.ToList(); // A result column that IS a source column keeps that column's DEFINITION — its type and its declared @@ -789,12 +997,15 @@ private int ExecuteSelectInto(SelectStatement statement) .ToList(); _database.CreateTable(target, specs); + // Written the way every inserted row is, so whatever the new table carries is checked as INSERT checks it. Table table = _database.OpenTable(target); + RowDefaults defaults = DefaultsOf(table.Definition); + var provided = Enumerable.Range(0, specs.Count).ToHashSet(); foreach (object?[] row in rows) { var values = new object?[specs.Count]; Array.Copy(row, values, Math.Min(row.Length, values.Length)); - table.Insert(values); + InsertNewRow(target, table, defaults, values, provided); } if (_session is not null) _session.RowCount = rows.Count; @@ -813,8 +1024,7 @@ private Dictionary SourceColumnsFor(SelectStatement statement { var available = new Dictionary(StringComparer.OrdinalIgnoreCase); foreach (string table in TablesIn(statement.From)) - if (_database.Catalog.Tables.FirstOrDefault( - t => string.Equals(t.Name, table, StringComparison.OrdinalIgnoreCase)) is { } def) + if (_database.Catalog.FindTable(table) is { } def) foreach (ColumnDef column in def.Columns) available.TryAdd(column.Name, column); @@ -874,7 +1084,7 @@ private sealed record RowDefaults(List<(int Index, Expression Expression)> Colum /// inserter (sequential counter, or a random Int32 for a GenUniqueID() "Random" AutoNumber), not by evaluating /// the DefaultValue — and GenUniqueID() is not a callable expression, so parsing it as a default would fail. /// - private RowDefaults DefaultsOf(TableDef definition) => new( + private RowDefaults DefaultsOf(TableDefinition definition) => new( definition.Columns .Where(c => c.DefaultValue is not null && !c.IsAutoNumber) .Select(c => (c.Index, Expression: ParseDefaultExpression(c.DefaultValue!))) @@ -940,6 +1150,11 @@ void InsertRow(ReadOnlySpan supplied) // would from INSERT INTO t DEFAULT VALUES. if (ReferenceEquals(supplied[i], DefaultRowValue)) continue; + // An explicit number is kept and the counter carries on after it; an explicit Null is refused + // (verified vs ACE) — to the row inserter a Null means "generate the next id". + if (supplied[i] is null && column.IsAutoNumber && column.Type == JetDataType.Int32) + throw new InvalidOperationException( + $"Cannot insert a Null value into AutoNumber field '{statement.Table}.{column.Name}'."); values[column.Index] = supplied[i]; provided.Add(column.Index); @@ -957,8 +1172,8 @@ void InsertRow(ReadOnlySpan supplied) // a table to itself otherwise feeds its own output back into the scan and never terminates. // Access's INSERT INTO t SELECT * FROM t doubles the table and stops, so the read completes // before the write begins. - ResultSet source = _scalarRunner.ExecuteQuery( - Planning.IndexSelection.Apply(Planning.QueryPlanner.PlanStatement(statement.Source), _database.Catalog)); + ResultSet source = _scalarRunner.ExecuteQuery(Planning.ColumnPruning.Apply( + Planning.IndexSelection.Apply(Planning.QueryPlanner.PlanStatement(statement.Source), _database.Catalog))); var rows = source.Rows.ToList(); // With no column list the source's output NAMES choose the target columns — ACE resolves by name, @@ -1104,7 +1319,7 @@ void EmitDerived(SubqueryTable sq, JoinKind kind, Expression? on) "Operation must use an updateable query: a derived table written to must select from one table, " + "with no grouping or DISTINCT."); var (columns, rows) = ExecuteDerivedSource(sq.Query, alias); - tables.Add(new SourceTable(alias, null, columns, rows)); + tables.Add(new SourceTable(alias, null, QueryExecutor.DerivedColumns(columns, alias, sq.Columns), rows)); kinds.Add(kind); ons.Add(on); groupBases.Add(null); @@ -1118,17 +1333,16 @@ void EmitLateral(SubqueryTable sq, JoinKind kind, Expression? on) // correlated predicate becomes a seek rather than a rescan, which matters here more than anywhere // because the body runs once per outer row. var outerColumns = tables.SelectMany(t => t.Columns).ToList(); - Plan.PlanNode plan = Planning.IndexSelection.Apply( + Plan.PlanNode plan = Planning.ColumnPruning.Apply(Planning.IndexSelection.Apply( Planning.QueryPlanner.PlanStatement(sq.Query), _database.Catalog, - tables.Select(t => t.Alias).ToHashSet(StringComparer.OrdinalIgnoreCase)); + tables.Select(t => t.Alias).ToHashSet(StringComparer.OrdinalIgnoreCase))); // The rows vary per outer row but the schema does not, and JoinRows needs it before reading any row. // Probe once against an all-null outer row and drop the rows unread, as QueryExecutor.ExecuteApply does. var (columns, _) = _scalarRunner.ExecuteCorrelated( plan, new EvalScope(outerColumns, new object?[outerColumns.Count], null)); - tables.Add(new SourceTable( - alias, null, columns.Select(c => c with { Qualifier = alias }).ToList(), null, plan)); + tables.Add(new SourceTable(alias, null, QueryExecutor.DerivedColumns(columns, alias, sq.Columns), null, plan)); kinds.Add(kind); ons.Add(on); groupBases.Add(null); @@ -1257,9 +1471,11 @@ void WalkGroup(JoinTable j, JoinTable group) /// private List? WritableDerivedTable(SubqueryTable sq) { - if (sq.Query is not SelectStatement + // A column list (which ACE has no syntax for) and WITH TIES leave it a query that is only read. + if (sq.Columns is not null + || sq.Query is not SelectStatement { - From: NamedTable or JoinTable, GroupBy.Count: 0, Having: null, Distinct: false, Into: null, + From: NamedTable or JoinTable, GroupBy.Count: 0, Having: null, Distinct: false, Into: null, WithTies: false, } select || select.Projection.Any(item => item.Value is StarExpression or QualifiedStarExpression && item.Alias is not null)) return null; @@ -1306,7 +1522,7 @@ ExpressionEvaluator Over(object?[] values) => { for (int i = 0; i < a.Length; i++) { - int order = ExpressionEvaluator.CompareForSort(a[i], b[i]); + int order = ExpressionEvaluator.CompareForSort(a[i], b[i], _scalarRunner.TextComparer); if (order != 0) return select.OrderBy[i].Direction == SortDirection.Descending ? -order : order; } @@ -1363,7 +1579,8 @@ ExpressionEvaluator Over(object?[] values) => // PlanStatement, not PlanSelect: EF puts a set operation here for Union/Except/Intersect/Concat before // an ExecuteUpdate, and a table value constructor for an inline collection. Same widening as the // subquery predicates needed, for the same reason - a derived table is a query expression. - var plan = Planning.IndexSelection.Apply(Planning.QueryPlanner.PlanStatement(query), _database.Catalog); + var plan = Planning.ColumnPruning.Apply( + Planning.IndexSelection.Apply(Planning.QueryPlanner.PlanStatement(query), _database.Catalog)); ResultSet result = _scalarRunner.ExecuteQuery(plan); var columns = result.ColumnNames.Select((name, i) => new OutputColumn(alias, name, result.ColumnTypes[i])).ToList(); return (columns, result.Rows.ToList()); @@ -1403,6 +1620,10 @@ ExpressionEvaluator Over(object?[] values) => if (tables.Count == 1) seekPlan[0] = SeekPlanFor(0, tables, where); + // Only the columns the statement names are decoded; the rows it then writes are read again in full by + // CompleteRows before anything is written. + bool[]?[] decode = [.. tables.Select(t => t.Table is { } table ? Planning.ColumnPruning.Mask(table.Definition, _readNames) : null)]; + // Precompute the accumulated columns visible when seeking/evaluating an ON at each depth. var colsUpTo = new List[tables.Count + 1]; colsUpTo[0] = []; @@ -1491,11 +1712,11 @@ void Recurse(int i) var keyEval = new ExpressionEvaluator(new EvalScope(colsUpTo[i], Flatten(acc, i), null), _scalarRunner, _parameters, _session); var keyValues = new object?[tables[i].Table!.Definition.Columns.Count]; keyValues[p.Index.Columns[0].Column.Index] = keyEval.Evaluate(p.Key); - rows = tables[i].Table!.SeekRowsWithIds(p.Index, keyValues); + rows = tables[i].Table!.SeekRowsWithIds(p.Index, keyValues, decode[i]); } else { - rows = tables[i].Table!.Rows().WithIds(); + rows = tables[i].Table!.Rows(decode[i]).WithIds(); } // The ON (also the seek's residual re-check — index keys can over-return) gates each candidate. @@ -1522,6 +1743,33 @@ void Recurse(int i) return result; } + // The column names the UPDATE or DELETE being executed reads anywhere (ON, WHERE, SET values, subqueries, a + // derived source), which is all JoinRows decodes; null decodes every column. + private HashSet? _readNames; + + /// + /// Reads each row the statement is about to write in full. JoinRows decoded only the columns the statement + /// names, which is all that choosing the rows and evaluating the SETs look at — but a row is rewritten, its + /// constraints checked and its index entries moved from all of its values. The rest are filled into the same + /// array, which every joined row of that physical row shares, before any of it is read. + /// + private void CompleteRows(List<(RowId Id, object?[] Values)[]> joinRows, List tables, IEnumerable targets) + { + int[] pruned = [.. targets.Distinct().Where(t => + tables[t].Table is { } table && Planning.ColumnPruning.Mask(table.Definition, _readNames) is not null)]; + if (pruned.Length == 0) return; + + var done = new HashSet(ReferenceEqualityComparer.Instance); + foreach (var combo in joinRows) + foreach (int ti in pruned) + { + if (IsNullExtended(combo[ti]) || !done.Add(combo[ti].Values)) continue; + object?[] full = tables[ti].Table!.GetRow(combo[ti].Id) + ?? throw new InvalidOperationException($"Row {combo[ti].Id} of '{tables[ti].Table!.Name}' vanished while it was being read."); + Array.Copy(full, combo[ti].Values, full.Length); + } + } + /// Concatenates the value arrays of the first accumulated join rows. private static object?[] Flatten((RowId, object?[])[] acc, int count) { @@ -1537,7 +1785,7 @@ private static (IndexDef Index, Expression Key)? SeekPlanFor(int i, List earlier = tables.Take(i).Select(t => t.Alias).ToHashSet(StringComparer.OrdinalIgnoreCase); @@ -1557,7 +1805,7 @@ private static IEnumerable Conjuncts(Expression e) => : [e]; private static (IndexDef Index, Expression Key)? MatchSeek( - Expression colSide, Expression keySide, string alias, TableDef def, HashSet earlier) + Expression colSide, Expression keySide, string alias, TableDefinition def, HashSet earlier) { if (colSide is not ColumnReference c || (c.Table is { } t && !string.Equals(t, alias, StringComparison.OrdinalIgnoreCase)) @@ -1616,6 +1864,19 @@ private static Table TargetTable(List tables, int ti) => /// Null). /// private int ExecuteUpdate(UpdateStatement statement) + { + _readNames = Planning.ColumnPruning.ReferencedNames(statement); + try + { + return Update(statement); + } + finally + { + _readNames = null; + } + } + + private int Update(UpdateStatement statement) { var (tables, kinds, ons, groupBases) = ResolveSource(statement.From); var columns = tables.SelectMany(t => t.Columns).ToList(); @@ -1637,12 +1898,19 @@ private int ExecuteUpdate(UpdateStatement statement) (int ti, int position) = found[0]; Table tt = TargetTable(tables, ti); ColumnDef col = tt.Definition.Columns[position]; + // An AutoNumber takes no UPDATE at all, even to its own value (verified vs ACE for the Int32 + // counter; a ReplicationID AutoNumber is not measured, so it is left alone). + if (col.IsAutoNumber && col.Type == JetDataType.Int32) + throw new InvalidOperationException( + $"Cannot update '{tt.Name}.{col.Name}': an AutoNumber field cannot be updated."); return (TableIndex: ti, Column: col, a.Value); }).ToList(); if (targets.GroupBy(t => (t.TableIndex, t.Column.Index)).FirstOrDefault(g => g.Count() > 1) is { } duplicate) throw new InvalidOperationException( $"Duplicate output destination '{tables[duplicate.Key.TableIndex].Alias}.{duplicate.First().Column.Name}'."); + CompleteRows(joinRows, tables, targets.Select(t => t.TableIndex)); + // Apply SETs to the shared value arrays; snapshot each touched row's original bytes on first touch. var dirty = new Dictionary<(string, RowId), (Table Table, RowId Id, object?[] Original, object?[] Values)>(); // A null-extended side's value array is the joined row's own, so it keys the new row it becomes. @@ -1673,50 +1941,26 @@ private int ExecuteUpdate(UpdateStatement statement) } } + // The rows this statement is still going to write, by table and row id. A cascade that reaches one of + // them must fold the new key into the values waiting here rather than write the row itself: the write + // waiting here was built from the snapshot taken before the cascade, so it would put the old key back + // while the index kept the cascaded one. A table whose parent and child ends are the same table is the + // shape that does this. An entry is dropped once its row has been written, after which a cascade + // reaching it writes the row as usual. + var stillToWrite = new Dictionary<(int Tdef, RowId Id), object?[]>(); + foreach (var (table, id, _, values) in dirty.Values) + stillToWrite[(table.Definition.DefinitionPage, id)] = values; + foreach (var (table, id, original, values) in dirty.Values) - { - var changed = new HashSet(); - for (int i = 0; i < values.Length; i++) - if (!Equals(original[i], values[i])) changed.Add(i); - if (changed.Count == 0) continue; // unchanged after all - - // UPDATE must preserve the same Required/NOT NULL invariant as INSERT. Check the complete - // post-assignment row before any referential action, row rewrite, or index mutation occurs. - EnforceRequired(table.Name, table.Definition.Columns, values); - - // Child side: a changed FK column must still reference an existing parent (like an insert). - if (_database.Catalog.ForeignKeysOf(table.Name).Any(f => f.IsEnforced && - f.Columns.Any(c => changed.Contains(table.Definition.FindColumn(c.Column)!.Index)))) - EnforceReferentialIntegrity(table.Name, table, values); - - // A changed UNIQUE/PRIMARY key must not collide with another row (null keys are distinct — a - // unique index permits multiple nulls, so they're skipped, matching the insert rule). - foreach (IndexDef index in table.Definition.Indexes - .Where(i => i.IsUnique && i.RootPage > 0 && i.Columns.Any(c => changed.Contains(c.Column.Index))) - .GroupBy(i => i.RootPage).Select(g => g.First())) - if (!index.Columns.Any(c => values[c.Column.Index] is null) && table.HasDuplicateKey(index, values, id)) - throw new ConstraintViolationException( - $"Cannot update '{table.Name}': a row with the same " + - $"{(index.IsPrimaryKey ? "primary key" : "unique key")} already exists (index '{index.Name}').", - index.Name, - index.IsPrimaryKey); - - // The updated row must still satisfy every CHECK constraint (evaluated against the full new row). - EnforceCheckConstraints(table.Definition, values); - - // Parent side: a changed referenced-key column triggers each relationship's ON UPDATE action - // (CASCADE rewrites children, NO ACTION rejects if children exist). - CascadeParentKeyUpdate(table.Name, original, values); - - table.Update(id, values, changed); - foreach (IndexDef index in table.Definition.Indexes - .Where(i => i.RootPage > 0 && i.Columns.Any(c => changed.Contains(c.Column.Index))) - .GroupBy(i => i.RootPage).Select(g => g.First())) - table.MoveIndexEntry(index, original, values, id); - } + UpdateRow(table, id, original, values, stillToWrite); + var newRowDefaults = new Dictionary(); foreach (var (values, (table, provided)) in newRows) - InsertNewRow(table.Name, table, DefaultsOf(table.Definition), values, provided); + { + if (!newRowDefaults.TryGetValue(table.Definition.DefinitionPage, out RowDefaults? defaults)) + newRowDefaults[table.Definition.DefinitionPage] = defaults = DefaultsOf(table.Definition); + InsertNewRow(table.Name, table, defaults, values, provided); + } int affected = joinRows.Count; if (_session is not null) _session.RowCount = affected; @@ -1735,6 +1979,19 @@ private int ExecuteUpdate(UpdateStatement statement) /// target without a row, which delete nothing but which ACE counts (verified). /// private int ExecuteDelete(DeleteStatement statement) + { + _readNames = Planning.ColumnPruning.ReferencedNames(statement); + try + { + return Delete(statement); + } + finally + { + _readNames = null; + } + } + + private int Delete(DeleteStatement statement) { var (tables, kinds, ons, groupBases) = ResolveSource(statement.From); var columns = tables.SelectMany(t => t.Columns).ToList(); @@ -1742,6 +1999,7 @@ private int ExecuteDelete(DeleteStatement statement) int ti = DeleteTarget(tables, statement.TargetTable); Table target = TargetTable(tables, ti); + CompleteRows(joinRows, tables, [ti]); var deleted = new Dictionary(); int withoutRow = 0; diff --git a/src/LibRed/LibRed.Engine/Execution/VbaEmpty.cs b/src/LibRed/LibRed.Engine/Execution/VbaEmpty.cs new file mode 100644 index 000000000..7f2ab15d2 --- /dev/null +++ b/src/LibRed/LibRed.Engine/Execution/VbaEmpty.cs @@ -0,0 +1,15 @@ +namespace LibRed.Engine.Execution; + +/// +/// VBA's Empty: what Access's one-argument Nz gives for a Null (verified vs Access). It is written out +/// as "", and read as 0 where a number is wanted and as "" where text is — so Nz(Null) + 2 is 2 +/// and Nz(Null) & 'x' is 'x'. Only Nz makes one. +/// +internal sealed class VbaEmpty +{ + public static readonly VbaEmpty Value = new(); + + private VbaEmpty() { } + + public override string ToString() => ""; +} diff --git a/src/LibRed/LibRed.Engine/Execution/WindowFunctions.cs b/src/LibRed/LibRed.Engine/Execution/WindowFunctions.cs index 0e3712f8b..24b917e96 100644 --- a/src/LibRed/LibRed.Engine/Execution/WindowFunctions.cs +++ b/src/LibRed/LibRed.Engine/Execution/WindowFunctions.cs @@ -186,7 +186,7 @@ internal static class WindowFunctions /// equal as GROUP BY takes values to be equal. private sealed class FrameAggregate(string name, WindowPartition p) { - private readonly RunningAggregate _aggregate = new(name, countRows: p.Call.Star, currency: p.Call.Currency); + private readonly RunningAggregate _aggregate = new(name, countRows: p.Call.Star, currency: p.Call.Currency, p.Text); private readonly HashSet? _seen = p.Call.Distinct ? [] : null; public object? Result => _aggregate.Result; @@ -201,7 +201,7 @@ public void Add(int position) return; } object? value = p.Call.Star ? null : p.Argument(position, 0); - if (_seen is not null && (value is null || !_seen.Add(new QueryExecutor.GroupKey([value])))) + if (_seen is not null && (value is null || !_seen.Add(new QueryExecutor.GroupKey([value], p.Text)))) return; _aggregate.Add(value); } @@ -224,7 +224,7 @@ public void Add(int position) o[i] = i > 0 && frame == previous && Equals(p.Argument(i, 0), p.Argument(i - 1, 0)) ? o[i - 1] : Percentile.Of(name, frame.Positions().Where(p.Includes).Select(k => p.Argument(k, 1)), - p.Argument(i, 0), p.Call.WithinGroup![0]); + p.Argument(i, 0), p.Call.WithinGroup![0], p.Text); previous = frame; } }, @@ -249,7 +249,7 @@ private static void ListAggOf(WindowPartition p, object?[] o) : ListAgg.Of( frame.Positions().Where(p.Includes).Select(k => (p.Argument(k, 0), Enumerable.Range(p.ArgumentCount - keys, keys).Select(a => p.Argument(k, a)).ToArray())), - separator, directions, p.Call.Distinct); + separator, directions, p.Call.Distinct, p.Text); previous = frame; } } diff --git a/src/LibRed/LibRed.Engine/Execution/WindowPartition.cs b/src/LibRed/LibRed.Engine/Execution/WindowPartition.cs index 51d307364..5e9aabe00 100644 --- a/src/LibRed/LibRed.Engine/Execution/WindowPartition.cs +++ b/src/LibRed/LibRed.Engine/Execution/WindowPartition.cs @@ -94,13 +94,18 @@ public IEnumerable Positions() /// Per row, the position the row's peer group starts at. /// Per row, the zero-based ordinal of the row's peer group within the partition. /// Per row, the already-evaluated arguments of the window call. +/// How text orders and compares in this database. /// How the call is written; null for a plain one. /// The frame clause; null for the default frame. /// Which rows an aggregate's FILTER lets in; null when it has none. internal sealed class WindowPartition( IReadOnlyList peerStart, IReadOnlyList peerOrdinal, IReadOnlyList arguments, + LibRed.Storage.JetTextComparer text, WindowCall? call = null, WindowFrameInput? frame = null, IReadOnlyList? included = null) { + /// How text orders and compares in this database, for the aggregates and lists computed here. + public LibRed.Storage.JetTextComparer Text { get; } = text; + /// Whether an aggregate takes in the row at : its FILTER holds there. public bool Includes(int position) => included is null || included[position]; diff --git a/src/LibRed/LibRed.Engine/LibRed.Engine.csproj b/src/LibRed/LibRed.Engine/LibRed.Engine.csproj index 75c320529..da2d8d098 100644 --- a/src/LibRed/LibRed.Engine/LibRed.Engine.csproj +++ b/src/LibRed/LibRed.Engine/LibRed.Engine.csproj @@ -1,7 +1,7 @@ - $(JetTargetFramework) + $(LibRedTargetFrameworks) LibRed.Engine LibRed.Engine Query engine: plans and executes bound SQL statements (select/insert/update/delete, joins, filters, expressions) against the LibRed.Core storage layer. @@ -13,6 +13,11 @@ + + + + + - - + diff --git a/test/LibRed.Core.Tests/LockManagerTests.cs b/test/LibRed.Core.Tests/LockManagerTests.cs index dc54f158c..9f48fcb4f 100644 --- a/test/LibRed.Core.Tests/LockManagerTests.cs +++ b/test/LibRed.Core.Tests/LockManagerTests.cs @@ -66,7 +66,7 @@ public void Locks_on_different_pages_do_not_block_each_other() public void Acquire_returns_one_shared_manager_per_path_and_frees_it_on_last_release() { MonitorLockManager a = MonitorLockManager.Acquire(@"C:\dir\db.accdb"); - MonitorLockManager b = MonitorLockManager.Acquire(@"C:\dir\DB.accdb"); // same file (case-insensitive) + MonitorLockManager b = MonitorLockManager.Acquire(@"C:\dir\db.accdb"); Assert.Same(a, b); // Two acquisitions → two releases; a fresh acquire after that is a new manager (the old was disposed). @@ -76,6 +76,27 @@ public void Acquire_returns_one_shared_manager_per_path_and_frees_it_on_last_rel MonitorLockManager.Release(@"C:\dir\db.accdb"); } + // Paths differing only in case name one file where file names are case-insensitive (Windows, macOS) and two on + // Linux, and the manager follows the file (FileIdentity): sharing one there would pool two files' pages. + [Fact] + public void Paths_differing_in_case_share_a_manager_only_where_file_names_ignore_case() + { + MonitorLockManager a = MonitorLockManager.Acquire(@"C:\dir\case.accdb"); + MonitorLockManager b = MonitorLockManager.Acquire(@"C:\dir\CASE.accdb"); + try + { + if (OperatingSystem.IsWindows() || OperatingSystem.IsMacOS()) + Assert.Same(a, b); + else + Assert.NotSame(a, b); + } + finally + { + MonitorLockManager.Release(@"C:\dir\case.accdb"); + MonitorLockManager.Release(@"C:\dir\CASE.accdb"); + } + } + [Fact] public void PageChannel_reads_and_writes_correctly_under_a_lock_manager() { diff --git a/test/LibRed.Core.Tests/LongValueCorruptionTests.cs b/test/LibRed.Core.Tests/LongValueCorruptionTests.cs index 13416a064..e6ce8adbc 100644 --- a/test/LibRed.Core.Tests/LongValueCorruptionTests.cs +++ b/test/LibRed.Core.Tests/LongValueCorruptionTests.cs @@ -1,5 +1,7 @@ using System.Buffers.Binary; using LibRed; +using LibRed.Formats; +using LibRed.IO; using LibRed.Pages; using LibRed.Storage; using Xunit; @@ -12,16 +14,15 @@ public class LongValueCorruptionTests public void Rejects_a_short_descriptor() { using var fixture = new Fixture(); - Assert.Throws(() => fixture.Reader.Resolve(new byte[11])); + Assert.Throws(() => + fixture.Reader.Resolve(new byte[fixture.Table.Channel.Format.LongValueDescriptorSize - 1])); } [Fact] public void Rejects_a_truncated_inline_value() { using var fixture = new Fixture(); - var descriptor = new byte[12]; - descriptor[0] = 1; - descriptor[3] = 0x80; + byte[] descriptor = LongValueStore.Descriptor(fixture.Table.Channel.Format, 1, LongValueStore.StorageKind.Inline); Assert.Throws(() => fixture.Reader.Resolve(descriptor)); } @@ -34,14 +35,15 @@ public void Rejects_a_truncated_inline_value() public void Rejects_a_malformed_chained_value(string corruption) { using var fixture = new Fixture(); - LongValueResult value = fixture.Writer.Write(new byte[5000]); + (byte[] Descriptor, IReadOnlyList OwnedPages, int FreePage) value = fixture.Writer.Write(new byte[5000]); byte[] descriptor = value.Descriptor.ToArray(); int first = value.OwnedPages[0]; switch (corruption) { case "outside-file": - WritePagePointer(descriptor.AsSpan(4, 4), fixture.Table.Channel.PageCount + 1); + PageBuffer.WriteRecordPointer(descriptor, fixture.Table.Channel.Format.LongValueDescriptorPointerOffset, + 0, fixture.Table.Channel.PageCount + 1); break; case "wrong-owner": byte[] wrongOwner = fixture.Table.Channel.ReadPage(first).Span.ToArray(); @@ -52,9 +54,8 @@ public void Rejects_a_malformed_chained_value(string corruption) break; case "short-chunk": byte[] shortChunk = fixture.Table.Channel.ReadPage(first).Span.ToArray(); - BinaryPrimitives.WriteUInt16LittleEndian( - shortChunk.AsSpan(fixture.Table.Channel.Format.DataRowDirectoryOffset, 2), - (ushort)(fixture.Table.Channel.PageSize - 3)); + DataPage.WriteSlot(shortChunk, fixture.Table.Channel.Format, 0, + fixture.Table.Channel.PageSize - (PageBuffer.RecordPointerSize - 1), RowSlotFlags.None); fixture.Table.Channel.WritePage(first, shortChunk); break; case "early-end": @@ -72,13 +73,13 @@ public void Rejects_a_malformed_chained_value(string corruption) public void Valid_inline_single_and_chained_values_round_trip() { using var fixture = new Fixture(); - byte[] inline = [3, 0, 0, 0x80, 0, 0, 0, 0, 0, 0, 0, 0, 1, 2, 3]; + byte[] inline = [.. LongValueStore.Descriptor(fixture.Table.Channel.Format, 3, LongValueStore.StorageKind.Inline), 1, 2, 3]; Assert.Equal(new byte[] { 1, 2, 3 }, fixture.Reader.Resolve(inline)); foreach (int size in new[] { 100, 5000 }) { byte[] payload = Enumerable.Range(0, size).Select(i => (byte)i).ToArray(); - LongValueResult stored = fixture.Writer.Write(payload); + (byte[] Descriptor, IReadOnlyList OwnedPages, int FreePage) stored = fixture.Writer.Write(payload); Assert.Equal(payload, fixture.Reader.Resolve(stored.Descriptor)); } } @@ -88,19 +89,11 @@ private static void RewriteNextPointer(Fixture fixture, int pageNumber, int next byte[] page = fixture.Table.Channel.ReadPage(pageNumber).Span.ToArray(); var parsed = new DataPage(); parsed.Read(fixture.Table.Channel.ReadPage(pageNumber), fixture.Table.Channel.Format); - RowSlot slot = parsed.Rows[0]; - WritePagePointer(page.AsSpan(slot.Offset, 4), nextPage); + DataPage.RowSlot slot = parsed.Rows[0]; + PageBuffer.WriteRecordPointer(page, slot.Offset, 0, nextPage); fixture.Table.Channel.WritePage(pageNumber, page); } - private static void WritePagePointer(Span pointer, int page) - { - pointer[0] = 0; - pointer[1] = (byte)page; - pointer[2] = (byte)(page >> 8); - pointer[3] = (byte)(page >> 16); - } - private sealed class Fixture : IDisposable { private readonly string _path = TemporaryDatabase.CopyPath(TestDatabases.NorthwindAccdb, "lval-corrupt-"); @@ -110,13 +103,13 @@ public Fixture() { _database = JetDatabase.Open(_path, readOnly: false); Table = _database.OpenTable("Categories"); - Writer = new LongValueWriter(Table.Channel); - Reader = new LongValueReader(Table.Channel); + Writer = new LongValueStore(Table.Channel); + Reader = new LongValueStore(Table.Channel); } public Table Table { get; } - public LongValueWriter Writer { get; } - public LongValueReader Reader { get; } + public LongValueStore Writer { get; } + public LongValueStore Reader { get; } public void Dispose() { @@ -124,4 +117,4 @@ public void Dispose() TemporaryDatabase.Delete(_path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/MultiPageDefinitionTests.cs b/test/LibRed.Core.Tests/MultiPageDefinitionTests.cs index d3be3683f..ce49b9a7b 100644 --- a/test/LibRed.Core.Tests/MultiPageDefinitionTests.cs +++ b/test/LibRed.Core.Tests/MultiPageDefinitionTests.cs @@ -1,6 +1,7 @@ using LibRed; using LibRed.Catalog; using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Core.Tests; @@ -29,9 +30,9 @@ public void Index_that_overflows_the_tdef_page_spills_to_a_continuation_and_roun using (var ch = PageChannel.Open(path, readOnly: true)) { var def = new JetCatalog(ch).FindTable("Wide")!; - var buf = ch.ReadPage(def.DefinitionPage); - Assert.True(buf.ReadInt32(0x08) > ch.Format.PageSize); // definition exceeds one page - Assert.NotEqual(0, buf.ReadInt32(ch.Format.TdefNextPageOffset)); // a continuation page exists + (PageBuffer definition, IReadOnlyList continuations) = TableDefinition.ReadChain(ch, def.DefinitionPage); + Assert.True(definition.Length > ch.Format.PageSize); // definition exceeds one page + Assert.NotEmpty(continuations); // a continuation page exists } using (var db = JetDatabase.Open(path)) @@ -70,7 +71,8 @@ public void A_definition_near_a_page_boundary_spills_its_reserve_and_reads_back( using (var ch = PageChannel.Open(path, readOnly: true)) { var chainFree = new List(); - for (int page = new JetCatalog(ch).FindTable("L")!.DefinitionPage; page != 0; page = ch.ReadPage(page).ReadInt32(ch.Format.TdefNextPageOffset)) + int first = new JetCatalog(ch).FindTable("L")!.DefinitionPage; + foreach (int page in (int[])[first, .. TableDefinition.ReadChain(ch, first).ContinuationPages]) chainFree.Add(ch.ReadPage(page).ReadUInt16(ch.Format.TdefFreeSpaceOffset)); Assert.Equal(free, chainFree); } @@ -83,4 +85,4 @@ public void A_definition_near_a_page_boundary_spills_its_reserve_and_reads_back( } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/MultiPageTableDefinitionTests.cs b/test/LibRed.Core.Tests/MultiPageTableDefinitionTests.cs index 92b9704e8..33d8e44be 100644 --- a/test/LibRed.Core.Tests/MultiPageTableDefinitionTests.cs +++ b/test/LibRed.Core.Tests/MultiPageTableDefinitionTests.cs @@ -61,15 +61,14 @@ public void Rejects_an_invalid_continuation_chain_before_assembly(string corrupt var channel = table.Channel; int firstPage = table.Definition.DefinitionPage; byte[] first = channel.ReadPage(firstPage).Span.ToArray(); - int continuation = BinaryPrimitives.ReadInt32LittleEndian( - first.AsSpan(db.Format.TdefNextPageOffset, 4)); + int continuation = db.ReadTableDefinition(firstPage).NextDefinitionPage; Assert.True(continuation > 0); switch (corruption) { case "wrong-type": byte[] wrongType = channel.ReadPage(continuation).Span.ToArray(); - wrongType[0] = (byte)PageType.DataPage; + PageHeader.WriteType(wrongType, PageType.DataPage); channel.WritePage(continuation, wrongType); break; case "outside-file": @@ -108,4 +107,4 @@ public void Rejects_an_invalid_continuation_chain_before_assembly(string corrupt } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/NameMapTests.cs b/test/LibRed.Core.Tests/NameMapTests.cs new file mode 100644 index 000000000..5ffb88a10 --- /dev/null +++ b/test/LibRed.Core.Tests/NameMapTests.cs @@ -0,0 +1,195 @@ +using LibRed; +using LibRed.Catalog; +using LibRed.Formats; +using Xunit; + +namespace LibRed.Core.Tests; + +// Access's Name AutoCorrect map, in both of the layouts Access keeps it in: the NameMap column of an MSysNameMap +// row and the NameMap property in the object's LvProp. The engine never maintains either; these pin that a caller +// can read and edit them, and that a map read and written back is unchanged down to the bytes nobody understands. +public class NameMapTests +{ + // An Access-written table's MSysNameMap blob: Table1, then its columns ID (Long) and Field1 (Memo). The last + // four fixed bytes of each record are uninitialised memory — 0 on Table1's, A6 on the others. + private static readonly byte[] Row = Convert.FromHexString( + "05000000300000003E0000000000000085276F677377BE40AC361424EE65776D000000004BDB0DFF959AE6400000000000000000" + + "01000000000000005400610062006C006500310000003600000000000000E51D15FE90C25B49920D5E7939D2700807000000" + + "85276F677377BE40AC361424EE65776D04000000A60000004900440000003E000000000000009BFDAC9F26E6A3458870B1AF" + + "8834DED60700000085276F677377BE40AC361424EE65776D0C000000A60000004600690065006C00640031000000"); + + // The same table's NameMap property: the same three records in the property layout, closed by a kind-12 + // record carrying version 5. + private static readonly byte[] Property = Convert.FromHexString( + "0ACC0E550000000085276F677377BE40AC361424EE65776D000000004BDB0DFF959AE64000000000000000005400610062006C00" + + "65003100000000000000E51D15FE90C25B49920D5E7939D270080700000085276F677377BE40AC361424EE65776D4900440000" + + "00000000009BFDAC9F26E6A3458870B1AF8834DED60700000085276F677377BE40AC361424EE65776D4600690065006C006400" + + "3100000000000000000000000000000000000000000000000C000000050000000000000000000000000000000000"); + + private static readonly Guid Table1 = Guid.Parse("676f2785-7773-40be-ac36-1424ee65776d"); + + // An Access-written file with one MSysNameMap row and a NameMap property, both for its Table1. Its NameMap + // column, like every Access-written one, has an owned-pages map and no free-pages map. + private static readonly string Fixture = Path.Combine(AppContext.BaseDirectory, "Data", "Hungarian.accdb"); + + [Fact] + public void A_row_map_reads_as_its_records_and_writes_back_unchanged() + { + NameMap map = NameMap.ReadRow(Row); + + Assert.Equal(5, map.Version); + Assert.Equal(new[] { "Table1", "ID", "Field1" }, map.Records.Select(r => r.Name)); + Assert.Equal(new[] { 0, 7, 7 }, map.Records.Select(r => r.Kind)); + Assert.Equal(new[] { 1, 4, 12 }, map.Records.Select(r => r.TypeCode)); + Assert.Equal(new[] { 0, 0xA6, 0xA6 }, map.Records.Select(r => r.Unused)); + Assert.Equal(Table1, map.Records[0].ItemGuid); + Assert.NotNull(map.Records[0].SlotDate); + Assert.All(map.Records.Skip(1), r => Assert.Equal(Table1, r.SlotGuid)); + Assert.Equal(Row, map.WriteRow()); + } + + [Fact] + public void A_property_map_reads_as_the_same_records_and_writes_back_unchanged() + { + NameMap property = NameMap.ReadProperty(Property); + NameMap row = NameMap.ReadRow(Row); + + Assert.Equal(5, property.Version); + Assert.Equal( + row.Records.Select(r => (r.ItemGuid, r.Kind, Convert.ToHexString(r.Slot), r.Name)), + property.Records.Select(r => (r.ItemGuid, r.Kind, Convert.ToHexString(r.Slot), r.Name))); + Assert.All(property.Records, r => Assert.Equal(0, r.TypeCode)); + Assert.Equal(Property, property.WriteProperty()); + } + + // A null GUID is an ordinary record in the property layout; only the kind closes the list. + [Fact] + public void A_null_guid_record_does_not_end_a_property_map() + { + NameMap map = new() + { + Records = [new NameMap.Entry(Guid.Empty, 0, "Unresolved"), new NameMap.Entry(Table1, 0, "Table1")], + }; + + NameMap read = NameMap.ReadProperty(map.WriteProperty()); + + Assert.Equal(new[] { "Unresolved", "Table1" }, read.Records.Select(r => r.Name)); + Assert.Equal(NameMap.RowVersion, read.Version); + } + + [Fact] + public void An_edited_map_keeps_every_record_it_did_not_touch() + { + NameMap map = NameMap.ReadRow(Row); + NameMap renamed = map with + { + Records = [.. map.Records.Select(r => r.Name == "Field1" ? r with { Name = "Notes" } : r)], + }; + + NameMap read = NameMap.ReadRow(renamed.WriteRow()); + + Assert.Equal(new[] { "Table1", "ID", "Notes" }, read.Records.Select(r => r.Name)); + Assert.Equal(map.Records.Take(2), read.Records.Take(2), NameMapRecordComparer.Instance); + Assert.Equal(0xA6, read.Records[2].Unused); + NameMap property = NameMap.ReadProperty(renamed.WriteProperty()); + Assert.Equal("Notes", property.Records[2].Name); + } + + [Fact] + public void A_blob_in_neither_layout_is_refused() + { + Assert.Throws(() => NameMap.ReadRow(Property)); + Assert.Throws(() => NameMap.ReadProperty(Row)); + } + + [Fact] + public void A_database_s_name_map_row_can_be_read_and_updated() + { + string path = TemporaryDatabase.CopyPath(Fixture, "namemap-row-"); + try + { + NameMap.CatalogRow row; + using (var db = JetDatabase.Open(path, readOnly: false)) + { + row = Assert.Single(db.ReadNameMaps()); + Assert.Equal("Table1", row.Name); + Assert.Equal(1, row.Type); + NameMap map = Assert.IsType(row.Map); + Assert.Equal("Table1", map.Records[0].Name); + + NameMap renamed = map with + { + Records = [map.Records[0] with { Name = "Renamed" }, .. map.Records.Skip(1)], + }; + Assert.True(db.UpdateNameMap(row.ObjectGuid, renamed, "Renamed")); + Assert.False(db.UpdateNameMap(Guid.NewGuid(), renamed)); + } + + using (var db = JetDatabase.Open(path, readOnly: true)) + { + NameMap.CatalogRow after = Assert.Single(db.ReadNameMaps()); + Assert.Equal("Renamed", after.Name); + Assert.Equal(row.Id, after.Id); + Assert.Equal("Renamed", after.Map!.Records[0].Name); + Assert.Equal(row.Map!.Records.Skip(1), after.Map.Records.Skip(1), NameMapRecordComparer.Instance); + } + } + finally { TemporaryDatabase.Delete(path); } + } + + [Fact] + public void A_table_s_name_map_property_can_be_read_updated_and_removed_without_touching_the_others() + { + string path = TemporaryDatabase.CopyPath(Fixture, "namemap-prop-"); + try + { + NameMap original; + using (var db = JetDatabase.Open(path, readOnly: false)) + { + original = Assert.IsType(db.ReadNameMapProperty("Table1", ObjectType.Table)); + Assert.Equal("Table1", original.Records[0].Name); + NameMap renamed = original with + { + Records = [original.Records[0] with { Name = "Renamed" }, .. original.Records.Skip(1)], + }; + db.UpdateNameMapProperty("Table1", ObjectType.Table, renamed); + Assert.Throws(() => db.ReadNameMapProperty("NoSuchTable", ObjectType.Table)); + } + + using (var db = JetDatabase.Open(path, readOnly: false)) + { + NameMap after = Assert.IsType(db.ReadNameMapProperty("Table1", ObjectType.Table)); + Assert.Equal("Renamed", after.Records[0].Name); + Assert.Equal(original.Version, after.Version); + Assert.NotNull(TableProperty(db, "GUID")); + + db.UpdateNameMapProperty("Table1", ObjectType.Table, null); + Assert.Null(db.ReadNameMapProperty("Table1", ObjectType.Table)); + Assert.NotNull(TableProperty(db, "GUID")); + } + } + finally { TemporaryDatabase.Delete(path); } + } + + private static PropertyBlob.Property? TableProperty(JetDatabase db, string name) + { + var objects = db.OpenTable("MSysObjects"); + var d = objects.Definition; + object?[] row = objects.Rows().Single(r => (string)r[d.FindColumn("Name")!.Index]! == "Table1"); + return PropertyBlob.Read((byte[])row[d.FindColumn("LvProp")!.Index]!) + .Cast().FirstOrDefault(p => p!.Value.IsOwnedBy("") && p.Value.Name == name); + } + + // Records hold their slot as an array, so record equality compares it by reference. + private sealed class NameMapRecordComparer : IEqualityComparer + { + public static readonly NameMapRecordComparer Instance = new(); + + public bool Equals(NameMap.Entry? x, NameMap.Entry? y) => + x is not null && y is not null && x with { Slot = NoSlot } == y with { Slot = NoSlot } && x.Slot.AsSpan().SequenceEqual(y.Slot); + + private static readonly byte[] NoSlot = []; + + public int GetHashCode(NameMap.Entry obj) => obj.ItemGuid.GetHashCode(); + } +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/OfficeStandardEncryptionTests.cs b/test/LibRed.Core.Tests/OfficeStandardEncryptionTests.cs index ff64e46b9..a3229fae0 100644 --- a/test/LibRed.Core.Tests/OfficeStandardEncryptionTests.cs +++ b/test/LibRed.Core.Tests/OfficeStandardEncryptionTests.cs @@ -1,5 +1,6 @@ using System.Buffers.Binary; using LibRed.Crypto; +using LibRed.Formats; using Xunit; namespace LibRed.Core.Tests; @@ -10,6 +11,8 @@ namespace LibRed.Core.Tests; // just the binary EncryptionInfo descriptor exercises the whole key-derivation + verifier path without the DB. public class OfficeStandardEncryptionTests { + private static readonly JetFormatBase Format = JetFormatBase.FromVersionByte(0x02); // ACE 12 + // db2007-oldenc.accdb — RC4-40, password "Test123" private const uint AlgRc4 = 0x6801; private static readonly byte[] Rc4Salt = Convert.FromHexString("78da7d5196c71492eed3b4471a479449"); @@ -24,25 +27,25 @@ public class OfficeStandardEncryptionTests [Fact] public void Rc4_authenticates_correct_password() => - Assert.NotNull(OfficeStandardEncryption.TryCreate(BuildPage0(AlgRc4, 0x04, 40, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash), 0x12345678, "Test123")); + Assert.NotNull(OfficeStandardEncryption.TryCreate(BuildPage0(AlgRc4, 0x04, 40, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash), 0x12345678, "Test123", Format)); [Fact] public void Aes_authenticates_correct_password() => - Assert.NotNull(OfficeStandardEncryption.TryCreate(BuildPage0(AlgAes256, 0x0C, 256, AesSalt, AesEncVerifier, AesEncVerifierHash), 0x12345678, "password")); + Assert.NotNull(OfficeStandardEncryption.TryCreate(BuildPage0(AlgAes256, 0x0C, 256, AesSalt, AesEncVerifier, AesEncVerifierHash), 0x12345678, "password", Format)); [Fact] public void Wrong_password_throws() => Assert.Throws(() => - OfficeStandardEncryption.TryCreate(BuildPage0(AlgAes256, 0x0C, 256, AesSalt, AesEncVerifier, AesEncVerifierHash), 0x12345678, "wrong")); + OfficeStandardEncryption.TryCreate(BuildPage0(AlgAes256, 0x0C, 256, AesSalt, AesEncVerifier, AesEncVerifierHash), 0x12345678, "wrong", Format)); [Fact] public void Missing_password_throws() => Assert.Throws(() => - OfficeStandardEncryption.TryCreate(BuildPage0(AlgRc4, 0x04, 40, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash), 0x12345678, null)); + OfficeStandardEncryption.TryCreate(BuildPage0(AlgRc4, 0x04, 40, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash), 0x12345678, null, Format)); [Fact] public void Unencrypted_returns_null() => - Assert.Null(OfficeStandardEncryption.TryCreate(new byte[4096], databaseKey: 0, password: "x")); + Assert.Null(OfficeStandardEncryption.TryCreate(new byte[Format.PageSize], databaseKey: 0, password: "x", Format)); [Theory] [InlineData(0)] @@ -52,7 +55,7 @@ public void Verifier_hash_size_must_exactly_match_the_declared_hash(int verifier { Assert.Throws(() => OfficeStandardEncryption.TryCreate( BuildPage0(AlgRc4, 0x04, 40, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash, verifierHashSize), - 0x12345678, "Test123")); + 0x12345678, "Test123", Format)); } [Theory] @@ -65,7 +68,7 @@ public void Rc4_key_size_must_be_a_supported_whole_byte_count(int keyBits) { Assert.Throws(() => OfficeStandardEncryption.TryCreate( BuildPage0(AlgRc4, 0x04, keyBits, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash), - 0x12345678, "Test123")); + 0x12345678, "Test123", Format)); } [Fact] @@ -73,7 +76,7 @@ public void Aes_key_size_must_be_a_supported_exact_size() { Assert.Throws(() => OfficeStandardEncryption.TryCreate( BuildPage0(AlgAes256, 0x0C, 257, AesSalt, AesEncVerifier, AesEncVerifierHash), - 0x12345678, "password")); + 0x12345678, "password", Format)); } [Theory] @@ -82,10 +85,10 @@ public void Aes_key_size_must_be_a_supported_exact_size() public void Encrypt_then_decrypt_round_trips(bool rc4) { var codec = rc4 - ? OfficeStandardEncryption.TryCreate(BuildPage0(AlgRc4, 0x04, 40, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash), 0x12345678, "Test123")! - : OfficeStandardEncryption.TryCreate(BuildPage0(AlgAes256, 0x0C, 256, AesSalt, AesEncVerifier, AesEncVerifierHash), 0x12345678, "password")!; + ? OfficeStandardEncryption.TryCreate(BuildPage0(AlgRc4, 0x04, 40, Rc4Salt, Rc4EncVerifier, Rc4EncVerifierHash), 0x12345678, "Test123", Format)! + : OfficeStandardEncryption.TryCreate(BuildPage0(AlgAes256, 0x0C, 256, AesSalt, AesEncVerifier, AesEncVerifierHash), 0x12345678, "password", Format)!; - var page = new byte[4096]; + var page = new byte[Format.PageSize]; new Random(5).NextBytes(page); var original = (byte[])page.Clone(); codec.EncryptPage(4, page); @@ -101,8 +104,8 @@ private static byte[] BuildPage0( uint algId, uint flags, int keyBits, byte[] salt, byte[] encVerifier, byte[] encVerifierHash, int verifierHashSize = 20) { - var page = new byte[4096]; - const int ei = 0x29B; + var page = new byte[Format.PageSize]; + int ei = Format.EncryptionInfoOffset; const int headerSize = 32; void U16(int o, ushort v) => BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(o), v); void U32(int o, uint v) => BinaryPrimitives.WriteUInt32LittleEndian(page.AsSpan(o), v); @@ -122,7 +125,7 @@ private static byte[] BuildPage0( U32(v + 4 + salt.Length + 16, (uint)verifierHashSize); // VerifierHashSize (SHA1 normally 20) encVerifierHash.CopyTo(page, v + 4 + salt.Length + 16 + 4); int descriptorLength = v + 4 + salt.Length + 16 + 4 + encVerifierHash.Length - ei; - U16(0x299, checked((ushort)descriptorLength)); + U16(Format.EncryptionInfoLengthOffset, checked((ushort)descriptorLength)); return page; } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/OfficeStandardVariantReadTests.cs b/test/LibRed.Core.Tests/OfficeStandardVariantReadTests.cs index 365820966..0b499ee60 100644 --- a/test/LibRed.Core.Tests/OfficeStandardVariantReadTests.cs +++ b/test/LibRed.Core.Tests/OfficeStandardVariantReadTests.cs @@ -1,6 +1,7 @@ using System.Buffers.Binary; using LibRed; using LibRed.Crypto; +using LibRed.Formats; using Xunit; namespace LibRed.Core.Tests; @@ -8,9 +9,9 @@ namespace LibRed.Core.Tests; /// Fixture-free rejection coverage for unsupported Office-Standard descriptor variants. public class OfficeStandardVariantReadTests { - private const int DescriptorOffset = 0x29B; - private const int AlgorithmOffset = DescriptorOffset + 12 + 8; - private const int HashOffset = DescriptorOffset + 12 + 12; + private static readonly JetFormatBase Format = TestDatabases.FormatOf(TestDatabases.WideTableAccdb); + private static readonly int AlgorithmOffset = Format.EncryptionInfoOffset + 12 + 8; + private static readonly int HashOffset = Format.EncryptionInfoOffset + 12 + 12; [Theory] [InlineData(0x6603u)] // 3DES-168 @@ -56,7 +57,7 @@ public void A_zero_length_descriptor_is_read_as_unencrypted_like_ace() string path = CreateEncryptedCopy(); try { - MutateUInt16(path, 0x299, 0); + MutateUInt16(path, Format.EncryptionInfoLengthOffset, 0); var error = Assert.Throws( () => JetDatabase.Open(path, readOnly: true, password: "Test123")); Assert.Contains("unsupported scheme", error.Message, StringComparison.OrdinalIgnoreCase); @@ -67,7 +68,8 @@ public void A_zero_length_descriptor_is_read_as_unencrypted_like_ace() private static string CreateEncryptedCopy() { string path = TemporaryDatabase.CopyPath(TestDatabases.WideTableAccdb, "office-standard-variant-"); - DatabaseEncryption.SetPasswordRc4(path, "Test123"); + using (var db = JetDatabase.Open(path, readOnly: false, exclusive: true)) + DatabaseEncryption.SetPasswordRc4(db, "Test123"); return path; } @@ -84,4 +86,4 @@ private static void MutateUInt16(string path, int offset, ushort value) BinaryPrimitives.WriteUInt16LittleEndian(file.AsSpan(offset, 2), value); File.WriteAllBytes(path, file); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/PageAndRowCorruptionTests.cs b/test/LibRed.Core.Tests/PageAndRowCorruptionTests.cs index 1bfe4924a..0bebc525c 100644 --- a/test/LibRed.Core.Tests/PageAndRowCorruptionTests.cs +++ b/test/LibRed.Core.Tests/PageAndRowCorruptionTests.cs @@ -15,6 +15,7 @@ public class PageAndRowCorruptionTests [Theory] [InlineData("wrong-page-type")] + [InlineData("wrong-page-type-high-byte")] [InlineData("short-page")] [InlineData("directory-past-page")] [InlineData("row-overlaps-directory")] @@ -26,7 +27,12 @@ public void Malformed_data_page_slots_are_rejected_as_corruption(string corrupti switch (corruption) { case "wrong-page-type": - page[0] = (byte)PageType.TableDefinition; + PageHeader.WriteType(page, PageType.TableDefinition); + break; + case "wrong-page-type-high-byte": + // The type is the whole word: 01 02 is not a data page, though its low byte is. ACE skips such a + // page rather than reading it as rows. + page[1] = 0x02; break; case "short-page": page = page[..100]; @@ -35,14 +41,14 @@ public void Malformed_data_page_slots_are_rejected_as_corruption(string corrupti BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowCountOffset, 2), ushort.MaxValue); break; case "row-overlaps-directory": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowDirectoryOffset, 2), - (ushort)(Format.DataRowDirectoryOffset + 2)); + DataPage.WriteSlot(page, Format, 0, + Format.DataRowDirectoryOffset + Format.DataRowDirectoryEntrySize, RowSlotFlags.None); break; case "row-offset-past-page": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowDirectoryOffset, 2), 5000); + DataPage.WriteSlot(page, Format, 0, 5000, RowSlotFlags.None); break; case "ascending-row-offsets": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowDirectoryOffset + 2, 2), 4050); + DataPage.WriteSlot(page, Format, 1, 4050, RowSlotFlags.None); break; } @@ -62,13 +68,12 @@ public void Direct_row_seek_applies_the_same_slot_validation() public void Zero_length_deleted_overflow_tombstone_remains_a_valid_slot_shape() { byte[] page = NewDataPage(rowCount: 2, firstOffset: 4000, secondOffset: 4000); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowDirectoryOffset + 2, 2), - (ushort)(4000 | 0x8000 | 0x4000)); + DataPage.WriteSlot(page, Format, 1, 4000, RowSlotFlags.Deleted | RowSlotFlags.Overflow); var dataPage = new DataPage(); dataPage.Read(new PageBuffer(page, 7), Format); - Assert.Equal(new RowSlot(4000, 0, IsDeleted: true, HasOverflow: true), dataPage.Rows[1]); + Assert.Equal(new DataPage.RowSlot(4000, 0, IsDeleted: true, HasOverflow: true), dataPage.Rows[1]); } [Fact] @@ -77,7 +82,7 @@ public void All_fixed_row_does_not_parse_fixed_bytes_as_a_variable_trailer() ColumnDef column = FixedColumn(JetDataType.Int32, length: 4); byte[] row = [1, 0, 0xFF, 0xFF, 0xFF, 0x7F, 1]; - object?[] values = new RowDecoder([column], Format).Decode(row); + object?[] values = new RowCodec([column], Format).Decode(row); Assert.Equal(int.MaxValue, values[0]); } @@ -113,7 +118,7 @@ public void Malformed_rows_are_rejected_as_corruption(string corruption) }; } - Assert.Throws(() => new RowDecoder([column], Format).Decode(row)); + Assert.Throws(() => new RowCodec([column], Format).Decode(row)); } [Fact] @@ -122,7 +127,7 @@ public void Column_added_after_an_old_row_decodes_as_null_when_its_bitmap_bit_do ColumnDef column = FixedColumn(JetDataType.Int32, length: 4, columnId: 8); byte[] oldRow = [1, 0, 0]; - object?[] values = new RowDecoder([column], Format).Decode(oldRow); + object?[] values = new RowCodec([column], Format).Decode(oldRow); Assert.Null(values[0]); } @@ -137,7 +142,7 @@ public void First_variable_column_added_after_an_old_all_fixed_row_decodes_as_nu }; byte[] oldRow = [1, 0, 0]; - object?[] values = new RowDecoder([column], Format).Decode(oldRow); + object?[] values = new RowCodec([column], Format).Decode(oldRow); Assert.Null(values[0]); } @@ -145,11 +150,10 @@ public void First_variable_column_added_after_an_old_all_fixed_row_decodes_as_nu private static byte[] NewDataPage(int rowCount, int firstOffset, int secondOffset) { var page = new byte[Format.PageSize]; - page[0] = (byte)PageType.DataPage; - page[1] = 1; + PageHeader.WriteType(page, PageType.DataPage); BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowCountOffset, 2), (ushort)rowCount); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowDirectoryOffset, 2), (ushort)firstOffset); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.DataRowDirectoryOffset + 2, 2), (ushort)secondOffset); + DataPage.WriteSlot(page, Format, 0, firstOffset, RowSlotFlags.None); + DataPage.WriteSlot(page, Format, 1, secondOffset, RowSlotFlags.None); return page; } @@ -164,4 +168,4 @@ private static JetFormatBase OpenFormat() using var db = JetDatabase.Open(TestDatabases.NorthwindAccdb); return db.Format; } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/PageCacheTests.cs b/test/LibRed.Core.Tests/PageCacheTests.cs index 7d2eb2766..c804798be 100644 --- a/test/LibRed.Core.Tests/PageCacheTests.cs +++ b/test/LibRed.Core.Tests/PageCacheTests.cs @@ -40,6 +40,48 @@ public void A_second_channels_cached_page_reflects_the_first_channels_write() finally { TemporaryDatabase.Delete(path); } } + // The pool is per FILE, and the key has to decide that the way the file system does. Two paths differing + // only in the case of the name are one file where names are case-insensitive and two files where they are + // not, so the same assertion reads both ways: the second channel sees the first's write exactly when the + // two paths name one file. Keying on the lower-cased path made them share everywhere, which on a + // case-sensitive file system writes one database's pages into another's. + [Fact] + public void Two_paths_share_a_pool_exactly_when_they_name_one_file() + { + string path = CopyNorthwind("case"); + string variant = Path.Combine(Path.GetDirectoryName(path)!, Path.GetFileName(path).ToUpperInvariant()); + Assert.NotEqual(path, variant); + + // Ground truth from the file system itself: where the upper-cased name already resolves, it is the file + // just written; where it does not, a second file is made so both paths exist either way. + bool sameFile = File.Exists(variant); + if (!sameFile) File.Copy(path, variant); + try + { + const int page = 1; + byte[] mutated; + using (var writer = PageChannel.Open(path, readOnly: false)) + using (var reader = PageChannel.Open(variant, readOnly: false)) + { + mutated = reader.ReadPage(page).Span.ToArray(); // the reader caches the current image + mutated[0x40] ^= 0xFF; + writer.WritePage(page, mutated); + + Assert.Equal(sameFile, reader.ReadPage(page).Span.SequenceEqual(mutated)); + } + + // And the same on disk once both have closed: one file carries the write, two files do not. + byte[] onDisk = File.ReadAllBytes(variant); + int pageSize = TestDatabases.FormatOf(variant).PageSize; + Assert.Equal(sameFile, onDisk.AsSpan(page * pageSize, pageSize).SequenceEqual(mutated)); + } + finally + { + if (!sameFile) TemporaryDatabase.Delete(variant); + TemporaryDatabase.Delete(path); + } + } + [Fact] public void Rollback_restores_the_cached_image_not_just_the_disk() { @@ -85,4 +127,4 @@ public void Reopening_after_the_last_channel_closes_reads_committed_bytes_from_d } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/PageChannelTests.cs b/test/LibRed.Core.Tests/PageChannelTests.cs index 2aebb43fd..ba46b4b3c 100644 --- a/test/LibRed.Core.Tests/PageChannelTests.cs +++ b/test/LibRed.Core.Tests/PageChannelTests.cs @@ -49,7 +49,7 @@ public void RollbackTransaction_restores_modified_pages_and_drops_allocated_ones channel.WritePage(1, page); // ...and allocate a couple of new ones. - channel.AllocatePage(); + channel.Allocator.Append(); channel.WritePage(channel.PageCount, new byte[channel.PageSize]); Assert.True(channel.PageCount > pagesBefore); Assert.True(channel.InTransaction); @@ -178,4 +178,36 @@ public void ReleaseSavepoint_merges_into_the_parent_so_an_outer_rollback_still_u } finally { TemporaryDatabase.Delete(path); } } -} + + // A dependency registered again under a key the transaction already holds is the same condition, so the + // commit checks it once. A savepoint rollback discards the dependencies made after it, and their keys with + // them: a write repeated after the rollback needs its condition held again. + [Fact] + public void A_dependency_is_held_once_per_key_and_a_savepoint_rollback_frees_its_key() + { + string path = TemporaryDatabase.CopyPath(TestDatabases.NorthwindAccdb, "libred-dep-"); + try + { + using var channel = PageChannel.Open(path, readOnly: false); + int checkedA = 0, checkedB = 0, checkedUnkeyed = 0; + + channel.BeginTransaction(); + channel.DependOn(() => ++checkedA > 0, "a", key: "A"); + channel.DependOn(() => ++checkedA > 0, "a", key: "A"); + channel.DependOn(() => ++checkedUnkeyed > 0, "u"); + channel.DependOn(() => ++checkedUnkeyed > 0, "u"); + + Savepoint sp = channel.CreateSavepoint(); + channel.DependOn(() => ++checkedB > 0, "b", key: "B"); + channel.RollbackToSavepoint(sp); + channel.DependOn(() => ++checkedB > 0, "b", key: "B"); + channel.DependOn(() => ++checkedB > 0, "b", key: "B"); + + channel.CommitTransaction(); + Assert.Equal(1, checkedA); + Assert.Equal(1, checkedB); + Assert.Equal(2, checkedUnkeyed); + } + finally { TemporaryDatabase.Delete(path); } + } +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/PageChannelWriteTests.cs b/test/LibRed.Core.Tests/PageChannelWriteTests.cs index 097eec8a2..9e86ca5a8 100644 --- a/test/LibRed.Core.Tests/PageChannelWriteTests.cs +++ b/test/LibRed.Core.Tests/PageChannelWriteTests.cs @@ -87,7 +87,7 @@ public void AllocatePage_grows_the_file_by_one_zeroed_page() using (var channel = PageChannel.Open(path, readOnly: false)) { countBefore = channel.PageCount; - allocated = channel.AllocatePage(); + allocated = channel.Allocator.Append(); Assert.Equal(countBefore, allocated); Assert.Equal(countBefore + 1, channel.PageCount); } @@ -112,4 +112,4 @@ public void Read_only_channel_refuses_writes() } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/PlainTextCollationTests.cs b/test/LibRed.Core.Tests/PlainTextCollationTests.cs new file mode 100644 index 000000000..f030e0c42 --- /dev/null +++ b/test/LibRed.Core.Tests/PlainTextCollationTests.cs @@ -0,0 +1,53 @@ +using LibRed.Catalog; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Core.Tests; + +/// +/// Under an untailored General order — either version — two texts of printable ASCII without a hyphen or +/// apostrophe are one value exactly when they are equal ignoring case. InStr leans on that to search such +/// text directly instead of trying every substring against the collation, and only under those orders: a tailored +/// one gives plain ASCII letters weights of their own (Czech ch, Danish aa), so nothing here holds +/// for it. Held over every string of up to two characters and a random sweep of longer ones. +/// +public class PlainTextCollationTests +{ + private static readonly char[] Plain = + [.. Enumerable.Range(0x20, 0x7F - 0x20).Select(c => (char)c).Where(c => c is not ('-' or '\''))]; + + public static TheoryData Versions => [0, Collation.GeneralVersion]; + + [Theory] + [MemberData(nameof(Versions))] + public void Plain_ascii_is_equal_exactly_when_equal_ignoring_case(byte version) + { + JetTextComparer comparer = JetTextComparer.For(new Collation(CollatingOrder.General, version)); + Assert.True(comparer.IsUntailoredGeneral); + + var random = new Random(20260927); + IEnumerable texts = Plain.Select(c => c.ToString()) + .Concat(Plain.SelectMany(a => Plain.Select(b => $"{a}{b}"))) + .Concat(Enumerable.Range(0, 100_000).Select(_ => + new string([.. Enumerable.Range(0, random.Next(1, 8)).Select(_ => Plain[random.Next(Plain.Length)])]))); + + var byKey = new Dictionary(); + var byFolded = new Dictionary(); + foreach (string text in texts.Distinct()) + { + string key = Convert.ToHexString(comparer.Key(text)), folded = text.TrimEnd(' ').ToUpperInvariant(); + if (byKey.TryGetValue(key, out string? sameKey)) + Assert.True(sameKey.TrimEnd(' ').ToUpperInvariant() == folded, $"'{sameKey}' and '{text}' share a key."); + else + byKey[key] = text; + if (byFolded.TryGetValue(folded, out string? sameText)) + Assert.True(Convert.ToHexString(comparer.Key(sameText)) == key, $"'{sameText}' and '{text}' key apart."); + else + byFolded[folded] = text; + } + } + + [Fact] + public void A_tailored_order_is_not_untailored_general() + => Assert.False(JetTextComparer.For(new Collation(CollatingOrder.Czech, 0)).IsUntailoredGeneral); +} diff --git a/test/LibRed.Core.Tests/PropertyBlobRoundTripTests.cs b/test/LibRed.Core.Tests/PropertyBlobRoundTripTests.cs index cdc254fe6..7b49995b1 100644 --- a/test/LibRed.Core.Tests/PropertyBlobRoundTripTests.cs +++ b/test/LibRed.Core.Tests/PropertyBlobRoundTripTests.cs @@ -6,9 +6,9 @@ namespace LibRed.Core.Tests; // The LvProp property blob must round-trip EVERY property faithfully — including ones LibRed does not model // (a numeric DecimalPlaces, a designer ValidationRule/Format) — because an ALTER that edits a table's defaults -// or nullability rewrites the whole blob (TableCreator.MutateLvPropForColumn does Read -> Write). Property.RawValue -// and IsDdl preserve each entry's exact stored value bytes and DDL classification, so an unmodelled property -// survives rather than being mangled or silently converted into a definition-protected property. +// or nullability rewrites the whole blob (SchemaEditor.MutateLvPropForColumn does Read -> Write). Property.RawValue +// and Flags preserve each entry's exact stored value bytes and flag byte, so an unmodelled property survives +// rather than being mangled or silently converted into a definition-protected property. public class PropertyBlobRoundTripTests { [Fact] @@ -20,7 +20,7 @@ public void Read_write_preserves_an_unmodelled_numeric_property_and_survives_an_ new("Price", "DecimalPlaces", "", JetDataType.Byte, [2]), // unmodelled, numeric new("Price", "Format", "Currency", JetDataType.Text, Encoding.Unicode.GetBytes("Currency")), // unmodelled, text new("Price", "Caption", "Retail price", JetDataType.Memo, - Encoding.Unicode.GetBytes("Retail price")) { IsDdl = false }, // ordinary designer property + Encoding.Unicode.GetBytes("Retail price")) { Flags = 0 }, // ordinary designer property PropertyBlob.Bool("Price", PropertyBlob.RequiredProperty, true), // modelled }; byte[] blob = PropertyBlob.Write(original); @@ -43,9 +43,262 @@ public void Read_write_preserves_an_unmodelled_numeric_property_and_survives_an_ Assert.Equal("Currency", format.Value); PropertyBlob.Property caption = Assert.Single(after, p => p.Name == "Caption"); - Assert.False(caption.IsDdl); + Assert.Equal(0, caption.Flags); Assert.Equal(Encoding.Unicode.GetBytes("Retail price"), caption.RawValue); Assert.Equal("42", Assert.Single(after, p => p.Name == PropertyBlob.DefaultValueProperty).Value); } -} + + // The table's own properties are owned by the empty string, and adding a CHECK constraint replaces one of + // them. Doing that by dropping the owner's whole block and writing back only CheckConstraints took every + // other table-level property with it — ValidationRule above all, which LibRed reads and reports but does + // not re-emit. No public API authors a designer ValidationRule, so this pins the invariant where the defect + // was: the read-modify-write over the property list. + [Fact] + public void Replacing_the_check_property_keeps_the_tables_other_properties() + { + byte[] blob = PropertyBlob.Write( + [ + new PropertyBlob.Property("", PropertyBlob.ValidationRuleProperty, "[V]>0"), + new PropertyBlob.Property("", PropertyBlob.ValidationTextProperty, "V must be positive"), + new PropertyBlob.Property("V", PropertyBlob.DefaultValueProperty, "0"), + ]); + + // The same shape the CHECK paths use: replace one table-owned property, leave the rest alone. + var properties = PropertyBlob.Read(blob).ToList(); + properties.RemoveAll(p => p.Owner.Length == 0 && p.Name == PropertyBlob.CheckConstraintsProperty); + properties.Add(new PropertyBlob.Property( + "", PropertyBlob.CheckConstraintsProperty, PropertyBlob.WriteCheckList([("CK_T", "V < 100")]))); + byte[] updated = PropertyBlob.Write(properties, blob.AsSpan(0, 4)); + + IReadOnlyList reread = PropertyBlob.Read(updated); + (string? rule, string? text) = PropertyBlob.ReadValidation(reread, ""); + Assert.Equal("[V]>0", rule); + Assert.Equal("V must be positive", text); + Assert.Contains(PropertyBlob.ReadCheckConstraints(reread), c => c.Name == "CK_T"); + Assert.Contains(PropertyBlob.Read(updated), p => p.Owner == "V" && p.Value == "0"); + Assert.Equal(blob[..4], updated[..4]); // and the signature is carried across, not restamped + } + + // Every type Access stores, as Access stores it (values taken from Access-written files): the stored length + // sets the width — a Boolean or Int16 can be four bytes, and a four-byte "Int16" is a signed 32-bit value — and + // a Boolean is true for 01 and for FF alike. + public static TheoryData AccessValues() => new() + { + { JetDataType.Boolean, "01", true, "1" }, + { JetDataType.Boolean, "FF", true, "1" }, + { JetDataType.Boolean, "00", false, "0" }, + { JetDataType.Boolean, "01000000", true, "1" }, + { JetDataType.Boolean, "00000000", false, "0" }, + { JetDataType.Byte, "FF", (byte)255, "255" }, + { JetDataType.Int16, "0A00", (short)10, "10" }, + { JetDataType.Int16, "FFFFFFFF", -1, "-1" }, + { JetDataType.Int16, "18060000", 1560, "1560" }, + { JetDataType.Int32, "9DFFFFFF", -99, "-99" }, + { JetDataType.Single, "0000C842", 100f, "100" }, + { JetDataType.Text, "4400650073006300", "Desc", "Desc" }, + { JetDataType.Binary, "85276F677377BE40AC361424EE65776D", + Convert.FromHexString("85276F677377BE40AC361424EE65776D"), "85276F677377BE40AC361424EE65776D" }, + }; + + [Theory] + [MemberData(nameof(AccessValues))] + public void A_property_of_every_type_reads_as_its_value_and_writes_back_unchanged( + JetDataType type, string hex, object expected, string text) + { + byte[] raw = Convert.FromHexString(hex); + byte[] blob = PropertyBlob.Write([new PropertyBlob.Property("C", "P", "", type, raw)]); + + PropertyBlob.Property read = Assert.Single(PropertyBlob.Read(blob)); + Assert.Equal(expected, read.TypedValue); + Assert.Equal(text, read.Value); + Assert.Equal(blob, PropertyBlob.Write([.. PropertyBlob.Read(blob)])); + } + + [Fact] + public void A_datetime_property_reads_as_its_date() + { + // MSysDb's local_ReportDate, as Access stored it: an OLE Automation date. + byte[] raw = Convert.FromHexString("214365C752E7E440"); + PropertyBlob.Property read = Assert.Single(PropertyBlob.Read( + PropertyBlob.Write([new PropertyBlob.Property("", "local_ReportDate", "", JetDataType.DateTime, raw)]))); + Assert.Equal(DateTime.FromOADate(BitConverter.ToDouble(raw)), read.TypedValue); + } + + // Of stores each CLR type as Access does; a GUID is 16 binary bytes in Guid.ToByteArray order — testf's + // Table1 GUID property holds exactly the GUID MSysNameMap records for the table. + [Fact] + public void Of_stores_every_clr_type_and_reads_it_back() + { + var guid = Guid.Parse("676f2785-7773-40be-ac36-1424ee65776d"); + var when = new DateTime(2024, 4, 25, 18, 16, 0); + PropertyBlob.Property[] built = + [ + PropertyBlob.Of("C", "Bool", true), + PropertyBlob.Of("C", "Byte", (byte)2), + PropertyBlob.Of("C", "Int16", (short)-1), + PropertyBlob.Of("C", "Int32", 1560), + PropertyBlob.Of("C", "Currency", 12.5m), + PropertyBlob.Of("C", "Single", 100f), + PropertyBlob.Of("C", "Double", 0.1), + PropertyBlob.Of("C", "Date", when), + PropertyBlob.Of("C", "GUID", guid), + PropertyBlob.Of("C", "Memo", "text"), + PropertyBlob.Of("C", "Text", "text", JetDataType.Text), + ]; + var read = PropertyBlob.Read(PropertyBlob.Write(built)).ToDictionary(p => p.Name); + + Assert.Equal(new byte[] { 1 }, read["Bool"].RawValue); + Assert.Equal(true, read["Bool"].TypedValue); + Assert.Equal((byte)2, read["Byte"].TypedValue); + Assert.Equal((short)-1, read["Int16"].TypedValue); + Assert.Equal(1560, read["Int32"].TypedValue); + Assert.Equal(12.5m, read["Currency"].TypedValue); + Assert.Equal("0000C842", Convert.ToHexString(read["Single"].RawValue!)); + Assert.Equal(0.1, read["Double"].TypedValue); + Assert.Equal(when, read["Date"].TypedValue); + Assert.Equal(JetDataType.Binary, read["GUID"].Type); + Assert.Equal("85276F677377BE40AC361424EE65776D", Convert.ToHexString(read["GUID"].RawValue!)); + Assert.Equal((JetDataType.Memo, "text"), (read["Memo"].Type, read["Memo"].Value)); + Assert.Equal((JetDataType.Text, "text"), (read["Text"].Type, read["Text"].Value)); + } + + // A property constructed from its Value text is encoded as its type, not as UTF-16 text whatever the type. + [Fact] + public void A_property_built_from_text_is_encoded_as_its_type() + { + byte[] blob = PropertyBlob.Write( + [ + new PropertyBlob.Property("C", "DecimalPlaces", "2", JetDataType.Byte), + new PropertyBlob.Property("C", "ColumnWidth", "1560", JetDataType.Int16), + new PropertyBlob.Property("C", "CurrencyLCID", "3081", JetDataType.Int32), + ]); + var read = PropertyBlob.Read(blob).ToDictionary(p => p.Name); + Assert.Equal(new byte[] { 2 }, read["DecimalPlaces"].RawValue); + Assert.Equal((short)1560, read["ColumnWidth"].TypedValue); + Assert.Equal(3081, read["CurrencyLCID"].TypedValue); + } + + // Every property blob in the Access-written fixtures decodes — each value as its type's CLR type — and a + // rewrite of it is byte for byte the original. + [Fact] + public void Every_property_in_the_access_fixtures_decodes_and_writes_back_unchanged() + { + var typesSeen = new HashSet(); + foreach (string path in Directory.GetFiles(Path.Combine(AppContext.BaseDirectory, "Data"), "*.accdb")) + { + JetDatabase db; + try { db = JetDatabase.Open(path); } + catch (InvalidOperationException) { continue; } // the password-encrypted fixture + using (db) + { + var objects = db.OpenTable("MSysObjects"); + int lvProp = objects.Definition.FindColumn("LvProp")!.Index; + foreach (object?[] row in objects.Rows()) + { + if (row[lvProp] is not byte[] blob) continue; + IReadOnlyList properties = PropertyBlob.Read(blob); + Assert.Equal(blob, PropertyBlob.Write(properties, blob)); + foreach (PropertyBlob.Property p in properties) + { + object? value = p.TypedValue; + Type expected = p.Type switch + { + JetDataType.Text or JetDataType.Memo => typeof(string), + JetDataType.Boolean => typeof(bool), + JetDataType.Single => typeof(float), + JetDataType.Binary or JetDataType.Ole => typeof(byte[]), + _ => value!.GetType(), + }; + Assert.IsType(expected, value); + typesSeen.Add(p.Type); + } + } + } + } + Assert.Superset( + new HashSet { JetDataType.Boolean, JetDataType.Byte, JetDataType.Int32, JetDataType.Text, JetDataType.Memo, JetDataType.Binary }, + typesSeen); + } + + // Access's name pool keeps the names of properties it has since deleted, in the order they were first + // defined. Rewriting a blob keeps that pool whole — a pool rebuilt from the surviving properties changed every + // such blob a rewrite touched — and appends only names that are new. + [Fact] + public void Rewriting_a_blob_keeps_its_name_pool_whole_and_in_order() + { + byte[] original = PropertyBlob.Write( + [ + new PropertyBlob.Property("", "Description", "gone soon", JetDataType.Text), + new PropertyBlob.Property("C", PropertyBlob.DefaultValueProperty, "1"), + PropertyBlob.Bool("C", PropertyBlob.RequiredProperty, true), + ]); + var props = PropertyBlob.Read(original).ToList(); + props.RemoveAll(p => p.Name == "Description"); // as Access leaves it: entry gone, name pooled + byte[] withoutDescription = PropertyBlob.Write(props, original); + + props.Add(PropertyBlob.Of("C", "ColumnWidth", (short)1560)); + byte[] rewritten = PropertyBlob.Write(props, withoutDescription); + + Assert.Equal( + ["Description", PropertyBlob.DefaultValueProperty, PropertyBlob.RequiredProperty, "ColumnWidth"], + NamePool(rewritten)); + Assert.Equal(withoutDescription, PropertyBlob.Write([.. PropertyBlob.Read(withoutDescription)], withoutDescription)); + } + + private static List NamePool(byte[] blob) + { + // The pool is the first block: [int length][short 0x80][ [short len][UTF-16] … ]. + int length = BitConverter.ToInt32(blob, 4); + var names = new List(); + for (int q = 10; q < 4 + length;) + { + int n = BitConverter.ToUInt16(blob, q); + names.Add(Encoding.Unicode.GetString(blob, q + 2, n)); + q += 2 + n; + } + return names; + } + + // The flag byte is a bit field, not a DDL boolean: Access writes 0x80 on its own account (every stored query + // in one of the example databases carries it). And an index's properties sit in a block of type 0x0002 under + // the index's name, which is usually its column's. Reading such a blob used to throw on the flag, and writing + // it back re-typed the index's block as the column's, where the column's accessors — and a DROP or RENAME of + // the column — took it for the column's own. + [Fact] + public void A_flag_byte_and_an_index_block_round_trip_and_stay_the_indexs() + { + const ushort IndexBlock = 0x0002; + byte[] blob = PropertyBlob.Write( + [ + new PropertyBlob.Property("", "Replicable", "T", JetDataType.Text) { Flags = 0x80 }, + new PropertyBlob.Property("CustomerID", PropertyBlob.DefaultValueProperty, "'X'"), + new PropertyBlob.Property("CustomerID", "Caption", "Customer", JetDataType.Text) { Flags = 0x81 }, + PropertyBlob.Bool("CustomerID", PropertyBlob.RequiredProperty, true) with { Block = IndexBlock }, + ]); + + IReadOnlyList read = PropertyBlob.Read(blob); + Assert.Equal(0x80, Assert.Single(read, p => p.Name == "Replicable").Flags); + Assert.Equal(0x81, Assert.Single(read, p => p.Name == "Caption").Flags); + Assert.Equal(IndexBlock, Assert.Single(read, p => p.Name == PropertyBlob.RequiredProperty).Block); + Assert.Equal(blob, PropertyBlob.Write([.. read], blob.AsSpan(0, 4))); + + // The index's Required is not the column's. + Assert.Empty(PropertyBlob.ReadRequiredColumns(read)); + Assert.Equal("'X'", PropertyBlob.ReadColumnDefaults(read)["CustomerID"]); + + // Clearing the column's Required, as ALTER COLUMN … NULL does, leaves the index's alone. + var edited = read.ToList(); + edited.RemoveAll(p => p.IsOwnedBy("CustomerID") && p.Name == PropertyBlob.RequiredProperty); + Assert.Equal(blob, PropertyBlob.Write(edited, blob.AsSpan(0, 4))); + + // Dropping or renaming the column moves only the column's block. + IReadOnlyList dropped = PropertyBlob.Read(PropertyBlob.RemoveOwner(blob, "CustomerID")); + Assert.DoesNotContain(dropped, p => p.Name == PropertyBlob.DefaultValueProperty); + Assert.Contains(dropped, p => p.Owner == "CustomerID" && p.Block == IndexBlock); + + IReadOnlyList renamed = PropertyBlob.Read(PropertyBlob.RenameOwner(blob, "CustomerID", "CustID")); + Assert.Equal("CustID", Assert.Single(renamed, p => p.Name == PropertyBlob.DefaultValueProperty).Owner); + Assert.Equal("CustomerID", Assert.Single(renamed, p => p.Block == IndexBlock).Owner); + } +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/PropertyBlobTests.cs b/test/LibRed.Core.Tests/PropertyBlobTests.cs index 49d00dc5a..5d1a7943c 100644 --- a/test/LibRed.Core.Tests/PropertyBlobTests.cs +++ b/test/LibRed.Core.Tests/PropertyBlobTests.cs @@ -32,7 +32,7 @@ public void Write_reproduces_the_ace_blob_byte_for_byte() [Fact] public void ReadColumnDefaults_parses_the_ace_blob() { - var defaults = PropertyBlob.ReadColumnDefaults(AceBlob); + var defaults = PropertyBlob.ReadColumnDefaults(PropertyBlob.Read(AceBlob)); Assert.Equal("42", defaults["Age"]); Assert.Equal("'hi'", defaults["Nm"]); Assert.False(defaults.ContainsKey("Id")); @@ -66,7 +66,7 @@ public void Write_reproduces_the_ace_check_blob_byte_for_byte() [Fact] public void ReadCheckConstraints_parses_the_ace_check_blob() { - var checks = PropertyBlob.ReadCheckConstraints(AceCheckBlob); + var checks = PropertyBlob.ReadCheckConstraints(PropertyBlob.Read(AceCheckBlob)); var (name, expr) = Assert.Single(checks); Assert.Equal("CK_BD", name); Assert.Equal("[BirthDate] < NOW()", expr); @@ -80,7 +80,7 @@ public void ReadColumnDefaults_round_trips_written_blob() new("A", PropertyBlob.DefaultValueProperty, "0"), new("B", PropertyBlob.DefaultValueProperty, "-1"), ]); - var defaults = PropertyBlob.ReadColumnDefaults(blob); + var defaults = PropertyBlob.ReadColumnDefaults(PropertyBlob.Read(blob)); Assert.Equal("0", defaults["A"]); Assert.Equal("-1", defaults["B"]); } diff --git a/test/LibRed.Core.Tests/PublicTableWriteTests.cs b/test/LibRed.Core.Tests/PublicTableWriteTests.cs new file mode 100644 index 000000000..f8c1e29dd --- /dev/null +++ b/test/LibRed.Core.Tests/PublicTableWriteTests.cs @@ -0,0 +1,42 @@ +using LibRed.Catalog; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Core.Tests; + +public class PublicTableWriteTests +{ + [Fact] + public void Updates_and_deletes_keep_indexes_consistent_even_with_an_incomplete_change_mask() + { + string path = Path.Combine(Path.GetTempPath(), $"public-table-{Guid.NewGuid():N}.accdb"); + try + { + JetDatabase.Create(path); + using var db = JetDatabase.Open(path, readOnly: false); + db.CreateTable("T", [new ColumnSpec("Id", JetDataType.Int32, 4, IsFixedLength: true)], primaryKey: ["Id"]); + Table table = db.OpenTable("T"); + IndexDef index = table.Definition.Indexes.Single(i => i.IsPrimaryKey); + table.Insert([1]); + table.Insert([2]); + RowId id = table.Rows().WithIds().First(r => Equals(r.Values[0], 1)).Id; + + db.CreateTable("Other", [new ColumnSpec("Id", JetDataType.Int32, 4, IsFixedLength: true)], primaryKey: ["Id"]); + Table other = db.OpenTable("Other"); + Assert.Throws(() => other.Update(id, [9])); + Assert.Throws(() => other.Delete(id)); + + Assert.Throws(() => table.Update(id, [2], new HashSet())); + Assert.Single(table.SeekRows(index, [1])); + Assert.Single(table.SeekRows(index, [2])); + + table.Update(id, [3], new HashSet()); + Assert.Empty(table.SeekRows(index, [1])); + Assert.Single(table.SeekRows(index, [3])); + table.Delete(id); + Assert.Empty(table.SeekRows(index, [3])); + Assert.Single(table.Rows()); + } + finally { TemporaryDatabase.Delete(path); } + } +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/ReleaseAtCloseTests.cs b/test/LibRed.Core.Tests/ReleaseAtCloseTests.cs index 9377d1e92..55f3d6dbe 100644 --- a/test/LibRed.Core.Tests/ReleaseAtCloseTests.cs +++ b/test/LibRed.Core.Tests/ReleaseAtCloseTests.cs @@ -1,6 +1,7 @@ -using System.Buffers.Binary; using LibRed.Catalog; +using LibRed.Formats; using LibRed.IO; +using LibRed.Pages; using LibRed.Storage; using Xunit; @@ -9,10 +10,12 @@ namespace LibRed.Core.Tests; // ACE holds the pages a session frees — a deleted row's long values, a dropped index, a dropped table — until the // session closes, then returns them and anything in the global released-pages map to the global free map and // clears the released map, lengthening it to cover the highest page released (docs/format/page-05-usage-maps.md -// §9.1). The long value an UPDATE replaces is the exception: its pages are free at once. Northwind's maps are page 1 -// rows 0 (free: 310 and 329 inside the file) and 1 (released, empty). +// §9.1). The long value an UPDATE replaces is the exception: its pages are free at once. Both maps are found through +// page 0's pointers; Northwind's free map has 310 and 329 inside the file, and its released map is empty. public class ReleaseAtCloseTests { + private static readonly JetFormatBase Format = TestDatabases.FormatOf(TestDatabases.NorthwindAccdb); + [Fact] public void Released_pages_stay_unreusable_until_close_then_become_free() { @@ -22,12 +25,12 @@ public void Released_pages_stay_unreusable_until_close_then_become_free() using (var db = JetDatabase.Open(path, readOnly: false)) { PageChannel channel = db.OpenTable("MSysObjects").Channel; - var allocator = new PageAllocator(channel); + var allocator = channel.Allocator; allocator.Release(300); allocator.Release(301); - Assert.False(MapBit(channel, 0, 300)); - Assert.False(MapBit(channel, 0, 301)); + Assert.False(FreeMapBit(channel,300)); + Assert.False(FreeMapBit(channel,301)); Assert.Equal(310, allocator.Allocate()); // the free pages, never the released ones Assert.Equal(329, allocator.Allocate()); Assert.Equal(353, allocator.Allocate()); @@ -36,9 +39,9 @@ public void Released_pages_stay_unreusable_until_close_then_become_free() using (var db = JetDatabase.Open(path)) { PageChannel channel = db.OpenTable("MSysObjects").Channel; - Assert.True(MapBit(channel, 0, 300)); - Assert.True(MapBit(channel, 0, 301)); - Assert.False(MapBit(channel, 0, 310)); + Assert.True(FreeMapBit(channel,300)); + Assert.True(FreeMapBit(channel,301)); + Assert.False(FreeMapBit(channel,310)); Assert.All(ReleasedBits(channel), b => Assert.Equal(0, b)); } } @@ -54,7 +57,7 @@ public void A_rolled_back_release_frees_nothing_and_a_savepoint_rollback_keeps_t using (var db = JetDatabase.Open(path, readOnly: false)) { PageChannel channel = db.OpenTable("MSysObjects").Channel; - var allocator = new PageAllocator(channel); + var allocator = channel.Allocator; channel.BeginTransaction(); allocator.Release(300); @@ -75,10 +78,10 @@ public void A_rolled_back_release_frees_nothing_and_a_savepoint_rollback_keeps_t using (var db = JetDatabase.Open(path)) { PageChannel channel = db.OpenTable("MSysObjects").Channel; - Assert.False(MapBit(channel, 0, 300)); - Assert.True(MapBit(channel, 0, 301)); - Assert.False(MapBit(channel, 0, 302)); - Assert.False(MapBit(channel, 0, 303)); + Assert.False(FreeMapBit(channel,300)); + Assert.True(FreeMapBit(channel,301)); + Assert.False(FreeMapBit(channel,302)); + Assert.False(FreeMapBit(channel,303)); } } finally { TemporaryDatabase.Delete(path); } @@ -94,7 +97,7 @@ public void The_close_lengthens_the_released_map_to_cover_the_highest_page_relea using (var db = JetDatabase.Open(path, readOnly: false)) { PageChannel channel = db.OpenTable("MSysObjects").Channel; - var allocator = new PageAllocator(channel); + var allocator = channel.Allocator; before = ReleasedBits(channel).Length; for (int i = 0; i < 400; i++) allocator.Allocate(); highest = channel.PageCount - 1; @@ -105,11 +108,12 @@ public void The_close_lengthens_the_released_map_to_cover_the_highest_page_relea { PageChannel channel = db.OpenTable("MSysObjects").Channel; // 5-byte header, then the bitmap to the highest page in whole 4-byte words. - int expected = ((highest / 8 + 1) + 3) / 4 * 4; + int growth = Format.UsageMapInlineGrowthSize; + int expected = (BitmapBits.ByteCount(highest + 1) + growth - 1) / growth * growth; Assert.True(expected > before); Assert.Equal(expected, ReleasedBits(channel).Length); Assert.All(ReleasedBits(channel), b => Assert.Equal(0, b)); - Assert.True(MapBit(channel, 0, highest)); + Assert.True(FreeMapBit(channel,highest)); } } finally { TemporaryDatabase.Delete(path); } @@ -121,12 +125,21 @@ public void Pages_already_in_the_released_map_are_merged_by_a_close_that_wrote_b string path = TemporaryDatabase.CopyPath(TestDatabases.NorthwindAccdb, "release-merge-"); try { - byte[] page1 = ReadPage(path, 1); - SetReleasedBit(page1, 300); - WritePage(path, 1, page1); + int holderPage; + byte[] holder; + using (var channel = PageChannel.Open(path, readOnly: true)) + { + (_, holderPage, holder, DataPage.RowSlot slot) = TestDatabases.GlobalMap(channel, Format.ReleasedPagesMapPointerOffset); + Span record = InlineMap(holder, slot); + int bit = 300 - UsageMap.StartPage(record, Format); + Span bits = UsageMap.InlineBits(record, Format); + Assert.InRange(bit, 0, bits.Length * 8 - 1); + BitmapBits.Set(bits, bit, true); + } + TestDatabases.WritePage(path, holderPage, holder); using (JetDatabase.Open(path, readOnly: false)) { } - Assert.Equal(page1, ReadPage(path, 1)); // an idle writable close changes nothing + Assert.Equal(holder, TestDatabases.ReadPage(path, holderPage)); // an idle writable close changes nothing using (var db = JetDatabase.Open(path, readOnly: false)) db.CreateTable("Wrote", [new("Id", JetDataType.Int32, 4, IsFixedLength: true)]); @@ -134,7 +147,7 @@ public void Pages_already_in_the_released_map_are_merged_by_a_close_that_wrote_b using (var db = JetDatabase.Open(path)) { PageChannel channel = db.OpenTable("MSysObjects").Channel; - Assert.True(MapBit(channel, 0, 300)); + Assert.True(FreeMapBit(channel,300)); Assert.All(ReleasedBits(channel), b => Assert.Equal(0, b)); } } @@ -158,66 +171,48 @@ public void Dropping_an_index_holds_its_root_but_a_memo_update_frees_the_old_cha int root = db.Catalog.FindTable("Paths")!.Indexes.Single().RootPage; db.DropIndex("Paths", db.Catalog.FindTable("Paths")!.Indexes.Single().Name); - Assert.False(MapBit(channel, 0, root)); + Assert.False(FreeMapBit(channel,root)); table = db.OpenTable("Paths"); (RowId id, object?[] values) = table.Rows().WithIds().Single(); - // The old chain is free at once, so the new value lands on it and the file does not grow. + // The old chain is freed after the new value is written, as ACE does, so this update grows the file... int pagesBefore = channel.PageCount; values[1] = new string('b', 20000); table.Update(id, values, new HashSet { 1 }); - Assert.Equal(pagesBefore, channel.PageCount); + Assert.True(channel.PageCount > pagesBefore); + + // ...but it is free at once, not held to close: the next update lands on it and the file does not grow. + int pagesAfterFirst = channel.PageCount; + values[1] = new string('c', 20000); + table.Update(id, values, new HashSet { 1 }); + Assert.Equal(pagesAfterFirst, channel.PageCount); } finally { TemporaryDatabase.Delete(path); } } // ---------------------------------------------------------------- helpers - private static (int Offset, int Length) Record(ReadOnlySpan page1, int row) + /// A global map's record on its holder page, which these tests expect in inline form. + private static Span InlineMap(byte[] holder, DataPage.RowSlot slot) { - int offset = BinaryPrimitives.ReadUInt16LittleEndian(page1[(14 + row * 2)..]) & 0x1FFF; - int end = row == 0 ? page1.Length : BinaryPrimitives.ReadUInt16LittleEndian(page1[(14 + (row - 1) * 2)..]) & 0x1FFF; - Assert.Equal(0x00, page1[offset]); // inline - return (offset, end - offset); + Span record = holder.AsSpan(slot.Offset, slot.Length); + Assert.Equal(UsageMapType.Inline, UsageMap.RecordType(record)); + return record; } - private static bool MapBit(PageChannel channel, int row, int page) + private static bool FreeMapBit(PageChannel channel, int page) { - byte[] page1 = channel.ReadPage(1).Span.ToArray(); - (int offset, int length) = Record(page1, row); - int bit = page - BinaryPrimitives.ReadInt32LittleEndian(page1.AsSpan(offset + 1)); - if (bit < 0 || bit / 8 >= length - 5) return false; - return (page1[offset + 5 + bit / 8] & (1 << (bit % 8))) != 0; + (_, _, byte[] holder, DataPage.RowSlot slot) = TestDatabases.GlobalMap(channel, Format.FreePagesMapPointerOffset); + Span record = InlineMap(holder, slot); + int bit = page - UsageMap.StartPage(record, Format); + Span bits = UsageMap.InlineBits(record, Format); + if (bit < 0 || bit / 8 >= bits.Length) return false; + return BitmapBits.Get(bits, bit); } private static byte[] ReleasedBits(PageChannel channel) { - byte[] page1 = channel.ReadPage(1).Span.ToArray(); - (int offset, int length) = Record(page1, 1); - return page1.AsSpan(offset + 5, length - 5).ToArray(); - } - - private static void SetReleasedBit(byte[] page1, int page) - { - (int offset, int length) = Record(page1, 1); - int bit = page - BinaryPrimitives.ReadInt32LittleEndian(page1.AsSpan(offset + 1)); - Assert.InRange(bit, 0, (length - 5) * 8 - 1); - page1[offset + 5 + bit / 8] |= (byte)(1 << (bit % 8)); - } - - private static byte[] ReadPage(string path, int page) - { - using var s = new FileStream(path, FileMode.Open, FileAccess.Read, FileShare.ReadWrite); - var bytes = new byte[4096]; - s.Position = page * 4096L; - s.ReadExactly(bytes); - return bytes; - } - - private static void WritePage(string path, int page, byte[] bytes) - { - using var s = new FileStream(path, FileMode.Open, FileAccess.ReadWrite); - s.Position = page * 4096L; - s.Write(bytes); + (_, _, byte[] holder, DataPage.RowSlot slot) = TestDatabases.GlobalMap(channel, Format.ReleasedPagesMapPointerOffset); + return UsageMap.InlineBits(InlineMap(holder, slot), Format).ToArray(); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/RelocationSlotWidthTests.cs b/test/LibRed.Core.Tests/RelocationSlotWidthTests.cs index 1cf4d8719..1823bbf74 100644 --- a/test/LibRed.Core.Tests/RelocationSlotWidthTests.cs +++ b/test/LibRed.Core.Tests/RelocationSlotWidthTests.cs @@ -46,19 +46,18 @@ [new ColumnSpec("Id", JetDataType.Int32, 4, IsFixedLength: true, IsNullable: fal private static (int Page, int Index, int Offset, int Length) FirstOverflowSlot(Table table) { PageChannel channel = table.Channel; - int dir = channel.Format.DataRowDirectoryOffset; + JetFormatBase format = channel.Format; foreach (int pageNumber in table.UsageMap.DataPages()) { PageBuffer page = channel.ReadPage(pageNumber); - int rowCount = page.ReadUInt16(channel.Format.DataRowCountOffset); + int rowCount = DataPage.ReadRowCount(page.Span, format); int prevEnd = page.Length; for (int i = 0; i < rowCount; i++) { - int raw = page.ReadUInt16(dir + i * 2); - int offset = raw & RowPointer.OffsetMask; + (int offset, RowSlotFlags flags) = DataPage.ReadSlot(page.Span, format, i); int length = prevEnd - offset; prevEnd = offset; - if ((raw & RowPointer.DeletedFlag) == 0 && (raw & RowPointer.OverflowFlag) != 0) + if ((flags & (RowSlotFlags.Deleted | RowSlotFlags.Overflow)) == RowSlotFlags.Overflow) return (pageNumber, i, offset, length); } } @@ -76,27 +75,26 @@ public void LibRed_writes_relocation_slots_exactly_four_bytes_wide() using var db = JetDatabase.Open(path, readOnly: true); Table table = db.OpenTable("R"); PageChannel channel = table.Channel; - int dir = channel.Format.DataRowDirectoryOffset; + JetFormatBase format = channel.Format; var widths = new List(); foreach (int pageNumber in table.UsageMap.DataPages()) { PageBuffer page = channel.ReadPage(pageNumber); - int rowCount = page.ReadUInt16(channel.Format.DataRowCountOffset); + int rowCount = DataPage.ReadRowCount(page.Span, format); int prevEnd = page.Length; for (int i = 0; i < rowCount; i++) { - int raw = page.ReadUInt16(dir + i * 2); - int offset = raw & RowPointer.OffsetMask; + (int offset, RowSlotFlags flags) = DataPage.ReadSlot(page.Span, format, i); int length = prevEnd - offset; prevEnd = offset; - if ((raw & RowPointer.DeletedFlag) == 0 && (raw & RowPointer.OverflowFlag) != 0) + if ((flags & (RowSlotFlags.Deleted | RowSlotFlags.Overflow)) == RowSlotFlags.Overflow) widths.Add(length); } } Assert.NotEmpty(widths); - Assert.All(widths, w => Assert.Equal(4, w)); + Assert.All(widths, w => Assert.Equal(PageBuffer.RecordPointerSize, w)); } finally { TemporaryDatabase.Delete(path); } } @@ -113,22 +111,22 @@ public void A_source_wider_than_the_pointer_resolves_to_the_same_row() Table table = db.OpenTable("R"); PageChannel channel = table.Channel; (int pageNumber, _, int offset, int length) = FirstOverflowSlot(table); - Assert.Equal(4, length); + Assert.Equal(PageBuffer.RecordPointerSize, length); - byte[] pointer = channel.ReadPage(pageNumber).Slice(offset, 4).ToArray(); + byte[] pointer = channel.ReadPage(pageNumber).Slice(offset, PageBuffer.RecordPointerSize).ToArray(); - byte[] trimmed = RowRelocationReader.Resolve( + byte[] trimmed = DataPage.ResolveRelocation( channel, table.Definition.DefinitionPage, - new RowSlot(offset, 4, IsDeleted: false, HasOverflow: true), pointer).Bytes.ToArray(); + new DataPage.RowSlot(offset, PageBuffer.RecordPointerSize, IsDeleted: false, HasOverflow: true), pointer).Bytes.ToArray(); // The Northwind shape: the pointer followed by 51 bytes of the row as it was before it moved. byte[] wide = new byte[55]; pointer.CopyTo(wide, 0); - for (int i = 4; i < wide.Length; i++) wide[i] = (byte)(i * 7); + for (int i = PageBuffer.RecordPointerSize; i < wide.Length; i++) wide[i] = (byte)(i * 7); - byte[] fromWide = RowRelocationReader.Resolve( + byte[] fromWide = DataPage.ResolveRelocation( channel, table.Definition.DefinitionPage, - new RowSlot(offset, wide.Length, IsDeleted: false, HasOverflow: true), wide).Bytes.ToArray(); + new DataPage.RowSlot(offset, wide.Length, IsDeleted: false, HasOverflow: true), wide).Bytes.ToArray(); Assert.Equal(trimmed, fromWide); } @@ -147,9 +145,10 @@ public void A_source_shorter_than_the_pointer_is_still_rejected() Table table = db.OpenTable("R"); (_, _, int offset, _) = FirstOverflowSlot(table); - var error = Assert.Throws(() => RowRelocationReader.Resolve( + var error = Assert.Throws(() => DataPage.ResolveRelocation( table.Channel, table.Definition.DefinitionPage, - new RowSlot(offset, 3, IsDeleted: false, HasOverflow: true), new byte[3])); + new DataPage.RowSlot(offset, PageBuffer.RecordPointerSize - 1, IsDeleted: false, HasOverflow: true), + new byte[PageBuffer.RecordPointerSize - 1])); Assert.Contains("4-byte pointer", error.Message); } finally { TemporaryDatabase.Delete(path); } @@ -166,4 +165,4 @@ public void The_northwind_system_storage_table_can_be_read() Table table = db.OpenTable("MSysAccessStorage"); Assert.NotEmpty(table.Rows().ToList()); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/RowCodecGappedIdTests.cs b/test/LibRed.Core.Tests/RowCodecGappedIdTests.cs index e819cc026..226f2da75 100644 --- a/test/LibRed.Core.Tests/RowCodecGappedIdTests.cs +++ b/test/LibRed.Core.Tests/RowCodecGappedIdTests.cs @@ -1,4 +1,3 @@ -using System.Buffers.Binary; using LibRed; using LibRed.Catalog; using LibRed.Storage; @@ -6,9 +5,12 @@ namespace LibRed.Core.Tests; -// The row's leading count and null-bitmap width are (max column id + 1), NOT the live column count — they -// diverge once ids have a gap (a burned type-change id, or a DROP COLUMN gap). Verified vs ACE (spec §5, -// AceModifyByteDiffProbe). These exercise that gapped-id case directly, which no contiguous-id table can. +// The row's leading count and null-bitmap width span every column id the table has handed out, NOT the live +// column count — they diverge once ids have a gap (a burned type-change id, or a DROP COLUMN gap). A dead +// id's own bit is CLEAR in a row written by an insert; only the ALTER COLUMN re-lay carries the old bit +// forward. Both measured against ACE (spec §5, RowByteParityAccessTests). These exercise the gapped-id case +// directly, which no contiguous-id table can. A standalone encoder has no TDEF, so the count here is the +// highest live id + 1; an insert passes the table's 0x29 high-water, which also outlives a dropped column. public class RowCodecGappedIdTests { private static List Int32Cols(params (string Name, int Id, int FixedOffset)[] cols) => @@ -19,20 +21,20 @@ private static List Int32Cols(params (string Name, int Id, int FixedO }).ToList(); [Fact] - public void Count_and_null_bitmap_use_max_id_plus_one_with_dead_id_bits_set() + public void Count_and_null_bitmap_span_the_dead_ids_whose_bits_stay_clear() { using var db = JetDatabase.Open(TestDatabases.NorthwindAccdb, readOnly: true); // Three live columns, id 1 is a DEAD gap (as a burned type-change would leave: ids 0, 2, 3). var cols = Int32Cols(("A", 0, 0), ("B", 2, 4), ("C", 3, 8)); - byte[] row = new RowEncoder(cols, db.Format).Encode([10, 20, 30]); + byte[] row = new RowCodec(cols, db.Format).Encode([10, 20, 30]); // Leading count = max id + 1 = 4 (NOT the live count 3). - Assert.Equal(4, BinaryPrimitives.ReadUInt16LittleEndian(row)); - // 1-byte null bitmap: live ids 0,2,3 present AND the dead id 1 present too → 0x0F. - Assert.Equal(0x0F, row[^1]); + Assert.Equal(4, RowCodec.Layout.ReadColumnCount(row, db.Format)); + // 1-byte null bitmap: live ids 0,2,3 present, the dead id 1 clear → 0x0D. + Assert.Equal(0x0D, row[^1]); // Round-trips (the decoder sizes the bitmap from the stored count, not the live count). - Assert.Equal(new object?[] { 10, 20, 30 }, new RowDecoder(cols, db.Format).Decode(row)); + Assert.Equal(new object?[] { 10, 20, 30 }, new RowCodec(cols, db.Format).Decode(row)); } [Fact] @@ -43,10 +45,10 @@ public void A_null_column_past_the_live_count_still_reads_null() // Two live columns with ids 0 and 3 (ids 1,2 dead); B (id 3) is null — its bit is beyond the live // count of 2, which is exactly the case that read back null before the fix. var cols = Int32Cols(("A", 0, 0), ("B", 3, 4)); - byte[] row = new RowEncoder(cols, db.Format).Encode([7, null]); + byte[] row = new RowCodec(cols, db.Format).Encode([7, null]); - Assert.Equal(4, BinaryPrimitives.ReadUInt16LittleEndian(row)); // count = max id + 1 - Assert.Equal(0x07, row[^1]); // A(0)+dead1+dead2 present, B(3) null - Assert.Equal(new object?[] { 7, null }, new RowDecoder(cols, db.Format).Decode(row)); + Assert.Equal(4, RowCodec.Layout.ReadColumnCount(row, db.Format)); // count spans id 3 + Assert.Equal(0x01, row[^1]); // A(0) present; dead 1,2 and null B(3) clear + Assert.Equal(new object?[] { 7, null }, new RowCodec(cols, db.Format).Decode(row)); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/RowDecoderTests.cs b/test/LibRed.Core.Tests/RowDecoderTests.cs index 080ca0fe3..260883c47 100644 --- a/test/LibRed.Core.Tests/RowDecoderTests.cs +++ b/test/LibRed.Core.Tests/RowDecoderTests.cs @@ -10,7 +10,10 @@ public class RowDecoderTests { var tdef = db.ReadTableDefinition(2); // MSysObjects schema columns = tdef.Columns; - var decoder = new RowDecoder(columns, db.Format); + // MSysObjects has a Memo column of its own — LvProp — so decoding one of its rows needs the pages + // that value lives on. Without the reader it came back as its 12-byte descriptor, which these + // assertions never look at and so never noticed. + var decoder = new RowCodec(columns, db.Format, longValues: new LongValueStore(db.OpenTable("MSysObjects").Channel)); var page = db.ReadDataPage(dataPage); var rows = new List(); @@ -56,4 +59,4 @@ public void Null_columns_decode_as_null() int connectIdx = columns.First(c => c.Name == "Connect").Index; Assert.Contains(rows, r => r[connectIdx] is null); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/RowEncoderTests.cs b/test/LibRed.Core.Tests/RowEncoderTests.cs index 93a25f9bb..dbabf594d 100644 --- a/test/LibRed.Core.Tests/RowEncoderTests.cs +++ b/test/LibRed.Core.Tests/RowEncoderTests.cs @@ -16,8 +16,8 @@ public void Encode_then_decode_round_trips_real_rows(string tableName) var table = db.OpenTable(tableName); var columns = table.Definition.Columns; - var encoder = new RowEncoder(columns, db.Format); - var decoder = new RowDecoder(columns, db.Format); + var encoder = new RowCodec(columns, db.Format); + var decoder = new RowCodec(columns, db.Format); int rows = 0; foreach (object?[] original in table.Rows()) @@ -41,8 +41,8 @@ public void Null_variable_columns_round_trip() var columns = table.Definition.Columns; int region = columns.Single(c => c.Name == "Region").Index; - var encoder = new RowEncoder(columns, db.Format); - var decoder = new RowDecoder(columns, db.Format); + var encoder = new RowCodec(columns, db.Format); + var decoder = new RowCodec(columns, db.Format); bool sawNull = false; foreach (object?[] original in table.Rows()) @@ -52,4 +52,4 @@ public void Null_variable_columns_round_trip() } Assert.True(sawNull, "expected at least one null Region"); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/RowRelocationCorruptionTests.cs b/test/LibRed.Core.Tests/RowRelocationCorruptionTests.cs index f8e409c9d..2088698fa 100644 --- a/test/LibRed.Core.Tests/RowRelocationCorruptionTests.cs +++ b/test/LibRed.Core.Tests/RowRelocationCorruptionTests.cs @@ -1,6 +1,9 @@ using System.Buffers.Binary; using LibRed; using LibRed.Catalog; +using LibRed.Formats; +using LibRed.IO; +using LibRed.Pages; using LibRed.Storage; using Xunit; @@ -8,12 +11,6 @@ namespace LibRed.Core.Tests; public class RowRelocationCorruptionTests { - private const int PageSize = 4096; - private const int RowDirectoryOffset = 0x0E; - private const int OffsetMask = 0x1FFF; - private const int DeletedFlag = 0x8000; - private const int OverflowFlag = 0x4000; - [Theory] [InlineData("short-source")] [InlineData("page-outside-file")] @@ -95,49 +92,41 @@ private static (string Path, RowId Source) CreateRelocatedRow() private static void Corrupt(string path, RowId source, string corruption) { + JetFormatBase format = TestDatabases.FormatOf(path); + int pageSize = format.PageSize; byte[] file = File.ReadAllBytes(path); - Span sourcePage = file.AsSpan(source.Page * PageSize, PageSize); - int sourceEntryPos = RowDirectoryOffset + source.Row * 2; - int raw = BinaryPrimitives.ReadUInt16LittleEndian(sourcePage[sourceEntryPos..]); - int sourceOffset = raw & OffsetMask; - int sourceEnd = source.Row == 0 - ? PageSize - : BinaryPrimitives.ReadUInt16LittleEndian(sourcePage[(sourceEntryPos - 2)..]) & OffsetMask; - int pointer = BinaryPrimitives.ReadInt32LittleEndian(sourcePage[sourceOffset..]); - int targetPageNumber = pointer >> 8; - int targetRow = pointer & 0xFF; + Span sourcePage = file.AsSpan(source.Page * pageSize, pageSize); + (int sourceOffset, RowSlotFlags sourceFlags) = DataPage.ReadSlot(sourcePage, format, source.Row); + int sourceEnd = source.Row == 0 ? pageSize : DataPage.ReadSlot(sourcePage, format, source.Row - 1).Offset; + (int targetRow, int targetPageNumber) = PageBuffer.ReadRecordPointer(sourcePage, sourceOffset); switch (corruption) { case "short-source": - BinaryPrimitives.WriteUInt16LittleEndian(sourcePage[sourceEntryPos..], - (ushort)((raw & ~OffsetMask) | (sourceEnd - 3))); + DataPage.WriteSlot(sourcePage, format, source.Row, sourceEnd - 3, sourceFlags); break; case "page-outside-file": - BinaryPrimitives.WriteInt32LittleEndian(sourcePage[sourceOffset..], - ((file.Length / PageSize) + 1) << 8); + PageBuffer.WriteRecordPointer(sourcePage, sourceOffset, row: 0, file.Length / pageSize + 1); break; case "row-outside-page": - BinaryPrimitives.WriteInt32LittleEndian(sourcePage[sourceOffset..], (targetPageNumber << 8) | 0xFF); + PageBuffer.WriteRecordPointer(sourcePage, sourceOffset, row: 0xFF, targetPageNumber); break; case "target-not-hidden": case "target-is-overflow": case "target-wrong-owner": - Span targetPage = file.AsSpan(targetPageNumber * PageSize, PageSize); + Span targetPage = file.AsSpan(targetPageNumber * pageSize, pageSize); if (corruption == "target-wrong-owner") { - BinaryPrimitives.WriteUInt32LittleEndian(targetPage[4..], 2); + BinaryPrimitives.WriteUInt32LittleEndian(targetPage[format.DataOwnerOffset..], 2); break; } - int targetEntryPos = RowDirectoryOffset + targetRow * 2; - int targetRaw = BinaryPrimitives.ReadUInt16LittleEndian(targetPage[targetEntryPos..]); - targetRaw = corruption == "target-not-hidden" - ? targetRaw & ~DeletedFlag - : targetRaw | DeletedFlag | OverflowFlag; - BinaryPrimitives.WriteUInt16LittleEndian(targetPage[targetEntryPos..], (ushort)targetRaw); + (int targetOffset, RowSlotFlags targetFlags) = DataPage.ReadSlot(targetPage, format, targetRow); + DataPage.WriteSlot(targetPage, format, targetRow, targetOffset, corruption == "target-not-hidden" + ? targetFlags & ~RowSlotFlags.Deleted + : targetFlags | RowSlotFlags.Deleted | RowSlotFlags.Overflow); break; } File.WriteAllBytes(path, file); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/SortOrderProvenanceProbeTest.cs b/test/LibRed.Core.Tests/SortOrderProvenanceProbeTest.cs index f8f8cacd8..a23a274b0 100644 --- a/test/LibRed.Core.Tests/SortOrderProvenanceProbeTest.cs +++ b/test/LibRed.Core.Tests/SortOrderProvenanceProbeTest.cs @@ -144,7 +144,7 @@ private static bool TryPrimaries(char c, out byte[] primaries) Name = "K", Type = JetDataType.Text, Index = 0, Collation = Collation.GeneralLegacy, }; byte[] key; - try { key = IndexKeyEncoder.Encode([(column, true)], [c.ToString()]); } + try { key = IndexKeyCodec.Encode([(column, true)], [c.ToString()]); } catch (NotSupportedException) { return false; } int split = key.Length - 2; @@ -175,4 +175,4 @@ private static bool TryPrimaries(char c, out byte[] primaries) private static string Describe(char c) => c is >= ' ' and <= '~' ? $"'{c}'" : $"U+{(int)c:X4}"; -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/TableCreatorTests.cs b/test/LibRed.Core.Tests/TableCreatorTests.cs index 485d8c84d..2d0a8c4e9 100644 --- a/test/LibRed.Core.Tests/TableCreatorTests.cs +++ b/test/LibRed.Core.Tests/TableCreatorTests.cs @@ -9,24 +9,6 @@ namespace LibRed.Core.Tests; public class TableCreatorTests { - // Reads the table's inline free-pages map (TDEF 0x3B) -> the set of pages marked as having room. - private static List ReadFreePagesMap(PageChannel ch, int tdefPage) - { - const int FreePtr = 0x3B; - var tdef = ch.ReadPage(tdefPage).Span; - int row = tdef[FreePtr]; - int mapPage = tdef[FreePtr + 1] | tdef[FreePtr + 2] << 8 | tdef[FreePtr + 3] << 16; - var dp = new LibRed.Pages.DataPage(); - dp.Read(ch.ReadPage(mapPage), ch.Format); - ReadOnlySpan m = dp.GetRow(row); - int start = BinaryPrimitives.ReadInt32LittleEndian(m.Slice(1, 4)); - var pages = new List(); - for (int i = 5; i < m.Length; i++) - for (int b = 0; b < 8; b++) - if ((m[i] & (1 << b)) != 0) pages.Add(start + (i - 5) * 8 + b); - return pages; - } - [Fact] public void Multi_page_insert_marks_only_the_tail_page_free() { @@ -49,12 +31,13 @@ public void Multi_page_insert_marks_only_the_tail_page_free() using (var db = JetDatabase.Open(path)) { var table = db.OpenTable("FM"); - var owned = new UsageMap(table.Channel, table.Definition).DataPages().ToList(); + var maps = new UsageMap(table.Channel, table.Definition); + var owned = maps.DataPages().ToList(); Assert.True(owned.Count > 1, "expected multiple owned data pages"); // Access keeps only the current append tail (highest owned page) in the free-pages // map — earlier full pages are cleared as it moves past them. LibRed must match. - var free = ReadFreePagesMap(table.Channel, table.Definition.DefinitionPage); + var free = maps.FreeDataPages().ToList(); Assert.Equal([owned.Max()], free); } } @@ -80,10 +63,8 @@ public void Insert_maintains_the_unique_index_stat_and_leaves_total_at_zero() using (var db = JetDatabase.Open(path)) { var table = db.OpenTable("Stats"); - var span = table.Channel.ReadPage(table.Definition.DefinitionPage).Span; - var format = table.Channel.Format; - int total = BinaryPrimitives.ReadInt32LittleEndian(span.Slice(format.TdefRealIndexBlockOffset, 4)); - int unique = BinaryPrimitives.ReadInt32LittleEndian(span.Slice(format.TdefRealIndexBlockOffset + 4, 4)); + (int total, int unique) = LibRed.Catalog.TableDefinition.ReadIndexCounts( + table.Channel.ReadPage(table.Definition.DefinitionPage).Span, table.Channel.Format, ordinal: 0); // Access maintains the cumulative unique-entry count live (one distinct key per row // for a unique index) but leaves the total-entry count at 0 until compact. @@ -361,4 +342,68 @@ public void Creating_a_table_leaves_existing_tables_readable() } finally { TemporaryDatabase.Delete(path); } } -} + + // Every JetDatabase DDL entry point takes a raw ColumnSpec and hands it to the writer, so the version gate + // has to be here and not only on the SQL path above it — otherwise a Core caller writes a BIGINT + // descriptor into an ACE 12 file, a column Access cannot read. + [Fact] + public void Creating_a_table_refuses_a_type_the_file_is_too_old_for() + { + string path = TemporaryDatabase.CreatePath("versiongate-"); + try + { + JetDatabase.Create(path, version: 0x02); // ACE 12 + using var db = JetDatabase.Open(path, readOnly: false); + + var error = Assert.Throws(() => db.CreateTable("T", + [new ColumnSpec("K", JetDataType.Int32, 4, IsFixedLength: true), + new ColumnSpec("Big", JetDataType.Int64, 8, IsFixedLength: false)])); + Assert.Contains("Access 2016", error.Message); + } + finally { TemporaryDatabase.Delete(path); } + } + + // Every lookup downstream resolves an index by name with First/FirstOrDefault, so two blocks sharing one + // name make DROP INDEX remove an arbitrary one. ACE refuses the duplicate outright. + [Fact] + public void A_duplicate_index_name_is_refused() + { + string path = TemporaryDatabase.CreatePath("dupindex-"); + try + { + JetDatabase.Create(path); + using var db = JetDatabase.Open(path, readOnly: false); + db.CreateTable("T", [Long("K"), Long("A"), Long("B")]); + + db.CreateIndex("T", "IX", [("A", false)], isUnique: false, isPrimary: false, + disallowNull: false, ignoreNulls: false); + Assert.Throws(() => db.CreateIndex("T", "IX", [("B", false)], + isUnique: false, isPrimary: false, disallowNull: false, ignoreNulls: false)); + } + finally { TemporaryDatabase.Delete(path); } + } + + // The TDEF header holds a single seed/increment pair, so a second AutoNumber column would carry the 0x04 + // flag with no counter of its own. ALTER's promote path always refused it; CREATE silently took the first + // and ignored the rest, and ADD COLUMN checked nothing. + [Fact] + public void A_table_may_only_have_one_autonumber_column() + { + string path = TemporaryDatabase.CreatePath("twocounters-"); + try + { + JetDatabase.Create(path); + using var db = JetDatabase.Open(path, readOnly: false); + + Assert.Throws(() => db.CreateTable("T", [Counter("A"), Counter("B")])); + + db.CreateTable("U", [Counter("A"), Long("K")]); + Assert.Throws(() => db.AddColumn("U", Counter("C"))); + } + finally { TemporaryDatabase.Delete(path); } + } + + private static ColumnSpec Long(string name) => new(name, JetDataType.Int32, 4, IsFixedLength: true); + private static ColumnSpec Counter(string name) => + new(name, JetDataType.Int32, 4, IsFixedLength: true, IsAutoNumber: true, Seed: 1, Increment: 1); +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/TableDefinitionPageTests.cs b/test/LibRed.Core.Tests/TableDefinitionPageTests.cs index e0aecd753..bc3e9982c 100644 --- a/test/LibRed.Core.Tests/TableDefinitionPageTests.cs +++ b/test/LibRed.Core.Tests/TableDefinitionPageTests.cs @@ -10,15 +10,12 @@ namespace LibRed.Core.Tests; public class TableDefinitionPageTests { - // In Jet 4 / ACE the system catalog table MSysObjects has its TDEF on page 2. - private const int MSysObjectsPage = 2; - [Fact] public void Reads_MSysObjects_table_definition() { using var db = JetDatabase.Open(TestDatabases.NorthwindAccdb); - var tdef = db.ReadTableDefinition(MSysObjectsPage); + var tdef = db.ReadTableDefinition(db.DefinitionPage.CatalogRootPage); Assert.Equal(TableType.System, tdef.TableType); Assert.Equal(17, tdef.ColumnCount); @@ -33,7 +30,7 @@ public void Decodes_MSysObjects_column_names() { using var db = JetDatabase.Open(TestDatabases.NorthwindAccdb); - var tdef = db.ReadTableDefinition(MSysObjectsPage); + var tdef = db.ReadTableDefinition(db.DefinitionPage.CatalogRootPage); var names = tdef.Columns.Select(c => c.Name).ToList(); // Known MSysObjects columns (a few stable ones, observed in the file). @@ -59,26 +56,28 @@ public void Rejects_counts_outside_jet_ace_table_geometry_before_parsing(string { using var db = JetDatabase.Open(TestDatabases.NorthwindAccdb); JetFormatBase format = db.Format; - byte[] page = TdefBuilder.Build(format, TableType.User, - [new ColumnSpec("C", JetDataType.Int32, 4, IsFixedLength: true)]).Page; + byte[] page = TableDefinition.Build(format, TableType.User, + [new ColumnSpec("C", JetDataType.Int32, 4, IsFixedLength: true)], Collation.GeneralLegacy).Page; switch (field) { case "columns": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.TdefColumnCountOffset, 2), 256); + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(format.TdefColumnCountOffset, 2), (ushort)(format.MaxColumnsPerTable + 1)); break; case "variable-columns": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(format.TdefVariableColumnsOffset, 2), 256); + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(format.TdefVariableColumnsOffset, 2), (ushort)(format.MaxColumnsPerTable + 1)); break; case "real-indexes": - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(format.TdefIndexCountOffset, 4), 33); + BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(format.TdefIndexCountOffset, 4), format.MaxIndexesPerTable + 1); break; case "logical-indexes": BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(format.TdefLogicalIndexCountOffset, 4), -1); break; } - var definition = new TableDefinitionPage(); + var definition = new TableDefinition(); Assert.Throws(() => definition.Read(new PageBuffer(page, 99), format)); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/TdefBuilderTests.cs b/test/LibRed.Core.Tests/TdefBuilderTests.cs index 974f14347..fc165474a 100644 --- a/test/LibRed.Core.Tests/TdefBuilderTests.cs +++ b/test/LibRed.Core.Tests/TdefBuilderTests.cs @@ -31,9 +31,9 @@ public void Built_tdef_round_trips_through_the_reader() new("Notes", JetDataType.Text, 510, IsFixedLength: false), }; - var result = TdefBuilder.Build(Format, TableType.User, specs); + var result = TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy); - var page = new TableDefinitionPage(); + var page = new TableDefinition(); page.Read(new PageBuffer(result.Page, 99), Format); Assert.Equal(TableType.User, page.TableType); @@ -73,9 +73,9 @@ public void Built_tdef_with_primary_key_index_round_trips() }; var indexes = new[] { new IndexSpec("PrimaryKey", ["Id"], IsPrimaryKey: true, IsUnique: true, RootPage: 42) }; - var result = TdefBuilder.Build(Format, TableType.User, specs, indexes); + var result = TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy, indexes); - var page = new TableDefinitionPage(); + var page = new TableDefinition(); page.Read(new PageBuffer(result.Page, 7), Format); var pk = Assert.Single(page.Indexes); @@ -99,10 +99,10 @@ public void Built_tdef_columns_can_encode_and_decode_a_row() new("Id", JetDataType.Int32, 4, IsFixedLength: true), new("Name", JetDataType.Text, 510, IsFixedLength: false), }; - var result = TdefBuilder.Build(Format, TableType.User, specs); + var result = TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy); - var encoder = new LibRed.Storage.RowEncoder(result.Columns, Format); - var decoder = new LibRed.Storage.RowDecoder(result.Columns, Format); + var encoder = new LibRed.Storage.RowCodec(result.Columns, Format); + var decoder = new LibRed.Storage.RowCodec(result.Columns, Format); object?[] row = [42, "hello"]; Assert.Equal(row, decoder.Decode(encoder.Encode(row))); @@ -115,7 +115,7 @@ public void Builder_rejects_more_than_255_columns_before_narrowing_the_count() .Select(i => new ColumnSpec($"C{i}", JetDataType.Int32, 4, IsFixedLength: true)) .ToArray(); - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, specs)); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy)); } [Theory] @@ -125,7 +125,7 @@ public void Builder_rejects_column_fields_that_do_not_fit_the_verified_domain(in { ColumnSpec[] specs = [new("C", JetDataType.Int32, length, IsFixedLength: true, ColumnId: columnId)]; - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, specs)); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy)); } [Fact] @@ -137,7 +137,7 @@ public void Builder_rejects_duplicate_column_ids() new("B", JetDataType.Int32, 4, IsFixedLength: true, ColumnId: 3), ]; - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, specs)); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy)); } [Fact] @@ -145,7 +145,7 @@ public void Builder_rejects_a_column_wider_than_ace_stores() { ColumnSpec[] specs = [new("A", JetDataType.Binary, 40000, IsFixedLength: true)]; - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, specs)); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy)); } // Every column within the per-field limit, yet the fixed region as a whole is past what a record can @@ -156,7 +156,7 @@ public void Builder_rejects_a_fixed_region_past_the_record_cap() ColumnSpec[] specs = [.. Enumerable.Range(0, 252) .Select(i => new ColumnSpec($"G{i}", JetDataType.Guid, 16, IsFixedLength: true))]; - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, specs)); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy)); } [Fact] @@ -165,10 +165,10 @@ public void Builder_sizes_the_definition_from_the_actual_encoded_names() ColumnSpec[] columns = Enumerable.Range(0, 255) .Select(i => new ColumnSpec($"C{i:D3}" + new string('N', 60), JetDataType.Boolean, 0, IsFixedLength: true)) .ToArray(); - var result = TdefBuilder.Build(Format, TableType.User, columns); + var result = TableDefinition.Build(Format, TableType.User, columns, Collation.GeneralLegacy); int declared = BinaryPrimitives.ReadInt32LittleEndian(result.Page.AsSpan(Format.TdefLengthOffset, 4)); - var definition = new TableDefinitionPage(); + var definition = new TableDefinition(); definition.Read(new PageBuffer(result.Page, 99), Format); Assert.Equal(255, definition.Columns.Count); Assert.Equal(declared, result.Page.Length); @@ -178,12 +178,12 @@ public void Builder_sizes_the_definition_from_the_actual_encoded_names() public void Builder_rejects_names_that_do_not_fit_their_16_bit_byte_lengths() { string tooLong = new('N', 65); - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, - [new(tooLong, JetDataType.Int32, 4, IsFixedLength: true)])); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, + [new(tooLong, JetDataType.Int32, 4, IsFixedLength: true)], Collation.GeneralLegacy)); ColumnSpec[] columns = [new("Id", JetDataType.Int32, 4, IsFixedLength: true)]; IndexSpec[] indexes = [new(tooLong, ["Id"], true, true, RootPage: 2)]; - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, columns, indexes)); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, columns, Collation.GeneralLegacy, indexes)); } [Theory] @@ -198,7 +198,7 @@ public void Builder_rejects_invalid_long_value_usage_map_fields( LongValueColumnSpec[] longValues = [new(columnId, usedRow, freeRow, mapPage)]; Assert.Throws(() => - TdefBuilder.Build(Format, TableType.User, columns, longValueColumns: longValues)); + TableDefinition.Build(Format, TableType.User, columns, Collation.GeneralLegacy, longValueColumns: longValues)); } [Fact] @@ -207,9 +207,9 @@ public void Builder_rejects_duplicate_long_value_entries_and_unknown_index_colum ColumnSpec[] columns = [new("M", JetDataType.Memo, 0, IsFixedLength: false)]; LongValueColumnSpec[] duplicate = [new(0, 1, 2, 5), new(0, 1, 2, 5)]; Assert.Throws(() => - TdefBuilder.Build(Format, TableType.User, columns, longValueColumns: duplicate)); + TableDefinition.Build(Format, TableType.User, columns, Collation.GeneralLegacy, longValueColumns: duplicate)); IndexSpec[] indexes = [new("IX", ["Missing"], false, false, RootPage: 2)]; - Assert.Throws(() => TdefBuilder.Build(Format, TableType.User, columns, indexes)); + Assert.Throws(() => TableDefinition.Build(Format, TableType.User, columns, Collation.GeneralLegacy, indexes)); } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/TdefVariableRegionTests.cs b/test/LibRed.Core.Tests/TdefVariableRegionTests.cs index 3accbb00c..03e2d4687 100644 --- a/test/LibRed.Core.Tests/TdefVariableRegionTests.cs +++ b/test/LibRed.Core.Tests/TdefVariableRegionTests.cs @@ -43,22 +43,27 @@ public void Malformed_variable_regions_are_rejected_with_a_corruption_error(stri ? [new("M", JetDataType.Memo, 0, IsFixedLength: false)] : [new("C", JetDataType.Int32, 4, IsFixedLength: true)]; - byte[] page = TdefBuilder.Build(Format, TableType.User, specs).Page; - int columnBlock = Format.TdefRealIndexBlockOffset; + byte[] page = TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy).Page; + // Where the regions are in the sound definition, before any corruption: with no index, the long-value + // list follows the column names directly. + TableDefinition.Regions regions = TableDefinition.Regions.Of(page, Format); + int columnBlock = regions.ColumnDescriptors; int namePos = columnBlock + specs.Length * Format.ColumnDescriptorSize; - int lvalPos = SkipNames(page, namePos, specs.Length); + int lvalPos = regions.IndexNames; int declaredLength = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(Format.TdefLengthOffset, 4)); + int overlongName = Format.MaxNameBytes + sizeof(char); switch (corruption) { case "column-name-too-long": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(namePos, 2), 130); - page.AsSpan(namePos + 2, 130).Fill((byte)'A'); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(namePos + 132, 2), 0xFFFF); - declaredLength = namePos + 134; + BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(namePos, Format.TdefNameLengthSize), (ushort)overlongName); + page.AsSpan(namePos + Format.TdefNameLengthSize, overlongName).Fill((byte)'A'); + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(namePos + Format.TdefNameLengthSize + overlongName, 2), JetFormatBase.TdefLongValueMapTerminator); + declaredLength = namePos + Format.TdefNameLengthSize + overlongName + sizeof(ushort); break; case "column-name-out-of-bounds": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(namePos, 2), 128); + BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(namePos, Format.TdefNameLengthSize), (ushort)Format.MaxNameBytes); break; case "duplicate-column-id": BinaryPrimitives.WriteUInt16LittleEndian( @@ -66,31 +71,33 @@ public void Malformed_variable_regions_are_rejected_with_a_corruption_error(stri break; case "column-id-255": BinaryPrimitives.WriteUInt16LittleEndian( - page.AsSpan(columnBlock + Format.ColumnNumberOffset, 2), 255); + page.AsSpan(columnBlock + Format.ColumnNumberOffset, 2), (ushort)Format.MaxColumnsPerTable); break; case "unknown-column-type": page[columnBlock + Format.ColumnTypeOffset] = 0xFF; break; case "missing-lval-terminator": - declaredLength -= 2; + declaredLength -= sizeof(ushort); break; case "lval-for-non-lval-column": WriteLvalEntry(page, lvalPos, 0); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(lvalPos + 10, 2), 0xFFFF); - declaredLength += 10; + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(lvalPos + Format.TdefLongValueMapEntrySize, 2), JetFormatBase.TdefLongValueMapTerminator); + declaredLength += Format.TdefLongValueMapEntrySize; break; case "duplicate-lval-column": WriteLvalEntry(page, lvalPos, 0); - WriteLvalEntry(page, lvalPos + 10, 0); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(lvalPos + 20, 2), 0xFFFF); - declaredLength += 20; + WriteLvalEntry(page, lvalPos + Format.TdefLongValueMapEntrySize, 0); + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(lvalPos + 2 * Format.TdefLongValueMapEntrySize, 2), JetFormatBase.TdefLongValueMapTerminator); + declaredLength += 2 * Format.TdefLongValueMapEntrySize; break; case "odd-name-length": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(namePos, 2), 1); + BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(namePos, Format.TdefNameLengthSize), 1); break; case "invalid-name-utf16": - page[namePos + 2] = 0x00; - page[namePos + 3] = 0xD8; // unpaired UTF-16 high surrogate + page[namePos + Format.TdefNameLengthSize] = 0x00; + page[namePos + Format.TdefNameLengthSize + 1] = 0xD8; // unpaired UTF-16 high surrogate break; case "trailing-after-lval": page[declaredLength] = 0; @@ -103,7 +110,7 @@ public void Malformed_variable_regions_are_rejected_with_a_corruption_error(stri } BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(Format.TdefLengthOffset, 4), declaredLength); - var definition = new TableDefinitionPage(); + var definition = new TableDefinition(); Assert.Throws(() => definition.Read(new PageBuffer(page.AsMemory(0, declaredLength), 99), Format)); } @@ -113,19 +120,18 @@ public void Overlong_index_name_is_rejected_before_decoding() { ColumnSpec[] specs = [new("C", JetDataType.Int32, 4, IsFixedLength: true)]; IndexSpec[] indexes = [new("I", ["C"], IsPrimaryKey: false, IsUnique: false, RootPage: 42)]; - byte[] page = TdefBuilder.Build(Format, TableType.User, specs, indexes).Page; - - int pos = Format.TdefRealIndexBlockOffset + Format.RealIndexEntrySize - + Format.ColumnDescriptorSize; - pos = SkipNames(page, pos, 1); - int indexNamePos = pos + 52 + 28; // §3.5 data block + §3.6 logical-info block - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(indexNamePos, 2), 130); - page.AsSpan(indexNamePos + 2, 130).Fill((byte)'I'); - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(indexNamePos + 132, 2), 0xFFFF); - int declaredLength = indexNamePos + 134; + byte[] page = TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy, indexes).Page; + + int indexNamePos = TableDefinition.Regions.Of(page, Format).IndexNames; + int overlongName = Format.MaxNameBytes + sizeof(char); + BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(indexNamePos, Format.TdefNameLengthSize), (ushort)overlongName); + page.AsSpan(indexNamePos + Format.TdefNameLengthSize, overlongName).Fill((byte)'I'); + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(indexNamePos + Format.TdefNameLengthSize + overlongName, 2), JetFormatBase.TdefLongValueMapTerminator); + int declaredLength = indexNamePos + Format.TdefNameLengthSize + overlongName + sizeof(ushort); BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(Format.TdefLengthOffset, 4), declaredLength); - var definition = new TableDefinitionPage(); + var definition = new TableDefinition(); Assert.Throws(() => definition.Read(new PageBuffer(page.AsMemory(0, declaredLength), 99), Format)); } @@ -142,8 +148,8 @@ public void Overlong_index_name_is_rejected_before_decoding() [InlineData("logical-index-region-overflow")] public void Malformed_header_counts_and_lengths_are_rejected_before_region_allocation(string corruption) { - byte[] page = TdefBuilder.Build(Format, TableType.User, - [new("C", JetDataType.Int32, 4, IsFixedLength: true)]).Page; + byte[] page = TableDefinition.Build(Format, TableType.User, + [new("C", JetDataType.Int32, 4, IsFixedLength: true)], Collation.GeneralLegacy).Page; int declaredLength = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(Format.TdefLengthOffset, 4)); switch (corruption) @@ -159,16 +165,18 @@ public void Malformed_header_counts_and_lengths_are_rejected_before_region_alloc BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(Format.TdefLengthOffset, 4), declaredLength + 1); break; case "too-many-columns": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.TdefColumnCountOffset, 2), 256); + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(Format.TdefColumnCountOffset, 2), (ushort)(Format.MaxColumnsPerTable + 1)); break; case "variable-column-high-water-overflow": - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(Format.TdefVariableColumnsOffset, 2), 256); + BinaryPrimitives.WriteUInt16LittleEndian( + page.AsSpan(Format.TdefVariableColumnsOffset, 2), (ushort)(Format.MaxColumnsPerTable + 1)); break; case "negative-index-count": BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(Format.TdefIndexCountOffset, 4), -1); break; case "too-many-indexes": - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(Format.TdefIndexCountOffset, 4), 33); + BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(Format.TdefIndexCountOffset, 4), Format.MaxIndexesPerTable + 1); break; case "negative-logical-index-count": BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(Format.TdefLogicalIndexCountOffset, 4), -1); @@ -178,7 +186,7 @@ public void Malformed_header_counts_and_lengths_are_rejected_before_region_alloc break; } - var definition = new TableDefinitionPage(); + var definition = new TableDefinition(); Assert.Throws(() => definition.Read(new PageBuffer(page.AsMemory(0, declaredLength), 99), Format)); } @@ -188,32 +196,16 @@ public void Valid_memo_usage_map_entry_remains_available() { ColumnSpec[] specs = [new("M", JetDataType.Memo, 0, IsFixedLength: false)]; LongValueColumnSpec[] maps = [new(ColumnId: 0, UsedRow: 2, FreeRow: 3, MapPage: 17)]; - byte[] page = TdefBuilder.Build(Format, TableType.User, specs, longValueColumns: maps).Page; + byte[] page = TableDefinition.Build(Format, TableType.User, specs, Collation.GeneralLegacy, longValueColumns: maps).Page; int declaredLength = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(Format.TdefLengthOffset, 4)); - var definition = new TableDefinitionPage(); + var definition = new TableDefinition(); definition.Read(new PageBuffer(page.AsMemory(0, declaredLength), 99), Format); Assert.Equal((2, 17), definition.LongValueOwnedMaps[0]); Assert.Equal((3, 17), definition.LongValueFreeMaps[0]); } - private static int SkipNames(byte[] page, int pos, int count) - { - for (int i = 0; i < count; i++) - { - int length = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(pos, 2)); - pos += 2 + length; - } - return pos; - } - - private static void WriteLvalEntry(byte[] page, int pos, ushort columnId) - { - BinaryPrimitives.WriteUInt16LittleEndian(page.AsSpan(pos, 2), columnId); - page[pos + 2] = 2; - page[pos + 3] = 17; - page[pos + 6] = 3; - page[pos + 7] = 17; - } -} + private static void WriteLvalEntry(byte[] page, int pos, ushort columnId) => + TableDefinition.LongValueMapEntry(Format, columnId, usedRow: 2, freeRow: 3, mapPage: 17).CopyTo(page, pos); +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/TransactionFailureRecoveryTests.cs b/test/LibRed.Core.Tests/TransactionFailureRecoveryTests.cs index 78349eac8..d97e8409b 100644 --- a/test/LibRed.Core.Tests/TransactionFailureRecoveryTests.cs +++ b/test/LibRed.Core.Tests/TransactionFailureRecoveryTests.cs @@ -5,6 +5,53 @@ namespace LibRed.Core.Tests; public class TransactionFailureRecoveryTests { + // An update is several writes, and the row is the last of them — so it is the last that can fail. Before + // it does, the old memo's pages have been freed and the new value's written. Through the SQL engine the + // statement's own transaction undoes all of that; a caller using the Core API directly had no such cover, + // and a refused update left the row naming a memo whose pages had been given away. + [Fact] + public void A_refused_core_update_leaves_the_row_and_its_memo_intact() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "update-undo-"); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var specs = new List + { + new("Id", Catalog.JetDataType.Int32, 4, IsFixedLength: true), + new("M", Catalog.JetDataType.Memo, 0, IsFixedLength: false), + }; + for (int i = 0; i < 9; i++) // 9 x 510 bytes of text takes the record past the 4060 cap + specs.Add(new Catalog.ColumnSpec($"T{i}", Catalog.JetDataType.Text, 510, IsFixedLength: false)); + db.CreateTable("U", specs, primaryKey: ["Id"]); + + string original = new('m', 4000); // long enough to own its own pages + Storage.Table table = db.OpenTable("U"); + var inserted = new object?[specs.Count]; + inserted[0] = 1; + inserted[1] = original; + for (int i = 0; i < 9; i++) inserted[2 + i] = "short"; + table.Insert(inserted); + + var (id, _) = table.Rows().WithIds().Single(); + // A new memo AND nine full text columns: the memo is written and the old one freed, and then the + // row itself is refused. + var oversized = new object?[specs.Count]; + oversized[0] = 1; + oversized[1] = new string('n', 4000); + for (int i = 0; i < 9; i++) oversized[2 + i] = new string('t', 255); + Assert.ThrowsAny(() => + table.Update(id, oversized, new HashSet(Enumerable.Range(1, specs.Count - 1)))); + + // Nothing moved: the row still reads, and its memo still resolves to the value it had. + object?[] after = db.OpenTable("U").Rows().Single(); + Assert.Equal(original, after[1]); + Assert.Equal("short", after[2]); + } + finally { TemporaryDatabase.Delete(path); } + } + [Fact] public void Conflicting_file_growth_can_rollback_and_retry_without_truncating_the_winner() { @@ -17,8 +64,8 @@ public void Conflicting_file_growth_can_rollback_and_retry_without_truncating_th first.BeginTransaction(); second.BeginTransaction(); - int firstPage = first.AllocatePage(); - int secondPage = second.AllocatePage(); + int firstPage = first.Allocator.Append(); + int secondPage = second.Allocator.Append(); Assert.Equal(originalCount, firstPage); Assert.Equal(firstPage, secondPage); WriteMarker(first, firstPage, 0x11); @@ -33,7 +80,7 @@ public void Conflicting_file_growth_can_rollback_and_retry_without_truncating_th Assert.Equal(0x11, second.ReadPage(firstPage).Span[100]); second.BeginTransaction(); - int retryPage = second.AllocatePage(); + int retryPage = second.Allocator.Append(); Assert.Equal(originalCount + 1, retryPage); WriteMarker(second, retryPage, 0x22); second.CommitTransaction(); @@ -55,7 +102,7 @@ public void Disposing_channel_with_outstanding_growth_discards_the_overlay() using (var channel = PageChannel.Open(path, readOnly: false)) { channel.BeginTransaction(); - int page = channel.AllocatePage(); + int page = channel.Allocator.Append(); WriteMarker(channel, page, 0x7E); Assert.Equal(before.Length / channel.PageSize + 1, channel.PageCount); } @@ -100,8 +147,8 @@ public void Failed_growth_publication_truncates_published_tail_and_same_transact using (var channel = PageChannel.Open(path, readOnly: false, locks: locks)) { channel.BeginTransaction(); - int firstPage = channel.AllocatePage(); - int secondPage = channel.AllocatePage(); + int firstPage = channel.Allocator.Append(); + int secondPage = channel.Allocator.Append(); WriteMarker(channel, firstPage, 0x31); WriteMarker(channel, secondPage, 0x32); @@ -140,4 +187,4 @@ public void EnterExclusive(int page) } public void ExitExclusive(int page) { } } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/UsageMapTests.cs b/test/LibRed.Core.Tests/UsageMapTests.cs index c0ed936bc..640283217 100644 --- a/test/LibRed.Core.Tests/UsageMapTests.cs +++ b/test/LibRed.Core.Tests/UsageMapTests.cs @@ -1,7 +1,8 @@ using LibRed; +using LibRed.Formats; using LibRed.IO; using LibRed.Pages; -using System.Buffers.Binary; +using LibRed.Storage; using Xunit; namespace LibRed.Core.Tests; @@ -41,18 +42,15 @@ public void Reference_usage_map_rejects_a_record_larger_than_the_format_shape() using var db = JetDatabase.Open(path, readOnly: false); var table = db.OpenTable("MSysObjects"); PageBuffer tdef = table.Channel.ReadPage(table.Definition.DefinitionPage); - int mapRow = tdef.ReadByte(db.Format.TdefOwnedPagesOffset); - int mapPage = tdef.ReadInt24(db.Format.TdefOwnedPagesOffset + 1); + (int mapRow, int mapPage) = tdef.ReadRecordPointer(db.Format.TdefOwnedPagesOffset); Assert.Equal(0, mapRow); // row 0 ends at the page boundary, making its length deterministic byte[] page = table.Channel.ReadPage(mapPage).Span.ToArray(); - int originalOffset = BinaryPrimitives.ReadUInt16LittleEndian( - page.AsSpan(db.Format.DataRowDirectoryOffset, 2)) & 0x1FFF; - int oversizedOffset = originalOffset - 4; - BinaryPrimitives.WriteUInt16LittleEndian( - page.AsSpan(db.Format.DataRowDirectoryOffset, 2), (ushort)oversizedOffset); + int originalOffset = DataPage.ReadSlot(page, db.Format, 0).Offset; + int oversizedOffset = originalOffset - sizeof(int); + DataPage.WriteSlot(page, db.Format, 0, oversizedOffset, RowSlotFlags.None); page.AsSpan(oversizedOffset, db.Format.PageSize - oversizedOffset).Clear(); - page[oversizedOffset] = 0x01; + page[oversizedOffset] = (byte)UsageMapType.Reference; table.Channel.WritePage(mapPage, page); Assert.Throws(() => table.UsageMap.DataPages().ToList()); @@ -69,13 +67,11 @@ public void Exact_empty_reference_usage_map_remains_valid() using var db = JetDatabase.Open(path, readOnly: false); var table = db.OpenTable("MSysObjects"); PageBuffer tdef = table.Channel.ReadPage(table.Definition.DefinitionPage); - int mapPage = tdef.ReadInt24(db.Format.TdefOwnedPagesOffset + 1); + int mapPage = tdef.ReadRecordPointer(db.Format.TdefOwnedPagesOffset).Page; byte[] page = table.Channel.ReadPage(mapPage).Span.ToArray(); - int offset = BinaryPrimitives.ReadUInt16LittleEndian( - page.AsSpan(db.Format.DataRowDirectoryOffset, 2)) & 0x1FFF; - Assert.Equal(69, db.Format.PageSize - offset); - page.AsSpan(offset, 69).Clear(); - page[offset] = 0x01; + int offset = DataPage.ReadSlot(page, db.Format, 0).Offset; + Assert.Equal(db.Format.UsageMapReferenceRecordSize, db.Format.PageSize - offset); + UsageMap.NewReferenceRecord(db.Format).CopyTo(page, offset); table.Channel.WritePage(mapPage, page); Assert.Empty(table.UsageMap.DataPages()); @@ -92,17 +88,16 @@ public void Reference_usage_map_rejects_a_pointer_to_a_non_bitmap_page() using var db = JetDatabase.Open(path, readOnly: false); var table = db.OpenTable("MSysObjects"); PageBuffer tdef = table.Channel.ReadPage(table.Definition.DefinitionPage); - int mapPage = tdef.ReadInt24(db.Format.TdefOwnedPagesOffset + 1); + int mapPage = tdef.ReadRecordPointer(db.Format.TdefOwnedPagesOffset).Page; byte[] page = table.Channel.ReadPage(mapPage).Span.ToArray(); - int offset = BinaryPrimitives.ReadUInt16LittleEndian( - page.AsSpan(db.Format.DataRowDirectoryOffset, 2)) & 0x1FFF; - page.AsSpan(offset, 69).Clear(); - page[offset] = 0x01; - BinaryPrimitives.WriteInt32LittleEndian(page.AsSpan(offset + 1, 4), mapPage); // data page, not 0x05 + int offset = DataPage.ReadSlot(page, db.Format, 0).Offset; + Span record = page.AsSpan(offset, db.Format.UsageMapReferenceRecordSize); + UsageMap.NewReferenceRecord(db.Format).CopyTo(record); + UsageMap.WriteReferencePointer(record, 0, db.Format, mapPage); // data page, not 0x05 table.Channel.WritePage(mapPage, page); Assert.Throws(() => table.UsageMap.DataPages().ToList()); } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Core.Tests/VersionByteDetectTests.cs b/test/LibRed.Core.Tests/VersionByteDetectTests.cs index 42bc9fab9..0d51043b1 100644 --- a/test/LibRed.Core.Tests/VersionByteDetectTests.cs +++ b/test/LibRed.Core.Tests/VersionByteDetectTests.cs @@ -77,4 +77,58 @@ public void Supported_jet4_identifiers_still_detect(string identifier) Assert.Equal(JetVersion.Version4, format.Version); Assert.False(format.IsAccdb); } -} + + // The fallback has to hold for the whole life of the channel, not just its open. Every rollback re-derives + // the format from the version byte then on disk, and re-deriving it STRICTLY contradicts the open: a file + // that opened and read perfectly well threw on its first rollback, because 0x07 has no format class. + [Fact] + public void A_future_version_byte_survives_a_rollback() + { + string path = TemporaryDatabase.CreatePath("future-version-"); + try + { + JetDatabase.Create(path); + byte[] file = File.ReadAllBytes(path); + file[JetFormatBase.VersionOffset] = 0x07; // an ACE this build has never heard of + File.WriteAllBytes(path, file); + + using var db = JetDatabase.Open(path, readOnly: false); + Assert.Equal(JetVersion.Version17_2019, db.Format.Version); // read as the latest known layout + + db.BeginTransaction(); + db.CreateTable("T", [new Catalog.ColumnSpec("Id", Catalog.JetDataType.Int32, 4, IsFixedLength: true)]); + db.Rollback(); + + Assert.Null(db.Catalog.FindTable("T")); + Assert.Equal(JetVersion.Version17_2019, db.Format.Version); // and still reads the same way + } + finally { TemporaryDatabase.Delete(path); } + } + + // A raise is something ANOTHER connection can do to the file, and the version it leaves decides which + // types this one will accept. A handle open across that change used to keep reporting the version the + // file had when it opened, so it would refuse a column the file had just been made able to hold. + [Fact] + public void A_raise_by_another_handle_is_picked_up() + { + string path = TemporaryDatabase.CreatePath("raise-other-handle-"); + try + { + JetDatabase.Create(path, version: 0x02); // ACE 12 + + using var watcher = JetDatabase.Open(path, readOnly: false); + Assert.Equal(JetVersion.Version12_2007, watcher.Format.Version); + + using (var raiser = JetDatabase.Open(path, readOnly: false)) + { + Assert.True(raiser.EnsureFormatAtLeast(JetVersion.Version16_2016)); + raiser.CreateTable("Big", [new Catalog.ColumnSpec("N", Catalog.JetDataType.Int64, 8, IsFixedLength: true)]); + } + + // The other handle sees the new table, and the version that came with it. + Assert.NotNull(watcher.Catalog.FindTable("Big")); + Assert.Equal(JetVersion.Version16_2016, watcher.Format.Version); + } + finally { TemporaryDatabase.Delete(path); } + } +} \ No newline at end of file diff --git a/test/LibRed.EFCore.Tests/CharacterConversionTests.cs b/test/LibRed.EFCore.Tests/CharacterConversionTests.cs new file mode 100644 index 000000000..474cbc3f7 --- /dev/null +++ b/test/LibRed.EFCore.Tests/CharacterConversionTests.cs @@ -0,0 +1,44 @@ +using EntityFrameworkCore.LibRed.Infrastructure; +using Microsoft.EntityFrameworkCore; +using Xunit; + +namespace LibRed.EFCore.Tests; + +public class CharacterConversionTests +{ + [Theory] + [InlineData(LibRedSqlMode.Compatible)] + [InlineData(LibRedSqlMode.Extended)] + public void Character_to_uint_preserves_unsigned_utf16_values(LibRedSqlMode mode) + { + string path = Path.Combine(Path.GetTempPath(), $"libred-char-{Guid.NewGuid():N}.accdb"); + File.Copy(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), path); + try + { + var options = new DbContextOptionsBuilder().UseLibRed(path, mode).Options; + using var context = new CharacterContext(options); + context.Database.ExecuteSqlRaw("CREATE TABLE CharRows (Id LONG PRIMARY KEY, Value TEXT(1))"); + char[] values = ['1', '\u7FFF', '\u8000', '\uFFFF']; + for (int i = 0; i < values.Length; i++) + context.Database.ExecuteSqlRaw("INSERT INTO CharRows (Id, Value) VALUES ({0}, {1})", i, values[i].ToString()); + + Assert.Equal(values.Select(c => (uint)c), + context.Rows.OrderBy(r => r.Id).Select(r => (uint)r.Value).ToArray()); + } + finally { File.Delete(path); } + } + + private sealed class CharacterRow + { + public int Id { get; set; } + public char Value { get; set; } + } + + private sealed class CharacterContext(DbContextOptions options) : DbContext(options) + { + public DbSet Rows => Set(); + + protected override void OnModelCreating(ModelBuilder modelBuilder) => + modelBuilder.Entity().ToTable("CharRows"); + } +} diff --git a/test/LibRed.EFCore.Tests/RoundTripTests.cs b/test/LibRed.EFCore.Tests/RoundTripTests.cs index 4a4035e7f..9ad59ffca 100644 --- a/test/LibRed.EFCore.Tests/RoundTripTests.cs +++ b/test/LibRed.EFCore.Tests/RoundTripTests.cs @@ -124,7 +124,7 @@ await Assert.ThrowsAsync( [Fact] public void EnsureCreated_creates_a_new_database_natively_and_round_trips() { - // A brand-new file (no copy): EnsureCreated must create the .accdb natively (DatabaseCreator), + // A brand-new file (no copy): EnsureCreated must create the .accdb natively (JetDatabase), // then create the model's schema, then the context is usable for insert + query. string path = Path.Combine(Path.GetTempPath(), $"libred-newdb-{Guid.NewGuid():N}.accdb"); var options = new DbContextOptionsBuilder() @@ -197,4 +197,4 @@ public void Find_by_key_round_trips() var c = context.Customers.Single(x => x.CustomerID == "AROUT"); Assert.Equal("Around the Horn", c.CompanyName); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/AceCoreApiShapeProbeTests.cs b/test/LibRed.Engine.AccessTests/AceCoreApiShapeProbeTests.cs index 87b43ec78..bf49f4cf4 100644 --- a/test/LibRed.Engine.AccessTests/AceCoreApiShapeProbeTests.cs +++ b/test/LibRed.Engine.AccessTests/AceCoreApiShapeProbeTests.cs @@ -10,7 +10,7 @@ namespace LibRed.Engine.Tests; // column it builds has already passed AccessTypeMapper — where the width, precision and scale caps live. This // goes in the other door, the one JetVersion.cs warns about: a ColumnSpec straight to the writer. // -// TdefBuilder already validates column count, names, ids, the 510-byte field cap and the widest record, and +// TableDefinition already validates column count, names, ids, the 510-byte field cap and the widest record, and // EnsureStorable covers version-gated types. This asks what is left OVER those guards. // // The contract is NOT "every shape must be accepted" — it is that whether LibRed accepts or refuses the spec, @@ -79,7 +79,7 @@ private sealed record Shape(string Why, Action Apply, JetVersion Ve private static readonly Dictionary Shapes = new() { // ---- Precision and scale, which nothing checked until these shapes found it: the width check covers - // only the fixed 17 bytes. TdefBuilder.ValidateNumericPrecision now refuses a wrong pair, and + // only the fixed 17 bytes. TableDefinition.ValidateNumericPrecision now refuses a wrong pair, and // EffectivePrecision resolves a declared 0 (which means "unspecified") to ACE's 18. ["decimal-precision-zero"] = new("FixedPoint declaring precision 0 — no digits at all", db => Numeric(db, precision: 0, scale: 0)), @@ -157,7 +157,7 @@ private sealed record Shape(string Why, Action Apply, JetVersion Ve table.Insert([null, 2]); }), - // ---- Undocumented descriptor flags Access sets on ITS OWN catalog columns. TdefBuilder's own comment + // ---- Undocumented descriptor flags Access sets on ITS OWN catalog columns. TableDefinition's own comment // says user-table columns leave these clear; nothing stops a caller setting them. ["system-flags-on-user-column"] = new("SystemFlags 0x10/0x20 (system-catalog, security-id) on a user column", db => db.CreateTable("S", @@ -252,4 +252,4 @@ private static void Numeric(JetDatabase db, byte precision, byte scale) ]); db.OpenTable("S").Insert([1.5m]); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/AceValidityLadder.cs b/test/LibRed.Engine.AccessTests/AceValidityLadder.cs index 315b76969..5de4f20a2 100644 --- a/test/LibRed.Engine.AccessTests/AceValidityLadder.cs +++ b/test/LibRed.Engine.AccessTests/AceValidityLadder.cs @@ -16,7 +16,7 @@ namespace LibRed.Engine.Tests; /// ACE inserts, updates and deletes through its own allocator and index maintenance /// LibRed reads the file back after ACE has written to it /// -/// Rung 4 earns its place: ACE writing means ACE trusting page 1's free map enough to allocate against it, and +/// Rung 4 earns its place: ACE writing means ACE trusting the global free map enough to allocate against it, and /// a file can pass 1-3 on a free map that is quietly wrong. /// /// Shared so both arms judge by one standard — a finding from either has to be comparable. @@ -163,7 +163,7 @@ private static bool ReadsAsText(OleDbConnection connection, string table) try { using var db = JetDatabase.Open(path, readOnly: true); - foreach (TableDef table in db.Catalog.UserTables) + foreach (TableDefinition table in db.Catalog.UserTables) foreach (object?[] row in db.OpenTable(table.Name).Rows()) _ = row.Length; return null; @@ -183,4 +183,4 @@ private static void Execute(OleDbConnection connection, string sql) public static string Flatten(string message) => string.Join(' ', message.Split('\n', '\r').Select(l => l.Trim()).Where(l => l.Length > 0)); -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/AceWriteValiditySweepProbeTests.cs b/test/LibRed.Engine.AccessTests/AceWriteValiditySweepProbeTests.cs index b57fc0b8c..1d182f3d9 100644 --- a/test/LibRed.Engine.AccessTests/AceWriteValiditySweepProbeTests.cs +++ b/test/LibRed.Engine.AccessTests/AceWriteValiditySweepProbeTests.cs @@ -116,7 +116,7 @@ public void Ace_opens_an_empty_database_at_every_format_libred_creates(JetVersio LibRedConnection.CreateDatabase($"Data Source={path}", collation: null, version: version); using var stream = File.OpenRead(path); - stream.Seek(0x14, SeekOrigin.Begin); + stream.Seek(JetFormatBase.VersionOffset, SeekOrigin.Begin); output.WriteLine($"{version}: version byte 0x{stream.ReadByte():X2}"); stream.Dispose(); @@ -150,7 +150,7 @@ public void Libred_refuses_to_create_an_ace_15_database() /// The measurement the creation guard rests on: ACE does not merely avoid the 0x04 version byte, /// it REFUSES a file carrying one, and restamping 0x14 to 0x03 makes the identical bytes open. The spec /// previously inferred "reserved" from absence; this is the stronger fact. - /// Built at 2010 and stamped by hand, since DatabaseCreator now refuses to write 0x04 — the + /// Built at 2010 and stamped by hand, since JetDatabase now refuses to write 0x04 — the /// evidence must not depend on the guard being absent. [Fact] public void Ace_refuses_the_0x04_version_byte_and_nothing_else_about_the_file() @@ -180,7 +180,7 @@ public void Ace_refuses_the_0x04_version_byte_and_nothing_else_about_the_file() private static void Stamp(string path, byte version) { using var stream = new FileStream(path, FileMode.Open, FileAccess.ReadWrite); - stream.Seek(0x14, SeekOrigin.Begin); + stream.Seek(JetFormatBase.VersionOffset, SeekOrigin.Begin); stream.WriteByte(version); } @@ -632,4 +632,4 @@ private static string Summarise(string sql, Dictionary? paramet _ => value.ToString() ?? "", }; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/BigIntCreatedDatabaseAccessTests.cs b/test/LibRed.Engine.AccessTests/BigIntCreatedDatabaseAccessTests.cs index c13c210ce..822cd7dc9 100644 --- a/test/LibRed.Engine.AccessTests/BigIntCreatedDatabaseAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/BigIntCreatedDatabaseAccessTests.cs @@ -41,10 +41,10 @@ public void Ace_reads_bigint_values_libred_wrote_into_a_database_libred_upgraded try { LibRedConnection.CreateDatabase($"Data Source={path}"); // the ACE 12 default, which cannot hold it - Assert.Equal(0x02, VersionByte(path)); using (var db = JetDatabase.Open(path, readOnly: false)) { + Assert.Equal(JetVersion.Version12_2007, db.Format.Version); var engine = new QueryEngine(db); engine.ExecuteNonQuery("CREATE TABLE `B` (`Id` INTEGER PRIMARY KEY, `V` BIGINT NULL)"); Assert.Equal(JetVersion.Version16_2016, db.Format.Version); @@ -54,7 +54,8 @@ public void Ace_reads_bigint_values_libred_wrote_into_a_database_libred_upgraded new Dictionary { ["id"] = i, ["v"] = values[i] }); } - Assert.Equal(0x05, VersionByte(path)); + using (var reopened = JetDatabase.Open(path)) + Assert.Equal(JetVersion.Version16_2016, reopened.Format.Version); using var connection = AceTestDatabase.Open(path); using var command = connection.CreateCommand(); @@ -67,11 +68,4 @@ public void Ace_reads_bigint_values_libred_wrote_into_a_database_libred_upgraded } finally { TemporaryDatabase.Delete(path); } } - - private static byte VersionByte(string path) - { - using var stream = File.OpenRead(path); - stream.Seek(0x14, SeekOrigin.Begin); - return (byte)stream.ReadByte(); - } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/CalculatedColumnSqlAccessTests.cs b/test/LibRed.Engine.AccessTests/CalculatedColumnSqlAccessTests.cs index 0301e6825..eca63a32e 100644 --- a/test/LibRed.Engine.AccessTests/CalculatedColumnSqlAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/CalculatedColumnSqlAccessTests.cs @@ -19,7 +19,7 @@ public void Creates_and_adds_calculated_columns_ace_can_read() string path = TemporaryDatabase.CreatePath("calc-sql-"); try { - DatabaseCreator.CreateEmpty(path); + JetDatabase.Create(path); using (var database = JetDatabase.Open(path, readOnly: false)) { var engine = new QueryEngine(database); @@ -58,7 +58,7 @@ public void Accepts_the_sql_ef_core_emits_for_a_computed_column() string path = TemporaryDatabase.CreatePath("calc-efsql-"); try { - DatabaseCreator.CreateEmpty(path); + JetDatabase.Create(path); using (var database = JetDatabase.Open(path, readOnly: false)) { var engine = new QueryEngine(database); @@ -98,7 +98,7 @@ public void Normalises_identifier_quoting_in_a_check_constraint() string path = TemporaryDatabase.CreatePath("check-tick-"); try { - DatabaseCreator.CreateEmpty(path); + JetDatabase.Create(path); using (var database = JetDatabase.Open(path, readOnly: false)) new QueryEngine(database).ExecuteNonQuery( "CREATE TABLE `T` (`Id` integer, `Qty` integer, CONSTRAINT `ck` CHECK (`Qty` > 0))"); @@ -142,7 +142,7 @@ public void Refuses_an_index_over_a_calculated_column(string sql) string path = TemporaryDatabase.CreatePath("calc-index-"); try { - DatabaseCreator.CreateEmpty(path); + JetDatabase.Create(path); using var database = JetDatabase.Open(path, readOnly: false); var engine = new QueryEngine(database); engine.ExecuteNonQuery("CREATE TABLE T (Id integer, Qty integer, C integer AS ([Qty]*2))"); @@ -164,7 +164,7 @@ public void Refuses_a_key_over_a_calculated_column(string sql) string path = TemporaryDatabase.CreatePath("calc-key-"); try { - DatabaseCreator.CreateEmpty(path); + JetDatabase.Create(path); using var database = JetDatabase.Open(path, readOnly: false); var ex = Assert.ThrowsAny(() => { new QueryEngine(database).ExecuteNonQuery(sql); }); output.WriteLine($" {ex.Message}"); @@ -188,7 +188,7 @@ public void Refuses_a_calculated_column_that_cannot_mean_what_it_says(string sql string path = TemporaryDatabase.CreatePath("calc-sqlbad-"); try { - DatabaseCreator.CreateEmpty(path); + JetDatabase.Create(path); using var database = JetDatabase.Open(path, readOnly: false); var engine = new QueryEngine(database); var ex = Assert.ThrowsAny(() => { engine.ExecuteNonQuery(sql); }); @@ -197,4 +197,4 @@ public void Refuses_a_calculated_column_that_cannot_mean_what_it_says(string sql } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/CatalogRowParityAccessTests.cs b/test/LibRed.Engine.AccessTests/CatalogRowParityAccessTests.cs index 63649062c..bb6f4a067 100644 --- a/test/LibRed.Engine.AccessTests/CatalogRowParityAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/CatalogRowParityAccessTests.cs @@ -31,18 +31,81 @@ public void The_catalog_row_matches_ace(string sql) Assert.Contains("Type=1", ace); // a user table } + // A view and a relationship are objects in the same catalog, written by the same two routines — their + // container, flags, owner and permission rows are as much a part of what Access reads as a table's. + [Fact] + public void A_views_catalog_row_matches_ace() + { + const string sql = "CREATE VIEW W AS SELECT CompanyName FROM Shippers"; + string ace = Describe(sql, AceRun); + Assert.Equal(ace, Describe(sql, LibRedRun)); + Assert.Contains("Type=5", ace); + } + + [Fact] + public void A_relationships_catalog_row_matches_ace() + { + const string sql = "CREATE TABLE P (Id LONG CONSTRAINT pk PRIMARY KEY);CREATE TABLE C (Id LONG, PId LONG);" + + "ALTER TABLE C ADD CONSTRAINT W FOREIGN KEY (PId) REFERENCES P (Id)"; + string ace = Describe(sql, AceRun); + Assert.Equal(ace, Describe(sql, LibRedRun)); + Assert.Contains("Type=8", ace); + } + + // An ALTER moves the object's DateUpdate and leaves DateCreate alone — measured through ACE, and the + // reason LibRed maintains the field at all: it used to write both stamps at CREATE and never touch them + // again, so its catalog said every table was last changed when it was made. Both engines are driven + // through the same statements here, because the rule is ACE's rather than a choice. + [Theory] + [InlineData(true)] + [InlineData(false)] + public void An_alter_moves_DateUpdate_and_leaves_DateCreate(bool throughAce) + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "dateupdate-"); + Action run = throughAce ? AceRun : LibRedRun; + try + { + run(path, "CREATE TABLE W (A LONG, B TEXT(20))"); + (DateTime created, DateTime updated) = Stamps(path); + + Thread.Sleep(1100); // the stamps are stored to the second + run(path, "ALTER TABLE W ADD COLUMN C LONG"); + (DateTime createdAfter, DateTime updatedAfter) = Stamps(path); + + Assert.Equal(created, createdAfter); + Assert.True(updatedAfter > updated, $"DateUpdate did not move: {updated:O} -> {updatedAfter:O}"); + } + finally { TemporaryDatabase.Delete(path); } + } + + private static (DateTime Created, DateTime Updated) Stamps(string path) + { + using var db = JetDatabase.Open(path); + var objects = db.OpenTable("MSysObjects"); + int name = objects.Definition.FindColumn("Name")!.Index; + int created = objects.Definition.FindColumn("DateCreate")!.Index; + int updated = objects.Definition.FindColumn("DateUpdate")!.Index; + object?[] row = objects.Rows().First(r => (string?)r[name] == "W"); + return ((DateTime)row[created]!, (DateTime)row[updated]!); + } + private static void AceRun(string path, string sql) { using OleDbConnection connection = AceTestDatabase.Open(path); - using OleDbCommand command = connection.CreateCommand(); - command.CommandText = sql; - command.ExecuteNonQuery(); + foreach (string statement in sql.Split(';')) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } } private static void LibRedRun(string path, string sql) { using var database = JetDatabase.Open(path, readOnly: false); - new QueryEngine(database).ExecuteNonQuery(sql); + var engine = new QueryEngine(database); + foreach (string statement in sql.Split(';')) engine.ExecuteNonQuery(statement); } /// Table W's MSysObjects row, column by column, with the per-file values left out and the @@ -56,7 +119,7 @@ private static string Describe(string sql, Action run) run(path, sql); using var database = JetDatabase.Open(path, readOnly: true); - TableDef objects = database.Catalog.FindTable("MSysObjects")!; + TableDefinition objects = database.Catalog.FindTable("MSysObjects")!; int Col(string n) => objects.Columns.Single(c => c.Name == n).Index; int name = Col("Name"), id = Col("Id"), parent = Col("ParentId"), type = Col("Type"); @@ -66,10 +129,21 @@ private static string Describe(string sql, Action run) .Select(r => $"{r[name]} (Type={r[type]})") .FirstOrDefault() ?? $"unknown id {table[parent]}"; + // The owner and the permission rows ARE comparable: both engines write into a copy of one file, so + // the per-file SID mask (page-00 §2.3) is the same for both and the masked account SIDs must be too. + TableDefinition aces = database.Catalog.FindTable("MSysACEs")!; + int aceObject = aces.Columns.Single(c => c.Name == "ObjectId").Index; + int aceSid = aces.Columns.Single(c => c.Name == "SID").Index; + int aceAcm = aces.Columns.Single(c => c.Name == "ACM").Index; + var grants = database.OpenTable("MSysACEs").Rows() + .Where(r => Equals(r[aceObject], table[id])) + .Select(r => $"{Format(r[aceSid])}:0x{Convert.ToInt32(r[aceAcm]):X}") + .Order(StringComparer.Ordinal); + return string.Join(", ", objects.Columns - .Where(c => c.Name is not ("DateCreate" or "DateUpdate" or "Id" or "ParentId" or "Owner")) + .Where(c => c.Name is not ("DateCreate" or "DateUpdate" or "Id" or "ParentId")) .Select(c => $"{c.Name}={Format(table[c.Index])}")) - + $", parent={container}"; + + $", parent={container}, grants=[{string.Join(" ", grants)}]"; } finally { TemporaryDatabase.Delete(path); } } @@ -77,7 +151,8 @@ private static string Describe(string sql, Action run) private static string Format(object? value) => value switch { null => "", - byte[] b => $"byte[{b.Length}]", + // A SID is short and is the point of the comparison; a property blob is not, and only its size is. + byte[] b => b.Length <= 8 ? Convert.ToHexString(b) : $"byte[{b.Length}]", _ => value.ToString() ?? "", }; -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ColumnDescriptorByteParityAccessTests.cs b/test/LibRed.Engine.AccessTests/ColumnDescriptorByteParityAccessTests.cs index 47cea7d7d..aa831f918 100644 --- a/test/LibRed.Engine.AccessTests/ColumnDescriptorByteParityAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/ColumnDescriptorByteParityAccessTests.cs @@ -1,8 +1,4 @@ -using System.Buffers.Binary; using System.Data.OleDb; -using LibRed.Catalog; -using LibRed.Formats; -using LibRed.IO; using Xunit; namespace LibRed.Engine.Tests; @@ -15,7 +11,7 @@ namespace LibRed.Engine.Tests; // // It found two, both of them beliefs rather than oversights: // -// 0x09, the repeated column id, on EVERY column of every type. TdefBuilder wrote zero, with a comment +// 0x09, the repeated column id, on EVERY column of every type. TableDefinition wrote zero, with a comment // saying real files store zero there - read off the system tables, where they do. Every genuine user // table in every fixture carries the id, and so does everything that creates one: ACE's SQL DDL, DAO's // object model, and DAO-executed SQL. Compaction preserves whatever is there, so it is set at creation. @@ -31,7 +27,7 @@ public class ColumnDescriptorByteParityAccessTests(ITestOutputHelper output) : T [ "BIT", "BYTE", "SMALLINT", "INTEGER", "COUNTER", "REAL", "FLOAT", "CURRENCY", "DATETIME", "GUID", "DECIMAL(18,4)", "CHAR(50)", "VARCHAR(50)", "TEXT(50)", - "BINARY(50)", "VARBINARY(50)", "LONGTEXT", "LONGBINARY", + "BINARY(50)", "VARBINARY(50)", "BIGBINARY(50)", "BIGBINARY", "LONGTEXT", "LONGBINARY", "BIGINT", // ACE 16+ "DATETIME2", // ACE 17+ ]; @@ -67,26 +63,13 @@ public void Descriptor_bytes_match_ace(string declaration) private static string Copy(string prefix) => TemporaryDatabase.CopyPath( Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), prefix); - /// The descriptor bytes for column "V" — the second column of table W. A two-column table's - /// definition fits one page, so the page is sliced directly rather than stitching a TDEF chain. + /// The descriptor bytes for column "V" of table W, as read from its definition. private static byte[]? Descriptor(string path, Action create) { try { create(path); } catch { return null; } - int definitionPage; - using (var database = JetDatabase.Open(path, readOnly: true)) - { - TableDef? table = database.Catalog.FindTable("W"); - if (table is null) return null; - definitionPage = table.DefinitionPage; - } - - using var channel = PageChannel.Open(path, readOnly: true); - JetFormatBase format = channel.Format; - byte[] page = channel.ReadPage(definitionPage).Span.ToArray(); - int dataCount = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(format.TdefIndexCountOffset, 4)); - int start = format.TdefRealIndexBlockOffset + dataCount * format.RealIndexEntrySize; - return page.AsSpan(start + format.ColumnDescriptorSize, format.ColumnDescriptorSize).ToArray(); + using var database = JetDatabase.Open(path, readOnly: true); + return database.Catalog.FindTable("W")?.Columns.Single(c => c.Name == "V").RawDescriptor; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ColumnLengthAccessTests.cs b/test/LibRed.Engine.AccessTests/ColumnLengthAccessTests.cs index 33d20db2b..60107caba 100644 --- a/test/LibRed.Engine.AccessTests/ColumnLengthAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/ColumnLengthAccessTests.cs @@ -4,7 +4,7 @@ namespace LibRed.Engine.Tests; // ACE rejects a value longer than a variable column's declared width — it neither stores nor clips it. -// RowEncoder.EnsureFitsDeclaredLength matches that; these are the measurements it holds to. +// RowCodec.EnsureFitsDeclaredLength matches that; these are the measurements it holds to. // // The text case uses a literal, not a parameter: parameter Size has its own clipping rule // (ParameterSizeAccessTests) and would confound the answer. @@ -54,12 +54,58 @@ public void Ace_variable_text_column_versus_an_overlong_value() Assert.True( outcome.StartsWith("rejected", StringComparison.Ordinal), $"ACE no longer rejects six characters in a TEXT(5) column - it {outcome}. " - + "RowEncoder.EnsureFitsDeclaredLength should then stop rejecting them too."); + + "RowCodec.EnsureFitsDeclaredLength should then stop rejecting them too."); // The wording LibRed's own rejection is modelled on. Assert.Contains("too small", outcome, StringComparison.OrdinalIgnoreCase); } + // WITH COMPRESSION stores a Latin-1 value one byte per character, so a limit checked on the stored bytes + // would let a TEXT(5) take up to 8 characters. ACE's limit is the declared character count either way. + [Theory] + [InlineData("abcdef")] + [InlineData("abcdefgh")] + public void Ace_compressed_text_column_versus_an_overlong_value(string value) + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "collencomp-"); + + using OleDbConnection connection = AceTestDatabase.Open(path); + using (OleDbCommand ddl = connection.CreateCommand()) + { + ddl.CommandText = "CREATE TABLE LenProbe (Id LONG PRIMARY KEY, V TEXT(5) WITH COMPRESSION)"; + ddl.ExecuteNonQuery(); + } + using (OleDbCommand ok = connection.CreateCommand()) + { + ok.CommandText = "INSERT INTO LenProbe (Id, V) VALUES (1, 'abcde')"; + ok.ExecuteNonQuery(); + } + + using OleDbCommand insert = connection.CreateCommand(); + insert.CommandText = $"INSERT INTO LenProbe (Id, V) VALUES (2, '{value}')"; + var error = Assert.Throws(() => insert.ExecuteNonQuery()); + Assert.Contains("too small", error.Message, StringComparison.OrdinalIgnoreCase); + } + + [Theory] + [InlineData("abcdef")] + [InlineData("abcdefgh")] + public void Libred_refuses_an_overlong_value_in_a_compressed_text_column(string value) + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "collencomp-lib-"); + + using var db = JetDatabase.Open(path, readOnly: false); + var engine = new QueryEngine(db); + engine.ExecuteNonQuery("CREATE TABLE LenProbe (Id LONG PRIMARY KEY, V TEXT(5) WITH COMPRESSION)"); + engine.ExecuteNonQuery("INSERT INTO LenProbe (Id, V) VALUES (1, 'abcde')"); + + var error = Assert.Throws(() => + engine.ExecuteNonQuery($"INSERT INTO LenProbe (Id, V) VALUES (2, '{value}')")); + Assert.Contains("too small", error.Message, StringComparison.OrdinalIgnoreCase); + } + /// Inserts , reporting what ACE did rather than throwing. private static string TryInsert(OleDbConnection connection, string table, int id, object value) { @@ -109,4 +155,4 @@ public void Ace_variable_binary_column_versus_an_overlong_value() $"ACE no longer rejects six bytes in a VARBINARY(5) column - it {overlong}, where five gave " + $"'{control}'. Text is still rejected, so the check would need splitting per column kind."); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ColumnNumberingAccessTests.cs b/test/LibRed.Engine.AccessTests/ColumnNumberingAccessTests.cs new file mode 100644 index 000000000..5a9e9e198 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/ColumnNumberingAccessTests.cs @@ -0,0 +1,154 @@ +using System.Buffers.Binary; +using System.Data.OleDb; +using System.Globalization; +using System.Reflection; +using LibRed; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// A column descriptor's id (0x05), variable-table index (0x07) and column number (0x09) come out as ACE +/// writes them through DROP COLUMN, ADD COLUMN, a retype and CREATE INDEX. The interesting case is ADD COLUMN +/// after a drop: it renumbers every column's 0x09 to its position, and an added fixed column's 0x07 counts the +/// dropped variable columns too. +/// +[Collection(AceCollection.Name)] +public class ColumnNumberingAccessTests +{ + private const string Create = "CREATE TABLE T (A LONG, B LONG, C TEXT(10), D LONG, E TEXT(10))"; + + public static TheoryData Sequences => new() + { + new[] { Create, "ALTER TABLE T DROP COLUMN B", "ALTER TABLE T ADD COLUMN F LONG" }, + new[] { Create, "ALTER TABLE T DROP COLUMN B", "ALTER TABLE T ALTER COLUMN D TEXT(5)" }, + new[] { Create, "ALTER TABLE T DROP COLUMN B", "CREATE INDEX ix ON T (D)" }, + new[] { Create, "ALTER TABLE T DROP COLUMN B", "ALTER TABLE T ADD COLUMN F TEXT(10)", "ALTER TABLE T DROP COLUMN C", "ALTER TABLE T ADD COLUMN G LONG" }, + new[] { Create, "ALTER TABLE T ALTER COLUMN B TEXT(5)", "ALTER TABLE T ADD COLUMN F LONG" }, + new[] { Create, "ALTER TABLE T DROP COLUMN E", "ALTER TABLE T ADD COLUMN F LONG" }, + new[] { Create, "ALTER TABLE T DROP COLUMN B", "ALTER TABLE T DROP COLUMN D" }, + new[] { Create, "ALTER TABLE T DROP COLUMN A", "ALTER TABLE T ADD COLUMN F LONG" }, + new[] { Create, "ALTER TABLE T DROP COLUMN B", "ALTER TABLE T DROP COLUMN D", "ALTER TABLE T ADD COLUMN F TEXT(10)" }, + new[] { Create, "ALTER TABLE T DROP COLUMN C", "ALTER TABLE T DROP COLUMN E", "ALTER TABLE T ADD COLUMN F LONG", "ALTER TABLE T ADD COLUMN G TEXT(10)" }, + }; + + [Theory] + [MemberData(nameof(Sequences))] + public void Descriptor_numbers_follow_ace(string[] statements) + { + string ace = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "colnum-ace-"); + string libred = TemporaryDatabase.CopyPath(ace, "colnum-libred-"); + RunBoth(ace, libred, statements); + } + + // 0x09 is DAO's Field.OrdinalPosition, which DAO sets freely — ties and gaps included — and keeps the + // descriptors sorted by. Moving E first and B last gives ordinals A 0, E 0, B 1, C 2, D 3 and then B 7, in + // descriptor order A E C D B. An ADD COLUMN after that ranks them, keeping the tie. + [Fact] + public void Descriptor_numbers_follow_ace_after_dao_reorders_the_columns() + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string ace = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "colnum-ace-"); + using (OleDbConnection conn = AceTestDatabase.Open(ace)) + using (OleDbCommand cmd = conn.CreateCommand()) + { + cmd.CommandText = Create; + cmd.ExecuteNonQuery(); + cmd.CommandText = "INSERT INTO T VALUES (1, 2, 'c1', 4, 'e1')"; + cmd.ExecuteNonQuery(); + } + object db = Invoke(engine!, "OpenDatabase", ace)!; + try + { + object fields = Get(Item(Get(db, "TableDefs")!, "T"), "Fields")!; + Set(Item(fields, "E"), "OrdinalPosition", 0); + Set(Item(fields, "B"), "OrdinalPosition", 7); + } + finally { Invoke(db, "Close"); } + AceTestDatabase.ReleaseAbandonedComObjects(); + + string libred = TemporaryDatabase.CopyPath(ace, "colnum-libred-"); + Assert.Equal("A:0/0/0 E:4/1/0 C:2/0/2 D:3/1/3 B:1/0/7", Describe(libred)); + RunBoth(ace, libred, ["ALTER TABLE T DROP COLUMN C", "ALTER TABLE T ADD COLUMN F LONG", + "INSERT INTO T (A, B, D, E, F) VALUES (2, 3, 5, 'e2', 6)", + "CREATE INDEX ixE ON T (E)", "ALTER TABLE T ALTER COLUMN D TEXT(5)", + "INSERT INTO T (A, B, D, E, F) VALUES (3, 4, 'd3', 'e3', 7)"]); + } + + private static void RunBoth(string ace, string libred, string[] statements) + { + try + { + foreach (string sql in statements) + { + using (OleDbConnection conn = AceTestDatabase.Open(ace)) + using (OleDbCommand cmd = conn.CreateCommand()) + { + cmd.CommandText = sql; + cmd.ExecuteNonQuery(); + } + using (var db = JetDatabase.Open(libred, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery(sql); + + Assert.Equal(Describe(ace), Describe(libred)); + } + + // And each engine reads the other's rows alike, whatever order the descriptors are in. + Assert.Equal(ReadThroughAce(ace), ReadThroughAce(libred)); + Assert.Equal(ReadThroughAce(ace), ReadThroughLibRed(ace)); + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(libred); + } + } + + // Unordered, since a sequence may drop any column; each reader sorts its rows after the header instead. + private const string ReadBack = "SELECT * FROM T"; + + private static string ReadThroughAce(string path) + { + using OleDbConnection conn = AceTestDatabase.Open(path); + using OleDbCommand cmd = conn.CreateCommand(); + cmd.CommandText = ReadBack; + using OleDbDataReader r = cmd.ExecuteReader(); + var rows = new List(); + while (r.Read()) + rows.Add(string.Join(" ", Enumerable.Range(0, r.FieldCount).Select(i => r.IsDBNull(i) ? "null" : Convert.ToString(r.GetValue(i), CultureInfo.InvariantCulture)))); + return string.Join(" ", Enumerable.Range(0, r.FieldCount).Select(r.GetName)) + " | " + string.Join(" | ", rows.Order(StringComparer.Ordinal)); + } + + private static string ReadThroughLibRed(string path) + { + using var db = JetDatabase.Open(path); + var result = new QueryEngine(db).ExecuteQuery(ReadBack); + var rows = new List(); + foreach (object?[] row in result.Rows) + rows.Add(string.Join(" ", row.Select(v => v is null ? "null" : Convert.ToString(v, CultureInfo.InvariantCulture)))); + return string.Join(" ", result.Columns.Select(c => c.Name)) + " | " + string.Join(" | ", rows.Order(StringComparer.Ordinal)); + } + + private static object Item(object collection, object key) => + collection.GetType().InvokeMember("Item", BindingFlags.GetProperty, null, collection, [key])!; + private static object? Get(object target, string name) => + target.GetType().InvokeMember(name, BindingFlags.GetProperty, null, target, null); + private static void Set(object target, string name, object value) => + target.GetType().InvokeMember(name, BindingFlags.SetProperty, null, target, [value]); + private static object? Invoke(object target, string name, params object?[] args) => + target.GetType().InvokeMember(name, BindingFlags.InvokeMethod, null, target, args); + + // name:0x05/0x07/0x09 for each descriptor, in descriptor order. + private static string Describe(string path) + { + using var db = JetDatabase.Open(path); + return string.Join(" ", db.Catalog.FindTable("T")!.Columns.Select(c => + { + byte[] d = c.RawDescriptor!; + return $"{c.Name}:{c.ColumnId}/" + + $"{c.VariableTableIndex}/" + + $"{BinaryPrimitives.ReadUInt16LittleEndian(d.AsSpan(db.Format.ColumnSecondaryNumberOffset))}"; + })); + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ColumnRetypeParityAccessTests.cs b/test/LibRed.Engine.AccessTests/ColumnRetypeParityAccessTests.cs new file mode 100644 index 000000000..4023d5d1f --- /dev/null +++ b/test/LibRed.Engine.AccessTests/ColumnRetypeParityAccessTests.cs @@ -0,0 +1,133 @@ +using System.Data.OleDb; +using LibRed; +using LibRed.Catalog; +using LibRed.Formats; +using LibRed.IO; +using LibRed.Pages; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +// ALTER COLUMN retypes a column in place, Memo/OLE and a text length included (page-02b-columns): the target's +// descriptor is edited, every row re-laid, the indexes over it rebuilt in name order into the slots they held, and a +// column becoming or ceasing to be a long value gets or loses its §3.3.2 maps as ADD and DROP COLUMN give and take +// them. ACE and LibRed run the same ALTER on copies of one file and must leave the same +// file: the same catalog and rows, and the same bytes on every page but MSysObjects' own (its DateUpdate is the +// clock) — and, when a Memo is re-declared as a Memo, the dead space below the live records on the table's +// usage-map page, where ACE leaves stale copies of the records it retired. +[Collection(AceCollection.Name)] +public class ColumnRetypeParityAccessTests +{ + private static readonly string[] Values = ["short", new string('a', 31), new string('b', 32), new string('c', 33), new string('d', 200)]; + + [Theory] + [InlineData("T", "ALTER TABLE T ALTER COLUMN B MEMO")] // Text -> Memo: short values inline, long on a page + [InlineData("M", "ALTER TABLE M ALTER COLUMN M TEXT(255)")] // Memo -> Text: the maps retired, the page released + [InlineData("M", "ALTER TABLE M ALTER COLUMN M MEMO")] // Memo re-declared: an id burned, maps replaced + [InlineData("T", "ALTER TABLE T ALTER COLUMN X DOUBLE")] // a fixed retype over a NULL: the dead bit and slot + [InlineData("T", "ALTER TABLE T ALTER COLUMN B TEXT(200)")] // a text narrowing: an id burned, every row re-laid + [InlineData("I", "ALTER TABLE I ALTER COLUMN E TEXT(100)")] // a text widening, under one index + [InlineData("I", "ALTER TABLE I ALTER COLUMN B TEXT(20)")] // two indexes whose names run against their slots + [InlineData("I", "ALTER TABLE I ALTER COLUMN ID DOUBLE")] // the primary key and IX_A: IX_A takes slot 0 + [InlineData("I", "ALTER TABLE I ALTER COLUMN B MEMO")] // the two indexes again, to Memo + public void A_retype_leaves_the_file_ACE_leaves(string table, string alter) + { + string basePath = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "retype-base-"); + string acePath = TemporaryDatabase.CopyPath(basePath, "retype-ace-"); + string libredPath = TemporaryDatabase.CopyPath(basePath, "retype-libred-"); + try + { + using (OleDbConnection c = AceTestDatabase.Open(acePath)) + { + Exec(c, "CREATE TABLE T (ID LONG PRIMARY KEY, B TEXT(255), X LONG)"); + Exec(c, "CREATE TABLE M (ID LONG PRIMARY KEY, M MEMO, X LONG)"); + for (int i = 0; i < Values.Length; i++) + foreach (string t in (string[])["T", "M"]) + Exec(c, $"INSERT INTO {t} VALUES ({i + 1}, '{Values[i]}', {i + 1})"); + Exec(c, "INSERT INTO T VALUES (9, NULL, 9)"); + Exec(c, "INSERT INTO M VALUES (9, NULL, 9)"); + Exec(c, "INSERT INTO T VALUES (10, 'z', NULL)"); + // Indexes over B in slots 1 and 3, out of name order, with IX_E between them; IX_Y shares IX_Z's. + Exec(c, "CREATE TABLE I (ID LONG CONSTRAINT PK_I PRIMARY KEY, B TEXT(50), E TEXT(50))"); + Exec(c, "CREATE INDEX IX_Z ON I (B)"); + Exec(c, "CREATE INDEX IX_Y ON I (B)"); + Exec(c, "CREATE INDEX IX_E ON I (E)"); + Exec(c, "CREATE UNIQUE INDEX IX_A ON I (B, ID)"); + for (int i = 0; i < Values.Length; i++) + Exec(c, $"INSERT INTO I VALUES ({i + 1}, 'b{Values.Length - i}', 'e{i}')"); + Exec(c, "INSERT INTO I VALUES (9, NULL, NULL)"); + } + OleDbConnection.ReleaseObjectPool(); + File.Copy(acePath, libredPath, overwrite: true); + + using (OleDbConnection c = AceTestDatabase.Open(acePath)) Exec(c, alter); + OleDbConnection.ReleaseObjectPool(); + JetFormatBase format; + using (var db = JetDatabase.Open(libredPath, readOnly: false)) + { + format = db.Format; + new QueryEngine(db).ExecuteNonQuery(alter); + } + + Assert.Equal(Describe(acePath, table), Describe(libredPath, table)); + + byte[] ace = File.ReadAllBytes(acePath), libred = File.ReadAllBytes(libredPath); + Assert.Equal(ace.Length, libred.Length); + (int catalogPage, int usageMapPage, int lowestRecord) = Landmarks(acePath, table); + int pageSize = format.PageSize; + for (int page = 1; page < ace.Length / pageSize; page++) + { + var differ = Enumerable.Range(0, pageSize).Where(i => ace[page * pageSize + i] != libred[page * pageSize + i]).ToList(); + if (differ.Count == 0 || (int)DataPage.ReadOwner(ace.AsSpan(page * pageSize, pageSize), format) == catalogPage) continue; + Assert.True(page == usageMapPage && differ.All(i => i < lowestRecord), + $"page {page} differs at 0x{differ[0]:X3} ({differ.Count} bytes)"); + } + } + finally + { + TemporaryDatabase.Delete(basePath); + TemporaryDatabase.Delete(acePath); + TemporaryDatabase.Delete(libredPath); + } + } + + // The table's columns (descriptors whole), long-value maps, and each row's raw record and value. + private static string Describe(string path, string tableName) + { + var lines = new List(); + using var db = JetDatabase.Open(path, readOnly: true); + TableDefinition t = db.Catalog.FindTable(tableName)!; + var tdef = db.ReadTableDefinition(t.DefinitionPage); + lines.AddRange(t.Columns.Select(c => $"{c.Name} {c.Type} id {c.ColumnId} {Convert.ToHexString(c.RawDescriptor ?? [])}")); + lines.AddRange(tdef.LongValueOwnedMaps.Select(m => $"maps {m.Key}: {m.Value} {tdef.LongValueFreeMaps[m.Key]}")); + var table = db.OpenTable(tableName); + var reader = new RowInserter(table.Channel, t); + foreach ((RowId id, object?[] values) in table.Rows().WithIds()) + lines.Add($"{id.Page}:{id.Row} {Convert.ToHexString(reader.ReadRow(id))} {string.Join("|", values.Select(v => v?.ToString() ?? "null"))}"); + return string.Join("\n", lines); + } + + // MSysObjects' definition page (its data pages carry it as owner), the table's usage-map page, and the offset of + // the lowest live record on that page. + private static (int CatalogPage, int UsageMapPage, int LowestRecord) Landmarks(string path, string tableName) + { + using var db = JetDatabase.Open(path, readOnly: true); + TableDefinition t = db.Catalog.FindTable(tableName)!; + byte[] file = File.ReadAllBytes(path); + int pageSize = db.Format.PageSize; + int holder = PageBuffer.ReadRecordPointer(file, t.DefinitionPage * pageSize + db.Format.TdefOwnedPagesOffset).Page; + int rows = DataPage.ReadRowCount(file.AsSpan(holder * pageSize, pageSize), db.Format); + int lowest = Enumerable.Range(0, rows) + .Select(i => DataPage.ReadSlot(file.AsSpan(holder * pageSize, pageSize), db.Format, i)) + .Where(s => (s.Flags & RowSlotFlags.Deleted) == 0).Min(s => s.Offset); + return (db.Catalog.FindTable("MSysObjects")!.DefinitionPage, holder, lowest); + } + + private static void Exec(OleDbConnection connection, string sql) + { + using var command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ComplexAttachmentWriteAccessTests.cs b/test/LibRed.Engine.AccessTests/ComplexAttachmentWriteAccessTests.cs index c1baff6b8..d06728129 100644 --- a/test/LibRed.Engine.AccessTests/ComplexAttachmentWriteAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/ComplexAttachmentWriteAccessTests.cs @@ -41,7 +41,7 @@ public void Ace_reads_an_attachment_libred_wrote(string fileName, bool expectCom ComplexColumn column = db.Catalog.ComplexColumns .First(c => c.OwnerTable.Name == "Table1" && c.IsAttachment); columnName = column.ColumnName; - TableDef owner = column.OwnerTable; + TableDefinition owner = column.OwnerTable; ColumnDef inRow = owner.FindColumn(column.ColumnName)!; object?[] record = db.OpenTable("Table1").Rows().First(); @@ -141,4 +141,4 @@ private static object Fields(object target, string name) => progId = "(none)"; return null; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ComplexDeleteByteParityProbeTest.cs b/test/LibRed.Engine.AccessTests/ComplexDeleteByteParityProbeTest.cs index 32349a909..31ce22a14 100644 --- a/test/LibRed.Engine.AccessTests/ComplexDeleteByteParityProbeTest.cs +++ b/test/LibRed.Engine.AccessTests/ComplexDeleteByteParityProbeTest.cs @@ -2,6 +2,9 @@ using System.Reflection; using LibRed; using LibRed.Catalog; +using LibRed.Formats; +using LibRed.Pages; +using LibRed.Storage; using Xunit; namespace LibRed.Engine.Tests; @@ -31,7 +34,7 @@ public void Probe_byte_parity_of_a_complex_delete() string where; using (var db = JetDatabase.Open(TemporaryDatabase.CopyPath(Source, "cx-parity-pick-"))) { - TableDef table = db.Catalog.FindTable("Table1")!; + TableDefinition table = db.Catalog.FindTable("Table1")!; ComplexColumn any = db.Catalog.ComplexColumns.First(c => c.OwnerTable.Name == "Table1"); int recordId = db.OpenTable(any.FlatTable.Name).Rows() .Select(r => Convert.ToInt32(r[any.OwnerLink.Index])).Distinct().First(); @@ -80,21 +83,28 @@ public void Probe_entry_counts_on_insert() new QueryEngine(d).ExecuteNonQuery(Insert); int page; - using (var d = JetDatabase.Open(orig)) page = d.Catalog.FindTable("Order Details")!.DefinitionPage; + JetFormatBase format; + using (var d = JetDatabase.Open(orig)) + { + page = d.Catalog.FindTable("Order Details")!.DefinitionPage; + format = d.Format; + } byte[] o = File.ReadAllBytes(orig), a = File.ReadAllBytes(aceCopy), l = File.ReadAllBytes(libredCopy); - const int PageSize = 4096; - var so = o.AsSpan(page * PageSize, PageSize); - var sa = a.AsSpan(page * PageSize, PageSize); - var sl = l.AsSpan(page * PageSize, PageSize); - - output.WriteLine($"PROBE rowCount(0x10) orig={I(so, 0x10)} ace={I(sa, 0x10)} libred={I(sl, 0x10)}"); - int realIndexes = I(so, 0x33); + int pageSize = format.PageSize; + var so = o.AsSpan(page * pageSize, pageSize); + var sa = a.AsSpan(page * pageSize, pageSize); + var sl = l.AsSpan(page * pageSize, pageSize); + + int rowCount = format.TdefRowCountOffset; + output.WriteLine($"PROBE rowCount(0x10) orig={I(so, rowCount)} ace={I(sa, rowCount)} libred={I(sl, rowCount)}"); + int realIndexes = I(so, format.TdefIndexCountOffset); for (int i = 0; i < realIndexes; i++) { - int at = 0x3F + i * 12; - output.WriteLine($"PROBE index {i} @0x{at:X2}: total orig={I(so, at)} ace={I(sa, at)} libred={I(sl, at)}" - + $" | unique orig={I(so, at + 4)} ace={I(sa, at + 4)} libred={I(sl, at + 4)}"); + var (origCounts, aceCounts, libredCounts) = (TableDefinition.ReadIndexCounts(so, format, i), + TableDefinition.ReadIndexCounts(sa, format, i), TableDefinition.ReadIndexCounts(sl, format, i)); + output.WriteLine($"PROBE index {i}: total orig={origCounts.Total} ace={aceCounts.Total} libred={libredCounts.Total}" + + $" | unique orig={origCounts.Unique} ace={aceCounts.Unique} libred={libredCounts.Unique}"); } } finally @@ -164,7 +174,7 @@ private static Dictionary RowCounts(string path) { using var db = JetDatabase.Open(path); var counts = new Dictionary(StringComparer.Ordinal); - foreach (TableDef t in db.Catalog.Tables) + foreach (TableDefinition t in db.Catalog.Tables) { try { counts[t.Name] = db.OpenTable(t.Name).Rows().Count(); } catch (Exception ex) { counts[t.Name] = -1; Console.WriteLine($"{t.Name}: {ex.Message}"); } @@ -246,28 +256,32 @@ private static Dictionary Owners(string path) { using var db = JetDatabase.Open(path); var map = new Dictionary(); - foreach (TableDef t in db.Catalog.Tables) map[t.DefinitionPage] = t.Name; + foreach (TableDefinition t in db.Catalog.Tables) map[t.DefinitionPage] = t.Name; return map; } - private static string Describe(byte[] file, int page, Dictionary owners) + private static string Describe(byte[] file, int page, Dictionary owners, JetFormatBase format) { - const int PageSize = 4096; - if ((page + 1) * PageSize > file.Length) return "(absent)"; - var span = file.AsSpan(page * PageSize, PageSize); - byte type = span[0]; + int pageSize = format.PageSize; + if ((page + 1) * pageSize > file.Length) return "(absent)"; + var span = file.AsSpan(page * pageSize, pageSize); + PageType type = PageHeader.ReadType(span); string kind = type switch { - 0x00 => "db-header", 0x01 => "data", 0x02 => "TDEF", 0x03 => "index-node", - 0x04 => "index-leaf", 0x05 => "usage-map", 0x08 => "long-value", _ => $"?0x{type:X2}", + PageType.DatabaseDefinition => "db-header", PageType.DataPage => "data", PageType.TableDefinition => "TDEF", + PageType.IntermediateIndexPage => "index-node", PageType.LeafIndexPage => "index-leaf", + PageType.PageUsageBitmap => "usage-bitmap", PageType.ReleasedTableDefinition => "released-TDEF", + PageType.ReleasedDataPage => "released-data", _ => $"?0x{(ushort)type:X4}", }; - if (type is 0x01 or 0x03 or 0x04) + if (type is PageType.DataPage or PageType.IntermediateIndexPage or PageType.LeafIndexPage) { - int tdef = BinaryPrimitives.ReadInt32LittleEndian(span[0x04..]); + int tdef = type == PageType.DataPage + ? (int)DataPage.ReadOwner(span, format) + : IndexTree.ReadOwner(span, format); string name = owners.TryGetValue(tdef, out string? t) ? t : $"tdef@{tdef}"; return $"{kind} of [{name}]"; } - if (type == 0x02 && owners.TryGetValue(page, out string? own)) return $"{kind} [{own}]"; + if (type == PageType.TableDefinition && owners.TryGetValue(page, out string? own)) return $"{kind} [{own}]"; if (span.TrimStart((byte)0).IsEmpty) return "all-zero"; return kind; } @@ -275,12 +289,13 @@ private static string Describe(byte[] file, int page, Dictionary ow private void Report(string label, byte[] o, byte[] a, byte[] n, byte[] l, Dictionary owners) { { + JetFormatBase format = JetFormatBase.Detect(new MemoryStream(o)); + int pageSize = format.PageSize; output.WriteLine($"PROBE [{label}] sizes: orig={o.Length} ace={a.Length} noise={n.Length} libred={l.Length}" - + $" (pages: orig={o.Length / 4096} ace={a.Length / 4096} libred={l.Length / 4096})"); + + $" (pages: orig={o.Length / pageSize} ace={a.Length / pageSize} libred={l.Length / pageSize})"); - const int PageSize = 4096; - HashSet acePages = ChangedPages(o, a, PageSize), noisePages = ChangedPages(o, n, PageSize), - libredPages = ChangedPages(o, l, PageSize); + HashSet acePages = ChangedPages(o, a, pageSize), noisePages = ChangedPages(o, n, pageSize), + libredPages = ChangedPages(o, l, pageSize); output.WriteLine($"PROBE pages changed: ace={acePages.Count} noise={noisePages.Count} libred={libredPages.Count}"); output.WriteLine($"PROBE ace-only (minus housekeeping): [{Join(acePages.Except(noisePages))}]"); @@ -294,39 +309,41 @@ private void Report(string label, byte[] o, byte[] a, byte[] n, byte[] l, Dictio // What every page in play actually is, on each side — a page number alone says nothing. output.WriteLine("PROBE page inventory (orig -> ace / libred):"); foreach (int page in aceReal.Union(libredPages).Order()) - output.WriteLine($"PROBE {page,5}: was {Describe(o, page, owners),-28}" - + $" ace={Describe(a, page, owners),-28} libred={Describe(l, page, owners)}"); + output.WriteLine($"PROBE {page,5}: was {Describe(o, page, owners, format),-28}" + + $" ace={Describe(a, page, owners, format),-28} libred={Describe(l, page, owners, format)}"); // For the pages both changed, are the resulting bytes identical — and where not, what exactly? foreach (int page in aceReal.Intersect(libredPages).Order()) { - var origSpan = o.AsSpan(page * PageSize, PageSize); - var aceSpan = a.AsSpan(page * PageSize, PageSize); - var libSpan = l.AsSpan(page * PageSize, PageSize); + var origSpan = o.AsSpan(page * pageSize, pageSize); + var aceSpan = a.AsSpan(page * pageSize, pageSize); + var libSpan = l.AsSpan(page * pageSize, pageSize); if (aceSpan.SequenceEqual(libSpan)) { output.WriteLine($"PROBE page {page}: IDENTICAL"); continue; } // Branch on what the page BECAME, not what it was: a recycled page changes type, and the // interesting structure is the new one. - byte type = aceSpan[0]; - output.WriteLine($"PROBE page {page}: was=0x{origSpan[0]:X2} now=0x{type:X2}" + PageType type = PageHeader.ReadType(aceSpan); + output.WriteLine($"PROBE page {page}: was=0x{origSpan[0]:X2} now=0x{aceSpan[0]:X2}" + $" ace-vs-libred differs in {Ranges(aceSpan, libSpan).Split(' ').Length} run(s)"); - if (type == 0x02) // TDEF — the per-index entry counters + if (type == PageType.TableDefinition) // TDEF — the per-index entry counters { - int realIndexes = BinaryPrimitives.ReadInt32LittleEndian(origSpan[0x33..]); - output.WriteLine($"PROBE rowCount(0x10) orig={I(origSpan, 0x10)} ace={I(aceSpan, 0x10)} libred={I(libSpan, 0x10)}"); - for (int i = 0; i < realIndexes && 0x3F + i * 12 + 8 <= PageSize; i++) + int realIndexes = I(origSpan, format.TdefIndexCountOffset); + int rowCount = format.TdefRowCountOffset; + output.WriteLine($"PROBE rowCount(0x10) orig={I(origSpan, rowCount)} ace={I(aceSpan, rowCount)} libred={I(libSpan, rowCount)}"); + for (int i = 0; i < realIndexes; i++) { - int at = 0x3F + i * 12; - if (I(origSpan, at) == I(aceSpan, at) && I(origSpan, at + 4) == I(aceSpan, at + 4) - && I(origSpan, at) == I(libSpan, at) && I(origSpan, at + 4) == I(libSpan, at + 4)) continue; - output.WriteLine($"PROBE index {i} @0x{at:X2}: total orig={I(origSpan, at)} ace={I(aceSpan, at)} libred={I(libSpan, at)}" - + $" | unique orig={I(origSpan, at + 4)} ace={I(aceSpan, at + 4)} libred={I(libSpan, at + 4)}"); + var (origCounts, aceCounts, libredCounts) = (TableDefinition.ReadIndexCounts(origSpan, format, i), + TableDefinition.ReadIndexCounts(aceSpan, format, i), TableDefinition.ReadIndexCounts(libSpan, format, i)); + if (origCounts == aceCounts && origCounts == libredCounts) continue; + output.WriteLine($"PROBE index {i}: total orig={origCounts.Total} ace={aceCounts.Total} libred={libredCounts.Total}" + + $" | unique orig={origCounts.Unique} ace={aceCounts.Unique} libred={libredCounts.Unique}"); } } - else if (type is 0x03 or 0x04) // index page — free space, and where each side starts to diverge + else if (type is PageType.IntermediateIndexPage or PageType.LeafIndexPage) // index page — free space, and where each side starts to diverge { - output.WriteLine($"PROBE freeSpace(0x02) orig={U(origSpan, 0x02)} ace={U(aceSpan, 0x02)} libred={U(libSpan, 0x02)}"); + output.WriteLine($"PROBE freeSpace(0x02) orig={IndexTree.ReadFreeSpace(origSpan, format)}" + + $" ace={IndexTree.ReadFreeSpace(aceSpan, format)} libred={IndexTree.ReadFreeSpace(libSpan, format)}"); output.WriteLine($"PROBE first diff vs orig: ace @0x{FirstDiff(origSpan, aceSpan):X3}" + $" libred @0x{FirstDiff(origSpan, libSpan):X3} ace-vs-libred @0x{FirstDiff(aceSpan, libSpan):X3}"); output.WriteLine($"PROBE last diff vs orig: ace @0x{LastDiff(origSpan, aceSpan):X3}" @@ -338,7 +355,8 @@ private void Report(string label, byte[] o, byte[] a, byte[] n, byte[] l, Dictio // The decisive question for tail handling: past each side's own last live byte, does the // page still hold what was there before, or has it been cleared? - int aceEnd = PageSize - U(aceSpan, 0x02), libEnd = PageSize - U(libSpan, 0x02); + int aceEnd = pageSize - IndexTree.ReadFreeSpace(aceSpan, format), + libEnd = pageSize - IndexTree.ReadFreeSpace(libSpan, format); output.WriteLine($"PROBE live ends: ace @0x{aceEnd:X3} libred @0x{libEnd:X3}" + $" tail preserved from orig? ace={origSpan[aceEnd..].SequenceEqual(aceSpan[aceEnd..])}" + $" libred={origSpan[libEnd..].SequenceEqual(libSpan[libEnd..])}" @@ -367,7 +385,6 @@ private static string Hex(ReadOnlySpan page, int at, int count) => Convert.ToHexString(page.Slice(Math.Max(0, at), Math.Min(count, page.Length - Math.Max(0, at)))); private static int I(ReadOnlySpan page, int at) => BinaryPrimitives.ReadInt32LittleEndian(page[at..]); - private static ushort U(ReadOnlySpan page, int at) => BinaryPrimitives.ReadUInt16LittleEndian(page[at..]); /// The byte ranges where two pages differ, as 0xSTART-0xEND(length). private static string Ranges(ReadOnlySpan left, ReadOnlySpan right) @@ -434,4 +451,4 @@ private static void DaoOpenClose(object engine, string path, string table) progId = "(none)"; return null; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ComplexWriteAceReadbackTests.cs b/test/LibRed.Engine.AccessTests/ComplexWriteAceReadbackTests.cs index 879d69c5c..acbf9e352 100644 --- a/test/LibRed.Engine.AccessTests/ComplexWriteAceReadbackTests.cs +++ b/test/LibRed.Engine.AccessTests/ComplexWriteAceReadbackTests.cs @@ -16,6 +16,162 @@ public class ComplexWriteAceReadbackTests(ITestOutputHelper output) { private const string Source = @"D:\exampleaccdb\complex1.accdb"; + // A complex column is held to its values by three links: its descriptor's 0x0B carries the + // MSysComplexColumns key (page-02b §3.4), where an ordinary column carries the collation LANGID; that + // catalog row names the owning table by TDEF page; and the values sit in an f_ flat table keyed to + // the row. A retype of some OTHER column takes the drop-and-recreate rebuild, which rewrites every + // descriptor and moves the table to a new TDEF page — so all three have to be carried over deliberately, + // and once left Table1 with no complex column resolving at all. ACE performs the same ALTER with every one + // intact, and it is the oracle here: the same statement is run through both engines and the wiring each + // leaves behind compared, column for column and value for value. + [Theory] + [InlineData(@"D:\exampleaccdb\complex1.accdb", "Table1", "Col1", "LONGTEXT")] + [InlineData(@"D:\exampleaccdb\LIBRARY.accdb", "Book", "BK_publisher", "LONGTEXT")] + public void An_alter_of_another_column_keeps_the_complex_columns_links( + string source, string table, string column, string newType) + { + if (!File.Exists(source)) { output.WriteLine($"Skipped: {source} not present."); return; } + string statement = $"ALTER TABLE [{table}] ALTER COLUMN [{column}] {newType}"; + + // What ACE itself does with the same statement is the oracle for what LibRed should do. + string acePath = TemporaryDatabase.CopyPath(source, "complex-alter-ace-"); + string aceOutcome; + using (OleDbConnection connection = AceTestDatabase.Open(acePath)) + using (OleDbCommand alter = connection.CreateCommand()) + { + alter.CommandText = statement; + try { alter.ExecuteNonQuery(); aceOutcome = "accepted"; } + catch (OleDbException e) { aceOutcome = e.Message.Trim(); } + } + output.WriteLine($"ACE: {aceOutcome}"); + Assert.Equal("accepted", aceOutcome); + string aceLeft = Remains(acePath); + + string path = TemporaryDatabase.CopyPath(source, "complex-alter-"); + try + { + string before = Remains(path); + using (var db = JetDatabase.Open(path, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery(statement); + + string libredLeft = Remains(path); + output.WriteLine($"before: {before}"); + output.WriteLine($"ACE: {aceLeft}"); + output.WriteLine($"LibRed: {libredLeft}"); + Assert.Equal(aceLeft, libredLeft); + Assert.Equal(before, libredLeft); // and neither engine changed the complex wiring at all + + // And ACE still reads the table LibRed rebuilt. + using OleDbConnection ace = AceTestDatabase.Open(path); + using OleDbCommand count = ace.CreateCommand(); + count.CommandText = $"SELECT COUNT(*) FROM [{table}]"; + output.WriteLine($"ACE reads {Convert.ToInt32(count.ExecuteScalar())} rows"); + } + finally { TemporaryDatabase.Delete(path); } + } + + // Dropping a table that owns complex columns. Each one has a row in MSysComplexColumns and a backing + // f_ flat table holding its values, neither of which the table's own pages account for — so what + // ACE takes with the table, LibRed has to take too, or the file keeps catalog rows pointing at a table + // that is gone and flat tables nothing will ever read. + [Fact] + public void Dropping_a_table_with_complex_columns_leaves_what_ace_leaves() + { + if (!File.Exists(Source)) { output.WriteLine($"Skipped: {Source} not present."); return; } + + string ace = TemporaryDatabase.CopyPath(Source, "complex-drop-ace-"); + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + using (OleDbCommand drop = connection.CreateCommand()) + { + drop.CommandText = "DROP TABLE Table1"; + drop.ExecuteNonQuery(); + } + + string libred = TemporaryDatabase.CopyPath(Source, "complex-drop-lib-"); + using (var db = JetDatabase.Open(libred, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery("DROP TABLE Table1"); + + string aceLeft = Remains(ace), libredLeft = Remains(libred); + output.WriteLine($"ACE: {aceLeft}"); + output.WriteLine($"LibRed: {libredLeft}"); + Assert.Equal(aceLeft, libredLeft); + } + + // And dropping one complex column rather than the whole table: the same row and flat table have to go, and + // whether ACE's SQL will even do it is the first question. + [Fact] + public void Dropping_a_complex_column_leaves_what_ace_leaves() + { + if (!File.Exists(Source)) { output.WriteLine($"Skipped: {Source} not present."); return; } + + string ace = TemporaryDatabase.CopyPath(Source, "complex-dropcol-ace-"); + string? refusal = null; + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + using (OleDbCommand drop = connection.CreateCommand()) + { + drop.CommandText = "ALTER TABLE Table1 DROP COLUMN att2"; + try { drop.ExecuteNonQuery(); } + catch (OleDbException e) { refusal = e.Message.Trim(); } + } + output.WriteLine($"ACE: {refusal ?? "accepted"}"); + + string libred = TemporaryDatabase.CopyPath(Source, "complex-dropcol-lib-"); + Exception? libredRefusal = null; + using (var db = JetDatabase.Open(libred, readOnly: false)) + { + try { new QueryEngine(db).ExecuteNonQuery("ALTER TABLE Table1 DROP COLUMN att2"); } + catch (Exception e) { libredRefusal = e; } + } + output.WriteLine($"LibRed: {libredRefusal?.Message ?? "accepted"}"); + + Assert.Equal(refusal is null, libredRefusal is null); + if (refusal is not null) return; + + string aceLeft = Remains(ace, "Table1"), libredLeft = Remains(libred, "Table1"); + output.WriteLine($"ACE: {aceLeft}"); + output.WriteLine($"LibRed: {libredLeft}"); + Assert.Equal(aceLeft, libredLeft); + } + + /// What the catalog still says about complex columns: the MSysComplexColumns rows by column name, + /// the names of the f_<GUID> flat tables still present, and — the part that says the wiring still + /// WORKS rather than merely still exists — each column that resolves, with the number of records holding + /// values and the number of values. Definition pages are left out deliberately: two engines rebuilding the + /// same table put it on different pages, and that is not a difference in the wiring. + /// With , also that table's own columns and indexes, so a drop that was accepted + /// and did nothing is visible as such. + private static string Remains(string path, string? table = null) + { + using var db = JetDatabase.Open(path); + if (table is not null) + { + TableDefinition owner = db.Catalog.FindTable(table)!; + string columns = string.Join(",", owner.Columns.Select(c => c.Name)); + string indexes = string.Join(",", owner.Indexes.Select(i => i.Name.Split('_')[0]).Order(StringComparer.Ordinal)); + return $"columns=[{columns}] indexes=[{indexes}] {Remains(path)}"; + } + TableDefinition complexColumns = db.Catalog.FindTable("MSysComplexColumns")!; + int name = complexColumns.FindColumn("ColumnName")!.Index; + var rows = db.OpenTable("MSysComplexColumns").Rows() + .Select(r => Convert.ToString(r[name]) ?? "?") + .Order(StringComparer.Ordinal); + var flat = db.Catalog.Tables + .Where(t => t.Name.StartsWith("f_", StringComparison.Ordinal)) + .Select(t => t.Name[^12..]) // the trailing _ part, which the GUID prefix buries + .Order(StringComparer.Ordinal); + var resolved = db.Catalog.ComplexColumns + .Select(c => + { + var values = db.OpenTable(c.FlatTable.Name).Rows() + .Select(r => Convert.ToInt32(r[c.OwnerLink.Index])) + .ToList(); + return $"{c.OwnerTable.Name}.{c.ColumnName}:{values.Distinct().Count()}/{values.Count}"; + }) + .Order(StringComparer.Ordinal); + return $"MSysComplexColumns=[{string.Join(",", rows)}] flat=[{string.Join(",", flat)}] " + + $"resolved=[{string.Join(",", resolved)}]"; + } + [Fact] public void Ace_reads_back_a_table_libred_inserted_into_and_deleted_from() { @@ -29,7 +185,7 @@ public void Ace_reads_back_a_table_libred_inserted_into_and_deleted_from() using (var db = JetDatabase.Open(path, readOnly: false)) { var engine = new QueryEngine(db); - TableDef table = db.Catalog.FindTable("Table1")!; + TableDefinition table = db.Catalog.FindTable("Table1")!; ColumnDef idColumn = table.Columns[0]; complexColumnCount = db.Catalog.ComplexColumns.Count(c => c.OwnerTable.Name == "Table1"); @@ -63,7 +219,7 @@ public void Ace_reads_back_a_table_libred_inserted_into_and_deleted_from() } // The new row took one complex id, shared by every complex column. - TableDef table = db.Catalog.FindTable("Table1")!; + TableDefinition table = db.Catalog.FindTable("Table1")!; object?[] newest = db.OpenTable("Table1").Rows() .Last(r => !idsBefore.Contains(Convert.ToInt32(r[table.Columns[0].Index]))); var ids = db.Catalog.ComplexColumns.Where(c => c.OwnerTable.Name == "Table1") @@ -91,4 +247,4 @@ public void Ace_reads_back_a_table_libred_inserted_into_and_deleted_from() } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ConcurrentAllocationTests.cs b/test/LibRed.Engine.AccessTests/ConcurrentAllocationTests.cs index e51472bd4..1a29c49ac 100644 --- a/test/LibRed.Engine.AccessTests/ConcurrentAllocationTests.cs +++ b/test/LibRed.Engine.AccessTests/ConcurrentAllocationTests.cs @@ -31,7 +31,8 @@ public void Encrypted_concurrent_allocations_conflict_then_retry_and_remain_read try { CreateTable(path); - DatabaseEncryption.SetPassword(path, password, AccessEncryption.Agile); + using (var db = JetDatabase.Open(path, readOnly: false, exclusive: true)) + DatabaseEncryption.SetPassword(db, password, AccessEncryption.Agile); RunConflictAndRetry(path, password); VerifyWithLibRed(path, password); diff --git a/test/LibRed.Engine.AccessTests/CreateTableUsageMapOrderAccessTests.cs b/test/LibRed.Engine.AccessTests/CreateTableUsageMapOrderAccessTests.cs new file mode 100644 index 000000000..81afc6104 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/CreateTableUsageMapOrderAccessTests.cs @@ -0,0 +1,89 @@ +using System.Data.OleDb; +using LibRed; +using LibRed.Catalog; +using LibRed.IO; +using LibRed.Pages; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Which usage-map row CREATE TABLE gives each index and each long-value column. Rows 0 and 1 are the table's +/// own owned/free maps; the rest follow the order the STATEMENT declares them — a long-value column takes two +/// rows where its column is written, an index takes one where its constraint is written. An index's row is +/// therefore not a function of the table's shape, and the same table written two ways lays out differently. +/// +/// +/// Both engines run the identical statement here, because the order is a property of the text: driving LibRed +/// through the Core API instead would have nothing to reproduce it from. +/// +[Collection(AceCollection.Name)] +public class CreateTableUsageMapOrderAccessTests(ITestOutputHelper output) +{ + [Theory] + // A constraint written before the long-value column takes the earlier row… + [InlineData("inline primary key", "CREATE TABLE T (Id LONG PRIMARY KEY, M MEMO)")] + [InlineData("named inline primary key", "CREATE TABLE T (Id LONG CONSTRAINT pk PRIMARY KEY, M MEMO)")] + // An inline constraint IS the tie case — it sits at the same point in the list as the column it is written + // on, and takes the row first. `CREATE TABLE T (Id LONG, CONSTRAINT pk PRIMARY KEY (Id), M MEMO)` would say + // the same thing about a TABLE-level constraint, and ACE gives it row 2 there, which is what rules out + // "table-level constraints come last". It is not a case here because LibRed's grammar does not accept a + // table constraint with columns still to come — every columnDefinition must precede every tableConstraint. + // …and one written after it takes the later row, which is the shape a migration emits. + [InlineData("table constraint after the memo", "CREATE TABLE T (Id LONG, M MEMO, CONSTRAINT pk PRIMARY KEY (Id))")] + [InlineData("memo column first", "CREATE TABLE T (M MEMO, Id LONG CONSTRAINT pk PRIMARY KEY)")] + // Several of each, interleaved both ways round. + [InlineData("two indexes then two memos", + "CREATE TABLE T (Id LONG CONSTRAINT pk PRIMARY KEY, A LONG CONSTRAINT u UNIQUE, M MEMO, N MEMO)")] + [InlineData("an index between two memos", "CREATE TABLE T (M MEMO, A LONG CONSTRAINT u UNIQUE, N MEMO)")] + [InlineData("a foreign key after a memo", + "CREATE TABLE T (Id LONG CONSTRAINT pk PRIMARY KEY, M MEMO, PId LONG CONSTRAINT fk REFERENCES T (Id))")] + // Controls: nothing to interleave. + [InlineData("memo only", "CREATE TABLE T (Id LONG, M MEMO)")] + [InlineData("index only", "CREATE TABLE T (Id LONG CONSTRAINT pk PRIMARY KEY, A LONG)")] + [InlineData("OLE rather than memo", "CREATE TABLE T (Id LONG CONSTRAINT pk PRIMARY KEY, B LONGBINARY)")] + public void Usage_map_rows_follow_the_order_the_statement_declares_them(string label, string sql) + { + string fixture = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); + string ace = TemporaryDatabase.CopyPath(fixture, "maporder-ace-"); + string libred = TemporaryDatabase.CopyPath(fixture, "maporder-lib-"); + try + { + using (var connection = AceTestDatabase.Open(ace)) + using (OleDbCommand command = connection.CreateCommand()) + { + command.CommandText = sql; + command.ExecuteNonQuery(); + } + + using (var db = JetDatabase.Open(libred, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery(sql); + + string expected = MapLayout(ace), actual = MapLayout(libred); + output.WriteLine($"{label}: ACE {expected} | LibRed {actual}"); + Assert.Equal(expected, actual); + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(libred); + } + } + + /// Each real index's usage-map row by data-block ordinal, then each long-value column's + /// owned/free rows by column id — the layout of the table's primary usage-map page. + private static string MapLayout(string path) + { + using var channel = PageChannel.Open(path, readOnly: true); + TableDefinition table = new JetCatalog(channel).FindTable("T")!; + var definition = new TableDefinition(); + definition.Read(channel, table.DefinitionPage); + + var parts = new List(); + foreach (IndexDef index in definition.Indexes.OrderBy(i => i.RealIndexOrdinal)) + parts.Add($"index{index.RealIndexOrdinal}=row{index.UsageMap.Row}"); + foreach ((int columnId, (int owned, int _)) in definition.LongValueOwnedMaps.OrderBy(e => e.Key)) + parts.Add($"column{columnId}=rows{owned}/{definition.LongValueFreeMaps[columnId].Item1}"); + return string.Join(" ", parts); + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/DateTime2CreatedDatabaseAccessTests.cs b/test/LibRed.Engine.AccessTests/DateTime2CreatedDatabaseAccessTests.cs index b9fb1b8d8..b7c2a15fb 100644 --- a/test/LibRed.Engine.AccessTests/DateTime2CreatedDatabaseAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/DateTime2CreatedDatabaseAccessTests.cs @@ -44,10 +44,10 @@ public void Ace_opens_a_database_libred_upgraded_in_place_and_reads_the_datetime { // Created at the DEFAULT format — ACE 12, the one that cannot hold the type. LibRedConnection.CreateDatabase($"Data Source={path}"); - Assert.Equal(0x02, VersionByte(path)); using (var db = JetDatabase.Open(path, readOnly: false)) { + Assert.Equal(JetVersion.Version12_2007, db.Format.Version); var engine = new QueryEngine(db); engine.ExecuteNonQuery("CREATE TABLE `E` (`Id` INTEGER PRIMARY KEY, `V` DATETIME2 NULL)"); Assert.Equal(JetVersion.Version17_2019, db.Format.Version); @@ -56,7 +56,8 @@ public void Ace_opens_a_database_libred_upgraded_in_place_and_reads_the_datetime new Dictionary { ["v"] = value }); } - Assert.Equal(0x06, VersionByte(path)); + using (var reopened = JetDatabase.Open(path)) + Assert.Equal(JetVersion.Version17_2019, reopened.Format.Version); using var connection = AceTestDatabase.Open(path); using var command = connection.CreateCommand(); @@ -72,13 +73,6 @@ public void Ace_opens_a_database_libred_upgraded_in_place_and_reads_the_datetime finally { TemporaryDatabase.Delete(path); } } - private static byte VersionByte(string path) - { - using var stream = File.OpenRead(path); - stream.Seek(0x14, SeekOrigin.Begin); - return (byte)stream.ReadByte(); - } - [Fact] public void Ace_reads_a_datetime2_value_libred_wrote_into_a_database_libred_created() { @@ -114,4 +108,4 @@ public void Ace_reads_a_datetime2_value_libred_wrote_into_a_database_libred_crea } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/DateTime2LocaleAccessTests.cs b/test/LibRed.Engine.AccessTests/DateTime2LocaleAccessTests.cs index 6b7d0d6d8..aafb06bfe 100644 --- a/test/LibRed.Engine.AccessTests/DateTime2LocaleAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/DateTime2LocaleAccessTests.cs @@ -1,8 +1,5 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Catalog; -using LibRed.Formats; -using LibRed.IO; using Xunit; namespace LibRed.Engine.Tests; @@ -46,11 +43,17 @@ public void Ace_gives_date_time_extended_only_the_primary_language_id(string loc foreach ((string name, byte[] bytes) in columns) output.WriteLine($"{locale} {name}: {Convert.ToHexString(bytes)}"); - byte[] text = columns["T"], extended = columns["V"]; - Assert.Equal(langId, text[0x0B] | (text[0x0C] << 8)); // the control: whole LANGID - Assert.Equal(langId & 0x00FF, extended[0x0B] | (extended[0x0C] << 8)); - Assert.Equal(0, extended[0x0D]); - Assert.Equal(0, extended[0x0E]); + Collation text, extended; + using (var database = JetDatabase.Open(path, readOnly: true)) + { + TableDefinition dt = database.Catalog.FindTable("DT")!; + text = dt.RequireColumn("T").Collation; + extended = dt.RequireColumn("V").Collation; + } + Assert.Equal(langId, (int)text.Order); // the control: whole LANGID + Assert.Equal(langId & 0x00FF, (int)extended.Order); + Assert.Equal(0, extended.SortId); + Assert.Equal(0, extended.Version); } finally { TemporaryDatabase.Delete(path); } } @@ -114,33 +117,12 @@ private static string Recollated(object engine, string locale) finally { TemporaryDatabase.Delete(source); } } - /// Every column's descriptor bytes, by name. A three-column table's definition fits one page. - /// + /// Every column's descriptor bytes, by name, as read from the table's definition. private static Dictionary Descriptors(string path, string table) { - int definitionPage; - using (var database = JetDatabase.Open(path, readOnly: true)) - definitionPage = database.Catalog.FindTable(table)!.DefinitionPage; - - using var channel = PageChannel.Open(path, readOnly: true); - JetFormatBase format = channel.Format; - byte[] page = channel.ReadPage(definitionPage).Span.ToArray(); - - int dataCount = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(format.TdefIndexCountOffset, 4)); - int colCount = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(format.TdefColumnCountOffset, 2)); - int start = format.TdefRealIndexBlockOffset + dataCount * format.RealIndexEntrySize; - int namePos = start + colCount * format.ColumnDescriptorSize; - - var result = new Dictionary(StringComparer.Ordinal); - for (int i = 0; i < colCount; i++) - { - int len = BinaryPrimitives.ReadUInt16LittleEndian(page.AsSpan(namePos, 2)); - string name = System.Text.Encoding.Unicode.GetString(page.AsSpan(namePos + 2, len)); - namePos += 2 + len; - result[name] = page.AsSpan(start + i * format.ColumnDescriptorSize, - format.ColumnDescriptorSize).ToArray(); - } - return result; + using var database = JetDatabase.Open(path, readOnly: true); + return database.Catalog.FindTable(table)!.Columns + .ToDictionary(c => c.Name, c => c.RawDescriptor!, StringComparer.Ordinal); } /// A DAO engine, once this ACE is known to take DATETIME2. An ACE below 17 cannot create the column @@ -155,4 +137,4 @@ private static object DateTime2AndDao() Assert.SkipWhen(engine is null, "DAO is unavailable in this process."); return engine!; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/DeleteByteParityTests.cs b/test/LibRed.Engine.AccessTests/DeleteByteParityTests.cs index 22016cfad..4603d40d6 100644 --- a/test/LibRed.Engine.AccessTests/DeleteByteParityTests.cs +++ b/test/LibRed.Engine.AccessTests/DeleteByteParityTests.cs @@ -17,8 +17,6 @@ namespace LibRed.Engine.Tests; [Collection(AceCollection.Name)] public class DeleteByteParityTests(ITestOutputHelper output) { - private const int PageSize = 4096; - [Fact] public void A_delete_writes_the_same_bytes_ace_writes() { @@ -52,13 +50,17 @@ public void A_delete_writes_the_same_bytes_ace_writes() } finally { Invoke(quiet, "Close"); } + int pageSize; using (var db = JetDatabase.Open(libredCopy, readOnly: false)) + { + pageSize = db.Format.PageSize; new QueryEngine(db).ExecuteNonQuery($"DELETE FROM [{Table}] WHERE {Where}"); + } byte[] o = File.ReadAllBytes(orig), a = File.ReadAllBytes(aceCopy), n = File.ReadAllBytes(noiseCopy), l = File.ReadAllBytes(libredCopy); - HashSet aceWrote = Changed(o, a), housekeeping = Changed(o, n), libredWrote = Changed(o, l); + HashSet aceWrote = Changed(o, a, pageSize), housekeeping = Changed(o, n, pageSize), libredWrote = Changed(o, l, pageSize); var aceDelete = aceWrote.Except(housekeeping).ToHashSet(); // Nothing LibRed touched is a page ACE left alone. @@ -70,8 +72,8 @@ public void A_delete_writes_the_same_bytes_ace_writes() Assert.NotEmpty(both); foreach (int page in both) Assert.True( - a.AsSpan(page * PageSize, PageSize).SequenceEqual(l.AsSpan(page * PageSize, PageSize)), - $"page {page} (type 0x{o[page * PageSize]:X2}) differs from ACE's"); + a.AsSpan(page * pageSize, pageSize).SequenceEqual(l.AsSpan(page * pageSize, pageSize)), + $"page {page} (type 0x{o[page * pageSize]:X2}) differs from ACE's"); output.WriteLine($"{both.Length} pages written by both engines, all identical"); } @@ -81,14 +83,83 @@ public void A_delete_writes_the_same_bytes_ace_writes() } } - private static HashSet Changed(byte[] left, byte[] right) + // A delete that gives space back to a page which had none. The page is the first of many, so it was full + // and its bit in the table's free-pages map had been cleared; freeing space in it is what puts the bit back, + // and a page whose bit stays clear is space no insert will ever use again. Both directions are asserted + // here: ACE writing a page LibRed leaves alone is exactly the shape of that leak. + [Fact] + public void A_delete_that_frees_space_in_a_full_page_writes_what_ace_writes() + { + object? engine = CreateDbEngine(); + if (engine is null) { output.WriteLine("Skipped: DAO unavailable."); return; } + + string northwind = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); + string orig = TemporaryDatabase.CopyPath(northwind, "delfree-orig-"); + try + { + using (var db = JetDatabase.Open(orig, readOnly: false)) + { + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE TABLE Filled (Id LONG CONSTRAINT pk PRIMARY KEY, V TEXT(200))"); + for (int i = 0; i < 200; i++) + e.ExecuteNonQuery($"INSERT INTO Filled (Id, V) VALUES ({i}, '{new string((char)('a' + i % 26), 200)}')"); + } + + string aceCopy = TemporaryDatabase.CopyPath(orig, "delfree-ace-"); + string noiseCopy = TemporaryDatabase.CopyPath(orig, "delfree-noise-"); + string libredCopy = TemporaryDatabase.CopyPath(orig, "delfree-libred-"); + + object ace = Invoke(engine, "OpenDatabase", aceCopy)!; + try + { + object rs = Invoke(ace, "OpenRecordset", "SELECT * FROM Filled WHERE Id = 0")!; + try { Invoke(rs, "Delete"); } + finally { Invoke(rs, "Close"); } + } + finally { Invoke(ace, "Close"); } + + object quiet = Invoke(engine, "OpenDatabase", noiseCopy)!; + try + { + object rs = Invoke(quiet, "OpenRecordset", "SELECT * FROM Filled")!; + Invoke(rs, "Close"); + } + finally { Invoke(quiet, "Close"); } + + int pageSize; + using (var db = JetDatabase.Open(libredCopy, readOnly: false)) + { + pageSize = db.Format.PageSize; + new QueryEngine(db).ExecuteNonQuery("DELETE FROM Filled WHERE Id = 0"); + } + + byte[] o = File.ReadAllBytes(orig), a = File.ReadAllBytes(aceCopy), + n = File.ReadAllBytes(noiseCopy), l = File.ReadAllBytes(libredCopy); + // Page 0 is left out: its modification counter moves for reasons that have nothing to do with the + // delete, which is why the whole-file comparisons skip it too. + var aceDelete = Changed(o, a, pageSize).Except(Changed(o, n, pageSize)).Where(p => p != 0).ToHashSet(); + var libredWrote = Changed(o, l, pageSize).Where(p => p != 0).ToHashSet(); + + output.WriteLine($"ACE wrote [{string.Join(",", aceDelete.Order())}], " + + $"LibRed wrote [{string.Join(",", libredWrote.Order())}]"); + Assert.Empty(libredWrote.Except(aceDelete).Order()); + Assert.Empty(aceDelete.Except(libredWrote).Order()); + foreach (int page in aceDelete) + Assert.True( + a.AsSpan(page * pageSize, pageSize).SequenceEqual(l.AsSpan(page * pageSize, pageSize)), + $"page {page} (type 0x{o[page * pageSize]:X2}) differs from ACE's"); + } + finally { TemporaryDatabase.Delete(orig); } + } + + private static HashSet Changed(byte[] left, byte[] right, int pageSize) { var changed = new HashSet(); - int pages = Math.Min(left.Length, right.Length) / PageSize; + int pages = Math.Min(left.Length, right.Length) / pageSize; for (int p = 0; p < pages; p++) - if (!left.AsSpan(p * PageSize, PageSize).SequenceEqual(right.AsSpan(p * PageSize, PageSize))) + if (!left.AsSpan(p * pageSize, pageSize).SequenceEqual(right.AsSpan(p * pageSize, pageSize))) changed.Add(p); - for (int p = pages; p < Math.Max(left.Length, right.Length) / PageSize; p++) changed.Add(p); + for (int p = pages; p < Math.Max(left.Length, right.Length) / pageSize; p++) changed.Add(p); return changed; } @@ -107,4 +178,4 @@ private static HashSet Changed(byte[] left, byte[] right) } return null; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/DeletedRowSpaceAccessTests.cs b/test/LibRed.Engine.AccessTests/DeletedRowSpaceAccessTests.cs index 9df26c0f0..c3de2101f 100644 --- a/test/LibRed.Engine.AccessTests/DeletedRowSpaceAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/DeletedRowSpaceAccessTests.cs @@ -1,7 +1,7 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Formats; using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Engine.Tests; @@ -100,18 +100,18 @@ private static string Directory(string[] statements, Action ru for (int page = 1; page < channel.PageCount; page++) { byte[] bytes = channel.ReadPage(page).Span.ToArray(); - if (bytes[0] != 0x01) continue; - if (BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(4, 4)) != definitionPage) continue; + if (PageHeader.ReadType(bytes) != PageType.DataPage) continue; + if ((int)DataPage.ReadOwner(bytes, format) != definitionPage) continue; - int slots = BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataRowCountOffset, 2)); + int slots = DataPage.ReadRowCount(bytes, format); var entries = Enumerable.Range(0, slots) - .Select(i => BinaryPrimitives.ReadUInt16LittleEndian( - bytes.AsSpan(format.DataRowDirectoryOffset + i * 2, 2)).ToString("X4")); - return $"free={BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataFreeSpaceOffset, 2))} " + .Select(i => DataPage.ReadSlot(bytes, format, i)) + .Select(s => ((int)s.Flags | s.Offset).ToString("X4")); + return $"free={DataPage.ReadFreeSpace(bytes, format)} " + $"dir=[{string.Join(" ", entries)}]"; } return "no data page"; } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/GuidColumnStorageAccessTests.cs b/test/LibRed.Engine.AccessTests/GuidColumnStorageAccessTests.cs index 24bab77c4..7312446eb 100644 --- a/test/LibRed.Engine.AccessTests/GuidColumnStorageAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/GuidColumnStorageAccessTests.cs @@ -22,7 +22,7 @@ public void Guid_and_uniqueidentifier_both_declare_a_variable_column() new QueryEngine(database).ExecuteNonQuery( "CREATE TABLE `W` (`Id` INTEGER PRIMARY KEY, `G` GUID NULL, `H` UNIQUEIDENTIFIER NULL)"); - TableDef table = database.Catalog.FindTable("W")!; + TableDefinition table = database.Catalog.FindTable("W")!; Assert.All(table.Columns.Where(c => c.Name is "G" or "H"), column => { Assert.False(column.IsFixedLength); @@ -55,4 +55,4 @@ public void A_wide_guid_table_no_longer_spends_fixed_record_budget() private static string Copy(string prefix) => TemporaryDatabase.CopyPath( Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), prefix); -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/IndexRootSplitByteParityTests.cs b/test/LibRed.Engine.AccessTests/IndexRootSplitByteParityTests.cs new file mode 100644 index 000000000..4c41e54e6 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/IndexRootSplitByteParityTests.cs @@ -0,0 +1,272 @@ +using System.Data.OleDb; +using LibRed; +using LibRed.Formats; +using LibRed.Pages; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Index splits that ACE and LibRed must write byte for byte alike: a root that splits (it keeps its page and +/// both halves move out), the cut at every position a new key can land, and the cut on a leaf whose entries +/// differ in length, where it is decided by bytes rather than by count. +/// +/// +/// Each case seeds a table through ACE, then applies one further insert to one copy through ACE and to another +/// through LibRed, and compares every page the table's indexes own — the dead bytes past each page's live end +/// included, since ACE's are reproducible. +/// +[Collection(AceCollection.Name)] +public class IndexRootSplitByteParityTests +{ + // 204-character keys, so a leaf holds 17 and splits on the 18th. + private static string Wide(int i) => $"K{i:D4}" + new string('x', 200); + + [Theory] + [InlineData(95, "a key in the upper half")] + [InlineData(5, "a key that sorts first")] + public void A_splitting_root_leaf_keeps_its_page(int key, string _) + { + string seeded = Seed("CREATE TABLE RootSplit (k TEXT(255) CONSTRAINT pk PRIMARY KEY)", + Enumerable.Range(1, 17).Select(i => $"INSERT INTO RootSplit (k) VALUES ('{Wide(i * 10)}')")); + try { AssertInsertMatchesAce(seeded, "RootSplit", $"INSERT INTO RootSplit (k) VALUES ('{Wide(key)}')"); } + finally { TemporaryDatabase.Delete(seeded); } + } + + [Fact] + public void A_splitting_root_node_keeps_its_page() + { + // Keys spread so every leaf keeps taking new ones: the root node fills at 16 separators and the 17th + // splits it. Found by inserting until the root's own children became nodes. + static string Spread(int i) => $"{(char)('A' + i * 7 % 26)}{i:D3}" + new string((char)('a' + i % 26), 200); + string seeded = Seed("CREATE TABLE NodeSplit (k TEXT(255) CONSTRAINT pk PRIMARY KEY)", + Enumerable.Range(1, 226).Select(i => $"INSERT INTO NodeSplit (k) VALUES ('{Spread(i)}')")); + try + { + byte[] before = File.ReadAllBytes(seeded); + int pageSize = JetFormatBase.Detect(new MemoryStream(before)).PageSize; + Assert.True(IndexPages(before, Tdef(seeded, "NodeSplit")) + .Count(p => PageHeader.ReadType(before.AsSpan(p * pageSize)) == PageType.IntermediateIndexPage) == 1, + "The seed was meant to leave a single root node."); + AssertInsertMatchesAce(seeded, "NodeSplit", $"INSERT INTO NodeSplit (k) VALUES ('{Spread(227)}')"); + } + finally { TemporaryDatabase.Delete(seeded); } + } + + [Fact] + public void A_leaf_is_cut_where_ace_cuts_it_at_every_position() + { + // A non-root leaf of 17 equal entries whose parent has room: 180…340 on the right of a first split. + string seeded = Seed("CREATE TABLE SplitPos (k TEXT(255) CONSTRAINT pk PRIMARY KEY)", + Enumerable.Range(1, 34).Select(i => $"INSERT INTO SplitPos (k) VALUES ('{Wide(i * 10)}')")); + try + { + for (int position = 0; position <= 17; position++) + AssertInsertMatchesAce(seeded, "SplitPos", $"INSERT INTO SplitPos (k) VALUES ('{Wide(175 + position * 10)}')"); + } + finally { TemporaryDatabase.Delete(seeded); } + } + + [Fact] + public void A_leaf_of_unequal_entries_is_cut_by_bytes() + { + // Labels 0…144 in id order land all over the text index; entries are 38 to 40 bytes stored, and the + // 146th splits a 91-entry leaf 45 and 47 where halving the count would keep 46. + static string Label(int i) => $"Bulk label number {i} " + new string('L', 30); + string seeded = Seed("CREATE TABLE Labels (Id LONG CONSTRAINT pkLabels PRIMARY KEY, Label TEXT(60))", + new[] { "CREATE INDEX ixLabel ON Labels (Label)" } + .Concat(Enumerable.Range(0, 145).Select(i => $"INSERT INTO Labels (Id, Label) VALUES ({i}, '{Label(i)}')"))); + try { AssertInsertMatchesAce(seeded, "Labels", $"INSERT INTO Labels (Id, Label) VALUES (145, '{Label(145)}')"); } + finally { TemporaryDatabase.Delete(seeded); } + } + + [Fact] + public void An_ascending_load_splits_nodes_at_the_right_edge_as_ace_does() + { + // 400 ascending keys fill 24 leaves of 17, so the root node overflows while each new separator is its + // last entry. The whole load goes through each engine from the same empty table. A load this size also + // takes ACE past the point where it places pages differently (in aligned groups), so the trees are + // compared by shape — each level left to right — rather than by page number. + string empty = Seed("CREATE TABLE Ascending (k TEXT(255) CONSTRAINT pk PRIMARY KEY)", []); + string ace = TemporaryDatabase.CopyPath(empty, "ascending-ace-"); + string libred = TemporaryDatabase.CopyPath(empty, "ascending-libred-"); + try + { + string[] inserts = [.. Enumerable.Range(1, 400).Select(i => $"INSERT INTO Ascending (k) VALUES ('{Wide(i)}')")]; + using (OleDbConnection conn = AceTestDatabase.Open(ace)) + foreach (string sql in inserts) + { + using OleDbCommand cmd = conn.CreateCommand(); + cmd.CommandText = sql; + cmd.ExecuteNonQuery(); + } + int pageSize; + using (var db = JetDatabase.Open(libred, readOnly: false)) + { + pageSize = db.Format.PageSize; + var engine = new QueryEngine(db); + foreach (string sql in inserts) engine.ExecuteNonQuery(sql); + } + + int tdef = Tdef(empty, "Ascending"); + byte[] a = File.ReadAllBytes(ace), l = File.ReadAllBytes(libred); + Assert.True(IndexPages(a, tdef).Count(p => PageHeader.ReadType(a.AsSpan(p * pageSize)) == PageType.IntermediateIndexPage) >= 3, + "The load was meant to split a node."); + Assert.Equal(Shape(a, tdef), Shape(l, tdef)); + } + finally + { + foreach (string path in new[] { empty, ace, libred }) TemporaryDatabase.Delete(path); + } + } + + // CREATE INDEX over existing rows writes the tree that inserting the keys one at a time in order would: the + // root keeps its page, full leaves are compressed before they split, nodes split at the right edge, and + // pages come from the allocator in the order those splits happen. 40 rows give one node over three leaves; + // 400 give two levels of nodes, the lower one split. + [Theory] + [InlineData(40)] + [InlineData(400)] + public void Create_index_writes_what_ace_writes(int rows) + { + string seeded = Seed("CREATE TABLE Indexed (Id LONG, K TEXT(255))", Enumerable.Range(1, rows).Select(i => + $"INSERT INTO Indexed VALUES ({i}, '{(char)('A' + i * 7 % 26)}{i:D4}{new string((char)('a' + i % 26), 200)}')")); + try + { + int tdef = Tdef(seeded, "Indexed"); + string ace = TemporaryDatabase.CopyPath(seeded, "createindex-ace-"); + string libred = TemporaryDatabase.CopyPath(seeded, "createindex-libred-"); + try + { + const string ddl = "CREATE INDEX ixK ON Indexed (K)"; + using (OleDbConnection conn = AceTestDatabase.Open(ace)) + using (OleDbCommand cmd = conn.CreateCommand()) + { + cmd.CommandText = ddl; + cmd.ExecuteNonQuery(); + } + int pageSize; + using (var db = JetDatabase.Open(libred, readOnly: false)) + { + pageSize = db.Format.PageSize; + new QueryEngine(db).ExecuteNonQuery(ddl); + } + + byte[] a = File.ReadAllBytes(ace), l = File.ReadAllBytes(libred); + Assert.True(a.Length == l.Length, $"ACE's file has {a.Length / pageSize} pages, LibRed's {l.Length / pageSize}."); + Assert.True(IndexPages(a, tdef).Any(p => PageHeader.ReadType(a.AsSpan(p * pageSize)) == PageType.IntermediateIndexPage), + "The index was meant to have a node."); + // The index's pages, and the definition that points at its root. + foreach (int page in IndexPages(a, tdef).Union(IndexPages(l, tdef)).Append(tdef)) + { + int differ = Enumerable.Range(0, pageSize).FirstOrDefault(b => a[page * pageSize + b] != l[page * pageSize + b], -1); + Assert.True(differ < 0, $"{rows} rows: page {page} differs from ACE's at 0x{differ:X3}."); + } + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(libred); + } + } + finally { TemporaryDatabase.Delete(seeded); } + } + + private static string Seed(string create, IEnumerable statements) + { + string seeded = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "rootsplit-seed-"); + using OleDbConnection conn = AceTestDatabase.Open(seeded); + foreach (string sql in statements.Prepend(create)) + { + using OleDbCommand cmd = conn.CreateCommand(); + cmd.CommandText = sql; + cmd.ExecuteNonQuery(); + } + return seeded; + } + + private static int Tdef(string path, string table) + { + using var db = JetDatabase.Open(path); + return db.Catalog.FindTable(table)!.DefinitionPage; + } + + private static void AssertInsertMatchesAce(string seeded, string table, string insert) + { + int tdef = Tdef(seeded, table); + string ace = TemporaryDatabase.CopyPath(seeded, "rootsplit-ace-"); + string libred = TemporaryDatabase.CopyPath(seeded, "rootsplit-libred-"); + try + { + using (OleDbConnection conn = AceTestDatabase.Open(ace)) + using (OleDbCommand cmd = conn.CreateCommand()) + { + cmd.CommandText = insert; + cmd.ExecuteNonQuery(); + } + int pageSize; + using (var db = JetDatabase.Open(libred, readOnly: false)) + { + pageSize = db.Format.PageSize; + new QueryEngine(db).ExecuteNonQuery(insert); + } + + byte[] a = File.ReadAllBytes(ace), l = File.ReadAllBytes(libred); + Assert.True(IndexPages(a, tdef).Count > IndexPages(File.ReadAllBytes(seeded), tdef).Count, + $"{insert[..Math.Min(60, insert.Length)]}…: the insert was meant to split a page."); + Assert.True(a.Length == l.Length, $"{insert}: ACE's file has {a.Length / pageSize} pages, LibRed's {l.Length / pageSize}."); + foreach (int page in IndexPages(a, tdef).Union(IndexPages(l, tdef))) + { + int differ = Enumerable.Range(0, pageSize).FirstOrDefault( + b => a[page * pageSize + b] != l[page * pageSize + b], -1); + Assert.True(differ < 0, $"{insert[..Math.Min(60, insert.Length)]}…: page {page} differs from ACE's at 0x{differ:X3}."); + } + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(libred); + } + } + + /// Each level of the tree, top down and left to right along the sibling links, as + /// entries/prefix/free per page — everything about the tree but where its pages are. + private static string Shape(byte[] file, int tdef) + { + List pages = IndexPages(file, tdef); + JetFormatBase format = JetFormatBase.Detect(new MemoryStream(file)); + int pageSize = format.PageSize; + var levels = new List(); + foreach (IGrouping level in pages.GroupBy(p => file[p * pageSize + format.IndexLevelOffset]).OrderByDescending(g => g.Key)) + { + var byPage = level.ToHashSet(); + var shapes = new List(); + for (int p = level.Single(q => IndexTree.ReadSiblings(file.AsSpan(q * pageSize, pageSize), format).Previous == 0); + p != 0 && byPage.Contains(p); + p = IndexTree.ReadSiblings(file.AsSpan(p * pageSize, pageSize), format).Next) + { + int entries = 0; + for (int i = format.IndexEntryMaskOffset; i < format.IndexEntryDataOffset; i++) + entries += System.Numerics.BitOperations.PopCount(file[p * pageSize + i]); + shapes.Add($"{entries}/{IndexTree.ReadCompressedByteCount(file.AsSpan(p * pageSize, pageSize), format)}/" + + $"{IndexTree.ReadFreeSpace(file.AsSpan(p * pageSize, pageSize), format)}"); + } + levels.Add($"level {level.Key}: {string.Join(" ", shapes)}"); + } + return string.Join("\n", levels); + } + + private static List IndexPages(byte[] file, int tdef) + { + JetFormatBase format = JetFormatBase.Detect(new MemoryStream(file)); + int pageSize = format.PageSize; + var pages = new List(); + for (int p = 0; p < file.Length / pageSize; p++) + if (PageHeader.ReadType(file.AsSpan(p * pageSize)) is PageType.IntermediateIndexPage or PageType.LeafIndexPage + && IndexTree.ReadOwner(file.AsSpan(p * pageSize, pageSize), format) == tdef) + pages.Add(p); + return pages; + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/IndexSplitByteParityTests.cs b/test/LibRed.Engine.AccessTests/IndexSplitByteParityTests.cs index dc351f892..c71d6a005 100644 --- a/test/LibRed.Engine.AccessTests/IndexSplitByteParityTests.cs +++ b/test/LibRed.Engine.AccessTests/IndexSplitByteParityTests.cs @@ -1,6 +1,8 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed; +using LibRed.Formats; +using LibRed.Pages; +using LibRed.Storage; using Xunit; namespace LibRed.Engine.Tests; @@ -18,7 +20,6 @@ namespace LibRed.Engine.Tests; [Collection(AceCollection.Name)] public class IndexSplitByteParityTests(ITestOutputHelper output) { - private const int PageSize = 4096; private const int EvenKeys = 900; [Fact] @@ -67,28 +68,34 @@ private void AssertSplitMatchesAce(string seeded, int key) new QueryEngine(db).ExecuteNonQuery($"INSERT INTO SplitGuard (k) VALUES ({key})"); int tdef; - using (var db = JetDatabase.Open(seeded)) tdef = db.Catalog.FindTable("SplitGuard")!.DefinitionPage; + JetFormatBase format; + using (var db = JetDatabase.Open(seeded)) + { + tdef = db.Catalog.FindTable("SplitGuard")!.DefinitionPage; + format = db.Format; + } + int pageSize = format.PageSize; byte[] a = File.ReadAllBytes(aceCopy), l = File.ReadAllBytes(libredCopy); - int pages = Math.Min(a.Length, l.Length) / PageSize; + int pages = Math.Min(a.Length, l.Length) / pageSize; int checkedPages = 0; for (int p = 0; p < pages; p++) { // Only this index's own pages: the rest of the file is ACE's bookkeeping. - if (!IsOurIndexPage(a, p, tdef) && !IsOurIndexPage(l, p, tdef)) continue; + if (!IsOurIndexPage(a, p, tdef, format) && !IsOurIndexPage(l, p, tdef, format)) continue; - int aceFree = BinaryPrimitives.ReadUInt16LittleEndian(a.AsSpan(p * PageSize + 2, 2)); - int libFree = BinaryPrimitives.ReadUInt16LittleEndian(l.AsSpan(p * PageSize + 2, 2)); + int aceFree = IndexTree.ReadFreeSpace(a.AsSpan(p * pageSize, pageSize), format); + int libFree = IndexTree.ReadFreeSpace(l.AsSpan(p * pageSize, pageSize), format); Assert.True(aceFree == libFree, $"key {key}, page {p}: free space {libFree} where ACE wrote {aceFree} — the leaf was cut elsewhere."); // Past the live end a page keeps whatever it held, and a page ACE appended past the old // end-of-file keeps ACE's uninitialised buffer, which is not reproducible. Compare the live // region, which is the split itself. - int live = PageSize - aceFree; + int live = pageSize - aceFree; Assert.True( - a.AsSpan(p * PageSize, live).SequenceEqual(l.AsSpan(p * PageSize, live)), + a.AsSpan(p * pageSize, live).SequenceEqual(l.AsSpan(p * pageSize, live)), $"key {key}, page {p}: live bytes differ from ACE's."); checkedPages++; } @@ -103,8 +110,8 @@ private void AssertSplitMatchesAce(string seeded, int key) } } - private static bool IsOurIndexPage(byte[] file, int page, int tdef) => - (page + 1) * PageSize <= file.Length - && file[page * PageSize] is 0x03 or 0x04 - && BinaryPrimitives.ReadInt32LittleEndian(file.AsSpan(page * PageSize + 4, 4)) == tdef; -} + private static bool IsOurIndexPage(byte[] file, int page, int tdef, JetFormatBase format) => + (page + 1) * format.PageSize <= file.Length + && PageHeader.ReadType(file.AsSpan(page * format.PageSize)) is PageType.IntermediateIndexPage or PageType.LeafIndexPage + && IndexTree.ReadOwner(file.AsSpan(page * format.PageSize, format.PageSize), format) == tdef; +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/IndexSplitPackingAccessTests.cs b/test/LibRed.Engine.AccessTests/IndexSplitPackingAccessTests.cs index cd2e63144..459190712 100644 --- a/test/LibRed.Engine.AccessTests/IndexSplitPackingAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/IndexSplitPackingAccessTests.cs @@ -1,6 +1,7 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.IO; +using LibRed.Pages; +using LibRed.Storage; using Xunit; namespace LibRed.Engine.Tests; @@ -152,12 +153,12 @@ private static (int Count, string Free, int Used, string Prefix) Leaves(string w for (int page = 1; page < channel.PageCount; page++) { byte[] bytes = channel.ReadPage(page).Span.ToArray(); - if (bytes[0] != 0x04) continue; - if (BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(4, 4)) != definitionPage) continue; + if (PageHeader.ReadType(bytes) != PageType.LeafIndexPage) continue; + if (IndexTree.ReadOwner(bytes, channel.Format) != definitionPage) continue; - int pageFree = BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(2, 2)); + int pageFree = IndexTree.ReadFreeSpace(bytes, channel.Format); free.Add(pageFree); - prefix.Add(BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(0x18, 2))); + prefix.Add(IndexTree.ReadCompressedByteCount(bytes, channel.Format)); used += channel.Format.PageSize - pageFree; } // Both lists stay in PAGE order. Sorting one and not the other made the two columns disagree @@ -167,4 +168,4 @@ private static (int Count, string Free, int Used, string Prefix) Leaves(string w } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/IndexSplitPointProbeTest.cs b/test/LibRed.Engine.AccessTests/IndexSplitPointProbeTest.cs index a3555c832..cf19a9fd0 100644 --- a/test/LibRed.Engine.AccessTests/IndexSplitPointProbeTest.cs +++ b/test/LibRed.Engine.AccessTests/IndexSplitPointProbeTest.cs @@ -1,8 +1,10 @@ -using System.Buffers.Binary; using System.Data.OleDb; using System.Globalization; using LibRed; using LibRed.Catalog; +using LibRed.Formats; +using LibRed.Pages; +using LibRed.Storage; using Xunit; namespace LibRed.Engine.Tests; @@ -24,7 +26,6 @@ namespace LibRed.Engine.Tests; [Collection(AceCollection.Name)] public class IndexSplitPointProbeTest(ITestOutputHelper output) { - private const int PageSize = 4096; private const int EvenKeys = 900; // enough for several full leaves at ~9 bytes an entry [Theory] @@ -62,16 +63,22 @@ public void Probe_where_ace_cuts_a_full_leaf(int oddKey) byte[] o = File.ReadAllBytes(seeded), a = File.ReadAllBytes(aceCopy), l = File.ReadAllBytes(libredCopy); int tdef; - using (var db = JetDatabase.Open(seeded)) tdef = db.Catalog.FindTable("SplitProbe")!.DefinitionPage; + JetFormatBase format; + using (var db = JetDatabase.Open(seeded)) + { + tdef = db.Catalog.FindTable("SplitProbe")!.DefinitionPage; + format = db.Format; + } + int pageSize = format.PageSize; output.WriteLine($"PROBE key {oddKey}: sizes orig={o.Length} ace={a.Length} libred={l.Length}"); - foreach (int page in Union(Changed(o, a), Changed(o, l)).Order()) + foreach (int page in Union(Changed(o, a, pageSize), Changed(o, l, pageSize)).Order()) { - if (!Owned(o, page, tdef) && !Owned(a, page, tdef) && !Owned(l, page, tdef)) continue; - output.WriteLine($"PROBE page {page} (was 0x{Type(o, page):X2}):" - + $" free orig={Free(o, page)} ace={Free(a, page)} libred={Free(l, page)}" - + $" | prefix orig={Prefix(o, page)} ace={Prefix(a, page)} libred={Prefix(l, page)}" - + $" | {(Same(a, l, page) ? "IDENTICAL" : $"differs first @0x{FirstDiff(a, l, page):X3}")}"); + if (!Owned(o, page, tdef, format) && !Owned(a, page, tdef, format) && !Owned(l, page, tdef, format)) continue; + output.WriteLine($"PROBE page {page} (was 0x{Type(o, page, pageSize):X2}):" + + $" free orig={Free(o, page, format)} ace={Free(a, page, format)} libred={Free(l, page, format)}" + + $" | prefix orig={Prefix(o, page, format)} ace={Prefix(a, page, format)} libred={Prefix(l, page, format)}" + + $" | {(Same(a, l, page, pageSize) ? "IDENTICAL" : $"differs first @0x{FirstDiff(a, l, page, pageSize):X3}")}"); } } finally @@ -88,40 +95,41 @@ private static void Exec(OleDbConnection conn, string sql) cmd.ExecuteNonQuery(); } - private static byte Type(byte[] f, int p) => (p + 1) * PageSize <= f.Length ? f[p * PageSize] : (byte)0xFF; + private static byte Type(byte[] f, int p, int pageSize) => (p + 1) * pageSize <= f.Length ? f[p * pageSize] : (byte)0xFF; - private static bool Owned(byte[] f, int p, int tdef) => - (p + 1) * PageSize <= f.Length && f[p * PageSize] is 0x03 or 0x04 - && BinaryPrimitives.ReadInt32LittleEndian(f.AsSpan(p * PageSize + 4, 4)) == tdef; + private static bool Owned(byte[] f, int p, int tdef, JetFormatBase format) => + (p + 1) * format.PageSize <= f.Length + && PageHeader.ReadType(f.AsSpan(p * format.PageSize)) is PageType.IntermediateIndexPage or PageType.LeafIndexPage + && IndexTree.ReadOwner(f.AsSpan(p * format.PageSize, format.PageSize), format) == tdef; - private static int Free(byte[] f, int p) => (p + 1) * PageSize <= f.Length - ? BinaryPrimitives.ReadUInt16LittleEndian(f.AsSpan(p * PageSize + 2, 2)) : -1; + private static int Free(byte[] f, int p, JetFormatBase format) => (p + 1) * format.PageSize <= f.Length + ? IndexTree.ReadFreeSpace(f.AsSpan(p * format.PageSize, format.PageSize), format) : -1; - private static int Prefix(byte[] f, int p) => (p + 1) * PageSize <= f.Length - ? BinaryPrimitives.ReadUInt16LittleEndian(f.AsSpan(p * PageSize + 0x18, 2)) : -1; + private static int Prefix(byte[] f, int p, JetFormatBase format) => (p + 1) * format.PageSize <= f.Length + ? IndexTree.ReadCompressedByteCount(f.AsSpan(p * format.PageSize, format.PageSize), format) : -1; - private static bool Same(byte[] x, byte[] y, int p) => - (p + 1) * PageSize <= Math.Min(x.Length, y.Length) - && x.AsSpan(p * PageSize, PageSize).SequenceEqual(y.AsSpan(p * PageSize, PageSize)); + private static bool Same(byte[] x, byte[] y, int p, int pageSize) => + (p + 1) * pageSize <= Math.Min(x.Length, y.Length) + && x.AsSpan(p * pageSize, pageSize).SequenceEqual(y.AsSpan(p * pageSize, pageSize)); private static IEnumerable Union(HashSet x, HashSet y) => x.Union(y); - private static int FirstDiff(byte[] x, byte[] y, int p) + private static int FirstDiff(byte[] x, byte[] y, int p, int pageSize) { - int limit = Math.Min(Math.Min(x.Length, y.Length) - p * PageSize, PageSize); + int limit = Math.Min(Math.Min(x.Length, y.Length) - p * pageSize, pageSize); for (int i = 0; i < limit; i++) - if (x[p * PageSize + i] != y[p * PageSize + i]) return i; + if (x[p * pageSize + i] != y[p * pageSize + i]) return i; return -1; } - private static HashSet Changed(byte[] left, byte[] right) + private static HashSet Changed(byte[] left, byte[] right, int pageSize) { var changed = new HashSet(); - int pages = Math.Min(left.Length, right.Length) / PageSize; + int pages = Math.Min(left.Length, right.Length) / pageSize; for (int p = 0; p < pages; p++) - if (!left.AsSpan(p * PageSize, PageSize).SequenceEqual(right.AsSpan(p * PageSize, PageSize))) + if (!left.AsSpan(p * pageSize, pageSize).SequenceEqual(right.AsSpan(p * pageSize, pageSize))) changed.Add(p); - for (int p = pages; p < Math.Max(left.Length, right.Length) / PageSize; p++) changed.Add(p); + for (int p = pages; p < Math.Max(left.Length, right.Length) / pageSize; p++) changed.Add(p); return changed; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/IndexUniqueEntryCountAccessTests.cs b/test/LibRed.Engine.AccessTests/IndexUniqueEntryCountAccessTests.cs index 66b5fca50..85d57d2a4 100644 --- a/test/LibRed.Engine.AccessTests/IndexUniqueEntryCountAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/IndexUniqueEntryCountAccessTests.cs @@ -1,5 +1,8 @@ using System.Data.OleDb; +using LibRed.Catalog; using LibRed; +using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Engine.Tests; @@ -142,6 +145,50 @@ public void An_index_built_over_existing_rows_is_counted_from_them(string sql) } } + // An UPDATE that changes a row's key drops the index's total by one and then holds the unique count to no more + // than the total — whether the old key had another holder does not matter, and the new key advances nothing. + // Visible only on an index built over rows: one built empty has a zero total, which is left alone. + [Theory] + [InlineData("UPDATE S SET A = 9 WHERE Id = 6")] // the last row holding A = 8 + [InlineData("UPDATE S SET A = 9 WHERE Id = 3")] // A = 6 is held by row 4 too + [InlineData("UPDATE S SET T = 'D' WHERE Id = 6")] // a collation-equal key: 'd' and 'D' are one + [InlineData("UPDATE S SET N = NULL WHERE Id = 3")] // leaves the IGNORE NULL index + [InlineData("UPDATE S SET N = 5 WHERE Id = 6")] // enters it from Null + [InlineData("UPDATE S SET A = A + 10")] + public void An_update_counts_the_old_key_out_as_a_delete_does(string sql) + { + string northwind = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); + string start = TemporaryDatabase.CopyPath(northwind, "uniqcount-upd-start-"); + string ace = "", libred = ""; + try + { + foreach (string statement in (string[]) + [ + "CREATE TABLE S (Id LONG CONSTRAINT pkS PRIMARY KEY, A LONG, T TEXT(10), N LONG)", + "INSERT INTO S VALUES (1, 5, 'a', NULL)", "INSERT INTO S VALUES (2, 5, 'A', NULL)", + "INSERT INTO S VALUES (3, 6, 'b', 1)", "INSERT INTO S VALUES (4, 6, 'b', 1)", + "INSERT INTO S VALUES (5, 7, 'c', 2)", "INSERT INTO S VALUES (6, 8, 'd', NULL)", + "CREATE INDEX ixA ON S (A)", "CREATE INDEX ixT ON S (T)", + "CREATE INDEX ixNi ON S (N) WITH IGNORE NULL", "CREATE INDEX ixAN ON S (A, N)", + ]) + Ace(start, statement); + + ace = TemporaryDatabase.CopyPath(start, "uniqcount-upd-ace-"); + libred = TemporaryDatabase.CopyPath(start, "uniqcount-upd-lib-"); + Ace(ace, sql); + LibRed(libred, sql); + string before = Statistics(start), aceStats = Statistics(ace), libredStats = Statistics(libred); + output.WriteLine($"{sql}\n before {before}\n ACE {aceStats}\n LibRed {libredStats}"); + Assert.Equal(aceStats, libredStats); + } + finally + { + TemporaryDatabase.Delete(start); + if (ace.Length > 0) TemporaryDatabase.Delete(ace); + if (libred.Length > 0) TemporaryDatabase.Delete(libred); + } + } + // A retype to or from Memo/OLE, which LibRed does by rebuilding the whole table, leaves the counts as ACE's ALTER // does: only an index over the changed column is rebuilt (a primary key included), every other index here keeps // its cumulative counts, and so does the foreign-key index of a table referencing this one. @@ -194,15 +241,17 @@ private static string Statistics(string path, params string[] tables) { if (tables.Length == 0) tables = ["S"]; using var db = JetDatabase.Open(path); - byte[] file = File.ReadAllBytes(path); return string.Join(" ", tables.SelectMany(name => { var table = db.Catalog.FindTable(name)!; - int block = table.DefinitionPage * 4096 + 0x3F; + PageBuffer tdef = TableDefinition.ReadChain(db.Channel, table.DefinitionPage).Buffer; return table.Indexes.GroupBy(i => i.RealIndexOrdinal).Select(g => g.First()) .OrderBy(i => i.Name, StringComparer.Ordinal) - .Select(i => $"{name}.{i.Name}={BitConverter.ToInt32(file, block + i.RealIndexOrdinal * 12)}" - + $"/{BitConverter.ToInt32(file, block + i.RealIndexOrdinal * 12 + 4)}"); + .Select(i => + { + (int total, int unique) = TableDefinition.ReadIndexCounts(tdef.Span, db.Format, i.RealIndexOrdinal); + return $"{name}.{i.Name}={total}/{unique}"; + }); })); } @@ -226,4 +275,4 @@ private static string Counts(string path, string table = "S") return string.Join(" ", db.Catalog.FindTable(table)!.Indexes.OrderBy(i => i.Name, StringComparer.Ordinal) .Select(i => $"{i.Name}={i.UniqueEntryCount}")); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/LikeCandidates.cs b/test/LibRed.Engine.AccessTests/LikeCandidates.cs new file mode 100644 index 000000000..c36bbaa58 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/LikeCandidates.cs @@ -0,0 +1,115 @@ +using System.Data.OleDb; +using LibRed; +using LibRed.Catalog; +using LibRed.Storage; + +namespace LibRed.Engine.Tests; + +/// +/// The pairs of text LIKE might count as the same character, and ACE's answer for them. The runtime only +/// PROPOSES pairs — each BMP character against its upper and lower case, the Turkish i forms, and each character +/// against the letters it decomposes to — and ACE decides every one. ACE's answer is the same in every collation +/// (measured across all 405 LibRed can create), so one database asks it. +/// +internal static class LikeCandidates +{ + public static (string Value, string Pattern)[] Pairs { get; } = [.. Generate()]; + + private static IEnumerable<(string Value, string Pattern)> Generate() + { + var seen = new HashSet<(string, string)>(); + IEnumerable<(string, string)> Both(string a, string b) => a == b ? [] : [(a, b), (b, a)]; + + for (int c = 0x20; c <= 0xFFFF; c++) + { + if (c is >= 0xD800 and <= 0xDFFF || c is '%' or '_' or '[' or ']') continue; + string s = ((char)c).ToString(); + foreach (string other in (string[])[s.ToUpperInvariant(), s.ToLowerInvariant(), + char.ToUpperInvariant((char)c).ToString(), char.ToLowerInvariant((char)c).ToString()]) + foreach (var p in Both(s, other)) + if (seen.Add(p)) yield return p; + + // Noncharacters have no decomposition to propose, and Normalize refuses them. + string decomposed; + try { decomposed = s.Normalize(System.Text.NormalizationForm.FormKD); } + catch (ArgumentException) { continue; } + if (decomposed.Length >= 2 && decomposed.All(char.IsLetter) && !decomposed.Any(ch => ch is '%' or '_' or '[')) + foreach (string spelling in (string[])[decomposed, decomposed.ToUpperInvariant(), decomposed.ToLowerInvariant()]) + foreach (var p in Both(s, spelling)) + if (seen.Add(p)) yield return p; + } + + (string, string)[] extra = + [ + ("i", "İ"), ("ı", "I"), ("i", "I"), ("ı", "i"), ("İ", "I"), ("ı", "İ"), + ("ß", "ss"), ("ß", "SS"), ("ẞ", "ss"), ("ẞ", "SS"), ("ẞ", "ß"), + ("æ", "ae"), ("Æ", "AE"), ("æ", "AE"), ("œ", "oe"), ("Œ", "OE"), ("œ", "OE"), + ("þ", "th"), ("Þ", "TH"), ("þ", "TH"), ("ð", "d"), ("ð", "dh"), ("Ð", "D"), ("ø", "o"), ("ø", "oe"), + ("ł", "l"), ("đ", "d"), ("ħ", "h"), ("ŋ", "ng"), ("ĸ", "q"), ("ſ", "s"), ("ſ", "S"), ("ʼn", "'n"), + ]; + foreach ((string a, string b) in extra) + foreach (var p in Both(a, b)) + if (seen.Add(p)) yield return p; + } + + /// Which pairs ACE says match, by index: all loaded into one table and asked in one query. + public static HashSet AceMatches(Collation collation) + { + string path = TemporaryDatabase.CreatePath("like-pairs-", ".accdb"); + try + { + JetDatabase.Create(path, collation: collation); + using OleDbConnection ace = AceTestDatabase.Open(path); + using (OleDbCommand create = ace.CreateCommand()) + { + create.CommandText = "CREATE TABLE Pairs (Id LONG, V TEXT(10), P TEXT(10))"; + create.ExecuteNonQuery(); + } + using (OleDbTransaction transaction = ace.BeginTransaction()) + { + using OleDbCommand insert = ace.CreateCommand(); + insert.Transaction = transaction; + insert.CommandText = "INSERT INTO Pairs (Id, V, P) VALUES (?, ?, ?)"; + OleDbParameter id = insert.Parameters.Add("i", OleDbType.Integer); + OleDbParameter v = insert.Parameters.Add("v", OleDbType.VarWChar, 10); + OleDbParameter p = insert.Parameters.Add("p", OleDbType.VarWChar, 10); + for (int i = 0; i < Pairs.Length; i++) + { + id.Value = i; + v.Value = Pairs[i].Value; + p.Value = Pairs[i].Pattern; + insert.ExecuteNonQuery(); + } + transaction.Commit(); + } + + using OleDbCommand query = ace.CreateCommand(); + query.CommandText = "SELECT Id FROM Pairs WHERE V LIKE P"; + var matched = new HashSet(); + using OleDbDataReader reader = query.ExecuteReader(); + while (reader.Read()) matched.Add(reader.GetInt32(0)); + return matched; + } + finally { TemporaryDatabase.Delete(path); } + } + + /// Which pairs LibRed says match, asked the same way. + public static HashSet LibRedMatches(Collation collation) + { + string path = TemporaryDatabase.CreatePath("like-pairs-libred-", ".accdb"); + try + { + JetDatabase.Create(path, collation: collation); + using var db = JetDatabase.Open(path, readOnly: false); + var engine = new QueryEngine(db); + engine.ExecuteNonQuery("CREATE TABLE Pairs (Id LONG, V TEXT(10), P TEXT(10))"); + engine.ExecuteNonQuery("BEGIN TRANSACTION"); + for (int i = 0; i < Pairs.Length; i++) + engine.ExecuteNonQuery("INSERT INTO Pairs (Id, V, P) VALUES (@i, @v, @p)", + new Dictionary { ["@i"] = i, ["@v"] = Pairs[i].Value, ["@p"] = Pairs[i].Pattern }); + engine.ExecuteNonQuery("COMMIT"); + return [.. engine.ExecuteQuery("SELECT Id FROM Pairs WHERE V LIKE P").Rows.Select(r => Convert.ToInt32(r[0]))]; + } + finally { TemporaryDatabase.Delete(path); } + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/LikeCollationProbeTests.cs b/test/LibRed.Engine.AccessTests/LikeCollationProbeTests.cs new file mode 100644 index 000000000..35e76770b --- /dev/null +++ b/test/LibRed.Engine.AccessTests/LikeCollationProbeTests.cs @@ -0,0 +1,163 @@ +using System.Data.OleDb; +using LibRed; +using LibRed.Catalog; +using LibRed.Engine; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +// PROBE: what ACE's LIKE treats as the same character, and whether the database's collation changes any of it. +// +// Report asks a handful of telling pairs in General v0, General v1 and Croatian v1 alongside LibRed. The screen +// asks every candidate pair (LikeCandidates) in every collation LibRed can create — 405 of them. What it found: +// every collation matches exactly the same pairs, Turkish, Azerbaijani and Lithuanian included, so LIKE is +// independent of the collation and one table describes it (LikeFoldTableGeneratorTest). +[Collection(AceCollection.Name)] +public class LikeCollationProbeTests(ITestOutputHelper output) +{ + private static readonly (string Value, string Pattern)[] Cases = + [ + // case beyond ASCII + ("é", "É"), ("σ", "Σ"), ("ς", "Σ"), ("ς", "σ"), ("ı", "I"), ("i", "İ"), ("dž", "DŽ"), ("Dž", "dž"), ("ÿ", "Ÿ"), + ("µ", "Μ"), ("ж", "Ж"), + // width and kana + ("A", "A"), ("a", "A"), ("カ", "カ"), ("カ", "か"), + // accents + ("é", "e"), ("e", "é"), ("é", "_"), ("é", "[e]"), + // expansions + ("ß", "ss"), ("ss", "ß"), ("ß", "SS"), ("ẞ", "ss"), ("æ", "ae"), ("œ", "oe"), ("Œ", "OE"), ("ij", "ij"), + ("þ", "th"), ("Þ", "TH"), ("fi", "fi"), ("dž", "dž"), ("lj", "lj"), + // ignorables + ("co-op", "coop"), ("coop", "co-op"), ("O'Brien", "OBrien"), ("a­b", "ab"), ("a\u0001b", "ab"), + // trailing spaces + ("a ", "a"), ("a", "a "), ("a ", "a_"), + // superscripts and fractions + ("²", "2"), ("½", "1/2"), + // ranges + ("é", "[a-z]"), ("É", "[A-Z]"), ("ß", "[a-z]"), ("ñ", "[m-o]"), ("Z", "[a-z]"), ("_", "[A-z]"), ("^", "[A-z]"), + ("1", "[a-z]"), ("²", "[0-9]"), ("č", "[c-d]"), ("č", "[a-c]"), ("ž", "[a-z]"), ("ω", "[α-ψ]"), + // a tailored letter + ("lj", "_"), ("lj", "__"), ("ča", "[c]a"), ("dž", "_"), + ]; + + [Fact] + public void Report() + { + var v0 = Ace(Collation.GeneralLegacy); + var v1 = Ace(Collation.General); + var hr = Ace(new Collation(CollatingOrder.Croatian, Collation.GeneralVersion)); + var libred = LibRed(Collation.GeneralLegacy); + + output.WriteLine($"{"value",-10} {"pattern",-10} v0 v1 hr | LibRed(v0)"); + for (int i = 0; i < Cases.Length; i++) + { + string flag = v0[i] != libred[i] ? " <-- differs" : ""; + output.WriteLine($"{Show(Cases[i].Value),-10} {Show(Cases[i].Pattern),-10} {v0[i],-2} {v1[i],-2} {hr[i],-2} | {libred[i]}{flag}"); + } + } + + /// Every candidate pair asked of ACE in every collation LibRed can create, one query per database. + /// Opt-in via LIBRED_SCREEN_LIKE=1: it creates 417 databases and takes the better part of an hour. + [Fact] + public void Screen_every_collation() + { + Assert.SkipUnless(Environment.GetEnvironmentVariable("LIBRED_SCREEN_LIKE") == "1", + "set LIBRED_SCREEN_LIKE=1 — this asks ACE 26,000 pairs in each of 417 collations"); + + var pairs = LikeCandidates.Pairs; + List collations = [.. Enum.GetValues().Distinct() + .Where(o => o != CollatingOrder.Undefined) + .SelectMany(o => (Collation[])[new(o, 0), new(o, Collation.GeneralVersion), new(o, 0, 1), + // the CJK orders reach sort id 4, at both versions + new(o, 0, 2), new(o, Collation.GeneralVersion, 2), new(o, 0, 3), new(o, Collation.GeneralVersion, 3), + new(o, Collation.GeneralVersion, 4)]) + .Where(c => c.IsIndexKeyEncodable) + .Distinct()]; + + string directory = Path.Combine(AppContext.BaseDirectory, "like-survey"); + Directory.CreateDirectory(directory); + string reportPath = Path.Combine(directory, "like-by-collation.txt"); + File.WriteAllText(reportPath, ""); + + HashSet? baseline = null; + foreach (Collation collation in collations) + { + HashSet matched; + try { matched = LikeCandidates.AceMatches(collation); } + catch (Exception e) when (e is OleDbException or InvalidOperationException or NotSupportedException) + { + File.AppendAllText(reportPath, $"{collation.Order} v{collation.Version} sort {collation.SortId}: not created ({e.Message}){Environment.NewLine}"); + continue; + } + + baseline ??= matched; + var extra = matched.Except(baseline).ToList(); + var missing = baseline.Except(matched).ToList(); + string line = $"{collation.Order,-28} v{collation.Version} sort {collation.SortId}: {matched.Count} match" + + (extra.Count + missing.Count == 0 ? "" : $" DIFFERS +{extra.Count} -{missing.Count}: " + + string.Join(" ", extra.Take(12).Select(i => "+" + Pair(pairs[i])).Concat(missing.Take(12).Select(i => "-" + Pair(pairs[i]))))); + // Written as it goes, so a long run shows its progress and a failure keeps what it had. + File.AppendAllText(reportPath, line + Environment.NewLine); + output.WriteLine(line); + } + } + + private static string Pair((string Value, string Pattern) p) => $"{Show(p.Value)}~{Show(p.Pattern)}"; + + private static string Show(string s) => + string.Concat(s.Select(c => c < 0x20 || c == 0xAD ? $"\\u{(int)c:X4}" : c.ToString())); + + private static string Sql((string Value, string Pattern) c) => + $"SELECT TOP 1 IIF('{c.Value.Replace("'", "''", StringComparison.Ordinal)}' LIKE " + + $"'{c.Pattern.Replace("'", "''", StringComparison.Ordinal)}', 'T', 'F') FROM One"; + + private static string[] Ace(Collation collation) + { + string path = TemporaryDatabase.CreatePath("like-probe-", ".accdb"); + try + { + JetDatabase.Create(path, collation: collation); + using OleDbConnection ace = AceTestDatabase.Open(path); + Exec(ace, "CREATE TABLE One (Id LONG)"); + Exec(ace, "INSERT INTO One (Id) VALUES (1)"); + return [.. Cases.Select(c => + { + try + { + using OleDbCommand command = ace.CreateCommand(); + command.CommandText = Sql(c); + return command.ExecuteScalar()?.ToString() ?? "N"; + } + catch (OleDbException) { return "E"; } + })]; + } + finally { TemporaryDatabase.Delete(path); } + } + + private static string[] LibRed(Collation collation) + { + string path = TemporaryDatabase.CreatePath("like-probe-libred-", ".accdb"); + try + { + JetDatabase.Create(path, collation: collation); + using var db = JetDatabase.Open(path, readOnly: false); + var engine = new QueryEngine(db); + engine.ExecuteNonQuery("CREATE TABLE One (Id LONG)"); + engine.ExecuteNonQuery("INSERT INTO One (Id) VALUES (1)"); + return [.. Cases.Select(c => + { + try { return engine.ExecuteQuery(Sql(c)).Rows.Single()[0]?.ToString() ?? "N"; } + catch (ArgumentException) { return "E"; } + })]; + } + finally { TemporaryDatabase.Delete(path); } + } + + private static void Exec(OleDbConnection connection, string sql) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/LikeFoldAccessTests.cs b/test/LibRed.Engine.AccessTests/LikeFoldAccessTests.cs new file mode 100644 index 000000000..21b5f5ddc --- /dev/null +++ b/test/LibRed.Engine.AccessTests/LikeFoldAccessTests.cs @@ -0,0 +1,28 @@ +using LibRed.Catalog; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// LibRed's LIKE matches exactly the pairs ACE's does, over every candidate — each BMP character against its upper +/// and lower case, the Turkish i forms, and each character against the letters it decomposes to (LikeCandidates). +/// One database is enough: ACE's answer is the same in every collation. +/// +[Collection(AceCollection.Name)] +public class LikeFoldAccessTests +{ + [Fact] + public void Every_candidate_pair_matches_as_ace_matches_it() + { + var pairs = LikeCandidates.Pairs; + HashSet ace = LikeCandidates.AceMatches(Collation.GeneralLegacy); + HashSet libred = LikeCandidates.LibRedMatches(Collation.GeneralLegacy); + + string Show(int i) => $"'{pairs[i].Value}' LIKE '{pairs[i].Pattern}'"; + var missed = ace.Except(libred).Take(20).Select(Show).ToList(); + var extra = libred.Except(ace).Take(20).Select(Show).ToList(); + Assert.True(missed.Count == 0 && extra.Count == 0, + $"ACE only: {string.Join(", ", missed)}; LibRed only: {string.Join(", ", extra)}"); + Assert.True(ace.Count > 1_900, $"ACE matched only {ace.Count} — the instrument is not measuring"); + } +} diff --git a/test/LibRed.Engine.AccessTests/LikeFoldTableGeneratorTest.cs b/test/LibRed.Engine.AccessTests/LikeFoldTableGeneratorTest.cs new file mode 100644 index 000000000..c455cbada --- /dev/null +++ b/test/LibRed.Engine.AccessTests/LikeFoldTableGeneratorTest.cs @@ -0,0 +1,150 @@ +using LibRed.Catalog; +using Xunit; + +namespace LibRed.Engine.Tests; + +// GENERATOR: builds LibRed.Engine's embedded LIKE folding table by asking ACE. +// +// ACE's LIKE ignores case through a table of its own — narrower than any runtime's: it leaves the micro sign, +// long s, final sigma, the Greek symbol variants, the titlecase digraphs and every letter Unicode added later +// unfolded — and spells four letters out: ß as ss, æ as ae, œ as oe, þ as th. The runtime's casing differs from +// that and between platforms, so the table is ACE's, measured: every candidate pair (LikeCandidates) asked in one +// database, since the answer is the same in every collation. +// +// Opt-in via LIBRED_GENERATE_LIKE=1: it asks ACE 26,000 pairs and rewrites a checked-in resource. +[Collection(AceCollection.Name)] +public class LikeFoldTableGeneratorTest(ITestOutputHelper output) +{ + private const string ResourcePath = "src/LibRed/LibRed.Engine/Resources/LikeFold.bin"; + + [Fact] + public void Generate_the_like_fold_resource() + { + Assert.SkipUnless(Environment.GetEnvironmentVariable("LIBRED_GENERATE_LIKE") == "1", + "set LIBRED_GENERATE_LIKE=1 — this asks ACE 26,000 pairs and rewrites a resource"); + + var pairs = LikeCandidates.Pairs; + HashSet matched = LikeCandidates.AceMatches(Collation.GeneralLegacy); + + // Case pairs: one character each side. Each must be matched both ways and pair with nothing else, or a + // simple map cannot describe it. + var partner = new Dictionary(); + foreach (int i in matched.Where(i => pairs[i].Value.Length == 1 && pairs[i].Pattern.Length == 1)) + { + (char a, char b) = (pairs[i].Value[0], pairs[i].Pattern[0]); + Assert.True(matched.Contains(Array.IndexOf(pairs, (pairs[i].Pattern, pairs[i].Value))), $"{a}~{b} one way only"); + Assert.True(!partner.TryGetValue(a, out char existing) || existing == b, $"{a} folds with {existing} and {b}"); + partner[a] = b; + } + + // Each pair folds to one member — the one the other upper-cases to, else the capital. Decided here, at + // generation, so nothing at run time consults the runtime's casing. + var folded = new SortedDictionary(); + foreach ((char a, char b) in partner) + { + if (a > b) continue; + char to = char.ToUpperInvariant(a) == b ? b + : char.ToUpperInvariant(b) == a ? a + : char.IsUpper(b) ? b : a; + char from = to == a ? b : a; + folded[from] = to; + } + char Fold(char c) => folded.TryGetValue(c, out char to) ? to : c; + + // Expansions: one character against a longer spelling, recorded as that spelling folded. + var expansions = new SortedDictionary(); + foreach (int i in matched.Where(i => pairs[i].Value.Length != pairs[i].Pattern.Length)) + { + (string one, string many) = pairs[i].Value.Length == 1 ? (pairs[i].Value, pairs[i].Pattern) : (pairs[i].Pattern, pairs[i].Value); + Assert.Equal(1, one.Length); + string spelling = string.Concat(many.Select(Fold)); + Assert.True(!expansions.TryGetValue(one[0], out string? had) || had == spelling, $"{one} spelt {had} and {spelling}"); + expansions[one[0]] = spelling; + } + + // The layout of the sort-key tables: the counts, then each section deflated on its own. + // case pairs: (character, what it folds to), a UTF-16 unit each + // expansions: (character, letter count, the letters folded) + var cases = new MemoryStream(); + using (var writer = new BinaryWriter(cases, System.Text.Encoding.UTF8, leaveOpen: true)) + foreach ((char from, char to) in folded) { writer.Write((ushort)from); writer.Write((ushort)to); } + var spelt = new MemoryStream(); + using (var writer = new BinaryWriter(spelt, System.Text.Encoding.UTF8, leaveOpen: true)) + foreach ((char c, string letters) in expansions) + { + writer.Write((ushort)c); + writer.Write((byte)letters.Length); + foreach (char letter in letters) writer.Write((ushort)letter); + } + + var blob = new MemoryStream(); + using (var writer = new BinaryWriter(blob, System.Text.Encoding.UTF8, leaveOpen: true)) + { + writer.Write(folded.Count); + writer.Write(expansions.Count); + WriteDeflated(writer, cases.ToArray()); + WriteDeflated(writer, spelt.ToArray()); + } + + string path = Path.Combine(RepositoryRoot(), ResourcePath); + Directory.CreateDirectory(Path.GetDirectoryName(path)!); + File.WriteAllBytes(path, blob.ToArray()); + output.WriteLine($"wrote {path} ({blob.Length:N0} bytes): {folded.Count} case pairs, {expansions.Count} expansions, " + + $"from {matched.Count} matches"); + + // Read it straight back, as the sort-key generators do: a structurally valid but empty file would + // otherwise ship silently. + (var reloadedFold, var reloadedExpansions) = Parse(blob.ToArray()); + Assert.Equal(folded, reloadedFold); + Assert.Equal(expansions, reloadedExpansions); + + // What one run found; a different count means ACE changed, or the candidates did. + Assert.Equal(973, folded.Count); + Assert.Equal(7, expansions.Count); + } + + private static void WriteDeflated(BinaryWriter writer, byte[] data) + { + var compressed = new MemoryStream(); + using (var deflate = new System.IO.Compression.ZLibStream(compressed, System.IO.Compression.CompressionLevel.SmallestSize, leaveOpen: true)) + deflate.Write(data); + writer.Write((int)compressed.Length); + writer.Write(compressed.ToArray()); + } + + private static (SortedDictionary Fold, SortedDictionary Expansions) Parse(byte[] blob) + { + var reader = new BinaryReader(new MemoryStream(blob)); + int caseCount = reader.ReadInt32(), expansionCount = reader.ReadInt32(); + var cases = new BinaryReader(new MemoryStream(Inflate(reader))); + var spelt = new BinaryReader(new MemoryStream(Inflate(reader))); + + var fold = new SortedDictionary(); + for (int i = 0; i < caseCount; i++) fold[(char)cases.ReadUInt16()] = (char)cases.ReadUInt16(); + var expansions = new SortedDictionary(); + for (int i = 0; i < expansionCount; i++) + { + char c = (char)spelt.ReadUInt16(); + int length = spelt.ReadByte(); + expansions[c] = new string([.. Enumerable.Range(0, length).Select(_ => (char)spelt.ReadUInt16())]); + } + return (fold, expansions); + + static byte[] Inflate(BinaryReader reader) + { + byte[] compressed = reader.ReadBytes(reader.ReadInt32()); + var output = new MemoryStream(); + using (var inflate = new System.IO.Compression.ZLibStream(new MemoryStream(compressed), System.IO.Compression.CompressionMode.Decompress)) + inflate.CopyTo(output); + return output.ToArray(); + } + } + + private static string RepositoryRoot() + { + var directory = new DirectoryInfo(AppContext.BaseDirectory); + while (directory is not null && !File.Exists(Path.Combine(directory.FullName, "EFCore.Jet.sln"))) + directory = directory.Parent; + return directory?.FullName ?? throw new InvalidOperationException("EFCore.Jet.sln not found above the test output."); + } +} diff --git a/test/LibRed.Engine.AccessTests/LongValueColumnLayoutAccessTests.cs b/test/LibRed.Engine.AccessTests/LongValueColumnLayoutAccessTests.cs index f05bf3847..7aa7f9ed2 100644 --- a/test/LibRed.Engine.AccessTests/LongValueColumnLayoutAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/LongValueColumnLayoutAccessTests.cs @@ -1,4 +1,3 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Catalog; using LibRed.Engine; @@ -88,8 +87,9 @@ private void Dump(string path, string stage) { using var channel = PageChannel.Open(path, readOnly: true); var table = new JetCatalog(channel).FindTable("LayoutProbe")!; - var header = channel.ReadPage(table.DefinitionPage); - output.WriteLine($"{stage}: highWater={BinaryPrimitives.ReadUInt16LittleEndian(header.Span.Slice(channel.Format.TdefMaxColumnsOffset, 2))}, varCount={BinaryPrimitives.ReadUInt16LittleEndian(header.Span.Slice(channel.Format.TdefVariableColumnsOffset, 2))}"); + var header = new TableDefinition(); + header.Read(channel, table.DefinitionPage); + output.WriteLine($"{stage}: highWater={header.ColumnIdHighWater}, varCount={header.VariableColumnCount}"); foreach (var c in table.Columns) output.WriteLine($"{c.Name}: id={c.ColumnId}, ordinal={c.Index}, var={c.VariableIndex}, fixedOffset={c.FixedOffset}, descriptor={Convert.ToHexString(c.RawDescriptor!)}"); foreach (int number in new UsageMap(channel, table).DataPages()) @@ -100,4 +100,4 @@ private void Dump(string path, string stage) if (!page.Rows[row].IsDeleted) output.WriteLine($"row {number}:{row}: {Convert.ToHexString(page.GetRow(row))}"); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/LongValueFreeMapAccessTests.cs b/test/LibRed.Engine.AccessTests/LongValueFreeMapAccessTests.cs new file mode 100644 index 000000000..e519b5e0d --- /dev/null +++ b/test/LibRed.Engine.AccessTests/LongValueFreeMapAccessTests.cs @@ -0,0 +1,72 @@ +using System.Data.OleDb; +using LibRed.Catalog; +using LibRed.Pages; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// A long-value page leaves its column's free-pages map once it cannot hold a 256-byte value and its slot — 257 +/// bytes free or fewer — the same through ACE and LibRed (docs/format/long-values.md). Two values fill the page to +/// just either side of the line: a 2000-byte one, then one that leaves remaining bytes. +/// +[Collection(AceCollection.Name)] +public class LongValueFreeMapAccessTests(ITestOutputHelper output) : TempDatabaseTest +{ + [Theory] + [InlineData(256)] + [InlineData(257)] + [InlineData(258)] + [InlineData(259)] + public void A_page_leaves_the_free_map_where_ace_takes_it_out(int remaining) + { + string[] statements = + [ + "CREATE TABLE L (Id LONG, M LONGBINARY)", + $"INSERT INTO L VALUES (1, 0x{new string('A', 4000)})", + $"INSERT INTO L VALUES (2, 0x{new string('B', 2 * (2078 - remaining))})", + ]; + string northwind = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); + string ace = TemporaryDatabase.CopyPath(northwind, "lvfree-ace-"), libred = TemporaryDatabase.CopyPath(northwind, "lvfree-lib-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + foreach (string sql in statements) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } + using (var db = JetDatabase.Open(libred, readOnly: false)) + { + var engine = new QueryEngine(db); + foreach (string sql in statements) engine.ExecuteNonQuery(sql); + } + + string expected = Maps(ace), actual = Maps(libred); + output.WriteLine($"ACE {expected}, LibRed {actual}"); + Assert.Equal(expected, actual); + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(libred); + } + } + + /// The column's long-value page, its free space, and whether its free-pages map still names it. + private static string Maps(string path) + { + using var db = JetDatabase.Open(path, readOnly: true); + Table table = db.OpenTable("L"); + var tdef = new TableDefinition(); + tdef.Read(table.Channel.ReadPage(table.Definition.DefinitionPage), table.Channel.Format); + int id = table.Definition.RequireColumn("M").ColumnId; + var maps = new UsageMap(table.Channel, table.Definition); + int page = maps.PagesInMap(tdef.LongValueOwnedMaps[id].Row, tdef.LongValueOwnedMaps[id].Page).Single(); + bool free = maps.PagesInMap(tdef.LongValueFreeMaps[id].Row, tdef.LongValueFreeMaps[id].Page).Contains(page); + int space = DataPage.ReadFreeSpace(table.Channel.ReadPage(page).Span, table.Channel.Format); + return $"{space} bytes free, {(free ? "in the free map" : "out of it")}"; + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/LongValueOrderProbeTest.cs b/test/LibRed.Engine.AccessTests/LongValueOrderProbeTest.cs new file mode 100644 index 000000000..11990f6d3 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/LongValueOrderProbeTest.cs @@ -0,0 +1,199 @@ +using System.Data.OleDb; +using System.Globalization; +using LibRed; +using LibRed.Catalog; +using LibRed.Formats; +using LibRed.IO; +using LibRed.Pages; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Probe: which pages ACE and LibRed give each long value of one inserted row, to see the order ACE writes a row's +/// long values in. Reports; asserts nothing. +/// +[Collection(AceCollection.Name)] +public class LongValueOrderProbeTest(ITestOutputHelper output) +{ + [Theory(Explicit = true)] + [InlineData("INSERT INTO Doc VALUES (1, String(5000, 'b'), String(300, 'c'), 'short and compressed', 0x{6000}, 0x{400})")] + [InlineData("INSERT INTO Doc (Id, Body, Blob) VALUES (1, String(5000, 'b'), 0x{6000})")] + [InlineData("INSERT INTO Doc (Id, Body, Blob) VALUES (1, String(300, 'b'), 0x{6000})")] + [InlineData("INSERT INTO Doc (Id, Body, Blob) VALUES (1, String(5000, 'b'), 0x{300})")] + [InlineData("INSERT INTO Doc (Id, Body, Packed) VALUES (1, String(5000, 'b'), String(300, 'c'))")] + [InlineData("INSERT INTO Doc (Id, Body, Blob) VALUES (1, String(5000, 'b'), 0x{12000})")] + [InlineData("INSERT INTO Doc (Id, Body, Blob) VALUES (1, String(200, 'b'), 0x{600})")] + [InlineData("INSERT INTO Doc (Id, Body, Blob) VALUES (1, String(1000, 'b'), 0x{600})")] + public void Long_value_pages_of_one_row(string insert) + { + string sql = System.Text.RegularExpressions.Regex.Replace(insert, @"0x\{(\d+)\}", + m => "0x" + Hex(int.Parse(m.Groups[1].Value, CultureInfo.InvariantCulture))); + string origin = TemporaryDatabase.CreatePath("lvorder-origin-"); + string ace = TemporaryDatabase.CreatePath("lvorder-ace-"); + string libred = TemporaryDatabase.CreatePath("lvorder-libred-"); + try + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = "CREATE TABLE Doc (Id LONG CONSTRAINT pkDoc PRIMARY KEY, Body MEMO, " + + "Packed MEMO WITH COMPRESSION, Brief TEXT(100) WITH COMPRESSION, Blob LONGBINARY, Big BIGBINARY(500))"; + command.ExecuteNonQuery(); + } + File.Copy(origin, ace, overwrite: true); + File.Copy(origin, libred, overwrite: true); + + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } + int pageSize; + using (var database = JetDatabase.Open(libred, readOnly: false)) + { + pageSize = database.Format.PageSize; + new QueryEngine(database).ExecuteNonQuery(sql); + } + + output.WriteLine($"{insert}\nbefore {new FileInfo(origin).Length / pageSize} pages\n" + + $"ACE: {Describe(ace)}\nLibRed: {Describe(libred)}"); + } + finally + { + foreach (string path in new[] { origin, ace, libred }) TemporaryDatabase.Delete(path); + } + } + + /// Compressed single-page memos: the bytes of every long-value page that differ between ACE and + /// LibRed, as ranges with ACE's bytes at the start of each — to see what ACE leaves in a page's free space. + [Theory(Explicit = true)] + [InlineData("INSERT INTO Doc (Id, Packed) VALUES (1, String(300, 'c'))")] + [InlineData("INSERT INTO Doc (Id, Packed) VALUES (1, String(300, 'c'))|INSERT INTO Doc (Id, Packed) VALUES (2, String(200, 'd'))")] + [InlineData("INSERT INTO Doc (Id, Packed) VALUES (1, String(300, ChrW(20013)))")] + [InlineData("INSERT INTO Doc (Id, Packed) VALUES (1, String(1800, 'e'))")] + [InlineData("INSERT INTO Doc (Id, Body) VALUES (1, String(300, 'c'))")] + [InlineData("INSERT INTO Doc (Id, Packed) VALUES (1, String(1800, 'e'))|INSERT INTO Doc (Id, Packed) VALUES (2, String(1000, 'f'))")] + [InlineData("INSERT INTO Doc (Id, Packed) VALUES (1, String(1800, 'e'))|INSERT INTO Doc (Id, Packed) VALUES (2, String(1200, 'f'))")] + public void Compressed_value_page_bytes(string statements) + { + string origin = TemporaryDatabase.CreatePath("lvbytes-origin-"); + string ace = TemporaryDatabase.CreatePath("lvbytes-ace-"); + string libred = TemporaryDatabase.CreatePath("lvbytes-libred-"); + try + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = "CREATE TABLE Doc (Id LONG CONSTRAINT pkDoc PRIMARY KEY, Body MEMO, " + + "Packed MEMO WITH COMPRESSION, Brief TEXT(100) WITH COMPRESSION, Blob LONGBINARY, Big BIGBINARY(500))"; + command.ExecuteNonQuery(); + } + File.Copy(origin, ace, overwrite: true); + File.Copy(origin, libred, overwrite: true); + + string[] sql = statements.Split('|'); + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + foreach (string statement in sql) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + JetFormatBase format; + using (var database = JetDatabase.Open(libred, readOnly: false)) + { + format = database.Format; + var queries = new QueryEngine(database); + foreach (string statement in sql) queries.ExecuteNonQuery(statement); + } + + int pageSize = format.PageSize; + byte[] a = File.ReadAllBytes(ace), l = File.ReadAllBytes(libred); + var report = new System.Text.StringBuilder($"{statements}\nACE {a.Length / pageSize} pages, LibRed {l.Length / pageSize} pages\n"); + for (int page = 1; page < Math.Min(a.Length, l.Length) / pageSize; page++) + { + int at = page * pageSize; + if (DataPage.ReadOwner(a.AsSpan(at, pageSize), format) != JetFormatBase.LongValuePageMarker) continue; + var ranges = new List(); + for (int i = 0; i < pageSize; i++) + { + if (a[at + i] == l[at + i]) continue; + int start = i; + while (i + 1 < pageSize && (a[at + i + 1] != l[at + i + 1] || (i + 2 < pageSize && a[at + i + 2] != l[at + i + 2]))) i++; + ranges.Add($"0x{start:X3}-0x{i:X3} ACE {Convert.ToHexString(a, at + start, Math.Min(8, i - start + 1))}"); + } + int rows = DataPage.ReadRowCount(a.AsSpan(at, pageSize), format); + string slots = string.Join(",", Enumerable.Range(0, rows) + .Select(r => $"0x{DataPage.ReadSlot(a.AsSpan(at, pageSize), format, r).Offset:X3}")); + report.AppendLine(CultureInfo.InvariantCulture, + $" LVAL page {page}: ACE rows at [{slots}]; {(ranges.Count == 0 ? "identical" : string.Join("; ", ranges))}"); + } + output.WriteLine(report.ToString()); + } + finally + { + foreach (string path in new[] { origin, ace, libred }) TemporaryDatabase.Delete(path); + } + } + + /// Each long-value column of Doc's one row: its form (inline, single page, chained) and the pages + /// its value occupies, first to last, with the file length. + private static string Describe(string path) + { + using var database = JetDatabase.Open(path, readOnly: true); + Table doc = database.OpenTable("Doc"); + (RowId id, _) = doc.Rows().WithIds().Single(); + var page = new DataPage(); + page.Read(doc.Channel.ReadPageShared(id.Page), doc.Channel.Format); + byte[] row = page.GetRow(id.Row).ToArray(); + var parts = new List(); + foreach ((int index, byte[] descriptor) in RowCodec.LongValueDescriptors(doc.Definition.Columns, doc.Channel.Format, row) + .OrderBy(d => d.Key)) + { + ColumnDef column = doc.Definition.Columns[index]; + var value = LongValueStore.Read(descriptor, doc.Channel.Format); + string where = value.Storage switch + { + LongValueStore.StorageKind.Inline => "inline", + LongValueStore.StorageKind.SinglePage => $"single {value.Page}:{value.Row}", + _ => $"chained {string.Join(">", Chain(doc, descriptor))}", + }; + parts.Add($"{column.Name}({value.Length}) {where}"); + } + return $"{new FileInfo(path).Length / database.Format.PageSize} pages, row on {id.Page}; " + string.Join("; ", parts); + } + + private static List Chain(Table table, byte[] descriptor) + { + var pages = new List(); + var (_, _, row, pageNumber, _) = LongValueStore.Read(descriptor, table.Channel.Format); + while (pageNumber != 0 && pages.Count < 64) + { + pages.Add($"{pageNumber}:{row}"); + var page = new DataPage(); + page.Read(table.Channel.ReadPageShared(pageNumber), table.Channel.Format); + ReadOnlySpan record = page.GetRow(row); + (row, pageNumber) = PageBuffer.ReadRecordPointer(record, 0); + } + return pages; + } + + private static string Hex(int bytes) => + Convert.ToHexString([.. Enumerable.Range(0, bytes).Select(i => (byte)((i * 7 + 1) % 251))]); + + private static object? Invoke(object target, string member, params object?[] args) => + target.GetType().InvokeMember(member, System.Reflection.BindingFlags.InvokeMethod, null, target, args); +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/LvPropByteParityAccessTests.cs b/test/LibRed.Engine.AccessTests/LvPropByteParityAccessTests.cs index 422c2658c..dd1a02265 100644 --- a/test/LibRed.Engine.AccessTests/LvPropByteParityAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/LvPropByteParityAccessTests.cs @@ -68,7 +68,7 @@ private static void LibRedRun(string path, string[] statements) catch (OleDbException) { return null; } using var database = JetDatabase.Open(path, readOnly: true); - TableDef objects = database.Catalog.FindTable("MSysObjects")!; + TableDefinition objects = database.Catalog.FindTable("MSysObjects")!; int nameCol = objects.Columns.Single(c => c.Name == "Name").Index; int lvCol = objects.Columns.Single(c => c.Name == "LvProp").Index; @@ -79,4 +79,4 @@ private static void LibRedRun(string path, string[] statements) } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/LvPropStorageAccessTests.cs b/test/LibRed.Engine.AccessTests/LvPropStorageAccessTests.cs new file mode 100644 index 000000000..5e34fe5ae --- /dev/null +++ b/test/LibRed.Engine.AccessTests/LvPropStorageAccessTests.cs @@ -0,0 +1,102 @@ +using System.Data.OleDb; +using LibRed.Catalog; +using LibRed.Pages; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Where a table's property blob (MSysObjects.LvProp) is stored: inline in the row up to 64 bytes and on an +/// LVAL page above that, the rule every long value follows — through ACE and through LibRed alike. Each property adds +/// its owner's name to the blob, so the length of a NOT NULL column's name walks the blob across the line. +/// +[Collection(AceCollection.Name)] +public class LvPropStorageAccessTests(ITestOutputHelper output) : TempDatabaseTest +{ + [Theory] + [InlineData("CREATE TABLE W (Id LONG, Cxxxxx LONG NOT NULL)")] // 61 bytes + [InlineData("CREATE TABLE W (Id LONG, Cxxxxxx LONG NOT NULL)")] // 63 bytes + [InlineData("CREATE TABLE W (Id LONG, Cxxxxxxx LONG NOT NULL)")] // 65 bytes + [InlineData("CREATE TABLE W (Id LONG, Cxxxxxxxx LONG NOT NULL)")] // 67 bytes + [InlineData("CREATE TABLE W (Id LONG, A LONG DEFAULT 7)")] // 60 bytes + [InlineData("CREATE TABLE W (Id LONG, A LONG DEFAULT 7, B LONG DEFAULT 8)")] // 84 bytes + public void A_property_blob_is_stored_where_ace_stores_it(string sql) + { + string ace = Stored(sql, AceRun), libred = Stored(sql, LibRedRun); + output.WriteLine($"ACE {ace}, LibRed {libred}"); + Assert.Equal(ace, libred); + } + + // What an inline blob is for: ACE reads it. A NOT NULL written inline by LibRed refuses a null in ACE, and a + // DEFAULT written inline fills an omitted column. + private static readonly string Northwind = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); + + [Fact] + public void Ace_honours_a_property_blob_libred_wrote_inline() + { + string path = TemporaryDatabase.CopyPath(Northwind, "lvpropinline-"); + try + { + LibRedRun(path, "CREATE TABLE W (Id LONG, A LONG NOT NULL)"); + LibRedRun(path, "CREATE TABLE V (Id LONG, A LONG DEFAULT 7)"); + Assert.StartsWith("inline", Stored(path, "W"), StringComparison.Ordinal); + Assert.StartsWith("inline", Stored(path, "V"), StringComparison.Ordinal); + + using OleDbConnection connection = AceTestDatabase.Open(path); + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = "INSERT INTO W (Id) VALUES (1)"; + Assert.Throws(() => command.ExecuteNonQuery()); + command.CommandText = "INSERT INTO V (Id) VALUES (1)"; + command.ExecuteNonQuery(); + command.CommandText = "SELECT A FROM V"; + Assert.Equal(7, command.ExecuteScalar()); + } + finally { TemporaryDatabase.Delete(path); } + } + + private static string Stored(string sql, Action create) + { + string path = TemporaryDatabase.CopyPath(Northwind, "lvpropstore-"); + try + { + create(path, sql); + return Stored(path, "W"); + } + finally { TemporaryDatabase.Delete(path); } + } + + /// How table 's blob is stored, and its length. + private static string Stored(string path, string table) + { + using var database = JetDatabase.Open(path, readOnly: true); + Table objects = database.OpenTable("MSysObjects"); + TableDefinition definition = objects.Definition; + int name = definition.RequireColumn("Name").Index, lvProp = definition.RequireColumn("LvProp").Index; + foreach ((RowId id, object?[] values) in objects.Rows().WithIds()) + { + if (values[name] as string != table) continue; + var page = new DataPage(); + page.Read(objects.Channel.ReadPageShared(id.Page), objects.Channel.Format); + byte[] descriptor = RowCodec.LongValueDescriptors( + definition.Columns, objects.Channel.Format, page.GetRow(id.Row))[lvProp]; + string form = descriptor[3] switch { 0x80 => "inline", 0x40 => "on a page", _ => $"0x{descriptor[3]:X2}" }; + return $"{form}, {((byte[])values[lvProp]!).Length} bytes"; + } + throw new InvalidOperationException($"No MSysObjects row for {table}."); + } + + private static void AceRun(string path, string sql) + { + using OleDbConnection connection = AceTestDatabase.Open(path); + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } + + private static void LibRedRun(string path, string sql) + { + using var database = JetDatabase.Open(path, readOnly: false); + new QueryEngine(database).ExecuteNonQuery(sql); + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/MemoCompressionAccessTests.cs b/test/LibRed.Engine.AccessTests/MemoCompressionAccessTests.cs index 63b1a26eb..11f875a00 100644 --- a/test/LibRed.Engine.AccessTests/MemoCompressionAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/MemoCompressionAccessTests.cs @@ -1,8 +1,9 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Catalog; using LibRed.Formats; +using LibRed.Storage; using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Engine.Tests; @@ -178,14 +179,13 @@ public void A_text_column_breaks_the_tie_the_other_way(string value, string expe for (int page = 1; page < channel.PageCount; page++) { byte[] bytes = channel.ReadPage(page).Span.ToArray(); - if (bytes[0] != 0x01) continue; - if (BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(4, 4)) != definitionPage) continue; - if (BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataRowCountOffset, 2)) == 0) + if (PageHeader.ReadType(bytes) != PageType.DataPage) continue; + if ((int)DataPage.ReadOwner(bytes, format) != definitionPage) continue; + if (DataPage.ReadRowCount(bytes, format) == 0) continue; - int start = BinaryPrimitives.ReadUInt16LittleEndian( - bytes.AsSpan(format.DataRowDirectoryOffset, 2)) & 0x1FFF; - int at = start + 2 + 4; + int start = DataPage.ReadSlot(bytes, format, 0).Offset; + int at = start + format.RowColumnCountSize + 4; return Convert.ToHexString(bytes.AsSpan(at, format.PageSize - 7 - at)); } return null; @@ -238,16 +238,15 @@ public void Libred_reads_back_what_ace_wrote(string value) for (int page = 1; page < channel.PageCount; page++) { byte[] bytes = channel.ReadPage(page).Span.ToArray(); - if (bytes[0] != 0x01) continue; - if (BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(4, 4)) != definitionPage) continue; - if (BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataRowCountOffset, 2)) == 0) + if (PageHeader.ReadType(bytes) != PageType.DataPage) continue; + if ((int)DataPage.ReadOwner(bytes, format) != definitionPage) continue; + if (DataPage.ReadRowCount(bytes, format) == 0) continue; - int start = BinaryPrimitives.ReadUInt16LittleEndian( - bytes.AsSpan(format.DataRowDirectoryOffset, 2)) & 0x1FFF; - int at = start + 2 + 4; - int length = (int)(BinaryPrimitives.ReadUInt32LittleEndian(bytes.AsSpan(at, 4)) & 0x3FFFFFFF); - return bytes.AsSpan(at + 12, length).ToArray(); + int start = DataPage.ReadSlot(bytes, format, 0).Offset; + int at = start + format.RowColumnCountSize + 4; + int length = LongValueStore.Read(bytes.AsSpan(at), format).Length; + return bytes.AsSpan(at + format.LongValueDescriptorSize, length).ToArray(); } return null; } @@ -291,21 +290,19 @@ private static void LibRedRun(string path, string[] statements) for (int page = 1; page < channel.PageCount; page++) { byte[] bytes = channel.ReadPage(page).Span.ToArray(); - if (bytes[0] != 0x01) continue; - if (BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(4, 4)) != definitionPage) continue; - if (BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataRowCountOffset, 2)) == 0) + if (PageHeader.ReadType(bytes) != PageType.DataPage) continue; + if ((int)DataPage.ReadOwner(bytes, format) != definitionPage) continue; + if (DataPage.ReadRowCount(bytes, format) == 0) continue; - int start = BinaryPrimitives.ReadUInt16LittleEndian( - bytes.AsSpan(format.DataRowDirectoryOffset, 2)) & 0x1FFF; - int at = start + 2 + 4; // past the row's column count and the LONG column - uint header = BinaryPrimitives.ReadUInt32LittleEndian(bytes.AsSpan(at, 4)); - byte flags = (byte)(bytes[at + 3] & 0xC0); - return $"{(flags == 0x80 ? "inline" : flags == 0x40 ? "page" : "chained")} " - + $"len={header & 0x3FFFFFFF}"; + int start = DataPage.ReadSlot(bytes, format, 0).Offset; + int at = start + format.RowColumnCountSize + 4; // past the row's column count and the LONG column + (int length, LongValueStore.StorageKind storage, _, _, _) = LongValueStore.Read(bytes.AsSpan(at), format); + return $"{(storage == LongValueStore.StorageKind.Inline ? "inline" : storage == LongValueStore.StorageKind.SinglePage ? "page" : "chained")} " + + $"len={length}"; } return null; } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/OleIndexRefusalAccessTests.cs b/test/LibRed.Engine.AccessTests/OleIndexRefusalAccessTests.cs index 866a40454..d8e7b7726 100644 --- a/test/LibRed.Engine.AccessTests/OleIndexRefusalAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/OleIndexRefusalAccessTests.cs @@ -8,7 +8,8 @@ namespace LibRed.Engine.Tests; /// An OLE column cannot be in an index — a key, a unique constraint, a relationship's — and ACE refuses every route /// that would put one there, up front, with "Invalid field definition '…' in definition of index or relationship.", /// leaving nothing behind (docs/format/page-03-04-index-btree.md). LibRed refuses the same statements at the same -/// point with the same message; it used to accept them on an empty table, after which every insert failed. +/// point with the same message; it used to accept them on an empty table, after which every insert failed. A +/// BigBinary column cannot be indexed either, and is refused the same way. /// [Collection(AceCollection.Name)] public class OleIndexRefusalAccessTests(ITestOutputHelper output) @@ -24,6 +25,12 @@ public class OleIndexRefusalAccessTests(ITestOutputHelper output) // Refused as OLE before the relationship's type match would refuse it. [InlineData("CREATE TABLE P (Id LONG CONSTRAINT pkP PRIMARY KEY)", "CREATE TABLE X (Id LONG, O LONGBINARY)", "ALTER TABLE X ADD CONSTRAINT fkXP FOREIGN KEY (O) REFERENCES P (Id)")] [InlineData("CREATE TABLE P (Id LONG CONSTRAINT pkP PRIMARY KEY)", "CREATE TABLE X (Id LONG, O LONGBINARY CONSTRAINT fkXP REFERENCES P (Id))")] + // A BigBinary column is refused on the same routes, with the same message. + [InlineData("CREATE TABLE X (Id LONG, B BIGBINARY)", "CREATE INDEX ixB ON X (B)")] + [InlineData("CREATE TABLE X (Id LONG, B BIGBINARY(20) CONSTRAINT pkX PRIMARY KEY)")] + [InlineData("CREATE TABLE X (Id LONG, B BIGBINARY(20) CONSTRAINT uxB UNIQUE)")] + [InlineData("CREATE TABLE X (Id LONG, B VARBINARY(20))", "CREATE INDEX ixB ON X (B)", "ALTER TABLE X ALTER COLUMN B BIGBINARY(20)")] + [InlineData("CREATE TABLE P (Id LONG CONSTRAINT pkP PRIMARY KEY)", "CREATE TABLE X (Id LONG, B BIGBINARY(20) CONSTRAINT fkXP REFERENCES P (Id))")] public void Libred_refuses_an_ole_column_in_an_index_as_ace_does(params string[] steps) { string northwind = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); diff --git a/test/LibRed.Engine.AccessTests/PageGroupProbeTest.cs b/test/LibRed.Engine.AccessTests/PageGroupProbeTest.cs new file mode 100644 index 000000000..e88baefb8 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/PageGroupProbeTest.cs @@ -0,0 +1,810 @@ +using System.Buffers.Binary; +using System.Data.OleDb; +using System.Globalization; +using System.Reflection; +using System.Text; +using LibRed; +using LibRed.Formats; +using LibRed.Pages; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Probe: how ACE chooses the page it allocates — the session extents it reserves in the global free map, and where +/// single pages go outside them — measured on a thousand-row INSERT … SELECT and variations of it. Reports; asserts +/// nothing. Every test is explicit: several take minutes, the reservation samples up to twenty. +/// +/// +/// A reservation is visible only while ACE's session is open — at close the unused rest is set free again — so the +/// *_reservations tests copy the file with the session still open, after waiting out ACE's flush timeout. +/// +[Collection(AceCollection.Name)] +public class PageGroupProbeTest(ITestOutputHelper output) +{ + [Theory(Explicit = true)] + [InlineData(0)] + [InlineData(40)] + public void Bulk_insert_page_map(int paddingRows) + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string origin = TemporaryDatabase.CreatePath("pagegroup-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + int pageSize; + using (FileStream stream = File.OpenRead(origin)) pageSize = JetFormatBase.Detect(stream).PageSize; + + var setup = new List + { + "CREATE TABLE Padding (Id LONG, Filler TEXT(255))", + "CREATE TABLE Digits (D BYTE)", + }; + for (int i = 0; i < paddingRows; i++) + setup.Add($"INSERT INTO Padding VALUES ({i}, String(255, 'x'))"); + for (int d = 0; d <= 9; d++) setup.Add($"INSERT INTO Digits (D) VALUES ({d})"); + setup.Add("CREATE TABLE Bulk (Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), " + + "Code BINARY(8), Amount DECIMAL(10,2))"); + setup.Add("CREATE INDEX ixBulkLabel ON Bulk (Label)"); + const string bulk = + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + "a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c ORDER BY 1"; + + string ace = TemporaryDatabase.CopyPath(origin, "pagegroup-ace-"); + string libred = TemporaryDatabase.CopyPath(origin, "pagegroup-libred-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + foreach (string statement in setup) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + File.Copy(origin, ace, overwrite: true); + File.Copy(origin, libred, overwrite: true); + + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = bulk; + command.ExecuteNonQuery(); + } + using (var database = JetDatabase.Open(libred, readOnly: false)) + new QueryEngine(database).ExecuteNonQuery(bulk); + + var report = new StringBuilder(); + report.AppendLine(CultureInfo.InvariantCulture, $"before: {new FileInfo(origin).Length / pageSize} pages"); + Describe(report, "ACE", ace); + Describe(report, "LibRed", libred); + output.WriteLine(report.ToString()); + } + finally + { + foreach (string path in new[] { origin, ace, libred }) TemporaryDatabase.Delete(path); + } + } + + /// The full table's insert, cut at a few row counts, with the file read while ACE's session is still + /// open (after its flush) and again after close: which pages the free and released maps hold at each point. + [Fact(Explicit = true)] + public void Bulk_insert_maps_while_the_session_is_open() + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string origin = TemporaryDatabase.CreatePath("pagegroup-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + var setup = new List { "CREATE TABLE Digits (D BYTE)" }; + for (int d = 0; d <= 9; d++) setup.Add($"INSERT INTO Digits (D) VALUES ({d})"); + setup.Add("CREATE TABLE Bulk (Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), " + + "Code BINARY(8), Amount DECIMAL(10,2))"); + setup.Add("CREATE INDEX ixBulkLabel ON Bulk (Label)"); + var report = new StringBuilder(); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + foreach (string statement in setup) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + + foreach (int rows in new[] { 240, 480, 560, 1000 }) + { + string ace = TemporaryDatabase.CopyPath(origin, "pagegroup-open-"); + string open = TemporaryDatabase.CreatePath("pagegroup-opencopy-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + $"a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c WHERE a.D * 100 + b.D * 10 + c.D < {rows} ORDER BY 1"; + command.ExecuteNonQuery(); + Thread.Sleep(3000); // past ACE's flush timeout, so the committed pages are on disk + using var source = new FileStream(ace, FileMode.Open, FileAccess.Read, FileShare.ReadWrite); + using var target = File.Create(open); + source.CopyTo(target); + } + report.AppendLine(CultureInfo.InvariantCulture, $"{rows} rows, session open: {Maps(open)}"); + report.AppendLine(CultureInfo.InvariantCulture, $"{rows} rows, after close: {Maps(ace)}"); + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(open); + } + } + output.WriteLine(report.ToString()); + } + finally { TemporaryDatabase.Delete(origin); } + } + + /// The full table's insert at sampled row counts — row by row across the known transitions — each read + /// with ACE's session still open: the pages newly in use since the last sample, and the pages ACE holds reserved + /// (neither in use nor free), with the free pages inside the file. + [Fact(Explicit = true)] + public void Bulk_insert_reservations() => + Reservations(FullColumns, LabelIndex, 0, 1000, + [(88, 102), (225, 245), (455, 482), (535, 610), (630, 690), (720, 780), (810, 820), (855, 925), (990, 1000)]); + + /// The same, with the first 200 rows inserted by the earlier session that built the table. + [Fact(Explicit = true)] + public void Bulk_insert_reservations_after_an_earlier_session() => + Reservations(FullColumns, LabelIndex, 200, 1000, [(201, 250), (455, 485), (540, 610), (630, 690), (720, 760)]); + + /// A heap table with no index, 200 rows from the earlier session: whether a data page that must grow + /// the file takes an extent when it is the second session's first growth. + [Fact(Explicit = true)] + public void Heap_insert_reservations_after_an_earlier_session() => + Reservations("Id LONG, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", "", + 200, 300, [(200, 240)]); + + /// An empty heap table, filled from a later session: whether its first data page takes an extent. + [Fact(Explicit = true)] + public void Heap_first_page_reservations() => + Reservations("Id LONG, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", "", + 0, 50, [(1, 20)]); + + /// A primary-keyed table, 600 rows from the earlier session, so the second session's first allocation + /// is the key's root split (at the 603rd row) while the last data page still has room. + [Fact(Explicit = true)] + public void Root_split_first_reservations() => + Reservations("Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", "", + 600, 625, [(600, 612)]); + + /// Whether growth lands at the end of the file or the next 8-page boundary depends on this session having + /// used the file's last group. Setup leaves Bulk (primary key only) at 600 rows, ending mid-group, and a heap + /// table Side with full pages; the second session first grows Side, then brings Bulk to its root split. + [Fact(Explicit = true)] + public void Growth_alignment_follows_the_session() + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string origin = TemporaryDatabase.CreatePath("pagegroup-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + + const string select = + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + "a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c WHERE a.D * 100 + b.D * 10 + c.D "; + var setup = new List { "CREATE TABLE Digits (D BYTE)" }; + for (int d = 0; d <= 9; d++) setup.Add($"INSERT INTO Digits (D) VALUES ({d})"); + setup.Add("CREATE TABLE Side (Id LONG, Filler TEXT(255))"); + for (int i = 0; i < 28; i++) setup.Add($"INSERT INTO Side VALUES ({i}, String(255, 's'))"); + setup.Add("CREATE TABLE Bulk (Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), " + + "Code BINARY(8), Amount DECIMAL(10,2))"); + setup.Add($"{select}< 600 ORDER BY 1"); + + string[] side = [.. Enumerable.Range(100, 20).Select(i => $"INSERT INTO Side VALUES ({i}, String(255, 't'))")]; + var runs = new List<(string Label, string[] Sql)> + { + ("nothing", []), + ("Side grown", side), + ("then Bulk split", [.. side, $"{select}>= 600 AND a.D * 100 + b.D * 10 + c.D < 603 ORDER BY 1"]), + }; + + var report = new StringBuilder(); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + foreach (string statement in setup) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + int pageSize; + using (FileStream stream = File.OpenRead(origin)) pageSize = JetFormatBase.Detect(stream).PageSize; + report.AppendLine(CultureInfo.InvariantCulture, $"after setup: {new FileInfo(origin).Length / pageSize} pages"); + + Dictionary previous = Kinds(origin); + foreach ((string label, string[] sql) in runs) + { + string ace = TemporaryDatabase.CopyPath(origin, "pagegroup-grow-"); + string open = TemporaryDatabase.CreatePath("pagegroup-growcopy-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + { + foreach (string statement in sql) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + Thread.Sleep(3000); // past ACE's flush timeout, so the committed pages are on disk + using var source = new FileStream(ace, FileMode.Open, FileAccess.Read, FileShare.ReadWrite); + using var target = File.Create(open); + source.CopyTo(target); + } + + Dictionary kinds = Kinds(open); + int pages = (int)(new FileInfo(open).Length / pageSize); + HashSet free; + string owners; + using (var database = JetDatabase.Open(open, readOnly: true)) + { + free = GlobalFree(database.OpenTable("MSysObjects"), 512); + owners = $"Side tdef {database.OpenTable("Side").Definition.DefinitionPage}, " + + $"Bulk tdef {database.OpenTable("Bulk").Definition.DefinitionPage}"; + } + string reserved = Ranges(Enumerable.Range(1, 511) + .Where(p => !free.Contains(p) && (p >= pages || !kinds.ContainsKey(p)))); + var changes = kinds.OrderBy(p => p.Key) + .Where(p => !previous.TryGetValue(p.Key, out string? before) || before != p.Value) + .Select(p => $"+{p.Key}{p.Value}"); + report.AppendLine(CultureInfo.InvariantCulture, + $"{label,-16} ({pages}; {owners}): {string.Join(" ", changes)}\n{"",18}reserved [{reserved}] " + + $"free in file [{Ranges(free.Where(p => p < pages))}]"); + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(open); + } + } + output.WriteLine(report.ToString()); + } + finally { TemporaryDatabase.Delete(origin); } + } + + private const string FullColumns = + "Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)"; + + private const string LabelIndex = "CREATE INDEX ixBulkLabel ON Bulk (Label)"; + + private void Reservations(string columns, string index, int preloaded, int last, (int From, int To)[] windows) + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string origin = TemporaryDatabase.CreatePath("pagegroup-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + var setup = new List { "CREATE TABLE Digits (D BYTE)" }; + for (int d = 0; d <= 9; d++) setup.Add($"INSERT INTO Digits (D) VALUES ({d})"); + setup.Add($"CREATE TABLE Bulk ({columns})"); + if (index.Length > 0) setup.Add(index); + const string select = + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + "a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c WHERE a.D * 100 + b.D * 10 + c.D "; + if (preloaded > 0) setup.Add($"{select}< {preloaded} ORDER BY 1"); + + var samples = new SortedSet(); + samples.Add(preloaded); // the session open, nothing inserted: the state it starts from + for (int rows = preloaded + 25; rows <= last; rows += 25) samples.Add(rows); + foreach ((int from, int to) in windows) + for (int rows = from; rows <= to; rows++) samples.Add(rows); + + var report = new StringBuilder(); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + foreach (string statement in setup) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + + int pageSize; + using (FileStream stream = File.OpenRead(origin)) pageSize = JetFormatBase.Detect(stream).PageSize; + Dictionary previous = []; + string previousReserved = ""; + foreach (int rows in samples) + { + string ace = TemporaryDatabase.CopyPath(origin, "pagegroup-res-"); + string open = TemporaryDatabase.CreatePath("pagegroup-rescopy-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + { + using OleDbCommand command = connection.CreateCommand(); + // Not BETWEEN: Jet swaps reversed bounds, so the empty range of the first sample would not be empty. + command.CommandText = $"{select}>= {preloaded} AND a.D * 100 + b.D * 10 + c.D < {rows} ORDER BY 1"; + command.ExecuteNonQuery(); + Thread.Sleep(3000); // past ACE's flush timeout, so the committed pages are on disk + using var source = new FileStream(ace, FileMode.Open, FileAccess.Read, FileShare.ReadWrite); + using var target = File.Create(open); + source.CopyTo(target); + } + + Dictionary used = Used(open); + byte[] file = File.ReadAllBytes(open); + int pages = file.Length / pageSize; + HashSet free; + using (var database = JetDatabase.Open(open, readOnly: true)) + free = GlobalFree(database.OpenTable("Bulk"), 512); + int first = used.Keys.Min(); + string reserved = Ranges(Enumerable.Range(first, 512 - first) + .Where(p => !free.Contains(p) && (p >= pages || PageHeader.ReadType(file.AsSpan(p * pageSize)) == 0))); + string freeInFile = Ranges(free.Where(p => p < pages)); + + var changes = new List(); + foreach (var (page, kind) in used.OrderBy(p => p.Key)) + if (!previous.TryGetValue(page, out string? before) || before != kind) changes.Add($"+{page}{kind}"); + if (changes.Count > 0 || reserved != previousReserved) + report.AppendLine(CultureInfo.InvariantCulture, + $"{rows,5} ({pages}): {string.Join(" ", changes),-28} reserved [{reserved}] free in file [{freeInFile}]"); + previous = used; + previousReserved = reserved; + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(open); + } + } + output.WriteLine(report.ToString()); + } + finally { TemporaryDatabase.Delete(origin); } + } + + /// The full table's 1,000-row insert under different Max Locks Per File limits: the page map each + /// leaves, to see whether intermediate commits are what start a new extent. + [Fact(Explicit = true)] + public void Bulk_insert_under_lock_limits() + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string origin = TemporaryDatabase.CreatePath("pagegroup-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + var setup = new List { "CREATE TABLE Digits (D BYTE)" }; + for (int d = 0; d <= 9; d++) setup.Add($"INSERT INTO Digits (D) VALUES ({d})"); + setup.Add("CREATE TABLE Bulk (Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), " + + "Code BINARY(8), Amount DECIMAL(10,2))"); + setup.Add("CREATE INDEX ixBulkLabel ON Bulk (Label)"); + var report = new StringBuilder(); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + foreach (string statement in setup) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + + report.AppendLine("(OLE DB default, for comparison, is the 1000-row line of Bulk_insert_snapshots)"); + foreach (int? limit in new int?[] { null, 1_000_000, 4000, 1000 }) + { + string ace = TemporaryDatabase.CopyPath(origin, "pagegroup-locks-"); + try + { + object dao = AceTestDatabase.CreateDaoEngine()!; + if (limit is not null) Invoke(dao, "SetOption", 62, limit.Value); // dbMaxLocksPerFile + object session = Invoke(dao, "CreateWorkspace", "", "admin", "", 2)!; + object database = Invoke(session, "OpenDatabase", ace)!; + Invoke(database, "Execute", + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + "a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c ORDER BY 1"); + Invoke(database, "Close"); + Invoke(session, "Close"); + report.AppendLine(CultureInfo.InvariantCulture, $"limit {limit?.ToString(CultureInfo.InvariantCulture) ?? "default"}: {Runs(ace)}"); + } + finally { TemporaryDatabase.Delete(ace); } + } + output.WriteLine(report.ToString()); + } + finally { TemporaryDatabase.Delete(origin); } + } + + /// The setup session itself — DDL, the Digits inserts, then the 200-row preload — each prefix run in one + /// session on a fresh DAO-made file and read with the session still open: the pages each step brings into use, + /// the pages reserved, and the free pages in the file. + [Fact(Explicit = true)] + public void Setup_session_reservations() + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string origin = TemporaryDatabase.CreatePath("pagegroup-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + + const string select = + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + "a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c WHERE a.D * 100 + b.D * 10 + c.D < "; + var steps = new List<(string Label, string Sql)> { ("CREATE Digits", "CREATE TABLE Digits (D BYTE)") }; + for (int d = 0; d <= 9; d++) steps.Add(($"Digits {d}", $"INSERT INTO Digits (D) VALUES ({d})")); + steps.Add(("CREATE Bulk", "CREATE TABLE Bulk (Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), " + + "Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2))")); + steps.Add(("CREATE INDEX", "CREATE INDEX ixBulkLabel ON Bulk (Label)")); + int ddl = steps.Count; + + // Prefixes: nothing, each DDL/insert step, then the preload at a few row counts after the whole DDL. + var runs = new List<(string Label, string[] Sql)> { ("session only", []) }; + for (int k = 1; k <= ddl; k++) runs.Add((steps[k - 1].Label, [.. steps.Take(k).Select(s => s.Sql)])); + foreach (int rows in new[] { 9, 10, 25, 100, 180, 185, 190, 191, 192, 195, 199, 200 }) + runs.Add(($"preload {rows}", [.. steps.Select(s => s.Sql), $"{select}{rows} ORDER BY 1"])); + + var report = new StringBuilder(); + int pageSize; + using (FileStream stream = File.OpenRead(origin)) pageSize = JetFormatBase.Detect(stream).PageSize; + report.AppendLine(CultureInfo.InvariantCulture, $"DAO-made origin: {new FileInfo(origin).Length / pageSize} pages"); + try + { + Dictionary previous = Kinds(origin); + foreach ((string label, string[] sql) in runs) + { + string ace = TemporaryDatabase.CopyPath(origin, "pagegroup-setup-"); + string open = TemporaryDatabase.CreatePath("pagegroup-setupcopy-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + { + foreach (string statement in sql) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + Thread.Sleep(3000); // past ACE's flush timeout, so the committed pages are on disk + using var source = new FileStream(ace, FileMode.Open, FileAccess.Read, FileShare.ReadWrite); + using var target = File.Create(open); + source.CopyTo(target); + } + + Dictionary kinds = Kinds(open); + byte[] file = File.ReadAllBytes(open); + int pages = file.Length / pageSize; + HashSet free; + using (var database = JetDatabase.Open(open, readOnly: true)) + free = GlobalFree(database.OpenTable("MSysObjects"), 512); + string reserved = Ranges(Enumerable.Range(1, 511) + .Where(p => !free.Contains(p) && (p >= pages || !kinds.ContainsKey(p)))); + var changes = kinds.OrderBy(p => p.Key) + .Where(p => !previous.TryGetValue(p.Key, out string? before) || before != p.Value) + .Select(p => $"+{p.Key}{p.Value}"); + report.AppendLine(CultureInfo.InvariantCulture, + $"{label,-14} ({pages}): {string.Join(" ", changes)}\n{"",16}reserved [{reserved}] free in file [{Ranges(free.Where(p => p < pages))}]"); + previous = kinds; + } + finally + { + TemporaryDatabase.Delete(ace); + TemporaryDatabase.Delete(open); + } + } + output.WriteLine(report.ToString()); + } + finally { TemporaryDatabase.Delete(origin); } + } + + /// Every page that holds something, as its kind: D/L/I with the owning TDEF page for data and index + /// pages, T for a table definition, otherwise the type byte. + private static Dictionary Kinds(string path) + { + byte[] file = File.ReadAllBytes(path); + JetFormatBase format = JetFormatBase.Detect(new MemoryStream(file)); + int pageSize = format.PageSize; + var kinds = new Dictionary(); + for (int page = 1; page < file.Length / pageSize; page++) + { + ReadOnlySpan p = file.AsSpan(page * pageSize, pageSize); + if (PageHeader.ReadType(p) == 0) continue; + int owner = PageHeader.ReadType(p) == PageType.DataPage + ? (int)DataPage.ReadOwner(p, format) + : IndexTree.ReadOwner(p, format); + kinds[page] = PageHeader.ReadType(p) switch + { + PageType.DataPage => $"D@{owner}", + PageType.LeafIndexPage => $"L@{owner}", + PageType.IntermediateIndexPage => $"I@{owner}", + PageType.TableDefinition => "T", + _ => $"x{p[0]:X2}", + }; + } + return kinds; + } + + /// File length, pages whose type byte is 0 (never written), and the pages set in the global free and + /// released maps — wherever page 0's pointers put them — whole map range. + private static string Maps(string path) + { + byte[] file = File.ReadAllBytes(path); + using var database = JetDatabase.Open(path, readOnly: true); + JetFormatBase format = database.Format; + int pages = file.Length / format.PageSize; + Table any = database.OpenTable("Bulk"); + (int freeRow, int freePage) = database.DefinitionPage.FreePagesMap; + (int releasedRow, int releasedPage) = database.DefinitionPage.ReleasedPagesMap; + var freeHolder = new DataPage(); + freeHolder.Read(any.Channel.ReadPageShared(freePage), any.Channel.Format); + var releasedHolder = new DataPage(); + releasedHolder.Read(any.Channel.ReadPageShared(releasedPage), any.Channel.Format); + var zero = Enumerable.Range(1, pages - 1).Where(p => PageHeader.ReadType(file.AsSpan(p * format.PageSize)) == 0); + return $"{pages} pages; zero [{Ranges(zero)}]; free [{Ranges(Bits(freeHolder.GetRow(freeRow), format))}]; " + + $"released [{Ranges(Bits(releasedHolder.GetRow(releasedRow), format))}]"; + + static List Bits(ReadOnlySpan map, JetFormatBase format) + { + var set = new List(); + if (UsageMap.RecordType(map) != UsageMapType.Inline) { set.Add(-1); return set; } + int start = UsageMap.StartPage(map, format); + int header = format.UsageMapInlineHeaderSize; + for (int i = header; i < map.Length; i++) + for (int bit = 0; bit < 8; bit++) + if ((map[i] & (1 << bit)) != 0) set.Add(start + (i - header) * 8 + bit); + return set; + } + } + + /// ACE only, the same insert cut at Id < rows for every row count, reporting the pages each + /// extra row brought into use: the order ACE allocates in. + [Theory(Explicit = true)] + [InlineData("Id LONG, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", "")] + [InlineData("Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", "")] + [InlineData("Id LONG, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", "CREATE INDEX ixBulkLabel ON Bulk (Label)")] + [InlineData("Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", "CREATE INDEX ixBulkLabel ON Bulk (Label)")] + public void Bulk_insert_snapshots(string columns, string index) => Snapshots(columns, index, 0); + + /// The full table, with the first 200 rows inserted by an earlier session. + [Fact(Explicit = true)] + public void Bulk_insert_snapshots_after_an_earlier_session() => + Snapshots("Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), Code BINARY(8), Amount DECIMAL(10,2)", + "CREATE INDEX ixBulkLabel ON Bulk (Label)", 200); + + private void Snapshots(string columns, string index, int preloaded) + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + string origin = TemporaryDatabase.CreatePath("pagegroup-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object db = Invoke(workspace, "CreateDatabase", origin, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(db, "Close"); + var setup = new List { "CREATE TABLE Digits (D BYTE)" }; + for (int d = 0; d <= 9; d++) setup.Add($"INSERT INTO Digits (D) VALUES ({d})"); + setup.Add($"CREATE TABLE Bulk ({columns})"); + if (index.Length > 0) setup.Add(index); + const string select = + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + "a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c WHERE a.D * 100 + b.D * 10 + c.D "; + if (preloaded > 0) setup.Add($"{select}< {preloaded} ORDER BY 1"); + var report = new StringBuilder(); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(origin)) + foreach (string statement in setup) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + int pageSize; + using (FileStream stream = File.OpenRead(origin)) pageSize = JetFormatBase.Detect(stream).PageSize; + report.AppendLine(CultureInfo.InvariantCulture, $"before: {new FileInfo(origin).Length / pageSize} pages"); + + Dictionary previous = []; + for (int rows = preloaded + 1; rows <= 1000; rows++) + { + string ace = TemporaryDatabase.CopyPath(origin, "pagegroup-snap-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = $"{select}BETWEEN {preloaded} AND {rows - 1} ORDER BY 1"; + command.ExecuteNonQuery(); + } + Dictionary used = Used(ace); + var changes = new List(); + foreach (var (page, kind) in used.OrderBy(p => p.Key)) + if (!previous.TryGetValue(page, out string? before) || before != kind) changes.Add($"+{page}{kind}"); + foreach (int page in previous.Keys.Where(p => !used.ContainsKey(p)).Order()) changes.Add($"-{page}"); + if (changes.Count > 0) + report.AppendLine(CultureInfo.InvariantCulture, + $"{rows,5} ({new FileInfo(ace).Length / pageSize}): {string.Join(" ", changes)}"); + previous = used; + if (rows % 100 == 0) report.AppendLine(CultureInfo.InvariantCulture, $" {Runs(ace)}"); + } + finally { TemporaryDatabase.Delete(ace); } + } + output.WriteLine(report.ToString()); + } + finally { TemporaryDatabase.Delete(origin); } + } + + /// The file from Bulk's first page on as runs: 51-61 D0 is data pages whose rows continue one + /// another from id 0, L leaf pages, . free, with the file length and the global free pages. + private static string Runs(string path) + { + byte[] file = File.ReadAllBytes(path); + using var database = JetDatabase.Open(path, readOnly: true); + Table bulk = database.OpenTable("Bulk"); + int tdef = bulk.Definition.DefinitionPage; + JetFormatBase format = database.Format; + int pageSize = format.PageSize; + HashSet free = GlobalFree(bulk, file.Length / pageSize); + var parts = new List(); + int start = -1, startId = -1, nextId = -1; string kind = ""; + for (int page = tdef + 1; page <= file.Length / pageSize; page++) + { + string k; int first = -1, last = -1; + if (page == file.Length / pageSize) k = "end"; + else + { + ReadOnlySpan p = file.AsSpan(page * pageSize, pageSize); + k = PageHeader.ReadType(p) switch + { + PageType.DataPage when (int)DataPage.ReadOwner(p, format) == tdef => "D", + PageType.LeafIndexPage => "L", + PageType.IntermediateIndexPage => "I", + _ when free.Contains(page) => ".", + _ => $"x{p[0]:X2}", + }; + if (k == "D") (first, last) = IdRange(bulk, page); + } + bool continues = k == kind && (k != "D" || first == nextId); + if (!continues) + { + if (start >= 0) parts.Add(Run(start, page - 1, kind)); + start = page; kind = k; startId = first; + } + if (k == "D") nextId = last + 1; + } + return $"{file.Length / pageSize} pages | " + string.Join(" ", parts); + + string Run(int from, int to, string what) => + (from == to ? $"{from}" : $"{from}-{to}") + (what == "D" ? $"D{startId}" : what); + } + + /// Every page past Bulk's definition that holds something: D for a Bulk data page, L leaf, I + /// intermediate, or the type byte. + private static Dictionary Used(string path) + { + byte[] file = File.ReadAllBytes(path); + using var database = JetDatabase.Open(path, readOnly: true); + int tdef = database.OpenTable("Bulk").Definition.DefinitionPage; + JetFormatBase format = database.Format; + int pageSize = format.PageSize; + var used = new Dictionary(); + for (int page = tdef + 1; page < file.Length / pageSize; page++) + { + ReadOnlySpan p = file.AsSpan(page * pageSize, pageSize); + string? kind = PageHeader.ReadType(p) switch + { + PageType.DataPage when (int)DataPage.ReadOwner(p, format) == tdef => "D", + PageType.LeafIndexPage => "L", + PageType.IntermediateIndexPage => "I", + _ when p[0] == 0 => null, + _ => $"x{p[0]:X2}", + }; + if (kind is not null) used[page] = kind; + } + return used; + } + + private static (int First, int Last) IdRange(Table bulk, int page) + { + var data = new DataPage(); + data.Read(bulk.Channel.ReadPageShared(page), bulk.Channel.Format); + var ids = new List(); + for (int row = 0; row < data.RowCount; row++) + if (data.Rows[row] is { IsDeleted: false, HasOverflow: false }) + ids.Add(BinaryPrimitives.ReadInt32LittleEndian(data.GetRow(row)[bulk.Channel.Format.RowColumnCountSize..])); + return ids.Count == 0 ? (-2, -2) : (ids.Min(), ids.Max()); + } + + private static void Describe(StringBuilder report, string name, string path) + { + byte[] file = File.ReadAllBytes(path); + using var database = JetDatabase.Open(path, readOnly: true); + Table bulk = database.OpenTable("Bulk"); + int bulkTdef = bulk.Definition.DefinitionPage; + var owned = new HashSet(bulk.UsageMap.DataPages()); + var freeSpace = new HashSet(bulk.UsageMap.FreeDataPages()); + JetFormatBase format = database.Format; + int pageSize = format.PageSize; + HashSet globalFree = GlobalFree(bulk, file.Length / pageSize); + + report.AppendLine(CultureInfo.InvariantCulture, + $"==== {name}: {file.Length / pageSize} pages; Bulk tdef {bulkTdef}; owned [{Ranges(owned)}]; " + + $"free-space [{Ranges(freeSpace)}]; global free (in file) [{Ranges(globalFree)}]"); + for (int page = 1; page < file.Length / pageSize; page++) + { + ReadOnlySpan p = file.AsSpan(page * pageSize, pageSize); + PageType type = PageHeader.ReadType(p); + int tdef = type == PageType.DataPage ? (int)DataPage.ReadOwner(p, format) : IndexTree.ReadOwner(p, format); + string what = type switch + { + PageType.DataPage when tdef == bulkTdef => $"D {BulkIds(bulk, page)}", + PageType.DataPage => $"d tdef {tdef}", + PageType.LeafIndexPage => $"{(tdef == bulkTdef ? "L" : "l")} tdef {tdef} entries-end 0x{IndexTree.ReadFreeSpace(p, format):X}", + PageType.IntermediateIndexPage => $"{(tdef == bulkTdef ? "I" : "i")} tdef {tdef}", + _ => $"type 0x{p[0]:X2}", + }; + string flags = (owned.Contains(page) ? " owned" : "") + (freeSpace.Contains(page) ? " has-space" : "") + + (globalFree.Contains(page) ? " FREE" : ""); + if (page < 8 && type != PageType.DataPage) continue; + report.AppendLine(CultureInfo.InvariantCulture, $" {page,4} {what}{flags}"); + } + } + + private static string BulkIds(Table bulk, int page) + { + var data = new DataPage(); + data.Read(bulk.Channel.ReadPageShared(page), bulk.Channel.Format); + var ids = new List(); + for (int row = 0; row < data.RowCount; row++) + { + if (data.Rows[row] is not { IsDeleted: false, HasOverflow: false }) continue; + ids.Add(BinaryPrimitives.ReadInt32LittleEndian(data.GetRow(row)[bulk.Channel.Format.RowColumnCountSize..])); + } + return ids.Count == 0 ? "empty" : $"{ids.Count} rows ids {ids.Min()}-{ids.Max()}"; + } + + /// The global free-pages map, wherever page 0's 0x18 puts it: an inline map with a set bit for + /// a free page. + private static HashSet GlobalFree(Table any, int pages) + { + var page0 = new DatabaseDefinitionPage(); + page0.Read(any.Channel.ReadPageShared(0), any.Channel.Format); + (int row, int mapPage) = page0.FreePagesMap; + var holder = new DataPage(); + holder.Read(any.Channel.ReadPageShared(mapPage), any.Channel.Format); + ReadOnlySpan map = holder.GetRow(row); + var free = new HashSet(); + if (UsageMap.RecordType(map) != UsageMapType.Inline) return free; + int start = UsageMap.StartPage(map, any.Channel.Format); + int header = any.Channel.Format.UsageMapInlineHeaderSize; + for (int i = header; i < map.Length; i++) + for (int bit = 0; bit < 8; bit++) + if ((map[i] & (1 << bit)) != 0 && start + (i - header) * 8 + bit < pages) free.Add(start + (i - header) * 8 + bit); + return free; + } + + private static string Ranges(IEnumerable pages) + { + var sorted = pages.Order().ToList(); + var parts = new List(); + for (int i = 0; i < sorted.Count;) + { + int j = i; + while (j + 1 < sorted.Count && sorted[j + 1] == sorted[j] + 1) j++; + parts.Add(i == j ? $"{sorted[i]}" : $"{sorted[i]}-{sorted[j]}"); + i = j + 1; + } + return string.Join(",", parts); + } + + private static object? Invoke(object target, string member, params object?[] args) => + target.GetType().InvokeMember(member, BindingFlags.InvokeMethod, null, target, args); +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/PermissionRowsAccessTests.cs b/test/LibRed.Engine.AccessTests/PermissionRowsAccessTests.cs new file mode 100644 index 000000000..9e956702f --- /dev/null +++ b/test/LibRed.Engine.AccessTests/PermissionRowsAccessTests.cs @@ -0,0 +1,93 @@ +using System.Data.OleDb; +using System.Reflection; +using LibRed.Catalog; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// The MSysACEs rows a new object gets are its container's inheritable grants, the Creator's becoming the +/// owner's (system-catalog §11) — so they differ from database to database. Each is checked in two whose Tables +/// containers grant differently: a fresh DAO database, whose owner row comes out 0xF00FE, and Northwind, +/// whose Tables container also grants the admin user 0xFFEFF. +/// +[Collection(AceCollection.Name)] +public class PermissionRowsAccessTests(ITestOutputHelper output) : TempDatabaseTest +{ + private static readonly string[] Statements = + [ + "CREATE TABLE PermT (Id LONG CONSTRAINT pkPermT PRIMARY KEY)", + "CREATE VIEW PermV AS SELECT Id FROM PermT", + "CREATE TABLE PermC (Id LONG, P LONG, CONSTRAINT fkPerm FOREIGN KEY (P) REFERENCES PermT (Id))", + ]; + + [Theory] + [InlineData("fresh")] + [InlineData("northwind")] + public void A_new_objects_permission_rows_are_the_ones_ace_writes(string database) + { + string origin = database == "fresh" + ? CreateEmptyThroughDao() + : TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "perm-nw-"); + string ace = TemporaryDatabase.CopyPath(origin, "perm-ace-"), libred = TemporaryDatabase.CopyPath(origin, "perm-libred-"); + try + { + using (OleDbConnection connection = AceTestDatabase.Open(ace)) + foreach (string sql in Statements) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = sql; + command.ExecuteNonQuery(); + } + using (var db = JetDatabase.Open(libred, readOnly: false)) + { + var engine = new QueryEngine(db); + foreach (string sql in Statements) engine.ExecuteNonQuery(sql); + } + + string expected = Grants(ace), actual = Grants(libred); + output.WriteLine($"ACE {expected}\nLibRed {actual}"); + Assert.Equal(expected, actual); + } + finally + { + foreach (string path in new[] { origin, ace, libred }) TemporaryDatabase.Delete(path); + } + } + + /// Each new object's MSysACEs rows in the order they are stored: its name, then each row's account, + /// mask and inheritability. + private static string Grants(string path) + { + using var db = JetDatabase.Open(path, readOnly: true); + Table objects = db.OpenTable("MSysObjects"); + int idCol = objects.Definition.RequireColumn("Id").Index, nameCol = objects.Definition.RequireColumn("Name").Index; + var names = objects.Rows() + .Where(r => r[nameCol] is string n && (n.StartsWith("Perm", StringComparison.Ordinal) || n.Contains("fkPerm", StringComparison.Ordinal))) + .ToDictionary(r => (int)r[idCol]!, r => (string)r[nameCol]!); + + Table aces = db.OpenTable("MSysACEs"); + TableDefinition def = aces.Definition; + int oid = def.RequireColumn("ObjectId").Index, sid = def.RequireColumn("SID").Index, + acm = def.RequireColumn("ACM").Index, inherit = def.RequireColumn("FInheritable").Index; + return string.Join("; ", aces.Rows() + .Where(r => names.ContainsKey((int)r[oid]!)) + .GroupBy(r => names[(int)r[oid]!]) + .OrderBy(g => g.Key, StringComparer.Ordinal) + .Select(g => $"{g.Key}: " + string.Join(", ", g.Select(r => + $"{Convert.ToHexString((byte[])r[sid]!)} 0x{(int)r[acm]!:X6}{((bool)r[inherit]! ? " inheritable" : "")}")))); + } + + private static string CreateEmptyThroughDao() + { + object engine = AceTestDatabase.CreateDaoEngine() ?? throw new InvalidOperationException("DAO is not registered."); + string path = TemporaryDatabase.CreatePath("perm-fresh-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + Invoke(Invoke(workspace, "CreateDatabase", path, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!, "Close"); + return path; + } + + private static object? Invoke(object target, string member, params object?[] args) => + target.GetType().InvokeMember(member, BindingFlags.InvokeMethod, null, target, args); +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/QueryNamesAccessTests.cs b/test/LibRed.Engine.AccessTests/QueryNamesAccessTests.cs new file mode 100644 index 000000000..1adbf3c7d --- /dev/null +++ b/test/LibRed.Engine.AccessTests/QueryNamesAccessTests.cs @@ -0,0 +1,219 @@ +using System.Data.OleDb; +using System.Globalization; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// How ACE reads the names in a query, through ACE over OLE DB and through LibRed: what Customers!CustomerID +/// resolves to and what the column is called, a declared form control as a parameter and a stored query that declares +/// one, a reserved word naming a column after a period or a bang, a name in any script written unbracketed, and Yes, +/// No, On and Off as True and False. +/// +[Collection(AceCollection.Name)] +public class QueryNamesAccessTests(ITestOutputHelper output) : TempDatabaseTest +{ + // Every reserved word LibRed knows, each a column of table K. ACE refuses Union and When even qualified. + private static readonly string[] Keywords = + [ + "Select", "From", "Where", "Top", "As", "And", "Or", "Not", "Xor", "Eqv", "Imp", "Band", "Bor", "Bxor", "Bnot", + "Like", "Mod", "Inner", "Left", "Right", "Full", "Outer", "Join", "In", "On", "Order", "Group", "Is", "By", + "Having", "Exists", "If", "Then", "Distinctrow", "Distinct", "Percent", "Cross", "Apply", "Over", "Partition", + "Case", "Else", "End", "Offset", "Fetch", "Next", "First", "Rows", "Row", "Only", "Between", "All", "Intersect", + "Except", "Create", "Table", "Begin", "Commit", "Rollback", "Transaction", "Work", "Alter", "Rename", "To", "Add", + "Drop", "Column", "Insert", "Into", "Values", "Primary", "Key", "Constraint", "Foreign", "References", "Delete", + "Update", "Cascade", "Restrict", "Action", "Set", "Default", "No", "Unique", "Clustered", "Identity", + "Nonclustered", "Index", "Temporary", "With", "Compression", "Comp", "Disallow", "Ignore", "Check", "View", + "Procedure", "Parameters", "Execute", "Exec", "Asc", "Desc", "True", "False", "Null", "Yes", "Off", "Language", + ]; + + // ANSI-92 reserved words Access's own SQL does not use. OLE DB runs ACE in ANSI-92 mode, and ACE before Access + // version 2311 (build 16.0.17029) refuses one as a name even qualified, with reserved error -1001, which has no + // message; 2311 fixed that, so the current engine takes each as any other name. The ACE 2016 redistributable and + // ACE 2010 are both older, and LibRed follows the current engine. + private static readonly HashSet RefusedBefore2311 = new(StringComparer.OrdinalIgnoreCase) + { + "Full", "Then", "Cross", "End", "Fetch", "Next", "Rows", "Only", "Intersect", "Except", "Restrict", "Temporary", + "Language", + }; + + // The operator words, which ACE refuses after a bang (as are Select and All). + private static readonly HashSet RefusedAfterBang = new(StringComparer.OrdinalIgnoreCase) + { + "Select", "All", "And", "Or", "Not", "Xor", "Eqv", "Imp", "Band", "Bor", "Bxor", "Bnot", "Like", "Mod", "In", + "Is", "Between", "Exists", + }; + + private static readonly string[] Setup = + [ + $"CREATE TABLE K ({string.Join(", ", Keywords.Select(k => $"[{k}] LONG"))})", + $"INSERT INTO K ({string.Join(", ", Keywords.Select(k => $"[{k}]"))}) VALUES ({string.Join(", ", Keywords.Select((_, i) => (i + 100).ToString(CultureInfo.InvariantCulture)))})", + "CREATE TABLE [Árú] ([Név] TEXT(10), [Номер] LONG, [名前] TEXT(10), [ΑΒΓ] LONG, [straße] LONG, [Ñandú] LONG, [n1] LONG)", + "INSERT INTO [Árú] VALUES ('x', 2, 'y', 3, 4, 5, 6)", + ]; + + public static TheoryData Queries + { + get + { + var queries = new TheoryData + { + "SELECT Customers!CustomerID, [Customers]![City], `Customers`!`Country` FROM Customers WHERE CustomerID = 'ALFKI'", + "SELECT Customers!CustomerID AS X FROM Customers WHERE CustomerID = 'ALFKI'", + "SELECT c!CustomerID FROM Customers AS c WHERE c!City = 'Berlin'", + "SELECT [c]!CustomerID FROM Customers AS [c] WHERE [c]![City] = 'Berlin'", + "SELECT City, COUNT(*) AS N FROM Customers GROUP BY Customers!City HAVING Customers!City > 'S' ORDER BY Customers!City DESC", + "SELECT TOP 3 CustomerID FROM Customers ORDER BY Customers!CustomerID DESC", + "SELECT CustomerID, (SELECT COUNT(*) FROM Orders WHERE Orders!CustomerID = Customers!CustomerID) AS N FROM Customers WHERE Country = 'Germany'", + "SELECT d!X FROM (SELECT CustomerID AS X FROM Customers WHERE Country = 'Germany') AS d ORDER BY d!X", + "SELECT * FROM (SELECT Customers!City, [Customers]![Country] FROM Customers WHERE CustomerID = 'ALFKI')", + "SELECT [Customers!City] FROM (SELECT Customers!City FROM Customers WHERE CustomerID = 'ALFKI')", + "SELECT Customers!City FROM Customers WHERE CustomerID = 'ALFKI' UNION SELECT Customers!City FROM Customers WHERE CustomerID = 'ANATR'", + "SELECT MAX(Customers!CustomerID) AS M FROM Customers", + "SELECT Név, Номер, 名前, ΑΒΓ, straße, Ñandú, n1 FROM Árú", + "SELECT Árú.Név, Árú!Номер FROM Árú", + "SELECT a.Név FROM Árú AS a WHERE a.ΑΒΓ = 3", + // Through CInt and IIF: ACE types True and False, and so these, as Int16 where LibRed has a Boolean. + "SELECT CInt(Yes) AS A, CInt(No) AS B, CInt(On) AS C, CInt(Off) AS D, CInt(True) AS E, CInt(False) AS F FROM K", + "SELECT Yes + 1 AS A, IIF(Not No, 1, 2) AS B, IIF(Yes AND No, 1, 2) AS C, IIF(Yes = True, 1, 2) AS D, IIF(Off = False, 1, 2) AS E FROM K", + "SELECT CustomerID FROM Customers WHERE (CustomerID = 'ALFKI') = Yes", + "SELECT CustomerID FROM Customers WHERE On AND CustomerID < 'B'", + // Unqualified, a name that is also a constant is the constant; qualified, it is the column. + "SELECT CInt(Yes) AS A, K.Yes AS B, CInt(No) AS C, K.No AS D, CInt(Off) AS E, K.Off AS F, K.On AS G FROM K", + }; + foreach (string k in Keywords) + { + queries.Add($"SELECT K.{k} FROM K"); + if (!RefusedAfterBang.Contains(k)) queries.Add($"SELECT K!{k} FROM K"); + } + return queries; + } + } + + // A declared parameter, bound by position in ACE and by name in LibRed. + public static TheoryData Declared => new() + { + { "PARAMETERS [Forms]![f]![c] Long; SELECT CustomerID, Forms!f!c FROM Customers WHERE CustomerID = 'ALFKI'", "Forms!f!c" }, + { "PARAMETERS Forms!f!c Long; SELECT [Forms]![f]![c] FROM Customers WHERE CustomerID = 'ALFKI'", "Forms!f!c" }, + { "PARAMETERS Forms!x Long; SELECT Forms!x, Forms.x FROM Customers WHERE CustomerID = 'ALFKI'", "Forms!x" }, + { "PARAMETERS Forms!f!c Long; SELECT Forms!f.c, Forms.f!c FROM Customers WHERE CustomerID = 'ALFKI'", "Forms!f!c" }, + { "PARAMETERS Customers!City Long; SELECT Customers!City FROM Customers WHERE CustomerID = 'ALFKI'", "Customers!City" }, + { "PARAMETERS p Long; SELECT p FROM Customers WHERE CustomerID = 'ALFKI'", "p" }, + { "PARAMETERS [Forms]![f]![c] Long; SELECT CustomerID FROM Orders WHERE OrderID = Forms!f!c + 10247", "Forms!f!c" }, + }; + + [Theory] + [MemberData(nameof(Queries))] + public void A_query_reads_and_names_as_ace_does(string query) => + Matches(query, command => { }, null, + refusedBefore2311: RefusedBefore2311.Any(k => query == $"SELECT K.{k} FROM K" || query == $"SELECT K!{k} FROM K")); + + [Theory] + [MemberData(nameof(Declared))] + public void A_declared_form_control_is_a_parameter_as_in_ace(string query, string name) => + Matches(query, command => command.Parameters.AddWithValue("?", 1), new Dictionary { [name] = 1 }); + + // A stored query declaring a form control, saved by Access, reads back and runs through LibRed; made by LibRed, it + // stores the same declaration and runs in ACE. ACE's own CREATE PROCEDURE will not take a chain as a parameter + // name, so Access's copy is saved through DAO, as the Access UI saves one. + [Fact] + public void A_stored_query_declaring_a_form_control_round_trips_with_ace() + { + const string body = "SELECT CustomerID FROM Customers WHERE City = Forms!frmMenu!txtCity"; + + string byAce = Copy(), byLibRed = Copy(); + try + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not registered in this bitness."); + object database = Invoke(engine!, "OpenDatabase", byAce, false, false, "")!; + Invoke(database, "CreateQueryDef", "ByCity", $"PARAMETERS [Forms]![frmMenu]![txtCity] Text ( 20 ); {body}"); + Invoke(database, "Close"); + + using (var db = JetDatabase.Open(byLibRed, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery($"CREATE PROCEDURE ByCity [Forms]![frmMenu]![txtCity] Text(20) AS {body}"); + + string aceSql, libredSql; + using (var db = JetDatabase.Open(byAce, readOnly: true)) + { + aceSql = db.Catalog.FindQuery("ByCity")!.Sql!; + Assert.Equal("Forms!frmMenu!txtCity", db.Catalog.FindQuery("ByCity")!.Parameters.Single().Name); + Assert.Equal(["ALFKI"], new QueryEngine(db).ExecuteQuery("EXECUTE ByCity 'Berlin'").Rows.Select(r => r[0])); + } + using (var db = JetDatabase.Open(byLibRed, readOnly: true)) + libredSql = db.Catalog.FindQuery("ByCity")!.Sql!; + output.WriteLine($"ACE {aceSql}\nLibRed {libredSql}"); + Assert.Equal(aceSql, libredSql); + + using OleDbConnection ace = AceTestDatabase.Open(byLibRed); + using OleDbCommand run = ace.CreateCommand(); + run.CommandText = "EXECUTE ByCity 'Berlin'"; + Assert.Equal("ALFKI", run.ExecuteScalar()); + } + finally + { + TemporaryDatabase.Delete(byAce); + TemporaryDatabase.Delete(byLibRed); + } + } + + /// Runs the query through ACE and through LibRed, over a copy of Northwind with tables K and Árú made + /// by ACE, and compares each column's name and type and every value. marks a + /// query an ACE older than version 2311 refuses (), which is skipped there. + private void Matches(string query, Action bind, IReadOnlyDictionary? parameters, + bool refusedBefore2311 = false) + { + string path = Copy(); + try + { + string ace; + using (OleDbConnection connection = AceTestDatabase.Open(path)) + { + foreach (string statement in Setup) + { + using OleDbCommand setup = connection.CreateCommand(); + setup.CommandText = statement; + setup.ExecuteNonQuery(); + } + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = query; + bind(command); + OleDbDataReader executed; + try { executed = command.ExecuteReader(); } + catch (OleDbException) when (refusedBefore2311) + { + Assert.Skip("This ACE predates Access version 2311 and refuses an unused ANSI-92 reserved word as a name " + + "(reserved error -1001)."); + throw; + } + using OleDbDataReader reader = executed; + var columns = Enumerable.Range(0, reader.FieldCount).Select(i => $"{reader.GetName(i)}:{reader.GetFieldType(i).Name}").ToList(); + var rows = new List(); + while (reader.Read()) + rows.Add([.. Enumerable.Range(0, reader.FieldCount).Select(i => reader.IsDBNull(i) ? null : reader.GetValue(i))]); + ace = Describe(columns, rows); + } + + string libred; + using (var database = JetDatabase.Open(path, readOnly: true)) + { + var result = new QueryEngine(database).ExecuteQuery(query, parameters); + libred = Describe([.. result.ColumnNames.Zip(result.ColumnTypes, (n, t) => $"{n}:{t.Name}")], result.Rows); + } + + output.WriteLine($"ACE {ace}\nLibRed {libred}"); + Assert.Equal(ace, libred); + } + finally { TemporaryDatabase.Delete(path); } + } + + private static object? Invoke(object target, string member, params object?[] args) => + target.GetType().InvokeMember(member, System.Reflection.BindingFlags.InvokeMethod, null, target, args); + + private static string Copy() => + TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "names-"); + + private static string Describe(IReadOnlyList columns, IEnumerable rows) => + $"[{string.Join(", ", columns)}] " + string.Join(" | ", rows.Select(row => + string.Join(", ", row.Select(v => v is null ? "NULL" : Convert.ToString(v, CultureInfo.InvariantCulture))))); +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ReferencesAndIdentityAccessTests.cs b/test/LibRed.Engine.AccessTests/ReferencesAndIdentityAccessTests.cs index 1fd695730..10c680a38 100644 --- a/test/LibRed.Engine.AccessTests/ReferencesAndIdentityAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/ReferencesAndIdentityAccessTests.cs @@ -182,6 +182,41 @@ public void A_relationship_between_column_types_matches_ace(string parentType, s public void Add_column_constraints_match_ace(string statements, bool accepted) => AssertSameOutcome(statements, "T", insertColumn: "V", accepted); + // A key's values are held to the same rules on every write, not only when the index is built: a primary key + // and a WITH DISALLOW NULL index take no Null, from INSERT or UPDATE. An AutoNumber takes no UPDATE at all — + // not even to its own value, and whether or not it is a key — nor an explicit Null on INSERT, while an + // explicit number is accepted and the counter carries on after it. + public static TheoryData KeyWrites => new() + { + { "CREATE TABLE T (Id LONG, V TEXT(10), CONSTRAINT pk PRIMARY KEY (Id));INSERT INTO T (V) VALUES ('a')", false }, + { "CREATE TABLE T (Id LONG CONSTRAINT pk PRIMARY KEY, V TEXT(10));INSERT INTO T (Id, V) VALUES (NULL, 'a')", false }, + { "CREATE TABLE T (Id LONG CONSTRAINT pk PRIMARY KEY, V TEXT(10));INSERT INTO T (Id, V) VALUES (1, 'a');UPDATE T SET Id = NULL", false }, + { "CREATE TABLE T (V TEXT(10), W LONG);CREATE INDEX ix ON T (W) WITH DISALLOW NULL;INSERT INTO T (V) VALUES ('a')", false }, + { "CREATE TABLE T (V TEXT(10), W LONG);CREATE INDEX ix ON T (W) WITH DISALLOW NULL;INSERT INTO T (V, W) VALUES ('a', 1);UPDATE T SET W = NULL", false }, + { "CREATE TABLE T (V TEXT(10), W LONG);CREATE INDEX ix ON T (W) WITH IGNORE NULL;INSERT INTO T (V) VALUES ('a')", true }, + { "CREATE TABLE T (Id COUNTER CONSTRAINT pk PRIMARY KEY, V TEXT(10));INSERT INTO T (V) VALUES ('a');UPDATE T SET Id = 5", false }, + { "CREATE TABLE T (Id COUNTER CONSTRAINT pk PRIMARY KEY, V TEXT(10));INSERT INTO T (V) VALUES ('a');UPDATE T SET Id = Id", false }, + { "CREATE TABLE T (Id COUNTER, V TEXT(10));INSERT INTO T (V) VALUES ('a');UPDATE T SET Id = Id", false }, + { "CREATE TABLE T (Id COUNTER, V TEXT(10));INSERT INTO T (V) VALUES ('a');UPDATE T SET Id = NULL", false }, + { "CREATE TABLE T (Id COUNTER CONSTRAINT pk PRIMARY KEY, V TEXT(10));INSERT INTO T (Id, V) VALUES (50, 'a')", true }, + { "CREATE TABLE T (Id COUNTER CONSTRAINT pk PRIMARY KEY, V TEXT(10));INSERT INTO T (Id, V) VALUES (NULL, 'a')", false }, + }; + + [Theory] + [MemberData(nameof(KeyWrites))] + public void Key_and_autonumber_writes_match_ace(string statements, bool accepted) + => AssertSameOutcome(statements, "T", insertColumn: "V", accepted); + + // An unnamed UNIQUE or CHECK gets a generated name, and a table name can use all 64 characters a name may + // have — so the generated name must still fit, or a table ACE creates is refused. + [Fact] + public void Unnamed_constraints_on_a_64_character_table_match_ace() + { + string table = new('T', 64); + AssertSameOutcome($"CREATE TABLE [{table}] (V TEXT(10) UNIQUE, W LONG)", table, insertColumn: null, accepted: true); + AssertSameOutcome($"CREATE TABLE [{table}] (V TEXT(10), W LONG, UNIQUE (V), CHECK (W > 0))", table, insertColumn: null, accepted: true); + } + [Fact] public void Identity_on_alter_table_matches_ace() { @@ -244,7 +279,7 @@ private static (string Description, string? Error) Run(string[] sql, string tabl string? counter; using (var db = JetDatabase.Open(path)) { - TableDef t = db.Catalog.FindTable(table)!; + TableDefinition t = db.Catalog.FindTable(table)!; description.AddRange(t.Columns.Select(c => $"{c.Name} {c.Type} autonumber={c.IsAutoNumber} required={!c.IsNullable}")); description.AddRange(db.Catalog.ForeignKeysOf(table).Select(fk => @@ -277,4 +312,4 @@ private static (string Description, string? Error) Run(string[] sql, string tabl return (string.Join("; ", description), null); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/ReferentialActionAccessTests.cs b/test/LibRed.Engine.AccessTests/ReferentialActionAccessTests.cs new file mode 100644 index 000000000..e536d5b57 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/ReferentialActionAccessTests.cs @@ -0,0 +1,111 @@ +using System.Data.OleDb; +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +// The ACE half of ReferentialActionTests: the two cascade shapes LibRed cannot decide on its own — a table +// whose parent and child ends are the SAME table, so a cascaded row is also a row the statement itself is +// updating, and a cascade that has to carry on into a grandchild. The same statements run through ACE and +// through LibRed on copies of one file, and the rows are read back through ACE either way, so a row that +// disagrees with its index shows up as a seek that misses. +[Collection(AceCollection.Name)] +public class ReferentialActionAccessTests(ITestOutputHelper output) : TempDatabaseTest +{ + private static string Northwind => Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); + + private static readonly string[] SelfReference = + [ + "CREATE TABLE Emp (Id LONG CONSTRAINT pk PRIMARY KEY, MgrId LONG, " + + "CONSTRAINT fk FOREIGN KEY (MgrId) REFERENCES Emp (Id) ON UPDATE CASCADE)", + "INSERT INTO Emp (Id, MgrId) VALUES (1, NULL)", + "INSERT INTO Emp (Id, MgrId) VALUES (2, 1)", + "UPDATE Emp SET Id = Id + 100", + ]; + + private static readonly string[] TwoLevels = + [ + "CREATE TABLE P (Id LONG CONSTRAINT pkP PRIMARY KEY)", + "CREATE TABLE C (Id LONG CONSTRAINT pkC PRIMARY KEY, PId LONG, " + + "CONSTRAINT fkC FOREIGN KEY (PId) REFERENCES P (Id) ON UPDATE CASCADE)", + "CREATE TABLE G (Id LONG CONSTRAINT pkG PRIMARY KEY, CId LONG, " + + "CONSTRAINT fkG FOREIGN KEY (CId) REFERENCES C (Id) ON UPDATE CASCADE)", + "INSERT INTO P (Id) VALUES (1)", + "INSERT INTO C (Id, PId) VALUES (10, 1)", + "INSERT INTO G (Id, CId) VALUES (100, 10)", + "UPDATE P SET Id = 5", + ]; + + [Fact] + public void A_self_referencing_cascade_update_leaves_what_ace_leaves() + { + string ace = Apply(SelfReference, throughAce: true); + string libred = Apply(SelfReference, throughAce: false); + + // Read back through ACE: the rows, then the FK index, which is where a row rewritten from the pre-cascade + // snapshot parts company with the entry the cascade already moved. + string aceRows = Read(ace, "SELECT Id, MgrId FROM Emp ORDER BY Id"); + string libredRows = Read(libred, "SELECT Id, MgrId FROM Emp ORDER BY Id"); + string aceSeek = Read(ace, "SELECT Id FROM Emp WHERE MgrId = 101"); + string libredSeek = Read(libred, "SELECT Id FROM Emp WHERE MgrId = 101"); + output.WriteLine($"ACE rows [{aceRows}] seek MgrId=101 [{aceSeek}]"); + output.WriteLine($"LibRed rows [{libredRows}] seek MgrId=101 [{libredSeek}]"); + + Assert.Equal("101,null | 102,101", aceRows); + Assert.Equal(aceRows, libredRows); + Assert.Equal(aceSeek, libredSeek); + } + + [Fact] + public void A_cascade_update_reaches_a_grandchild_as_it_does_in_ace() + { + string ace = Apply(TwoLevels, throughAce: true); + string libred = Apply(TwoLevels, throughAce: false); + + string aceRows = Read(ace, "SELECT P.Id, C.Id, C.PId, G.CId FROM (P INNER JOIN C ON P.Id = C.PId) INNER JOIN G ON C.Id = G.CId"); + string libredRows = Read(libred, "SELECT P.Id, C.Id, C.PId, G.CId FROM (P INNER JOIN C ON P.Id = C.PId) INNER JOIN G ON C.Id = G.CId"); + output.WriteLine($"ACE [{aceRows}]"); + output.WriteLine($"LibRed [{libredRows}]"); + + Assert.Equal("5,10,5,10", aceRows); + Assert.Equal(aceRows, libredRows); + } + + /// Runs the statements on a fresh copy, through ACE or through LibRed, and returns the copy's path. + private static string Apply(string[] statements, bool throughAce) + { + string path = TemporaryDatabase.CopyPath(Northwind, throughAce ? "refaction-ace-" : "refaction-lib-"); + if (throughAce) + { + using OleDbConnection connection = AceTestDatabase.Open(path); + foreach (string statement in statements) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statement; + command.ExecuteNonQuery(); + } + } + else + { + using var db = JetDatabase.Open(path, readOnly: false); + var engine = new QueryEngine(db); + foreach (string statement in statements) engine.ExecuteNonQuery(statement); + } + return path; + } + + /// The rows ACE reads for a query, as a,b | c,d. + private static string Read(string path, string sql) + { + using OleDbConnection connection = AceTestDatabase.Open(path); + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = sql; + using OleDbDataReader reader = command.ExecuteReader(); + var rows = new List(); + while (reader.Read()) + rows.Add(string.Join(",", Enumerable.Range(0, reader.FieldCount) + .Select(i => reader.IsDBNull(i) ? "null" : Convert.ToString(reader.GetValue(i))))); + return string.Join(" | ", rows); + } +} diff --git a/test/LibRed.Engine.AccessTests/RelationTypeAccessTests.cs b/test/LibRed.Engine.AccessTests/RelationTypeAccessTests.cs new file mode 100644 index 000000000..f8615c5eb --- /dev/null +++ b/test/LibRed.Engine.AccessTests/RelationTypeAccessTests.cs @@ -0,0 +1,62 @@ +using System.Reflection; +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +// A one-to-one relationship is the grbit bit 0x01, which only Access's dialog and DAO set (SQL cannot). LibRed +// reports it as DAO does — RELATION_TYPE "ONE" — whether or not the relationship is enforced; an unenforced one +// has no child index for anything to be inferred from. +[Collection(AceCollection.Name)] +public class RelationTypeAccessTests +{ + private const int UseJet = 2, Ace12 = 128, OneToOne = 1, DontEnforce = 2; + + [Fact] + public void A_one_to_one_created_by_dao_reads_as_one_enforced_or_not() + { + object? engine = AceTestDatabase.CreateDaoEngine(); + Assert.SkipWhen(engine is null, "DAO is not available in this process."); + object workspace = Invoke(engine!, "CreateWorkspace", "", "admin", "", UseJet)!; + + string path = TemporaryDatabase.CreatePath("reltype-", ".accdb"); + try + { + object database = Invoke(workspace, "CreateDatabase", path, ";LANGID=0x0409;CP=1252;COUNTRY=0", Ace12)!; + foreach (string sql in (string[]) + ["CREATE TABLE P (ID LONG PRIMARY KEY)", "CREATE TABLE E (ID LONG PRIMARY KEY)", "CREATE TABLE U (ID LONG PRIMARY KEY)"]) + Invoke(database, "Execute", sql, 128); + Relate(database, "enforced_one", "E", OneToOne); + Relate(database, "unenforced_one", "U", OneToOne | DontEnforce); + Invoke(database, "Close"); + + using var db = JetDatabase.Open(path, readOnly: true); + Assert.True(db.Catalog.Relationships.Single(r => r.Name == "enforced_one").IsOneToOne); + Assert.True(db.Catalog.Relationships.Single(r => r.Name == "unenforced_one").IsOneToOne); + + var types = new QueryEngine(db) + .ExecuteQuery("SELECT `RELATION_NAME`, `RELATION_TYPE` FROM `INFORMATION_SCHEMA.RELATIONS`") + .Rows.ToDictionary(r => (string)r[0]!, r => (string)r[1]!); + Assert.Equal("ONE", types["enforced_one"]); + Assert.Equal("ONE", types["unenforced_one"]); + } + finally { TemporaryDatabase.Delete(path); } + } + + // DAO: CreateRelation(name, primary table, foreign table, attributes); Field.Name = primary field, + // Field.ForeignName = foreign field. + private static void Relate(object database, string name, string child, int attributes) + { + object relation = Invoke(database, "CreateRelation", name, "P", child, attributes)!; + object field = Invoke(relation, "CreateField", "ID")!; + field.GetType().InvokeMember("ForeignName", BindingFlags.SetProperty, null, field, ["ID"]); + object fields = relation.GetType().InvokeMember("Fields", BindingFlags.GetProperty, null, relation, null)!; + Invoke(fields, "Append", field); + object relations = database.GetType().InvokeMember("Relations", BindingFlags.GetProperty, null, database, null)!; + Invoke(relations, "Append", relation); + } + + private static object? Invoke(object target, string member, params object?[] args) => + target.GetType().InvokeMember(member, BindingFlags.InvokeMethod, null, target, args); +} diff --git a/test/LibRed.Engine.AccessTests/RelationshipByteParityAccessTests.cs b/test/LibRed.Engine.AccessTests/RelationshipByteParityAccessTests.cs index 016aa2284..41093c71b 100644 --- a/test/LibRed.Engine.AccessTests/RelationshipByteParityAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/RelationshipByteParityAccessTests.cs @@ -1,7 +1,7 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Catalog; using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Engine.Tests; @@ -82,7 +82,7 @@ private static void LibRedRun(string path, string[] statements) try { using var database = JetDatabase.Open(path, readOnly: true); - TableDef rel = database.Catalog.FindTable("MSysRelationships")!; + TableDefinition rel = database.Catalog.FindTable("MSysRelationships")!; int Col(string name) => rel.Columns.Single(c => c.Name == name).Index; int szObject = Col("szObject"), szReferenced = Col("szReferencedObject"); int szColumn = Col("szColumn"), szReferencedColumn = Col("szReferencedColumn"); @@ -112,16 +112,13 @@ private static (byte[] Bytes, string IndexNames)? Definition( string names; using (var database = JetDatabase.Open(path, readOnly: true)) { - TableDef def = database.Catalog.FindTable(table)!; + TableDefinition def = database.Catalog.FindTable(table)!; definitionPage = def.DefinitionPage; names = string.Join(", ", def.Indexes.Select(i => i.Name)); } using var channel = PageChannel.Open(path, readOnly: true); - byte[] page = channel.ReadPage(definitionPage).Span.ToArray(); - int length = BinaryPrimitives.ReadInt32LittleEndian( - page.AsSpan(channel.Format.TdefLengthOffset, 4)); - return (length > 0 && length <= page.Length ? page.AsSpan(0, length).ToArray() : page, names); + return (TableDefinition.ReadChain(channel, definitionPage).Buffer.Span.ToArray(), names); } finally { TemporaryDatabase.Delete(path); } } @@ -135,4 +132,4 @@ private static string Build(string childSql, Action create) catch (OleDbException) { TemporaryDatabase.Delete(path); return null!; } return path; } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/RelocatedRowDeleteAccessTests.cs b/test/LibRed.Engine.AccessTests/RelocatedRowDeleteAccessTests.cs index cd9b776ff..d9b518170 100644 --- a/test/LibRed.Engine.AccessTests/RelocatedRowDeleteAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/RelocatedRowDeleteAccessTests.cs @@ -1,7 +1,7 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Formats; using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Engine.Tests; @@ -114,18 +114,18 @@ private static string Describe(string[] statements, Action run for (int page = 1; page < channel.PageCount; page++) { byte[] bytes = channel.ReadPage(page).Span.ToArray(); - if (bytes[0] != 0x01) continue; - if (BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(4, 4)) != definitionPage) continue; + if (PageHeader.ReadType(bytes) != PageType.DataPage) continue; + if ((int)DataPage.ReadOwner(bytes, format) != definitionPage) continue; - int slots = BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataRowCountOffset, 2)); + int slots = DataPage.ReadRowCount(bytes, format); var entries = Enumerable.Range(0, slots) - .Select(i => BinaryPrimitives.ReadUInt16LittleEndian( - bytes.AsSpan(format.DataRowDirectoryOffset + i * 2, 2)).ToString("X4")); - pages.Add($"free={BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataFreeSpaceOffset, 2))}" + .Select(i => DataPage.ReadSlot(bytes, format, i)) + .Select(s => ((int)s.Flags | s.Offset).ToString("X4")); + pages.Add($"free={DataPage.ReadFreeSpace(bytes, format)}" + $" [{string.Join(" ", entries)}]"); } return string.Join(" | ", pages); } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/RelocatedRowGrowthAccessTests.cs b/test/LibRed.Engine.AccessTests/RelocatedRowGrowthAccessTests.cs new file mode 100644 index 000000000..e02ff92a9 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/RelocatedRowGrowthAccessTests.cs @@ -0,0 +1,130 @@ +using System.Data.OleDb; +using LibRed.Formats; +using LibRed.IO; +using LibRed.Pages; +using Xunit; + +namespace LibRed.Engine.Tests; + +// A row that has already been relocated and then grows AGAIN, past the page it was moved to. The first move +// leaves a 4-byte forward pointer in the original slot and the row itself hidden on another page; the second +// has nothing to rewrite in place, so the row has to move a second time and the pointer follow it. ACE is the +// oracle for what that leaves on disk, and for whether it happens at all rather than the row being packed +// differently. +[Collection(AceCollection.Name)] +public class RelocatedRowGrowthAccessTests(ITestOutputHelper output) : TempDatabaseTest +{ + // Wide rows fill the first page; row 4 is then grown in steps. The first step moves it, and the later ones + // have to keep working on a page that other rows have since filled up. + private static string[] Statements() + { + var s = new List + { + "CREATE TABLE W (A LONG, B TEXT(255), C TEXT(255), CONSTRAINT pk PRIMARY KEY (A))", + }; + // Narrow rows fill the first page, exactly as the relocation test does. + for (int i = 1; i <= 18; i++) + s.Add($"INSERT INTO W (A, B) VALUES ({i}, '{Text('a', i, 100)}')"); + + // Widening one by more than the page's free space moves it to a page of its own. + s.Add($"UPDATE W SET B = '{Text('z', 4, 255)}' WHERE A = 4"); + + // Fill that page up behind it, so the row has nowhere left to grow where it now lies. + for (int i = 19; i <= 30; i++) + s.Add($"INSERT INTO W (A, B, C) VALUES ({i}, '{Text('d', i, 255)}', '{Text('e', i, 255)}')"); + + // And grow it again: the second column is 510 more bytes with nothing to rewrite in place. + s.Add($"UPDATE W SET C = '{Text('y', 4, 255)}' WHERE A = 4"); + return [.. s]; + } + + private static string Text(char seed, int row, int length) => new((char)(seed + row % 5), length); + + [Fact] + public void Growing_an_already_relocated_row_past_its_new_page_matches_ace() + { + string ace = Describe(AceRun); + string libred = Describe(LibRedRun); + output.WriteLine($"ACE: {ace}"); + output.WriteLine($"LibRed: {libred}"); + Assert.Equal(ace, libred); + } + + // And the row still reads back through ACE, whichever page it ended up on. + [Fact] + public void Ace_reads_the_twice_moved_row_back() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "relocgrow-read-"); + try + { + LibRedRun(path, Statements()); + + using OleDbConnection connection = AceTestDatabase.Open(path); + using (OleDbCommand count = connection.CreateCommand()) + { + count.CommandText = "SELECT COUNT(*) FROM W"; + Assert.Equal(30, Convert.ToInt32(count.ExecuteScalar())); + } + using OleDbCommand read = connection.CreateCommand(); + read.CommandText = "SELECT B, C FROM W WHERE A = 4"; + using OleDbDataReader reader = read.ExecuteReader(); + Assert.True(reader.Read()); + Assert.Equal(Text('z', 4, 255), reader.GetString(0)); + Assert.Equal(Text('y', 4, 255), reader.GetString(1)); + } + finally { TemporaryDatabase.Delete(path); } + } + + private static void AceRun(string path, string[] statements) + { + using OleDbConnection connection = AceTestDatabase.Open(path); + foreach (string s in statements) + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = s; + command.ExecuteNonQuery(); + } + } + + private static void LibRedRun(string path, string[] statements) + { + using var database = JetDatabase.Open(path, readOnly: false); + var engine = new QueryEngine(database); + foreach (string s in statements) engine.ExecuteNonQuery(s); + } + + /// Every data page the table owns: free space and slot directory, in page order. + private static string Describe(Action run) + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "relocgrow-"); + try + { + run(path, Statements()); + + int definitionPage; + using (var database = JetDatabase.Open(path, readOnly: true)) + definitionPage = database.Catalog.FindTable("W")!.DefinitionPage; + + using var channel = PageChannel.Open(path, readOnly: true); + JetFormatBase format = channel.Format; + var pages = new List(); + for (int page = 1; page < channel.PageCount; page++) + { + byte[] bytes = channel.ReadPage(page).Span.ToArray(); + if (PageHeader.ReadType(bytes) != PageType.DataPage) continue; + if ((int)DataPage.ReadOwner(bytes, format) != definitionPage) continue; + + int slots = DataPage.ReadRowCount(bytes, format); + var entries = Enumerable.Range(0, slots) + .Select(i => DataPage.ReadSlot(bytes, format, i)) + .Select(s => ((int)s.Flags | s.Offset).ToString("X4")); + pages.Add($"free={DataPage.ReadFreeSpace(bytes, format)}" + + $" [{string.Join(" ", entries)}]"); + } + return string.Join(" | ", pages); + } + finally { TemporaryDatabase.Delete(path); } + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/RowByteParityAccessTests.cs b/test/LibRed.Engine.AccessTests/RowByteParityAccessTests.cs index 6770b2bfe..b74736e1e 100644 --- a/test/LibRed.Engine.AccessTests/RowByteParityAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/RowByteParityAccessTests.cs @@ -1,7 +1,7 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Formats; using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Engine.Tests; @@ -36,20 +36,91 @@ public class RowByteParityAccessTests(ITestOutputHelper output) : TempDatabaseTe "INSERT INTO W (A, M) VALUES (1, 'xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx')" }, { "CREATE TABLE W (A LONG, M LONGTEXT)", "INSERT INTO W (A, M) VALUES (1, '中中中中中中中中中中中中中中中中中中中中中中中中中中中中中中')" }, + // A BigBinary value stays inline at any size, where VARBINARY stops at 510. + { "CREATE TABLE W (A LONG, B BIGBINARY)", "INSERT INTO W (A, B) VALUES (1, 0x0102030405)" }, + { "CREATE TABLE W (A LONG, B BIGBINARY)", $"INSERT INTO W (A, B) VALUES (1, 0x{string.Concat(Enumerable.Repeat("0A1B2C", 1000))})" }, }; [Theory] [MemberData(nameof(Shapes))] public void The_row_bytes_match_ace(string ddl, string insert) { - byte[]? ace = Row(ddl, insert, AceRun); + byte[]? ace = Row([ddl, insert], AceRun); Assert.SkipWhen(ace is null, "ACE wrote no row for this shape."); - byte[] libred = Row(ddl, insert, LibRedRun)!; + byte[] libred = Row([ddl, insert], LibRedRun)!; output.WriteLine(Convert.ToHexString(ace!)); Assert.Equal(Convert.ToHexString(ace!), Convert.ToHexString(libred)); } + // A row written after the table's variable columns have been dropped. The rows already on the page still + // carry the variable trailer those columns needed, while the schema no longer says they are variable — so + // a writer that measures its fixed-region length off an existing row measures that trailer as fixed data + // and pads every later row to it. Both end states are asked: no variable column left at all, and one + // added back afterwards. + public static TheoryData DroppedVariableColumns => new() + { + { "none left", + [ + "CREATE TABLE W (A LONG, T TEXT(20), U TEXT(20))", + "INSERT INTO W (A, T, U) VALUES (1, 'aa', 'bbbb')", + "ALTER TABLE W DROP COLUMN T", + "ALTER TABLE W DROP COLUMN U", + "INSERT INTO W (A) VALUES (2)", + ] + }, + // The two single-drop shapes the colCount rule turns on: dropping the highest-id column when it is + // fixed, and when it is variable. + { "highest-id fixed column dropped", + [ + "CREATE TABLE W (A LONG, B LONG)", + "INSERT INTO W (A, B) VALUES (1, 2)", + "ALTER TABLE W DROP COLUMN B", + "INSERT INTO W (A) VALUES (2)", + ] + }, + { "highest-id variable column dropped", + [ + "CREATE TABLE W (A LONG, T TEXT(20))", + "INSERT INTO W (A, T) VALUES (1, 'aa')", + "ALTER TABLE W DROP COLUMN T", + "INSERT INTO W (A) VALUES (2)", + ] + }, + // A type-change ALTER burns the old column id the same way a drop does. The row the ALTER itself + // re-lays carries the dead id's bit SET (§5); this asks what a row INSERTED afterwards carries. + { "an id burned by a retype", + [ + "CREATE TABLE W (A LONG, B LONG, C LONG)", + "INSERT INTO W (A, B, C) VALUES (1, 2, 3)", + "ALTER TABLE W ALTER COLUMN B DOUBLE", + "INSERT INTO W (A, B, C) VALUES (4, 5, 6)", + ] + }, + { "one added back", + [ + "CREATE TABLE W (A LONG, T TEXT(20), U TEXT(20))", + "INSERT INTO W (A, T, U) VALUES (1, 'aa', 'bbbb')", + "ALTER TABLE W DROP COLUMN T", + "ALTER TABLE W DROP COLUMN U", + "ALTER TABLE W ADD COLUMN V TEXT(20)", + "INSERT INTO W (A, V) VALUES (2, 'cc')", + ] + }, + }; + + [Theory] + [MemberData(nameof(DroppedVariableColumns))] + public void A_row_written_after_the_variable_columns_were_dropped_matches_ace(string label, string[] statements) + { + byte[]? ace = Row(statements, AceRun, row: 1); + Assert.SkipWhen(ace is null, "ACE wrote no row for this shape."); + + byte[] libred = Row(statements, LibRedRun, row: 1)!; + output.WriteLine($"{label}: ace={Convert.ToHexString(ace!)} libred={Convert.ToHexString(libred)}"); + Assert.Equal(Convert.ToHexString(ace!), Convert.ToHexString(libred)); + } + private static void AceRun(string path, string[] statements) { using OleDbConnection connection = AceTestDatabase.Open(path); @@ -68,15 +139,15 @@ private static void LibRedRun(string path, string[] statements) foreach (string s in statements) engine.ExecuteNonQuery(s); } - /// The bytes of the single row on table W's data page — the page the table owns, found by its - /// owner stamp rather than by walking the usage map. - private static byte[]? Row(string ddl, string insert, Action run) + /// The bytes of row on table W's data page — the page the table owns, + /// found by its owner stamp rather than by walking the usage map. + private static byte[]? Row(string[] statements, Action run, int row = 0) { string path = TemporaryDatabase.CopyPath( Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "rowbytes-"); try { - try { run(path, [ddl, insert]); } + try { run(path, statements); } catch (OleDbException) { return null; } int definitionPage; @@ -88,17 +159,19 @@ private static void LibRedRun(string path, string[] statements) for (int page = 1; page < channel.PageCount; page++) { byte[] bytes = channel.ReadPage(page).Span.ToArray(); - if (bytes[0] != 0x01) continue; - if (BinaryPrimitives.ReadInt32LittleEndian(bytes.AsSpan(4, 4)) != definitionPage) continue; - if (BinaryPrimitives.ReadUInt16LittleEndian(bytes.AsSpan(format.DataRowCountOffset, 2)) == 0) + if (PageHeader.ReadType(bytes) != PageType.DataPage) continue; + if ((int)DataPage.ReadOwner(bytes, format) != definitionPage) continue; + if (DataPage.ReadRowCount(bytes, format) <= row) continue; - int start = BinaryPrimitives.ReadUInt16LittleEndian( - bytes.AsSpan(format.DataRowDirectoryOffset, 2)) & 0x1FFF; - return bytes.AsSpan(start, format.PageSize - start).ToArray(); + // Slot offsets are non-increasing, so a row runs from its own offset to the previous slot's + // (the page end for row 0). + int start = DataPage.ReadSlot(bytes, format, row).Offset; + int end = row == 0 ? format.PageSize : DataPage.ReadSlot(bytes, format, row - 1).Offset; + return bytes.AsSpan(start, end - start).ToArray(); } return null; } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/SelectIntoAccessTests.cs b/test/LibRed.Engine.AccessTests/SelectIntoAccessTests.cs index dae9e98a1..e6ebb051b 100644 --- a/test/LibRed.Engine.AccessTests/SelectIntoAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/SelectIntoAccessTests.cs @@ -76,7 +76,7 @@ private static void AssertSameAsAce(string makeTable, string verify) private static string Describe(JetDatabase db, string table) { - TableDef? def = db.Catalog.Tables.FirstOrDefault(t => t.Name == table); + TableDefinition? def = db.Catalog.Tables.FirstOrDefault(t => t.Name == table); if (def is null) return "(not created)"; return string.Join(", ", def.Columns.Select(c => $"{c.Name} {c.Type}({c.Length})")) + " | indexes: " + (def.Indexes.Count == 0 ? "(none)" : string.Join(", ", def.Indexes.Select(i => i.Name))); @@ -106,4 +106,4 @@ private static void Exec(OleDbConnection connection, string sql) command.CommandText = sql; command.ExecuteNonQuery(); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/SelectIntoShapeProbeTest.cs b/test/LibRed.Engine.AccessTests/SelectIntoShapeProbeTest.cs index c54344714..5dc83afa0 100644 --- a/test/LibRed.Engine.AccessTests/SelectIntoShapeProbeTest.cs +++ b/test/LibRed.Engine.AccessTests/SelectIntoShapeProbeTest.cs @@ -84,7 +84,7 @@ public void Probe_make_table_shape() using var db = JetDatabase.Open(path); foreach (string table in (string[])["SiSrc", "SiA", "SiB", "SiC", "SiD", "SiE", "SiF"]) { - TableDef? def = db.Catalog.Tables.FirstOrDefault(t => t.Name == table); + TableDefinition? def = db.Catalog.Tables.FirstOrDefault(t => t.Name == table); if (def is null) { output.WriteLine($" {table}: not created"); continue; } output.WriteLine($" {table}: " + string.Join(", ", def.Columns.Select(c => @@ -106,4 +106,4 @@ private static void Exec(OleDbConnection connection, string sql) command.CommandText = sql; command.ExecuteNonQuery(); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/StoredActionQueryWriteAccessTests.cs b/test/LibRed.Engine.AccessTests/StoredActionQueryWriteAccessTests.cs index ce27cca51..f5e4c5426 100644 --- a/test/LibRed.Engine.AccessTests/StoredActionQueryWriteAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/StoredActionQueryWriteAccessTests.cs @@ -25,6 +25,9 @@ private static string Copy() => TemporaryDatabase.CopyPath( [InlineData("DELETE FROM Shippers WHERE ShipperID > 900")] [InlineData("DELETE Shippers.* FROM Shippers WHERE ShipperID > 900")] [InlineData("SELECT ShipperID, CompanyName INTO ShipperCopy FROM Shippers WHERE ShipperID > 1")] + [InlineData("SELECT DISTINCT Country INTO CountryCopy FROM Customers")] + [InlineData("SELECT TOP 5 CompanyName INTO CustomerCopy FROM Customers")] + [InlineData("INSERT INTO Shippers (CompanyName) SELECT DISTINCT Country FROM Customers")] [InlineData("INSERT INTO Shippers (CompanyName, Phone) SELECT CompanyName, Phone FROM Customers WHERE Country = 'UK'")] public void A_libred_written_action_query_stores_the_rows_ace_stores(string body) { @@ -175,6 +178,114 @@ public void A_from_less_body_is_stored_as_ace_stores_it(string sql) } } + /// WITH OWNERACCESS OPTION is one more option row in a stored query, which ACE writes for a view and for + /// every kind of action query alike. LibRed acts on nothing it says, but keeps it: the rows are ACE's, ACE runs the + /// query LibRed stored, and LibRed reads the declaration back and runs the query itself. + [Theory] + [InlineData("CREATE VIEW [Q] AS SELECT CompanyName FROM Shippers WITH OWNERACCESS OPTION")] + // ACE's CREATE VIEW takes no ORDER BY ("Only simple SELECT queries are allowed in VIEWS"); a procedure does. + [InlineData("CREATE PROCEDURE [Q] AS SELECT DISTINCT CompanyName FROM Shippers ORDER BY CompanyName WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS SELECT TOP 2 CompanyName FROM Shippers ORDER BY CompanyName WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS SELECT DISTINCT TOP 2 CompanyName FROM Shippers ORDER BY CompanyName WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS SELECT DISTINCT TOP 50 PERCENT CompanyName FROM Shippers ORDER BY CompanyName " + + "WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS SELECT CompanyName FROM Shippers WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS UPDATE Customers SET ContactTitle = 'Owner' WHERE Country = 'UK' WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS DELETE FROM Shippers WHERE ShipperID > 900 WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS SELECT ShipperID, CompanyName INTO ShipperCopy FROM Shippers WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS INSERT INTO Shippers (CompanyName, Phone) SELECT CompanyName, Phone FROM Customers " + + "WHERE Country = 'UK' WITH OWNERACCESS OPTION")] + [InlineData("CREATE PROCEDURE [Q] AS INSERT INTO Shippers (CompanyName) VALUES ('Owner') WITH OWNERACCESS OPTION")] + public void An_owneraccess_query_is_stored_as_ace_stores_it(string sql) + { + string ourPath = Copy(), acePath = Copy(), readPath = Copy(); + bool action = !sql.Contains(" AS SELECT CompanyName", StringComparison.Ordinal) + && !sql.Contains(" AS SELECT DISTINCT", StringComparison.Ordinal) && !sql.Contains(" AS SELECT TOP", StringComparison.Ordinal); + try + { + using (var db = TemporaryDatabase.OpenTracked(ourPath, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery(sql); + + using (var connection = AceTestDatabase.Open(acePath)) + { + using var create = connection.CreateCommand(); + create.CommandText = sql; + create.ExecuteNonQuery(); + } + + Assert.Equal(QueryRows(acePath, "Q"), QueryRows(ourPath, "Q")); + Assert.Equal(ObjectFlags(acePath, "Q"), ObjectFlags(ourPath, "Q")); + + // ACE runs the query LibRed stored — on a copy, so an action query leaves the file LibRed reads next alone. + File.Copy(ourPath, readPath, overwrite: true); + using (var connection = AceTestDatabase.Open(readPath)) + { + using var run = connection.CreateCommand(); + run.CommandText = "Q"; + run.CommandType = CommandType.StoredProcedure; + run.ExecuteNonQuery(); + } + + // LibRed reads the declaration back with the query, and runs it. + using var ours = TemporaryDatabase.OpenTracked(ourPath, readOnly: false); + StoredQuery query = ours.Catalog.FindQuery("Q")!; + Assert.Equal(action, query.IsAction); + string? stored = query.Sql; + Assert.EndsWith(" WITH OWNERACCESS OPTION", stored); + var engine = new QueryEngine(ours); + if (action) engine.ExecuteStoredActionQuery("Q"); + else Assert.NotEmpty(engine.ExecuteQuery("SELECT * FROM [Q]").Rows); + } + finally + { + TemporaryDatabase.Delete(ourPath); + TemporaryDatabase.Delete(acePath); + TemporaryDatabase.Delete(readPath); + } + } + + /// A stored SELECT's DISTINCT, TOP and PERCENT are bits of one option row, as ACE writes them — and PERCENT + /// survives the round trip: ACE and LibRed return the same rows from the query LibRed stored. + [Theory] + [InlineData("CREATE PROCEDURE [Q] AS SELECT DISTINCT TOP 2 Country FROM Customers ORDER BY Country")] + [InlineData("CREATE PROCEDURE [Q] AS SELECT TOP 10 PERCENT CompanyName FROM Customers ORDER BY CompanyName")] + [InlineData("CREATE PROCEDURE [Q] AS SELECT DISTINCT TOP 25 PERCENT Country FROM Customers ORDER BY Country")] + public void A_queries_options_share_one_row_as_ace_writes_them(string sql) + { + string ourPath = Copy(), acePath = Copy(); + try + { + using (var db = TemporaryDatabase.OpenTracked(ourPath, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery(sql); + + using (var connection = AceTestDatabase.Open(acePath)) + { + using var create = connection.CreateCommand(); + create.CommandText = sql; + create.ExecuteNonQuery(); + } + + Assert.Equal(QueryRows(acePath, "Q"), QueryRows(ourPath, "Q")); + + int aceRows = 0; + using (var connection = AceTestDatabase.Open(ourPath)) + { + using var run = connection.CreateCommand(); + run.CommandText = "Q"; + run.CommandType = CommandType.StoredProcedure; + using var reader = run.ExecuteReader(); + while (reader.Read()) aceRows++; + } + using var ours = TemporaryDatabase.OpenTracked(ourPath, readOnly: true); + Assert.Equal(aceRows, new QueryEngine(ours).ExecuteQuery("SELECT * FROM [Q]").Rows.Count()); + } + finally + { + TemporaryDatabase.Delete(ourPath); + TemporaryDatabase.Delete(acePath); + } + } + [Fact] public void A_written_action_query_reads_back_as_the_statement_it_was_written_from() { @@ -187,7 +298,7 @@ public void A_written_action_query_reads_back_as_the_statement_it_was_written_fr "CREATE PROCEDURE [P] AS UPDATE Customers SET ContactTitle = 'Owner' WHERE Country = 'UK'"); // Round trip: what was written is read back as runnable SQL, and running it by name works. - StoredActionQuery stored = db.Catalog.ActionQueries["P"]; + StoredQuery stored = db.Catalog.FindQuery("P")!; Assert.Null(stored.UnsupportedReason); Assert.Equal("UPDATE [Customers] SET [ContactTitle] = 'Owner' WHERE Country = 'UK'", stored.Sql); Assert.Equal(7, engine.ExecuteNonQuery("EXECUTE [P]")); @@ -207,7 +318,7 @@ public void A_written_action_query_reads_back_as_the_statement_it_was_written_fr private static List QueryRows(string path, string queryName) { using var db = JetDatabase.Open(path); - TableDef queries = db.Catalog.FindTable("MSysQueries")!; + TableDefinition queries = db.Catalog.FindTable("MSysQueries")!; int objectIdIndex = Index(queries, "ObjectId"); int attributeIndex = Index(queries, "Attribute"); int id = QueryObjectId(db, queryName); @@ -233,7 +344,7 @@ private static List QueryRows(string path, string queryName) private static int ObjectFlags(string path, string queryName) { using var db = JetDatabase.Open(path); - TableDef objects = db.Catalog.FindTable("MSysObjects")!; + TableDefinition objects = db.Catalog.FindTable("MSysObjects")!; int flagsIndex = Index(objects, "Flags"); int idIndex = Index(objects, "Id"); int id = QueryObjectId(db, queryName); @@ -242,11 +353,11 @@ private static int ObjectFlags(string path, string queryName) private static int QueryObjectId(JetDatabase db, string queryName) { - TableDef objects = db.Catalog.FindTable("MSysObjects")!; + TableDefinition objects = db.Catalog.FindTable("MSysObjects")!; int idIndex = Index(objects, "Id"), nameIndex = Index(objects, "Name"); return (int)db.OpenTable("MSysObjects").Rows() .Single(row => string.Equals(row[nameIndex] as string, queryName, StringComparison.OrdinalIgnoreCase))[idIndex]!; } - private static int Index(TableDef table, string column) => table.FindColumn(column)!.Index; -} + private static int Index(TableDefinition table, string column) => table.FindColumn(column)!.Index; +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/StoredParameterNameAccessTests.cs b/test/LibRed.Engine.AccessTests/StoredParameterNameAccessTests.cs new file mode 100644 index 000000000..82f8c3a9d --- /dev/null +++ b/test/LibRed.Engine.AccessTests/StoredParameterNameAccessTests.cs @@ -0,0 +1,74 @@ +using System.Data.OleDb; +using LibRed; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// A declared parameter of a stored query. ACE stores its name as declared — a bracketed one keeps its brackets, +/// a bare @name loses the @ — and LibRed must store the same bytes, and read either engine's back by +/// the name it binds to, rebuilding a query that still parses and runs. +/// +[Collection(AceCollection.Name)] +public class StoredParameterNameAccessTests +{ + private static string Create(string declared) => + $"CREATE PROCEDURE P ({declared} TEXT(50)) AS SELECT CustomerID FROM Customers WHERE ContactName = {declared}"; + + [Theory] + [InlineData("[@firstName]", "@firstName")] + [InlineData("[firstName]", "firstName")] + [InlineData("firstName", "firstName")] + [InlineData("[first name]", "first name")] + [InlineData("@firstName", "firstName")] + public void A_parameter_is_stored_as_ace_stores_it_and_reads_back_by_its_name(string declared, string name) + { + string northwind = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); + string acePath = TemporaryDatabase.CopyPath(northwind, "stored-param-ace-"); + string libredPath = TemporaryDatabase.CopyPath(northwind, "stored-param-libred-"); + try + { + using (OleDbConnection conn = AceTestDatabase.Open(acePath)) + { + using OleDbCommand cmd = conn.CreateCommand(); + cmd.CommandText = Create(declared); + cmd.ExecuteNonQuery(); + } + using (var db = JetDatabase.Open(libredPath, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery(Create(declared)); + + Assert.Equal(StoredName(acePath), StoredName(libredPath)); + + foreach (string path in new[] { acePath, libredPath }) + { + using var db = JetDatabase.Open(path); + Assert.Equal([name], db.Catalog.FindQuery("P")!.Parameters.Select(p => p.Name)); + var engine = new QueryEngine(db); + Assert.Equal(["ALFKI"], engine.ExecuteQuery("EXECUTE P 'Maria Anders'").Rows.Select(r => r[0])); + // As a table source, bound by the name reported for it — what a consumer reading the schema does. + Assert.Equal(["ALFKI"], engine.ExecuteQuery("SELECT * FROM [P]", + new Dictionary { [name] = "Maria Anders" }).Rows.Select(r => r[0])); + } + } + finally + { + TemporaryDatabase.Delete(acePath); + TemporaryDatabase.Delete(libredPath); + } + } + + /// The raw Name1 of query P's parameter row. + private static string? StoredName(string path) + { + using var db = JetDatabase.Open(path); + var objects = db.Catalog.FindTable("MSysObjects")!; + int id = objects.FindColumn("Id")!.Index, objectName = objects.FindColumn("Name")!.Index; + object procId = db.OpenTable("MSysObjects").Rows().Single(r => r[objectName] as string == "P")[id]!; + + var queries = db.Catalog.FindTable("MSysQueries")!; + int attr = queries.FindColumn("Attribute")!.Index, name1 = queries.FindColumn("Name1")!.Index, + owner = queries.FindColumn("ObjectId")!.Index; + return db.OpenTable("MSysQueries").Rows() + .Single(r => r[attr] is byte a && a == 2 && Equals(r[owner], procId))[name1] as string; + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/TdefByteParityAccessTests.cs b/test/LibRed.Engine.AccessTests/TdefByteParityAccessTests.cs index b2b9d39ec..e53c4b921 100644 --- a/test/LibRed.Engine.AccessTests/TdefByteParityAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/TdefByteParityAccessTests.cs @@ -1,8 +1,8 @@ -using System.Buffers.Binary; using System.Data.OleDb; using LibRed.Catalog; using LibRed.Formats; using LibRed.IO; +using LibRed.Pages; using Xunit; namespace LibRed.Engine.Tests; @@ -16,7 +16,7 @@ namespace LibRed.Engine.Tests; // with one measured exception recorded in Usage_map_rows_follow_declaration_order below. // // Constraints are named deliberately. An unnamed primary key makes ACE generate `Index_` from nothing -// reproducible while LibRed picks the stable "PrimaryKey" — an engine choice documented in TableCreator — +// reproducible while LibRed picks the stable "PrimaryKey" — an engine choice documented in SchemaEditor — // so an unnamed key would only ever measure that known difference. [Collection(AceCollection.Name)] public class TdefByteParityAccessTests(ITestOutputHelper output) : TempDatabaseTest @@ -30,6 +30,7 @@ public class TdefByteParityAccessTests(ITestOutputHelper output) : TempDatabaseT { "notnull", "CREATE TABLE W (Id LONG, A LONG NOT NULL, B TEXT(20) NOT NULL, CONSTRAINT pk PRIMARY KEY (Id))" }, { "counter", "CREATE TABLE W (Id COUNTER, A TEXT(30), CONSTRAINT pk PRIMARY KEY (Id))" }, { "unique", "CREATE TABLE W (Id LONG, A LONG, CONSTRAINT pk PRIMARY KEY (Id), CONSTRAINT u UNIQUE (A))" }, + { "self-reference", "CREATE TABLE W (Id LONG, P LONG, CONSTRAINT pk PRIMARY KEY (Id), CONSTRAINT fk FOREIGN KEY (P) REFERENCES W (Id))" }, { "guid+decimal", "CREATE TABLE W (Id LONG, G GUID, D DECIMAL(18,4), CONSTRAINT pk PRIMARY KEY (Id))" }, { "many-columns", "CREATE TABLE W (Id LONG, A BYTE, B SMALLINT, C REAL, D FLOAT, E CURRENCY, " + "F DATETIME, G BIT, H CHAR(10), I VARCHAR(40), J BINARY(8), CONSTRAINT pk PRIMARY KEY (Id))" }, @@ -45,12 +46,12 @@ public void The_whole_definition_matches_ace(string label, string sql) var aceDef = Definition(sql, AceCreate); Assert.SkipWhen(aceDef is null, $"ACE would not create {label}."); - (byte[] ace, JetFormatBase format) = aceDef!.Value; + (byte[] ace, JetFormatBase format, TableDefinition aceParsed) = aceDef!.Value; byte[] libred = Definition(sql, LibRedCreate)!.Value.Bytes; // An inline key leaves ACE naming the index unreproducibly; compare those shapes by every region // except the names, and the named-constraint shapes whole. - bool generatedName = IndexNames(ace, format).Any(n => n.StartsWith("Index_", StringComparison.Ordinal)); + bool generatedName = IndexNames(aceParsed).Any(n => n.StartsWith("Index_", StringComparison.Ordinal)); foreach ((string region, int start, int end) in Regions(ace, format)) { if (generatedName && region is "index-names" or "header") continue; @@ -60,29 +61,29 @@ public void The_whole_definition_matches_ace(string label, string sql) } if (!generatedName) Assert.Equal(ace.Length, libred.Length); - output.WriteLine($"{label}: {ace.Length} bytes, index names [{string.Join(", ", IndexNames(ace, format))}]"); + output.WriteLine($"{label}: {ace.Length} bytes, index names [{string.Join(", ", IndexNames(aceParsed))}]"); } - // The one measured divergence. ACE assigns usage-map rows in DECLARATION order: an inline PRIMARY KEY on - // the first column is created before the long-value columns and takes row 2, while a trailing CONSTRAINT - // clause is created after them and lands past their rows. LibRed cannot tell the two spellings apart — - // the position is lost between the parser and CreateTable — so it always uses the inline order. - // - // Both files are self-consistent and ACE reads either, so this is faithfulness, not corruption. Asserted - // rather than ignored so that closing the gap fails here and this note gets updated with it. + // Usage-map rows follow DECLARATION order: an inline PRIMARY KEY on the first column is declared before the + // long-value columns and takes row 2, while a trailing CONSTRAINT clause is declared after them and lands + // past their rows. The same table written the two ways lays its maps out differently, and both engines now + // agree on each. (This used to record the divergence, LibRed always using the inline order because the + // constraint's position was lost between the parser and CreateTable; it is carried through now, and the + // rule across inline, table-level, interleaved and foreign-key shapes is in + // CreateTableUsageMapOrderAccessTests.) [Fact] public void Usage_map_rows_follow_declaration_order() { const string inline = "CREATE TABLE W (Id LONG PRIMARY KEY, M LONGTEXT, N LONGTEXT)"; const string named = "CREATE TABLE W (Id LONG, M LONGTEXT, N LONGTEXT, CONSTRAINT pk PRIMARY KEY (Id))"; - // Inline: the index is declared first and gets row 2, the columns follow. Both engines agree. + // Inline: the index is declared first and gets row 2, the columns follow. Assert.Equal("index [2] long-value [3,4 5,6]", Rows(inline, AceCreate)); Assert.Equal("index [2] long-value [3,4 5,6]", Rows(inline, LibRedCreate)); - // Named: ACE creates the columns first, so they take rows 2..5 and the index lands on 6. + // Named: the columns are declared first, so they take rows 2..5 and the index lands on 6. Assert.Equal("index [6] long-value [2,3 4,5]", Rows(named, AceCreate)); - Assert.Equal("index [2] long-value [3,4 5,6]", Rows(named, LibRedCreate)); // the divergence + Assert.Equal("index [6] long-value [2,3 4,5]", Rows(named, LibRedCreate)); } private static void AceCreate(string path, string sql) @@ -102,77 +103,38 @@ private static void LibRedCreate(string path, string sql) /// Which usage-map row each index and long-value column was given. private static string Rows(string sql, Action create) { - (byte[] def, JetFormatBase format) = Definition(sql, create)!.Value; - int dataCount = BinaryPrimitives.ReadInt32LittleEndian(def.AsSpan(format.TdefIndexCountOffset, 4)); - (string _, int dataStart, int _) = Regions(def, format).Single(r => r.Name == "index-data-blocks"); - (string _, int lvalStart, int lvalEnd) = Regions(def, format).Single(r => r.Name == "long-value-region"); - - var indexes = Enumerable.Range(0, dataCount) - .Select(i => def[dataStart + i * DataBlockSize + UsageMapRowOffset].ToString()); - - var columns = new List(); - for (int pos = lvalStart; pos + 10 <= lvalEnd - && BinaryPrimitives.ReadUInt16LittleEndian(def.AsSpan(pos, 2)) != 0xFFFF; pos += 10) - columns.Add($"{def[pos + 2]},{def[pos + 6]}"); - + TableDefinition definition = Definition(sql, create)!.Value.Parsed; + var indexes = definition.Indexes.OrderBy(i => i.RealIndexOrdinal).Select(i => i.UsageMap.Row); + var columns = definition.LongValueOwnedMaps.OrderBy(e => e.Key) + .Select(e => $"{e.Value.Row},{definition.LongValueFreeMaps[e.Key].Row}"); return $"index [{string.Join(",", indexes)}] long-value [{string.Join(" ", columns)}]"; } - // IndexBlockFormat is internal to Core; these are its DataBlockSize / InfoBlockSize / UsageMapRowOffset. - private const int DataBlockSize = 52; - private const int InfoBlockSize = 28; - private const int UsageMapRowOffset = 0x22; + private static List IndexNames(TableDefinition definition) => + [.. definition.LogicalIndexes.Select(l => l.Name)]; - private static List IndexNames(byte[] def, JetFormatBase format) - { - (string _, int start, int end) = Regions(def, format).Single(r => r.Name == "index-names"); - var names = new List(); - for (int pos = start; pos + 2 <= end;) - { - int len = BinaryPrimitives.ReadUInt16LittleEndian(def.AsSpan(pos, 2)); - if (len == 0 || pos + 2 + len > end) break; - names.Add(System.Text.Encoding.Unicode.GetString(def.AsSpan(pos + 2, len))); - pos += 2 + len; - } - return names; - } - - /// The TDEF's regions in order, derived from its own counts. + /// The TDEF's regions in order, through the walk the engine reads and writes it with. private static IEnumerable<(string Name, int Start, int End)> Regions(byte[] def, JetFormatBase format) { - int dataCount = BinaryPrimitives.ReadInt32LittleEndian(def.AsSpan(format.TdefIndexCountOffset, 4)); - int logicalCount = BinaryPrimitives.ReadInt32LittleEndian( - def.AsSpan(format.TdefLogicalIndexCountOffset, 4)); - int colCount = BinaryPrimitives.ReadUInt16LittleEndian(def.AsSpan(format.TdefColumnCountOffset, 2)); - - int pos = format.TdefRealIndexBlockOffset; - yield return ("header", 0, pos); - - yield return ("index-stats", pos, pos + dataCount * format.RealIndexEntrySize); - pos += dataCount * format.RealIndexEntrySize; - - yield return ("column-descriptors", pos, pos + colCount * format.ColumnDescriptorSize); - pos += colCount * format.ColumnDescriptorSize; - - int namesStart = pos; - for (int i = 0; i < colCount; i++) pos += 2 + BinaryPrimitives.ReadUInt16LittleEndian(def.AsSpan(pos, 2)); - yield return ("column-names", namesStart, pos); - - yield return ("index-data-blocks", pos, pos + dataCount * DataBlockSize); - pos += dataCount * DataBlockSize; - - yield return ("index-info-blocks", pos, pos + logicalCount * InfoBlockSize); - pos += logicalCount * InfoBlockSize; - - int indexNames = pos; - for (int i = 0; i < logicalCount && pos + 2 <= def.Length; i++) - pos += 2 + BinaryPrimitives.ReadUInt16LittleEndian(def.AsSpan(pos, 2)); - yield return ("index-names", indexNames, pos); - - yield return ("long-value-region", pos, def.Length); + TableDefinition.Regions regions = TableDefinition.Regions.Of(def, format); + int columnNames = regions.ColumnDescriptors + regions.ColumnCount * format.ColumnDescriptorSize; + int indexNamesEnd = regions.IndexNames; + for (int i = 0; i < regions.LogicalCount; i++) + indexNamesEnd = TableDefinition.NameEntryEnd(def, indexNamesEnd, format, $"index {i}"); + + yield return ("header", 0, regions.Stats); + yield return ("index-stats", regions.Stats, regions.ColumnDescriptors); + yield return ("column-descriptors", regions.ColumnDescriptors, columnNames); + yield return ("column-names", columnNames, regions.DataBlocks); + yield return ("index-data-blocks", regions.DataBlocks, regions.InfoBlocks); + yield return ("index-info-blocks", regions.InfoBlocks, regions.IndexNames); + yield return ("index-names", regions.IndexNames, indexNamesEnd); + yield return ("long-value-region", indexNamesEnd, def.Length); } - private static (byte[] Bytes, JetFormatBase Format)? Definition(string sql, Action create) + /// Table W's whole definition, continuation pages stitched in, and that definition parsed. + private static (byte[] Bytes, JetFormatBase Format, TableDefinition Parsed)? Definition( + string sql, Action create) { string path = TemporaryDatabase.CopyPath( Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "tdef-parity-"); @@ -181,16 +143,12 @@ private static (byte[] Bytes, JetFormatBase Format)? Definition(string sql, Acti try { create(path, sql); } catch (OleDbException) { return null; } - int definitionPage; - using (var database = JetDatabase.Open(path, readOnly: true)) - definitionPage = database.Catalog.FindTable("W")!.DefinitionPage; - - using var channel = PageChannel.Open(path, readOnly: true); - JetFormatBase format = channel.Format; - byte[] page = channel.ReadPage(definitionPage).Span.ToArray(); - int length = BinaryPrimitives.ReadInt32LittleEndian(page.AsSpan(format.TdefLengthOffset, 4)); - return (length > 0 && length <= page.Length ? page.AsSpan(0, length).ToArray() : page, format); + using var database = JetDatabase.Open(path, readOnly: true); + (PageBuffer buffer, _) = TableDefinition.ReadChain(database.Channel, database.Catalog.FindTable("W")!.DefinitionPage); + var parsed = new TableDefinition(); + parsed.Read(buffer, database.Format); + return (buffer.Span.ToArray(), database.Format, parsed); } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/TopWithTiesAccessTests.cs b/test/LibRed.Engine.AccessTests/TopWithTiesAccessTests.cs new file mode 100644 index 000000000..c296b0789 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/TopWithTiesAccessTests.cs @@ -0,0 +1,70 @@ +using System.Data.OleDb; +using System.Globalization; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// ACE's own TOP n keeps every row that ties the last on the ORDER BY keys, so LibRed's TOP n WITH TIES +/// must choose the same rows. ACE orders tied rows unstably, so the rows are compared as a set, not as a sequence. +/// +[Collection(AceCollection.Name)] +public class TopWithTiesAccessTests : TempDatabaseTest +{ + private static readonly string[] Setup = + [ + "CREATE TABLE T (Id LONG, K LONG, K2 TEXT(10))", + "INSERT INTO T VALUES (1, 1, 'a')", + "INSERT INTO T VALUES (2, 2, 'b')", + "INSERT INTO T VALUES (3, 2, 'a')", + "INSERT INTO T VALUES (4, 2, 'b')", + "INSERT INTO T VALUES (5, 3, 'a')", + "INSERT INTO T VALUES (6, 3, 'a')", + "INSERT INTO T VALUES (7, 10, 'c')", + "INSERT INTO T VALUES (8, NULL, NULL)", + "INSERT INTO T VALUES (9, NULL, 'z')", + ]; + + [Theory] + [InlineData("TOP 1", "ORDER BY K")] + [InlineData("TOP 3", "ORDER BY K")] + [InlineData("TOP 4", "ORDER BY K")] + [InlineData("TOP 6", "ORDER BY K")] + [InlineData("TOP 3", "ORDER BY K, K2")] + [InlineData("TOP 2", "ORDER BY K DESC")] + [InlineData("TOP 5", "ORDER BY K2")] + [InlineData("TOP 2", "ORDER BY K MOD 2")] + [InlineData("TOP 30 PERCENT", "ORDER BY K")] + [InlineData("TOP 50 PERCENT", "ORDER BY K")] + public void Libred_with_ties_takes_the_rows_ace_top_takes(string top, string orderBy) + { + string path = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "ties-"); + try + { + List ace; + using (OleDbConnection connection = AceTestDatabase.Open(path)) + { + foreach (string statement in Setup) + { + using OleDbCommand setup = connection.CreateCommand(); + setup.CommandText = statement; + setup.ExecuteNonQuery(); + } + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = $"SELECT {top} Id FROM T {orderBy}"; + using OleDbDataReader reader = command.ExecuteReader(); + ace = []; + while (reader.Read()) ace.Add(reader.GetInt32(0)); + } + OleDbConnection.ReleaseObjectPool(); + + List libred; + using (var database = JetDatabase.Open(path, readOnly: true)) + libred = [.. new QueryEngine(database).ExecuteQuery($"SELECT {top} WITH TIES Id FROM T {orderBy}") + .Rows.Select(row => Convert.ToInt32(row[0], CultureInfo.InvariantCulture))]; + + Assert.Equal(ace.Order(), libred.Order()); + } + finally { TemporaryDatabase.Delete(path); } + } +} diff --git a/test/LibRed.Engine.AccessTests/TransactionalDdlRollbackAccessTests.cs b/test/LibRed.Engine.AccessTests/TransactionalDdlRollbackAccessTests.cs index f272560b4..033f33d14 100644 --- a/test/LibRed.Engine.AccessTests/TransactionalDdlRollbackAccessTests.cs +++ b/test/LibRed.Engine.AccessTests/TransactionalDdlRollbackAccessTests.cs @@ -27,14 +27,14 @@ public void Full_rollback_removes_created_index_view_table_and_column_alter_byte e.ExecuteNonQuery("CREATE TABLE TransientDdl (Id LONG PRIMARY KEY)"); e.ExecuteNonQuery("ALTER TABLE DdlTxn ALTER COLUMN V TEXT(80)"); - TableDef changed = db.Catalog.FindTable("DdlTxn")!; + TableDefinition changed = db.Catalog.FindTable("DdlTxn")!; Assert.Contains(changed.Indexes, i => i.Name == "UX_DdlTxn_Code"); Assert.Equal(160, changed.FindColumn("V")!.Length); Assert.NotNull(db.Catalog.FindTable("TransientDdl")); Assert.Single(e.ExecuteQuery("SELECT Id FROM DdlTxnView").Rows); e.ExecuteNonQuery("ROLLBACK"); - TableDef restored = db.Catalog.FindTable("DdlTxn")!; + TableDefinition restored = db.Catalog.FindTable("DdlTxn")!; Assert.DoesNotContain(restored.Indexes, i => i.Name == "UX_DdlTxn_Code"); Assert.Equal(40, restored.FindColumn("V")!.Length); Assert.Null(db.Catalog.FindTable("TransientDdl")); @@ -75,7 +75,7 @@ public void Full_rollback_restores_dropped_objects_and_rebuilt_column_byte_for_b Assert.Throws(() => e.ExecuteQuery("SELECT Id FROM DdlTxnView")); e.ExecuteNonQuery("ROLLBACK"); - TableDef restored = db.Catalog.FindTable("DdlTxn")!; + TableDefinition restored = db.Catalog.FindTable("DdlTxn")!; Assert.Contains(restored.Indexes, i => i.Name == "UX_DdlTxn_Code"); Assert.Equal(JetDataType.Int32, restored.FindColumn("N")!.Type); Assert.NotNull(db.Catalog.FindTable("KeptTable")); @@ -113,7 +113,7 @@ public void Inner_ddl_rollback_restores_outer_created_objects_which_then_commit_ e.ExecuteNonQuery("CREATE TABLE InnerOnly (Id LONG PRIMARY KEY)"); e.ExecuteNonQuery("ROLLBACK"); - TableDef restoredOuter = db.Catalog.FindTable("DdlTxn")!; + TableDefinition restoredOuter = db.Catalog.FindTable("DdlTxn")!; Assert.Contains(restoredOuter.Indexes, i => i.Name == "UX_DdlTxn_Code"); Assert.Equal(JetDataType.Int32, restoredOuter.FindColumn("N")!.Type); Assert.Null(db.Catalog.FindTable("InnerOnly")); @@ -163,4 +163,4 @@ private static void Execute(OleDbConnection connection, string sql) private static void AssertScalar(OleDbConnection connection, string sql, int expected) => Assert.Equal(expected, Convert.ToInt32(ExecuteScalar(connection, sql))); -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.AccessTests/VariantAccessTests.cs b/test/LibRed.Engine.AccessTests/VariantAccessTests.cs new file mode 100644 index 000000000..06f5c215a --- /dev/null +++ b/test/LibRed.Engine.AccessTests/VariantAccessTests.cs @@ -0,0 +1,213 @@ +using System.Data.OleDb; +using System.Globalization; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Variants and Mixed values, as ACE's expression service types them: CVar(x) and what keeps one, and a +/// choice whose values disagree in kind. Each query runs through ACE over OLE DB and through LibRed, and the column +/// types and every value must agree — which covers where each becomes text (a result, a derived table, a scalar +/// subquery, a union), where it counts as a Double, and where it sorts, groups and takes Max as its text. +/// +/// ACE writes a date out in the Windows user locale, so LibRed runs under that culture here; the rest of the +/// suite pins en-US. +[Collection(AceCollection.Name)] +public class VariantAccessTests(ITestOutputHelper output) : TempDatabaseTest +{ + private static readonly string[] Setup = + [ + "CREATE TABLE V (Id LONG, B BYTE, F DOUBLE, M CURRENCY, D DATETIME, Y BIT, T TEXT(20), G GUID, N VARBINARY(4))", + "INSERT INTO V VALUES (1, 3, 2.25, 10.5, #2020-01-02 03:04:05#, TRUE, '8', {00112233-4455-6677-8899-AABBCCDDEEFF}, 0x41004200)", + "INSERT INTO V VALUES (2, 10, 7.5, 1.25, #2021-05-06#, FALSE, '9', NULL, NULL)", + "INSERT INTO V VALUES (3, 25, 0.5, 3, #2019-12-31#, TRUE, '10', NULL, NULL)", + ]; + + public static TheoryData Expressions => + [ + // A Variant is written out as text, as CStr writes it. + "CVar(B)", "CVar(F)", "CVar(M)", "CVar(D)", "CVar(Y)", "CVar(T)", "CVar(G)", "CVar(N)", "CVar(CVar(B))", + "-CVar(B)", + // + keeps a Variant beside a Variant or text; it adds or concatenates as the values are. + "CVar(B) + CVar(B)", "CVar(D) + CVar(B)", "CVar(T) + CVar(T)", "CVar(T) + CVar(B)", "CVar(B) + '1'", + "CVar(B) + T", + // Anything else counts a Variant as a Double. + "CVar(B) + 1", "1 + CVar(B)", "CVar(B) * 2", "CVar(B) - B", "CVar(B) / 2", "CVar(B) ^ 2", "CVar(B) \\ 2", + "CVar(B) MOD 2", "CVar(B) - CVar(B)", "CVar(B) * CVar(B)", "CVar(B) / CVar(B)", "CVar(D) + 1", "CVar(D) - D", + "CVar(M) + 1", "CVar(M) * 2", "CVar(B) + M", "CVar(M) + M", "CVar(F) + 1", "CVar(T) + 1", "CVar(B) & 'x'", + "CInt(CVar(B))", "CStr(CVar(B))", "Abs(CVar(B))", "Round(CVar(F), 1)", "Len(CVar(B))", "Left(CVar(T), 1)", + "DateAdd('d', 1, CVar(D))", "Int(CVar(F))", + // A choice: a Variant when all its values are, Mixed when only some are or text meets another kind. + "IIF(Id = 1, CVar(B), 5)", "IIF(Id = 1, 5, CVar(B))", "IIF(Id = 1, CVar(B), CVar(B))", "IIF(Id = 1, CVar(B), NULL)", + "IIF(Id = 1, IIF(Id = 2, CVar(B), 1), 2)", "IIF(Id = 1, CVar(B), 5) + 1", + "IIF(Id = 1, CVar(B), 5) + IIF(Id = 1, CVar(B), 5)", + "IIF(Id = 1, CVar(B), CVar(B)) + IIF(Id = 1, CVar(B), CVar(B))", "SWITCH(Id = 1, CVar(B), TRUE, 7)", + "CHOOSE(Id, CVar(B), 5, 6)", "IIF(Id = 1, T, 2)", "IIF(Id = 1, T, 2) + 1", "IIF(Id = 1, T, 2) + IIF(Id = 1, T, 2)", + "IIF(Id = 1, T, T) + IIF(Id = 1, T, T)", "IIF(Id = 1, D, T)", "IIF(Id = 1, Y, T)", "IIF(Id = 1, G, T)", + "IIF(Id = 1, N, T)", "IIF(Id = 1, T, NULL)", "SWITCH(Id = 1, T, TRUE, 2)", "CHOOSE(Id, T, 2, 3)", + "-IIF(Id = 1, T, 2)", "IIF(Id = 1, T, 2) + CVar(B)", "IIF(Id = 1, T, 2) + T", "IIF(Id = 1, T, 2) * 2", + "Abs(IIF(Id = 1, T, 2))", "CStr(IIF(Id = 1, CVar(B), 5))", "IIF(Id = 1, CVar(B), 5) & ''", + "SWITCH(Id = 1, CVar(B), TRUE, CVar(B))", "CHOOSE(Id, CVar(B), CVar(B), CVar(B))", + "IIF(Id = 1, IIF(Id = 1, CVar(B), CVar(B)), 2)", "IIF(Id = 1, CVar(B) + CVar(B), 2)", + "IIF(Id = 1, IIF(Id = 1, CVar(B), CVar(B)), 2) + 1", + // A date beside a number is a date; a Boolean beside one a Long. + "IIF(Id = 1, D, 2)", "IIF(Id = 1, D, 2) + 1", "IIF(Id = 1, Y, 2)", + // Aggregates: Min, Max, First and Last over text; the others count a Variant as a Double. + "SUM(CVar(B))", "MAX(CVar(B))", "MIN(CVar(B))", "MIN(CVar(T))", "AVG(CVar(F))", "COUNT(CVar(B))", + "FIRST(CVar(B))", "LAST(CVar(B))", "MAX(CVar(D))", "STDEV(CVar(F))", "MAX(CVar(B) + 1)", "MAX(-CVar(B))", + "MAX(IIF(Id = 0, 1, T))", "MAX(IIF(Id = 0, 1, B))", "FIRST(IIF(Id = 1, T, 2))", "SUM(IIF(Id = 0, 1, T))", + ]; + + public static TheoryData Queries => + [ + // Grouping and ordering compare the text; a filter compares the value itself. + "SELECT CVar(B), COUNT(*) FROM V GROUP BY CVar(B)", + "SELECT IIF(Id = 0, 1, T), COUNT(*) FROM V GROUP BY IIF(Id = 0, 1, T)", + "SELECT Id FROM V ORDER BY CVar(B)", "SELECT Id FROM V ORDER BY -CVar(B)", "SELECT Id FROM V ORDER BY CVar(B) + 1", + "SELECT Id FROM V ORDER BY IIF(Id = 0, 1, T)", "SELECT Id FROM V ORDER BY CVar(T)", + "SELECT Id FROM V WHERE CVar(B) > 5", "SELECT Id FROM V WHERE CVar(B) = '3'", + "SELECT Id FROM V WHERE CVar(T) > 9", "SELECT Id FROM V WHERE CVar(T) > '9'", + // A derived table keeps a Variant and writes a Mixed value out as text. + "SELECT X + X FROM (SELECT CVar(B) AS X FROM V)", "SELECT X + 1 FROM (SELECT CVar(B) AS X FROM V)", + "SELECT X FROM (SELECT CVar(B) AS X FROM V) ORDER BY X", + "SELECT X + X FROM (SELECT IIF(Id = 1, CVar(B), 5) AS X FROM V)", + "SELECT X + X FROM (SELECT IIF(Id = 1, T, 2) AS X FROM V)", + "SELECT X + X FROM (SELECT IIF(Id = 1, CVar(B), CVar(B)) AS X FROM V)", + "SELECT X + X FROM (SELECT IIF(Id = 1, CVar(B), 5) + 0 AS X FROM V)", + "SELECT X + X FROM (SELECT -CVar(B) AS X FROM V)", "SELECT X + X FROM (SELECT CVar(B) + CVar(B) AS X FROM V)", + "SELECT X + X FROM (SELECT CVar(B) + '1' AS X FROM V)", "SELECT X + X FROM (SELECT CVar(B) + T AS X FROM V)", + "SELECT X + X FROM (SELECT IIF(Id = 1, T, 2) + CVar(B) AS X FROM V)", + "SELECT X + X FROM (SELECT SWITCH(Id = 1, CVar(B), TRUE, CVar(B)) AS X FROM V)", + "SELECT X + X FROM (SELECT CHOOSE(Id, CVar(B), CVar(B), CVar(B)) AS X FROM V)", + "SELECT X + X FROM (SELECT IIF(Id = 1, IIF(Id = 1, CVar(B), CVar(B)), 2) AS X FROM V)", + "SELECT X + X FROM (SELECT IIF(Id = 1, CVar(B) + CVar(B), 2) AS X FROM V)", + "SELECT X, COUNT(*) FROM (SELECT CVar(B) AS X FROM V) GROUP BY X", "SELECT MAX(X) FROM (SELECT CVar(B) AS X FROM V)", + // A scalar subquery and a union write a Variant out as text. + "SELECT (SELECT CVar(B) FROM V WHERE Id = 2) FROM V", "SELECT (SELECT CVar(B) FROM V WHERE Id = 2) + 1 FROM V", + "SELECT (SELECT CVar(B) FROM V WHERE Id = 2) + (SELECT CVar(B) FROM V WHERE Id = 2) FROM V", + "SELECT (SELECT MAX(CVar(B)) FROM V) FROM V", + "SELECT CVar(B) FROM V UNION ALL SELECT B FROM V", "SELECT B FROM V UNION ALL SELECT CVar(B) FROM V", + "SELECT X + X FROM (SELECT CVar(B) AS X FROM V UNION ALL SELECT B FROM V)", + ]; + + [Theory] + [MemberData(nameof(Expressions))] + public void An_expression_is_typed_and_valued_as_ace_does(string expression) => + Matches($"SELECT {expression} FROM V"); + + [Theory] + [MemberData(nameof(Queries))] + public void A_query_is_typed_and_valued_as_ace_does(string query) => Matches(query); + + // A make-table query writes a Variant or a Mixed value out as text, into a Text(255) column; a value over one that + // counts as a number keeps its number column. + [Fact] + public void Select_into_makes_a_variant_a_text_column_as_ace_does() + { + const string query = "SELECT CVar(B) AS XB, CVar(D) AS XD, IIF(Id = 1, T, 2) AS XM, CVar(B) + 1 AS XP INTO W FROM V"; + string ace = MadeTable(path => + { + using OleDbConnection connection = AceTestDatabase.Open(path); + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = query; + return command.ExecuteNonQuery(); + }); + string libred = MadeTable(path => UnderUserCulture(() => + { + using var database = JetDatabase.Open(path, readOnly: false); + return new QueryEngine(database).ExecuteNonQuery(query); + })); + + output.WriteLine($"ACE {ace}\nLibRed {libred}"); + Assert.Equal(ace, libred); + } + + /// Table W as a make-table query left it: each column's type and length, and its rows. + private static string MadeTable(Func run) + { + string path = Seeded(); + try + { + run(path); + using var database = JetDatabase.Open(path, readOnly: true); + var table = database.Catalog.FindTable("W")!; + return $"[{string.Join(", ", table.Columns.Select(c => $"{c.Name} {c.Type}({c.Length})"))}] " + + Rows(new QueryEngine(database).ExecuteQuery("SELECT * FROM W").Rows); + } + finally { TemporaryDatabase.Delete(path); } + } + + private void Matches(string query) + { + string path = Seeded(); + try + { + string ace; + using (OleDbConnection connection = AceTestDatabase.Open(path)) + using (OleDbCommand command = connection.CreateCommand()) + { + command.CommandText = query; + using OleDbDataReader reader = command.ExecuteReader(); + var types = Enumerable.Range(0, reader.FieldCount).Select(reader.GetFieldType).ToList(); + var rows = new List(); + while (reader.Read()) + rows.Add([.. Enumerable.Range(0, reader.FieldCount).Select(i => reader.IsDBNull(i) ? null : reader.GetValue(i))]); + ace = Describe(types, rows); + } + + string libred = UnderUserCulture(() => + { + using var database = JetDatabase.Open(path, readOnly: true); + var result = new QueryEngine(database).ExecuteQuery(query); + return Describe(result.ColumnTypes, result.Rows); + }); + + output.WriteLine($"ACE {ace}\nLibRed {libred}"); + Assert.Equal(ace, libred); + } + finally { TemporaryDatabase.Delete(path); } + } + + /// A copy of Northwind with table V made and filled through ACE. + private static string Seeded() + { + string path = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "variant-"); + using OleDbConnection connection = AceTestDatabase.Open(path); + foreach (string statement in Setup) + { + using OleDbCommand setup = connection.CreateCommand(); + setup.CommandText = statement; + setup.ExecuteNonQuery(); + } + return path; + } + + /// Runs LibRed under the Windows user's own locale and formats, which ACE writes a date out in. + private static T UnderUserCulture(Func run) + { + CultureInfo previous = CultureInfo.CurrentCulture; + CultureInfo.CurrentCulture = + Microsoft.Win32.Registry.CurrentUser.OpenSubKey(@"Control Panel\International")?.GetValue("LocaleName") is string name + ? new CultureInfo(name, useUserOverride: true) + : CultureInfo.InstalledUICulture; + try { return run(); } + finally { CultureInfo.CurrentCulture = previous; } + } + + private static string Describe(IReadOnlyList types, IEnumerable rows) => + $"[{string.Join(", ", types.Select(t => t.Name))}] " + Rows(rows); + + private static string Rows(IEnumerable rows) => + string.Join(" | ", rows.Select(row => string.Join(", ", row.Select(Value)))); + + private static string Value(object? value) => value switch + { + null => "NULL", + byte[] bytes => $"{Convert.ToHexString(bytes)}:Byte[]", + DateTime date => $"{date:o}:DateTime", + double d => $"{d.ToString("R", CultureInfo.InvariantCulture)}:Double", + // A Decimal's scale is not compared: LibRed's Currency sums keep .NET's (21.0) where ACE's do not (21). + decimal m => $"{(m / 1.0000000000000000000000000000m).ToString(CultureInfo.InvariantCulture)}:Decimal", + _ => $"{Convert.ToString(value, CultureInfo.InvariantCulture)}:{value.GetType().Name}", + }; +} diff --git a/test/LibRed.Engine.AccessTests/WholeFileParityProbeTest.cs b/test/LibRed.Engine.AccessTests/WholeFileParityProbeTest.cs new file mode 100644 index 000000000..a9d197ed1 --- /dev/null +++ b/test/LibRed.Engine.AccessTests/WholeFileParityProbeTest.cs @@ -0,0 +1,475 @@ +using System.Buffers.Binary; +using System.Data.OleDb; +using System.Globalization; +using System.Reflection; +using System.Text; +using LibRed; +using LibRed.Catalog; +using LibRed.Formats; +using LibRed.Pages; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// About 140 statements run through ACE over OLE DB and through LibRed on copies of one ACE-created empty database, +/// and the files compared byte for byte. +/// +/// +/// Two measurements. The whole run applies every statement in one session per engine and compares the two files at +/// the end — how close LibRed comes over a realistic workload; it reports, since one early placement difference +/// leaves every later page misaligned. Step by step applies one statement at a time, both engines starting from the +/// same file — ACE's result of the step before — so each statement is compared on its own, and it asserts: every +/// statement leaves identical files except the , each of which must still differ. +/// Only what differs for reasons other than the format is masked: page 0's user commit-byte table, which +/// counts a lock-file user's writes and LibRed does not keep; the wall-clock DateCreate and DateUpdate of each +/// MSysObjects row; and a data page's write stamp at 0x08, with the copy of it a chained long value's descriptor +/// carries. A copy opened and closed through ACE with nothing +/// done to it shows what ACE writes merely for holding a session; a byte that alone changes is counted apart rather +/// than as a difference. Calculated and complex columns are left out — ACE's OLE DB DDL cannot create either. +/// +[Collection(AceCollection.Name)] +public class WholeFileParityProbeTest(ITestOutputHelper output) +{ + [Fact] + public void The_whole_run_leaves_libred_and_ace_with_the_same_file() + { + string[] statements = Statements(); + string? origin = CreateEmptyThroughDao(); + Assert.SkipWhen(origin is null, "DAO is not registered in this bitness."); + string ace = TemporaryDatabase.CopyPath(origin, "wholefile-ace-"); + string noise = TemporaryDatabase.CopyPath(origin, "wholefile-noise-"); + string libred = TemporaryDatabase.CopyPath(origin, "wholefile-libred-"); + try + { + string?[] aceErrors = RunThroughAce(ace, statements); + using (AceTestDatabase.Open(noise)) { } + string?[] libredErrors = RunThroughLibRed(libred, statements); + + var report = new StringBuilder(); + report.AppendLine(CultureInfo.InvariantCulture, $"{statements.Length} statements"); + for (int i = 0; i < statements.Length; i++) + if (aceErrors[i] is not null || libredErrors[i] is not null) + report.AppendLine(CultureInfo.InvariantCulture, + $" #{i + 1} {statements[i]}\n ACE: {aceErrors[i] ?? "ok"}\n LibRed: {libredErrors[i] ?? "ok"}"); + report.Append(Compare(origin, ace, noise, libred).Report); + output.WriteLine(report.ToString()); + } + finally + { + foreach (string path in new[] { origin, ace, noise, libred }) TemporaryDatabase.Delete(path); + } + } + + [Fact] + public void Each_statement_leaves_libred_and_ace_with_the_same_file() + { + string[] statements = Statements(); + string? origin = CreateEmptyThroughDao(); + Assert.SkipWhen(origin is null, "DAO is not registered in this bitness."); + string current = origin; + var summary = new StringBuilder(); + var details = new StringBuilder(); + var outcomes = new string[statements.Length]; + int same = 0; + try + { + for (int i = 0; i < statements.Length; i++) + { + string ace = TemporaryDatabase.CopyPath(current, $"stepfile-ace{i}-"); + string noise = TemporaryDatabase.CopyPath(current, $"stepfile-noise{i}-"); + string libred = TemporaryDatabase.CopyPath(current, $"stepfile-libred{i}-"); + string? aceError = RunThroughAce(ace, [statements[i]])[0]; + using (AceTestDatabase.Open(noise)) { } + string? libredError = RunThroughLibRed(libred, [statements[i]])[0]; + + (string report, int pages) = Compare(current, ace, noise, libred); + string outcome = aceError is not null || libredError is not null + ? $"ACE: {aceError ?? "ok"} | LibRed: {libredError ?? "ok"}" + : pages == 0 ? "identical" : $"{pages} page(s) differ"; + if (outcome == "identical") same++; + outcomes[i] = outcome; + summary.AppendLine(CultureInfo.InvariantCulture, $"#{i + 1,-3} {outcome,-22} {Abbreviate(statements[i])}"); + if (pages > 0) + details.AppendLine(CultureInfo.InvariantCulture, $"==== #{i + 1} {statements[i]}").Append(report); + + // The next statement starts from ACE's result, so each is measured on its own. + if (current != origin) TemporaryDatabase.Delete(current); + current = ace; + TemporaryDatabase.Delete(noise); + TemporaryDatabase.Delete(libred); + } + + output.WriteLine($"{same} of {statements.Length} statements left identical files\n{summary}\n{details}"); + + var unexpected = new List(); + for (int i = 0; i < statements.Length; i++) + { + string? known = KnownDifferences.FirstOrDefault(k => statements[i].StartsWith(k.Statement, StringComparison.Ordinal)).Reason; + if (outcomes[i] == "identical" && known is not null) + unexpected.Add($"#{i + 1} is now identical; remove it from the known differences: {Abbreviate(statements[i])}"); + else if (outcomes[i] != "identical" && (known is null || !outcomes[i].EndsWith("differ", StringComparison.Ordinal))) + unexpected.Add($"#{i + 1} {outcomes[i]}: {Abbreviate(statements[i])}"); + } + Assert.True(unexpected.Count == 0, string.Join("\n", unexpected)); + } + finally + { + if (current != origin) TemporaryDatabase.Delete(current); + TemporaryDatabase.Delete(origin); + } + } + + /// The statements, by how each begins, that ACE and LibRed are known to leave differently — each a + /// placement difference with every pointer correct — and why. + private static readonly (string Statement, string Reason)[] KnownDifferences = + [ + ("INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT", + "ACE allocates from session extents and 8-page groups (PageGroupProbeTest); LibRed takes the lowest free page"), + ]; + + /// A hundred statements over six tables: creates with every ordinary column type and each kind of + /// constraint, inserts, updates and deletes, more inserts into the space the deletes freed, a column added, + /// retyped and dropped, indexes made and dropped, and a table dropped and another made after it. Then what + /// spans pages: a thousand rows from one INSERT … SELECT over many data pages with a text index grown + /// past one leaf, a 253-column table whose definition runs onto continuation pages, and long values chained + /// across pages, grown, shrunk and deleted. + private static string[] Statements() + { + var s = new List + { + "CREATE TABLE Customer (Id COUNTER CONSTRAINT pkCustomer PRIMARY KEY, Name TEXT(50) NOT NULL, City TEXT(30), " + + "Joined DATETIME, Credit CURRENCY, Active BIT)", + "CREATE TABLE Product (Code TEXT(10) CONSTRAINT pkProduct PRIMARY KEY, Title TEXT(100), Price DOUBLE, " + + "Weight REAL, Stock SHORT, Rating BYTE, Notes MEMO)", + "CREATE TABLE Orders (OrderId LONG CONSTRAINT pkOrders PRIMARY KEY, CustomerId LONG, Placed DATETIME, " + + "Total DECIMAL(18,4), Ref GUID, CONSTRAINT fkOrderCustomer FOREIGN KEY (CustomerId) REFERENCES Customer (Id))", + "CREATE TABLE OrderLine (OrderId LONG, LineNo SHORT, ProductCode TEXT(10), Qty LONG, " + + "CONSTRAINT pkOrderLine PRIMARY KEY (OrderId, LineNo))", + "CREATE TABLE Scratch (Id LONG, Data VARBINARY(100), Flag BIT, Label TEXT(20))", + "CREATE INDEX ixCustomerCity ON Customer (City)", + "CREATE UNIQUE INDEX ixProductTitle ON Product (Title)", + "CREATE INDEX ixOrderPlaced ON Orders (Placed DESC)", + }; + + string[] cities = ["London", "Paris", "Berlin", "Madrid", "Rome", "Oslo", "Vienna", "Prague", "Lisbon", "Dublin"]; + for (int i = 1; i <= 12; i++) + s.Add($"INSERT INTO Customer (Name, City, Joined, Credit, Active) VALUES ('Customer {i:D2}', " + + $"'{cities[i % cities.Length]}', #2020-{(i % 12) + 1:D2}-{(i % 27) + 1:D2} {i % 24:D2}:15:00#, {i * 125.5m}, {(i % 3 == 0 ? "FALSE" : "TRUE")})"); + + for (int i = 1; i <= 10; i++) + s.Add($"INSERT INTO Product (Code, Title, Price, Weight, Stock, Rating, Notes) VALUES ('P{i:D3}', 'Product number {i}', " + + $"{i * 9.99}, {i * 0.25}, {i * 7}, {i * 20 % 256}, '{Notes(i)}')"); + + for (int i = 1; i <= 12; i++) + s.Add($"INSERT INTO Orders (OrderId, CustomerId, Placed, Total, Ref) VALUES ({1000 + i}, {(i % 9) + 1}, " + + $"#2021-{(i % 12) + 1:D2}-15#, {i * 101.1234m}, {{guid {{{new Guid(i, 0, 0, [1, 2, 3, 4, 5, 6, 7, (byte)i])}}}}})"); + + for (int i = 1; i <= 16; i++) + s.Add($"INSERT INTO OrderLine (OrderId, LineNo, ProductCode, Qty) VALUES ({1000 + (i % 8) + 1}, {i}, 'P{(i % 10) + 1:D3}', {i % 6})"); + + for (int i = 1; i <= 6; i++) + s.Add($"INSERT INTO Scratch (Id, Data, Flag, Label) VALUES ({i}, 0x{Convert.ToHexString([.. Enumerable.Range(i, 20).Select(b => (byte)b)])}, " + + $"{(i % 2 == 0 ? "TRUE" : "FALSE")}, 'scratch {i}')"); + + s.AddRange( + [ + "UPDATE Customer SET City = 'Paris' WHERE Id = 3", + "UPDATE Customer SET Credit = Credit * 2 WHERE Active = TRUE", + "UPDATE Product SET Price = Price + 1, Stock = Stock - 1", + "UPDATE Product SET Notes = 'short note' WHERE Code = 'P002'", + "UPDATE Orders SET Total = Total + 0.5 WHERE CustomerId = 2", + "DELETE FROM OrderLine WHERE Qty < 2", + "DELETE FROM Scratch WHERE Id > 4", + "DELETE FROM Customer WHERE Id = 12", + "DELETE FROM Product WHERE Code = 'P010'", + ]); + + for (int i = 17; i <= 22; i++) + s.Add($"INSERT INTO OrderLine (OrderId, LineNo, ProductCode, Qty) VALUES ({1000 + (i % 8) + 1}, {i}, 'P{(i % 9) + 1:D3}', {i})"); + + s.AddRange( + [ + "ALTER TABLE Customer ADD COLUMN Email TEXT(80)", + "UPDATE Customer SET Email = 'c' & Id & '@example.com' WHERE Id < 6", + "ALTER TABLE Product ALTER COLUMN Stock LONG", + "ALTER TABLE Scratch DROP COLUMN Flag", + "INSERT INTO Scratch (Id, Data, Label) VALUES (7, 0x0A0B0C, 'after drop')", + "DROP INDEX ixCustomerCity ON Customer", + "CREATE INDEX ixCustomerName ON Customer (Name)", + "DROP TABLE Scratch", + "CREATE TABLE Audit (Id COUNTER CONSTRAINT pkAudit PRIMARY KEY, Detail TEXT(255), LoggedOn DATETIME, Charge CURRENCY)", + ]); + + for (int i = 1; i <= 8; i++) + s.Add($"INSERT INTO Audit (Detail, LoggedOn, Charge) VALUES ('Audit entry {i} {new string('x', i * 20)}', #2022-03-{i:D2} 08:00:00#, {i * 3.25m})"); + + s.AddRange( + [ + "DELETE FROM Audit WHERE Id IN (2, 4, 6)", + "INSERT INTO Audit (Detail, LoggedOn, Charge) VALUES ('refill one', #2022-04-01#, 1)", + "INSERT INTO Audit (Detail, LoggedOn, Charge) VALUES ('refill two', #2022-04-02#, 2)", + "ALTER TABLE Audit ADD COLUMN Severity BYTE", + "UPDATE Audit SET Severity = Id MOD 4", + "UPDATE Customer SET Name = 'Renamed ' & Name WHERE Id MOD 2 = 0", + "DELETE FROM Orders WHERE OrderId = 1012", + "INSERT INTO Customer (Name, City, Joined, Credit, Active, Email) VALUES ('Late comer', 'Oslo', #2023-01-01#, 10, TRUE, 'late@example.com')", + ]); + + // A table over many data pages, filled by one INSERT … SELECT over a cross join, with an index that grows + // past one leaf; deletes across its pages; and updates that grow rows until they have to move. + s.Add("CREATE TABLE Digits (D BYTE)"); + for (int d = 0; d <= 9; d++) s.Add($"INSERT INTO Digits (D) VALUES ({d})"); + s.AddRange( + [ + "CREATE TABLE Bulk (Id LONG CONSTRAINT pkBulk PRIMARY KEY, Grp LONG, Label TEXT(60), Payload TEXT(255), " + + "Code BINARY(8), Amount DECIMAL(10,2))", + "CREATE INDEX ixBulkLabel ON Bulk (Label)", + "INSERT INTO Bulk (Id, Grp, Label, Payload, Amount) SELECT a.D * 100 + b.D * 10 + c.D, a.D, " + + "'Bulk label number ' & (a.D * 100 + b.D * 10 + c.D) & ' ' & String(30, 'L'), String(150, 'p'), " + + "a.D + b.D * 0.5 FROM Digits AS a, Digits AS b, Digits AS c ORDER BY 1", + "DELETE FROM Bulk WHERE Id MOD 7 = 0", + "UPDATE Bulk SET Payload = Payload & String(100, 'g') WHERE Id MOD 10 = 1", + "UPDATE Bulk SET Label = 'R ' & Label WHERE Id < 50", + "UPDATE Bulk SET Code = 0x0102030405060708 WHERE Id < 5", + "INSERT INTO Bulk (Id, Grp, Label, Payload) VALUES (5000, 1, 'Late bulk', 'late')", + ]); + + // A 253-column table, whose definition runs onto continuation pages, leaving room for one more column id + // after a drop — ids are never reused before a compact, and 255 is the lifetime limit. + string[] wideTypes = ["LONG", "TEXT(20)", "DOUBLE", "DATETIME", "CURRENCY", "BIT"]; + s.Add("CREATE TABLE Wide (Id LONG CONSTRAINT pkWide PRIMARY KEY, " + + string.Join(", ", Enumerable.Range(1, 252).Select(i => $"Column{i:D3} {wideTypes[i % wideTypes.Length]}")) + ")"); + for (int i = 1; i <= 3; i++) + s.Add($"INSERT INTO Wide (Id, Column001, Column002, Column252) VALUES ({i}, 'wide {i}', {i * 1.5}, {i})"); + s.AddRange( + [ + "ALTER TABLE Wide DROP COLUMN Column005", + "ALTER TABLE Wide ADD COLUMN Extra TEXT(10)", + "UPDATE Wide SET Extra = 'extra' WHERE Id = 2", + ]); + + // Long values across pages, the other binary types, and compressed text. + s.AddRange( + [ + "CREATE TABLE Doc (Id LONG CONSTRAINT pkDoc PRIMARY KEY, Body MEMO, Packed MEMO WITH COMPRESSION, " + + "Brief TEXT(100) WITH COMPRESSION, Blob LONGBINARY, Big BIGBINARY(500))", + $"INSERT INTO Doc VALUES (1, String(5000, 'b'), String(300, 'c'), 'short and compressed', 0x{Hex(6000, 1)}, 0x{Hex(400, 2)})", + $"INSERT INTO Doc VALUES (2, String(2500, 'd'), 'packed two', 'two', 0x{Hex(100, 3)}, 0x{Hex(50, 4)})", + "INSERT INTO Doc (Id, Body, Brief) VALUES (3, 'small', 'three')", + "UPDATE Doc SET Body = String(8000, 'e') WHERE Id = 2", + "UPDATE Doc SET Body = 'now small' WHERE Id = 1", + $"UPDATE Doc SET Blob = 0x{Hex(9000, 5)} WHERE Id = 3", + "DELETE FROM Doc WHERE Id = 2", + "INSERT INTO Doc (Id, Body) VALUES (4, String(4000, 'f'))", + ]); + + return [.. s]; + } + + /// bytes as hex, a repeating pattern seeded by . + private static string Hex(int bytes, int seed) => + Convert.ToHexString([.. Enumerable.Range(0, bytes).Select(i => (byte)((i * 7 + seed) % 251))]); + + private static string Notes(int i) => string.Concat(Enumerable.Repeat($"Note {i}. ", i * 12)); + + private static string Abbreviate(string statement) => statement.Length <= 90 ? statement : statement[..87] + "..."; + + private static string?[] RunThroughAce(string path, string[] statements) + { + var errors = new string?[statements.Length]; + using OleDbConnection connection = AceTestDatabase.Open(path); + for (int i = 0; i < statements.Length; i++) + { + try + { + using OleDbCommand command = connection.CreateCommand(); + command.CommandText = statements[i]; + command.ExecuteNonQuery(); + } + catch (OleDbException e) { errors[i] = e.Message; } + } + return errors; + } + + private static string?[] RunThroughLibRed(string path, string[] statements) + { + var errors = new string?[statements.Length]; + using var database = JetDatabase.Open(path, readOnly: false); + var engine = new QueryEngine(database); + for (int i = 0; i < statements.Length; i++) + { + try { engine.ExecuteNonQuery(statements[i]); } +#pragma warning disable CA1031 // a probe records every refusal, whatever its type, and carries on + catch (Exception e) { errors[i] = $"{e.GetType().Name}: {e.Message}"; } +#pragma warning restore CA1031 + } + return errors; + } + + /// Compares the two results of the same statements applied to : page + /// counts, then every differing page — page 0 on its own, the rest grouped by page type and owner — with its + /// differing byte count and the first few differences. Returns the report and how many pages differ. + private static (string Report, int Pages) Compare(string originPath, string acePath, string noisePath, string libredPath) + { + byte[] origin = File.ReadAllBytes(originPath), ace = File.ReadAllBytes(acePath), + noise = File.ReadAllBytes(noisePath), libred = File.ReadAllBytes(libredPath); + JetFormatBase format = JetFormatBase.Detect(new MemoryStream(origin)); + int pageSize = format.PageSize; + Dictionary owners = Owners(acePath, libredPath); + HashSet[] masks = [.. Enumerable.Range(0, Math.Max(ace.Length, libred.Length) / pageSize).Select(_ => new HashSet())]; + Mask(acePath, ace, masks); + Mask(libredPath, libred, masks); + + int acePages = ace.Length / pageSize, libredPages = libred.Length / pageSize; + int identical = 0, housekeeping = 0, differingPages = 0; + var groups = new SortedDictionary>(StringComparer.Ordinal); + for (int page = 0; page < Math.Max(acePages, libredPages); page++) + { + int at = page * pageSize; + if (page >= acePages || page >= libredPages) + { + byte[] only = page >= acePages ? libred : ace; + Add(groups, "pages in one file only", $"page {page}: {(page >= acePages ? "LibRed" : "ACE")} only, " + + $"type 0x{only[at]:X2} {Owner(owners, format, only, at)}"); + differingPages++; + continue; + } + + var differing = new List(); + for (int i = 0; i < pageSize; i++) + { + if (ace[at + i] == libred[at + i] || masks[page].Contains(i)) continue; + // A byte ACE writes merely for holding a session, which LibRed need not reproduce. + if (at + i < origin.Length && at + i < noise.Length && origin[at + i] != noise[at + i]) { housekeeping++; continue; } + differing.Add(i); + } + if (differing.Count == 0) { identical++; continue; } + + differingPages++; + string detail = string.Join(" ", differing.Take(8).Select(i => $"+0x{i:X3}:{ace[at + i]:X2}/{libred[at + i]:X2}")); + string kind = page == 0 ? "page 0" + : $"type 0x{ace[at]:X2}{(ace[at] != libred[at] ? $"/0x{libred[at]:X2}" : "")} {Owner(owners, format, ace, at)}"; + Add(groups, kind, $"page {page}: {differing.Count} bytes {detail}"); + } + + var report = new StringBuilder(); + report.AppendLine(CultureInfo.InvariantCulture, + $"pages: before {origin.Length / pageSize}, ACE {acePages}, LibRed {libredPages}; identical {identical}; " + + $"{housekeeping} byte(s) differ only where ACE's session alone writes (ace/libred shown)"); + foreach (var (kind, lines) in groups) + { + report.AppendLine(CultureInfo.InvariantCulture, $"-- {kind}: {lines.Count} page(s)"); + foreach (string line in lines) report.AppendLine(CultureInfo.InvariantCulture, $" {line}"); + } + return (report.ToString(), differingPages); + } + + private static void Add(SortedDictionary> groups, string kind, string line) + { + if (!groups.TryGetValue(kind, out var lines)) groups[kind] = lines = []; + lines.Add(line); + } + + /// What owns a page: for a data or index page the table whose TDEF page its header names, for a TDEF + /// page its own table, where either file's catalog knows the name. + private static string Owner(Dictionary owners, JetFormatBase format, byte[] file, int at) + { + if (PageHeader.ReadType(file.AsSpan(at)) is PageType.TableDefinition or PageType.ReleasedTableDefinition) + return owners.TryGetValue(at / format.PageSize, out string? self) ? $"({self})" : ""; + if (PageHeader.ReadType(file.AsSpan(at)) is not (PageType.DataPage or PageType.IntermediateIndexPage or PageType.LeafIndexPage)) + return ""; + ReadOnlySpan page = file.AsSpan(at, format.PageSize); + uint ownerField = PageHeader.ReadType(page) == PageType.DataPage + ? DataPage.ReadOwner(page, format) + : (uint)IndexTree.ReadOwner(page, format); + if (ownerField == JetFormatBase.LongValuePageMarker) return "(long values)"; + int owner = (int)ownerField; + return owners.TryGetValue(owner, out string? name) ? $"({name})" : $"(tdef {owner})"; + } + + private static Dictionary Owners(params string[] paths) + { + var owners = new Dictionary(); + foreach (string path in paths) + { + using var database = JetDatabase.Open(path, readOnly: true); + foreach (TableDefinition table in database.Catalog.Tables) owners.TryAdd(table.DefinitionPage, table.Name); + } + return owners; + } + + /// Masks the bytes known to differ for reasons other than the format: page 0's user commit-byte table, + /// a count of each lock-file user's committed writes that LibRed, keeping no lock file, leaves alone + /// (page-00-database.md §2.2); each MSysObjects row's wall-clock DateCreate and DateUpdate; and every data page's + /// write stamp. + private static void Mask(string path, byte[] file, HashSet[] masks) + { + using var database = JetDatabase.Open(path, readOnly: true); + JetFormatBase format = database.Format; + int pageSize = format.PageSize; + + for (int i = format.CommitByteTableOffset; i < pageSize; i++) masks[0].Add(i); + + TableDefinition objects = database.Catalog.FindTable("MSysObjects")!; + int[] dates = [.. new[] { "DateCreate", "DateUpdate" }.Select(c => objects.FindColumn(c)!.FixedOffset)]; + + for (int page = 1; page < file.Length / pageSize; page++) + { + int at = page * pageSize; + if (PageHeader.ReadType(file.AsSpan(at)) != PageType.DataPage) continue; + for (int i = format.DataChainStampOffset; i < format.DataChainStampOffset + sizeof(int); i++) masks[page].Add(i); + ReadOnlySpan bytes = file.AsSpan(at, pageSize); + if ((int)DataPage.ReadOwner(bytes, format) != objects.DefinitionPage) continue; + + int rows = DataPage.ReadRowCount(bytes, format); + for (int row = 0; row < rows; row++) + { + (int start, RowSlotFlags flags) = DataPage.ReadSlot(bytes, format, row); + if ((flags & (RowSlotFlags.Deleted | RowSlotFlags.Overflow)) != 0) continue; + foreach (int offset in dates) + for (int i = 0; i < 8; i++) masks[page].Add(start + format.RowColumnCountSize + offset + i); + } + } + + // A chained long value's descriptor carries the same write stamp as the chain's first page, bytes 8–11 of + // its twelve (long-values.md). + foreach (TableDefinition definition in database.Catalog.Tables) + { + if (!definition.Columns.Any(c => c.Type is JetDataType.Memo or JetDataType.Ole)) continue; + Table table = database.OpenTable(definition.Name); + foreach ((RowId id, _) in table.Rows().WithIds()) + { + var page = new DataPage(); + page.Read(table.Channel.ReadPageShared(id.Page), table.Channel.Format); + if (page.Rows[id.Row] is not { IsDeleted: false, HasOverflow: false } slot) continue; + ReadOnlySpan row = page.GetRow(id.Row); + foreach (byte[] descriptor in RowCodec.LongValueDescriptors(definition.Columns, table.Channel.Format, row).Values) + { + if (descriptor.Length < format.LongValueDescriptorSize + || LongValueStore.Read(descriptor, format).Storage != LongValueStore.StorageKind.Chained) continue; + int at = row.IndexOf(descriptor); + int stamp = format.LongValueDescriptorChainStampOffset; + if (at >= 0) + for (int i = stamp; i < stamp + sizeof(int); i++) masks[id.Page].Add(slot.Offset + at + i); + } + } + } + } + + /// An empty ACE 12 database made by DAO, the path Access itself takes; null where DAO is missing. + private static string? CreateEmptyThroughDao() + { + object? engine = AceTestDatabase.CreateDaoEngine(); + if (engine is null) return null; + string path = TemporaryDatabase.CreatePath("wholefile-origin-"); + object workspace = Invoke(engine, "CreateWorkspace", "", "admin", "", 2)!; + object database = Invoke(workspace, "CreateDatabase", path, ";LANGID=0x0409;CP=1252;COUNTRY=0", 128)!; + Invoke(database, "Close"); + return path; + } + + private static object? Invoke(object target, string member, params object?[] args) => + target.GetType().InvokeMember(member, BindingFlags.InvokeMethod, null, target, args); +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/AddedFunctionsTests.cs b/test/LibRed.Engine.Tests/AddedFunctionsTests.cs index 64fb86e84..383a9d371 100644 --- a/test/LibRed.Engine.Tests/AddedFunctionsTests.cs +++ b/test/LibRed.Engine.Tests/AddedFunctionsTests.cs @@ -28,6 +28,10 @@ private static QueryEngine Fresh() [InlineData("Hex(255)", "FF")] [InlineData("Oct(8)", "10")] [InlineData("MonthName(1)", "January")] + // The second argument abbreviates. It was accepted by the arity table and then ignored, so this returned + // the full name — a silently wrong value rather than a refused call. + [InlineData("MonthName(1, True)", "Jan")] + [InlineData("MonthName(1, False)", "January")] [InlineData("TypeName(5)", "Long")] public void String_returning(string expr, string expected) => Assert.Equal(expected, Convert.ToString(Eval(expr))); diff --git a/test/LibRed.Engine.Tests/AggregateSurfaceTests.cs b/test/LibRed.Engine.Tests/AggregateSurfaceTests.cs new file mode 100644 index 000000000..87d08d156 --- /dev/null +++ b/test/LibRed.Engine.Tests/AggregateSurfaceTests.cs @@ -0,0 +1,42 @@ +using LibRed.Engine.Execution; +using LibRed.Engine.Planning; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// The aggregate surface is answered in three places — computes the value, +/// decides the name is an aggregate at all, and +/// declares the type it comes back as. The first two already +/// derive from ; the third is a switch beside it, whose own comment asks to be +/// kept in lock-step. This binds them, so adding an aggregate to one and not the others fails here. +/// +public class AggregateSurfaceTests +{ + [Fact] + public void Every_computed_aggregate_is_recognised_by_the_planner() + { + foreach (string name in RunningAggregate.SupportedNames) + { + Assert.True(RunningAggregate.Supports(name), $"{name} is listed but Supports says otherwise."); + Assert.True(QueryPlanner.IsAggregate(name), $"{name} computes but the planner does not treat it as an aggregate."); + } + } + + [Theory] + [InlineData(typeof(int))] + [InlineData(typeof(decimal))] + [InlineData(typeof(double))] + [InlineData(typeof(string))] + public void Every_computed_aggregate_declares_a_result_type(Type argument) + { + foreach (string name in RunningAggregate.SupportedNames) + { + // Null is what AggregateResultType returns for a name it has no case for — the silent failure + // this test exists to catch, since such a name still plans and still passes the arity check. + Assert.True( + QueryExecutor.AggregateResultType(name, argument) is not null, + $"{name} over {argument.Name} declares no result type — give it a case in AggregateResultType."); + } + } +} diff --git a/test/LibRed.Engine.Tests/AlterColumnDefaultTests.cs b/test/LibRed.Engine.Tests/AlterColumnDefaultTests.cs index 1bd9c7a9e..46e2ea142 100644 --- a/test/LibRed.Engine.Tests/AlterColumnDefaultTests.cs +++ b/test/LibRed.Engine.Tests/AlterColumnDefaultTests.cs @@ -46,4 +46,22 @@ public void Drop_default_on_a_column_without_one_is_a_noop() e.ExecuteNonQuery("INSERT INTO T (Id, V) VALUES (1, 42)"); Assert.Equal(42, Convert.ToInt32(ReadV(e, 1))); } + + // GenUniqueID() is a LONG-only default, which CREATE TABLE and ADD COLUMN both refuse on any other type. + // Both ALTER forms wrote the property straight through instead, so the default was persisted and then + // evaluated — putting a random Int32 into a Text column on the next omit-insert. + [Fact] + public void GenUniqueID_is_refused_as_a_default_on_a_text_column() + { + var e = Fresh(); + e.ExecuteNonQuery("ALTER TABLE T ADD COLUMN S TEXT(10)"); + + Assert.Throws(() => + e.ExecuteNonQuery("ALTER TABLE T ALTER COLUMN S TEXT(10) DEFAULT GenUniqueID()")); + Assert.Throws(() => + e.ExecuteNonQuery("ALTER TABLE T ALTER COLUMN S SET DEFAULT GenUniqueID()")); + + // And the legitimate case still works, so the guard is not simply refusing everything. + e.ExecuteNonQuery("ALTER TABLE T ALTER COLUMN V SET DEFAULT GenUniqueID()"); + } } diff --git a/test/LibRed.Engine.Tests/AlterColumnTests.cs b/test/LibRed.Engine.Tests/AlterColumnTests.cs index 8d0499b5a..f77498f47 100644 --- a/test/LibRed.Engine.Tests/AlterColumnTests.cs +++ b/test/LibRed.Engine.Tests/AlterColumnTests.cs @@ -35,6 +35,30 @@ public void Widen_a_text_column_keeps_existing_data() e.ExecuteNonQuery("INSERT INTO T (K, V) VALUES (2, 'a much longer value than twenty chars')"); } + // A Memo retype edits the definition in place, as ACE's does — it once rebuilt the table and recreated the primary + // key first. A key added after another index keeps its place, and so does every other index, and the + // relationship a child table holds to it. + [Fact] + public void A_memo_retype_keeps_the_indexes_in_their_order() + { + var (e, db) = Fresh(); + e.ExecuteNonQuery("CREATE TABLE P ( ID LONG NOT NULL, A LONG, B TEXT(20) )"); + e.ExecuteNonQuery("CREATE INDEX IX_A ON P (A)"); + e.ExecuteNonQuery("ALTER TABLE P ADD CONSTRAINT PK_P PRIMARY KEY (ID)"); + e.ExecuteNonQuery("CREATE INDEX IX_B ON P (B)"); + e.ExecuteNonQuery("CREATE TABLE C ( CID LONG, PID LONG, CONSTRAINT FK_C_P FOREIGN KEY (PID) REFERENCES P (ID) )"); + e.ExecuteNonQuery("INSERT INTO P (ID, A, B) VALUES (1, 10, 'x')"); + + e.ExecuteNonQuery("ALTER TABLE P ALTER COLUMN B MEMO"); + + var p = db.Catalog.FindTable("P")!; + Assert.Equal(["IX_A", "PK_P", "IX_B"], p.Indexes.Select(i => i.Name)); + Assert.True(p.Indexes[1].IsPrimaryKey); + Assert.Equal(["A", "ID", "B"], p.RealIndexes.Select(i => i.Columns.Single().Column.Name)); + Assert.Contains(db.Catalog.Relationships, r => r.Name == "FK_C_P" && r.ReferencedTable == "P"); + Assert.Equal("x", e.ExecuteQuery("SELECT B FROM P WHERE ID = 1").Rows.Single()[0]); + } + [Fact] public void Narrow_a_text_column_is_a_metadata_change() { @@ -78,6 +102,54 @@ public void Change_text_to_number_converts_and_keeps_other_columns() Assert.Equal([("1", "one", 42), ("2", "two", 100)], rows); } + // Recreating a table writes a property blob built from the column specs — DefaultValue, Required, the + // calculated triple and the CHECK constraints — which is everything LibRed models and nothing else, and + // the two permission rows a brand-new table gets. Access keeps far more in that blob (ValidationRule, + // Format, Description, AllowZeroLength, the table's own properties) and a secured database grants to more + // accounts than two, so both are carried across verbatim. The assertion is byte-identity rather than a + // list of the properties LibRed happens to know the names of, because the ones at risk are the others. + [Fact] + public void A_rebuild_keeps_the_tables_property_blob_and_permission_rows() + { + var (e, db) = Fresh(); + byte[] properties = PropertyBlobOf(db, "Order Details"); + var permissions = PermissionsOf(db, "Order Details"); + Assert.NotEmpty(properties); + Assert.NotEmpty(permissions); + + e.ExecuteNonQuery("ALTER TABLE [Order Details] ALTER COLUMN Quantity LONG"); // SHORT -> LONG: a rebuild + + Assert.Equal(properties, PropertyBlobOf(db, "Order Details")); + Assert.Equal(permissions, PermissionsOf(db, "Order Details")); + } + + /// The table's extended-property blob, looked up afresh each time — a rebuild moves the table to + /// a new definition page, which is the id the blob is filed under. + private static byte[] PropertyBlobOf(JetDatabase db, string table) + { + int id = db.Catalog.FindTable(table)!.DefinitionPage; + var objects = db.OpenTable("MSysObjects"); + int idIndex = objects.Definition.FindColumn("Id")!.Index; + int lvProp = objects.Definition.FindColumn("LvProp")!.Index; + return objects.Rows() + .Where(r => r[idIndex] is not null && Convert.ToInt32(r[idIndex]) == id) + .Select(r => r[lvProp] as byte[] ?? []).FirstOrDefault() ?? []; + } + + /// The table's MSysACEs grants, as account + mask pairs. + private static List PermissionsOf(JetDatabase db, string table) + { + int id = db.Catalog.FindTable(table)!.DefinitionPage; + var aces = db.OpenTable("MSysACEs"); + int idIndex = aces.Definition.FindColumn("ObjectId")!.Index; + int sid = aces.Definition.FindColumn("SID")!.Index; + int acm = aces.Definition.FindColumn("ACM")!.Index; + return [.. aces.Rows() + .Where(r => r[idIndex] is not null && Convert.ToInt32(r[idIndex]) == id) + .Select(r => $"{Convert.ToHexString(r[sid] as byte[] ?? [])}:{Convert.ToInt32(r[acm])}") + .Order(StringComparer.Ordinal)]; + } + [Fact] public void Rewrite_preserves_primary_key_uniqueness() { diff --git a/test/LibRed.Engine.Tests/AlterTableAddForeignKeyTests.cs b/test/LibRed.Engine.Tests/AlterTableAddForeignKeyTests.cs index fdf12a809..d7287d3ad 100644 --- a/test/LibRed.Engine.Tests/AlterTableAddForeignKeyTests.cs +++ b/test/LibRed.Engine.Tests/AlterTableAddForeignKeyTests.cs @@ -39,6 +39,79 @@ public void Add_foreign_key_to_existing_table() finally { TemporaryDatabase.Delete(path); } } + // The relationship is refused when the rows already in the table break it, and nothing of it is left behind: + // an orphan inserted afterwards is accepted. The same rows ACE refuses and accepts (probed over OLE DB): MATCH + // FULL, so a composite key that is partly null is refused and one entirely null is not, and the ON DELETE + // action makes no difference. + [Theory] + [InlineData("10, 1, NULL", "(ParentId) REFERENCES Parents (Id)", true)] + [InlineData("10, NULL, NULL", "(ParentId) REFERENCES Parents (Id)", true)] + [InlineData("10, 2, NULL", "(ParentId) REFERENCES Parents (Id)", false)] + [InlineData("10, 2, NULL", "(ParentId) REFERENCES Parents", false)] + [InlineData("10, 2, NULL", "(ParentId) REFERENCES Parents (Id) ON DELETE CASCADE", false)] + [InlineData("10, 2, NULL", "(ParentId) REFERENCES Parents (Id) ON DELETE SET NULL", false)] + [InlineData("10, 1, 1", "(ParentId, Code) REFERENCES Parents (Id, Code)", true)] + [InlineData("10, NULL, NULL", "(ParentId, Code) REFERENCES Parents (Id, Code)", true)] + [InlineData("10, 1, NULL", "(ParentId, Code) REFERENCES Parents (Id, Code)", false)] + [InlineData("10, 1, 2", "(ParentId, Code) REFERENCES Parents (Id, Code)", false)] + public void Add_foreign_key_checks_the_rows_already_in_the_table(string row, string constraint, bool accepted) + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY, Code LONG, CONSTRAINT UQ_Code UNIQUE (Id, Code))"); + e.ExecuteNonQuery("CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG, Code LONG)"); + e.ExecuteNonQuery("INSERT INTO Parents (Id, Code) VALUES (1, 1)"); + e.ExecuteNonQuery($"INSERT INTO Children (Id, ParentId, Code) VALUES ({row})"); + + const string add = "ALTER TABLE Children ADD CONSTRAINT FK_Child FOREIGN KEY "; + if (accepted) + { + e.ExecuteNonQuery(add + constraint); + Assert.Single(db.Catalog.ForeignKeysOf("Children")); + Assert.Throws( + () => e.ExecuteNonQuery("INSERT INTO Children (Id, ParentId, Code) VALUES (20, 3, 3)")); + } + else + { + var refused = Assert.Throws(() => e.ExecuteNonQuery(add + constraint)); + Assert.Equal( + "Cannot create relationships to enforce referential integrity. Existing data in table " + + "'Children' violates referential integrity rules in table 'Parents'.", refused.Message); + Assert.Empty(db.Catalog.ForeignKeysOf("Children")); + e.ExecuteNonQuery("INSERT INTO Children (Id, ParentId, Code) VALUES (20, 3, 3)"); + } + } + finally { TemporaryDatabase.Delete(path); } + } + + // A self-reference is checked against the table's own rows: a row may point at itself or at another row that + // is there, and not at one that isn't. + [Theory] + [InlineData("(1, 1), (2, 1)", true)] + [InlineData("(1, NULL), (2, 3)", false)] + public void Add_self_referencing_foreign_key_checks_the_rows_already_in_the_table(string rows, bool accepted) + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE TABLE Staff (EmployeeID LONG PRIMARY KEY, ReportsTo LONG)"); + e.ExecuteNonQuery($"INSERT INTO Staff (EmployeeID, ReportsTo) VALUES {rows}"); + + const string add = "ALTER TABLE Staff ADD CONSTRAINT FK_Staff_Staff FOREIGN KEY (ReportsTo) REFERENCES Staff (EmployeeID)"; + if (accepted) + e.ExecuteNonQuery(add); + else + Assert.Throws(() => e.ExecuteNonQuery(add)); + Assert.Equal(accepted ? 1 : 0, db.Catalog.ForeignKeysOf("Staff").Count()); + } + finally { TemporaryDatabase.Delete(path); } + } + // A self-referencing foreign key (Northwind's Employees.ReportsTo → Employees.EmployeeID). [Fact] public void Add_self_referencing_foreign_key() diff --git a/test/LibRed.Engine.Tests/AuditRegressionTests.cs b/test/LibRed.Engine.Tests/AuditRegressionTests.cs deleted file mode 100644 index b9eaeb26b..000000000 --- a/test/LibRed.Engine.Tests/AuditRegressionTests.cs +++ /dev/null @@ -1,128 +0,0 @@ -using LibRed.Engine; -using Xunit; - -namespace LibRed.Engine.Tests; - -// Engine-side regressions from the spec-vs-code audit. Each is a check that one statement path applied and -// its sibling did not — the shape that produced most of the findings. -public class AuditRegressionTests : TempDatabaseTest -{ - private static QueryEngine Fresh() - { - string path = TemporaryDatabase.CopyPath( - Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "audit-engine-"); - return new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); - } - - // GenUniqueID() is a LONG-only default; ACE rejects it elsewhere at DDL time. CREATE TABLE and ADD COLUMN - // validated it, but both ALTER forms wrote the property straight through — so the default was persisted - // and then evaluated, putting a random Int32 into a Text column on the next omit-insert. - [Fact] - public void Alter_column_default_rejects_GenUniqueID_on_a_text_column() - { - var engine = Fresh(); - engine.ExecuteNonQuery("CREATE TABLE T (K LONG, V TEXT(10))"); - - Assert.Throws(() => - engine.ExecuteNonQuery("ALTER TABLE T ALTER COLUMN V TEXT(10) DEFAULT GenUniqueID()")); - Assert.Throws(() => - engine.ExecuteNonQuery("ALTER TABLE T ALTER COLUMN V SET DEFAULT GenUniqueID()")); - - // And the legitimate case still works, so the guard is not simply refusing everything. - engine.ExecuteNonQuery("ALTER TABLE T ALTER COLUMN K SET DEFAULT GenUniqueID()"); - } - - // A cascade rewrites a child row, so it owes that row the same invariants an UPDATE does. It applied - // none: ON DELETE SET NULL would write NULL into a Required column, and Table.Update carries no - // enforcement of its own, so nothing underneath caught it either. - [Fact] - public void On_delete_set_null_refuses_to_null_a_required_child_column() - { - var engine = Fresh(); - engine.ExecuteNonQuery("CREATE TABLE P (A LONG CONSTRAINT PKP PRIMARY KEY)"); - engine.ExecuteNonQuery("CREATE TABLE C (K LONG CONSTRAINT PKC PRIMARY KEY, B LONG NOT NULL)"); - engine.ExecuteNonQuery("ALTER TABLE C ADD CONSTRAINT FK FOREIGN KEY (B) REFERENCES P (A) ON DELETE SET NULL"); - engine.ExecuteNonQuery("INSERT INTO P (A) VALUES (1)"); - engine.ExecuteNonQuery("INSERT INTO C (K, B) VALUES (10, 1)"); - - Assert.ThrowsAny(() => engine.ExecuteNonQuery("DELETE FROM P WHERE A = 1")); - - // The child row is intact, not half-nulled. - Assert.Equal(1, Convert.ToInt32(engine.ExecuteQuery("SELECT B FROM C WHERE K = 10").Rows.Single()[0])); - } - - // ON UPDATE CASCADE can drive two children onto the same unique key — the collision the UPDATE path - // explicitly guards against, reached by a statement that never names the child table. - [Fact] - public void On_update_cascade_refuses_to_create_a_duplicate_child_key() - { - var engine = Fresh(); - engine.ExecuteNonQuery("CREATE TABLE P (A LONG CONSTRAINT PKP PRIMARY KEY)"); - engine.ExecuteNonQuery("CREATE TABLE C (K LONG CONSTRAINT PKC PRIMARY KEY, B LONG)"); - engine.ExecuteNonQuery("CREATE UNIQUE INDEX UXB ON C (B)"); - engine.ExecuteNonQuery("ALTER TABLE C ADD CONSTRAINT FK FOREIGN KEY (B) REFERENCES P (A) ON UPDATE CASCADE"); - engine.ExecuteNonQuery("INSERT INTO P (A) VALUES (1)"); - engine.ExecuteNonQuery("INSERT INTO P (A) VALUES (2)"); - engine.ExecuteNonQuery("INSERT INTO C (K, B) VALUES (10, 1)"); - engine.ExecuteNonQuery("INSERT INTO C (K, B) VALUES (20, 2)"); - - // Moving parent 1 onto 2 would cascade child 10 onto child 20's unique key. - Assert.ThrowsAny(() => engine.ExecuteNonQuery("UPDATE P SET A = 2 WHERE A = 1")); - } - - // ViewExpander recursed with no visited-set and the binder never validated a CREATE VIEW body's sources, - // so a self-referencing view was accepted and then selected from — a StackOverflowException, which .NET - // cannot catch: it takes the host process down rather than failing the statement. - [Fact] - public void A_self_referencing_view_is_reported_rather_than_overflowing_the_stack() - { - var engine = Fresh(); - engine.ExecuteNonQuery("CREATE TABLE T (K LONG)"); - engine.ExecuteNonQuery("CREATE VIEW V AS SELECT * FROM T"); - engine.ExecuteNonQuery("DROP VIEW V"); - engine.ExecuteNonQuery("CREATE VIEW V AS SELECT * FROM V"); - - var error = Assert.Throws(() => engine.ExecuteQuery("SELECT * FROM V")); - Assert.Contains("defined in terms of itself", error.Message); - } - - // Two views that reference each other — the same cycle one hop longer, which a naive "is this view the - // one we started from" check would miss. - [Fact] - public void A_mutually_recursive_view_pair_is_reported_too() - { - var engine = Fresh(); - engine.ExecuteNonQuery("CREATE TABLE T (K LONG)"); - engine.ExecuteNonQuery("CREATE VIEW V1 AS SELECT * FROM T"); - engine.ExecuteNonQuery("CREATE VIEW V2 AS SELECT * FROM V1"); - engine.ExecuteNonQuery("DROP VIEW V1"); - engine.ExecuteNonQuery("CREATE VIEW V1 AS SELECT * FROM V2"); - - Assert.Throws(() => engine.ExecuteQuery("SELECT * FROM V1")); - } - - // A view referenced twice in one query is NOT a cycle, and must still expand. - [Fact] - public void The_same_view_used_twice_in_one_query_still_expands() - { - var engine = Fresh(); - engine.ExecuteNonQuery("CREATE TABLE T (K LONG)"); - engine.ExecuteNonQuery("INSERT INTO T (K) VALUES (1)"); - engine.ExecuteNonQuery("CREATE VIEW V AS SELECT * FROM T"); - - var result = engine.ExecuteQuery("SELECT A.K FROM V AS A INNER JOIN V AS B ON A.K = B.K"); - Assert.Equal(1, Convert.ToInt32(result.Rows.Single()[0])); - } - - // MonthName's second argument was accepted by the arity table and then ignored, so MonthName(1, True) - // returned "January" where ACE returns "Jan" — a silently wrong value, and the arity check that would - // have caught a stray argument is exactly what let it through. - [Fact] - public void MonthName_honours_its_abbreviate_argument() - { - var engine = Fresh(); - Assert.Equal("January", engine.ExecuteQuery("SELECT MonthName(1)").Rows.Single()[0]); - Assert.Equal("Jan", engine.ExecuteQuery("SELECT MonthName(1, True)").Rows.Single()[0]); - Assert.Equal("January", engine.ExecuteQuery("SELECT MonthName(1, False)").Rows.Single()[0]); - } -} diff --git a/test/LibRed.Engine.Tests/CaseExpressionTests.cs b/test/LibRed.Engine.Tests/CaseExpressionTests.cs index f34afbaf8..f465b5b80 100644 --- a/test/LibRed.Engine.Tests/CaseExpressionTests.cs +++ b/test/LibRed.Engine.Tests/CaseExpressionTests.cs @@ -168,8 +168,7 @@ public void Numeric_branches_widen_to_the_larger_type() => Assert.Equal(typeof(double), ColumnType("CASE WHEN 1 = 1 THEN 1 ELSE 2.5E0 END")); [Fact] - public void Irreconcilable_branches_declare_nothing() - // A string arm and a numeric arm have no common type; declaring one would be a guess, so the column - // stays untyped rather than claiming something wrong. - => Assert.Equal(typeof(object), ColumnType("CASE WHEN 1 = 1 THEN 'a' ELSE 1 END")); + public void Text_beside_a_number_declares_text() + // A string arm beside a numeric one is a Mixed choice, which declares text as ACE's IIF does. + => Assert.Equal(typeof(string), ColumnType("CASE WHEN 1 = 1 THEN 'a' ELSE 1 END")); } diff --git a/test/LibRed.Engine.Tests/CoalesceTests.cs b/test/LibRed.Engine.Tests/CoalesceTests.cs index c2c801d9d..85b278cbf 100644 --- a/test/LibRed.Engine.Tests/CoalesceTests.cs +++ b/test/LibRed.Engine.Tests/CoalesceTests.cs @@ -111,7 +111,8 @@ public void A_null_argument_does_not_erase_the_declared_type() public void Numeric_arguments_widen_to_the_larger_type() => Assert.Equal(typeof(double), ColumnType("COALESCE(1, 2.5E0)")); + // Text beside a number is a Mixed choice, which declares text as ACE's IIF does. [Fact] - public void Irreconcilable_arguments_declare_nothing() - => Assert.Equal(typeof(object), ColumnType("COALESCE('a', 1)")); + public void Text_beside_a_number_declares_text() + => Assert.Equal(typeof(string), ColumnType("COALESCE('a', 1)")); } diff --git a/test/LibRed.Engine.Tests/CollationComparisonTests.cs b/test/LibRed.Engine.Tests/CollationComparisonTests.cs new file mode 100644 index 000000000..84691efaf --- /dev/null +++ b/test/LibRed.Engine.Tests/CollationComparisonTests.cs @@ -0,0 +1,134 @@ +using LibRed; +using LibRed.Catalog; +using LibRed.Engine; +using LibRed.Storage; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Every comparison a query makes orders and equates text in the database's collation (page-02b §3.4), and only +/// by LibRed's own weight tables, so a database answers the same on every platform. Croatian v1 against General v1 +/// is the instrument, as in ACE's own measurement: 'č' is a letter after 'c' and 'lj' one after 'l', so General has +/// 'ča' < 'cb' and 'lja' < 'lm' and Croatian the reverse. The expected orders are the ones ACE returns for the +/// same values (CollationQueryProbeTests). +/// +public class CollationComparisonTests : TempDatabaseTest +{ + private static readonly Collation GeneralV1 = Collation.General; + private static readonly Collation CroatianV1 = new(CollatingOrder.Croatian, Collation.GeneralVersion); + + private static readonly string[] Values = ["cb", "ča", "ca", "d", "lm", "lja", "l"]; + + private static readonly string[] GeneralOrder = ["ca", "ča", "cb", "d", "l", "lja", "lm"]; + private static readonly string[] CroatianOrder = ["ca", "cb", "ča", "d", "l", "lm", "lja"]; + + public static TheoryData Orders => ["general", "croatian"]; + + private static (QueryEngine Engine, string[] Order) Seeded(string order) + { + Collation collation = order == "croatian" ? CroatianV1 : GeneralV1; + string path = TemporaryDatabase.CreatePath("collation-comparison-"); + JetDatabase.Create(path, collation: collation); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path)); + e.ExecuteNonQuery("CREATE TABLE T (Id LONG, S TEXT(20))"); + e.ExecuteNonQuery("CREATE TABLE U (K TEXT(20), N LONG)"); + for (int i = 0; i < Values.Length; i++) + { + e.ExecuteNonQuery($"INSERT INTO T (Id, S) VALUES ({i}, '{Values[i]}')"); + e.ExecuteNonQuery($"INSERT INTO U (K, N) VALUES ('{Values[i].ToUpperInvariant()}', {i})"); + } + return (e, order == "croatian" ? CroatianOrder : GeneralOrder); + } + + private static string[] Read(QueryEngine e, string sql) => + [.. e.ExecuteQuery(sql).Rows.Select(r => r[0] is null ? "NULL" : Convert.ToString(r[0], System.Globalization.CultureInfo.InvariantCulture)!)]; + + [Theory] + [MemberData(nameof(Orders))] + public void Order_by_sorts_in_the_database_collation(string order) + { + (QueryEngine e, string[] expected) = Seeded(order); + Assert.Equal(expected, Read(e, "SELECT S FROM T ORDER BY S")); + Assert.Equal([.. expected.Reverse()], Read(e, "SELECT S FROM T ORDER BY S DESC")); + Assert.Equal(expected, Read(e, "SELECT S FROM T ORDER BY S & ''")); + Assert.Equal(expected.Take(3), Read(e, "SELECT TOP 3 S FROM T ORDER BY S")); + } + + [Theory] + [MemberData(nameof(Orders))] + public void Group_by_and_distinct_come_back_in_the_database_collation(string order) + { + (QueryEngine e, string[] expected) = Seeded(order); + Assert.Equal(expected, Read(e, "SELECT S FROM T GROUP BY S")); + Assert.Equal(expected, Read(e, "SELECT DISTINCT S FROM T ORDER BY S")); + Assert.Equal(expected, Read(e, "SELECT S FROM T UNION SELECT S FROM T ORDER BY 1")); + } + + [Theory] + [MemberData(nameof(Orders))] + public void Min_and_max_are_the_database_collations(string order) + { + (QueryEngine e, string[] expected) = Seeded(order); + Assert.Equal([$"{expected[0]} / {expected[^1]}"], Read(e, "SELECT MIN(S) & ' / ' & MAX(S) FROM T")); + } + + [Theory] + [MemberData(nameof(Orders))] + public void A_comparison_against_a_literal_either_way_round_uses_the_database_collation(string order) + { + (QueryEngine e, string[] expected) = Seeded(order); + string[] below = [.. Values.Where(v => Array.IndexOf(expected, v) < Array.IndexOf(expected, "cb"))]; + Assert.Equal(below, Read(e, "SELECT S FROM T WHERE S < 'cb' ORDER BY Id")); + Assert.Equal(below, Read(e, "SELECT S FROM T WHERE 'cb' > S ORDER BY Id")); + } + + [Theory] + [InlineData("general", "general")] + [InlineData("croatian", "croatian")] + public void Two_literals_compare_in_the_database_collation(string order, string answer) + { + (QueryEngine e, _) = Seeded(order); + Assert.Equal([answer], Read(e, "SELECT TOP 1 IIF('ča' < 'cb', 'general', 'croatian') FROM T")); + Assert.Equal([answer], Read(e, "SELECT TOP 1 IIF('lja' < 'lm', 'general', 'croatian') FROM T")); + } + + [Theory] + [MemberData(nameof(Orders))] + public void Text_equality_folds_case_in_joins_in_lists_and_lookups(string order) + { + (QueryEngine e, _) = Seeded(order); + // U holds each value upper-cased: an unindexed equi-join is a hash join, whose hash has to agree with '='. + Assert.Equal(Values.Length, Read(e, "SELECT T.S FROM T INNER JOIN U ON T.S = U.K").Length); + Assert.Equal(["ča", "lja"], Read(e, "SELECT S FROM T WHERE S IN ('ČA', 'LJA') ORDER BY Id")); + Assert.Equal(["ča"], Read(e, "SELECT S FROM T WHERE S IN (SELECT K FROM U WHERE N = 1)")); + } + + [Theory] + [InlineData(0)] + [InlineData(Collation.GeneralVersion)] + public void Group_by_folds_what_the_collation_folds(byte version) + { + // As ACE groups them (CollationQueryProbeTests): 'ß' is 'ss', case and trailing spaces fold, an accent + // separates — cafe, café, the three spellings of strasse, and x. + string path = TemporaryDatabase.CreatePath("collation-group-"); + JetDatabase.Create(path, collation: new Collation(CollatingOrder.General, version)); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path)); + e.ExecuteNonQuery("CREATE TABLE G (Id LONG, S TEXT(20))"); + string[] values = ["Straße", "STRASSE", "strasse ", "x", "X", "café", "cafe"]; + for (int i = 0; i < values.Length; i++) + e.ExecuteNonQuery($"INSERT INTO G (Id, S) VALUES ({i}, '{values[i]}')"); + + Assert.Equal(["1", "1", "3", "2"], Read(e, "SELECT COUNT(*) FROM G GROUP BY S")); + Assert.Equal(["4"], Read(e, "SELECT COUNT(*) FROM (SELECT DISTINCT S FROM G)")); + } + + [Fact] + public void The_comparer_for_a_collation_orders_as_that_collation() + { + JetTextComparer croatian = JetTextComparer.For(CroatianV1); + Assert.Same(croatian, JetTextComparer.For(CroatianV1)); + Assert.Equal(-1, croatian.Compare("cb", "ča")); + Assert.Equal(1, JetTextComparer.For(GeneralV1).Compare("cb", "ča")); + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/ColumnPruningTests.cs b/test/LibRed.Engine.Tests/ColumnPruningTests.cs new file mode 100644 index 000000000..1df53003a --- /dev/null +++ b/test/LibRed.Engine.Tests/ColumnPruningTests.cs @@ -0,0 +1,353 @@ +using LibRed; +using LibRed.Engine; +using LibRed.Engine.Plan; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// A table read decodes only the columns its statement names (ColumnPruning, and JoinRows for +/// UPDATE and DELETE), and an undecoded column reads as null. So what needs pinning is that nothing which reads +/// a column goes without it: every place a column can be named, every shape that passes a whole row to the +/// output, and every write — which has to rewrite, check and re-index a row from all of its values, including +/// the ones the statement never mentioned. +/// +public class ColumnPruningTests : TempDatabaseTest +{ + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "column-pruning-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + + e.ExecuteNonQuery("CREATE TABLE T (Id LONG PRIMARY KEY, Grp LONG, V LONG, Label TEXT(20), Note TEXT(50))"); + e.ExecuteNonQuery("CREATE INDEX IX_Label ON T (Label)"); + e.ExecuteNonQuery("CREATE INDEX IX_V ON T (V)"); + e.ExecuteNonQuery("CREATE TABLE U (K LONG PRIMARY KEY, X LONG, Descr TEXT(20))"); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + for (int i = 1; i <= 12; i++) + e.ExecuteNonQuery($"INSERT INTO T (Id, Grp, V, Label, Note) VALUES ({i}, {i % 3}, {i * 10}, 'L{i}', 'N{i}')"); + for (int k = 0; k <= 2; k++) + e.ExecuteNonQuery($"INSERT INTO U (K, X, Descr) VALUES ({k}, {k * 100}, 'D{k}')"); + e.ExecuteNonQuery("COMMIT"); + return e; + } + + private static List Rows(QueryEngine e, string sql) => [.. e.ExecuteQuery(sql).Rows]; + + private static int[] Ints(QueryEngine e, string sql) => [.. Rows(e, sql).Select(r => Convert.ToInt32(r[0]))]; + + private static int[] Sorted(QueryEngine e, string sql) => [.. Ints(e, sql).Order()]; + + private static object?[] RowOfT(QueryEngine e, int id) => Rows(e, $"SELECT * FROM T WHERE Id = {id}").Single(); + + // --- every place a column can be read from, each the only place that names it ----------------------- + + [Fact] + public void A_column_named_only_in_the_where_is_read() + => Assert.Equal([10, 11, 12], Sorted(Seeded(), "SELECT Id FROM T WHERE V > 90")); + + [Fact] + public void A_column_named_only_in_the_order_by_is_read() + => Assert.Equal([12, 11, 10, 9, 8, 7, 6, 5, 4, 3, 2, 1], Ints(Seeded(), "SELECT Id FROM T ORDER BY V DESC")); + + [Fact] + public void A_column_named_only_in_the_group_by_is_read() + => Assert.Equal([4, 4, 4], Ints(Seeded(), "SELECT COUNT(*) FROM T GROUP BY Grp")); + + [Fact] + public void A_column_named_only_in_an_aggregate_is_read() + => Assert.Equal([780], Ints(Seeded(), "SELECT SUM(V) FROM T")); + + [Fact] + public void A_column_named_only_in_the_having_is_read() + // Group sums: 0 → 300, 1 → 220, 2 → 260. + => Assert.Equal([0], Ints(Seeded(), "SELECT Grp FROM T GROUP BY Grp HAVING SUM(V) > 260")); + + [Fact] + public void Columns_named_only_in_a_join_are_read() + => Assert.Equal([1, 4, 7, 10], Sorted(Seeded(), "SELECT T.Id FROM T INNER JOIN U ON T.Grp = U.K WHERE U.X = 100")); + + [Fact] + public void Columns_named_only_in_a_hash_join_and_its_projection_are_read() + { + var rows = Rows(Seeded(), "SELECT T.Id, U.Descr FROM T INNER JOIN U ON T.Grp = U.X") + .Select(r => (Convert.ToInt32(r[0]), (string)r[1]!)).Order().ToArray(); + Assert.Equal([(3, "D0"), (6, "D0"), (9, "D0"), (12, "D0")], rows); + } + + [Fact] + public void An_outer_column_named_only_inside_a_correlated_subquery_is_read() + { + QueryEngine e = Seeded(); + Assert.Equal([10], Ints(e, "SELECT t.Id FROM T AS t WHERE EXISTS (SELECT 1 FROM U AS u WHERE u.X = t.V)")); + + var rows = Rows(e, "SELECT t.Id, (SELECT u.Descr FROM U AS u WHERE u.K = t.Grp) FROM T AS t WHERE t.Id <= 3") + .Select(r => (Convert.ToInt32(r[0]), (string)r[1]!)).ToArray(); + Assert.Equal([(1, "D1"), (2, "D2"), (3, "D0")], rows); + } + + // --- a join passes on only the columns its statement names, and builds its rows that narrow ----------- + + [Fact] + public void An_aggregate_over_a_join_reads_what_it_names() + { + // Group sums of V over T.Grp = U.K: 0 → 300, 1 → 220, 2 → 260. + var rows = Rows(Seeded(), "SELECT U.Descr, SUM(T.V) FROM T INNER JOIN U ON T.Grp = U.K GROUP BY U.Descr") + .Select(r => ((string)r[0]!, Convert.ToInt32(r[1]))).ToArray(); + Assert.Equal([("D0", 300), ("D1", 220), ("D2", 260)], rows); + } + + [Fact] + public void A_narrowed_left_join_pads_its_unmatched_rows() + { + // Only V = 100 meets an X; every other T row comes back with a null Descr. + var rows = Rows(Seeded(), "SELECT T.Id, U.Descr FROM T LEFT JOIN U ON T.V = U.X") + .ToDictionary(r => Convert.ToInt32(r[0]), r => r[1]); + Assert.Equal(12, rows.Count); + Assert.Equal("D1", rows[10]); + Assert.All(rows.Where(p => p.Key != 10), p => Assert.Null(p.Value)); + } + + [Fact] + public void A_narrowed_full_join_pads_either_side() + { + // U rows 0 and 2 (X 0 and 200) meet no V, so they come back with a null Id; T row 10 meets K 1. + var rows = Rows(Seeded(), "SELECT T.Id, U.K FROM T FULL JOIN U ON T.V = U.X") + .Select(r => (r[0] is null ? (int?)null : Convert.ToInt32(r[0]), r[1] is null ? (int?)null : Convert.ToInt32(r[1]))) + .ToList(); + Assert.Equal(14, rows.Count); + Assert.Contains((10, 1), rows); + Assert.Contains((null, 0), rows); + Assert.Contains((null, 2), rows); + } + + [Fact] + public void A_name_both_sides_share_is_still_ambiguous_when_the_join_is_narrowed() + // Every column a reference could mean is kept, so an unqualified Id still finds two. + => Assert.ThrowsAny(() => + Rows(Seeded(), "SELECT Id FROM T AS a INNER JOIN T AS b ON a.Grp = b.Grp")); + + [Fact] + public void A_column_named_only_in_an_in_subquery_is_read() + => Assert.Equal([10], Ints(Seeded(), "SELECT Id FROM T WHERE V IN (SELECT X FROM U)")); + + [Fact] + public void Columns_named_only_in_a_window_are_read() + { + var numbers = Rows(Seeded(), "SELECT Id, ROW_NUMBER() OVER (PARTITION BY Grp ORDER BY V DESC) FROM T") + .ToDictionary(r => Convert.ToInt32(r[0]), r => Convert.ToInt32(r[1])); + Assert.Equal(1, numbers[12]); + Assert.Equal(2, numbers[9]); + Assert.Equal(4, numbers[3]); + Assert.Equal(1, numbers[10]); + Assert.Equal(4, numbers[1]); + } + + [Fact] + public void A_derived_star_is_read_for_the_columns_its_outer_query_names() + => Assert.Equal([11, 12], Sorted(Seeded(), "SELECT d.Id FROM (SELECT * FROM T) AS d WHERE d.V > 100")); + + [Fact] + public void Seeks_decode_what_their_residual_and_projection_name() + { + QueryEngine e = Seeded(); + Assert.Equal([5], Ints(e, "SELECT Id FROM T WHERE Label = 'L5' AND V = 50")); + Assert.Equal(["L3", "L4", "L5"], + Rows(e, "SELECT Label FROM T WHERE V BETWEEN 30 AND 50").Select(r => (string)r[0]!).Order().ToArray()); + } + + [Fact] + public void A_bare_count_reads_no_column_and_still_counts_every_row() + => Assert.Equal([12], Ints(Seeded(), "SELECT COUNT(*) FROM T")); + + // --- shapes that pass whole rows to the output, which must not be pruned --------------------------- + + [Theory] + [InlineData("SELECT Grp FROM (SELECT DISTINCT * FROM T) AS d")] + [InlineData("SELECT Grp FROM (SELECT * FROM T UNION SELECT * FROM T) AS d")] + [InlineData("SELECT Grp FROM (SELECT * FROM T INTERSECT SELECT * FROM T) AS d")] + [InlineData("SELECT Grp FROM (SELECT * FROM T EXCEPT SELECT * FROM T WHERE Grp < 0) AS d")] + public void An_outer_projection_preserves_columns_used_for_inner_duplicate_elimination(string sql) + => Assert.Equal([0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2], Sorted(Seeded(), sql)); + + // U's columns reach the outer Id by position, under names the statement never writes. + [Fact] + public void A_union_all_preserves_positionally_matched_columns_with_different_names() + => Assert.Equal([0, 1, 1, 2], Sorted(Seeded(), + "SELECT Id FROM (SELECT Id, V, Grp FROM T WHERE Id = 1 UNION ALL SELECT * FROM U) AS d")); + + [Fact] + public void Select_star_returns_every_column() + => Assert.Equal([5, 2, 50, "L5", "N5"], RowOfT(Seeded(), 5).Select(Normalise)); + + [Fact] + public void A_qualified_star_returns_every_column_of_its_table() + { + object?[] row = Rows(Seeded(), "SELECT T.*, U.Descr FROM T INNER JOIN U ON T.Grp = U.K WHERE T.Id = 5").Single(); + Assert.Equal([5, 2, 50, "L5", "N5", "D2"], row.Select(Normalise)); + } + + [Fact] + public void A_union_of_stars_returns_every_column() + { + var rows = Rows(Seeded(), "SELECT * FROM T WHERE Id <= 2 UNION SELECT * FROM T WHERE Id <= 2"); + Assert.Equal(["N1", "N2"], rows.Select(r => (string)r[4]!).Order()); + } + + [Fact] + public void Distinctrow_still_tells_underlying_rows_apart_by_columns_it_does_not_output() + { + // Grp 0 joins three U rows, 1 two, 2 one: 24 joined rows over the 12 of T. DISTINCTROW keeps one per T + // row. Deciding that from Grp alone — all a pruned read would have — collapses them to three. + Assert.Equal(24, Rows(Seeded(), "SELECT T.Grp FROM T INNER JOIN U ON U.K >= T.Grp").Count); + Assert.Equal(12, Rows(Seeded(), "SELECT DISTINCTROW T.Grp FROM T INNER JOIN U ON U.K >= T.Grp").Count); + } + + // --- the plan: which reads are pruned, and to what --------------------------------------------------- + + private static IEnumerable Nodes(PlanNode node) => [node, .. node.Children.SelectMany(Nodes)]; + + private static IReadOnlySet? DecodeOf(PlanNode node) => node switch + { + ScanNode s => s.Decode, + IndexSeekNode s => s.Decode, + IndexRangeSeekNode s => s.Decode, + _ => throw new InvalidOperationException($"{node.GetType().Name} is not a table read."), + }; + + private static IEnumerable Reads(PlanNode plan) => + Nodes(plan).Where(n => n is ScanNode or IndexSeekNode or IndexRangeSeekNode); + + [Fact] + public void A_projection_prunes_its_scan_to_the_names_it_reads() + { + IReadOnlySet? decode = DecodeOf(Reads(Seeded().PlanFor("SELECT Id, Label FROM T WHERE Grp = 1")).Single()); + Assert.NotNull(decode); + Assert.Contains("Id", decode); + Assert.Contains("Label", decode); + Assert.Contains("Grp", decode); + Assert.DoesNotContain("Note", decode); + } + + private static IReadOnlySet? KeepOf(PlanNode join) => join switch + { + JoinNode j => j.Keep, + HashJoinNode h => h.Keep, + _ => throw new InvalidOperationException($"{join.GetType().Name} is not a join."), + }; + + [Fact] + public void A_join_under_an_aggregate_keeps_the_names_it_reads() + { + PlanNode join = Nodes(Seeded().PlanFor( + "SELECT U.Descr, SUM(T.V) FROM T INNER JOIN U ON T.Grp = U.K GROUP BY U.Descr")) + .Single(n => n is JoinNode or HashJoinNode); + IReadOnlySet? keep = KeepOf(join); + Assert.NotNull(keep); + Assert.Superset(new HashSet(["Descr", "V", "Grp", "K"]), new HashSet(keep, StringComparer.OrdinalIgnoreCase)); + Assert.DoesNotContain("Note", keep); + } + + [Theory] + [InlineData("SELECT * FROM T INNER JOIN U ON T.Grp = U.K")] + [InlineData("SELECT T.*, U.Descr FROM T INNER JOIN U ON T.Grp = U.K")] + [InlineData("SELECT DISTINCTROW T.Grp FROM T INNER JOIN U ON U.K >= T.Grp")] + public void A_join_whose_rows_can_reach_the_output_whole_keeps_every_column(string sql) + => Assert.All(Nodes(Seeded().PlanFor(sql)).Where(n => n is JoinNode or HashJoinNode), + join => Assert.Null(KeepOf(join))); + + [Theory] + [InlineData("SELECT * FROM T")] + [InlineData("SELECT DISTINCT * FROM T")] + [InlineData("SELECT * FROM T WHERE Id = 3")] + [InlineData("SELECT DISTINCTROW T.Grp FROM T INNER JOIN U ON U.K >= T.Grp")] + public void A_read_whose_rows_can_reach_the_output_whole_is_not_pruned(string sql) + => Assert.All(Reads(Seeded().PlanFor(sql)), read => Assert.Null(DecodeOf(read))); + + // --- writes: a row is rewritten, checked and re-indexed from every value it has ----------------------- + + [Fact] + public void An_update_keeps_the_columns_it_never_names() + { + QueryEngine e = Seeded(); + Assert.Equal(4, e.ExecuteNonQuery("UPDATE T SET V = V + 1 WHERE Grp = 1")); + Assert.Equal([1, 1, 11, "L1", "N1"], RowOfT(e, 1).Select(Normalise)); + Assert.Equal([2, 2, 20, "L2", "N2"], RowOfT(e, 2).Select(Normalise)); + } + + [Fact] + public void A_joined_update_keeps_both_tables_columns() + { + QueryEngine e = Seeded(); + e.ExecuteNonQuery("UPDATE T INNER JOIN U ON T.Grp = U.K SET T.V = U.X WHERE T.Id = 4"); + Assert.Equal([4, 1, 100, "L4", "N4"], RowOfT(e, 4).Select(Normalise)); + Assert.Equal([1, 100, "D1"], Rows(e, "SELECT * FROM U WHERE K = 1").Single().Select(Normalise)); + } + + [Fact] + public void An_update_moves_the_index_entry_of_a_column_it_only_assigns() + { + // Label appears only as a SET target, so the read that finds the row does not decode it; the old index + // entry can only be removed if the row is read in full before it is rewritten. + QueryEngine e = Seeded(); + e.ExecuteNonQuery("UPDATE T SET Label = 'Z' WHERE Id = 2"); + Assert.Equal([2], Ints(e, "SELECT Id FROM T WHERE Label = 'Z'")); + Assert.Empty(Ints(e, "SELECT Id FROM T WHERE Label = 'L2'")); + Assert.Equal([2, 2, 20, "Z", "N2"], RowOfT(e, 2).Select(Normalise)); + } + + [Fact] + public void A_delete_removes_the_index_entries_of_columns_it_never_names() + { + QueryEngine e = Seeded(); + Assert.Equal(1, e.ExecuteNonQuery("DELETE FROM T WHERE V = 30")); + Assert.Empty(Ints(e, "SELECT Id FROM T WHERE Label = 'L3'")); + Assert.Equal([11], Ints(e, "SELECT COUNT(*) FROM T")); + } + + [Fact] + public void An_update_through_a_derived_table_keeps_the_columns_it_does_not_select() + { + QueryEngine e = Seeded(); + Assert.Equal(4, e.ExecuteNonQuery("UPDATE (SELECT Id, V FROM T WHERE Grp = 2) SET V = 0")); + Assert.Equal([5, 2, 0, "L5", "N5"], RowOfT(e, 5).Select(Normalise)); + } + + [Fact] + public void Cascades_and_key_checks_read_what_they_need() + { + QueryEngine e = Seeded(); + e.ExecuteNonQuery("CREATE TABLE Par (Id LONG PRIMARY KEY, Nm TEXT(20))"); + e.ExecuteNonQuery("CREATE TABLE Ch (Id LONG PRIMARY KEY, Pid LONG, Info TEXT(20), " + + "CONSTRAINT fk FOREIGN KEY (Pid) REFERENCES Par (Id) ON UPDATE CASCADE ON DELETE CASCADE)"); + e.ExecuteNonQuery("INSERT INTO Par (Id, Nm) VALUES (1, 'p1')"); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Pid, Info) VALUES (1, 1, 'c1')"); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Pid, Info) VALUES (2, 1, 'c2')"); + + // The parent check reads the key alone, and still finds a missing parent. + Assert.ThrowsAny(() => e.ExecuteNonQuery("INSERT INTO Ch (Id, Pid, Info) VALUES (3, 99, 'x')")); + + // An ON UPDATE CASCADE rewrites each child from all of its values, not just its key. + e.ExecuteNonQuery("UPDATE Par SET Id = 10 WHERE Id = 1"); + Assert.Equal([[1, 10, "c1"], [2, 10, "c2"]], + Rows(e, "SELECT * FROM Ch ORDER BY Id").Select(r => r.Select(Normalise).ToArray()).ToArray()); + + e.ExecuteNonQuery("DELETE FROM Par WHERE Id = 10"); + Assert.Empty(Rows(e, "SELECT * FROM Ch")); + } + + [Fact] + public void Insert_select_and_select_into_carry_the_columns_they_name() + { + QueryEngine e = Seeded(); + e.ExecuteNonQuery("CREATE TABLE T2 (Id LONG, Note TEXT(50))"); + e.ExecuteNonQuery("INSERT INTO T2 (Id, Note) SELECT Id, Note FROM T WHERE Grp = 0"); + Assert.Equal(["N12", "N3", "N6", "N9"], Rows(e, "SELECT Note FROM T2").Select(r => (string)r[0]!).Order()); + + e.ExecuteNonQuery("SELECT Id, Label INTO T3 FROM T WHERE Grp = 1"); + Assert.Equal(["L1", "L10", "L4", "L7"], Rows(e, "SELECT Label FROM T3").Select(r => (string)r[0]!).Order()); + } + + private static object? Normalise(object? value) => value is int or short or long or byte ? Convert.ToInt32(value) : value; +} diff --git a/test/LibRed.Engine.Tests/ColumnSizeLimitTests.cs b/test/LibRed.Engine.Tests/ColumnSizeLimitTests.cs index c21f72069..73def5d3c 100644 --- a/test/LibRed.Engine.Tests/ColumnSizeLimitTests.cs +++ b/test/LibRed.Engine.Tests/ColumnSizeLimitTests.cs @@ -6,7 +6,8 @@ namespace LibRed.Engine.Tests; // Jet/ACE caps a char/varchar column at 255 characters and a binary/varbinary column at 510 bytes (verified // vs ACE: char(255)/binary(510) accepted, char(256)/binary(511) rejected "Size of field is too long"). LibRed -// enforces the same caps at CREATE so it never writes a fixed column Access can't open. +// enforces the same caps at CREATE so it never writes a fixed column Access can't open. A bigbinary column's cap +// is 4000 bytes (bigbinary(4000) accepted, bigbinary(4001) rejected the same way). public class ColumnSizeLimitTests : TempDatabaseTest { private static QueryEngine Fresh() @@ -21,6 +22,7 @@ private static QueryEngine Fresh() [InlineData("nchar(255)")] [InlineData("binary(510)")] [InlineData("varbinary(510)")] + [InlineData("bigbinary(4000)")] public void Sizes_at_the_limit_are_accepted(string type) { var e = Fresh(); @@ -34,6 +36,7 @@ public void Sizes_at_the_limit_are_accepted(string type) [InlineData("binary(511)")] [InlineData("varbinary(511)")] [InlineData("binary(8000)")] + [InlineData("bigbinary(4001)")] public void Sizes_over_the_limit_are_rejected(string type) { var e = Fresh(); diff --git a/test/LibRed.Engine.Tests/ConversionFunctionTests.cs b/test/LibRed.Engine.Tests/ConversionFunctionTests.cs index d39c85256..3f4e6bee7 100644 --- a/test/LibRed.Engine.Tests/ConversionFunctionTests.cs +++ b/test/LibRed.Engine.Tests/ConversionFunctionTests.cs @@ -276,11 +276,14 @@ public void Time_text_may_have_a_fraction_of_a_second(string expression, string public void Only_seconds_take_a_short_fraction(string expression) => Assert.Throws(() => Scalar(expression)); + // A Variant keeps its argument's own type through an expression, and a result writes it out as text. [Theory] - [InlineData("CVAR(1)", 1)] + [InlineData("CVAR(1)", "1")] [InlineData("CVAR('abc')", "abc")] - [InlineData("CVAR(TRUE)", true)] - public void Cvar_passes_its_argument_through(string expression, object expected) => + [InlineData("CVAR(TRUE)", "-1")] + [InlineData("CVAR(1) + CVAR(1)", "2")] + [InlineData("CVAR(1) + 1", 2.0)] + public void Cvar_keeps_its_argument_and_is_written_out_as_text(string expression, object expected) => Assert.Equal(expected, Scalar(expression)); [Theory] diff --git a/test/LibRed.Engine.Tests/CorrelatedScalarAggregateTests.cs b/test/LibRed.Engine.Tests/CorrelatedScalarAggregateTests.cs index df63b14c4..4302b8a5d 100644 --- a/test/LibRed.Engine.Tests/CorrelatedScalarAggregateTests.cs +++ b/test/LibRed.Engine.Tests/CorrelatedScalarAggregateTests.cs @@ -102,6 +102,25 @@ public void A_predicate_over_the_aggregate_selects_the_right_rows() => Assert.Equal([1], Ids(Fresh(), "SELECT o.Id FROM O AS o WHERE (SELECT COUNT(*) FROM I AS i WHERE i.K = o.K) > 1")); + [Fact] + public void A_count_over_rows_that_exist_without_a_table_types_without_the_outer_row() + { + // EF's inline collection, with I standing in for its one-row #Dual: the body's rows come from a + // `SELECT COUNT(*)`, which has a row even with nothing read. Typing the scalar describes its plan with no + // outer row, so a count there that filtered those rows on o.Id asked for a column that does not exist yet. + const string sql = + """ + SELECT o.Id FROM O AS o + WHERE ( + SELECT COUNT(*) + FROM (SELECT CLNG(2) AS `Value` FROM (SELECT COUNT(*) FROM I) AS `v_0` + UNION + SELECT 999 AS `Value` FROM (SELECT COUNT(*) FROM I) AS `v_1`) AS `v` + WHERE `v`.`Value` > o.Id) = 1 + """; + Assert.Equal([2, 3, 4, 5], Ids(Fresh(), sql)); + } + [Fact] public void A_residual_predicate_still_applies() { diff --git a/test/LibRed.Engine.Tests/CreateIndexTests.cs b/test/LibRed.Engine.Tests/CreateIndexTests.cs index a82fa2de0..224905fbe 100644 --- a/test/LibRed.Engine.Tests/CreateIndexTests.cs +++ b/test/LibRed.Engine.Tests/CreateIndexTests.cs @@ -96,7 +96,7 @@ public void With_ignore_null_creates_sparse_index_and_skips_null_rows() } // A descending index: the index-data block records the column as descending (Ascending = false), - // and inserts encode reversed key bytes (IndexKeyEncoder handles the inversion). + // and inserts encode reversed key bytes (IndexKeyCodec handles the inversion). [Fact] public void Descending_index_records_direction_and_inserts() { @@ -147,4 +147,4 @@ public void Create_index_on_non_empty_table_backfills() } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/CreateProcedureTests.cs b/test/LibRed.Engine.Tests/CreateProcedureTests.cs index cf85949c5..18b03cdb8 100644 --- a/test/LibRed.Engine.Tests/CreateProcedureTests.cs +++ b/test/LibRed.Engine.Tests/CreateProcedureTests.cs @@ -231,6 +231,17 @@ public void Unsupported_procedure_body_is_rejected() "SELECT ShipperID, CompanyName INTO [ShipperCopy] FROM [Shippers] WHERE ShipperID > 1", 2)] [InlineData("INSERT INTO Shippers (CompanyName) SELECT ContactName FROM Customers WHERE Country = 'UK'", "INSERT INTO [Shippers] ([CompanyName]) SELECT ContactName FROM [Customers] WHERE Country = 'UK'", 7)] + // A make-table or append query's DISTINCT and TOP [PERCENT] are kept on the option row, and read back. + [InlineData("SELECT DISTINCT Country INTO CountryCopy FROM Customers", + "SELECT DISTINCT Country INTO [CountryCopy] FROM [Customers]", 21)] + [InlineData("SELECT TOP 5 CompanyName INTO CustomerCopy FROM Customers", + "SELECT TOP 5 CompanyName INTO [CustomerCopy] FROM [Customers]", 5)] + [InlineData("SELECT TOP 10 PERCENT CompanyName INTO CustomerCopy FROM Customers", + "SELECT TOP 10 PERCENT CompanyName INTO [CustomerCopy] FROM [Customers]", 10)] + [InlineData("INSERT INTO Shippers (CompanyName) SELECT DISTINCT Country FROM Customers WHERE Country = 'UK'", + "INSERT INTO [Shippers] ([CompanyName]) SELECT DISTINCT Country FROM [Customers] WHERE Country = 'UK'", 1)] + [InlineData("UPDATE Customers SET City = 'X' WHERE Country = 'UK' WITH OWNERACCESS OPTION", + "UPDATE [Customers] SET [City] = 'X' WHERE Country = 'UK' WITH OWNERACCESS OPTION", 7)] public void Action_query_body_round_trips_through_the_file(string body, string expected, int affected) { string path = Fresh(); @@ -241,7 +252,7 @@ public void Action_query_body_round_trips_through_the_file(string body, string e using (var db = JetDatabase.Open(path, readOnly: false)) // fresh open: read from the file { - Assert.Equal(expected, db.Catalog.ActionQueries["P"].Sql); + Assert.Equal(expected, db.Catalog.FindQuery("P")!.Sql); Assert.Equal(affected, new QueryEngine(db).ExecuteNonQuery("EXECUTE [P]")); } } @@ -267,7 +278,7 @@ public void A_written_action_querys_parameters_bind_when_it_is_executed() Assert.Equal( "PARAMETERS [pCity] TEXT(50), [pCountry] TEXT(20); " + "UPDATE [Customers] SET [City] = pCity WHERE Country = pCountry", - db.Catalog.ActionQueries["ByCountry"].Sql); + db.Catalog.FindQuery("ByCountry")!.Sql); Assert.Equal(7, engine.ExecuteNonQuery("EXECUTE [ByCountry] 'Ankh-Morpork', 'UK'")); Assert.Equal( @@ -292,4 +303,4 @@ public void Procedure_name_colliding_with_an_object_throws() } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/CreateTableDefaultTests.cs b/test/LibRed.Engine.Tests/CreateTableDefaultTests.cs index 20a565771..8d2c08304 100644 --- a/test/LibRed.Engine.Tests/CreateTableDefaultTests.cs +++ b/test/LibRed.Engine.Tests/CreateTableDefaultTests.cs @@ -188,6 +188,29 @@ public void Duplicate_table_name_throws() finally { TemporaryDatabase.Delete(path); } } + // A table shares its name space with the queries and linked tables of the Tables container, and with nothing + // else — ACE's rule, and its message. A relationship's name is free for a table. + [Fact] + public void Table_name_colliding_with_a_query_throws_but_a_relationship_name_is_free() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE VIEW `WidgetView` AS SELECT `ShipperID` FROM `Shippers`"); + var ex = Assert.Throws(() => + e.ExecuteNonQuery("CREATE TABLE `widgetview` (`Id` INTEGER)")); + Assert.Equal("Table 'widgetview' already exists.", ex.Message); + + e.ExecuteNonQuery("CREATE TABLE `Widget` (`Id` INTEGER PRIMARY KEY, `ShipperID` INTEGER, " + + "CONSTRAINT `WidgetShipper` FOREIGN KEY (`ShipperID`) REFERENCES `Shippers` (`ShipperID`))"); + e.ExecuteNonQuery("CREATE TABLE `WidgetShipper` (`Id` INTEGER)"); + Assert.Contains(db.Catalog.Tables, t => t.Name == "WidgetShipper"); + } + finally { TemporaryDatabase.Delete(path); } + } + [Fact] public void Temporary_table_throws_not_supported() { diff --git a/test/LibRed.Engine.Tests/CreateViewTests.cs b/test/LibRed.Engine.Tests/CreateViewTests.cs index cc0d4f1b4..34de6803d 100644 --- a/test/LibRed.Engine.Tests/CreateViewTests.cs +++ b/test/LibRed.Engine.Tests/CreateViewTests.cs @@ -42,13 +42,31 @@ public void A_from_less_body_is_stored_and_queryable(string sql) using (var db = JetDatabase.Open(path)) // fresh open: read from the file { - Assert.Equal("SELECT 1 AS [n]", db.Catalog.Views["Const"]); + Assert.Equal("SELECT 1 AS [n]", db.Catalog.FindQuery("Const")!.Sql); Assert.Equal(1, new QueryEngine(db).ExecuteQuery("SELECT `n` FROM `Const`").Rows.Single()[0]); } } finally { TemporaryDatabase.Delete(path); } } + // CREATE VIEW drops a leading space from the name, where CREATE PROCEDURE refuses one — both measured against + // ACE (RenameNameValidationAccessTests). + [Fact] + public void A_leading_space_is_dropped_from_a_view_name_and_refused_in_a_procedure_name() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE VIEW [ Spaced] AS SELECT 1 AS [n]"); + Assert.Contains("Spaced", db.Catalog.Queries.Keys); + + Assert.Throws(() => e.ExecuteNonQuery("CREATE PROCEDURE [ Proc] AS SELECT 1 AS [n]")); + } + finally { TemporaryDatabase.Delete(path); } + } + // A view is read back from the file (its MSysQueries rows), reconstructed to SQL, and resolved as a // derived table when queried through LibRed's own engine. [Fact] @@ -179,9 +197,28 @@ public void View_name_colliding_with_an_object_throws() { using var db = JetDatabase.Open(path, readOnly: false); // Northwind already has a Customers table. - Assert.Throws(() => + var ex = Assert.Throws(() => new QueryEngine(db).ExecuteNonQuery("CREATE VIEW `Customers` AS SELECT `CustomerID` FROM `Customers`")); + Assert.Equal("Object 'Customers' already exists.", ex.Message); + } + finally { TemporaryDatabase.Delete(path); } + } + + // A view collides only with the tables, queries and linked tables of the Tables container, as in ACE — a + // relationship lives in another container, so its name is free for a view. + [Fact] + public void View_may_take_a_relationships_name() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE TABLE `Widget` (`Id` INTEGER PRIMARY KEY, `ShipperID` INTEGER, " + + "CONSTRAINT `WidgetShipper` FOREIGN KEY (`ShipperID`) REFERENCES `Shippers` (`ShipperID`))"); + e.ExecuteNonQuery("CREATE VIEW `WidgetShipper` AS SELECT `Id` FROM `Widget`"); + Assert.Empty(e.ExecuteQuery("SELECT * FROM `WidgetShipper`").Rows); } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/DdlDmlTests.cs b/test/LibRed.Engine.Tests/DdlDmlTests.cs index ab3b91b73..88a5b0c52 100644 --- a/test/LibRed.Engine.Tests/DdlDmlTests.cs +++ b/test/LibRed.Engine.Tests/DdlDmlTests.cs @@ -38,7 +38,8 @@ public void Creating_a_column_of_a_newer_type_raises_the_file_format( Assert.Equal(expectedByte, db.DefinitionPage.JetVersion); // page 0 was re-read, not left stale } - Assert.Equal(expectedByte, VersionByte(path)); // and it reached the file + using (var reopened = JetDatabase.Open(path)) + Assert.Equal(expected, reopened.Format.Version); // and it reached the file } finally { TemporaryDatabase.Delete(path); } } @@ -59,7 +60,8 @@ public void Altering_a_column_to_bigint_raises_the_file_format() Assert.Equal(JetVersion.Version16_2016, db.Format.Version); } - Assert.Equal(0x05, VersionByte(path)); + using (var reopened = JetDatabase.Open(path)) + Assert.Equal(JetVersion.Version16_2016, reopened.Format.Version); } finally { TemporaryDatabase.Delete(path); } } @@ -87,18 +89,12 @@ public void A_failed_statement_does_not_leave_the_format_raised() Assert.Equal(0x02, db.DefinitionPage.JetVersion); } - Assert.Equal(0x02, VersionByte(path)); + using (var reopened = JetDatabase.Open(path)) + Assert.Equal(JetVersion.Version12_2007, reopened.Format.Version); } finally { TemporaryDatabase.Delete(path); } } - private static byte VersionByte(string path) - { - using var stream = File.OpenRead(path); - stream.Seek(0x14, SeekOrigin.Begin); - return (byte)stream.ReadByte(); - } - // BIGINT written by LibRed rather than read from an ACE fixture, including through an index so the key // encoder runs on our own writes. Both extremes and both signs: the key transform is a sign-bit flip, so // positives alone would pass against almost any encoding. @@ -174,10 +170,9 @@ public void Datetime2_round_trips_through_libred_on_the_ace17_format() string path = CopyToTemp(); try { - SetVersionByte(path, 0x06); - using (var db = JetDatabase.Open(path, readOnly: false)) { + db.EnsureFormatAtLeast(JetVersion.Version17_2019); var e = new QueryEngine(db); e.ExecuteNonQuery("CREATE TABLE `E` (`Id` INTEGER PRIMARY KEY, `V` DATETIME2 NULL)"); foreach ((int id, DateTime? value) in cases) @@ -208,8 +203,8 @@ public void Datetime2_values_within_one_millisecond_stay_distinct() string path = CopyToTemp(); try { - SetVersionByte(path, 0x06); using var db = JetDatabase.Open(path, readOnly: false); + db.EnsureFormatAtLeast(JetVersion.Version17_2019); var e = new QueryEngine(db); e.ExecuteNonQuery("CREATE TABLE `E` (`Id` INTEGER PRIMARY KEY, `V` DATETIME2 NULL)"); foreach ((int id, int ticks) in new[] { (1, 3), (2, 1), (3, 2) }) @@ -225,15 +220,6 @@ public void Datetime2_values_within_one_millisecond_stay_distinct() finally { TemporaryDatabase.Delete(path); } } - /// Raises a copied file to the ACE 17 format. Page 0 offset 0x14 is the entire upgrade — see - /// docs/format/page-00-database.md and AceDateTime2UpgradeTests. - private static void SetVersionByte(string path, byte version) - { - using var stream = new FileStream(path, FileMode.Open, FileAccess.Write); - stream.Seek(0x14, SeekOrigin.Begin); - stream.WriteByte(version); - } - // Creating at a chosen format, rather than upgrading someone else's file. The default stays ACE 12 so an // ordinary database keeps opening in every Access from 2007; asking for ACE 17 up front is how you get a // DATETIME2 database without the file ever having been at an older version. @@ -253,7 +239,8 @@ public void A_natively_created_database_is_stamped_at_the_requested_format( try { LibRed.Data.LibRedConnection.CreateDatabase($"Data Source={path}", version: version); - Assert.Equal(createdByte, VersionByte(path)); + using (var created = JetDatabase.Open(path)) + Assert.Equal(createdByte, (byte)created.Format.Version); using (var db = JetDatabase.Open(path, readOnly: false)) { @@ -265,7 +252,8 @@ public void A_natively_created_database_is_stamped_at_the_requested_format( Assert.Equal(value, e.ExecuteQuery("SELECT `V` FROM `E`").Rows.Single()[0]); } - Assert.Equal(afterDatetime2, VersionByte(path)); + using (var reopened = JetDatabase.Open(path)) + Assert.Equal(afterDatetime2, (byte)reopened.Format.Version); } finally { TemporaryDatabase.Delete(path); } } @@ -530,4 +518,4 @@ public void Create_without_primary_key_is_allowed() } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/DeleteCompactionTests.cs b/test/LibRed.Engine.Tests/DeleteCompactionTests.cs index 12746afc7..b443f52dd 100644 --- a/test/LibRed.Engine.Tests/DeleteCompactionTests.cs +++ b/test/LibRed.Engine.Tests/DeleteCompactionTests.cs @@ -4,11 +4,11 @@ namespace LibRed.Engine.Tests; // Deleting several rows from one page has to leave every surviving row readable. // -// Reclaiming a deleted row's space (RowInserter.ReclaimRow) closes the gap by sliding the rows below it up +// Reclaiming a deleted row's space (DataPage.ReclaimRow) closes the gap by sliding the rows below it up // and turning the emptied slot into a zero-length tombstone. The first version decided which slots to move // by comparing offsets, which leaves behind a tombstone sitting at exactly the deleted row's offset — it -// then absorbs that row's length and starves the next live row to zero, so a later scan fails with "Row is -// too short to be an inline record". Slot offsets are non-increasing with slot index, so the rows below are +// then absorbs that row's length and starves the next live row to zero, so a later scan fails reading it as +// a row too short for its own column count. Slot offsets are non-increasing with slot index, so the rows below are // simply the LATER slots; moving by index is what makes tombstones travel with them. // // Single deletes could not catch it: the two rules agree until a tombstone is already on the page. @@ -77,4 +77,4 @@ public void Churn_reuses_the_reclaimed_space() } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/DerivedColumnListTests.cs b/test/LibRed.Engine.Tests/DerivedColumnListTests.cs new file mode 100644 index 000000000..e3e3bba88 --- /dev/null +++ b/test/LibRed.Engine.Tests/DerivedColumnListTests.cs @@ -0,0 +1,75 @@ +using System.Globalization; +using LibRed.Sql.Parsing; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// A derived table's column list, the standard's (query) AS t(a, b): it names the table's columns in order, a +/// table value constructor's included, as SQL Server and PostgreSQL take it and EF Core emits it for an inline +/// collection. ACE has no such syntax, so this is a LibRed extension. +/// +public class DerivedColumnListTests(DerivedColumnListTests.Database database) + : TempDatabaseTest, IClassFixture +{ + private static readonly string[] Setup = + [ + "CREATE TABLE T (Id LONG, K LONG)", + "INSERT INTO T VALUES (1, 10)", + "INSERT INTO T VALUES (2, 20)", + "INSERT INTO T VALUES (7, 70)", + ]; + + public sealed class Database() : SharedDatabase("derived-columns-", Setup); + + private string Run(string sql) + { + var result = database.Engine.ExecuteQuery(sql); + return $"[{string.Join(",", result.ColumnNames)}] " + string.Join(" | ", result.Rows.Select(row => + string.Join(",", row.Select(v => Convert.ToString(v, CultureInfo.InvariantCulture))))); + } + + [Theory] + [InlineData("SELECT * FROM (VALUES (1, 'a'), (2, 'b')) AS v(n, s)", "[n,s] 1,a | 2,b")] + [InlineData("SELECT v.s FROM (VALUES (1, 'a'), (2, 'b')) AS v(n, s) WHERE v.n = 2", "[s] b")] + [InlineData("SELECT x FROM (VALUES (1), (2)) v(x) ORDER BY x DESC", "[x] 2 | 1")] + [InlineData("SELECT t.x FROM (SELECT Id, K FROM T) AS t(x, y) WHERE t.y = 70", "[x] 7")] + [InlineData("SELECT x FROM (SELECT * FROM T) AS t(x, y) ORDER BY x", "[x] 1 | 2 | 7")] + [InlineData("SELECT t.x FROM (SELECT * FROM T) AS t(x, y) WHERE t.y = 70", "[x] 7")] + [InlineData("SELECT x FROM (SELECT * FROM (SELECT * FROM T) AS t(x, y)) AS d ORDER BY x", "[x] 1 | 2 | 7")] + [InlineData("SELECT x FROM (SELECT a.* FROM T AS a INNER JOIN T AS b ON a.Id = b.Id) AS t(x, y) ORDER BY x", "[x] 1 | 2 | 7")] + [InlineData("SELECT * FROM (SELECT Id AS a, K AS b FROM T WHERE Id = 1) AS t(c, d)", "[c,d] 1,10")] + [InlineData("SELECT * FROM (SELECT Id FROM T WHERE Id = 1 UNION ALL VALUES (5)) AS u(n)", "[n] 1 | 5")] + [InlineData("SELECT a.Id, v.label FROM T AS a INNER JOIN (VALUES (1, 'one'), (7, 'seven')) AS v(id, label) ON a.Id = v.id ORDER BY a.Id", + "[Id,label] 1,one | 7,seven")] + public void The_list_names_the_columns_in_order(string sql, string expected) => + Assert.Equal(expected, Run(sql)); + + [Fact] + public void The_old_names_are_gone() => + Assert.ThrowsAny(() => Run("SELECT t.Id FROM (SELECT Id FROM T) AS t(x)")); + + [Fact] + public void The_list_must_name_every_column() => + Assert.Throws(() => Run("SELECT * FROM (VALUES (1)) AS v(a, b)")); + + [Fact] + public void A_name_may_appear_once() => + Assert.Throws(() => Run("SELECT * FROM (VALUES (1, 2)) AS v(a, A)")); + + [Fact] + public void An_update_can_read_through_one() + { + QueryEngine engine = SharedDatabase.Fresh("derived-columns-update-", Setup); + engine.ExecuteNonQuery("UPDATE T AS a INNER JOIN (VALUES (1, 11), (7, 77)) AS v(id, k) ON a.Id = v.id SET a.K = v.k"); + Assert.Equal([11, 20, 77], engine.ExecuteQuery("SELECT K FROM T ORDER BY Id").Rows.Select(row => (int)row[0]!)); + } + + [Fact] + public void A_view_cannot_store_one() + { + QueryEngine engine = SharedDatabase.Fresh("derived-columns-view-", Setup); + Assert.Throws(() => + engine.ExecuteNonQuery("CREATE VIEW V AS SELECT t.x FROM (SELECT Id FROM T) AS t(x)")); + } +} diff --git a/test/LibRed.Engine.Tests/DropViewProcedureTests.cs b/test/LibRed.Engine.Tests/DropViewProcedureTests.cs index d57eb4645..280e1fbcb 100644 --- a/test/LibRed.Engine.Tests/DropViewProcedureTests.cs +++ b/test/LibRed.Engine.Tests/DropViewProcedureTests.cs @@ -18,10 +18,10 @@ public void Drop_view_removes_it_and_frees_the_name() { var e = Fresh(); e.ExecuteNonQuery("CREATE VIEW V AS SELECT ProductID FROM Products"); - Assert.True(e.Database.Catalog.Views.ContainsKey("V")); + Assert.NotNull(e.Database.Catalog.FindQuery("V")); e.ExecuteNonQuery("DROP VIEW V"); - Assert.False(e.Database.Catalog.Views.ContainsKey("V")); // gone + Assert.Null(e.Database.Catalog.FindQuery("V")); // gone Assert.Throws(() => e.ExecuteQuery("SELECT * FROM V")); Assert.Equal(0, e.ExecuteNonQuery("CREATE VIEW V AS SELECT ProductID FROM Products")); // name reusable } @@ -32,8 +32,7 @@ public void Drop_procedure_removes_it() var e = Fresh(); e.ExecuteNonQuery("CREATE PROCEDURE Pr n LONG AS SELECT ProductID FROM Products WHERE ProductID = n"); e.ExecuteNonQuery("DROP PROCEDURE Pr"); - Assert.False(e.Database.Catalog.ActionQueries.ContainsKey("Pr")); - Assert.False(e.Database.Catalog.Views.ContainsKey("Pr")); + Assert.Null(e.Database.Catalog.FindQuery("Pr")); } [Fact] @@ -43,8 +42,8 @@ public void Drop_view_and_procedure_are_interchangeable_and_missing_is_rejected( // ACE lets either statement drop either object; LibRed matches. e.ExecuteNonQuery("CREATE VIEW V2 AS SELECT ProductID FROM Products"); e.ExecuteNonQuery("DROP PROCEDURE V2"); // drop a view via DROP PROCEDURE - Assert.False(e.Database.Catalog.Views.ContainsKey("V2")); + Assert.Null(e.Database.Catalog.FindQuery("V2")); Assert.Throws(() => e.ExecuteNonQuery("DROP VIEW Nope")); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/ForeignKeyDdlTests.cs b/test/LibRed.Engine.Tests/ForeignKeyDdlTests.cs index 158a4616d..08ae3724c 100644 --- a/test/LibRed.Engine.Tests/ForeignKeyDdlTests.cs +++ b/test/LibRed.Engine.Tests/ForeignKeyDdlTests.cs @@ -31,6 +31,11 @@ public void Foreign_key_persists_enforces_and_round_trips() // Orphan reference is rejected. AssertForeignKeyViolation(engine, "INSERT INTO `Child` (`Id`, `ParentId`) VALUES (3, 99)", "FK_Child_Parent", "Parent"); + + // So is moving a child to one, and the message names the statement that did it. + var moved = Assert.Throws( + () => engine.ExecuteNonQuery("UPDATE `Child` SET `ParentId` = 99 WHERE `Id` = 1")); + Assert.Contains("UPDATE of 'Child' violates foreign key 'FK_Child_Parent'", moved.Message, StringComparison.Ordinal); } using (var db = JetDatabase.Open(path)) diff --git a/test/LibRed.Engine.Tests/ForeignKeySeekTests.cs b/test/LibRed.Engine.Tests/ForeignKeySeekTests.cs new file mode 100644 index 000000000..6fa41ca0b --- /dev/null +++ b/test/LibRed.Engine.Tests/ForeignKeySeekTests.cs @@ -0,0 +1,71 @@ +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Referential integrity finds a parent row, and a cascade its children, through an index on the key where +/// there is one, rather than scanning the table. A seek is only a faster way to the same rows: the key +/// comparison folds case and trailing spaces, so the index — whose keys fold them too — has to reach every row +/// the scan's comparison would have accepted. These hold it to that. +/// +public class ForeignKeySeekTests : TempDatabaseTest +{ + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "fk-seek-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + + e.ExecuteNonQuery("CREATE TABLE Par (Code TEXT(10) PRIMARY KEY, Nm TEXT(20))"); + e.ExecuteNonQuery("CREATE TABLE Ch (Id LONG PRIMARY KEY, Code TEXT(10), Info TEXT(20), " + + "CONSTRAINT fk FOREIGN KEY (Code) REFERENCES Par (Code) ON UPDATE CASCADE ON DELETE CASCADE)"); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + for (int i = 0; i < 50; i++) + e.ExecuteNonQuery($"INSERT INTO Par (Code, Nm) VALUES ('P{i}', 'parent {i}')"); + e.ExecuteNonQuery("INSERT INTO Par (Code, Nm) VALUES ('ABC', 'mixed')"); + e.ExecuteNonQuery("COMMIT"); + return e; + } + + private static int Count(QueryEngine e, string sql) => Convert.ToInt32(e.ExecuteQuery(sql).Rows.Single()[0]); + + [Fact] + public void A_child_finds_its_parent_whatever_the_case_and_trailing_spaces_of_its_key() + { + QueryEngine e = Seeded(); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (1, 'abc', 'lower')"); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (2, 'Abc ', 'padded')"); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (3, 'P7', 'plain')"); + Assert.Equal(3, Count(e, "SELECT COUNT(*) FROM Ch")); + } + + [Fact] + public void A_child_with_no_parent_is_still_refused() + => Assert.ThrowsAny(() => Seeded().ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (1, 'P99', 'x')")); + + [Fact] + public void A_cascade_reaches_every_child_its_key_comparison_matches() + { + QueryEngine e = Seeded(); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (1, 'abc', 'lower')"); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (2, 'ABC ', 'padded')"); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (3, 'P1', 'other')"); + + e.ExecuteNonQuery("DELETE FROM Par WHERE Code = 'ABC'"); + Assert.Equal(["P1"], e.ExecuteQuery("SELECT Code FROM Ch").Rows.Select(r => (string)r[0]!)); + } + + [Fact] + public void An_update_cascade_rewrites_the_children_it_finds_by_seek_from_all_their_values() + { + QueryEngine e = Seeded(); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (1, 'P3', 'first')"); + e.ExecuteNonQuery("INSERT INTO Ch (Id, Code, Info) VALUES (2, 'p3', 'second')"); + + e.ExecuteNonQuery("UPDATE Par SET Code = 'Q3' WHERE Code = 'P3'"); + Assert.Equal([("Q3", "first"), ("Q3", "second")], + e.ExecuteQuery("SELECT Code, Info FROM Ch ORDER BY Id").Rows.Select(r => ((string)r[0]!, (string)r[1]!))); + } +} diff --git a/test/LibRed.Engine.Tests/GroupKeyEqualityTests.cs b/test/LibRed.Engine.Tests/GroupKeyEqualityTests.cs new file mode 100644 index 000000000..18287b4d4 --- /dev/null +++ b/test/LibRed.Engine.Tests/GroupKeyEqualityTests.cs @@ -0,0 +1,60 @@ +using System.Linq; +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// What makes two grouping keys one key. Access folds them on the same terms it compares values on, so text +/// folds case and trailing spaces and a number folds across its types — measured against ACE, which returns a +/// single group for each case here. A column alone never mixes numeric types, so the numeric fold only shows +/// up through an expression. +/// +public class GroupKeyEqualityTests : TempDatabaseTest +{ + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "gke-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + e.ExecuteNonQuery("CREATE TABLE GK (Id LONG PRIMARY KEY, L LONG, D DOUBLE, C CURRENCY, S TEXT(10))"); + e.ExecuteNonQuery("INSERT INTO GK (Id, L, D, C, S) VALUES (1, 1, 1.0, 1.0, 'abc')"); + e.ExecuteNonQuery("INSERT INTO GK (Id, L, D, C, S) VALUES (2, 2, 2.0, 2.0, 'ABC ')"); + return e; + } + + [Fact] + public void An_integer_and_a_float_of_the_same_value_are_one_group() + { + var e = Seeded(); + // Row 1 yields LONG 1, row 2 yields DOUBLE 1.0 — different CLR types, one key. + var rows = e.ExecuteQuery("SELECT IIF(`L`=1, 1, 1.0) AS `E`, COUNT(*) FROM `GK` GROUP BY IIF(`L`=1, 1, 1.0)").Rows; + Assert.Equal(2, Convert.ToInt32(Assert.Single(rows)[1])); + } + + [Fact] + public void A_double_and_a_currency_of_the_same_value_are_one_group() + { + var e = Seeded(); + var rows = e.ExecuteQuery( + "SELECT `V`, COUNT(*) FROM (SELECT `D` AS `V` FROM `GK` UNION ALL SELECT `C` FROM `GK`) AS `U` GROUP BY `V`").Rows; + Assert.Equal([2, 2], rows.Select(r => Convert.ToInt32(r[1])).ToArray()); + } + + [Fact] + public void Text_folds_case_and_trailing_spaces() + { + var e = Seeded(); + Assert.Equal(2, Convert.ToInt32( + Assert.Single(e.ExecuteQuery("SELECT `S`, COUNT(*) FROM `GK` GROUP BY `S`").Rows)[1])); + } + + [Fact] + public void Distinct_folds_the_same_way() + { + var e = Seeded(); + // DISTINCT shares the grouping key, so the fold has to reach it too. + Assert.Single(e.ExecuteQuery("SELECT DISTINCT IIF(`L`=1, 1, 1.0) AS `E` FROM `GK`").Rows); + } +} diff --git a/test/LibRed.Engine.Tests/GuidLiteralAndTextTests.cs b/test/LibRed.Engine.Tests/GuidLiteralAndTextTests.cs new file mode 100644 index 000000000..eb1f92035 --- /dev/null +++ b/test/LibRed.Engine.Tests/GuidLiteralAndTextTests.cs @@ -0,0 +1,49 @@ +using System.Linq; +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// An indexed GUID column against GUID literals and text. Expected values are ACE's, measured with the same +/// statements: {guid {…}} and {…} are GUID literals, and text that reads as a GUID is stored as one. +/// +public class GuidLiteralAndTextTests : TempDatabaseTest +{ + private const string G = "6F9619FF-8B86-D011-B42D-00C04FC964FF"; + + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "guid-lit-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + e.ExecuteNonQuery("CREATE TABLE Q (Id LONG, G GUID)"); + e.ExecuteNonQuery("CREATE INDEX IxG ON Q (G)"); + e.ExecuteNonQuery($"INSERT INTO Q (Id, G) VALUES (1, {{{G}}})"); + return e; + } + + private static int Count(QueryEngine e, string where) => + Convert.ToInt32(e.ExecuteQuery($"SELECT COUNT(*) FROM Q WHERE {where}").Rows.First()[0]); + + [Theory] + [InlineData("[G] = {guid {" + G + "}}", 1)] + [InlineData("[G] = {" + G + "}", 1)] + [InlineData("[G] = '{" + G + "}'", 1)] + [InlineData("[G] = '{guid {" + G + "}}'", 0)] + public void An_indexed_guid_column_compares_with_literals_and_text(string where, int expected) => + Assert.Equal(expected, Count(Seeded(), where)); + + [Theory] + [InlineData("{guid {" + G + "}}")] + [InlineData("{" + G + "}")] + [InlineData("'{" + G + "}'")] + [InlineData("'" + G + "'")] + [InlineData("'{guid {" + G + "}}'")] + public void A_guid_literal_or_guid_text_inserts_into_an_indexed_guid_column(string value) + { + var e = Seeded(); + e.ExecuteNonQuery($"INSERT INTO Q (Id, G) VALUES (2, {value})"); + Assert.Equal(Guid.Parse(G), e.ExecuteQuery("SELECT G FROM Q WHERE Id = 2").Rows.First()[0]); + } +} diff --git a/test/LibRed.Engine.Tests/IndexSeekKindTests.cs b/test/LibRed.Engine.Tests/IndexSeekKindTests.cs new file mode 100644 index 000000000..33028130c --- /dev/null +++ b/test/LibRed.Engine.Tests/IndexSeekKindTests.cs @@ -0,0 +1,94 @@ +using System.Linq; +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// A comparison must answer the same whether or not the column happens to be indexed. An index can only be +/// searched in its own column's kind, so a seek keyed by a value of another kind either cannot be encoded at +/// all or would find a different set of rows than the comparison defines — S = 1 on text compares as a +/// number (), matching ' 1 ' and '1.0' as well as +/// '1', which are nowhere near each other in a text index. Such a seek falls back to a scan; the +/// filter the seek was planned under re-checks every row regardless, so only the reading strategy changes. +/// +public class IndexSeekKindTests : TempDatabaseTest +{ + /// The same two rows in a table with an index on each column and one with none. + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "seek-kind-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + foreach (string table in new[] { "Plain", "Indexed" }) + { + e.ExecuteNonQuery($"CREATE TABLE {table} (Id LONG, S TEXT(10), N LONG)"); + e.ExecuteNonQuery($"INSERT INTO {table} (Id, S, N) VALUES (1, 'abc', 10)"); + e.ExecuteNonQuery($"INSERT INTO {table} (Id, S, N) VALUES (2, 'def', 20)"); + } + e.ExecuteNonQuery("CREATE INDEX IxS ON Indexed (S)"); + e.ExecuteNonQuery("CREATE INDEX IxN ON Indexed (N)"); + return e; + } + + private static string Run(QueryEngine e, string sql, IReadOnlyDictionary? p = null) + { + try { return string.Join(",", e.ExecuteQuery(sql, p).Rows.Select(r => $"{r[0]}")); } + catch (Exception ex) { return $"{ex.GetType().Name}: {ex.Message}"; } + } + + [Theory] + // A numeric literal against a text column: compares as a number, so 'abc' is a type mismatch. The indexed + // table used to raise the storage layer's own cast error instead. + [InlineData("SELECT `Id` FROM `{0}` WHERE `S` = 1")] + // A text literal against a numeric column, the same the other way round. + [InlineData("SELECT `Id` FROM `{0}` WHERE `N` = 'abc'")] + // Same kinds, so the seek is used on the indexed table — the control that proves these cases are reached. + [InlineData("SELECT `Id` FROM `{0}` WHERE `S` = 'abc'")] + [InlineData("SELECT `Id` FROM `{0}` WHERE `N` = 10")] + [InlineData("SELECT `Id` FROM `{0}` WHERE `S` = NULL")] + public void An_index_does_not_change_the_answer(string sql) + { + var e = Seeded(); + Assert.Equal( + Run(e, string.Format(null, sql, "Plain")), + Run(e, string.Format(null, sql, "Indexed"))); + } + + [Theory] + // A parameter against text takes the text's type (CompareAsKinds), so a numeric parameter is compared as + // text — '1' against 'abc' is simply no match, which is what ACE answers too. + [InlineData("SELECT `Id` FROM `{0}` WHERE `S` = @p", 1)] + [InlineData("SELECT `Id` FROM `{0}` WHERE `S` = @p", "abc")] + [InlineData("SELECT `Id` FROM `{0}` WHERE `N` = @p", "10")] + [InlineData("SELECT `Id` FROM `{0}` WHERE `N` = @p", 10)] + public void An_index_does_not_change_the_answer_for_a_parameter(string sql, object value) + { + var e = Seeded(); + var p = new Dictionary { ["p"] = value }; + Assert.Equal( + Run(e, string.Format(null, sql, "Plain"), p), + Run(e, string.Format(null, sql, "Indexed"), p)); + } + + [Fact] + public void A_numeric_parameter_against_an_indexed_text_column_matches_the_text_it_reads_as() + { + var e = Seeded(); + e.ExecuteNonQuery("INSERT INTO `Indexed` (`Id`, `S`, `N`) VALUES (3, '1', 30)"); + Assert.Equal("3", Run(e, "SELECT `Id` FROM `Indexed` WHERE `S` = @p", + new Dictionary { ["p"] = 1 })); + } + + [Fact] + public void A_cross_kind_join_key_does_not_crash_the_index_nested_loop() + { + var e = Seeded(); + // Indexed.S is text and Plain.N is a number, so the inner seek cannot be keyed by it; the join falls + // back to scanning the inner table and its ON — kept whole — decides the rows. + Assert.Equal( + Run(e, "SELECT `p`.`Id` FROM `Plain` AS `p` INNER JOIN `Plain` AS `q` ON `q`.`S` = `p`.`N`"), + Run(e, "SELECT `p`.`Id` FROM `Plain` AS `p` INNER JOIN `Indexed` AS `q` ON `q`.`S` = `p`.`N`")); + } +} diff --git a/test/LibRed.Engine.Tests/IndexSplitTests.cs b/test/LibRed.Engine.Tests/IndexSplitTests.cs index 22f751ab8..a7e999ce8 100644 --- a/test/LibRed.Engine.Tests/IndexSplitTests.cs +++ b/test/LibRed.Engine.Tests/IndexSplitTests.cs @@ -38,12 +38,13 @@ public void Leaf_splitting_keeps_every_key_in_order_and_findable() for (int i = 1; i <= N; i++) t.Insert([i, $"r{i}"]); } - using (var ch = PageChannel.Open(path, readOnly: true)) + using (var db = JetDatabase.Open(path)) { - var pk = new LibRed.Catalog.JetCatalog(ch).FindTable("Big")!.Indexes.Single(x => x.IsPrimaryKey); + var pk = db.OpenTable("Big").Definition.Indexes.Single(x => x.IsPrimaryKey); // The root grew into a node — the tree is genuinely multi-level, not a single fat leaf. - Assert.Equal(PageType.IntermediateIndexPage, (PageType)ch.ReadPage(pk.RootPage).ReadByte(0)); + using var ch = PageChannel.Open(path, readOnly: true); + Assert.Equal(PageType.IntermediateIndexPage, PageHeader.ReadType(ch.ReadPage(pk.RootPage).Span)); var cursor = new IndexCursor(ch, pk.RootPage); Assert.Equal(N, cursor.RowIds().Count()); @@ -63,4 +64,4 @@ public void Leaf_splitting_keeps_every_key_in_order_and_findable() } finally { TemporaryDatabase.Delete(path); } } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/InformationSchemaTests.cs b/test/LibRed.Engine.Tests/InformationSchemaTests.cs index f6abad6d0..047baa5e8 100644 --- a/test/LibRed.Engine.Tests/InformationSchemaTests.cs +++ b/test/LibRed.Engine.Tests/InformationSchemaTests.cs @@ -26,6 +26,17 @@ public void Tables_view_lists_user_tables_and_filters_by_name() Assert.Equal(0, miss); } + // RELATION_TYPE is grbit's one-to-one bit, which SQL never sets — not the uniqueness of whatever index covers + // the child columns. A foreign key from one primary key to another is still one-to-many. + [Fact] + public void Relations_view_reports_a_sql_foreign_key_between_primary_keys_as_many() + { + var e = Seeded(); + e.ExecuteNonQuery("CREATE TABLE Part (Id LONG PRIMARY KEY, CONSTRAINT fk_part FOREIGN KEY (Id) REFERENCES Widget (Id))"); + var row = e.ExecuteQuery("SELECT `RELATION_TYPE` FROM `INFORMATION_SCHEMA.RELATIONS` WHERE `RELATION_NAME` = 'fk_part'").Rows.Single(); + Assert.Equal("MANY", row[0]); + } + [Fact] public void Columns_view_lists_columns() { diff --git a/test/LibRed.Engine.Tests/InsertSelectTests.cs b/test/LibRed.Engine.Tests/InsertSelectTests.cs index 270b2d29b..620a7df71 100644 --- a/test/LibRed.Engine.Tests/InsertSelectTests.cs +++ b/test/LibRed.Engine.Tests/InsertSelectTests.cs @@ -38,6 +38,24 @@ public void Appends_every_row_the_source_produces() Assert.Equal("two", engine.ExecuteQuery("SELECT Name FROM Dst WHERE Id = 2").Rows.Single()[0]); } + // A VALUES list that is only the first operand of a set operation is a query source, not the single-record + // form. The grammar has VALUES as a query term alone, so it is the builder that tells the two apart — by + // whether anything follows the constructor. + [Fact] + public void A_values_list_inside_a_set_operation_appends_as_a_query() + { + QueryEngine engine = Fresh(); + engine.ExecuteNonQuery("CREATE TABLE Src (Id LONG, Name TEXT(50))"); + engine.ExecuteNonQuery("CREATE TABLE Dst (Id LONG, Name TEXT(50))"); + engine.ExecuteNonQuery("INSERT INTO Src (Id, Name) VALUES (2, 'two')"); + + int affected = engine.ExecuteNonQuery( + "INSERT INTO Dst (Id, Name) VALUES (1, 'one') UNION ALL SELECT Id, Name FROM Src"); + + Assert.Equal(2, affected); + Assert.Equal([1, 2], engine.ExecuteQuery("SELECT Id FROM Dst ORDER BY Id").Rows.Select(r => Convert.ToInt32(r[0]))); + } + [Fact] public void Applies_the_sources_where_and_order() { diff --git a/test/LibRed.Engine.Tests/IsTruthAndDistinctFromTests.cs b/test/LibRed.Engine.Tests/IsTruthAndDistinctFromTests.cs new file mode 100644 index 000000000..70eac0402 --- /dev/null +++ b/test/LibRed.Engine.Tests/IsTruthAndDistinctFromTests.cs @@ -0,0 +1,45 @@ +using System.Linq; +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// The standard's IS [NOT] TRUE, IS [NOT] FALSE and IS [NOT] DISTINCT FROM, which ACE rejects. +/// Each is never Null. +/// +public class IsTruthAndDistinctFromTests : TempDatabaseTest +{ + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "is-truth-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + e.ExecuteNonQuery("CREATE TABLE T (Id LONG, X LONG, Y LONG)"); + foreach (string row in new[] { "1, 1, 1", "2, 1, 2", "3, NULL, 1", "4, NULL, NULL", "5, 2, 2" }) + e.ExecuteNonQuery($"INSERT INTO T (Id, X, Y) VALUES ({row})"); + return e; + } + + private static int[] Ids(QueryEngine e, string where) => + [.. e.ExecuteQuery($"SELECT Id FROM T WHERE {where}").Rows.Select(r => Convert.ToInt32(r[0])).Order()]; + + // X = 1 is True, True, Null, Null, False over the five rows. + [Theory] + [InlineData("(X = 1) IS TRUE", new[] { 1, 2 })] + [InlineData("(X = 1) IS NOT TRUE", new[] { 3, 4, 5 })] + [InlineData("(X = 1) IS FALSE", new[] { 5 })] + [InlineData("(X = 1) IS NOT FALSE", new[] { 1, 2, 3, 4 })] + [InlineData("X IS DISTINCT FROM Y", new[] { 2, 3 })] + [InlineData("X IS NOT DISTINCT FROM Y", new[] { 1, 4, 5 })] + public void Truth_tests_and_distinct_from_treat_null_as_a_value(string where, int[] expected) => + Assert.Equal(expected, Ids(Seeded(), where)); + + [Fact] + public void They_are_never_null() + { + object?[] row = Seeded().ExecuteQuery( + "SELECT (X = 1) IS TRUE, (X = 1) IS FALSE, X IS DISTINCT FROM Y FROM T WHERE Id = 4").Rows.Single(); + Assert.Equal([false, false, false], row); + } +} diff --git a/test/LibRed.Engine.Tests/LibRed.Engine.Tests.csproj b/test/LibRed.Engine.Tests/LibRed.Engine.Tests.csproj index 6c5a346e6..601b70639 100644 --- a/test/LibRed.Engine.Tests/LibRed.Engine.Tests.csproj +++ b/test/LibRed.Engine.Tests/LibRed.Engine.Tests.csproj @@ -13,8 +13,8 @@ true $(MSBuildThisFileDirectory)..\..\Key.snk AnyCPU;x86;x64 - - $(NoWarn);CA1416 + diff --git a/test/LibRed.Engine.Tests/LikeTests.cs b/test/LibRed.Engine.Tests/LikeTests.cs index 9c36b0e8b..c0ba255a9 100644 --- a/test/LibRed.Engine.Tests/LikeTests.cs +++ b/test/LibRed.Engine.Tests/LikeTests.cs @@ -40,6 +40,43 @@ public void Hash_star_and_question_mark_are_plain_characters() Assert.Equal(["A?"], Match(e, "A?")); } + // ACE's answers (LikeCollationProbeTests), the same in every collation: case folds by ACE's own table, which + // leaves out what a runtime's casing adds; four letters are spelt out; nothing the collation folds besides. + [Theory] + [InlineData("é", "É", true)] + [InlineData("σ", "Σ", true)] + [InlineData("ж", "Ж", true)] + [InlineData("dž", "DŽ", true)] + [InlineData("ÿ", "Ÿ", true)] + [InlineData("ς", "Σ", false)] // final sigma + [InlineData("ς", "σ", false)] + [InlineData("µ", "Μ", false)] // micro sign against capital mu + [InlineData("ſ", "S", false)] // long s + [InlineData("Dž", "dž", false)] // titlecase digraph + [InlineData("ı", "I", false)] // dotless i, in every collation — Turkish too + [InlineData("i", "İ", false)] + [InlineData("ß", "ss", true)] + [InlineData("ss", "ß", true)] + [InlineData("ß", "SS", true)] + [InlineData("æ", "AE", true)] + [InlineData("œ", "oe", true)] + [InlineData("Œ", "OE", true)] + [InlineData("œ", "OE", true)] + [InlineData("þ", "th", true)] + [InlineData("Þ", "TH", true)] + [InlineData("ẞ", "ss", false)] // capital sharp s is not spelt out + [InlineData("ij", "ij", false)] + [InlineData("fi", "fi", false)] + [InlineData("é", "e", false)] // accents count + [InlineData("A", "A", false)] // so does width, which the collation folds + [InlineData("co-op", "coop", false)] + [InlineData("a ", "a", false)] // and trailing spaces, which '=' ignores + [InlineData("Z", "[a-z]", true)] + [InlineData("é", "[a-z]", false)] // a range runs in character order, not the collation's + [InlineData("_", "[A-z]", false)] + public void Case_and_spelling_fold_as_ace_folds_them(string value, string pattern, bool matches) + => Assert.Equal(matches ? [value] : [], Match(Fresh(value), pattern)); + [Fact] public void Bracket_class_and_negation_and_ranges() { diff --git a/test/LibRed.Engine.Tests/MSysAcesViewTests.cs b/test/LibRed.Engine.Tests/MSysAcesViewTests.cs index 7e90749a6..130f4e28b 100644 --- a/test/LibRed.Engine.Tests/MSysAcesViewTests.cs +++ b/test/LibRed.Engine.Tests/MSysAcesViewTests.cs @@ -7,8 +7,11 @@ namespace LibRed.Engine.Tests; // A LibRed-created view must get the same MSysACEs permission rows Access writes for a query object, or -// Access warns about permissions when opening it (verified against Northwind: owner 0x690C = ACM 0xF00FE, -// admin/users 0x680C = ACM 0xFFEFF — distinct from a table's, where both SIDs get full 0xFFEFF). +// Access warns about permissions when opening it. Measured against ACE's own CREATE VIEW, row for row +// (CatalogRowParityAccessTests): both rows get full access, as a table's do. The SIDs are this file's — +// Northwind masks the Users account as 0x690C and admin as 0x680C — and LibRed reads that mask out of the +// file rather than carrying one of its own, so the same view in a LibRed-created database gets that +// database's SIDs instead. public class MSysAcesViewTests { [Fact] @@ -39,8 +42,8 @@ public void Created_view_gets_the_query_permission_rows_access_writes() .ToList(); Assert.Equal(2, rows.Count); - Assert.Equal(("680C", 0xFFEFF), (rows[0].Sid, rows[0].Acm)); // admin/users → full - Assert.Equal(("690C", 0xF00FE), (rows[1].Sid, rows[1].Acm)); // owner → query mask + Assert.Equal(("680C", 0xFFEFF), (rows[0].Sid, rows[0].Acm)); // admin → full + Assert.Equal(("690C", 0xFFEFF), (rows[1].Sid, rows[1].Acm)); // owner (Users) → full Assert.All(rows, r => Assert.Equal(false, r.Inh)); } finally { TemporaryDatabase.Delete(path); } diff --git a/test/LibRed.Engine.Tests/MemoIndexTests.cs b/test/LibRed.Engine.Tests/MemoIndexTests.cs index 8102a82c0..fc80567e5 100644 --- a/test/LibRed.Engine.Tests/MemoIndexTests.cs +++ b/test/LibRed.Engine.Tests/MemoIndexTests.cs @@ -5,7 +5,7 @@ namespace LibRed.Engine.Tests; // A Memo (Long Text) column is indexable in Access; its index key is the text collation key over the first -// 255 characters. Exercise the insert path (RowInserter → IndexKeyEncoder) end-to-end through the engine. +// 255 characters. Exercise the write paths (RowInserter → IndexKeyCodec) end-to-end through the engine. public class MemoIndexTests : TempDatabaseTest { private static QueryEngine Fresh() @@ -40,4 +40,52 @@ public void Two_memos_differing_only_past_255_chars_share_a_key_but_both_insert( e.ExecuteNonQuery($"INSERT INTO MK (Id, M) VALUES (2, '{new string('z', 255)}B')"); Assert.Equal(2, Convert.ToInt32(e.ExecuteQuery("SELECT COUNT(*) FROM MK").Rows.Single()[0])); } -} + + // Deleting asks each index whether the row being removed held the last copy of its key, which means + // encoding that key from the row. The decode behind that question was the one on the table path without a + // long-value reader, so the Memo came back as its 12-byte on-disk descriptor and the key encoder's text + // path cast a byte[] to string — deleting ANY row of a memo-indexed table threw. + [Theory] + [InlineData(5)] // inline: the value sits in the row beside its descriptor + [InlineData(4000)] // chained onto its own long-value pages + public void A_row_deletes_from_a_memo_indexed_table(int length) + { + string memo = new('m', length); + var e = Fresh(); + e.ExecuteNonQuery($"INSERT INTO MK (Id, M) VALUES (1, '{memo}')"); + e.ExecuteNonQuery("INSERT INTO MK (Id, M) VALUES (2, 'second')"); + + Assert.Equal(1, e.ExecuteNonQuery("DELETE FROM MK WHERE Id = 1")); + + Assert.Equal(2, Convert.ToInt32(e.ExecuteQuery("SELECT Id FROM MK").Rows.Single()[0])); + // The survivor is still reachable through the memo index, and the deleted row is not — so the delete + // maintained the index rather than leaving an entry behind. + Assert.Single(e.ExecuteQuery("SELECT Id FROM MK WHERE M = 'second'").Rows); + Assert.Empty(e.ExecuteQuery($"SELECT Id FROM MK WHERE M = '{memo}'").Rows); + } + + // The same question on the UPDATE path, and the harder half of it: moving the index entry needs the OLD + // key as well as the new one, so the old row's Memo has to be resolved rather than read as the 12-byte + // descriptor standing in for it. Both storage forms, because the descriptor is all the row holds either + // way and only the reader knows the difference. + [Theory] + [InlineData(5, 7)] // inline to inline + [InlineData(5, 4000)] // inline to chained + [InlineData(4000, 5)] // chained to inline + [InlineData(4000, 3000)] // chained to chained + public void A_memo_indexed_row_updates_and_keeps_its_index(int before, int after) + { + string old = new('m', before), replacement = new('n', after); + var e = Fresh(); + e.ExecuteNonQuery($"INSERT INTO MK (Id, M) VALUES (1, '{old}')"); + e.ExecuteNonQuery("INSERT INTO MK (Id, M) VALUES (2, 'second')"); + + Assert.Equal(1, e.ExecuteNonQuery($"UPDATE MK SET M = '{replacement}' WHERE Id = 1")); + + Assert.Equal(replacement, e.ExecuteQuery("SELECT M FROM MK WHERE Id = 1").Rows.Single()[0]); + // The index moved with the value: the new one is reachable through it and the old one is gone. + Assert.Single(e.ExecuteQuery($"SELECT Id FROM MK WHERE M = '{replacement}'").Rows); + Assert.Empty(e.ExecuteQuery($"SELECT Id FROM MK WHERE M = '{old}'").Rows); + Assert.Single(e.ExecuteQuery("SELECT Id FROM MK WHERE M = 'second'").Rows); + } +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/NzTests.cs b/test/LibRed.Engine.Tests/NzTests.cs new file mode 100644 index 000000000..b3e599a46 --- /dev/null +++ b/test/LibRed.Engine.Tests/NzTests.cs @@ -0,0 +1,69 @@ +using System.Globalization; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Access's Nz(value [, valueIfNull]), which the Access application has and ACE's OLE DB provider does not. Every +/// expected value here is what Access itself returned for the same query over the same rows: a Variant, written out as +/// text, sorted and grouped as its text, and kept as its own value where it is an operand; and, with one argument, VBA's +/// Empty for a Null — written out as "", read as 0 by arithmetic. +/// +public class NzTests(NzTests.Database database) : TempDatabaseTest, IClassFixture +{ + private static readonly string[] Setup = + [ + "CREATE TABLE T (Id LONG, K LONG, K2 TEXT(10))", + "INSERT INTO T VALUES (1, 1, 'a')", + "INSERT INTO T VALUES (2, 2, 'b')", + "INSERT INTO T VALUES (7, 10, 'c')", + "INSERT INTO T VALUES (8, NULL, NULL)", + ]; + + public sealed class Database() : SharedDatabase("nz-", Setup); + + private string ById(string expression) + { + var (types, rows) = database.Query($"SELECT Id, {expression} AS r FROM T ORDER BY Id", CultureInfo.InvariantCulture); + return $"{types[1].Name} " + string.Join(" ", rows.Select(row => + $"{row[0]}:{(row[1] is null ? "Null" : Convert.ToString(row[1], CultureInfo.InvariantCulture))}")); + } + + [Theory] + [InlineData("Nz(K, 0)", "String 1:1 2:2 7:10 8:0")] + [InlineData("Nz(K2, '-')", "String 1:a 2:b 7:c 8:-")] + [InlineData("Nz(K, 'none')", "String 1:1 2:2 7:10 8:none")] + [InlineData("Nz(K, 0.5)", "String 1:1 2:2 7:10 8:0.5")] + [InlineData("Nz(Null, 5)", "String 1:5 2:5 7:5 8:5")] + [InlineData("Nz(K)", "String 1:1 2:2 7:10 8:")] + [InlineData("Nz(K2)", "String 1:a 2:b 7:c 8:")] + [InlineData("Nz(Null)", "String 1: 2: 7: 8:")] + [InlineData("Nz(Null, Null)", "String 1:Null 2:Null 7:Null 8:Null")] + public void Nz_is_written_out_as_text(string expression, string expected) => + Assert.Equal(expected, ById(expression)); + + [Theory] + [InlineData("Nz(K, 0) + 1", "Double 1:2 2:3 7:11 8:1")] + [InlineData("Nz(K) + 2", "Double 1:3 2:4 7:12 8:2")] + [InlineData("Nz(K) * 3", "Double 1:3 2:6 7:30 8:0")] + [InlineData("Nz(K2) & 'x'", "String 1:ax 2:bx 7:cx 8:x")] + [InlineData("Len(Nz(K))", "Int32 1:1 2:1 7:2 8:0")] + [InlineData("Nz(K) = 0", "Boolean 1:False 2:False 7:False 8:True")] + [InlineData("Nz(Nz(K))", "String 1:1 2:2 7:10 8:")] + public void As_an_operand_nz_is_its_value_and_empty_is_zero_or_empty_text(string expression, string expected) => + Assert.Equal(expected, ById(expression)); + + [Fact] + public void Nz_sorts_and_groups_as_its_text() => + Assert.Equal([8, 1, 7, 2], database.Query("SELECT Id FROM T ORDER BY Nz(K, 0)", CultureInfo.InvariantCulture) + .Rows.Select(row => (int)row[0]!)); + + [Fact] + public void Nz_compares_as_a_number_beside_one() => + Assert.Equal([7], database.Query("SELECT Id FROM T WHERE Nz(K, 0) > 2 ORDER BY Id", CultureInfo.InvariantCulture) + .Rows.Select(row => (int)row[0]!)); + + [Fact] + public void Nz_takes_one_or_two_arguments() => + Assert.Throws(() => database.Query("SELECT Nz(K, 0, 1) FROM T", CultureInfo.InvariantCulture)); +} diff --git a/test/LibRed.Engine.Tests/OwnerAccessTests.cs b/test/LibRed.Engine.Tests/OwnerAccessTests.cs new file mode 100644 index 000000000..0adbf9ac3 --- /dev/null +++ b/test/LibRed.Engine.Tests/OwnerAccessTests.cs @@ -0,0 +1,105 @@ +using System.Globalization; +using LibRed.Sql.Parsing; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// WITH OWNERACCESS OPTION: run a query with its owner's permissions. LibRed has no users, so a statement with it +/// does exactly what it does without it — it is accepted wherever ACE accepts it and refused wherever ACE refuses it +/// (verified). A stored query keeps it; StoredActionQueryWriteAccessTests checks that against ACE. +/// +public class OwnerAccessTests : TempDatabaseTest +{ + private static readonly string[] Setup = + [ + "CREATE TABLE T (Id LONG, N TEXT(10))", + "INSERT INTO T VALUES (1, 'a')", + "INSERT INTO T VALUES (2, 'b')", + "INSERT INTO T VALUES (3, 'b')", + ]; + + private static QueryEngine Fresh() => SharedDatabase.Fresh("owneraccess-", Setup); + + private static string Rows(QueryEngine engine, string sql, IReadOnlyDictionary? parameters = null) => + string.Join(" | ", engine.ExecuteQuery(sql, parameters).Rows.Select(row => + string.Join(",", row.Select(v => Convert.ToString(v, CultureInfo.InvariantCulture))))); + + // Each with the declaration returns what it returns without: it ends a SELECT, a query after its ORDER BY, each + // arm of a UNION and a subquery. + [Theory] + [InlineData("SELECT Id FROM T WITH OWNERACCESS OPTION", "SELECT Id FROM T")] + [InlineData("SELECT Id FROM T WITH OWNERACCESS OPTION;", "SELECT Id FROM T")] + [InlineData("SELECT Id FROM T ORDER BY Id DESC WITH OWNERACCESS OPTION", "SELECT Id FROM T ORDER BY Id DESC")] + [InlineData("SELECT Id FROM T WHERE Id > 1 WITH OWNERACCESS OPTION", "SELECT Id FROM T WHERE Id > 1")] + [InlineData("SELECT N, COUNT(*) FROM T GROUP BY N HAVING COUNT(*) > 1 WITH OWNERACCESS OPTION", + "SELECT N, COUNT(*) FROM T GROUP BY N HAVING COUNT(*) > 1")] + [InlineData("SELECT N FROM T UNION SELECT N FROM T ORDER BY N WITH OWNERACCESS OPTION", + "SELECT N FROM T UNION SELECT N FROM T ORDER BY N")] + [InlineData("SELECT N FROM T WITH OWNERACCESS OPTION UNION SELECT N FROM T ORDER BY N", + "SELECT N FROM T UNION SELECT N FROM T ORDER BY N")] + [InlineData("SELECT N FROM T WITH OWNERACCESS OPTION UNION SELECT N FROM T WITH OWNERACCESS OPTION", + "SELECT N FROM T UNION SELECT N FROM T")] + [InlineData("SELECT N FROM T WITH OWNERACCESS OPTION UNION SELECT N FROM T ORDER BY N WITH OWNERACCESS OPTION", + "SELECT N FROM T UNION SELECT N FROM T ORDER BY N")] + [InlineData("SELECT * FROM (SELECT Id FROM T WITH OWNERACCESS OPTION) WITH OWNERACCESS OPTION", "SELECT * FROM (SELECT Id FROM T)")] + [InlineData("SELECT * FROM (SELECT Id FROM T WITH OWNERACCESS OPTION) ORDER BY Id", "SELECT * FROM (SELECT Id FROM T) ORDER BY Id")] + [InlineData("SELECT Id FROM T WHERE Id IN (SELECT Id FROM T WHERE N = 'b' WITH OWNERACCESS OPTION)", + "SELECT Id FROM T WHERE Id IN (SELECT Id FROM T WHERE N = 'b')")] + [InlineData("SELECT Id FROM T with owneraccess\r\n option", "SELECT Id FROM T")] + public void A_query_with_it_returns_what_it_returns_without(string with, string without) + { + QueryEngine engine = Fresh(); + Assert.Equal(Rows(engine, without), Rows(engine, with)); + } + + [Fact] + public void It_follows_a_parameters_clause() => + Assert.Equal("2 | 3", Rows(Fresh(), "PARAMETERS p Long; SELECT Id FROM T WHERE Id > p ORDER BY Id WITH OWNERACCESS OPTION", + new Dictionary { ["p"] = 1 })); + + // An INSERT, UPDATE or DELETE with it changes the rows it changes without. + [Theory] + [InlineData("INSERT INTO T (Id, N) VALUES (4, 'c') WITH OWNERACCESS OPTION", 1, "1,a | 2,b | 3,b | 4,c")] + [InlineData("INSERT INTO T (Id, N) SELECT Id + 10, N FROM T WHERE N = 'b' WITH OWNERACCESS OPTION", 2, + "1,a | 2,b | 3,b | 12,b | 13,b")] + [InlineData("UPDATE T SET N = 'z' WHERE Id = 2 WITH OWNERACCESS OPTION", 1, "1,a | 2,z | 3,b")] + [InlineData("DELETE FROM T WHERE N = 'b' WITH OWNERACCESS OPTION", 2, "1,a")] + public void A_change_with_it_changes_what_it_changes_without(string sql, int affected, string after) + { + QueryEngine engine = Fresh(); + Assert.Equal(affected, engine.ExecuteNonQuery(sql)); + Assert.Equal(after, Rows(engine, "SELECT Id, N FROM T ORDER BY Id")); + } + + [Fact] + public void A_make_table_query_takes_it() + { + QueryEngine engine = Fresh(); + Assert.Equal(3, engine.ExecuteNonQuery("SELECT Id, N INTO T2 FROM T WITH OWNERACCESS OPTION")); + Assert.Equal("1,a | 2,b | 3,b", Rows(engine, "SELECT Id, N FROM T2 ORDER BY Id")); + } + + // ACE refuses each of these: a query's ORDER BY belongs to its last SELECT, so the option ending that SELECT cannot + // come before it; it ends a query once; it needs all three words; and it is not part of DDL. + [Theory] + [InlineData("SELECT Id FROM T WITH OWNERACCESS OPTION ORDER BY Id")] + [InlineData("SELECT Id FROM T UNION SELECT Id FROM T WITH OWNERACCESS OPTION ORDER BY Id")] + [InlineData("SELECT Id FROM T WITH OWNERACCESS OPTION WITH OWNERACCESS OPTION")] + [InlineData("SELECT Id FROM T WITH OWNERACCESS")] + [InlineData("SELECT Id FROM T WITH OPTION")] + [InlineData("SELECT Id FROM T OWNERACCESS OPTION")] + [InlineData("CREATE TABLE T3 (Id LONG) WITH OWNERACCESS OPTION")] + public void Where_ace_refuses_it_so_does_libred(string sql) => + Assert.Throws(() => Fresh().ExecuteNonQuery(sql)); + + // Neither word is reserved: a column may still be named either, unbracketed. + [Fact] + public void Its_words_still_name_columns() + { + QueryEngine engine = Fresh(); + engine.ExecuteNonQuery("CREATE TABLE W (Option LONG, OwnerAccess LONG)"); + engine.ExecuteNonQuery("INSERT INTO W (Option, OwnerAccess) VALUES (1, 2)"); + Assert.Equal("1,2", Rows(engine, "SELECT Option, OwnerAccess FROM W WITH OWNERACCESS OPTION")); + } +} diff --git a/test/LibRed.Engine.Tests/ParameterComparedWithColumnTests.cs b/test/LibRed.Engine.Tests/ParameterComparedWithColumnTests.cs new file mode 100644 index 000000000..82a7c3f96 --- /dev/null +++ b/test/LibRed.Engine.Tests/ParameterComparedWithColumnTests.cs @@ -0,0 +1,76 @@ +using System.Linq; +using LibRed; +using LibRed.Engine; +using LibRed.Engine.Execution; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// A parameter takes the type of the column it is compared with, as ACE's does: a number against a text column is +/// compared as text, text against a number column is read as a number. Expected values are ACE's, measured with +/// the same data over OLE DB parameters. +/// +public class ParameterComparedWithColumnTests : TempDatabaseTest +{ + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "param-cmp-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + e.ExecuteNonQuery("CREATE TABLE T (Id LONG, S TEXT(50))"); + e.ExecuteNonQuery("INSERT INTO T (Id, S) VALUES (1, '')"); + e.ExecuteNonQuery("INSERT INTO T (Id, S) VALUES (2, 'abc')"); + e.ExecuteNonQuery("INSERT INTO T (Id, S) VALUES (3, '11')"); + + // Text order and number order disagree over these. + e.ExecuteNonQuery("CREATE TABLE W (Id LONG, S TEXT(50), N LONG)"); + foreach (string row in new[] { "1, '9', 9", "2, '10', 10", "3, '100', 100", "4, '011', 11", "5, '1.5', NULL", "6, '-1', NULL" }) + e.ExecuteNonQuery($"INSERT INTO W (Id, S, N) VALUES ({row})"); + return e; + } + + private static int[] Ids(QueryEngine e, string sql, params object[] values) + { + var parameters = new Dictionary(); + for (int i = 0; i < values.Length; i++) parameters[$"@p{i}"] = values[i]; + return [.. e.ExecuteQuery(sql, parameters).Rows.Select(r => Convert.ToInt32(r[0])).Order()]; + } + + [Fact] + public void A_number_parameter_against_a_text_column_holding_non_numbers_does_not_throw() + { + var e = Seeded(); + Assert.Equal([3], Ids(e, "SELECT Id FROM T WHERE S = @p0", 11)); + Assert.Equal([3], Ids(e, "SELECT Id FROM T WHERE S IN (@p0, @p1, @p2)", 11, 18, 19)); + Assert.Equal([1, 2], Ids(e, "SELECT Id FROM T WHERE S NOT IN (@p0, @p1, @p2)", 11, 18, 19)); + } + + [Fact] + public void A_number_parameter_against_a_text_column_compares_as_text() + { + var e = Seeded(); + // As numbers '011' would be 11, and only 9 and 10 would lie between 9 and 10. + Assert.Empty(Ids(e, "SELECT Id FROM W WHERE S IN (@p0, @p1)", 11, 99)); + Assert.Equal([1, 2, 3], Ids(e, "SELECT Id FROM W WHERE S BETWEEN @p0 AND @p1", 9, 10)); + Assert.Equal([5], Ids(e, "SELECT Id FROM W WHERE S = @p0", 1.5)); + Assert.Equal([6], Ids(e, "SELECT Id FROM W WHERE S = @p0", true)); + } + + // ACE refuses this query outright; LibRed reads the column as a number row by row instead, as SQL Server does. + [Fact] + public void A_number_literal_against_a_text_column_is_not_converted() + { + var e = Seeded(); + Assert.Equal([4], Ids(e, "SELECT Id FROM W WHERE S = 11")); + } + + [Fact] + public void A_text_parameter_against_a_number_column_is_read_as_a_number() + { + var e = Seeded(); + Assert.Equal([4], Ids(e, "SELECT Id FROM W WHERE N = @p0", "11")); + Assert.Equal([1, 4], Ids(e, "SELECT Id FROM W WHERE N IN (@p0, @p1)", "11", "9")); + Assert.Equal([1, 2], Ids(e, "SELECT Id FROM W WHERE N BETWEEN @p0 AND @p1", "9", "10")); + Assert.Throws(() => Ids(e, "SELECT Id FROM W WHERE N = @p0", "abc")); + } +} diff --git a/test/LibRed.Engine.Tests/PreEpochDateOrderingTests.cs b/test/LibRed.Engine.Tests/PreEpochDateOrderingTests.cs index f701c6284..6de1a777b 100644 --- a/test/LibRed.Engine.Tests/PreEpochDateOrderingTests.cs +++ b/test/LibRed.Engine.Tests/PreEpochDateOrderingTests.cs @@ -18,7 +18,7 @@ namespace LibRed.Engine.Tests; /// live data in the suite. /// /// What matters most here is INTERNAL CONSISTENCY. LibRed has two paths to an ordered or filtered result: the -/// index (IndexKeyEncoder writes the raw OA serial as the sort key, so it inherits ACE's ordering) and the +/// index (IndexKeyCodec writes the raw OA serial as the sort key, so it inherits ACE's ordering) and the /// evaluator (which compares CLR DateTime values, and is therefore chronologically correct). If those two /// disagree, the same query returns different answers depending on whether the planner picks a seek or a scan. /// These tests pin that they agree, and record which convention the agreement follows. @@ -110,4 +110,4 @@ public void Comparison_within_a_pre_epoch_day_is_consistent_with_ordering() Assert.Equal(morningFirst, morningIsLess); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/QueryNamesTests.cs b/test/LibRed.Engine.Tests/QueryNamesTests.cs new file mode 100644 index 000000000..de2dcfaaa --- /dev/null +++ b/test/LibRed.Engine.Tests/QueryNamesTests.cs @@ -0,0 +1,165 @@ +using LibRed; +using LibRed.Engine; +using LibRed.Sql.Parsing; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// How LibRed reads the names in a query as Access writes them: the bang, where Customers!CustomerID is +/// Customers.CustomerID and a longer chain such as Forms!frmMenu!cmbGroup names a form control, which a +/// query reaches only as a parameter it declares; a reserved word naming a column after a period or a bang; a name in +/// any script written unbracketed; and Yes, No, On and Off as True and False. ACE's own answers are in +/// QueryNamesAccessTests. +/// +public class QueryNamesTests : TempDatabaseTest +{ + private static QueryEngine Northwind(bool readOnly = true) + { + string path = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "names-"); + return new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly)); + } + + /// Northwind with a table K whose columns are named by reserved words and constants, and a table Árú + /// named, like its columns, outside ASCII. + private static QueryEngine Seeded() + { + QueryEngine engine = Northwind(readOnly: false); + foreach (string sql in new[] + { + "CREATE TABLE K ([Key] LONG, [Select] LONG, [Yes] LONG, [No] LONG, [On] LONG, [Off] LONG)", + "INSERT INTO K VALUES (1, 2, 3, 4, 5, 6)", + "CREATE TABLE [Árú] ([Név] TEXT(10), [Номер] LONG, [名前] TEXT(10), [ΑΒΓ] LONG)", + "INSERT INTO [Árú] VALUES ('x', 2, 'y', 3)", + }) + engine.ExecuteNonQuery(sql); + return engine; + } + + private static List Column(QueryEngine e, string sql, IReadOnlyDictionary? parameters = null) => + e.ExecuteQuery(sql, parameters).Rows.Select(r => r[0]).ToList(); + + private static object?[] Row(QueryEngine e, string sql) => e.ExecuteQuery(sql).Rows.Single(); + + [Theory] + [InlineData("SELECT Customers!CustomerID FROM Customers", "SELECT Customers.CustomerID FROM Customers")] + [InlineData("SELECT [Customers]![City] FROM Customers", "SELECT Customers.City FROM Customers")] + [InlineData("SELECT `Customers`!`City` FROM Customers", "SELECT Customers.City FROM Customers")] + [InlineData("SELECT c!CustomerID FROM Customers AS c WHERE c!City = 'London'", + "SELECT c.CustomerID FROM Customers AS c WHERE c.City = 'London'")] + [InlineData("SELECT City FROM Customers GROUP BY Customers!City HAVING Customers!City > 'M' ORDER BY Customers!City DESC", + "SELECT City FROM Customers GROUP BY Customers.City HAVING Customers.City > 'M' ORDER BY Customers.City DESC")] + [InlineData("SELECT o!OrderID FROM Customers AS c INNER JOIN Orders AS o ON c!CustomerID = o!CustomerID", + "SELECT o.OrderID FROM Customers AS c INNER JOIN Orders AS o ON c.CustomerID = o.CustomerID")] + [InlineData("SELECT (SELECT COUNT(*) FROM Orders WHERE Orders!CustomerID = Customers!CustomerID) FROM Customers", + "SELECT (SELECT COUNT(*) FROM Orders WHERE Orders.CustomerID = Customers.CustomerID) FROM Customers")] + [InlineData("SELECT d!X FROM (SELECT CustomerID AS X FROM Customers) AS d", "SELECT d.X FROM (SELECT CustomerID AS X FROM Customers) AS d")] + public void Two_parts_joined_by_a_bang_are_table_and_column(string bang, string dot) + { + QueryEngine e = Northwind(); + var viaBang = Column(e, bang); + Assert.NotEmpty(viaBang); + Assert.Equal(Column(e, dot), viaBang); + } + + [Theory] + [InlineData("SELECT Customers!CustomerID FROM Customers", "Customers!CustomerID")] + [InlineData("SELECT [Customers]![CustomerID] FROM Customers", "Customers]![CustomerID")] + [InlineData("SELECT Customers!CustomerID AS X FROM Customers", "X")] + [InlineData("SELECT Customers.CustomerID FROM Customers", "CustomerID")] + public void An_unaliased_bang_column_is_named_as_written(string sql, string name) => + Assert.Equal(name, Northwind().ExecuteQuery(sql).ColumnNames.Single()); + + // The name is the column's, so an outer query reaches it by that name and not by the column's own. + [Fact] + public void A_derived_tables_bang_column_goes_by_its_written_name() + { + QueryEngine e = Northwind(); + const string inner = "(SELECT Customers!City FROM Customers WHERE CustomerID = 'ALFKI')"; + Assert.Equal(["Berlin"], Column(e, $"SELECT [Customers!City] FROM {inner}")); + Assert.Throws(() => Column(e, $"SELECT City FROM {inner}")); + } + + [Theory] + [InlineData("SELECT Customers ! CustomerID FROM Customers")] + [InlineData("SELECT Customers !CustomerID FROM Customers")] + [InlineData("SELECT Customers! CustomerID FROM Customers")] + public void A_bang_takes_no_space_either_side(string sql) => + Assert.Throws(() => Northwind().ExecuteQuery(sql)); + + // A form control is nothing a query can reach unless the query declares it. + [Fact] + public void An_undeclared_form_control_is_not_found() + { + var ex = Assert.Throws( + () => Column(Northwind(), "SELECT CustomerID FROM Customers WHERE CustomerID = Forms!frmMenu!cmbCustomer")); + Assert.Contains("Forms!frmMenu!cmbCustomer", ex.Message, StringComparison.Ordinal); + } + + [Theory] + [InlineData("PARAMETERS [Forms]![frmMenu]![cmbCustomer] Text(5); SELECT CustomerID FROM Customers WHERE CustomerID = Forms!frmMenu!cmbCustomer", + "Forms!frmMenu!cmbCustomer")] + [InlineData("PARAMETERS Forms!frmMenu!cmbCustomer Text(5); SELECT CustomerID FROM Customers WHERE CustomerID = [Forms]![frmMenu]![cmbCustomer]", + "Forms!frmMenu!cmbCustomer")] + // A period in the chain counts as a bang, as the Form property of a subform control is written. + [InlineData("PARAMETERS [Forms]![frmMenu]![sub].[Form]![cmbCustomer] Text(5); SELECT CustomerID FROM Customers WHERE CustomerID = Forms!frmMenu!sub!Form!cmbCustomer", + "Forms!frmMenu!sub!Form!cmbCustomer")] + public void A_declared_form_control_is_a_parameter(string sql, string name) => + Assert.Equal(["ALFKI"], Column(Northwind(), sql, new Dictionary { [name] = "ALFKI" })); + + // A two-part control reads as a table and a column, and its declaration wins over a real column of that name. + [Theory] + [InlineData("PARAMETERS Forms!txtCity Text(20); SELECT CustomerID FROM Customers WHERE City = Forms!txtCity", "Forms!txtCity")] + [InlineData("PARAMETERS Forms!txtCity Text(20); SELECT CustomerID FROM Customers WHERE City = Forms.txtCity", "Forms!txtCity")] + [InlineData("PARAMETERS Customers!City Text(20); SELECT CustomerID FROM Customers WHERE City = Customers!City", "Customers!City")] + public void A_declared_two_part_control_is_a_parameter(string sql, string name) => + Assert.Equal(["ALFKI"], Column(Northwind(), sql, new Dictionary { [name] = "Berlin" })); + + [Fact] + public void An_update_may_set_a_column_written_with_a_bang() + { + QueryEngine e = Northwind(readOnly: false); + Assert.Equal(1, e.ExecuteNonQuery("UPDATE Customers SET Customers!City = 'Ankh-Morpork' WHERE Customers!CustomerID = 'ALFKI'")); + Assert.Equal(["Ankh-Morpork"], Column(e, "SELECT City FROM Customers WHERE CustomerID = 'ALFKI'")); + } + + // Access stores a declared form control as written, brackets and all, and binds the stored query's EXECUTE + // arguments to it by position. + [Fact] + public void A_stored_query_declaring_a_form_control_executes() + { + string path = TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "names-proc-"); + using (var db = JetDatabase.Open(path, readOnly: false)) + new QueryEngine(db).ExecuteNonQuery( + "CREATE PROCEDURE ByCity [Forms]![frmMenu]![txtCity] Text(20) AS " + + "SELECT CustomerID FROM Customers WHERE City = Forms!frmMenu!txtCity"); + + using (var db = JetDatabase.Open(path, readOnly: true)) + { + Assert.Equal("Forms!frmMenu!txtCity", db.Catalog.FindQuery("ByCity")!.Parameters.Single().Name); + Assert.Equal(["ALFKI"], Column(new QueryEngine(db), "EXECUTE ByCity 'Berlin'")); + } + } + + [Fact] + public void A_reserved_word_names_a_column_after_a_period_or_a_bang() => + Assert.Equal([1, 2, 1, 2], Row(Seeded(), "SELECT K.Key, K.Select, K!Key, K!Select FROM K")); + + [Fact] + public void A_name_in_any_script_needs_no_brackets() + { + QueryEngine e = Seeded(); + Assert.Equal(["x", 2, "y", 3], Row(e, "SELECT Név, Номер, 名前, ΑΒΓ FROM Árú")); + Assert.Equal(["x"], Row(e, "SELECT a.Név FROM Árú AS a WHERE a.ΑΒΓ = 3")); + } + + [Fact] + public void Yes_no_on_and_off_are_true_and_false() => + Assert.Equal([true, false, true, false], Row(Seeded(), "SELECT Yes, No, On, Off FROM K")); + + // Unqualified, the constant wins over a column of that name; qualified, it is the column. + [Fact] + public void A_qualified_yes_is_the_column() => + Assert.Equal([true, 3, false, 4, true, 5, false, 6], + Row(Seeded(), "SELECT Yes, K.Yes, No, K.No, On, K.On, Off, K!Off FROM K")); +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/ReaderWriterIsolationTests.cs b/test/LibRed.Engine.Tests/ReaderWriterIsolationTests.cs index 9de9d8c41..29a75b1d1 100644 --- a/test/LibRed.Engine.Tests/ReaderWriterIsolationTests.cs +++ b/test/LibRed.Engine.Tests/ReaderWriterIsolationTests.cs @@ -108,13 +108,201 @@ public void Preloaded_reader_cache_sees_committed_multi_page_update_but_not_unco finally { TemporaryDatabase.Delete(path); } } + // Write skew across two connections. The commit's conflict check covers the pages a transaction WROTE, and + // a foreign-key check writes nothing — it reads the parent row. So one connection can insert a child while + // another deletes the parent, each finding the other's precondition satisfied, and the file ends up with a + // child referencing nothing. The insert's dependency is re-checked at commit, which is where it is caught. + [Fact] + public void A_child_insert_cannot_commit_after_its_parent_is_deleted_by_another_connection() + { + string path = CreateDatabase("write-skew-"); + try + { + using var childDb = JetDatabase.Open(path, readOnly: false); + using var parentDb = JetDatabase.Open(path, readOnly: false); + var child = new QueryEngine(childDb); + var parent = new QueryEngine(parentDb); + + child.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY)"); + child.ExecuteNonQuery( + "CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG REFERENCES Parents (Id))"); + child.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (1)"); + child.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (2)"); + + child.ExecuteNonQuery("BEGIN TRANSACTION"); + child.ExecuteNonQuery("INSERT INTO Children (Id, ParentId) VALUES (10, 1)"); // parent 1 is there + + parent.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 1"); // ... until now + + var conflict = Assert.Throws(() => child.ExecuteNonQuery("COMMIT")); + Assert.Contains("no longer exists", conflict.Message, StringComparison.Ordinal); + + // Nothing of the transaction survived, so the file has no child row referencing the deleted parent. + child.ExecuteNonQuery("ROLLBACK"); + Assert.Empty(parent.ExecuteQuery("SELECT Id FROM Children").Rows); + + // And the check is specific: deleting a DIFFERENT parent does not hold up the same insert. + child.ExecuteNonQuery("BEGIN TRANSACTION"); + child.ExecuteNonQuery("INSERT INTO Children (Id, ParentId) VALUES (11, 2)"); + parent.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (3)"); + parent.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 3"); + child.ExecuteNonQuery("COMMIT"); + Assert.Single(parent.ExecuteQuery("SELECT Id FROM Children").Rows); + } + finally { TemporaryDatabase.Delete(path); } + } + + // The same write skew from the other end: the delete found no children, and another connection's child insert + // found the parent it had not seen deleted. The delete's dependency is re-checked at its commit. + [Fact] + public void A_parent_delete_cannot_commit_after_another_connection_adds_a_child_for_it() + { + string path = CreateDatabase("write-skew-parent-"); + try + { + using var parentDb = JetDatabase.Open(path, readOnly: false); + using var childDb = JetDatabase.Open(path, readOnly: false); + var parent = new QueryEngine(parentDb); + var child = new QueryEngine(childDb); + + parent.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY)"); + parent.ExecuteNonQuery( + "CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG REFERENCES Parents (Id))"); + parent.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (1)"); + parent.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (2)"); + + parent.ExecuteNonQuery("BEGIN TRANSACTION"); + parent.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 1"); // no children yet + + child.ExecuteNonQuery("INSERT INTO Children (Id, ParentId) VALUES (10, 1)"); // ... until now + + var conflict = Assert.Throws(() => parent.ExecuteNonQuery("COMMIT")); + Assert.Contains("was added for the row", conflict.Message, StringComparison.Ordinal); + + parent.ExecuteNonQuery("ROLLBACK"); + Assert.Equal(2, child.ExecuteQuery("SELECT Id FROM Parents").Rows.Count()); + + // Specific to the key: a child added for a DIFFERENT parent does not hold up the delete. + parent.ExecuteNonQuery("BEGIN TRANSACTION"); + parent.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 2"); + child.ExecuteNonQuery("INSERT INTO Children (Id, ParentId) VALUES (11, 1)"); + parent.ExecuteNonQuery("COMMIT"); + Assert.Single(child.ExecuteQuery("SELECT Id FROM Parents").Rows); + } + finally { TemporaryDatabase.Delete(path); } + } + + // The dependency is on the parent being needed, not on it having been checked: once the child has moved to + // another parent, the one it was inserted against can go, and the commit has nothing to object to. + [Fact] + public void A_child_moved_to_another_parent_commits_after_the_first_parent_is_deleted_by_another_connection() + { + string path = CreateDatabase("write-skew-moved-"); + try + { + using var childDb = JetDatabase.Open(path, readOnly: false); + using var parentDb = JetDatabase.Open(path, readOnly: false); + var child = new QueryEngine(childDb); + var parent = new QueryEngine(parentDb); + + child.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY)"); + child.ExecuteNonQuery( + "CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG REFERENCES Parents (Id))"); + child.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (1)"); + child.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (2)"); + + child.ExecuteNonQuery("BEGIN TRANSACTION"); + child.ExecuteNonQuery("INSERT INTO Children (Id, ParentId) VALUES (10, 1)"); + child.ExecuteNonQuery("UPDATE Children SET ParentId = 2 WHERE Id = 10"); + + parent.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 1"); + + child.ExecuteNonQuery("COMMIT"); + Assert.Equal(2, Convert.ToInt32(parent.ExecuteQuery("SELECT ParentId FROM Children").Rows.Single()[0])); + } + finally { TemporaryDatabase.Delete(path); } + } + + // Renaming inside the transaction changes every name the dependency was registered under — the child table, + // its column, the parent table — but not the relationship, which still joins the same tables over the same + // columns. The parent it relied on is still needed, so its deletion elsewhere still stops the commit. + [Fact] + public void A_child_insert_cannot_commit_after_its_parent_is_deleted_even_when_the_tables_were_renamed() + { + string path = CreateDatabase("write-skew-renamed-"); + try + { + using var childDb = JetDatabase.Open(path, readOnly: false); + using var parentDb = JetDatabase.Open(path, readOnly: false); + var child = new QueryEngine(childDb); + var parent = new QueryEngine(parentDb); + + child.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY)"); + child.ExecuteNonQuery( + "CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG REFERENCES Parents (Id))"); + child.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (1)"); + + child.ExecuteNonQuery("BEGIN TRANSACTION"); + child.ExecuteNonQuery("INSERT INTO Children (Id, ParentId) VALUES (10, 1)"); + child.ExecuteNonQuery("ALTER TABLE Children RENAME TO Kids"); + child.ExecuteNonQuery("ALTER TABLE Kids RENAME COLUMN ParentId TO Pid"); + child.ExecuteNonQuery("ALTER TABLE Parents RENAME TO Folks"); + + parent.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 1"); + + var conflict = Assert.Throws(() => child.ExecuteNonQuery("COMMIT")); + Assert.Contains("no longer exists", conflict.Message, StringComparison.Ordinal); + child.ExecuteNonQuery("ROLLBACK"); + Assert.Empty(parent.ExecuteQuery("SELECT Id FROM Children").Rows); + } + finally { TemporaryDatabase.Delete(path); } + } + + // Adding a relationship checks the child rows already there against their parents, which is the same write + // skew as an insert's check: nothing it read is written. ACE closes it by holding both tables exclusively + // until the transaction ends; LibRed checks the rows again when the transaction commits. The parent goes by a + // key UPDATE rather than a DELETE: a delete rewrites the parent TDEF's row count, and the ADD wrote that page + // too (its incoming relationship block), so the page check alone already refuses that commit. + [Fact] + public void An_added_relationship_cannot_commit_after_a_parent_it_was_checked_against_is_gone() + { + string path = CreateDatabase("write-skew-add-"); + try + { + using var childDb = JetDatabase.Open(path, readOnly: false); + using var parentDb = JetDatabase.Open(path, readOnly: false); + var child = new QueryEngine(childDb); + var parent = new QueryEngine(parentDb); + + child.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY)"); + child.ExecuteNonQuery("CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG)"); + child.ExecuteNonQuery("INSERT INTO Parents (Id) VALUES (1)"); + child.ExecuteNonQuery("INSERT INTO Children (Id, ParentId) VALUES (10, 1)"); + + child.ExecuteNonQuery("BEGIN TRANSACTION"); + child.ExecuteNonQuery( + "ALTER TABLE Children ADD CONSTRAINT FK_Child FOREIGN KEY (ParentId) REFERENCES Parents (Id)"); + + parent.ExecuteNonQuery("UPDATE Parents SET Id = 5 WHERE Id = 1"); + + var conflict = Assert.Throws(() => child.ExecuteNonQuery("COMMIT")); + Assert.Contains("no longer exists", conflict.Message, StringComparison.Ordinal); + child.ExecuteNonQuery("ROLLBACK"); + Assert.Empty(childDb.Catalog.ForeignKeysOf("Children")); + } + finally { TemporaryDatabase.Delete(path); } + } + private static void RunCrossingCommit(string? password) { string path = CreateDatabase("reader-crossing-"); try { if (password is not null) - DatabaseEncryption.SetPassword(path, password, AccessEncryption.Agile); + { + using var encrypt = JetDatabase.Open(path, readOnly: false, exclusive: true); + DatabaseEncryption.SetPassword(encrypt, password, AccessEncryption.Agile); + } using var writerDb = JetDatabase.Open(path, readOnly: false, password: password); using var readerDb = JetDatabase.Open(path, readOnly: false, password: password); diff --git a/test/LibRed.Engine.Tests/RecursiveViewTests.cs b/test/LibRed.Engine.Tests/RecursiveViewTests.cs new file mode 100644 index 000000000..ba8a78863 --- /dev/null +++ b/test/LibRed.Engine.Tests/RecursiveViewTests.cs @@ -0,0 +1,57 @@ +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +// A view whose definition reaches itself, directly or round a longer loop. Nothing stops one being stored — +// the body's sources are resolved when the view is USED, and a file can carry a cycle that Access wrote or +// that a DROP + CREATE left behind — so the expansion is where it has to be caught. Expanding blind is not a +// failed statement but a StackOverflowException, which .NET cannot catch: it takes the process down. +public class RecursiveViewTests : TempDatabaseTest +{ + private static QueryEngine Fresh() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "recursive-view-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + e.ExecuteNonQuery("CREATE TABLE T (K LONG)"); + return e; + } + + [Fact] + public void A_view_defined_in_terms_of_itself_is_reported() + { + var e = Fresh(); + e.ExecuteNonQuery("CREATE VIEW V AS SELECT * FROM T"); + e.ExecuteNonQuery("DROP VIEW V"); + e.ExecuteNonQuery("CREATE VIEW V AS SELECT * FROM V"); + + var error = Assert.Throws(() => e.ExecuteQuery("SELECT * FROM V")); + Assert.Contains("defined in terms of itself", error.Message); + } + + // The same cycle one hop longer, which a naive "is this the view we started from" check would miss. + [Fact] + public void A_mutually_recursive_view_pair_is_reported_too() + { + var e = Fresh(); + e.ExecuteNonQuery("CREATE VIEW V1 AS SELECT * FROM T"); + e.ExecuteNonQuery("CREATE VIEW V2 AS SELECT * FROM V1"); + e.ExecuteNonQuery("DROP VIEW V1"); + e.ExecuteNonQuery("CREATE VIEW V1 AS SELECT * FROM V2"); + + Assert.Throws(() => e.ExecuteQuery("SELECT * FROM V1")); + } + + // And the shape that must NOT be mistaken for a cycle: one view used twice in a query still expands. + [Fact] + public void The_same_view_used_twice_in_one_query_still_expands() + { + var e = Fresh(); + e.ExecuteNonQuery("INSERT INTO T (K) VALUES (1)"); + e.ExecuteNonQuery("CREATE VIEW V AS SELECT * FROM T"); + + var result = e.ExecuteQuery("SELECT A.K FROM V AS A INNER JOIN V AS B ON A.K = B.K"); + Assert.Equal(1, Convert.ToInt32(result.Rows.Single()[0])); + } +} diff --git a/test/LibRed.Engine.Tests/ReferencesAndIdentityTests.cs b/test/LibRed.Engine.Tests/ReferencesAndIdentityTests.cs index 268977f39..e08072cda 100644 --- a/test/LibRed.Engine.Tests/ReferencesAndIdentityTests.cs +++ b/test/LibRed.Engine.Tests/ReferencesAndIdentityTests.cs @@ -121,7 +121,7 @@ public void Add_column_creates_its_primary_key_and_unique_index() JetDatabase db = Run("CREATE TABLE T (V TEXT(10))", "ALTER TABLE T ADD COLUMN Id LONG CONSTRAINT pk PRIMARY KEY", "ALTER TABLE T ADD COLUMN Code TEXT(10) CONSTRAINT uq UNIQUE"); - TableDef t = db.Catalog.FindTable("T")!; + TableDefinition t = db.Catalog.FindTable("T")!; IndexDef pk = Assert.Single(t.Indexes, i => i.IsPrimaryKey); Assert.Equal("pk", pk.Name); IndexDef uq = Assert.Single(t.Indexes, i => i.Name == "uq"); @@ -239,4 +239,4 @@ public void Alter_table_takes_identity_too() db = Run("CREATE TABLE T (Id INT, V TEXT(10))", "ALTER TABLE T ALTER COLUMN Id INT IDENTITY"); Assert.True(db.Catalog.FindTable("T")!.FindColumn("Id")!.IsAutoNumber); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/ReferentialActionTests.cs b/test/LibRed.Engine.Tests/ReferentialActionTests.cs index 97cefe91d..60cee541c 100644 --- a/test/LibRed.Engine.Tests/ReferentialActionTests.cs +++ b/test/LibRed.Engine.Tests/ReferentialActionTests.cs @@ -95,6 +95,70 @@ public void Set_null_delete_nulls_the_children_fk() }); } + // SET NULL on a SELF-referencing table, where the row being nulled and the row being deleted are rows of + // one table. The action reads each child from the snapshot it took before the delete, so a child that the + // delete has already removed — or that an earlier child's nulling has already rewritten — is looked up by + // a key the table no longer holds, and the index says "entry not found". + [Fact] + public void Set_null_delete_on_a_self_referencing_table_nulls_the_children() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery( + "CREATE TABLE T (Id long PRIMARY KEY, ParentId long, " + + "CONSTRAINT FK_T FOREIGN KEY (ParentId) REFERENCES T (Id) ON DELETE SET NULL)"); + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (1, NULL)"); // the root + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (2, 1)"); // three children of it + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (3, 1)"); + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (4, 1)"); + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (5, 2)"); // and a grandchild + + Assert.Equal(1, e.ExecuteNonQuery("DELETE FROM T WHERE Id = 1")); + + // The root is gone, its three children point at nothing, and the grandchild is untouched. + var rows = e.ExecuteQuery("SELECT Id, ParentId FROM T ORDER BY Id").Rows + .Select(r => (Convert.ToInt32(r[0]), r[1] is null ? -1 : Convert.ToInt32(r[1]))).ToArray(); + Assert.Equal([(2, -1), (3, -1), (4, -1), (5, 2)], rows); + } + finally { TemporaryDatabase.Delete(path); } + } + + // The harder half: the row the action nulls is ALSO one the statement is deleting. Deleting the parent + // rewrites the child's FK, and the statement then reaches that child carrying the values it read before — + // whose key no longer names anything in the index. + [Fact] + public void Set_null_delete_reaches_a_child_the_same_statement_is_deleting() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery( + "CREATE TABLE T (Id long PRIMARY KEY, ParentId long, " + + "CONSTRAINT FK_T FOREIGN KEY (ParentId) REFERENCES T (Id) ON DELETE SET NULL)"); + e.ExecuteNonQuery("CREATE INDEX IX_Parent ON T (ParentId)"); + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (1, NULL)"); + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (2, 1)"); + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (3, 2)"); + e.ExecuteNonQuery("INSERT INTO T (Id, ParentId) VALUES (4, 1)"); + + // 1 and 2 go together, and 2 is 1's child — so the action rewrites a row the delete also removes. + Assert.Equal(2, e.ExecuteNonQuery("DELETE FROM T WHERE Id = 1 OR Id = 2")); + + var rows = e.ExecuteQuery("SELECT Id, ParentId FROM T ORDER BY Id").Rows + .Select(r => (Convert.ToInt32(r[0]), r[1] is null ? -1 : Convert.ToInt32(r[1]))).ToArray(); + Assert.Equal([(3, -1), (4, -1)], rows); + + // And the index agrees with the rows: a seek on the nulled column finds both survivors. + Assert.Equal(2, e.ExecuteQuery("SELECT Id FROM T WHERE ParentId IS NULL").Rows.Count()); + } + finally { TemporaryDatabase.Delete(path); } + } + // ON UPDATE SET NULL is a pathway only — its Jet storage bytes are unverified (ACE's OLE DB provider // rejects the DDL), so creating one throws NotImplemented rather than guessing the bytes. [Fact] @@ -127,4 +191,135 @@ public void Set_null_action_persists_and_reads_back() } finally { TemporaryDatabase.Delete(path); } } + + // A cascade rewrites a child row, so it owes that row every invariant an UPDATE of it would: SET NULL may + // not null a Required column, and CASCADE may not drive two children onto one unique key. Table.Update + // enforces nothing itself, so without these checks the statement — which never names the child table — + // writes what the UPDATE path a few lines away explicitly refuses. + [Fact] + public void Set_null_refuses_to_null_a_required_child_column() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE TABLE P (Id long PRIMARY KEY)"); + e.ExecuteNonQuery("CREATE TABLE C (Id long PRIMARY KEY, ParentId long NOT NULL, " + + "CONSTRAINT FK_C FOREIGN KEY (ParentId) REFERENCES P (Id) ON DELETE SET NULL)"); + e.ExecuteNonQuery("INSERT INTO P (Id) VALUES (1)"); + e.ExecuteNonQuery("INSERT INTO C (Id, ParentId) VALUES (100, 1)"); + + Assert.ThrowsAny(() => e.ExecuteNonQuery("DELETE FROM P WHERE Id = 1")); + + // The child row is intact, not half-nulled. + Assert.Equal(1, Convert.ToInt32(e.ExecuteQuery("SELECT ParentId FROM C WHERE Id = 100").Rows.Single()[0])); + } + finally { TemporaryDatabase.Delete(path); } + } + + [Fact] + public void Cascade_refuses_to_create_a_duplicate_child_key() + { + Run(" ON UPDATE CASCADE", e => + { + // One child per parent, and a unique index over the FK column: moving parent 1 onto 2 would + // cascade its child onto the other child's key. + e.ExecuteNonQuery("DELETE FROM C WHERE Id = 101"); + e.ExecuteNonQuery("INSERT INTO C (Id, ParentId) VALUES (102, 2)"); + e.ExecuteNonQuery("CREATE UNIQUE INDEX UX_C ON C (ParentId)"); + + Assert.ThrowsAny(() => e.ExecuteNonQuery("UPDATE P SET Id = 2 WHERE Id = 1")); + }); + } + + // SET NULL may not null a primary-key column either — the same rule an UPDATE of the child meets. + [Fact] + public void Set_null_refuses_to_null_a_primary_key_column() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE TABLE P (Id long PRIMARY KEY)"); + e.ExecuteNonQuery("CREATE TABLE C (ParentId long, X long, CONSTRAINT PK_C PRIMARY KEY (ParentId, X), " + + "CONSTRAINT FK_C FOREIGN KEY (ParentId) REFERENCES P (Id) ON DELETE SET NULL)"); + e.ExecuteNonQuery("INSERT INTO P (Id) VALUES (1)"); + e.ExecuteNonQuery("INSERT INTO C (ParentId, X) VALUES (1, 1)"); + + var error = Assert.Throws(() => e.ExecuteNonQuery("DELETE FROM P WHERE Id = 1")); + Assert.Contains("cannot contain a Null value", error.Message, StringComparison.Ordinal); + Assert.Equal(1, Convert.ToInt32(e.ExecuteQuery("SELECT ParentId FROM C").Rows.Single()[0])); + } + finally { TemporaryDatabase.Delete(path); } + } + + // A child whose key a cascade rewrites is a parent in its own right: A -> B -> C, each ON UPDATE CASCADE over + // B's unique Aid, carries A's new key all the way down. + [Fact] + public void Cascade_update_continues_down_a_chain() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = ChainOf(db, " ON UPDATE CASCADE", " ON UPDATE CASCADE"); + + e.ExecuteNonQuery("UPDATE A SET Id = 5"); + + Assert.Equal(5, Convert.ToInt32(e.ExecuteQuery("SELECT Aid FROM B").Rows.Single()[0])); + Assert.Equal(5, Convert.ToInt32(e.ExecuteQuery("SELECT Baid FROM C").Rows.Single()[0])); + } + finally { TemporaryDatabase.Delete(path); } + } + + // ON DELETE SET NULL changes the child's key, so the rows referencing THAT key follow their own rule: a + // cascade takes them to null too, and NO ACTION refuses — neither leaves a row pointing at a key that is gone. + [Fact] + public void Set_null_onto_a_referenced_key_cascades_to_its_children() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = ChainOf(db, " ON DELETE SET NULL", " ON UPDATE CASCADE"); + + e.ExecuteNonQuery("DELETE FROM A"); + + Assert.Null(e.ExecuteQuery("SELECT Aid FROM B").Rows.Single()[0]); + Assert.Null(e.ExecuteQuery("SELECT Baid FROM C").Rows.Single()[0]); + } + finally { TemporaryDatabase.Delete(path); } + } + + [Fact] + public void Set_null_onto_a_referenced_key_with_no_action_children_is_refused() + { + string path = Fresh(); + try + { + using var db = JetDatabase.Open(path, readOnly: false); + var e = ChainOf(db, " ON DELETE SET NULL", ""); + + var error = Assert.Throws(() => e.ExecuteNonQuery("DELETE FROM A")); + Assert.Contains("table 'C' includes related records", error.Message, StringComparison.Ordinal); + Assert.Equal(1, Convert.ToInt32(e.ExecuteQuery("SELECT Aid FROM B").Rows.Single()[0])); + } + finally { TemporaryDatabase.Delete(path); } + } + + // A(Id) <- B(Aid, unique) <- C(Baid), one row each, all holding key 1. + private static QueryEngine ChainOf(JetDatabase db, string bRule, string cRule) + { + var e = new QueryEngine(db); + e.ExecuteNonQuery("CREATE TABLE A (Id long PRIMARY KEY)"); + e.ExecuteNonQuery($"CREATE TABLE B (Id long PRIMARY KEY, Aid long, CONSTRAINT FK_B FOREIGN KEY (Aid) REFERENCES A (Id){bRule})"); + e.ExecuteNonQuery("CREATE UNIQUE INDEX UX_B ON B (Aid)"); + e.ExecuteNonQuery($"CREATE TABLE C (Id long PRIMARY KEY, Baid long, CONSTRAINT FK_C FOREIGN KEY (Baid) REFERENCES B (Aid){cRule})"); + e.ExecuteNonQuery("INSERT INTO A (Id) VALUES (1)"); + e.ExecuteNonQuery("INSERT INTO B (Id, Aid) VALUES (1, 1)"); + e.ExecuteNonQuery("INSERT INTO C (Id, Baid) VALUES (1, 1)"); + return e; + } } diff --git a/test/LibRed.Engine.Tests/ResultColumnTypeTests.cs b/test/LibRed.Engine.Tests/ResultColumnTypeTests.cs index 1147d14ef..702690917 100644 --- a/test/LibRed.Engine.Tests/ResultColumnTypeTests.cs +++ b/test/LibRed.Engine.Tests/ResultColumnTypeTests.cs @@ -132,6 +132,73 @@ public void A_money_sum_with_a_written_zero_stays_money() Assert.IsType(rows.Single()[0]); } + // A scalar subquery declares its column's type, so a choice over it widens with it. Untyped, the Integer 0 alone + // declared IIF(x IS NULL, 0, x): the money sum doubled came back a Decimal under a declared Integer, and the + // other column's value was rounded into one. + [Fact] + public void A_scalar_subquery_declares_its_type_through_a_choice() + { + var (types, rows) = database.Query( + "SELECT IIF(t3.x < 0, 9, t3.x + 8), t3.x + t3.x FROM (SELECT IIF(t2.x IS NULL, 0, t2.x) AS x " + + "FROM (SELECT (SELECT SUM(M) FROM T) AS x FROM T q) t2) t3", CultureInfo.GetCultureInfo("en-US")); + Assert.Equal([typeof(decimal), typeof(decimal)], types); + Assert.All(rows, row => Assert.Equal(new object?[] { 20.5m, 25m }, row)); + } + + [Theory] + // Rows taking different arms still come back in the one declared type. + [InlineData("IIF(Id = 1, 9, (SELECT SUM(M) FROM T))", 9, 12.5)] + // ... and a choice inside arithmetic too: linq2db's coalesced sum, taking the 0 arm, came back an Integer. + [InlineData("1000 - IIF(Id = 1, 0, (SELECT SUM(M) FROM T))", 1000, 987.5)] + // A correlated subquery is typed without the outer row. + [InlineData("(SELECT SUM(i.M) FROM T i WHERE i.Id = o.Id)", 10.5, 2)] + public void A_scalar_subquery_column_is_its_type_on_every_row(string expression, double first, double second) + { + (Type declared, object?[] values) = Column($"SELECT o.Id, {expression} AS c FROM T o ORDER BY o.Id"); + Assert.Equal(typeof(decimal), declared); + Assert.Equal(new object?[] { (decimal)first, (decimal)second }, values); + } + + // A choice with an arm nothing can type declares nothing, rather than letting its typed arms declare alone. + [Fact] + public void A_choice_with_an_untyped_arm_declares_nothing() => + Assert.Equal(typeof(object), + database.Query("SELECT IIF(Id = 1, 0, (SELECT o.M FROM T i WHERE i.Id = 1)) FROM T o", + CultureInfo.GetCultureInfo("en-US")).ColumnTypes[0]); + + // A Variant keeps its own type through an expression and is written out as text; a Mixed choice — text beside + // another kind — is text; either as an operand counts as a Double (verified vs ACE in VariantAccessTests). + [Theory] + [InlineData("CVar(B)", typeof(string), "3", "4")] + [InlineData("CVar(B) + CVar(B)", typeof(string), "6", "8")] + [InlineData("CVar(B) + 1", typeof(double), 4.0, 5.0)] + [InlineData("CVar(B) * CVar(B)", typeof(double), 9.0, 16.0)] + [InlineData("IIF(Id = 1, CVar(B), 5)", typeof(string), "3", "5")] + [InlineData("IIF(Id = 1, X, 2)", typeof(string), "abc", "2")] + [InlineData("IIF(Id = 1, Y, 2)", typeof(int), -1, 2)] + public void A_variant_or_a_mixed_choice_is_typed_as_ace_types_it(string expression, Type declared, object? first, object? second) + { + (Type type, object?[] values) = Column($"SELECT Id, {expression} AS c FROM T ORDER BY Id"); + Assert.Equal(declared, type); + Assert.Equal([first, second], values); + } + + // A Variant sorts, groups and takes Max as its text, so 200000 comes before 70000. + [Fact] + public void A_variant_sorts_and_takes_max_as_its_text() + { + var culture = CultureInfo.GetCultureInfo("en-US"); + Assert.Equal([2, 1], database.Query("SELECT Id FROM T ORDER BY CVar(L)", culture).Rows.Select(r => r[0])); + Assert.Equal("70000", database.Query("SELECT MAX(CVar(L)) FROM T", culture).Rows.Single()[0]); + } + + // CVar(Null) is left untyped, as a bare Null is. ACE makes a union with an arm of them text, but EFCore.Jet writes + // one for every projected Null, and LibRed keeps the union's values as the other arm has them. + [Fact] + public void A_union_with_a_cvar_null_arm_keeps_the_other_arms_values() => + Assert.Equal([null, (byte)3], + database.Query(Union("CVar(NULL)", "B"), CultureInfo.GetCultureInfo("en-US")).Rows.Select(r => r[0])); + [Fact] public void Values_keep_their_widened_value() { diff --git a/test/LibRed.Engine.Tests/RowScopeReuseTests.cs b/test/LibRed.Engine.Tests/RowScopeReuseTests.cs new file mode 100644 index 000000000..08474cf43 --- /dev/null +++ b/test/LibRed.Engine.Tests/RowScopeReuseTests.cs @@ -0,0 +1,95 @@ +using LibRed; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// Filter, projection, sort, grouping and aggregation evaluate every row through one scope and evaluator, +/// rebound per row, instead of allocating a pair per row. The rebind is only sound if nothing outlives its row +/// and no two enumerations share a scope, so these pin both: a result enumerated twice, or two enumerations +/// interleaved, each see their own rows, and a correlated subquery sees the row it was asked about. +/// +public class RowScopeReuseTests : TempDatabaseTest +{ + private static QueryEngine Seeded() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "row-scope-"); + var e = new QueryEngine(TemporaryDatabase.OpenTracked(path, readOnly: false)); + + e.ExecuteNonQuery("CREATE TABLE T (Id LONG PRIMARY KEY, Grp LONG, V LONG)"); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + for (int i = 1; i <= 12; i++) + e.ExecuteNonQuery($"INSERT INTO T (Id, Grp, V) VALUES ({i}, {i % 3}, {i * 10})"); + e.ExecuteNonQuery("COMMIT"); + return e; + } + + private const string FilteredProjection = "SELECT Id, V + Id FROM T WHERE V + Id > 40"; + + private static readonly (int Id, int Sum)[] Expected = + [.. Enumerable.Range(4, 9).Select(i => (i, i * 11))]; + + private static (int, int) Read(object?[] row) => (Convert.ToInt32(row[0]), Convert.ToInt32(row[1])); + + [Fact] + public void A_result_enumerated_twice_gives_the_same_rows() + { + IEnumerable rows = Seeded().ExecuteQuery(FilteredProjection).Rows; + Assert.Equal(Expected, rows.Select(Read).ToArray()); + Assert.Equal(Expected, rows.Select(Read).ToArray()); + } + + [Fact] + public void Interleaved_enumerations_do_not_move_each_others_row() + { + IEnumerable rows = Seeded().ExecuteQuery(FilteredProjection).Rows; + using IEnumerator ahead = rows.GetEnumerator(); + using IEnumerator behind = rows.GetEnumerator(); + + var fromAhead = new List<(int, int)>(); + var fromBehind = new List<(int, int)>(); + Assert.True(ahead.MoveNext()); + fromAhead.Add(Read(ahead.Current)); + while (true) + { + bool a = ahead.MoveNext(); + if (a) fromAhead.Add(Read(ahead.Current)); + bool b = behind.MoveNext(); + if (b) fromBehind.Add(Read(behind.Current)); + if (!a && !b) break; + } + + Assert.Equal(Expected, fromAhead); + Assert.Equal(Expected, fromBehind); + } + + [Fact] + public void A_correlated_subquery_sees_the_row_being_filtered_and_projected() + { + var rows = Seeded().ExecuteQuery( + "SELECT t.Id, (SELECT COUNT(*) FROM T AS i WHERE i.Grp = t.Grp AND i.Id < t.Id) FROM T AS t " + + "WHERE (SELECT MAX(i.Id) FROM T AS i WHERE i.Grp = t.Grp) > t.Id").Rows + .Select(Read).ToArray(); + + // Ids 1..9 each have a later row in their group; each is preceded by (Id - 1) / 3 earlier ones. + Assert.Equal([.. Enumerable.Range(1, 9).Select(i => (i, (i - 1) / 3))], rows); + } + + [Fact] + public void Aggregates_evaluate_each_row_of_their_own_group() + { + var rows = Seeded().ExecuteQuery( + "SELECT Grp, SUM(V), FIRST(Id), LAST(Id), COUNT(*) FILTER (WHERE V > 60) FROM T GROUP BY Grp").Rows + .Select(r => r.Select(Convert.ToInt32).ToArray()).ToArray(); + + Assert.Equal( + [ + [0, 300, 3, 12, 2], // 3, 6, 9, 12 + [1, 220, 1, 10, 2], // 1, 4, 7, 10 + [2, 260, 2, 11, 2], // 2, 5, 8, 11 + ], + rows); + } +} diff --git a/test/LibRed.Engine.Tests/SchemaVisibilityTests.cs b/test/LibRed.Engine.Tests/SchemaVisibilityTests.cs index af8a801ed..763c50dcb 100644 --- a/test/LibRed.Engine.Tests/SchemaVisibilityTests.cs +++ b/test/LibRed.Engine.Tests/SchemaVisibilityTests.cs @@ -73,7 +73,7 @@ public void Ordinary_dml_on_one_connection_does_not_invalidate_another_catalog() first.ExecuteNonQuery("INSERT INTO Rows1 (Id) VALUES (1)"); - // Same TableDef instance: the DML did not force the second connection to re-read the catalog. + // Same TableDefinition instance: the DML did not force the second connection to re-read the catalog. Assert.Same(cached, secondDb.Catalog.Tables.Single(t => t.Name == "Rows1")); } finally { TemporaryDatabase.Delete(path); } @@ -81,4 +81,4 @@ public void Ordinary_dml_on_one_connection_does_not_invalidate_another_catalog() private static string Fresh(string prefix) => TemporaryDatabase.CopyPath(Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), prefix); -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/SelectIntoTests.cs b/test/LibRed.Engine.Tests/SelectIntoTests.cs index d2b500374..df8f5647e 100644 --- a/test/LibRed.Engine.Tests/SelectIntoTests.cs +++ b/test/LibRed.Engine.Tests/SelectIntoTests.cs @@ -54,8 +54,8 @@ public void Does_not_copy_the_primary_key_or_indexes() QueryEngine engine = WithSource(); engine.ExecuteNonQuery("SELECT * INTO SiNew FROM SiSrc"); - TableDef source = engine.Database.Catalog.Tables.Single(t => t.Name == "SiSrc"); - TableDef made = engine.Database.Catalog.Tables.Single(t => t.Name == "SiNew"); + TableDefinition source = engine.Database.Catalog.Tables.Single(t => t.Name == "SiSrc"); + TableDefinition made = engine.Database.Catalog.Tables.Single(t => t.Name == "SiNew"); Assert.NotEmpty(source.Indexes); // the source has a PK and an index Assert.Empty(made.Indexes); // the copy has neither @@ -102,6 +102,18 @@ public void An_existing_target_is_an_error() Assert.Equal("SiNew", error.ObjectName); } + // A query's name is taken too: tables and queries share the Tables container's name space. + [Fact] + public void A_query_named_like_the_target_is_an_error() + { + QueryEngine engine = WithSource(); + engine.ExecuteNonQuery("CREATE VIEW SiView AS SELECT Id FROM SiSrc"); + + var error = Assert.Throws( + () => engine.ExecuteNonQuery("SELECT Id INTO SiView FROM SiSrc")); + Assert.Equal("Table 'SiView' already exists.", error.Message); + } + // An expression column is typed from the expression, since there is no source column to copy. [Fact] public void An_expression_column_is_typed_from_the_expression() @@ -109,7 +121,7 @@ public void An_expression_column_is_typed_from_the_expression() QueryEngine engine = WithSource(); engine.ExecuteNonQuery("SELECT Id, Qty * 2 AS Doubled, Label & '!' AS Shout INTO SiNew FROM SiSrc"); - TableDef made = engine.Database.Catalog.Tables.Single(t => t.Name == "SiNew"); + TableDefinition made = engine.Database.Catalog.Tables.Single(t => t.Name == "SiNew"); Assert.Equal(["Id", "Doubled", "Shout"], made.Columns.Select(c => c.Name)); Assert.Equal(JetDataType.Int32, made.Columns[1].Type); Assert.Equal(JetDataType.Text, made.Columns[2].Type); @@ -140,4 +152,4 @@ public void Returns_no_rows_to_the_caller() Assert.Empty(result.Rows); Assert.Equal(2, Convert.ToInt32(engine.ExecuteQuery("SELECT COUNT(*) FROM SiNew").Rows.Single()[0])); } -} +} \ No newline at end of file diff --git a/test/LibRed.Engine.Tests/SqlTransactionControlTests.cs b/test/LibRed.Engine.Tests/SqlTransactionControlTests.cs index 0535d8543..314831ee4 100644 --- a/test/LibRed.Engine.Tests/SqlTransactionControlTests.cs +++ b/test/LibRed.Engine.Tests/SqlTransactionControlTests.cs @@ -68,6 +68,40 @@ public void A_nested_commit_leaves_its_work_under_the_outer_which_can_still_roll Assert.Empty(Ids(e)); } + // A statement that fails inside a transaction runs under a savepoint of its own, which is rolled back — and + // has to be closed as well. Left open it sits above the savepoint the enclosing BEGIN pushed, so that + // level's COMMIT is no longer releasing the innermost savepoint and refuses: one failed statement would + // make every later nested COMMIT throw, on a transaction the failure was supposed to leave intact. + [Fact] + public void A_failed_statement_leaves_the_enclosing_levels_committable() + { + var e = Fresh(); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + e.ExecuteNonQuery("INSERT INTO t (id) VALUES (1)"); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + e.ExecuteNonQuery("INSERT INTO t (id) VALUES (2)"); + Assert.Throws(() => e.ExecuteNonQuery("INSERT INTO t (id) VALUES (2)")); + + e.ExecuteNonQuery("COMMIT"); // inner: releases its savepoint + e.ExecuteNonQuery("COMMIT"); // outer: commits the transaction + Assert.Equal([1, 2], Ids(e)); + } + + [Fact] + public void A_failed_statement_leaves_the_enclosing_levels_rollable() + { + var e = Fresh(); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + e.ExecuteNonQuery("INSERT INTO t (id) VALUES (1)"); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + e.ExecuteNonQuery("INSERT INTO t (id) VALUES (2)"); + Assert.Throws(() => e.ExecuteNonQuery("INSERT INTO t (id) VALUES (1)")); + + e.ExecuteNonQuery("ROLLBACK"); // inner: undoes id=2 only + e.ExecuteNonQuery("COMMIT"); // outer: keeps id=1 + Assert.Equal([1], Ids(e)); + } + [Fact] public void Commit_with_no_transaction_open_throws() { @@ -75,6 +109,41 @@ public void Commit_with_no_transaction_open_throws() Assert.Throws(() => e.ExecuteNonQuery("COMMIT")); } + [Theory] + [InlineData("DELETE FROM Children WHERE Id = 10", 0)] + [InlineData("UPDATE Children SET ParentId = 2 WHERE Id = 10", 1)] + [InlineData("UPDATE Children SET ParentId = NULL WHERE Id = 10", 1)] + public void Commit_does_not_require_a_parent_after_its_child_reference_is_removed(string change, int children) + { + QueryEngine e = Fresh(); + e.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY)"); + e.ExecuteNonQuery("CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG REFERENCES Parents (Id))"); + e.ExecuteNonQuery("INSERT INTO Parents VALUES (1), (2)"); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + e.ExecuteNonQuery("INSERT INTO Children VALUES (10, 1)"); + e.ExecuteNonQuery(change); + e.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 1"); + e.ExecuteNonQuery("COMMIT"); + Assert.Equal(children, Convert.ToInt32(e.ExecuteQuery("SELECT COUNT(*) FROM Children").Rows.Single()[0])); + Assert.Equal(2, Convert.ToInt32(e.ExecuteQuery("SELECT Id FROM Parents").Rows.Single()[0])); + } + + [Fact] + public void Commit_does_not_require_a_parent_after_its_relationship_is_dropped() + { + QueryEngine e = Fresh(); + e.ExecuteNonQuery("CREATE TABLE Parents (Id LONG PRIMARY KEY)"); + e.ExecuteNonQuery("CREATE TABLE Children (Id LONG PRIMARY KEY, ParentId LONG, CONSTRAINT FK_Child FOREIGN KEY (ParentId) REFERENCES Parents (Id))"); + e.ExecuteNonQuery("INSERT INTO Parents VALUES (1)"); + e.ExecuteNonQuery("BEGIN TRANSACTION"); + e.ExecuteNonQuery("INSERT INTO Children VALUES (10, 1)"); + e.ExecuteNonQuery("ALTER TABLE Children DROP CONSTRAINT FK_Child"); + e.ExecuteNonQuery("DELETE FROM Parents WHERE Id = 1"); + e.ExecuteNonQuery("COMMIT"); + Assert.Single(e.ExecuteQuery("SELECT * FROM Children").Rows); + Assert.Empty(e.ExecuteQuery("SELECT * FROM Parents").Rows); + } + [Fact] public void Sql_outer_transaction_and_ado_inner_transaction_share_one_controller() { diff --git a/test/LibRed.Engine.Tests/TopWithTiesTests.cs b/test/LibRed.Engine.Tests/TopWithTiesTests.cs new file mode 100644 index 000000000..e16d275b2 --- /dev/null +++ b/test/LibRed.Engine.Tests/TopWithTiesTests.cs @@ -0,0 +1,93 @@ +using System.Globalization; +using Xunit; + +namespace LibRed.Engine.Tests; + +/// +/// TOP n [PERCENT] WITH TIES and FETCH … WITH TIES: the n rows, and every further row whose ORDER BY keys +/// equal the last one's. ACE's own TOP n always keeps the ties, and the rows each case expects here are the ones +/// ACE's plain TOP returned over the same data (verified; TopWithTiesAccessTests checks it). LibRed's plain TOP +/// stays exactly n. +/// +public class TopWithTiesTests(TopWithTiesTests.Database database) : TempDatabaseTest, IClassFixture +{ + // K ties three ways at 2 and two ways at 3 and at Null (which sorts first); K2 breaks some of those ties. + private static readonly string[] Setup = + [ + "CREATE TABLE T (Id LONG, K LONG, K2 TEXT(10))", + "INSERT INTO T VALUES (1, 1, 'a')", + "INSERT INTO T VALUES (2, 2, 'b')", + "INSERT INTO T VALUES (3, 2, 'a')", + "INSERT INTO T VALUES (4, 2, 'b')", + "INSERT INTO T VALUES (5, 3, 'a')", + "INSERT INTO T VALUES (6, 3, 'a')", + "INSERT INTO T VALUES (7, 10, 'c')", + "INSERT INTO T VALUES (8, NULL, NULL)", + "INSERT INTO T VALUES (9, NULL, 'z')", + ]; + + public sealed class Database() : SharedDatabase("ties-", Setup); + + private string Ids(string sql) => + string.Join(",", database.Query(sql, CultureInfo.InvariantCulture).Rows.Select(row => Convert.ToString(row[0], CultureInfo.InvariantCulture))); + + [Theory] + [InlineData("SELECT TOP 1 WITH TIES Id FROM T ORDER BY K", "8,9")] + [InlineData("SELECT TOP 2 WITH TIES Id FROM T ORDER BY K", "8,9")] + [InlineData("SELECT TOP 3 WITH TIES Id FROM T ORDER BY K", "8,9,1")] + [InlineData("SELECT TOP 4 WITH TIES Id FROM T ORDER BY K", "8,9,1,2,3,4")] + [InlineData("SELECT TOP 6 WITH TIES Id FROM T ORDER BY K", "8,9,1,2,3,4")] + [InlineData("SELECT TOP 3 WITH TIES Id FROM T ORDER BY K, K2", "8,9,1")] + [InlineData("SELECT TOP 1 WITH TIES Id FROM T ORDER BY K DESC", "7")] + [InlineData("SELECT TOP 2 WITH TIES Id FROM T ORDER BY K DESC", "7,5,6")] + [InlineData("SELECT TOP 5 WITH TIES Id FROM T ORDER BY K2", "8,1,3,5,6")] + [InlineData("SELECT TOP 2 WITH TIES Id FROM T ORDER BY K MOD 2", "8,9")] + [InlineData("SELECT TOP 2 WITH TIES Id FROM T WHERE K IS NOT NULL ORDER BY K", "1,2,3,4")] + [InlineData("SELECT TOP 30 PERCENT WITH TIES Id FROM T ORDER BY K", "8,9,1")] + [InlineData("SELECT TOP 50 PERCENT WITH TIES Id FROM T ORDER BY K", "8,9,1,2,3,4")] + [InlineData("SELECT TOP 0 WITH TIES Id FROM T ORDER BY K", "")] + public void The_ties_of_the_last_row_are_kept(string sql, string expected) => + Assert.Equal(expected, Ids(sql)); + + [Fact] + public void Plain_top_stays_exactly_n() => + Assert.Equal("8,9,1,2", Ids("SELECT TOP 4 Id FROM T ORDER BY K")); + + [Fact] + public void Ties_are_cut_across_a_join() => + Assert.Equal("5,6", Ids("SELECT TOP 1 WITH TIES a.Id FROM T AS a INNER JOIN T AS b ON a.Id = b.Id WHERE a.K < 10 ORDER BY a.K DESC")); + + // Groups by K2: a has 4 rows, b 2, and c, z and Null 1 each — so the third place is a three-way tie. + [Fact] + public void A_grouped_query_ties_its_groups() + { + var rows = database.Query( + "SELECT TOP 3 WITH TIES K2, COUNT(*) AS N FROM T GROUP BY K2 ORDER BY COUNT(*) DESC", CultureInfo.InvariantCulture).Rows; + Assert.Equal(["a:4", "b:2"], rows.Take(2).Select(row => $"{row[0]}:{row[1]}")); + Assert.Equal(["(null):1", "c:1", "z:1"], rows.Skip(2).Select(row => $"{row[0] ?? "(null)"}:{row[1]}").Order()); + } + + [Theory] + [InlineData("SELECT Id FROM T ORDER BY K FETCH FIRST 3 ROWS WITH TIES", "8,9,1")] + [InlineData("SELECT Id FROM T ORDER BY K OFFSET 2 ROWS FETCH NEXT 2 ROWS WITH TIES", "1,2,3,4")] + [InlineData("SELECT Id FROM T ORDER BY K OFFSET 2 ROWS FETCH NEXT 1 ROW WITH TIES", "1")] + [InlineData("SELECT Id FROM T UNION ALL SELECT Id FROM T ORDER BY Id FETCH FIRST 1 ROWS WITH TIES", "1,1")] + public void Fetch_takes_with_ties_as_well(string sql, string expected) => + Assert.Equal(expected, Ids(sql)); + + [Fact] + public void With_ties_needs_an_order_by() => + Assert.Throws(() => Ids("SELECT TOP 2 WITH TIES Id FROM T")); + + [Fact] + public void With_ties_is_refused_beside_distinct() => + Assert.Throws(() => Ids("SELECT DISTINCT TOP 2 WITH TIES K FROM T ORDER BY K")); + + [Fact] + public void A_view_cannot_store_with_ties() + { + QueryEngine engine = SharedDatabase.Fresh("ties-view-", Setup); + Assert.Throws(() => + engine.ExecuteNonQuery("CREATE VIEW V AS SELECT TOP 2 WITH TIES Id FROM T ORDER BY K")); + } +} diff --git a/test/LibRed.Engine.Tests/UnnamedConstraintNameTests.cs b/test/LibRed.Engine.Tests/UnnamedConstraintNameTests.cs new file mode 100644 index 000000000..7e5134780 --- /dev/null +++ b/test/LibRed.Engine.Tests/UnnamedConstraintNameTests.cs @@ -0,0 +1,62 @@ +using LibRed; +using LibRed.Catalog; +using LibRed.Engine; +using Xunit; + +namespace LibRed.Engine.Tests; + +// A UNIQUE or CHECK constraint written without a name gets one made from the table's: UQ_
_. A name +// over 64 characters is not merely rejected by ACE — it makes ACE refuse the whole file ("Unrecognized +// database format"), so the generated name has to fit whatever the table is called, and a table may use all 64 +// characters itself. The table's name is truncated to make room rather than the constraint being refused. +public class UnnamedConstraintNameTests : TempDatabaseTest +{ + private static (QueryEngine Engine, JetDatabase Db) Fresh() + { + string path = TemporaryDatabase.CopyPath( + Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"), "unnamed-constraint-"); + var db = TemporaryDatabase.OpenTracked(path, readOnly: false); + return (new QueryEngine(db), db); + } + + [Theory] + [InlineData(4)] + [InlineData(JetName.MaxLength)] // a table using every character it is allowed + public void An_unnamed_unique_constraint_gets_a_name_that_fits(int tableNameLength) + { + var (e, db) = Fresh(); + string table = new('T', tableNameLength); + + e.ExecuteNonQuery($"CREATE TABLE [{table}] (Id LONG PRIMARY KEY, V TEXT(10), UNIQUE (V))"); + + IndexDef unique = Assert.Single( + db.Catalog.FindTable(table)!.Indexes.Where(i => i.IsUnique && !i.IsPrimaryKey)); + Assert.StartsWith("UQ_", unique.Name, StringComparison.Ordinal); + Assert.True(unique.Name.Length <= JetName.MaxLength, + $"the generated name is {unique.Name.Length} characters: '{unique.Name}'"); + + // And it is a working unique index, not just a name. + e.ExecuteNonQuery($"INSERT INTO [{table}] (Id, V) VALUES (1, 'x')"); + Assert.Throws(() => + e.ExecuteNonQuery($"INSERT INTO [{table}] (Id, V) VALUES (2, 'x')")); + } + + [Theory] + [InlineData(4)] + [InlineData(JetName.MaxLength)] + public void An_unnamed_check_constraint_gets_a_name_that_fits(int tableNameLength) + { + var (e, db) = Fresh(); + string table = new('C', tableNameLength); + + e.ExecuteNonQuery($"CREATE TABLE [{table}] (Id LONG PRIMARY KEY, V LONG, CHECK (V > 0))"); + + var check = Assert.Single(db.Catalog.FindTable(table)!.CheckConstraints); + Assert.StartsWith("CK_", check.Name, StringComparison.Ordinal); + Assert.True(check.Name.Length <= JetName.MaxLength, + $"the generated name is {check.Name.Length} characters: '{check.Name}'"); + + e.ExecuteNonQuery($"INSERT INTO [{table}] (Id, V) VALUES (1, 5)"); + Assert.ThrowsAny(() => e.ExecuteNonQuery($"INSERT INTO [{table}] (Id, V) VALUES (2, -1)")); + } +} diff --git a/test/LibRed.Shared/AceTestDatabase.cs b/test/LibRed.Shared/AceTestDatabase.cs index acc316d4a..fbc0f4070 100644 --- a/test/LibRed.Shared/AceTestDatabase.cs +++ b/test/LibRed.Shared/AceTestDatabase.cs @@ -66,7 +66,7 @@ public static OleDbConnection Open(string path, string? password = null, int att /// ACE faults when two threads are inside it at once (see AceCollection), and serialising the tests /// does not stop that on its own. DAO is apartment-threaded, so an engine created from a test's thread lives /// on a COM-created thread of its own, and the probes release none of what they create. Each Workspace, - /// Database, TableDef and Field is torn down when the finalizer gets to it — inside ACE, on that COM thread, + /// Database, TableDefinition and Field is torn down when the finalizer gets to it — inside ACE, on that COM thread, /// at whatever moment a GC happens to run, which is usually in the middle of a later test that is itself /// inside ACE. An OLE DB object left undisposed does the same the other way round, from the finalizer thread /// into a later DAO call. @@ -149,4 +149,4 @@ public static bool SupportsColumnType(string sourceDatabase, string typeName) public static string UnsupportedColumnTypeReason(string typeName) => $"The installed ACE cannot create a {typeName} column - it predates the type. " + $"DATETIME2 needs ACE 17 (Access 2019+/365); BIGINT needs ACE 16 (Access 2016)."; -} +} \ No newline at end of file diff --git a/test/LibRed.Shared/TestDatabases.cs b/test/LibRed.Shared/TestDatabases.cs index 7b2dc3c4b..14bd510ca 100644 --- a/test/LibRed.Shared/TestDatabases.cs +++ b/test/LibRed.Shared/TestDatabases.cs @@ -1,6 +1,9 @@ +using Xunit; + namespace LibRed.Core.Tests; -/// Paths to the real database files copied alongside the test assembly. +/// Paths to the real database files copied alongside the test assembly, and what reads and writes them +/// directly. internal static class TestDatabases { /// The path to a checked-in fixture by file name — every Data\*.accdb is copied @@ -9,6 +12,90 @@ internal static class TestDatabases public static string Data(string fileName) => Path.Combine(AppContext.BaseDirectory, "Data", fileName); + /// The format of the database at , read from its own page 0 as an open + /// reads it — never assumed from what the fixture is believed to be. + public static Formats.JetFormatBase FormatOf(string path) + { + // Shared for writing too: a test may ask while its own connection still holds the file open. + using var stream = new FileStream(path, FileMode.Open, FileAccess.Read, FileShare.ReadWrite); + return Formats.JetFormatBase.Detect(stream); + } + + // Raw page I/O on a closed file, for a test that stands in for another writer — planting released pages, moving or + // breaking page 0's map pointers. A writable channel cannot do it: its close merges the released map and checks + // those very pointers, undoing or refusing the state being planted. + + /// Page of the file at , straight off disk. + public static byte[] ReadPage(string path, int page) + { + int pageSize = FormatOf(path).PageSize; + using var stream = new FileStream(path, FileMode.Open, FileAccess.Read, FileShare.ReadWrite); + var bytes = new byte[pageSize]; + stream.Position = (long)page * pageSize; + stream.ReadExactly(bytes); + return bytes; + } + + /// Overwrites page of the file at . + public static void WritePage(string path, int page, byte[] bytes) + { + int pageSize = FormatOf(path).PageSize; + using var stream = new FileStream(path, FileMode.Open, FileAccess.ReadWrite); + stream.Position = (long)page * pageSize; + stream.Write(bytes); + } + + /// Adds as a new page at the end of the file at . + public static void AppendPage(string path, byte[] bytes) + { + using var stream = new FileStream(path, FileMode.Append, FileAccess.Write); + stream.Write(bytes); + } + + /// Points page 0's global map pointer at at + /// :, under the header mask. + public static void WriteMapPointer(string path, int pointerOffset, int row, int page) + { + Formats.JetFormatBase format = FormatOf(path); + byte[] page0 = ReadPage(path, 0); + var pointer = new byte[IO.PageBuffer.RecordPointerSize]; + IO.PageBuffer.WriteRecordPointer(pointer, 0, row, page); + Pages.DatabaseDefinitionPage.WriteMasked(page0, pointerOffset, pointer, format); + WritePage(path, 0, page0); + } + + /// Whether 's bit is set in the inline usage map at of a + /// holder page's bytes. + public static bool MapBit(byte[] holder, Formats.JetFormatBase format, int row, int page) => + Storage.BitmapBits.Get(InlineMapBits(holder, format, row, page, out int bit), bit); + + /// Sets or clears 's bit in the inline usage map at of a + /// holder page's bytes. + public static void SetMapBit(byte[] holder, Formats.JetFormatBase format, int row, int page, bool set) => + Storage.BitmapBits.Set(InlineMapBits(holder, format, row, page, out int bit), bit, set); + + private static Span InlineMapBits(byte[] holder, Formats.JetFormatBase format, int row, int page, out int bit) + { + var parsed = new Pages.DataPage(); + parsed.Read(new IO.PageBuffer(holder, 0), format); + Pages.DataPage.RowSlot slot = parsed.Rows[row]; + Span record = holder.AsSpan(slot.Offset, slot.Length); + Assert.Equal(Formats.UsageMapType.Inline, Storage.UsageMap.RecordType(record)); + bit = page - Storage.UsageMap.StartPage(record, format); + Span bits = Storage.UsageMap.InlineBits(record, format); + Assert.InRange(bit, 0, bits.Length * 8 - 1); + return bits; + } + + /// The global usage map page 0's pointer at names, found as the allocator + /// finds it, wherever it is: the pointer, the holder page's bytes, and the record's slot on it. + public static (int Row, int Page, byte[] Holder, Pages.DataPage.RowSlot Slot) GlobalMap(IO.PageChannel channel, int pointerOffset) + { + (int row, int page) = Pages.DatabaseDefinitionPage.ReadMapPointer(channel.ReadPage(0).Span, pointerOffset, channel.Format); + (IO.PageBuffer holder, _, Pages.DataPage.RowSlot slot) = Storage.UsageMap.ReadRecord(channel, row, page, "Global map pointer"); + return (row, page, holder.Span.ToArray(), slot); + } + /// An Access 2007 (ACE 12 / ACCDB) Northwind sample. public static string NorthwindAccdb { get; } = Path.Combine(AppContext.BaseDirectory, "Data", "Northwind.accdb"); @@ -55,4 +142,4 @@ public static string Data(string fileName) => /// The password for . public const string EncryptedPassword = "Test"; -} +} \ No newline at end of file diff --git a/tools/JetLockTrace/README.md b/tools/JetLockTrace/README.md index ebc5f8605..adbe85f34 100644 --- a/tools/JetLockTrace/README.md +++ b/tools/JetLockTrace/README.md @@ -78,8 +78,10 @@ plus enough beyond that to identify the holder). It also decodes I/O against the database file — page numbers, sub-page field writes, and the **commit-byte table** at page 0 `0xE00`–`0xFFF` (256 users × 2 bytes), which Access writes immediately before and after a -batch of page writes. A nonzero commit byte with no matching user lock is what makes Access declare a file -suspect and demand a repair. +batch of page writes. Each slot is one little-endian 16-bit value: the bytes `00 00` (mid-write) or `01 00` +(accessed a corrupted page) with no matching user lock are what make Access declare the file suspect and demand +a repair. The idle value is `00 01` (256), and above that the slot counts the user's committed writes — see +[page-00 §2.2](../../src/LibRed/docs/format/page-00-database.md). Region names are the Jet development team's own, from `docs/JetWhitePapers_UPDATE1/Jetlock.docx`.