diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index d145a3a..0f38f76 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -37,7 +37,7 @@ jobs: with: toolchain: 1.98.1 components: rustfmt, clippy - targets: x86_64-unknown-linux-musl, aarch64-unknown-linux-musl, x86_64-pc-windows-gnu, i686-pc-windows-gnu, wasm32-wasip1 + targets: x86_64-unknown-linux-musl, aarch64-unknown-linux-musl, x86_64-apple-darwin, aarch64-apple-darwin, x86_64-pc-windows-gnu, i686-pc-windows-gnu, wasm32-wasip1, wasm32-unknown-unknown - name: Cache cargo uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 @@ -69,8 +69,10 @@ jobs: sudo apt-get update -qq sudo apt-get install -y --no-install-recommends gcc-mingw-w64 musl-tools p7zip-full qemu-user-static - # zig, because mimalloc is C and apt has no aarch64-musl compiler, plus - # cargo-zigbuild for that cross-build and wasmtime for tests/wasm.sh. + # zig, because mimalloc is C and apt has no aarch64-musl compiler and the + # macOS build needs a Mach-O linker, plus cargo-zigbuild for those + # cross-builds, wasmtime for tests/wasm.sh and + # wasm-bindgen for the npm package. # Versions and digests live in the script; ci.sh has already installed the # gate tools it shares with this step. - name: Install build tools @@ -107,15 +109,18 @@ jobs: run: | cp target/x86_64-unknown-linux-musl/release/dat3 dat3 cp target/aarch64-unknown-linux-musl/release/dat3 dat3-arm64 + cp target/universal2-apple-darwin/release/dat3 dat3-macos cp target/x86_64-pc-windows-gnu/release/dat3.exe dat3.exe cp target/i686-pc-windows-gnu/release/dat3.exe dat3-win32.exe cp target/wasm32-wasip1/release/dat3.wasm dat3.wasm + cp target/dat3-wasm.tgz dat3-wasm.tgz - name: Prepare debug assets if: ${{ !startsWith(github.ref, 'refs/tags/v') }} run: | cp target/x86_64-unknown-linux-musl/debug/dat3 dat3-debug cp target/aarch64-unknown-linux-musl/debug/dat3 dat3-arm64-debug + cp target/universal2-apple-darwin/debug/dat3 dat3-macos-debug cp target/x86_64-pc-windows-gnu/debug/dat3.exe dat3-debug.exe cp target/i686-pc-windows-gnu/debug/dat3.exe dat3-win32-debug.exe cp target/wasm32-wasip1/debug/dat3.wasm dat3-debug.wasm @@ -130,9 +135,11 @@ jobs: path: | dat3 dat3-arm64 + dat3-macos dat3.exe dat3-win32.exe dat3.wasm + dat3-wasm.tgz if-no-files-found: error - name: Upload debug artifacts @@ -143,6 +150,7 @@ jobs: path: | dat3-debug dat3-arm64-debug + dat3-macos-debug dat3-debug.exe dat3-win32-debug.exe dat3-debug.wasm @@ -150,8 +158,8 @@ jobs: # The declared rust-version is a promise to users, and nothing checked it: the # build job pins a toolchain many releases newer, so code using a feature - # stabilised after 1.87 would compile there and fail for anyone honouring the - # README. The version is read from Cargo.toml rather than repeated here. + # stabilised after 1.87 would compile there and fail for anyone honouring + # docs/building.md. The version is read from Cargo.toml rather than repeated here. msrv: name: msrv runs-on: ubuntu-latest @@ -171,18 +179,52 @@ jobs: uses: dtolnay/rust-toolchain@6c977a6ca4077a0ceb28ffbe03f59d46e9ac8772 # master with: toolchain: ${{ steps.msrv.outputs.version }} + targets: wasm32-unknown-unknown # --locked so the committed lockfile is what gets checked, and --all-targets # so the test code is covered too. - name: Check against the declared MSRV run: cargo check --all-targets --locked + # Not a default workspace member: it builds only for its wasm target + - name: Check the wasm library against the declared MSRV + run: cargo check -p dat3-wasm --target wasm32-unknown-unknown --locked + + # The macOS binary is cross-built on Linux, where nothing can run it. One + # runner per architecture, so each slice of the universal binary is run. + macos: + name: macos (${{ matrix.runner }}) + needs: ci + strategy: + matrix: + runner: [macos-26, macos-26-intel] + runs-on: ${{ matrix.runner }} + timeout-minutes: 10 + permissions: + contents: read + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Download release artifacts + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + with: + name: dat3 + path: assets + + # Artifacts do not keep the executable bit + - name: Test the macOS binary + env: + MACOS_DAT3: assets/dat3-macos + run: chmod +x "$MACOS_DAT3" && ./tests/macos.sh + # Separate job so contents:write exists only on a tag run, and only for the # step that publishes. It builds nothing: the binaries come from the build # job's artifact. release: name: release - needs: ci + needs: [ci, macos] if: startsWith(github.ref, 'refs/tags/v') runs-on: ubuntu-latest timeout-minutes: 10 @@ -208,9 +250,11 @@ jobs: files: | assets/dat3 assets/dat3-arm64 + assets/dat3-macos assets/dat3.exe assets/dat3-win32.exe assets/dat3.wasm + assets/dat3-wasm.tgz assets/SHA256SUMS fail_on_unmatched_files: true env: diff --git a/CHANGELOG.md b/CHANGELOG.md index 9b83245..1f3c9c5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,8 +2,9 @@ ## Unreleased +- New: releases ship a macOS binary (`dat3-macos`), a universal binary that runs natively on Intel and Apple Silicon Macs. It is not notarized, so a copy downloaded in a browser has to be allowed in System Settings, or cleared with `xattr -d com.apple.quarantine dat3-macos`, before its first run. - Fixed: a crafted Fallout 2 or Arcanum archive could make dat3 reserve gigabytes of memory for a single file name, and a crafted entry in any format could expand on extraction far past the size it declares. Entry paths are now limited to 1024 bytes, and an entry must decompress to exactly its declared size. -- Errors for damaged Fallout 1, Fallout 2 and Arcanum archives now report missing data in bytes rather than bits, and name the Fallout 2 entry that failed. +- Errors for damaged Fallout 1, Fallout 2 and Arcanum archives now report missing data in bytes rather than bits, and name the Fallout 1 or Fallout 2 entry that failed. - Fixed: `l`, `x` and `e` reported a requested name as not found, and failed, when an earlier glob in the same command had already selected that file. - Fixed: adding a file whose name or directory is longer than 255 bytes to a Fallout 1 archive wrote an archive that could not be opened again. The add now fails and leaves the archive untouched. - Fixed: adding a file of 4 GiB or more wrote a corrupt archive. Such files are now refused. @@ -16,7 +17,11 @@ - Entry names Windows cannot create as named are now refused on every platform, both when extracting and when adding: a `:` inside a name (which writes an NTFS alternate data stream), device names such as `CON`, `NUL`, `COM1` or `LPT1` with or without an extension, and names ending in a dot or space. Extraction checks every name first, so a refused name leaves nothing half-extracted. - Fixed: when `e` met several files of the same name (ignoring case) in different directories, which copy ended up on disk varied from run to run. The last one in archive order now wins, and `e` warns how many were skipped. - Fixed: `a` printed each "Skipping symlink" warning twice. +- Fixed: `x` and `e` wrote through a symlink already present at a file's destination, overwriting the file the link pointed to. The link is now replaced by the extracted file. - Saving an archive now flushes it to disk before replacing the old file, and on Linux and macOS keeps the old file's permissions. Two dat3 runs saving the same archive at once no longer write into each other's temporary file, though the one that finishes last still replaces the other's changes. Archives with very long file names, which could not be saved, now save. +- New: the archive code is available to other Rust programs as the `dat3-core` library, used as a git dependency on this repository. Its API may still change between releases. +- New: releases ship `dat3-wasm.tgz`, an npm package of the same library for Node.js and Electron: open archives from bytes, list, read, add and remove entries, and write the result back out. It runs on every platform, with TypeScript types. +- Changed: `.bgforge.yml` sets the default format with a single flat key, `dat3.default_format: arcanum`. The nested form (`dat3:` with `default_format:` under it) is no longer read, so a config using it falls back to `dat2` until it is rewritten. ## v0.10.1 diff --git a/Cargo.lock b/Cargo.lock index f40dbc0..b47d866 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -109,6 +109,12 @@ dependencies = [ "wyz", ] +[[package]] +name = "bumpalo" +version = "3.20.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" + [[package]] name = "byteorder" version = "1.5.0" @@ -246,6 +252,29 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "dat3-core" +version = "0.10.1" +dependencies = [ + "anyhow", + "byteorder", + "clap", + "deku", + "flate2", + "glob", + "proptest", + "rayon", +] + +[[package]] +name = "dat3-wasm" +version = "0.10.1" +dependencies = [ + "dat3-core", + "js-sys", + "wasm-bindgen", +] + [[package]] name = "deku" version = "0.20.3" @@ -298,14 +327,9 @@ name = "fallout-dat3" version = "0.10.1" dependencies = [ "anyhow", - "byteorder", "clap", - "deku", - "flate2", - "glob", + "dat3-core", "mimalloc", - "proptest", - "rayon", "yaml-rust2", ] @@ -350,6 +374,30 @@ version = "2.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "e6d5a32815ae3f33302d95fdcb2ce17862f8c65363dcfd29360480ba1001fc9c" +[[package]] +name = "futures-core" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" + +[[package]] +name = "futures-task" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" + +[[package]] +name = "futures-util" +version = "0.3.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" +dependencies = [ + "futures-core", + "futures-task", + "pin-project-lite", + "slab", +] + [[package]] name = "getrandom" version = "0.3.4" @@ -425,6 +473,17 @@ version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a6cb138bb79a146c1bd460005623e142ef0181e3d0219cb493e02f7d08a35695" +[[package]] +name = "js-sys" +version = "0.3.104" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" +dependencies = [ + "cfg-if", + "futures-util", + "wasm-bindgen", +] + [[package]] name = "libc" version = "0.2.189" @@ -501,6 +560,12 @@ version = "1.70.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "384b8ab6d37215f3c5301a95a4accb5d64aa607f1fcb26a11b5303878451b4fe" +[[package]] +name = "pin-project-lite" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" + [[package]] name = "ppv-lite86" version = "0.2.21" @@ -707,6 +772,12 @@ version = "0.3.10" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" +[[package]] +name = "slab" +version = "0.4.12" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c790de23124f9ab44544d7ac05d60440adc586479ce501c1d6d7da3cd8c9cf5" + [[package]] name = "strsim" version = "0.11.1" @@ -820,6 +891,51 @@ dependencies = [ "wit-bindgen", ] +[[package]] +name = "wasm-bindgen" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" +dependencies = [ + "cfg-if", + "once_cell", + "rustversion", + "wasm-bindgen-macro", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-macro" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" +dependencies = [ + "quote", + "wasm-bindgen-macro-support", +] + +[[package]] +name = "wasm-bindgen-macro-support" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" +dependencies = [ + "bumpalo", + "proc-macro2", + "quote", + "syn 2.0.119", + "wasm-bindgen-shared", +] + +[[package]] +name = "wasm-bindgen-shared" +version = "0.2.127" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" +dependencies = [ + "unicode-ident", +] + [[package]] name = "windows-link" version = "0.2.1" diff --git a/Cargo.toml b/Cargo.toml index bc8aae3..5658bfb 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,57 +1,35 @@ -[package] -name = "fallout-dat3" +[workspace] +members = ["crates/*"] +# dat3-wasm builds only for wasm32-unknown-unknown (a cdylib cannot link on the +# static musl target), so plain cargo commands leave it out; its gates name it. +default-members = ["crates/dat3", "crates/dat3-core"] +resolver = "3" + +[workspace.package] version = "0.10.1" edition = "2024" # Floor set by u32::is_multiple_of (stabilized 1.87); 1.85 fails to compile, 1.87 verified clean. rust-version = "1.87" authors = ["BGforge"] -description = "Cross-platform CLI for Fallout and Troika DAT archives." license = "GPL-3.0-only" +# Consumed as a git dependency, never from crates.io. +publish = false -[[bin]] -name = "dat3" -path = "src/main.rs" - -[dependencies] +[workspace.dependencies] # Error handling - makes error management much easier anyhow = "1.0" -# Binary data handling -byteorder = "1.5" # Read/write integers in different byte orders -# Declarative record layouts for DAT1, DAT2 and Arcanum; ToEE reads with byteorder. Kept rather than ported to -# byteorder or binrw: its one hazard, sizing a name buffer from an untrusted `count`, is closed by an assert on each -# length field, and a port would at most lift the clap cap below (binrw is on syn 2 as well). -deku = "0.20" - # Command-line interface # <4.6.4: clap_derive 4.6.4 moved to syn 3 while deku's derive macros still use syn 2, and [bans] multiple-versions = # "deny" forbids carrying both. Nothing in 4.6.x is needed; if a needed fix ships past the cap, skip syn 2 in # deny.toml before porting off deku. clap = { version = ">=4.4, <4.6.4", features = ["derive"] } -# Compression and performance -# zlib-rs backend: measurably faster than the default miniz_oxide (level-9 archive -# creation -17%, extraction -11% on a 129 MB archive) and pure Rust, so the -# mingw/musl cross-builds need no C toolchain. -flate2 = { version = "1.1", features = ["zlib-rs"] } # zlib compression for DAT2 format -rayon = "1.8" # Parallel processing for faster extraction - -# Cross-platform path handling -glob = "0.3" # Glob pattern matching for cross-platform support -# Reads .bgforge.yml for the default-format setting. Default features off: -# they only add encoding_rs for non-UTF-8 input detection, which a config -# file doesn't need (and whose BSD-3-Clause part the license gate rejects). -yaml-rust2 = { version = "0.12", default-features = false } - -# Optional: Use mimalloc on Linux for better performance -[target.'cfg(target_os = "linux")'.dependencies] -mimalloc = "0.1" - -[lints.rust] +[workspace.lints.rust] unsafe_op_in_unsafe_fn = "deny" missing_debug_implementations = "warn" -[lints.clippy] +[workspace.lints.clippy] # `expect` stays a warning: it is allowed on an invariant that cannot fail, with the message saying why. unwrap_used = "deny" expect_used = "warn" @@ -71,6 +49,3 @@ strip = true [profile.release.package."*"] opt-level = 3 - -[dev-dependencies] -proptest = "1.11.0" diff --git a/README.md b/README.md index 239562f..c325927 100644 --- a/README.md +++ b/README.md @@ -6,6 +6,7 @@ Crossplatform, static Rust re-implementation of DAT2, with minor differences. Al - [Usage](#usage) - [Differences from DAT2](#differences-from-dat2) +- [Using dat3 as a library](#using-dat3-as-a-library) - [Verifying a release](#verifying-a-release) - [Building](#building) @@ -202,8 +203,7 @@ dat3 a master.dat /tmp/patch000/file.txt When `a` creates a new archive and no `--format` is given, an optional `.bgforge.yml` in the current directory picks the default: ```yaml -dat3: - default_format: arcanum +dat3.default_format: arcanum ``` Supported values: `dat1`, `dat2`, `arcanum`, `toee`. An unrecognized value prints a warning and `dat2` is used. An explicit `--format` always wins, and existing archives always keep their format. @@ -234,9 +234,14 @@ dat3 d master.dat @files_to_delete.txt - Names are matched, listed, extracted and added in lowercase unless `--case-sensitive` is given (see [Letter case in entry names](#letter-case-in-entry-names)). +## Using dat3 as a library + +Other Rust programs can use the archive code as the `dat3-core` crate, and Node.js and Electron applications as the +`dat3-wasm` npm package. See [docs/api.md](docs/api.md). + ## Verifying a release -Every release ships a `SHA256SUMS` file covering its binaries. Download it +Every release ships a `SHA256SUMS` file covering its assets. Download it alongside the assets and check them: ```bash @@ -248,29 +253,4 @@ as missing. ## Building -### Requirements - -- Rust 1.87 or newer -- Target-specific toolchains (install as needed) -- `./install-tools.sh` for the pinned tooling, including [Zig](https://ziglang.org/), which the aarch64 - target needs: mimalloc is C, and no aarch64-musl C compiler is packaged for common distros -- Node 24 or newer, to run the integration suite (`./test.sh`): its helpers under `tests/` are TypeScript, run - by Node's own type stripping. `npm ci && npm run typecheck` typechecks them. Neither is needed to build dat3 - -### Build - -```bash -./build.sh -``` - -Builds are static. - -Binaries will be at: - -```bash -target/x86_64-unknown-linux-musl/release/dat3 -target/aarch64-unknown-linux-musl/release/dat3 -target/x86_64-pc-windows-gnu/release/dat3.exe -target/i686-pc-windows-gnu/release/dat3.exe -target/wasm32-wasip1/release/dat3.wasm -``` +See [docs/building.md](docs/building.md). diff --git a/build.sh b/build.sh index 10f1735..594e7b1 100755 --- a/build.sh +++ b/build.sh @@ -13,9 +13,11 @@ CARGO_TARGETS=( ) # mimalloc is C, and no aarch64-musl C compiler ships in apt; zig provides one. -# Only this target needs it, so the others stay on plain cargo. +# macOS needs a Mach-O linker, which zig also provides, and cargo-zigbuild's +# universal2 target merges the x86_64 and arm64 builds into one binary. ZIG_TARGETS=( aarch64-unknown-linux-musl + universal2-apple-darwin ) ALL_TARGETS=("${CARGO_TARGETS[@]}" "${ZIG_TARGETS[@]}") @@ -23,8 +25,13 @@ ALL_TARGETS=("${CARGO_TARGETS[@]}" "${ZIG_TARGETS[@]}") # Install targets if not already installed. Tolerated failure: a distro rustc has # no rustup, and its targets come from packages instead. A missing target still # fails loudly at the cargo build below. -for target in "${ALL_TARGETS[@]}"; do - rustup target add "$target" 2>/dev/null || true +# wasm32-unknown-unknown is the npm package's target, built by its own script below. +for target in "${ALL_TARGETS[@]}" wasm32-unknown-unknown; do + case "$target" in + # Not a rustup target: the two it merges are + universal2-apple-darwin) rustup target add x86_64-apple-darwin aarch64-apple-darwin 2>/dev/null || true ;; + *) rustup target add "$target" 2>/dev/null || true ;; + esac done # Build all targets in parallel - both debug and release. @@ -70,6 +77,9 @@ binary_name() { esac } +# The npm package for Node and Electron +crates/dat3-wasm/package.sh + echo "" echo "Cross-compile completed. Static binaries:" for profile in debug release; do @@ -79,3 +89,5 @@ for profile in debug release; do done echo "" done +echo "npm package:" +ls -lh target/dat3-wasm.tgz diff --git a/ci.sh b/ci.sh index ad4f27f..dbd8bca 100755 --- a/ci.sh +++ b/ci.sh @@ -25,10 +25,15 @@ cargo fmt --all -- --check # Clippy lints, test targets included - without --all-targets the #[cfg(test)] # modules are never compiled under clippy cargo clippy --all-targets -- -D warnings +# dat3-wasm is not a default member: it builds only for its wasm target +cargo clippy -p dat3-wasm --target wasm32-unknown-unknown -- -D warnings # Tests cargo test --verbose +# Library API docs: a broken doc link or other rustdoc warning fails the build +RUSTDOCFLAGS="-D warnings" cargo doc --no-deps --package dat3-core + # License, advisory (RustSec) and duplicate-dependency checks cargo deny check -D parse-error licenses cargo deny check advisories diff --git a/crates/dat3-core/Cargo.toml b/crates/dat3-core/Cargo.toml new file mode 100644 index 0000000..42f9545 --- /dev/null +++ b/crates/dat3-core/Cargo.toml @@ -0,0 +1,43 @@ +[package] +name = "dat3-core" +description = "Read and write Fallout and Troika DAT archives." +version.workspace = true +edition.workspace = true +rust-version.workspace = true +authors.workspace = true +# LGPL rather than the workspace's GPL, so programs under other licenses can use the library. +license = "LGPL-3.0-only" +publish.workspace = true + +[features] +# Derives clap::ValueEnum on ArchiveFormat, so a CLI can take it as an argument. +clap = ["dep:clap"] +# Exposes the test_support module to other workspace crates' tests. +test-support = [] + +[dependencies] +anyhow.workspace = true +clap = { workspace = true, optional = true } + +# Binary data handling +byteorder = "1.5" # Read/write integers in different byte orders +# Declarative record layouts for DAT1, DAT2 and Arcanum; ToEE reads with byteorder. Kept rather than ported to +# byteorder or binrw: its one hazard, sizing a name buffer from an untrusted `count`, is closed by an assert on each +# length field, and a port would at most lift the clap cap (binrw is on syn 2 as well). +deku = "0.20" + +# Compression and performance +# zlib-rs backend: measurably faster than the default miniz_oxide (level-9 archive +# creation -17%, extraction -11% on a 129 MB archive) and pure Rust, so the +# mingw/musl cross-builds need no C toolchain. +flate2 = { version = "1.1", features = ["zlib-rs"] } # zlib compression for DAT2 format +rayon = "1.8" # Parallel processing for faster extraction + +# Cross-platform path handling +glob = "0.3" # Glob pattern matching for cross-platform support + +[dev-dependencies] +proptest = "1.11.0" + +[lints] +workspace = true diff --git a/proptest-regressions/arcanum.txt b/crates/dat3-core/proptest-regressions/arcanum.txt similarity index 100% rename from proptest-regressions/arcanum.txt rename to crates/dat3-core/proptest-regressions/arcanum.txt diff --git a/proptest-regressions/dat2.txt b/crates/dat3-core/proptest-regressions/dat2.txt similarity index 100% rename from proptest-regressions/dat2.txt rename to crates/dat3-core/proptest-regressions/dat2.txt diff --git a/src/arcanum.rs b/crates/dat3-core/src/arcanum.rs similarity index 87% rename from src/arcanum.rs rename to crates/dat3-core/src/arcanum.rs index 0a93be9..1c475fd 100644 --- a/src/arcanum.rs +++ b/crates/dat3-core/src/arcanum.rs @@ -17,6 +17,7 @@ little-endian, flat entry table at the end of the file, zlib compression. use anyhow::{Context, Result, bail}; use byteorder::{ByteOrder, LittleEndian, ReadBytesExt, WriteBytesExt}; use deku::prelude::*; +use std::borrow::Cow; use std::io::{Cursor, Write}; use std::path::Path; @@ -93,6 +94,12 @@ pub fn is_arcanum_format(data: &[u8]) -> bool { data.len() >= FOOTER_SIZE + 4 && data[data.len() - 12..data.len() - 8] == MAGIC } +impl Default for ArcanumArchive { + fn default() -> Self { + Self::new() + } +} + impl ArcanumArchive { /// Create a new empty Arcanum archive pub fn new() -> Self { @@ -228,22 +235,46 @@ impl ArcanumArchive { ) } - /// Delete a file from the archive by name - pub fn delete_file(&mut self, file_name: &str) -> Result<()> { - common::delete_file_from_list(&mut self.files, file_name) + /// An entry's contents, decompressed + pub fn contents<'a>(&'a self, file: &'a FileEntry) -> Result> { + common::entry_contents(&self.data, file, common::decompress_zlib) } - /// Save the archive to an Arcanum DAT file. - /// - /// Layout: file data, table marker, entry table, 28-byte footer. + /// Add a prepared entry, replacing the entries of its name as `case` compares + pub fn insert_entry(&mut self, entry: FileEntry, case: CaseMode) { + common::merge_entries(&mut self.files, vec![entry], case); + } + + /// Remove the entry named exactly `file_name`, reporting whether there was one + pub fn remove_entry(&mut self, file_name: &str) -> bool { + common::remove_from_list(&mut self.files, file_name) + } + + /// Save the archive to an Arcanum DAT file pub fn save(&self, path: &Path) -> Result<()> { + self.prepare_save()?; + utils::write_atomically(path, |out| self.write_prepared(out)).context(WRITE_CONTEXT) + } + + /// Write the archive as an Arcanum DAT file to `out` + pub fn write_to(&self, out: &mut dyn Write) -> Result<()> { + self.prepare_save()?; + self.write_prepared(out).context(WRITE_CONTEXT) + } + + /// Checks that fail before any output exists, so a refused save leaves no temp file + fn prepare_save(&self) -> Result<()> { // Offsets are u32 and the table marker adds 4 bytes past the data. // Entries keep data.len() == packed_size, bounding the accumulation. let total_payload: u64 = self.files.iter().map(|f| f.packed_size as u64).sum(); if total_payload > u32::MAX as u64 - 4 { bail!("Arcanum archive would exceed the format's 4 GiB offset limit"); } + Ok(()) + } + /// Layout: file data, table marker, entry table, 28-byte footer. + fn write_prepared(&self, out: &mut dyn Write) -> Result<()> { // The table stores explicit directory entries interleaved with files // in one flat case-insensitive path order, matching the original // tool's layout. Directories are synthesized from file paths, so @@ -275,85 +306,82 @@ impl ArcanumArchive { // dominant cost on a large table. table.sort_by_cached_key(|(name, _)| name.to_lowercase()); - utils::write_atomically(path, |out| { - // Step 1: file data, written in table order like the original tool - let mut file_offsets = vec![0u32; self.files.len()]; - let mut current_offset = 0u32; - for (_, index) in &table { - if let Some(i) = index { - let file = &self.files[*i]; + // Step 1: file data, written in table order like the original tool + let mut file_offsets = vec![0u32; self.files.len()]; + let mut current_offset = 0u32; + for (_, index) in &table { + if let Some(i) = index { + let file = &self.files[*i]; - let data = self.read_file_data(file)?; + let data = self.read_file_data(file)?; - file_offsets[*i] = current_offset; - out.write_all(data)?; - current_offset += data.len() as u32; - } + file_offsets[*i] = current_offset; + out.write_all(data)?; + current_offset += data.len() as u32; } + } - // Step 2: table marker - the entry table's absolute offset, - // which sits just past this u32 - out.write_u32::(current_offset + 4)?; - - // Step 3: entry table, tracking its size for the footer - out.write_u32::(table.len() as u32)?; - let mut table_size: u64 = 4; - let mut names_len: u64 = 0; - - for (name, index) in &table { - let mut name_bytes = name.as_bytes().to_vec(); - name_bytes.push(0); - // unknown is 0: shipped archives carry junk there (see the - // field's doc) and no reader is known to use it - let entry = match index { - Some(i) => { - let f = &self.files[*i]; - ArcanumFileEntry { - name_len: name.len() as u32 + 1, - name_bytes, - unknown: 0, - flags: if f.compressed { FLAG_ZLIB } else { FLAG_RAW }, - real_size: f.size, - packed_size: f.packed_size, - offset: file_offsets[*i], - } - } - None => ArcanumFileEntry { + // Step 2: table marker - the entry table's absolute offset, + // which sits just past this u32 + out.write_u32::(current_offset + 4)?; + + // Step 3: entry table, tracking its size for the footer + out.write_u32::(table.len() as u32)?; + let mut table_size: u64 = 4; + let mut names_len: u64 = 0; + + for (name, index) in &table { + let mut name_bytes = name.as_bytes().to_vec(); + name_bytes.push(0); + // unknown is 0: shipped archives carry junk there (see the + // field's doc) and no reader is known to use it + let entry = match index { + Some(i) => { + let f = &self.files[*i]; + ArcanumFileEntry { name_len: name.len() as u32 + 1, name_bytes, unknown: 0, - flags: FLAG_DIR, - real_size: 0, - packed_size: 0, - offset: 0, - }, - }; - - let entry_bytes = entry.to_bytes()?; - out.write_all(&entry_bytes)?; - table_size += entry_bytes.len() as u64; - names_len += name.len() as u64 + 1; - } - - // Step 4: footer - let footer = ArcanumFooter { - guid: self.guid, - magic: MAGIC, - filename_total_bytes: u32::try_from(names_len) - .context("Arcanum archive filenames exceed the format's u32 limit")?, - table_from_end: u32::try_from(table_size + FOOTER_SIZE as u64) - .context("Arcanum entry table would exceed the format's u32 limit")?, + flags: if f.compressed { FLAG_ZLIB } else { FLAG_RAW }, + real_size: f.size, + packed_size: f.packed_size, + offset: file_offsets[*i], + } + } + None => ArcanumFileEntry { + name_len: name.len() as u32 + 1, + name_bytes, + unknown: 0, + flags: FLAG_DIR, + real_size: 0, + packed_size: 0, + offset: 0, + }, }; - out.write_all(&footer.to_bytes()?)?; - Ok(()) - }) - .context("Failed to write Arcanum DAT file")?; + let entry_bytes = entry.to_bytes()?; + out.write_all(&entry_bytes)?; + table_size += entry_bytes.len() as u64; + names_len += name.len() as u64 + 1; + } + + // Step 4: footer + let footer = ArcanumFooter { + guid: self.guid, + magic: MAGIC, + filename_total_bytes: u32::try_from(names_len) + .context("Arcanum archive filenames exceed the format's u32 limit")?, + table_from_end: u32::try_from(table_size + FOOTER_SIZE as u64) + .context("Arcanum entry table would exceed the format's u32 limit")?, + }; + out.write_all(&footer.to_bytes()?)?; Ok(()) } } +const WRITE_CONTEXT: &str = "Failed to write Arcanum DAT file"; + #[cfg(test)] mod tests { use super::*; diff --git a/crates/dat3-core/src/archive.rs b/crates/dat3-core/src/archive.rs new file mode 100644 index 0000000..53fb879 --- /dev/null +++ b/crates/dat3-core/src/archive.rs @@ -0,0 +1,901 @@ +/*! +# Archive formats + +`ArchiveFormat` names the supported formats, and `DatArchive` wraps their +implementations behind one interface so callers don't need to know which +format they're working with. Kept out of `common`, which the format modules +build on, so module dependencies run one way. +*/ + +use anyhow::{Context, Result, bail}; +use std::fs; +use std::io::Write; +use std::path::Path; + +use crate::arcanum::{self, ArcanumArchive}; +use crate::common::{ + self, CaseMode, CompressionLevel, ExtractionMode, ListFormat, Selection, utils, +}; +use crate::dat1::Dat1Archive; +use crate::dat2::Dat2Archive; +use crate::toee::{self, ToeeArchive}; + +// DAT1 format detection: big-endian header, no signature to key on +const DAT1_MAX_DIRECTORIES: u32 = 1000; + +/// A supported archive format, as selected by `a --format` or `.bgforge.yml` +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[cfg_attr(feature = "clap", derive(clap::ValueEnum))] +pub enum ArchiveFormat { + /// Fallout 1 (big-endian, LZSS; created uncompressed) + Dat1, + /// Fallout 2 (little-endian, zlib) - the default for new archives + Dat2, + /// Arcanum (little-endian, zlib) + Arcanum, + /// The Temple of Elemental Evil (hierarchical Troika DAT, zlib) + Toee, +} + +impl ArchiveFormat { + /// Every format + pub const ALL: [Self; 4] = [Self::Dat1, Self::Dat2, Self::Arcanum, Self::Toee]; + + /// The format whose [`arg_name`](Self::arg_name) is `name` + pub fn from_arg_name(name: &str) -> Option { + Self::ALL + .into_iter() + .find(|format| format.arg_name() == name) + } + + /// The value as typed on the command line, for error messages + pub fn arg_name(self) -> &'static str { + match self { + Self::Dat1 => "dat1", + Self::Dat2 => "dat2", + Self::Arcanum => "arcanum", + Self::Toee => "toee", + } + } + + /// Human-readable format name for messages + pub fn display_name(self) -> &'static str { + match self { + Self::Dat1 => "DAT1", + Self::Dat2 => "DAT2", + Self::Arcanum => "Arcanum", + Self::Toee => "ToEE", + } + } +} + +/// Unified interface for DAT1, DAT2, Arcanum, and ToEE archives. +/// +/// Uses an enum instead of trait objects because the set of known formats +/// is small and fixed - this gives us static dispatch, exhaustive matching, +/// and no heap allocation for the wrapper. +/// +/// **Memory**: The entire archive is loaded into memory on open, whatever the +/// format. Fallout archives typically stay under ~200MB; retail Arcanum +/// archives run considerably larger and are held in RAM the same way. +/// +/// ```no_run +/// use dat3_core::{ArchiveFormat, DatArchive}; +/// +/// let archive = DatArchive::open("master.dat")?; // auto-detects format +/// let dat1 = DatArchive::new(ArchiveFormat::Dat1); // create new DAT1 +/// # anyhow::Ok(()) +/// ``` +#[derive(Debug)] +pub enum DatArchive { + /// Fallout 1 format (big-endian, hierarchical dirs, LZSS compression) + Dat1(Dat1Archive), + /// Fallout 2 format (little-endian, flat file list, zlib compression) + Dat2(Dat2Archive), + /// Arcanum format (little-endian, flat entry table, zlib compression) + Arcanum(ArcanumArchive), + /// ToEE format (little-endian, hierarchical entry table, zlib compression) + Toee(ToeeArchive), +} + +impl DatArchive { + /// Open an existing DAT archive, auto-detecting the format. + /// + /// Reads the whole file into memory. DAT2 has no signature, so a file that is + /// none of the other formats is parsed as DAT2 and fails there if it is not one. + pub fn open>(path: P) -> Result { + let data = fs::read(&path) + .with_context(|| format!("Failed to read DAT file: {}", path.as_ref().display()))?; + Self::from_bytes(data) + } + + /// Parse an archive held in memory, auto-detecting the format as [`open`](Self::open) does + pub fn from_bytes(data: Vec) -> Result { + // Troika formats share a real magic and go first. ToEE's hierarchical + // entry-table size distinguishes it from Arcanum's flat table. DAT1 is + // a header heuristic and DAT2 (no signature at all) is the fallback. + if toee::is_toee_format(&data) { + Ok(Self::Toee(ToeeArchive::from_bytes(data)?)) + } else if arcanum::is_arcanum_format(&data) { + Ok(Self::Arcanum(ArcanumArchive::from_bytes(data)?)) + } else if Self::is_dat1_format(&data) { + Ok(Self::Dat1(Dat1Archive::from_bytes(data)?)) + } else { + Ok(Self::Dat2(Dat2Archive::from_bytes(data)?)) + } + } + + /// Create a new empty archive of the given format. Nothing is written until [`save`](Self::save). + pub fn new(format: ArchiveFormat) -> Self { + match format { + ArchiveFormat::Dat1 => Self::Dat1(Dat1Archive::new()), + ArchiveFormat::Dat2 => Self::Dat2(Dat2Archive::new()), + ArchiveFormat::Arcanum => Self::Arcanum(ArcanumArchive::new()), + ArchiveFormat::Toee => Self::Toee(ToeeArchive::new()), + } + } + + /// The format this archive is stored in + pub fn format(&self) -> ArchiveFormat { + match self { + Self::Dat1(_) => ArchiveFormat::Dat1, + Self::Dat2(_) => ArchiveFormat::Dat2, + Self::Arcanum(_) => ArchiveFormat::Arcanum, + Self::Toee(_) => ArchiveFormat::Toee, + } + } + + /// Detect DAT1 format by examining the big-endian header. + /// + /// The header opens with a directory count and the engine's allocation hint + /// for that directory list, which is never below the count. Both are checked: + /// DAT2 carries no signature at all and is the fallback, so this heuristic is + /// what keeps a DAT2 archive from being parsed as DAT1. + /// + /// The second field is NOT a format identifier, despite reading like one in + /// the shipped archives - `critter.dat` carries 10 and `master.dat` 94, which + /// are simply their own hints. Matching those two values exactly rejects every + /// other real DAT1 archive, including the Fallout 1 demo's (hint 46). + fn is_dat1_format(data: &[u8]) -> bool { + let Some(header) = data.get(..16) else { + return false; + }; + let field = + |i: usize| u32::from_be_bytes([header[i], header[i + 1], header[i + 2], header[i + 3]]); + let dir_count = field(0); + let allocation_hint = field(4); + + dir_count > 0 + && dir_count < DAT1_MAX_DIRECTORIES + && allocation_hint >= dir_count + && allocation_hint < DAT1_MAX_DIRECTORIES + } + + /// Print the entries `selection` picks to stdout, as columns or a JSON array. + /// + /// Patterns that match nothing are printed to stderr after the listing, and + /// fail the call when `selection.on_missing` is [`MissingFiles::Fail`](crate::common::MissingFiles::Fail). + pub fn list(&self, selection: &Selection, format: ListFormat) -> Result<()> { + match self { + Self::Dat1(a) => a.list(selection, format), + Self::Dat2(a) => a.list(selection, format), + Self::Arcanum(a) => a.list(selection, format), + Self::Toee(a) => a.list(selection, format), + } + } + + /// Extract the entries `selection` picks into `output_dir`, in parallel. + /// + /// Every selected name is checked before anything is written: a name that is + /// unsafe to create, or a pattern that matches nothing under + /// [`MissingFiles::Fail`](crate::common::MissingFiles::Fail), fails with no files written. Names are + /// written as `selection.case` shows them. Progress goes to stdout; in + /// [`ExtractionMode::Flat`] entries whose file name a later entry reuses are + /// skipped with a warning on stderr. + pub fn extract>( + &self, + output_dir: P, + mode: ExtractionMode, + selection: &Selection, + ) -> Result<()> { + let output_dir = output_dir.as_ref(); + match self { + Self::Dat1(a) => a.extract(output_dir, mode, selection), + Self::Dat2(a) => a.extract(output_dir, mode, selection), + Self::Arcanum(a) => a.extract(output_dir, mode, selection), + Self::Toee(a) => a.extract(output_dir, mode, selection), + } + } + + /// Add a file, or a directory recursively (symlinks are skipped), in memory. + /// + /// The entry name comes from `file_path`: + /// - with `source_root`, its path relative to that root, which must prefix it + /// (pass both canonicalized); + /// - otherwise the path as given without a leading `./`, or with `target_dir` + /// just the file's own name, or a directory's name and its contents. + /// + /// `target_dir` is prefixed to the name in either case, and a name the archive + /// cannot hold fails (see [`common::utils::validate_add_archive_path`]). An entry whose + /// name equals an existing one, compared as `case` says, replaces it; under + /// [`CaseMode::Insensitive`] new names are stored in lowercase. DAT1 stores + /// entries uncompressed whatever `compression` says. Prints each added name to stdout. + pub fn add_file>( + &mut self, + file_path: P, + compression: CompressionLevel, + target_dir: Option<&str>, + source_root: Option<&Path>, + case: CaseMode, + ) -> Result<()> { + let path = file_path.as_ref(); + match self { + Self::Dat1(a) => a.add_file(path, compression, target_dir, source_root, case), + Self::Dat2(a) => a.add_file(path, compression, target_dir, source_root, case), + Self::Arcanum(a) => a.add_file(path, compression, target_dir, source_root, case), + Self::Toee(a) => a.add_file(path, compression, target_dir, source_root, case), + } + } + + /// Names of every file entry, as stored: backslash-separated, in their stored case + pub fn entry_names(&self) -> Vec { + self.file_entries() + .into_iter() + .map(|entry| entry.name.clone()) + .collect() + } + + /// Every file entry with its sizes, names as stored (see [`entry_names`](Self::entry_names)) + pub fn entries(&self) -> Vec { + self.file_entries() + .into_iter() + .map(|file| Entry { + name: file.name.clone(), + size: file.size, + packed_size: file.packed_size, + compressed: file.compressed, + }) + .collect() + } + + fn file_entries(&self) -> Vec<&common::FileEntry> { + match self { + Self::Dat1(a) => a.entries(), + Self::Dat2(a) => a.entries(), + Self::Arcanum(a) => a.entries(), + Self::Toee(a) => a.entries(), + } + } + + /// The decompressed contents of the entry `name` refers to. + /// + /// `name` may use `/` or `\`. The entry with exactly that name wins; failing + /// that, under [`CaseMode::Insensitive`], the one entry whose name equals it + /// ignoring case. Several such entries, or none, is an error. + pub fn read(&self, name: &str, case: CaseMode) -> Result> { + let files = self.file_entries(); + let index = common::find_stored_index(files.iter().map(|f| f.name.as_str()), name, case)? + .with_context(|| { + format!( + "File not found: {}", + utils::normalize_path_for_display(name) + ) + })?; + let file = files[index]; + let contents = match self { + Self::Dat1(a) => a.contents(file)?, + Self::Dat2(a) => a.contents(file)?, + Self::Arcanum(a) => a.contents(file)?, + Self::Toee(a) => a.contents(file)?, + }; + Ok(contents.into_owned()) + } + + /// Add `data` as the entry `name` (`/` or `\` separated), in memory. + /// + /// The name is stored as given, and replaces any entry equal to it as `case` + /// compares. A name the archive cannot hold fails (see + /// [`utils::stored_name_for_insert`]), as does data over 4 GiB. The zlib + /// formats compress at `compression` when that saves space; DAT1 stores it + /// uncompressed. Prints nothing. + pub fn insert( + &mut self, + name: &str, + data: Vec, + compression: CompressionLevel, + case: CaseMode, + ) -> Result<()> { + let stored = utils::stored_name_for_insert(name)?; + if u32::try_from(data.len()).is_err() { + bail!( + "{} is larger than the 4 GiB a DAT archive entry can hold", + utils::normalize_path_for_display(&stored) + ); + } + match self { + Self::Dat1(a) => { + let mut entry = common::FileEntry::with_data(stored, data, false); + entry.size = entry.packed_size; + a.insert_entry(entry, case); + } + Self::Dat2(a) => a.insert_entry(common::zlib_entry(stored, data, compression)?, case), + Self::Arcanum(a) => { + a.insert_entry(common::zlib_entry(stored, data, compression)?, case) + } + Self::Toee(a) => a.insert_entry(common::zlib_entry(stored, data, compression)?, case), + } + Ok(()) + } + + /// Remove the entry `name` refers to, found as [`read`](Self::read) finds it, + /// in memory. Returns whether an entry was removed. Prints nothing. + pub fn remove(&mut self, name: &str, case: CaseMode) -> Result { + let names = self.entry_names(); + let Some(index) = common::find_stored_index(names.iter().map(String::as_str), name, case)? + else { + return Ok(false); + }; + Ok(self.remove_entry(&names[index])) + } + + fn remove_entry(&mut self, stored_name: &str) -> bool { + match self { + Self::Dat1(a) => a.remove_entry(stored_name), + Self::Dat2(a) => a.remove_entry(stored_name), + Self::Arcanum(a) => a.remove_entry(stored_name), + Self::Toee(a) => a.remove_entry(stored_name), + } + } + + /// Delete every entry `patterns` select, in memory. + /// + /// A glob deletes every entry it matches; a plain name deletes only the entry + /// with that whole name. If any pattern matches nothing, nothing is deleted + /// (see [`resolve_delete_targets`](crate::common::resolve_delete_targets)). + /// Prints each deleted name to stdout. + pub fn delete(&mut self, patterns: &[String], case: CaseMode) -> Result<()> { + let names = self.entry_names(); + let names: Vec<&str> = names.iter().map(String::as_str).collect(); + for target in common::resolve_delete_targets(&names, patterns, case)? { + self.delete_file(&target)?; + } + Ok(()) + } + + /// Delete the entry named exactly `file_name` (`/` or `\` separated), in memory. + /// Prints the deleted name to stdout. + pub fn delete_file(&mut self, file_name: &str) -> Result<()> { + if !self.remove_entry(file_name) { + bail!( + "File not found: {}", + utils::normalize_path_for_display(file_name) + ); + } + let normalized = utils::normalize_user_path(file_name); + common::print_stdout(format_args!( + "Deleting: {}", + utils::normalize_path_for_display(&normalized) + )); + Ok(()) + } + + /// Write the archive to `path`, replacing any file there only once the new one + /// is complete and flushed to disk. + pub fn save>(&self, path: P) -> Result<()> { + match self { + Self::Dat1(a) => a.save(path.as_ref()), + Self::Dat2(a) => a.save(path.as_ref()), + Self::Arcanum(a) => a.save(path.as_ref()), + Self::Toee(a) => a.save(path.as_ref()), + } + } + + /// Write the archive to `out`, byte for byte what [`save`](Self::save) writes + pub fn write_to(&self, out: &mut dyn Write) -> Result<()> { + match self { + Self::Dat1(a) => a.write_to(out), + Self::Dat2(a) => a.write_to(out), + Self::Arcanum(a) => a.write_to(out), + Self::Toee(a) => a.write_to(out), + } + } + + /// The archive as it would be saved + pub fn to_bytes(&self) -> Result> { + let mut out = Vec::new(); + self.write_to(&mut out)?; + Ok(out) + } +} + +/// One file entry's name and sizes, as [`DatArchive::entries`] reports it +#[derive(Debug, Clone, PartialEq, Eq)] +#[non_exhaustive] +pub struct Entry { + /// Name as stored: backslash-separated, in its stored case + pub name: String, + /// Size of the contents in bytes + pub size: u32, + /// Size as stored in the archive, compressed or not + pub packed_size: u32, + /// Whether the stored data is compressed + pub compressed: bool, +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::test_support::ScratchPath; + + const ALL_FORMATS: [ArchiveFormat; 4] = ArchiveFormat::ALL; + + #[test] + fn from_arg_name_names_every_format_and_nothing_else() { + for format in ALL_FORMATS { + assert_eq!( + ArchiveFormat::from_arg_name(format.arg_name()), + Some(format) + ); + } + assert_eq!(ArchiveFormat::from_arg_name("zip"), None); + assert_eq!(ArchiveFormat::from_arg_name("DAT2"), None); + } + + /// A new archive holding one small file per name (`/`-separated), stored in + /// exactly the case given + fn archive_with(format: ArchiveFormat, names: &[&str]) -> DatArchive { + let mut archive = DatArchive::new(format); + add_names(&mut archive, names, CaseMode::Sensitive); + archive + } + + /// Add one small file per name, as `a` would under `case` + fn add_names(archive: &mut DatArchive, names: &[&str], case: CaseMode) { + let src = ScratchPath::dir("archive_names_src"); + for name in names { + let path = src.join(name); + std::fs::create_dir_all(path.parent().unwrap()).unwrap(); + std::fs::write(&path, name.as_bytes()).unwrap(); + archive + .add_file( + &path, + CompressionLevel::new(0).unwrap(), + None, + Some(&src), + case, + ) + .unwrap(); + } + } + + /// Save and reopen, so assertions see what the format actually stored + fn reopened(archive: &DatArchive) -> DatArchive { + let path = ScratchPath::new("archive_case_roundtrip"); + archive.save(&path).unwrap(); + DatArchive::open(&path).unwrap() + } + + fn all_entries(case: CaseMode) -> Selection<'static> { + Selection { + patterns: &[], + on_missing: common::MissingFiles::Fail, + case, + } + } + + #[test] + fn by_default_new_entries_are_stored_in_lowercase() { + for format in ALL_FORMATS { + let mut archive = DatArchive::new(format); + add_names(&mut archive, &["Art/Hero.FRM"], CaseMode::Insensitive); + assert_eq!( + sorted_names(&reopened(&archive)), + ["art\\hero.frm"], + "{format:?}" + ); + } + } + + #[test] + fn with_case_sensitive_new_entries_keep_their_case() { + for format in ALL_FORMATS { + let archive = archive_with(format, &["Art/Hero.FRM"]); + assert_eq!( + sorted_names(&reopened(&archive)), + ["Art\\Hero.FRM"], + "{format:?}" + ); + } + } + + #[test] + fn by_default_extraction_writes_lowercase_paths() { + for format in ALL_FORMATS { + let archive = reopened(&archive_with(format, &["ART/HERO.FRM"])); + let out = ScratchPath::dir("archive_case_extract"); + archive + .extract( + &out, + ExtractionMode::PreserveStructure, + &all_entries(CaseMode::Insensitive), + ) + .unwrap(); + assert!(out.join("art").join("hero.frm").is_file(), "{format:?}"); + assert!(!out.join("ART").exists(), "{format:?}"); + } + } + + #[test] + fn with_case_sensitive_extraction_keeps_stored_paths() { + for format in ALL_FORMATS { + let archive = reopened(&archive_with(format, &["ART/HERO.FRM"])); + let out = ScratchPath::dir("archive_case_extract_exact"); + archive + .extract( + &out, + ExtractionMode::PreserveStructure, + &all_entries(CaseMode::Sensitive), + ) + .unwrap(); + assert!(out.join("ART").join("HERO.FRM").is_file(), "{format:?}"); + } + } + + /// Re-adding a file under another case replaces the entry rather than + /// keeping two names that differ only in case. + #[test] + fn by_default_adding_replaces_an_entry_differing_only_in_case() { + for format in ALL_FORMATS { + let mut archive = reopened(&archive_with(format, &["ART/HERO.FRM", "ART/OTHER.FRM"])); + add_names(&mut archive, &["art/hero.frm"], CaseMode::Insensitive); + // DAT1 and ToEE keep one stored spelling per directory, so the new + // lowercase file sits under the existing ART; the flat formats store + // the whole lowercase path. + let expected = match format { + ArchiveFormat::Dat1 | ArchiveFormat::Toee => ["ART\\OTHER.FRM", "ART\\hero.frm"], + ArchiveFormat::Dat2 | ArchiveFormat::Arcanum => ["ART\\OTHER.FRM", "art\\hero.frm"], + }; + assert_eq!(sorted_names(&reopened(&archive)), expected, "{format:?}"); + } + } + + #[test] + fn by_default_delete_ignores_case() { + for format in ALL_FORMATS { + let mut archive = reopened(&archive_with(format, &["ART/HERO.FRM", "ART/OTHER.FRM"])); + archive + .delete(&["art/hero.frm".to_string()], CaseMode::Insensitive) + .unwrap(); + assert_eq!(sorted_names(&archive), ["ART\\OTHER.FRM"], "{format:?}"); + } + } + + #[test] + fn with_case_sensitive_delete_needs_the_exact_case() { + for format in ALL_FORMATS { + let mut archive = reopened(&archive_with(format, &["ART/HERO.FRM"])); + assert!( + archive + .delete(&["art/hero.frm".to_string()], CaseMode::Sensitive) + .is_err(), + "{format:?}" + ); + } + } + + /// DAT1 stores one name per directory, so a file added under a differently + /// cased directory joins the stored one instead of creating a second. + #[test] + fn dat1_adding_into_a_directory_stored_in_another_case_round_trips() { + let mut archive = reopened(&archive_with(ArchiveFormat::Dat1, &["ART/HERO.FRM"])); + add_names(&mut archive, &["art/new.frm"], CaseMode::Insensitive); + let archive = reopened(&archive); + assert_eq!(sorted_names(&archive), ["ART\\HERO.FRM", "ART\\new.frm"]); + + let out = ScratchPath::dir("dat1_case_dir"); + archive + .extract( + &out, + ExtractionMode::PreserveStructure, + &all_entries(CaseMode::Insensitive), + ) + .unwrap(); + assert_eq!( + std::fs::read(out.join("art").join("new.frm")).unwrap(), + b"art/new.frm" + ); + } + + fn sorted_names(archive: &DatArchive) -> Vec { + let mut names = archive.entry_names(); + names.sort(); + names + } + + #[test] + fn a_glob_deletes_every_matching_entry_and_nothing_else() { + for format in ALL_FORMATS { + let mut archive = archive_with( + format, + &["ART/A.FRM", "ART/B.FRM", "ART/A.TXT", "TEXT/A.FRM"], + ); + archive + .delete(&["ART/*.FRM".to_string()], CaseMode::Sensitive) + .unwrap(); + assert_eq!( + sorted_names(&archive), + ["ART\\A.TXT", "TEXT\\A.FRM"], + "{format:?}" + ); + } + } + + #[test] + fn a_plain_name_deletes_only_the_entry_with_that_exact_name() { + for format in ALL_FORMATS { + let mut archive = archive_with(format, &["DATA/A.TXT", "DATA/BIGA.TXT"]); + archive + .delete(&["DATA/A.TXT".to_string()], CaseMode::Sensitive) + .unwrap(); + assert_eq!(sorted_names(&archive), ["DATA\\BIGA.TXT"], "{format:?}"); + } + } + + #[test] + fn an_unmatched_pattern_fails_before_anything_is_deleted() { + for format in ALL_FORMATS { + let mut archive = archive_with(format, &["DATA/A.TXT", "DATA/B.TXT"]); + let err = archive + .delete( + &["DATA/A.TXT".to_string(), "*.ZZZ".to_string()], + CaseMode::Sensitive, + ) + .unwrap_err(); + assert_eq!( + err.to_string(), + "Some requested files were not found", + "{format:?}" + ); + assert_eq!( + sorted_names(&archive), + ["DATA\\A.TXT", "DATA\\B.TXT"], + "{format:?}" + ); + } + } + + // -- In-memory API -- + + fn level(n: u8) -> CompressionLevel { + CompressionLevel::new(n).unwrap() + } + + /// Serialize and parse again, without touching the filesystem + fn through_bytes(archive: &DatArchive) -> DatArchive { + DatArchive::from_bytes(archive.to_bytes().unwrap()).unwrap() + } + + #[test] + fn inserted_entries_survive_to_bytes_and_from_bytes() { + let compressible = b"frame ".repeat(500); + for format in ALL_FORMATS { + let mut archive = DatArchive::new(format); + archive + .insert( + "art/Hero.FRM", + compressible.clone(), + level(9), + CaseMode::Insensitive, + ) + .unwrap(); + archive + .insert( + "README.TXT", + b"hi".to_vec(), + level(0), + CaseMode::Insensitive, + ) + .unwrap(); + + let parsed = through_bytes(&archive); + assert_eq!(parsed.format(), format); + assert_eq!( + sorted_names(&parsed), + ["README.TXT", "art\\Hero.FRM"], + "{format:?}" + ); + for opened in [&archive, &parsed] { + assert_eq!( + opened.read("art/Hero.FRM", CaseMode::Insensitive).unwrap(), + compressible, + "{format:?}" + ); + assert_eq!( + opened.read("README.TXT", CaseMode::Insensitive).unwrap(), + b"hi" + ); + } + } + } + + #[test] + fn to_bytes_matches_what_save_writes() { + for format in ALL_FORMATS { + let archive = archive_with(format, &["ART/HERO.FRM", "TEXT/A.TXT"]); + let path = ScratchPath::new("archive_to_bytes"); + archive.save(&path).unwrap(); + assert_eq!( + archive.to_bytes().unwrap(), + std::fs::read(&path).unwrap(), + "{format:?}" + ); + } + } + + #[test] + fn entries_report_sizes_and_compression() { + let compressible = b"frame ".repeat(500); + for format in ALL_FORMATS { + let mut archive = DatArchive::new(format); + archive + .insert( + "a.frm", + compressible.clone(), + level(9), + CaseMode::Insensitive, + ) + .unwrap(); + let entries = through_bytes(&archive).entries(); + assert_eq!(entries.len(), 1, "{format:?}"); + let entry = &entries[0]; + assert_eq!(entry.name, "a.frm"); + assert_eq!(entry.size as usize, compressible.len(), "{format:?}"); + // DAT1 writing is uncompressed; the zlib formats keep compression that saves space + match format { + ArchiveFormat::Dat1 => { + assert!(!entry.compressed); + assert_eq!(entry.packed_size, entry.size); + } + ArchiveFormat::Dat2 | ArchiveFormat::Arcanum | ArchiveFormat::Toee => { + assert!(entry.compressed, "{format:?}"); + assert!(entry.packed_size < entry.size, "{format:?}"); + } + } + } + } + + #[test] + fn insert_replaces_a_name_as_case_compares_and_stores_the_name_as_given() { + for format in ALL_FORMATS { + let mut archive = DatArchive::new(format); + archive + .insert( + "Data/A.txt", + b"old".to_vec(), + level(0), + CaseMode::Insensitive, + ) + .unwrap(); + archive + .insert( + "data\\A.TXT", + b"new".to_vec(), + level(0), + CaseMode::Insensitive, + ) + .unwrap(); + let parsed = through_bytes(&archive); + assert_eq!(parsed.entry_names().len(), 1, "{format:?}"); + assert_eq!( + parsed.read("DATA/A.TXT", CaseMode::Insensitive).unwrap(), + b"new" + ); + + archive + .insert( + "data/a.txt", + b"twin".to_vec(), + level(0), + CaseMode::Sensitive, + ) + .unwrap(); + assert_eq!(archive.entry_names().len(), 2, "{format:?}"); + } + } + + #[test] + fn insert_refuses_names_an_archive_cannot_hold() { + let too_long = "a".repeat(common::MAX_PATH_BYTES + 1); + for format in ALL_FORMATS { + let mut archive = DatArchive::new(format); + for name in [ + "", + "../escape.txt", + "/abs.txt", + "dir/con.txt", + "a:b.txt", + too_long.as_str(), + ] { + assert!( + archive + .insert(name, b"x".to_vec(), level(0), CaseMode::Insensitive) + .is_err(), + "{format:?} accepted {name:?}" + ); + } + assert!(archive.entry_names().is_empty(), "{format:?}"); + } + } + + #[test] + fn read_takes_the_exact_name_first_then_any_case() { + for format in [ArchiveFormat::Dat2, ArchiveFormat::Arcanum] { + let mut archive = DatArchive::new(format); + archive + .insert( + "README.TXT", + b"upper".to_vec(), + level(0), + CaseMode::Sensitive, + ) + .unwrap(); + archive + .insert( + "readme.txt", + b"lower".to_vec(), + level(0), + CaseMode::Sensitive, + ) + .unwrap(); + archive + .insert( + "Other.txt", + b"other".to_vec(), + level(0), + CaseMode::Sensitive, + ) + .unwrap(); + let archive = through_bytes(&archive); + let case = CaseMode::Insensitive; + + assert_eq!(archive.read("README.TXT", case).unwrap(), b"upper"); + assert_eq!(archive.read("readme.txt", case).unwrap(), b"lower"); + assert_eq!(archive.read("OTHER.TXT", case).unwrap(), b"other"); + let ambiguous = archive.read("Readme.txt", case).unwrap_err().to_string(); + assert!(ambiguous.contains("README.TXT"), "{format:?}: {ambiguous}"); + let missing = archive.read("missing.txt", case).unwrap_err().to_string(); + assert!(missing.contains("not found"), "{format:?}: {missing}"); + assert!(archive.read("OTHER.TXT", CaseMode::Sensitive).is_err()); + } + } + + #[test] + fn remove_reports_whether_an_entry_was_removed() { + for format in ALL_FORMATS { + let mut archive = + through_bytes(&archive_with(format, &["ART/HERO.FRM", "ART/OTHER.FRM"])); + assert!( + archive + .remove("art/hero.frm", CaseMode::Insensitive) + .unwrap(), + "{format:?}" + ); + assert!( + !archive + .remove("art/hero.frm", CaseMode::Insensitive) + .unwrap(), + "{format:?}" + ); + assert_eq!( + sorted_names(&through_bytes(&archive)), + ["ART\\OTHER.FRM"], + "{format:?}" + ); + } + } + + #[test] + fn from_bytes_rejects_data_that_is_no_archive() { + assert!(DatArchive::from_bytes(b"not an archive".to_vec()).is_err()); + } +} diff --git a/src/common.rs b/crates/dat3-core/src/common.rs similarity index 91% rename from src/common.rs rename to crates/dat3-core/src/common.rs index b82c06b..b89d305 100644 --- a/src/common.rs +++ b/crates/dat3-core/src/common.rs @@ -144,7 +144,8 @@ pub enum ExtractionMode { /// What to do about a requested name or glob that matches nothing in the archive #[derive(Debug, Clone, Copy)] pub enum MissingFiles { - /// Report the misses and fail without listing or extracting anything + /// Report the misses and fail. Extraction writes nothing; a listing has + /// already printed whatever did match. Fail, /// Report the misses as a warning and carry on with whatever did match Warn, @@ -186,6 +187,8 @@ pub struct NameView { } impl NameView { + /// A view over every stored name in one archive; `names` must be all of + /// them, since a case-only twin anywhere keeps a name in its stored case. pub fn new<'a>(mode: CaseMode, names: impl IntoIterator) -> Self { let ambiguous = match mode { CaseMode::Sensitive => std::collections::HashSet::new(), @@ -268,7 +271,9 @@ pub enum ListFormat { pub struct Selection<'a> { /// Names and globs as given; empty selects every entry pub patterns: &'a [String], + /// What to do about a pattern that matches no entry pub on_missing: MissingFiles, + /// How patterns compare with entry names, and how names are shown and extracted pub case: CaseMode, } @@ -459,17 +464,16 @@ pub fn extract_archive_parallel( utils::ensure_dir_exists(&output_path)?; - // Read and optionally decompress - let file_data = utils::read_file_slice(archive_data, file) - .with_context(|| format!("Failed to read data for file '{}'", file.name))?; - let write_result = if file.compressed { - let decompressed = decompress(file_data, file.size as usize) - .with_context(|| format!("Failed to decompress {}", file.name))?; - fs::write(&output_path, decompressed) - } else { - fs::write(&output_path, file_data) - }; - write_result.with_context(|| format!("Failed to write {}", output_path.display()))?; + let contents = entry_contents(archive_data, file, &decompress)?; + // Writing through a link left at the destination would overwrite its + // target outside the output directory, so the link itself is replaced. + if fs::symlink_metadata(&output_path).is_ok_and(|m| m.file_type().is_symlink()) { + fs::remove_file(&output_path).with_context(|| { + format!("Failed to replace symlink {}", output_path.display()) + })?; + } + fs::write(&output_path, contents) + .with_context(|| format!("Failed to write {}", output_path.display()))?; // Counted once written, every 1000 files and at the end let count = completed.fetch_add(1, Ordering::Relaxed) + 1; @@ -491,6 +495,90 @@ pub fn extract_archive_parallel( Ok(()) } +/// An entry's contents: borrowed when stored uncompressed, decompressed with +/// `decompress` otherwise +pub fn entry_contents<'a>( + archive_data: &'a [u8], + file: &'a FileEntry, + decompress: impl Fn(&[u8], usize) -> Result>, +) -> Result> { + let stored = utils::read_file_slice(archive_data, file) + .with_context(|| format!("Failed to read data for file '{}'", file.name))?; + if file.compressed { + let decompressed = decompress(stored, file.size as usize) + .with_context(|| format!("Failed to decompress {}", file.name))?; + Ok(std::borrow::Cow::Owned(decompressed)) + } else { + Ok(std::borrow::Cow::Borrowed(stored)) + } +} + +/// The position among stored `names` of the one `name` (`/` or `\` separated) +/// refers to. +/// +/// The exact name wins. Failing that, under [`CaseMode::Insensitive`] the one +/// name equal ignoring case; several such names are an error, since only the +/// exact spelling can tell them apart. +pub fn find_stored_index<'a>( + names: impl IntoIterator, + name: &str, + case: CaseMode, +) -> Result> { + let names = names.into_iter(); + let wanted = utils::normalize_user_path(name); + // Exact pass first, without allocating: folding each name the scan passed + // cost an allocation per entry on every lookup. + if let Some(index) = names.clone().position(|stored| stored == wanted) { + return Ok(Some(index)); + } + if case == CaseMode::Sensitive { + return Ok(None); + } + let folded = case.fold(&wanted); + let matches: Vec<(usize, &str)> = names + .enumerate() + .filter(|(_, stored)| case.fold(stored) == folded) + .collect(); + match matches.as_slice() { + [] => Ok(None), + [(index, _)] => Ok(Some(*index)), + several => { + let shown: Vec = several + .iter() + .map(|(_, stored)| utils::normalize_path_for_display(stored)) + .collect(); + bail!( + "{} matches entries that differ only in case; name one exactly: {}", + utils::normalize_path_for_display(name), + shown.join(" / ") + ) + } + } +} + +/// Merge `new_entries` into a zlib-format entry list: each replaces the +/// entries of its name, compared as `case` says, a batch naming one path twice +/// keeps its first entry, and the list stays sorted as the formats require. +pub fn merge_entries(entries: &mut Vec, new_entries: Vec, case: CaseMode) { + use std::collections::HashSet; + + let key = |name: &str| case.fold(name).into_owned(); + let new_file_names: HashSet = new_entries.iter().map(|e| key(&e.name)).collect(); + entries.retain(|existing_file| !new_file_names.contains(&key(&existing_file.name))); + + let mut seen_names = HashSet::new(); + for entry in new_entries { + if seen_names.insert(key(&entry.name)) { + entries.push(entry); + } + } + + // The formats require entries sorted alphabetically (case-insensitive). + // Cached: the key allocates, and sort_by_key recomputes it per comparison + // rather than per element. + entries.sort_by_cached_key(|f| f.name.to_lowercase()); +} + /// Read files from disk into an entry list: zlib-compress when it saves /// space, replace same-named entries, dedupe the batch, and keep the list /// sorted case-insensitively as the zlib-based formats require. @@ -505,7 +593,6 @@ pub fn add_files_zlib( case: CaseMode, ) -> Result<()> { use rayon::prelude::*; - use std::collections::HashSet; let base_path = file_path; let files = utils::collect_files(file_path).with_context(|| { @@ -531,25 +618,7 @@ pub fn add_files_zlib( .collect(); let new_entries = results?; // Collect results, propagating the first error if any file failed - - // An added file replaces every existing entry of its name, compared as - // `case` says, and a batch naming one path twice keeps its first file. - let key = |name: &str| case.fold(name).into_owned(); - let new_file_names: HashSet = new_entries.iter().map(|e| key(&e.name)).collect(); - entries.retain(|existing_file| !new_file_names.contains(&key(&existing_file.name))); - - let mut seen_names = HashSet::new(); - for entry in new_entries { - if seen_names.insert(key(&entry.name)) { - entries.push(entry); - } - } - - // The formats require entries sorted alphabetically (case-insensitive). - // Cached: the key allocates, and sort_by_key recomputes it per comparison - // rather than per element. - entries.sort_by_cached_key(|f| f.name.to_lowercase()); - + merge_entries(entries, new_entries, case); Ok(()) } @@ -565,33 +634,30 @@ fn process_single_file_for_adding( let data = utils::read_entry_data(file)?; let archive_path = utils::calculate_archive_path(file, base_path, target_dir, source_root)?; let archive_path = case.fold(&archive_path).into_owned(); - let display_path = utils::normalize_path_for_display(&archive_path); - // The readers reject longer names, so writing one would produce an archive - // dat3 itself cannot open. - if archive_path.len() > MAX_PATH_BYTES { - bail!("Archive path is longer than {MAX_PATH_BYTES} bytes: {display_path}"); - } - print_stdout(format_args!("Adding: {display_path}")); + utils::check_entry_name_len(&archive_path)?; + print_stdout(format_args!( + "Adding: {}", + utils::normalize_path_for_display(&archive_path) + )); + zlib_entry(archive_path, data, compression) +} +/// An entry for a zlib format, compressed at `compression` when that saves space +pub fn zlib_entry(name: String, data: Vec, compression: CompressionLevel) -> Result { if compression.level() > 0 { let compressed_data = compress_zlib(&data, compression.level())?; // Only use compression if it actually saves space if compressed_data.len() < data.len() { - Ok(FileEntry::with_compression_data( - archive_path, + return Ok(FileEntry::with_compression_data( + name, data, compressed_data, - )) - } else { - let mut entry = FileEntry::with_data(archive_path, data, false); - entry.size = entry.packed_size; - Ok(entry) + )); } - } else { - let mut entry = FileEntry::with_data(archive_path, data, false); - entry.size = entry.packed_size; - Ok(entry) } + let mut entry = FileEntry::with_data(name, data, false); + entry.size = entry.packed_size; + Ok(entry) } /// Compress data using zlib @@ -652,23 +718,19 @@ pub fn decompress_zlib(data: &[u8], expected_size: usize) -> Result> { Ok(decompressed) } -/// Delete a file from a list by normalized name. +/// Remove the entry named exactly `file_name` (`/` or `\` separated) from a +/// list, reporting whether there was one. /// -/// Shared by the DAT2, Arcanum, and ToEE delete implementations; DAT1 keeps its -/// files per directory and deletes through its own. -pub fn delete_file_from_list(files: &mut Vec, file_name: &str) -> Result<()> { - let normalized_name = utils::normalize_user_path(file_name).into_owned(); - - if let Some(pos) = files.iter().position(|f| f.name == normalized_name) { - let display_name = utils::normalize_path_for_display(&normalized_name); - print_stdout(format_args!("Deleting: {display_name}")); - files.remove(pos); - Ok(()) - } else { - bail!( - "File not found: {}", - utils::normalize_path_for_display(file_name) - ); +/// Shared by the DAT2, Arcanum, and ToEE archives; DAT1 keeps its files per +/// directory and removes through its own. +pub fn remove_from_list(files: &mut Vec, file_name: &str) -> bool { + let normalized_name = utils::normalize_user_path(file_name); + match files.iter().position(|f| f.name == normalized_name) { + Some(pos) => { + files.remove(pos); + true + } + None => false, } } @@ -1072,11 +1134,13 @@ pub mod utils { /// A pattern with glob metacharacters is a glob; one without a path separator /// matches the file name alone. Any other pattern matches as a substring, for /// backward compatibility. Both ignore case unless the `CaseMode` is sensitive. + #[derive(Debug)] pub struct NamePattern { source: String, kind: PatternKind, } + #[derive(Debug)] enum PatternKind { Glob { glob: glob::Pattern, @@ -1121,10 +1185,12 @@ pub mod utils { &self.source } + /// Whether the pattern has glob metacharacters and matches as a glob pub fn is_glob(&self) -> bool { matches!(self.kind, PatternKind::Glob { .. }) } + /// Whether the stored entry name `file_name` matches, comparing as `mode` says pub fn matches(&self, file_name: &str, mode: CaseMode) -> bool { match &self.kind { PatternKind::Substring => mode @@ -1393,6 +1459,26 @@ pub mod utils { Ok(()) } + /// Refuse a stored name longer than [`MAX_PATH_BYTES`]: the readers reject + /// longer names, so writing one would produce an archive dat3 cannot open. + pub fn check_entry_name_len(archive_path: &str) -> Result<()> { + if archive_path.len() > MAX_PATH_BYTES { + bail!( + "Archive path is longer than {MAX_PATH_BYTES} bytes: {}", + normalize_path_for_display(archive_path) + ); + } + Ok(()) + } + + /// The stored, backslash-separated form of a name given for a new entry, + /// refusing names an archive cannot hold + pub fn stored_name_for_insert(name: &str) -> Result { + let stored = normalize_path_for_archive(&validate_add_archive_path(name)?); + check_entry_name_len(&stored)?; + Ok(stored) + } + /// Validate and normalize a path to be stored in a new archive. /// /// - Rejects `..` (ParentDir), absolute roots, and Windows drive prefixes. @@ -1482,8 +1568,6 @@ pub mod utils { PathBuf::from(dat_path.replace('\\', std::path::MAIN_SEPARATOR_STR)) } - /// Get just the filename (basename) from a path. - /// Handles both forward and backward slashes. /// Split entries for flat extraction into the ones to write and the ones a /// later entry of the same file name replaces. /// @@ -1516,6 +1600,8 @@ pub mod utils { ) } + /// Get just the filename (basename) from a path. + /// Handles both forward and backward slashes. pub fn get_filename_from_dat_path(path: &str) -> &str { path.rfind(['/', '\\']) .map(|pos| &path[pos + 1..]) diff --git a/src/common_tests.rs b/crates/dat3-core/src/common_tests.rs similarity index 95% rename from src/common_tests.rs rename to crates/dat3-core/src/common_tests.rs index 9ca62f2..f22709d 100644 --- a/src/common_tests.rs +++ b/crates/dat3-core/src/common_tests.rs @@ -350,6 +350,37 @@ mod tests { .unwrap(); assert_eq!(std::fs::read(out.join("SAME.TXT")).unwrap(), b"second"); } + + /// A link already sitting at an entry's destination is replaced, not + /// written through: following it would overwrite its target, outside the + /// output directory. + #[cfg(unix)] + #[test] + fn replaces_a_symlink_at_the_destination_instead_of_writing_through_it() { + let entries = [entry("SUB\\EVIL.TXT", b"from archive")]; + let refs: Vec<&FileEntry> = entries.iter().collect(); + let victim = crate::test_support::ScratchPath::new("link_victim"); + std::fs::write(&victim, b"original").unwrap(); + let out = crate::test_support::ScratchPath::dir("link_destination"); + let destination = out.join("SUB").join("EVIL.TXT"); + std::fs::create_dir(out.join("SUB")).unwrap(); + std::os::unix::fs::symlink(victim.path(), &destination).unwrap(); + + extract_archive_parallel( + &[], + &refs, + &out, + ExtractionMode::PreserveStructure, + &NameView::new(CaseMode::Sensitive, []), + |d, _| Ok(d.to_vec()), + ) + .unwrap(); + + assert_eq!(std::fs::read(&victim).unwrap(), b"original"); + let metadata = std::fs::symlink_metadata(&destination).unwrap(); + assert!(metadata.is_file(), "the destination is still a link"); + assert_eq!(std::fs::read(&destination).unwrap(), b"from archive"); + } } // ── read_file_slice ──────────────────────────────────────────── @@ -582,13 +613,10 @@ mod tests { #[test] fn round_trips_a_detected_archive() { let bytes = dat1_bytes(46, 1, "A.TXT"); - let path = - std::env::temp_dir().join(format!("dat3_detect_rt_{}.dat", std::process::id())); + let path = crate::test_support::ScratchPath::new("detect_rt"); std::fs::write(&path, &bytes).unwrap(); let archive = DatArchive::open(&path).unwrap(); - std::fs::remove_file(&path).ok(); - let dir = std::env::temp_dir().join(format!("dat3_detect_x_{}", std::process::id())); - let _ = std::fs::remove_dir_all(&dir); + let dir = crate::test_support::ScratchPath::new("detect_x"); archive .extract( &dir, @@ -596,54 +624,7 @@ mod tests { &crate::test_support::exact(&[], MissingFiles::Fail), ) .unwrap(); - let got = std::fs::read(dir.join("A.TXT")).unwrap(); - std::fs::remove_dir_all(&dir).unwrap(); - assert_eq!(got, b"hi"); - } - } - - // ── CLI argument parsing ─────────────────────────────────────── - - mod cli_args { - use clap::Parser; - - #[test] - fn rejects_out_of_range_compression_at_parse_time() { - let result = crate::Cli::try_parse_from(["dat3", "a", "test.dat", "-c", "10", "file"]); - assert!( - result.is_err(), - "compression level 10 should be rejected during argument parsing" - ); - } - - #[test] - fn accepts_maximum_compression_level() { - let result = crate::Cli::try_parse_from(["dat3", "a", "test.dat", "-c", "9", "file"]); - assert!(result.is_ok()); - } - - #[test] - fn accepts_each_archive_format() { - for format in ["dat1", "dat2", "arcanum", "toee"] { - let result = crate::Cli::try_parse_from([ - "dat3", "a", "test.dat", "--format", format, "file", - ]); - assert!(result.is_ok(), "--format {format} should parse"); - } - } - - #[test] - fn rejects_unknown_format_and_removed_format_flags() { - for args in [ - ["dat3", "a", "test.dat", "--format", "zip", "file"].as_slice(), - ["dat3", "a", "test.dat", "--dat1", "file"].as_slice(), - ["dat3", "a", "test.dat", "--arcanum", "file"].as_slice(), - ] { - assert!( - crate::Cli::try_parse_from(args.iter().copied()).is_err(), - "{args:?} should be rejected" - ); - } + assert_eq!(std::fs::read(dir.join("A.TXT")).unwrap(), b"hi"); } } @@ -761,6 +742,20 @@ mod tests { fn null_only_input() { assert_eq!(utils::decode_filename(b"\0\0").unwrap(), ""); } + + /// The other formats name the entry whose name failed to decode; DAT1 + /// names it by directory, since its entries are numbered per directory. + #[test] + fn a_dat1_error_names_the_entry_with_the_bad_name() { + let error = + crate::dat1::Dat1Archive::from_bytes(super::dat1_bytes(46, 1, "\u{e9}.TXT")) + .unwrap_err(); + let message = format!("{error:#}"); + assert!( + message.contains("Failed to decode name for file entry 0 in directory '.'"), + "got: {message}" + ); + } } // ── validate_filename_ascii ──────────────────────────────────── @@ -1236,19 +1231,15 @@ mod tests { /// prove the extract path calls it. #[test] fn extraction_writes_nothing_outside_the_output_directory() { - let pid = std::process::id(); - let escape = std::env::temp_dir().join(format!("dat3_escape_{pid}.txt")); - std::fs::remove_file(&escape).ok(); - assert!(!escape.exists(), "stale probe file from an earlier run"); - - // Stored the way a hostile archive would: a leading separator, which - // is what makes Path::join discard the output directory. - let entry = format!("\\tmp\\dat3_escape_{pid}.txt"); - let archive_path = std::env::temp_dir().join(format!("dat3_escape_src_{pid}.dat")); + let escape = crate::test_support::ScratchPath::new("escape"); + + // Stored the way a hostile archive would: the absolute path, which is + // what makes Path::join discard the output directory. + let entry = escape.display().to_string().replace('/', "\\"); + let archive_path = crate::test_support::ScratchPath::new("escape_src"); std::fs::write(&archive_path, super::dat1_bytes(46, 1, &entry)).unwrap(); - let out = std::env::temp_dir().join(format!("dat3_escape_out_{pid}")); - let _ = std::fs::remove_dir_all(&out); + let out = crate::test_support::ScratchPath::new("escape_out"); let archive = DatArchive::open(&archive_path).unwrap(); let result = archive.extract( &out, @@ -1256,12 +1247,10 @@ mod tests { &crate::test_support::exact(&[], MissingFiles::Fail), ); - let escaped = escape.exists(); - std::fs::remove_file(&archive_path).ok(); - std::fs::remove_file(&escape).ok(); - let _ = std::fs::remove_dir_all(&out); - - assert!(!escaped, "extraction wrote outside the output directory"); + assert!( + !escape.exists(), + "extraction wrote outside the output directory" + ); assert!(result.is_err(), "extraction of a hostile entry should fail"); } diff --git a/src/dat1.rs b/crates/dat3-core/src/dat1.rs similarity index 86% rename from src/dat1.rs rename to crates/dat3-core/src/dat1.rs index 0d10925..1c4bc0b 100644 --- a/src/dat1.rs +++ b/crates/dat3-core/src/dat1.rs @@ -14,6 +14,7 @@ LZSS compression for writing is not implemented - files are stored uncompressed. use anyhow::{Context, Result, bail}; use deku::prelude::*; +use std::borrow::Cow; use std::io::Write; use std::path::Path; @@ -96,6 +97,12 @@ pub struct Dat1Archive { data: Vec, } +impl Default for Dat1Archive { + fn default() -> Self { + Self::new() + } +} + impl Dat1Archive { /// Create a new empty DAT1 archive with just a root directory pub fn new() -> Self { @@ -121,7 +128,8 @@ impl Dat1Archive { })?; rest = r; dir_names.push( - utils::decode_filename(&name.bytes).context("Failed to decode directory name")?, + utils::decode_filename(&name.bytes) + .with_context(|| format!("Failed to decode name for directory {i}"))?, ); } @@ -146,8 +154,9 @@ impl Dat1Archive { })?; rest = r; - let name = utils::decode_filename(&entry.name_bytes) - .context("Failed to decode file name")?; + let name = utils::decode_filename(&entry.name_bytes).with_context(|| { + format!("Failed to decode name for file entry {j} in directory '{dir_name}'") + })?; let compressed = entry.attributes & DAT1_COMPRESSED_FLAG != 0; let actual_packed_size = if entry.packed_size == 0 { entry.size @@ -222,9 +231,6 @@ impl Dat1Archive { source_root: Option<&Path>, case: CaseMode, ) -> Result<()> { - use std::collections::HashSet; - - let key = |name: &str| case.fold(name).into_owned(); let base_path = file_path; let files = utils::collect_files(file_path).with_context(|| { format!( @@ -254,6 +260,27 @@ impl Dat1Archive { new_entries.push(file_entry); } + self.insert_entries(new_entries, case); + Ok(()) + } + + /// An entry's contents, decompressed + pub fn contents<'a>(&'a self, file: &'a FileEntry) -> Result> { + common::entry_contents(&self.data, file, lzss::decompress) + } + + /// Add a prepared entry, replacing the entries of its name as `case` compares + pub fn insert_entry(&mut self, entry: FileEntry, case: CaseMode) { + self.insert_entries(vec![entry], case); + } + + /// Place new entries in their directories, replacing existing entries of + /// the same name as `case` compares + fn insert_entries(&mut self, new_entries: Vec, case: CaseMode) { + use std::collections::HashSet; + + let key = |name: &str| case.fold(name).into_owned(); + // Every directory is swept, not just the ones the new entries land in: // an archive read from disk can hold an entry whose name does not match // the bucket it sits in, and both copies would then be written out. @@ -289,31 +316,37 @@ impl Dat1Archive { }; self.directories[dir_index].files.push(entry); } - - Ok(()) } - /// Delete a file from the archive by name - pub fn delete_file(&mut self, file_name: &str) -> Result<()> { - let normalized_name = utils::normalize_user_path(file_name).into_owned(); - + /// Remove the entry named exactly `file_name`, reporting whether there was one + pub fn remove_entry(&mut self, file_name: &str) -> bool { + let normalized_name = utils::normalize_user_path(file_name); for dir in &mut self.directories { if let Some(pos) = dir.files.iter().position(|f| f.name == normalized_name) { - let display_name = utils::normalize_path_for_display(&normalized_name); - common::print_stdout(format_args!("Deleting: {display_name}")); dir.files.remove(pos); - return Ok(()); + return true; } } - - bail!( - "File not found: {}", - utils::normalize_path_for_display(file_name) - ); + false } - /// Save the archive to a file + /// Save the archive to a DAT1 file pub fn save(&self, path: &Path) -> Result<()> { + let (dirs, data_offset) = self.prepare_save()?; + utils::write_atomically(path, |out| self.write_prepared(&dirs, data_offset, out)) + .context(WRITE_CONTEXT) + } + + /// Write the archive as a DAT1 file to `out` + pub fn write_to(&self, out: &mut dyn Write) -> Result<()> { + let (dirs, data_offset) = self.prepare_save()?; + self.write_prepared(&dirs, data_offset, out) + .context(WRITE_CONTEXT) + } + + /// The directories to write and where file data starts, checked before any + /// output exists so a refused save leaves no temp file + fn prepare_save(&self) -> Result<(Vec<&Directory>, u32)> { // Calculate where file data starts: header, directory names, then // directory content blocks. Computed up front so entry offsets are // known before anything is written. @@ -359,82 +392,87 @@ impl Dat1Archive { if data_offset as u64 + total_payload > u32::MAX as u64 { bail!("DAT1 archive would exceed the format's 4 GiB offset limit"); } + Ok((dirs_to_write, data_offset)) + } + + fn write_prepared( + &self, + dirs_to_write: &[&Directory], + data_offset: u32, + output: &mut dyn Write, + ) -> Result<()> { + output.write_all( + &Dat1Header { + dir_count: dirs_to_write.len() as u32, + // The hint must cover the count, and the reader keys on that + // to recognise the header; the count itself always satisfies it. + folder_allocation_hint: dirs_to_write.len() as u32, + reserved: 0, + // Zero rather than the clock: nothing reads it back, and a + // constant keeps repacking the same tree byte-reproducible. + timestamp: 0, + } + .to_bytes()?, + )?; - utils::write_atomically(path, |output| { + // Write directory names + for dir in dirs_to_write { output.write_all( - &Dat1Header { - dir_count: dirs_to_write.len() as u32, - // The hint must cover the count, and the reader keys on that - // to recognise the header; the count itself always satisfies it. - folder_allocation_hint: dirs_to_write.len() as u32, - reserved: 0, - // Zero rather than the clock: nothing reads it back, and a - // constant keeps repacking the same tree byte-reproducible. - timestamp: 0, + &Dat1Name { + len: name_len_u8("directory name", &dir.name)?, + bytes: dir.name.as_bytes().to_vec(), } .to_bytes()?, )?; + } - // Write directory names - for dir in &dirs_to_write { - output.write_all( - &Dat1Name { - len: name_len_u8("directory name", &dir.name)?, - bytes: dir.name.as_bytes().to_vec(), - } - .to_bytes()?, - )?; - } + let mut current_offset = data_offset; - let mut current_offset = data_offset; - - // Write directory content headers and file entries - for dir in &dirs_to_write { - output.write_all( - &Dat1DirHeader { - file_count: dir.files.len() as u32, - file_allocation_hint: dir.files.len() as u32, - fixed_metadata_size: DAT1_ENTRY_METADATA_SIZE, - timestamp: 0, - } - .to_bytes()?, - )?; - - for file in &dir.files { - let stored_name = stored_file_name(&dir.name, &file.name); - let entry = Dat1FileEntry { - name_len: name_len_u8("file name", stored_name)?, - name_bytes: stored_name.as_bytes().to_vec(), - attributes: if file.compressed { - DAT1_COMPRESSED_FLAG - } else { - DAT1_UNCOMPRESSED_FLAG - }, - offset: current_offset, - size: file.size, - packed_size: if file.compressed { file.packed_size } else { 0 }, - }; - output.write_all(&entry.to_bytes()?)?; - - current_offset += file.packed_size; + // Write directory content headers and file entries + for dir in dirs_to_write { + output.write_all( + &Dat1DirHeader { + file_count: dir.files.len() as u32, + file_allocation_hint: dir.files.len() as u32, + fixed_metadata_size: DAT1_ENTRY_METADATA_SIZE, + timestamp: 0, } - } + .to_bytes()?, + )?; - // Write file data, borrowed from memory or from the original archive - for dir in &dirs_to_write { - for file in &dir.files { - output.write_all(self.read_file_data(file)?)?; - } + for file in &dir.files { + let stored_name = stored_file_name(&dir.name, &file.name); + let entry = Dat1FileEntry { + name_len: name_len_u8("file name", stored_name)?, + name_bytes: stored_name.as_bytes().to_vec(), + attributes: if file.compressed { + DAT1_COMPRESSED_FLAG + } else { + DAT1_UNCOMPRESSED_FLAG + }, + offset: current_offset, + size: file.size, + packed_size: if file.compressed { file.packed_size } else { 0 }, + }; + output.write_all(&entry.to_bytes()?)?; + + current_offset += file.packed_size; } + } - Ok(()) - }) - .context("Failed to write DAT1 file")?; + // Write file data, borrowed from memory or from the original archive + for dir in dirs_to_write { + for file in &dir.files { + output.write_all(self.read_file_data(file)?)?; + } + } Ok(()) } } +const WRITE_CONTEXT: &str = "Failed to write DAT1 file"; + /// Name as stored in a directory's content block: the directory prefix is /// stripped for real directories; root (".") entries are stored as-is. /// DAT1 prefixes names with a one-byte length, so a longer name cannot be stored. diff --git a/src/dat2.rs b/crates/dat3-core/src/dat2.rs similarity index 84% rename from src/dat2.rs rename to crates/dat3-core/src/dat2.rs index e73ca19..62e8380 100644 --- a/src/dat2.rs +++ b/crates/dat3-core/src/dat2.rs @@ -12,6 +12,7 @@ Little-endian, flat file list, zlib compression, parallel extraction via rayon. use anyhow::{Context, Result, bail}; use byteorder::{LittleEndian, ReadBytesExt, WriteBytesExt}; use deku::prelude::*; +use std::borrow::Cow; use std::io::{Cursor, Write}; use std::path::Path; @@ -56,6 +57,12 @@ pub struct Dat2Archive { data: Vec, } +impl Default for Dat2Archive { + fn default() -> Self { + Self::new() + } +} + impl Dat2Archive { /// Create a new empty DAT2 archive pub fn new() -> Self { @@ -191,77 +198,98 @@ impl Dat2Archive { ) } - /// Delete a file from the archive by name - pub fn delete_file(&mut self, file_name: &str) -> Result<()> { - common::delete_file_from_list(&mut self.files, file_name) + /// An entry's contents, decompressed + pub fn contents<'a>(&'a self, file: &'a FileEntry) -> Result> { + common::entry_contents(&self.data, file, common::decompress_zlib) } - /// Save the archive to a DAT2 file. - /// - /// DAT2 layout: file data, then directory tree, then 8-byte footer. + /// Add a prepared entry, replacing the entries of its name as `case` compares + pub fn insert_entry(&mut self, entry: FileEntry, case: CaseMode) { + common::merge_entries(&mut self.files, vec![entry], case); + } + + /// Remove the entry named exactly `file_name`, reporting whether there was one + pub fn remove_entry(&mut self, file_name: &str) -> bool { + common::remove_from_list(&mut self.files, file_name) + } + + /// Save the archive to a DAT2 file pub fn save(&self, path: &Path) -> Result<()> { + self.prepare_save()?; + utils::write_atomically(path, |out| self.write_prepared(out)).context(WRITE_CONTEXT) + } + + /// Write the archive as a DAT2 file to `out` + pub fn write_to(&self, out: &mut dyn Write) -> Result<()> { + self.prepare_save()?; + self.write_prepared(out).context(WRITE_CONTEXT) + } + + /// Checks that fail before any output exists, so a refused save leaves no temp file + fn prepare_save(&self) -> Result<()> { // DAT2 stores file offsets as u32. Entries keep data.len() == packed_size, // so this bounds the u32 offset accumulation below. let total_payload: u64 = self.files.iter().map(|f| f.packed_size as u64).sum(); if total_payload > u32::MAX as u64 { bail!("DAT2 archive would exceed the format's 4 GiB offset limit"); } + Ok(()) + } - utils::write_atomically(path, |cursor| { - // Step 1: Write all file data - let mut current_offset = 0u32; - let mut file_offsets = Vec::new(); + /// DAT2 layout: file data, then directory tree, then 8-byte footer. + fn write_prepared(&self, cursor: &mut dyn Write) -> Result<()> { + // Step 1: Write all file data + let mut current_offset = 0u32; + let mut file_offsets = Vec::new(); - for file in &self.files { - file_offsets.push(current_offset); + for file in &self.files { + file_offsets.push(current_offset); - // In memory for a newly added file, borrowed from the original archive otherwise - let data = self.read_file_data(file)?; + // In memory for a newly added file, borrowed from the original archive otherwise + let data = self.read_file_data(file)?; - cursor.write_all(data)?; - current_offset += data.len() as u32; - } + cursor.write_all(data)?; + current_offset += data.len() as u32; + } - // Step 2: Write directory tree, tracking its size since a file - // writer has no cheap position() like the old in-memory cursor - let tree_start = current_offset as u64; - cursor.write_u32::(self.files.len() as u32)?; - let mut tree_size: u64 = 4; - - for (i, file) in self.files.iter().enumerate() { - let entry = Dat2FileEntry { - filename_size: file.name.len() as u32, - filename_bytes: file.name.as_bytes().to_vec(), - compression_type: if file.compressed { 1 } else { 0 }, - real_size: file.size, - packed_size: file.packed_size, - offset: file_offsets[i], - }; - - let entry_bytes = entry.to_bytes()?; - cursor.write_all(&entry_bytes)?; - tree_size += entry_bytes.len() as u64; - } + // Step 2: Write directory tree, tracking its size since a file + // writer has no cheap position() like the old in-memory cursor + let tree_start = current_offset as u64; + cursor.write_u32::(self.files.len() as u32)?; + let mut tree_size: u64 = 4; + + for (i, file) in self.files.iter().enumerate() { + let entry = Dat2FileEntry { + filename_size: file.name.len() as u32, + filename_bytes: file.name.as_bytes().to_vec(), + compression_type: if file.compressed { 1 } else { 0 }, + real_size: file.size, + packed_size: file.packed_size, + offset: file_offsets[i], + }; - // Step 3: Write the footer - let total_size = tree_start + tree_size + FOOTER_SIZE as u64; + let entry_bytes = entry.to_bytes()?; + cursor.write_all(&entry_bytes)?; + tree_size += entry_bytes.len() as u64; + } - let footer = Dat2Footer { - tree_size: tree_size as u32, - dat_size: u32::try_from(total_size) - .context("DAT2 archive would exceed the format's 4 GiB size limit")?, - }; - let footer_bytes = footer.to_bytes()?; - cursor.write_all(&footer_bytes)?; + // Step 3: Write the footer + let total_size = tree_start + tree_size + FOOTER_SIZE as u64; - Ok(()) - }) - .context("Failed to write DAT2 file")?; + let footer = Dat2Footer { + tree_size: tree_size as u32, + dat_size: u32::try_from(total_size) + .context("DAT2 archive would exceed the format's 4 GiB size limit")?, + }; + let footer_bytes = footer.to_bytes()?; + cursor.write_all(&footer_bytes)?; Ok(()) } } +const WRITE_CONTEXT: &str = "Failed to write DAT2 file"; + #[cfg(test)] mod tests { use super::*; diff --git a/crates/dat3-core/src/lib.rs b/crates/dat3-core/src/lib.rs new file mode 100644 index 0000000..053fd5f --- /dev/null +++ b/crates/dat3-core/src/lib.rs @@ -0,0 +1,84 @@ +/*! +# dat3-core + +Read, list, extract, add to and delete from Fallout (DAT1, DAT2) and Troika +(Arcanum, ToEE) archives. + +## Entry points + +- [`DatArchive`] is an open or new archive: [`open`](DatArchive::open) detects the + format, [`new`](DatArchive::new) starts an empty one, and + [`list`](DatArchive::list), [`extract`](DatArchive::extract), + [`add_file`](DatArchive::add_file), [`delete`](DatArchive::delete) and + [`save`](DatArchive::save) work on it. Adds and deletes change only the + in-memory archive until it is saved. +- For archives and files held in memory, [`from_bytes`](DatArchive::from_bytes), + [`entries`](DatArchive::entries), [`read`](DatArchive::read), + [`insert`](DatArchive::insert), [`remove`](DatArchive::remove) and + [`to_bytes`](DatArchive::to_bytes) touch no filesystem and print nothing. +- [`ArchiveFormat`] names a format. +- The options those methods take live in [`common`]: [`Selection`](common::Selection), + [`CaseMode`](common::CaseMode), [`MissingFiles`](common::MissingFiles), + [`ExtractionMode`](common::ExtractionMode), [`ListFormat`](common::ListFormat) and + [`CompressionLevel`](common::CompressionLevel). + +The remaining items in [`common`] and [`common::utils`] are the helpers the +formats and the `dat3` command line share. They are public so that command line +can use them, and change with it. + +## Conventions + +- Entry names are stored with `\` separators; every method taking a name or + pattern also accepts `/`. +- A pattern with glob metacharacters (`*`, `?`, `[`) is a glob, matched against the + whole name if it contains a separator and against the file name otherwise. Any + other pattern matches as a substring. See [`NamePattern`](common::utils::NamePattern). +- [`CaseMode::Insensitive`](common::CaseMode::Insensitive) matches regardless of + case and shows, extracts and adds names in lowercase, except names in the archive + that differ only in case, which keep their stored case. + [`CaseMode::Sensitive`](common::CaseMode::Sensitive) uses names exactly as stored. +- Errors are [`anyhow::Error`], worded for a person to read. +- The path-based operations print as the `dat3` command line does: listings and + progress to stdout, warnings and patterns that matched nothing to stderr. +- [`DatArchive::open`] and [`DatArchive::from_bytes`] hold the whole archive in memory. + +## Example + +```no_run +use dat3_core::common::{CaseMode, CompressionLevel, ExtractionMode, MissingFiles, Selection}; +use dat3_core::{ArchiveFormat, DatArchive}; + +// Build a Fallout 2 archive from the `art` directory +let mut archive = DatArchive::new(ArchiveFormat::Dat2); +archive.add_file("art", CompressionLevel::new(9)?, None, None, CaseMode::Insensitive)?; +archive.save("patch000.dat")?; + +// Open it again and extract the FRM files, keeping their directories +let archive = DatArchive::open("patch000.dat")?; +let patterns = ["*.frm".to_string()]; +let selection = Selection { + patterns: &patterns, + on_missing: MissingFiles::Fail, + case: CaseMode::Insensitive, +}; +archive.extract("out", ExtractionMode::PreserveStructure, &selection)?; +# anyhow::Ok(()) +``` +*/ + +#![warn(missing_docs)] + +mod arcanum; // Arcanum (Troika) DAT format implementation +pub mod archive; // ArchiveFormat and the unified DatArchive interface +pub mod common; // Shared types, archive operations, and path utilities +mod dat1; // Fallout 1 DAT format implementation +mod dat2; // Fallout 2 DAT format implementation +mod lzss; // LZSS decompression for DAT1 files +mod toee; // The Temple of Elemental Evil (Troika) DAT format implementation + +#[cfg(test)] +mod common_tests; +#[cfg(any(test, feature = "test-support"))] +pub mod test_support; // Self-cleaning scratch paths for the test modules + +pub use archive::{ArchiveFormat, DatArchive}; diff --git a/src/lzss.rs b/crates/dat3-core/src/lzss.rs similarity index 100% rename from src/lzss.rs rename to crates/dat3-core/src/lzss.rs diff --git a/src/test_support.rs b/crates/dat3-core/src/test_support.rs similarity index 95% rename from src/test_support.rs rename to crates/dat3-core/src/test_support.rs index 99fcfde..b9fa92b 100644 --- a/src/test_support.rs +++ b/crates/dat3-core/src/test_support.rs @@ -25,6 +25,7 @@ pub fn exact(patterns: &[String], on_missing: MissingFiles) -> Selection<'_> { } /// A temp-directory path that deletes itself when it goes out of scope. +#[derive(Debug)] pub struct ScratchPath(PathBuf); impl ScratchPath { @@ -44,12 +45,17 @@ impl ScratchPath { } /// The same, with the directory created and empty. + #[expect( + clippy::expect_used, + reason = "test-only helper: a failed setup should fail the test that asked for it" + )] pub fn dir(tag: &str) -> Self { let scratch = Self::new(tag); std::fs::create_dir_all(&scratch.0).expect("could not create scratch directory"); scratch } + /// The path itself pub fn path(&self) -> &Path { &self.0 } diff --git a/src/toee.rs b/crates/dat3-core/src/toee.rs similarity index 89% rename from src/toee.rs rename to crates/dat3-core/src/toee.rs index 7fe6362..feb3925 100644 --- a/src/toee.rs +++ b/crates/dat3-core/src/toee.rs @@ -16,6 +16,7 @@ components and every record has parent, first-child, and next-sibling indices. use anyhow::{Context, Result, bail}; use byteorder::{ByteOrder, LittleEndian, ReadBytesExt, WriteBytesExt}; +use std::borrow::Cow; use std::collections::HashMap; use std::io::{Cursor, Read, Write}; use std::path::Path; @@ -139,6 +140,12 @@ pub fn is_toee_format(data: &[u8]) -> bool { expected == Some(table_from_end as u64) } +impl Default for ToeeArchive { + fn default() -> Self { + Self::new() + } +} + impl ToeeArchive { pub fn new() -> Self { Self { @@ -483,8 +490,19 @@ impl ToeeArchive { ) } - pub fn delete_file(&mut self, file_name: &str) -> Result<()> { - common::delete_file_from_list(&mut self.files, file_name) + /// An entry's contents, decompressed + pub fn contents<'a>(&'a self, file: &'a FileEntry) -> Result> { + common::entry_contents(&self.data, file, common::decompress_zlib) + } + + /// Add a prepared entry, replacing the entries of its name as `case` compares + pub fn insert_entry(&mut self, entry: FileEntry, case: CaseMode) { + common::merge_entries(&mut self.files, vec![entry], case); + } + + /// Remove the entry named exactly `file_name`, reporting whether there was one + pub fn remove_entry(&mut self, file_name: &str) -> bool { + common::remove_from_list(&mut self.files, file_name) } /// Add `raw` and each of its ancestors to the node set. @@ -608,7 +626,21 @@ impl ToeeArchive { Ok(nodes) } + /// Save the archive to a ToEE DAT file pub fn save(&self, path: &Path) -> Result<()> { + let nodes = self.prepare_save()?; + utils::write_atomically(path, |out| self.write_prepared(&nodes, out)).context(WRITE_CONTEXT) + } + + /// Write the archive as a ToEE DAT file to `out` + pub fn write_to(&self, out: &mut dyn Write) -> Result<()> { + let nodes = self.prepare_save()?; + self.write_prepared(&nodes, out).context(WRITE_CONTEXT) + } + + /// Checks and the entry tree, built before any output exists so a refused + /// save leaves no temp file + fn prepare_save(&self) -> Result> { let total_payload: u64 = self.files.iter().map(|f| f.packed_size as u64).sum(); if total_payload > u32::MAX as u64 - 4 { bail!("ToEE archive would exceed the format's 4 GiB offset limit"); @@ -618,89 +650,85 @@ impl ToeeArchive { if nodes.len() > i32::MAX as usize { bail!("ToEE archive has too many entries for signed tree indices"); } + Ok(nodes) + } - utils::write_atomically(path, |out| { - let mut file_offsets = vec![0u32; self.files.len()]; - let mut current_offset = 0u32; - for node in &nodes { - let Some(file_index) = node.file_index else { - continue; - }; - let file = &self.files[file_index]; - let bytes = self.read_file_data(file)?; - if bytes.len() != file.packed_size as usize { - bail!("Stored size does not match data for {}", file.name); - } - file_offsets[file_index] = current_offset; - out.write_all(bytes)?; - current_offset += file.packed_size; + fn write_prepared(&self, nodes: &[SaveNode], out: &mut dyn Write) -> Result<()> { + let mut file_offsets = vec![0u32; self.files.len()]; + let mut current_offset = 0u32; + for node in nodes { + let Some(file_index) = node.file_index else { + continue; + }; + let file = &self.files[file_index]; + let bytes = self.read_file_data(file)?; + if bytes.len() != file.packed_size as usize { + bail!("Stored size does not match data for {}", file.name); } + file_offsets[file_index] = current_offset; + out.write_all(bytes)?; + current_offset += file.packed_size; + } - out.write_u32::(current_offset + 4)?; - out.write_u32::(u32::try_from(nodes.len())?)?; - let mut table_size = 4u64; - let mut names_len = 0u64; + out.write_u32::(current_offset + 4)?; + out.write_u32::(u32::try_from(nodes.len())?)?; + let mut table_size = 4u64; + let mut names_len = 0u64; - let link = |index: Option| -> Result { - Ok(index.map(i32::try_from).transpose()?.unwrap_or(-1)) - }; - for node in &nodes { - let mut name_bytes = node.name.as_bytes().to_vec(); - name_bytes.push(0); - out.write_u32::(u32::try_from(name_bytes.len())?)?; - out.write_all(&name_bytes)?; - out.write_u32::(0)?; // original tools wrote an in-memory pointer - - if let Some(file_index) = node.file_index { - let file = &self.files[file_index]; - out.write_u32::(if file.compressed { - FLAG_ZLIB - } else { - FLAG_RAW - })?; - out.write_u32::(file.size)?; - out.write_u32::(file.packed_size)?; - out.write_u32::(file_offsets[file_index])?; - } else { - out.write_u32::(FLAG_DIR)?; - out.write_u32::(0)?; - out.write_u32::(0)?; - out.write_u32::(0)?; - } - out.write_i32::(link(node.parent)?)?; - out.write_i32::(link(node.first_child)?)?; - out.write_i32::(link(node.next_sibling)?)?; + let link = |index: Option| -> Result { + Ok(index.map(i32::try_from).transpose()?.unwrap_or(-1)) + }; + for node in nodes { + let mut name_bytes = node.name.as_bytes().to_vec(); + name_bytes.push(0); + out.write_u32::(u32::try_from(name_bytes.len())?)?; + out.write_all(&name_bytes)?; + out.write_u32::(0)?; // original tools wrote an in-memory pointer - names_len += name_bytes.len() as u64; - table_size += ENTRY_FIXED_BYTES + name_bytes.len() as u64; + if let Some(file_index) = node.file_index { + let file = &self.files[file_index]; + out.write_u32::(if file.compressed { FLAG_ZLIB } else { FLAG_RAW })?; + out.write_u32::(file.size)?; + out.write_u32::(file.packed_size)?; + out.write_u32::(file_offsets[file_index])?; + } else { + out.write_u32::(FLAG_DIR)?; + out.write_u32::(0)?; + out.write_u32::(0)?; + out.write_u32::(0)?; } + out.write_i32::(link(node.parent)?)?; + out.write_i32::(link(node.first_child)?)?; + out.write_i32::(link(node.next_sibling)?)?; - let footer_size = match self.version { - ToeeVersion::V0 => { - out.write_all(&V0_MAGIC)?; - V0_FOOTER_SIZE - } - ToeeVersion::V1 => { - out.write_all(&self.guid)?; - out.write_all(&MAGIC)?; - FOOTER_SIZE - } - }; - out.write_u32::( - u32::try_from(names_len).context("ToEE archive filenames exceed the u32 limit")?, - )?; - out.write_u32::( - u32::try_from(table_size + footer_size as u64) - .context("ToEE entry table exceeds the u32 limit")?, - )?; - Ok(()) - }) - .context("Failed to write ToEE DAT file")?; + names_len += name_bytes.len() as u64; + table_size += ENTRY_FIXED_BYTES + name_bytes.len() as u64; + } + let footer_size = match self.version { + ToeeVersion::V0 => { + out.write_all(&V0_MAGIC)?; + V0_FOOTER_SIZE + } + ToeeVersion::V1 => { + out.write_all(&self.guid)?; + out.write_all(&MAGIC)?; + FOOTER_SIZE + } + }; + out.write_u32::( + u32::try_from(names_len).context("ToEE archive filenames exceed the u32 limit")?, + )?; + out.write_u32::( + u32::try_from(table_size + footer_size as u64) + .context("ToEE entry table exceeds the u32 limit")?, + )?; Ok(()) } } +const WRITE_CONTEXT: &str = "Failed to write ToEE DAT file"; + #[cfg(test)] mod tests { use super::*; @@ -992,7 +1020,7 @@ mod tests { let mut parsed = ToeeArchive::from_bytes(std::fs::read(&first).unwrap()).unwrap(); assert_eq!(parsed.dirs, vec!["gone".to_string(), "keep".to_string()]); - parsed.delete_file("gone\\b.txt").unwrap(); + assert!(parsed.remove_entry("gone\\b.txt")); let second = ScratchPath::new("toee_dirs2"); parsed.save(&second).unwrap(); diff --git a/crates/dat3-wasm/Cargo.toml b/crates/dat3-wasm/Cargo.toml new file mode 100644 index 0000000..2a9c6d9 --- /dev/null +++ b/crates/dat3-wasm/Cargo.toml @@ -0,0 +1,26 @@ +[package] +name = "dat3-wasm" +description = "Read and write Fallout and Troika DAT archives from Node.js and Electron." +version.workspace = true +edition.workspace = true +rust-version.workspace = true +authors.workspace = true +# Same license as the dat3-core library it exposes. +license = "LGPL-3.0-only" +publish.workspace = true + +[lib] +crate-type = ["cdylib"] + +[dependencies] +dat3-core = { path = "../dat3-core" } +# Builds entries() as plain JS objects; released with wasm-bindgen, whose exact +# version it requires. +js-sys = "0.3" +# Exact: the JS glue comes from the wasm-bindgen CLI, which must be this same +# version; install-tools.sh pins it. 0.2.127, not 0.2.128: 0.2.128's macros moved +# to syn 3 while deku's are on syn 2, the conflict behind the workspace's clap cap. +wasm-bindgen = "=0.2.127" + +[lints] +workspace = true diff --git a/crates/dat3-wasm/package.json b/crates/dat3-wasm/package.json new file mode 100644 index 0000000..1f481b2 --- /dev/null +++ b/crates/dat3-wasm/package.json @@ -0,0 +1,13 @@ +{ + "name": "dat3-wasm", + "version": "0.0.0", + "description": "Read and write Fallout and Troika DAT archives from Node.js and Electron.", + "license": "LGPL-3.0-only", + "repository": { + "type": "git", + "url": "git+https://github.com/BGforgeNet/dat3.git" + }, + "main": "dat3_wasm.js", + "types": "dat3_wasm.d.ts", + "files": ["dat3_wasm.js", "dat3_wasm.d.ts", "dat3_wasm_bg.wasm", "dat3_wasm_bg.wasm.d.ts"] +} diff --git a/crates/dat3-wasm/package.sh b/crates/dat3-wasm/package.sh new file mode 100755 index 0000000..6a99a3f --- /dev/null +++ b/crates/dat3-wasm/package.sh @@ -0,0 +1,42 @@ +#!/bin/bash + +set -xeu -o pipefail + +# Builds the dat3-wasm npm package: the crate compiled for wasm32-unknown-unknown, +# wasm-bindgen's nodejs glue, and package.json stamped with the workspace version. +# Leaves the package in target/dat3-wasm/pkg and its tarball at target/dat3-wasm.tgz. + +ROOT="$(cd "$(dirname "$0")/../.." && pwd)" +PKG_DIR="$ROOT/target/dat3-wasm/pkg" +TARBALL="$ROOT/target/dat3-wasm.tgz" + +cd "$ROOT" + +# The glue only works with the crate version it was generated for, so a stale +# or missing CLI fails here rather than producing a package that breaks at load. +wanted="$(sed -n 's/^wasm-bindgen = "=\(.*\)"/\1/p' crates/dat3-wasm/Cargo.toml)" +if [[ "$(wasm-bindgen --version)" != "wasm-bindgen $wanted" ]]; then + echo "Error: wasm-bindgen $wanted is required (./install-tools.sh wasm-bindgen)" >&2 + exit 1 +fi + +cargo build --release --target wasm32-unknown-unknown -p dat3-wasm + +rm -rf "$PKG_DIR" +# --weak-refs frees wasm memory held by an Archive once JS garbage-collects it +wasm-bindgen --target nodejs --weak-refs --out-dir "$PKG_DIR" \ + target/wasm32-unknown-unknown/release/dat3_wasm.wasm + +version="$(sed -n 's/^version = "\(.*\)"/\1/p' Cargo.toml)" +if [ -z "$version" ]; then + echo "Error: no version found in the workspace Cargo.toml" >&2 + exit 1 +fi +cp crates/dat3-wasm/package.json "$PKG_DIR/package.json" +(cd "$PKG_DIR" && npm pkg set "version=$version") + +# npm pack names the tarball after name and version; the release asset keeps one fixed name +pack_dir="$(mktemp -d)" +npm pack "$PKG_DIR" --pack-destination "$pack_dir" +mv "$pack_dir/dat3-wasm-$version.tgz" "$TARBALL" +rmdir "$pack_dir" diff --git a/crates/dat3-wasm/src/lib.rs b/crates/dat3-wasm/src/lib.rs new file mode 100644 index 0000000..096a956 --- /dev/null +++ b/crates/dat3-wasm/src/lib.rs @@ -0,0 +1,130 @@ +/*! +# dat3-wasm + +dat3-core compiled to WebAssembly for Node.js and Electron's main process and +workers. `package.sh` turns the module into an npm package with wasm-bindgen's +`nodejs` glue. + +Everything works on bytes: the JS side reads and writes files. Names use `/` in +both directions (`\` is accepted too), and are looked up regardless of case, +preferring an exact match. +*/ + +use dat3_core::common::{CaseMode, CompressionLevel}; +use dat3_core::{ArchiveFormat, DatArchive}; +use wasm_bindgen::prelude::*; + +/// Level `insert` uses when none is given, the dat3 command line's default +const DEFAULT_COMPRESSION: u8 = 1; + +/// Games look names up regardless of case; `find_stored_index` still prefers +/// the exact name, so names differing only in case stay reachable. +const CASE: CaseMode = CaseMode::Insensitive; + +/// A JS `Error` carrying the whole context chain +fn js_error(error: impl std::fmt::Display) -> JsError { + JsError::new(&format!("{error:#}")) +} + +/// A Fallout (DAT1, DAT2) or Troika (Arcanum, ToEE) archive held in memory +#[wasm_bindgen] +#[derive(Debug)] +pub struct Archive { + inner: DatArchive, +} + +#[wasm_bindgen] +impl Archive { + /// A new, empty archive in `format`: `"dat1"`, `"dat2"`, `"arcanum"` or `"toee"` + #[wasm_bindgen(constructor)] + pub fn new(format: &str) -> Result { + let format = ArchiveFormat::from_arg_name(format).ok_or_else(|| { + js_error(format_args!( + "unsupported archive format {format:?} (expected dat1, dat2, arcanum, or toee)" + )) + })?; + Ok(Self { + inner: DatArchive::new(format), + }) + } + + /// Parse archive bytes, detecting the format + #[wasm_bindgen(js_name = fromBytes)] + pub fn from_bytes(bytes: Vec) -> Result { + let inner = DatArchive::from_bytes(bytes).map_err(js_error)?; + Ok(Self { inner }) + } + + /// The archive's format: `"dat1"`, `"dat2"`, `"arcanum"` or `"toee"` + #[wasm_bindgen(getter)] + pub fn format(&self) -> String { + self.inner.format().arg_name().to_string() + } + + /// Every file entry, names `/`-separated in their stored case. + /// + /// Plain objects rather than wasm-backed classes, so they survive + /// `JSON.stringify` and `postMessage` between Electron processes. + #[wasm_bindgen(unchecked_return_type = "Entry[]")] + pub fn entries(&self) -> Result { + let array = js_sys::Array::new(); + for entry in self.inner.entries() { + let object = js_sys::Object::new(); + let fields: [(&str, JsValue); 4] = [ + ("name", entry.name.replace('\\', "/").into()), + ("size", entry.size.into()), + ("packedSize", entry.packed_size.into()), + ("compressed", entry.compressed.into()), + ]; + for (key, value) in fields { + js_sys::Reflect::set(&object, &key.into(), &value)?; + } + array.push(&object); + } + Ok(array) + } + + /// The decompressed contents of entry `name` + pub fn read(&self, name: &str) -> Result, JsError> { + self.inner.read(name, CASE).map_err(js_error) + } + + /// Add `data` as entry `name`, replacing an entry of that name in any case. + /// `compression` is 0-9 (default 1); DAT1 always stores uncompressed. + pub fn insert( + &mut self, + name: &str, + data: Vec, + compression: Option, + ) -> Result<(), JsError> { + let level = + CompressionLevel::new(compression.unwrap_or(DEFAULT_COMPRESSION)).map_err(js_error)?; + self.inner.insert(name, data, level, CASE).map_err(js_error) + } + + /// Remove entry `name`; returns whether there was one + pub fn remove(&mut self, name: &str) -> Result { + self.inner.remove(name, CASE).map_err(js_error) + } + + /// The archive file's bytes, as dat3 would save it + #[wasm_bindgen(js_name = toBytes)] + pub fn to_bytes(&self) -> Result, JsError> { + self.inner.to_bytes().map_err(js_error) + } +} + +#[wasm_bindgen(typescript_custom_section)] +const ENTRY_TYPE: &str = r#" +/** One file entry's name and sizes */ +export interface Entry { + /** `/`-separated name in its stored case */ + name: string; + /** Size of the contents in bytes */ + size: number; + /** Size as stored in the archive */ + packedSize: number; + /** Whether the stored data is compressed */ + compressed: boolean; +} +"#; diff --git a/crates/dat3/Cargo.toml b/crates/dat3/Cargo.toml new file mode 100644 index 0000000..bea1071 --- /dev/null +++ b/crates/dat3/Cargo.toml @@ -0,0 +1,33 @@ +[package] +name = "fallout-dat3" +description = "Cross-platform CLI for Fallout and Troika DAT archives." +version.workspace = true +edition.workspace = true +rust-version.workspace = true +authors.workspace = true +license.workspace = true +publish.workspace = true + +[[bin]] +name = "dat3" +path = "src/main.rs" + +[dependencies] +anyhow.workspace = true +clap.workspace = true +dat3-core = { path = "../dat3-core", features = ["clap"] } + +# Reads .bgforge.yml for the default-format setting. Default features off: +# they only add encoding_rs for non-UTF-8 input detection, which a config +# file doesn't need (and whose BSD-3-Clause part the license gate rejects). +yaml-rust2 = { version = "0.12", default-features = false } + +# Optional: Use mimalloc on Linux for better performance +[target.'cfg(target_os = "linux")'.dependencies] +mimalloc = "0.1" + +[dev-dependencies] +dat3-core = { path = "../dat3-core", features = ["clap", "test-support"] } + +[lints] +workspace = true diff --git a/src/config.rs b/crates/dat3/src/config.rs similarity index 74% rename from src/config.rs rename to crates/dat3/src/config.rs index 12244ff..a5106f5 100644 --- a/src/config.rs +++ b/crates/dat3/src/config.rs @@ -4,8 +4,7 @@ Optional `.bgforge.yml` in the current directory. Only one key is read: ```yaml -dat3: - default_format: arcanum +dat3.default_format: arcanum ``` It sets the format used when `a` creates a new archive and no `--format` @@ -15,7 +14,7 @@ the built-in default, so a foreign or broken config never blocks the tool. use std::path::Path; -use crate::archive::ArchiveFormat; +use dat3_core::ArchiveFormat; /// Config file name, looked up in the process working directory pub const CONFIG_FILE: &str = ".bgforge.yml"; @@ -53,16 +52,13 @@ fn parse_default_format(text: &str) -> Result, String> { return Ok(None); }; - let value = &doc["dat3"]["default_format"]; + // One flat top-level key; a nested `dat3:` mapping is not read + let value = &doc["dat3.default_format"]; match value { yaml_rust2::Yaml::BadValue => Ok(None), - yaml_rust2::Yaml::String(s) => ::from_str(s, false) - .map(Some) - .map_err(|_| { - format!( - "unsupported dat3.default_format {s:?} (expected dat1, dat2, arcanum, or toee)" - ) - }), + yaml_rust2::Yaml::String(s) => ArchiveFormat::from_arg_name(s).map(Some).ok_or_else(|| { + format!("unsupported dat3.default_format {s:?} (expected dat1, dat2, arcanum, or toee)") + }), other @ (yaml_rust2::Yaml::Real(_) | yaml_rust2::Yaml::Integer(_) | yaml_rust2::Yaml::Boolean(_) @@ -87,7 +83,7 @@ mod tests { ("arcanum", ArchiveFormat::Arcanum), ("toee", ArchiveFormat::Toee), ] { - let text = format!("dat3:\n default_format: {name}\n"); + let text = format!("dat3.default_format: {name}\n"); assert_eq!(parse_default_format(&text), Ok(Some(expected))); } } @@ -95,37 +91,45 @@ mod tests { #[test] fn ignores_missing_key_and_unrelated_content() { assert_eq!(parse_default_format(""), Ok(None)); - assert_eq!(parse_default_format("other_tool:\n key: 1\n"), Ok(None)); - assert_eq!(parse_default_format("dat3:\n other: x\n"), Ok(None)); + assert_eq!(parse_default_format("other_tool.key: 1\n"), Ok(None)); + assert_eq!(parse_default_format("dat3.other: x\n"), Ok(None)); + } + + #[test] + fn ignores_the_nested_form() { + assert_eq!( + parse_default_format("dat3:\n default_format: arcanum\n"), + Ok(None) + ); } #[test] fn warns_on_unsupported_value() { - let err = parse_default_format("dat3:\n default_format: zip\n").unwrap_err(); + let err = parse_default_format("dat3.default_format: zip\n").unwrap_err(); assert!(err.contains("unsupported"), "got: {err}"); assert!(err.contains("zip"), "got: {err}"); } #[test] fn warns_on_non_string_value() { - let err = parse_default_format("dat3:\n default_format: 2\n").unwrap_err(); + let err = parse_default_format("dat3.default_format: 2\n").unwrap_err(); assert!(err.contains("must be a string"), "got: {err}"); } #[test] fn warns_on_invalid_yaml() { - let err = parse_default_format("dat3: [unclosed\n").unwrap_err(); + let err = parse_default_format("dat3.default_format: [unclosed\n").unwrap_err(); assert!(err.contains("not valid YAML"), "got: {err}"); } #[test] fn missing_file_is_silent_none() { - let dir = crate::test_support::ScratchPath::new("cfg"); + let dir = dat3_core::test_support::ScratchPath::new("cfg"); let _ = std::fs::remove_dir_all(&dir); std::fs::create_dir_all(&dir).unwrap(); assert_eq!(default_format(&dir), None); - std::fs::write(dir.join(CONFIG_FILE), "dat3:\n default_format: arcanum\n").unwrap(); + std::fs::write(dir.join(CONFIG_FILE), "dat3.default_format: arcanum\n").unwrap(); assert_eq!(default_format(&dir), Some(ArchiveFormat::Arcanum)); std::fs::remove_dir_all(&dir).unwrap(); } diff --git a/src/main.rs b/crates/dat3/src/main.rs similarity index 86% rename from src/main.rs rename to crates/dat3/src/main.rs index 281511c..cb64c08 100644 --- a/src/main.rs +++ b/crates/dat3/src/main.rs @@ -13,24 +13,12 @@ use std::path::{Path, PathBuf}; #[global_allocator] static GLOBAL: mimalloc::MiMalloc = mimalloc::MiMalloc; -mod arcanum; // Arcanum (Troika) DAT format implementation -mod archive; // ArchiveFormat and the unified DatArchive interface -mod common; // Shared types, archive operations, and path utilities mod config; // Optional .bgforge.yml defaults -mod dat1; // Fallout 1 DAT format implementation -mod dat2; // Fallout 2 DAT format implementation -mod lzss; // LZSS decompression for DAT1 files -mod toee; // The Temple of Elemental Evil (Troika) DAT format implementation -#[cfg(test)] -mod common_tests; -#[cfg(test)] -mod test_support; // Self-cleaning scratch paths for the test modules - -use archive::{ArchiveFormat, DatArchive}; -use common::{ - CaseMode, CompressionLevel, ExtractionMode, ListFormat, MissingFiles, Selection, utils, +use dat3_core::common::{ + self, CaseMode, CompressionLevel, ExtractionMode, ListFormat, MissingFiles, Selection, utils, }; +use dat3_core::{ArchiveFormat, DatArchive}; /// Command-line interface definition. /// The `clap` crate uses these derive macros to automatically parse arguments. @@ -309,3 +297,46 @@ fn main() -> Result<()> { Ok(()) } + +#[cfg(test)] +mod cli_args { + use clap::Parser; + + #[test] + fn rejects_out_of_range_compression_at_parse_time() { + let result = crate::Cli::try_parse_from(["dat3", "a", "test.dat", "-c", "10", "file"]); + assert!( + result.is_err(), + "compression level 10 should be rejected during argument parsing" + ); + } + + #[test] + fn accepts_maximum_compression_level() { + let result = crate::Cli::try_parse_from(["dat3", "a", "test.dat", "-c", "9", "file"]); + assert!(result.is_ok()); + } + + #[test] + fn accepts_each_archive_format() { + for format in ["dat1", "dat2", "arcanum", "toee"] { + let result = + crate::Cli::try_parse_from(["dat3", "a", "test.dat", "--format", format, "file"]); + assert!(result.is_ok(), "--format {format} should parse"); + } + } + + #[test] + fn rejects_unknown_format_and_removed_format_flags() { + for args in [ + ["dat3", "a", "test.dat", "--format", "zip", "file"].as_slice(), + ["dat3", "a", "test.dat", "--dat1", "file"].as_slice(), + ["dat3", "a", "test.dat", "--arcanum", "file"].as_slice(), + ] { + assert!( + crate::Cli::try_parse_from(args.iter().copied()).is_err(), + "{args:?} should be rejected" + ); + } + } +} diff --git a/deny.toml b/deny.toml index 636a90a..9ea13f6 100644 --- a/deny.toml +++ b/deny.toml @@ -9,9 +9,10 @@ [licenses] # A license that stops being needed fails the build rather than lingering here. unused-allowed-license = "deny" -# GPL-3.0 compatible licenses actually used by dependencies. +# The workspace crates' own licenses, and GPL-3.0 compatible licenses actually used by dependencies. allow = [ "GPL-3.0-only", + "LGPL-3.0-only", "MIT", "Apache-2.0", "Unlicense", diff --git a/docs/api.md b/docs/api.md new file mode 100644 index 0000000..ac8492b --- /dev/null +++ b/docs/api.md @@ -0,0 +1,44 @@ +# Using dat3-core as a library + +The archive code is the `dat3-core` crate in this repository. It is not published on crates.io; depend on it +through git, pinned to a release tag: + +```toml +[dependencies] +dat3-core = { git = "https://github.com/BGforgeNet/dat3", tag = "" } +``` + +Its API may still change between releases. The optional `clap` feature derives `clap::ValueEnum` on +`ArchiveFormat`. The API documentation, with an example, builds with `cargo doc --no-deps --package dat3-core --open` +in a checkout, or with `cargo doc --open` in a project that depends on it. + +## From Node.js and Electron + +Releases also ship `dat3-wasm.tgz`, an npm package of the same library compiled to WebAssembly. It runs in Node and in +Electron's main process and workers on every platform, with TypeScript types included. Install it from the release: + +```bash +npm install https://github.com/BGforgeNet/dat3/releases/download//dat3-wasm.tgz +``` + +It works on bytes, so the application reads and writes the files: + +```ts +import { readFileSync, writeFileSync } from "node:fs"; +import { Archive, type Entry } from "dat3-wasm"; + +const archive = Archive.fromBytes(readFileSync("patch000.dat")); +const entries: Entry[] = archive.entries(); +for (const entry of entries) { + console.log(entry.name, entry.size, entry.packedSize, entry.compressed); // names use "/" +} +const frm: Uint8Array = archive.read("art/critters/haenroaa.frm"); // any letter case +archive.insert("text/english/game/new.msg", readFileSync("new.msg"), 9); // compression 0-9 +archive.remove("data/old.txt"); +writeFileSync("patch000.dat", archive.toBytes()); +archive.free(); // releases the archive's memory now rather than at garbage collection +``` + +`new Archive("dat2")` starts an empty archive (`"dat1"`, `"dat2"`, `"arcanum"` or `"toee"`). Names are looked up +regardless of case, preferring an exact match. Failures throw an `Error` with the reason. Calls run on the calling +thread, so run long operations on large archives in a worker to keep a window responsive. diff --git a/docs/building.md b/docs/building.md new file mode 100644 index 0000000..81c5529 --- /dev/null +++ b/docs/building.md @@ -0,0 +1,32 @@ +# Building dat3 + +## Requirements + +- Rust 1.87 or newer +- Target-specific toolchains (install as needed) +- `./install-tools.sh` for the pinned tooling, including [Zig](https://ziglang.org/), which the aarch64 + target needs: mimalloc is C, and no aarch64-musl C compiler is packaged for common distros. It also links the + macOS build +- Node 24 or newer, to run the integration suite (`./test.sh`): its helpers under `tests/` are TypeScript, run + by Node's own type stripping. `npm ci && npm run typecheck` typechecks them. Neither is needed to build dat3 + +## Build + +```bash +./build.sh +``` + +Builds are static, except for macOS, where every program links the system library dynamically. + +Binaries will be at: + +```bash +target/x86_64-unknown-linux-musl/release/dat3 +target/aarch64-unknown-linux-musl/release/dat3 +target/universal2-apple-darwin/release/dat3 +target/x86_64-pc-windows-gnu/release/dat3.exe +target/i686-pc-windows-gnu/release/dat3.exe +target/wasm32-wasip1/release/dat3.wasm +``` + +The npm package for Node and Electron is at `target/dat3-wasm.tgz`; `crates/dat3-wasm/package.sh` builds it alone. diff --git a/install-tools.sh b/install-tools.sh index f033457..a466ac9 100755 --- a/install-tools.sh +++ b/install-tools.sh @@ -13,7 +13,7 @@ set -xeu -o pipefail BIN_DIR="$HOME/.cargo/bin" -ALL_TOOLS=(actionlint cargo-deny cargo-machete cargo-zigbuild shellcheck shfmt wasmtime zig zizmor) +ALL_TOOLS=(actionlint cargo-deny cargo-machete cargo-zigbuild shellcheck shfmt wasm-bindgen wasmtime zig zizmor) # zig lives as a whole tree; only a symlink to it goes in BIN_DIR ZIG_DIR="$HOME/.local/share/zig" @@ -49,6 +49,11 @@ ZIGBUILD_SHA256="9e3cf73485edbd45905c8aadbc0fdf869c7ddc3848f0c898229f2680db52e44 WASMTIME_VERSION="47.0.4" WASMTIME_SHA256="446e8641ba372333670ba0373d5d3083e5cf0dd001b66088afbb3983db0f768f" +# Must equal the wasm-bindgen version crates/dat3-wasm pins: the glue it +# generates only works with the matching crate. +WASM_BINDGEN_VERSION="0.2.127" +WASM_BINDGEN_SHA256="61d4a7dc85acfa0d2354ccc0b8361928c7e52a746d17f28ebaa795ed3dc1614a" + # Digest as published in ziglang.org's download index ZIG_VERSION="0.16.0" ZIG_SHA256="70e49664a74374b48b51e6f3fdfbf437f6395d42509050588bd49abe52ba3d00" @@ -99,6 +104,12 @@ tool_spec() { "$ZIGBUILD_SHA256" \ "cargo-zigbuild-x86_64-unknown-linux-musl/cargo-zigbuild" ;; + wasm-bindgen) + printf '%s|%s|%s|%s' "$WASM_BINDGEN_VERSION" \ + "https://github.com/wasm-bindgen/wasm-bindgen/releases/download/${WASM_BINDGEN_VERSION}/wasm-bindgen-${WASM_BINDGEN_VERSION}-x86_64-unknown-linux-musl.tar.gz" \ + "$WASM_BINDGEN_SHA256" \ + "wasm-bindgen-${WASM_BINDGEN_VERSION}-x86_64-unknown-linux-musl/wasm-bindgen" + ;; wasmtime) printf '%s|%s|%s|%s' "$WASMTIME_VERSION" \ "https://github.com/bytecodealliance/wasmtime/releases/download/v${WASMTIME_VERSION}/wasmtime-v${WASMTIME_VERSION}-x86_64-linux.tar.xz" \ diff --git a/package-lock.json b/package-lock.json index 0387f99..8ac144e 100644 --- a/package-lock.json +++ b/package-lock.json @@ -9,7 +9,7 @@ "version": "0.0.0", "devDependencies": { "@types/node": "24.13.3", - "typescript": "5.9.3" + "typescript": "6.0.3" } }, "node_modules/@types/node": { @@ -23,9 +23,9 @@ } }, "node_modules/typescript": { - "version": "5.9.3", - "resolved": "https://registry.npmjs.org/typescript/-/typescript-5.9.3.tgz", - "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", + "version": "6.0.3", + "resolved": "https://registry.npmjs.org/typescript/-/typescript-6.0.3.tgz", + "integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==", "dev": true, "license": "Apache-2.0", "bin": { diff --git a/package.json b/package.json index 830c56a..c755f73 100644 --- a/package.json +++ b/package.json @@ -9,6 +9,6 @@ }, "devDependencies": { "@types/node": "24.13.3", - "typescript": "5.9.3" + "typescript": "6.0.3" } } diff --git a/src/archive.rs b/src/archive.rs deleted file mode 100644 index edd9abe..0000000 --- a/src/archive.rs +++ /dev/null @@ -1,473 +0,0 @@ -/*! -# Archive formats - -`ArchiveFormat` names the supported formats, and `DatArchive` wraps their -implementations behind one interface so callers don't need to know which -format they're working with. Kept out of `common`, which the format modules -build on, so module dependencies run one way. -*/ - -use anyhow::{Context, Result}; -use clap::ValueEnum; -use std::fs; -use std::path::Path; - -use crate::arcanum::{self, ArcanumArchive}; -use crate::common::{self, CaseMode, CompressionLevel, ExtractionMode, ListFormat, Selection}; -use crate::dat1::Dat1Archive; -use crate::dat2::Dat2Archive; -use crate::toee::{self, ToeeArchive}; - -// DAT1 format detection: big-endian header, no signature to key on -const DAT1_MAX_DIRECTORIES: u32 = 1000; - -/// A supported archive format, as selected by `a --format` or `.bgforge.yml` -#[derive(Debug, Clone, Copy, PartialEq, Eq, ValueEnum)] -pub enum ArchiveFormat { - /// Fallout 1 (big-endian, LZSS; created uncompressed) - Dat1, - /// Fallout 2 (little-endian, zlib) - the default for new archives - Dat2, - /// Arcanum (little-endian, zlib) - Arcanum, - /// The Temple of Elemental Evil (hierarchical Troika DAT, zlib) - Toee, -} - -impl ArchiveFormat { - /// The value as typed on the command line, for error messages - pub fn arg_name(self) -> &'static str { - match self { - Self::Dat1 => "dat1", - Self::Dat2 => "dat2", - Self::Arcanum => "arcanum", - Self::Toee => "toee", - } - } - - /// Human-readable format name for messages - pub fn display_name(self) -> &'static str { - match self { - Self::Dat1 => "DAT1", - Self::Dat2 => "DAT2", - Self::Arcanum => "Arcanum", - Self::Toee => "ToEE", - } - } -} - -/// Unified interface for DAT1, DAT2, Arcanum, and ToEE archives. -/// -/// Uses an enum instead of trait objects because the set of known formats -/// is small and fixed - this gives us static dispatch, exhaustive matching, -/// and no heap allocation for the wrapper. -/// -/// **Memory**: The entire archive is loaded into memory on open, whatever the -/// format. Fallout archives typically stay under ~200MB; retail Arcanum -/// archives run considerably larger and are held in RAM the same way. -/// -/// ```ignore -/// let archive = DatArchive::open("master.dat")?; // auto-detects format -/// let dat1 = DatArchive::new(ArchiveFormat::Dat1); // create new DAT1 -/// ``` -pub enum DatArchive { - /// Fallout 1 format (big-endian, hierarchical dirs, LZSS compression) - Dat1(Dat1Archive), - /// Fallout 2 format (little-endian, flat file list, zlib compression) - Dat2(Dat2Archive), - /// Arcanum format (little-endian, flat entry table, zlib compression) - Arcanum(ArcanumArchive), - /// ToEE format (little-endian, hierarchical entry table, zlib compression) - Toee(ToeeArchive), -} - -impl DatArchive { - /// Open an existing DAT archive, auto-detecting the format - pub fn open>(path: P) -> Result { - let data = fs::read(&path) - .with_context(|| format!("Failed to read DAT file: {}", path.as_ref().display()))?; - - // Troika formats share a real magic and go first. ToEE's hierarchical - // entry-table size distinguishes it from Arcanum's flat table. DAT1 is - // a header heuristic and DAT2 (no signature at all) is the fallback. - if toee::is_toee_format(&data) { - Ok(Self::Toee(ToeeArchive::from_bytes(data)?)) - } else if arcanum::is_arcanum_format(&data) { - Ok(Self::Arcanum(ArcanumArchive::from_bytes(data)?)) - } else if Self::is_dat1_format(&data) { - Ok(Self::Dat1(Dat1Archive::from_bytes(data)?)) - } else { - Ok(Self::Dat2(Dat2Archive::from_bytes(data)?)) - } - } - - /// Create a new empty archive of the given format - pub fn new(format: ArchiveFormat) -> Self { - match format { - ArchiveFormat::Dat1 => Self::Dat1(Dat1Archive::new()), - ArchiveFormat::Dat2 => Self::Dat2(Dat2Archive::new()), - ArchiveFormat::Arcanum => Self::Arcanum(ArcanumArchive::new()), - ArchiveFormat::Toee => Self::Toee(ToeeArchive::new()), - } - } - - /// The format this archive is stored in - pub fn format(&self) -> ArchiveFormat { - match self { - Self::Dat1(_) => ArchiveFormat::Dat1, - Self::Dat2(_) => ArchiveFormat::Dat2, - Self::Arcanum(_) => ArchiveFormat::Arcanum, - Self::Toee(_) => ArchiveFormat::Toee, - } - } - - /// Detect DAT1 format by examining the big-endian header. - /// - /// The header opens with a directory count and the engine's allocation hint - /// for that directory list, which is never below the count. Both are checked: - /// DAT2 carries no signature at all and is the fallback, so this heuristic is - /// what keeps a DAT2 archive from being parsed as DAT1. - /// - /// The second field is NOT a format identifier, despite reading like one in - /// the shipped archives - `critter.dat` carries 10 and `master.dat` 94, which - /// are simply their own hints. Matching those two values exactly rejects every - /// other real DAT1 archive, including the Fallout 1 demo's (hint 46). - fn is_dat1_format(data: &[u8]) -> bool { - let Some(header) = data.get(..16) else { - return false; - }; - let field = - |i: usize| u32::from_be_bytes([header[i], header[i + 1], header[i + 2], header[i + 3]]); - let dir_count = field(0); - let allocation_hint = field(4); - - dir_count > 0 - && dir_count < DAT1_MAX_DIRECTORIES - && allocation_hint >= dir_count - && allocation_hint < DAT1_MAX_DIRECTORIES - } - - /// List files in the archive (all or filtered by patterns) - pub fn list(&self, selection: &Selection, format: ListFormat) -> Result<()> { - match self { - Self::Dat1(a) => a.list(selection, format), - Self::Dat2(a) => a.list(selection, format), - Self::Arcanum(a) => a.list(selection, format), - Self::Toee(a) => a.list(selection, format), - } - } - - /// Extract files from the archive - pub fn extract>( - &self, - output_dir: P, - mode: ExtractionMode, - selection: &Selection, - ) -> Result<()> { - let output_dir = output_dir.as_ref(); - match self { - Self::Dat1(a) => a.extract(output_dir, mode, selection), - Self::Dat2(a) => a.extract(output_dir, mode, selection), - Self::Arcanum(a) => a.extract(output_dir, mode, selection), - Self::Toee(a) => a.extract(output_dir, mode, selection), - } - } - - /// Add a file to the archive (directories are processed recursively) - pub fn add_file>( - &mut self, - file_path: P, - compression: CompressionLevel, - target_dir: Option<&str>, - source_root: Option<&Path>, - case: CaseMode, - ) -> Result<()> { - let path = file_path.as_ref(); - match self { - Self::Dat1(a) => a.add_file(path, compression, target_dir, source_root, case), - Self::Dat2(a) => a.add_file(path, compression, target_dir, source_root, case), - Self::Arcanum(a) => a.add_file(path, compression, target_dir, source_root, case), - Self::Toee(a) => a.add_file(path, compression, target_dir, source_root, case), - } - } - - /// Names of every file entry in the archive - pub fn entry_names(&self) -> Vec { - let entries = match self { - Self::Dat1(a) => a.entries(), - Self::Dat2(a) => a.entries(), - Self::Arcanum(a) => a.entries(), - Self::Toee(a) => a.entries(), - }; - entries - .into_iter() - .map(|entry| entry.name.clone()) - .collect() - } - - /// Delete every entry the `d` operands select (see `resolve_delete_targets`) - pub fn delete(&mut self, patterns: &[String], case: CaseMode) -> Result<()> { - let names = self.entry_names(); - let names: Vec<&str> = names.iter().map(String::as_str).collect(); - for target in common::resolve_delete_targets(&names, patterns, case)? { - self.delete_file(&target)?; - } - Ok(()) - } - - /// Delete a file from the archive - pub fn delete_file(&mut self, file_name: &str) -> Result<()> { - match self { - Self::Dat1(a) => a.delete_file(file_name), - Self::Dat2(a) => a.delete_file(file_name), - Self::Arcanum(a) => a.delete_file(file_name), - Self::Toee(a) => a.delete_file(file_name), - } - } - - /// Save the archive to a file - pub fn save>(&self, path: P) -> Result<()> { - match self { - Self::Dat1(a) => a.save(path.as_ref()), - Self::Dat2(a) => a.save(path.as_ref()), - Self::Arcanum(a) => a.save(path.as_ref()), - Self::Toee(a) => a.save(path.as_ref()), - } - } -} - -#[cfg(test)] -mod tests { - use super::*; - use crate::test_support::ScratchPath; - - const ALL_FORMATS: [ArchiveFormat; 4] = [ - ArchiveFormat::Dat1, - ArchiveFormat::Dat2, - ArchiveFormat::Arcanum, - ArchiveFormat::Toee, - ]; - - /// A new archive holding one small file per name (`/`-separated), stored in - /// exactly the case given - fn archive_with(format: ArchiveFormat, names: &[&str]) -> DatArchive { - let mut archive = DatArchive::new(format); - add_names(&mut archive, names, CaseMode::Sensitive); - archive - } - - /// Add one small file per name, as `a` would under `case` - fn add_names(archive: &mut DatArchive, names: &[&str], case: CaseMode) { - let src = ScratchPath::dir("archive_names_src"); - for name in names { - let path = src.join(name); - std::fs::create_dir_all(path.parent().unwrap()).unwrap(); - std::fs::write(&path, name.as_bytes()).unwrap(); - archive - .add_file( - &path, - CompressionLevel::new(0).unwrap(), - None, - Some(&src), - case, - ) - .unwrap(); - } - } - - /// Save and reopen, so assertions see what the format actually stored - fn reopened(archive: &DatArchive) -> DatArchive { - let path = ScratchPath::new("archive_case_roundtrip"); - archive.save(&path).unwrap(); - DatArchive::open(&path).unwrap() - } - - fn all_entries(case: CaseMode) -> Selection<'static> { - Selection { - patterns: &[], - on_missing: common::MissingFiles::Fail, - case, - } - } - - #[test] - fn by_default_new_entries_are_stored_in_lowercase() { - for format in ALL_FORMATS { - let mut archive = DatArchive::new(format); - add_names(&mut archive, &["Art/Hero.FRM"], CaseMode::Insensitive); - assert_eq!( - sorted_names(&reopened(&archive)), - ["art\\hero.frm"], - "{format:?}" - ); - } - } - - #[test] - fn with_case_sensitive_new_entries_keep_their_case() { - for format in ALL_FORMATS { - let archive = archive_with(format, &["Art/Hero.FRM"]); - assert_eq!( - sorted_names(&reopened(&archive)), - ["Art\\Hero.FRM"], - "{format:?}" - ); - } - } - - #[test] - fn by_default_extraction_writes_lowercase_paths() { - for format in ALL_FORMATS { - let archive = reopened(&archive_with(format, &["ART/HERO.FRM"])); - let out = ScratchPath::dir("archive_case_extract"); - archive - .extract( - &out, - ExtractionMode::PreserveStructure, - &all_entries(CaseMode::Insensitive), - ) - .unwrap(); - assert!(out.join("art").join("hero.frm").is_file(), "{format:?}"); - assert!(!out.join("ART").exists(), "{format:?}"); - } - } - - #[test] - fn with_case_sensitive_extraction_keeps_stored_paths() { - for format in ALL_FORMATS { - let archive = reopened(&archive_with(format, &["ART/HERO.FRM"])); - let out = ScratchPath::dir("archive_case_extract_exact"); - archive - .extract( - &out, - ExtractionMode::PreserveStructure, - &all_entries(CaseMode::Sensitive), - ) - .unwrap(); - assert!(out.join("ART").join("HERO.FRM").is_file(), "{format:?}"); - } - } - - /// Re-adding a file under another case replaces the entry rather than - /// keeping two names that differ only in case. - #[test] - fn by_default_adding_replaces_an_entry_differing_only_in_case() { - for format in ALL_FORMATS { - let mut archive = reopened(&archive_with(format, &["ART/HERO.FRM", "ART/OTHER.FRM"])); - add_names(&mut archive, &["art/hero.frm"], CaseMode::Insensitive); - // DAT1 and ToEE keep one stored spelling per directory, so the new - // lowercase file sits under the existing ART; the flat formats store - // the whole lowercase path. - let expected = match format { - ArchiveFormat::Dat1 | ArchiveFormat::Toee => ["ART\\OTHER.FRM", "ART\\hero.frm"], - ArchiveFormat::Dat2 | ArchiveFormat::Arcanum => ["ART\\OTHER.FRM", "art\\hero.frm"], - }; - assert_eq!(sorted_names(&reopened(&archive)), expected, "{format:?}"); - } - } - - #[test] - fn by_default_delete_ignores_case() { - for format in ALL_FORMATS { - let mut archive = reopened(&archive_with(format, &["ART/HERO.FRM", "ART/OTHER.FRM"])); - archive - .delete(&["art/hero.frm".to_string()], CaseMode::Insensitive) - .unwrap(); - assert_eq!(sorted_names(&archive), ["ART\\OTHER.FRM"], "{format:?}"); - } - } - - #[test] - fn with_case_sensitive_delete_needs_the_exact_case() { - for format in ALL_FORMATS { - let mut archive = reopened(&archive_with(format, &["ART/HERO.FRM"])); - assert!( - archive - .delete(&["art/hero.frm".to_string()], CaseMode::Sensitive) - .is_err(), - "{format:?}" - ); - } - } - - /// DAT1 stores one name per directory, so a file added under a differently - /// cased directory joins the stored one instead of creating a second. - #[test] - fn dat1_adding_into_a_directory_stored_in_another_case_round_trips() { - let mut archive = reopened(&archive_with(ArchiveFormat::Dat1, &["ART/HERO.FRM"])); - add_names(&mut archive, &["art/new.frm"], CaseMode::Insensitive); - let archive = reopened(&archive); - assert_eq!(sorted_names(&archive), ["ART\\HERO.FRM", "ART\\new.frm"]); - - let out = ScratchPath::dir("dat1_case_dir"); - archive - .extract( - &out, - ExtractionMode::PreserveStructure, - &all_entries(CaseMode::Insensitive), - ) - .unwrap(); - assert_eq!( - std::fs::read(out.join("art").join("new.frm")).unwrap(), - b"art/new.frm" - ); - } - - fn sorted_names(archive: &DatArchive) -> Vec { - let mut names = archive.entry_names(); - names.sort(); - names - } - - #[test] - fn a_glob_deletes_every_matching_entry_and_nothing_else() { - for format in ALL_FORMATS { - let mut archive = archive_with( - format, - &["ART/A.FRM", "ART/B.FRM", "ART/A.TXT", "TEXT/A.FRM"], - ); - archive - .delete(&["ART/*.FRM".to_string()], CaseMode::Sensitive) - .unwrap(); - assert_eq!( - sorted_names(&archive), - ["ART\\A.TXT", "TEXT\\A.FRM"], - "{format:?}" - ); - } - } - - #[test] - fn a_plain_name_deletes_only_the_entry_with_that_exact_name() { - for format in ALL_FORMATS { - let mut archive = archive_with(format, &["DATA/A.TXT", "DATA/BIGA.TXT"]); - archive - .delete(&["DATA/A.TXT".to_string()], CaseMode::Sensitive) - .unwrap(); - assert_eq!(sorted_names(&archive), ["DATA\\BIGA.TXT"], "{format:?}"); - } - } - - #[test] - fn an_unmatched_pattern_fails_before_anything_is_deleted() { - for format in ALL_FORMATS { - let mut archive = archive_with(format, &["DATA/A.TXT", "DATA/B.TXT"]); - let err = archive - .delete( - &["DATA/A.TXT".to_string(), "*.ZZZ".to_string()], - CaseMode::Sensitive, - ) - .unwrap_err(); - assert_eq!( - err.to_string(), - "Some requested files were not found", - "{format:?}" - ); - assert_eq!( - sorted_names(&archive), - ["DATA\\A.TXT", "DATA\\B.TXT"], - "{format:?}" - ); - } - } -} diff --git a/test.sh b/test.sh index de4ef33..832e066 100755 --- a/test.sh +++ b/test.sh @@ -48,9 +48,14 @@ cd tests ./extract_missing.sh ./default_format.sh ./case_handling.sh +./readme_help.sh # TypeScript: the assertions are about a parsed document, so a real parser runs them node ./json_listing.ts +# The npm package for Node and Electron, installed as an application would and +# checked against the native build on the fixtures fetched above +run_if_available wasm-bindgen ./wasm_library.sh + # The WebAssembly build, run under a WASI runtime, and the arm64 build under # qemu. The arm64 binary comes from build.sh, which needs zig to produce it. run_if_available wasmtime ./wasm.sh diff --git a/tests/default_format.sh b/tests/default_format.sh index 970d87d..a71a63b 100755 --- a/tests/default_format.sh +++ b/tests/default_format.sh @@ -31,7 +31,7 @@ is_dat2() { [ "$size" -eq "$stored" ] } -printf 'dat3:\n default_format: arcanum\n' >.bgforge.yml +printf 'dat3.default_format: arcanum\n' >.bgforge.yml "$DAT3" a from_config.dat file.txt if ! is_arcanum from_config.dat; then @@ -45,7 +45,7 @@ if is_arcanum from_flag.dat || ! is_dat2 from_flag.dat; then exit 1 fi -printf 'dat3:\n default_format: zip\n' >.bgforge.yml +printf 'dat3.default_format: zip\n' >.bgforge.yml warning=$("$DAT3" a from_bad_config.dat file.txt 2>&1 >/dev/null) if [[ "$warning" != *'unsupported dat3.default_format "zip"'* ]]; then echo "Error: an unsupported default_format was not reported: $warning" diff --git a/tests/macos.sh b/tests/macos.sh new file mode 100755 index 0000000..3e44db3 --- /dev/null +++ b/tests/macos.sh @@ -0,0 +1,52 @@ +#!/bin/bash + +set -xeu -o pipefail + +# Smoke-test the macOS universal binary. Nothing on Linux can run it, so CI runs +# this on an Intel and an Apple Silicon runner, each exercising its own slice. +# MACOS_DAT3 is the binary to test. + +# Absolute, and resolved before any cd, so a path relative to the caller works +MACOS_BIN="$(cd "$(dirname "${MACOS_DAT3:?set MACOS_DAT3 to the binary to test}")" && pwd)/$(basename "$MACOS_DAT3")" + +# Work inside tests directory +cd "$(dirname "$0")" + +TEST_DIR="test_macos" +ARCHIVE="test.dat" + +# Without a native slice an Apple Silicon runner with Rosetta would run the +# x86_64 one instead and still pass +ARCH="$(uname -m)" +if ! lipo "$MACOS_BIN" -verify_arch "$ARCH"; then + echo "Error: $MACOS_BIN has no $ARCH slice" >&2 + exit 1 +fi + +rm -rf "$TEST_DIR" +mkdir -p "$TEST_DIR/data/sub" +cd "$TEST_DIR" + +echo "hello macos" >data/a.txt +seq 3000 | awk '{print "compressible line"}' >data/sub/text.txt + +# Write an archive and read it back: the tree must come out unchanged +"$MACOS_BIN" a -c 9 "$ARCHIVE" data +"$MACOS_BIN" x "$ARCHIVE" -o out +diff -r data out/data + +# Deleting rewrites the archive, so it exercises the save path too +"$MACOS_BIN" d "$ARCHIVE" "data/a.txt" +rm -rf out +"$MACOS_BIN" x "$ARCHIVE" -o out +if [ -e "out/data/a.txt" ]; then + echo "Error: deleted file still present in archive" + exit 1 +fi +diff data/sub/text.txt out/data/sub/text.txt + +# Clean up +cd .. +rm -rf "$TEST_DIR" + +echo "macOS test passed" diff --git a/tests/readme_help.sh b/tests/readme_help.sh new file mode 100755 index 0000000..b347532 --- /dev/null +++ b/tests/readme_help.sh @@ -0,0 +1,28 @@ +#!/bin/bash + +set -xeu -o pipefail + +# README.md shows the top-level --help output. It is a copy of the clap +# definitions in crates/dat3/src/main.rs, so a changed flag or description +# would otherwise leave it silently stale. + +# Work inside tests directory +cd "$(dirname "$0")" + +# Load common variables and functions +# shellcheck source=tests/common.sh +source ./common.sh + +# The first block whose prompt line is a bare `dat3`, without the blank lines +# around the output. An empty result still differs from the help text below. +readme_help=$(awk 'copying && /^```$/ { exit } copying { print } /^dat3$/ { copying = 1 }' ../README.md | sed '/./,$!d') +help=$("$DAT3" --help) + +if [ "$readme_help" != "$help" ]; then + echo "Error: the --help block in README.md differs from dat3 --help:" + # diff exits 1 on the difference it is here to show; the exit below reports it + diff <(printf '%s\n' "$readme_help") <(printf '%s\n' "$help") || true + exit 1 +fi + +echo "README help block test passed" diff --git a/tests/tsconfig.json b/tests/tsconfig.json index 0c72535..3a70e24 100644 --- a/tests/tsconfig.json +++ b/tests/tsconfig.json @@ -1,7 +1,7 @@ { "compilerOptions": { - "target": "es2023", - "lib": ["es2023"], + "target": "es2025", + "lib": ["es2025"], "module": "nodenext", "moduleResolution": "nodenext", "types": ["node"], diff --git a/tests/wasm_library.sh b/tests/wasm_library.sh new file mode 100755 index 0000000..5d5ca13 --- /dev/null +++ b/tests/wasm_library.sh @@ -0,0 +1,10 @@ +#!/bin/bash + +set -xeu -o pipefail + +# Builds the dat3-wasm npm package, then runs its integration test, which +# installs the tarball the way an application would. + +cd "$(dirname "$0")" +../crates/dat3-wasm/package.sh +node ./wasm_library.ts diff --git a/tests/wasm_library.ts b/tests/wasm_library.ts new file mode 100644 index 0000000..3722909 --- /dev/null +++ b/tests/wasm_library.ts @@ -0,0 +1,231 @@ +/** + * Integration test for the dat3-wasm npm package: dat3-core for Node and Electron. + * + * Installs the packed tarball into a scratch project, as an application would, and + * checks the library against the native dat3 on the fixture archives the other + * tests fetch. Run by wasm_library.sh after the package build; typechecked by + * `npm run typecheck`. + */ + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { spawnSync } from "node:child_process"; +import { createRequire } from "node:module"; +import { mkdirSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import path from "node:path"; + +const TESTS_DIR = import.meta.dirname; +const WORK_DIR = path.join(TESTS_DIR, "test_wasm_library"); +const TARBALL = path.join(TESTS_DIR, "..", "target", "dat3-wasm.tgz"); +const DAT3 = path.join(TESTS_DIR, "..", "target", "x86_64-unknown-linux-musl", "release", "dat3"); + +/** Bound on each child process: a sync spawn blocks the runner, so its own timeout is the only one */ +const SPAWN_TIMEOUT_MS = 120_000; +/** Room for a fixture's JSON listing, which passes spawnSync's 1 MiB default */ +const SPAWN_MAX_BUFFER = 64 * 1024 * 1024; + +const FORMATS = ["dat1", "dat2", "arcanum", "toee"] as const; + +/** Real archives of every format, one with compressed entries each, fetched by the shell tests */ +const FIXTURES = [ + { file: "FalloutDemo.dat", format: "dat1" }, + { file: "rpu.dat", format: "dat2" }, + { file: "ArcanumDemo.dat", format: "arcanum" }, + { file: "tpgamefiles.dat", format: "toee" }, +] as const; + +/** + * The package API as these tests use it. The generated dat3_wasm.d.ts is the real + * declaration, but it exists only after a build, and the typecheck runs without one. + */ +interface Entry { + name: string; + size: number; + packedSize: number; + compressed: boolean; +} + +interface Archive { + readonly format: string; + entries(): Entry[]; + read(name: string): Uint8Array; + insert(name: string, data: Uint8Array, compression?: number | null): void; + remove(name: string): boolean; + toBytes(): Uint8Array; + free(): void; +} + +interface Dat3Wasm { + Archive: { + new (format: string): Archive; + fromBytes(bytes: Uint8Array): Archive; + }; +} + +/** One object in a `dat3 l --json` array */ +interface ListingEntry { + name: string; + size: number; + packed_size: number; + compressed: boolean; +} + +function run(command: string, args: string[]): string { + const result = spawnSync(command, args, { + cwd: WORK_DIR, + encoding: "utf8", + timeout: SPAWN_TIMEOUT_MS, + maxBuffer: SPAWN_MAX_BUFFER, + }); + if (result.error) { + throw result.error; + } + assert.equal(result.status, 0, `${command} ${args.join(" ")} failed:\n${result.stderr}`); + return result.stdout; +} + +/** The native listing, in the library's field names */ +function nativeEntries(archive: string): Entry[] { + // Unchecked here: json_listing.ts verifies this document's shape field by field + const listing = JSON.parse(run(DAT3, ["l", "--json", "--case-sensitive", archive])) as ListingEntry[]; + return listing.map((entry) => ({ + name: entry.name, + size: entry.size, + packedSize: entry.packed_size, + compressed: entry.compressed, + })); +} + +function isDat3Wasm(value: unknown): value is Dat3Wasm { + if (typeof value !== "object" || value === null || !("Archive" in value)) { + return false; + } + const archive: unknown = value.Archive; + return typeof archive === "function" && "fromBytes" in archive && typeof archive.fromBytes === "function"; +} + +rmSync(WORK_DIR, { recursive: true, force: true }); +mkdirSync(path.join(WORK_DIR, "src"), { recursive: true }); +writeFileSync(path.join(WORK_DIR, "package.json"), JSON.stringify({ name: "consumer", private: true })); +// --offline: the package has no dependencies, so installing it must not need the network +run("npm", ["install", "--offline", "--no-audit", "--no-fund", TARBALL]); +const loaded: unknown = createRequire(path.join(WORK_DIR, "package.json"))("dat3-wasm"); +assert.ok(isDat3Wasm(loaded), "dat3-wasm does not export the Archive class"); +const { Archive } = loaded; + +const BIG = Buffer.from("frame ".repeat(500)); +const TINY = Buffer.from("hi\n"); + +for (const { file, format } of FIXTURES) { + test(`${format}: ${file} lists and reads exactly as the native dat3 extracts it`, () => { + const archivePath = path.join(TESTS_DIR, file); + const archive = Archive.fromBytes(readFileSync(archivePath)); + try { + assert.equal(archive.format, format); + const entries = archive.entries(); + assert.deepEqual(entries, nativeEntries(archivePath)); + assert.ok( + entries.some((entry) => entry.compressed), + `${file} has no compressed entry, so decompression goes untested`, + ); + + const outDir = path.join(WORK_DIR, `native_${format}`); + run(DAT3, ["x", "--case-sensitive", archivePath, "-o", outDir]); + const differing = entries + .filter((entry) => Buffer.compare(archive.read(entry.name), readFileSync(path.join(outDir, entry.name))) !== 0) + .map((entry) => entry.name); + assert.deepEqual(differing, []); + } finally { + archive.free(); + } + }); +} + +for (const format of FORMATS) { + test(`${format}: an archive built in JS reads back, and the native dat3 reads it too`, () => { + const archive = new Archive(format); + archive.insert("art/Hero.FRM", BIG, 9); + archive.insert("sub\\tiny.txt", TINY); + const bytes = archive.toBytes(); + archive.free(); + + const file = path.join(WORK_DIR, `built_${format}.dat`); + writeFileSync(file, bytes); + assert.deepEqual( + nativeEntries(file) + .map((entry) => entry.name) + .sort(), + ["art/Hero.FRM", "sub/tiny.txt"], + ); + + const reopened = Archive.fromBytes(bytes); + assert.deepEqual(Buffer.from(reopened.read("ART/hero.frm")), BIG); + assert.deepEqual(Buffer.from(reopened.read("sub/tiny.txt")), TINY); + reopened.free(); + }); +} + +test("entries are plain objects that survive JSON and structured cloning", () => { + const archive = new Archive("dat2"); + archive.insert("a.txt", TINY); + const entries = archive.entries(); + archive.free(); + const expected = [{ name: "a.txt", size: TINY.length, packedSize: TINY.length, compressed: false }]; + assert.deepEqual(entries, expected); + assert.deepEqual(JSON.parse(JSON.stringify(entries)), expected); + assert.deepEqual(structuredClone(entries), expected); +}); + +test("insert replaces a name in any case, and remove reports whether it removed", () => { + const archive = new Archive("dat2"); + archive.insert("Data/A.txt", Buffer.from("old")); + archive.insert("data/a.TXT", Buffer.from("new")); + assert.deepEqual( + archive.entries().map((entry) => entry.name), + ["data/a.TXT"], + ); + assert.equal(Buffer.from(archive.read("DATA/A.TXT")).toString(), "new"); + assert.equal(archive.remove("data/a.txt"), true); + assert.equal(archive.remove("data/a.txt"), false); + assert.deepEqual(archive.entries(), []); + archive.free(); +}); + +test("names differing only in case stay reachable by their exact spelling", () => { + writeFileSync(path.join(WORK_DIR, "src", "README.TXT"), "upper"); + writeFileSync(path.join(WORK_DIR, "src", "readme.txt"), "lower"); + run(DAT3, ["a", "--case-sensitive", "twins.dat", "-C", "src", "README.TXT", "readme.txt"]); + + const archive = Archive.fromBytes(readFileSync(path.join(WORK_DIR, "twins.dat"))); + assert.equal(Buffer.from(archive.read("README.TXT")).toString(), "upper"); + assert.equal(Buffer.from(archive.read("readme.txt")).toString(), "lower"); + assert.throws(() => archive.read("Readme.txt"), { + name: "Error", + message: "Readme.txt matches entries that differ only in case; name one exactly: README.TXT / readme.txt", + }); + archive.free(); +}); + +test("failures throw an Error carrying the reason", () => { + assert.throws(() => new Archive("zip"), { + name: "Error", + message: 'unsupported archive format "zip" (expected dat1, dat2, arcanum, or toee)', + }); + assert.throws(() => Archive.fromBytes(new Uint8Array([1, 2, 3])), { + name: "Error", + message: "DAT2 file too small", + }); + + const archive = new Archive("dat2"); + assert.throws(() => archive.read("missing.txt"), { name: "Error", message: "File not found: missing.txt" }); + assert.throws(() => archive.insert("../escape.txt", TINY), { + name: "Error", + message: /^Invalid archive path for add operation \('\.\.' component\)/, + }); + assert.throws(() => archive.insert("a.txt", TINY, 10), { + name: "Error", + message: "Compression level must be 0-9, got 10", + }); + assert.deepEqual(archive.entries(), []); + archive.free(); +});