diff --git a/AGENTS.md b/AGENTS.md index b5f0a7f..ec7e70c 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,13 +4,18 @@ Guidance for AI coding agents (and human contributors) working in this repo. ## Project -Senbei is a static unpacker for Crackproof-protected PE files: a Cargo -workspace with a pure, panic-free, no-I/O unpacker core (`senbei-pe/`, built -on `senbei-crypto/`), an il2cpp metadata de-obfuscator (`senbei-metadata/`), -filesystem/CLI orchestration (`senbei-io/`), the `senbei` binary -(`senbei-cli/`), WebAssembly bindings (`senbei-wasm/`, outside the workspace; -builds into `web/pkg/`), and the static browser frontend assets (`web/`). -Read `docs/design.md` first. +Senbei is a static unpacker for Crackproof-protected PE files and protected +Android (AArch64) shared libraries: a Cargo workspace with a pure, panic-free, +no-I/O PE unpacker core (`senbei-pe/`, built on `senbei-crypto/`), il2cpp +metadata de-obfuscators (`senbei-metadata/` for the Windows structural +variant, `senbei-android-metadata/` for the Android seeded-permutation and +embedded-blob variants), the native-only Android pipeline +(`senbei-android-crypto/`, `senbei-android-engine/`, `senbei-android-elf/`), +filesystem/CLI orchestration (`senbei-io/`, including the Android +single-library/package glue in `senbei-io/src/android.rs`), the `senbei` +binary (`senbei-cli/`), WebAssembly bindings (`senbei-wasm/`, outside the +workspace; builds into `web/pkg/`), and the static browser frontend assets +(`web/`). Read `docs/design.md` first. ## Commands @@ -24,15 +29,26 @@ cd senbei-wasm && wasm-pack build --target web --release --out-dir ../web/pkg The `samples/` corpus is user-managed and absent on CI; without it the samples test is a no-op pass. `SENBEI_REQUIRE_SAMPLES=1` makes an absent -corpus fail (use this on a private CI that *does* have the corpus). Do not +corpus fail (use this on a private CI that *does* have the corpus). The +Android corpus lives in `samples/android/` (one extracted app tree per +subdirectory) and is covered by `tests/android_samples.rs`; +`SENBEI_ANDROID_SAMPLES` overrides that location. Do not delete `samples/` with `rm -rf` — it may be a junction; use git worktree-aware cleanup. ## Hard rules -- **The unpacker core stays pure**: no file I/O, no `unsafe`, no panics across - the public boundary, no platform-specific code. It must keep compiling to - `wasm32-unknown-unknown` (`cargo check --target wasm32-unknown-unknown`). +- **The PE unpacker core stays pure**: `senbei-pe` and `senbei-crypto` have no + file I/O, no `unsafe`, no panics across the public boundary, no + platform-specific code. Everything `senbei-wasm` compiles must keep building + for `wasm32-unknown-unknown` (`cargo check --target wasm32-unknown-unknown` + at the workspace root covers it — the Android crates do compile to wasm, but + nothing on the wasm path calls them). +- **The Android crates are native-only orchestration-style crates**: + `senbei-android-engine`/`senbei-android-elf` memory-map inputs and write a + module workspace to disk (the restore is a two-phase design consuming that + workspace). Keep them off the web app's code paths; `senbei-io`'s + `android.rs` is the only caller the CLI uses. - **`catch_unwind` does not work on wasm** (the prebuilt std can't unwind; a caught panic becomes a fatal `unreachable` trap). Native code may rely on `catch_unpack`, but any routing decision must also work without a catchable diff --git a/Cargo.lock b/Cargo.lock index 350f66f..35c098c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2,6 +2,23 @@ # It is not intended for manual editing. version = 4 +[[package]] +name = "adler2" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" + +[[package]] +name = "aes" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b169f7a6d4742236a0a00c541b845991d0ac43e546831af1249753ab4c3aa3a0" +dependencies = [ + "cfg-if", + "cipher", + "cpufeatures", +] + [[package]] name = "anyhow" version = "1.0.104" @@ -14,18 +31,43 @@ version = "2.13.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + [[package]] name = "bumpalo" version = "3.20.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" +[[package]] +name = "byteorder" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" + [[package]] name = "cfg-if" version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" +[[package]] +name = "cipher" +version = "0.4.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad" +dependencies = [ + "crypto-common", + "inout", +] + [[package]] name = "console" version = "0.16.4" @@ -38,6 +80,50 @@ dependencies = [ "windows-sys", ] +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "crc32fast" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "crossbeam-utils" +version = "0.8.22" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + [[package]] name = "encode_unicode" version = "1.0.0" @@ -60,6 +146,17 @@ version = "2.5.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" +[[package]] +name = "flate2" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" +dependencies = [ + "crc32fast", + "miniz_oxide", + "zlib-rs", +] + [[package]] name = "futures-core" version = "0.3.34" @@ -84,6 +181,16 @@ dependencies = [ "slab", ] +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + [[package]] name = "getrandom" version = "0.4.3" @@ -95,6 +202,17 @@ dependencies = [ "r-efi", ] +[[package]] +name = "goblin" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "17582616a7718cca54cec18e534a76c7c4aec11a8b9a85695712f262fd15a4c8" +dependencies = [ + "log", + "plain", + "scroll", +] + [[package]] name = "indicatif" version = "0.18.6" @@ -108,6 +226,21 @@ dependencies = [ "web-time", ] +[[package]] +name = "inout" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01" +dependencies = [ + "generic-array", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + [[package]] name = "js-sys" version = "0.3.104" @@ -131,6 +264,37 @@ version = "0.12.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" +[[package]] +name = "log" +version = "0.4.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "memmap2" +version = "0.9.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" +dependencies = [ + "libc", +] + +[[package]] +name = "miniz_oxide" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" +dependencies = [ + "adler2", + "simd-adler32", +] + [[package]] name = "once_cell" version = "1.21.4" @@ -149,6 +313,12 @@ version = "0.2.17" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" +[[package]] +name = "plain" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b4596b6d070b27117e987119b4dac604f3c58cfb0b191112e24771b2faeac1a6" + [[package]] name = "portable-atomic" version = "1.15.0" @@ -207,49 +377,179 @@ dependencies = [ "winapi-util", ] +[[package]] +name = "scroll" +version = "0.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c1257cd4248b4132760d6524d6dda4e053bc648c9070b960929bf50cfb1e7add" +dependencies = [ + "scroll_derive", +] + +[[package]] +name = "scroll_derive" +version = "0.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e1a36a382ed65dbcc0ab47fd5e9a94112417ccd34560a392ef3b7b0f0ec39148" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.4", +] + +[[package]] +name = "senbei-android-crypto" +version = "1.2.0" +dependencies = [ + "aes", + "thiserror", +] + +[[package]] +name = "senbei-android-elf" +version = "1.2.0" +dependencies = [ + "memmap2", + "senbei-android-crypto", + "serde", + "serde_json", + "sha2", + "tempfile", + "thiserror", +] + +[[package]] +name = "senbei-android-engine" +version = "1.2.0" +dependencies = [ + "goblin", + "memmap2", + "senbei-android-crypto", + "serde", + "serde_json", + "sha2", + "tempfile", + "thiserror", +] + +[[package]] +name = "senbei-android-metadata" +version = "1.2.0" +dependencies = [ + "serde", + "thiserror", +] + [[package]] name = "senbei-cli" -version = "1.1.0" +version = "1.2.0" dependencies = [ "senbei-io", "senbei-metadata", + "sha2", "tempfile", ] [[package]] name = "senbei-crypto" -version = "1.1.0" +version = "1.2.0" dependencies = [ "thiserror", ] [[package]] name = "senbei-io" -version = "1.1.0" +version = "1.2.0" dependencies = [ "anyhow", + "flate2", "indicatif", "libc", "owo-colors", + "senbei-android-elf", + "senbei-android-engine", + "senbei-android-metadata", "senbei-metadata", "senbei-pe", + "sha2", "tempfile", "walkdir", "windows", + "zip", ] [[package]] name = "senbei-metadata" -version = "1.1.0" +version = "1.2.0" [[package]] name = "senbei-pe" -version = "1.1.0" +version = "1.2.0" dependencies = [ "senbei-crypto", "thiserror", ] +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.4", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures", + "digest", +] + +[[package]] +name = "simd-adler32" +version = "0.3.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" + [[package]] name = "slab" version = "0.4.12" @@ -311,6 +611,12 @@ dependencies = [ "syn 3.0.4", ] +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + [[package]] name = "unicode-ident" version = "1.0.24" @@ -329,6 +635,12 @@ version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "81e544489bf3d8ef66c953931f56617f423cd4b5494be343d9b9d3dda037b9a3" +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + [[package]] name = "walkdir" version = "2.5.0" @@ -521,3 +833,27 @@ checksum = "3949bd5b99cafdf1c7ca86b43ca564028dfe27d66958f2470940f73d86d75b37" dependencies = [ "windows-link", ] + +[[package]] +name = "zip" +version = "0.6.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "760394e246e4c28189f19d488c058bf16f564016aefac5d32bb1f3b51d5e9261" +dependencies = [ + "byteorder", + "crc32fast", + "crossbeam-utils", + "flate2", +] + +[[package]] +name = "zlib-rs" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/Cargo.toml b/Cargo.toml index 32f4bb8..d539d14 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,5 +1,9 @@ [workspace] members = [ + "senbei-android-crypto", + "senbei-android-elf", + "senbei-android-engine", + "senbei-android-metadata", "senbei-cli", "senbei-crypto", "senbei-io", @@ -13,15 +17,23 @@ exclude = ["senbei-wasm"] resolver = "2" [workspace.package] -version = "1.1.0" +version = "1.2.0" edition = "2024" +rust-version = "1.85" license = "AGPL-3.0-only" [workspace.dependencies] +aes = "0.8" anyhow = "1" +flate2 = "1" +goblin = "0.10" indicatif = "0.18" libc = "0.2" +memmap2 = "0.9" owo-colors = "4" +serde = { version = "1", features = ["derive"] } +serde_json = "1" +sha2 = "0.10" tempfile = "3" thiserror = "2" walkdir = "2" @@ -30,11 +42,25 @@ windows = { version = "0.62", features = [ "Win32_System_Console", "Win32_System_SystemInformation", ] } +zip = { version = "0.6.6", default-features = false, features = ["deflate"] } +senbei-android-crypto = { path = "senbei-android-crypto" } +senbei-android-elf = { path = "senbei-android-elf" } +senbei-android-engine = { path = "senbei-android-engine" } +senbei-android-metadata = { path = "senbei-android-metadata" } senbei-crypto = { path = "senbei-crypto" } senbei-io = { path = "senbei-io" } senbei-metadata = { path = "senbei-metadata" } senbei-pe = { path = "senbei-pe" } +[workspace.lints.rust] +unsafe_op_in_unsafe_fn = "deny" + +[workspace.lints.clippy] +correctness = { level = "deny", priority = -1 } +suspicious = { level = "warn", priority = -1 } +complexity = { level = "warn", priority = -1 } +perf = { level = "warn", priority = -1 } + [profile.release] opt-level = 3 lto = true diff --git a/README.md b/README.md index d481381..30fc6f9 100644 --- a/README.md +++ b/README.md @@ -1,7 +1,8 @@ # Senbei -A static unpacker for Crackproof-protected 64-bit and 32-bit PE files. Point it -at a file or a folder and it writes decrypted copies — no launch of the +A static unpacker for Crackproof-protected 64-bit and 32-bit PE files and +protected Android (AArch64) shared libraries. Point it at a file, an app +package, or a folder and it writes decrypted copies — no launch of the protected program, no kernel driver, no code runs out of the protected binary. > _"Crackproof"? It's senbei (煎餅 — rice cracker). Cracks itself._ @@ -27,8 +28,12 @@ lives in [`web/`](web/). is lawful. - Senbei does not bypass any access control for you: it performs a purely static transformation of a file already on your disk. It derives everything - it needs from the input file itself, contains no vendor code or secrets, and - distributes no keys, cracks, or copyrighted content. + it needs from the input file itself, contains no vendor code, and + distributes no cracks or copyrighted content. (One Android packaging + variant's embedded metadata layer is unwrapped with an XOR keystream + recovered from a ciphertext/plaintext pair during analysis of a single + build; that keystream is research output shipped with the unpacker, not a + vendor-distributed key, and builds it doesn't match are left alone.) - Senbei does not enable online play, license fraud, or cheating, and must not be used to redistribute decrypted binaries. Do not upload outputs anywhere. - The authors provide this software "as is", without warranty of any kind, and @@ -47,9 +52,13 @@ lives in [`web/`](web/). | `ManagedDll` | Protected .NET assembly (has a CLR data directory). | | `._` companion | Stub + external encrypted payload layout, spliced automatically. | | `global-metadata.dat` | il2cpp metadata with obfuscated method tokens, de-obfuscated in place. | +| Android `.so` | Protected AArch64 shared library, statically restored (hollowed sections + stripped dynamic tables rebuilt). | +| `.apk` / `.apks` / `.xapk` | App packages; protected entries inside are restored, preserving the package's internal layout. | Detection is content-based (header key-table at offset 4096, magic `KONN`), -not extension-based. Anything unrecognized is left untouched. +not extension-based — app packages are the one exception, recognised by +extension plus the zip magic because they are containers. Anything +unrecognized is left untouched. ## Quick start @@ -59,6 +68,9 @@ cargo build --release senbei protected.exe :: -> unpack\protected.unpack.exe +senbei game.apk +:: -> unpack\game.apk\lib\arm64-v8a\libil2cpp.unpack.so + senbei "C:\Games\MyGame" :: -> C:\Games\MyGame\unpack\... (recursive, skips non-targets) ``` diff --git a/docs/design.md b/docs/design.md index 3b75cd8..4cb88b3 100644 --- a/docs/design.md +++ b/docs/design.md @@ -18,10 +18,25 @@ Senbei is a Cargo workspace split into a pure core and thin shells around it: primitives the core is built from. Same purity rules as `senbei-pe`. - **`senbei-metadata/`** — il2cpp `global-metadata.dat` method-token de-obfuscation (format version 31; other versions are left untouched). +- **`senbei-android-crypto/`** — container primitives of the Android + (AArch64) protection scheme: the word/record ciphers, the GF(2³²) + transform, the AES-augmented segment transform, and the Huffman/LZ decoder. +- **`senbei-android-engine/`** — stage-1/stage-2 extraction: finds the + appended payload section, decrypts the stage-1 header and stage-2 payload, + and walks the recursive record streams to decode every module. Native-only + (memory-maps the input, writes the module set to a workspace directory). +- **`senbei-android-elf/`** — the restore: replays the decoded target-image + and fixup containers onto a hollowed ELF and rebuilds the dynamic-linker + tables (hash tables, symbols, relocations) the protector stripped. + Native-only. +- **`senbei-android-metadata/`** — the Android metadata variants: the seeded + five-round MethodDef-RID permutation restore (v31), seed discovery, and the + embedded-metadata XOR unwrap (`keystream.rs`). - **`senbei-io/`** — filesystem and orchestration: recursive folder scanning, - per-run log file, progress bar, Explorer-friendly exit pause, and the + per-run log file, progress bar, Explorer-friendly exit pause, the single-file/folder orchestration in `job.rs` (incl. the wasm-safe in-memory - byte API used by the web frontend). + byte API used by the web frontend), and `android.rs` — the Android + single-library / folder / app-package orchestration. - **`senbei-cli/`** — the `senbei` binary: argument parsing + dispatch. The integration test suite (incl. the golden corpus test) lives in `senbei-cli/tests/`. @@ -33,7 +48,9 @@ senbei-io/src/ ├── job.rs single-file + folder orchestration, out-naming, │ companion splice, stub overlay/TLS restore, │ pipeline routing (incl. the wasm-safe byte API) -├── scan.rs recursive Crackproof + metadata discovery +├── android.rs Android single-library / folder / package +│ orchestration, cross-source dedup +├── scan.rs recursive target discovery (PE + metadata + Android) ├── logfile.rs per-run timestamped log ├── ui.rs progress bar + status lines └── pause.rs Explorer-friendly exit pause @@ -58,6 +75,23 @@ senbei-pe/src/engine/ pure, panic-free, no-I/O core │ └── pipeline/pe32.rs PE32-specific EXE restore └── dll/ └── pipeline.rs native + managed DLL pipeline +senbei-android-crypto/src/ +└── protector.rs container ciphers, GF(2^32), Huffman/LZ decoder +senbei-android-engine/src/ +├── stage1.rs payload-section discovery + stage-1 header/payload +├── stream.rs record-stream parsing +├── extract.rs recursive module extraction (writes the workspace) +├── probe.rs protected-library content probe +└── report.rs machine-readable extraction report +senbei-android-elf/src/ +├── restore.rs image restore + dynamic-table rebuild +├── layout.rs ELF layout parsing +├── artifact.rs module-workspace index loading +└── hash.rs SysV/GNU hash table rebuild +senbei-android-metadata/src/ +├── method_tokens.rs seeded RID permutation restore + seed discovery +├── embedded.rs embedded-metadata blob locate + XOR unwrap +└── keystream.rs recovered keystream table (one observed build) ``` ## Detection and routing @@ -66,7 +100,11 @@ Detection is content-based (`unpacker::detect`), never extension-based: the key table is derived from the file header and checked against the format magic, then the PE characteristics classify the input as EXE or DLL and the CLR data directory splits each into native vs managed (`NativeExe` / -`ManagedExe` / `NativeDll` / `ManagedDll`). +`ManagedExe` / `NativeDll` / `ManagedDll`). The folder scan additionally +classifies Android targets: an ELF64/AArch64 prefix promotes the file to a +full protection probe (`senbei_android_engine::is_protected_libil2cpp`), and a +package extension plus zip magic marks an app package for container +extraction. `unpack_auto` then dispatches: @@ -115,6 +153,47 @@ Several protected stages are themselves little bytecode programs. The core includes a small VM (`bytecode.rs`) that generates and interprets those programs rather than hardcoding each variant's constants. +## The Android pipeline + +The Android scheme hollows an ELF64/AArch64 shared object: section bodies are +zeroed in the file and the original bytes move into an encrypted payload +appended as a `SHT_LOUSER` section (invisible to the dynamic loader). Restore +is two-phase: + +1. **Extract** (`senbei-android-engine`): decrypt the stage-1 parameter block + and stage-2 payload from the payload section, then walk the recursive + record streams — each decoded module may interpret a further nested stream + — into a temporary module workspace with a JSON index. +2. **Restore** (`senbei-android-elf`): decode the target-image container onto + a copy of the hollowed file, apply the compact fixup database (the + relocations stripped from `.rela.dyn`), and rebuild the dynamic-linker + tables the loader needs (SysV/GNU hash, symbol and string tables, + `.rela.dyn`/`.rela.plt`). Validation is structural and total: mismatched + container sizes, descriptor bounds, or a rebuilt table overhanging its + section fail the restore rather than emit a broken image. + +il2cpp metadata comes in three shapes, all routed through +`job::deobfuscate_metadata_to` / `android::restore_metadata_bytes`: + +- **structural (Windows `-GMD`)**: sparse method tokens remapped to the + contiguous per-module range, keyless, idempotent (`senbei-metadata`). +- **seeded permutation (Android v31)**: MethodDef RIDs permuted by a keyed + five-round transform; the seed is recovered by intersecting per-image key + residues, and the restore validates every RID — a wrong seed errors and the + structural remap takes over (`senbei-android-metadata`). +- **embedded blob**: no metadata file in the app at all; a slim blob sits in + the library's data section under a per-word XOR layer. After a restore the + blob is located by content (two known plaintext header words against the + embedded keystream) and unwrapped to a standalone `global-metadata.dat`. + Key derivation is untraced — the shipped keystream covers the one observed + build, and other builds simply never match the probe. + +Packages (`.apk`/`.apks`/`.xapk`) are containers, not targets: entries are +extracted to a temporary workspace and content-probed like loose files. +Cross-source duplicates (a library loose in the tree *and* inside its +package) are restored once, preferring the loose file, then the `.apk`, then +bundle splits. + ## Integrity check Every produced image passes through `integrity::check` — a static, execution- diff --git a/docs/development.md b/docs/development.md index 66b77d2..0206cec 100644 --- a/docs/development.md +++ b/docs/development.md @@ -57,6 +57,10 @@ since binaries are not committed). - `SENBEI_THREADS` — cap the block-parallel fan-out (`1` forces the fully sequential path). - `SENBEI_SCAN_ALL` — same as `--scan-all` (probe every file in a folder). +- `SENBEI_ANDROID_SAMPLES` — override the Android corpus location (default + `samples/android/`; see `samples/README.md`). The Android corpus test pins + restored outputs with SHA-256 sidecar files next to each protected input + and documents known restore gaps with empty `.restore-fails` markers. ## Conventions diff --git a/docs/usage.md b/docs/usage.md index 9bfcc3f..32824e6 100644 --- a/docs/usage.md +++ b/docs/usage.md @@ -30,13 +30,44 @@ expects; the output is `global-metadata.unpack.dat`, written only when tokens actually changed. Only metadata format version 31 is rewritten; other versions are reported and left untouched. +## Android targets + +Senbei also restores Android (AArch64) protected shared libraries and app +packages: + +- **`.so`** — a protected library is hollowed out on disk: its original + sections live in an encrypted payload appended to the file, and senbei + rebuilds the static image from it. Output: `libil2cpp.unpack.so`. +- **`.apk`** — entries are extracted to a temporary workspace and + content-probed like loose files; protected libraries and metadata blobs + inside are restored to `//`. +- **`.apks` / `.xapk`** — split-package bundles; each nested `.apk` is opened + and searched the same way, under `///...`. + +When a restored il2cpp library carries its metadata embedded in its data +section (no standalone `global-metadata.dat` in the app at all), senbei +unwraps the blob and writes it next to the library as +`global-metadata.unpack.dat`. One observed packaging variant wraps the blob in +a per-word XOR layer whose keys are generated at runtime and stored nowhere; +senbei ships the keystream recovered from the one build known to use it and +content-probes for it — builds with a different keystream are silently +skipped (the library itself is still fully restored). + +The same content may appear loose in a folder, in its `.apk`, and in a bundle +side by side: identical content is restored once, at the loose file's +destination. A restored library is validated structurally by the restore +itself (the rebuild refuses inconsistent layouts); a protected library that +fails validation counts as an error, not a suspect. + ## Folder mode Senbei walks the directory recursively, skips any subdirectory literally named -`unpack`, and unpacks every file it recognises as Crackproof-protected (by -content, not extension — renamed files and `.bak` backups are still found). -Results land under `/unpack/` (or `--out DIR`), mirroring the input -tree's relative paths. The run log is written **in that same out directory**: +`unpack`, and unpacks every file it recognises as protected (by content, not +extension — renamed files and `.bak` backups are still found; packages are the +one exception, recognised by extension plus the zip magic because they are +containers). Results land under `/unpack/` (or `--out DIR`), mirroring +the input tree's relative paths. The run log is written **in that same out +directory**: ```cmd senbei "C:\Games\MyGame" @@ -58,8 +89,14 @@ line, then duration: done in 1234 ms ``` +The `packages` count appears (as `· N packages`) only when Android app +packages were processed. + ## Integrity check +(PE outputs only — Android restores carry their own structural validation; see +[Android targets](#android-targets).) + A successful unpack is not always a runnable one: a layout heuristic can pick the wrong offset and leave the entry-point stub or import strings encrypted, so the pipeline reports success but the OS loader faults at runtime (typically @@ -98,7 +135,7 @@ line, adds a `SUSPECT` entry to the run log, and counts it in the summary's | Flag | Behavior | | --- | --- | | `--out DIR` | Write outputs (and the log, unless `--no-log`) under `DIR`. | -| `-v`, `--verbose` | Print detailed `[N/9]` per-stage unpack progress (and the destination path) for each file. In folder mode this replaces the progress bar. | +| `-v`, `--verbose` | Print detailed per-stage progress (and the destination path) for each file — `[N/9]` stages for PE targets, container/segment lines for Android libraries. In folder mode this replaces the progress bar. | | `-q`, `--quiet` | Once: hide progress bar and per-file lines; keep banner, summary, and duration. Twice (`-q -q`): suppress all stdio (exit code only). | | `--no-log` | Do not write `senbei-*.log`. Console output is unchanged by this flag alone. | | `--scan-all` | Probe every file in a folder, including ones the scan pre-filter skips (under 4128 bytes, or a bulk-asset extension like `.ab`/`.xml`/`.acb`). Much slower on large game trees; finds the same targets in practice. | @@ -115,7 +152,7 @@ process. | Code | Meaning | | --- | --- | -| `0` | Success (single file unpacked, or folder run with no errors). | +| `0` | Success (single file restored, or folder run with no errors). | | `1` | At least one file failed, a scan probe was unreadable, or a single-file unpack errored. | | `2` | Usage error: no path given, unknown option, missing `--out` value, or multiple input paths (help printed). | diff --git a/samples/README.md b/samples/README.md index f6f0de3..07fed79 100644 --- a/samples/README.md +++ b/samples/README.md @@ -82,3 +82,24 @@ cargo test --release --test samples -- --nocapture - A **companion** is `._` (e.g. `stub.dll._` for `stub.dll`). Its extension is `_`, so it is never picked up as an input of its own; it is read only when its base module is processed. + +## Android corpus (`samples/android/`) + +The `android/` subfolder holds Android samples, one **extracted app tree** per +subdirectory (the layout an APK unpacks to: `lib//*.so`, +`assets/.../global-metadata.dat`, ...). The test +(`tests/android_samples.rs`) finds protected AArch64 libraries by content and +restores them through the real pipeline. `SENBEI_ANDROID_SAMPLES` overrides +the corpus location. + +Sidecar conventions (all next to the protected `.so` input): + +| File | Meaning | +| ---- | ------- | +| `.golden.so.sha256` | Expected SHA-256 of the restored library | +| `.golden.metadata.sha256` | Expected SHA-256 of the unwrapped embedded metadata blob (when the library carries one) | +| `.restore-fails` | Empty marker: this input's restore is a known gap and *must* fail (a future fix fails the test, prompting marker removal) | + +A missing sidecar is a warning (with the computed digest printed, ready to +promote), never a failure. App packages (`.apk`/`.apks`/`.xapk`) dropped into +a tree are exercised by folder mode as containers. diff --git a/senbei-android-crypto/Cargo.toml b/senbei-android-crypto/Cargo.toml new file mode 100644 index 0000000..269e2aa --- /dev/null +++ b/senbei-android-crypto/Cargo.toml @@ -0,0 +1,14 @@ +[package] +name = "senbei-android-crypto" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +description = "Protector container primitives for Senbei Android" + +[dependencies] +aes.workspace = true +thiserror.workspace = true + +[lints] +workspace = true diff --git a/senbei-android-crypto/src/lib.rs b/senbei-android-crypto/src/lib.rs new file mode 100644 index 0000000..36c6876 --- /dev/null +++ b/senbei-android-crypto/src/lib.rs @@ -0,0 +1,8 @@ +//! Cryptographic and container primitives used by Senbei Android. + +mod protector; + +pub use protector::{ + ContainerHeader, EncodedSegment, Error, HuffmanLzDecoder, Module9bConfig, ProtectedDescriptor, + decode_container, gf32_mul_fixed, transform_segment, +}; diff --git a/senbei-android-crypto/src/protector.rs b/senbei-android-crypto/src/protector.rs new file mode 100644 index 0000000..49fe6c9 --- /dev/null +++ b/senbei-android-crypto/src/protector.rs @@ -0,0 +1,717 @@ +//! Cryptographic and compression primitives used by the Android protector. + +use aes::Aes256; +use aes::cipher::{Block, BlockDecrypt, KeyInit}; + +const RECORD_SIZE: usize = 0x5c; + +/// Errors raised while parsing or decoding protector containers. +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("{0}")] + Invalid(String), +} + +type Result = std::result::Result; + +fn invalid(message: impl Into) -> Result { + Err(Error::Invalid(message.into())) +} + +fn range(data: &[u8], offset: usize, size: usize) -> Result<&[u8]> { + let end = offset + .checked_add(size) + .ok_or_else(|| Error::Invalid("byte range overflow".to_owned()))?; + data.get(offset..end).ok_or_else(|| { + Error::Invalid(format!( + "byte range 0x{offset:x}..0x{end:x} is out of bounds" + )) + }) +} + +fn read_u16(data: &[u8], offset: usize) -> Result { + let bytes: [u8; 2] = range(data, offset, 2)? + .try_into() + .map_err(|_| Error::Invalid("invalid u16 range".to_owned()))?; + Ok(u16::from_le_bytes(bytes)) +} + +fn read_u32(data: &[u8], offset: usize) -> Result { + let bytes: [u8; 4] = range(data, offset, 4)? + .try_into() + .map_err(|_| Error::Invalid("invalid u32 range".to_owned()))?; + Ok(u32::from_le_bytes(bytes)) +} + +fn align_up(value: usize, alignment: usize) -> Result { + let mask = alignment + .checked_sub(1) + .ok_or_else(|| Error::Invalid("zero alignment".to_owned()))?; + value + .checked_add(mask) + .map(|v| v & !mask) + .ok_or_else(|| Error::Invalid("alignment overflow".to_owned())) +} + +/// Multiply by the fixed element used by the native GF(2^32) transform. +#[must_use] +pub fn gf32_mul_fixed(mut value: u32) -> u32 { + let mut multiplier = 0x9451_1dd2_u32; + let mut result = 0_u32; + while multiplier != 0 { + if multiplier & 1 != 0 { + result ^= value; + } + let carry = value >> 31; + value = value.wrapping_shl(1); + if carry != 0 { + value ^= 0x5793_57eb; + } + multiplier >>= 1; + } + result +} + +fn mix_columns(block: [u8; 16]) -> [u8; 16] { + const fn xtime(value: u8) -> u8 { + (value << 1) ^ if value & 0x80 != 0 { 0x1b } else { 0 } + } + + let mut output = [0_u8; 16]; + for offset in (0..16).step_by(4) { + let [a, b, c, d] = block[offset..offset + 4] else { + unreachable!("fixed four-byte AES column") + }; + output[offset] = xtime(a) ^ (xtime(b) ^ b) ^ c ^ d; + output[offset + 1] = a ^ xtime(b) ^ (xtime(c) ^ c) ^ d; + output[offset + 2] = a ^ b ^ xtime(c) ^ (xtime(d) ^ d); + output[offset + 3] = (xtime(a) ^ a) ^ b ^ c ^ xtime(d); + } + output +} + +/// Static configuration recovered from module `0x9B`. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct Module9bConfig { + pub header_seed: u32, + pub container_seed: u32, + pub aes_key: [u8; 32], + pub skip_aes: bool, + pub schedule_offset: usize, +} + +impl Module9bConfig { + /// Parse the unique AES-256 decryption schedule and adjacent configuration. + pub fn parse(image: &[u8]) -> Result { + Self::parse_inner(image, true) + } + + /// Parse the decoder configuration embedded in the raw Stage 2 image. + /// + /// The embedded decoder ends before the interpreter-only `skip_aes` + /// field, so that flag is definitionally false for this layout. + pub fn parse_embedded(image: &[u8]) -> Result { + Self::parse_inner(image, false) + } + + fn parse_inner(image: &[u8], has_skip_aes: bool) -> Result { + const MARKER: [u8; 4] = [0x00, 0x01, 0x0e, 0x00]; + let mut matches = image + .windows(MARKER.len()) + .enumerate() + .filter_map(|(offset, bytes)| (bytes == MARKER).then_some(offset)); + let schedule_offset = matches + .next() + .ok_or_else(|| Error::Invalid("cannot locate the 0x9B AES-256 schedule".to_owned()))?; + if schedule_offset < 8 || matches.next().is_some() { + return invalid("cannot uniquely locate the 0x9B AES-256 schedule"); + } + + let header_seed = read_u32(image, schedule_offset - 8)?; + let schedule_size = read_u32(image, schedule_offset - 4)?; + if !matches!(schedule_size, 0 | 0xf4) { + return invalid(format!( + "unexpected 0x9B AES schedule size 0x{schedule_size:x}" + )); + } + let bits = read_u16(image, schedule_offset)?; + let rounds = read_u16(image, schedule_offset + 2)?; + if (bits, rounds) != (0x100, 14) { + return invalid(format!( + "unexpected AES schedule header 0x{bits:x}/{rounds}" + )); + } + + let schedule = range(image, schedule_offset + 4, 15 * 16)?; + let mut round_keys = [[0_u8; 16]; 15]; + for (round, output) in round_keys.iter_mut().enumerate() { + let source = &schedule[round * 16..round * 16 + 16]; + for word in 0..4 { + let start = word * 4; + for byte in 0..4 { + output[start + byte] = source[start + 3 - byte]; + } + } + } + let mut aes_key = [0_u8; 32]; + aes_key[..16].copy_from_slice(&round_keys[14]); + aes_key[16..].copy_from_slice(&mix_columns(round_keys[13])); + + let container_seed_offset = schedule_offset + .checked_add(0x100) + .ok_or_else(|| Error::Invalid("container seed offset overflow".to_owned()))?; + let skip_aes = if has_skip_aes { + let skip_aes_offset = schedule_offset + .checked_add(0x240) + .ok_or_else(|| Error::Invalid("skip-AES offset overflow".to_owned()))?; + *image.get(skip_aes_offset).ok_or_else(|| { + Error::Invalid("module static configuration exceeds its image".to_owned()) + })? != 0 + } else { + false + }; + + Ok(Self { + header_seed, + container_seed: if has_skip_aes { + read_u32(image, container_seed_offset)? + } else { + header_seed + }, + aes_key, + skip_aes, + schedule_offset, + }) + } +} + +/// Decrypted header at the start of direct-data object `0x9D`. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ProtectedDescriptor { + pub command_id: u32, + pub flags: u32, + pub outer_offset: u32, + pub outer_expected_size: u32, + pub auxiliary_offset: u32, + pub auxiliary_expected_size: u32, +} + +impl ProtectedDescriptor { + /// Decrypt the `0x5c`-byte descriptor with the module header seed. + pub fn decrypt(data: &[u8], seed: u32) -> Result { + if data.len() < RECORD_SIZE { + return invalid("0x9D descriptor is truncated"); + } + let base0 = seed.wrapping_add(0xd3e8_7144).wrapping_mul(seed); + let base1 = base0.wrapping_add(seed.wrapping_mul(0x0bd9_418d)); + let mut words = [0_u32; RECORD_SIZE / 4]; + for (index, word) in words.iter_mut().enumerate() { + let cipher = read_u32(data, index * 4)?; + let subtractor = base0.wrapping_shl(if index & 1 != 0 { 4 } else { 0 }); + *word = cipher.wrapping_sub(subtractor) + ^ base1.wrapping_shr((seed.wrapping_add((index as u32).wrapping_mul(4))) & 7); + } + if words[6..].iter().any(|&word| word != 0) { + return invalid("unexpected nonzero reserved words in the 0x9D descriptor"); + } + let descriptor = Self { + command_id: words[0], + flags: words[1], + outer_offset: words[2], + outer_expected_size: words[3], + auxiliary_offset: words[4], + auxiliary_expected_size: words[5], + }; + if descriptor.command_id != 0x9d || descriptor.outer_offset as usize != RECORD_SIZE { + return invalid("unexpected decrypted 0x9D descriptor"); + } + Ok(descriptor) + } +} + +/// One encrypted segment in a decoded `0x9D` container header. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct EncodedSegment { + pub offset: u32, + pub size: u32, +} + +/// Parsed primary or auxiliary `0x9D` container. +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct ContainerHeader { + pub start: usize, + pub output_size: u32, + pub skip_aes: bool, + pub tree: Vec, + pub segments: Vec, +} + +impl ContainerHeader { + /// Parse and decrypt a container header, Huffman tree, and segment table. + pub fn parse(data: &[u8], start: usize, seed: u32) -> Result { + range(data, start, 12)?; + let seed_square = seed.wrapping_mul(seed); + let state = seed_square.wrapping_shr(17) ^ seed_square.wrapping_shl(11); + let raw0 = read_u32(data, start)?; + let raw1 = read_u32(data, start + 4)?; + let raw2 = read_u32(data, start + 8)?; + let output_size = 0xa21d_fb3a_u32 + .wrapping_shl(state & 7) + .wrapping_add(state.wrapping_mul(0xf87b_337c)) + .wrapping_add(gf32_mul_fixed(raw0)); + let flag_word = gf32_mul_fixed(raw1) + ^ state + .wrapping_add(0xbd19_c63c) + .wrapping_add(0x416e_2af2_u32.wrapping_shr(state & 0x0d)); + let segment_count = (flag_word & 0xff) as usize; + let skip_aes = (flag_word >> 8) & 0xff == 1; + let tree_size = 0x643a_3a3b_u32 + .wrapping_shl(state & 0x0b) + .wrapping_sub(state ^ 0x3b2b_f538) + .wrapping_add(gf32_mul_fixed(raw2)) as usize; + if segment_count == 0 || tree_size > 0x1b00 { + return invalid(format!( + "invalid container fields: segments={segment_count}, tree=0x{tree_size:x}" + )); + } + + let tree_start = start + .checked_add(12) + .ok_or_else(|| Error::Invalid("tree offset overflow".to_owned()))?; + let mut tree = range(data, tree_start, tree_size)?.to_vec(); + for offset in (0..tree_size & !3).step_by(4) { + let value = read_u32(&tree, offset)?; + tree[offset..offset + 4].copy_from_slice(&gf32_mul_fixed(value).to_le_bytes()); + } + let tree_state = state.wrapping_add(0xf1cb_5b81).wrapping_mul(state); + let tree_delta = tree_state.wrapping_sub(0x23b3_2203_u32.wrapping_mul(state)); + for (index, byte) in tree.iter_mut().enumerate() { + let shift = u32::try_from(index & 0x1b) + .map_err(|_| Error::Invalid("tree shift conversion failed".to_owned()))?; + let left = gf32_mul_fixed(tree_state.wrapping_shl(shift)); + let right = tree_delta.wrapping_shr((index & 0x17) as u32); + let adjustment = left.wrapping_sub(right).wrapping_shr((index & 0x1f) as u32); + *byte = byte.wrapping_add(adjustment as u8); + } + + let table_start = start + .checked_add(align_up(12 + tree_size, 4)?) + .ok_or_else(|| Error::Invalid("segment table offset overflow".to_owned()))?; + let table_size = segment_count + .checked_mul(8) + .ok_or_else(|| Error::Invalid("segment table size overflow".to_owned()))?; + let mut table = range(data, table_start, table_size)?.to_vec(); + let table_state = state.wrapping_add(0xb31f_451c).wrapping_mul(state); + let table_xor = table_state.wrapping_shl(3); + let table_add = table_state.wrapping_sub(0x822f_e82d_u32.wrapping_mul(state)); + for offset in (0..table_size).step_by(4) { + let value = read_u32(&table, offset)?; + let decoded = gf32_mul_fixed(value ^ table_xor) + .wrapping_add(table_add.wrapping_shr(((offset & 7) + 5) as u32)); + table[offset..offset + 4].copy_from_slice(&decoded.to_le_bytes()); + } + let mut segments = Vec::with_capacity(segment_count); + for index in 0..segment_count { + let offset = read_u32(&table, index * 8)?; + let size = read_u32(&table, index * 8 + 4)?; + let absolute = start + .checked_add(offset as usize) + .and_then(|value| value.checked_add(size as usize)); + if size == 0 || absolute.is_none_or(|end| end > data.len()) { + return invalid(format!("container segment {index} lies outside 0x9D")); + } + segments.push(EncodedSegment { offset, size }); + } + Ok(Self { + start, + output_size, + skip_aes, + tree, + segments, + }) + } + + /// End offset of the furthest encrypted segment. + pub fn encoded_end(&self) -> Result { + self.segments + .iter() + .map(|segment| { + self.start + .checked_add(segment.offset as usize) + .and_then(|value| value.checked_add(segment.size as usize)) + .ok_or_else(|| Error::Invalid("encoded segment end overflow".to_owned())) + }) + .collect::>>()? + .into_iter() + .max() + .ok_or_else(|| Error::Invalid("container has no encoded segments".to_owned())) + } +} + +/// Decoder for the protector's Huffman/LZ writer streams. +#[derive(Debug, Clone)] +pub struct HuffmanLzDecoder { + tree: Vec, + lookup_symbols: Vec, + lookup_bits: Vec, +} + +impl HuffmanLzDecoder { + /// Build the full 16-bit prefix lookup used by the static decoder. + pub fn new(tree: &[u8]) -> Result { + if tree.len() < 256 * 3 || tree.len() % 3 != 0 { + return invalid(format!("invalid Huffman tree size 0x{:x}", tree.len())); + } + let mut result = Self { + tree: tree.to_vec(), + lookup_symbols: vec![0; 0x1_0000], + lookup_bits: vec![0; 0x1_0000], + }; + for word in 0..0x1_0000_u32 { + let (symbol, bits) = result.decode_symbol(word)?; + if bits <= 16 { + result.lookup_symbols[word as usize] = symbol; + result.lookup_bits[word as usize] = bits; + } + } + Ok(result) + } + + fn entry(&self, index: usize) -> Result<(u16, bool, u8)> { + let offset = index + .checked_mul(3) + .ok_or_else(|| Error::Invalid("Huffman node offset overflow".to_owned()))?; + let bytes = range(&self.tree, offset, 3)?; + let raw = u16::from(bytes[0]) | (u16::from(bytes[1]) << 8); + Ok((raw & 0x7fff, raw & 0x8000 != 0, bytes[2])) + } + + fn decode_symbol(&self, word: u32) -> Result<(u16, u8)> { + let (mut value, leaf, extra) = self.entry((word & 0xff) as usize)?; + if leaf { + if extra == 0 { + return invalid("zero-width Huffman leaf"); + } + return Ok((value, extra)); + } + let mut bits = extra + .checked_add(1) + .ok_or_else(|| Error::Invalid("Huffman bit count overflow".to_owned()))?; + let mut mask = 1_u32.wrapping_shl(u32::from(extra)); + loop { + let branch = usize::from(word & mask != 0); + let (next, is_leaf, _) = self.entry(usize::from(value) + branch)?; + value = next; + if is_leaf { + return Ok((value, bits)); + } + mask = mask.wrapping_shl(1); + bits = bits + .checked_add(1) + .ok_or_else(|| Error::Invalid("Huffman bit count overflow".to_owned()))?; + if bits > 31 { + return invalid("Huffman code exceeds the native 32-bit window"); + } + } + } + + /// Decode one compressed writer payload to its exact expected size. + pub fn decode(&self, source: &[u8], output_size: usize) -> Result> { + let mut output = vec![0_u8; output_size]; + let mut source_pos = 0_usize; + let mut bit_buffer = 0_u64; + let mut available = 0_u8; + let mut consumed_bits = 0_usize; + let mut output_pos = 0_usize; + let mut prefix = 0_usize; + + while output_pos < output_size { + while available < 24 && source_pos < source.len() { + bit_buffer |= u64::from(source[source_pos]) << available; + source_pos += 1; + available += 8; + } + let key = (bit_buffer & 0xffff) as usize; + let mut bits = self.lookup_bits[key]; + let symbol = if bits != 0 { + self.lookup_symbols[key] + } else { + let mut value_offset = ((bit_buffer & 0xff) as usize) * 3; + let mut node = range(&self.tree, value_offset, 3)?; + let mut raw = u16::from(node[0]) | (u16::from(node[1]) << 8); + if raw & 0x8000 != 0 { + bits = node[2]; + raw & 0x7fff + } else { + let extra = node[2]; + bits = extra + 1; + let mut mask = 1_u64 << extra; + loop { + let branch = usize::from(bit_buffer & mask != 0); + let index = usize::from(raw & 0x7fff) + branch; + value_offset = index + .checked_mul(3) + .ok_or_else(|| Error::Invalid("Huffman node overflow".to_owned()))?; + node = range(&self.tree, value_offset, 3)?; + raw = u16::from(node[0]) | (u16::from(node[1]) << 8); + if raw & 0x8000 != 0 { + break raw & 0x7fff; + } + mask <<= 1; + bits += 1; + } + } + }; + if bits == 0 || bits > available { + return invalid("compressed stream ends inside a Huffman code"); + } + bit_buffer >>= bits; + available -= bits; + consumed_bits = consumed_bits + .checked_add(usize::from(bits)) + .ok_or_else(|| Error::Invalid("consumed bit count overflow".to_owned()))?; + + let kind = symbol & 0x300; + let value = usize::from(symbol & 0xff); + match kind { + 0 => { + output[output_pos] = value as u8; + output_pos += 1; + } + 0x100 => { + if prefix > 0xff { + return invalid("compressed prefix exceeds 16 bits"); + } + prefix = if prefix == 0 { + value + } else { + value | (prefix << 8) + }; + } + 0x200 => { + if prefix == 0 { + prefix = 1; + } + let count = value + .checked_mul(prefix) + .ok_or_else(|| Error::Invalid("repeat count overflow".to_owned()))?; + if !matches!(value, 1 | 2 | 4) + || value > output_pos + || output_pos + .checked_add(count) + .is_none_or(|end| end > output_size) + { + return invalid("invalid compressed repeated-pattern command"); + } + let pattern = output[output_pos - value..output_pos].to_vec(); + for chunk in output[output_pos..output_pos + count].chunks_exact_mut(value) { + chunk.copy_from_slice(&pattern); + } + output_pos += count; + prefix = 0; + } + 0x300 => { + let length = value; + let distance = prefix.checked_add(length).ok_or_else(|| { + Error::Invalid("back-reference distance overflow".to_owned()) + })?; + if distance > output_pos + || output_pos + .checked_add(length) + .is_none_or(|end| end > output_size) + { + return invalid("invalid compressed back-reference"); + } + let source_start = output_pos - distance; + output.copy_within(source_start..source_start + length, output_pos); + output_pos += length; + prefix = 0; + } + _ => unreachable!("masked Huffman symbol kind"), + } + } + if consumed_bits.div_ceil(8) != source.len() { + return invalid(format!( + "compressed input consumption mismatch: used=0x{:x}, size=0x{:x}", + consumed_bits.div_ceil(8), + source.len() + )); + } + Ok(output) + } +} + +/// Apply the native word transform and optional AES-256-CBC decryption. +pub fn transform_segment( + data: &[u8], + seed: u32, + aes_key: &[u8; 32], + decrypt_aes: bool, +) -> Result> { + let mut transformed = data.to_vec(); + let mut state = seed; + let mut left = 0xe34e_ac63_u32; + let mut right = 0x07b4_8238_u32; + for (index, chunk) in transformed.chunks_exact_mut(4).enumerate() { + let index32 = u32::try_from(index) + .map_err(|_| Error::Invalid("segment word index exceeds u32".to_owned()))?; + left = state + .wrapping_add(0x72f6_fcbe) + .wrapping_add(left.wrapping_add(0x4f8b_1bca).wrapping_mul(left)) + .wrapping_shr(index32.wrapping_mul(index32) & 0x0f); + right = state + .wrapping_sub(0x71b6_a98d) + .wrapping_add(right.wrapping_sub(0x1605_a81c).wrapping_mul(right)) + .wrapping_shl(index32 & 7); + state = left ^ right; + let bytes: [u8; 4] = chunk + .try_into() + .map_err(|_| Error::Invalid("invalid transformed word".to_owned()))?; + let mut value = u32::from_le_bytes(bytes); + value = value.wrapping_add(0xb43b_9baf_u32.wrapping_mul(index32 & 0x0d)); + value ^= 0xaf57_f7fb_u32.wrapping_mul(index32 & 3); + value = value.wrapping_sub(state) ^ state; + chunk.copy_from_slice(&value.to_le_bytes()); + } + + if decrypt_aes { + let cipher = Aes256::new_from_slice(aes_key) + .map_err(|_| Error::Invalid("invalid AES-256 key length".to_owned()))?; + let aligned_size = transformed.len() & !0x0f; + let mut previous = [0_u8; 16]; + for chunk in transformed[..aligned_size].chunks_exact_mut(16) { + let mut ciphertext = [0_u8; 16]; + ciphertext.copy_from_slice(chunk); + cipher.decrypt_block(Block::::from_mut_slice(chunk)); + for (byte, prior) in chunk.iter_mut().zip(previous) { + *byte ^= prior; + } + previous = ciphertext; + } + } + Ok(transformed) +} + +/// Decode one complete protector container into its flat output buffer. +/// +/// This is the static equivalent of the decoder entrypoint embedded in Stage +/// 2 and in each nested interpreter module. +pub fn decode_container( + data: &[u8], + config: &Module9bConfig, + expected_size: usize, +) -> Result> { + let header = ContainerHeader::parse(data, 0, config.container_seed)?; + let header_size = usize::try_from(header.output_size) + .map_err(|_| Error::Invalid("container output size exceeds usize".to_owned()))?; + if header_size != expected_size { + return invalid(format!( + "container output size 0x{header_size:x} != expected 0x{expected_size:x}" + )); + } + let decoder = HuffmanLzDecoder::new(&header.tree)?; + let decrypt_aes = !(config.skip_aes || header.skip_aes); + let mut output = vec![0_u8; expected_size]; + + for (segment_index, encoded) in header.segments.iter().enumerate() { + let start = header + .start + .checked_add(encoded.offset as usize) + .ok_or_else(|| Error::Invalid("encoded segment start overflow".to_owned()))?; + let encoded_data = range(data, start, encoded.size as usize)?; + let transformed = transform_segment( + encoded_data, + config.container_seed, + &config.aes_key, + decrypt_aes, + )?; + if transformed.len() < 16 { + return invalid(format!( + "decoded segment {segment_index} is shorter than its header" + )); + } + let base_offset = read_u32(&transformed, 0)? as usize; + let writer_count = read_u32(&transformed, 4)? as usize; + let table_offset = read_u32(&transformed, 8)? as usize; + let data_offset = read_u32(&transformed, 12)? as usize; + let table_size = writer_count + .checked_mul(16) + .ok_or_else(|| Error::Invalid("writer table size overflow".to_owned()))?; + let table_end = table_offset + .checked_add(table_size) + .ok_or_else(|| Error::Invalid("writer table end overflow".to_owned()))?; + if table_end > transformed.len() || data_offset > transformed.len() { + return invalid(format!( + "decoded segment {segment_index} has invalid writer offsets" + )); + } + + let mut data_cursor = data_offset; + for writer_index in 0..writer_count { + let record = + table_offset + .checked_add(writer_index.checked_mul(16).ok_or_else(|| { + Error::Invalid("writer record offset overflow".to_owned()) + })?) + .ok_or_else(|| Error::Invalid("writer record offset overflow".to_owned()))?; + let output_offset = read_u32(&transformed, record)? as usize; + let output_size = read_u32(&transformed, record + 4)? as usize; + let encoded_size = read_u32(&transformed, record + 8)? as usize; + let reserved = read_u32(&transformed, record + 12)?; + let encoded_end = data_cursor + .checked_add(encoded_size) + .ok_or_else(|| Error::Invalid("writer data end overflow".to_owned()))?; + if reserved != 0 || encoded_end > transformed.len() { + return invalid(format!( + "segment {segment_index} writer {writer_index} has invalid bounds" + )); + } + let source = &transformed[data_cursor..encoded_end]; + let decoded = if encoded_size == output_size { + None + } else { + Some(decoder.decode(source, output_size)?) + }; + let decoded = decoded.as_deref().unwrap_or(source); + let target = base_offset + .checked_add(output_offset) + .ok_or_else(|| Error::Invalid("writer target offset overflow".to_owned()))?; + let target_end = target + .checked_add(decoded.len()) + .ok_or_else(|| Error::Invalid("writer target end overflow".to_owned()))?; + let destination = output.get_mut(target..target_end).ok_or_else(|| { + Error::Invalid(format!( + "segment {segment_index} writer {writer_index} target is out of range" + )) + })?; + destination.copy_from_slice(decoded); + data_cursor = encoded_end; + } + } + Ok(output) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn aes_mix_columns_matches_fips_example() { + let input = [ + 0xdb, 0x13, 0x53, 0x45, 0xf2, 0x0a, 0x22, 0x5c, 0x01, 0x01, 0x01, 0x01, 0xc6, 0xc6, + 0xc6, 0xc6, + ]; + assert_eq!( + mix_columns(input), + [ + 0x8e, 0x4d, 0xa1, 0xbc, 0x9f, 0xdc, 0x58, 0x9d, 0x01, 0x01, 0x01, 0x01, 0xc6, 0xc6, + 0xc6, 0xc6, + ] + ); + } + + #[test] + fn descriptor_rejects_truncated_input() { + assert!(ProtectedDescriptor::decrypt(&[0_u8; 16], 1).is_err()); + } +} diff --git a/senbei-android-elf/Cargo.toml b/senbei-android-elf/Cargo.toml new file mode 100644 index 0000000..6d84996 --- /dev/null +++ b/senbei-android-elf/Cargo.toml @@ -0,0 +1,19 @@ +[package] +name = "senbei-android-elf" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +description = "AArch64 ELF restoration for Senbei Android" + +[dependencies] +memmap2.workspace = true +serde.workspace = true +serde_json.workspace = true +sha2.workspace = true +tempfile.workspace = true +thiserror.workspace = true +senbei-android-crypto.workspace = true + +[lints] +workspace = true diff --git a/senbei-android-elf/src/artifact.rs b/senbei-android-elf/src/artifact.rs new file mode 100644 index 0000000..464e51d --- /dev/null +++ b/senbei-android-elf/src/artifact.rs @@ -0,0 +1,105 @@ +use std::collections::BTreeMap; +use std::path::{Path, PathBuf}; + +use serde_json::Value; + +use crate::error::{Error, Result, invalid}; + +const REQUIRED_IDS: [u32; 3] = [0x9b, 0x9d, 0x9e]; + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct Artifact { + pub path: PathBuf, + pub size: u64, +} + +pub(crate) fn load_artifacts(index_path: &Path) -> Result> { + let text = std::fs::read_to_string(index_path) + .map_err(|error| Error::io("read module index", index_path, error))?; + let document: Value = serde_json::from_str(&text)?; + let root = index_path.parent().unwrap_or_else(|| Path::new(".")); + let mut result = BTreeMap::new(); + + if let Some(items) = document.get("module_registry").and_then(Value::as_array) { + for item in items { + let Some(command_id) = item.get("command_id").and_then(Value::as_u64) else { + continue; + }; + let command_id = u32::try_from(command_id) + .map_err(|_| Error::Invalid("module command ID exceeds u32".to_owned()))?; + if !REQUIRED_IDS.contains(&command_id) { + continue; + } + let Some(path) = item.get("image_path").and_then(Value::as_str) else { + continue; + }; + let size = item + .get("size") + .and_then(Value::as_u64) + .ok_or_else(|| Error::Invalid(format!("module 0x{command_id:02X} lacks size")))?; + result.insert( + command_id, + Artifact { + path: root.join(path), + size, + }, + ); + } + } + if let Some(streams) = document.get("streams").and_then(Value::as_array) { + for stream in streams { + let Some(records) = stream.get("records").and_then(Value::as_array) else { + continue; + }; + for record in records { + let Some(command_id) = record.get("command_id").and_then(Value::as_u64) else { + continue; + }; + let command_id = u32::try_from(command_id) + .map_err(|_| Error::Invalid("record command ID exceeds u32".to_owned()))?; + if !REQUIRED_IDS.contains(&command_id) { + continue; + } + let Some(image) = record.get("image") else { + continue; + }; + let Some(path) = image.get("path").and_then(Value::as_str) else { + continue; + }; + let size = image.get("size").and_then(Value::as_u64).ok_or_else(|| { + Error::Invalid(format!("record 0x{command_id:02X} lacks image size")) + })?; + result.insert( + command_id, + Artifact { + path: root.join(path), + size, + }, + ); + } + } + } + + let missing = REQUIRED_IDS + .iter() + .filter(|id| !result.contains_key(id)) + .map(|id| format!("0x{id:02X}")) + .collect::>(); + if !missing.is_empty() { + return invalid(format!( + "module index lacks required IDs: {}", + missing.join(", ") + )); + } + for (&command_id, artifact) in &result { + let metadata = std::fs::metadata(&artifact.path) + .map_err(|error| Error::io("inspect artifact", &artifact.path, error))?; + if !metadata.is_file() || metadata.len() != artifact.size { + return invalid(format!( + "invalid artifact for module 0x{command_id:02X}: {}", + artifact.path.display() + )); + } + } + Ok(result) +} diff --git a/senbei-android-elf/src/error.rs b/senbei-android-elf/src/error.rs new file mode 100644 index 0000000..058db5e --- /dev/null +++ b/senbei-android-elf/src/error.rs @@ -0,0 +1,35 @@ +use std::path::{Path, PathBuf}; + +/// ELF restoration failure. +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("{action} `{path}`: {source}")] + Io { + action: &'static str, + path: PathBuf, + #[source] + source: std::io::Error, + }, + #[error("cannot parse module index: {0}")] + Json(#[from] serde_json::Error), + #[error(transparent)] + Crypto(#[from] senbei_android_crypto::Error), + #[error("{0}")] + Invalid(String), +} + +impl Error { + pub(crate) fn io(action: &'static str, path: &Path, source: std::io::Error) -> Self { + Self::Io { + action, + path: path.to_path_buf(), + source, + } + } +} + +pub(crate) type Result = std::result::Result; + +pub(crate) fn invalid(message: impl Into) -> Result { + Err(Error::Invalid(message.into())) +} diff --git a/senbei-android-elf/src/hash.rs b/senbei-android-elf/src/hash.rs new file mode 100644 index 0000000..550e012 --- /dev/null +++ b/senbei-android-elf/src/hash.rs @@ -0,0 +1,107 @@ +use crate::error::{Error, Result, invalid}; + +#[must_use] +pub(crate) fn elf_hash(name: &[u8]) -> u32 { + let mut value = 0_u32; + for &byte in name { + value = value.wrapping_shl(4).wrapping_add(u32::from(byte)); + let high = value & 0xf000_0000; + if high != 0 { + value ^= high >> 24; + value &= !high; + } + } + value +} + +#[must_use] +pub(crate) fn gnu_hash(name: &[u8]) -> u32 { + name.iter().fold(5381_u32, |value, &byte| { + value.wrapping_mul(33).wrapping_add(u32::from(byte)) + }) +} + +pub(crate) fn build_sysv_hash(names: &[Vec]) -> Result> { + if names.len() < 2 { + return invalid("dynamic symbol table is unexpectedly empty"); + } + let bucket_count = names.len(); + let symbol_count = names.len(); + let mut buckets = vec![0_u32; bucket_count]; + let mut chains = vec![0_u32; symbol_count]; + for (symbol_index, name) in names.iter().enumerate().skip(1) { + let bucket_index = elf_hash(name) as usize % bucket_count; + let symbol_index32 = u32::try_from(symbol_index) + .map_err(|_| Error::Invalid("dynamic symbol index exceeds u32".to_owned()))?; + if buckets[bucket_index] == 0 { + buckets[bucket_index] = symbol_index32; + continue; + } + let mut chain_index = buckets[bucket_index] as usize; + while chains[chain_index] != 0 { + chain_index = chains[chain_index] as usize; + } + chains[chain_index] = symbol_index32; + } + let mut output = Vec::with_capacity((2 + bucket_count + symbol_count) * 4); + output.extend_from_slice( + &u32::try_from(bucket_count) + .map_err(|_| Error::Invalid("SysV bucket count exceeds u32".to_owned()))? + .to_le_bytes(), + ); + output.extend_from_slice( + &u32::try_from(symbol_count) + .map_err(|_| Error::Invalid("SysV symbol count exceeds u32".to_owned()))? + .to_le_bytes(), + ); + for value in buckets.into_iter().chain(chains) { + output.extend_from_slice(&value.to_le_bytes()); + } + Ok(output) +} + +pub(crate) fn build_gnu_hash(names: &[Vec]) -> Result> { + let hashes = names + .iter() + .skip(1) + .map(|name| gnu_hash(name)) + .collect::>(); + if hashes.is_empty() { + return invalid("GNU hash requires at least one dynamic symbol"); + } + let bloom_shift = 5_u32; + let mut bloom_word = 0_u64; + for &value in &hashes { + bloom_word |= 1_u64 << (value & 63); + bloom_word |= 1_u64 << ((value >> bloom_shift) & 63); + } + let mut chains = hashes + .into_iter() + .map(|value| value & !1) + .collect::>(); + let last = chains + .last_mut() + .ok_or_else(|| Error::Invalid("GNU hash chain is empty".to_owned()))?; + *last |= 1; + let mut output = Vec::with_capacity(28 + chains.len() * 4); + for value in [1_u32, 1, 1, bloom_shift] { + output.extend_from_slice(&value.to_le_bytes()); + } + output.extend_from_slice(&bloom_word.to_le_bytes()); + output.extend_from_slice(&1_u32.to_le_bytes()); + for value in chains { + output.extend_from_slice(&value.to_le_bytes()); + } + Ok(output) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn standard_elf_hash_is_stable() { + assert_eq!(elf_hash(b"printf"), 0x0779_05a6); + assert_eq!(gnu_hash(b"printf"), 0x156b_2bb8); + } +} diff --git a/senbei-android-elf/src/layout.rs b/senbei-android-elf/src/layout.rs new file mode 100644 index 0000000..5ffc7e9 --- /dev/null +++ b/senbei-android-elf/src/layout.rs @@ -0,0 +1,301 @@ +use crate::error::{Error, Result, invalid}; + +pub(crate) const SHT_NOBITS: u32 = 8; +pub(crate) const SHT_LOUSER: u32 = 0x8000_0000; +pub(crate) const SHF_ALLOC: u64 = 2; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct LoadSegment { + pub offset: u64, + pub virtual_address: u64, + pub file_size: u64, + pub memory_size: u64, + pub flags: u32, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct SectionHeader { + pub name: u32, + pub section_type: u32, + pub flags: u64, + pub address: u64, + pub offset: u64, + pub size: u64, + pub link: u32, + pub info: u32, + pub alignment: u64, + pub entry_size: u64, +} + +impl SectionHeader { + pub const SIZE: usize = 0x40; + + fn parse(data: &[u8], offset: usize) -> Result { + Ok(Self { + name: read_u32(data, offset)?, + section_type: read_u32(data, offset + 4)?, + flags: read_u64(data, offset + 8)?, + address: read_u64(data, offset + 0x10)?, + offset: read_u64(data, offset + 0x18)?, + size: read_u64(data, offset + 0x20)?, + link: read_u32(data, offset + 0x28)?, + info: read_u32(data, offset + 0x2c)?, + alignment: read_u64(data, offset + 0x30)?, + entry_size: read_u64(data, offset + 0x38)?, + }) + } + + pub fn encode(self) -> [u8; Self::SIZE] { + let mut output = [0_u8; Self::SIZE]; + output[0..4].copy_from_slice(&self.name.to_le_bytes()); + output[4..8].copy_from_slice(&self.section_type.to_le_bytes()); + output[8..0x10].copy_from_slice(&self.flags.to_le_bytes()); + output[0x10..0x18].copy_from_slice(&self.address.to_le_bytes()); + output[0x18..0x20].copy_from_slice(&self.offset.to_le_bytes()); + output[0x20..0x28].copy_from_slice(&self.size.to_le_bytes()); + output[0x28..0x2c].copy_from_slice(&self.link.to_le_bytes()); + output[0x2c..0x30].copy_from_slice(&self.info.to_le_bytes()); + output[0x30..0x38].copy_from_slice(&self.alignment.to_le_bytes()); + output[0x38..0x40].copy_from_slice(&self.entry_size.to_le_bytes()); + output + } +} + +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct ElfLayout { + pub entrypoint: u64, + pub program_headers: Vec, + pub section_headers: Vec, + pub section_name_index: usize, + pub private_section_index: usize, +} + +impl ElfLayout { + pub fn parse(data: &[u8], require_private: bool) -> Result { + let ident = slice(data, 0, 6)?; + if ident[..4] != *b"\x7fELF" || ident[4] != 2 || ident[5] != 1 { + return invalid("input is not a little-endian ELF64 file"); + } + if read_u16(data, 0x12)? != 0xb7 { + return invalid("input is not an AArch64 ELF"); + } + let entrypoint = read_u64(data, 0x18)?; + let program_header_offset = usize_from_u64(read_u64(data, 0x20)?, "program header offset")?; + let section_header_offset = usize_from_u64(read_u64(data, 0x28)?, "section header offset")?; + let program_header_size = usize::from(read_u16(data, 0x36)?); + let program_header_count = usize::from(read_u16(data, 0x38)?); + let section_header_size = usize::from(read_u16(data, 0x3a)?); + let section_header_count = usize::from(read_u16(data, 0x3c)?); + let section_name_index = usize::from(read_u16(data, 0x3e)?); + if program_header_size != 0x38 || section_header_size != SectionHeader::SIZE { + return invalid("unexpected ELF program/section header size"); + } + + let mut program_headers = Vec::new(); + for index in 0..program_header_count { + let offset = checked_index(program_header_offset, index, program_header_size)?; + if read_u32(data, offset)? != 1 { + continue; + } + let segment = LoadSegment { + flags: read_u32(data, offset + 4)?, + offset: read_u64(data, offset + 8)?, + virtual_address: read_u64(data, offset + 0x10)?, + file_size: read_u64(data, offset + 0x20)?, + memory_size: read_u64(data, offset + 0x28)?, + }; + let file_end = segment + .offset + .checked_add(segment.file_size) + .ok_or_else(|| Error::Invalid(format!("PT_LOAD {index} file range overflow")))?; + if file_end > data.len() as u64 { + return invalid(format!("PT_LOAD {index} exceeds input file")); + } + program_headers.push(segment); + } + if program_headers.is_empty() { + return invalid("input ELF contains no PT_LOAD segments"); + } + + let mut section_headers = Vec::with_capacity(section_header_count); + for index in 0..section_header_count { + let offset = checked_index(section_header_offset, index, section_header_size)?; + section_headers.push(SectionHeader::parse(data, offset)?); + } + if section_name_index >= section_headers.len() { + return invalid("ELF section-name index is out of range"); + } + let private = section_headers + .iter() + .enumerate() + .filter_map(|(index, section)| (section.section_type == SHT_LOUSER).then_some(index)) + .collect::>(); + let private_section_index = match private.as_slice() { + [index] => *index, + [] if !require_private => usize::MAX, + _ => { + return invalid(format!( + "expected {} SHT_LOUSER section, found {}", + if require_private { + "one" + } else { + "at most one" + }, + private.len() + )); + } + }; + Ok(Self { + entrypoint, + program_headers, + section_headers, + section_name_index, + private_section_index, + }) + } + + pub fn private_section(&self) -> Result { + self.section_headers + .get(self.private_section_index) + .copied() + .ok_or_else(|| Error::Invalid("ELF has no private section".to_owned())) + } + + pub fn load_end(&self) -> Result { + self.program_headers + .iter() + .map(|segment| { + segment + .virtual_address + .checked_add(segment.memory_size) + .ok_or_else(|| Error::Invalid("PT_LOAD memory end overflow".to_owned())) + }) + .collect::>>()? + .into_iter() + .max() + .ok_or_else(|| Error::Invalid("ELF has no PT_LOAD memory range".to_owned())) + } + + pub fn file_load_end(&self) -> Result { + self.program_headers + .iter() + .map(|segment| { + segment + .offset + .checked_add(segment.file_size) + .ok_or_else(|| Error::Invalid("PT_LOAD file end overflow".to_owned())) + }) + .collect::>>()? + .into_iter() + .max() + .ok_or_else(|| Error::Invalid("ELF has no PT_LOAD file range".to_owned())) + } + + pub fn section_names(&self, data: &[u8]) -> Result> { + let table = self.section_headers[self.section_name_index]; + let strings = slice_u64(data, table.offset, table.size)?; + self.section_headers + .iter() + .map(|section| { + let offset = section.name as usize; + if offset >= strings.len() { + return Ok(String::new()); + } + let end = strings[offset..] + .iter() + .position(|&byte| byte == 0) + .map_or(strings.len(), |length| offset + length); + Ok(String::from_utf8_lossy(&strings[offset..end]).into_owned()) + }) + .collect() + } + + pub fn file_offset_to_virtual_address(&self, offset: u64, size: u64) -> Result { + let end = offset + .checked_add(size) + .ok_or_else(|| Error::Invalid("file range overflow".to_owned()))?; + for segment in &self.program_headers { + let segment_end = segment + .offset + .checked_add(segment.file_size) + .ok_or_else(|| Error::Invalid("PT_LOAD file range overflow".to_owned()))?; + if segment.offset <= offset && end <= segment_end { + return segment + .virtual_address + .checked_add(offset - segment.offset) + .ok_or_else(|| Error::Invalid("virtual address overflow".to_owned())); + } + } + invalid(format!( + "file range 0x{offset:x}..0x{end:x} is not in PT_LOAD" + )) + } +} + +pub(crate) fn slice(data: &[u8], offset: usize, size: usize) -> Result<&[u8]> { + let end = offset + .checked_add(size) + .ok_or_else(|| Error::Invalid("byte range overflow".to_owned()))?; + data.get(offset..end).ok_or_else(|| { + Error::Invalid(format!( + "byte range 0x{offset:x}..0x{end:x} is out of bounds" + )) + }) +} + +pub(crate) fn slice_u64(data: &[u8], offset: u64, size: u64) -> Result<&[u8]> { + slice( + data, + usize_from_u64(offset, "file offset")?, + usize_from_u64(size, "file size")?, + ) +} + +pub(crate) fn read_u16(data: &[u8], offset: usize) -> Result { + let bytes: [u8; 2] = slice(data, offset, 2)? + .try_into() + .map_err(|_| Error::Invalid("invalid u16 range".to_owned()))?; + Ok(u16::from_le_bytes(bytes)) +} + +pub(crate) fn read_u32(data: &[u8], offset: usize) -> Result { + let bytes: [u8; 4] = slice(data, offset, 4)? + .try_into() + .map_err(|_| Error::Invalid("invalid u32 range".to_owned()))?; + Ok(u32::from_le_bytes(bytes)) +} + +pub(crate) fn read_u64(data: &[u8], offset: usize) -> Result { + let bytes: [u8; 8] = slice(data, offset, 8)? + .try_into() + .map_err(|_| Error::Invalid("invalid u64 range".to_owned()))?; + Ok(u64::from_le_bytes(bytes)) +} + +pub(crate) fn read_i64(data: &[u8], offset: usize) -> Result { + let bytes: [u8; 8] = slice(data, offset, 8)? + .try_into() + .map_err(|_| Error::Invalid("invalid i64 range".to_owned()))?; + Ok(i64::from_le_bytes(bytes)) +} + +pub(crate) fn usize_from_u64(value: u64, field: &str) -> Result { + usize::try_from(value).map_err(|_| Error::Invalid(format!("{field} 0x{value:x} exceeds usize"))) +} + +pub(crate) fn checked_index(base: usize, index: usize, stride: usize) -> Result { + index + .checked_mul(stride) + .and_then(|value| base.checked_add(value)) + .ok_or_else(|| Error::Invalid("table index overflow".to_owned())) +} + +pub(crate) fn align_up(value: u64, alignment: u64) -> Result { + if alignment == 0 || !alignment.is_power_of_two() { + return invalid(format!("invalid alignment {alignment}")); + } + value + .checked_add(alignment - 1) + .map(|aligned| aligned & !(alignment - 1)) + .ok_or_else(|| Error::Invalid("alignment overflow".to_owned())) +} diff --git a/senbei-android-elf/src/lib.rs b/senbei-android-elf/src/lib.rs new file mode 100644 index 0000000..4c57c0a --- /dev/null +++ b/senbei-android-elf/src/lib.rs @@ -0,0 +1,10 @@ +//! Static restoration of the current protected AArch64 `libil2cpp.so`. + +mod artifact; +mod error; +mod hash; +mod layout; +mod restore; + +pub use error::Error; +pub use restore::{RestoreOptions, RestoreReport, restore_libil2cpp}; diff --git a/senbei-android-elf/src/restore.rs b/senbei-android-elf/src/restore.rs new file mode 100644 index 0000000..e311318 --- /dev/null +++ b/senbei-android-elf/src/restore.rs @@ -0,0 +1,1482 @@ +use std::collections::{BTreeMap, HashMap, HashSet}; +use std::fs::File; +use std::io::{Read, Seek, SeekFrom, Write}; +use std::path::{Path, PathBuf}; +use std::time::Instant; + +use memmap2::{Mmap, MmapMut, MmapOptions}; +use senbei_android_crypto::{ + ContainerHeader, HuffmanLzDecoder, Module9bConfig, ProtectedDescriptor, transform_segment, +}; +use serde::Serialize; +use sha2::{Digest, Sha256}; +use tempfile::NamedTempFile; + +use crate::artifact::load_artifacts; +use crate::error::{Error, Result, invalid}; +use crate::hash::{build_gnu_hash, build_sysv_hash}; +use crate::layout::{ + ElfLayout, SHF_ALLOC, SHT_LOUSER, SHT_NOBITS, SectionHeader, align_up, read_i64, read_u32, + read_u64, slice, slice_u64, usize_from_u64, +}; + +const CHUNK_SIZE: usize = 16 * 1024 * 1024; +const ELF64_SYMBOL_SIZE: usize = 0x18; +const ELF64_RELA_SIZE: usize = 0x18; +const R_AARCH64_ABS64: u32 = 0x101; +const R_AARCH64_GLOB_DAT: u32 = 0x401; +const R_AARCH64_JUMP_SLOT: u32 = 0x402; +const R_AARCH64_RELATIVE: u32 = 0x403; +const VER_NDX_GLOBAL: u16 = 1; + +const DT_PLTRELSZ: u64 = 2; +const DT_HASH: u64 = 4; +const DT_STRTAB: u64 = 5; +const DT_SYMTAB: u64 = 6; +const DT_RELA: u64 = 7; +const DT_RELASZ: u64 = 8; +const DT_STRSZ: u64 = 10; +const DT_JMPREL: u64 = 23; +const DT_GNU_HASH: u64 = 0x6fff_fef5; +const DT_VERSYM: u64 = 0x6fff_fff0; +const DT_RELACOUNT: u64 = 0x6fff_fff9; +const DT_VERNEED: u64 = 0x6fff_fffe; + +/// Inputs and optional diagnostics for one `libil2cpp.so` restoration. +#[derive(Debug, Clone)] +pub struct RestoreOptions { + pub input: PathBuf, + pub output: PathBuf, + pub index: PathBuf, + pub dump_auxiliary: Option, + pub outer_only: bool, + pub preserve_entrypoint: bool, + /// Print per-phase progress lines to stderr. Off for quiet/batch drivers. + pub verbose: bool, +} + +/// Container decoding counters. +#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize)] +pub struct DecodeStatistics { + pub segments: usize, + pub writers: usize, + pub compressed_writers: usize, + pub encoded_bytes: u64, + pub decoded_bytes: u64, + pub file_bytes_written: u64, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct StaticConfigReport { + pub header_seed: String, + pub container_seed: String, + pub aes_key_sha256: String, + pub schedule_offset: String, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct DescriptorReport { + pub command_id: String, + pub flags: String, + pub outer_offset: String, + pub outer_expected_size: String, + pub auxiliary_offset: String, + pub auxiliary_expected_size: String, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct HiddenSymbolReport { + pub patch_blob_size: u32, + pub patched_symbols: u32, + pub copied_strings: usize, + pub secondary_record_count: u32, + pub first_target_index: u32, + pub last_target_index: u32, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct PlacementReport { + pub offset: u64, + pub size: usize, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct ElfMaterializationReport { + pub hidden_symbols: HiddenSymbolReport, + pub old_symbol_count: usize, + pub auxiliary_symbol_count: u32, + pub appended_symbols: usize, + pub new_symbol_count: usize, + pub old_dynstr_size: usize, + pub auxiliary_dynstr_size: u32, + pub new_dynstr_size: usize, + pub rela_dyn_count: usize, + pub rela_plt_count: usize, + pub relative_prefix_count: usize, + pub metadata_start: u64, + pub metadata_end: u64, + pub metadata_capacity_end: u64, + pub metadata_slack: u64, + pub placements: BTreeMap, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct CleaningReport { + pub private_section_index: usize, + pub private_offset: u64, + pub private_size: u64, + pub input_entrypoint: u64, + pub output_entrypoint: u64, + pub output_section_count: usize, + pub retained_sections: Vec, + pub section_header_offset: u64, +} + +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct ValidationReport { + pub format: String, + pub machine: String, + pub sections: usize, + pub segments: usize, + pub dynamic_symbols: usize, + pub dynamic_relocations: usize, + pub pltgot_relocations: usize, + pub has_louser: bool, +} + +/// Machine-readable result of the restoration. +#[derive(Debug, Clone, PartialEq, Serialize)] +pub struct RestoreReport { + pub input: String, + pub input_sha256: String, + pub output: String, + pub output_sha256: String, + pub output_size: u64, + pub module_index: String, + pub static_config: StaticConfigReport, + pub descriptor: DescriptorReport, + pub primary: DecodeStatistics, + pub auxiliary: Option, + pub elf_materialization: Option, + pub cleaning: CleaningReport, + pub validation: ValidationReport, + pub elapsed_seconds: f64, +} + +fn map_read_only(file: &File, path: &Path) -> Result { + // SAFETY: the mapping is read-only and `file` remains open for the mapping's + // lifetime. The restoration never mutates or truncates the mapped source. + unsafe { MmapOptions::new().map(file) }.map_err(|error| Error::io("map", path, error)) +} + +fn map_mut(file: &File, length: usize, path: &Path) -> Result { + // SAFETY: `length` is set on the private temporary output immediately + // before this call. No other handle mutates or truncates it while mapped. + unsafe { MmapOptions::new().len(length).map_mut(file) } + .map_err(|error| Error::io("map temporary output", path, error)) +} + +fn read_file(path: &Path) -> Result> { + std::fs::read(path).map_err(|error| Error::io("read", path, error)) +} + +fn sha256_bytes(data: &[u8]) -> String { + let mut digest = Sha256::new(); + digest.update(data); + format!("{:x}", digest.finalize()) +} + +fn sha256_file(path: &Path) -> Result { + let mut file = File::open(path).map_err(|error| Error::io("open", path, error))?; + let mut digest = Sha256::new(); + let mut buffer = vec![0_u8; CHUNK_SIZE]; + loop { + let read = file + .read(&mut buffer) + .map_err(|error| Error::io("hash", path, error))?; + if read == 0 { + break; + } + digest.update(&buffer[..read]); + } + Ok(format!("{:x}", digest.finalize())) +} + +fn copy_range(source: &[u8], output: &mut File, size: usize, path: &Path) -> Result<()> { + for chunk in source[..size].chunks(CHUNK_SIZE) { + output + .write_all(chunk) + .map_err(|error| Error::io("write temporary output", path, error))?; + } + Ok(()) +} + +fn checked_add(base: usize, value: usize, field: &str) -> Result { + base.checked_add(value) + .ok_or_else(|| Error::Invalid(format!("{field} overflow"))) +} + +struct FileLayoutWriter<'a> { + output: &'a mut [u8], + layout: &'a ElfLayout, + load_end: u64, +} + +impl FileLayoutWriter<'_> { + fn write(&mut self, virtual_address: u64, data: &[u8]) -> Result { + let data_len = u64::try_from(data.len()) + .map_err(|_| Error::Invalid("decoded write length exceeds u64".to_owned()))?; + let end = virtual_address + .checked_add(data_len) + .ok_or_else(|| Error::Invalid("decoded write range overflow".to_owned()))?; + if end > self.load_end { + return invalid(format!( + "decoded write 0x{virtual_address:x}..0x{end:x} exceeds target load image" + )); + } + let mut written = 0_u64; + let mut covered_memory = 0_u64; + for segment in &self.layout.program_headers { + let memory_end = segment + .virtual_address + .checked_add(segment.memory_size) + .ok_or_else(|| Error::Invalid("PT_LOAD memory end overflow".to_owned()))?; + let overlap_start = virtual_address.max(segment.virtual_address); + let overlap_end = end.min(memory_end); + if overlap_start >= overlap_end { + continue; + } + covered_memory = covered_memory + .checked_add(overlap_end - overlap_start) + .ok_or_else(|| Error::Invalid("covered memory count overflow".to_owned()))?; + let file_end_va = segment + .virtual_address + .checked_add(segment.file_size) + .ok_or_else(|| Error::Invalid("PT_LOAD file VA end overflow".to_owned()))?; + let file_overlap_end = overlap_end.min(file_end_va); + if overlap_start < file_overlap_end { + let source_offset = + usize_from_u64(overlap_start - virtual_address, "write source offset")?; + let file_offset = usize_from_u64( + segment + .offset + .checked_add(overlap_start - segment.virtual_address) + .ok_or_else(|| Error::Invalid("write file offset overflow".to_owned()))?, + "write file offset", + )?; + let count = usize_from_u64(file_overlap_end - overlap_start, "write size")?; + let destination = self + .output + .get_mut(file_offset..file_offset + count) + .ok_or_else(|| { + Error::Invalid("decoded write exceeds temporary output".to_owned()) + })?; + destination.copy_from_slice(&data[source_offset..source_offset + count]); + written += count as u64; + } + } + if covered_memory != data_len { + return invalid(format!( + "decoded write 0x{virtual_address:x}..0x{end:x} is not covered by PT_LOAD memory" + )); + } + usize_from_u64(written, "written byte count") + } +} + +fn decode_container( + payload: &[u8], + header: &ContainerHeader, + config: &Module9bConfig, + verbose: bool, + mut writer: F, +) -> Result +where + F: FnMut(u64, &[u8]) -> Result, +{ + let decoder = HuffmanLzDecoder::new(&header.tree)?; + let mut statistics = DecodeStatistics { + segments: header.segments.len(), + ..DecodeStatistics::default() + }; + let decrypt_aes = !(config.skip_aes || header.skip_aes); + for (segment_index, encoded) in header.segments.iter().enumerate() { + let start = checked_add( + header.start, + encoded.offset as usize, + "encoded segment start", + )?; + let encoded_data = slice(payload, start, encoded.size as usize)?; + let transformed = transform_segment( + encoded_data, + config.container_seed, + &config.aes_key, + decrypt_aes, + )?; + if transformed.len() < 16 { + return invalid(format!( + "decoded segment {segment_index} is shorter than its header" + )); + } + let base_offset = u64::from(read_u32(&transformed, 0)?); + let writer_count = read_u32(&transformed, 4)? as usize; + let table_offset = read_u32(&transformed, 8)? as usize; + let data_offset = read_u32(&transformed, 12)? as usize; + let table_end = table_offset + .checked_add( + writer_count + .checked_mul(16) + .ok_or_else(|| Error::Invalid("writer table size overflow".to_owned()))?, + ) + .ok_or_else(|| Error::Invalid("writer table end overflow".to_owned()))?; + if table_end > transformed.len() || data_offset > transformed.len() { + return invalid(format!( + "decoded segment {segment_index} has invalid writer offsets" + )); + } + let mut data_cursor = data_offset; + for writer_index in 0..writer_count { + let record = table_offset + writer_index * 16; + let output_offset = u64::from(read_u32(&transformed, record)?); + let output_size = read_u32(&transformed, record + 4)? as usize; + let encoded_size = read_u32(&transformed, record + 8)? as usize; + let reserved = read_u32(&transformed, record + 12)?; + let encoded_end = data_cursor + .checked_add(encoded_size) + .ok_or_else(|| Error::Invalid("writer data end overflow".to_owned()))?; + if reserved != 0 || encoded_end > transformed.len() { + return invalid(format!( + "segment {segment_index} writer {writer_index} has invalid bounds" + )); + } + let source = &transformed[data_cursor..encoded_end]; + let decoded = if encoded_size == output_size { + None + } else { + statistics.compressed_writers += 1; + Some(decoder.decode(source, output_size)?) + }; + let decoded_slice = decoded.as_deref().unwrap_or(source); + let target = base_offset + .checked_add(output_offset) + .ok_or_else(|| Error::Invalid("writer target address overflow".to_owned()))?; + statistics.file_bytes_written += writer(target, decoded_slice)? as u64; + statistics.writers += 1; + statistics.encoded_bytes += encoded_size as u64; + statistics.decoded_bytes += output_size as u64; + data_cursor = encoded_end; + } + if verbose { + eprintln!( + "[{current:02}/{total:02}] writers={writer_count} encoded=0x{size:x}", + current = segment_index + 1, + total = header.segments.len(), + size = encoded.size, + ); + } + } + Ok(statistics) +} + +fn read_c_string(data: &[u8], offset: usize, limit: usize) -> Result<&[u8]> { + if offset >= limit || limit > data.len() { + return invalid(format!("invalid string offset 0x{offset:x}/0x{limit:x}")); + } + let relative_end = data[offset..limit] + .iter() + .position(|&byte| byte == 0) + .ok_or_else(|| Error::Invalid(format!("unterminated string at 0x{offset:x}")))?; + Ok(&data[offset..offset + relative_end]) +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct AuxiliaryElfImage { + dynstr_offset: u32, + dynstr_size: u32, + dynsym_offset: u32, + dynsym_count: u32, + relocation1_offset: u32, + relocation1_count: u32, + relocation2_offset: u32, + relocation2_count: u32, +} + +impl AuxiliaryElfImage { + fn parse(data: &[u8]) -> Result { + if data.len() < 0x40 { + return invalid("decoded auxiliary ELF image is truncated"); + } + let mut words = [0_u32; 16]; + for (index, word) in words.iter_mut().enumerate() { + *word = read_u32(data, index * 4)?; + } + if [1, 3, 12, 13, 15] + .into_iter() + .any(|index| words[index] != 0) + { + return invalid("unexpected nonzero auxiliary ELF header field"); + } + if words[14] != 0xb7 { + return invalid(format!( + "unexpected auxiliary ELF machine 0x{:x}", + words[14] + )); + } + let result = Self { + dynstr_offset: words[4], + dynstr_size: words[5], + dynsym_offset: words[6], + dynsym_count: words[7], + relocation1_offset: words[8], + relocation1_count: words[9], + relocation2_offset: words[10], + relocation2_count: words[11], + }; + if result.relocation1_offset != 0x40 { + return invalid("auxiliary relocation table does not follow its header"); + } + let relocation1_end = u64::from(result.relocation1_offset) + + u64::from(result.relocation1_count) * ELF64_RELA_SIZE as u64; + let relocation2_end = u64::from(result.relocation2_offset) + + u64::from(result.relocation2_count) * ELF64_RELA_SIZE as u64; + let dynsym_end = u64::from(result.dynsym_offset) + + u64::from(result.dynsym_count) * ELF64_SYMBOL_SIZE as u64; + let dynstr_end = u64::from(result.dynstr_offset) + u64::from(result.dynstr_size); + let expected_relocation2 = align_up(relocation1_end, 0x10)?; + let expected_dynsym = align_up(relocation2_end, 0x10)?; + if expected_relocation2 != u64::from(result.relocation2_offset) + || expected_dynsym != u64::from(result.dynsym_offset) + || dynsym_end != u64::from(result.dynstr_offset) + || dynstr_end != data.len() as u64 + { + return invalid(format!( + "auxiliary ELF layout mismatch: rela1_end=0x{relocation1_end:x}/rela2=0x{:x}, rela2_end=0x{relocation2_end:x}/dynsym=0x{:x}, dynsym_end=0x{dynsym_end:x}/dynstr=0x{:x}, dynstr_end=0x{dynstr_end:x}/size=0x{:x}", + result.relocation2_offset, + result.dynsym_offset, + result.dynstr_offset, + data.len() + )); + } + for (start, end) in [ + (relocation1_end, expected_relocation2), + (relocation2_end, expected_dynsym), + ] { + if slice_u64(data, start, end - start)? + .iter() + .any(|&byte| byte != 0) + { + return invalid(format!( + "auxiliary ELF alignment padding 0x{start:x}..0x{end:x} is nonzero" + )); + } + } + if result.dynsym_count < 2 { + return invalid("auxiliary dynamic symbol table is empty"); + } + if slice(data, result.dynsym_offset as usize, ELF64_SYMBOL_SIZE)? + .iter() + .any(|&byte| byte != 0) + { + return invalid("auxiliary dynamic symbol zero entry is not empty"); + } + Ok(result) + } +} + +fn restore_hidden_symbols( + output: &mut [u8], + dynsym: SectionHeader, + dynstr: SectionHeader, + patch_data: &[u8], +) -> Result<(Vec, Vec, HiddenSymbolReport)> { + if dynsym.entry_size != ELF64_SYMBOL_SIZE as u64 || dynsym.size % ELF64_SYMBOL_SIZE as u64 != 0 + { + return invalid("unexpected .dynsym entry layout"); + } + let symbol_count = usize_from_u64( + dynsym.size / ELF64_SYMBOL_SIZE as u64, + "dynamic symbol count", + )?; + let mut symbols = slice_u64(output, dynsym.offset, dynsym.size)?.to_vec(); + let mut strings = slice_u64(output, dynstr.offset, dynstr.size)?.to_vec(); + if patch_data.len() < 0x18 { + return invalid("0x9E symbol patch data is truncated"); + } + let blob_size = read_u32(patch_data, 0)?; + let secondary_record_count = read_u32(patch_data, 4)?; + let table_base = 8_usize; + let table_end = table_base + .checked_add(blob_size as usize) + .ok_or_else(|| Error::Invalid("0x9E patch blob end overflow".to_owned()))?; + if table_end > patch_data.len() { + return invalid("0x9E primary symbol patch blob exceeds its artifact"); + } + let count = read_u32(patch_data, table_base)?; + let symbol_offset = read_u32(patch_data, table_base + 4)? as usize; + let index_offset = read_u32(patch_data, table_base + 8)? as usize; + let string_offset = read_u32(patch_data, table_base + 12)? as usize; + let count_usize = count as usize; + if symbol_offset + .checked_add(count_usize * ELF64_SYMBOL_SIZE) + .is_none_or(|end| end > blob_size as usize) + || index_offset + .checked_add(count_usize * 4) + .is_none_or(|end| end > blob_size as usize) + || string_offset >= blob_size as usize + { + return invalid("0x9E symbol patch table has invalid offsets"); + } + let mut cursor = table_base + string_offset; + let mut patched_indices = HashSet::with_capacity(count_usize); + let mut copied_strings = 0_usize; + for index in 0..count_usize { + let source_offset = table_base + symbol_offset + index * ELF64_SYMBOL_SIZE; + let source_symbol = slice(patch_data, source_offset, ELF64_SYMBOL_SIZE)?; + let target_index = read_u32(patch_data, table_base + index_offset + index * 4)?; + if target_index == 0 || target_index as usize >= symbol_count { + return invalid(format!( + "0x9E target symbol index {target_index} is invalid" + )); + } + if !patched_indices.insert(target_index) { + return invalid(format!("0x9E patches symbol {target_index} more than once")); + } + let name = read_c_string(patch_data, cursor, table_end)?; + cursor += name.len() + 1; + let name_offset = read_u32(source_symbol, 0)? as usize; + if name_offset + .checked_add(name.len() + 1) + .is_none_or(|end| end > strings.len()) + { + return invalid(format!("0x9E symbol {target_index} name exceeds .dynstr")); + } + let existing = read_c_string(&strings, name_offset, strings.len())?; + if existing.is_empty() { + strings[name_offset..name_offset + name.len()].copy_from_slice(name); + strings[name_offset + name.len()] = 0; + copied_strings += 1; + } else if existing != name { + return invalid(format!( + "0x9E symbol {target_index} conflicts with existing .dynstr data" + )); + } + let target_offset = target_index as usize * ELF64_SYMBOL_SIZE; + symbols[target_offset..target_offset + ELF64_SYMBOL_SIZE].copy_from_slice(source_symbol); + } + let string_padding = patch_data.get(cursor..table_end).ok_or_else(|| { + Error::Invalid("0x9E symbol strings exceed the primary patch blob".to_owned()) + })?; + if string_padding.len() > 3 || string_padding.iter().any(|&byte| byte != 0) { + return invalid(format!( + "0x9E symbol strings have invalid padding at 0x{cursor:x}..0x{table_end:x}" + )); + } + let first_target_index = patched_indices + .iter() + .copied() + .min() + .ok_or_else(|| Error::Invalid("0x9E patch table is empty".to_owned()))?; + let last_target_index = patched_indices + .iter() + .copied() + .max() + .ok_or_else(|| Error::Invalid("0x9E patch table is empty".to_owned()))?; + Ok(( + symbols, + strings, + HiddenSymbolReport { + patch_blob_size: blob_size, + patched_symbols: count, + copied_strings, + secondary_record_count, + first_target_index, + last_target_index, + }, + )) +} + +fn dynamic_symbol_names(symbols: &[u8], strings: &[u8]) -> Result>> { + if symbols.len() % ELF64_SYMBOL_SIZE != 0 { + return invalid("dynamic symbol table is not entry-aligned"); + } + symbols + .chunks_exact(ELF64_SYMBOL_SIZE) + .map(|symbol| { + let name_offset = read_u32(symbol, 0)? as usize; + Ok(read_c_string(strings, name_offset, strings.len())?.to_vec()) + }) + .collect() +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +struct Rela { + offset: u64, + info: u64, + addend: i64, +} + +impl Rela { + fn parse(data: &[u8], offset: usize) -> Result { + Ok(Self { + offset: read_u64(data, offset)?, + info: read_u64(data, offset + 8)?, + addend: read_i64(data, offset + 0x10)?, + }) + } + + fn kind(self) -> u32 { + self.info as u32 + } + + fn symbol(self) -> u64 { + self.info >> 32 + } + + fn encode(self, output: &mut Vec) { + output.extend_from_slice(&self.offset.to_le_bytes()); + output.extend_from_slice(&self.info.to_le_bytes()); + output.extend_from_slice(&self.addend.to_le_bytes()); + } +} + +#[derive(Clone, Copy)] +struct RelocationTable<'a> { + data: &'a [u8], + offset: usize, + count: usize, + remap: Option<(usize, u32)>, +} + +impl RelocationTable<'_> { + fn relocation(self, index: usize) -> Result { + if index >= self.count { + return invalid("relocation index is out of range"); + } + let offset = self + .offset + .checked_add( + index + .checked_mul(ELF64_RELA_SIZE) + .ok_or_else(|| Error::Invalid("relocation index overflow".to_owned()))?, + ) + .ok_or_else(|| Error::Invalid("relocation offset overflow".to_owned()))?; + let mut relocation = Rela::parse(self.data, offset)?; + if let Some((old_symbol_count, auxiliary_symbol_count)) = self.remap { + let symbol = relocation.symbol(); + if symbol >= u64::from(auxiliary_symbol_count) { + return invalid("auxiliary relocation references an invalid symbol"); + } + if symbol != 0 { + let base = u64::try_from(old_symbol_count.checked_sub(1).ok_or_else(|| { + Error::Invalid("old dynamic symbol table is empty".to_owned()) + })?) + .map_err(|_| Error::Invalid("old symbol count exceeds u64".to_owned()))?; + let remapped = symbol + .checked_add(base) + .ok_or_else(|| Error::Invalid("remapped symbol index overflow".to_owned()))?; + relocation.info = (remapped << 32) | u64::from(relocation.kind()); + } + } + Ok(relocation) + } + + fn validate(self, allowed: &[u32], description: &str) -> Result<()> { + for index in 0..self.count { + let kind = self.relocation(index)?.kind(); + if !allowed.contains(&kind) { + return invalid(format!( + "{description} contains unsupported relocation type 0x{kind:x}" + )); + } + } + Ok(()) + } + + fn append_where(self, output: &mut Vec, predicate: impl Fn(u32) -> bool) -> Result { + let mut count = 0_usize; + for index in 0..self.count { + let relocation = self.relocation(index)?; + if predicate(relocation.kind()) { + relocation.encode(output); + count += 1; + } + } + Ok(count) + } +} + +fn patch_dynamic_tags( + output: &mut [u8], + dynamic: SectionHeader, + values: &BTreeMap, +) -> Result<()> { + if dynamic.size % 0x10 != 0 { + return invalid(".dynamic size is not entry-aligned"); + } + let start = usize_from_u64(dynamic.offset, ".dynamic offset")?; + let size = usize_from_u64(dynamic.size, ".dynamic size")?; + let end = start + .checked_add(size) + .ok_or_else(|| Error::Invalid(".dynamic end overflow".to_owned()))?; + slice(output, start, size)?; + let mut found = HashSet::with_capacity(values.len()); + for offset in (start..end).step_by(0x10) { + let tag = read_u64(output, offset)?; + if let Some(&value) = values.get(&tag) { + if !found.insert(tag) { + return invalid(format!("dynamic tag 0x{tag:x} occurs more than once")); + } + output[offset + 8..offset + 0x10].copy_from_slice(&value.to_le_bytes()); + } + if tag == 0 { + break; + } + } + let missing = values + .keys() + .filter(|tag| !found.contains(tag)) + .map(|tag| format!("0x{tag:x}")) + .collect::>(); + if !missing.is_empty() { + return invalid(format!("missing dynamic tags: {}", missing.join(", "))); + } + Ok(()) +} + +fn required_section_indices(names: &[String]) -> Result> { + const REQUIRED: [&str; 9] = [ + ".dynsym", + ".gnu.version", + ".gnu.version_r", + ".gnu.hash", + ".dynstr", + ".rela.dyn", + ".rela.plt", + ".dynamic", + ".rodata", + ]; + let mut result = HashMap::with_capacity(REQUIRED.len()); + for required in REQUIRED { + let indices = names + .iter() + .enumerate() + .filter_map(|(index, name)| (name == required).then_some(index)) + .collect::>(); + match indices.as_slice() { + [index] => { + result.insert(required, *index); + } + [] => return invalid(format!("ELF lacks required section {required}")), + _ => return invalid(format!("ELF contains duplicate section {required}")), + } + } + let sysv_hash = names + .iter() + .enumerate() + .filter_map(|(index, name)| (name == ".hash").then_some(index)) + .collect::>(); + match sysv_hash.as_slice() { + [index] => { + result.insert(".hash", *index); + } + [] => {} + _ => return invalid("ELF contains duplicate section .hash"), + } + Ok(result) +} + +fn materialize_static_elf_tables( + output: &mut [u8], + source: &[u8], + layout: &ElfLayout, + symbol_patch_data: &[u8], + auxiliary_data: &[u8], +) -> Result<(ElfLayout, ElfMaterializationReport)> { + let names = layout.section_names(source)?; + let indices = required_section_indices(&names)?; + let section = |name: &'static str| -> SectionHeader { layout.section_headers[indices[name]] }; + let dynsym = section(".dynsym"); + let dynstr = section(".dynstr"); + let versym = section(".gnu.version"); + let verneed = section(".gnu.version_r"); + let rela_dyn = section(".rela.dyn"); + let rela_plt = section(".rela.plt"); + let dynamic = section(".dynamic"); + let rodata = section(".rodata"); + + let (old_symbols, old_strings, hidden_symbols) = + restore_hidden_symbols(output, dynsym, dynstr, symbol_patch_data)?; + let old_symbol_count = old_symbols.len() / ELF64_SYMBOL_SIZE; + if versym.size != (old_symbol_count * 2) as u64 { + return invalid(".gnu.version count does not match .dynsym"); + } + let old_versions = slice_u64(output, versym.offset, versym.size)?.to_vec(); + let version_requirements = slice_u64(output, verneed.offset, verneed.size)?.to_vec(); + + let auxiliary = AuxiliaryElfImage::parse(auxiliary_data)?; + let auxiliary_strings = slice( + auxiliary_data, + auxiliary.dynstr_offset as usize, + auxiliary.dynstr_size as usize, + )?; + let appended_count = auxiliary.dynsym_count as usize - 1; + let mut appended_symbols = Vec::with_capacity(appended_count * ELF64_SYMBOL_SIZE); + for index in 1..auxiliary.dynsym_count as usize { + let offset = auxiliary.dynsym_offset as usize + index * ELF64_SYMBOL_SIZE; + let symbol = slice(auxiliary_data, offset, ELF64_SYMBOL_SIZE)?; + let name_offset = read_u32(symbol, 0)? as usize; + read_c_string(auxiliary_strings, name_offset, auxiliary_strings.len())?; + let section_index = u16::from_le_bytes([symbol[6], symbol[7]]); + if section_index != 0 { + return invalid("auxiliary dynamic symbol is unexpectedly defined"); + } + let merged_name_offset = old_strings + .len() + .checked_add(name_offset) + .ok_or_else(|| Error::Invalid("merged dynamic string offset overflow".to_owned()))?; + let merged_name_offset = u32::try_from(merged_name_offset) + .map_err(|_| Error::Invalid("merged dynamic string offset exceeds u32".to_owned()))?; + appended_symbols.extend_from_slice(&merged_name_offset.to_le_bytes()); + appended_symbols.extend_from_slice(&symbol[4..]); + } + let mut merged_symbols = Vec::with_capacity(old_symbols.len() + appended_symbols.len()); + merged_symbols.extend_from_slice(&old_symbols); + merged_symbols.extend_from_slice(&appended_symbols); + let mut merged_strings = Vec::with_capacity(old_strings.len() + auxiliary_strings.len()); + merged_strings.extend_from_slice(&old_strings); + merged_strings.extend_from_slice(auxiliary_strings); + let mut merged_versions = Vec::with_capacity(old_versions.len() + appended_count * 2); + merged_versions.extend_from_slice(&old_versions); + for _ in 0..appended_count { + merged_versions.extend_from_slice(&VER_NDX_GLOBAL.to_le_bytes()); + } + let merged_names = dynamic_symbol_names(&merged_symbols, &merged_strings)?; + let sysv_hash = indices + .contains_key(".hash") + .then(|| build_sysv_hash(&merged_names)) + .transpose()?; + let gnu_hash_table = build_gnu_hash(&merged_names)?; + let new_symbol_count = merged_names.len(); + let new_dynstr_size = merged_strings.len(); + + if rela_dyn.entry_size != ELF64_RELA_SIZE as u64 + || rela_plt.entry_size != ELF64_RELA_SIZE as u64 + || rela_dyn.size % ELF64_RELA_SIZE as u64 != 0 + || rela_plt.size % ELF64_RELA_SIZE as u64 != 0 + { + return invalid("unexpected relocation entry layout"); + } + let old_dyn = RelocationTable { + data: output, + offset: usize_from_u64(rela_dyn.offset, ".rela.dyn offset")?, + count: usize_from_u64(rela_dyn.size / ELF64_RELA_SIZE as u64, ".rela.dyn count")?, + remap: None, + }; + let old_plt = RelocationTable { + data: output, + offset: usize_from_u64(rela_plt.offset, ".rela.plt offset")?, + count: usize_from_u64(rela_plt.size / ELF64_RELA_SIZE as u64, ".rela.plt count")?, + remap: None, + }; + let auxiliary1 = RelocationTable { + data: auxiliary_data, + offset: auxiliary.relocation1_offset as usize, + count: auxiliary.relocation1_count as usize, + remap: Some((old_symbol_count, auxiliary.dynsym_count)), + }; + let auxiliary2 = RelocationTable { + data: auxiliary_data, + offset: auxiliary.relocation2_offset as usize, + count: auxiliary.relocation2_count as usize, + remap: Some((old_symbol_count, auxiliary.dynsym_count)), + }; + old_dyn.validate( + &[R_AARCH64_RELATIVE, R_AARCH64_GLOB_DAT, R_AARCH64_ABS64], + "existing .rela.dyn", + )?; + old_plt.validate(&[R_AARCH64_JUMP_SLOT], "existing .rela.plt")?; + auxiliary1.validate( + &[R_AARCH64_RELATIVE, R_AARCH64_GLOB_DAT, R_AARCH64_ABS64], + "auxiliary relocation table 1", + )?; + auxiliary2.validate( + &[R_AARCH64_RELATIVE, R_AARCH64_JUMP_SLOT], + "auxiliary relocation table 2", + )?; + + let estimated_dyn = (old_dyn.count + auxiliary1.count + auxiliary2.count) + .checked_mul(ELF64_RELA_SIZE) + .ok_or_else(|| Error::Invalid("merged .rela.dyn capacity overflow".to_owned()))?; + let mut merged_rela_dyn = Vec::with_capacity(estimated_dyn); + let mut relative_count = 0_usize; + for table in [old_dyn, auxiliary1, auxiliary2] { + relative_count += + table.append_where(&mut merged_rela_dyn, |kind| kind == R_AARCH64_RELATIVE)?; + } + for table in [old_dyn, auxiliary1] { + table.append_where(&mut merged_rela_dyn, |kind| kind != R_AARCH64_RELATIVE)?; + } + let mut merged_rela_plt = Vec::with_capacity( + (old_plt.count + auxiliary2.count) + .checked_mul(ELF64_RELA_SIZE) + .ok_or_else(|| Error::Invalid("merged .rela.plt capacity overflow".to_owned()))?, + ); + old_plt.append_where(&mut merged_rela_plt, |_| true)?; + auxiliary2.append_where(&mut merged_rela_plt, |kind| kind == R_AARCH64_JUMP_SLOT)?; + let rela_dyn_count = merged_rela_dyn.len() / ELF64_RELA_SIZE; + let rela_plt_count = merged_rela_plt.len() / ELF64_RELA_SIZE; + + struct TablePayload { + name: &'static str, + alignment: u64, + data: Vec, + } + let mut tables = vec![ + TablePayload { + name: ".dynsym", + alignment: 8, + data: merged_symbols, + }, + TablePayload { + name: ".gnu.version", + alignment: 2, + data: merged_versions, + }, + TablePayload { + name: ".gnu.version_r", + alignment: 4, + data: version_requirements, + }, + TablePayload { + name: ".gnu.hash", + alignment: 8, + data: gnu_hash_table, + }, + ]; + if let Some(data) = sysv_hash { + tables.push(TablePayload { + name: ".hash", + alignment: 4, + data, + }); + } + tables.extend([ + TablePayload { + name: ".dynstr", + alignment: 1, + data: merged_strings, + }, + TablePayload { + name: ".rela.dyn", + alignment: 8, + data: merged_rela_dyn, + }, + TablePayload { + name: ".rela.plt", + alignment: 8, + data: merged_rela_plt, + }, + ]); + let metadata_start = dynsym.offset; + let mut cursor = metadata_start; + let mut placements = BTreeMap::new(); + for table in &tables { + cursor = align_up(cursor, table.alignment)?; + placements.insert( + table.name.to_owned(), + PlacementReport { + offset: cursor, + size: table.data.len(), + }, + ); + cursor = cursor + .checked_add(table.data.len() as u64) + .ok_or_else(|| Error::Invalid("rebuilt ELF metadata end overflow".to_owned()))?; + } + if cursor > rodata.offset { + return invalid(format!( + "rebuilt ELF tables end at 0x{cursor:x}, beyond .rodata 0x{:x}", + rodata.offset + )); + } + let zero_start = usize_from_u64(metadata_start, "metadata start")?; + let zero_end = usize_from_u64(rodata.offset, ".rodata offset")?; + output + .get_mut(zero_start..zero_end) + .ok_or_else(|| Error::Invalid("metadata capacity exceeds output mapping".to_owned()))? + .fill(0); + + let mut updated_sections = layout.section_headers.clone(); + for table in &tables { + let placement = placements + .get(table.name) + .ok_or_else(|| Error::Invalid("table placement disappeared".to_owned()))?; + let offset = usize_from_u64(placement.offset, "table placement offset")?; + let end = offset + .checked_add(table.data.len()) + .ok_or_else(|| Error::Invalid("table placement end overflow".to_owned()))?; + output + .get_mut(offset..end) + .ok_or_else(|| Error::Invalid("table placement exceeds output mapping".to_owned()))? + .copy_from_slice(&table.data); + let index = indices[table.name]; + let mut updated = updated_sections[index]; + updated.address = + layout.file_offset_to_virtual_address(placement.offset, table.data.len() as u64)?; + updated.offset = placement.offset; + updated.size = table.data.len() as u64; + updated_sections[index] = updated; + } + + let section_address = |name: &'static str| -> u64 { updated_sections[indices[name]].address }; + let mut dynamic_values = BTreeMap::from([ + (DT_PLTRELSZ, (rela_plt_count * ELF64_RELA_SIZE) as u64), + (DT_STRTAB, section_address(".dynstr")), + (DT_SYMTAB, section_address(".dynsym")), + (DT_RELA, section_address(".rela.dyn")), + (DT_RELASZ, (rela_dyn_count * ELF64_RELA_SIZE) as u64), + (DT_STRSZ, new_dynstr_size as u64), + (DT_JMPREL, section_address(".rela.plt")), + (DT_GNU_HASH, section_address(".gnu.hash")), + (DT_VERSYM, section_address(".gnu.version")), + (DT_RELACOUNT, relative_count as u64), + (DT_VERNEED, section_address(".gnu.version_r")), + ]); + if indices.contains_key(".hash") { + dynamic_values.insert(DT_HASH, section_address(".hash")); + } + patch_dynamic_tags(output, dynamic, &dynamic_values)?; + + let mut restored_layout = layout.clone(); + restored_layout.section_headers = updated_sections; + Ok(( + restored_layout, + ElfMaterializationReport { + hidden_symbols, + old_symbol_count, + auxiliary_symbol_count: auxiliary.dynsym_count, + appended_symbols: appended_count, + new_symbol_count, + old_dynstr_size: old_strings.len(), + auxiliary_dynstr_size: auxiliary.dynstr_size, + new_dynstr_size, + rela_dyn_count, + rela_plt_count, + relative_prefix_count: relative_count, + metadata_start, + metadata_end: cursor, + metadata_capacity_end: rodata.offset, + metadata_slack: rodata.offset - cursor, + placements, + }, + )) +} + +fn write_padding(file: &mut File, size: u64, path: &Path) -> Result<()> { + const ZEROES: [u8; 4096] = [0; 4096]; + let mut remaining = size; + while remaining != 0 { + let count = usize::try_from(remaining.min(ZEROES.len() as u64)) + .map_err(|_| Error::Invalid("padding size exceeds usize".to_owned()))?; + file.write_all(&ZEROES[..count]) + .map_err(|error| Error::io("write padding", path, error))?; + remaining -= count as u64; + } + Ok(()) +} + +fn finalize_clean_elf( + stream: &mut File, + temporary_path: &Path, + source: &[u8], + layout: &ElfLayout, + preserve_entrypoint: bool, +) -> Result { + let private = layout.private_section()?; + let names = layout.section_names(source)?; + if layout.private_section_index + 1 != layout.section_headers.len() { + return invalid("SHT_LOUSER section is not the final section"); + } + let retained = &layout.section_headers[..layout.private_section_index]; + let mut updated = Vec::with_capacity(retained.len()); + stream + .seek(SeekFrom::Start(private.offset)) + .map_err(|error| Error::io("seek temporary output", temporary_path, error))?; + for §ion in retained { + if section.section_type == SHT_NOBITS || section.flags & SHF_ALLOC != 0 || section.size == 0 + { + updated.push(section); + continue; + } + let section_data = slice_u64(source, section.offset, section.size)?; + let alignment = section.alignment.max(1); + let position = stream + .stream_position() + .map_err(|error| Error::io("query temporary output position", temporary_path, error))?; + let padding = (alignment - position % alignment) % alignment; + write_padding(stream, padding, temporary_path)?; + let new_offset = stream + .stream_position() + .map_err(|error| Error::io("query temporary output position", temporary_path, error))?; + stream + .write_all(section_data) + .map_err(|error| Error::io("append ELF section", temporary_path, error))?; + let mut relocated = section; + relocated.offset = new_offset; + updated.push(relocated); + } + let position = stream + .stream_position() + .map_err(|error| Error::io("query temporary output position", temporary_path, error))?; + write_padding(stream, (8 - position % 8) % 8, temporary_path)?; + let section_header_offset = stream + .stream_position() + .map_err(|error| Error::io("query section-header position", temporary_path, error))?; + for section in &updated { + stream + .write_all(§ion.encode()) + .map_err(|error| Error::io("write section header", temporary_path, error))?; + } + let mut elf_header = [0_u8; 0x40]; + stream + .seek(SeekFrom::Start(0)) + .and_then(|_| stream.read_exact(&mut elf_header)) + .map_err(|error| Error::io("read ELF header", temporary_path, error))?; + if !preserve_entrypoint { + elf_header[0x18..0x20].copy_from_slice(&0_u64.to_le_bytes()); + } + elf_header[0x28..0x30].copy_from_slice(§ion_header_offset.to_le_bytes()); + let section_count = u16::try_from(updated.len()) + .map_err(|_| Error::Invalid("output section count exceeds u16".to_owned()))?; + elf_header[0x3c..0x3e].copy_from_slice(§ion_count.to_le_bytes()); + stream + .seek(SeekFrom::Start(0)) + .and_then(|_| stream.write_all(&elf_header)) + .map_err(|error| Error::io("patch ELF header", temporary_path, error))?; + stream + .flush() + .and_then(|_| stream.sync_all()) + .map_err(|error| Error::io("flush temporary output", temporary_path, error))?; + Ok(CleaningReport { + private_section_index: layout.private_section_index, + private_offset: private.offset, + private_size: private.size, + input_entrypoint: layout.entrypoint, + output_entrypoint: if preserve_entrypoint { + layout.entrypoint + } else { + 0 + }, + output_section_count: updated.len(), + retained_sections: names[..layout.private_section_index].to_vec(), + section_header_offset, + }) +} + +fn section_by_name<'a>( + layout: &'a ElfLayout, + names: &[String], + wanted: &str, +) -> Result<&'a SectionHeader> { + let indices = names + .iter() + .enumerate() + .filter_map(|(index, name)| (name == wanted).then_some(index)) + .collect::>(); + match indices.as_slice() { + [index] => Ok(&layout.section_headers[*index]), + [] => invalid(format!("restored ELF lacks {wanted}")), + _ => invalid(format!("restored ELF contains duplicate {wanted}")), + } +} + +fn validate_restored_binary( + data: &[u8], + preserve_entrypoint: bool, + materialization: Option<&ElfMaterializationReport>, +) -> Result { + let layout = ElfLayout::parse(data, false)?; + let has_louser = layout + .section_headers + .iter() + .any(|section| section.section_type == SHT_LOUSER); + if has_louser { + return invalid("restored output still contains SHT_LOUSER"); + } + if !preserve_entrypoint && layout.entrypoint != 0 { + return invalid("restored output retains the protector entrypoint"); + } + let names = layout.section_names(data)?; + let dynsym = section_by_name(&layout, &names, ".dynsym")?; + let rela_dyn = section_by_name(&layout, &names, ".rela.dyn")?; + let rela_plt = section_by_name(&layout, &names, ".rela.plt")?; + let dynamic_symbols = usize_from_u64( + dynsym.size / ELF64_SYMBOL_SIZE as u64, + "restored dynamic symbol count", + )?; + let dynamic_relocations = usize_from_u64( + rela_dyn.size / ELF64_RELA_SIZE as u64, + "restored dynamic relocation count", + )?; + let pltgot_relocations = usize_from_u64( + rela_plt.size / ELF64_RELA_SIZE as u64, + "restored PLT relocation count", + )?; + if let Some(expected) = materialization { + if dynamic_symbols != expected.new_symbol_count + || dynamic_relocations != expected.rela_dyn_count + || pltgot_relocations != expected.rela_plt_count + { + return invalid("restored ELF table counts do not match materialization report"); + } + } + Ok(ValidationReport { + format: "ELF64".to_owned(), + machine: "AARCH64".to_owned(), + sections: layout.section_headers.len(), + segments: layout.program_headers.len(), + dynamic_symbols, + dynamic_relocations, + pltgot_relocations, + has_louser, + }) +} + +fn absolute(path: &Path) -> Result { + if path.is_absolute() { + Ok(path.to_path_buf()) + } else { + std::env::current_dir() + .map(|current| current.join(path)) + .map_err(|error| Error::io("query current directory", path, error)) + } +} + +fn write_atomic(path: &Path, data: &[u8]) -> Result<()> { + let parent = path.parent().unwrap_or_else(|| Path::new(".")); + std::fs::create_dir_all(parent) + .map_err(|error| Error::io("create output directory", parent, error))?; + let mut temporary = NamedTempFile::new_in(parent) + .map_err(|error| Error::io("create temporary file", parent, error))?; + temporary + .write_all(data) + .and_then(|_| temporary.as_file().sync_all()) + .map_err(|error| Error::io("write temporary file", temporary.path(), error))?; + temporary + .persist(path) + .map_err(|error| Error::io("replace output", path, error.error))?; + Ok(()) +} + +/// Restore the current protected `libil2cpp.so` without executing protector code. +pub fn restore_libil2cpp(options: &RestoreOptions) -> Result { + let started = Instant::now(); + let input_path = absolute(&options.input)?; + let output_path = absolute(&options.output)?; + let index_path = absolute(&options.index)?; + if input_path == output_path + || (output_path.exists() + && std::fs::canonicalize(&input_path).ok() == std::fs::canonicalize(&output_path).ok()) + { + return invalid("refusing to overwrite the protected input in place"); + } + let artifacts = load_artifacts(&index_path)?; + let module = read_file(&artifacts[&0x9b].path)?; + let symbol_patch_data = read_file(&artifacts[&0x9e].path)?; + let config = Module9bConfig::parse(&module)?; + + let input_file = File::open(&input_path) + .map_err(|error| Error::io("open protected input", &input_path, error))?; + let source = map_read_only(&input_file, &input_path)?; + let payload_path = &artifacts[&0x9d].path; + let payload_file = File::open(payload_path) + .map_err(|error| Error::io("open 0x9D artifact", payload_path, error))?; + let payload = map_read_only(&payload_file, payload_path)?; + let layout = ElfLayout::parse(&source, true)?; + let private = layout.private_section()?; + let file_load_end = layout.file_load_end()?; + let aligned_load_end = align_up(file_load_end, 0x10)?; + if private.offset != aligned_load_end { + return invalid(format!( + "SHT_LOUSER offset 0x{:x} != aligned file-backed PT_LOAD end 0x{aligned_load_end:x} (raw 0x{file_load_end:x})", + private.offset + )); + } + let load_padding = slice_u64(&source, file_load_end, private.offset - file_load_end)?; + if load_padding.iter().any(|&byte| byte != 0) { + return invalid(format!( + "nonzero padding between PT_LOAD end 0x{file_load_end:x} and SHT_LOUSER 0x{:x}", + private.offset + )); + } + let descriptor = ProtectedDescriptor::decrypt(&payload, config.header_seed)?; + let load_end = layout.load_end()?; + if u64::from(descriptor.outer_expected_size) != load_end { + return invalid(format!( + "0x9D target size 0x{:x} != ELF load size 0x{load_end:x}", + descriptor.outer_expected_size + )); + } + let outer = ContainerHeader::parse( + &payload, + descriptor.outer_offset as usize, + config.container_seed, + )?; + if u64::from(outer.output_size) != load_end { + return invalid(format!( + "primary container output 0x{:x} != ELF load size 0x{load_end:x}", + outer.output_size + )); + } + let auxiliary_header = ContainerHeader::parse( + &payload, + descriptor.auxiliary_offset as usize, + config.container_seed, + )?; + if auxiliary_header.output_size != descriptor.auxiliary_expected_size { + return invalid("auxiliary container output size does not match the 0x9D descriptor"); + } + if outer.encoded_end()? != descriptor.auxiliary_offset as usize { + return invalid("primary and auxiliary 0x9D containers are not contiguous"); + } + + let parent = output_path.parent().unwrap_or_else(|| Path::new(".")); + std::fs::create_dir_all(parent) + .map_err(|error| Error::io("create output directory", parent, error))?; + let mut temporary = NamedTempFile::new_in(parent) + .map_err(|error| Error::io("create temporary output", parent, error))?; + let temporary_path = temporary.path().to_path_buf(); + let private_size = usize_from_u64(private.offset, "private section offset")?; + copy_range( + &source, + temporary.as_file_mut(), + private_size, + &temporary_path, + )?; + temporary + .as_file_mut() + .flush() + .map_err(|error| Error::io("flush initial output", &temporary_path, error))?; + temporary + .as_file() + .set_len(private.offset) + .map_err(|error| Error::io("size temporary output", &temporary_path, error))?; + + let mut restored_layout = layout.clone(); + let mut auxiliary_data = None; + let mut auxiliary_stats = None; + let mut materialization = None; + let primary_stats; + { + let mut output = map_mut(temporary.as_file(), private_size, &temporary_path)?; + if options.verbose { + eprintln!("Decoding primary 0x9D target-image container..."); + } + let mut writer = FileLayoutWriter { + output: &mut output, + layout: &layout, + load_end, + }; + primary_stats = decode_container( + &payload, + &outer, + &config, + options.verbose, + |address, data| writer.write(address, data), + )?; + if !options.outer_only { + if options.verbose { + eprintln!("Decoding auxiliary 0x9D ELF materialization container..."); + } + let mut decoded = vec![0_u8; auxiliary_header.output_size as usize]; + let stats = decode_container( + &payload, + &auxiliary_header, + &config, + options.verbose, + |offset, data| { + let start = usize_from_u64(offset, "auxiliary write offset")?; + let end = start.checked_add(data.len()).ok_or_else(|| { + Error::Invalid("auxiliary decoded write overflow".to_owned()) + })?; + let destination = decoded.get_mut(start..end).ok_or_else(|| { + Error::Invalid("auxiliary decoded write is out of range".to_owned()) + })?; + destination.copy_from_slice(data); + Ok(data.len()) + }, + )?; + if let Some(path) = &options.dump_auxiliary { + write_atomic(&absolute(path)?, &decoded)?; + } + if options.verbose { + eprintln!("Rebuilding static ELF dynamic-linker tables..."); + } + let (new_layout, report) = materialize_static_elf_tables( + &mut output, + &source, + &layout, + &symbol_patch_data, + &decoded, + )?; + restored_layout = new_layout; + materialization = Some(report); + auxiliary_stats = Some(stats); + auxiliary_data = Some(decoded); + } + output + .flush() + .map_err(|error| Error::io("flush restored image", &temporary_path, error))?; + } + drop(auxiliary_data); + + let cleaning = finalize_clean_elf( + temporary.as_file_mut(), + &temporary_path, + &source, + &restored_layout, + options.preserve_entrypoint, + )?; + let validation = { + let restored = map_read_only(temporary.as_file(), &temporary_path)?; + validate_restored_binary( + &restored, + options.preserve_entrypoint, + materialization.as_ref(), + )? + }; + temporary + .persist(&output_path) + .map_err(|error| Error::io("replace restored output", &output_path, error.error))?; + + Ok(RestoreReport { + input: input_path.display().to_string(), + input_sha256: sha256_file(&input_path)?, + output: output_path.display().to_string(), + output_sha256: sha256_file(&output_path)?, + output_size: std::fs::metadata(&output_path) + .map_err(|error| Error::io("inspect restored output", &output_path, error))? + .len(), + module_index: index_path.display().to_string(), + static_config: StaticConfigReport { + header_seed: format!("0x{:08X}", config.header_seed), + container_seed: format!("0x{:08X}", config.container_seed), + aes_key_sha256: sha256_bytes(&config.aes_key), + schedule_offset: format!("0x{:X}", config.schedule_offset), + }, + descriptor: DescriptorReport { + command_id: format!("0x{:X}", descriptor.command_id), + flags: format!("0x{:X}", descriptor.flags), + outer_offset: format!("0x{:X}", descriptor.outer_offset), + outer_expected_size: format!("0x{:X}", descriptor.outer_expected_size), + auxiliary_offset: format!("0x{:X}", descriptor.auxiliary_offset), + auxiliary_expected_size: format!("0x{:X}", descriptor.auxiliary_expected_size), + }, + primary: primary_stats, + auxiliary: auxiliary_stats, + elf_materialization: materialization, + cleaning, + validation, + elapsed_seconds: started.elapsed().as_secs_f64(), + }) +} diff --git a/senbei-android-engine/Cargo.toml b/senbei-android-engine/Cargo.toml new file mode 100644 index 0000000..c164822 --- /dev/null +++ b/senbei-android-engine/Cargo.toml @@ -0,0 +1,20 @@ +[package] +name = "senbei-android-engine" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +description = "Static Stage 1 and Stage 2 extraction for Senbei Android" + +[dependencies] +goblin.workspace = true +memmap2.workspace = true +serde.workspace = true +serde_json.workspace = true +sha2.workspace = true +tempfile.workspace = true +thiserror.workspace = true +senbei-android-crypto.workspace = true + +[lints] +workspace = true diff --git a/senbei-android-engine/src/error.rs b/senbei-android-engine/src/error.rs new file mode 100644 index 0000000..2711d68 --- /dev/null +++ b/senbei-android-engine/src/error.rs @@ -0,0 +1,63 @@ +use std::path::{Path, PathBuf}; + +/// Stage 1 or Stage 2 extraction failure. +#[derive(Debug, thiserror::Error)] +pub enum Error { + #[error("{action} `{path}`: {source}")] + Io { + action: &'static str, + path: PathBuf, + #[source] + source: std::io::Error, + }, + #[error("parse ELF `{path}`: {source}")] + Elf { + path: PathBuf, + #[source] + source: goblin::error::Error, + }, + #[error("serialize extraction index: {0}")] + Json(#[from] serde_json::Error), + #[error("embedded Stage 2 decoder configuration: {0}")] + EmbeddedConfig(#[source] senbei_android_crypto::Error), + #[error( + "depth {depth} stream 0x{stream_id:02X} interpreter 0x{interpreter_id:02X} configuration: {source}" + )] + InterpreterConfig { + depth: usize, + stream_id: u32, + interpreter_id: u32, + #[source] + source: senbei_android_crypto::Error, + }, + #[error( + "depth {depth} stream 0x{stream_id:02X} record {record_index} command 0x{command_id:02X} {part}: {source}" + )] + RecordDecode { + depth: usize, + stream_id: u32, + record_index: usize, + command_id: u32, + part: &'static str, + #[source] + source: senbei_android_crypto::Error, + }, + #[error("{0}")] + Invalid(String), +} + +impl Error { + pub(crate) fn io(action: &'static str, path: &Path, source: std::io::Error) -> Self { + Self::Io { + action, + path: path.to_path_buf(), + source, + } + } +} + +pub(crate) type Result = std::result::Result; + +pub(crate) fn invalid(message: impl Into) -> Result { + Err(Error::Invalid(message.into())) +} diff --git a/senbei-android-engine/src/extract.rs b/senbei-android-engine/src/extract.rs new file mode 100644 index 0000000..c79c598 --- /dev/null +++ b/senbei-android-engine/src/extract.rs @@ -0,0 +1,529 @@ +use std::collections::{BTreeMap, BTreeSet, HashSet}; +use std::fs::{File, create_dir_all}; +use std::io::Write; +use std::path::{Path, PathBuf}; + +use memmap2::MmapOptions; +use senbei_android_crypto::{Module9bConfig, decode_container}; +use serde_json::to_vec_pretty; +use sha2::{Digest, Sha256}; +use tempfile::NamedTempFile; + +use crate::error::{Error, Result, invalid}; +use crate::report::{ + ArtifactReport, DecoderReport, ExtractionReport, ModuleRegistryEntry, RecordReport, + Stage1Report, StreamParent, StreamReport, +}; +use crate::stage1::{ + DEFAULT_CIPHER_CONSTANT, DEFAULT_OUTER_SIZE, SHT_LOUSER, Stage1Result, inspect, +}; +use crate::stream::{DIRECT_FLAG, Record, parse_record_stream}; + +/// Inputs and output locations for one complete static Stage 2 extraction. +#[derive(Debug, Clone)] +pub struct ExtractOptions { + pub input: PathBuf, + pub output_dir: PathBuf, + pub stage2_output: Option, + pub outer_size: usize, + pub cipher_constant: u32, +} + +impl ExtractOptions { + #[must_use] + pub fn with_defaults(input: PathBuf, output_dir: PathBuf) -> Self { + Self { + input, + output_dir, + stage2_output: None, + outer_size: DEFAULT_OUTER_SIZE, + cipher_constant: DEFAULT_CIPHER_CONSTANT, + } + } +} + +#[derive(Debug)] +struct LoadedModule { + image: Vec, + metadata: Option>, + image_path: String, + metadata_path: Option, + sha256: String, + depth: usize, + record_index: usize, + command_id: u32, + init_offset: u32, + entry_offset: u32, +} + +#[derive(Debug, Clone, Copy)] +struct ArtifactSpec<'a> { + suffix: &'a str, + kind: &'a str, + classification: &'a str, +} + +struct Extractor { + output_dir: PathBuf, + streams: Vec, + artifacts: Vec, + registry: BTreeMap, + seen_streams: HashSet<(u32, String)>, +} + +pub fn extract_stage2(options: &ExtractOptions) -> Result { + let input_path = absolute(&options.input)?; + let output_dir = absolute(&options.output_dir)?; + if !input_path.is_file() { + return invalid(format!( + "protected ELF does not exist: {}", + input_path.display() + )); + } + if let Some(stage2_output) = &options.stage2_output { + let stage2_output = absolute(stage2_output)?; + if stage2_output == input_path { + return invalid("refusing to overwrite the protected ELF with Stage 2 output"); + } + } + create_dir_all(&output_dir) + .map_err(|source| Error::io("create Stage 2 output directory", &output_dir, source))?; + + let file = File::open(&input_path) + .map_err(|source| Error::io("open protected ELF", &input_path, source))?; + // SAFETY: the mapping is read-only, the file remains open for the mapping + // lifetime, and extraction never mutates or truncates the source. + let source = unsafe { MmapOptions::new().map(&file) } + .map_err(|source| Error::io("map protected ELF", &input_path, source))?; + let stage1 = inspect( + &source, + &input_path, + options.outer_size, + options.cipher_constant, + )?; + if let Some(stage2_output) = &options.stage2_output { + write_atomic(&absolute(stage2_output)?, &stage1.plaintext)?; + } + + let core_config = + Module9bConfig::parse_embedded(&stage1.plaintext).map_err(Error::EmbeddedConfig)?; + let bootstrap_end = stage1 + .remaining_file_offset + .checked_add(stage1.remaining_size) + .ok_or_else(|| Error::Invalid("Stage 2 bootstrap range overflow".to_owned()))?; + let bootstrap = source + .get(stage1.remaining_file_offset..bootstrap_end) + .ok_or_else(|| Error::Invalid("Stage 2 bootstrap range is outside the ELF".to_owned()))?; + let mut extractor = Extractor { + output_dir: output_dir.clone(), + streams: Vec::new(), + artifacts: Vec::new(), + registry: BTreeMap::new(), + seen_streams: HashSet::new(), + }; + extractor.extract_stream( + bootstrap, + 0xe2, + 0, + None, + Some(stage1.remaining_file_offset), + core_config, + )?; + + let module_registry = extractor + .registry + .values() + .map(|module| ModuleRegistryEntry { + command_id: module.command_id, + size: module.image.len(), + sha256: module.sha256.clone(), + depth: module.depth, + record_index: module.record_index, + image_path: module.image_path.clone(), + metadata_path: module.metadata_path.clone(), + init_offset: module.init_offset, + entry_offset: module.entry_offset, + classification: if module.metadata.is_some() { + "module_image".to_owned() + } else { + "decoded_data".to_owned() + }, + }) + .collect::>(); + let report = ExtractionReport { + format_version: 4, + protected_elf: input_path.display().to_string(), + output_dir: output_dir.display().to_string(), + stage1: stage1_report(&stage1, options.outer_size), + streams: extractor.streams, + artifacts: extractor.artifacts, + errors: Vec::new(), + module_registry, + }; + write_json_atomic(&output_dir.join("index.json"), &report)?; + Ok(report) +} + +impl Extractor { + fn extract_stream( + &mut self, + stream: &[u8], + stream_id: u32, + depth: usize, + parent: Option, + source_file_offset: Option, + config: Module9bConfig, + ) -> Result<()> { + let digest = sha256(stream); + if !self.seen_streams.insert((stream_id, digest.clone())) { + return Ok(()); + } + let (header, records, table_size) = + parse_record_stream(stream, stream_id).map_err(|source| { + Error::Invalid(format!( + "depth {depth} stream 0x{stream_id:02X} record table: {source}" + )) + })?; + let mut stream_report = StreamReport { + depth, + stream_id, + parent, + source_file_offset, + available_size: stream.len(), + descriptor_table_size: table_size, + encrypted_header_words: header.encrypted_words, + decrypted_header_words: header.decrypted_words, + record_state: header.record_state, + sha256: digest, + decoder: decoder_report( + if depth == 0 { + "embedded_stage2" + } else { + "decoded_interpreter" + }, + (depth != 0).then_some(stream_id), + &config, + ), + records: Vec::with_capacity(records.len()), + }; + let mut direct_records = Vec::new(); + let mut modules_at_level = BTreeSet::new(); + + for record in records { + let mut result = record_report(record); + let mut image_data = None; + let mut metadata_data = None; + + if !record.direct() && record.image_size != 0 { + let image_source = record_tail(stream, record.image_offset)?; + let image = decode_container(image_source, &config, record.image_size as usize) + .map_err(|source| Error::RecordDecode { + depth, + stream_id, + record_index: record.index, + command_id: record.command_id, + part: "image decode", + source, + })?; + let classification = if record.metadata_size != 0 { + "module_image" + } else { + "decoded_data" + }; + let artifact = self.write_artifact( + &record, + depth, + stream_id, + ArtifactSpec { + suffix: "module.bin", + kind: "decoded_container", + classification, + }, + &image, + )?; + result.image = Some(artifact.clone()); + image_data = Some((image, artifact)); + } + if record.metadata_size != 0 { + let metadata_source = record_tail(stream, record.metadata_offset)?; + let metadata = + decode_container(metadata_source, &config, record.metadata_size as usize) + .map_err(|source| Error::RecordDecode { + depth, + stream_id, + record_index: record.index, + command_id: record.command_id, + part: "metadata decode", + source, + })?; + let artifact = self.write_artifact( + &record, + depth, + stream_id, + ArtifactSpec { + suffix: "metadata.bin", + kind: "decoded_metadata", + classification: "decoded_metadata", + }, + &metadata, + )?; + result.metadata = Some(artifact.clone()); + metadata_data = Some((metadata, artifact)); + } + if let Some((image, image_artifact)) = image_data { + let (metadata, metadata_path) = if let Some((data, artifact)) = metadata_data { + (Some(data), Some(artifact.path)) + } else { + (None, None) + }; + self.register_module(LoadedModule { + sha256: image_artifact.sha256.clone(), + image_path: image_artifact.path.clone(), + metadata_path, + image, + metadata, + depth, + record_index: record.index, + command_id: record.command_id, + init_offset: record.init_offset, + entry_offset: record.entry_offset, + })?; + modules_at_level.insert(record.command_id); + } + if record.direct() && record.image_size != 0 { + direct_records.push((record, stream_report.records.len())); + } + stream_report.records.push(result); + } + + let mut children = Vec::new(); + for (record, report_index) in direct_records { + let next_stream_id = record.command_id.wrapping_sub(0x10); + if modules_at_level.contains(&next_stream_id) { + stream_report.records[report_index].nested_stream_id = Some(next_stream_id); + children.push((record, next_stream_id)); + continue; + } + let direct_data = record_slice(stream, record.image_offset, record.image_size)?; + let artifact = self.write_artifact( + &record, + depth, + stream_id, + ArtifactSpec { + suffix: "direct.bin", + kind: "direct", + classification: "direct_data", + }, + direct_data, + )?; + stream_report.records[report_index].image = Some(artifact); + } + + self.streams.push(stream_report); + for (record, next_stream_id) in children { + let child_data = record_slice(stream, record.image_offset, record.image_size)?; + let parent = StreamParent { + stream_id, + record_index: record.index, + command_id: record.command_id, + }; + let interpreter = self.registry.get(&next_stream_id).ok_or_else(|| { + Error::Invalid(format!( + "depth {depth} stream 0x{stream_id:02X} child 0x{next_stream_id:02X} has no interpreter module" + )) + })?; + let interpreter_config = + Module9bConfig::parse(&interpreter.image).map_err(|source| { + Error::InterpreterConfig { + depth: depth + 1, + stream_id: next_stream_id, + interpreter_id: next_stream_id, + source, + } + })?; + self.extract_stream( + child_data, + next_stream_id, + depth + 1, + Some(parent), + None, + interpreter_config, + )?; + } + Ok(()) + } + + fn register_module(&mut self, module: LoadedModule) -> Result<()> { + if let Some(previous) = self.registry.get(&module.command_id) { + if previous.sha256 != module.sha256 { + return invalid(format!( + "module 0x{:02X} produced conflicting images: {} and {}", + module.command_id, previous.sha256, module.sha256 + )); + } + return Ok(()); + } + self.registry.insert(module.command_id, module); + Ok(()) + } + + fn write_artifact( + &mut self, + record: &Record, + depth: usize, + stream_id: u32, + spec: ArtifactSpec<'_>, + data: &[u8], + ) -> Result { + let digest = sha256(data); + let filename = format!( + "d{depth:02}_s{stream_id:02X}_r{:03}_id{:08X}_{}.{}", + record.index, + record.command_id, + &digest[..12], + spec.suffix + ); + let path = self.output_dir.join(filename); + write_atomic(&path, data)?; + let artifact = ArtifactReport { + kind: spec.kind.to_owned(), + path: path + .file_name() + .ok_or_else(|| Error::Invalid("artifact path has no file name".to_owned()))? + .to_string_lossy() + .into_owned(), + size: data.len(), + sha256: digest, + depth, + stream_id, + record_index: Some(record.index), + command_id: Some(record.command_id), + classification: spec.classification.to_owned(), + }; + self.artifacts.push(artifact.clone()); + Ok(artifact) + } +} + +fn record_report(record: Record) -> RecordReport { + RecordReport { + index: record.index, + command_id: record.command_id, + flags: record.flags, + image_offset: record.image_offset, + image_size: record.image_size, + metadata_offset: record.metadata_offset, + metadata_size: record.metadata_size, + id_copy: record.id_copy, + entry_offset: record.entry_offset, + init_offset: record.init_offset, + direct: record.flags & DIRECT_FLAG != 0, + extraction_status: "complete".to_owned(), + image: None, + metadata: None, + nested_stream_id: None, + } +} + +fn decoder_report( + kind: &str, + interpreter_id: Option, + config: &Module9bConfig, +) -> DecoderReport { + DecoderReport { + kind: kind.to_owned(), + interpreter_id, + header_seed: config.header_seed, + container_seed: config.container_seed, + schedule_offset: config.schedule_offset, + aes_key_sha256: sha256(&config.aes_key), + skip_aes: config.skip_aes, + } +} + +fn record_slice(stream: &[u8], offset: u32, size: u32) -> Result<&[u8]> { + let offset = usize::try_from(offset) + .map_err(|_| Error::Invalid("record payload offset exceeds usize".to_owned()))?; + let size = usize::try_from(size) + .map_err(|_| Error::Invalid("record payload size exceeds usize".to_owned()))?; + let end = offset + .checked_add(size) + .ok_or_else(|| Error::Invalid("record payload range overflows usize".to_owned()))?; + stream.get(offset..end).ok_or_else(|| { + Error::Invalid(format!( + "record payload range 0x{offset:x}..0x{end:x} exceeds stream 0x{:x}", + stream.len() + )) + }) +} + +fn record_tail(stream: &[u8], offset: u32) -> Result<&[u8]> { + let offset = usize::try_from(offset) + .map_err(|_| Error::Invalid("record container offset exceeds usize".to_owned()))?; + stream.get(offset..).ok_or_else(|| { + Error::Invalid(format!( + "record container offset 0x{offset:x} exceeds stream 0x{:x}", + stream.len() + )) + }) +} + +fn stage1_report(stage1: &Stage1Result, outer_size: usize) -> Stage1Report { + Stage1Report { + section_index: stage1.section_index, + section_type: SHT_LOUSER, + section_offset: stage1.section_offset, + section_size: stage1.section_size, + outer_size, + header_offset: stage1.header_offset, + header_key: stage1.header.key, + payload_offset: stage1.header.payload_offset, + payload_size: stage1.header.payload_size, + payload_key: stage1.header.payload_key, + entry_offset: stage1.header.entry_offset, + protect_size: stage1.header.protect_size, + stage2_file_offset: stage1.payload_file_offset, + stage2_size: stage1.plaintext.len(), + stage2_sha256: sha256(&stage1.plaintext), + remaining_file_offset: stage1.remaining_file_offset, + remaining_size: stage1.remaining_size, + } +} + +fn write_json_atomic(path: &Path, value: &impl serde::Serialize) -> Result<()> { + let mut bytes = to_vec_pretty(value)?; + bytes.push(b'\n'); + write_atomic(path, &bytes) +} + +fn write_atomic(path: &Path, data: &[u8]) -> Result<()> { + let parent = path.parent().unwrap_or_else(|| Path::new(".")); + create_dir_all(parent) + .map_err(|source| Error::io("create output directory", parent, source))?; + let mut temporary = NamedTempFile::new_in(parent) + .map_err(|source| Error::io("create temporary output", parent, source))?; + temporary + .write_all(data) + .and_then(|()| temporary.as_file().sync_all()) + .map_err(|source| Error::io("write temporary output", temporary.path(), source))?; + temporary + .persist(path) + .map_err(|error| Error::io("replace output", path, error.error))?; + Ok(()) +} + +fn absolute(path: &Path) -> Result { + if path.is_absolute() { + Ok(path.to_path_buf()) + } else { + std::env::current_dir() + .map(|current| current.join(path)) + .map_err(|source| Error::io("query current directory", path, source)) + } +} + +fn sha256(data: &[u8]) -> String { + let mut digest = Sha256::new(); + digest.update(data); + format!("{:x}", digest.finalize()) +} diff --git a/senbei-android-engine/src/lib.rs b/senbei-android-engine/src/lib.rs new file mode 100644 index 0000000..386eb55 --- /dev/null +++ b/senbei-android-engine/src/lib.rs @@ -0,0 +1,14 @@ +//! Pure-static Stage 1 decryption and recursive Stage 2 module extraction. + +mod error; +mod extract; +mod probe; +mod report; +mod stage1; +mod stream; + +pub use error::Error; +pub use extract::{ExtractOptions, extract_stage2}; +pub use probe::is_protected_libil2cpp; +pub use report::ExtractionReport; +pub use stage1::{DEFAULT_CIPHER_CONSTANT, DEFAULT_OUTER_SIZE}; diff --git a/senbei-android-engine/src/probe.rs b/senbei-android-engine/src/probe.rs new file mode 100644 index 0000000..73c250d --- /dev/null +++ b/senbei-android-engine/src/probe.rs @@ -0,0 +1,22 @@ +use std::path::Path; + +use senbei_android_crypto::Module9bConfig; + +use crate::stage1::{self, DEFAULT_CIPHER_CONSTANT, DEFAULT_OUTER_SIZE}; + +/// Return whether `data` has a supported protected AArch64 IL2CPP layout. +#[must_use] +pub fn is_protected_libil2cpp(data: &[u8]) -> bool { + if !stage1::looks_protected(data) { + return false; + } + let Ok(stage1) = stage1::inspect( + data, + Path::new(""), + DEFAULT_OUTER_SIZE, + DEFAULT_CIPHER_CONSTANT, + ) else { + return false; + }; + Module9bConfig::parse_embedded(&stage1.plaintext).is_ok() +} diff --git a/senbei-android-engine/src/report.rs b/senbei-android-engine/src/report.rs new file mode 100644 index 0000000..9435067 --- /dev/null +++ b/senbei-android-engine/src/report.rs @@ -0,0 +1,115 @@ +use serde::Serialize; + +#[derive(Debug, Clone, Serialize)] +pub struct Stage1Report { + pub section_index: usize, + pub section_type: u32, + pub section_offset: usize, + pub section_size: usize, + pub outer_size: usize, + pub header_offset: usize, + pub header_key: u32, + pub payload_offset: u32, + pub payload_size: u32, + pub payload_key: u32, + pub entry_offset: u32, + pub protect_size: u32, + pub stage2_file_offset: usize, + pub stage2_size: usize, + pub stage2_sha256: String, + pub remaining_file_offset: usize, + pub remaining_size: usize, +} + +#[derive(Debug, Clone, Serialize)] +pub struct DecoderReport { + pub kind: String, + pub interpreter_id: Option, + pub header_seed: u32, + pub container_seed: u32, + pub schedule_offset: usize, + pub aes_key_sha256: String, + pub skip_aes: bool, +} + +#[derive(Debug, Clone, Serialize)] +pub struct ArtifactReport { + pub kind: String, + pub path: String, + pub size: usize, + pub sha256: String, + pub depth: usize, + pub stream_id: u32, + pub record_index: Option, + pub command_id: Option, + pub classification: String, +} + +#[derive(Debug, Clone, Serialize)] +pub struct RecordReport { + pub index: usize, + pub command_id: u32, + pub flags: u32, + pub image_offset: u32, + pub image_size: u32, + pub metadata_offset: u32, + pub metadata_size: u32, + pub id_copy: u32, + pub entry_offset: u32, + pub init_offset: u32, + pub direct: bool, + pub extraction_status: String, + pub image: Option, + pub metadata: Option, + pub nested_stream_id: Option, +} + +#[derive(Debug, Clone, Serialize)] +pub struct StreamParent { + pub stream_id: u32, + pub record_index: usize, + pub command_id: u32, +} + +#[derive(Debug, Clone, Serialize)] +pub struct StreamReport { + pub depth: usize, + pub stream_id: u32, + pub parent: Option, + pub source_file_offset: Option, + pub available_size: usize, + pub descriptor_table_size: usize, + pub encrypted_header_words: [u32; 2], + pub decrypted_header_words: [u32; 2], + pub record_state: u32, + pub sha256: String, + pub decoder: DecoderReport, + pub records: Vec, +} + +#[derive(Debug, Clone, Serialize)] +pub struct ModuleRegistryEntry { + pub command_id: u32, + pub size: usize, + pub sha256: String, + pub depth: usize, + pub record_index: usize, + pub image_path: String, + pub metadata_path: Option, + pub init_offset: u32, + pub entry_offset: u32, + pub classification: String, +} + +/// Machine-readable output of one complete static Stage 2 extraction. +#[derive(Debug, Clone, Serialize)] +pub struct ExtractionReport { + pub format_version: u32, + pub protected_elf: String, + pub output_dir: String, + pub stage1: Stage1Report, + pub streams: Vec, + pub artifacts: Vec, + pub errors: Vec, + pub module_registry: Vec, +} diff --git a/senbei-android-engine/src/stage1.rs b/senbei-android-engine/src/stage1.rs new file mode 100644 index 0000000..8b52c68 --- /dev/null +++ b/senbei-android-engine/src/stage1.rs @@ -0,0 +1,262 @@ +use std::path::Path; + +use goblin::elf::{Elf, header::EM_AARCH64}; + +use crate::error::{Error, Result, invalid}; + +pub(crate) const SHT_LOUSER: u32 = 0x8000_0000; +pub const DEFAULT_CIPHER_CONSTANT: u32 = 0xbf20_165d; +pub const DEFAULT_OUTER_SIZE: usize = 0x23c; + +pub(crate) fn looks_protected(data: &[u8]) -> bool { + let Ok(elf) = Elf::parse(data) else { + return false; + }; + if elf.header.e_machine != EM_AARCH64 + || elf + .section_headers + .iter() + .filter(|section| section.sh_type == SHT_LOUSER) + .count() + != 1 + { + return false; + } + [ + ".dynsym", + ".dynstr", + ".gnu.hash", + ".gnu.version", + ".gnu.version_r", + ] + .into_iter() + .all(|wanted| { + elf.section_headers.iter().any(|section| { + elf.shdr_strtab + .get_at(section.sh_name) + .is_some_and(|name| name == wanted) + }) + }) +} + +#[derive(Debug, Clone, Copy)] +pub(crate) struct Stage1Header { + pub key: u32, + pub reserved: u32, + pub payload_offset: u32, + pub payload_size: u32, + pub payload_key: u32, + pub entry_offset: u32, + pub protect_size: u32, + pub size_copy: u32, +} + +#[derive(Debug)] +pub(crate) struct Stage1Result { + pub section_index: usize, + pub section_offset: usize, + pub section_size: usize, + pub header_offset: usize, + pub payload_file_offset: usize, + pub remaining_file_offset: usize, + pub remaining_size: usize, + pub header: Stage1Header, + pub plaintext: Vec, +} + +pub(crate) fn inspect( + data: &[u8], + path: &Path, + outer_size: usize, + cipher_constant: u32, +) -> Result { + let elf = Elf::parse(data).map_err(|source| Error::Elf { + path: path.to_path_buf(), + source, + })?; + if elf.header.e_machine != EM_AARCH64 { + return invalid(format!( + "expected AArch64 ELF (machine 0x{EM_AARCH64:X}), got 0x{:X}", + elf.header.e_machine + )); + } + let matches = elf + .section_headers + .iter() + .enumerate() + .filter(|(_, section)| section.sh_type == SHT_LOUSER) + .collect::>(); + if matches.len() != 1 { + return invalid(format!( + "expected exactly one SHT_LOUSER section, found {}", + matches.len() + )); + } + let (section_index, section) = matches[0]; + let section_offset = usize::try_from(section.sh_offset) + .map_err(|_| Error::Invalid("SHT_LOUSER offset exceeds usize".to_owned()))?; + let section_size = usize::try_from(section.sh_size) + .map_err(|_| Error::Invalid("SHT_LOUSER size exceeds usize".to_owned()))?; + let section_end = section_offset + .checked_add(section_size) + .ok_or_else(|| Error::Invalid("SHT_LOUSER range overflows usize".to_owned()))?; + if section_end > data.len() { + return invalid("SHT_LOUSER range extends beyond the input file"); + } + let header_relative = outer_size; + if outer_size + .checked_add(0x1000) + .is_none_or(|end| end > section_size) + { + return invalid("Stage 1 outer header leaves no complete parameter area"); + } + let header_offset = section_offset + .checked_add(header_relative) + .ok_or_else(|| Error::Invalid("Stage 1 header offset overflow".to_owned()))?; + let header_raw = bytes(data, header_offset, 0x1000)?; + let header = decrypt_header(header_raw, cipher_constant)?; + if header.reserved != 0 { + return invalid(format!( + "Stage 1 header reserved word is nonzero: 0x{:x}", + header.reserved + )); + } + if header.size_copy != header.payload_size { + return invalid(format!( + "Stage 1 payload size copy 0x{:x} != size 0x{:x}", + header.size_copy, header.payload_size + )); + } + let private_size = section_size - outer_size; + let payload_offset = usize::try_from(header.payload_offset) + .map_err(|_| Error::Invalid("Stage 1 payload offset exceeds usize".to_owned()))?; + let payload_size = usize::try_from(header.payload_size) + .map_err(|_| Error::Invalid("Stage 1 payload size exceeds usize".to_owned()))?; + let payload_end = payload_offset + .checked_add(payload_size) + .ok_or_else(|| Error::Invalid("Stage 1 payload range overflow".to_owned()))?; + if payload_offset < 0x20 || payload_end > private_size { + return invalid(format!( + "Stage 1 payload range 0x{payload_offset:x}..0x{payload_end:x} exceeds private size 0x{private_size:x}" + )); + } + if payload_size == 0 || payload_size % 4 != 0 { + return invalid(format!( + "Stage 1 payload size must be nonzero and word aligned: 0x{payload_size:x}" + )); + } + let entry_offset = usize::try_from(header.entry_offset) + .map_err(|_| Error::Invalid("Stage 1 entry offset exceeds usize".to_owned()))?; + if entry_offset >= payload_size { + return invalid("Stage 1 entry offset is outside the payload"); + } + let protect_size = usize::try_from(header.protect_size) + .map_err(|_| Error::Invalid("Stage 1 protect size exceeds usize".to_owned()))?; + if protect_size > payload_size { + return invalid("Stage 1 mprotect length exceeds the payload"); + } + let payload_file_offset = header_offset + .checked_add(payload_offset) + .ok_or_else(|| Error::Invalid("Stage 1 payload file offset overflow".to_owned()))?; + let encrypted = bytes(data, payload_file_offset, payload_size)?; + let plaintext = decrypt_words(encrypted, header.payload_key, cipher_constant)?; + let aligned_payload_end = (payload_end + 3) & !3; + let remaining_relative = aligned_payload_end; + if remaining_relative > private_size { + return invalid("aligned Stage 2 cursor exceeds SHT_LOUSER"); + } + let remaining_file_offset = section_offset + .checked_add(outer_size) + .and_then(|value| value.checked_add(remaining_relative)) + .ok_or_else(|| Error::Invalid("Stage 2 stream offset overflow".to_owned()))?; + Ok(Stage1Result { + section_index, + section_offset, + section_size, + header_offset, + payload_file_offset, + remaining_file_offset, + remaining_size: private_size - remaining_relative, + header, + plaintext, + }) +} + +fn decrypt_header(raw: &[u8], constant: u32) -> Result { + let key = read_u32(raw, 0)?; + let mut decoded = decrypt_words(&raw[..0x20], key, constant)?; + decoded[..4].copy_from_slice(&key.to_le_bytes()); + Ok(Stage1Header { + key, + reserved: read_u32(&decoded, 4)?, + payload_offset: read_u32(&decoded, 8)?, + payload_size: read_u32(&decoded, 12)?, + payload_key: read_u32(&decoded, 16)?, + entry_offset: read_u32(&decoded, 20)?, + protect_size: read_u32(&decoded, 24)?, + size_copy: read_u32(&decoded, 28)?, + }) +} + +fn decrypt_words(ciphertext: &[u8], key: u32, constant: u32) -> Result> { + if ciphertext.len() % 4 != 0 { + return invalid("Stage 1 word cipher input is not 4-byte aligned"); + } + let mut plaintext = ciphertext.to_vec(); + for (index, chunk) in plaintext.chunks_exact_mut(4).enumerate() { + let index = u32::try_from(index) + .map_err(|_| Error::Invalid("Stage 1 word index exceeds u32".to_owned()))?; + let mut word = u32::from_le_bytes( + chunk + .try_into() + .map_err(|_| Error::Invalid("Stage 1 word has an invalid size".to_owned()))?, + ); + word = word.wrapping_add(index.wrapping_add(3).wrapping_mul(key)); + word ^= constant.wrapping_mul(index.wrapping_add(1)); + chunk.copy_from_slice(&word.to_le_bytes()); + } + Ok(plaintext) +} + +fn bytes(data: &[u8], offset: usize, size: usize) -> Result<&[u8]> { + let end = offset + .checked_add(size) + .ok_or_else(|| Error::Invalid("byte range overflow".to_owned()))?; + data.get(offset..end).ok_or_else(|| { + Error::Invalid(format!( + "byte range 0x{offset:x}..0x{end:x} is outside the input" + )) + }) +} + +fn read_u32(data: &[u8], offset: usize) -> Result { + let bytes = bytes(data, offset, 4)?; + Ok(u32::from_le_bytes(bytes.try_into().map_err(|_| { + Error::Invalid("invalid u32 byte range".to_owned()) + })?)) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn stage1_word_transform_round_trips() { + let key = 0x1234_5678; + let constant = DEFAULT_CIPHER_CONSTANT; + let plain = [0x1122_3344_u32, 0xaabb_ccdd, 0x0102_0304]; + let mut cipher = Vec::new(); + for (index, value) in plain.into_iter().enumerate() { + let index = index as u32; + let word = (value ^ constant.wrapping_mul(index + 1)) + .wrapping_sub((index + 3).wrapping_mul(key)); + cipher.extend_from_slice(&word.to_le_bytes()); + } + let decoded = decrypt_words(&cipher, key, constant).unwrap(); + let expected = plain + .into_iter() + .flat_map(u32::to_le_bytes) + .collect::>(); + assert_eq!(decoded, expected); + } +} diff --git a/senbei-android-engine/src/stream.rs b/senbei-android-engine/src/stream.rs new file mode 100644 index 0000000..b1d4e19 --- /dev/null +++ b/senbei-android-engine/src/stream.rs @@ -0,0 +1,168 @@ +use senbei_android_crypto::gf32_mul_fixed; + +use crate::error::{Error, Result, invalid}; + +pub(crate) const RECORD_SIZE: usize = 0x5c; +pub(crate) const DIRECT_FLAG: u32 = 2; + +#[derive(Debug, Clone, Copy)] +pub(crate) struct Record { + pub index: usize, + pub command_id: u32, + pub flags: u32, + pub image_offset: u32, + pub image_size: u32, + pub metadata_offset: u32, + pub metadata_size: u32, + pub id_copy: u32, + pub entry_offset: u32, + pub init_offset: u32, +} + +impl Record { + pub(crate) fn direct(self) -> bool { + self.flags & DIRECT_FLAG != 0 + } +} + +#[derive(Debug, Clone, Copy)] +pub(crate) struct StreamHeader { + pub encrypted_words: [u32; 2], + pub decrypted_words: [u32; 2], + pub record_state: u32, +} + +pub(crate) fn parse_record_stream( + stream: &[u8], + stream_id: u32, +) -> Result<(StreamHeader, Vec, usize)> { + if stream.len() < 8 { + return invalid(format!( + "stream 0x{stream_id:02X} is shorter than its 8-byte header" + )); + } + let cipher0 = read_u32(stream, 0)?; + let cipher1 = read_u32(stream, 4)?; + let key = stream_id.wrapping_mul(0x9d32_3cd7); + let shift = stream_id & 7; + let base = (key >> shift) + .wrapping_add(0x5e72_7d74) + .wrapping_add(key.wrapping_shl(stream_id & 0xb)) + .wrapping_add(0xf71e_3005); + let plain0 = + gf32_mul_fixed(cipher0.wrapping_add(0xcbf0_c1d8)) ^ 0xeb_e81dba_u32.wrapping_add(base); + let plain1 = gf32_mul_fixed(cipher1.wrapping_add(cipher0)) + ^ 0xeb_e81dba_u32.wrapping_mul(5).wrapping_add(base); + let header = StreamHeader { + encrypted_words: [cipher0, cipher1], + decrypted_words: [plain0, plain1], + record_state: plain1.wrapping_add(base), + }; + + let mut records = Vec::new(); + let mut first_payload = stream.len(); + for index in 0..256_usize { + let start = + 8_usize + .checked_add(index.checked_mul(RECORD_SIZE).ok_or_else(|| { + Error::Invalid("record descriptor offset overflow".to_owned()) + })?) + .ok_or_else(|| Error::Invalid("record descriptor offset overflow".to_owned()))?; + let end = start + .checked_add(RECORD_SIZE) + .ok_or_else(|| Error::Invalid("record descriptor end overflow".to_owned()))?; + if end > stream.len() { + return invalid(format!( + "stream 0x{stream_id:02X} descriptor table is truncated at record {index}" + )); + } + let record = decrypt_record(&stream[start..end], index, header.record_state)?; + if record.id_copy != 0 && record.command_id != record.id_copy { + return invalid(format!( + "stream 0x{stream_id:02X} record {index} command/id mismatch: 0x{:X} != 0x{:X}", + record.command_id, record.id_copy + )); + } + for (offset, size) in [ + (record.image_offset, record.image_size), + (record.metadata_offset, record.metadata_size), + ] { + if offset != 0 && size != 0 { + let offset = usize::try_from(offset).map_err(|_| { + Error::Invalid(format!( + "stream 0x{stream_id:02X} record {index} payload offset exceeds usize" + )) + })?; + if offset >= stream.len() { + return invalid(format!( + "stream 0x{stream_id:02X} record {index} payload offset 0x{offset:x} exceeds stream 0x{:x}", + stream.len() + )); + } + first_payload = first_payload.min(offset); + } + } + records.push(record); + if end == first_payload { + return Ok((header, records, first_payload)); + } + if end > first_payload { + return invalid(format!( + "stream 0x{stream_id:02X} descriptor table crosses first payload at 0x{first_payload:x}" + )); + } + } + invalid(format!( + "stream 0x{stream_id:02X} has no descriptor boundary in 256 records" + )) +} + +fn decrypt_record(raw: &[u8], index: usize, state: u32) -> Result { + if raw.len() != RECORD_SIZE { + return invalid(format!( + "record {index} has size 0x{:x}, expected 0x{RECORD_SIZE:x}", + raw.len() + )); + } + let product = state.wrapping_add(0x96f6_0b71).wrapping_mul(state); + let index_mask = product.wrapping_shl(((index + 1) & 3) as u32); + let mix = state.wrapping_mul(0x06a5_5bcc).wrapping_add(product); + let mut accumulator = 0x7993_4cf6_u32; + let mut feedback = 0xf02f_7685_u32; + let mut words = [0_u32; RECORD_SIZE / 4]; + for (word_index, chunk) in raw.chunks_exact(4).enumerate() { + feedback = feedback.wrapping_mul(feedback); + let cipher = u32::from_le_bytes( + chunk + .try_into() + .map_err(|_| Error::Invalid("record word has an invalid size".to_owned()))?, + ); + let mut value = gf32_mul_fixed(cipher ^ (feedback >> 3)) ^ index_mask; + value = value.wrapping_add(accumulator).wrapping_add(state); + value = value.wrapping_sub(mix >> ((word_index * 4 + 3) & 5)); + words[word_index] = value; + accumulator = accumulator.wrapping_add(0xe64d_33d8); + feedback = cipher; + } + Ok(Record { + index, + command_id: words[0], + flags: words[1], + image_offset: words[2], + image_size: words[3], + metadata_offset: words[4], + metadata_size: words[5], + id_copy: words[6], + entry_offset: words[7], + init_offset: words[8], + }) +} + +fn read_u32(data: &[u8], offset: usize) -> Result { + let bytes = data.get(offset..offset + 4).ok_or_else(|| { + Error::Invalid(format!("record header range 0x{offset:x} is out of bounds")) + })?; + Ok(u32::from_le_bytes(bytes.try_into().map_err(|_| { + Error::Invalid("invalid record u32 range".to_owned()) + })?)) +} diff --git a/senbei-android-metadata/Cargo.toml b/senbei-android-metadata/Cargo.toml new file mode 100644 index 0000000..2dbaeb7 --- /dev/null +++ b/senbei-android-metadata/Cargo.toml @@ -0,0 +1,14 @@ +[package] +name = "senbei-android-metadata" +version.workspace = true +edition.workspace = true +rust-version.workspace = true +license.workspace = true +description = "IL2CPP metadata restoration for Senbei Android" + +[dependencies] +serde.workspace = true +thiserror.workspace = true + +[lints] +workspace = true diff --git a/senbei-android-metadata/src/embedded.rs b/senbei-android-metadata/src/embedded.rs new file mode 100644 index 0000000..78138be --- /dev/null +++ b/senbei-android-metadata/src/embedded.rs @@ -0,0 +1,137 @@ +//! Extraction of the embedded-metadata packaging variant. +//! +//! Some protected il2cpp builds ship no `global-metadata.dat` in the app's +//! assets at all. Instead a slim metadata blob (an older header format with +//! custom record layouts) is embedded in the protected library's data section +//! and wrapped in a per-word XOR layer: a 0x100-byte header whose 64 words each +//! carry their own key, followed by exactly 256 segments with one u32 key each +//! at irregular boundaries. At runtime the protector's il2cpp-side modules +//! regenerate the keys and unwrap the blob in place; the keys are stored +//! nowhere in the image. +//! +//! For the one observed build using this variant the full keystream was +//! recovered from a ciphertext/plaintext pair and is embedded in +//! [`crate::keystream`]. Extraction is therefore content-gated: the wrapped +//! header's first plaintext words are known constants, so a restored image that +//! does not contain them (every other build) is skipped cheaply and nothing is +//! written. +//! +//! The unwrapped blob stores its patched sanity/version fields byte-swapped; +//! they are rewritten to the standard il2cpp metadata magic and version so the +//! output is a well-formed `global-metadata.dat`. + +use crate::keystream::{HEADER_KEYS, SEGMENTS}; + +/// Standard il2cpp metadata sanity magic written over the patched header. +const STANDARD_MAGIC: u32 = 0xfab1_1baf; +/// Standard header version matching the blob's record layout. +const STANDARD_VERSION: u32 = 24; + +/// Plaintext of the first two wrapped header words (the byte-swapped patched +/// sanity/version pair). Also the probe pattern: a restored image contains the +/// embedded blob iff `word[0] ^ HEADER_KEYS[0]` and `word[1] ^ HEADER_KEYS[1]` +/// equal these constants at some 4-aligned offset. +const PROBE_WORDS: [u32; 2] = [0x9732_ca38, 0xbac4_374f]; + +/// Size of the wrapped blob: the last segment's end offset. +pub fn embedded_metadata_size() -> usize { + SEGMENTS[SEGMENTS.len() - 1].0 as usize +} + +/// Locate and unwrap the embedded metadata blob in a restored library image. +/// +/// Returns a standalone, well-formed `global-metadata.dat`, or `None` when the +/// image carries no blob wrapped with the known keystream. +pub fn extract_embedded_metadata(image: &[u8]) -> Option> { + let total = embedded_metadata_size(); + let offset = find_wrapped_header(image)?; + let blob = image.get(offset..offset.checked_add(total)?)?; + + let mut out = blob.to_vec(); + for (i, &key) in HEADER_KEYS.iter().enumerate() { + xor_word(&mut out, 4 * i, key); + } + let mut pos = 0x100_usize; + for &(end, key) in &SEGMENTS { + let end = end as usize; + let mut o = pos; + while o + 4 <= end { + xor_word(&mut out, o, key); + o += 4; + } + pos = end; + } + out[0..4].copy_from_slice(&STANDARD_MAGIC.to_le_bytes()); + out[4..8].copy_from_slice(&STANDARD_VERSION.to_le_bytes()); + Some(out) +} + +/// Scan `image` for the wrapped header probe pattern (4-aligned). +fn find_wrapped_header(image: &[u8]) -> Option { + let mut off = 0; + while off + 8 <= image.len() { + let word = u32::from_le_bytes(image[off..off + 4].try_into().ok()?); + if word ^ HEADER_KEYS[0] == PROBE_WORDS[0] { + let next = u32::from_le_bytes(image[off + 4..off + 8].try_into().ok()?); + if next ^ HEADER_KEYS[1] == PROBE_WORDS[1] { + return Some(off); + } + } + off += 4; + } + None +} + +fn xor_word(data: &mut [u8], offset: usize, key: u32) { + let word = u32::from_le_bytes(data[offset..offset + 4].try_into().expect("word in bounds")); + data[offset..offset + 4].copy_from_slice(&(word ^ key).to_le_bytes()); +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Wrap a synthetic blob with the keystream, then unwrap it back. + #[test] + fn roundtrip_wrapped_blob() { + let total = embedded_metadata_size(); + let mut image = vec![0_u8; total + 0x40]; + // Plaintext blob: standard probe words, then a ramp. + image[0..4].copy_from_slice(&PROBE_WORDS[0].to_le_bytes()); + image[4..8].copy_from_slice(&PROBE_WORDS[1].to_le_bytes()); + for o in (8..total).step_by(4) { + let v = (o as u32).wrapping_mul(0x9e37_79b1); + image[o..o + 4].copy_from_slice(&v.to_le_bytes()); + } + // Wrap with the keystream. + for (i, &key) in HEADER_KEYS.iter().enumerate() { + xor_word(&mut image, 4 * i, key); + } + let mut pos = 0x100_usize; + for &(end, key) in &SEGMENTS { + let mut o = pos; + while o + 4 <= end as usize { + xor_word(&mut image, o, key); + o += 4; + } + pos = end as usize; + } + + let out = extract_embedded_metadata(&image).expect("blob found"); + assert_eq!(out.len(), total); + // Header rewritten to the standard magic/version… + assert_eq!(&out[0..4], &STANDARD_MAGIC.to_le_bytes()); + assert_eq!(&out[4..8], &STANDARD_VERSION.to_le_bytes()); + // …and the body round-trips. + for o in (8..total).step_by(4) { + let v = (o as u32).wrapping_mul(0x9e37_79b1); + assert_eq!(&out[o..o + 4], &v.to_le_bytes(), "word at {o:#x}"); + } + } + + #[test] + fn no_blob_in_plain_data() { + let image = vec![0xAB_u8; 0x1000]; + assert!(extract_embedded_metadata(&image).is_none()); + } +} diff --git a/senbei-android-metadata/src/keystream.rs b/senbei-android-metadata/src/keystream.rs new file mode 100644 index 0000000..17ba499 --- /dev/null +++ b/senbei-android-metadata/src/keystream.rs @@ -0,0 +1,274 @@ +/// Per-word XOR keystream for the embedded-metadata packaging variant, +/// recovered from a ciphertext/plaintext pair of one observed build. +/// Key derivation for future builds is untraced; other builds simply do +/// not match the header probe and are left untouched. +pub(crate) const HEADER_KEYS: [u32; 64] = [ + 0x39184c70, 0xd901afd4, 0x19b98815, 0x132906ed, 0x663e8ace, 0x299b1952, 0xe5404ab8, 0xd93b331c, + 0xb67d3761, 0x42da9259, 0xc29c7a59, 0x17cb841c, 0xd0bcb9c6, 0x21db779b, 0x43874deb, 0x89bf697b, + 0x0b7f97b4, 0xbe1c59f7, 0xc653ad92, 0x8cdf4336, 0x5e0b6b68, 0x1bd4d668, 0x7250ed61, 0x31a36491, + 0xaf144dcd, 0xc1e387d0, 0x9d6df5b7, 0x78514f32, 0xc2648cbf, 0x3b8272a5, 0xd2053679, 0x4b18af77, + 0x71b9ebdd, 0x0094daaa, 0xf3adfed8, 0xc0d082bc, 0xae5e523c, 0xa8dec0be, 0x090a7784, 0x2c0483d6, + 0x95f0e8f7, 0x234de6d4, 0xa7464527, 0x3b1c531d, 0xc2b31d82, 0xe1c60be0, 0x3d65a0c2, 0x2ea7d77a, + 0x4ababadb, 0xce484b16, 0x59ab3f99, 0x10a9a463, 0x70e2f78a, 0x0ed71c9c, 0xf8996b2e, 0xff637928, + 0xf413313d, 0x77c57bf9, 0xdab41dba, 0x0cd2ccbc, 0x3b2fbde3, 0x0b19b14d, 0xd2645dbc, 0x318113d4, +]; + +/// (segment end offset, segment key) pairs; offsets relative to blob start. +pub(crate) const SEGMENTS: [(u32, u32); 256] = [ + (0x3915c, 0xbb5dda1a), + (0x736b4, 0x906dbe0f), + (0xaedc4, 0x1e4ca8bd), + (0xc506c, 0xe603cb21), + (0xdfe34, 0x93bec702), + (0xe8f10, 0x9a9f429f), + (0x139088, 0x125a8b3f), + (0x20bd30, 0x69a6395f), + (0x20c548, 0xca807b9a), + (0x20c774, 0x8b8880c4), + (0x300518, 0x48087852), + (0x36c07c, 0x32aa7b5b), + (0x448b7c, 0x2e668589), + (0x4592b0, 0x292e07d9), + (0x45d374, 0x83b0a0ef), + (0x474520, 0x8983245d), + (0x47a1d4, 0xefb941b7), + (0x4bdc74, 0x7c3b3458), + (0x4c34c8, 0xee0a87b3), + (0x4f1068, 0xf6a2069f), + (0x51601c, 0x2e83612b), + (0x549d40, 0xb413a58f), + (0x56a714, 0x95596da3), + (0x573c98, 0x68513e8d), + (0x59d058, 0xc4ff5f9a), + (0x5c4b34, 0x249ed022), + (0x5f19bc, 0xc27272d3), + (0x5f47c8, 0xd73aa37b), + (0x627df4, 0x002334ba), + (0x648f54, 0x868bb6c9), + (0x6718a4, 0x17ff0ef4), + (0x6a88b8, 0x22cfbc5f), + (0x742dbc, 0x152072dd), + (0x75603c, 0xbd31be45), + (0x783238, 0x45911d6a), + (0x7b94d8, 0x6f281add), + (0x800060, 0xcf58c8d0), + (0x819b50, 0xb40f0276), + (0x82dea4, 0x1a1a8402), + (0x880210, 0xf2c0824a), + (0x8e4f08, 0x86c9ba90), + (0x8ea544, 0x0e928544), + (0x931454, 0xc3fa017b), + (0x94eb70, 0x1dbe612a), + (0x95993c, 0x902498fe), + (0x98d2b0, 0xb7760451), + (0x992034, 0x711cddfc), + (0x9e14a0, 0x8bd95e64), + (0xa1d2d0, 0xbfdce920), + (0xa21ed8, 0x90cf0372), + (0xa4d2b8, 0x91e88c9c), + (0xa76f8c, 0x9c721e61), + (0xac5ba4, 0xbda16e3e), + (0xaf070c, 0xe02b6799), + (0xaf3a78, 0x32953b4e), + (0xb32510, 0x47ea48db), + (0xb46550, 0x1443e512), + (0xb54998, 0x9e123a75), + (0xb5e24c, 0xe11a8efd), + (0xb625cc, 0x3facfbf4), + (0xb66ec8, 0x76c0c452), + (0xb67ad0, 0x4de4ed6c), + (0xb7ba5c, 0xe622d97a), + (0xb85a90, 0x6f564f8b), + (0xbff8b8, 0x3e25d671), + (0xc03e50, 0x3563fc2b), + (0xc6e958, 0xda8bc3b0), + (0xc87f7c, 0x5a9d2269), + (0xcb36a4, 0x0ab420cc), + (0xcbe4d0, 0x9bbb091e), + (0xccd7dc, 0x9e4fd577), + (0xd078c0, 0x4b655ae1), + (0xd275dc, 0x5ca2a2f4), + (0xd2c840, 0xdb437f0d), + (0xd3296c, 0x66487f75), + (0xd7cbe8, 0xf5427945), + (0xd8e0a0, 0x9a65bdb6), + (0xda2ed4, 0x46dea4b3), + (0xda6f1c, 0xb9916a02), + (0xdee9ac, 0x18800a5c), + (0xe3673c, 0x4afab3cd), + (0xe65420, 0x52e80204), + (0xe861f8, 0x639a02d7), + (0xeb61c0, 0x21077eba), + (0xed51dc, 0x17be91d8), + (0xf048a4, 0xd30cc8cb), + (0xf3b274, 0xdfb43f3f), + (0xf76ac8, 0x63a8b363), + (0xf84b64, 0x16508a16), + (0xf8bb2c, 0x22ce110d), + (0xfb3390, 0xf09a4eb2), + (0xff07bc, 0xd2bb0e2c), + (0x102d32c, 0xb424012c), + (0x10795b0, 0x07338bb9), + (0x108d65c, 0x5d68f86e), + (0x10ce528, 0x1826c952), + (0x10d3528, 0xa7473860), + (0x10dca58, 0x92435967), + (0x1115c78, 0x061200f4), + (0x1171098, 0x94f538a1), + (0x117ebf0, 0xd8731d88), + (0x1186638, 0x4381b3f9), + (0x118afdc, 0xf25ff376), + (0x11ee3b4, 0x29605488), + (0x11f182c, 0x04367932), + (0x11f41dc, 0xaeaccadd), + (0x11ff0f4, 0x7c4d358e), + (0x120caac, 0xacbc8412), + (0x12437cc, 0x3e0ac7e9), + (0x124edf8, 0x06f523fd), + (0x1263ba0, 0x0a1b9763), + (0x12943f0, 0x24a86ba4), + (0x12d9230, 0x0cd82e2e), + (0x12fd9b4, 0xf3903fb9), + (0x135a198, 0x2887f4a3), + (0x1366180, 0x9f0d7ca5), + (0x13680a4, 0x61e9a459), + (0x13a1b44, 0xe61623a4), + (0x13a860c, 0xdc44c798), + (0x13c024c, 0xc90f7be6), + (0x1475c00, 0xc2f338b3), + (0x1480aa8, 0xb7b0609e), + (0x14fb82c, 0x748e3939), + (0x1511184, 0x98426fcf), + (0x153c144, 0x1a452d5d), + (0x1547838, 0x7dd360e9), + (0x15565d4, 0x1d8f093b), + (0x156e298, 0x102a1524), + (0x159df70, 0xe42613f7), + (0x15a13d0, 0xafc5fdc6), + (0x15e7f24, 0x84fcd342), + (0x15f0878, 0x55038958), + (0x1614210, 0xe0602ae4), + (0x1631b3c, 0xce2765f6), + (0x164eb70, 0xf772dac5), + (0x1688b68, 0x5f1a72c9), + (0x16d5f8c, 0x7c77747d), + (0x16e76dc, 0xac0e16fb), + (0x1726374, 0x4a1e7fd7), + (0x173455c, 0x870856b4), + (0x17697d8, 0xbb2f0a5c), + (0x176ed60, 0xc937b386), + (0x1784fa8, 0x5e676ab2), + (0x17ae3a0, 0xdbf662a1), + (0x1866c7c, 0x4e3f1a7d), + (0x186b844, 0xe30fce60), + (0x18b507c, 0xdfc73c88), + (0x18c2f64, 0xb7ee08e0), + (0x18c8010, 0xd1471a25), + (0x18de290, 0x292e6310), + (0x19140d0, 0x9f346f05), + (0x192c590, 0xf1eb61bf), + (0x194fca0, 0x8888b1df), + (0x1959d34, 0x92b89d15), + (0x196c0c4, 0x5e152de5), + (0x19a5710, 0x866e7bfa), + (0x19abfd4, 0x3084ae26), + (0x19b1550, 0x0581836f), + (0x19b7214, 0xeefc34eb), + (0x19c523c, 0xc980335d), + (0x19dc4c0, 0x019084e6), + (0x19dfb8c, 0xdb1a21a7), + (0x19fbf3c, 0xec84cc17), + (0x1a29b18, 0xcb31da7d), + (0x1a4c670, 0xc5fe570e), + (0x1a97024, 0xbbd80964), + (0x1ac33bc, 0xe186586d), + (0x1acd124, 0x1e413252), + (0x1ad9bac, 0x48fc4c75), + (0x1b1b728, 0x8071d7a5), + (0x1b31d78, 0x9d958013), + (0x1badb24, 0x2f236951), + (0x1bccc00, 0x7023c620), + (0x1bdab2c, 0x88b1e4b8), + (0x1c000dc, 0x9e43291a), + (0x1c9f0cc, 0x27a7d592), + (0x1cd1328, 0x9c0bcc88), + (0x1cd79a0, 0x63e0ed75), + (0x1d0e484, 0xf51a0d3d), + (0x1d17b10, 0xbfd2a7ac), + (0x1d930c4, 0xf6b9e877), + (0x1db115c, 0xf3eb7e37), + (0x1df16b4, 0x682326ff), + (0x1e389c0, 0xea11f566), + (0x1eb7e48, 0x3dc5fa76), + (0x1ec38fc, 0x296ffc1d), + (0x1ee87a0, 0x1b9f7fd4), + (0x1f19f88, 0x78972e8f), + (0x1f33a0c, 0x390c2deb), + (0x1f4e0fc, 0xe05e8c6b), + (0x1f5a718, 0x367432ae), + (0x1f61dcc, 0x7063e58a), + (0x1f85878, 0x21c00cea), + (0x1fc043c, 0x2676aaaa), + (0x1ffdb94, 0xc270eb02), + (0x202a618, 0x3a98aed2), + (0x2037b34, 0x115d5afc), + (0x203d92c, 0x11bced76), + (0x203da14, 0xf2628105), + (0x2066014, 0x97f32700), + (0x208a908, 0xa68e2f71), + (0x20ab8ac, 0x1daa2a78), + (0x20ba504, 0x73919ef6), + (0x20e71e0, 0x0b3fd1d3), + (0x2102278, 0x6c123def), + (0x21166dc, 0xee354161), + (0x2126478, 0x299493f4), + (0x2137090, 0x05ae2007), + (0x2148270, 0x34b52663), + (0x21482ac, 0xe381b5b6), + (0x21813b8, 0x94244de1), + (0x21a41e8, 0x02c38df5), + (0x21a8c4c, 0xf72700dd), + (0x21abbac, 0x34c2e7b5), + (0x21bcb24, 0x442739ad), + (0x21cbe84, 0x6e40d22c), + (0x21e2798, 0xdbf774d0), + (0x21f892c, 0xe90f1e0c), + (0x222beec, 0xa27f27f3), + (0x22394f4, 0x7f999a4f), + (0x22437ec, 0xf12d28f8), + (0x22480b0, 0xf58f3a7d), + (0x2261a0c, 0x89b28301), + (0x22a76c8, 0x1fe501e2), + (0x22b2018, 0xf079db5f), + (0x22cc610, 0xaf7d17b7), + (0x22cd4f8, 0x71c010cb), + (0x22d016c, 0x4a8daed0), + (0x22e1c04, 0xe1201aca), + (0x22f9994, 0xf3f0e4ee), + (0x2384f5c, 0x6b8a5eb1), + (0x23d5ecc, 0x5298a9c4), + (0x23e15b0, 0xc7bf0afb), + (0x23e248c, 0x2d67fecf), + (0x2407898, 0x4eef422a), + (0x241695c, 0x33ba9ce8), + (0x243e8b0, 0x833c1d2c), + (0x2460b64, 0x819c96ee), + (0x247caec, 0x0ebccbd6), + (0x24832b4, 0xf789d4b6), + (0x24938b8, 0x9f63baeb), + (0x24a7c64, 0x3384e552), + (0x24bce94, 0x7bfec208), + (0x24bd8f4, 0x9b5260cc), + (0x24ce8ec, 0xaf854888), + (0x24e741c, 0xda82f062), + (0x254401c, 0xbb1a5d5a), + (0x25a6a64, 0x24d202d3), + (0x2617878, 0x6ac71e5f), + (0x2617df0, 0x0e76bd90), + (0x268e9b8, 0x874d931c), + (0x26a8848, 0xefefd680), + (0x26e4f10, 0xfc2799f7), + (0x26e93d8, 0x1930ad55), + (0x26eea84, 0xfa2f742f), + (0x2701f30, 0xed9b92c4), +]; diff --git a/senbei-android-metadata/src/lib.rs b/senbei-android-metadata/src/lib.rs new file mode 100644 index 0000000..0251c69 --- /dev/null +++ b/senbei-android-metadata/src/lib.rs @@ -0,0 +1,11 @@ +//! Static IL2CPP metadata restoration interfaces. + +mod embedded; +mod keystream; +mod method_tokens; + +pub use embedded::{embedded_metadata_size, extract_embedded_metadata}; +pub use method_tokens::{ + DEFAULT_METHOD_TOKEN_SEED, Error, ImageKeyDiscovery, Report, SeedDiscoveryReport, + discover_method_token_seeds, restore_method_tokens, +}; diff --git a/senbei-android-metadata/src/method_tokens.rs b/senbei-android-metadata/src/method_tokens.rs new file mode 100644 index 0000000..1a13817 --- /dev/null +++ b/senbei-android-metadata/src/method_tokens.rs @@ -0,0 +1,659 @@ +//! Static restoration of protected IL2CPP v31 method tokens. + +use serde::Serialize; + +/// Seed embedded in the current `libil2cpp` module `0x0C`. +pub const DEFAULT_METHOD_TOKEN_SEED: u32 = 0xa6fa_e968; + +const MAGIC: u32 = 0xfab1_1baf; +const SUPPORTED_VERSION: u32 = 31; +const HDR_METHODS: usize = 0x30; +const HDR_TYPES: usize = 0xa0; +const HDR_IMAGES: usize = 0xa8; +const METHOD_STRIDE: usize = 0x24; +const METHOD_TOKEN_OFFSET: usize = 0x18; +const TYPE_STRIDE: usize = 0x58; +const TYPE_METHOD_START_OFFSET: usize = 0x24; +const TYPE_METHOD_COUNT_OFFSET: usize = 0x40; +const IMAGE_STRIDE: usize = 0x28; +const IMAGE_TYPE_START_OFFSET: usize = 0x08; +const IMAGE_TYPE_COUNT_OFFSET: usize = 0x0c; +const METHOD_TOKEN_TABLE: u32 = 0x0600_0000; + +/// Summary of one metadata restoration pass. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct Report { + pub version: u32, + pub seed: String, + pub encryption_status: String, + pub images: usize, + pub images_with_methods: usize, + pub types: usize, + pub methods: usize, + pub visited_methods: usize, + pub already_correct_before: usize, + pub correct_after: usize, + pub changed_tokens: usize, + pub transformed_images: usize, +} + +/// Per-image constraints recovered from the encrypted MethodDef RID +/// permutation. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct ImageKeyDiscovery { + pub image: usize, + pub method_count: u32, + pub modulus: u32, + pub clean: bool, + pub seed_residues: Vec, +} + +/// Result of statically testing the known five-round permutation against a +/// metadata file without assuming a seed. +#[derive(Debug, Clone, PartialEq, Eq, Serialize)] +pub struct SeedDiscoveryReport { + pub version: u32, + pub images: Vec, + pub seed_candidates: Vec, +} + +/// Metadata parsing or validation failure. +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum Error { + #[error("not an IL2CPP global-metadata.dat")] + NotMetadata, + #[error("unsupported metadata version {0}")] + UnsupportedVersion(u32), + #[error("malformed metadata: {0}")] + Malformed(String), + #[error("method-token restoration failed: {0}")] + Validation(String), +} + +type Result = std::result::Result; + +fn malformed(message: impl Into) -> Result { + Err(Error::Malformed(message.into())) +} + +fn validation(message: impl Into) -> Result { + Err(Error::Validation(message.into())) +} + +fn bytes(data: &[u8], offset: usize, size: usize) -> Result<&[u8]> { + let end = offset + .checked_add(size) + .ok_or_else(|| Error::Malformed("byte range overflow".to_owned()))?; + data.get(offset..end).ok_or_else(|| { + Error::Malformed(format!( + "byte range 0x{offset:x}..0x{end:x} is out of bounds" + )) + }) +} + +fn read_u16(data: &[u8], offset: usize) -> Result { + let value: [u8; 2] = bytes(data, offset, 2)? + .try_into() + .map_err(|_| Error::Malformed("invalid u16 range".to_owned()))?; + Ok(u16::from_le_bytes(value)) +} + +fn read_u32(data: &[u8], offset: usize) -> Result { + let value: [u8; 4] = bytes(data, offset, 4)? + .try_into() + .map_err(|_| Error::Malformed("invalid u32 range".to_owned()))?; + Ok(u32::from_le_bytes(value)) +} + +fn read_i32(data: &[u8], offset: usize) -> Result { + let value: [u8; 4] = bytes(data, offset, 4)? + .try_into() + .map_err(|_| Error::Malformed("invalid i32 range".to_owned()))?; + Ok(i32::from_le_bytes(value)) +} + +fn table(data: &[u8], header_offset: usize) -> Result<(usize, usize)> { + let offset = read_u32(data, header_offset)? as usize; + let size = read_u32(data, header_offset + 4)? as usize; + bytes(data, offset, size)?; + Ok((offset, size)) +} + +#[inline] +fn inverse_round(mut value: u32, count: u32, key: u32) -> u32 { + let mirror = count.wrapping_mul(2).wrapping_sub(1); + if value & 1 != 0 { + value = mirror.wrapping_sub(value); + } + value >>= 1; + if value >= count { + value = mirror.wrapping_sub(value); + } + let value = value.wrapping_sub(key); + if value > count { + value.wrapping_add(count) + } else { + value + } +} + +fn decrypt_rid(rid: u32, low: u32, high: u32, seed: u32) -> Result { + let count = high + .checked_add(1) + .and_then(|value| value.checked_sub(low)) + .ok_or_else(|| Error::Validation("invalid image RID interval".to_owned()))?; + if count < 2 { + return validation("RID inverse permutation requires at least two entries"); + } + let half = count / 2; + if half == 0 { + return validation("RID inverse permutation has a zero divisor"); + } + let key = seed % half + count / 4; + let mut value = rid + .checked_sub(low) + .ok_or_else(|| Error::Validation("encrypted RID lies below image minimum".to_owned()))?; + for _ in 0..5 { + value = inverse_round(value, count, key); + } + value + .checked_add(low) + .ok_or_else(|| Error::Validation("restored RID overflow".to_owned())) +} + +/// Restore MethodDef RID values exactly as module `0x0C` does. +/// +/// The operation is idempotent for tooling purposes: an image whose tokens are +/// already canonical is detected and left untouched instead of applying the +/// native inverse permutation a second time. +pub fn restore_method_tokens(data: &[u8], seed: u32) -> Result<(Vec, Report)> { + if read_u32(data, 0).ok() != Some(MAGIC) { + return Err(Error::NotMetadata); + } + let version = read_u32(data, 4)?; + if version != SUPPORTED_VERSION { + return Err(Error::UnsupportedVersion(version)); + } + + let (method_offset, method_size) = table(data, HDR_METHODS)?; + let (type_offset, type_size) = table(data, HDR_TYPES)?; + let (image_offset, image_size) = table(data, HDR_IMAGES)?; + if method_size % METHOD_STRIDE != 0 + || type_size % TYPE_STRIDE != 0 + || image_size % IMAGE_STRIDE != 0 + { + return malformed("v31 table size is not divisible by its entry stride"); + } + let method_count = method_size / METHOD_STRIDE; + let type_count = type_size / TYPE_STRIDE; + let image_count = image_size / IMAGE_STRIDE; + let mut owners = vec![u32::MAX; method_count]; + let mut output = data.to_vec(); + let mut images_with_methods = 0_usize; + let mut visited_methods = 0_usize; + let mut already_correct_before = 0_usize; + let mut correct_after = 0_usize; + let mut changed_tokens = 0_usize; + let mut transformed_images = 0_usize; + + for image_index in 0..image_count { + let image_base = image_offset + image_index * IMAGE_STRIDE; + let type_start = read_i32(data, image_base + IMAGE_TYPE_START_OFFSET)?; + let type_start = usize::try_from(type_start) + .map_err(|_| Error::Malformed(format!("image {image_index} has negative typeStart")))?; + let type_entries = read_u32(data, image_base + IMAGE_TYPE_COUNT_OFFSET)? as usize; + let type_end = type_start + .checked_add(type_entries) + .ok_or_else(|| Error::Malformed("image type range overflow".to_owned()))?; + if type_end > type_count { + return malformed(format!("image {image_index} type range exceeds the table")); + } + + let mut methods = Vec::new(); + for type_index in type_start..type_end { + let type_base = type_offset + type_index * TYPE_STRIDE; + let method_entries = read_u16(data, type_base + TYPE_METHOD_COUNT_OFFSET)? as usize; + if method_entries == 0 { + continue; + } + let method_start = read_i32(data, type_base + TYPE_METHOD_START_OFFSET)?; + let method_start = usize::try_from(method_start).map_err(|_| { + Error::Malformed(format!( + "type {type_index} has methods but negative methodStart" + )) + })?; + let method_end = method_start + .checked_add(method_entries) + .ok_or_else(|| Error::Malformed("type method range overflow".to_owned()))?; + if method_end > method_count { + return malformed(format!("type {type_index} method range exceeds the table")); + } + for (method_index, owner) in owners + .iter_mut() + .enumerate() + .take(method_end) + .skip(method_start) + { + if *owner != u32::MAX { + return malformed(format!("method {method_index} belongs to multiple images")); + } + *owner = u32::try_from(image_index) + .map_err(|_| Error::Malformed("image index exceeds u32".to_owned()))?; + methods.push(method_index); + } + } + if methods.is_empty() { + continue; + } + images_with_methods += 1; + visited_methods += methods.len(); + let method_base = *methods + .iter() + .min() + .ok_or_else(|| Error::Malformed("nonempty image lost its method minimum".to_owned()))?; + let method_last = *methods + .iter() + .max() + .ok_or_else(|| Error::Malformed("nonempty image lost its method maximum".to_owned()))?; + if method_last - method_base + 1 != methods.len() { + return malformed(format!( + "image {image_index} method block is not contiguous" + )); + } + + let mut tokens = Vec::with_capacity(methods.len()); + let mut image_already_clean = true; + for &method_index in &methods { + let token_offset = method_offset + method_index * METHOD_STRIDE + METHOD_TOKEN_OFFSET; + let token = read_u32(data, token_offset)?; + if token & 0xff00_0000 != METHOD_TOKEN_TABLE { + return malformed(format!( + "method {method_index} has non-MethodDef token 0x{token:08x}" + )); + } + let expected = u32::try_from(method_index - method_base + 1) + .map_err(|_| Error::Validation("local method RID exceeds u32".to_owned()))?; + let rid = token & 0x00ff_ffff; + if rid == expected { + already_correct_before += 1; + } else { + image_already_clean = false; + } + tokens.push((method_index, token_offset, token, expected)); + } + + if image_already_clean { + correct_after += tokens.len(); + continue; + } + transformed_images += 1; + let low = tokens + .iter() + .map(|(_, _, token, _)| token & 0x00ff_ffff) + .min() + .ok_or_else(|| Error::Validation("image has no MethodDef RID".to_owned()))?; + let high = tokens + .iter() + .map(|(_, _, token, _)| token & 0x00ff_ffff) + .max() + .ok_or_else(|| Error::Validation("image has no MethodDef RID".to_owned()))?; + if high <= 1 { + return validation(format!( + "image {image_index} is noncanonical but native R > 1 gate would skip it" + )); + } + let interval = high - low + 1; + if interval as usize != tokens.len() { + return validation(format!( + "image {image_index} RID interval {low}..={high} is not a permutation" + )); + } + for (method_index, token_offset, token, expected) in tokens { + let restored_rid = decrypt_rid(token & 0x00ff_ffff, low, high, seed)?; + if restored_rid != expected { + return validation(format!( + "method {method_index} restored RID {restored_rid} != expected {expected}" + )); + } + let restored_token = METHOD_TOKEN_TABLE | restored_rid; + if restored_token != token { + output[token_offset..token_offset + 4] + .copy_from_slice(&restored_token.to_le_bytes()); + changed_tokens += 1; + } + correct_after += 1; + } + } + + if owners.contains(&u32::MAX) { + return malformed("one or more method definitions are not owned by an image"); + } + if visited_methods != method_count || correct_after != method_count { + return validation(format!( + "method coverage mismatch: visited={visited_methods}, correct={correct_after}, total={method_count}" + )); + } + + Ok(( + output, + Report { + version, + seed: format!("0x{seed:08X}"), + encryption_status: if changed_tokens == 0 { + "clean".to_owned() + } else { + "encrypted".to_owned() + }, + images: image_count, + images_with_methods, + types: type_count, + methods: method_count, + visited_methods, + already_correct_before, + correct_after, + changed_tokens, + transformed_images, + }, + )) +} + +/// Discover seeds compatible with the known v31 five-round RID permutation. +/// +/// This is diagnostic and does not modify metadata. It enumerates the only +/// possible per-image key residues and intersects them over the 32-bit seed +/// domain. An empty candidate list means that the sample changed the +/// permutation itself rather than merely embedding a different seed. +pub fn discover_method_token_seeds(data: &[u8]) -> Result { + if read_u32(data, 0).ok() != Some(MAGIC) { + return Err(Error::NotMetadata); + } + let version = read_u32(data, 4)?; + if version != SUPPORTED_VERSION { + return Ok(SeedDiscoveryReport { + version, + images: Vec::new(), + seed_candidates: Vec::new(), + }); + } + let (method_offset, method_size) = table(data, HDR_METHODS)?; + let (type_offset, type_size) = table(data, HDR_TYPES)?; + let (image_offset, image_size) = table(data, HDR_IMAGES)?; + if method_size % METHOD_STRIDE != 0 + || type_size % TYPE_STRIDE != 0 + || image_size % IMAGE_STRIDE != 0 + { + return malformed("v31 table size is not divisible by its entry stride"); + } + let method_count = method_size / METHOD_STRIDE; + let type_count = type_size / TYPE_STRIDE; + let image_count = image_size / IMAGE_STRIDE; + let mut reports = Vec::with_capacity(image_count); + for image_index in 0..image_count { + let image_base = image_offset + image_index * IMAGE_STRIDE; + let type_start = usize::try_from(read_i32(data, image_base + IMAGE_TYPE_START_OFFSET)?) + .map_err(|_| Error::Malformed(format!("image {image_index} has negative typeStart")))?; + let type_entries = read_u32(data, image_base + IMAGE_TYPE_COUNT_OFFSET)? as usize; + let type_end = type_start + .checked_add(type_entries) + .ok_or_else(|| Error::Malformed("image type range overflow".to_owned()))?; + if type_end > type_count { + return malformed(format!("image {image_index} type range exceeds the table")); + } + let mut methods = Vec::new(); + for type_index in type_start..type_end { + let type_base = type_offset + type_index * TYPE_STRIDE; + let method_entries = read_u16(data, type_base + TYPE_METHOD_COUNT_OFFSET)? as usize; + if method_entries == 0 { + continue; + } + let method_start = + usize::try_from(read_i32(data, type_base + TYPE_METHOD_START_OFFSET)?).map_err( + |_| Error::Malformed(format!("type {type_index} has negative methodStart")), + )?; + let method_end = method_start + .checked_add(method_entries) + .ok_or_else(|| Error::Malformed("type method range overflow".to_owned()))?; + if method_end > method_count { + return malformed(format!("type {type_index} method range exceeds the table")); + } + methods.extend(method_start..method_end); + } + if methods.is_empty() { + reports.push(ImageKeyDiscovery { + image: image_index, + method_count: 0, + modulus: 0, + clean: true, + seed_residues: Vec::new(), + }); + continue; + } + let method_base = *methods + .iter() + .min() + .ok_or_else(|| Error::Malformed("image method minimum is missing".to_owned()))?; + let method_last = *methods + .iter() + .max() + .ok_or_else(|| Error::Malformed("image method maximum is missing".to_owned()))?; + if method_last - method_base + 1 != methods.len() { + return validation(format!( + "image {image_index} method block is not contiguous" + )); + } + let mut values = Vec::with_capacity(methods.len()); + let mut clean = true; + for method_index in methods { + let token = read_u32( + data, + method_offset + method_index * METHOD_STRIDE + METHOD_TOKEN_OFFSET, + )?; + if token & 0xff00_0000 != METHOD_TOKEN_TABLE { + return validation(format!( + "method {method_index} has non-MethodDef token 0x{token:08x}" + )); + } + let expected = u32::try_from(method_index - method_base + 1) + .map_err(|_| Error::Validation("local method RID exceeds u32".to_owned()))?; + let rid = token & 0x00ff_ffff; + clean &= rid == expected; + values.push((rid, expected)); + } + let count = u32::try_from(values.len()) + .map_err(|_| Error::Validation("image method count exceeds u32".to_owned()))?; + if clean { + reports.push(ImageKeyDiscovery { + image: image_index, + method_count: count, + modulus: count / 2, + clean, + seed_residues: Vec::new(), + }); + continue; + } + let low = values + .iter() + .map(|(rid, _)| *rid) + .min() + .ok_or_else(|| Error::Validation("image has no encrypted RID".to_owned()))?; + let high = values + .iter() + .map(|(rid, _)| *rid) + .max() + .ok_or_else(|| Error::Validation("image has no encrypted RID".to_owned()))?; + if high - low + 1 != count || count < 2 { + return validation(format!( + "image {image_index} RID interval is not a permutation" + )); + } + let half = count / 2; + let quarter = count / 4; + let mut residues = Vec::new(); + for key_delta in 0..half { + let key = quarter + key_delta; + let valid = values + .iter() + .all(|(rid, expected)| decrypt_rid_with_key(*rid, low, high, key) == *expected); + if valid { + residues.push(key_delta); + } + } + reports.push(ImageKeyDiscovery { + image: image_index, + method_count: count, + modulus: half, + clean, + seed_residues: residues, + }); + } + + let constraints = reports + .iter() + .filter(|report| !report.clean) + .collect::>(); + let mut seeds = Vec::new(); + if let Some(anchor) = constraints.iter().max_by_key(|report| report.modulus) { + for &residue in &anchor.seed_residues { + let mut candidate = u64::from(residue); + let modulus = u64::from(anchor.modulus); + while candidate <= u64::from(u32::MAX) { + let valid = constraints.iter().all(|report| { + report.modulus != 0 + && !report.seed_residues.is_empty() + && report + .seed_residues + .iter() + .any(|&value| candidate % u64::from(report.modulus) == u64::from(value)) + }); + if valid { + seeds.push(candidate as u32); + } + candidate = candidate.saturating_add(modulus); + } + } + } + seeds.sort_unstable(); + seeds.dedup(); + Ok(SeedDiscoveryReport { + version, + images: reports, + seed_candidates: seeds, + }) +} + +fn decrypt_rid_with_key(rid: u32, low: u32, high: u32, key: u32) -> u32 { + let count = high - low + 1; + let mut value = rid - low; + for _ in 0..5 { + value = inverse_round(value, count, key); + } + value + low +} + +#[cfg(test)] +mod tests { + use super::*; + + fn put_u16(data: &mut [u8], offset: usize, value: u16) { + data[offset..offset + 2].copy_from_slice(&value.to_le_bytes()); + } + + fn put_u32(data: &mut [u8], offset: usize, value: u32) { + data[offset..offset + 4].copy_from_slice(&value.to_le_bytes()); + } + + fn encrypted_rid(expected: u32, count: u32, seed: u32) -> u32 { + (1..=count) + .find(|&candidate| decrypt_rid(candidate, 1, count, seed) == Ok(expected)) + .expect("inverse permutation must be bijective") + } + + fn build(tokens: &[u32]) -> (Vec, usize) { + let header_size = 0x100; + let images = header_size; + let types = images + IMAGE_STRIDE; + let methods = types + 2 * TYPE_STRIDE; + let mut data = vec![0_u8; methods + tokens.len() * METHOD_STRIDE]; + put_u32(&mut data, 0, MAGIC); + put_u32(&mut data, 4, SUPPORTED_VERSION); + put_u32(&mut data, HDR_METHODS, methods as u32); + put_u32( + &mut data, + HDR_METHODS + 4, + (tokens.len() * METHOD_STRIDE) as u32, + ); + put_u32(&mut data, HDR_TYPES, types as u32); + put_u32(&mut data, HDR_TYPES + 4, (2 * TYPE_STRIDE) as u32); + put_u32(&mut data, HDR_IMAGES, images as u32); + put_u32(&mut data, HDR_IMAGES + 4, IMAGE_STRIDE as u32); + put_u32(&mut data, images + IMAGE_TYPE_START_OFFSET, 0); + put_u32(&mut data, images + IMAGE_TYPE_COUNT_OFFSET, 2); + // Deliberately traverse the high method indices first. + put_u32(&mut data, types + TYPE_METHOD_START_OFFSET, 4); + put_u16(&mut data, types + TYPE_METHOD_COUNT_OFFSET, 3); + put_u32(&mut data, types + TYPE_STRIDE + TYPE_METHOD_START_OFFSET, 0); + put_u16(&mut data, types + TYPE_STRIDE + TYPE_METHOD_COUNT_OFFSET, 4); + for (index, &token) in tokens.iter().enumerate() { + put_u32( + &mut data, + methods + index * METHOD_STRIDE + METHOD_TOKEN_OFFSET, + token, + ); + } + (data, methods) + } + + #[test] + fn restores_five_round_permutation_by_physical_method_index() { + let tokens = (1..=7) + .map(|expected| { + METHOD_TOKEN_TABLE | encrypted_rid(expected, 7, DEFAULT_METHOD_TOKEN_SEED) + }) + .collect::>(); + let (data, methods) = build(&tokens); + let (restored, report) = + restore_method_tokens(&data, DEFAULT_METHOD_TOKEN_SEED).expect("restore"); + assert_eq!(report.encryption_status, "encrypted"); + assert!(report.changed_tokens > 0); + assert_eq!(report.correct_after, 7); + for index in 0..7 { + assert_eq!( + read_u32( + &restored, + methods + index * METHOD_STRIDE + METHOD_TOKEN_OFFSET + ) + .expect("token"), + METHOD_TOKEN_TABLE | (index as u32 + 1) + ); + } + } + + #[test] + fn clean_metadata_is_idempotent() { + let tokens = (1..=7) + .map(|rid| METHOD_TOKEN_TABLE | rid) + .collect::>(); + let (data, _) = build(&tokens); + let (restored, report) = + restore_method_tokens(&data, DEFAULT_METHOD_TOKEN_SEED).expect("restore"); + assert_eq!(report.encryption_status, "clean"); + assert_eq!(report.changed_tokens, 0); + assert_eq!(restored, data); + } + + #[test] + fn encrypted_metadata_rejects_the_wrong_seed() { + let tokens = (1..=7) + .map(|expected| { + METHOD_TOKEN_TABLE | encrypted_rid(expected, 7, DEFAULT_METHOD_TOKEN_SEED) + }) + .collect::>(); + let (data, _) = build(&tokens); + let wrong_seed = DEFAULT_METHOD_TOKEN_SEED.wrapping_add(1); + + assert!(matches!( + restore_method_tokens(&data, wrong_seed), + Err(Error::Validation(_)) + )); + } +} diff --git a/senbei-cli/Cargo.toml b/senbei-cli/Cargo.toml index b6b9da6..8ee1b2a 100644 --- a/senbei-cli/Cargo.toml +++ b/senbei-cli/Cargo.toml @@ -17,4 +17,5 @@ senbei-io.workspace = true [dev-dependencies] senbei-io.workspace = true senbei-metadata.workspace = true +sha2.workspace = true tempfile.workspace = true diff --git a/senbei-cli/src/main.rs b/senbei-cli/src/main.rs index 2c226bd..9379e19 100644 --- a/senbei-cli/src/main.rs +++ b/senbei-cli/src/main.rs @@ -74,14 +74,7 @@ fn main() -> std::process::ExitCode { match result { Ok(summary) => { if quiet < 2 { - println!( - "{} unpacked · {} skipped · {} errors · {} suspect · {} metadata", - summary.unpacked, - summary.skipped, - summary.errors, - summary.suspect, - summary.metadata - ); + println!("{}", summary.line()); println!("done in {} ms", summary.duration_ms); } if summary.errors > 0 { 1 } else { 0 } @@ -104,6 +97,12 @@ fn print_help() { println!( "senbei [--out DIR] [-v|--verbose] [-q|--quiet]... [--scan-all] [--no-log] [--no-pause] [-V|--version] [-h|--help]" ); + println!( + " input a Crackproof PE (.exe/.dll), an il2cpp global-metadata.dat,\n\ + \x20 a protected Android AArch64 library (.so), an Android app\n\ + \x20 package (.apk/.apks/.xapk), or a folder containing any of\n\ + \x20 these" + ); println!( " --scan-all probe every file in a folder, including ones the scan\n\ \x20 pre-filter skips (under 4128 bytes, extensionless,\n\ diff --git a/senbei-cli/tests/android_samples.rs b/senbei-cli/tests/android_samples.rs new file mode 100644 index 0000000..2d223e0 --- /dev/null +++ b/senbei-cli/tests/android_samples.rs @@ -0,0 +1,227 @@ +//! Corpus test over the user-managed Android samples. +//! +//! Each immediate subdirectory of `samples/android/` that contains a `lib/` +//! tree is one app-package sample (an extracted APK layout). For every +//! protected AArch64 `.so` found by content probe, the test runs the real +//! restore pipeline and checks the result: +//! +//! - `.golden.so.sha256` next to the input pins the restored bytes +//! (byte-identity through the digest; absent sidecar -> WARNING). +//! - `.restore-fails` (empty marker) documents an input whose restore +//! is known to fail; the test then *requires* failure, so a future fix +//! surfaces as a test failure too. Without the marker a failed restore is +//! a test failure. +//! - A restored library carrying an unwrappable embedded metadata blob must +//! produce one, pinned by `.golden.metadata.sha256`. +//! +//! The folder-mode driver is then run over each app dir to exercise the +//! scan/restore/write path end to end; its error count must equal the number +//! of marked known-failures. +//! +//! The corpus is git-ignored and absent on CI (no binaries in the repo); +//! `SENBEI_REQUIRE_SAMPLES=1` turns an absent corpus into a failure, and +//! `SENBEI_ANDROID_SAMPLES` overrides the corpus location. + +mod common; + +use std::path::{Path, PathBuf}; + +use senbei_io::{android, job}; + +fn corpus_dir() -> PathBuf { + if let Some(dir) = std::env::var_os("SENBEI_ANDROID_SAMPLES") { + return PathBuf::from(dir); + } + common::samples_dir().join("android") +} + +/// Immediate subdirectories of `root` that hold an app tree (a `lib/` +/// folder) — research notes, dumps, and other non-app material in the corpus +/// never match. +fn app_dirs(root: &Path) -> Vec { + let mut dirs: Vec = std::fs::read_dir(root) + .unwrap_or_else(|e| panic!("read {}: {e}", root.display())) + .filter_map(|entry| entry.ok().map(|entry| entry.path())) + .filter(|path| path.is_dir() && path.join("lib").is_dir()) + .collect(); + dirs.sort(); + dirs +} + +/// Every regular `.so` below `dir`, skipping previous output trees. +fn collect_so_files(dir: &Path, out: &mut Vec) { + let mut entries: Vec<_> = std::fs::read_dir(dir) + .unwrap_or_else(|e| panic!("read {}: {e}", dir.display())) + .filter_map(|entry| entry.ok().map(|entry| entry.path())) + .collect(); + entries.sort(); + for path in entries { + if path.is_dir() { + if path + .file_name() + .is_some_and(|name| !name.eq_ignore_ascii_case("unpack")) + { + collect_so_files(&path, out); + } + } else if path + .extension() + .and_then(|ext| ext.to_str()) + .is_some_and(|ext| ext.eq_ignore_ascii_case("so")) + { + out.push(path); + } + } +} + +fn sha256_hex(data: &[u8]) -> String { + use sha2::Digest; + let mut digest = sha2::Sha256::new(); + digest.update(data); + format!("{:x}", digest.finalize()) +} + +/// `.golden.so.sha256` next to `input`. +fn golden_sidecar(input: &Path, artifact: &str) -> PathBuf { + let file = input.file_name().unwrap().to_string_lossy(); + let stem = file.strip_suffix(".so").unwrap_or(&file); + input.with_file_name(format!("{stem}.golden.{artifact}.sha256")) +} + +fn read_sidecar(path: &Path) -> Option { + std::fs::read_to_string(path) + .ok() + .map(|text| text.trim().to_ascii_lowercase()) +} + +#[test] +fn android_samples_restore_against_goldens() { + let root = corpus_dir(); + // Same opt-in gate as the PE corpus test: an absent corpus is a no-op + // pass unless CI explicitly requires it. + let require = std::env::var_os("SENBEI_REQUIRE_SAMPLES").is_some(); + if !root.is_dir() { + assert!( + !require, + "android samples: {} does not exist — corpus required (CI)", + root.display() + ); + eprintln!( + "android samples: {} does not exist, nothing to test", + root.display() + ); + return; + } + let apps = app_dirs(&root); + if apps.is_empty() { + assert!( + !require, + "android samples: no app trees under {} — corpus required (CI)", + root.display() + ); + eprintln!("android samples: no app trees under {}", root.display()); + return; + } + + let mut passed = 0usize; + let mut warnings: Vec = Vec::new(); + let mut failures: Vec = Vec::new(); + + for app in &apps { + let mut so_files = Vec::new(); + collect_so_files(app, &mut so_files); + let protected: Vec = so_files + .into_iter() + .filter(|path| android::is_protected_so_file(path)) + .collect(); + let mut known_failures = 0usize; + + for input in &protected { + let name = input.file_name().unwrap().to_string_lossy().to_string(); + let known_fails = input.with_file_name(format!( + "{}.restore-fails", + name.strip_suffix(".so").unwrap_or(&name) + )); + let temp = tempfile::tempdir().expect("tempdir"); + let dest = temp.path().join("restored.so"); + match android::restore_so_file(input, &dest, false) { + Ok(embedded) => { + if known_fails.is_file() { + failures.push(format!( + "{name}: restore succeeded but a restore-fails marker exists \ + (delete the marker — the gap is fixed)" + )); + continue; + } + let bytes = std::fs::read(&dest).expect("read restored output"); + match read_sidecar(&golden_sidecar(input, "so")) { + Some(expected) if expected == sha256_hex(&bytes) => passed += 1, + Some(expected) => failures.push(format!( + "{name}: restored bytes differ from golden\n expected sha256 {expected}\n actual sha256 {}", + sha256_hex(&bytes) + )), + None => warnings.push(format!( + "{name}: no golden sidecar — restored sha256 {}", + sha256_hex(&bytes) + )), + } + if let Some(blob) = embedded { + match read_sidecar(&golden_sidecar(input, "metadata")) { + Some(expected) if expected == sha256_hex(&blob) => {} + Some(expected) => failures.push(format!( + "{name}: embedded metadata differs from golden\n expected sha256 {expected}\n actual sha256 {}", + sha256_hex(&blob) + )), + None => warnings.push(format!( + "{name}: no embedded-metadata sidecar — sha256 {}", + sha256_hex(&blob) + )), + } + } + } + Err(error) => { + if known_fails.is_file() { + known_failures += 1; + } else { + failures.push(format!("{name}: restore failed: {error:#}")); + } + } + } + } + + // Folder-mode smoke run: the scan must route every protected library, + // and only the marked known-failures may error. + let out_temp = tempfile::tempdir().expect("tempdir"); + match job::run_folder_opts(app, Some(out_temp.path()), 2, false, true, false) { + Ok(summary) => { + if summary.errors != known_failures { + failures.push(format!( + "{}: folder run errors {} != known-failure markers {known_failures}", + app.display(), + summary.errors + )); + } + if summary.unpacked < protected.len().saturating_sub(known_failures) { + failures.push(format!( + "{}: folder run restored {} libraries, per-file pass found {} ({} known-failing)", + app.display(), + summary.unpacked, + protected.len(), + known_failures + )); + } + } + Err(error) => failures.push(format!("{}: folder run failed: {error:#}", app.display())), + } + } + + for warning in &warnings { + eprintln!("WARNING: {warning}"); + } + eprintln!( + "android samples: {} app tree(s) — {passed} pass, {} warning(s), {} failure(s)", + apps.len(), + warnings.len(), + failures.len() + ); + assert!(failures.is_empty(), "{}", failures.join("\n")); +} diff --git a/senbei-io/Cargo.toml b/senbei-io/Cargo.toml index 6cc3e0f..d092b0b 100644 --- a/senbei-io/Cargo.toml +++ b/senbei-io/Cargo.toml @@ -7,11 +7,18 @@ description = "Filesystem, scanning, logging, and CLI orchestration for Senbei" [dependencies] anyhow.workspace = true +flate2.workspace = true indicatif.workspace = true owo-colors.workspace = true +senbei-android-elf.workspace = true +senbei-android-engine.workspace = true +senbei-android-metadata.workspace = true senbei-metadata.workspace = true senbei-pe.workspace = true +sha2.workspace = true +tempfile.workspace = true walkdir.workspace = true +zip.workspace = true [target.'cfg(windows)'.dependencies] windows.workspace = true diff --git a/senbei-io/src/android.rs b/senbei-io/src/android.rs new file mode 100644 index 0000000..4dfb249 --- /dev/null +++ b/senbei-io/src/android.rs @@ -0,0 +1,432 @@ +//! Android target orchestration: protected AArch64 shared libraries (`.so`), +//! app packages (`.apk` / `.apks` / `.xapk`), and the Android variant of the +//! il2cpp method-token obfuscation. +//! +//! The protection scheme hollows out an ELF64/AArch64 shared object and moves +//! the original bytes into an encrypted payload appended as a `SHT_LOUSER` +//! section; restoration extracts the stage-2 module set +//! ([`senbei_android_engine`]) and rebuilds the static image +//! ([`senbei_android_elf`]). Some il2cpp builds additionally embed their +//! metadata blob — XOR-wrapped, with no standalone `global-metadata.dat` in +//! the assets — inside the library's data section; after a successful restore +//! the blob is located by content and unwrapped +//! ([`senbei_android_metadata::extract_embedded_metadata`]). +//! +//! All functions in this module are native filesystem orchestration; the web +//! app (wasm) never touches them. + +use std::collections::HashSet; +use std::io::Read; +use std::path::{Path, PathBuf}; + +use anyhow::{Context, Result, bail}; +use flate2::read::DeflateDecoder; +use senbei_android_elf::{RestoreOptions, restore_libil2cpp}; +use senbei_android_engine::{ExtractOptions, extract_stage2, is_protected_libil2cpp}; +use sha2::{Digest, Sha256}; +use zip::ZipArchive; + +/// File name of an il2cpp metadata blob (a platform-standard technology name). +pub const METADATA_FILE_NAME: &str = "global-metadata.dat"; + +/// Package extensions recognised as Android app packages. Packages are +/// *containers*: membership is decided by extension plus the zip magic, while +/// every file pulled out of one is still content-probed like a loose file. +const PACKAGE_EXTENSIONS: [&str; 3] = ["apk", "apks", "xapk"]; + +/// Whether `prefix` (the first bytes of a file) is an ELF64/AArch64 image. +/// Only those can be protected Android libraries, so the folder scan uses this +/// cheap check to decide when the full-file protection probe is worth its +/// read. +pub fn is_elf64_aarch64(prefix: &[u8]) -> bool { + prefix.len() >= 20 + && prefix[0..4] == [0x7f, b'E', b'L', b'F'] + && prefix[4] == 2 // ELFCLASS64 + && prefix[5] == 1 // ELFDATA2LSB + && u16::from_le_bytes([prefix[18], prefix[19]]) == 0xB7 // EM_AARCH64 +} + +/// Whether `path` is an Android app package: a recognised package extension +/// and the local-file-header zip magic in `prefix`. +pub fn is_app_package(path: &Path, prefix: &[u8]) -> bool { + let is_package_ext = path + .extension() + .and_then(|value| value.to_str()) + .is_some_and(|value| { + PACKAGE_EXTENSIONS + .iter() + .any(|ext| value.eq_ignore_ascii_case(ext)) + }); + is_package_ext && prefix.starts_with(b"PK\x03\x04") +} + +/// Probe a file on disk: true when it is a protected AArch64 library. +/// Reads the whole file (the payload section is found through the +/// section-header table at the end); call only after [`is_elf64_aarch64`] +/// has matched a prefix. +pub fn is_protected_so_file(path: &Path) -> bool { + let Ok(bytes) = std::fs::read(path) else { + return false; + }; + is_elf64_aarch64(&bytes) && is_protected_libil2cpp(&bytes) +} + +/// Restore one protected `.so` to `dest`. +/// +/// The stage-2 module set is extracted into a temporary workspace (it is an +/// implementation detail of the two-phase restore, not user-facing output). +/// Returns the unwrapped embedded metadata blob when the restored image +/// carries one (see the module docs); the caller decides where to write it. +pub fn restore_so_file(input: &Path, dest: &Path, verbose: bool) -> Result>> { + let temporary = tempfile::tempdir().context("create stage-2 workspace")?; + let stage2_dir = temporary.path().join("stage2"); + extract_stage2(&ExtractOptions::with_defaults( + input.to_path_buf(), + stage2_dir.clone(), + )) + .context("extract stage-1/stage-2 payload")?; + restore_libil2cpp(&RestoreOptions { + input: input.to_path_buf(), + output: dest.to_path_buf(), + index: stage2_dir.join("index.json"), + dump_auxiliary: None, + outer_only: false, + preserve_entrypoint: false, + verbose, + }) + .context("restore protected library")?; + let restored = + std::fs::read(dest).with_context(|| format!("read restored `{}`", dest.display()))?; + Ok(senbei_android_metadata::extract_embedded_metadata( + &restored, + )) +} + +/// Content identity for cross-source deduplication: the same library may +/// appear loose in a tree, in its `.apk`, and again in an `.apks`/`.xapk` +/// bundle — restore it once, at the highest-priority source's destination. +pub fn content_identity(data: &[u8]) -> String { + let mut digest = Sha256::new(); + digest.update(data); + format!("{:x}", digest.finalize()) +} + +/// Restore an il2cpp metadata blob (Android seeded permutation first, then the +/// structural remap used by the Windows builds). +/// +/// The Android variant obfuscates MethodDef RIDs with a keyed five-round +/// permutation; the correct seed is recovered by intersecting per-image key +/// residues, and the restore *validates* every restored RID against its +/// canonical per-module index — so an unusable seed fails loudly and the +/// caller falls through to the structural remap, which targets the same +/// canonical form. Both paths are no-ops (`remapped == 0`) on an +/// already-clean blob. +pub fn restore_metadata_bytes(data: &[u8]) -> anyhow::Result<(Vec, senbei_metadata::Report)> { + if let Ok(discovery) = senbei_android_metadata::discover_method_token_seeds(data) + && discovery.version == 31 + && discovery.images.iter().any(|image| !image.clean) + { + let mut seeds = discovery.seed_candidates.clone(); + if seeds.is_empty() { + seeds.push(senbei_android_metadata::DEFAULT_METHOD_TOKEN_SEED); + } + // Trial-and-validate: a wrong seed fails the restore's full-coverage + // RID check, so ambiguous candidates cost one extra pass each and a + // build with an unseeded permutation falls through to the structural + // remap rather than producing a silently wrong file. + for seed in seeds { + if let Ok((out, report)) = senbei_android_metadata::restore_method_tokens(data, seed) { + return Ok(( + out, + senbei_metadata::Report { + version: report.version, + methods: report.methods, + remapped: report.changed_tokens, + modules: report.images_with_methods, + }, + )); + } + } + } + let (out, report) = senbei_metadata::deobfuscate(data).map_err(anyhow::Error::new)?; + Ok((out, report)) +} + +/// What happened to one archive entry (or one loose Android target). +#[derive(Debug)] +pub struct EntryOutcome { + /// Human-readable source label, e.g. `base.apk::lib/arm64-v8a/libil2cpp.so`. + pub label: String, + /// Where the restored bytes were written (meaningless unless `status` is + /// `Restored`). + pub dest: PathBuf, + pub kind: EntryKind, + pub status: EntryStatus, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum EntryKind { + /// A protected shared library, restored. + So, + /// An il2cpp metadata blob, de-obfuscated (`remapped` tokens changed). + Metadata { remapped: usize }, + /// A metadata blob unwrapped from a restored library's data section. + EmbeddedMetadata, +} + +#[derive(Debug)] +pub enum EntryStatus { + Restored, + /// Byte-identical content was already restored from a higher-priority + /// source; no output written. + Duplicate, + /// Content-probed but not a target (unprotected library). + NotTarget, + /// A metadata blob whose tokens were already canonical; no copy written. + Unchanged, + /// Recognised as a target but the restore failed. + Failed(anyhow::Error), +} + +/// Restore every protected library and metadata blob inside one app package. +/// +/// `rel` is the package's path relative to the scanned root (or its bare file +/// name in single-file mode); outputs mirror the package's internal layout +/// under `out_root/rel/`, with [`crate::job::out_name`] renaming. `seen` +/// carries content identities already restored from higher-priority sources +/// (loose files first, then `.apk`, then bundles) across the whole run. +pub fn restore_package( + package: &Path, + rel: &Path, + out_root: &Path, + seen: &mut HashSet, + verbose: bool, +) -> Result> { + let bundle = package + .extension() + .and_then(|value| value.to_str()) + .is_some_and(|value| { + value.eq_ignore_ascii_case("apks") || value.eq_ignore_ascii_case("xapk") + }); + let mut archive = open_package(package)?; + let temporary = tempfile::tempdir().context("create package workspace")?; + let mut outcomes = Vec::new(); + + let mut direct = Vec::new(); + let mut nested = Vec::new(); + for index in 0..archive.len() { + let (name, is_dir) = { + let entry = archive.by_index(index)?; + (entry.enclosed_name().map(PathBuf::from), entry.is_dir()) + }; + if is_dir { + continue; + } + let Some(name) = name else { + bail!("unsafe entry path in package `{}`", package.display()); + }; + if bundle { + if name + .extension() + .and_then(|value| value.to_str()) + .is_some_and(|value| value.eq_ignore_ascii_case("apk")) + { + nested.push((index, name)); + } + } else { + direct.push((index, name)); + } + } + drop(archive); + + for (index, name) in direct { + let label = format!("{}::{}", rel.display(), name.display()); + let dest = out_root.join(rel).join(crate::job::out_name(&name)); + let mut entry_outcomes = + restore_package_entry(package, index, &label, &dest, &temporary, seen, verbose) + .with_context(|| format!("extract `{label}`"))?; + outcomes.append(&mut entry_outcomes); + } + for (index, name) in nested { + let nested_label = rel.join(&name); + let nested_path = extract_entry(package, index, &temporary, &nested_label) + .with_context(|| format!("extract `{}`", nested_label.display()))?; + let mut nested_archive = open_package(&nested_path)?; + let mut entries = Vec::new(); + for nested_index in 0..nested_archive.len() { + let (entry_name, is_dir) = { + let entry = nested_archive.by_index(nested_index)?; + (entry.enclosed_name().map(PathBuf::from), entry.is_dir()) + }; + if !is_dir { + let Some(entry_name) = entry_name else { + bail!("unsafe entry path in `{}`", nested_label.display()); + }; + entries.push((nested_index, entry_name)); + } + } + drop(nested_archive); + // Keep the nested package's stem in the output layout so two splits + // carrying same-named entries cannot collide. + let base = rel.join(name.with_extension("")); + for (nested_index, entry_name) in entries { + let label = format!("{}::{}", nested_label.display(), entry_name.display()); + let dest = out_root.join(&base).join(crate::job::out_name(&entry_name)); + let mut entry_outcomes = restore_package_entry( + &nested_path, + nested_index, + &label, + &dest, + &temporary, + seen, + verbose, + ) + .with_context(|| format!("extract `{label}`"))?; + outcomes.append(&mut entry_outcomes); + } + } + Ok(outcomes) +} + +/// Probe one extracted package entry and restore it when it is a target. +/// Returns one outcome per produced/consumed artifact: the entry itself, plus +/// an `EmbeddedMetadata` outcome when the restored library carried a blob. +fn restore_package_entry( + package: &Path, + index: usize, + label: &str, + dest: &Path, + temporary: &tempfile::TempDir, + seen: &mut HashSet, + verbose: bool, +) -> Result> { + let entry_path = extract_entry(package, index, temporary, Path::new(label))?; + let data = std::fs::read(&entry_path).with_context(|| format!("read extracted `{label}`"))?; + + let is_so = is_elf64_aarch64(&data) && is_protected_libil2cpp(&data); + let is_meta = !is_so && senbei_metadata::is_metadata(&data); + let outcome = |kind, status| EntryOutcome { + label: label.to_owned(), + dest: dest.to_path_buf(), + kind, + status, + }; + if !is_so && !is_meta { + return Ok(vec![outcome(EntryKind::So, EntryStatus::NotTarget)]); + } + if !seen.insert(content_identity(&data)) { + let kind = if is_so { + EntryKind::So + } else { + EntryKind::Metadata { remapped: 0 } + }; + return Ok(vec![outcome(kind, EntryStatus::Duplicate)]); + } + + if is_so { + return Ok(match restore_so_file(&entry_path, dest, verbose) { + Ok(embedded) => { + let mut outcomes = vec![outcome(EntryKind::So, EntryStatus::Restored)]; + if let Some(blob) = embedded { + let meta_dest = embedded_metadata_dest(dest); + let status = match write_metadata_blob(&meta_dest, &blob) { + Ok(()) => EntryStatus::Restored, + Err(error) => EntryStatus::Failed(error), + }; + outcomes.push(EntryOutcome { + label: format!("{label} (embedded metadata)"), + dest: meta_dest, + kind: EntryKind::EmbeddedMetadata, + status, + }); + } + outcomes + } + Err(error) => vec![outcome(EntryKind::So, EntryStatus::Failed(error))], + }); + } + + // Metadata entry: write only when the restore actually changed tokens — + // a clean blob needs no copy (same contract as loose metadata files). + let kind_and_status = match restore_metadata_bytes(&data) { + Ok((out, report)) if report.remapped > 0 => { + let kind = EntryKind::Metadata { + remapped: report.remapped, + }; + match write_metadata_blob(dest, &out) { + Ok(()) => (kind, EntryStatus::Restored), + Err(error) => (kind, EntryStatus::Failed(error)), + } + } + Ok(_) => (EntryKind::Metadata { remapped: 0 }, EntryStatus::Unchanged), + Err(error) => ( + EntryKind::Metadata { remapped: 0 }, + EntryStatus::Failed(error), + ), + }; + Ok(vec![outcome(kind_and_status.0, kind_and_status.1)]) +} + +/// Output path for a metadata blob unwrapped from a restored library: next to +/// the library, under the standard file name (with the usual `.unpack` infix). +pub fn embedded_metadata_dest(restored_so: &Path) -> PathBuf { + let dir = restored_so.parent().unwrap_or_else(|| Path::new(".")); + dir.join(crate::job::out_name(Path::new(METADATA_FILE_NAME))) +} + +/// Write a metadata blob, creating the parent directory. The restore writes +/// its own output atomically; metadata blobs go through the job layer's +/// atomic write to share the mid-write failure semantics. +fn write_metadata_blob(dest: &Path, data: &[u8]) -> Result<()> { + if let Some(parent) = dest.parent() { + std::fs::create_dir_all(parent) + .with_context(|| format!("create `{}`", parent.display()))?; + } + crate::job::write_atomic(dest, data) + .map_err(anyhow::Error::from) + .context("write metadata output") +} + +fn open_package(path: &Path) -> Result> { + let file = std::fs::File::open(path).with_context(|| format!("open `{}`", path.display()))?; + ZipArchive::new(file).with_context(|| format!("read package `{}`", path.display())) +} + +/// Extract one package entry to the temporary workspace, streaming stored +/// entries and inflating deflated ones by hand so compression-method +/// surprises fail loudly instead of producing a truncated file. +fn extract_entry( + package: &Path, + index: usize, + temporary: &tempfile::TempDir, + label: &Path, +) -> Result { + let mut archive = open_package(package)?; + let mut entry = archive.by_index_raw(index)?; + let key = format!("{}-{index:08x}", label.display()); + // `:` appears in `package::entry` labels and is invalid in Windows file + // names; sanitize every path-ish separator. + let destination = temporary.path().join(key.replace(['\\', '/', ':'], "_")); + let compressed_size = usize::try_from(entry.compressed_size()) + .map_err(|_| anyhow::anyhow!("entry compressed size exceeds usize"))?; + let output_size = + usize::try_from(entry.size()).map_err(|_| anyhow::anyhow!("entry size exceeds usize"))?; + let mut compressed = vec![0_u8; compressed_size]; + entry.read_exact(&mut compressed)?; + let mut output = Vec::with_capacity(output_size); + match entry.compression() { + zip::CompressionMethod::Stored => output.extend_from_slice(&compressed), + zip::CompressionMethod::Deflated => { + DeflateDecoder::new(compressed.as_slice()).read_to_end(&mut output)?; + } + method => bail!("unsupported compression method {method:?} in entry `{key}`"), + } + if output.len() != output_size { + bail!( + "entry `{key}` decompressed to 0x{:x}, expected 0x{output_size:x}", + output.len() + ); + } + std::fs::write(&destination, &output)?; + Ok(destination) +} diff --git a/senbei-io/src/job.rs b/senbei-io/src/job.rs index 44b08ac..7405e94 100644 --- a/senbei-io/src/job.rs +++ b/senbei-io/src/job.rs @@ -323,6 +323,7 @@ fn splice_companion(stub: &[u8], comp: &[u8]) -> Option> { } /// Summary of a folder-mode run. +#[derive(Default)] pub struct Summary { pub unpacked: usize, pub skipped: usize, @@ -331,12 +332,29 @@ pub struct Summary { /// — likely to crash at runtime (e.g. 0xC0000005). Counted in addition to /// `unpacked` (a suspect file is still written). pub suspect: usize, - /// il2cpp `global-metadata.dat` files de-obfuscated (method tokens remapped). + /// il2cpp `global-metadata.dat` files de-obfuscated (method tokens remapped), + /// including blobs unwrapped from restored Android libraries. pub metadata: usize, + /// Android app packages (`.apk`/`.apks`/`.xapk`) opened and searched. + pub packages: usize, /// Wall-clock duration of the folder run in milliseconds. pub duration_ms: u128, } +impl Summary { + /// The summary line shared by CLI output and the log file. + pub fn line(&self) -> String { + let mut line = format!( + "{} unpacked · {} skipped · {} errors · {} suspect · {} metadata", + self.unpacked, self.skipped, self.errors, self.suspect, self.metadata + ); + if self.packages > 0 { + line.push_str(&format!(" · {} packages", self.packages)); + } + line + } +} + /// Default output root for a folder unpack: `/unpack`. pub fn default_out_root_for_folder(root: &Path) -> PathBuf { root.join("unpack") @@ -416,12 +434,15 @@ pub fn run_folder_opts( log.step(&format!("out {}", out_root.display())); Some(log) }; - // Single merged directory walk: returns Crackproof unpack candidates and - // il2cpp metadata blobs from one traversal (see + // Single merged directory walk: returns Crackproof unpack candidates, il2cpp + // metadata blobs, and Android targets from one traversal (see // [`crate::scan::find_targets_opts`]). Files the free directory metadata // already rules out are never opened — on asset-heavy trees the per-file // open+read latency, not the traversal, is the whole cost. - let (candidates, metas, scan_stats) = crate::scan::find_targets_opts(root, scan_all); + let scan = crate::scan::find_targets_opts(root, scan_all); + let candidates = scan.crackproof.as_slice(); + let metas = scan.metadata.as_slice(); + let scan_stats = &scan.stats; // Files the scan could not classify are potential missed targets, not // clean skips: an unreadable directory or a locked il2cpp game assembly must // fail the run (exit 1) rather than report "0 errors" over a partial scan. @@ -453,21 +474,22 @@ pub fn run_folder_opts( // Verbose mode prints multi-line `[N/9]` step output per file straight to // stdout; an active progress bar would be clobbered by it, so hide the bar // (its per-file ok/err lines still print) when verbose is on. - let bar = crate::ui::progress(candidates.len() as u64, quiet >= 1 || verbose); + let android_targets = scan.android_so.len() + scan.android_packages.len(); + let bar = crate::ui::progress( + (candidates.len() + android_targets) as u64, + quiet >= 1 || verbose, + ); let mut s = Summary { - unpacked: 0, skipped: scan_stats.skipped, errors: scan_failed, - suspect: 0, - metadata: 0, - duration_ms: 0, + ..Summary::default() }; // Silence the default panic hook's stderr spew during per-file processing. let default_hook = std::panic::take_hook(); std::panic::set_hook(Box::new(|_| {})); // suppress "thread panicked" messages - for input in &candidates { + for input in candidates { let rel = rel_in_tree(root, input); let dest = out_root.join(out_name(&rel)); @@ -515,14 +537,146 @@ pub fn run_folder_opts( bar.inc(1); } + // Android pass: protected AArch64 libraries and app packages. Loose `.so` + // files restore first so the cross-source dedup keeps them over a copy + // inside a package (loose beats `.apk` beats `.apks`/`.xapk` bundle). + let mut android_seen = std::collections::HashSet::new(); + // Hashing a protected library costs a full read, so only pay it when a + // duplicate source can actually exist in this run. + let android_dedup = scan.android_so.len() > 1 || !scan.android_packages.is_empty(); + for input in &scan.android_so { + let rel = rel_in_tree(root, input); + let dest = out_root.join(out_name(&rel)); + // Unreadable here is fine: the restore reports the same error. + if android_dedup + && let Ok(bytes) = std::fs::read(input) + && !android_seen.insert(crate::android::content_identity(&bytes)) + { + s.skipped += 1; + if let Some(log) = &log { + log.step(&format!("SKIP {rel:?}: duplicate of an earlier target")); + } + bar.inc(1); + continue; + } + let input_owned = input.clone(); + let dest_owned = dest.clone(); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + crate::android::restore_so_file(&input_owned, &dest_owned, verbose_steps) + })); + match result { + Ok(Ok(embedded)) => { + s.unpacked += 1; + crate::ui::ok_label( + &bar, + suppress_file_lines, + &rel.display().to_string(), + "So", + &dest, + ); + if let Some(log) = &log { + log.step(&format!("OK {rel:?} -> {dest:?} (Android SO)")); + } + match write_embedded_metadata(embedded, &dest) { + Ok(Some(meta_dest)) => { + s.metadata += 1; + crate::ui::ok_label( + &bar, + suppress_file_lines, + &format!("{} (embedded metadata)", rel.display()), + "metadata", + &meta_dest, + ); + if let Some(log) = &log { + log.step(&format!("META {rel:?} (embedded) -> {meta_dest:?}")); + } + } + Ok(None) => {} + Err(e) => { + s.errors += 1; + crate::ui::err(&bar, suppress_file_lines, &rel, &e); + if let Some(log) = &log { + log.step(&format!("ERR {rel:?}: embedded metadata: {e:#}")); + } + } + } + } + Ok(Err(e)) => { + s.errors += 1; + crate::ui::err(&bar, suppress_file_lines, &rel, &e); + if let Some(log) = &log { + log.step(&format!("ERR {rel:?}: {e:#}")); + } + } + Err(panic) => { + s.errors += 1; + let e = anyhow::anyhow!("unexpected panic: {}", panic_payload(&panic)); + crate::ui::err(&bar, suppress_file_lines, &rel, &e); + if let Some(log) = &log { + log.step(&format!( + "ERR {rel:?}: panic during restore: {}", + panic_payload(&panic) + )); + } + } + } + bar.inc(1); + } + for package in &scan.android_packages { + let rel = rel_in_tree(root, package); + s.packages += 1; + let package_owned = package.clone(); + let rel_owned = rel.clone().into_owned(); + let out_root_owned = out_root.clone(); + let mut seen_taken = std::mem::take(&mut android_seen); + let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + let outcomes = crate::android::restore_package( + &package_owned, + &rel_owned, + &out_root_owned, + &mut seen_taken, + verbose_steps, + ); + (outcomes, seen_taken) + })); + match result { + Ok((Ok(outcomes), seen_back)) => { + android_seen = seen_back; + apply_package_outcomes(outcomes, &mut s, &bar, suppress_file_lines, &log); + } + Ok((Err(e), seen_back)) => { + android_seen = seen_back; + s.errors += 1; + crate::ui::err(&bar, suppress_file_lines, &rel, &e); + if let Some(log) = &log { + log.step(&format!("ERR {rel:?}: {e:#}")); + } + } + Err(panic) => { + // The dedup set may be in an unknown state after a panic; a + // re-scan costs a duplicate restore at worst, never corruption. + let e = anyhow::anyhow!("unexpected panic: {}", panic_payload(&panic)); + s.errors += 1; + crate::ui::err(&bar, suppress_file_lines, &rel, &e); + if let Some(log) = &log { + log.step(&format!( + "ERR {rel:?}: panic during package restore: {}", + panic_payload(&panic) + )); + } + } + } + bar.inc(1); + } + // il2cpp metadata pass. Crackproof's `-GMD` option obfuscates the method // tokens in `global-metadata.dat`; de-obfuscate any we find so the unpacked // il2cpp game assembly resolves methods instead of indexing its per-module // tables out of bounds (see [`senbei_metadata`]). This is additive to the // Crackproof module unpack above — the metadata blob is not itself a // Crackproof file. - for meta in metas { - let rel = rel_in_tree(root, &meta); + for meta in metas.iter() { + let rel = rel_in_tree(root, meta); let dest = out_root.join(out_name(&rel)); let meta_owned = meta.clone(); let dest_owned = dest.clone(); @@ -604,10 +758,7 @@ pub fn run_folder_opts( s.duration_ms = t0.elapsed().as_millis(); if let Some(log) = &log { log.step(&format!("done in {} ms", s.duration_ms)); - log.step(&format!( - "summary: {} unpacked · {} skipped · {} errors · {} suspect · {} metadata", - s.unpacked, s.skipped, s.errors, s.suspect, s.metadata - )); + log.step(&format!("summary: {}", s.line())); } Ok(s) } @@ -646,23 +797,30 @@ pub fn run_file_v( let name = out_name(Path::new(input.file_name().unwrap_or_default())); let dest = out_root.join(name); - let mut s = Summary { - unpacked: 0, - skipped: 0, - errors: 0, - suspect: 0, - metadata: 0, - duration_ms: 0, - }; + let mut s = Summary::default(); - let is_meta = { + let prefix = { use std::io::Read; - let mut buf = [0u8; 4]; - std::fs::File::open(input) - .and_then(|mut f| f.read_exact(&mut buf)) - .map(|_| senbei_metadata::is_metadata(&buf)) - .unwrap_or(false) + let mut buf = vec![0u8; 8 * 1024]; + match std::fs::File::open(input).and_then(|mut f| f.read(&mut buf).map(|n| (buf, n))) { + Ok((buf, n)) => { + let mut b = buf; + b.truncate(n); + b + } + Err(_) => Vec::new(), + } }; + let is_meta = senbei_metadata::is_metadata(&prefix); + // Android single-file targets are routed by content: a protected AArch64 + // library probe needs the whole file (its payload section is found through + // the section-header table at the end), while a package is a container + // handled entry-by-entry. Anything else falls through to the PE pipeline. + let is_android_so = crate::android::is_elf64_aarch64(&prefix) + && std::fs::read(input) + .map(|bytes| senbei_android_engine::is_protected_libil2cpp(&bytes)) + .unwrap_or(false); + let is_android_package = !is_android_so && crate::android::is_app_package(input, &prefix); if is_meta { match deobfuscate_metadata_to(input, &dest, verbose && quiet == 0) { @@ -706,6 +864,77 @@ pub fn run_file_v( } } } + } else if is_android_so { + match crate::android::restore_so_file(input, &dest, verbose && quiet == 0) { + Ok(embedded) => { + s.unpacked = 1; + if let Some(log) = &log { + log.step(&format!("OK {:?} -> {:?} (Android SO)", input, dest)); + } + if quiet == 0 { + println!("✓ So {} -> {}", input.display(), dest.display()); + } + match write_embedded_metadata(embedded, &dest) { + Ok(Some(meta_dest)) => { + s.metadata += 1; + if let Some(log) = &log { + log.step(&format!("META {:?} (embedded) -> {:?}", input, meta_dest)); + } + if quiet == 0 { + println!( + "✓ metadata {} (embedded) -> {}", + input.display(), + meta_dest.display() + ); + } + } + Ok(None) => {} + Err(e) => { + s.errors += 1; + if let Some(log) = &log { + log.step(&format!("ERR {:?}: embedded metadata: {e:#}", input)); + } + if quiet == 0 { + eprintln!("error: embedded metadata: {e:#}"); + } + } + } + } + Err(e) => { + s.errors = 1; + if let Some(log) = &log { + log.step(&format!("ERR {:?}: {e:#}", input)); + } + if quiet == 0 { + eprintln!("error: {e:#}"); + } + } + } + } else if is_android_package { + s.packages = 1; + let rel = PathBuf::from(input.file_name().unwrap_or_default()); + let mut seen = std::collections::HashSet::new(); + match crate::android::restore_package( + input, + &rel, + &out_root, + &mut seen, + verbose && quiet == 0, + ) { + Ok(outcomes) => { + let bar = crate::ui::progress(0, true); + apply_package_outcomes(outcomes, &mut s, &bar, quiet >= 1, &log); + } + Err(e) => { + s.errors = 1; + if let Some(log) = &log { + log.step(&format!("ERR {:?}: {e:#}", input)); + } + if quiet == 0 { + eprintln!("error: {e:#}"); + } + } + } } else { match unpack_one_v(input, &dest, verbose && quiet == 0) { Ok((kind, report)) => { @@ -748,10 +977,7 @@ pub fn run_file_v( s.duration_ms = t0.elapsed().as_millis(); if let Some(log) = &log { log.step(&format!("done in {} ms", s.duration_ms)); - log.step(&format!( - "summary: {} unpacked · {} skipped · {} errors · {} suspect · {} metadata", - s.unpacked, s.skipped, s.errors, s.suspect, s.metadata - )); + log.step(&format!("summary: {}", s.line())); } Ok(s) } @@ -820,7 +1046,7 @@ fn unsupported_version(e: &anyhow::Error) -> Option { /// failure (disk full, AV lock, quota) destroys a previously good unpack at /// the same path; the temp+rename keeps the old file until the new one is /// complete. Best-effort temp cleanup on failure. -fn write_atomic(dest: &Path, bytes: &[u8]) -> std::io::Result<()> { +pub(crate) fn write_atomic(dest: &Path, bytes: &[u8]) -> std::io::Result<()> { let mut tmp_name = dest.as_os_str().to_os_string(); tmp_name.push(".senbei-tmp"); let tmp = PathBuf::from(tmp_name); @@ -966,10 +1192,13 @@ pub fn deobfuscate_metadata_to( verbose: bool, ) -> anyhow::Result { let data = std::fs::read(input)?; - // Preserve the metadata::Error in the chain (rather than stringifying it) - // so the folder driver can apply its unsupported-version policy. - let (out, report) = senbei_metadata::deobfuscate(&data) - .map_err(|e| anyhow::Error::new(e).context(format!("{input:?}")))?; + // The Android seeded-permutation variant is tried first (it validates + // every restored RID); the structural remap is the fallback and the + // Windows path. The [`senbei_metadata::Error`] is preserved in the chain + // (rather than stringified) so the folder driver can apply its + // unsupported-version policy. + let (out, report) = crate::android::restore_metadata_bytes(&data) + .map_err(|e| e.context(format!("{input:?}")))?; if report.remapped > 0 { if let Some(parent) = dest.parent() { std::fs::create_dir_all(parent)?; @@ -982,6 +1211,77 @@ pub fn deobfuscate_metadata_to( Ok(report) } +/// Write an embedded metadata blob (unwrapped from a restored Android +/// library) next to the restored library. Returns the destination when a +/// blob was written. +fn write_embedded_metadata( + embedded: Option>, + so_dest: &Path, +) -> anyhow::Result> { + let Some(blob) = embedded else { + return Ok(None); + }; + let dest = crate::android::embedded_metadata_dest(so_dest); + if let Some(parent) = dest.parent() { + std::fs::create_dir_all(parent)?; + } + write_atomic(&dest, &blob)?; + Ok(Some(dest)) +} + +/// Fold one package's per-entry outcomes into the run summary, UI, and log. +fn apply_package_outcomes( + outcomes: Vec, + s: &mut Summary, + bar: &indicatif::ProgressBar, + quiet: bool, + log: &Option, +) { + use crate::android::{EntryKind, EntryStatus}; + for outcome in outcomes { + match outcome.status { + EntryStatus::Restored => { + match outcome.kind { + EntryKind::So => { + s.unpacked += 1; + crate::ui::ok_label(bar, quiet, &outcome.label, "So", &outcome.dest); + } + EntryKind::Metadata { remapped } => { + s.metadata += 1; + crate::ui::metadata( + bar, + quiet, + Path::new(&outcome.label), + remapped, + &outcome.dest, + ); + } + EntryKind::EmbeddedMetadata => { + s.metadata += 1; + crate::ui::ok_label(bar, quiet, &outcome.label, "metadata", &outcome.dest); + } + } + if let Some(log) = log { + log.step(&format!("OK {} -> {:?}", outcome.label, outcome.dest)); + } + } + EntryStatus::Duplicate | EntryStatus::NotTarget | EntryStatus::Unchanged => { + s.skipped += 1; + if let Some(log) = log { + log.step(&format!("SKIP {} ({:?})", outcome.label, outcome.kind)); + } + } + EntryStatus::Failed(e) => { + s.errors += 1; + crate::ui::err(bar, quiet, Path::new(&outcome.label), &e); + if let Some(log) = log { + log.step(&format!("ERR {}: {e:#}", outcome.label)); + } + } + } + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/senbei-io/src/lib.rs b/senbei-io/src/lib.rs index 5147a04..68dfd4d 100644 --- a/senbei-io/src/lib.rs +++ b/senbei-io/src/lib.rs @@ -1,5 +1,6 @@ //! Filesystem and command-line orchestration. +pub mod android; pub mod job; pub mod logfile; pub mod pause; diff --git a/senbei-io/src/scan.rs b/senbei-io/src/scan.rs index 8f3bbed..ebf06f8 100644 --- a/senbei-io/src/scan.rs +++ b/senbei-io/src/scan.rs @@ -115,6 +115,25 @@ enum Class { Crackproof, /// An il2cpp `global-metadata.dat` (de-obfuscation target). Metadata, + /// A protected AArch64 shared library (Android restore target). + AndroidSo, + /// An Android app package (`.apk`/`.apks`/`.xapk`) — a container whose + /// entries are content-probed individually during the Android pass. + AndroidPackage, +} + +/// Everything one [`find_targets_opts`] walk found, plus non-target tallies. +#[derive(Default)] +pub struct ScanResult { + /// Crackproof-protected PE files. + pub crackproof: Vec, + /// il2cpp `global-metadata.dat` blobs. + pub metadata: Vec, + /// Protected AArch64 shared libraries. + pub android_so: Vec, + /// Android app packages (containers restored entry-by-entry). + pub android_packages: Vec, + pub stats: ScanStats, } /// Walk `root` recursively (skipping any directory literally named `"unpack"`) @@ -146,7 +165,7 @@ enum Class { /// thread count: each worker owns a disjoint contiguous slice of the path list /// and writes the matching disjoint slice of the class list, so results are /// deterministic. -pub fn find_targets(root: &Path) -> (Vec, Vec, ScanStats) { +pub fn find_targets(root: &Path) -> ScanResult { find_targets_opts(root, scan_all_env()) } @@ -169,7 +188,7 @@ pub struct ScanStats { /// [`find_targets`], but with the pre-filter explicitly controlled. When /// `scan_all` is true every regular file is probed, restoring the exhaustive /// (and on asset-heavy trees, far slower) behavior. -pub fn find_targets_opts(root: &Path, scan_all: bool) -> (Vec, Vec, ScanStats) { +pub fn find_targets_opts(root: &Path, scan_all: bool) -> ScanResult { // Phase 1: serial traversal collecting regular-file paths only. No file is // opened here; `readdir` is fast relative to the content probe that follows, // and `entry.metadata()` is served from the directory entry on Windows, so @@ -246,19 +265,23 @@ pub fn find_targets_opts(root: &Path, scan_all: bool) -> (Vec, Vec candidates.push(p), - Some(Class::Metadata) => metadata.push(p), - Some(Class::None) => stats.skipped += 1, + Some(Class::Crackproof) => result.crackproof.push(p), + Some(Class::Metadata) => result.metadata.push(p), + Some(Class::AndroidSo) => result.android_so.push(p), + Some(Class::AndroidPackage) => result.android_packages.push(p), + Some(Class::None) => result.stats.skipped += 1, // Unreadable / panicking probe: NOT skipped — the scan could not // classify it, so it may be a target we failed to unpack. - None => stats.probe_errors += 1, + None => result.stats.probe_errors += 1, } } - (candidates, metadata, stats) + result } /// True if a walked directory entry is a reparse point (junction or symlink). @@ -292,10 +315,11 @@ pub fn scan_all_env() -> bool { } /// Classify one file by content. Reads a short prefix once and tests the -/// Crackproof detector first, then the il2cpp metadata magic. Returns `None` -/// when the file could not be classified at all — an I/O error opening it -/// (locked, permissions) or a panic inside a detector — so the caller counts -/// it as a probe error rather than a clean "not a target" skip. +/// Crackproof detector first, then the il2cpp metadata magic, then the +/// Android probes. Returns `None` when the file could not be classified at +/// all — an I/O error opening it (locked, permissions) or a panic inside a +/// detector — so the caller counts it as a probe error rather than a clean +/// "not a target" skip. /// /// The detector is wrapped in `catch_unwind` because a panic in a scan worker /// thread would otherwise abort the whole folder run (a scoped-thread panic @@ -304,16 +328,32 @@ pub fn scan_all_env() -> bool { /// /// A Crackproof PE never matches the metadata magic (it is a PE, not a /// metadata blob) and vice versa, so the order is immaterial. +/// +/// The Android library probe needs more than the prefix: the protection +/// payload lives in a section found via the section-header table at the *end* +/// of the file, so an ELF64/AArch64 prefix triggers a full-file read. Only +/// aarch64 images pay for it — a handful of `.so` files per app tree, against +/// tens of thousands of assets the free name/size checks already rejected. fn classify(path: &Path) -> Option { let head = read_prefix(path, DETECT_PREFIX)?; let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { if detect(&head).is_some() { - Class::Crackproof - } else if senbei_metadata::is_metadata(&head) { - Class::Metadata - } else { - Class::None + return Class::Crackproof; } + if senbei_metadata::is_metadata(&head) { + return Class::Metadata; + } + if crate::android::is_elf64_aarch64(&head) + && std::fs::read(path) + .map(|bytes| senbei_android_engine::is_protected_libil2cpp(&bytes)) + .unwrap_or(false) + { + return Class::AndroidSo; + } + if crate::android::is_app_package(path, &head) { + return Class::AndroidPackage; + } + Class::None })); r.ok() } @@ -370,11 +410,11 @@ mod tests { blob[..4].copy_from_slice(&0xFAB1_1BAFu32.to_le_bytes()); std::fs::write(root.join("metadata"), &blob).unwrap(); - let (_, filtered, _) = find_targets_opts(root, false); - assert!(filtered.is_empty()); + let filtered = find_targets_opts(root, false); + assert!(filtered.metadata.is_empty()); - let (_, exhaustive, _) = find_targets_opts(root, true); - assert_eq!(exhaustive.len(), 1); + let exhaustive = find_targets_opts(root, true); + assert_eq!(exhaustive.metadata.len(), 1); } /// A file below the Crackproof key-table bound is skipped without being @@ -389,10 +429,10 @@ mod tests { // None of them are Crackproof, so both modes find nothing; the point is // that the filtered walk does not panic and honors `scan_all`. - let (c, m, _) = find_targets_opts(root, false); - assert!(c.is_empty() && m.is_empty()); - let (c, m, _) = find_targets_opts(root, true); - assert!(c.is_empty() && m.is_empty()); + let scan = find_targets_opts(root, false); + assert!(scan.crackproof.is_empty() && scan.metadata.is_empty()); + let scan = find_targets_opts(root, true); + assert!(scan.crackproof.is_empty() && scan.metadata.is_empty()); } /// An il2cpp metadata blob is found by the filtered scan: `.dat` is not on @@ -407,9 +447,9 @@ mod tests { // Same magic but too small to be processable — skipped by the size floor. std::fs::write(root.join("stub.dat"), &blob[..64]).unwrap(); - let (_, m, _) = find_targets_opts(root, false); - assert_eq!(m.len(), 1); - assert!(m[0].ends_with("global-metadata.dat")); + let scan = find_targets_opts(root, false); + assert_eq!(scan.metadata.len(), 1); + assert!(scan.metadata[0].ends_with("global-metadata.dat")); } /// Review regression: a previous output tree is pruned case-insensitively @@ -428,12 +468,15 @@ mod tests { // A big non-target file at the root: probed, then skipped. std::fs::write(root.join("plain.dll"), vec![0u8; 100_000]).unwrap(); - let (c, m, stats) = find_targets_opts(root, false); + let scan = find_targets_opts(root, false); assert!( - c.is_empty() && m.is_empty(), + scan.crackproof.is_empty() && scan.metadata.is_empty(), "old output tree must be pruned" ); - assert_eq!(stats.skipped, 1, "the probed non-target counts as skipped"); - assert_eq!(stats.walk_errors, 0); + assert_eq!( + scan.stats.skipped, 1, + "the probed non-target counts as skipped" + ); + assert_eq!(scan.stats.walk_errors, 0); } } diff --git a/senbei-io/src/ui.rs b/senbei-io/src/ui.rs index 8d49ce7..2d04e7b 100644 --- a/senbei-io/src/ui.rs +++ b/senbei-io/src/ui.rs @@ -19,16 +19,22 @@ pub fn progress(n: u64, quiet: bool) -> ProgressBar { /// Print a green success line, suspending the progress bar. pub fn ok(bar: &ProgressBar, quiet: bool, rel: &Path, kind: Kind, dest: &Path) { + ok_label( + bar, + quiet, + &rel.display().to_string(), + &format!("{kind:?}"), + dest, + ); +} + +/// Print a green success line with a free-form kind label (Android targets), +/// suspending the progress bar. +pub fn ok_label(bar: &ProgressBar, quiet: bool, rel: &str, label: &str, dest: &Path) { if quiet { return; } - let msg = format!( - "{} {:?} {} -> {}", - "✓".green(), - kind, - rel.display(), - dest.display() - ); + let msg = format!("{} {} {} -> {}", "✓".green(), label, rel, dest.display()); bar.suspend(|| println!("{msg}")); } diff --git a/senbei-wasm/Cargo.lock b/senbei-wasm/Cargo.lock index c9f4efc..d57a7fe 100644 --- a/senbei-wasm/Cargo.lock +++ b/senbei-wasm/Cargo.lock @@ -2,24 +2,72 @@ # It is not intended for manual editing. version = 4 +[[package]] +name = "adler2" +version = "2.0.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "320119579fcad9c21884f5c4861d16174d0e06250625266f50fe6898340abefa" + +[[package]] +name = "aes" +version = "0.8.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b169f7a6d4742236a0a00c541b845991d0ac43e546831af1249753ab4c3aa3a0" +dependencies = [ + "cfg-if", + "cipher", + "cpufeatures", +] + [[package]] name = "anyhow" version = "1.0.104" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "330a5ed07fa54e4702c9d6c4174f74427fc0ef6e214bbd677ae50a5099946470" +[[package]] +name = "bitflags" +version = "2.13.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b588b76d00fde79687d7646a9b5bdf3cc0f655e0bbd080335a95d7e96f3587da" + +[[package]] +name = "block-buffer" +version = "0.10.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3078c7629b62d3f0439517fa394996acacc5cbc91c5a20d8c658e77abd503a71" +dependencies = [ + "generic-array", +] + [[package]] name = "bumpalo" version = "3.20.3" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "72f5acc6cb2ba439de613abc23857ec3d78374d8ed5ac84e9d11336e87da8649" +[[package]] +name = "byteorder" +version = "1.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1fd0f2584146f6f2ef48085050886acf353beff7305ebd1ae69500e27c67f64b" + [[package]] name = "cfg-if" version = "1.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801" +[[package]] +name = "cipher" +version = "0.4.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "773f3b9af64447d2ce9850330c473515014aa235e6a783b02db81ff39e4a3dad" +dependencies = [ + "crypto-common", + "inout", +] + [[package]] name = "console" version = "0.16.4" @@ -42,12 +90,83 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "cpufeatures" +version = "0.2.17" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "59ed5838eebb26a2bb2e58f6d5b5316989ae9d08bab10e0e6d103e656d1b0280" +dependencies = [ + "libc", +] + +[[package]] +name = "crc32fast" +version = "1.5.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8498c871161e1742aaa9d52551b2d6ebdd4c3d45a3be423e3728f33b955be550" +dependencies = [ + "cfg-if", +] + +[[package]] +name = "crossbeam-utils" +version = "0.8.22" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "61803da095bee82a81bb1a452ecc25d3b2f1416d1897eb86430c6159ef717c17" + +[[package]] +name = "crypto-common" +version = "0.1.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "78c8292055d1c1df0cce5d180393dc8cce0abec0a7102adb6c7b1eef6016d60a" +dependencies = [ + "generic-array", + "typenum", +] + +[[package]] +name = "digest" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9ed9a281f7bc9b7576e61468ba615a66a5c8cfdff42420a70aa82701a3b1e292" +dependencies = [ + "block-buffer", + "crypto-common", +] + [[package]] name = "encode_unicode" version = "1.0.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0" +[[package]] +name = "errno" +version = "0.3.14" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" +dependencies = [ + "libc", + "windows-sys", +] + +[[package]] +name = "fastrand" +version = "2.5.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" + +[[package]] +name = "flate2" +version = "1.1.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e634e2e0ebac1ee034020da1ca582e17ffe4e0f5e985823721e168928136dcb" +dependencies = [ + "crc32fast", + "miniz_oxide", + "zlib-rs", +] + [[package]] name = "futures-core" version = "0.3.34" @@ -72,6 +191,38 @@ dependencies = [ "slab", ] +[[package]] +name = "generic-array" +version = "0.14.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "85649ca51fd72272d7821adaf274ad91c288277713d9c18820d8499a7ff69e9a" +dependencies = [ + "typenum", + "version_check", +] + +[[package]] +name = "getrandom" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099" +dependencies = [ + "cfg-if", + "libc", + "r-efi", +] + +[[package]] +name = "goblin" +version = "0.10.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "17582616a7718cca54cec18e534a76c7c4aec11a8b9a85695712f262fd15a4c8" +dependencies = [ + "log", + "plain", + "scroll", +] + [[package]] name = "indicatif" version = "0.18.6" @@ -85,6 +236,21 @@ dependencies = [ "web-time", ] +[[package]] +name = "inout" +version = "0.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "879f10e63c20629ecabbb64a8010319738c66a5cd0c29b02d63d272b03751d01" +dependencies = [ + "generic-array", +] + +[[package]] +name = "itoa" +version = "1.0.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" + [[package]] name = "js-sys" version = "0.3.104" @@ -102,6 +268,43 @@ version = "0.2.189" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2" +[[package]] +name = "linux-raw-sys" +version = "0.12.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32a66949e030da00e8c7d4434b251670a91556f4144941d37452769c25d58a53" + +[[package]] +name = "log" +version = "0.4.34" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9f8bd3e56ce4dfc153cf470fffbfa98c7620958b312ca5c3a4b8d5181fd13c6" + +[[package]] +name = "memchr" +version = "2.8.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "cf8baf1c55e62ffcace7a9f06f4bd9cd3f0c4beb022d3b367256b91b87513d98" + +[[package]] +name = "memmap2" +version = "0.9.11" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" +dependencies = [ + "libc", +] + +[[package]] +name = "miniz_oxide" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b63fbc4a50860e98e7b2aa7804ded1db5cbc3aff9193adaff57a6931bf7c4b4c" +dependencies = [ + "adler2", + "simd-adler32", +] + [[package]] name = "once_cell" version = "1.21.4" @@ -120,6 +323,12 @@ version = "0.2.17" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" +[[package]] +name = "plain" +version = "0.2.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b4596b6d070b27117e987119b4dac604f3c58cfb0b191112e24771b2faeac1a6" + [[package]] name = "portable-atomic" version = "1.15.0" @@ -144,6 +353,25 @@ dependencies = [ "proc-macro2", ] +[[package]] +name = "r-efi" +version = "6.0.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf" + +[[package]] +name = "rustix" +version = "1.1.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6fe4565b9518b83ef4f91bb47ce29620ca828bd32cb7e408f0062e9930ba190" +dependencies = [ + "bitflags", + "errno", + "libc", + "linux-raw-sys", + "windows-sys", +] + [[package]] name = "rustversion" version = "1.0.23" @@ -159,34 +387,104 @@ dependencies = [ "winapi-util", ] +[[package]] +name = "scroll" +version = "0.13.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c1257cd4248b4132760d6524d6dda4e053bc648c9070b960929bf50cfb1e7add" +dependencies = [ + "scroll_derive", +] + +[[package]] +name = "scroll_derive" +version = "0.13.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e1a36a382ed65dbcc0ab47fd5e9a94112417ccd34560a392ef3b7b0f0ec39148" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.4", +] + +[[package]] +name = "senbei-android-crypto" +version = "1.2.0" +dependencies = [ + "aes", + "thiserror", +] + +[[package]] +name = "senbei-android-elf" +version = "1.2.0" +dependencies = [ + "memmap2", + "senbei-android-crypto", + "serde", + "serde_json", + "sha2", + "tempfile", + "thiserror", +] + +[[package]] +name = "senbei-android-engine" +version = "1.2.0" +dependencies = [ + "goblin", + "memmap2", + "senbei-android-crypto", + "serde", + "serde_json", + "sha2", + "tempfile", + "thiserror", +] + +[[package]] +name = "senbei-android-metadata" +version = "1.2.0" +dependencies = [ + "serde", + "thiserror", +] + [[package]] name = "senbei-crypto" -version = "1.1.0" +version = "1.2.0" dependencies = [ "thiserror", ] [[package]] name = "senbei-io" -version = "1.1.0" +version = "1.2.0" dependencies = [ "anyhow", + "flate2", "indicatif", "libc", "owo-colors", + "senbei-android-elf", + "senbei-android-engine", + "senbei-android-metadata", "senbei-metadata", "senbei-pe", + "sha2", + "tempfile", "walkdir", "windows", + "zip", ] [[package]] name = "senbei-metadata" -version = "1.1.0" +version = "1.2.0" [[package]] name = "senbei-pe" -version = "1.1.0" +version = "1.2.0" dependencies = [ "senbei-crypto", "thiserror", @@ -194,7 +492,7 @@ dependencies = [ [[package]] name = "senbei-wasm" -version = "1.1.0" +version = "1.2.0" dependencies = [ "console_error_panic_hook", "senbei-io", @@ -203,6 +501,66 @@ dependencies = [ "wasm-bindgen", ] +[[package]] +name = "serde" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "4148590afebada386688f18773da617792bf2ef03ffc1e4cbd2b1d45b023e0ba" +dependencies = [ + "serde_core", + "serde_derive", +] + +[[package]] +name = "serde_core" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "67dca2c9c51e58a4791a4b1ed58308b39c64224d349a935ab5039aa360942a48" +dependencies = [ + "serde_derive", +] + +[[package]] +name = "serde_derive" +version = "1.0.229" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e7a5d71263a5a7d47b41f6b3f06ba276f10cc18b0931f1799f710578e2309348" +dependencies = [ + "proc-macro2", + "quote", + "syn 3.0.4", +] + +[[package]] +name = "serde_json" +version = "1.0.151" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c841b55ecdae098c80dcae9cf767f6f8a0c2cdb3416bbef72181df4d0fe73f14" +dependencies = [ + "itoa", + "memchr", + "serde", + "serde_core", + "zmij", +] + +[[package]] +name = "sha2" +version = "0.10.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a7507d819769d01a365ab707794a4084392c824f54a7a6a7862f8c3d0892b283" +dependencies = [ + "cfg-if", + "cpufeatures", + "digest", +] + +[[package]] +name = "simd-adler32" +version = "0.3.10" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "3a219298ac11a56ea9a6d2120044824d6f01aeb034955e7af7bc16858527deea" + [[package]] name = "slab" version = "0.4.12" @@ -231,6 +589,19 @@ dependencies = [ "unicode-ident", ] +[[package]] +name = "tempfile" +version = "3.27.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd" +dependencies = [ + "fastrand", + "getrandom", + "once_cell", + "rustix", + "windows-sys", +] + [[package]] name = "thiserror" version = "2.0.20" @@ -251,6 +622,12 @@ dependencies = [ "syn 3.0.4", ] +[[package]] +name = "typenum" +version = "1.20.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b6f5e870be6c3b371b77fe0ee0bafb859fa4964b4404c27de1d380043c4dda20" + [[package]] name = "unicode-ident" version = "1.0.24" @@ -269,6 +646,12 @@ version = "0.5.2" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "81e544489bf3d8ef66c953931f56617f423cd4b5494be343d9b9d3dda037b9a3" +[[package]] +name = "version_check" +version = "0.9.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0b928f33d975fc6ad9f86c8f283853ad26bdd5b10b7f1542aa2fa15e2289105a" + [[package]] name = "walkdir" version = "2.5.0" @@ -461,3 +844,27 @@ checksum = "3949bd5b99cafdf1c7ca86b43ca564028dfe27d66958f2470940f73d86d75b37" dependencies = [ "windows-link", ] + +[[package]] +name = "zip" +version = "0.6.6" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "760394e246e4c28189f19d488c058bf16f564016aefac5d32bb1f3b51d5e9261" +dependencies = [ + "byteorder", + "crc32fast", + "crossbeam-utils", + "flate2", +] + +[[package]] +name = "zlib-rs" +version = "0.6.7" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "34b31d188d9d685a4f9c7b46d6e36631b07058d2cfe190267adce54dc230bf12" + +[[package]] +name = "zmij" +version = "1.0.23" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "29666d0abbfad1e3dc4dcf6144730dd3a3ab225bbbdac83319345b1b44ccfc1b" diff --git a/senbei-wasm/Cargo.toml b/senbei-wasm/Cargo.toml index dd1fa69..39a6a2c 100644 --- a/senbei-wasm/Cargo.toml +++ b/senbei-wasm/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "senbei-wasm" -version = "1.1.0" +version = "1.2.0" edition = "2024" description = "WebAssembly bindings for senbei (browser frontend assets live in web/)" license = "AGPL-3.0-only"