diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 65d0959..efca40c 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -19,7 +19,7 @@ jobs: runs-on: windows-latest steps: - uses: actions/checkout@v4 - - run: cargo clippy --all-targets -- -D warnings + - run: cargo clippy --workspace --all-targets -- -D warnings test: # The test suite exercises Windows path semantics, so it runs on Windows. @@ -28,7 +28,7 @@ jobs: runs-on: windows-latest steps: - uses: actions/checkout@v4 - - run: cargo test --release + - run: cargo test --release --workspace check-portable: # Build-only portability gate: non-Windows host and the wasm target the @@ -39,8 +39,8 @@ jobs: runs-on: ubuntu-latest steps: - uses: actions/checkout@v4 - - run: cargo clippy --all-targets -- -D warnings - - run: cargo check --target wasm32-unknown-unknown + - run: cargo clippy --workspace --all-targets -- -D warnings + - run: cargo check --workspace --target wasm32-unknown-unknown cli: strategy: diff --git a/AGENTS.md b/AGENTS.md index 55d7781..d0ab548 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,17 +4,19 @@ Guidance for AI coding agents (and human contributors) working in this repo. ## Project -Senbei is a static unpacker for Crackproof-protected PE files: a pure, -panic-free, no-I/O unpacker core (`src/unpacker/`) plus a thin CLI shell -(`src/`), an il2cpp metadata de-obfuscator (`src/metadata.rs`), and a -WebAssembly browser frontend (`web/`). Read `docs/design.md` first. +Senbei is a static unpacker for Crackproof-protected PE files: a Cargo +workspace with a pure, panic-free, no-I/O unpacker core (`senbei-pe/`, built +on `senbei-crypto/`), an il2cpp metadata de-obfuscator (`senbei-metadata/`), +filesystem/CLI orchestration (`senbei-io/`), the `senbei` binary +(`senbei-cli/`), and a WebAssembly browser frontend (`web/`, outside the +workspace). Read `docs/design.md` first. ## Commands ```cmd -cargo build --release :: CLI -cargo test --release :: full suite (golden corpus: samples/, git-ignored) -cargo clippy --all-targets -- -D warnings +cargo build --release :: CLI (default member: senbei-cli) +cargo test --release --workspace :: full suite (golden corpus: samples/, git-ignored) +cargo clippy --workspace --all-targets -- -D warnings cargo fmt --all cd web && wasm-pack build --target web --release :: browser build ``` diff --git a/Cargo.lock b/Cargo.lock index e1917b5..350f66f 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -62,21 +62,21 @@ checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223" [[package]] name = "futures-core" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" [[package]] name = "futures-task" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" [[package]] name = "futures-util" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" dependencies = [ "futures-core", "futures-task", @@ -110,9 +110,9 @@ dependencies = [ [[package]] name = "js-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" +checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" dependencies = [ "cfg-if", "futures-util", @@ -139,9 +139,9 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] name = "owo-colors" -version = "4.3.0" +version = "4.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d211803b9b6b570f68772237e415a029d5a50c65d382910b879fb19d3271f94d" +checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8" [[package]] name = "pin-project-lite" @@ -151,9 +151,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] name = "portable-atomic" -version = "1.14.0" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "proc-macro2" @@ -208,19 +208,48 @@ dependencies = [ ] [[package]] -name = "senbei" -version = "1.0.1" +name = "senbei-cli" +version = "1.1.0" +dependencies = [ + "senbei-io", + "senbei-metadata", + "tempfile", +] + +[[package]] +name = "senbei-crypto" +version = "1.1.0" +dependencies = [ + "thiserror", +] + +[[package]] +name = "senbei-io" +version = "1.1.0" dependencies = [ "anyhow", "indicatif", "libc", "owo-colors", + "senbei-metadata", + "senbei-pe", "tempfile", - "thiserror", "walkdir", "windows", ] +[[package]] +name = "senbei-metadata" +version = "1.1.0" + +[[package]] +name = "senbei-pe" +version = "1.1.0" +dependencies = [ + "senbei-crypto", + "thiserror", +] + [[package]] name = "slab" version = "0.4.12" @@ -240,9 +269,9 @@ dependencies = [ [[package]] name = "syn" -version = "3.0.3" +version = "3.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" dependencies = [ "proc-macro2", "quote", @@ -264,22 +293,22 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.19" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ "thiserror-impl", ] [[package]] name = "thiserror-impl" -version = "2.0.19" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn 3.0.3", + "syn 3.0.4", ] [[package]] @@ -312,9 +341,9 @@ dependencies = [ [[package]] name = "wasm-bindgen" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" +checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" dependencies = [ "cfg-if", "once_cell", @@ -325,9 +354,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" +checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -335,9 +364,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" +checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" dependencies = [ "bumpalo", "proc-macro2", @@ -348,9 +377,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-shared" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" +checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" dependencies = [ "unicode-ident", ] diff --git a/Cargo.toml b/Cargo.toml index 88e468d..f70d511 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -1,39 +1,39 @@ -[package] -name = "senbei" -version = "1.0.1" +[workspace] +members = [ + "senbei-cli", + "senbei-crypto", + "senbei-io", + "senbei-metadata", + "senbei-pe", +] +default-members = ["senbei-cli"] +# The wasm frontend is its own crate (own Cargo.lock, cdylib) and stays outside +# the workspace. +exclude = ["web"] +resolver = "2" + +[workspace.package] +version = "1.1.0" edition = "2024" -description = "Static unpacker for Crackproof-protected PE files" license = "AGPL-3.0-only" -keywords = ["unpacker", "reverse-engineering", "pe", "security-research"] -categories = ["command-line-utilities"] -[lib] -name = "senbei" -path = "src/lib.rs" - -[[bin]] -name = "senbei" -path = "src/main.rs" - -[dependencies] +[workspace.dependencies] anyhow = "1" +indicatif = "0.18" +libc = "0.2" +owo-colors = "4" +tempfile = "3" thiserror = "2" walkdir = "2" -indicatif = "0.18" -owo-colors = "4" - -[target.'cfg(windows)'.dependencies] windows = { version = "0.62", features = [ "Win32_Foundation", "Win32_System_Console", "Win32_System_SystemInformation", ] } - -[target.'cfg(all(not(windows), not(target_arch = "wasm32")))'.dependencies] -libc = "0.2" - -[dev-dependencies] -tempfile = "3" +senbei-crypto = { path = "senbei-crypto" } +senbei-io = { path = "senbei-io" } +senbei-metadata = { path = "senbei-metadata" } +senbei-pe = { path = "senbei-pe" } [profile.release] opt-level = 3 diff --git a/docs/design.md b/docs/design.md index eec8d5b..3b75cd8 100644 --- a/docs/design.md +++ b/docs/design.md @@ -7,41 +7,57 @@ no driver or proxy DLL is involved. ## Crate layout -The crate is split into a pure core and a thin CLI shell: +Senbei is a Cargo workspace split into a pure core and thin shells around it: -- **`src/unpacker/`** — the core. Pure functions over byte slices: no file - I/O, no environment access (beyond a few debugging overrides, see +- **`senbei-pe/`** — the core. Pure functions over byte slices: no file I/O, + no environment access (beyond a few debugging overrides, see [development.md](development.md)), panic-free at the public boundary (all internal panics are trapped and converted to `UnpackError::Corrupt`). This is what the WebAssembly build embeds. -- **`src/` (top level)** — the CLI shell: argument parsing, recursive folder - scanning, per-run log file, progress bar, Explorer-friendly exit pause, and - the single-file/folder orchestration in `job.rs`. -- **`src/metadata.rs`** — il2cpp `global-metadata.dat` method-token +- **`senbei-crypto/`** — cryptographic, checksum, compression, and bytecode + primitives the core is built from. Same purity rules as `senbei-pe`. +- **`senbei-metadata/`** — il2cpp `global-metadata.dat` method-token de-obfuscation (format version 31; other versions are left untouched). +- **`senbei-io/`** — filesystem and orchestration: recursive folder scanning, + per-run log file, progress bar, Explorer-friendly exit pause, and the + single-file/folder orchestration in `job.rs` (incl. the wasm-safe in-memory + byte API used by the web frontend). +- **`senbei-cli/`** — the `senbei` binary: argument parsing + dispatch. The + integration test suite (incl. the golden corpus test) lives in + `senbei-cli/tests/`. ``` -src/ -├── main.rs argument parsing + dispatch -├── lib.rs module roots +senbei-cli/ +└── src/main.rs argument parsing + dispatch +senbei-io/src/ ├── job.rs single-file + folder orchestration, out-naming, │ companion splice, stub overlay/TLS restore, │ pipeline routing (incl. the wasm-safe byte API) ├── scan.rs recursive Crackproof + metadata discovery -├── metadata.rs il2cpp global-metadata.dat de-obfuscation ├── logfile.rs per-run timestamped log ├── ui.rs progress bar + status lines -├── pause.rs Explorer-friendly exit pause -└── unpacker/ pure, panic-free, no-I/O core - ├── mod.rs detection + unpack_auto dispatch - ├── exe.rs EXE pipeline (PE32+ and PE32) - ├── dll.rs native + managed DLL pipeline - ├── integrity.rs static post-unpack sanity check - ├── primitives.rs decrypt_data* steps, key/shift selection - ├── bytecode.rs bytecode VM - ├── parallel.rs deterministic block-parallel fan-out - ├── tables.rs constant tables - └── crc32.rs checksum +└── pause.rs Explorer-friendly exit pause +senbei-metadata/src/ +└── metadata.rs il2cpp global-metadata.dat de-obfuscation +senbei-crypto/src/ +├── primitives.rs decrypt_data* steps, key derivation +├── bytecode.rs bytecode VM +├── tables.rs constant tables +└── crc32.rs checksum +senbei-pe/src/engine/ pure, panic-free, no-I/O core +├── mod.rs detection + unpack_auto dispatch +├── error.rs structured error taxonomy +├── integrity.rs static post-unpack sanity check +├── parallel.rs deterministic block-parallel fan-out +├── layout/ layout discovery + validation +│ ├── dd8.rs .text dd8 key-formula + shift selection +│ ├── discovery.rs layout candidate discovery (trial-and-validate) +│ └── image.rs PE image reconstruction helpers +├── exe/ +│ ├── pipeline.rs EXE pipeline (PE32+ and PE32 orchestration) +│ └── pipeline/pe32.rs PE32-specific EXE restore +└── dll/ + └── pipeline.rs native + managed DLL pipeline ``` ## Detection and routing diff --git a/docs/development.md b/docs/development.md index 0dbaad8..0d72365 100644 --- a/docs/development.md +++ b/docs/development.md @@ -60,9 +60,9 @@ since binaries are not committed). ## Conventions -- The `src/unpacker/` core is pure: no file I/O, no panics across the public - boundary, no `unsafe`. Keep it that way — it is what the WebAssembly build - embeds. +- The `senbei-pe/` core (and its `senbei-crypto/` base) is pure: no file I/O, + no panics across the public boundary, no `unsafe`. Keep it that way — it is + what the WebAssembly build embeds. - Layout heuristics must **trial-and-validate**: never pick a candidate offset on shape alone and trust it; validate by decryption/checksum and fall through to the next candidate on failure. A silent wrong offset produces a diff --git a/senbei-cli/Cargo.toml b/senbei-cli/Cargo.toml new file mode 100644 index 0000000..b6b9da6 --- /dev/null +++ b/senbei-cli/Cargo.toml @@ -0,0 +1,20 @@ +[package] +name = "senbei-cli" +version.workspace = true +edition.workspace = true +description = "Command-line entry point for Senbei" +license.workspace = true +keywords = ["unpacker", "reverse-engineering", "pe", "security-research"] +categories = ["command-line-utilities"] + +[[bin]] +name = "senbei" +path = "src/main.rs" + +[dependencies] +senbei-io.workspace = true + +[dev-dependencies] +senbei-io.workspace = true +senbei-metadata.workspace = true +tempfile.workspace = true diff --git a/src/main.rs b/senbei-cli/src/main.rs similarity index 76% rename from src/main.rs rename to senbei-cli/src/main.rs index da93dcc..2c226bd 100644 --- a/src/main.rs +++ b/senbei-cli/src/main.rs @@ -1,4 +1,4 @@ -use senbei::{job, pause}; +use senbei_io::{job, pause, scan}; use std::path::Path; fn main() -> std::process::ExitCode { @@ -27,9 +27,6 @@ fn main() -> std::process::ExitCode { "--no-log" => no_log = true, "--scan-all" => scan_all = true, "--out" => match args.next() { - // Reject a missing value (and a following flag swallowed as the - // value): previously `--out` at end of argv silently fell back - // to the default output directory. Some(v) if !v.starts_with('-') => out = Some(v), _ => { eprintln!("error: --out requires a directory argument"); @@ -42,7 +39,6 @@ fn main() -> std::process::ExitCode { return std::process::ExitCode::from(2); } other => { - // Previously the last positional silently won. if let Some(prev) = &path { eprintln!("error: multiple input paths given ('{prev}' and '{other}')"); return std::process::ExitCode::from(2); @@ -63,33 +59,36 @@ fn main() -> std::process::ExitCode { } let p = Path::new(&p); let out_path = out.as_deref().map(Path::new); - let r = if p.is_dir() { + let result = if p.is_dir() { job::run_folder_opts( p, out_path, quiet, verbose, no_log, - scan_all || senbei::scan::scan_all_env(), + scan_all || scan::scan_all_env(), ) } else { job::run_file_v(p, out_path, quiet, verbose, no_log) }; - match r { - Ok(s) => { + match result { + Ok(summary) => { if quiet < 2 { println!( "{} unpacked · {} skipped · {} errors · {} suspect · {} metadata", - s.unpacked, s.skipped, s.errors, s.suspect, s.metadata + summary.unpacked, + summary.skipped, + summary.errors, + summary.suspect, + summary.metadata ); - println!("done in {} ms", s.duration_ms); + println!("done in {} ms", summary.duration_ms); } - if s.errors > 0 { 1 } else { 0 } + if summary.errors > 0 { 1 } else { 0 } } - Err(e) => { - // Fatal: out-dir/log create, etc. + Err(error) => { if quiet < 2 { - eprintln!("error: {e:#}"); + eprintln!("error: {error:#}"); } 1 } @@ -107,7 +106,7 @@ fn print_help() { ); println!( " --scan-all probe every file in a folder, including ones the scan\n\ - \x20 pre-filter skips (under 4128 bytes, or a bulk-asset\n\ - \x20 extension like .ab/.xml/.acb). Much slower on game trees." + \x20 pre-filter skips (under 4128 bytes, extensionless,\n\ + \x20 or a bulk-asset extension). Much slower on large trees." ); } diff --git a/senbei-cli/tests/common/mod.rs b/senbei-cli/tests/common/mod.rs new file mode 100644 index 0000000..8db840c --- /dev/null +++ b/senbei-cli/tests/common/mod.rs @@ -0,0 +1,11 @@ +//! Shared test fixtures. +#![allow(dead_code)] + +use std::path::PathBuf; + +/// Path to the workspace-root `samples/` — the user-managed corpus dropped in +/// by hand. Git-ignored except its README; tests here run against whatever is +/// present. `CARGO_MANIFEST_DIR` is `senbei-cli/`, so go one level up. +pub fn samples_dir() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../samples") +} diff --git a/tests/job.rs b/senbei-cli/tests/job.rs similarity index 93% rename from tests/job.rs rename to senbei-cli/tests/job.rs index 8fb8a9c..be2c44a 100644 --- a/tests/job.rs +++ b/senbei-cli/tests/job.rs @@ -1,4 +1,4 @@ -use senbei::job::{default_out_root_for_file, out_name}; +use senbei_io::job::{default_out_root_for_file, out_name}; use std::path::Path; #[test] diff --git a/tests/logfile.rs b/senbei-cli/tests/logfile.rs similarity index 95% rename from tests/logfile.rs rename to senbei-cli/tests/logfile.rs index c8250ea..3a753a2 100644 --- a/tests/logfile.rs +++ b/senbei-cli/tests/logfile.rs @@ -1,4 +1,4 @@ -use senbei::logfile::{Log, local_stamp_compact, local_stamp_display}; +use senbei_io::logfile::{Log, local_stamp_compact, local_stamp_display}; #[test] fn local_stamp_compact_matches_shape() { diff --git a/tests/run_log.rs b/senbei-cli/tests/run_log.rs similarity index 99% rename from tests/run_log.rs rename to senbei-cli/tests/run_log.rs index 3a909e7..fc69bb4 100644 --- a/tests/run_log.rs +++ b/senbei-cli/tests/run_log.rs @@ -1,4 +1,4 @@ -use senbei::job; +use senbei_io::job; use std::path::Path; fn list_logs(dir: &Path) -> Vec { diff --git a/tests/samples.rs b/senbei-cli/tests/samples.rs similarity index 95% rename from tests/samples.rs rename to senbei-cli/tests/samples.rs index 7adf8fe..1102445 100644 --- a/tests/samples.rs +++ b/senbei-cli/tests/samples.rs @@ -8,7 +8,7 @@ //! - golden present, bytes differ -> FAIL (the test fails) //! - no golden -> WARNING (printed; needs a manual check) //! -//! Inputs go through [`senbei::job::unpack_bytes`], the same routing the CLI +//! Inputs go through [`senbei_io::job::unpack_bytes`], the same routing the CLI //! uses, **not** `unpack_auto` directly. That matters: `unpack_auto` alone //! cannot reach the external-companion layout, whose stub is meaningless //! without its `._` payload — a corpus wired to `unpack_auto` silently @@ -17,7 +17,7 @@ //! samples folder is picked up automatically, exactly as it is on disk. //! //! An input whose bytes carry the il2cpp metadata magic is routed through -//! [`senbei::metadata::deobfuscate`] instead, giving the method-token remap +//! [`senbei_metadata::deobfuscate`] instead, giving the method-token remap //! real-world coverage (its unit tests only build synthetic layouts). //! //! The folder is git-ignored (see `senbei/samples/README.md`), so the set of @@ -116,10 +116,10 @@ fn samples_unpack_against_goldens() { } }; - let got = if senbei::metadata::is_metadata(&bytes) { + let got = if senbei_metadata::is_metadata(&bytes) { // il2cpp metadata: method-token de-obfuscation, no PE pipeline and // no integrity check (the output is not a PE image). - match senbei::metadata::deobfuscate(&bytes) { + match senbei_metadata::deobfuscate(&bytes) { Ok((out, _report)) => out, Err(e) => { failures.push(format!("{name}: de-obfuscation failed: {e}")); @@ -140,7 +140,7 @@ fn samples_unpack_against_goldens() { }, None => None, }; - let image = match senbei::job::unpack_bytes(&bytes, companion.as_deref()) { + let image = match senbei_io::job::unpack_bytes(&bytes, companion.as_deref()) { Ok(img) => img, Err(e) => { failures.push(format!("{name}: unpack failed: {e:?}")); diff --git a/senbei-crypto/Cargo.toml b/senbei-crypto/Cargo.toml new file mode 100644 index 0000000..3ee7e4b --- /dev/null +++ b/senbei-crypto/Cargo.toml @@ -0,0 +1,9 @@ +[package] +name = "senbei-crypto" +version.workspace = true +edition.workspace = true +license.workspace = true +description = "Cryptographic and compression primitives for Senbei" + +[dependencies] +thiserror.workspace = true diff --git a/src/unpacker/bytecode.rs b/senbei-crypto/src/bytecode.rs similarity index 95% rename from src/unpacker/bytecode.rs rename to senbei-crypto/src/bytecode.rs index f662c6e..30c4a13 100644 --- a/src/unpacker/bytecode.rs +++ b/senbei-crypto/src/bytecode.rs @@ -61,8 +61,8 @@ impl OpsLut { pub fn generate(data: &[u8], offset: u32) -> Option> { // Bounds-checked cursor: a corrupt `data_offset` (bad decrypt_data6 / the // alignment fallback) must yield `None`, not an out-of-bounds panic — the - // panic path would surface as a misleading `UnpackError::Corrupt` instead - // of the precise `BytecodeGenFailed`, and any future caller without a + // panic path would surface as a misleading `UnpackError::InternalPanic` instead + // of the precise `BytecodeGenerationFailed`, and any future caller without a // `catch_unwind` wrapper would abort outright. let mut pos = offset as usize; let mut next = move || { diff --git a/src/unpacker/crc32.rs b/senbei-crypto/src/crc32.rs similarity index 100% rename from src/unpacker/crc32.rs rename to senbei-crypto/src/crc32.rs diff --git a/senbei-crypto/src/lib.rs b/senbei-crypto/src/lib.rs new file mode 100644 index 0000000..2f99e7f --- /dev/null +++ b/senbei-crypto/src/lib.rs @@ -0,0 +1,77 @@ +//! Cryptographic, checksum, compression, and bytecode primitives. + +pub mod bytecode; +pub mod crc32; +pub mod primitives; +mod tables; + +/// Maximum buffer size accepted by allocation-sensitive transforms. +pub const MAX_IMAGE_SIZE: u64 = 1 << 30; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BufferOperation { + Read, + CopySource, + CopyDestination, + ZeroFill, +} + +impl std::fmt::Display for BufferOperation { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::Read => "read", + Self::CopySource => "copy source", + Self::CopyDestination => "copy destination", + Self::ZeroFill => "zero-fill", + }) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +pub enum Error { + #[error( + "{operation} range out of bounds (offset {offset}, size {size}, buffer length {buffer_len})" + )] + BufferRangeOutOfBounds { + operation: BufferOperation, + offset: usize, + size: usize, + buffer_len: usize, + }, +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)] +#[non_exhaustive] +pub enum DecompressionFailure { + #[error("compressed source size {size} exceeds limit {max}")] + SourceTooLarge { size: u32, max: u64 }, + #[error("Huffman code length {bits} is invalid")] + InvalidCodeLength { bits: u8 }, + #[error("Huffman tree traversal exceeded 64 levels")] + HuffmanTraversalLimit, + #[error("pending length accumulator overflowed at {pending}")] + PendingLengthOverflow { pending: u32 }, + #[error("output step {step} at byte {written} exceeds expected size {expected}")] + OutputOverflow { + written: u32, + step: u32, + expected: u32, + }, + #[error("run-fill width {width} reads before output offset 0x{destination:08X}")] + RunFillBeforeOutput { width: u32, destination: u32 }, + #[error("run-fill width {width} is unsupported")] + InvalidRunFillWidth { width: u32 }, + #[error("back-reference distance {distance} exceeds {written} written bytes")] + InvalidBackReference { distance: u32, written: u32 }, + #[error("Huffman symbol consumed no input and produced no output")] + NoProgress, + #[error( + "output size mismatch (wrote {written}/{expected} bytes after consuming {consumed}/{source_size})" + )] + OutputSizeMismatch { + written: u32, + expected: u32, + consumed: u32, + source_size: u32, + }, +} diff --git a/senbei-crypto/src/primitives.rs b/senbei-crypto/src/primitives.rs new file mode 100644 index 0000000..99b0774 --- /dev/null +++ b/senbei-crypto/src/primitives.rs @@ -0,0 +1,1133 @@ +//! Shared crypto primitives and helper utilities. +//! +//! These functions form the low-level API used by the PE pipelines. +//! Each free function is self-contained: it takes the relevant byte buffer(s) +//! and parameters explicitly, with no coupling to the EXE `Unpacker` struct. + +use crate::bytecode::{Op, OpsLut}; +use crate::crc32; +use crate::tables::{COLUMMIX1, COLUMMIX2, COLUMMIX3, COLUMMIX4, SBOX}; +use std::cell::RefCell; + +thread_local! { + /// Reusable scratch for `decompress`. A single unpack runs `decompress` + /// hundreds of times over small blocks; reusing one growable buffer avoids a + /// fresh allocation each call. Thread-local, so it stays correct (one buffer + /// per worker) under the parallel block fan-out. + static DECOMPRESS_SCRATCH: RefCell> = const { RefCell::new(Vec::new()) }; +} + +// --------------------------------------------------------------------------- +// Byte-order accessors +// --------------------------------------------------------------------------- + +pub fn get_u16(data: &[u8], offset: u32) -> u16 { + let i = offset as usize; + u16::from_le_bytes([data[i], data[i + 1]]) +} + +pub fn get_u32(data: &[u8], offset: u32) -> u32 { + let i = offset as usize; + u32::from_le_bytes([data[i], data[i + 1], data[i + 2], data[i + 3]]) +} + +pub fn get_u64(data: &[u8], offset: u32) -> u64 { + let i = offset as usize; + u64::from_le_bytes([ + data[i], + data[i + 1], + data[i + 2], + data[i + 3], + data[i + 4], + data[i + 5], + data[i + 6], + data[i + 7], + ]) +} + +pub fn write_u16(data: &mut [u8], offset: u32, value: u32) { + let i = offset as usize; + let v = value as u16; + let b = v.to_le_bytes(); + data[i] = b[0]; + data[i + 1] = b[1]; +} + +pub fn write_u32(data: &mut [u8], offset: u32, value: u32) { + let i = offset as usize; + let b = value.to_le_bytes(); + data[i] = b[0]; + data[i + 1] = b[1]; + data[i + 2] = b[2]; + data[i + 3] = b[3]; +} + +// --------------------------------------------------------------------------- +// Checked accessors (return Err instead of panicking on OOB) +// --------------------------------------------------------------------------- + +#[allow(dead_code)] +pub fn try_u32(d: &[u8], off: usize) -> Result { + let end = off + .checked_add(4) + .ok_or(crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::Read, + offset: off, + size: 4, + buffer_len: d.len(), + })?; + d.get(off..end) + .map(|s| u32::from_le_bytes(s.try_into().unwrap())) + .ok_or(crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::Read, + offset: off, + size: 4, + buffer_len: d.len(), + }) +} + +#[allow(dead_code)] +pub fn try_i32(d: &[u8], off: usize) -> Result { + try_u32(d, off).map(|v| v as i32) +} + +/// Checked copy with distinct source and destination range errors. +pub fn try_copy_from_slice( + dst: &mut [u8], + dst_off: usize, + dst_len: usize, + src: &[u8], + src_off: usize, +) -> Result<(), crate::Error> { + let dst_end = dst_off + .checked_add(dst_len) + .ok_or(crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::CopyDestination, + offset: dst_off, + size: dst_len, + buffer_len: dst.len(), + })?; + let src_end = src_off + .checked_add(dst_len) + .ok_or(crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::CopySource, + offset: src_off, + size: dst_len, + buffer_len: src.len(), + })?; + if dst_end > dst.len() { + return Err(crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::CopyDestination, + offset: dst_off, + size: dst_len, + buffer_len: dst.len(), + }); + } + if src_end > src.len() { + return Err(crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::CopySource, + offset: src_off, + size: dst_len, + buffer_len: src.len(), + }); + } + dst[dst_off..dst_end].copy_from_slice(&src[src_off..src_end]); + Ok(()) +} + +/// Reproduce the LFSR keystream that decrypt_data6 XORs in. Used to +/// trial-decrypt candidate bytecode positions without mutating the buffer. +pub fn lfsr_keystream(out: &mut [u8]) { + let mut state: u32 = 1; + for byte in out.iter_mut() { + let mut b: u8 = 0; + for k in 0..8u32 { + b |= ((state & 1) << k) as u8; + state <<= 1; + if state & 0x8000 != 0 { + state ^= 0x8003; + } + } + *byte = b; + } +} + +// --------------------------------------------------------------------------- +// AES primitives +// --------------------------------------------------------------------------- + +/// One AES-CBC-like round over a 16-byte block in `d` at `pos`, using the +/// expanded key schedule stored in `d` at `key_offset`. Works entirely within +/// the single `d` buffer (both ciphertext and key schedule live there). +pub fn aes_round(d: &mut [u8], pos: u32, key_offset: u32, round: u32) { + let cm1 = &COLUMMIX1; + let cm2 = &COLUMMIX2; + let cm3 = &COLUMMIX3; + let cm4 = &COLUMMIX4; + let sbox = &SBOX; + + let mut n0 = get_u32(d, pos).swap_bytes() ^ get_u32(d, key_offset); + let mut n1 = + get_u32(d, pos.wrapping_add(4)).swap_bytes() ^ get_u32(d, key_offset.wrapping_add(4)); + let mut n2 = + get_u32(d, pos.wrapping_add(8)).swap_bytes() ^ get_u32(d, key_offset.wrapping_add(8)); + let mut n3 = + get_u32(d, pos.wrapping_add(12)).swap_bytes() ^ get_u32(d, key_offset.wrapping_add(12)); + + let mut r = 1u32; + while r < round { + let off = key_offset.wrapping_add(r.wrapping_mul(16)); + let a = get_u32(cm2, ((n3 >> 16) & 0xFF) * 4) + ^ get_u32(cm3, ((n2 >> 8) & 0xFF) * 4) + ^ get_u32(cm1, ((n0 >> 24) & 0xFF) * 4) + ^ get_u32(cm4, (n1 & 0xFF) * 4) + ^ get_u32(d, off); + let b = get_u32(cm2, ((n0 >> 16) & 0xFF) * 4) + ^ get_u32(cm1, ((n1 >> 24) & 0xFF) * 4) + ^ get_u32(cm3, ((n3 >> 8) & 0xFF) * 4) + ^ get_u32(cm4, (n2 & 0xFF) * 4) + ^ get_u32(d, off.wrapping_add(4)); + let c = get_u32(cm2, ((n1 >> 16) & 0xFF) * 4) + ^ get_u32(cm3, ((n0 >> 8) & 0xFF) * 4) + ^ get_u32(cm1, ((n2 >> 24) & 0xFF) * 4) + ^ get_u32(cm4, (n3 & 0xFF) * 4) + ^ get_u32(d, off.wrapping_add(8)); + let e = get_u32(cm3, ((n1 >> 8) & 0xFF) * 4) + ^ get_u32(cm2, ((n2 >> 16) & 0xFF) * 4) + ^ get_u32(cm1, ((n3 >> 24) & 0xFF) * 4) + ^ get_u32(cm4, (n0 & 0xFF) * 4) + ^ get_u32(d, off.wrapping_add(12)); + n0 = a; + n1 = b; + n2 = c; + n3 = e; + r = r.wrapping_add(1); + } + + let s0 = (get_u32(sbox, ((n0 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n3 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n2 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n1 & 0xFF) * 4) & 0x0000_00FF); + let s1 = (get_u32(sbox, ((n1 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n0 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n3 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n2 & 0xFF) * 4) & 0x0000_00FF); + let s2 = (get_u32(sbox, ((n2 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n1 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n0 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n3 & 0xFF) * 4) & 0x0000_00FF); + let s3 = (get_u32(sbox, ((n3 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n2 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n1 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n0 & 0xFF) * 4) & 0x0000_00FF); + + let last = key_offset.wrapping_add(round.wrapping_mul(16)); + n0 = s0 ^ get_u32(d, last); + n1 = s1 ^ get_u32(d, last.wrapping_add(4)); + n2 = s2 ^ get_u32(d, last.wrapping_add(8)); + n3 = s3 ^ get_u32(d, last.wrapping_add(12)); + + write_u32(d, pos, n0.swap_bytes()); + write_u32(d, pos.wrapping_add(4), n1.swap_bytes()); + write_u32(d, pos.wrapping_add(8), n2.swap_bytes()); + write_u32(d, pos.wrapping_add(12), n3.swap_bytes()); +} + +/// AES-CBC-like decryption over `size` bytes starting at `pos` in `d`. +/// The key schedule lives at `key_offset` within the same buffer `d`. +pub fn aes_decrypt(d: &mut [u8], pos: u32, size: u32, key_offset: u32) { + let mut prev = [0u8; 16]; + let mut cur = [0u8; 16]; + let round = get_u16(d, key_offset.wrapping_add(2)) as u32; + let blocks = size >> 4; + for i in 0..blocks { + let p = pos.wrapping_add(i.wrapping_mul(16)); + let pi = p as usize; + cur.copy_from_slice(&d[pi..pi + 16]); + aes_round(d, p, key_offset.wrapping_add(4), round); + for j in 0..16 { + d[pi + j] ^= prev[j]; + } + prev = cur; + } +} + +/// [`aes_decrypt`] variant reading the key schedule from a separate snapshot +/// slice instead of the data buffer. `ks` is a snapshot of `d[key_offset..]` +/// taken by [`aes_schedule_snapshot`] (round count at `ks[2]`, round keys from +/// `ks[4]`), so the schedule extent is exactly right by construction. Used by +/// the parallel block fan-out, where each worker owns a disjoint `&mut` span +/// of the image and cannot read the schedule out of the shared buffer. +pub fn aes_decrypt_ks(ks: &[u8], d: &mut [u8], pos: u32, size: u32) { + let mut prev = [0u8; 16]; + let mut cur = [0u8; 16]; + let round = u16::from_le_bytes([ks[2], ks[3]]) as u32; + let sched = &ks[4..]; + let blocks = size >> 4; + for i in 0..blocks { + let p = pos.wrapping_add(i.wrapping_mul(16)); + let pi = p as usize; + cur.copy_from_slice(&d[pi..pi + 16]); + aes_round_ks(sched, d, p, round); + for j in 0..16 { + d[pi + j] ^= prev[j]; + } + prev = cur; + } +} + +/// [`aes_round`] with the round keys in a separate slice (see +/// [`aes_decrypt_ks`]). Identical math; only the key source differs. +fn aes_round_ks(ks: &[u8], d: &mut [u8], pos: u32, round: u32) { + let cm1 = &COLUMMIX1; + let cm2 = &COLUMMIX2; + let cm3 = &COLUMMIX3; + let cm4 = &COLUMMIX4; + let sbox = &SBOX; + let k = |i: u32| get_u32(ks, i); + + let mut n0 = get_u32(d, pos).swap_bytes() ^ k(0); + let mut n1 = get_u32(d, pos.wrapping_add(4)).swap_bytes() ^ k(4); + let mut n2 = get_u32(d, pos.wrapping_add(8)).swap_bytes() ^ k(8); + let mut n3 = get_u32(d, pos.wrapping_add(12)).swap_bytes() ^ k(12); + + let mut r = 1u32; + while r < round { + let off = r.wrapping_mul(16); + let a = get_u32(cm2, ((n3 >> 16) & 0xFF) * 4) + ^ get_u32(cm3, ((n2 >> 8) & 0xFF) * 4) + ^ get_u32(cm1, ((n0 >> 24) & 0xFF) * 4) + ^ get_u32(cm4, (n1 & 0xFF) * 4) + ^ k(off); + let b = get_u32(cm2, ((n0 >> 16) & 0xFF) * 4) + ^ get_u32(cm1, ((n1 >> 24) & 0xFF) * 4) + ^ get_u32(cm3, ((n3 >> 8) & 0xFF) * 4) + ^ get_u32(cm4, (n2 & 0xFF) * 4) + ^ k(off.wrapping_add(4)); + let c = get_u32(cm2, ((n1 >> 16) & 0xFF) * 4) + ^ get_u32(cm3, ((n0 >> 8) & 0xFF) * 4) + ^ get_u32(cm1, ((n2 >> 24) & 0xFF) * 4) + ^ get_u32(cm4, (n3 & 0xFF) * 4) + ^ k(off.wrapping_add(8)); + let e = get_u32(cm3, ((n1 >> 8) & 0xFF) * 4) + ^ get_u32(cm2, ((n2 >> 16) & 0xFF) * 4) + ^ get_u32(cm1, ((n3 >> 24) & 0xFF) * 4) + ^ get_u32(cm4, (n0 & 0xFF) * 4) + ^ k(off.wrapping_add(12)); + n0 = a; + n1 = b; + n2 = c; + n3 = e; + r = r.wrapping_add(1); + } + + let s0 = (get_u32(sbox, ((n0 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n3 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n2 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n1 & 0xFF) * 4) & 0x0000_00FF); + let s1 = (get_u32(sbox, ((n1 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n0 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n3 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n2 & 0xFF) * 4) & 0x0000_00FF); + let s2 = (get_u32(sbox, ((n2 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n1 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n0 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n3 & 0xFF) * 4) & 0x0000_00FF); + let s3 = (get_u32(sbox, ((n3 >> 24) & 0xFF) * 4) & 0xFF00_0000) + | (get_u32(sbox, ((n2 >> 16) & 0xFF) * 4) & 0x00FF_0000) + | (get_u32(sbox, ((n1 >> 8) & 0xFF) * 4) & 0x0000_FF00) + | (get_u32(sbox, (n0 & 0xFF) * 4) & 0x0000_00FF); + + let last = round.wrapping_mul(16); + n0 = s0 ^ k(last); + n1 = s1 ^ k(last.wrapping_add(4)); + n2 = s2 ^ k(last.wrapping_add(8)); + n3 = s3 ^ k(last.wrapping_add(12)); + + write_u32(d, pos, n0.swap_bytes()); + write_u32(d, pos.wrapping_add(4), n1.swap_bytes()); + write_u32(d, pos.wrapping_add(8), n2.swap_bytes()); + write_u32(d, pos.wrapping_add(12), n3.swap_bytes()); +} + +/// Snapshot the AES key schedule at `key_offset` for [`aes_decrypt_ks`]: +/// `d[key_offset .. key_offset + 4 + (round+1)*16]` where `round` is read from +/// the schedule header. Returns `None` when the header is truncated or the +/// round count is implausible (corrupt input — the same bytes would otherwise +/// drive reads past the buffer). +pub fn aes_schedule_snapshot(d: &[u8], key_offset: u32) -> Option> { + let base = key_offset as usize; + let round = u16::from_le_bytes([*d.get(base + 2)?, *d.get(base + 3)?]) as usize; + if round > 64 { + return None; + } + let end = base.checked_add(4 + (round + 1) * 16)?; + if end > d.len() { + return None; + } + Some(d[base..end].to_vec()) +} + +// --------------------------------------------------------------------------- +// Checksum primitives +// --------------------------------------------------------------------------- + +/// CRC32-based checksum over a (offset, length) descriptor pair embedded in +/// `d` at `pos`. Returns `crc32(d[offset..offset+length]) ^ length`. +pub fn calculate_checksum(d: &[u8], pos: u32) -> u32 { + let offset = get_u32(d, pos); + let length = get_u32(d, pos.wrapping_add(4)); + crc32::compute(&d[offset as usize..(offset + length) as usize]) ^ length +} + +/// CRC32 chained checksum. The (offset, length) descriptor at `pos` is read +/// from `d`; the bytes themselves are read from the separate `clean` buffer +/// (the original file image). `start` is the initial CRC accumulator. Returns +/// a range error instead of panicking when a descriptor points past `clean`. +pub fn calculate_checksum2( + d: &[u8], + clean: &[u8], + pos: u32, + start: u32, +) -> Result { + let offset = get_u32(d, pos); + let length = get_u32(d, pos.wrapping_add(4)); + let data_start = offset as usize; + let size = length as usize; + let end = data_start.checked_add(size); + let Some(end) = end.filter(|&end| end <= clean.len()) else { + return Err(crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::Read, + offset: data_start, + size, + buffer_len: clean.len(), + }); + }; + Ok(crc32::append(start, &clean[data_start..end])) +} + +// --------------------------------------------------------------------------- +// Decompression (Huffman/LZ) +// --------------------------------------------------------------------------- + +/// Huffman/LZ decompression operating entirely within a single `d` buffer. +/// Reads `s_size` bytes from `src`, writes `d_size` bytes to `dest`. +/// The Huffman table lives at `key_offset` within `d`. +/// +/// Returns a structured reason when the stream cannot produce exactly +/// `d_size` bytes. +pub fn decompress_detailed( + d: &mut [u8], + src: u32, + mut dest: u32, + key_offset: u32, + s_size: u32, + d_size: u32, +) -> Result<(), crate::DecompressionFailure> { + use crate::DecompressionFailure; + + // Bound the scratch allocation: a corrupt descriptor could request a + // multi-gigabyte source size, and an allocation failure aborts the process + // (uncatchable). Real payloads are far below this. + if s_size as u64 > crate::MAX_IMAGE_SIZE { + return Err(DecompressionFailure::SourceTooLarge { + size: s_size, + max: crate::MAX_IMAGE_SIZE, + }); + } + DECOMPRESS_SCRATCH.with_borrow_mut(|buf| -> Result<(), DecompressionFailure> { + let mut bit_pos: i32 = 0; + let need = (s_size as usize).saturating_add(3); + if buf.len() < need { + buf.resize(need, 0); + } + let mut buf_off: u32 = 0; + let mut src_consumed: i32 = 0; + let mut pending: u32 = 0; + let mut written: u32 = 0; + let src_u = src as usize; + let s_size_u = s_size as usize; + // The bit-reader's final get_u32 may read up to 3 bytes past s_size; those + // must be zero. Reused scratch can hold stale bytes there, so zero them + // before copying the (exactly s_size) source over the head. + buf[s_size_u] = 0; + buf[s_size_u + 1] = 0; + buf[s_size_u + 2] = 0; + buf[..s_size_u].copy_from_slice(&d[src_u..src_u + s_size_u]); + + while (src_consumed as u32) < s_size && written < d_size { + let word = get_u32(&buf[..], buf_off) >> bit_pos; + let tab_addr = key_offset.wrapping_add((word & 0xFF).wrapping_mul(3)); + let mut tab = get_u16(d, tab_addr); + let bits: u8; + if (tab & 0x8000) != 0 { + tab &= 0x7FFF; + bits = d[tab_addr as usize + 2]; + } else { + let mut b2 = d[tab_addr as usize + 2]; + // A Huffman code longer than 32 bits cannot exist; a larger + // length byte comes from a corrupt table, and `1 << b2` would + // panic (debug) or wrap (release) on it. + if b2 >= 32 { + return Err(DecompressionFailure::InvalidCodeLength { bits: b2 }); + } + let mut mask: u32 = 1u32 << b2; + b2 = b2.wrapping_add(1); + let mut idx = (tab & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; + let mut t2 = get_u16(d, key_offset.wrapping_add(idx.wrapping_mul(3))); + // A corrupt table can form a non-terminal cycle; cap the walk so it + // fails instead of spinning forever. + let mut depth = 0u32; + while (t2 & 0x8000) == 0 { + depth += 1; + if depth > 64 { + return Err(DecompressionFailure::HuffmanTraversalLimit); + } + mask <<= 1; + b2 = b2.wrapping_add(1); + idx = (t2 & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; + t2 = get_u16(d, key_offset.wrapping_add(idx.wrapping_mul(3))); + } + tab = t2 & 0x7FFF; + bits = b2; + } + bit_pos += bits as i32; + let advance = bit_pos / 8; + buf_off = buf_off.wrapping_add(advance as u32); + src_consumed += advance; + bit_pos %= 8; + + let mode = (tab as u32) & 0x300; + let payload = (tab as u32) & 0xFF; + let step: u32; + match mode { + 0 => { + step = 1; + d[dest as usize] = payload as u8; + } + 0x100 => { + step = 0; + if pending >= 256 { + return Err(DecompressionFailure::PendingLengthOverflow { pending }); + } + pending = if pending == 0 { + payload + } else { + (pending << 8) | payload + }; + } + 0x200 => { + if pending == 0 { + pending = 1; + } + step = pending.wrapping_mul(payload); + if step.wrapping_add(written) > d_size { + return Err(DecompressionFailure::OutputOverflow { + written, + step, + expected: d_size, + }); + } + // Run-fill replicates the unit just written before `dest`. A + // corrupt stream can emit one of these before anything has been + // written, so guard against reading before the buffer start + // (an unsigned underflow would index astronomically far OOB). + match payload { + 1 => { + if dest < 1 { + return Err(DecompressionFailure::RunFillBeforeOutput { + width: payload, + destination: dest, + }); + } + let v = d[(dest as usize) - 1]; + for k in 0..pending { + d[(dest + k) as usize] = v; + } + } + 2 => { + if dest < 2 { + return Err(DecompressionFailure::RunFillBeforeOutput { + width: payload, + destination: dest, + }); + } + let v = get_u16(d, dest.wrapping_sub(2)); + for k in 0..pending { + write_u16(d, dest.wrapping_add(k.wrapping_mul(2)), v as u32); + } + } + 4 => { + if dest < 4 { + return Err(DecompressionFailure::RunFillBeforeOutput { + width: payload, + destination: dest, + }); + } + let v = get_u32(d, dest.wrapping_sub(4)); + for k in 0..pending { + write_u32(d, dest.wrapping_add(k.wrapping_mul(4)), v); + } + } + _ => { + // Only unit widths 1/2/4 exist. Any other payload comes + // from a corrupt stream: previously this wrote nothing + // yet still counted `step` bytes as written, leaving + // stale-buffer holes that later stages treated as + // plaintext. Report corruption instead. + return Err(DecompressionFailure::InvalidRunFillWidth { + width: payload, + }); + } + } + pending = 0; + } + _ => { + step = payload; + if written.wrapping_add(payload) > d_size + || pending.wrapping_add(payload) > written + { + let distance = pending.wrapping_add(payload); + if distance > written { + return Err(DecompressionFailure::InvalidBackReference { + distance, + written, + }); + } + return Err(DecompressionFailure::OutputOverflow { + written, + step: payload, + expected: d_size, + }); + } + let back = pending.wrapping_add(payload); + for k in 0..payload { + d[(dest + k) as usize] = d[(dest + k - back) as usize]; + } + pending = 0; + } + } + + dest = dest.wrapping_add(step); + written = written.wrapping_add(step); + if bits == 0 && step == 0 { + // Corrupt table: no input bits consumed and no output bytes + // written, so the loop condition can never advance — an + // infinite loop (and `catch_unpack` traps panics, not hangs). + // Every real symbol consumes ≥ 1 bit, so a valid stream can + // never hit this. + return Err(DecompressionFailure::NoProgress); + } + } + src_consumed += if bit_pos != 0 { 1 } else { 0 }; + if written != d_size { + return Err(DecompressionFailure::OutputSizeMismatch { + written, + expected: d_size, + consumed: src_consumed.max(0) as u32, + source_size: s_size, + }); + } + Ok(()) + }) +} + +/// Boolean compatibility wrapper used by candidate searches and block fan-out. +pub fn decompress( + d: &mut [u8], + src: u32, + dest: u32, + key_offset: u32, + s_size: u32, + d_size: u32, +) -> bool { + decompress_detailed(d, src, dest, key_offset, s_size, d_size).is_ok() +} + +/// Walk the Huffman table at `key_offset` and snapshot its bytes for +/// [`decompress_tbl`]. The table is a forest of 256 root entries (3 bytes +/// each); non-terminal entries point at a child index pair. Returns `None` +/// when the table is truncated or self-referential past the buffer (corrupt +/// input — the same bytes would otherwise drive reads out of bounds). +pub fn huffman_table_snapshot(d: &[u8], key_offset: u32) -> Option> { + let mut visited = vec![false; 0x1_0000usize]; + let mut stack: Vec = (0..256).collect(); + let mut max_idx: u32 = 255; + while let Some(idx) = stack.pop() { + if idx >= 0x1_0000 || visited[idx as usize] { + continue; + } + visited[idx as usize] = true; + let off = key_offset as usize + idx as usize * 3; + if off + 3 > d.len() { + return None; + } + let t = get_u16(d, key_offset.wrapping_add(idx.wrapping_mul(3))); + if (t & 0x8000) == 0 { + let child = (t & 0x7FFF) as u32; + max_idx = max_idx.max(child).max(child.wrapping_add(1)); + stack.push(child); + stack.push(child.wrapping_add(1)); + } + } + let end = key_offset as usize + (max_idx as usize + 1) * 3; + if end > d.len() { + return None; + } + Some(d[key_offset as usize..end].to_vec()) +} + +/// [`decompress`] variant reading the Huffman table from a separate snapshot +/// slice (see [`huffman_table_snapshot`]) instead of the data buffer. Used by +/// the parallel block fan-out, where each worker owns a disjoint `&mut` span +/// and cannot read the table out of the shared image. Table reads are bounds +/// checked against the snapshot — past-the-end means corrupt table, reported +/// as `false` rather than a panic. +pub fn decompress_tbl( + tab: &[u8], + d: &mut [u8], + src: u32, + mut dest: u32, + s_size: u32, + d_size: u32, +) -> bool { + if s_size as u64 > crate::MAX_IMAGE_SIZE { + return false; + } + DECOMPRESS_SCRATCH.with_borrow_mut(|buf| { + // Table reads, bounds-checked against the snapshot. + let tab16 = |addr: usize| -> Option { + let b = tab.get(addr..addr + 3)?; + Some(u16::from_le_bytes([b[0], b[1]])) + }; + let tab8 = |addr: usize| -> Option { tab.get(addr + 2).copied() }; + + let mut bit_pos: i32 = 0; + let need = (s_size as usize).saturating_add(3); + if buf.len() < need { + buf.resize(need, 0); + } + let mut buf_off: u32 = 0; + let mut src_consumed: i32 = 0; + let mut pending: u32 = 0; + let mut written: u32 = 0; + let src_u = src as usize; + let s_size_u = s_size as usize; + buf[s_size_u] = 0; + buf[s_size_u + 1] = 0; + buf[s_size_u + 2] = 0; + buf[..s_size_u].copy_from_slice(&d[src_u..src_u + s_size_u]); + + while (src_consumed as u32) < s_size && written < d_size { + let word = get_u32(&buf[..], buf_off) >> bit_pos; + let tab_addr = ((word & 0xFF).wrapping_mul(3)) as usize; + let mut tab = match tab16(tab_addr) { + Some(t) => t, + None => { + return false; + } + }; + let bits: u8; + if (tab & 0x8000) != 0 { + tab &= 0x7FFF; + bits = match tab8(tab_addr) { + Some(b) => b, + None => return false, + }; + } else { + let mut b2 = match tab8(tab_addr) { + Some(b) => b, + None => return false, + }; + if b2 >= 32 { + return false; + } + let mut mask: u32 = 1u32 << b2; + b2 = b2.wrapping_add(1); + let mut idx = (tab & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; + let mut t2 = match tab16(idx as usize * 3) { + Some(t) => t, + None => { + return false; + } + }; + // A corrupt table can form a non-terminal cycle; cap the walk so it + // fails instead of spinning forever. + let mut depth = 0u32; + while (t2 & 0x8000) == 0 { + depth += 1; + if depth > 64 { + return false; + } + mask <<= 1; + b2 = b2.wrapping_add(1); + idx = (t2 & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; + t2 = match tab16(idx as usize * 3) { + Some(t) => t, + None => { + return false; + } + }; + } + tab = t2 & 0x7FFF; + bits = b2; + } + bit_pos += bits as i32; + let advance = bit_pos / 8; + buf_off = buf_off.wrapping_add(advance as u32); + src_consumed += advance; + bit_pos %= 8; + + let mode = (tab as u32) & 0x300; + let payload = (tab as u32) & 0xFF; + let step: u32; + match mode { + 0 => { + step = 1; + d[dest as usize] = payload as u8; + } + 0x100 => { + step = 0; + if pending >= 256 { + return false; + } + pending = if pending == 0 { + payload + } else { + (pending << 8) | payload + }; + } + 0x200 => { + if pending == 0 { + pending = 1; + } + step = pending.wrapping_mul(payload); + if step.wrapping_add(written) > d_size { + return false; + } + // Run-fill replicates the unit just written before `dest` + // (see `decompress` for the underflow rationale). + match payload { + 1 => { + if dest < 1 { + return false; + } + let v = d[(dest as usize) - 1]; + for k in 0..pending { + d[(dest + k) as usize] = v; + } + } + 2 => { + if dest < 2 { + return false; + } + let v = get_u16(d, dest.wrapping_sub(2)); + for k in 0..pending { + write_u16(d, dest.wrapping_add(k.wrapping_mul(2)), v as u32); + } + } + 4 => { + if dest < 4 { + return false; + } + let v = get_u32(d, dest.wrapping_sub(4)); + for k in 0..pending { + write_u32(d, dest.wrapping_add(k.wrapping_mul(4)), v); + } + } + _ => { + return false; + } + } + pending = 0; + } + _ => { + step = payload; + if written.wrapping_add(payload) > d_size + || pending.wrapping_add(payload) > written + { + return false; + } + let back = pending.wrapping_add(payload); + for k in 0..payload { + d[(dest + k) as usize] = d[(dest + k - back) as usize]; + } + pending = 0; + } + } + + dest = dest.wrapping_add(step); + written = written.wrapping_add(step); + if bits == 0 && step == 0 { + return false; + } + } + src_consumed += if bit_pos != 0 { 1 } else { 0 }; + let _ = src_consumed; + written == d_size + }) +} +// Decrypt primitives (free-function wrappers) +// --------------------------------------------------------------------------- + +/// decrypt_data3: XOR+rotate cipher. Reads/writes dwords in `d` starting at +/// the address stored at `d[pos]`, for `d[pos+4]>>2` words. `shift` is the +/// right-rotate amount (19 or 21 depending on caller). +pub fn decrypt_data3(d: &mut [u8], pos: u32, mut key: u32, shift: u32) { + let base_addr = get_u32(d, pos); + let length = get_u32(d, pos.wrapping_add(4)); + let words = length >> 2; + for i in 0..words { + let off = base_addr.wrapping_add(i.wrapping_mul(4)); + let v = get_u32(d, off) ^ key; + key = key.wrapping_add(i); + let rotated = v.rotate_right(shift); + write_u32(d, off, rotated.wrapping_sub(i)); + } +} + +/// decrypt_data1 (called `decrypt_data` in the original): decode the 8-dword +/// info header from `file_data` at offset 4096 and write results into `info`. +pub fn decrypt_data1(file_data: &[u8], info: &mut [u32; 8]) { + info[0] = get_u32(file_data, 4096); + let mut k = get_u32(file_data, 4096); + for i in 0..7u32 { + let off = i.wrapping_mul(4).wrapping_add(4); + let cell = get_u32(file_data, 4096u32.wrapping_add(off)); + info[(i + 1) as usize] = k ^ cell; + k = i.wrapping_mul(i) ^ (k.wrapping_add(cell).wrapping_sub(i)); + } +} + +/// decrypt_data6: LFSR XOR decryption of a bytecode block at `pos` in `d`. +/// The block length is read from `d[pos + 95]`. +pub fn decrypt_data6(d: &mut [u8], pos: u32) { + let len = d[(pos + 95) as usize] as usize; + // The keystream is exactly `lfsr_keystream`'s — generate it once (len is a + // byte, so 256 always covers it) instead of keeping a second copy of the + // LFSR that a future poly fix would have to update separately. + let mut ks = [0u8; 256]; + lfsr_keystream(&mut ks); + let pos = pos as usize; + for i in 0..len { + d[pos + i] ^= ks[i]; + } +} + +/// decrypt_data7: nibble-swap + key-rolling byte cipher applied to a +/// null-terminated string in `d` starting at `pos`. +pub fn decrypt_data7(d: &mut [u8], pos: u32, mut key: u8) { + let mut i: u32 = 0; + loop { + let idx = (pos + i) as usize; + if d[idx] == 0 { + break; + } + let mut b = d[idx]; + b = b.rotate_right(4); + b = b.wrapping_sub(key); + if b == 0 { + b = 0u8.wrapping_sub(key); + } + d[idx] = b; + key = key.wrapping_add(67); + i += 1; + } +} + +// --------------------------------------------------------------------------- +// Higher-level composite: AES + decrypt3 + optional bytecode + decompress +// --------------------------------------------------------------------------- + +/// Decrypt and optionally decompress a stage payload descriptor. +/// `pos` points to a (src, src_len, dest, dest_len) quad of dwords in `d`. +/// - AES-decrypts `src..src+src_len` using key at `key3_offset` +/// - XOR+rotate-decrypts with `decrypt_data3(pos, key, 19)` +/// - Applies optional custom `ops` bytecode per-byte +/// - If `src_len != dest_len`, Huffman/LZ-decompresses `src..` → `dest..` +/// +/// Returns the decompression success status (always `true` when no +/// decompression was needed). The PE32 eighth-stage key search relies on this. +pub fn decrypt_and_decompress_data_detailed( + d: &mut [u8], + pos: u32, + key: u32, + key1_offset: u32, + key3_offset: u32, + ops: Option<&[Op]>, +) -> Result<(), crate::DecompressionFailure> { + let src = get_u32(d, pos); + let src_len = get_u32(d, pos.wrapping_add(4)); + aes_decrypt(d, src, src_len, key3_offset); + decrypt_data3(d, pos, key, 19); + if let Some(ops) = ops + && src_len != 0 + { + OpsLut::new(ops).map_region(d, src as usize, src_len as usize); + } + let dest = get_u32(d, pos.wrapping_add(8)); + let dest_len = get_u32(d, pos.wrapping_add(12)); + if src_len != dest_len { + return decompress_detailed(d, src, dest, key1_offset, src_len, dest_len); + } + Ok(()) +} + +/// Boolean compatibility wrapper used by key searches that trial candidates. +pub fn decrypt_and_decompress_data( + d: &mut [u8], + pos: u32, + key: u32, + key1_offset: u32, + key3_offset: u32, + ops: Option<&[Op]>, +) -> bool { + decrypt_and_decompress_data_detailed(d, pos, key, key1_offset, key3_offset, ops).is_ok() +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn checked_copy_distinguishes_source_and_destination_ranges() { + let mut short_destination = [0u8; 2]; + let source = [1u8; 4]; + let error = try_copy_from_slice(&mut short_destination, 0, 3, &source, 0) + .expect_err("destination must be rejected"); + assert!(matches!( + error, + crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::CopyDestination, + offset: 0, + size: 3, + buffer_len: 2, + } + )); + + let mut destination = [0u8; 4]; + let short_source = [1u8; 2]; + let error = try_copy_from_slice(&mut destination, 0, 3, &short_source, 0) + .expect_err("source must be rejected"); + assert!(matches!( + error, + crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::CopySource, + offset: 0, + size: 3, + buffer_len: 2, + } + )); + } + + #[test] + fn aes_ks_variant_matches_single_buffer() { + // Random-ish key schedule at ko and data block; both variants must + // produce identical output. + let ko: usize = 0x40; + let mut d = vec![0u8; 0x400]; + let mut x: u32 = 0x12345678; + for b in d.iter_mut() { + x = x.wrapping_mul(1664525).wrapping_add(1013904223); + *b = (x >> 24) as u8; + } + d[ko + 2] = 10; // round count = 10 + d[ko + 3] = 0; + let snap = aes_schedule_snapshot(&d, ko as u32).expect("snapshot"); + + let mut a = d.clone(); + aes_decrypt(&mut a, 0x100, 0x80, ko as u32); + let mut b = d.clone(); + aes_decrypt_ks(&snap, &mut b, 0x100, 0x80); + if a != b { + let idx = (0..a.len()).find(|&i| a[i] != b[i]).unwrap(); + panic!( + "first diff at {idx:#x}: a={:02x} b={:02x}\n a[..]: {:02x?}\n b[..]: {:02x?}", + a[idx], + b[idx], + &a[idx..idx + 16], + &b[idx..idx + 16] + ); + } + } + + #[test] + fn dtbl_variant_matches_single_buffer() { + // Real table + real compressed block lifted from an actual unpack is + // covered by the golden suite; here we just check a trivial stream: + // build a table where every byte is a literal (mode 0, 8 bits), then + // a source stream of N bytes should expand to N identical bytes. + let ko: usize = 0x100; + let mut d = vec![0u8; 0x1000]; + for e in 0..256usize { + let off = ko + e * 3; + let sym = 0x8000u16 | (e as u16 & 0xFF); // terminal, mode 0, payload=e + d[off] = (sym & 0xFF) as u8; + d[off + 1] = (sym >> 8) as u8; + d[off + 2] = 8; // 8 bits per symbol + } + // Source: 16 bytes 0x00..0x0F at src. + let src = 0x600u32; + for i in 0..16u32 { + d[(src + i) as usize] = i as u8; + } + let snap = huffman_table_snapshot(&d, ko as u32).expect("table snapshot"); + + let mut a = vec![0u8; 0x1000]; + a[..d.len()].copy_from_slice(&d); + assert!(decompress(&mut a, src, 0x800, ko as u32, 16, 16)); + let mut b = d.clone(); + assert!(decompress_tbl(&snap, &mut b, src, 0x800, 16, 16)); + assert_eq!(&a[0x800..0x810], &b[0x800..0x810]); + assert_eq!(&b[0x800..0x810], &(0u8..16).collect::>()[..]); + } + + /// Review regression: a run-fill token with a unit width other than 1/2/4 + /// comes from a corrupt stream and must report failure — previously it + /// wrote nothing yet still counted the bytes as written, leaving stale + /// holes that later stages treated as plaintext. + #[test] + fn decompress_rejects_unknown_run_fill_width() { + // Huffman table at key_offset 0, entry 0: terminal symbol with + // mode 0x200 (run-fill), payload 3 (invalid width), code length 8. + let mut d = vec![0u8; 0x100]; + let sym: u16 = 0x8000 | 0x203; + d[0..2].copy_from_slice(&sym.to_le_bytes()); + d[2] = 8; + // All-zero source -> symbol index 0 -> the invalid run-fill. + assert_eq!( + decompress_detailed(&mut d, 0x40, 0x80, 0, 4, 3), + Err(crate::DecompressionFailure::InvalidRunFillWidth { width: 3 }) + ); + } + + /// Control for the above: a width-1 run-fill is legal and succeeds. + #[test] + fn decompress_accepts_width1_run_fill() { + let mut d = vec![0u8; 0x100]; + d[0x7F] = 0x5A; // unit to replicate + let sym: u16 = 0x8000 | 0x201; + d[0..2].copy_from_slice(&sym.to_le_bytes()); + d[2] = 8; + assert!(decompress(&mut d, 0x40, 0x80, 0, 4, 3)); + assert_eq!(&d[0x80..0x83], &[0x5A, 0x5A, 0x5A]); + } + + #[test] + fn checksum2_rejects_source_range_outside_clean_image() { + let mut descriptor = [0u8; 8]; + descriptor[0..4].copy_from_slice(&448u32.to_le_bytes()); + descriptor[4..8].copy_from_slice(&634_432u32.to_le_bytes()); + let clean = vec![0u8; 590_896]; + let error = calculate_checksum2(&descriptor, &clean, 0, 0).expect_err("range must fail"); + assert!(matches!( + error, + crate::Error::BufferRangeOutOfBounds { + operation: crate::BufferOperation::Read, + offset: 448, + size: 634_432, + buffer_len: 590_896, + } + )); + } +} diff --git a/src/unpacker/tables.rs b/senbei-crypto/src/tables.rs similarity index 92% rename from src/unpacker/tables.rs rename to senbei-crypto/src/tables.rs index 04dcc9b..29fb9fd 100644 --- a/src/unpacker/tables.rs +++ b/senbei-crypto/src/tables.rs @@ -146,11 +146,11 @@ mod tests { #[test] fn generated_tables_match_committed_bytes() { assert_eq!(COLUMMIX1.len(), 1024); - assert_eq!(super::super::crc32::compute(&COLUMMIX1), 0x7e8d_5d5f); - assert_eq!(super::super::crc32::compute(&COLUMMIX2), 0xfcc4_acfc); - assert_eq!(super::super::crc32::compute(&COLUMMIX3), 0x637a_f0cd); - assert_eq!(super::super::crc32::compute(&COLUMMIX4), 0x1e7b_c381); - assert_eq!(super::super::crc32::compute(&SBOX), 0x10fd_6dc1); + assert_eq!(crate::crc32::compute(&COLUMMIX1), 0x7e8d_5d5f); + assert_eq!(crate::crc32::compute(&COLUMMIX2), 0xfcc4_acfc); + assert_eq!(crate::crc32::compute(&COLUMMIX3), 0x637a_f0cd); + assert_eq!(crate::crc32::compute(&COLUMMIX4), 0x1e7b_c381); + assert_eq!(crate::crc32::compute(&SBOX), 0x10fd_6dc1); // Spot-check the first dword of each (matches the original first row). assert_eq!(&COLUMMIX1[..4], &[0x50, 0xa7, 0xf4, 0x51]); assert_eq!(&COLUMMIX2[..4], &[0xa7, 0xf4, 0x51, 0x50]); diff --git a/senbei-io/Cargo.toml b/senbei-io/Cargo.toml new file mode 100644 index 0000000..6cc3e0f --- /dev/null +++ b/senbei-io/Cargo.toml @@ -0,0 +1,23 @@ +[package] +name = "senbei-io" +version.workspace = true +edition.workspace = true +license.workspace = true +description = "Filesystem, scanning, logging, and CLI orchestration for Senbei" + +[dependencies] +anyhow.workspace = true +indicatif.workspace = true +owo-colors.workspace = true +senbei-metadata.workspace = true +senbei-pe.workspace = true +walkdir.workspace = true + +[target.'cfg(windows)'.dependencies] +windows.workspace = true + +[target.'cfg(all(not(windows), not(target_arch = "wasm32")))'.dependencies] +libc.workspace = true + +[dev-dependencies] +tempfile.workspace = true diff --git a/src/job.rs b/senbei-io/src/job.rs similarity index 96% rename from src/job.rs rename to senbei-io/src/job.rs index 5d3d22d..44b08ac 100644 --- a/src/job.rs +++ b/senbei-io/src/job.rs @@ -1,4 +1,4 @@ -use crate::unpacker; +use senbei_pe as unpacker; use std::path::{Path, PathBuf}; /// Crackproof header key table lives at this fixed file offset. For the @@ -387,9 +387,9 @@ pub fn run_folder_v( /// /// When `scan_all` is true every regular file under `root` is opened and /// content-probed, instead of skipping ones the free directory metadata already -/// rules out (too small to hold a Crackproof key table, or a bulk-asset -/// extension). See [`crate::scan::find_targets_opts`] — exhaustive scanning is -/// dramatically slower on asset-heavy game trees and finds the same targets. +/// rules out (extensionless, too small to hold a Crackproof key table, or a +/// bulk-asset extension). See [`crate::scan::find_targets_opts`] — exhaustive +/// scanning is dramatically slower on asset-heavy trees. pub fn run_folder_opts( root: &Path, out_dir: Option<&Path>, @@ -518,7 +518,7 @@ pub fn run_folder_opts( // il2cpp metadata pass. Crackproof's `-GMD` option obfuscates the method // tokens in `global-metadata.dat`; de-obfuscate any we find so the unpacked // il2cpp game assembly resolves methods instead of indexing its per-module - // tables out of bounds (see [`crate::metadata`]). This is additive to the + // tables out of bounds (see [`senbei_metadata`]). This is additive to the // Crackproof module unpack above — the metadata blob is not itself a // Crackproof file. for meta in metas { @@ -660,7 +660,7 @@ pub fn run_file_v( let mut buf = [0u8; 4]; std::fs::File::open(input) .and_then(|mut f| f.read_exact(&mut buf)) - .map(|_| crate::metadata::is_metadata(&buf)) + .map(|_| senbei_metadata::is_metadata(&buf)) .unwrap_or(false) }; @@ -801,13 +801,13 @@ fn panic_payload(panic: &(dyn std::any::Any + Send)) -> String { } } -/// If `e`'s chain contains [`crate::metadata::Error::UnsupportedVersion`], +/// If `e`'s chain contains [`senbei_metadata::Error::UnsupportedVersion`], /// return the version. Used to apply the folder-mode "leave untouched, don't /// fail the run" policy to metadata versions this build can't de-obfuscate. fn unsupported_version(e: &anyhow::Error) -> Option { for cause in e.chain() { - if let Some(crate::metadata::Error::UnsupportedVersion(v)) = - cause.downcast_ref::() + if let Some(senbei_metadata::Error::UnsupportedVersion(v)) = + cause.downcast_ref::() { return Some(*v); } @@ -831,19 +831,14 @@ fn write_atomic(dest: &Path, bytes: &[u8]) -> std::io::Result<()> { r } -/// Detect `bytes` and run the right pipeline. The EXE pipeline is invoked -/// directly (no DLL-pipeline probe) when the input was spliced from an -/// external companion (`spliced`) or when the caller forces it (`force_exe` -/// — the web app's recovery path after a DLL-probe trap; see -/// [`unpack_bytes_force_exe`]). +/// Detect `bytes` and run the right pipeline. Spliced external companions use +/// the EXE pipeline directly because that layout is definitionally EXE-style. /// /// Routing spliced inputs straight to the EXE pipeline is safe: the /// companion layout is definitionally the EXE-style shell (the runtime /// loader maps the companion and runs the standard shell unpack), so the DLL -/// pipeline probe can never be right for it — and probing is not a no-op on -/// targets without unwinding (wasm), where the probe's caught panic becomes -/// a fatal trap. Output bytes are identical to the dll-first + exe-fallback -/// route for every input that route handles. +/// pipeline probe can never be right for it. Output bytes are identical to the +/// DLL-first + EXE-fallback route for every input that route handles. fn unpack_spliced_or_auto( bytes: &[u8], spliced: bool, @@ -880,10 +875,9 @@ pub struct UnpackedImage { /// Unpack in-memory `input` bytes, optionally paired with an external /// companion payload `companion` (the `._` file's contents). /// -/// This is the I/O-free counterpart of [`unpack_one_v`], used by the -/// WebAssembly build: splice (when the companion's first 32 bytes match the -/// stub header), unpack, overlay the export table and TLS directory from the -/// stub, then run the static integrity check. +/// This is the in-memory counterpart of [`unpack_one_v`]: splice a matching +/// companion, unpack, overlay the export table and TLS directory from the stub, +/// then run the static integrity check. pub fn unpack_bytes( input: &[u8], companion: Option<&[u8]>, @@ -960,7 +954,7 @@ pub fn unpack_one_v( /// into a sparse, original-metadata-style value; il2cpp expects the contiguous /// per-module index it indexes its codegen tables with, so a statically-unpacked /// il2cpp game assembly reads garbage and crashes during init. This rewrites -/// the tokens back to their canonical form (see [`crate::metadata::deobfuscate`]). +/// the tokens back to their canonical form (see [`senbei_metadata::deobfuscate`]). /// /// The output is written only when something actually changed /// (`report.remapped > 0`); an already-clean metadata is left untouched and no @@ -970,11 +964,11 @@ pub fn deobfuscate_metadata_to( input: &Path, dest: &Path, verbose: bool, -) -> anyhow::Result { +) -> anyhow::Result { let data = std::fs::read(input)?; // Preserve the metadata::Error in the chain (rather than stringifying it) // so the folder driver can apply its unsupported-version policy. - let (out, report) = crate::metadata::deobfuscate(&data) + let (out, report) = senbei_metadata::deobfuscate(&data) .map_err(|e| anyhow::Error::new(e).context(format!("{input:?}")))?; if report.remapped > 0 { if let Some(parent) = dest.parent() { diff --git a/src/lib.rs b/senbei-io/src/lib.rs similarity index 59% rename from src/lib.rs rename to senbei-io/src/lib.rs index 270c372..5147a04 100644 --- a/src/lib.rs +++ b/senbei-io/src/lib.rs @@ -1,7 +1,7 @@ +//! Filesystem and command-line orchestration. + pub mod job; pub mod logfile; -pub mod metadata; pub mod pause; pub mod scan; pub mod ui; -pub mod unpacker; diff --git a/src/logfile.rs b/senbei-io/src/logfile.rs similarity index 100% rename from src/logfile.rs rename to senbei-io/src/logfile.rs diff --git a/src/pause.rs b/senbei-io/src/pause.rs similarity index 100% rename from src/pause.rs rename to senbei-io/src/pause.rs diff --git a/src/scan.rs b/senbei-io/src/scan.rs similarity index 87% rename from src/scan.rs rename to senbei-io/src/scan.rs index d0308f4..8f3bbed 100644 --- a/src/scan.rs +++ b/senbei-io/src/scan.rs @@ -1,4 +1,4 @@ -use crate::unpacker::detect; +use senbei_pe::detect; use std::io::Read; use std::path::{Path, PathBuf}; use walkdir::WalkDir; @@ -16,24 +16,23 @@ const DETECT_PREFIX: u64 = 8 * 1024; /// Smallest file that can possibly be a target, so anything shorter is skipped /// without ever being opened. /// -/// A Crackproof module needs ≥ 4128 bytes for [`crate::unpacker::detect`]'s key +/// A Crackproof module needs ≥ 4128 bytes for [`senbei_pe::detect`]'s key /// table (it reads the dword at 4124), so the bound is exact for the unpack /// path. An il2cpp `global-metadata.dat` only needs 4 bytes to match its magic, /// but its header alone runs to offset 0xB0 and the images/types/methods tables /// it indexes make every real one megabytes long — a sub-4 KiB "metadata" could -/// only ever fail [`crate::metadata::deobfuscate`] with `Malformed`, so nothing +/// only ever fail [`senbei_metadata::deobfuscate`] with `Malformed`, so nothing /// processable is lost. const MIN_SIZE: u64 = 4128; /// File extensions that are bulk data by construction and can never be a PE /// image or an il2cpp metadata blob. /// -/// This is deliberately a **deny**-list, not an allow-list: the default is to -/// probe, so anything unrecognised is still opened. Targets are recognised by -/// content, not extension, and can carry arbitrary names — there is no closed -/// set of target extensions an allow-list of `exe`/`dll` could enumerate. -/// Only extensions that are bulk asset or text formats by construction appear -/// here. +/// This is deliberately a **deny**-list, not an executable allow-list: unknown +/// extensions are still probed. Extensionless files are handled separately by +/// [`denied_name`] because asset stores commonly contain tens of thousands of +/// extensionless chunks; exhaustive probing remains available through +/// `--scan-all`. /// /// Set `SENBEI_SCAN_ALL=1` (or pass `--scan-all`) to probe every file regardless. const DENY_EXT: &[&str] = &[ @@ -90,11 +89,12 @@ const DENY_EXT: &[&str] = &[ "sr", ]; -/// Whether `path`'s extension is on [`DENY_EXT`]. Extensionless files are never -/// denied (they could be anything). -fn denied_ext(path: &Path) -> bool { +/// Whether `path` can be skipped from its name alone. Extensionless files and +/// files whose extension is on [`DENY_EXT`] are not opened during a default +/// scan. `--scan-all` remains available when exhaustive probing is required. +fn denied_name(path: &Path) -> bool { let Some(ext) = path.extension() else { - return false; + return true; }; let Some(ext) = ext.to_str() else { return false; @@ -206,12 +206,17 @@ pub fn find_targets_opts(root: &Path, scan_all: bool) -> (Vec, Vec (Vec, Vec> = vec![Some(Class::None); n]; - let workers = crate::unpacker::parallel::thread_cap().clamp(1, n.max(1)); + let workers = senbei_pe::thread_cap().clamp(1, n.max(1)); if workers <= 1 { for (p, c) in paths.iter().zip(class.iter_mut()) { *c = classify(p); @@ -304,7 +309,7 @@ fn classify(path: &Path) -> Option { let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { if detect(&head).is_some() { Class::Crackproof - } else if crate::metadata::is_metadata(&head) { + } else if senbei_metadata::is_metadata(&head) { Class::Metadata } else { Class::None @@ -329,29 +334,49 @@ mod tests { #[test] fn denies_bulk_asset_extensions_case_insensitively() { for p in ["a.ab", "a.XML", "a.Acb", "a.ma2", "a.manifest", "a.PNG"] { - assert!(denied_ext(Path::new(p)), "{p} should be denied"); + assert!(denied_name(Path::new(p)), "{p} should be denied"); + } + } + + #[test] + fn denies_extensionless_files() { + for p in ["asset", "level0", "0123456789abcdef"] { + assert!(denied_name(Path::new(p)), "{p} should be denied"); } } #[test] fn never_denies_what_a_target_can_be_named() { - // Targets are recognised by content, not name — a protected module - // can carry any extension, or none — so names like these must always - // be probed. An allow-list would have skipped them. + // Unknown extensions must still be probed. This keeps the filter a + // narrow deny-list rather than an executable-extension allow-list. for p in [ "app.exe.bak", "managed.dll.bak", "daemon.exe", "GameLib.dll", "global-metadata.dat", - "noextension", "a.so", "a.bin", ] { - assert!(!denied_ext(Path::new(p)), "{p} must still be probed"); + assert!(!denied_name(Path::new(p)), "{p} must still be probed"); } } + #[test] + fn extensionless_targets_require_exhaustive_scan() { + let td = tempfile::tempdir().unwrap(); + let root = td.path(); + let mut blob = vec![0u8; MIN_SIZE as usize + 1]; + blob[..4].copy_from_slice(&0xFAB1_1BAFu32.to_le_bytes()); + std::fs::write(root.join("metadata"), &blob).unwrap(); + + let (_, filtered, _) = find_targets_opts(root, false); + assert!(filtered.is_empty()); + + let (_, exhaustive, _) = find_targets_opts(root, true); + assert_eq!(exhaustive.len(), 1); + } + /// A file below the Crackproof key-table bound is skipped without being /// opened, but a large non-asset file is still probed. #[test] diff --git a/src/ui.rs b/senbei-io/src/ui.rs similarity index 97% rename from src/ui.rs rename to senbei-io/src/ui.rs index d882745..8d49ce7 100644 --- a/src/ui.rs +++ b/senbei-io/src/ui.rs @@ -1,6 +1,6 @@ -use crate::unpacker::{IntegrityReport, Kind}; use indicatif::{ProgressBar, ProgressStyle}; use owo_colors::OwoColorize; +use senbei_pe::{IntegrityReport, Kind}; use std::path::Path; /// Create a progress bar for `n` items. Hidden when `quiet` is true. diff --git a/senbei-metadata/Cargo.toml b/senbei-metadata/Cargo.toml new file mode 100644 index 0000000..c9d49e2 --- /dev/null +++ b/senbei-metadata/Cargo.toml @@ -0,0 +1,6 @@ +[package] +name = "senbei-metadata" +version.workspace = true +edition.workspace = true +license.workspace = true +description = "Unity il2cpp metadata de-obfuscation for Senbei" diff --git a/senbei-metadata/src/lib.rs b/senbei-metadata/src/lib.rs new file mode 100644 index 0000000..18e6fc7 --- /dev/null +++ b/senbei-metadata/src/lib.rs @@ -0,0 +1,5 @@ +//! Unity il2cpp metadata de-obfuscation. + +mod metadata; + +pub use metadata::*; diff --git a/src/metadata.rs b/senbei-metadata/src/metadata.rs similarity index 100% rename from src/metadata.rs rename to senbei-metadata/src/metadata.rs diff --git a/senbei-pe/Cargo.toml b/senbei-pe/Cargo.toml new file mode 100644 index 0000000..8790c8b --- /dev/null +++ b/senbei-pe/Cargo.toml @@ -0,0 +1,10 @@ +[package] +name = "senbei-pe" +version.workspace = true +edition.workspace = true +license.workspace = true +description = "PE detection, unpacking, and validation for Senbei" + +[dependencies] +senbei-crypto.workspace = true +thiserror.workspace = true diff --git a/senbei-pe/src/engine/dll/mod.rs b/senbei-pe/src/engine/dll/mod.rs new file mode 100644 index 0000000..bf70ed1 --- /dev/null +++ b/senbei-pe/src/engine/dll/mod.rs @@ -0,0 +1,3 @@ +mod pipeline; + +pub use pipeline::*; diff --git a/src/unpacker/dll.rs b/senbei-pe/src/engine/dll/pipeline.rs similarity index 88% rename from src/unpacker/dll.rs rename to senbei-pe/src/engine/dll/pipeline.rs index 6bc0894..b14c4b0 100644 --- a/src/unpacker/dll.rs +++ b/senbei-pe/src/engine/dll/pipeline.rs @@ -13,9 +13,12 @@ //! CalculateChecksumWithSizeXor -> primitives::calculate_checksum //! CalculateCrc32 -> crc32::compute (via above) -use super::UnpackError; -use super::bytecode::{Op, OpsLut, generate}; -use super::primitives::{self, *}; +use super::super::{ + BufferOperation, BytecodeStage, DecompressionStage, DescriptorTable, SectionPipeline, + UnpackError, +}; +use senbei_crypto::bytecode::{Op, OpsLut, generate}; +use senbei_crypto::primitives::{self, *}; /// Read a signed 32-bit little-endian value. fn get_i32(d: &[u8], offset: i32) -> i32 { @@ -68,6 +71,7 @@ fn decrypt_data4( key: i32, decomp_params: &[i32; 4], transform: Option<&[Op]>, + stage: DecompressionStage, ) -> Result<(), UnpackError> { let addr = get_i32(d, offset); let size = get_i32(d, offset + 4); @@ -84,19 +88,17 @@ fn decrypt_data4( OpsLut::new(ops).map_region(d, addr as usize, size as usize); } - if size != decompressed_size { - // decompress reports corruption (after partial writes) via its bool; - // surface it instead of shipping a garbage block. - if !decompress( + if size != decompressed_size + && let Err(reason) = primitives::decompress_detailed( d, addr as u32, compressed_addr as u32, decomp_params[1] as u32, size as u32, decompressed_size as u32, - ) { - return Err(UnpackError::DecompressFailed); - } + ) + { + return Err(UnpackError::StageDecompressionFailed { stage, reason }); } Ok(()) } @@ -218,7 +220,11 @@ fn decrypt_and_decompress_data( // Guard: need 16 bytes at section_data_offset in `d` let off = section_data_offset as usize; if off.saturating_add(16) > d.len() { - return Err(UnpackError::OutOfBounds(off)); + return Err(UnpackError::DescriptorOutOfBounds { + table: DescriptorTable::DllSectionBlocks, + offset: off, + image_len: d.len(), + }); } decrypt_data6_shift6(d, section_data_offset, 16); let dest_offset = get_i32(d, section_data_offset); @@ -246,10 +252,10 @@ fn decrypt_and_decompress_data( let lut = OpsLut::new(decrypt_func); let ko0 = decomp_params[0]; let ko2 = decomp_params[2]; - let ks_snap = - primitives::aes_schedule_snapshot(d, ko2 as u32).ok_or(UnpackError::Corrupt)?; + let ks_snap = primitives::aes_schedule_snapshot(d, ko2 as u32) + .ok_or(UnpackError::InvalidAesKeySchedule { offset: ko2 as u32 })?; let tab_snap = primitives::huffman_table_snapshot(d, ko0 as u32) - .ok_or(UnpackError::DecompressFailed)?; + .ok_or(UnpackError::InvalidHuffmanTable { offset: ko0 as u32 })?; let spans: Vec<(usize, usize)> = blocks .iter() .map(|b| { @@ -277,12 +283,15 @@ fn decrypt_and_decompress_data( b.size as u32, b.expected_crc as u32, ) { - return Err(UnpackError::DecompressFailed); + return Err(UnpackError::SectionDecompressionFailed { + pipeline: SectionPipeline::Dll, + block: i, + }); } } Ok(()) }; - super::parallel::parallel_for(d, &spans, 1, do_block)?; + super::super::parallel::parallel_for(d, &spans, 1, do_block)?; } // Zero-fill loop. @@ -292,7 +301,11 @@ fn decrypt_and_decompress_data( // decrypts 16 too, so guard 16 (an 8-byte guard would let // decrypt_data6_shift6 index past the end of a truncated descriptor). if off.saturating_add(16) > d.len() { - return Err(UnpackError::OutOfBounds(off)); + return Err(UnpackError::DescriptorOutOfBounds { + table: DescriptorTable::DllZeroFill, + offset: off, + image_len: d.len(), + }); } decrypt_data6_shift6(d, section_data_offset, 16); let zero_offset = get_i32(d, section_data_offset); @@ -305,7 +318,12 @@ fn decrypt_and_decompress_data( for i in 0..zero_size { let idx = (zero_offset + i) as usize; if idx >= d.len() { - return Err(UnpackError::OutOfBounds(idx)); + return Err(UnpackError::BufferRangeOutOfBounds { + operation: BufferOperation::ZeroFill, + offset: idx, + size: 1, + buffer_len: d.len(), + }); } d[idx] = 0; } @@ -324,12 +342,16 @@ pub fn unpack_dll(input: &[u8]) -> Result, UnpackError> { pub fn unpack_dll_v(input: &[u8], verbose: bool) -> Result, UnpackError> { // Trap any out-of-bounds panic from a truncated/garbled file and report it // as a clean error so the public API stays panic-free. - super::catch_unpack(move || unpack_dll_inner(input, verbose)) + super::super::catch_unpack(move || unpack_dll_inner(input, verbose)) } fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> { - if input.len() < 4096 { - return Err(UnpackError::InputTooShort(input.len())); + const HEADER_LEN: usize = 4128; + if input.len() < HEADER_LEN { + return Err(UnpackError::InputTooShort { + actual: input.len(), + required: HEADER_LEN, + }); } // `file_data` and `original_file_data` both borrow the same protected input. @@ -347,15 +369,18 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> println!(" keys[6] anchor = 0x{:08X}", keys[6] as u32); } - if !super::is_supported_magic(keys[1] as u32) { - return Err(UnpackError::DllUnpack( - "Not a Crackproof protected file (KONN magic mismatch)".into(), - )); + if !super::super::is_supported_magic(keys[1] as u32) { + return Err(UnpackError::HeaderMagicMismatch { + found: keys[1] as u32, + }); } let pe_offset = get_i32(file_data, 60); if pe_offset < 0 || (pe_offset as usize).saturating_add(84) > file_data.len() { - return Err(UnpackError::DllUnpack("implausible PE offset".into())); + return Err(UnpackError::InvalidPeOffset { + offset: i64::from(pe_offset), + input_len: file_data.len(), + }); } // This pipeline is PE32+-only: its header fixups write the data // directories at PE32+ offsets (pe+144..180, pe+136 for the DD blob). On a @@ -363,14 +388,18 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> // structurally plausible but unloadable file. Reject early with a clear // error so `unpack_auto`'s EXE-pipeline fallback handles PE32 DLLs (that // path is PE32-aware — see run_pe32), instead of us mangling them here. - if get_i32(file_data, pe_offset + 24) & 0xFFFF != 0x20B { - return Err(UnpackError::DllUnpack( - "not a PE32+ image (the DLL pipeline handles 64-bit only)".into(), - )); + let optional_magic = get_u16(file_data, (pe_offset + 24) as u32); + if optional_magic != 0x20B { + return Err(UnpackError::UnsupportedDllPeMagic { + found: optional_magic, + }); } let size_of_image = get_i32(file_data, pe_offset + 80); - if size_of_image <= 0 || size_of_image as u64 > super::MAX_IMAGE_SIZE { - return Err(UnpackError::DllUnpack("implausible SizeOfImage".into())); + if size_of_image <= 0 || size_of_image as u64 > super::super::MAX_IMAGE_SIZE { + return Err(UnpackError::InvalidImageSize { + size: i64::from(size_of_image), + max: super::super::MAX_IMAGE_SIZE, + }); } let mut out = vec![0u8; size_of_image as usize]; let base_offset = keys[6] - keys[3] + 0x2000; @@ -438,6 +467,16 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> println!(" checksum1 = 0x{:08X}", checksum1 as u32); println!(" decrypted_addr1 = 0x{:08X}", decrypted_addr1 as u32); } + let primary_end = decrypted_addr1.checked_add(3856); + if decrypted_addr1 < keys[3] + || primary_end.is_none_or(|end| end < 0 || end as usize > out.len()) + { + return Err(UnpackError::InvalidDllPrimaryDescriptor { + address: decrypted_addr1 as u32, + minimum: keys[3] as u32, + image_len: out.len(), + }); + } let import_offset = get_i32(&out, decrypted_addr1 + 3444); let decrypted_addr2_size = get_i32(&out, decrypted_addr1 + 3632); decrypt_data3( @@ -512,6 +551,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> table_val ^ checksum2 ^ (xor_accumulator as i32), &decomp_params, None, + DecompressionStage::DllCodeBlock1, )?; let addr3b = get_i32(&out, decrypted_addr1 + 3728); @@ -527,7 +567,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> let crc_val = { let a = crc_data_addr as usize; let n = crc_data_size as usize; - super::crc32::compute(&out[a..a + n]) as i32 + senbei_crypto::crc32::compute(&out[a..a + n]) as i32 }; let crc_xored = crc_data_size ^ crc_val; let trailing_val = get_i32(&out, crc_data_addr + crc_data_size - 4); @@ -537,6 +577,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> crc_xored ^ (xor_accumulator as i32) ^ trailing_val, &decomp_params, None, + DecompressionStage::DllCodeBlock2, )?; let checksum3 = calculate_checksum(&out, (decrypted_addr1 + 3480) as u32) as i32; @@ -549,6 +590,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> (not_val ^ (xor_key as u32)) as i32, &decomp_params, None, + DecompressionStage::DllCodeBlock3, )?; let addr4 = get_i32(&out, addr4_offset); @@ -574,8 +616,9 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> lfsr_seed_val = lfsr_seed_val.wrapping_add(k); } - let decrypt_func = generate(&out, lfsr as u32) - .ok_or_else(|| UnpackError::DllUnpack("Failed to build decryption expression".into()))?; + let decrypt_func = generate(&out, lfsr as u32).ok_or(UnpackError::BytecodeGenerationFailed( + BytecodeStage::DllPrimaryDecryptor, + ))?; let addr5_offset = decrypted_addr1 + 3840; let addr5 = get_i32(&out, addr5_offset); @@ -585,6 +628,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> lfsr_seed_val ^ xor_key ^ checksum4, &decomp_params, Some(&decrypt_func), + DecompressionStage::DllCodeBlock4, )?; if verbose { println!("[7/9] Decrypting code block 4 (addr5)..."); @@ -603,9 +647,9 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result, UnpackError> let lfsr2 = metadata_offset + 88; decrypt_data6(&mut out, lfsr2 as u32); - let decrypt_func2 = generate(&out, lfsr2 as u32).ok_or_else(|| { - UnpackError::DllUnpack("Failed to build second decryption expression".into()) - })?; + let decrypt_func2 = generate(&out, lfsr2 as u32).ok_or( + UnpackError::BytecodeGenerationFailed(BytecodeStage::DllSectionDecryptor), + )?; let section_image_base = 4095 - get_i32(original_file_data, 4224); let section_data_offset = get_i32(&out, addr5 + 11976); diff --git a/senbei-pe/src/engine/error.rs b/senbei-pe/src/engine/error.rs new file mode 100644 index 0000000..11ef75c --- /dev/null +++ b/senbei-pe/src/engine/error.rs @@ -0,0 +1,253 @@ +pub use senbei_crypto::{BufferOperation, DecompressionFailure}; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DecompressionStage { + ExeStage3, + ExeStage3Secondary, + ExeStage4, + ExeStage5, + Pe32FourthStage, + Pe32FifthStage, + Pe32SeventhStage, + DllCodeBlock1, + DllCodeBlock2, + DllCodeBlock3, + DllCodeBlock4, +} + +impl std::fmt::Display for DecompressionStage { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::ExeStage3 => "EXE stage3", + Self::ExeStage3Secondary => "EXE secondary stage3", + Self::ExeStage4 => "EXE stage4", + Self::ExeStage5 => "EXE stage5", + Self::Pe32FourthStage => "PE32 fourth stage", + Self::Pe32FifthStage => "PE32 fifth stage", + Self::Pe32SeventhStage => "PE32 seventh stage", + Self::DllCodeBlock1 => "DLL code block 1", + Self::DllCodeBlock2 => "DLL code block 2", + Self::DllCodeBlock3 => "DLL code block 3", + Self::DllCodeBlock4 => "DLL code block 4", + }) + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum BytecodeStage { + ExeStage4, + ExeStage5, + Pe32CustomDecryptor, + Pe32FileDecryptor, + DllPrimaryDecryptor, + DllSectionDecryptor, +} + +impl std::fmt::Display for BytecodeStage { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::ExeStage4 => "EXE stage4", + Self::ExeStage5 => "EXE stage5", + Self::Pe32CustomDecryptor => "PE32 custom decryptor", + Self::Pe32FileDecryptor => "PE32 file decryptor", + Self::DllPrimaryDecryptor => "DLL primary decryptor", + Self::DllSectionDecryptor => "DLL section decryptor", + }) + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum SectionPipeline { + ExePe32Plus, + ExePe32, + Dll, +} + +impl std::fmt::Display for SectionPipeline { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::ExePe32Plus => "PE32+ EXE", + Self::ExePe32 => "PE32 EXE", + Self::Dll => "DLL", + }) + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum DescriptorTable { + DllSectionBlocks, + DllZeroFill, +} + +impl std::fmt::Display for DescriptorTable { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.write_str(match self { + Self::DllSectionBlocks => "DLL section-block", + Self::DllZeroFill => "DLL zero-fill", + }) + } +} + +#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)] +#[non_exhaustive] +pub enum UnpackError { + #[error("input too short (need at least {required} bytes, got {actual})")] + InputTooShort { actual: usize, required: usize }, + + #[error("decrypted header magic mismatch (got 0x{found:08X})")] + HeaderMagicMismatch { found: u32 }, + + #[error("anchor field not found — corrupt data or wrong offset")] + AnchorNotFound, + + #[error("stage1 descriptor not found near anchor 0x{anchor:08X}")] + Stage1DescriptorNotFound { anchor: u32 }, + + #[error("stage2 field not found — corrupt data or wrong offset")] + Stage2NotFound, + + #[error("chk_src_start not found — corrupt data or wrong offset")] + ChkSrcStartNotFound, + + #[error("table_start not found — corrupt data or wrong offset")] + TableStartNotFound, + + #[error("{0} bytecode generation failed — corrupt data or wrong offset")] + BytecodeGenerationFailed(BytecodeStage), + + #[error("stage5 marker not found — this build's layout is not supported by this unpacker")] + Stage5MarkerNotFound, + + #[error("not a Crackproof-protected file")] + NotCrackproof, + + #[error("invalid PE header offset {offset} for {input_len}-byte input")] + InvalidPeOffset { offset: i64, input_len: usize }, + + #[error("DLL pipeline requires PE32+ optional-header magic, got 0x{found:04X}")] + UnsupportedDllPeMagic { found: u16 }, + + #[error( + "DLL primary descriptor address 0x{address:08X} is below layout base 0x{minimum:08X} or outside {image_len}-byte image" + )] + InvalidDllPrimaryDescriptor { + address: u32, + minimum: u32, + image_len: usize, + }, + + #[error("invalid SizeOfImage {size}; expected 1..={max}")] + InvalidImageSize { size: i64, max: u64 }, + + #[error( + "{operation} range out of bounds (offset {offset}, size {size}, buffer length {buffer_len})" + )] + BufferRangeOutOfBounds { + operation: BufferOperation, + offset: usize, + size: usize, + buffer_len: usize, + }, + + #[error( + "EXE checksum descriptor at 0x{descriptor:08X} points outside input (offset {offset}, size {size}, input length {image_len})" + )] + ExeChecksumRangeOutOfBounds { + descriptor: u32, + offset: usize, + size: usize, + image_len: usize, + }, + + #[error( + "{table} descriptor out of bounds (offset {offset}, size 16, image length {image_len})" + )] + DescriptorOutOfBounds { + table: DescriptorTable, + offset: usize, + image_len: usize, + }, + + #[error("PE32 tbl not found — corrupt data or wrong offset")] + Pe32TblNotFound, + + #[error("PE32 thirdStage decrypt failed — corrupt data or wrong offset")] + Pe32ThirdStageFailed, + + #[error("PE32 customDecryptor not found in sevenStage")] + Pe32CustomDecryptorNotFound, + + #[error("PE32 eighthStageKey not found")] + Pe32EighthKeyNotFound, + + #[error("PE32 file LFSR not found in eighthStage")] + Pe32FileLfsrNotFound, + + #[error("{stage} decompression failed: {reason}")] + StageDecompressionFailed { + stage: DecompressionStage, + reason: DecompressionFailure, + }, + + #[error("{pipeline} section block {block} decompression failed")] + SectionDecompressionFailed { + pipeline: SectionPipeline, + block: usize, + }, + + #[error("AES key schedule is outside the image at offset {offset}")] + InvalidAesKeySchedule { offset: u32 }, + + #[error("Huffman table is outside the image at offset {offset}")] + InvalidHuffmanTable { offset: u32 }, + + #[error("DLL pipeline failed: {dll}; EXE fallback failed: {exe}")] + PipelineFallbackFailed { + dll: Box, + exe: Box, + }, + + #[error( + "PE32 second-stage range is invalid (offset {offset}, size {size}, image length {image_len})" + )] + Pe32SecondStageRangeInvalid { + offset: u32, + size: u32, + image_len: usize, + }, + + #[error("PE32 relocation-data descriptor not found")] + Pe32RelocationDataNotFound, + + #[error("file decryptor candidate failed structural validation")] + FileDecryptorValidationFailed, + + #[error("PE32 memory image could not be rebuilt as a file-layout PE")] + Pe32OutputLayoutInvalid, + + #[error("internal panic at {file}:{line}:{column}: {message}")] + InternalPanic { + message: String, + file: String, + line: u32, + column: u32, + }, +} + +impl From for UnpackError { + fn from(error: senbei_crypto::Error) -> Self { + match error { + senbei_crypto::Error::BufferRangeOutOfBounds { + operation, + offset, + size, + buffer_len, + } => Self::BufferRangeOutOfBounds { + operation, + offset, + size, + buffer_len, + }, + } + } +} diff --git a/senbei-pe/src/engine/exe/mod.rs b/senbei-pe/src/engine/exe/mod.rs new file mode 100644 index 0000000..bf70ed1 --- /dev/null +++ b/senbei-pe/src/engine/exe/mod.rs @@ -0,0 +1,3 @@ +mod pipeline; + +pub use pipeline::*; diff --git a/src/unpacker/exe.rs b/senbei-pe/src/engine/exe/pipeline.rs similarity index 60% rename from src/unpacker/exe.rs rename to senbei-pe/src/engine/exe/pipeline.rs index 0c4def1..a7ec023 100644 --- a/src/unpacker/exe.rs +++ b/senbei-pe/src/engine/exe/pipeline.rs @@ -1,69 +1,11 @@ -use super::bytecode::{Op, OpsLut, generate}; -use super::primitives; -use super::primitives::*; +use senbei_crypto::bytecode::{Op, OpsLut, generate}; +use senbei_crypto::primitives; +use senbei_crypto::primitives::*; -#[derive(Debug, thiserror::Error)] -pub enum UnpackError { - #[error("input too short for header (need at least 4096 bytes, got {0})")] - InputTooShort(usize), +use super::super::error::*; +use super::super::layout::{self, *}; - #[error("info[1] mismatch — corrupt data or wrong offset")] - HeaderMismatch, - - #[error("anchor field not found — corrupt data or wrong offset")] - AnchorNotFound, - - #[error("stage2 field not found — corrupt data or wrong offset")] - Stage2NotFound, - - #[error("chk_src_start not found — corrupt data or wrong offset")] - ChkSrcStartNotFound, - - #[error("table_start not found — corrupt data or wrong offset")] - TableStartNotFound, - - #[error("stage4 bytecode generation failed — corrupt data or wrong offset")] - BytecodeGenFailed, - - #[error("stage5 marker not found — this build's layout is not supported by this unpacker")] - Stage5MarkerNotFound, - - #[error("stage5 bytecode generation failed — corrupt data or wrong offset")] - Stage5BytecodeGenFailed, - - #[error("DLL unpack failed: {0}")] - DllUnpack(String), - - #[error("not a Crackproof-protected file")] - NotCrackproof, - - #[error("out-of-bounds access at offset {0}")] - OutOfBounds(usize), - - #[error("PE32 tbl not found — corrupt data or wrong offset")] - Pe32TblNotFound, - - #[error("PE32 thirdStage decrypt failed — corrupt data or wrong offset")] - Pe32ThirdStageFailed, - - #[error("PE32 customDecryptor not found in sevenStage")] - Pe32CustomDecryptorNotFound, - - #[error("PE32 stage bytecode generation failed")] - Pe32BytecodeGenFailed, - - #[error("PE32 eighthStageKey not found")] - Pe32EighthKeyNotFound, - - #[error("PE32 file LFSR not found in eighthStage")] - Pe32FileLfsrNotFound, - - #[error("decompression failed — corrupt data or wrong offset")] - DecompressFailed, - - #[error("input is corrupt or not a supported Crackproof layout")] - Corrupt, -} +mod pe32; pub fn unpack(input: &[u8]) -> Result, UnpackError> { unpack_v(input, false) @@ -93,8 +35,12 @@ fn prot_rva_to_off(file_data: &[u8], pe_header: u32, rva: u32) -> Option { } pub fn unpack_v(input: &[u8], verbose: bool) -> Result, UnpackError> { - if input.len() < 4096 { - return Err(UnpackError::InputTooShort(input.len())); + const HEADER_LEN: usize = 4128; + if input.len() < HEADER_LEN { + return Err(UnpackError::InputTooShort { + actual: input.len(), + required: HEADER_LEN, + }); } // The pipeline chases offsets read out of the decrypted image; on a // truncated/garbled-but-detected file those run out of bounds. Trap any @@ -104,7 +50,7 @@ pub fn unpack_v(input: &[u8], verbose: bool) -> Result, UnpackError> { // separate `decompressed` buffer), so the unpacker borrows it directly — no // owned copy is made here. catch_unwind uses AssertUnwindSafe, so a borrowing // (non-'static) closure is fine. - super::catch_unpack(move || Unpacker::run(input, verbose)) + super::super::catch_unpack(move || Unpacker::run(input, verbose)) } struct Unpacker<'a> { @@ -127,8 +73,22 @@ impl<'a> Unpacker<'a> { } // Strategy (a): delegate to primitives::calculate_checksum2 - fn calculate_checksum2(&self, pos: u32, start: u32) -> u32 { - primitives::calculate_checksum2(&self.decompressed, self.file_data, pos, start) + fn calculate_checksum2(&self, pos: u32, start: u32) -> Result { + primitives::calculate_checksum2(&self.decompressed, self.file_data, pos, start).map_err( + |error| match error { + senbei_crypto::Error::BufferRangeOutOfBounds { + offset, + size, + buffer_len, + .. + } => UnpackError::ExeChecksumRangeOutOfBounds { + descriptor: pos, + offset, + size, + image_len: buffer_len, + }, + }, + ) } // Strategy (a): delegate to primitives::decrypt_data1 @@ -228,8 +188,13 @@ impl<'a> Unpacker<'a> { } // Strategy (a): delegate to primitives::decrypt_and_decompress_data - fn decrypt_and_decompress_data(&mut self, pos: u32, key: u32, custom: Option<&[Op]>) -> bool { - primitives::decrypt_and_decompress_data( + fn decrypt_and_decompress_data( + &mut self, + pos: u32, + key: u32, + custom: Option<&[Op]>, + ) -> Result<(), DecompressionFailure> { + primitives::decrypt_and_decompress_data_detailed( &mut self.decompressed, pos, key, @@ -375,14 +340,26 @@ impl<'a> Unpacker<'a> { println!(" info[6] end_mark = 0x{:08X}", u.info[6]); println!(" info[7] = 0x{:08X}", u.info[7]); } - if !super::is_supported_magic(u.info[1]) { - return Err(UnpackError::HeaderMismatch); + if !super::super::is_supported_magic(u.info[1]) { + return Err(UnpackError::HeaderMagicMismatch { found: u.info[1] }); } let pe_off = get_u32(u.file_data, 60); + if (pe_off as usize) + .checked_add(84) + .is_none_or(|end| end > u.file_data.len()) + { + return Err(UnpackError::InvalidPeOffset { + offset: i64::from(pe_off), + input_len: u.file_data.len(), + }); + } let size_of_image = get_u32(u.file_data, pe_off.wrapping_add(80)); - if size_of_image == 0 || size_of_image as u64 > super::MAX_IMAGE_SIZE { - return Err(UnpackError::Corrupt); + if size_of_image == 0 || size_of_image as u64 > super::super::MAX_IMAGE_SIZE { + return Err(UnpackError::InvalidImageSize { + size: i64::from(size_of_image), + max: super::super::MAX_IMAGE_SIZE, + }); } u.decompressed = vec![0u8; size_of_image as usize]; u.decrypt_size = u.info[6].wrapping_sub(u.info[3]).wrapping_add(8192); @@ -445,26 +422,34 @@ impl<'a> Unpacker<'a> { } }; - // Detect config-block layout version. Newer Crackproof builds (observed - // across several EXE families) shift every anchor-relative field from - // offset 40 onward by +8 bytes. The config-version stamp sits at - // anchor+104 in the old layout and anchor+112 in the new one. Across - // the whole corpus the stamp's top nibble is always 0x4 (top byte 0x40 - // or 0x44), whereas the +8 layout's anchor+104 holds an inserted small - // count (top nibble 0), so the stamp position is a reliable layout - // discriminator. - let stamp_at = |off: u32| -> bool { - (anchor + off + 4) as usize <= u.decompressed.len() - && (get_u32(&u.decompressed, anchor + off) >> 28) == 0x4 + // Layouts shift the anchor-relative fields by either zero or eight + // bytes. The nearby version-like word is not stable across all build + // families, so validate the stage1 (RVA, length) descriptor itself. + let descriptor_is_valid = |extra: u32| -> bool { + let pos = anchor.wrapping_add(120 + extra); + let Some(end) = (pos as usize).checked_add(8) else { + return false; + }; + if end > u.decompressed.len() { + return false; + } + let base = get_u32(&u.decompressed, pos); + let length = get_u32(&u.decompressed, pos.wrapping_add(4)); + base >= u.info[3] + && length >= 16 + && (base as usize) + .checked_add(length as usize) + .is_some_and(|stage_end| stage_end <= u.decompressed.len()) }; - let magic_off: u32 = if stamp_at(104) { - 104 - } else if stamp_at(112) { - 112 - } else { - 104 - }; - let anchor_extra: u32 = magic_off - 104; + let anchor_extra = [0u32, 8] + .into_iter() + .find(|&extra| descriptor_is_valid(extra)) + .ok_or(UnpackError::Stage1DescriptorNotFound { anchor })?; + + if verbose { + println!(" anchor = 0x{anchor:08X}"); + println!(" anchor layout offset = +0x{anchor_extra:X}"); + } let p1 = get_u32(&u.decompressed, anchor.wrapping_add(8)); let p2 = get_u32(&u.decompressed, anchor.wrapping_add(4)); @@ -492,11 +477,22 @@ impl<'a> Unpacker<'a> { let v_at = anchor.wrapping_add(20); let v = get_u32(&u.decompressed, v_at); let tgt = anchor.wrapping_add(120 + anchor_extra); + let stage1_descriptor = [ + get_u32(&u.decompressed, tgt), + get_u32(&u.decompressed, tgt.wrapping_add(4)), + ]; u.decrypt_data3(tgt, xor_acc ^ chk1 ^ v, 21); let stage1 = get_u32(&u.decompressed, tgt); + let stage1_len = get_u32(&u.decompressed, tgt.wrapping_add(4)); if verbose { println!("[3/9] Locating config layout..."); println!(" stage1 = 0x{:08X}", stage1); + println!(" stage1_len = 0x{stage1_len:08X}"); + println!( + " stage1 descriptor = [0x{:08X}, 0x{:08X}]", + stage1_descriptor[0], stage1_descriptor[1] + ); + println!(" stage1 key = xor 0x{xor_acc:08X} ^ chk 0x{chk1:08X} ^ val 0x{v:08X}"); } // Field offsets inside stage1 vary between Crackproof versions. Locate @@ -505,7 +501,6 @@ impl<'a> Unpacker<'a> { // derive every other field as fixed offsets from there. Observed // stage2_off: 3632 (older EXE builds), 3616 (another old-layout build), // 3624 (managed-assembly builds). - let stage1_len = get_u32(&u.decompressed, tgt.wrapping_add(4)); let info3 = u.info[3]; let info5 = u.info[5]; // Use the full info[3]..info[3]+info[5] range: stage entries may live in @@ -577,6 +572,9 @@ impl<'a> Unpacker<'a> { if verbose { println!("[4/9] Decrypting stage2..."); println!(" stage2 = 0x{:08X}", stage2); + println!(" stage2_off = 0x{stage2_off:04X}"); + println!(" checksum table = stage1+0x{chk_src_start:04X}"); + println!(" stage2 key = 0x{key2:08X}"); } // The stage2 head/walk2 tables shift between Crackproof versions. The 4-entry @@ -608,6 +606,20 @@ impl<'a> Unpacker<'a> { }; let head_off = table_start.wrapping_add(32); let walk2_off = head_off.wrapping_sub(88); + if verbose { + println!(" operation table = stage2+0x{table_start:04X}"); + println!(" head/walk = +0x{head_off:04X}/+0x{walk2_off:04X}"); + for index in 0..2u32 { + let entry = stage2.wrapping_add(head_off + index * 16); + println!( + " operation[{index}] = [0x{:08X}, 0x{:08X}, 0x{:08X}, 0x{:08X}]", + get_u32(&u.decompressed, entry), + get_u32(&u.decompressed, entry.wrapping_add(4)), + get_u32(&u.decompressed, entry.wrapping_add(8)), + get_u32(&u.decompressed, entry.wrapping_add(12)), + ); + } + } let mut head = stage2.wrapping_add(head_off); for _iter in 0..2 { @@ -650,28 +662,110 @@ impl<'a> Unpacker<'a> { } walk2 = walk2.wrapping_add(32); } + if verbose { + println!( + " key offsets = [0x{:08X}, 0x{:08X}, 0x{:08X}, 0x{:08X}]", + u.key_offsets[0], u.key_offsets[1], u.key_offsets[2], u.key_offsets[3] + ); + } let chk2 = u.calculate_checksum(anchor.wrapping_add(48 + anchor_extra)); let accum_at = stage1.wrapping_add(chk_src_start.wrapping_sub(16)); - let mut accum = get_u32(&u.decompressed, accum_at); - for l in 0..4u32 { + let accum_seed = get_u32(&u.decompressed, accum_at); + let mut running_accum = accum_seed; + let mut accum_candidates = vec![(0u32, accum_seed)]; + for l in 0..8u32 { let bound = (l + 1).wrapping_mul(25) << 2; let mut i: u32 = 1; while i <= bound { - accum = accum.wrapping_add(i); + running_accum = running_accum.wrapping_add(i); i = i.wrapping_add(1); } + accum_candidates.push((l + 1, running_accum)); } + let accum = accum_candidates[4].1; let at1 = stage1.wrapping_add(stage2_off.wrapping_add(88)); let stage3_field = get_u32(&u.decompressed, at1); + let stage3_slen = get_u32(&u.decompressed, at1.wrapping_add(4)); + let stage3_dest = get_u32(&u.decompressed, at1.wrapping_add(8)); let stage3_dlen = get_u32(&u.decompressed, at1.wrapping_add(12)); if verbose { println!("[5/9] Decrypting stages 3-5..."); println!(" stage3 = 0x{:08X}", stage3_field); + println!( + " stage3 descriptor = [0x{stage3_field:08X}, 0x{stage3_slen:08X}, 0x{stage3_dest:08X}, 0x{stage3_dlen:08X}]" + ); + println!(" stage3 key = xor 0x{xor_acc:08X} ^ chk 0x{chk2:08X} ^ val 0x{accum:08X}"); + println!(" stage3 accum seed = 0x{accum_seed:08X}"); } - if !u.decrypt_and_decompress_data(at1, xor_acc ^ chk2 ^ accum, None) { - return Err(UnpackError::DecompressFailed); + let stage3_key = xor_acc ^ chk2 ^ accum; + let stage3_source_end = (stage3_field as usize) + .checked_add(stage3_slen as usize) + .filter(|&end| end <= u.decompressed.len()) + .ok_or(UnpackError::BufferRangeOutOfBounds { + operation: BufferOperation::Read, + offset: stage3_field as usize, + size: stage3_slen as usize, + buffer_len: u.decompressed.len(), + })?; + let stage3_dest_end = (stage3_dest as usize) + .checked_add(stage3_dlen as usize) + .filter(|&end| end <= u.decompressed.len()) + .ok_or(UnpackError::BufferRangeOutOfBounds { + operation: BufferOperation::CopyDestination, + offset: stage3_dest as usize, + size: stage3_dlen as usize, + buffer_len: u.decompressed.len(), + })?; + const MAX_STAGE3_TRIAL_BYTES: usize = 16 * 1024 * 1024; + let trial_size = (stage3_slen as usize).checked_add(stage3_dlen as usize); + let stage3_backups = trial_size + .filter(|&size| size <= MAX_STAGE3_TRIAL_BYTES) + .map(|_| { + ( + u.decompressed[stage3_field as usize..stage3_source_end].to_vec(), + u.decompressed[stage3_dest as usize..stage3_dest_end].to_vec(), + ) + }); + let default_result = u.decrypt_and_decompress_data(at1, stage3_key, None); + if let Err(reason) = default_result { + let Some((source_backup, dest_backup)) = stage3_backups else { + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::ExeStage3, + reason, + }); + }; + let restore_stage3 = |data: &mut [u8]| { + data[stage3_field as usize..stage3_source_end].copy_from_slice(&source_backup); + data[stage3_dest as usize..stage3_dest_end].copy_from_slice(&dest_backup); + }; + let mut selected = None; + for (rounds, candidate_accum) in &accum_candidates { + if *rounds == 4 { + continue; + } + restore_stage3(&mut u.decompressed); + let candidate_key = xor_acc ^ chk2 ^ candidate_accum; + let result = u.decrypt_and_decompress_data(at1, candidate_key, None); + if result.is_ok() + && find_v4_offset(&u.decompressed, stage3_field, stage3_dlen).is_some() + { + selected = Some(*rounds); + break; + } + } + if let Some(rounds) = selected { + if verbose { + println!(" selected stage3 accumulator rounds = {rounds}"); + } + } else { + restore_stage3(&mut u.decompressed); + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::ExeStage3, + reason, + }); + } } let at2 = stage1.wrapping_add(stage2_off.wrapping_add(104)); @@ -688,8 +782,11 @@ impl<'a> Unpacker<'a> { let v4 = find_v4_offset(&u.decompressed, stage3_field, stage3_dlen) .unwrap_or_else(|| stage3_field.wrapping_add(4692)); let v4_val = get_u32(&u.decompressed, v4); - if !u.decrypt_and_decompress_data(at2, xor_acc ^ chk3 ^ v4_val, None) { - return Err(UnpackError::DecompressFailed); + if let Err(reason) = u.decrypt_and_decompress_data(at2, xor_acc ^ chk3 ^ v4_val, None) { + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::ExeStage3Secondary, + reason, + }); } let chk4 = u.calculate_checksum(stage1.wrapping_add(chk_src_start.wrapping_add(16))); @@ -707,8 +804,11 @@ impl<'a> Unpacker<'a> { if verbose { println!(" stage4 = 0x{:08X}", stage4_field); } - if !u.decrypt_and_decompress_data(at3, xor_acc ^ chk4 ^ v5_val, None) { - return Err(UnpackError::DecompressFailed); + if let Err(reason) = u.decrypt_and_decompress_data(at3, xor_acc ^ chk4 ^ v5_val, None) { + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::ExeStage4, + reason, + }); } // Inside stage4, two locations vary by build: @@ -745,33 +845,82 @@ impl<'a> Unpacker<'a> { // (i.e., the 4 bytes immediately after the last instance of that pattern). let v6 = find_v_after_pad(&u.decompressed, stage4_field, stage4_dlen) .unwrap_or_else(|| idb_pos.wrapping_sub(24)); - let mut accum2 = get_u32(&u.decompressed, v6); - for m in 0..3u32 { + let accum2_seed = get_u32(&u.decompressed, v6); + let mut accum2 = accum2_seed; + let mut accum2_candidates = vec![(0u32, accum2_seed)]; + for m in 0..8u32 { let bound = (m + 1).wrapping_mul(25) << 2; let mut i: u32 = 1; while i <= bound { accum2 = accum2.wrapping_add(i); i = i.wrapping_add(1); } + accum2_candidates.push((m + 1, accum2)); } let ops1 = match generate(&u.decompressed, data_offset) { Some(v) => v, None => { - return Err(UnpackError::BytecodeGenFailed); + return Err(UnpackError::BytecodeGenerationFailed( + BytecodeStage::ExeStage4, + )); } }; let at4 = stage1.wrapping_add(stage2_off.wrapping_add(216)); let stage5_field = get_u32(&u.decompressed, at4); - // at4 is a (src, src_len, dest, dest_len) quad; only src and dest_len - // are needed here, the other two are consumed by the decrypt below. + let stage5_slen = get_u32(&u.decompressed, at4.wrapping_add(4)); + let stage5_dest = get_u32(&u.decompressed, at4.wrapping_add(8)); let stage5_dlen = get_u32(&u.decompressed, at4.wrapping_add(12)); if verbose { println!(" stage5 = 0x{:08X}", stage5_field); + println!( + " stage5 descriptor = [0x{stage5_field:08X}, 0x{stage5_slen:08X}, 0x{stage5_dest:08X}, 0x{stage5_dlen:08X}]" + ); + println!(" stage5 bytecode = 0x{data_offset:08X}"); + println!(" stage5 accumulator seed = 0x{accum2_seed:08X} at 0x{v6:08X}"); } - if !u.decrypt_and_decompress_data(at4, xor_acc ^ chk4 ^ chk5 ^ accum2, Some(&ops1)) { - return Err(UnpackError::DecompressFailed); + let stage5_lo = stage5_field.min(stage5_dest) as usize; + let stage5_hi = stage5_field + .checked_add(stage5_slen) + .zip(stage5_dest.checked_add(stage5_dlen)) + .map(|(source_end, dest_end)| source_end.max(dest_end) as usize) + .filter(|&end| stage5_lo <= end && end <= u.decompressed.len()) + .ok_or(UnpackError::BufferRangeOutOfBounds { + operation: BufferOperation::Read, + offset: stage5_lo, + size: stage5_slen.max(stage5_dlen) as usize, + buffer_len: u.decompressed.len(), + })?; + let stage5_backup = u.decompressed[stage5_lo..stage5_hi].to_vec(); + let mut first_failure = None; + let mut selected_rounds = None; + for rounds in std::iter::once(3u32).chain((0..=8).filter(|&rounds| rounds != 3)) { + u.decompressed[stage5_lo..stage5_hi].copy_from_slice(&stage5_backup); + let candidate_accum = accum2_candidates[rounds as usize].1; + match u.decrypt_and_decompress_data( + at4, + xor_acc ^ chk4 ^ chk5 ^ candidate_accum, + Some(&ops1), + ) { + Ok(()) => { + selected_rounds = Some(rounds); + break; + } + Err(reason) => { + first_failure.get_or_insert(reason); + } + } + } + let Some(selected_rounds) = selected_rounds else { + u.decompressed[stage5_lo..stage5_hi].copy_from_slice(&stage5_backup); + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::ExeStage5, + reason: first_failure.expect("at least one Stage5 candidate was tried"), + }); + }; + if verbose { + println!(" selected stage5 accumulator rounds = {selected_rounds}"); } // Inside stage5, the loader stores a table of (ptr, size) pairs at a @@ -843,7 +992,7 @@ impl<'a> Unpacker<'a> { // structurally. fileCS stays at bc2_off-0x58 as in the old layout. if new_layout { let compress_data_offset = (!get_u32(u.file_data, 0x1080)).wrapping_add(0x1000); - let slots = primitives::discover_eighth_slots( + let slots = layout::discover_eighth_slots( &u.decompressed, stage5_field, stage5_dlen, @@ -883,7 +1032,19 @@ impl<'a> Unpacker<'a> { let n = get_u32(&u.decompressed, p.wrapping_add(4)); walk3 = walk3.wrapping_add(16); if n != 0 { - chain_crc = u.calculate_checksum2(walk3.wrapping_sub(16), chain_crc); + match u.calculate_checksum2(walk3.wrapping_sub(16), chain_crc) { + Ok(next) => chain_crc = next, + Err(UnpackError::ExeChecksumRangeOutOfBounds { .. }) => { + if verbose { + println!( + " checksum chain terminates at 0x{:08X}: descriptor payload is outside protected input", + walk3.wrapping_sub(16) + ); + } + break; + } + Err(error) => return Err(error), + } } if get_u32(&u.decompressed, walk3.wrapping_sub(16).wrapping_add(4)) == 0 { break; @@ -921,7 +1082,9 @@ impl<'a> Unpacker<'a> { let ops2 = match generate(&u.decompressed, data_offset2) { Some(v) => v, None => { - return Err(UnpackError::Stage5BytecodeGenFailed); + return Err(UnpackError::BytecodeGenerationFailed( + BytecodeStage::ExeStage5, + )); } }; // The new layout picked its file decryptor by distance (no marker, no @@ -930,7 +1093,7 @@ impl<'a> Unpacker<'a> { // garbling into the output without any error (see the validator). let rebase = (!get_u32(u.file_data, 4224)).wrapping_add(4096); if new_layout && !u.new_layout_file_ops_validate(walk4_slot, &ops2, rebase) { - return Err(UnpackError::DecompressFailed); + return Err(UnpackError::FileDecryptorValidationFailed); } let at6 = walk4_slot; @@ -980,9 +1143,9 @@ impl<'a> Unpacker<'a> { // Snapshot the shared tables before the fan-out: workers get // disjoint span slices, not the whole buffer. let ks_snap = primitives::aes_schedule_snapshot(&u.decompressed, ko[2]) - .ok_or(UnpackError::Corrupt)?; + .ok_or(UnpackError::InvalidAesKeySchedule { offset: ko[2] })?; let tab_snap = primitives::huffman_table_snapshot(&u.decompressed, ko[0]) - .ok_or(UnpackError::DecompressFailed)?; + .ok_or(UnpackError::InvalidHuffmanTable { offset: ko[0] })?; let spans: Vec<(usize, usize)> = blocks .iter() .map(|b| { @@ -1009,12 +1172,15 @@ impl<'a> Unpacker<'a> { b.len, b.plain_len, ) { - return Err(UnpackError::DecompressFailed); + return Err(UnpackError::SectionDecompressionFailed { + pipeline: SectionPipeline::ExePe32Plus, + block: i, + }); } } Ok(()) }; - super::parallel::parallel_for(&mut u.decompressed, &spans, 1, do_block)?; + super::super::parallel::parallel_for(&mut u.decompressed, &spans, 1, do_block)?; } loop { u.decrypt_data5(walk4, 16); @@ -1211,9 +1377,7 @@ impl<'a> Unpacker<'a> { let dd8_shift: u32 = match std::env::var("DD8_SHIFT").ok().and_then(|s| s.parse().ok()) { Some(s) => s, - None => { - primitives::select_dd8_shift(&u.decompressed, text_va, text_size, u.info[3]) - } + None => layout::select_dd8_shift(&u.decompressed, text_va, text_size, u.info[3]), }; if dd8_shift != 99 { let mut page = text_va >> 12; @@ -1339,7 +1503,7 @@ impl<'a> Unpacker<'a> { let shift = match std::env::var("DD8_SHIFT").ok().and_then(|s| s.parse().ok()) { Some(s) => s, None => { - primitives::select_dd8_shift(&u.decompressed, text_va, text_size, u.info[3]) + layout::select_dd8_shift(&u.decompressed, text_va, text_size, u.info[3]) } }; if shift != 99 { @@ -1541,7 +1705,7 @@ impl<'a> Unpacker<'a> { /// whose fileCS pointer sits just past `info[3]`), not by content. A /// coincidental LFSR-shaped block at a shorter distance would decode to a /// wrong `ops2` translate and silently garble every section block — raw - /// blocks never hit `DecompressFailed`, so the failure would ship as a + /// raw blocks never enter the decompressor, so the failure would ship as a /// plausible but wrong image. Replay the first *compressed* block's full /// transform (raw copy, AES, translate, decompress) on a snapshot and /// require decompression to succeed; restore the region afterwards. @@ -1582,7 +1746,8 @@ impl<'a> Unpacker<'a> { self.aes_decrypt(dst, len, self.key_offsets[2]); for k in 0..len { let idx = (dst + k) as usize; - self.decompressed[idx] = super::bytecode::apply(ops, self.decompressed[idx]); + self.decompressed[idx] = + senbei_crypto::bytecode::apply(ops, self.decompressed[idx]); } let ok = primitives::decompress( &mut self.decompressed, @@ -1674,7 +1839,8 @@ impl<'a> Unpacker<'a> { self.aes_decrypt(dst2, s_sz2, self.key_offsets[2]); for k in 0..s_sz2 { let idx = (dst2 + k) as usize; - self.decompressed[idx] = super::bytecode::apply(&ops, self.decompressed[idx]); + self.decompressed[idx] = + senbei_crypto::bytecode::apply(&ops, self.decompressed[idx]); } let ok = primitives::decompress( &mut self.decompressed, @@ -1835,1044 +2001,42 @@ impl<'a> Unpacker<'a> { self.decompressed[d..d + 24].copy_from_slice(&self.file_data[src..src + 24]); true } +} - /// PE32 (32-bit) unpack pipeline. The shared Stage 1/2 setup (info decrypt, - /// payload decrypt, raw copy, header restore) has already run in `run()` - /// before dispatch; this takes over from "Locating shell offsets". - fn run_pe32(&mut self, pe_off: u32, verbose: bool) -> Result, UnpackError> { - let info = self.info; - let info3 = info[3]; +#[cfg(test)] +mod error_tests { + use super::*; - // advance_key: replays the packer's per-iteration key walk. - let advance_key = |mut key: u32, iterations: u32| -> u32 { - for m in 0..iterations { - let bound = (m + 1).wrapping_mul(25) << 2; - let mut n: u32 = 1; - while n <= bound { - key = key.wrapping_add(n); - n += 1; - } + #[test] + fn short_input_reports_actual_and_required_lengths() { + let error = unpack(&[0; 4096]).expect_err("header must be rejected"); + assert_eq!( + error, + UnpackError::InputTooShort { + actual: 4096, + required: 4128, } - key + ); + } + + #[test] + fn structured_errors_include_stage_and_block_context() { + let stage = UnpackError::StageDecompressionFailed { + stage: DecompressionStage::ExeStage4, + reason: DecompressionFailure::NoProgress, }; - - // ---- Locate tbl in shell ---- - let tbl = primitives::find_tbl_pe32(&self.decompressed, &info) - .ok_or(UnpackError::Pe32TblNotFound)?; - if verbose { - println!("[3/9] Locating config layout (PE32)..."); - println!(" tbl = 0x{:X}", tbl); - } - - // ---- PE header restore ---- - let val_bc = get_u32(&self.decompressed, tbl.wrapping_add(0xBC)); - let val_c8 = get_u32(&self.decompressed, tbl.wrapping_add(0xC8)); - let val_cc = get_u32(&self.decompressed, tbl.wrapping_add(0xCC)); - write_u32(&mut self.decompressed, pe_off.wrapping_add(0x80), val_bc); - write_u32(&mut self.decompressed, pe_off.wrapping_add(0x88), val_c8); - write_u32(&mut self.decompressed, pe_off.wrapping_add(0x8C), val_cc); - write_u32(&mut self.decompressed, pe_off.wrapping_add(0xB0), 0); - write_u32(&mut self.decompressed, pe_off.wrapping_add(0xB4), 0); - - // ---- Header-independent checksum inputs ---- - let first_stage_cs = self.calculate_checksum(tbl.wrapping_add(0xA8)); - let second_stage_key = get_u32(&self.decompressed, tbl.wrapping_add(0x40)); - - // ---- Stage 3: SecondStage ---- - // - // ss_key = headerChecksum ^ firstStageCS ^ secondStageKey, where the - // header checksum (a XOR of crc32(region)^size over the sub-regions at - // tbl+0x58) is taken over the *original* pre-pack PE header. For EXEs the - // import/resource restore above reconstructs that header exactly. Native - // DLLs additionally carry a packer-added BaseReloc data-directory entry - // (dir 5) that was absent from the checksummed original, so the header - // checksum only matches once that entry is treated as zero. EXEs have no - // dir-5 entry, so zeroing it is a no-op for them. - // - // Rather than branch on EXE-vs-DLL, try the header as-is and, on failure, - // with the BaseReloc entry zeroed; keep whichever ss_key decrypts a - // SecondStage whose ThirdStage (off,size) pair lands inside the image. - // This uses the same shift/key trial-and-validate the later stages - // already use, and keeps EXE output byte-identical (the as-is variant - // wins first). - let ss_pair = tbl.wrapping_add(0x98); - let ss = get_u32(&self.decompressed, ss_pair); - let ss_size = get_u32(&self.decompressed, ss_pair.wrapping_add(4)); - let ss_shift = ss_size.wrapping_sub(0xBC0); - // Back up the SecondStage ciphertext so a failed trial can be retried. - let ss_lo = ss as usize; - let ss_hi = ss_lo.wrapping_add(ss_size as usize); - if ss_hi < ss_lo || ss_hi > self.decompressed.len() { - return Err(UnpackError::Corrupt); - } - let ss_ct: Vec = self.decompressed[ss_lo..ss_hi].to_vec(); - // PE32 data dir 5 (BaseReloc) = optional_header(pe+24) + 0x60 + 5*8 = pe+0xA0. - let reloc_dir = pe_off.wrapping_add(0xA0); - let len = self.decompressed.len() as u64; - let pair_off = 0xB8Cu32.wrapping_add(ss_shift); - let mut found = false; - // Holds the winning variant's header checksum; the later stages - // (Forth/Fifth/Seven/Eighth) reuse it as a key component. - let mut header_checksum: u32 = 0; - for zero_reloc in [false, true] { - if zero_reloc { - write_u32(&mut self.decompressed, reloc_dir, 0); - write_u32(&mut self.decompressed, reloc_dir.wrapping_add(4), 0); - } - let mut hcs_addr = tbl.wrapping_add(0x58); - header_checksum = 0; - while get_u32(&self.decompressed, hcs_addr.wrapping_add(4)) != 0 { - header_checksum ^= self.calculate_checksum(hcs_addr); - hcs_addr = hcs_addr.wrapping_add(8); - } - let ss_key = header_checksum ^ first_stage_cs ^ second_stage_key; - self.decompressed[ss_lo..ss_hi].copy_from_slice(&ss_ct); - self.decrypt_data3(ss_pair, ss_key, 21); - // Validate: the ThirdStage (off,size) pair must reference the image. - let pair = ss.wrapping_add(pair_off); - let off = get_u32(&self.decompressed, pair) as u64; - let sz = get_u32(&self.decompressed, pair.wrapping_add(4)) as u64; - if off > 0x1000 && off < len && sz >= 4 && off.saturating_add(sz) <= len { - found = true; - break; - } - } - if !found { - return Err(UnpackError::Corrupt); - } - if verbose { - println!( - " ss = 0x{:08X}, size = 0x{:X}, shift = 0x{:X}", - ss, ss_size, ss_shift - ); - } - - // ---- PE32 fixed offsets ---- - let third_key_off = 0x968u32.wrapping_add(ss_shift); - let forth_key_off = 0x964u32.wrapping_add(ss_shift); - let cs_base_off = 0x96Cu32.wrapping_add(ss_shift); - let dp_base_off = 0xA9Cu32.wrapping_add(ss_shift); - - // ---- Stage 4: ThirdStage (brute-force the rotate shift) ---- - let third_pair_off = 0xB8Cu32.wrapping_add(ss_shift); - let key = get_u32(&self.decompressed, ss.wrapping_add(third_key_off)); - let pair_addr = ss.wrapping_add(third_pair_off); - let ts_addr = get_u32(&self.decompressed, pair_addr); - let ts_size_raw = get_u32(&self.decompressed, pair_addr.wrapping_add(4)); - let backup: Vec = - self.decompressed[ts_addr as usize..(ts_addr + ts_size_raw) as usize].to_vec(); - let mut info_table: Option = None; - let mut keys_addr: u32 = 0; - let mut ts: u32 = 0; - for &shift in &[19u32, 21, 17, 23, 15, 25, 13, 11] { - self.decompressed[ts_addr as usize..(ts_addr + ts_size_raw) as usize] - .copy_from_slice(&backup); - write_u32(&mut self.decompressed, pair_addr, ts_addr); - write_u32( - &mut self.decompressed, - pair_addr.wrapping_add(4), - ts_size_raw, - ); - self.decrypt_data3(pair_addr, key, shift); - let mut off = 0u32; - while off + 32 < ts_size_raw { - let t0 = get_u32(&self.decompressed, ts_addr.wrapping_add(off)); - if t0 == 1 || t0 == 0x11 { - let t1 = get_u32(&self.decompressed, ts_addr.wrapping_add(off + 16)); - if t1 == 2 { - let addr0 = get_u32(&self.decompressed, ts_addr.wrapping_add(off + 4)); - if 0x1000 < addr0 && (addr0 as usize) < self.decompressed.len() { - let it = ts_addr.wrapping_add(off); - info_table = Some(it); - keys_addr = it.wrapping_sub(0x58); - ts = ts_addr; - break; - } - } - } - off = off.wrapping_add(4); - } - if info_table.is_some() { - break; - } - } - let info_table = info_table.ok_or(UnpackError::Pe32ThirdStageFailed)?; - if verbose { - println!("[4/9] Decrypting stages (PE32)..."); - println!( - " thirdStage start = 0x{:X}, infoTable = 0x{:X}", - ts, info_table - ); - } - - // ---- Process infoTable ---- - let mut it_addr = info_table; - for _ in 0..2 { - let tval = get_u32(&self.decompressed, it_addr); - if tval == 1 || tval == 0x11 { - self.decrypt_data4(it_addr.wrapping_add(4)); - } else if tval == 2 { - let mut copy_addr = get_u32(&self.decompressed, it_addr.wrapping_add(4)); - loop { - self.decrypt_data5(copy_addr, 16); - let s_a = get_u32(&self.decompressed, copy_addr); - let s_sz = get_u32(&self.decompressed, copy_addr.wrapping_add(4)); - let d_a = get_u32(&self.decompressed, copy_addr.wrapping_add(8)); - let d_sz = get_u32(&self.decompressed, copy_addr.wrapping_add(12)); - copy_addr = copy_addr.wrapping_add(16); - if s_sz == 0 { - break; - } - if s_a != 0 && d_a != 0 && d_sz == s_sz { - let sa = s_a as usize; - let da = d_a as usize; - let n = s_sz as usize; - self.decompressed.copy_within(sa..sa + n, da); - } - } - } - it_addr = it_addr.wrapping_add(16); - } - - // ---- keyOffsets ---- - let mut ka = keys_addr; - for k in 0..2usize { - let mut ka2 = ka; - for l in 0..2usize { - self.decrypt_data4(ka2); - self.key_offsets[k * 2 + l] = get_u32(&self.decompressed, ka2); - ka2 = ka2.wrapping_add(8); - } - ka = ka.wrapping_add(32); - } - - // ---- Checksum addresses ---- - let second_stage_cs_addr = tbl.wrapping_add(0xB0); - let forth_stage_cs_addr = ss.wrapping_add(cs_base_off); - let fifth_stage_cs_addr = ss.wrapping_add(cs_base_off).wrapping_add(0x08); - let seven_stage_cs_addr = ss.wrapping_add(cs_base_off).wrapping_add(0x10); - - // ---- ForthStage ---- - let second_stage_cs = self.calculate_checksum(second_stage_cs_addr); - let forth_stage_key = advance_key( - get_u32(&self.decompressed, ss.wrapping_add(forth_key_off)), - 4, + assert_eq!( + stage.to_string(), + "EXE stage4 decompression failed: Huffman symbol consumed no input and produced no output" ); - let dp_base = ss.wrapping_add(dp_base_off); - let forth_addr = dp_base.wrapping_add(0x40); - let fk = header_checksum ^ second_stage_cs ^ forth_stage_key; - if !self.decrypt_and_decompress_data(forth_addr, fk, None) { - return Err(UnpackError::DecompressFailed); - } - // ---- FifthStage ---- - let fifth_addr = dp_base.wrapping_add(0x50); - let forth_cs = self.calculate_checksum(forth_stage_cs_addr); - let forth_region_off = get_u32(&self.decompressed, forth_stage_cs_addr); - let forth_region_sz = get_u32(&self.decompressed, forth_stage_cs_addr.wrapping_add(4)); - let fifth_key = get_u32( - &self.decompressed, - forth_region_off - .wrapping_add(forth_region_sz) - .wrapping_sub(4), - ); - let fk5 = header_checksum ^ forth_cs ^ fifth_key; - if !self.decrypt_and_decompress_data(fifth_addr, fk5, None) { - return Err(UnpackError::DecompressFailed); - } - - // ---- SevenStage ---- - let seven_addr = dp_base.wrapping_add(0x70); - let seven_dsz = get_u32(&self.decompressed, seven_addr.wrapping_add(12)); - let fifth_cs = self.calculate_checksum(fifth_stage_cs_addr); - let cs1_addr = get_u32( - &self.decompressed, - ss.wrapping_add(cs_base_off).wrapping_add(0x08), - ); - let cs1_size = get_u32( - &self.decompressed, - ss.wrapping_add(cs_base_off) - .wrapping_add(0x08) - .wrapping_add(4), - ); - let seven_key = !get_u32( - &self.decompressed, - cs1_addr.wrapping_add(cs1_size).wrapping_sub(0x10), - ); - let fk7 = header_checksum ^ fifth_cs ^ seven_key; - if !self.decrypt_and_decompress_data(seven_addr, fk7, None) { - return Err(UnpackError::DecompressFailed); - } - - // ---- EighthStage ---- - let seven_start_actual = get_u32(&self.decompressed, seven_addr); - if verbose { - println!("[5/9] Decrypting eighthStage (PE32)..."); - println!( - " sevenStart = 0x{:X}, sevenDsz = 0x{:X}", - seven_start_actual, seven_dsz - ); - } - // Locate the customDecryptor LFSR block (scan backward from middle, then - // forward as fallback). - let scan_start = seven_dsz / 2; - let custom_dec_off = primitives::find_lfsr_block( - &self.decompressed, - seven_start_actual, - seven_dsz, - scan_start, - true, - ) - .or_else(|| { - primitives::find_lfsr_block(&self.decompressed, seven_start_actual, seven_dsz, 0, false) - }) - .ok_or(UnpackError::Pe32CustomDecryptorNotFound)?; - let custom_dec_addr = seven_start_actual.wrapping_add(custom_dec_off); - self.decrypt_data6(custom_dec_addr); - let custom_ops = generate(&self.decompressed, custom_dec_addr) - .ok_or(UnpackError::Pe32BytecodeGenFailed)?; - - let seven_cs = self.calculate_checksum(seven_stage_cs_addr); - let eighth_addr = dp_base.wrapping_add(0xC0); - let eighth_dsz = get_u32(&self.decompressed, eighth_addr.wrapping_add(12)); - let eighth_src = get_u32(&self.decompressed, eighth_addr); - let eighth_ssz = get_u32(&self.decompressed, eighth_addr.wrapping_add(4)); - let eighth_backup: Vec = - self.decompressed[eighth_src as usize..(eighth_src + eighth_ssz) as usize].to_vec(); - let eighth_pair_bak: Vec = - self.decompressed[eighth_addr as usize..(eighth_addr + 16) as usize].to_vec(); - let data_len = self.decompressed.len() as u32; - - // Build the eighthStageKey candidate list (offsets relative to - // sevenStart) using gap heuristics + scan. - let mut candidates: Vec = Vec::new(); - let push_cand = |c: &mut Vec, off: u32| { - if !c.contains(&off) { - c.push(off); - } + let block = UnpackError::SectionDecompressionFailed { + pipeline: SectionPipeline::ExePe32, + block: 7, }; - for &end_gap in &[0xD0u32, 0xC0, 0xE0, 0xB0, 0xA0, 0xF0, 0x100] { - if end_gap <= seven_dsz { - let off = seven_dsz - end_gap; - if off < seven_dsz { - let val = get_u32(&self.decompressed, seven_start_actual.wrapping_add(off)); - if val != 0 && val != 0xCCCC_CCCC { - push_cand(&mut candidates, off); - } - } - } - } - for &gap in &[ - 0x70u32, 0xD0, 0x28, 0x50, 0x48, 0x30, 0x40, 0x58, 0x60, 0x20, 0x38, 0x80, 0x90, 0xA0, - 0xB0, - ] { - if gap <= custom_dec_off { - let off = custom_dec_off - gap; - if off + 4 <= seven_dsz && !candidates.contains(&off) { - let val = get_u32(&self.decompressed, seven_start_actual.wrapping_add(off)); - if val != 0 && val != 0xCCCC_CCCC { - push_cand(&mut candidates, off); - } - } - } - } - let scan_lo = custom_dec_off.saturating_sub(0x100); - let mut off = scan_lo; - while off < custom_dec_off { - if !candidates.contains(&off) { - let val = get_u32(&self.decompressed, seven_start_actual.wrapping_add(off)); - let all_printable = (0..4u32).all(|i| { - let b = (val >> (i * 8)) & 0xFF; - (32..127).contains(&b) - }); - if val != 0 && val != 0xCCCC_CCCC && !all_printable { - push_cand(&mut candidates, off); - } - } - off = off.wrapping_add(4); - } - - let k1 = self.key_offsets[1]; - let k3 = self.key_offsets[3]; - let mut eighth_ok = false; - for ek_off in candidates { - self.decompressed[eighth_src as usize..(eighth_src + eighth_ssz) as usize] - .copy_from_slice(&eighth_backup); - self.decompressed[eighth_addr as usize..(eighth_addr + 16) as usize] - .copy_from_slice(&eighth_pair_bak); - let raw = get_u32(&self.decompressed, seven_start_actual.wrapping_add(ek_off)); - let test_key = advance_key(raw, 3); - let fk8 = header_checksum ^ fifth_cs ^ seven_cs ^ test_key; - let result = primitives::decrypt_and_decompress_data( - &mut self.decompressed, - eighth_addr, - fk8, - k1, - k3, - Some(&custom_ops), - ); - if result { - let est = get_u32(&self.decompressed, eighth_addr); - if 0x1000 < est && est < data_len { - eighth_ok = true; - break; - } - } - } - if !eighth_ok { - return Err(UnpackError::Pe32EighthKeyNotFound); - } - let eighth_start = get_u32(&self.decompressed, eighth_addr); - if verbose { - println!( - " eighthStart = 0x{:08X}, dsz = 0x{:X}", - eighth_start, eighth_dsz - ); - } - - // ---- Final processing offsets (anchored on the eighthStage config cluster) ---- - // - // The eighthStage holds a config cluster — importTable, fileCS, - // compressedInfo, zeroList — at fixed offsets from a cluster base - // (base+0x18 / +0x30 / +0x40 / +0x48) with the fileLFSR at +0x4B4. - // Classic builds stamp a 0x00007679 dword at that base; native DLLs and - // some older PE32 EXEs (ss_size=0xBE8) omit the stamp. - // Locate the cluster by stamp when present (validated by fileCS at - // base+0x30 pointing past info[3]); otherwise fall back to finding the - // fileCS slot by shape — (addr, size) with addr just past info[3] and a - // small 16-aligned size — and back-derive base = fileCS_off - 0x30. - // Hardcoded eighthStart-relative constants remain as a last-resort - // fallback for builds where neither discovery path fires. - let marker = { - let mut m: Option = None; - let hi = eighth_dsz.saturating_sub(0x4C); - let mut o = 0u32; - while o < hi { - if get_u32(&self.decompressed, eighth_start.wrapping_add(o)) == 0x7679 { - let fc = get_u32(&self.decompressed, eighth_start.wrapping_add(o + 0x30)); - if fc > info3 && (fc as usize) < self.decompressed.len() { - m = Some(o); - break; - } - } - o = o.wrapping_add(4); - } - if m.is_none() { - // fileCS-shaped slot: addr in (info3, info3+0x2000], size in - // 0x10..=0x200 and 16-aligned. Prefer the candidate whose addr - // is closest to (but past) info3 — matches every observed - // build (one PE32 EXE family dist ~0x1C0, another ~0x1A0). - let mut best: Option<(u32 /*dist*/, u32 /*off*/)> = None; - let mut o = 0u32; - let dlen = self.decompressed.len() as u32; - while o + 8 <= eighth_dsz.saturating_sub(0x4B4u32.saturating_sub(0x30)) { - let fc = get_u32(&self.decompressed, eighth_start.wrapping_add(o)); - let sz = get_u32(&self.decompressed, eighth_start.wrapping_add(o + 4)); - if fc > info3 - && fc <= info3.wrapping_add(0x2000) - && fc < dlen - && (0x10..=0x200).contains(&sz) - && (sz & 0xF) == 0 - { - // Cluster base must leave room for the +0x4B4 LFSR slot - // (even if the exact LFSR is later adjusted by scan). - if o >= 0x30 { - let base = o - 0x30; - if base.wrapping_add(0x4C) <= eighth_dsz { - let dist = fc - info3; - match best { - None => best = Some((dist, base)), - Some((bd, _)) if dist < bd => best = Some((dist, base)), - _ => {} - } - } - } - } - o = o.wrapping_add(4); - } - if let Some((dist, base)) = best { - if verbose { - println!( - " pe32 cluster via fileCS (no 0x7679): base=+0x{:X} dist_info3=0x{:X}", - base, dist - ); - } - m = Some(base); - } - } - m - }; - let (off_import_table, off_file_cs, off_compressed_info, off_zero_list, off_file_lfsr) = - match marker { - Some(m) => (m + 0x18, m + 0x30, m + 0x40, m + 0x48, m + 0x4B4), - None => ( - 0x3C50u32.wrapping_add(ss_shift), - 0x3C68u32.wrapping_add(ss_shift), - 0x3C78u32.wrapping_add(ss_shift), - 0x3C80u32.wrapping_add(ss_shift), - 0x40ECu32.wrapping_add(ss_shift), - ), - }; - - // ---- File checksums (permanent decrypt) ---- - let file_cs_addr_ptr = eighth_start.wrapping_add(off_file_cs); - let mut file_cs_addr = get_u32(&self.decompressed, file_cs_addr_ptr); - let file_cs_size = get_u32(&self.decompressed, file_cs_addr_ptr.wrapping_add(4)); - if file_cs_size > 0 { - let file_cs_end = file_cs_addr.wrapping_add(file_cs_size); - while file_cs_addr < file_cs_end { - self.decrypt_data5(file_cs_addr, 16); - file_cs_addr = file_cs_addr.wrapping_add(16); - } - } else { - while get_u32(&self.decompressed, file_cs_addr.wrapping_add(4)) != 0 { - self.decrypt_data5(file_cs_addr, 16); - file_cs_addr = file_cs_addr.wrapping_add(16); - } - } - - // ---- File decryptor LFSR ---- - // - // When the marker-relative off_file_lfsr is in range, try that slot - // first (exact). If it is not a valid LFSR block, trial-and-validate - // candidates from off_zero_list forward — required for older PE32 EXEs - // without the 0x7679 stamp where the expected slot is empty and a loose - // decoded[0]+0xC3 nearest-hit picks the wrong decryptor. Fall back to - // the legacy loose scan only if no candidate trial-decompresses. Native - // DLLs have a smaller eighthStage where off_file_lfsr lands out of range - // and use the same trial-validate scan from just past the cluster (else - // branch). - let lfsr_off = if off_file_lfsr.wrapping_add(96) <= eighth_dsz { - let mut lfsr_off = off_file_lfsr; - let exact = primitives::find_lfsr_block( - &self.decompressed, - eighth_start, - eighth_dsz, - off_file_lfsr, - false, - ); - if exact != Some(off_file_lfsr) { - // Prefer trial-and-validate (same as the DLL branch): a loose - // decoded[0]+0xC3 scan can land on coincidental LFSR-shaped - // blocks that decode to a wrong file_ops and scramble every - // compressed block. Observed on older PE32 EXEs without the - // 0x7679 cluster stamp: the expected slot is empty and the - // nearest loose hit is not the real decryptor. - let ci_slot = eighth_start.wrapping_add(off_compressed_info); - let mut scan = off_zero_list; - let mut chosen: Option = None; - let mut considered = 0u32; - while let Some(cand) = primitives::find_lfsr_block( - &self.decompressed, - eighth_start, - eighth_dsz, - scan, - false, - ) { - considered = considered.wrapping_add(1); - if self.pe32_file_lfsr_validates(eighth_start.wrapping_add(cand), ci_slot) { - chosen = Some(cand); - break; - } - scan = cand + 1; - } - if let Some(c) = chosen { - if verbose { - println!( - " pe32 fileLFSR via trial-validate: +0x{:X} (expected +0x{:X}, considered {})", - c, off_file_lfsr, considered - ); - } - lfsr_off = c; - } else { - // No candidate trial-decompresses: fail loudly. The old - // "legacy loose scan" picked the nearest LFSR-shaped block - // by offset distance without any validation — that is - // exactly how a wrong file_ops got applied to every data - // block (uncompressed blocks never hit DecompressFailed), - // producing a plausible but fully wrong image (the PE32 - // .text scramble root cause). Trial-and-validate or error. - return Err(UnpackError::Pe32FileLfsrNotFound); - } - } - lfsr_off - } else { - // Native DLL: the marker-relative off_file_lfsr (EXE-tuned, marker + - // 0x4B4) overshoots the smaller DLL eighthStage, so the exact slot is - // unavailable. A plain forward scan returns the FIRST valid-opcode - // block, but the DLL eighthStage contains coincidental valid-opcode - // blocks that decode to trivial programs (e.g. a constant byte add) - // ahead of the real file decryptor. A wrong file_ops corrupts the - // per-block translate (applied before decompression), so every data - // block fails to decompress. Enumerate every candidate forward and - // keep the first whose decoded file_ops actually decompresses the - // first compressed data block — trial-and-validate, same idea as the - // D1/D2 fixes. Non-DLL (EXE) builds never reach this branch. - let ci_slot = eighth_start.wrapping_add(off_compressed_info); - let mut scan = off_zero_list.wrapping_add(8); - let mut chosen: Option = None; - while let Some(cand) = primitives::find_lfsr_block( - &self.decompressed, - eighth_start, - eighth_dsz, - scan, - false, - ) { - if self.pe32_file_lfsr_validates(eighth_start.wrapping_add(cand), ci_slot) { - chosen = Some(cand); - break; - } - scan = cand + 1; - } - chosen.ok_or(UnpackError::Pe32FileLfsrNotFound)? - }; - let file_dec_addr = eighth_start.wrapping_add(lfsr_off); - self.decrypt_data6(file_dec_addr); - let file_ops = generate(&self.decompressed, file_dec_addr) - .ok_or(UnpackError::Pe32BytecodeGenFailed)?; - - // ---- PE32 metadata: EP and data dirs from info[3] ---- - let test_val = get_u32(&self.decompressed, info3.wrapping_add(0x10)); - let metadata_ep: u32; - let mut metadata_dirs = [0u8; 128]; - if test_val > 0x10000 { - // Layout B - let s = info3.wrapping_add(0x10) as usize; - let backup_meta = self.decompressed[s..s + 0x290].to_vec(); - self.decrypt_data5(info3.wrapping_add(0x10), 0x290); - metadata_ep = get_u32(&self.decompressed, info3.wrapping_add(0x20)); - let d = info3.wrapping_add(0x30) as usize; - metadata_dirs.copy_from_slice(&self.decompressed[d..d + 128]); - self.decompressed[s..s + 0x290].copy_from_slice(&backup_meta); - } else { - // Layout A - let s = info3.wrapping_add(0x40) as usize; - let backup_meta = self.decompressed[s..s + 144].to_vec(); - self.decrypt_data5(info3.wrapping_add(0x40), 144); - metadata_ep = get_u32(&self.decompressed, info3.wrapping_add(0x40)); - let d = info3.wrapping_add(0x50) as usize; - metadata_dirs.copy_from_slice(&self.decompressed[d..d + 128]); - self.decompressed[s..s + 144].copy_from_slice(&backup_meta); - } - - // ---- Zero-out list (runs BEFORE decompression) ---- - let zero_list_addr = eighth_start.wrapping_add(off_zero_list); - let mut zero_ptr = get_u32(&self.decompressed, zero_list_addr); - loop { - self.decrypt_data5(zero_ptr, 16); - let src3 = get_u32(&self.decompressed, zero_ptr); - let s_sz3 = get_u32(&self.decompressed, zero_ptr.wrapping_add(4)); - zero_ptr = zero_ptr.wrapping_add(16); - if s_sz3 == 0 { - break; - } - if src3.wrapping_add(s_sz3) as usize > self.decompressed.len() { - break; - } - for b in &mut self.decompressed[src3 as usize..(src3 + s_sz3) as usize] { - *b = 0; - } - } - - // ---- File data decompression ---- - if verbose { - println!("[6/9] Loading and decompressing file data (PE32)..."); - } - let compress_data_offset = (!get_u32(self.file_data, 0x1080)).wrapping_add(0x1000); - let compressed_info_addr = eighth_start.wrapping_add(off_compressed_info); - let mut compressed_info = get_u32(&self.decompressed, compressed_info_addr); - // Pass 1 (sequential): position-keyed descriptor chain (decrypt_data5), - // terminated by a zero source-size record. - struct Blk { - src: u32, - ssz: u32, - dst: u32, - dsz: u32, - } - let mut blocks: Vec = Vec::new(); - loop { - self.decrypt_data5(compressed_info, 16); - let src2 = get_u32(&self.decompressed, compressed_info); - let s_sz2 = get_u32(&self.decompressed, compressed_info.wrapping_add(4)); - let dst2 = get_u32(&self.decompressed, compressed_info.wrapping_add(8)); - let d_sz2 = get_u32(&self.decompressed, compressed_info.wrapping_add(12)); - compressed_info = compressed_info.wrapping_add(16); - if s_sz2 == 0 { - break; - } - blocks.push(Blk { - src: src2, - ssz: s_sz2, - dst: dst2, - dsz: d_sz2, - }); - } - // Pass 2: independent per-block work over disjoint dst spans (see - // `parallel_for` for how the spans are carved safely). - { - let lut = OpsLut::new(&file_ops); - let clean = &self.file_data; - let ko = self.key_offsets; - let ks_snap = primitives::aes_schedule_snapshot(&self.decompressed, ko[2]) - .ok_or(UnpackError::Corrupt)?; - let tab_snap = primitives::huffman_table_snapshot(&self.decompressed, ko[0]) - .ok_or(UnpackError::DecompressFailed)?; - let spans: Vec<(usize, usize)> = blocks - .iter() - .map(|b| { - let s = b.dst as usize; - (s, s + b.ssz.max(b.dsz) as usize) - }) - .collect(); - let do_block = |i: usize, base: usize, span: &mut [u8]| -> Result<(), UnpackError> { - let b = &blocks[i]; - let file_src = b.src.wrapping_add(compress_data_offset) as usize; - let rel = b.dst as usize - base; - let n = b.ssz as usize; - span[rel..rel + n].copy_from_slice(&clean[file_src..file_src + n]); - primitives::aes_decrypt_ks(&ks_snap, span, rel as u32, b.ssz); - lut.map_region(span, rel, n); - if b.ssz != b.dsz { - // decompress reports corruption (after partial writes) via - // its bool; surface it instead of shipping a garbage block. - if !primitives::decompress_tbl( - &tab_snap, span, rel as u32, rel as u32, b.ssz, b.dsz, - ) { - return Err(UnpackError::DecompressFailed); - } - } - Ok(()) - }; - super::parallel::parallel_for(&mut self.decompressed, &spans, 1, do_block)?; - } - - // ---- Section fixup ---- - self.decompressed[..0x1000].copy_from_slice(&self.file_data[..0x1000]); - let opt_hdr_size = get_u16(self.file_data, pe_off.wrapping_add(20)) as u32; - let sec_hdr_table = pe_off.wrapping_add(24).wrapping_add(opt_hdr_size); - let export_va = get_u32( - self.file_data, - pe_off - .wrapping_add(24) - .wrapping_add(opt_hdr_size) - .wrapping_sub(128), + assert_eq!( + block.to_string(), + "PE32 EXE section block 7 decompression failed" ); - let export_size = get_u32( - self.file_data, - pe_off - .wrapping_add(24) - .wrapping_add(opt_hdr_size) - .wrapping_sub(124), - ); - let mut export_file_off: u32 = 0; - let mut text_off: u32 = 0; - let mut text_size: u32 = 0; - // Walk by NumberOfSections (PE has no zero-VS sentinel; a real - // VirtualSize==0 section would truncate these fixups early), stopping - // at the all-zero padding in case NumberOfSections is overstated. - let num_sections = get_u16(self.file_data, pe_off.wrapping_add(6)) as u32; - for i in 0..num_sections.min(96) { - let sec_hdr = sec_hdr_table.wrapping_add(i.wrapping_mul(40)); - if self.file_data[sec_hdr as usize..sec_hdr as usize + 8] - .iter() - .all(|&b| b == 0) - { - break; - } - let va = get_u32(self.file_data, sec_hdr.wrapping_add(12)); - let sz = get_u32(self.file_data, sec_hdr.wrapping_add(8)); - let f_off = get_u32(self.file_data, sec_hdr.wrapping_add(20)); - let name = section_name(self.file_data, sec_hdr); - if name.starts_with(".text") { - text_size = sz; - text_off = va; - } - if export_size != 0 - && export_va >= va - && export_va.wrapping_add(export_size) <= va.wrapping_add(sz) - { - export_file_off = export_va.wrapping_sub(va).wrapping_add(f_off); - } - write_u32(&mut self.decompressed, sec_hdr.wrapping_add(16), sz); - write_u32(&mut self.decompressed, sec_hdr.wrapping_add(20), va); - if name.starts_with(".idata") { - write_u32( - &mut self.decompressed, - sec_hdr.wrapping_add(36), - 0xC000_0040, - ); - } - } - if export_size != 0 && export_file_off != 0 { - let d = export_va as usize; - let s = export_file_off as usize; - let n = export_size as usize; - self.decompressed[d..d + n].copy_from_slice(&self.file_data[s..s + n]); - } - - // ---- .text decrypt with decrypt_data8 (PE32 auto-detected formula) ---- - // `select_dd8_formula_pe32` returns None when `.text` was not packer-dd8- - // encrypted (native DLLs leave it plaintext); applying dd8 there would - // scramble valid code, so skip it entirely in that case. - if text_size > 0 && text_off > 0 { - if let Some(big) = - primitives::select_dd8_formula_pe32(&self.decompressed, text_off, text_size) - { - if verbose { - println!( - "[7/9] Decrypting .text (PE32 dd8, formula={})...", - if big { "0x8000*(page+1)" } else { "page+1" } - ); - } - let num_pages = text_size / 0x1000; - for page in 0..num_pages { - let pk = if big { - 0x8000u32.wrapping_mul(page.wrapping_add(1)) - } else { - page.wrapping_add(1) - }; - let pa = text_off.wrapping_add(page.wrapping_mul(0x1000)); - let mut k = pk; - let rk = k.rotate_right(15); - k = rk; - for bi in 1..256u32 { - let rk = k.rotate_right(15); - let ri = rk.wrapping_add(bi); - k = ri.wrapping_add(bi); - let tidx = - pa.wrapping_add(bi.wrapping_mul(16)).wrapping_add(ri & 0xF) as usize; - self.decompressed[tidx] ^= k as u8; - } - } - } else if verbose { - println!("[7/9] Skipping .text dd8 (already plaintext)..."); - } - } - - // ---- Fix data directories (PE32: data dirs at pe+0x78) ---- - let exe_pe = get_u32(&self.decompressed, 60); - for i in 0..128u32 { - self.decompressed[(exe_pe + 0x78 + i) as usize] = metadata_dirs[i as usize]; - } - // DLL-aware reloc / DllCharacteristics handling. An EXE's packer rebuilds - // the relocation table and clears DllCharacteristics, so the loader needs - // no relocations. A DLL, by contrast, is almost always mapped at a - // non-preferred base, so it MUST keep its base-relocation directory - // (restored above from metadata_dirs) and a valid DllCharacteristics - // (DYNAMIC_BASE) — zeroing them leaves the DLL unrelocatable and its - // imports pinned to the packer stub, so it fails to load (which looks - // like a missing/broken export table). - let is_dll = (get_u16(&self.decompressed, exe_pe.wrapping_add(22)) & 0x2000) != 0; - if !is_dll { - // EXE: clear BaseReloc (index 5 = pe+0xA0) and DllCharacteristics (pe+0x5E). - write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xA0), 0); - write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xA4), 0); - write_u16(&mut self.decompressed, exe_pe.wrapping_add(0x5E), 0); - } else { - // DLL: keep the BaseReloc dir from metadata; ensure DYNAMIC_BASE. - let mut dll_chars = get_u16(&self.decompressed, exe_pe.wrapping_add(0x5E)); - if dll_chars == 0 { - dll_chars = 0x0040; // IMAGE_DLLCHARACTERISTICS_DYNAMIC_BASE - } - write_u16( - &mut self.decompressed, - exe_pe.wrapping_add(0x5E), - dll_chars as u32, - ); - } - - // ---- TLS directory reconstruction (PE32: index 9 = pe+0xC0) ---- - let tls_dir_rva = get_u32(&self.decompressed, exe_pe.wrapping_add(0xC0)); - let tls_dir_sz = get_u32(&self.decompressed, exe_pe.wrapping_add(0xC4)); - if tls_dir_rva > 0 - && tls_dir_sz >= 24 - && (tls_dir_rva as usize + 24) <= self.decompressed.len() - { - let all_zero = (0..6u32) - .all(|i| get_u32(&self.decompressed, tls_dir_rva.wrapping_add(i * 4)) == 0); - if all_zero { - let image_base = get_u32(&self.decompressed, exe_pe.wrapping_add(52)); - // Prefer the module's real TLS directory, which survives in the - // loader stub's plaintext `.rdata`/`.tls`. Only when the stub - // cannot supply one does a placeholder get synthesized: it keeps - // the image loadable, but drops the initialized TLS template, - // `_tls_index` and the TLS callback array, so any module that - // actually uses `thread_local` faults once it runs. - if !self.restore_pe32_tls_from_stub(pe_off, tls_dir_rva, image_base) { - let mut tls_sec_va: u32 = 0; - let mut data_sec_va: u32 = 0; - let mut data_sec_sz: u32 = 0; - let sh = exe_pe - .wrapping_add(24) - .wrapping_add(get_u16(&self.decompressed, exe_pe.wrapping_add(20)) as u32); - let ns = get_u16(&self.decompressed, exe_pe.wrapping_add(6)) as u32; - for i in 0..ns { - let s = sh.wrapping_add(i * 40); - let nm = get_string_to_null(&self.decompressed, s); - let va = get_u32(&self.decompressed, s.wrapping_add(12)); - let sz = get_u32(&self.decompressed, s.wrapping_add(16)); - if nm.starts_with(".tls") { - tls_sec_va = va; - } - if nm.starts_with(".data") { - data_sec_va = va; - data_sec_sz = sz; - } - } - if tls_sec_va > 0 && data_sec_va > 0 { - let start_raw = image_base.wrapping_add(tls_sec_va); - let end_raw = start_raw; - let idx_addr = image_base - .wrapping_add(data_sec_va) - .wrapping_add(data_sec_sz) - .wrapping_sub(16); - let cb_addr = image_base - .wrapping_add(data_sec_va) - .wrapping_add(data_sec_sz) - .wrapping_sub(8); - let scratch = (data_sec_va + data_sec_sz - 16) as usize; - for b in &mut self.decompressed[scratch..scratch + 16] { - *b = 0; - } - write_u32(&mut self.decompressed, tls_dir_rva, start_raw); - write_u32(&mut self.decompressed, tls_dir_rva.wrapping_add(4), end_raw); - write_u32( - &mut self.decompressed, - tls_dir_rva.wrapping_add(8), - idx_addr, - ); - write_u32( - &mut self.decompressed, - tls_dir_rva.wrapping_add(12), - cb_addr, - ); - write_u32(&mut self.decompressed, tls_dir_rva.wrapping_add(16), 0); - write_u32( - &mut self.decompressed, - tls_dir_rva.wrapping_add(20), - 0x30_0000, - ); - } else { - write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xC0), 0); - write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xC4), 0); - } - } - } - } - - // ---- Import table (PE32, 4-byte thunks) ---- - if verbose { - println!("[8/9] Decrypting import strings (PE32)..."); - } - let import_table_addr = eighth_start.wrapping_add(off_import_table); - let mut import_table_ptr = get_u32(&self.decompressed, import_table_addr); - let mut idt_size = get_u32(&self.decompressed, import_table_addr.wrapping_add(4)); - - let metadata_import_rva = get_u32(&metadata_dirs, 8); - let metadata_import_size = get_u32(&metadata_dirs, 12); - let dlen = self.decompressed.len() as u32; - - let mut eighth_import_valid = false; - if 0 < import_table_ptr && import_table_ptr < dlen && 0 < idt_size && idt_size < 0x10000 { - let test_name = if import_table_ptr + 20 <= dlen { - get_u32(&self.decompressed, import_table_ptr.wrapping_add(12)) - } else { - 0 - }; - let test_ilt = if import_table_ptr + 4 <= dlen { - get_u32(&self.decompressed, import_table_ptr) - } else { - 0 - }; - if 0x1000 < test_name && test_name < dlen && 0x1000 < test_ilt && test_ilt < dlen { - eighth_import_valid = true; - } - } - let mut metadata_import_valid = false; - if 0x1000 < metadata_import_rva && metadata_import_rva < dlen.wrapping_sub(20) { - let test_name2 = get_u32(&self.decompressed, metadata_import_rva.wrapping_add(12)); - let test_ilt2 = get_u32(&self.decompressed, metadata_import_rva); - if 0x1000 < test_name2 && test_name2 < dlen && 0x1000 < test_ilt2 && test_ilt2 < dlen { - metadata_import_valid = true; - } - } - if metadata_import_valid - && (!eighth_import_valid || metadata_import_rva != import_table_ptr) - { - import_table_ptr = metadata_import_rva; - idt_size = metadata_import_size; - } - - if 0 < import_table_ptr && import_table_ptr < dlen && 0 < idt_size && idt_size < 0x10000 { - let mut idt_pos = import_table_ptr; - let idt_end = import_table_ptr.wrapping_add(idt_size); - while idt_pos.wrapping_add(20) <= idt_end { - let ilt_rva = get_u32(&self.decompressed, idt_pos); - let name_rva = get_u32(&self.decompressed, idt_pos.wrapping_add(12)); - let iat_rva = get_u32(&self.decompressed, idt_pos.wrapping_add(16)); - if ilt_rva == 0 && name_rva == 0 && iat_rva == 0 { - break; - } - if 0 < name_rva && name_rva < dlen { - self.decrypt_data7(name_rva, name_rva as u8); - } - let thunk_base = if 0 < ilt_rva && ilt_rva < dlen { - ilt_rva - } else { - iat_rva - }; - if 0 < thunk_base && thunk_base < dlen.wrapping_sub(4) { - let mut thunk_pos = thunk_base; - while thunk_pos.wrapping_add(4) <= dlen { - let thunk_val = get_u32(&self.decompressed, thunk_pos); - if thunk_val == 0 { - break; - } - if thunk_val & 0x8000_0000 == 0 && thunk_val.wrapping_add(2) < dlen { - self.decrypt_data7(thunk_val.wrapping_add(2), thunk_val as u8); - write_u16(&mut self.decompressed, thunk_val, 0); - } - thunk_pos = thunk_pos.wrapping_add(4); - } - } - idt_pos = idt_pos.wrapping_add(20); - } - } - - // Update PE header: Import directory (index 1 = pe+0x80), clear IAT - // directory (index 12 = pe+0xD8). - write_u32( - &mut self.decompressed, - exe_pe.wrapping_add(0x80), - import_table_ptr, - ); - write_u32(&mut self.decompressed, exe_pe.wrapping_add(0x84), idt_size); - write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xD8), 0); - write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xDC), 0); - - // ---- EP (from metadata) ---- - if metadata_ep > 0 { - write_u32(&mut self.decompressed, exe_pe.wrapping_add(40), metadata_ep); - } else { - let real_ep = get_u32(self.file_data, pe_off.wrapping_add(40)); - write_u32(&mut self.decompressed, exe_pe.wrapping_add(40), real_ep); - } - - // ---- Output transforms ---- - if verbose { - println!("[9/9] Rebuilding PE file layout (PE32)..."); - } - let mut out = std::mem::take(&mut self.decompressed); - // kmiat import relocation is an EXE-only fixup: it discards the original - // import directory in favour of the loader-written IAT stub. A DLL keeps - // its real import table (restored above from metadata), so skip kmiat for - // DLLs (`!is_dll` guard). - let is_dll = (get_u16(&out, pe_off.wrapping_add(22)) & 0x2000) != 0; - if !is_dll && !primitives::pe32_imports_already_match_idata_layout(&mut out, pe_off) { - primitives::move_pe32_imports_to_kmiat(&mut out, pe_off); - } - let compact = - primitives::compact_memory_image_to_pe(&out, pe_off).ok_or(UnpackError::Corrupt)?; - Ok(compact) } } diff --git a/senbei-pe/src/engine/exe/pipeline/pe32.rs b/senbei-pe/src/engine/exe/pipeline/pe32.rs new file mode 100644 index 0000000..0b319ba --- /dev/null +++ b/senbei-pe/src/engine/exe/pipeline/pe32.rs @@ -0,0 +1,1063 @@ +use super::super::super::layout; +use super::*; + +impl<'a> Unpacker<'a> { + /// PE32 (32-bit) unpack pipeline. The shared Stage 1/2 setup (info decrypt, + /// payload decrypt, raw copy, header restore) has already run in `run()` + /// before dispatch; this takes over from "Locating shell offsets". + pub(super) fn run_pe32(&mut self, pe_off: u32, verbose: bool) -> Result, UnpackError> { + let info = self.info; + let info3 = info[3]; + + // advance_key: replays the packer's per-iteration key walk. + let advance_key = |mut key: u32, iterations: u32| -> u32 { + for m in 0..iterations { + let bound = (m + 1).wrapping_mul(25) << 2; + let mut n: u32 = 1; + while n <= bound { + key = key.wrapping_add(n); + n += 1; + } + } + key + }; + + // ---- Locate tbl in shell ---- + let tbl = + layout::find_tbl_pe32(&self.decompressed, &info).ok_or(UnpackError::Pe32TblNotFound)?; + if verbose { + println!("[3/9] Locating config layout (PE32)..."); + println!(" tbl = 0x{:X}", tbl); + } + + // ---- PE header restore ---- + let val_bc = get_u32(&self.decompressed, tbl.wrapping_add(0xBC)); + let val_c8 = get_u32(&self.decompressed, tbl.wrapping_add(0xC8)); + let val_cc = get_u32(&self.decompressed, tbl.wrapping_add(0xCC)); + write_u32(&mut self.decompressed, pe_off.wrapping_add(0x80), val_bc); + write_u32(&mut self.decompressed, pe_off.wrapping_add(0x88), val_c8); + write_u32(&mut self.decompressed, pe_off.wrapping_add(0x8C), val_cc); + write_u32(&mut self.decompressed, pe_off.wrapping_add(0xB0), 0); + write_u32(&mut self.decompressed, pe_off.wrapping_add(0xB4), 0); + + // ---- Header-independent checksum inputs ---- + let first_stage_cs = self.calculate_checksum(tbl.wrapping_add(0xA8)); + let second_stage_key = get_u32(&self.decompressed, tbl.wrapping_add(0x40)); + + // ---- Stage 3: SecondStage ---- + // + // ss_key = headerChecksum ^ firstStageCS ^ secondStageKey, where the + // header checksum (a XOR of crc32(region)^size over the sub-regions at + // tbl+0x58) is taken over the *original* pre-pack PE header. For EXEs the + // import/resource restore above reconstructs that header exactly. Native + // DLLs additionally carry a packer-added BaseReloc data-directory entry + // (dir 5) that was absent from the checksummed original, so the header + // checksum only matches once that entry is treated as zero. EXEs have no + // dir-5 entry, so zeroing it is a no-op for them. + // + // Rather than branch on EXE-vs-DLL, try the header as-is and, on failure, + // with the BaseReloc entry zeroed; keep whichever ss_key decrypts a + // SecondStage whose ThirdStage (off,size) pair lands inside the image. + // This uses the same shift/key trial-and-validate the later stages + // already use, and keeps EXE output byte-identical (the as-is variant + // wins first). + let ss_pair = tbl.wrapping_add(0x98); + let ss = get_u32(&self.decompressed, ss_pair); + let ss_size = get_u32(&self.decompressed, ss_pair.wrapping_add(4)); + let ss_shift = ss_size.wrapping_sub(0xBC0); + // Back up the SecondStage ciphertext so a failed trial can be retried. + let ss_lo = ss as usize; + let ss_hi = ss_lo.wrapping_add(ss_size as usize); + if ss_hi < ss_lo || ss_hi > self.decompressed.len() { + return Err(UnpackError::Pe32SecondStageRangeInvalid { + offset: ss, + size: ss_size, + image_len: self.decompressed.len(), + }); + } + let ss_ct: Vec = self.decompressed[ss_lo..ss_hi].to_vec(); + // PE32 data dir 5 (BaseReloc) = optional_header(pe+24) + 0x60 + 5*8 = pe+0xA0. + let reloc_dir = pe_off.wrapping_add(0xA0); + let len = self.decompressed.len() as u64; + let pair_off = 0xB8Cu32.wrapping_add(ss_shift); + let mut found = false; + // Holds the winning variant's header checksum; the later stages + // (Forth/Fifth/Seven/Eighth) reuse it as a key component. + let mut header_checksum: u32 = 0; + for zero_reloc in [false, true] { + if zero_reloc { + write_u32(&mut self.decompressed, reloc_dir, 0); + write_u32(&mut self.decompressed, reloc_dir.wrapping_add(4), 0); + } + let mut hcs_addr = tbl.wrapping_add(0x58); + header_checksum = 0; + while get_u32(&self.decompressed, hcs_addr.wrapping_add(4)) != 0 { + header_checksum ^= self.calculate_checksum(hcs_addr); + hcs_addr = hcs_addr.wrapping_add(8); + } + let ss_key = header_checksum ^ first_stage_cs ^ second_stage_key; + self.decompressed[ss_lo..ss_hi].copy_from_slice(&ss_ct); + self.decrypt_data3(ss_pair, ss_key, 21); + // Validate: the ThirdStage (off,size) pair must reference the image. + let pair = ss.wrapping_add(pair_off); + let off = get_u32(&self.decompressed, pair) as u64; + let sz = get_u32(&self.decompressed, pair.wrapping_add(4)) as u64; + if off > 0x1000 && off < len && sz >= 4 && off.saturating_add(sz) <= len { + found = true; + break; + } + } + if !found { + return Err(UnpackError::Pe32RelocationDataNotFound); + } + if verbose { + println!( + " ss = 0x{:08X}, size = 0x{:X}, shift = 0x{:X}", + ss, ss_size, ss_shift + ); + } + + // ---- PE32 fixed offsets ---- + let third_key_off = 0x968u32.wrapping_add(ss_shift); + let forth_key_off = 0x964u32.wrapping_add(ss_shift); + let cs_base_off = 0x96Cu32.wrapping_add(ss_shift); + let dp_base_off = 0xA9Cu32.wrapping_add(ss_shift); + + // ---- Stage 4: ThirdStage (brute-force the rotate shift) ---- + let third_pair_off = 0xB8Cu32.wrapping_add(ss_shift); + let key = get_u32(&self.decompressed, ss.wrapping_add(third_key_off)); + let pair_addr = ss.wrapping_add(third_pair_off); + let ts_addr = get_u32(&self.decompressed, pair_addr); + let ts_size_raw = get_u32(&self.decompressed, pair_addr.wrapping_add(4)); + let backup: Vec = + self.decompressed[ts_addr as usize..(ts_addr + ts_size_raw) as usize].to_vec(); + let mut info_table: Option = None; + let mut keys_addr: u32 = 0; + let mut ts: u32 = 0; + for &shift in &[19u32, 21, 17, 23, 15, 25, 13, 11] { + self.decompressed[ts_addr as usize..(ts_addr + ts_size_raw) as usize] + .copy_from_slice(&backup); + write_u32(&mut self.decompressed, pair_addr, ts_addr); + write_u32( + &mut self.decompressed, + pair_addr.wrapping_add(4), + ts_size_raw, + ); + self.decrypt_data3(pair_addr, key, shift); + let mut off = 0u32; + while off + 32 < ts_size_raw { + let t0 = get_u32(&self.decompressed, ts_addr.wrapping_add(off)); + if t0 == 1 || t0 == 0x11 { + let t1 = get_u32(&self.decompressed, ts_addr.wrapping_add(off + 16)); + if t1 == 2 { + let addr0 = get_u32(&self.decompressed, ts_addr.wrapping_add(off + 4)); + if 0x1000 < addr0 && (addr0 as usize) < self.decompressed.len() { + let it = ts_addr.wrapping_add(off); + info_table = Some(it); + keys_addr = it.wrapping_sub(0x58); + ts = ts_addr; + break; + } + } + } + off = off.wrapping_add(4); + } + if info_table.is_some() { + break; + } + } + let info_table = info_table.ok_or(UnpackError::Pe32ThirdStageFailed)?; + if verbose { + println!("[4/9] Decrypting stages (PE32)..."); + println!( + " thirdStage start = 0x{:X}, infoTable = 0x{:X}", + ts, info_table + ); + } + + // ---- Process infoTable ---- + let mut it_addr = info_table; + for _ in 0..2 { + let tval = get_u32(&self.decompressed, it_addr); + if tval == 1 || tval == 0x11 { + self.decrypt_data4(it_addr.wrapping_add(4)); + } else if tval == 2 { + let mut copy_addr = get_u32(&self.decompressed, it_addr.wrapping_add(4)); + loop { + self.decrypt_data5(copy_addr, 16); + let s_a = get_u32(&self.decompressed, copy_addr); + let s_sz = get_u32(&self.decompressed, copy_addr.wrapping_add(4)); + let d_a = get_u32(&self.decompressed, copy_addr.wrapping_add(8)); + let d_sz = get_u32(&self.decompressed, copy_addr.wrapping_add(12)); + copy_addr = copy_addr.wrapping_add(16); + if s_sz == 0 { + break; + } + if s_a != 0 && d_a != 0 && d_sz == s_sz { + let sa = s_a as usize; + let da = d_a as usize; + let n = s_sz as usize; + self.decompressed.copy_within(sa..sa + n, da); + } + } + } + it_addr = it_addr.wrapping_add(16); + } + + // ---- keyOffsets ---- + let mut ka = keys_addr; + for k in 0..2usize { + let mut ka2 = ka; + for l in 0..2usize { + self.decrypt_data4(ka2); + self.key_offsets[k * 2 + l] = get_u32(&self.decompressed, ka2); + ka2 = ka2.wrapping_add(8); + } + ka = ka.wrapping_add(32); + } + + // ---- Checksum addresses ---- + let second_stage_cs_addr = tbl.wrapping_add(0xB0); + let forth_stage_cs_addr = ss.wrapping_add(cs_base_off); + let fifth_stage_cs_addr = ss.wrapping_add(cs_base_off).wrapping_add(0x08); + let seven_stage_cs_addr = ss.wrapping_add(cs_base_off).wrapping_add(0x10); + + // ---- ForthStage ---- + let second_stage_cs = self.calculate_checksum(second_stage_cs_addr); + let forth_stage_key = advance_key( + get_u32(&self.decompressed, ss.wrapping_add(forth_key_off)), + 4, + ); + let dp_base = ss.wrapping_add(dp_base_off); + let forth_addr = dp_base.wrapping_add(0x40); + let fk = header_checksum ^ second_stage_cs ^ forth_stage_key; + if let Err(reason) = self.decrypt_and_decompress_data(forth_addr, fk, None) { + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::Pe32FourthStage, + reason, + }); + } + + // ---- FifthStage ---- + let fifth_addr = dp_base.wrapping_add(0x50); + let forth_cs = self.calculate_checksum(forth_stage_cs_addr); + let forth_region_off = get_u32(&self.decompressed, forth_stage_cs_addr); + let forth_region_sz = get_u32(&self.decompressed, forth_stage_cs_addr.wrapping_add(4)); + let fifth_key = get_u32( + &self.decompressed, + forth_region_off + .wrapping_add(forth_region_sz) + .wrapping_sub(4), + ); + let fk5 = header_checksum ^ forth_cs ^ fifth_key; + if let Err(reason) = self.decrypt_and_decompress_data(fifth_addr, fk5, None) { + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::Pe32FifthStage, + reason, + }); + } + + // ---- SevenStage ---- + let seven_addr = dp_base.wrapping_add(0x70); + let seven_dsz = get_u32(&self.decompressed, seven_addr.wrapping_add(12)); + let fifth_cs = self.calculate_checksum(fifth_stage_cs_addr); + let cs1_addr = get_u32( + &self.decompressed, + ss.wrapping_add(cs_base_off).wrapping_add(0x08), + ); + let cs1_size = get_u32( + &self.decompressed, + ss.wrapping_add(cs_base_off) + .wrapping_add(0x08) + .wrapping_add(4), + ); + let seven_key = !get_u32( + &self.decompressed, + cs1_addr.wrapping_add(cs1_size).wrapping_sub(0x10), + ); + let fk7 = header_checksum ^ fifth_cs ^ seven_key; + if let Err(reason) = self.decrypt_and_decompress_data(seven_addr, fk7, None) { + return Err(UnpackError::StageDecompressionFailed { + stage: DecompressionStage::Pe32SeventhStage, + reason, + }); + } + + // ---- EighthStage ---- + let seven_start_actual = get_u32(&self.decompressed, seven_addr); + if verbose { + println!("[5/9] Decrypting eighthStage (PE32)..."); + println!( + " sevenStart = 0x{:X}, sevenDsz = 0x{:X}", + seven_start_actual, seven_dsz + ); + } + // Locate the customDecryptor LFSR block (scan backward from middle, then + // forward as fallback). + let scan_start = seven_dsz / 2; + let custom_dec_off = layout::find_lfsr_block( + &self.decompressed, + seven_start_actual, + seven_dsz, + scan_start, + true, + ) + .or_else(|| { + layout::find_lfsr_block(&self.decompressed, seven_start_actual, seven_dsz, 0, false) + }) + .ok_or(UnpackError::Pe32CustomDecryptorNotFound)?; + let custom_dec_addr = seven_start_actual.wrapping_add(custom_dec_off); + self.decrypt_data6(custom_dec_addr); + let custom_ops = generate(&self.decompressed, custom_dec_addr).ok_or( + UnpackError::BytecodeGenerationFailed(BytecodeStage::Pe32CustomDecryptor), + )?; + + let seven_cs = self.calculate_checksum(seven_stage_cs_addr); + let eighth_addr = dp_base.wrapping_add(0xC0); + let eighth_dsz = get_u32(&self.decompressed, eighth_addr.wrapping_add(12)); + let eighth_src = get_u32(&self.decompressed, eighth_addr); + let eighth_ssz = get_u32(&self.decompressed, eighth_addr.wrapping_add(4)); + let eighth_backup: Vec = + self.decompressed[eighth_src as usize..(eighth_src + eighth_ssz) as usize].to_vec(); + let eighth_pair_bak: Vec = + self.decompressed[eighth_addr as usize..(eighth_addr + 16) as usize].to_vec(); + let data_len = self.decompressed.len() as u32; + + // Build the eighthStageKey candidate list (offsets relative to + // sevenStart) using gap heuristics + scan. + let mut candidates: Vec = Vec::new(); + let push_cand = |c: &mut Vec, off: u32| { + if !c.contains(&off) { + c.push(off); + } + }; + for &end_gap in &[0xD0u32, 0xC0, 0xE0, 0xB0, 0xA0, 0xF0, 0x100] { + if end_gap <= seven_dsz { + let off = seven_dsz - end_gap; + if off < seven_dsz { + let val = get_u32(&self.decompressed, seven_start_actual.wrapping_add(off)); + if val != 0 && val != 0xCCCC_CCCC { + push_cand(&mut candidates, off); + } + } + } + } + for &gap in &[ + 0x70u32, 0xD0, 0x28, 0x50, 0x48, 0x30, 0x40, 0x58, 0x60, 0x20, 0x38, 0x80, 0x90, 0xA0, + 0xB0, + ] { + if gap <= custom_dec_off { + let off = custom_dec_off - gap; + if off + 4 <= seven_dsz && !candidates.contains(&off) { + let val = get_u32(&self.decompressed, seven_start_actual.wrapping_add(off)); + if val != 0 && val != 0xCCCC_CCCC { + push_cand(&mut candidates, off); + } + } + } + } + let scan_lo = custom_dec_off.saturating_sub(0x100); + let mut off = scan_lo; + while off < custom_dec_off { + if !candidates.contains(&off) { + let val = get_u32(&self.decompressed, seven_start_actual.wrapping_add(off)); + let all_printable = (0..4u32).all(|i| { + let b = (val >> (i * 8)) & 0xFF; + (32..127).contains(&b) + }); + if val != 0 && val != 0xCCCC_CCCC && !all_printable { + push_cand(&mut candidates, off); + } + } + off = off.wrapping_add(4); + } + + let k1 = self.key_offsets[1]; + let k3 = self.key_offsets[3]; + let mut eighth_ok = false; + for ek_off in candidates { + self.decompressed[eighth_src as usize..(eighth_src + eighth_ssz) as usize] + .copy_from_slice(&eighth_backup); + self.decompressed[eighth_addr as usize..(eighth_addr + 16) as usize] + .copy_from_slice(&eighth_pair_bak); + let raw = get_u32(&self.decompressed, seven_start_actual.wrapping_add(ek_off)); + let test_key = advance_key(raw, 3); + let fk8 = header_checksum ^ fifth_cs ^ seven_cs ^ test_key; + let result = primitives::decrypt_and_decompress_data( + &mut self.decompressed, + eighth_addr, + fk8, + k1, + k3, + Some(&custom_ops), + ); + if result { + let est = get_u32(&self.decompressed, eighth_addr); + if 0x1000 < est && est < data_len { + eighth_ok = true; + break; + } + } + } + if !eighth_ok { + return Err(UnpackError::Pe32EighthKeyNotFound); + } + let eighth_start = get_u32(&self.decompressed, eighth_addr); + if verbose { + println!( + " eighthStart = 0x{:08X}, dsz = 0x{:X}", + eighth_start, eighth_dsz + ); + } + + // ---- Final processing offsets (anchored on the eighthStage config cluster) ---- + // + // The eighthStage holds a config cluster — importTable, fileCS, + // compressedInfo, zeroList — at fixed offsets from a cluster base + // (base+0x18 / +0x30 / +0x40 / +0x48) with the fileLFSR at +0x4B4. + // Classic builds stamp a 0x00007679 dword at that base; native DLLs and + // some older PE32 EXEs (ss_size=0xBE8) omit the stamp. + // Locate the cluster by stamp when present (validated by fileCS at + // base+0x30 pointing past info[3]); otherwise fall back to finding the + // fileCS slot by shape — (addr, size) with addr just past info[3] and a + // small 16-aligned size — and back-derive base = fileCS_off - 0x30. + // Hardcoded eighthStart-relative constants remain as a last-resort + // fallback for builds where neither discovery path fires. + let marker = { + let mut m: Option = None; + let hi = eighth_dsz.saturating_sub(0x4C); + let mut o = 0u32; + while o < hi { + if get_u32(&self.decompressed, eighth_start.wrapping_add(o)) == 0x7679 { + let fc = get_u32(&self.decompressed, eighth_start.wrapping_add(o + 0x30)); + if fc > info3 && (fc as usize) < self.decompressed.len() { + m = Some(o); + break; + } + } + o = o.wrapping_add(4); + } + if m.is_none() { + // fileCS-shaped slot: addr in (info3, info3+0x2000], size in + // 0x10..=0x200 and 16-aligned. Prefer the candidate whose addr + // is closest to (but past) info3 — matches every observed + // build (one PE32 EXE family dist ~0x1C0, another ~0x1A0). + let mut best: Option<(u32 /*dist*/, u32 /*off*/)> = None; + let mut o = 0u32; + let dlen = self.decompressed.len() as u32; + while o + 8 <= eighth_dsz.saturating_sub(0x4B4u32.saturating_sub(0x30)) { + let fc = get_u32(&self.decompressed, eighth_start.wrapping_add(o)); + let sz = get_u32(&self.decompressed, eighth_start.wrapping_add(o + 4)); + if fc > info3 + && fc <= info3.wrapping_add(0x2000) + && fc < dlen + && (0x10..=0x200).contains(&sz) + && (sz & 0xF) == 0 + { + // Cluster base must leave room for the +0x4B4 LFSR slot + // (even if the exact LFSR is later adjusted by scan). + if o >= 0x30 { + let base = o - 0x30; + if base.wrapping_add(0x4C) <= eighth_dsz { + let dist = fc - info3; + match best { + None => best = Some((dist, base)), + Some((bd, _)) if dist < bd => best = Some((dist, base)), + _ => {} + } + } + } + } + o = o.wrapping_add(4); + } + if let Some((dist, base)) = best { + if verbose { + println!( + " pe32 cluster via fileCS (no 0x7679): base=+0x{:X} dist_info3=0x{:X}", + base, dist + ); + } + m = Some(base); + } + } + m + }; + let (off_import_table, off_file_cs, off_compressed_info, off_zero_list, off_file_lfsr) = + match marker { + Some(m) => (m + 0x18, m + 0x30, m + 0x40, m + 0x48, m + 0x4B4), + None => ( + 0x3C50u32.wrapping_add(ss_shift), + 0x3C68u32.wrapping_add(ss_shift), + 0x3C78u32.wrapping_add(ss_shift), + 0x3C80u32.wrapping_add(ss_shift), + 0x40ECu32.wrapping_add(ss_shift), + ), + }; + + // ---- File checksums (permanent decrypt) ---- + let file_cs_addr_ptr = eighth_start.wrapping_add(off_file_cs); + let mut file_cs_addr = get_u32(&self.decompressed, file_cs_addr_ptr); + let file_cs_size = get_u32(&self.decompressed, file_cs_addr_ptr.wrapping_add(4)); + if file_cs_size > 0 { + let file_cs_end = file_cs_addr.wrapping_add(file_cs_size); + while file_cs_addr < file_cs_end { + self.decrypt_data5(file_cs_addr, 16); + file_cs_addr = file_cs_addr.wrapping_add(16); + } + } else { + while get_u32(&self.decompressed, file_cs_addr.wrapping_add(4)) != 0 { + self.decrypt_data5(file_cs_addr, 16); + file_cs_addr = file_cs_addr.wrapping_add(16); + } + } + + // ---- File decryptor LFSR ---- + // + // When the marker-relative off_file_lfsr is in range, try that slot + // first (exact). If it is not a valid LFSR block, trial-and-validate + // candidates from off_zero_list forward — required for older PE32 EXEs + // without the 0x7679 stamp where the expected slot is empty and a loose + // decoded[0]+0xC3 nearest-hit picks the wrong decryptor. Fall back to + // the legacy loose scan only if no candidate trial-decompresses. Native + // DLLs have a smaller eighthStage where off_file_lfsr lands out of range + // and use the same trial-validate scan from just past the cluster (else + // branch). + let lfsr_off = if off_file_lfsr.wrapping_add(96) <= eighth_dsz { + let mut lfsr_off = off_file_lfsr; + let exact = layout::find_lfsr_block( + &self.decompressed, + eighth_start, + eighth_dsz, + off_file_lfsr, + false, + ); + if exact != Some(off_file_lfsr) { + // Prefer trial-and-validate (same as the DLL branch): a loose + // decoded[0]+0xC3 scan can land on coincidental LFSR-shaped + // blocks that decode to a wrong file_ops and scramble every + // compressed block. Observed on older PE32 EXEs without the + // 0x7679 cluster stamp: the expected slot is empty and the + // nearest loose hit is not the real decryptor. + let ci_slot = eighth_start.wrapping_add(off_compressed_info); + let mut scan = off_zero_list; + let mut chosen: Option = None; + let mut considered = 0u32; + while let Some(cand) = layout::find_lfsr_block( + &self.decompressed, + eighth_start, + eighth_dsz, + scan, + false, + ) { + considered = considered.wrapping_add(1); + if self.pe32_file_lfsr_validates(eighth_start.wrapping_add(cand), ci_slot) { + chosen = Some(cand); + break; + } + scan = cand + 1; + } + if let Some(c) = chosen { + if verbose { + println!( + " pe32 fileLFSR via trial-validate: +0x{:X} (expected +0x{:X}, considered {})", + c, off_file_lfsr, considered + ); + } + lfsr_off = c; + } else { + // No candidate trial-decompresses: fail loudly. The old + // "legacy loose scan" picked the nearest LFSR-shaped block + // by offset distance without any validation — that is + // exactly how a wrong file_ops got applied to every data + // block (uncompressed blocks never enter the decompressor), + // producing a plausible but fully wrong image (the PE32 + // .text scramble root cause). Trial-and-validate or error. + return Err(UnpackError::Pe32FileLfsrNotFound); + } + } + lfsr_off + } else { + // Native DLL: the marker-relative off_file_lfsr (EXE-tuned, marker + + // 0x4B4) overshoots the smaller DLL eighthStage, so the exact slot is + // unavailable. A plain forward scan returns the FIRST valid-opcode + // block, but the DLL eighthStage contains coincidental valid-opcode + // blocks that decode to trivial programs (e.g. a constant byte add) + // ahead of the real file decryptor. A wrong file_ops corrupts the + // per-block translate (applied before decompression), so every data + // block fails to decompress. Enumerate every candidate forward and + // keep the first whose decoded file_ops actually decompresses the + // first compressed data block — trial-and-validate, same idea as the + // D1/D2 fixes. Non-DLL (EXE) builds never reach this branch. + let ci_slot = eighth_start.wrapping_add(off_compressed_info); + let mut scan = off_zero_list.wrapping_add(8); + let mut chosen: Option = None; + while let Some(cand) = + layout::find_lfsr_block(&self.decompressed, eighth_start, eighth_dsz, scan, false) + { + if self.pe32_file_lfsr_validates(eighth_start.wrapping_add(cand), ci_slot) { + chosen = Some(cand); + break; + } + scan = cand + 1; + } + chosen.ok_or(UnpackError::Pe32FileLfsrNotFound)? + }; + let file_dec_addr = eighth_start.wrapping_add(lfsr_off); + self.decrypt_data6(file_dec_addr); + let file_ops = generate(&self.decompressed, file_dec_addr).ok_or( + UnpackError::BytecodeGenerationFailed(BytecodeStage::Pe32FileDecryptor), + )?; + + // ---- PE32 metadata: EP and data dirs from info[3] ---- + let test_val = get_u32(&self.decompressed, info3.wrapping_add(0x10)); + let metadata_ep: u32; + let mut metadata_dirs = [0u8; 128]; + if test_val > 0x10000 { + // Layout B + let s = info3.wrapping_add(0x10) as usize; + let backup_meta = self.decompressed[s..s + 0x290].to_vec(); + self.decrypt_data5(info3.wrapping_add(0x10), 0x290); + metadata_ep = get_u32(&self.decompressed, info3.wrapping_add(0x20)); + let d = info3.wrapping_add(0x30) as usize; + metadata_dirs.copy_from_slice(&self.decompressed[d..d + 128]); + self.decompressed[s..s + 0x290].copy_from_slice(&backup_meta); + } else { + // Layout A + let s = info3.wrapping_add(0x40) as usize; + let backup_meta = self.decompressed[s..s + 144].to_vec(); + self.decrypt_data5(info3.wrapping_add(0x40), 144); + metadata_ep = get_u32(&self.decompressed, info3.wrapping_add(0x40)); + let d = info3.wrapping_add(0x50) as usize; + metadata_dirs.copy_from_slice(&self.decompressed[d..d + 128]); + self.decompressed[s..s + 144].copy_from_slice(&backup_meta); + } + + // ---- Zero-out list (runs BEFORE decompression) ---- + let zero_list_addr = eighth_start.wrapping_add(off_zero_list); + let mut zero_ptr = get_u32(&self.decompressed, zero_list_addr); + loop { + self.decrypt_data5(zero_ptr, 16); + let src3 = get_u32(&self.decompressed, zero_ptr); + let s_sz3 = get_u32(&self.decompressed, zero_ptr.wrapping_add(4)); + zero_ptr = zero_ptr.wrapping_add(16); + if s_sz3 == 0 { + break; + } + if src3.wrapping_add(s_sz3) as usize > self.decompressed.len() { + break; + } + for b in &mut self.decompressed[src3 as usize..(src3 + s_sz3) as usize] { + *b = 0; + } + } + + // ---- File data decompression ---- + if verbose { + println!("[6/9] Loading and decompressing file data (PE32)..."); + } + let compress_data_offset = (!get_u32(self.file_data, 0x1080)).wrapping_add(0x1000); + let compressed_info_addr = eighth_start.wrapping_add(off_compressed_info); + let mut compressed_info = get_u32(&self.decompressed, compressed_info_addr); + // Pass 1 (sequential): position-keyed descriptor chain (decrypt_data5), + // terminated by a zero source-size record. + struct Blk { + src: u32, + ssz: u32, + dst: u32, + dsz: u32, + } + let mut blocks: Vec = Vec::new(); + loop { + self.decrypt_data5(compressed_info, 16); + let src2 = get_u32(&self.decompressed, compressed_info); + let s_sz2 = get_u32(&self.decompressed, compressed_info.wrapping_add(4)); + let dst2 = get_u32(&self.decompressed, compressed_info.wrapping_add(8)); + let d_sz2 = get_u32(&self.decompressed, compressed_info.wrapping_add(12)); + compressed_info = compressed_info.wrapping_add(16); + if s_sz2 == 0 { + break; + } + blocks.push(Blk { + src: src2, + ssz: s_sz2, + dst: dst2, + dsz: d_sz2, + }); + } + // Pass 2: independent per-block work over disjoint dst spans (see + // `parallel_for` for how the spans are carved safely). + { + let lut = OpsLut::new(&file_ops); + let clean = &self.file_data; + let ko = self.key_offsets; + let ks_snap = primitives::aes_schedule_snapshot(&self.decompressed, ko[2]) + .ok_or(UnpackError::InvalidAesKeySchedule { offset: ko[2] })?; + let tab_snap = primitives::huffman_table_snapshot(&self.decompressed, ko[0]) + .ok_or(UnpackError::InvalidHuffmanTable { offset: ko[0] })?; + let spans: Vec<(usize, usize)> = blocks + .iter() + .map(|b| { + let s = b.dst as usize; + (s, s + b.ssz.max(b.dsz) as usize) + }) + .collect(); + let do_block = |i: usize, base: usize, span: &mut [u8]| -> Result<(), UnpackError> { + let b = &blocks[i]; + let file_src = b.src.wrapping_add(compress_data_offset) as usize; + let rel = b.dst as usize - base; + let n = b.ssz as usize; + span[rel..rel + n].copy_from_slice(&clean[file_src..file_src + n]); + primitives::aes_decrypt_ks(&ks_snap, span, rel as u32, b.ssz); + lut.map_region(span, rel, n); + if b.ssz != b.dsz { + // decompress reports corruption (after partial writes) via + // its bool; surface it instead of shipping a garbage block. + if !primitives::decompress_tbl( + &tab_snap, span, rel as u32, rel as u32, b.ssz, b.dsz, + ) { + return Err(UnpackError::SectionDecompressionFailed { + pipeline: SectionPipeline::ExePe32, + block: i, + }); + } + } + Ok(()) + }; + super::super::super::parallel::parallel_for( + &mut self.decompressed, + &spans, + 1, + do_block, + )?; + } + + // ---- Section fixup ---- + self.decompressed[..0x1000].copy_from_slice(&self.file_data[..0x1000]); + let opt_hdr_size = get_u16(self.file_data, pe_off.wrapping_add(20)) as u32; + let sec_hdr_table = pe_off.wrapping_add(24).wrapping_add(opt_hdr_size); + let export_va = get_u32( + self.file_data, + pe_off + .wrapping_add(24) + .wrapping_add(opt_hdr_size) + .wrapping_sub(128), + ); + let export_size = get_u32( + self.file_data, + pe_off + .wrapping_add(24) + .wrapping_add(opt_hdr_size) + .wrapping_sub(124), + ); + let mut export_file_off: u32 = 0; + let mut text_off: u32 = 0; + let mut text_size: u32 = 0; + // Walk by NumberOfSections (PE has no zero-VS sentinel; a real + // VirtualSize==0 section would truncate these fixups early), stopping + // at the all-zero padding in case NumberOfSections is overstated. + let num_sections = get_u16(self.file_data, pe_off.wrapping_add(6)) as u32; + for i in 0..num_sections.min(96) { + let sec_hdr = sec_hdr_table.wrapping_add(i.wrapping_mul(40)); + if self.file_data[sec_hdr as usize..sec_hdr as usize + 8] + .iter() + .all(|&b| b == 0) + { + break; + } + let va = get_u32(self.file_data, sec_hdr.wrapping_add(12)); + let sz = get_u32(self.file_data, sec_hdr.wrapping_add(8)); + let f_off = get_u32(self.file_data, sec_hdr.wrapping_add(20)); + let name = section_name(self.file_data, sec_hdr); + if name.starts_with(".text") { + text_size = sz; + text_off = va; + } + if export_size != 0 + && export_va >= va + && export_va.wrapping_add(export_size) <= va.wrapping_add(sz) + { + export_file_off = export_va.wrapping_sub(va).wrapping_add(f_off); + } + write_u32(&mut self.decompressed, sec_hdr.wrapping_add(16), sz); + write_u32(&mut self.decompressed, sec_hdr.wrapping_add(20), va); + if name.starts_with(".idata") { + write_u32( + &mut self.decompressed, + sec_hdr.wrapping_add(36), + 0xC000_0040, + ); + } + } + if export_size != 0 && export_file_off != 0 { + let d = export_va as usize; + let s = export_file_off as usize; + let n = export_size as usize; + self.decompressed[d..d + n].copy_from_slice(&self.file_data[s..s + n]); + } + + // ---- .text decrypt with decrypt_data8 (PE32 auto-detected formula) ---- + // `select_dd8_formula_pe32` returns None when `.text` was not packer-dd8- + // encrypted (native DLLs leave it plaintext); applying dd8 there would + // scramble valid code, so skip it entirely in that case. + if text_size > 0 && text_off > 0 { + if let Some(big) = + layout::select_dd8_formula_pe32(&self.decompressed, text_off, text_size) + { + if verbose { + println!( + "[7/9] Decrypting .text (PE32 dd8, formula={})...", + if big { "0x8000*(page+1)" } else { "page+1" } + ); + } + let num_pages = text_size / 0x1000; + for page in 0..num_pages { + let pk = if big { + 0x8000u32.wrapping_mul(page.wrapping_add(1)) + } else { + page.wrapping_add(1) + }; + let pa = text_off.wrapping_add(page.wrapping_mul(0x1000)); + let mut k = pk; + let rk = k.rotate_right(15); + k = rk; + for bi in 1..256u32 { + let rk = k.rotate_right(15); + let ri = rk.wrapping_add(bi); + k = ri.wrapping_add(bi); + let tidx = + pa.wrapping_add(bi.wrapping_mul(16)).wrapping_add(ri & 0xF) as usize; + self.decompressed[tidx] ^= k as u8; + } + } + } else if verbose { + println!("[7/9] Skipping .text dd8 (already plaintext)..."); + } + } + + // ---- Fix data directories (PE32: data dirs at pe+0x78) ---- + let exe_pe = get_u32(&self.decompressed, 60); + for i in 0..128u32 { + self.decompressed[(exe_pe + 0x78 + i) as usize] = metadata_dirs[i as usize]; + } + // DLL-aware reloc / DllCharacteristics handling. An EXE's packer rebuilds + // the relocation table and clears DllCharacteristics, so the loader needs + // no relocations. A DLL, by contrast, is almost always mapped at a + // non-preferred base, so it MUST keep its base-relocation directory + // (restored above from metadata_dirs) and a valid DllCharacteristics + // (DYNAMIC_BASE) — zeroing them leaves the DLL unrelocatable and its + // imports pinned to the packer stub, so it fails to load (which looks + // like a missing/broken export table). + let is_dll = (get_u16(&self.decompressed, exe_pe.wrapping_add(22)) & 0x2000) != 0; + if !is_dll { + // EXE: clear BaseReloc (index 5 = pe+0xA0) and DllCharacteristics (pe+0x5E). + write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xA0), 0); + write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xA4), 0); + write_u16(&mut self.decompressed, exe_pe.wrapping_add(0x5E), 0); + } else { + // DLL: keep the BaseReloc dir from metadata; ensure DYNAMIC_BASE. + let mut dll_chars = get_u16(&self.decompressed, exe_pe.wrapping_add(0x5E)); + if dll_chars == 0 { + dll_chars = 0x0040; // IMAGE_DLLCHARACTERISTICS_DYNAMIC_BASE + } + write_u16( + &mut self.decompressed, + exe_pe.wrapping_add(0x5E), + dll_chars as u32, + ); + } + + // ---- TLS directory reconstruction (PE32: index 9 = pe+0xC0) ---- + let tls_dir_rva = get_u32(&self.decompressed, exe_pe.wrapping_add(0xC0)); + let tls_dir_sz = get_u32(&self.decompressed, exe_pe.wrapping_add(0xC4)); + if tls_dir_rva > 0 + && tls_dir_sz >= 24 + && (tls_dir_rva as usize + 24) <= self.decompressed.len() + { + let all_zero = (0..6u32) + .all(|i| get_u32(&self.decompressed, tls_dir_rva.wrapping_add(i * 4)) == 0); + if all_zero { + let image_base = get_u32(&self.decompressed, exe_pe.wrapping_add(52)); + // Prefer the module's real TLS directory, which survives in the + // loader stub's plaintext `.rdata`/`.tls`. Only when the stub + // cannot supply one does a placeholder get synthesized: it keeps + // the image loadable, but drops the initialized TLS template, + // `_tls_index` and the TLS callback array, so any module that + // actually uses `thread_local` faults once it runs. + if !self.restore_pe32_tls_from_stub(pe_off, tls_dir_rva, image_base) { + let mut tls_sec_va: u32 = 0; + let mut data_sec_va: u32 = 0; + let mut data_sec_sz: u32 = 0; + let sh = exe_pe + .wrapping_add(24) + .wrapping_add(get_u16(&self.decompressed, exe_pe.wrapping_add(20)) as u32); + let ns = get_u16(&self.decompressed, exe_pe.wrapping_add(6)) as u32; + for i in 0..ns { + let s = sh.wrapping_add(i * 40); + let nm = get_string_to_null(&self.decompressed, s); + let va = get_u32(&self.decompressed, s.wrapping_add(12)); + let sz = get_u32(&self.decompressed, s.wrapping_add(16)); + if nm.starts_with(".tls") { + tls_sec_va = va; + } + if nm.starts_with(".data") { + data_sec_va = va; + data_sec_sz = sz; + } + } + if tls_sec_va > 0 && data_sec_va > 0 { + let start_raw = image_base.wrapping_add(tls_sec_va); + let end_raw = start_raw; + let idx_addr = image_base + .wrapping_add(data_sec_va) + .wrapping_add(data_sec_sz) + .wrapping_sub(16); + let cb_addr = image_base + .wrapping_add(data_sec_va) + .wrapping_add(data_sec_sz) + .wrapping_sub(8); + let scratch = (data_sec_va + data_sec_sz - 16) as usize; + for b in &mut self.decompressed[scratch..scratch + 16] { + *b = 0; + } + write_u32(&mut self.decompressed, tls_dir_rva, start_raw); + write_u32(&mut self.decompressed, tls_dir_rva.wrapping_add(4), end_raw); + write_u32( + &mut self.decompressed, + tls_dir_rva.wrapping_add(8), + idx_addr, + ); + write_u32( + &mut self.decompressed, + tls_dir_rva.wrapping_add(12), + cb_addr, + ); + write_u32(&mut self.decompressed, tls_dir_rva.wrapping_add(16), 0); + write_u32( + &mut self.decompressed, + tls_dir_rva.wrapping_add(20), + 0x30_0000, + ); + } else { + write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xC0), 0); + write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xC4), 0); + } + } + } + } + + // ---- Import table (PE32, 4-byte thunks) ---- + if verbose { + println!("[8/9] Decrypting import strings (PE32)..."); + } + let import_table_addr = eighth_start.wrapping_add(off_import_table); + let mut import_table_ptr = get_u32(&self.decompressed, import_table_addr); + let mut idt_size = get_u32(&self.decompressed, import_table_addr.wrapping_add(4)); + + let metadata_import_rva = get_u32(&metadata_dirs, 8); + let metadata_import_size = get_u32(&metadata_dirs, 12); + let dlen = self.decompressed.len() as u32; + + let mut eighth_import_valid = false; + if 0 < import_table_ptr && import_table_ptr < dlen && 0 < idt_size && idt_size < 0x10000 { + let test_name = if import_table_ptr + 20 <= dlen { + get_u32(&self.decompressed, import_table_ptr.wrapping_add(12)) + } else { + 0 + }; + let test_ilt = if import_table_ptr + 4 <= dlen { + get_u32(&self.decompressed, import_table_ptr) + } else { + 0 + }; + if 0x1000 < test_name && test_name < dlen && 0x1000 < test_ilt && test_ilt < dlen { + eighth_import_valid = true; + } + } + let mut metadata_import_valid = false; + if 0x1000 < metadata_import_rva && metadata_import_rva < dlen.wrapping_sub(20) { + let test_name2 = get_u32(&self.decompressed, metadata_import_rva.wrapping_add(12)); + let test_ilt2 = get_u32(&self.decompressed, metadata_import_rva); + if 0x1000 < test_name2 && test_name2 < dlen && 0x1000 < test_ilt2 && test_ilt2 < dlen { + metadata_import_valid = true; + } + } + if metadata_import_valid + && (!eighth_import_valid || metadata_import_rva != import_table_ptr) + { + import_table_ptr = metadata_import_rva; + idt_size = metadata_import_size; + } + + if 0 < import_table_ptr && import_table_ptr < dlen && 0 < idt_size && idt_size < 0x10000 { + let mut idt_pos = import_table_ptr; + let idt_end = import_table_ptr.wrapping_add(idt_size); + while idt_pos.wrapping_add(20) <= idt_end { + let ilt_rva = get_u32(&self.decompressed, idt_pos); + let name_rva = get_u32(&self.decompressed, idt_pos.wrapping_add(12)); + let iat_rva = get_u32(&self.decompressed, idt_pos.wrapping_add(16)); + if ilt_rva == 0 && name_rva == 0 && iat_rva == 0 { + break; + } + if 0 < name_rva && name_rva < dlen { + self.decrypt_data7(name_rva, name_rva as u8); + } + let thunk_base = if 0 < ilt_rva && ilt_rva < dlen { + ilt_rva + } else { + iat_rva + }; + if 0 < thunk_base && thunk_base < dlen.wrapping_sub(4) { + let mut thunk_pos = thunk_base; + while thunk_pos.wrapping_add(4) <= dlen { + let thunk_val = get_u32(&self.decompressed, thunk_pos); + if thunk_val == 0 { + break; + } + if thunk_val & 0x8000_0000 == 0 && thunk_val.wrapping_add(2) < dlen { + self.decrypt_data7(thunk_val.wrapping_add(2), thunk_val as u8); + write_u16(&mut self.decompressed, thunk_val, 0); + } + thunk_pos = thunk_pos.wrapping_add(4); + } + } + idt_pos = idt_pos.wrapping_add(20); + } + } + + // Update PE header: Import directory (index 1 = pe+0x80), clear IAT + // directory (index 12 = pe+0xD8). + write_u32( + &mut self.decompressed, + exe_pe.wrapping_add(0x80), + import_table_ptr, + ); + write_u32(&mut self.decompressed, exe_pe.wrapping_add(0x84), idt_size); + write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xD8), 0); + write_u32(&mut self.decompressed, exe_pe.wrapping_add(0xDC), 0); + + // ---- EP (from metadata) ---- + if metadata_ep > 0 { + write_u32(&mut self.decompressed, exe_pe.wrapping_add(40), metadata_ep); + } else { + let real_ep = get_u32(self.file_data, pe_off.wrapping_add(40)); + write_u32(&mut self.decompressed, exe_pe.wrapping_add(40), real_ep); + } + + // ---- Output transforms ---- + if verbose { + println!("[9/9] Rebuilding PE file layout (PE32)..."); + } + let mut out = std::mem::take(&mut self.decompressed); + // kmiat import relocation is an EXE-only fixup: it discards the original + // import directory in favour of the loader-written IAT stub. A DLL keeps + // its real import table (restored above from metadata), so skip kmiat for + // DLLs (`!is_dll` guard). + let is_dll = (get_u16(&out, pe_off.wrapping_add(22)) & 0x2000) != 0; + if !is_dll && !layout::pe32_imports_already_match_idata_layout(&mut out, pe_off) { + layout::move_pe32_imports_to_kmiat(&mut out, pe_off); + } + let compact = layout::compact_memory_image_to_pe(&out, pe_off) + .ok_or(UnpackError::Pe32OutputLayoutInvalid)?; + Ok(compact) + } +} diff --git a/src/unpacker/integrity.rs b/senbei-pe/src/engine/integrity.rs similarity index 85% rename from src/unpacker/integrity.rs rename to senbei-pe/src/engine/integrity.rs index b49cecb..c768511 100644 --- a/src/unpacker/integrity.rs +++ b/senbei-pe/src/engine/integrity.rs @@ -77,6 +77,49 @@ fn rva_to_off(secs: &[Section], file_len: usize, rva: u32, need: u32) -> Option< None } +fn is_executable_rva(secs: &[Section], rva: u32) -> bool { + secs.iter().any(|section| { + let span = section.vsize.max(section.raw_size); + rva >= section.va + && rva < section.va.wrapping_add(span) + && (section.chars & 0x2000_0000) != 0 + }) +} + +fn check_common_entry_branches( + stub: &[u8], + ep: u32, + secs: &[Section], + report: &mut IntegrityReport, +) { + if stub.len() < 18 + || stub[0..3] != [0x48, 0x83, 0xEC] + || stub[4] != 0xE8 + || stub[9..12] != [0x48, 0x83, 0xC4] + || stub[12] != stub[3] + || stub[13] != 0xE9 + { + return; + } + for (name, rel_off, instruction_len) in [("call", 5usize, 9i64), ("jump", 14usize, 18i64)] { + let rel = i32::from_le_bytes([ + stub[rel_off], + stub[rel_off + 1], + stub[rel_off + 2], + stub[rel_off + 3], + ]) as i64; + let target = i64::from(ep) + instruction_len + rel; + let valid = u32::try_from(target) + .ok() + .is_some_and(|rva| is_executable_rva(secs, rva)); + if !valid { + report.issues.push(format!( + "entry point {name} target 0x{target:X} is outside executable sections (DD8 selection is likely wrong)" + )); + } + } +} + /// Inspect an unpacked PE image and report any defect that would make the OS /// loader fault at runtime. `out` is the bytes the unpacker produced. pub fn check(out: &[u8]) -> IntegrityReport { @@ -253,6 +296,9 @@ pub fn check(out: &[u8]) -> IntegrityReport { "entry point RVA 0x{ep:X} is not in an executable section" )); } + if let Some(entry_stub) = out.get(off as usize..off as usize + 18) { + check_common_entry_branches(entry_stub, ep, &secs, &mut r); + } } } } @@ -363,3 +409,40 @@ fn looks_like_dll_name(d: &[u8], off: u32) -> bool { } d[start..end].iter().all(|&b| (0x20..0x7F).contains(&b)) } + +#[cfg(test)] +mod tests { + use super::*; + + fn executable_text() -> Vec
{ + vec![Section { + va: 0x1000, + vsize: 0x4000, + raw_ptr: 0x1000, + raw_size: 0x4000, + chars: 0x6000_0020, + }] + } + + #[test] + fn common_entry_stub_rejects_out_of_image_branches() { + let stub = [ + 0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x41, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9, + 0x7A, 0xFE, 0x54, 0xFF, + ]; + let mut report = IntegrityReport::default(); + check_common_entry_branches(&stub, 0x1264, &executable_text(), &mut report); + assert_eq!(report.issues.len(), 2); + } + + #[test] + fn common_entry_stub_accepts_executable_branches() { + let stub = [ + 0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x00, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9, + 0x7A, 0xFE, 0xFF, 0xFF, + ]; + let mut report = IntegrityReport::default(); + check_common_entry_branches(&stub, 0x1264, &executable_text(), &mut report); + assert!(report.ok()); + } +} diff --git a/senbei-pe/src/engine/layout.rs b/senbei-pe/src/engine/layout.rs new file mode 100644 index 0000000..a9ef00a --- /dev/null +++ b/senbei-pe/src/engine/layout.rs @@ -0,0 +1,14 @@ +//! Internal PE layout discovery and image reconstruction. + +mod dd8; +mod discovery; +mod image; + +pub(super) use dd8::{select_dd8_formula_pe32, select_dd8_shift}; +pub(super) use discovery::{ + discover_eighth_slots, find_bytecode_offset, find_lfsr_block, find_str_pos, find_tbl_pe32, + find_v_after_pad, find_v4_offset, get_string_to_null, section_name, trial_decrypt5_u32, +}; +pub(super) use image::{ + compact_memory_image_to_pe, move_pe32_imports_to_kmiat, pe32_imports_already_match_idata_layout, +}; diff --git a/senbei-pe/src/engine/layout/dd8.rs b/senbei-pe/src/engine/layout/dd8.rs new file mode 100644 index 0000000..5d9ef08 --- /dev/null +++ b/senbei-pe/src/engine/layout/dd8.rs @@ -0,0 +1,609 @@ +//! Validation-driven selection for per-page text transforms. + +use super::discovery::trial_decrypt5_u32; + +/// PE32 `.text` dd8 key-formula selection with a skip decision. The packer keys +/// the per-page XOR either with `page+1` or `0x8000*(page+1)`; the formula is +/// not recorded. Replays the dd8 page pass on a scratch copy of sample pages +/// (25/50/75% of `.text`) under each formula and counts how many positions +/// decode to `0xCC` (int3 padding). +/// +/// Returns `Some(true)` for the `0x8000*(page+1)` formula, `Some(false)` for +/// `page+1`, or `None` when `.text` must NOT be dd8-decrypted at all. The packer +/// dd8-encrypts `.text` on EXEs (so unpacking must replay it) but leaves a native +/// DLL's `.text` plaintext; replaying dd8 there scrambles ~1 byte per 16-byte +/// block. The decision: dd8 only *restores* int3 padding when `.text` was +/// genuinely encrypted, so apply it only when the chosen formula's whole-page +/// 0xCC count rises *clearly* above the no-dd8 baseline; otherwise skip. +/// +/// "Clearly" matters: dd8 XORs 255 positions per page with pseudo-random bytes, +/// so on an already-plaintext `.text` it manufactures ~1 spurious `0xCC` per +/// sampled page for free (255/256 expected). A bare `best > baseline` test is +/// therefore biased towards *applying* dd8 on exactly the inputs that must skip +/// it — and a wrongly-applied dd8 is silent: it scrambles ~1 byte per 16 with no +/// error and nothing downstream (not even `integrity::check`, which only reads +/// 16 bytes at the entry point) notices. The [`MIN_DD8_NET_GAIN`] floor below is +/// the PE32 counterpart of the margin+floor `select_dd8_shift` already applies +/// on PE32+ for the same failure mode. +pub fn select_dd8_formula_pe32(data: &[u8], text_off: u32, text_size: u32) -> Option { + let num_pages_total = text_size / 0x1000; + let mut sample_pages: Vec = Vec::new(); + for frac in [0.25f64, 0.5, 0.75] { + let pg = (num_pages_total as f64 * frac) as u32; + if pg > 0 && pg < num_pages_total { + sample_pages.push(pg); + } + } + if sample_pages.is_empty() && num_pages_total > 1 { + sample_pages.push(num_pages_total / 2); + } + let score = |big: bool| -> i64 { + let mut total = 0i64; + for &sp in &sample_pages { + let pg_off = (text_off + sp * 0x1000) as usize; + if pg_off + 0x1000 > data.len() { + continue; + } + let mut buf = [0u8; 0x1000]; + buf.copy_from_slice(&data[pg_off..pg_off + 0x1000]); + let pk = if big { + 0x8000u32.wrapping_mul(sp.wrapping_add(1)) + } else { + sp.wrapping_add(1) + }; + let mut k = pk; + let rk = k.rotate_right(15); + k = rk; + for bi in 1..256u32 { + let rk = k.rotate_right(15); + let ri = rk.wrapping_add(bi); + k = ri.wrapping_add(bi); + let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize; + if tidx < buf.len() { + buf[tidx] ^= k as u8; + } + } + total += buf.iter().filter(|&&b| b == 0xCC).count() as i64; + } + total + }; + let s_small = score(false); + let s_big = score(true); + // Baseline: whole-page 0xCC over the same sample pages with NO dd8. dd8 only + // rewrites 255 bytes per page, so comparing the chosen formula's whole-page + // 0xCC against this baseline reveals whether dd8 *restores* int3 padding + // (count rises -> .text was packer-encrypted, apply) or merely scrambles + // already-plaintext code (count falls -> native-DLL .text left intact, skip). + let mut baseline: i64 = 0; + for &sp in &sample_pages { + let pg_off = (text_off + sp * 0x1000) as usize; + if pg_off + 0x1000 > data.len() { + continue; + } + baseline += data[pg_off..pg_off + 0x1000] + .iter() + .filter(|&&b| b == 0xCC) + .count() as i64; + } + let big = s_big > s_small; + let best = s_small.max(s_big); + // Minimum net 0xCC gain over the baseline before dd8 is applied. Noise on an + // already-plaintext `.text` is ~1 manufactured 0xCC per sampled page (3 pages + // -> ~3); every corpus build that genuinely needs dd8 gains +154 or more + // (observed +154 and +312), and the one native DLL that must skip scores -18. + // A floor of 32 sits ~10x above the noise and ~5x below the smallest true + // positive, so it changes no existing decision. + const MIN_DD8_NET_GAIN: i64 = 32; + let apply = best.saturating_sub(baseline) >= MIN_DD8_NET_GAIN; + if std::env::var("SEL_DIAG").is_ok() { + eprintln!( + "SEL pe32 dd8 s_small={} s_big={} baseline={} gain={} big={} apply={}", + s_small, + s_big, + baseline, + best - baseline, + big, + apply + ); + } + // When no interior pages could be sampled (tiny .text) we cannot measure the + // effect; preserve the historical behavior of applying dd8. + if sample_pages.is_empty() || apply { + Some(big) + } else { + None + } +} + +// --------------------------------------------------------------------------- +// dd8 page-XOR shift selection. +// +// The packer scrambles ~1 byte per 16-byte block of .text via decrypt_data8, +// keyed by `page_idx << shift` (absolute page index = text_va >> 12). Observed +// shifts are 0 and 15. The shift is NOT stored in any header/config field, so +// the decision must be validated against the resulting .text content. +// +// A recognised CRT entry stub is the strongest oracle: decode skip/0/15 and +// require both of its direct rel32 branches to land in executable .text. This +// includes the call/jump displacement bytes themselves; an older entry oracle +// wildcarded those bytes and could accept a stub whose opcodes looked right but +// whose branch targets were outside the image. +// +// Other entry shapes fall back to padding statistics over a few sample pages +// (head/tail margin skipped: entry/exit regions have atypical padding density). +// The primary signal is a *structural* fingerprint: the MSVC function-end +// padding pattern, a 0xC3 RET opcode followed by a run of >= 4 0xCC int3 bytes. +// dd8 XORs one pseudo-random byte per 16-byte block, so an already-plaintext +// page keeps its padding runs only under "no dd8", while a packer-encrypted +// page restores them only under the correct shift — a wrong candidate destroys +// every run it touches and essentially never manufactures a RET followed by a +// long int3 run by chance. This separates the states far more cleanly than a +// bare 0xCC count, which a wrong candidate inflates for free (~255 coincidences +// per page at p=1/256). +// +// When no candidate produces any RET-anchored padding (sampled pages with +// dense code and no padded epilogues), the fingerprint is silent, so the +// decision falls back to the older mutated-position 0xCC count. Both signals +// use the same decision rule: a candidate must beat the no-dd8 baseline by a +// clear margin AND an absolute floor, otherwise dd8 is skipped — a wrongly +// applied dd8 scrambles ~1 byte per 16 with no error surfaced downstream. +// --------------------------------------------------------------------------- +pub fn select_dd8_shift(data: &[u8], text_va: u32, text_size: u32, info3: u32) -> u32 { + if let Some((shift, scores)) = select_dd8_by_entry_stub(data, text_va, text_size, info3) { + if std::env::var("SEL_DIAG").is_ok() { + eprintln!( + "SEL dd8 entry best_shift={} none={} s0={} s15={}", + shift, scores[0], scores[1], scores[2] + ); + } + return shift; + } + let num_pages_total = text_size >> 12; + // Fewer than two pages: nothing meaningful to sample; preserve the + // historical behavior (shift 0 — the dd8 loop is empty or single-page). + if num_pages_total < 2 { + return 0; + } + let text_off = text_va as usize; + + // Sample up to 4 pages, skipping a head/tail margin (entry/exit regions + // have atypical padding density). Small .text: sample every page. + let mut sample_pages: Vec = Vec::new(); + if num_pages_total <= 4 { + sample_pages.extend(0..num_pages_total); + } else { + let margin = (num_pages_total / 8).max(1); + let lo = margin; + let hi = num_pages_total - margin; + if hi <= lo { + sample_pages.extend(0..num_pages_total); + } else { + let step = ((hi - lo) / 4).max(1); + let mut i = 0; + while i < 4 { + let p = lo + i * step; + if p < num_pages_total { + sample_pages.push(p); + } + i += 1; + } + } + } + if sample_pages.is_empty() { + return 0; + } + + let abs_base = text_va >> 12; + // Require a clear 2x margin over the already-plaintext baseline AND an + // absolute floor. The 2x test alone trips on noise when the counts are + // tiny: an external-companion DLL whose .text is already plaintext scores + // s15=4 vs none=1 — a spurious 4x — and gets dd8 wrongly applied, + // corrupting ~1 byte per 16. The floor rejects that noise while sitting + // far below every genuinely-encrypted build's score. + const MIN_DD8_HITS: u32 = 8; + let margin_pick = |none: u32, s0: u32, s15: u32| -> u32 { + let mut best_score = none; + let mut best_shift = 99u32; // 99 == skip dd8 + for (shift, hits) in [(0u32, s0), (15u32, s15)] { + if hits > best_score { + best_score = hits; + best_shift = shift; + } + } + if best_shift != 99 && (best_score < none * 2 || best_score < MIN_DD8_HITS) { + best_shift = 99; + } + best_shift + }; + + // Primary: RET+int3 padding fingerprint. The fingerprint is diluted across + // the whole page (dd8 touches only 255 of 4096 bytes, so even an encrypted + // page keeps most of its padding runs), so instead of the fallback's 2x + // margin the gate is a *positive delta* over the no-dd8 baseline: on an + // already-plaintext .text each wrong shift destroys runs (scores below the + // baseline), while the correct shift on an encrypted page restores them + // (scores above it). The floor on the delta rejects noise-level gains. + let r_none = fingerprint_score(data, text_off, abs_base, &sample_pages, None); + let r0 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(0)); + let r15 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(15)); + // Fallback: mutated-position 0xCC count, for pages whose code has no + // RET-anchored padding at all (the fingerprint is silent there). + let (none_hits, s0, s15); + let best_shift = if r_none != 0 || r0 != 0 || r15 != 0 { + none_hits = 0; + s0 = 0; + s15 = 0; + let mut best_score = r_none; + let mut shift = 99u32; + for (s, score) in [(0u32, r0), (15u32, r15)] { + if score > best_score { + best_score = score; + shift = s; + } + } + if shift != 99 && best_score.saturating_sub(r_none) < MIN_DD8_HITS { + shift = 99; + } + shift + } else { + none_hits = score_dd8_baseline(data, text_off, &sample_pages); + s0 = score_dd8_shift(data, text_off, text_va, &sample_pages, 0); + s15 = score_dd8_shift(data, text_off, text_va, &sample_pages, 15); + margin_pick(none_hits, s0, s15) + }; + if std::env::var("SEL_DIAG").is_ok() { + eprintln!( + "SEL dd8 best_shift={} fp=({},{},{}) cc=({},{},{}) samples={:?}", + best_shift, r_none, r0, r15, none_hits, s0, s15, sample_pages + ); + } + best_shift +} + +/// Minimum 0xCC run length after a RET for the run to count as MSVC +/// function-end padding. +const MIN_CC_RUN: u32 = 4; + +/// Total length of MSVC function-end padding runs in a page: each 0xC3 byte +/// followed by >= [`MIN_CC_RUN`] 0xCC bytes contributes the run length. +fn ret_int3_score(page: &[u8]) -> u32 { + let mut total = 0u32; + let mut i = 0; + while i < page.len() { + if page[i] == 0xC3 { + let mut j = i + 1; + while j < page.len() && page[j] == 0xCC { + j += 1; + } + let run = (j - i - 1) as u32; + if run >= MIN_CC_RUN { + total += run; + } + i = j; + } else { + i += 1; + } + } + total +} + +/// Replay the dd8 page-XOR in place on one sample page. +fn dd8_apply(buf: &mut [u8; 0x1000], abs_page: u32, shift: u32) { + let mut key = abs_page << shift; + for bi in 0..256u32 { + let mixed = key.rotate_right(15).wrapping_add(bi); + key = mixed.wrapping_add(bi); + // The packer's dd8 loop does not XOR block i=0 (see decrypt_data8). + if bi == 0 { + continue; + } + let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize; + buf[tidx] ^= key as u8; + } +} + +/// Sum the RET+int3 fingerprint over the sample pages for one candidate +/// (`None` = the no-dd8 baseline, page as-is). +fn fingerprint_score( + data: &[u8], + text_off: usize, + abs_base: u32, + sample_pages: &[u32], + shift: Option, +) -> u32 { + let mut total = 0u32; + for &sp in sample_pages { + let pg_off = text_off + (sp as usize) * 0x1000; + if pg_off + 0x1000 > data.len() { + continue; + } + let mut page = [0u8; 0x1000]; + page.copy_from_slice(&data[pg_off..pg_off + 0x1000]); + if let Some(sh) = shift { + dd8_apply(&mut page, abs_base.wrapping_add(sp), sh); + } + total += ret_int3_score(&page); + } + total +} + +/// Select DD8 from the common CRT entry stub when its direct call and jump +/// provide a stronger oracle than sparse padding statistics. The candidate is +/// accepted only when it is the sole one whose two branch targets stay inside +/// `.text`; unrecognised entry code falls through to the padding selector. +fn select_dd8_by_entry_stub( + data: &[u8], + text_va: u32, + text_size: u32, + info3: u32, +) -> Option<(u32, [u8; 3])> { + for entry in entry_candidates(data, text_va, text_size, info3) { + let [Some(none), Some(s0), Some(s15)] = [None, Some(0), Some(15)] + .map(|shift| entry_stub_branch_score(data, text_va, text_size, entry, shift)) + else { + continue; + }; + let scores = [none, s0, s15]; + let best = scores.iter().copied().max()?; + if best == 2 && scores.iter().filter(|&&score| score == best).count() == 1 { + let index = scores.iter().position(|&score| score == best)?; + return Some(([99, 0, 15][index], scores)); + } + } + None +} + +fn entry_candidates(data: &[u8], text_va: u32, text_size: u32, info3: u32) -> Vec { + let text_end = text_va.saturating_add(text_size); + let mut entries = Vec::with_capacity(3); + if let Some(pe) = read_u32(data, 0x3C) + && let Some(entry) = pe.checked_add(40).and_then(|offset| read_u32(data, offset)) + && (text_va..text_end).contains(&entry) + { + entries.push(entry); + } + for metadata_off in [32u32, 64] { + let Some(end) = info3 + .checked_add(metadata_off) + .and_then(|offset| offset.checked_add(8)) + else { + continue; + }; + if end as usize > data.len() { + continue; + } + let entry = trial_decrypt5_u32(data, info3 + metadata_off); + let image_base = trial_decrypt5_u32(data, info3 + metadata_off + 4); + if image_base == info3 && (text_va..text_end).contains(&entry) && !entries.contains(&entry) + { + entries.push(entry); + } + } + entries +} + +fn entry_stub_branch_score( + data: &[u8], + text_va: u32, + text_size: u32, + entry: u32, + shift: Option, +) -> Option { + let text_end = text_va.checked_add(text_size)?; + if entry < text_va || entry.checked_add(18)? > text_end { + return None; + } + + let mut stub = [0u8; 18]; + for (offset, byte) in stub.iter_mut().enumerate() { + *byte = dd8_candidate_byte(data, entry + offset as u32, shift)?; + } + if stub[0..3] != [0x48, 0x83, 0xEC] + || stub[4] != 0xE8 + || stub[9..12] != [0x48, 0x83, 0xC4] + || stub[12] != stub[3] + || stub[13] != 0xE9 + { + return None; + } + + let call_rel = i32::from_le_bytes(stub[5..9].try_into().ok()?) as i64; + let jump_rel = i32::from_le_bytes(stub[14..18].try_into().ok()?) as i64; + let call_target = i64::from(entry) + 9 + call_rel; + let jump_target = i64::from(entry) + 18 + jump_rel; + let in_text = |target: i64| target >= i64::from(text_va) && target < i64::from(text_end); + Some(u8::from(in_text(call_target)) + u8::from(in_text(jump_target))) +} + +fn dd8_candidate_byte(data: &[u8], rva: u32, shift: Option) -> Option { + let mut byte = *data.get(rva as usize)?; + let Some(shift) = shift else { + return Some(byte); + }; + let page = rva >> 12; + let block = (rva & 0xFFF) >> 4; + let mut key = page << shift; + for index in 0..=block { + let mixed = key.rotate_right(15).wrapping_add(index); + key = mixed.wrapping_add(index); + if index != 0 { + let target = (page << 12) + .wrapping_add(index << 4) + .wrapping_add(mixed & 0xF); + if target == rva { + byte ^= key as u8; + } + } + } + Some(byte) +} + +fn read_u32(data: &[u8], offset: u32) -> Option { + let start = offset as usize; + let bytes = data.get(start..start.checked_add(4)?)?; + Some(u32::from_le_bytes(bytes.try_into().ok()?)) +} + +// Baseline: count int3 pads already present at the first byte of each 16-byte +// block, i.e. the positions dd8 would target if its in-block offset were 0. +fn score_dd8_baseline(data: &[u8], text_off: usize, sample_pages: &[u32]) -> u32 { + let mut hits = 0u32; + for &sp in sample_pages { + let pg_off = text_off + (sp as usize) * 0x1000; + if pg_off + 0x1000 > data.len() { + continue; + } + for bi in 1..256usize { + if data[pg_off + bi * 16] == 0xCC { + hits += 1; + } + } + } + hits +} + +// Replay decrypt_data8 on each sample page under `shift` and count how many of +// the 255 mutated positions decode to 0xCC. +fn score_dd8_shift( + data: &[u8], + text_off: usize, + text_va: u32, + sample_pages: &[u32], + shift: u32, +) -> u32 { + let abs_base = text_va >> 12; + let mut hits = 0u32; + for &sp in sample_pages { + let pg_off = text_off + (sp as usize) * 0x1000; + if pg_off + 0x1000 > data.len() { + continue; + } + let abs_page = abs_base.wrapping_add(sp); + let mut key = abs_page << shift; + for bi in 0..256u32 { + let mixed = key.rotate_right(15).wrapping_add(bi); + key = mixed.wrapping_add(bi); + if bi == 0 { + continue; + } + let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize; + if tidx < 0x1000 { + let mutated = data[pg_off + tidx] ^ (key as u8); + if mutated == 0xCC { + hits += 1; + } + } + } + } + hits +} + +#[cfg(test)] +mod tests { + use super::*; + + fn entry_stub_fixture() -> Vec { + let mut data = vec![0u8; 0x5000]; + data[0x3C..0x40].copy_from_slice(&0x100u32.to_le_bytes()); + data[0x128..0x12C].copy_from_slice(&0x1264u32.to_le_bytes()); + data[0x1264..0x1276].copy_from_slice(&[ + 0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x00, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9, + 0x7A, 0xFE, 0xFF, 0xFF, + ]); + data + } + + fn apply_dd8_page(data: &mut [u8], page_rva: u32, shift: u32) { + let mut key = (page_rva >> 12) << shift; + for index in 0..256u32 { + let mixed = key.rotate_right(15).wrapping_add(index); + key = mixed.wrapping_add(index); + if index == 0 { + continue; + } + let target = page_rva.wrapping_add(index << 4).wrapping_add(mixed & 0xF) as usize; + data[target] ^= key as u8; + } + } + + #[test] + fn entry_stub_selects_plaintext_and_both_dd8_shifts() { + let plain = entry_stub_fixture(); + assert_eq!(select_dd8_shift(&plain, 0x1000, 0x4000, 0), 99); + + for expected in [0u32, 15] { + let mut encrypted = plain.clone(); + apply_dd8_page(&mut encrypted, 0x1000, expected); + assert_eq!(select_dd8_shift(&encrypted, 0x1000, 0x4000, 0), expected); + } + } + + /// Seed the first `count` dd8-targeted positions of each sampled page with + /// the byte that decodes to `0xCC` under the `page+1` formula — i.e. an + /// encrypted `.text` whose plaintext is int3 padding. Positions whose key + /// byte would make the *ciphertext* itself `0xCC` are skipped so the + /// fixture contains no `0xCC` at all and every post-dd8 `0xCC` is a genuine + /// gain over a zero baseline. + fn seed_dd8_int3(data: &mut [u8], text_off: u32, pages: &[u32], count: u32) { + for &sp in pages { + let pg_off = (text_off + sp * 0x1000) as usize; + let mut k = sp.wrapping_add(1); + k = k.rotate_right(15); + let mut planted = 0u32; + for bi in 1..256u32 { + let ri = k.rotate_right(15).wrapping_add(bi); + k = ri.wrapping_add(bi); + if planted >= count { + continue; + } + let ct = 0xCCu8 ^ (k as u8); + if ct == 0xCC { + continue; + } + let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize; + data[pg_off + tidx] = ct; + planted += 1; + } + } + } + + /// Review regression: a near-plaintext `.text` must NOT be dd8-decrypted. + /// dd8 XORs 255 positions per page with pseudo-random bytes, so it + /// manufactures a few `0xCC` for free — under the old bare + /// `best > baseline` test any positive gain was enough to "apply" dd8 and + /// scramble ~1 byte per 16 of a native DLL's already-plaintext code, + /// silently (nothing downstream, including the integrity check, notices). + /// Here the gain is real but small; the floor must still reject it. + #[test] + fn pe32_dd8_skips_text_whose_gain_is_only_noise_sized() { + let text_off: u32 = 0x1000; + let text_size: u32 = 8 * 0x1000; + let mut data = vec![0u8; (text_off + text_size) as usize]; + seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 5); + assert!( + !data.contains(&0xCC), + "fixture must have a zero 0xCC baseline" + ); + assert_eq!( + select_dd8_formula_pe32(&data, text_off, text_size), + None, + "a gain this small is indistinguishable from dd8's own noise" + ); + } + + /// Control for the above: a `.text` whose dd8 pass restores a large amount + /// of int3 padding clears the floor and is decrypted. Same fixture shape, + /// only the amount of restored padding differs. + #[test] + fn pe32_dd8_applies_when_padding_is_restored() { + let text_off: u32 = 0x1000; + let text_size: u32 = 8 * 0x1000; + let mut data = vec![0u8; (text_off + text_size) as usize]; + seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 255); + assert_eq!( + select_dd8_formula_pe32(&data, text_off, text_size), + Some(false), + "encrypted .text must be decrypted with the page+1 formula" + ); + } +} diff --git a/senbei-pe/src/engine/layout/discovery.rs b/senbei-pe/src/engine/layout/discovery.rs new file mode 100644 index 0000000..830b48a --- /dev/null +++ b/senbei-pe/src/engine/layout/discovery.rs @@ -0,0 +1,507 @@ +//! Structural locators for protected PE stages. + +use senbei_crypto::primitives::{get_u32, lfsr_keystream}; + +/// Find the 4-byte v_val that follows the LAST occurrence of `48 EB 01 B9` +/// (REX.W jmp+1; mov ecx,imm32) plus any 0xCC padding. Used to locate +/// stage4's accum2 seed. Works across builds even when API-name anchors are +/// absent. +pub fn find_v_after_pad(data: &[u8], base: u32, len: u32) -> Option { + let start = base as usize; + let end = (base.saturating_add(len)) as usize; + if end > data.len() { + return None; + } + let sig = [0x48u8, 0xEB, 0x01, 0xB9]; + let slice = &data[start..end]; + // last occurrence + let mut last = None; + let mut i = 0usize; + while i + sig.len() <= slice.len() { + if slice[i..i + sig.len()] == sig { + last = Some(i); + } + i += 1; + } + let pos = last?; + // skip CCs after the `48 EB 01 B9` + let mut after = pos + sig.len(); + while after < slice.len() && slice[after] == 0xCC { + after += 1; + } + if after + 4 > slice.len() { + return None; + } + Some((start + after) as u32) +} + +/// Predict the 4 bytes that DecryptData5(va, size) would produce at va+0..va+4 +/// without mutating the buffer. The cipher's per-byte transform depends only +/// on the byte itself and the low 8 bits of (va+i), with no cross-byte state, +/// so each byte can be decrypted in isolation. Used to detect the EP/DD layout +/// offset before committing to the actual call. +pub fn trial_decrypt5_u32(data: &[u8], va: u32) -> u32 { + let mut out = [0u8; 4]; + for i in 0..4u32 { + let b3 = data[(va + i) as usize]; + let b = (va + i) as u8; + let b2 = b.wrapping_add(1); + let b4 = b3.rotate_left(2) ^ b2; + let b5 = b4.rotate_left(2) ^ b; + out[i as usize] = b5.rotate_left(2); + } + u32::from_le_bytes(out) +} + +/// Scan stage4/stage5 for the encrypted custom-decryptor bytecode block. The +/// raw byte at p+95 is used by decrypt_data6 as the iteration count. We trial- +/// decrypt that many bytes with the LFSR keystream and accept the first +/// position where the byte stream parses as a valid opcode sequence ending in +/// 195 (ret). +pub fn find_bytecode_offset(data: &[u8], base: u32, len: u32) -> Option { + let start = base as usize; + let end = (base.saturating_add(len)) as usize; + if end > data.len() { + return None; + } + let mut ks = [0u8; 256]; + lfsr_keystream(&mut ks); + // Scan forward from `start+16` on 16-byte boundaries relative to `start`. + // The bytecode block is positioned a fixed offset into stage4/stage5; the + // lowest parseable candidate is the real one (later ones are coincidental + // parses of trailing filler bytes that happen to map to valid opcodes). + // The enclosing buffer isn't necessarily 16-aligned to its absolute + // address in newer builds, so we anchor the stride to `start`. + let mut p = start + 16; + while p + 96 <= end { + let count = data[p + 95] as usize; + if count >= 8 && p + count <= end { + let mut buf = [0u8; 256]; + let take = count.min(256); + for i in 0..take { + buf[i] = data[p + i] ^ ks[i]; + } + if let Some(nops) = parse_bytecode_check(&buf[..take]) + && nops >= 4 + { + return Some(p as u32); + } + } + p += 16; + } + None +} + +/// Validate bytecode structure without allocating a `Vec` of ops. Returns +/// `Some(non_nop_op_count)` if the byte stream parses successfully as a valid +/// opcode sequence ending in 195 (ret), `None` otherwise. Allows non-trivial +/// bytecode filtering by op count. +pub fn parse_bytecode_check(buf: &[u8]) -> Option { + let mut i = 0usize; + let mut nops: usize = 0; + while i < buf.len() { + let b = buf[i]; + i += 1; + match b { + 4 | 44 | 52 => { + if i >= buf.len() { + return None; + } + i += 1; + nops += 1; + } + 144 => {} + 192 | 254 => { + if i >= buf.len() { + return None; + } + let mb = buf[i]; + i += 1; + let rm = mb & 7; + let mod_ = (mb >> 6) & 3; + let reg = (mb >> 3) & 7; + if mod_ != 3 || rm != 0 { + return None; + } + if reg > 1 { + return None; + } + if b == 192 { + if i >= buf.len() { + return None; + } + i += 1; + } + nops += 1; + } + 195 => return Some(nops), + _ => return None, + } + } + None +} + +/// Locate stage3's v4_val: the last non-zero dword in the buffer, anchored +/// by the `C3 CC CC CC` (ret + 3 int3) immediately before it. +pub fn find_v4_offset(data: &[u8], base: u32, len: u32) -> Option { + let start = base as usize; + let end = (base.saturating_add(len)) as usize; + if end > data.len() || end < start + 4 { + return None; + } + // walk backwards looking for the first non-zero byte + let mut i = end; + while i > start && data[i - 1] == 0 { + i -= 1; + } + if i < start + 4 { + return None; + } + // v_val occupies the 4 bytes ending at i (rounded up to dword boundary) + let v_end = i; + let v_start = ((v_end + 3) & !3).saturating_sub(4); + // require that the 4 bytes preceding v_val match `C3 CC CC CC` + if v_start < start + 4 || data[v_start - 4..v_start] != [0xC3, 0xCC, 0xCC, 0xCC] { + return None; + } + Some(v_start as u32) +} + +/// Scan a sub-buffer for an ASCII needle; return its absolute position. +pub fn find_str_pos(data: &[u8], base: u32, len: u32, needle: &[u8]) -> Option { + let start = base as usize; + let end = (base.saturating_add(len)) as usize; + if end > data.len() || needle.is_empty() { + return None; + } + data[start..end] + .windows(needle.len()) + .position(|w| w == needle) + .map(|rel| (start + rel) as u32) +} + +pub fn get_string_to_null(data: &[u8], offset: u32) -> String { + let start = offset as usize; + if start >= data.len() { + return String::new(); + } + // Bounded: an unterminated run must never walk off the end of the buffer + // (panic) or scan unboundedly into unrelated data. + let limit = start.saturating_add(4096).min(data.len()); + let mut i = start; + while i < limit && data[i] != 0 { + i += 1; + } + String::from_utf8_lossy(&data[start..i]).into_owned() +} + +/// Read a PE section-name field: exactly 8 bytes, NOT necessarily +/// NUL-terminated (a full-width name like `.textbss` has no NUL at all). +/// Returns the name with trailing NULs stripped. Using `get_string_to_null` +/// here would run past the field into the VirtualSize/VirtualAddress dwords. +pub fn section_name(data: &[u8], offset: u32) -> String { + let start = offset as usize; + let Some(field) = data.get(start..start + 8) else { + return String::new(); + }; + let end = field.iter().position(|&b| b == 0).unwrap_or(8); + String::from_utf8_lossy(&field[..end]).into_owned() +} + +// --------------------------------------------------------------------------- +// PE32 (32-bit) helpers +// --------------------------------------------------------------------------- + +/// PE32 shell-table locator. Walks the shell region (`info[6]`) for a dword +/// equal to `info[6]` followed by a plausible shell size, returning the table +/// base (`candidate = off - 0x88`) when `candidate+0x58` holds a valid pointer. +pub fn find_tbl_pe32(data: &[u8], info: &[u32; 8]) -> Option { + let shell = info[6]; + if (data.len() as u64) < 0x100 { + return None; + } + let hi = (shell as u64) + .saturating_add(0x3000) + .min(data.len() as u64 - 0x100) as u32; + let mut off = shell; + while off < hi { + if off as usize + 8 <= data.len() { + let candidate = off.wrapping_sub(0x88); + if candidate >= shell && get_u32(data, off) == info[6] { + let shell_size_val = get_u32(data, off.wrapping_add(4)); + if shell_size_val > 0x1000 && shell_size_val < 0x100000 { + let v58_off = candidate.wrapping_add(0x58); + if (v58_off as usize + 4) <= data.len() { + let v58 = get_u32(data, v58_off); + if v58 > 0 && (v58 as usize) < data.len() { + return Some(candidate); + } + } + } + } + } + off = off.wrapping_add(4); + } + None +} + +/// Locate an LFSR-encrypted bytecode block (decrypt_data6 form) in a region. +/// `start_off` is the byte offset to begin scanning at, `scan_backward` +/// controls direction. Returns the relative offset of the block. Includes full +/// opcode-walk validation of candidate blocks. +pub fn find_lfsr_block( + data: &[u8], + base: u32, + size: u32, + start_off: u32, + scan_backward: bool, +) -> Option { + if size < 96 { + return None; + } + let mut ks = [0u8; 128]; + lfsr_keystream(&mut ks); + let check = |scan_off: u32| -> bool { + let abs_off = base.wrapping_add(scan_off) as usize; + if abs_off + 96 > data.len() { + return false; + } + let sz = data[abs_off + 95] as usize; + if !(10..=95).contains(&sz) { + return false; + } + let mut decoded = [0u8; 95]; + for bi in 0..sz { + decoded[bi] = data[abs_off + bi] ^ ks[bi]; + } + // Full bytecode validation (shared with the stage4/5 locator): every + // opcode must decode with a valid ModR/M and the stream must REACH a + // RET (0xC3) as an opcode. The previous check only required a 0xC3 + // byte *anywhere* in the window and accepted a walk that ran off the + // end without hitting RET — a `0x04 0xC3` (ADD 0xC3) tail passed, so + // coincidental LFSR-shaped garbage was accepted as a decryptor block. + parse_bytecode_check(&decoded[..sz]).is_some() + }; + if scan_backward { + let hi = size - 96; + if hi >= start_off { + let mut scan_off = hi; + loop { + if check(scan_off) { + return Some(scan_off); + } + if scan_off == start_off { + break; + } + scan_off -= 1; + } + } + } else { + let hi = size - 95; + let mut scan_off = start_off; + while scan_off < hi { + if check(scan_off) { + return Some(scan_off); + } + scan_off += 1; + } + } + None +} + +/// Slots discovered in the eighthStage for the marker-less layout. +pub struct EighthSlots { + /// Absolute address of the file-data decryptor LFSR bytecode block. The + /// fileCS chain pointer is derived downstream as `file_lfsr - 0x58`. + pub file_lfsr: u32, + /// Absolute address of the compressedInfo (ptr,size) table pointer slot. + pub compressed_info_ptr: u32, +} + +/// Marker-independent eighthStage slot discovery (PE32+ branch). +/// +/// Newer Crackproof builds (e.g. some native/managed DLLs) omit the +/// `pm\0\0cm\0\0` and `00 00 00 40 01 00 00 00` markers that the older layout's +/// walk3/walk4/walk5 slot derivation relies on. Instead this discovers the +/// slots structurally: +/// * Scan the eighthStage for every LFSR (decrypt_data6) bytecode block. +/// * The file decryptor is the LFSR block whose `fileCS = lfsr - 0x58` holds +/// a pointer sitting just past `info[3]` (smallest positive distance). +/// * `compressedInfo` is the pointer slot whose 16-byte target, after a +/// trial `decrypt_data5`, parses as a plausible (src,sSize,dst,dSize) +/// descriptor. +/// +/// Returns `None` if no plausible file LFSR is found. `eighth_start`/`eighth_dsz` +/// bound the search region; `info3` is `info[3]`; `compress_data_offset` is +/// `(!u32(file_data,0x1080)) + 0x1000`; `file_data_len` is the protected file +/// length. +#[allow(clippy::too_many_arguments)] +pub fn discover_eighth_slots( + data: &[u8], + eighth_start: u32, + eighth_dsz: u32, + info3: u32, + compress_data_offset: u32, + file_data_len: u32, +) -> Option { + // Collect all LFSR candidates (forward scan). + // + // Advance by 1 after each hit, NOT by 96. A false-positive LFSR match can sit + // just before the real file-decryptor block (observed on an il2cpp game + // assembly build, 2026-07-13: junk at rel=0x31C1, real block at 0x3210). + // Stepping by the LFSR body size then skips the real block and discovery + // fails. Byte-stepping is cheap: eighthStage is only a few KB. + let mut all_lfsrs: Vec = Vec::new(); + let mut scan_off: u32 = 0; + while scan_off + 95 < eighth_dsz { + match find_lfsr_block(data, eighth_start, eighth_dsz, scan_off, false) { + Some(found) => { + all_lfsrs.push(found); + scan_off = found + 1; + } + None => break, + } + } + + // Pick the file LFSR: prefer the candidate whose fileCS pointer sits the + // smallest positive distance past info[3]. + let mut off_file_lfsr: Option = None; + let mut best_dist: Option = None; + for &lfsr_off in &all_lfsrs { + if lfsr_off < 0x58 { + continue; + } + let cs_off = lfsr_off - 0x58; + let cs_val = get_u32(data, eighth_start.wrapping_add(cs_off)); + if !(0x1000 < cs_val && (cs_val as usize) < data.len()) { + continue; + } + if cs_val < info3 { + continue; + } + let dist = cs_val - info3; + if best_dist.is_none_or(|b| dist < b) { + best_dist = Some(dist); + off_file_lfsr = Some(lfsr_off); + } + } + // Fallback: last LFSR with any in-image fileCS pointer. + if off_file_lfsr.is_none() { + for &lfsr_off in all_lfsrs.iter().rev() { + if lfsr_off < 0x58 { + continue; + } + let cs_val = get_u32(data, eighth_start.wrapping_add(lfsr_off - 0x58)); + if 0x1000 < cs_val && (cs_val as usize) < data.len() { + off_file_lfsr = Some(lfsr_off); + break; + } + } + } + let off_file_lfsr = off_file_lfsr?; + let off_file_cs = off_file_lfsr - 0x58; + + // Trial-decrypt to find compressedInfo: the pointer slot in the data area + // (between fileCS region start and the LFSR) whose target parses as a valid + // (src,sSize,dst,dSize) descriptor after a transient decrypt_data5. + let scan_from = off_file_lfsr.saturating_sub(0x400); + let mut off_compressed_info: Option = None; + let mut doff = scan_from; + while doff < off_file_lfsr { + if doff == off_file_cs { + doff += 4; + continue; + } + let ptr_val = get_u32(data, eighth_start.wrapping_add(doff)); + if !(0x1000 < ptr_val && (ptr_val as usize) < data.len().saturating_sub(16)) { + doff += 4; + continue; + } + // Predict decrypt_data5(ptr_val, 16) without mutating: each dword is + // position-keyed and independent, so trial_decrypt5_u32 per dword. + let src2 = trial_decrypt5_u32(data, ptr_val); + let s_sz2 = trial_decrypt5_u32(data, ptr_val + 4); + let dst2 = trial_decrypt5_u32(data, ptr_val + 8); + let d_sz2 = trial_decrypt5_u32(data, ptr_val + 12); + let src_file_off = src2.wrapping_add(compress_data_offset); + let valid = s_sz2 > 0 + && s_sz2 < 0x200000 + && (src_file_off as u64 + s_sz2 as u64) <= file_data_len as u64 + && dst2 >= 0x1000 + && (dst2 as u64 + d_sz2 as u64) <= data.len() as u64 + && d_sz2 >= s_sz2 + && d_sz2 < 0x200000; + if valid { + off_compressed_info = Some(doff); + break; + } + doff += 4; + } + let off_compressed_info = off_compressed_info?; + + Some(EighthSlots { + file_lfsr: eighth_start.wrapping_add(off_file_lfsr), + compressed_info_ptr: eighth_start.wrapping_add(off_compressed_info), + }) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Task 4.1 regression: build a synthetic buffer whose valid bytecode block + /// sits PAST `len` but within `len*2`. Assert that the smaller window misses + /// it and the doubled window finds it. + #[test] + fn bytecode_locate_double_window_retry() { + // We place the block at offset (base + len + 16) which is inside + // the len*2 window but outside the len window. + let base: u32 = 0; + let len: u32 = 256; + // Block sits at base + len + 16 = 272, aligned to 16. + let block_pos: usize = (base + len + 16) as usize; // 272 + + // The buffer must be large enough for the block (block_pos + 96 bytes). + let buf_len = block_pos + 256; + let mut buf = vec![0u8; buf_len]; + + // Build a valid plaintext op stream: + // [4, 0, 4, 0, 4, 0, 4, 0, 195] (4 ADD-AL ops then RET) + // Padded to 10 bytes total; count >= 8. + let count: usize = 10; + let mut plain = [0u8; 256]; + plain[0] = 4; + plain[1] = 0; + plain[2] = 4; + plain[3] = 0; + plain[4] = 4; + plain[5] = 0; + plain[6] = 4; + plain[7] = 0; + plain[8] = 195; // ret + + // Compute the LFSR keystream and XOR the first `count` bytes to get the + // encrypted representation that the scanner would decrypt back. + let mut ks = [0u8; 256]; + lfsr_keystream(&mut ks); + for i in 0..count { + buf[block_pos + i] = plain[i] ^ ks[i]; + } + // Raw count byte at block_pos+95 (outside the XOR range since count=10 < 95). + buf[block_pos + 95] = count as u8; + + // Verify our construction: find_bytecode_offset with len should NOT find it. + assert_eq!( + find_bytecode_offset(&buf, base, len), + None, + "smaller window should not find the block" + ); + + // The doubled window should find it at block_pos. + assert_eq!( + find_bytecode_offset(&buf, base, len.saturating_mul(2)), + Some(block_pos as u32), + "doubled window should locate the block" + ); + } +} diff --git a/senbei-pe/src/engine/layout/image.rs b/senbei-pe/src/engine/layout/image.rs new file mode 100644 index 0000000..04065cd --- /dev/null +++ b/senbei-pe/src/engine/layout/image.rs @@ -0,0 +1,481 @@ +//! PE import reconstruction and memory-image compaction. + +use senbei_crypto::primitives::{get_u16, get_u32, write_u16, write_u32}; + +use super::super::MAX_IMAGE_SIZE; + +/// Read a NUL-terminated byte string starting at `off`, bounded to 512 bytes. +/// Returns the raw bytes up to the terminator (excluding it). +fn read_cstr_bounded(data: &[u8], off: u32) -> Vec { + let start = off as usize; + if start >= data.len() { + return Vec::new(); + } + let limit = (start + 512).min(data.len()); + let mut end = start; + while end < limit && data[end] != 0 { + end += 1; + } + data[start..end].to_vec() +} + +fn align_up_u32(value: u32, alignment: u32) -> u32 { + ((value.wrapping_add(alignment - 1)) / alignment).wrapping_mul(alignment) +} + +fn align_up_u64(value: u64, alignment: u64) -> u64 { + value.div_ceil(alignment) * alignment +} + +#[derive(Clone)] +enum ImportFunc { + Ordinal(u32), + Name(u16, Vec), +} + +struct ImportDesc { + time_date: u32, + fwd_chain: u32, + dll_name: Vec, + iat_rva: u32, + functions: Vec, +} + +/// Return true when PE32 imports already sit in the original `.idata` layout +/// (so no relocation to `.kmiat` is needed). May write the IAT data directory +/// (pe+0xD8). +pub fn pe32_imports_already_match_idata_layout(data: &mut [u8], pe_header: u32) -> bool { + let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32; + let sec_table = pe_header.wrapping_add(24).wrapping_add(opt_hdr_size); + let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32; + let import_rva = get_u32(data, pe_header.wrapping_add(0x80)); + let import_size = get_u32(data, pe_header.wrapping_add(0x84)); + let len = data.len() as u32; + if !(import_rva > 0 && import_size > 0) { + return false; + } + for idx in 0..num_sections { + let sec_off = sec_table.wrapping_add(idx * 40); + if (sec_off as usize + 40) > data.len() { + return false; + } + if &data[sec_off as usize..sec_off as usize + 6] != b".idata" { + continue; + } + let sec_va = get_u32(data, sec_off.wrapping_add(12)); + let sec_size = + get_u32(data, sec_off.wrapping_add(8)).max(get_u32(data, sec_off.wrapping_add(16))); + let sec_end = sec_va.wrapping_add(sec_size); + if !(sec_va <= import_rva + && import_rva < sec_end + && import_rva.wrapping_add(import_size) <= sec_end) + { + continue; + } + let first_oft = get_u32(data, import_rva); + let first_name = get_u32(data, import_rva.wrapping_add(12)); + let first_iat = get_u32(data, import_rva.wrapping_add(16)); + if !(sec_va <= first_oft + && first_oft < sec_end + && sec_va <= first_iat + && first_iat < sec_end) + { + return false; + } + if !(0x1000 < first_name && first_name < len) { + return false; + } + let dll_name = read_cstr_bounded(data, first_name); + let lower: Vec = dll_name.iter().map(|b| b.to_ascii_lowercase()).collect(); + if !lower.ends_with(b".dll") { + return false; + } + let mut iat_min = first_iat; + let mut iat_max = first_iat; + let mut idt_pos = import_rva; + while idt_pos.wrapping_add(20) <= len { + let oft_rva = get_u32(data, idt_pos); + let name_rva = get_u32(data, idt_pos.wrapping_add(12)); + let iat_rva = get_u32(data, idt_pos.wrapping_add(16)); + if oft_rva == 0 && name_rva == 0 && iat_rva == 0 { + break; + } + if !(sec_va <= oft_rva && oft_rva < sec_end && sec_va <= iat_rva && iat_rva < sec_end) { + return false; + } + let mut thunk = iat_rva; + while thunk.wrapping_add(4) <= sec_end { + let tv = get_u32(data, thunk); + thunk = thunk.wrapping_add(4); + if tv == 0 { + break; + } + } + iat_min = iat_min.min(iat_rva); + iat_max = iat_max.max(thunk); + idt_pos = idt_pos.wrapping_add(20); + } + if iat_max > iat_min { + write_u32(data, pe_header.wrapping_add(0xD8), iat_min); + write_u32(data, pe_header.wrapping_add(0xDC), iat_max - iat_min); + } + return true; + } + false +} + +/// Rebuild PE32 import metadata (descriptors, lookup tables, names) into the +/// last section as `.kmiat`, leaving the loader-written IAT in place. Mutates +/// `data` (may grow it). +pub fn move_pe32_imports_to_kmiat(data: &mut Vec, pe_header: u32) { + const SECTION_SIZE: u32 = 0x7000; + let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32; + let opt_hdr = pe_header.wrapping_add(24); + let sec_table = opt_hdr.wrapping_add(opt_hdr_size); + let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32; + if num_sections == 0 { + return; + } + let import_rva = get_u32(data, pe_header.wrapping_add(0x80)); + let import_size = get_u32(data, pe_header.wrapping_add(0x84)); + let len = data.len() as u32; + if !(0x1000 < import_rva && import_rva < len && import_size > 0 && import_size < SECTION_SIZE) { + return; + } + + let mut descriptors: Vec = Vec::new(); + let mut idt_pos = import_rva; + while idt_pos.wrapping_add(20) <= len { + let oft_rva = get_u32(data, idt_pos); + let time_date = get_u32(data, idt_pos.wrapping_add(4)); + let fwd_chain = get_u32(data, idt_pos.wrapping_add(8)); + let name_rva = get_u32(data, idt_pos.wrapping_add(12)); + let iat_rva = get_u32(data, idt_pos.wrapping_add(16)); + if oft_rva == 0 && name_rva == 0 && iat_rva == 0 { + break; + } + if !(0x1000 < name_rva && name_rva < len) { + break; + } + let dll_name = read_cstr_bounded(data, name_rva); + let thunk_rva = if 0x1000 < oft_rva && oft_rva < len { + oft_rva + } else { + iat_rva + }; + let mut functions: Vec = Vec::new(); + let mut thunk_pos = thunk_rva; + while 0x1000 < thunk_pos.wrapping_add(4) && thunk_pos.wrapping_add(4) <= len { + let thunk_val = get_u32(data, thunk_pos); + if thunk_val == 0 { + break; + } + if thunk_val & 0x8000_0000 != 0 { + functions.push(ImportFunc::Ordinal(thunk_val & 0xFFFF)); + } else { + let hint = if thunk_val.wrapping_add(2) <= len { + get_u16(data, thunk_val) + } else { + 0 + }; + let func_name = if thunk_val.wrapping_add(2) < len { + read_cstr_bounded(data, thunk_val.wrapping_add(2)) + } else { + Vec::new() + }; + functions.push(ImportFunc::Name(hint, func_name)); + } + thunk_pos = thunk_pos.wrapping_add(4); + } + descriptors.push(ImportDesc { + time_date, + fwd_chain, + dll_name, + iat_rva, + functions, + }); + idt_pos = idt_pos.wrapping_add(20); + } + if descriptors.is_empty() { + return; + } + + for desc in &mut descriptors { + let lower: Vec = desc + .dll_name + .iter() + .map(|b| b.to_ascii_lowercase()) + .collect(); + if lower.starts_with(b"api-ms-win-crt-") { + desc.dll_name = b"ucrtbase.dll".to_vec(); + } else { + desc.dll_name = lower; + } + } + descriptors.sort_by_key(|d| d.iat_rva); + + let last_sec = sec_table.wrapping_add((num_sections - 1) * 40); + let kmiat_rva = get_u32(data, last_sec.wrapping_add(12)); + // A zero last-section VA means a corrupt section table: building .kmiat at + // RVA 0 would zero the DOS/PE headers and emit a structurally broken image + // with no error. Bail and keep the original import table. + if kmiat_rva == 0 { + return; + } + // Grow the image when .kmiat overruns it, but cap the growth: a corrupt VA + // could otherwise request a multi-gigabyte allocation, which aborts the + // process (uncatchable). Use u64 math so a near-u32::MAX VA cannot wrap the + // end calculation the way the previous wrapping/plain-add mix could. + let kmiat_end = kmiat_rva as u64 + SECTION_SIZE as u64; + if kmiat_end > MAX_IMAGE_SIZE { + return; + } + if kmiat_end > data.len() as u64 { + data.resize(kmiat_end as usize, 0); + } + // Zero the .kmiat region. + for b in &mut data[kmiat_rva as usize..kmiat_end as usize] { + *b = 0; + } + + let idt_size = (descriptors.len() as u32 + 1) * 20; + let oft_start = kmiat_rva; + let mut idt_rva = oft_start; + for desc in &descriptors { + idt_rva = idt_rva.wrapping_add((desc.functions.len() as u32 + 1) * 4); + } + idt_rva = align_up_u32(idt_rva.wrapping_add(0x2C), 4); + + // Size check: compute the final name_pos and bail if it overruns .kmiat. + let mut name_pos_check = idt_rva.wrapping_add(idt_size); + for desc in &descriptors { + name_pos_check = name_pos_check.wrapping_add(desc.dll_name.len() as u32 + 1); + for func in &desc.functions { + if let ImportFunc::Name(_, fname) = func { + name_pos_check = name_pos_check.wrapping_add(2 + fname.len() as u32 + 1); + } + } + } + if name_pos_check > kmiat_rva.wrapping_add(SECTION_SIZE) { + // Section too small; keep existing import table untouched. + return; + } + + let mut oft_pos = oft_start; + let mut name_pos = idt_rva.wrapping_add(idt_size); + for (idx, desc) in descriptors.iter().enumerate() { + let idt_entry = idt_rva.wrapping_add(idx as u32 * 20); + let current_oft = oft_pos; + write_u32(data, idt_entry, current_oft); + write_u32(data, idt_entry.wrapping_add(4), desc.time_date); + write_u32(data, idt_entry.wrapping_add(8), desc.fwd_chain); + let dll_name_pos = name_pos; + write_u32(data, idt_entry.wrapping_add(12), dll_name_pos); + write_u32(data, idt_entry.wrapping_add(16), desc.iat_rva); + + let dnp = dll_name_pos as usize; + data[dnp..dnp + desc.dll_name.len()].copy_from_slice(&desc.dll_name); + data[dnp + desc.dll_name.len()] = 0; + name_pos = name_pos.wrapping_add(desc.dll_name.len() as u32 + 1); + + for func in &desc.functions { + match func { + ImportFunc::Ordinal(ord) => { + write_u32(data, oft_pos, 0x8000_0000 | ord); + } + ImportFunc::Name(hint, fname) => { + let hint_name_rva = name_pos; + write_u32(data, oft_pos, hint_name_rva); + write_u16(data, hint_name_rva, *hint as u32); + let fp = (hint_name_rva + 2) as usize; + data[fp..fp + fname.len()].copy_from_slice(fname); + data[fp + fname.len()] = 0; + name_pos = name_pos.wrapping_add(2 + fname.len() as u32 + 1); + } + } + oft_pos = oft_pos.wrapping_add(4); + } + write_u32(data, oft_pos, 0); + oft_pos = oft_pos.wrapping_add(4); + } + // Null-terminator IDT entry (20 zero bytes) after the last descriptor. + let term = idt_rva.wrapping_add(descriptors.len() as u32 * 20) as usize; + for b in &mut data[term..term + 20] { + *b = 0; + } + + let ls = last_sec as usize; + data[ls..ls + 8].copy_from_slice(b".kmiat\x00\x00"); + write_u32(data, last_sec.wrapping_add(8), SECTION_SIZE); + write_u32(data, last_sec.wrapping_add(16), SECTION_SIZE); + write_u32(data, last_sec.wrapping_add(36), 0xE000_0060); + write_u32(data, pe_header.wrapping_add(0x80), idt_rva); + write_u32(data, pe_header.wrapping_add(0x84), idt_size); + write_u32( + data, + pe_header.wrapping_add(80), + kmiat_rva.wrapping_add(SECTION_SIZE), + ); +} + +/// Convert the unpacked RVA-addressed image back to a compact PE file layout +/// (headers at 0x400, sections packed consecutively, FileAlignment 0x200). +/// Returns `None` if the accumulated output size wraps or exceeds +/// [`MAX_IMAGE_SIZE`]: the final allocation is sized from header-derived +/// section data, and an uncapped `vec![0; n]` from a corrupt header would abort +/// the process (which `catch_unpack` cannot trap). +pub fn compact_memory_image_to_pe(data: &[u8], pe_header: u32) -> Option> { + const FILE_ALIGNMENT: u32 = 0x200; + const HEADER_SIZE: u32 = 0x400; + let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32; + let opt_hdr = pe_header.wrapping_add(24); + let sec_table = opt_hdr.wrapping_add(opt_hdr_size); + let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32; + + struct SecLayout { + sec_off: u32, + va: u32, + vsize: u32, + raw_ptr: u32, + raw_size: u32, + } + + let mut raw_cursor: u64 = HEADER_SIZE as u64; + let mut raw_layout: Vec = Vec::new(); + for idx in 0..num_sections { + let sec_off = sec_table.wrapping_add(idx * 40); + let vsize = get_u32(data, sec_off.wrapping_add(8)); + let va = get_u32(data, sec_off.wrapping_add(12)); + let sd_start = va as usize; + let sd_end = if (va.wrapping_add(vsize) as usize) <= data.len() { + va.wrapping_add(vsize) as usize + } else { + data.len() + }; + let section_data: &[u8] = if sd_start <= sd_end && sd_start <= data.len() { + &data[sd_start..sd_end] + } else { + &[] + }; + + let mut last_nonzero: i64 = -1; + for pos in (0..section_data.len()).rev() { + if section_data[pos] != 0 { + last_nonzero = pos as i64; + break; + } + } + let meaningful = if last_nonzero >= 0 { + (last_nonzero + 1) as u32 + } else { + 0 + }; + let mut raw_size = if meaningful != 0 { + align_up_u32(meaningful, FILE_ALIGNMENT) + } else { + 0 + }; + if vsize != 0 && raw_size == 0 { + raw_size = FILE_ALIGNMENT; + } + raw_size = raw_size.min(align_up_u32(section_data.len() as u32, FILE_ALIGNMENT)); + + let raw_ptr = if raw_size != 0 { raw_cursor as u32 } else { 0 }; + raw_layout.push(SecLayout { + sec_off, + va, + vsize, + raw_ptr, + raw_size, + }); + if raw_size != 0 { + // Accumulate in u64 and cap: section sizes are header-derived, and + // a corrupt table could otherwise wrap raw_cursor (small alloc, + // huge recorded raw_ptrs → OOB panic) or request an abort-sized + // allocation. + raw_cursor = align_up_u64(raw_cursor + raw_size as u64, FILE_ALIGNMENT as u64); + if raw_cursor > MAX_IMAGE_SIZE { + return None; + } + } + } + + let mut compact = vec![0u8; raw_cursor as usize]; + let hdr_copy = (HEADER_SIZE as usize).min(data.len()); + compact[..hdr_copy].copy_from_slice(&data[..hdr_copy]); + write_u32(&mut compact, opt_hdr.wrapping_add(36), FILE_ALIGNMENT); + write_u32(&mut compact, opt_hdr.wrapping_add(60), HEADER_SIZE); + + for sl in &raw_layout { + write_u32(&mut compact, sl.sec_off.wrapping_add(16), sl.raw_size); + write_u32(&mut compact, sl.sec_off.wrapping_add(20), sl.raw_ptr); + if sl.raw_size != 0 { + let sd_start = sl.va as usize; + let sd_end = if (sl.va.wrapping_add(sl.vsize) as usize) <= data.len() { + sl.va.wrapping_add(sl.vsize) as usize + } else { + data.len() + }; + let section_data: &[u8] = if sd_start <= sd_end { + &data[sd_start..sd_end] + } else { + &[] + }; + let copy_size = (sl.raw_size as usize).min(section_data.len()); + let rp = sl.raw_ptr as usize; + compact[rp..rp + copy_size].copy_from_slice(§ion_data[..copy_size]); + } + } + Some(compact) +} + +#[cfg(test)] +mod tests { + use super::*; + + /// Review regression: a zero last-section VA (corrupt section table) must + /// bail instead of building .kmiat at RVA 0 — the old code zeroed + /// `[0, 0x7000)`, wiping the DOS/PE headers, and returned the broken image + /// as a success. A near-2 GiB VA must likewise refuse to grow the image + /// past [`MAX_IMAGE_SIZE`]. + #[test] + fn kmiat_bogus_section_va_bails_without_wiping_headers() { + for last_sec_va in [0u32, 0x5000_0000] { + let pe: u32 = 0x80; + let mut data = vec![0xAAu8; 0x8000]; + // COFF header: 1 section, optional header size 0xE0 (PE32). + write_u16(&mut data, pe + 6, 1); + write_u16(&mut data, pe + 20, 0xE0); + // Import directory at pe+0x80: one descriptor + null terminator. + write_u32(&mut data, pe + 0x80, 0x1100); + write_u32(&mut data, pe + 0x84, 0x28); + write_u32(&mut data, 0x1100, 0x1200); // OFT rva + write_u32(&mut data, 0x1100 + 12, 0x1300); // name rva + write_u32(&mut data, 0x1100 + 16, 0x1400); // IAT rva + for b in &mut data[0x1100 + 20..0x1100 + 40] { + *b = 0; // null terminator descriptor + } + data[0x1300..0x1300 + 13].copy_from_slice(b"KERNEL32.dll\0"); + write_u32(&mut data, 0x1200, 0x1500); // thunk -> hint/name + write_u32(&mut data, 0x1204, 0); // thunk terminator + data[0x1500..0x1502].copy_from_slice(&0u16.to_le_bytes()); + data[0x1502..0x1502 + 12].copy_from_slice(b"ExitProcess\0"); + // Section table at pe+24+0xE0 = 0x178; VA field at +12. + write_u32(&mut data, 0x178 + 12, last_sec_va); + + let head_before: Vec = data[..0x400].to_vec(); + let len_before = data.len(); + move_pe32_imports_to_kmiat(&mut data, pe); + assert_eq!( + data.len(), + len_before, + "VA 0x{last_sec_va:08X}: image must not grow" + ); + assert_eq!( + &data[..0x400], + &head_before[..], + "VA 0x{last_sec_va:08X}: headers must be untouched" + ); + } + } +} diff --git a/src/unpacker/mod.rs b/senbei-pe/src/engine/mod.rs similarity index 51% rename from src/unpacker/mod.rs rename to senbei-pe/src/engine/mod.rs index e133173..3eb4bcb 100644 --- a/src/unpacker/mod.rs +++ b/senbei-pe/src/engine/mod.rs @@ -1,30 +1,157 @@ -//! Pure, panic-free Crackproof unpacker core. No file I/O lives here. +//! PE detection, unpacking, and structural validation. -mod bytecode; -mod crc32; pub mod dll; +mod error; pub mod exe; pub mod integrity; +mod layout; pub(crate) mod parallel; -pub(crate) mod primitives; -mod tables; + +use senbei_crypto::primitives; +use std::cell::RefCell; +use std::sync::{Arc, Mutex}; pub use dll::{unpack_dll, unpack_dll_v}; -pub use exe::{UnpackError, unpack as unpack_exe, unpack_v as unpack_exe_v}; +pub use error::*; +pub use exe::{unpack as unpack_exe, unpack_v as unpack_exe_v}; pub use integrity::{IntegrityReport, check as check_integrity}; +pub use parallel::thread_cap; /// Maximum plausible PE `SizeOfImage` we are willing to allocate a zero buffer /// for. Guards against a corrupt/crafted header requesting a multi-gigabyte /// (or, as a sign-extended negative `i32`, multi-exabyte) allocation, which /// would abort the process — an abort that `catch_unpack` below cannot trap. /// Real protected binaries are far below this. -pub(crate) const MAX_IMAGE_SIZE: u64 = 1 << 30; // 1 GiB +pub(crate) const MAX_IMAGE_SIZE: u64 = senbei_crypto::MAX_IMAGE_SIZE; + +#[derive(Clone)] +pub(crate) struct PanicCapture(Arc>>); + +#[derive(Clone)] +struct PanicDetails { + message: String, + file: String, + line: u32, + column: u32, +} + +thread_local! { + static ACTIVE_PANIC_CAPTURE: RefCell> = const { RefCell::new(None) }; +} + +struct PanicCaptureGuard(Option); + +impl Drop for PanicCaptureGuard { + fn drop(&mut self) { + ACTIVE_PANIC_CAPTURE.with(|slot| { + slot.replace(self.0.take()); + }); + } +} + +impl PanicCapture { + fn new() -> Self { + Self(Arc::new(Mutex::new(None))) + } + + fn record(&self, info: &std::panic::PanicHookInfo<'_>) { + let location = info.location(); + let details = PanicDetails { + message: panic_message(info.payload()), + file: location + .map(|value| value.file().to_owned()) + .unwrap_or_else(|| "".to_owned()), + line: location.map_or(0, std::panic::Location::line), + column: location.map_or(0, std::panic::Location::column), + }; + let mut captured = self + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if captured.is_none() { + *captured = Some(details); + } + } + + fn into_error(self, payload: &(dyn std::any::Any + Send)) -> UnpackError { + let details = self + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .clone() + .unwrap_or_else(|| PanicDetails { + message: panic_message(payload), + file: "".to_owned(), + line: 0, + column: 0, + }); + UnpackError::InternalPanic { + message: details.message, + file: details.file, + line: details.line, + column: details.column, + } + } + + fn merge_from(&self, other: &Self) { + let details = other + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner) + .clone(); + let Some(details) = details else { return }; + let mut captured = self + .0 + .lock() + .unwrap_or_else(std::sync::PoisonError::into_inner); + if captured.is_none() { + *captured = Some(details); + } + } +} + +fn panic_message(payload: &(dyn std::any::Any + Send)) -> String { + if let Some(message) = payload.downcast_ref::<&str>() { + (*message).to_owned() + } else if let Some(message) = payload.downcast_ref::() { + message.clone() + } else { + "non-string panic payload".to_owned() + } +} + +fn install_panic_capture_hook() { + static INSTALL: std::sync::Once = std::sync::Once::new(); + INSTALL.call_once(|| { + let previous = std::panic::take_hook(); + std::panic::set_hook(Box::new(move |info| { + let capture = ACTIVE_PANIC_CAPTURE + .try_with(|slot| slot.borrow().clone()) + .ok() + .flatten(); + if let Some(capture) = capture { + capture.record(info); + } else { + previous(info); + } + })); + }); +} + +pub(crate) fn current_panic_capture() -> Option { + ACTIVE_PANIC_CAPTURE.with(|slot| slot.borrow().clone()) +} + +pub(crate) fn with_panic_capture(capture: Option, f: impl FnOnce() -> R) -> R { + let previous = ACTIVE_PANIC_CAPTURE.with(|slot| slot.replace(capture)); + let _guard = PanicCaptureGuard(previous); + f() +} /// Run an unpack pipeline, converting any internal panic into a clean -/// [`UnpackError::Corrupt`] so the public API stays panic-free on any input -/// (truncated/garbled files chase offsets out of bounds). The default panic -/// hook is suppressed transiently so a trapped panic does not spill a -/// backtrace to stderr. +/// [`UnpackError::InternalPanic`] so the public API stays panic-free on any input +/// (truncated/garbled files chase offsets out of bounds). The panic location and +/// payload are captured for diagnostics without printing a backtrace to stderr. /// /// Note: allocation *failures* abort the process and are NOT caught here; size /// requests are bounds-checked against [`MAX_IMAGE_SIZE`] before allocating. @@ -32,17 +159,15 @@ pub(crate) fn catch_unpack(f: F) -> Result, UnpackError> where F: FnOnce() -> Result, UnpackError>, { - // Hook suppression is skipped on wasm: the prebuilt std cannot unwind - // there, so a panic traps immediately — and the suppressed hook would - // hide the panic message, leaving a bare `unreachable` with no clue. - #[cfg(not(target_arch = "wasm32"))] - let prev = std::panic::take_hook(); - #[cfg(not(target_arch = "wasm32"))] - std::panic::set_hook(Box::new(|_| {})); - let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(f)); - #[cfg(not(target_arch = "wasm32"))] - std::panic::set_hook(prev); - r.unwrap_or(Err(UnpackError::Corrupt)) + install_panic_capture_hook(); + let capture = PanicCapture::new(); + let r = with_panic_capture(Some(capture.clone()), || { + std::panic::catch_unwind(std::panic::AssertUnwindSafe(f)) + }); + match r { + Ok(result) => result, + Err(payload) => Err(capture.into_error(payload.as_ref())), + } } /// Crackproof header magic stored in `keys[1]`/`info[1]`. @@ -80,11 +205,8 @@ fn key_table(input: &[u8]) -> Option<[u32; 8]> { if input.len() < 4128 { return None; } - // Validate PE signature. `checked_add`, not `+`: `usize` is 32-bit on - // wasm32, where an `e_lfanew` of 0xFFFF_FFFC..=0xFFFF_FFFF wraps the bound - // check, and the slice below then panics with start > end. `detect` runs on - // the folder-scan threads and (in the web app) on the main thread outside - // the disposable-worker isolation, so it must not panic on any input. + // Validate the PE signature with checked arithmetic so a crafted offset + // cannot wrap the bounds check on a narrower target. let e_lfanew = primitives::get_u32(input, 0x3C); let pe_start = e_lfanew as usize; if pe_start.checked_add(4).is_none_or(|end| end > input.len()) { @@ -194,13 +316,76 @@ pub fn unpack_auto_v(input: &[u8], verbose: bool) -> Result<(Kind, Vec), Unp Ok(out) => out, Err(dll_err) => match exe::unpack_v(input, verbose) { Ok(out) => out, - // Surface the DLL-pipeline error, not the EXE one: for a - // genuinely corrupt DLL the DLL error is the more relevant - // diagnostic, and the EXE fallback is best-effort. - Err(_) => return Err(dll_err), + Err(exe_err) => { + return Err(UnpackError::PipelineFallbackFailed { + dll: Box::new(dll_err), + exe: Box::new(exe_err), + }); + } }, } } }; Ok((detected.kind, out)) } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn caught_panic_reports_location_and_message() { + let error = catch_unpack(|| -> Result, UnpackError> { + panic!("test panic"); + }) + .expect_err("panic must become an error"); + let UnpackError::InternalPanic { + message, + file, + line, + column, + } = error + else { + panic!("unexpected error: {error}"); + }; + assert_eq!(message, "test panic"); + assert!( + file.ends_with("senbei-pe/src/engine/mod.rs") + || file.ends_with("senbei-pe\\src\\engine\\mod.rs") + ); + assert!(line > 0); + assert!(column > 0); + } + + #[test] + fn worker_panic_keeps_the_worker_source_location() { + let error = catch_unpack(|| -> Result, UnpackError> { + let capture = current_panic_capture(); + let result = std::thread::spawn(move || { + with_panic_capture(capture, || panic!("worker panic")); + }) + .join(); + if let Err(payload) = result { + std::panic::resume_unwind(payload); + } + Ok(Vec::new()) + }) + .expect_err("worker panic must become an error"); + let UnpackError::InternalPanic { + message, + file, + line, + column, + } = error + else { + panic!("unexpected error: {error}"); + }; + assert_eq!(message, "worker panic"); + assert!( + file.ends_with("senbei-pe/src/engine/mod.rs") + || file.ends_with("senbei-pe\\src\\engine\\mod.rs") + ); + assert!(line > 0); + assert!(column > 0); + } +} diff --git a/src/unpacker/parallel.rs b/senbei-pe/src/engine/parallel.rs similarity index 89% rename from src/unpacker/parallel.rs rename to senbei-pe/src/engine/parallel.rs index e6e0a8a..fad1132 100644 --- a/src/unpacker/parallel.rs +++ b/senbei-pe/src/engine/parallel.rs @@ -21,7 +21,7 @@ use std::sync::atomic::{AtomicBool, Ordering}; /// Worker-thread cap. `SENBEI_THREADS` overrides it (`1` forces the sequential /// path); otherwise the host's available parallelism; otherwise 1. -pub(crate) fn thread_cap() -> usize { +pub fn thread_cap() -> usize { if let Ok(v) = std::env::var("SENBEI_THREADS") && let Ok(n) = v.trim().parse::() && n >= 1 @@ -46,7 +46,7 @@ pub(crate) fn thread_cap() -> usize { /// /// Returns the first `Err` any block produces; re-raises the first block panic /// on the calling thread (so the pipeline's existing `catch_unpack` still -/// converts it to `UnpackError::Corrupt`). +/// converts it to `UnpackError::InternalPanic`). pub(crate) fn parallel_for( buf: &mut [u8], spans: &[(usize, usize)], @@ -125,6 +125,7 @@ where let stop = AtomicBool::new(false); let first_err: Mutex> = Mutex::new(None); let first_panic: Mutex>> = Mutex::new(None); + let panic_capture = super::current_panic_capture(); std::thread::scope(|scope| { for _ in 0..workers { @@ -133,6 +134,7 @@ where let first_err = &first_err; let first_panic = &first_panic; let f = &f; + let panic_capture = panic_capture.clone(); scope.spawn(move || { loop { if stop.load(Ordering::Relaxed) { @@ -141,9 +143,15 @@ where let next = iter.lock().unwrap().next(); let Some((i, piece)) = next else { break }; let span = piece.unwrap(); - let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { - f(i, spans[i].0, span) - })); + // Keep details local until this panic wins `first_panic`; + // otherwise simultaneous workers could pair one worker's + // location with another worker's propagated payload. + let block_capture = panic_capture.as_ref().map(|_| super::PanicCapture::new()); + let r = super::with_panic_capture(block_capture.clone(), || { + std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { + f(i, spans[i].0, span) + })) + }); match r { Ok(Ok(())) => {} Ok(Err(e)) => { @@ -157,6 +165,11 @@ where Err(panic) => { let mut slot = first_panic.lock().unwrap(); if slot.is_none() { + if let (Some(parent), Some(block)) = + (&panic_capture, &block_capture) + { + parent.merge_from(block); + } *slot = Some(panic); } stop.store(true, Ordering::Relaxed); diff --git a/senbei-pe/src/lib.rs b/senbei-pe/src/lib.rs new file mode 100644 index 0000000..5db4e9a --- /dev/null +++ b/senbei-pe/src/lib.rs @@ -0,0 +1,5 @@ +//! PE detection, unpacking, and structural validation. + +mod engine; + +pub use engine::*; diff --git a/src/unpacker/primitives.rs b/src/unpacker/primitives.rs deleted file mode 100644 index a0a6adc..0000000 --- a/src/unpacker/primitives.rs +++ /dev/null @@ -1,2397 +0,0 @@ -//! Shared crypto primitives and helper utilities. -//! -//! All functions here are `pub(crate)` so that both the EXE unpacker (`exe.rs`) -//! and the future DLL unpacker (`dll.rs`) can call them without duplication. -//! Each free function is self-contained: it takes the relevant byte buffer(s) -//! and parameters explicitly, with no coupling to the EXE `Unpacker` struct. - -use super::bytecode::{Op, OpsLut}; -use super::crc32; -use super::tables::{COLUMMIX1, COLUMMIX2, COLUMMIX3, COLUMMIX4, SBOX}; -use std::cell::RefCell; - -thread_local! { - /// Reusable scratch for `decompress`. A single unpack runs `decompress` - /// hundreds of times over small blocks; reusing one growable buffer avoids a - /// fresh allocation each call. Thread-local, so it stays correct (one buffer - /// per worker) under the parallel block fan-out. - static DECOMPRESS_SCRATCH: RefCell> = const { RefCell::new(Vec::new()) }; -} - -// --------------------------------------------------------------------------- -// Byte-order accessors -// --------------------------------------------------------------------------- - -pub(crate) fn get_u16(data: &[u8], offset: u32) -> u16 { - let i = offset as usize; - u16::from_le_bytes([data[i], data[i + 1]]) -} - -pub(crate) fn get_u32(data: &[u8], offset: u32) -> u32 { - let i = offset as usize; - u32::from_le_bytes([data[i], data[i + 1], data[i + 2], data[i + 3]]) -} - -pub(crate) fn get_u64(data: &[u8], offset: u32) -> u64 { - let i = offset as usize; - u64::from_le_bytes([ - data[i], - data[i + 1], - data[i + 2], - data[i + 3], - data[i + 4], - data[i + 5], - data[i + 6], - data[i + 7], - ]) -} - -pub(crate) fn write_u16(data: &mut [u8], offset: u32, value: u32) { - let i = offset as usize; - let v = value as u16; - let b = v.to_le_bytes(); - data[i] = b[0]; - data[i + 1] = b[1]; -} - -pub(crate) fn write_u32(data: &mut [u8], offset: u32, value: u32) { - let i = offset as usize; - let b = value.to_le_bytes(); - data[i] = b[0]; - data[i + 1] = b[1]; - data[i + 2] = b[2]; - data[i + 3] = b[3]; -} - -// --------------------------------------------------------------------------- -// Checked accessors (return Err instead of panicking on OOB) -// --------------------------------------------------------------------------- - -#[allow(dead_code)] -pub(crate) fn try_u32(d: &[u8], off: usize) -> Result { - d.get(off..off + 4) - .map(|s| u32::from_le_bytes(s.try_into().unwrap())) - .ok_or(super::UnpackError::OutOfBounds(off)) -} - -#[allow(dead_code)] -pub(crate) fn try_i32(d: &[u8], off: usize) -> Result { - try_u32(d, off).map(|v| v as i32) -} - -/// Checked copy: returns OutOfBounds if src or dst ranges exceed their respective slices. -pub(crate) fn try_copy_from_slice( - dst: &mut [u8], - dst_off: usize, - dst_len: usize, - src: &[u8], - src_off: usize, -) -> Result<(), super::UnpackError> { - let dst_end = dst_off - .checked_add(dst_len) - .ok_or(super::UnpackError::OutOfBounds(dst_off))?; - let src_end = src_off - .checked_add(dst_len) - .ok_or(super::UnpackError::OutOfBounds(src_off))?; - if dst_end > dst.len() { - return Err(super::UnpackError::OutOfBounds(dst_off)); - } - if src_end > src.len() { - return Err(super::UnpackError::OutOfBounds(src_off)); - } - dst[dst_off..dst_end].copy_from_slice(&src[src_off..src_end]); - Ok(()) -} - -// --------------------------------------------------------------------------- -// Locator helpers -// --------------------------------------------------------------------------- - -/// Find the 4-byte v_val that follows the LAST occurrence of `48 EB 01 B9` -/// (REX.W jmp+1; mov ecx,imm32) plus any 0xCC padding. Used to locate -/// stage4's accum2 seed. Works across builds even when API-name anchors are -/// absent. -pub(crate) fn find_v_after_pad(data: &[u8], base: u32, len: u32) -> Option { - let start = base as usize; - let end = (base.saturating_add(len)) as usize; - if end > data.len() { - return None; - } - let sig = [0x48u8, 0xEB, 0x01, 0xB9]; - let slice = &data[start..end]; - // last occurrence - let mut last = None; - let mut i = 0usize; - while i + sig.len() <= slice.len() { - if slice[i..i + sig.len()] == sig { - last = Some(i); - } - i += 1; - } - let pos = last?; - // skip CCs after the `48 EB 01 B9` - let mut after = pos + sig.len(); - while after < slice.len() && slice[after] == 0xCC { - after += 1; - } - if after + 4 > slice.len() { - return None; - } - Some((start + after) as u32) -} - -/// Predict the 4 bytes that DecryptData5(va, size) would produce at va+0..va+4 -/// without mutating the buffer. The cipher's per-byte transform depends only -/// on the byte itself and the low 8 bits of (va+i), with no cross-byte state, -/// so each byte can be decrypted in isolation. Used to detect the EP/DD layout -/// offset before committing to the actual call. -pub(crate) fn trial_decrypt5_u32(data: &[u8], va: u32) -> u32 { - let mut out = [0u8; 4]; - for i in 0..4u32 { - let b3 = data[(va + i) as usize]; - let b = (va + i) as u8; - let b2 = b.wrapping_add(1); - let b4 = b3.rotate_left(2) ^ b2; - let b5 = b4.rotate_left(2) ^ b; - out[i as usize] = b5.rotate_left(2); - } - u32::from_le_bytes(out) -} - -/// Reproduce the LFSR keystream that decrypt_data6 XORs in. Used to -/// trial-decrypt candidate bytecode positions without mutating the buffer. -pub(crate) fn lfsr_keystream(out: &mut [u8]) { - let mut state: u32 = 1; - for byte in out.iter_mut() { - let mut b: u8 = 0; - for k in 0..8u32 { - b |= ((state & 1) << k) as u8; - state <<= 1; - if state & 0x8000 != 0 { - state ^= 0x8003; - } - } - *byte = b; - } -} - -/// Scan stage4/stage5 for the encrypted custom-decryptor bytecode block. The -/// raw byte at p+95 is used by decrypt_data6 as the iteration count. We trial- -/// decrypt that many bytes with the LFSR keystream and accept the first -/// position where the byte stream parses as a valid opcode sequence ending in -/// 195 (ret). -pub(crate) fn find_bytecode_offset(data: &[u8], base: u32, len: u32) -> Option { - let start = base as usize; - let end = (base.saturating_add(len)) as usize; - if end > data.len() { - return None; - } - let mut ks = [0u8; 256]; - lfsr_keystream(&mut ks); - // Scan forward from `start+16` on 16-byte boundaries relative to `start`. - // The bytecode block is positioned a fixed offset into stage4/stage5; the - // lowest parseable candidate is the real one (later ones are coincidental - // parses of trailing filler bytes that happen to map to valid opcodes). - // The enclosing buffer isn't necessarily 16-aligned to its absolute - // address in newer builds, so we anchor the stride to `start`. - let mut p = start + 16; - while p + 96 <= end { - let count = data[p + 95] as usize; - if count >= 8 && p + count <= end { - let mut buf = [0u8; 256]; - let take = count.min(256); - for i in 0..take { - buf[i] = data[p + i] ^ ks[i]; - } - if let Some(nops) = parse_bytecode_check(&buf[..take]) - && nops >= 4 - { - return Some(p as u32); - } - } - p += 16; - } - None -} - -/// Validate bytecode structure without allocating a `Vec` of ops. Returns -/// `Some(non_nop_op_count)` if the byte stream parses successfully as a valid -/// opcode sequence ending in 195 (ret), `None` otherwise. Allows non-trivial -/// bytecode filtering by op count. -pub(crate) fn parse_bytecode_check(buf: &[u8]) -> Option { - let mut i = 0usize; - let mut nops: usize = 0; - while i < buf.len() { - let b = buf[i]; - i += 1; - match b { - 4 | 44 | 52 => { - if i >= buf.len() { - return None; - } - i += 1; - nops += 1; - } - 144 => {} - 192 | 254 => { - if i >= buf.len() { - return None; - } - let mb = buf[i]; - i += 1; - let rm = mb & 7; - let mod_ = (mb >> 6) & 3; - let reg = (mb >> 3) & 7; - if mod_ != 3 || rm != 0 { - return None; - } - if reg > 1 { - return None; - } - if b == 192 { - if i >= buf.len() { - return None; - } - i += 1; - } - nops += 1; - } - 195 => return Some(nops), - _ => return None, - } - } - None -} - -/// Locate stage3's v4_val: the last non-zero dword in the buffer, anchored -/// by the `C3 CC CC CC` (ret + 3 int3) immediately before it. -pub(crate) fn find_v4_offset(data: &[u8], base: u32, len: u32) -> Option { - let start = base as usize; - let end = (base.saturating_add(len)) as usize; - if end > data.len() || end < start + 4 { - return None; - } - // walk backwards looking for the first non-zero byte - let mut i = end; - while i > start && data[i - 1] == 0 { - i -= 1; - } - if i < start + 4 { - return None; - } - // v_val occupies the 4 bytes ending at i (rounded up to dword boundary) - let v_end = i; - let v_start = ((v_end + 3) & !3).saturating_sub(4); - // require that the 4 bytes preceding v_val match `C3 CC CC CC` - if v_start < start + 4 || data[v_start - 4..v_start] != [0xC3, 0xCC, 0xCC, 0xCC] { - return None; - } - Some(v_start as u32) -} - -/// Scan a sub-buffer for an ASCII needle; return its absolute position. -pub(crate) fn find_str_pos(data: &[u8], base: u32, len: u32, needle: &[u8]) -> Option { - let start = base as usize; - let end = (base.saturating_add(len)) as usize; - if end > data.len() || needle.is_empty() { - return None; - } - data[start..end] - .windows(needle.len()) - .position(|w| w == needle) - .map(|rel| (start + rel) as u32) -} - -pub(crate) fn get_string_to_null(data: &[u8], offset: u32) -> String { - let start = offset as usize; - if start >= data.len() { - return String::new(); - } - // Bounded: an unterminated run must never walk off the end of the buffer - // (panic) or scan unboundedly into unrelated data. - let limit = start.saturating_add(4096).min(data.len()); - let mut i = start; - while i < limit && data[i] != 0 { - i += 1; - } - String::from_utf8_lossy(&data[start..i]).into_owned() -} - -/// Read a PE section-name field: exactly 8 bytes, NOT necessarily -/// NUL-terminated (a full-width name like `.textbss` has no NUL at all). -/// Returns the name with trailing NULs stripped. Using `get_string_to_null` -/// here would run past the field into the VirtualSize/VirtualAddress dwords. -pub(crate) fn section_name(data: &[u8], offset: u32) -> String { - let start = offset as usize; - let Some(field) = data.get(start..start + 8) else { - return String::new(); - }; - let end = field.iter().position(|&b| b == 0).unwrap_or(8); - String::from_utf8_lossy(&field[..end]).into_owned() -} - -// --------------------------------------------------------------------------- -// AES primitives -// --------------------------------------------------------------------------- - -/// One AES-CBC-like round over a 16-byte block in `d` at `pos`, using the -/// expanded key schedule stored in `d` at `key_offset`. Works entirely within -/// the single `d` buffer (both ciphertext and key schedule live there). -pub(crate) fn aes_round(d: &mut [u8], pos: u32, key_offset: u32, round: u32) { - let cm1 = &COLUMMIX1; - let cm2 = &COLUMMIX2; - let cm3 = &COLUMMIX3; - let cm4 = &COLUMMIX4; - let sbox = &SBOX; - - let mut n0 = get_u32(d, pos).swap_bytes() ^ get_u32(d, key_offset); - let mut n1 = - get_u32(d, pos.wrapping_add(4)).swap_bytes() ^ get_u32(d, key_offset.wrapping_add(4)); - let mut n2 = - get_u32(d, pos.wrapping_add(8)).swap_bytes() ^ get_u32(d, key_offset.wrapping_add(8)); - let mut n3 = - get_u32(d, pos.wrapping_add(12)).swap_bytes() ^ get_u32(d, key_offset.wrapping_add(12)); - - let mut r = 1u32; - while r < round { - let off = key_offset.wrapping_add(r.wrapping_mul(16)); - let a = get_u32(cm2, ((n3 >> 16) & 0xFF) * 4) - ^ get_u32(cm3, ((n2 >> 8) & 0xFF) * 4) - ^ get_u32(cm1, ((n0 >> 24) & 0xFF) * 4) - ^ get_u32(cm4, (n1 & 0xFF) * 4) - ^ get_u32(d, off); - let b = get_u32(cm2, ((n0 >> 16) & 0xFF) * 4) - ^ get_u32(cm1, ((n1 >> 24) & 0xFF) * 4) - ^ get_u32(cm3, ((n3 >> 8) & 0xFF) * 4) - ^ get_u32(cm4, (n2 & 0xFF) * 4) - ^ get_u32(d, off.wrapping_add(4)); - let c = get_u32(cm2, ((n1 >> 16) & 0xFF) * 4) - ^ get_u32(cm3, ((n0 >> 8) & 0xFF) * 4) - ^ get_u32(cm1, ((n2 >> 24) & 0xFF) * 4) - ^ get_u32(cm4, (n3 & 0xFF) * 4) - ^ get_u32(d, off.wrapping_add(8)); - let e = get_u32(cm3, ((n1 >> 8) & 0xFF) * 4) - ^ get_u32(cm2, ((n2 >> 16) & 0xFF) * 4) - ^ get_u32(cm1, ((n3 >> 24) & 0xFF) * 4) - ^ get_u32(cm4, (n0 & 0xFF) * 4) - ^ get_u32(d, off.wrapping_add(12)); - n0 = a; - n1 = b; - n2 = c; - n3 = e; - r = r.wrapping_add(1); - } - - let s0 = (get_u32(sbox, ((n0 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n3 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n2 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n1 & 0xFF) * 4) & 0x0000_00FF); - let s1 = (get_u32(sbox, ((n1 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n0 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n3 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n2 & 0xFF) * 4) & 0x0000_00FF); - let s2 = (get_u32(sbox, ((n2 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n1 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n0 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n3 & 0xFF) * 4) & 0x0000_00FF); - let s3 = (get_u32(sbox, ((n3 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n2 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n1 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n0 & 0xFF) * 4) & 0x0000_00FF); - - let last = key_offset.wrapping_add(round.wrapping_mul(16)); - n0 = s0 ^ get_u32(d, last); - n1 = s1 ^ get_u32(d, last.wrapping_add(4)); - n2 = s2 ^ get_u32(d, last.wrapping_add(8)); - n3 = s3 ^ get_u32(d, last.wrapping_add(12)); - - write_u32(d, pos, n0.swap_bytes()); - write_u32(d, pos.wrapping_add(4), n1.swap_bytes()); - write_u32(d, pos.wrapping_add(8), n2.swap_bytes()); - write_u32(d, pos.wrapping_add(12), n3.swap_bytes()); -} - -/// AES-CBC-like decryption over `size` bytes starting at `pos` in `d`. -/// The key schedule lives at `key_offset` within the same buffer `d`. -pub(crate) fn aes_decrypt(d: &mut [u8], pos: u32, size: u32, key_offset: u32) { - let mut prev = [0u8; 16]; - let mut cur = [0u8; 16]; - let round = get_u16(d, key_offset.wrapping_add(2)) as u32; - let blocks = size >> 4; - for i in 0..blocks { - let p = pos.wrapping_add(i.wrapping_mul(16)); - let pi = p as usize; - cur.copy_from_slice(&d[pi..pi + 16]); - aes_round(d, p, key_offset.wrapping_add(4), round); - for j in 0..16 { - d[pi + j] ^= prev[j]; - } - prev = cur; - } -} - -/// [`aes_decrypt`] variant reading the key schedule from a separate snapshot -/// slice instead of the data buffer. `ks` is a snapshot of `d[key_offset..]` -/// taken by [`aes_schedule_snapshot`] (round count at `ks[2]`, round keys from -/// `ks[4]`), so the schedule extent is exactly right by construction. Used by -/// the parallel block fan-out, where each worker owns a disjoint `&mut` span -/// of the image and cannot read the schedule out of the shared buffer. -pub(crate) fn aes_decrypt_ks(ks: &[u8], d: &mut [u8], pos: u32, size: u32) { - let mut prev = [0u8; 16]; - let mut cur = [0u8; 16]; - let round = u16::from_le_bytes([ks[2], ks[3]]) as u32; - let sched = &ks[4..]; - let blocks = size >> 4; - for i in 0..blocks { - let p = pos.wrapping_add(i.wrapping_mul(16)); - let pi = p as usize; - cur.copy_from_slice(&d[pi..pi + 16]); - aes_round_ks(sched, d, p, round); - for j in 0..16 { - d[pi + j] ^= prev[j]; - } - prev = cur; - } -} - -/// [`aes_round`] with the round keys in a separate slice (see -/// [`aes_decrypt_ks`]). Identical math; only the key source differs. -fn aes_round_ks(ks: &[u8], d: &mut [u8], pos: u32, round: u32) { - let cm1 = &COLUMMIX1; - let cm2 = &COLUMMIX2; - let cm3 = &COLUMMIX3; - let cm4 = &COLUMMIX4; - let sbox = &SBOX; - let k = |i: u32| get_u32(ks, i); - - let mut n0 = get_u32(d, pos).swap_bytes() ^ k(0); - let mut n1 = get_u32(d, pos.wrapping_add(4)).swap_bytes() ^ k(4); - let mut n2 = get_u32(d, pos.wrapping_add(8)).swap_bytes() ^ k(8); - let mut n3 = get_u32(d, pos.wrapping_add(12)).swap_bytes() ^ k(12); - - let mut r = 1u32; - while r < round { - let off = r.wrapping_mul(16); - let a = get_u32(cm2, ((n3 >> 16) & 0xFF) * 4) - ^ get_u32(cm3, ((n2 >> 8) & 0xFF) * 4) - ^ get_u32(cm1, ((n0 >> 24) & 0xFF) * 4) - ^ get_u32(cm4, (n1 & 0xFF) * 4) - ^ k(off); - let b = get_u32(cm2, ((n0 >> 16) & 0xFF) * 4) - ^ get_u32(cm1, ((n1 >> 24) & 0xFF) * 4) - ^ get_u32(cm3, ((n3 >> 8) & 0xFF) * 4) - ^ get_u32(cm4, (n2 & 0xFF) * 4) - ^ k(off.wrapping_add(4)); - let c = get_u32(cm2, ((n1 >> 16) & 0xFF) * 4) - ^ get_u32(cm3, ((n0 >> 8) & 0xFF) * 4) - ^ get_u32(cm1, ((n2 >> 24) & 0xFF) * 4) - ^ get_u32(cm4, (n3 & 0xFF) * 4) - ^ k(off.wrapping_add(8)); - let e = get_u32(cm3, ((n1 >> 8) & 0xFF) * 4) - ^ get_u32(cm2, ((n2 >> 16) & 0xFF) * 4) - ^ get_u32(cm1, ((n3 >> 24) & 0xFF) * 4) - ^ get_u32(cm4, (n0 & 0xFF) * 4) - ^ k(off.wrapping_add(12)); - n0 = a; - n1 = b; - n2 = c; - n3 = e; - r = r.wrapping_add(1); - } - - let s0 = (get_u32(sbox, ((n0 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n3 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n2 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n1 & 0xFF) * 4) & 0x0000_00FF); - let s1 = (get_u32(sbox, ((n1 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n0 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n3 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n2 & 0xFF) * 4) & 0x0000_00FF); - let s2 = (get_u32(sbox, ((n2 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n1 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n0 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n3 & 0xFF) * 4) & 0x0000_00FF); - let s3 = (get_u32(sbox, ((n3 >> 24) & 0xFF) * 4) & 0xFF00_0000) - | (get_u32(sbox, ((n2 >> 16) & 0xFF) * 4) & 0x00FF_0000) - | (get_u32(sbox, ((n1 >> 8) & 0xFF) * 4) & 0x0000_FF00) - | (get_u32(sbox, (n0 & 0xFF) * 4) & 0x0000_00FF); - - let last = round.wrapping_mul(16); - n0 = s0 ^ k(last); - n1 = s1 ^ k(last.wrapping_add(4)); - n2 = s2 ^ k(last.wrapping_add(8)); - n3 = s3 ^ k(last.wrapping_add(12)); - - write_u32(d, pos, n0.swap_bytes()); - write_u32(d, pos.wrapping_add(4), n1.swap_bytes()); - write_u32(d, pos.wrapping_add(8), n2.swap_bytes()); - write_u32(d, pos.wrapping_add(12), n3.swap_bytes()); -} - -/// Snapshot the AES key schedule at `key_offset` for [`aes_decrypt_ks`]: -/// `d[key_offset .. key_offset + 4 + (round+1)*16]` where `round` is read from -/// the schedule header. Returns `None` when the header is truncated or the -/// round count is implausible (corrupt input — the same bytes would otherwise -/// drive reads past the buffer). -pub(crate) fn aes_schedule_snapshot(d: &[u8], key_offset: u32) -> Option> { - let base = key_offset as usize; - let round = u16::from_le_bytes([*d.get(base + 2)?, *d.get(base + 3)?]) as usize; - if round > 64 { - return None; - } - let end = base.checked_add(4 + (round + 1) * 16)?; - if end > d.len() { - return None; - } - Some(d[base..end].to_vec()) -} - -// --------------------------------------------------------------------------- -// Checksum primitives -// --------------------------------------------------------------------------- - -/// CRC32-based checksum over a (offset, length) descriptor pair embedded in -/// `d` at `pos`. Returns `crc32(d[offset..offset+length]) ^ length`. -pub(crate) fn calculate_checksum(d: &[u8], pos: u32) -> u32 { - let offset = get_u32(d, pos); - let length = get_u32(d, pos.wrapping_add(4)); - crc32::compute(&d[offset as usize..(offset + length) as usize]) ^ length -} - -/// CRC32 chained checksum. The (offset, length) descriptor at `pos` is read -/// from `d`; the bytes themselves are read from the separate `clean` buffer -/// (the original file image). `start` is the initial CRC accumulator. -pub(crate) fn calculate_checksum2(d: &[u8], clean: &[u8], pos: u32, start: u32) -> u32 { - let offset = get_u32(d, pos); - let length = get_u32(d, pos.wrapping_add(4)); - crc32::append(start, &clean[offset as usize..(offset + length) as usize]) -} - -// --------------------------------------------------------------------------- -// Decompression (Huffman/LZ) -// --------------------------------------------------------------------------- - -/// Huffman/LZ decompression operating entirely within a single `d` buffer. -/// Reads `s_size` bytes from `src`, writes `d_size` bytes to `dest`. -/// The Huffman table lives at `key_offset` within `d`. -/// -/// Returns `true` when exactly `d_size` bytes were written (full success), -/// `false` on any corruption-triggered early exit. The PE32 eighth-stage key -/// brute force uses this status to discriminate the correct key. -pub(crate) fn decompress( - d: &mut [u8], - src: u32, - mut dest: u32, - key_offset: u32, - s_size: u32, - d_size: u32, -) -> bool { - // Bound the scratch allocation: a corrupt descriptor could request a - // multi-gigabyte source size, and an allocation failure aborts the process - // (uncatchable). Real payloads are far below this. - if s_size as u64 > super::MAX_IMAGE_SIZE { - return false; - } - DECOMPRESS_SCRATCH.with_borrow_mut(|buf| { - let mut bit_pos: i32 = 0; - let need = (s_size as usize).saturating_add(3); - if buf.len() < need { - buf.resize(need, 0); - } - let mut buf_off: u32 = 0; - let mut src_consumed: i32 = 0; - let mut pending: u32 = 0; - let mut written: u32 = 0; - let src_u = src as usize; - let s_size_u = s_size as usize; - // The bit-reader's final get_u32 may read up to 3 bytes past s_size; those - // must be zero. Reused scratch can hold stale bytes there, so zero them - // before copying the (exactly s_size) source over the head. - buf[s_size_u] = 0; - buf[s_size_u + 1] = 0; - buf[s_size_u + 2] = 0; - buf[..s_size_u].copy_from_slice(&d[src_u..src_u + s_size_u]); - - while (src_consumed as u32) < s_size && written < d_size { - let word = get_u32(&buf[..], buf_off) >> bit_pos; - let tab_addr = key_offset.wrapping_add((word & 0xFF).wrapping_mul(3)); - let mut tab = get_u16(d, tab_addr); - let bits: u8; - if (tab & 0x8000) != 0 { - tab &= 0x7FFF; - bits = d[tab_addr as usize + 2]; - } else { - let mut b2 = d[tab_addr as usize + 2]; - // A Huffman code longer than 32 bits cannot exist; a larger - // length byte comes from a corrupt table, and `1 << b2` would - // panic (debug) or wrap (release) on it. - if b2 >= 32 { - return false; - } - let mut mask: u32 = 1u32 << b2; - b2 = b2.wrapping_add(1); - let mut idx = (tab & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; - let mut t2 = get_u16(d, key_offset.wrapping_add(idx.wrapping_mul(3))); - // A corrupt table can form a non-terminal cycle; cap the walk so it - // fails instead of spinning forever. - let mut depth = 0u32; - while (t2 & 0x8000) == 0 { - depth += 1; - if depth > 64 { - return false; - } - mask <<= 1; - b2 = b2.wrapping_add(1); - idx = (t2 & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; - t2 = get_u16(d, key_offset.wrapping_add(idx.wrapping_mul(3))); - } - tab = t2 & 0x7FFF; - bits = b2; - } - bit_pos += bits as i32; - let advance = bit_pos / 8; - buf_off = buf_off.wrapping_add(advance as u32); - src_consumed += advance; - bit_pos %= 8; - - let mode = (tab as u32) & 0x300; - let payload = (tab as u32) & 0xFF; - let step: u32; - match mode { - 0 => { - step = 1; - d[dest as usize] = payload as u8; - } - 0x100 => { - step = 0; - if pending >= 256 { - // corrupt input: stop decompressing (diagnostics go to caller/log, not stdout) - return false; - } - pending = if pending == 0 { - payload - } else { - (pending << 8) | payload - }; - } - 0x200 => { - if pending == 0 { - pending = 1; - } - step = pending.wrapping_mul(payload); - if step.wrapping_add(written) > d_size { - return false; - } - // Run-fill replicates the unit just written before `dest`. A - // corrupt stream can emit one of these before anything has been - // written, so guard against reading before the buffer start - // (an unsigned underflow would index astronomically far OOB). - match payload { - 1 => { - if dest < 1 { - return false; - } - let v = d[(dest as usize) - 1]; - for k in 0..pending { - d[(dest + k) as usize] = v; - } - } - 2 => { - if dest < 2 { - return false; - } - let v = get_u16(d, dest.wrapping_sub(2)); - for k in 0..pending { - write_u16(d, dest.wrapping_add(k.wrapping_mul(2)), v as u32); - } - } - 4 => { - if dest < 4 { - return false; - } - let v = get_u32(d, dest.wrapping_sub(4)); - for k in 0..pending { - write_u32(d, dest.wrapping_add(k.wrapping_mul(4)), v); - } - } - _ => { - // Only unit widths 1/2/4 exist. Any other payload comes - // from a corrupt stream: previously this wrote nothing - // yet still counted `step` bytes as written, leaving - // stale-buffer holes that later stages treated as - // plaintext. Report corruption instead. - return false; - } - } - pending = 0; - } - _ => { - step = payload; - if written.wrapping_add(payload) > d_size - || pending.wrapping_add(payload) > written - { - return false; - } - let back = pending.wrapping_add(payload); - for k in 0..payload { - d[(dest + k) as usize] = d[(dest + k - back) as usize]; - } - pending = 0; - } - } - - dest = dest.wrapping_add(step); - written = written.wrapping_add(step); - if bits == 0 && step == 0 { - // Corrupt table: no input bits consumed and no output bytes - // written, so the loop condition can never advance — an - // infinite loop (and `catch_unpack` traps panics, not hangs). - // Every real symbol consumes ≥ 1 bit, so a valid stream can - // never hit this. - return false; - } - } - src_consumed += if bit_pos != 0 { 1 } else { 0 }; - // Mismatch in consumed/written sizes indicates corrupt input; the unpack - // result will then fail downstream checks. No stdout diagnostics here — - // the pure core stays I/O-free; surface errors via the caller/logfile. - let _ = src_consumed; - written == d_size - }) -} - -/// Walk the Huffman table at `key_offset` and snapshot its bytes for -/// [`decompress_tbl`]. The table is a forest of 256 root entries (3 bytes -/// each); non-terminal entries point at a child index pair. Returns `None` -/// when the table is truncated or self-referential past the buffer (corrupt -/// input — the same bytes would otherwise drive reads out of bounds). -pub(crate) fn huffman_table_snapshot(d: &[u8], key_offset: u32) -> Option> { - let mut visited = vec![false; 0x1_0000usize]; - let mut stack: Vec = (0..256).collect(); - let mut max_idx: u32 = 255; - while let Some(idx) = stack.pop() { - if idx >= 0x1_0000 || visited[idx as usize] { - continue; - } - visited[idx as usize] = true; - let off = key_offset as usize + idx as usize * 3; - if off + 3 > d.len() { - return None; - } - let t = get_u16(d, key_offset.wrapping_add(idx.wrapping_mul(3))); - if (t & 0x8000) == 0 { - let child = (t & 0x7FFF) as u32; - max_idx = max_idx.max(child).max(child.wrapping_add(1)); - stack.push(child); - stack.push(child.wrapping_add(1)); - } - } - let end = key_offset as usize + (max_idx as usize + 1) * 3; - if end > d.len() { - return None; - } - Some(d[key_offset as usize..end].to_vec()) -} - -/// [`decompress`] variant reading the Huffman table from a separate snapshot -/// slice (see [`huffman_table_snapshot`]) instead of the data buffer. Used by -/// the parallel block fan-out, where each worker owns a disjoint `&mut` span -/// and cannot read the table out of the shared image. Table reads are bounds -/// checked against the snapshot — past-the-end means corrupt table, reported -/// as `false` rather than a panic. -pub(crate) fn decompress_tbl( - tab: &[u8], - d: &mut [u8], - src: u32, - mut dest: u32, - s_size: u32, - d_size: u32, -) -> bool { - if s_size as u64 > super::MAX_IMAGE_SIZE { - return false; - } - DECOMPRESS_SCRATCH.with_borrow_mut(|buf| { - // Table reads, bounds-checked against the snapshot. - let tab16 = |addr: usize| -> Option { - let b = tab.get(addr..addr + 3)?; - Some(u16::from_le_bytes([b[0], b[1]])) - }; - let tab8 = |addr: usize| -> Option { tab.get(addr + 2).copied() }; - - let mut bit_pos: i32 = 0; - let need = (s_size as usize).saturating_add(3); - if buf.len() < need { - buf.resize(need, 0); - } - let mut buf_off: u32 = 0; - let mut src_consumed: i32 = 0; - let mut pending: u32 = 0; - let mut written: u32 = 0; - let src_u = src as usize; - let s_size_u = s_size as usize; - buf[s_size_u] = 0; - buf[s_size_u + 1] = 0; - buf[s_size_u + 2] = 0; - buf[..s_size_u].copy_from_slice(&d[src_u..src_u + s_size_u]); - - while (src_consumed as u32) < s_size && written < d_size { - let word = get_u32(&buf[..], buf_off) >> bit_pos; - let tab_addr = ((word & 0xFF).wrapping_mul(3)) as usize; - let mut tab = match tab16(tab_addr) { - Some(t) => t, - None => { - return false; - } - }; - let bits: u8; - if (tab & 0x8000) != 0 { - tab &= 0x7FFF; - bits = match tab8(tab_addr) { - Some(b) => b, - None => return false, - }; - } else { - let mut b2 = match tab8(tab_addr) { - Some(b) => b, - None => return false, - }; - if b2 >= 32 { - return false; - } - let mut mask: u32 = 1u32 << b2; - b2 = b2.wrapping_add(1); - let mut idx = (tab & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; - let mut t2 = match tab16(idx as usize * 3) { - Some(t) => t, - None => { - return false; - } - }; - // A corrupt table can form a non-terminal cycle; cap the walk so it - // fails instead of spinning forever. - let mut depth = 0u32; - while (t2 & 0x8000) == 0 { - depth += 1; - if depth > 64 { - return false; - } - mask <<= 1; - b2 = b2.wrapping_add(1); - idx = (t2 & 0x7FFF) as u32 + if (word & mask) != 0 { 1 } else { 0 }; - t2 = match tab16(idx as usize * 3) { - Some(t) => t, - None => { - return false; - } - }; - } - tab = t2 & 0x7FFF; - bits = b2; - } - bit_pos += bits as i32; - let advance = bit_pos / 8; - buf_off = buf_off.wrapping_add(advance as u32); - src_consumed += advance; - bit_pos %= 8; - - let mode = (tab as u32) & 0x300; - let payload = (tab as u32) & 0xFF; - let step: u32; - match mode { - 0 => { - step = 1; - d[dest as usize] = payload as u8; - } - 0x100 => { - step = 0; - if pending >= 256 { - return false; - } - pending = if pending == 0 { - payload - } else { - (pending << 8) | payload - }; - } - 0x200 => { - if pending == 0 { - pending = 1; - } - step = pending.wrapping_mul(payload); - if step.wrapping_add(written) > d_size { - return false; - } - // Run-fill replicates the unit just written before `dest` - // (see `decompress` for the underflow rationale). - match payload { - 1 => { - if dest < 1 { - return false; - } - let v = d[(dest as usize) - 1]; - for k in 0..pending { - d[(dest + k) as usize] = v; - } - } - 2 => { - if dest < 2 { - return false; - } - let v = get_u16(d, dest.wrapping_sub(2)); - for k in 0..pending { - write_u16(d, dest.wrapping_add(k.wrapping_mul(2)), v as u32); - } - } - 4 => { - if dest < 4 { - return false; - } - let v = get_u32(d, dest.wrapping_sub(4)); - for k in 0..pending { - write_u32(d, dest.wrapping_add(k.wrapping_mul(4)), v); - } - } - _ => { - return false; - } - } - pending = 0; - } - _ => { - step = payload; - if written.wrapping_add(payload) > d_size - || pending.wrapping_add(payload) > written - { - return false; - } - let back = pending.wrapping_add(payload); - for k in 0..payload { - d[(dest + k) as usize] = d[(dest + k - back) as usize]; - } - pending = 0; - } - } - - dest = dest.wrapping_add(step); - written = written.wrapping_add(step); - if bits == 0 && step == 0 { - return false; - } - } - src_consumed += if bit_pos != 0 { 1 } else { 0 }; - let _ = src_consumed; - written == d_size - }) -} -// Decrypt primitives (free-function wrappers) -// --------------------------------------------------------------------------- - -// --------------------------------------------------------------------------- -// PE32 (32-bit) helpers -// --------------------------------------------------------------------------- - -/// PE32 shell-table locator. Walks the shell region (`info[6]`) for a dword -/// equal to `info[6]` followed by a plausible shell size, returning the table -/// base (`candidate = off - 0x88`) when `candidate+0x58` holds a valid pointer. -pub(crate) fn find_tbl_pe32(data: &[u8], info: &[u32; 8]) -> Option { - let shell = info[6]; - if (data.len() as u64) < 0x100 { - return None; - } - let hi = (shell as u64) - .saturating_add(0x3000) - .min(data.len() as u64 - 0x100) as u32; - let mut off = shell; - while off < hi { - if off as usize + 8 <= data.len() { - let candidate = off.wrapping_sub(0x88); - if candidate >= shell && get_u32(data, off) == info[6] { - let shell_size_val = get_u32(data, off.wrapping_add(4)); - if shell_size_val > 0x1000 && shell_size_val < 0x100000 { - let v58_off = candidate.wrapping_add(0x58); - if (v58_off as usize + 4) <= data.len() { - let v58 = get_u32(data, v58_off); - if v58 > 0 && (v58 as usize) < data.len() { - return Some(candidate); - } - } - } - } - } - off = off.wrapping_add(4); - } - None -} - -/// Locate an LFSR-encrypted bytecode block (decrypt_data6 form) in a region. -/// `start_off` is the byte offset to begin scanning at, `scan_backward` -/// controls direction. Returns the relative offset of the block. Includes full -/// opcode-walk validation of candidate blocks. -pub(crate) fn find_lfsr_block( - data: &[u8], - base: u32, - size: u32, - start_off: u32, - scan_backward: bool, -) -> Option { - if size < 96 { - return None; - } - let mut ks = [0u8; 128]; - lfsr_keystream(&mut ks); - let check = |scan_off: u32| -> bool { - let abs_off = base.wrapping_add(scan_off) as usize; - if abs_off + 96 > data.len() { - return false; - } - let sz = data[abs_off + 95] as usize; - if !(10..=95).contains(&sz) { - return false; - } - let mut decoded = [0u8; 95]; - for bi in 0..sz { - decoded[bi] = data[abs_off + bi] ^ ks[bi]; - } - // Full bytecode validation (shared with the stage4/5 locator): every - // opcode must decode with a valid ModR/M and the stream must REACH a - // RET (0xC3) as an opcode. The previous check only required a 0xC3 - // byte *anywhere* in the window and accepted a walk that ran off the - // end without hitting RET — a `0x04 0xC3` (ADD 0xC3) tail passed, so - // coincidental LFSR-shaped garbage was accepted as a decryptor block. - parse_bytecode_check(&decoded[..sz]).is_some() - }; - if scan_backward { - let hi = size - 96; - if hi >= start_off { - let mut scan_off = hi; - loop { - if check(scan_off) { - return Some(scan_off); - } - if scan_off == start_off { - break; - } - scan_off -= 1; - } - } - } else { - let hi = size - 95; - let mut scan_off = start_off; - while scan_off < hi { - if check(scan_off) { - return Some(scan_off); - } - scan_off += 1; - } - } - None -} - -/// Slots discovered in the eighthStage for the marker-less layout. -pub(crate) struct EighthSlots { - /// Absolute address of the file-data decryptor LFSR bytecode block. The - /// fileCS chain pointer is derived downstream as `file_lfsr - 0x58`. - pub file_lfsr: u32, - /// Absolute address of the compressedInfo (ptr,size) table pointer slot. - pub compressed_info_ptr: u32, -} - -/// Marker-independent eighthStage slot discovery (PE32+ branch). -/// -/// Newer Crackproof builds (e.g. some native/managed DLLs) omit the -/// `pm\0\0cm\0\0` and `00 00 00 40 01 00 00 00` markers that the older layout's -/// walk3/walk4/walk5 slot derivation relies on. Instead this discovers the -/// slots structurally: -/// * Scan the eighthStage for every LFSR (decrypt_data6) bytecode block. -/// * The file decryptor is the LFSR block whose `fileCS = lfsr - 0x58` holds -/// a pointer sitting just past `info[3]` (smallest positive distance). -/// * `compressedInfo` is the pointer slot whose 16-byte target, after a -/// trial `decrypt_data5`, parses as a plausible (src,sSize,dst,dSize) -/// descriptor. -/// -/// Returns `None` if no plausible file LFSR is found. `eighth_start`/`eighth_dsz` -/// bound the search region; `info3` is `info[3]`; `compress_data_offset` is -/// `(!u32(file_data,0x1080)) + 0x1000`; `file_data_len` is the protected file -/// length. -#[allow(clippy::too_many_arguments)] -pub(crate) fn discover_eighth_slots( - data: &[u8], - eighth_start: u32, - eighth_dsz: u32, - info3: u32, - compress_data_offset: u32, - file_data_len: u32, -) -> Option { - // Collect all LFSR candidates (forward scan). - // - // Advance by 1 after each hit, NOT by 96. A false-positive LFSR match can sit - // just before the real file-decryptor block (observed on an il2cpp game - // assembly build, 2026-07-13: junk at rel=0x31C1, real block at 0x3210). - // Stepping by the LFSR body size then skips the real block and discovery - // fails. Byte-stepping is cheap: eighthStage is only a few KB. - let mut all_lfsrs: Vec = Vec::new(); - let mut scan_off: u32 = 0; - while scan_off + 95 < eighth_dsz { - match find_lfsr_block(data, eighth_start, eighth_dsz, scan_off, false) { - Some(found) => { - all_lfsrs.push(found); - scan_off = found + 1; - } - None => break, - } - } - - // Pick the file LFSR: prefer the candidate whose fileCS pointer sits the - // smallest positive distance past info[3]. - let mut off_file_lfsr: Option = None; - let mut best_dist: Option = None; - for &lfsr_off in &all_lfsrs { - if lfsr_off < 0x58 { - continue; - } - let cs_off = lfsr_off - 0x58; - let cs_val = get_u32(data, eighth_start.wrapping_add(cs_off)); - if !(0x1000 < cs_val && (cs_val as usize) < data.len()) { - continue; - } - if cs_val < info3 { - continue; - } - let dist = cs_val - info3; - if best_dist.is_none_or(|b| dist < b) { - best_dist = Some(dist); - off_file_lfsr = Some(lfsr_off); - } - } - // Fallback: last LFSR with any in-image fileCS pointer. - if off_file_lfsr.is_none() { - for &lfsr_off in all_lfsrs.iter().rev() { - if lfsr_off < 0x58 { - continue; - } - let cs_val = get_u32(data, eighth_start.wrapping_add(lfsr_off - 0x58)); - if 0x1000 < cs_val && (cs_val as usize) < data.len() { - off_file_lfsr = Some(lfsr_off); - break; - } - } - } - let off_file_lfsr = off_file_lfsr?; - let off_file_cs = off_file_lfsr - 0x58; - - // Trial-decrypt to find compressedInfo: the pointer slot in the data area - // (between fileCS region start and the LFSR) whose target parses as a valid - // (src,sSize,dst,dSize) descriptor after a transient decrypt_data5. - let scan_from = off_file_lfsr.saturating_sub(0x400); - let mut off_compressed_info: Option = None; - let mut doff = scan_from; - while doff < off_file_lfsr { - if doff == off_file_cs { - doff += 4; - continue; - } - let ptr_val = get_u32(data, eighth_start.wrapping_add(doff)); - if !(0x1000 < ptr_val && (ptr_val as usize) < data.len().saturating_sub(16)) { - doff += 4; - continue; - } - // Predict decrypt_data5(ptr_val, 16) without mutating: each dword is - // position-keyed and independent, so trial_decrypt5_u32 per dword. - let src2 = trial_decrypt5_u32(data, ptr_val); - let s_sz2 = trial_decrypt5_u32(data, ptr_val + 4); - let dst2 = trial_decrypt5_u32(data, ptr_val + 8); - let d_sz2 = trial_decrypt5_u32(data, ptr_val + 12); - let src_file_off = src2.wrapping_add(compress_data_offset); - let valid = s_sz2 > 0 - && s_sz2 < 0x200000 - && (src_file_off as u64 + s_sz2 as u64) <= file_data_len as u64 - && dst2 >= 0x1000 - && (dst2 as u64 + d_sz2 as u64) <= data.len() as u64 - && d_sz2 >= s_sz2 - && d_sz2 < 0x200000; - if valid { - off_compressed_info = Some(doff); - break; - } - doff += 4; - } - let off_compressed_info = off_compressed_info?; - - Some(EighthSlots { - file_lfsr: eighth_start.wrapping_add(off_file_lfsr), - compressed_info_ptr: eighth_start.wrapping_add(off_compressed_info), - }) -} - -/// PE32 `.text` dd8 key-formula selection with a skip decision. The packer keys -/// the per-page XOR either with `page+1` or `0x8000*(page+1)`; the formula is -/// not recorded. Replays the dd8 page pass on a scratch copy of sample pages -/// (25/50/75% of `.text`) under each formula and counts how many positions -/// decode to `0xCC` (int3 padding). -/// -/// Returns `Some(true)` for the `0x8000*(page+1)` formula, `Some(false)` for -/// `page+1`, or `None` when `.text` must NOT be dd8-decrypted at all. The packer -/// dd8-encrypts `.text` on EXEs (so unpacking must replay it) but leaves a native -/// DLL's `.text` plaintext; replaying dd8 there scrambles ~1 byte per 16-byte -/// block. The decision: dd8 only *restores* int3 padding when `.text` was -/// genuinely encrypted, so apply it only when the chosen formula's whole-page -/// 0xCC count rises *clearly* above the no-dd8 baseline; otherwise skip. -/// -/// "Clearly" matters: dd8 XORs 255 positions per page with pseudo-random bytes, -/// so on an already-plaintext `.text` it manufactures ~1 spurious `0xCC` per -/// sampled page for free (255/256 expected). A bare `best > baseline` test is -/// therefore biased towards *applying* dd8 on exactly the inputs that must skip -/// it — and a wrongly-applied dd8 is silent: it scrambles ~1 byte per 16 with no -/// error and nothing downstream (not even `integrity::check`, which only reads -/// 16 bytes at the entry point) notices. The [`MIN_DD8_NET_GAIN`] floor below is -/// the PE32 counterpart of the margin+floor `select_dd8_shift` already applies -/// on PE32+ for the same failure mode. -pub(crate) fn select_dd8_formula_pe32(data: &[u8], text_off: u32, text_size: u32) -> Option { - let num_pages_total = text_size / 0x1000; - let mut sample_pages: Vec = Vec::new(); - for frac in [0.25f64, 0.5, 0.75] { - let pg = (num_pages_total as f64 * frac) as u32; - if pg > 0 && pg < num_pages_total { - sample_pages.push(pg); - } - } - if sample_pages.is_empty() && num_pages_total > 1 { - sample_pages.push(num_pages_total / 2); - } - let score = |big: bool| -> i64 { - let mut total = 0i64; - for &sp in &sample_pages { - let pg_off = (text_off + sp * 0x1000) as usize; - if pg_off + 0x1000 > data.len() { - continue; - } - let mut buf = [0u8; 0x1000]; - buf.copy_from_slice(&data[pg_off..pg_off + 0x1000]); - let pk = if big { - 0x8000u32.wrapping_mul(sp.wrapping_add(1)) - } else { - sp.wrapping_add(1) - }; - let mut k = pk; - let rk = k.rotate_right(15); - k = rk; - for bi in 1..256u32 { - let rk = k.rotate_right(15); - let ri = rk.wrapping_add(bi); - k = ri.wrapping_add(bi); - let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize; - if tidx < buf.len() { - buf[tidx] ^= k as u8; - } - } - total += buf.iter().filter(|&&b| b == 0xCC).count() as i64; - } - total - }; - let s_small = score(false); - let s_big = score(true); - // Baseline: whole-page 0xCC over the same sample pages with NO dd8. dd8 only - // rewrites 255 bytes per page, so comparing the chosen formula's whole-page - // 0xCC against this baseline reveals whether dd8 *restores* int3 padding - // (count rises -> .text was packer-encrypted, apply) or merely scrambles - // already-plaintext code (count falls -> native-DLL .text left intact, skip). - let mut baseline: i64 = 0; - for &sp in &sample_pages { - let pg_off = (text_off + sp * 0x1000) as usize; - if pg_off + 0x1000 > data.len() { - continue; - } - baseline += data[pg_off..pg_off + 0x1000] - .iter() - .filter(|&&b| b == 0xCC) - .count() as i64; - } - let big = s_big > s_small; - let best = s_small.max(s_big); - // Minimum net 0xCC gain over the baseline before dd8 is applied. Noise on an - // already-plaintext `.text` is ~1 manufactured 0xCC per sampled page (3 pages - // -> ~3); every corpus build that genuinely needs dd8 gains +154 or more - // (observed +154 and +312), and the one native DLL that must skip scores -18. - // A floor of 32 sits ~10x above the noise and ~5x below the smallest true - // positive, so it changes no existing decision. - const MIN_DD8_NET_GAIN: i64 = 32; - let apply = best.saturating_sub(baseline) >= MIN_DD8_NET_GAIN; - if std::env::var("SEL_DIAG").is_ok() { - eprintln!( - "SEL pe32 dd8 s_small={} s_big={} baseline={} gain={} big={} apply={}", - s_small, - s_big, - baseline, - best - baseline, - big, - apply - ); - } - // When no interior pages could be sampled (tiny .text) we cannot measure the - // effect; preserve the historical behavior of applying dd8. - if sample_pages.is_empty() || apply { - Some(big) - } else { - None - } -} - -/// Read a NUL-terminated byte string starting at `off`, bounded to 512 bytes. -/// Returns the raw bytes up to the terminator (excluding it). -fn read_cstr_bounded(data: &[u8], off: u32) -> Vec { - let start = off as usize; - if start >= data.len() { - return Vec::new(); - } - let limit = (start + 512).min(data.len()); - let mut end = start; - while end < limit && data[end] != 0 { - end += 1; - } - data[start..end].to_vec() -} - -fn align_up_u32(value: u32, alignment: u32) -> u32 { - ((value.wrapping_add(alignment - 1)) / alignment).wrapping_mul(alignment) -} - -fn align_up_u64(value: u64, alignment: u64) -> u64 { - value.div_ceil(alignment) * alignment -} - -#[derive(Clone)] -enum ImportFunc { - Ordinal(u32), - Name(u16, Vec), -} - -struct ImportDesc { - time_date: u32, - fwd_chain: u32, - dll_name: Vec, - iat_rva: u32, - functions: Vec, -} - -/// Return true when PE32 imports already sit in the original `.idata` layout -/// (so no relocation to `.kmiat` is needed). May write the IAT data directory -/// (pe+0xD8). -pub(crate) fn pe32_imports_already_match_idata_layout(data: &mut [u8], pe_header: u32) -> bool { - let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32; - let sec_table = pe_header.wrapping_add(24).wrapping_add(opt_hdr_size); - let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32; - let import_rva = get_u32(data, pe_header.wrapping_add(0x80)); - let import_size = get_u32(data, pe_header.wrapping_add(0x84)); - let len = data.len() as u32; - if !(import_rva > 0 && import_size > 0) { - return false; - } - for idx in 0..num_sections { - let sec_off = sec_table.wrapping_add(idx * 40); - if (sec_off as usize + 40) > data.len() { - return false; - } - if &data[sec_off as usize..sec_off as usize + 6] != b".idata" { - continue; - } - let sec_va = get_u32(data, sec_off.wrapping_add(12)); - let sec_size = - get_u32(data, sec_off.wrapping_add(8)).max(get_u32(data, sec_off.wrapping_add(16))); - let sec_end = sec_va.wrapping_add(sec_size); - if !(sec_va <= import_rva - && import_rva < sec_end - && import_rva.wrapping_add(import_size) <= sec_end) - { - continue; - } - let first_oft = get_u32(data, import_rva); - let first_name = get_u32(data, import_rva.wrapping_add(12)); - let first_iat = get_u32(data, import_rva.wrapping_add(16)); - if !(sec_va <= first_oft - && first_oft < sec_end - && sec_va <= first_iat - && first_iat < sec_end) - { - return false; - } - if !(0x1000 < first_name && first_name < len) { - return false; - } - let dll_name = read_cstr_bounded(data, first_name); - let lower: Vec = dll_name.iter().map(|b| b.to_ascii_lowercase()).collect(); - if !lower.ends_with(b".dll") { - return false; - } - let mut iat_min = first_iat; - let mut iat_max = first_iat; - let mut idt_pos = import_rva; - while idt_pos.wrapping_add(20) <= len { - let oft_rva = get_u32(data, idt_pos); - let name_rva = get_u32(data, idt_pos.wrapping_add(12)); - let iat_rva = get_u32(data, idt_pos.wrapping_add(16)); - if oft_rva == 0 && name_rva == 0 && iat_rva == 0 { - break; - } - if !(sec_va <= oft_rva && oft_rva < sec_end && sec_va <= iat_rva && iat_rva < sec_end) { - return false; - } - let mut thunk = iat_rva; - while thunk.wrapping_add(4) <= sec_end { - let tv = get_u32(data, thunk); - thunk = thunk.wrapping_add(4); - if tv == 0 { - break; - } - } - iat_min = iat_min.min(iat_rva); - iat_max = iat_max.max(thunk); - idt_pos = idt_pos.wrapping_add(20); - } - if iat_max > iat_min { - write_u32(data, pe_header.wrapping_add(0xD8), iat_min); - write_u32(data, pe_header.wrapping_add(0xDC), iat_max - iat_min); - } - return true; - } - false -} - -/// Rebuild PE32 import metadata (descriptors, lookup tables, names) into the -/// last section as `.kmiat`, leaving the loader-written IAT in place. Mutates -/// `data` (may grow it). -pub(crate) fn move_pe32_imports_to_kmiat(data: &mut Vec, pe_header: u32) { - const SECTION_SIZE: u32 = 0x7000; - let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32; - let opt_hdr = pe_header.wrapping_add(24); - let sec_table = opt_hdr.wrapping_add(opt_hdr_size); - let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32; - if num_sections == 0 { - return; - } - let import_rva = get_u32(data, pe_header.wrapping_add(0x80)); - let import_size = get_u32(data, pe_header.wrapping_add(0x84)); - let len = data.len() as u32; - if !(0x1000 < import_rva && import_rva < len && import_size > 0 && import_size < SECTION_SIZE) { - return; - } - - let mut descriptors: Vec = Vec::new(); - let mut idt_pos = import_rva; - while idt_pos.wrapping_add(20) <= len { - let oft_rva = get_u32(data, idt_pos); - let time_date = get_u32(data, idt_pos.wrapping_add(4)); - let fwd_chain = get_u32(data, idt_pos.wrapping_add(8)); - let name_rva = get_u32(data, idt_pos.wrapping_add(12)); - let iat_rva = get_u32(data, idt_pos.wrapping_add(16)); - if oft_rva == 0 && name_rva == 0 && iat_rva == 0 { - break; - } - if !(0x1000 < name_rva && name_rva < len) { - break; - } - let dll_name = read_cstr_bounded(data, name_rva); - let thunk_rva = if 0x1000 < oft_rva && oft_rva < len { - oft_rva - } else { - iat_rva - }; - let mut functions: Vec = Vec::new(); - let mut thunk_pos = thunk_rva; - while 0x1000 < thunk_pos.wrapping_add(4) && thunk_pos.wrapping_add(4) <= len { - let thunk_val = get_u32(data, thunk_pos); - if thunk_val == 0 { - break; - } - if thunk_val & 0x8000_0000 != 0 { - functions.push(ImportFunc::Ordinal(thunk_val & 0xFFFF)); - } else { - let hint = if thunk_val.wrapping_add(2) <= len { - get_u16(data, thunk_val) - } else { - 0 - }; - let func_name = if thunk_val.wrapping_add(2) < len { - read_cstr_bounded(data, thunk_val.wrapping_add(2)) - } else { - Vec::new() - }; - functions.push(ImportFunc::Name(hint, func_name)); - } - thunk_pos = thunk_pos.wrapping_add(4); - } - descriptors.push(ImportDesc { - time_date, - fwd_chain, - dll_name, - iat_rva, - functions, - }); - idt_pos = idt_pos.wrapping_add(20); - } - if descriptors.is_empty() { - return; - } - - for desc in &mut descriptors { - let lower: Vec = desc - .dll_name - .iter() - .map(|b| b.to_ascii_lowercase()) - .collect(); - if lower.starts_with(b"api-ms-win-crt-") { - desc.dll_name = b"ucrtbase.dll".to_vec(); - } else { - desc.dll_name = lower; - } - } - descriptors.sort_by_key(|d| d.iat_rva); - - let last_sec = sec_table.wrapping_add((num_sections - 1) * 40); - let kmiat_rva = get_u32(data, last_sec.wrapping_add(12)); - // A zero last-section VA means a corrupt section table: building .kmiat at - // RVA 0 would zero the DOS/PE headers and emit a structurally broken image - // with no error. Bail and keep the original import table. - if kmiat_rva == 0 { - return; - } - // Grow the image when .kmiat overruns it, but cap the growth: a corrupt VA - // could otherwise request a multi-gigabyte allocation, which aborts the - // process (uncatchable). Use u64 math so a near-u32::MAX VA cannot wrap the - // end calculation the way the previous wrapping/plain-add mix could. - let kmiat_end = kmiat_rva as u64 + SECTION_SIZE as u64; - if kmiat_end > super::MAX_IMAGE_SIZE { - return; - } - if kmiat_end > data.len() as u64 { - data.resize(kmiat_end as usize, 0); - } - // Zero the .kmiat region. - for b in &mut data[kmiat_rva as usize..kmiat_end as usize] { - *b = 0; - } - - let idt_size = (descriptors.len() as u32 + 1) * 20; - let oft_start = kmiat_rva; - let mut idt_rva = oft_start; - for desc in &descriptors { - idt_rva = idt_rva.wrapping_add((desc.functions.len() as u32 + 1) * 4); - } - idt_rva = align_up_u32(idt_rva.wrapping_add(0x2C), 4); - - // Size check: compute the final name_pos and bail if it overruns .kmiat. - let mut name_pos_check = idt_rva.wrapping_add(idt_size); - for desc in &descriptors { - name_pos_check = name_pos_check.wrapping_add(desc.dll_name.len() as u32 + 1); - for func in &desc.functions { - if let ImportFunc::Name(_, fname) = func { - name_pos_check = name_pos_check.wrapping_add(2 + fname.len() as u32 + 1); - } - } - } - if name_pos_check > kmiat_rva.wrapping_add(SECTION_SIZE) { - // Section too small; keep existing import table untouched. - return; - } - - let mut oft_pos = oft_start; - let mut name_pos = idt_rva.wrapping_add(idt_size); - for (idx, desc) in descriptors.iter().enumerate() { - let idt_entry = idt_rva.wrapping_add(idx as u32 * 20); - let current_oft = oft_pos; - write_u32(data, idt_entry, current_oft); - write_u32(data, idt_entry.wrapping_add(4), desc.time_date); - write_u32(data, idt_entry.wrapping_add(8), desc.fwd_chain); - let dll_name_pos = name_pos; - write_u32(data, idt_entry.wrapping_add(12), dll_name_pos); - write_u32(data, idt_entry.wrapping_add(16), desc.iat_rva); - - let dnp = dll_name_pos as usize; - data[dnp..dnp + desc.dll_name.len()].copy_from_slice(&desc.dll_name); - data[dnp + desc.dll_name.len()] = 0; - name_pos = name_pos.wrapping_add(desc.dll_name.len() as u32 + 1); - - for func in &desc.functions { - match func { - ImportFunc::Ordinal(ord) => { - write_u32(data, oft_pos, 0x8000_0000 | ord); - } - ImportFunc::Name(hint, fname) => { - let hint_name_rva = name_pos; - write_u32(data, oft_pos, hint_name_rva); - write_u16(data, hint_name_rva, *hint as u32); - let fp = (hint_name_rva + 2) as usize; - data[fp..fp + fname.len()].copy_from_slice(fname); - data[fp + fname.len()] = 0; - name_pos = name_pos.wrapping_add(2 + fname.len() as u32 + 1); - } - } - oft_pos = oft_pos.wrapping_add(4); - } - write_u32(data, oft_pos, 0); - oft_pos = oft_pos.wrapping_add(4); - } - // Null-terminator IDT entry (20 zero bytes) after the last descriptor. - let term = idt_rva.wrapping_add(descriptors.len() as u32 * 20) as usize; - for b in &mut data[term..term + 20] { - *b = 0; - } - - let ls = last_sec as usize; - data[ls..ls + 8].copy_from_slice(b".kmiat\x00\x00"); - write_u32(data, last_sec.wrapping_add(8), SECTION_SIZE); - write_u32(data, last_sec.wrapping_add(16), SECTION_SIZE); - write_u32(data, last_sec.wrapping_add(36), 0xE000_0060); - write_u32(data, pe_header.wrapping_add(0x80), idt_rva); - write_u32(data, pe_header.wrapping_add(0x84), idt_size); - write_u32( - data, - pe_header.wrapping_add(80), - kmiat_rva.wrapping_add(SECTION_SIZE), - ); -} - -/// Convert the unpacked RVA-addressed image back to a compact PE file layout -/// (headers at 0x400, sections packed consecutively, FileAlignment 0x200). -/// Returns `None` if the accumulated output size wraps or exceeds -/// [`super::MAX_IMAGE_SIZE`]: the final allocation is sized from header-derived -/// section data, and an uncapped `vec![0; n]` from a corrupt header would abort -/// the process (which `catch_unpack` cannot trap). -pub(crate) fn compact_memory_image_to_pe(data: &[u8], pe_header: u32) -> Option> { - const FILE_ALIGNMENT: u32 = 0x200; - const HEADER_SIZE: u32 = 0x400; - let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32; - let opt_hdr = pe_header.wrapping_add(24); - let sec_table = opt_hdr.wrapping_add(opt_hdr_size); - let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32; - - struct SecLayout { - sec_off: u32, - va: u32, - vsize: u32, - raw_ptr: u32, - raw_size: u32, - } - - let mut raw_cursor: u64 = HEADER_SIZE as u64; - let mut raw_layout: Vec = Vec::new(); - for idx in 0..num_sections { - let sec_off = sec_table.wrapping_add(idx * 40); - let vsize = get_u32(data, sec_off.wrapping_add(8)); - let va = get_u32(data, sec_off.wrapping_add(12)); - let sd_start = va as usize; - let sd_end = if (va.wrapping_add(vsize) as usize) <= data.len() { - va.wrapping_add(vsize) as usize - } else { - data.len() - }; - let section_data: &[u8] = if sd_start <= sd_end && sd_start <= data.len() { - &data[sd_start..sd_end] - } else { - &[] - }; - - let mut last_nonzero: i64 = -1; - for pos in (0..section_data.len()).rev() { - if section_data[pos] != 0 { - last_nonzero = pos as i64; - break; - } - } - let meaningful = if last_nonzero >= 0 { - (last_nonzero + 1) as u32 - } else { - 0 - }; - let mut raw_size = if meaningful != 0 { - align_up_u32(meaningful, FILE_ALIGNMENT) - } else { - 0 - }; - if vsize != 0 && raw_size == 0 { - raw_size = FILE_ALIGNMENT; - } - raw_size = raw_size.min(align_up_u32(section_data.len() as u32, FILE_ALIGNMENT)); - - let raw_ptr = if raw_size != 0 { raw_cursor as u32 } else { 0 }; - raw_layout.push(SecLayout { - sec_off, - va, - vsize, - raw_ptr, - raw_size, - }); - if raw_size != 0 { - // Accumulate in u64 and cap: section sizes are header-derived, and - // a corrupt table could otherwise wrap raw_cursor (small alloc, - // huge recorded raw_ptrs → OOB panic) or request an abort-sized - // allocation. - raw_cursor = align_up_u64(raw_cursor + raw_size as u64, FILE_ALIGNMENT as u64); - if raw_cursor > super::MAX_IMAGE_SIZE { - return None; - } - } - } - - let mut compact = vec![0u8; raw_cursor as usize]; - let hdr_copy = (HEADER_SIZE as usize).min(data.len()); - compact[..hdr_copy].copy_from_slice(&data[..hdr_copy]); - write_u32(&mut compact, opt_hdr.wrapping_add(36), FILE_ALIGNMENT); - write_u32(&mut compact, opt_hdr.wrapping_add(60), HEADER_SIZE); - - for sl in &raw_layout { - write_u32(&mut compact, sl.sec_off.wrapping_add(16), sl.raw_size); - write_u32(&mut compact, sl.sec_off.wrapping_add(20), sl.raw_ptr); - if sl.raw_size != 0 { - let sd_start = sl.va as usize; - let sd_end = if (sl.va.wrapping_add(sl.vsize) as usize) <= data.len() { - sl.va.wrapping_add(sl.vsize) as usize - } else { - data.len() - }; - let section_data: &[u8] = if sd_start <= sd_end { - &data[sd_start..sd_end] - } else { - &[] - }; - let copy_size = (sl.raw_size as usize).min(section_data.len()); - let rp = sl.raw_ptr as usize; - compact[rp..rp + copy_size].copy_from_slice(§ion_data[..copy_size]); - } - } - Some(compact) -} - -/// decrypt_data3: XOR+rotate cipher. Reads/writes dwords in `d` starting at -/// the address stored at `d[pos]`, for `d[pos+4]>>2` words. `shift` is the -/// right-rotate amount (19 or 21 depending on caller). -pub(crate) fn decrypt_data3(d: &mut [u8], pos: u32, mut key: u32, shift: u32) { - let base_addr = get_u32(d, pos); - let length = get_u32(d, pos.wrapping_add(4)); - let words = length >> 2; - for i in 0..words { - let off = base_addr.wrapping_add(i.wrapping_mul(4)); - let v = get_u32(d, off) ^ key; - key = key.wrapping_add(i); - let rotated = v.rotate_right(shift); - write_u32(d, off, rotated.wrapping_sub(i)); - } -} - -/// decrypt_data1 (called `decrypt_data` in the original): decode the 8-dword -/// info header from `file_data` at offset 4096 and write results into `info`. -pub(crate) fn decrypt_data1(file_data: &[u8], info: &mut [u32; 8]) { - info[0] = get_u32(file_data, 4096); - let mut k = get_u32(file_data, 4096); - for i in 0..7u32 { - let off = i.wrapping_mul(4).wrapping_add(4); - let cell = get_u32(file_data, 4096u32.wrapping_add(off)); - info[(i + 1) as usize] = k ^ cell; - k = i.wrapping_mul(i) ^ (k.wrapping_add(cell).wrapping_sub(i)); - } -} - -/// decrypt_data6: LFSR XOR decryption of a bytecode block at `pos` in `d`. -/// The block length is read from `d[pos + 95]`. -pub(crate) fn decrypt_data6(d: &mut [u8], pos: u32) { - let len = d[(pos + 95) as usize] as usize; - // The keystream is exactly `lfsr_keystream`'s — generate it once (len is a - // byte, so 256 always covers it) instead of keeping a second copy of the - // LFSR that a future poly fix would have to update separately. - let mut ks = [0u8; 256]; - lfsr_keystream(&mut ks); - let pos = pos as usize; - for i in 0..len { - d[pos + i] ^= ks[i]; - } -} - -/// decrypt_data7: nibble-swap + key-rolling byte cipher applied to a -/// null-terminated string in `d` starting at `pos`. -pub(crate) fn decrypt_data7(d: &mut [u8], pos: u32, mut key: u8) { - let mut i: u32 = 0; - loop { - let idx = (pos + i) as usize; - if d[idx] == 0 { - break; - } - let mut b = d[idx]; - b = b.rotate_right(4); - b = b.wrapping_sub(key); - if b == 0 { - b = 0u8.wrapping_sub(key); - } - d[idx] = b; - key = key.wrapping_add(67); - i += 1; - } -} - -// --------------------------------------------------------------------------- -// Higher-level composite: AES + decrypt3 + optional bytecode + decompress -// --------------------------------------------------------------------------- - -/// Decrypt and optionally decompress a stage payload descriptor. -/// `pos` points to a (src, src_len, dest, dest_len) quad of dwords in `d`. -/// - AES-decrypts `src..src+src_len` using key at `key3_offset` -/// - XOR+rotate-decrypts with `decrypt_data3(pos, key, 19)` -/// - Applies optional custom `ops` bytecode per-byte -/// - If `src_len != dest_len`, Huffman/LZ-decompresses `src..` → `dest..` -/// -/// Returns the decompression success status (always `true` when no -/// decompression was needed). The PE32 eighth-stage key search relies on this. -pub(crate) fn decrypt_and_decompress_data( - d: &mut [u8], - pos: u32, - key: u32, - key1_offset: u32, - key3_offset: u32, - ops: Option<&[Op]>, -) -> bool { - let src = get_u32(d, pos); - let src_len = get_u32(d, pos.wrapping_add(4)); - aes_decrypt(d, src, src_len, key3_offset); - decrypt_data3(d, pos, key, 19); - if let Some(ops) = ops - && src_len != 0 - { - OpsLut::new(ops).map_region(d, src as usize, src_len as usize); - } - let dest = get_u32(d, pos.wrapping_add(8)); - let dest_len = get_u32(d, pos.wrapping_add(12)); - if src_len != dest_len { - return decompress(d, src, dest, key1_offset, src_len, dest_len); - } - true -} - -// --------------------------------------------------------------------------- -// dd8 page-XOR shift selection. -// -// The packer scrambles ~1 byte per 16-byte block of .text via decrypt_data8, -// keyed by `page_idx << shift` (absolute page index = text_va >> 12). Observed -// shifts are 0 and 15. The shift is NOT stored in any header/config field: -// two otherwise-unrelated builds can carry byte-identical config-version stamps -// (0x40327253) yet require different shifts, so the only reliable discriminator -// is the .text content itself. -// -// Detection replays the three candidate states — no dd8 (already plaintext), -// shift 0, shift 15 — over a few sample pages (head/tail margin skipped: -// entry/exit regions have atypical padding density) and picks the state whose -// decoded pages look most like real x64 code. The primary signal is a -// *structural* fingerprint: the MSVC function-end padding pattern, a 0xC3 RET -// opcode followed by a run of >= 4 0xCC int3 bytes. dd8 XORs one pseudo-random -// byte per 16-byte block, so an already-plaintext page keeps its padding runs -// only under "no dd8", while a packer-encrypted page restores them only under -// the correct shift — a wrong candidate destroys every run it touches and -// essentially never manufactures a RET followed by a long int3 run by chance. -// This separates the states far more cleanly than a bare 0xCC count, which a -// wrong candidate inflates for free (~255 coincidences per page at p=1/256). -// -// When no candidate produces any RET-anchored padding (sampled pages with -// dense code and no padded epilogues), the fingerprint is silent, so the -// decision falls back to the older mutated-position 0xCC count. Both signals -// use the same decision rule: a candidate must beat the no-dd8 baseline by a -// clear 2x margin AND an absolute floor, otherwise dd8 is skipped — a wrongly -// applied dd8 scrambles ~1 byte per 16 with no error surfaced downstream. -// -// This replaces an earlier entry-stub oracle that matched the 14 fixed CRT-stub -// bytes at the AEP. That oracle false-positived on a newer EXE-64 build: dd8 -// corrupted only the call rel32 (bytes 5-8, the wildcard region), so the stub -// matched under BOTH shifts and the selector defaulted to 0 when the truth was -// 15. A whole-page padding statistic samples hundreds of positions per page -// and is not fooled by a stub whose fixed bytes happen to survive. -// --------------------------------------------------------------------------- - -/// Minimum 0xCC run length after a RET for the run to count as MSVC -/// function-end padding. -const MIN_CC_RUN: u32 = 4; - -/// Total length of MSVC function-end padding runs in a page: each 0xC3 byte -/// followed by >= [`MIN_CC_RUN`] 0xCC bytes contributes the run length. -fn ret_int3_score(page: &[u8]) -> u32 { - let mut total = 0u32; - let mut i = 0; - while i < page.len() { - if page[i] == 0xC3 { - let mut j = i + 1; - while j < page.len() && page[j] == 0xCC { - j += 1; - } - let run = (j - i - 1) as u32; - if run >= MIN_CC_RUN { - total += run; - } - i = j; - } else { - i += 1; - } - } - total -} - -/// Replay the dd8 page-XOR in place on one sample page. -fn dd8_apply(buf: &mut [u8; 0x1000], abs_page: u32, shift: u32) { - let mut key = abs_page << shift; - for bi in 0..256u32 { - let mixed = key.rotate_right(15).wrapping_add(bi); - key = mixed.wrapping_add(bi); - // The packer's dd8 loop does not XOR block i=0 (see decrypt_data8). - if bi == 0 { - continue; - } - let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize; - buf[tidx] ^= key as u8; - } -} - -/// Sum the RET+int3 fingerprint over the sample pages for one candidate -/// (`None` = the no-dd8 baseline, page as-is). -fn fingerprint_score( - data: &[u8], - text_off: usize, - abs_base: u32, - sample_pages: &[u32], - shift: Option, -) -> u32 { - let mut total = 0u32; - for &sp in sample_pages { - let pg_off = text_off + (sp as usize) * 0x1000; - if pg_off + 0x1000 > data.len() { - continue; - } - let mut page = [0u8; 0x1000]; - page.copy_from_slice(&data[pg_off..pg_off + 0x1000]); - if let Some(sh) = shift { - dd8_apply(&mut page, abs_base.wrapping_add(sp), sh); - } - total += ret_int3_score(&page); - } - total -} - -pub(crate) fn select_dd8_shift(data: &[u8], text_va: u32, text_size: u32, _info3: u32) -> u32 { - let num_pages_total = text_size >> 12; - // Fewer than two pages: nothing meaningful to sample; preserve the - // historical behavior (shift 0 — the dd8 loop is empty or single-page). - if num_pages_total < 2 { - return 0; - } - let text_off = text_va as usize; - - // Sample up to 4 pages, skipping a head/tail margin. Small .text: sample - // every page. - let mut sample_pages: Vec = Vec::new(); - if num_pages_total <= 4 { - sample_pages.extend(0..num_pages_total); - } else { - let margin = (num_pages_total / 8).max(1); - let lo = margin; - let hi = num_pages_total - margin; - if hi <= lo { - sample_pages.extend(0..num_pages_total); - } else { - let step = ((hi - lo) / 4).max(1); - let mut i = 0; - while i < 4 { - let p = lo + i * step; - if p < num_pages_total { - sample_pages.push(p); - } - i += 1; - } - } - } - if sample_pages.is_empty() { - return 0; - } - - let abs_base = text_va >> 12; - // Require a clear 2x margin over the already-plaintext baseline AND an - // absolute floor. The 2x test alone trips on noise when the counts are - // tiny: an external-companion DLL whose .text is already plaintext scores - // s15=4 vs none=1 — a spurious 4x — and gets dd8 wrongly applied, - // corrupting ~1 byte per 16. The floor rejects that noise while sitting - // far below every genuinely-encrypted build's score. - const MIN_DD8_HITS: u32 = 8; - let margin_pick = |none: u32, s0: u32, s15: u32| -> u32 { - let mut best_score = none; - let mut best_shift = 99u32; // 99 == skip dd8 - for (shift, hits) in [(0u32, s0), (15u32, s15)] { - if hits > best_score { - best_score = hits; - best_shift = shift; - } - } - if best_shift != 99 && (best_score < none * 2 || best_score < MIN_DD8_HITS) { - best_shift = 99; - } - best_shift - }; - - // Primary: RET+int3 padding fingerprint. The fingerprint is diluted across - // the whole page (dd8 touches only 255 of 4096 bytes, so even an encrypted - // page keeps most of its padding runs), so instead of the fallback's 2x - // margin the gate is a *positive delta* over the no-dd8 baseline: on an - // already-plaintext .text each wrong shift destroys runs (scores below the - // baseline), while the correct shift on an encrypted page restores them - // (scores above it). The floor on the delta rejects noise-level gains. - let r_none = fingerprint_score(data, text_off, abs_base, &sample_pages, None); - let r0 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(0)); - let r15 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(15)); - // Fallback: mutated-position 0xCC count, for pages whose code has no - // RET-anchored padding at all (the fingerprint is silent there). - let (none_hits, s0, s15); - let best_shift = if r_none != 0 || r0 != 0 || r15 != 0 { - none_hits = 0; - s0 = 0; - s15 = 0; - let mut best_score = r_none; - let mut shift = 99u32; - for (s, score) in [(0u32, r0), (15u32, r15)] { - if score > best_score { - best_score = score; - shift = s; - } - } - if shift != 99 && best_score.saturating_sub(r_none) < MIN_DD8_HITS { - shift = 99; - } - shift - } else { - none_hits = score_dd8_baseline(data, text_off, &sample_pages); - s0 = score_dd8_shift(data, text_off, text_va, &sample_pages, 0); - s15 = score_dd8_shift(data, text_off, text_va, &sample_pages, 15); - margin_pick(none_hits, s0, s15) - }; - if std::env::var("SEL_DIAG").is_ok() { - eprintln!( - "SEL dd8 best_shift={} fp=({},{},{}) cc=({},{},{}) samples={:?}", - best_shift, r_none, r0, r15, none_hits, s0, s15, sample_pages - ); - } - best_shift -} - -// Baseline: count int3 pads already present at the first byte of each 16-byte -// block, i.e. the positions dd8 would target if its in-block offset were 0. -fn score_dd8_baseline(data: &[u8], text_off: usize, sample_pages: &[u32]) -> u32 { - let mut hits = 0u32; - for &sp in sample_pages { - let pg_off = text_off + (sp as usize) * 0x1000; - if pg_off + 0x1000 > data.len() { - continue; - } - for bi in 1..256usize { - if data[pg_off + bi * 16] == 0xCC { - hits += 1; - } - } - } - hits -} - -// Replay decrypt_data8 on each sample page under `shift` and count how many of -// the 255 mutated positions decode to 0xCC. -fn score_dd8_shift( - data: &[u8], - text_off: usize, - text_va: u32, - sample_pages: &[u32], - shift: u32, -) -> u32 { - let abs_base = text_va >> 12; - let mut hits = 0u32; - for &sp in sample_pages { - let pg_off = text_off + (sp as usize) * 0x1000; - if pg_off + 0x1000 > data.len() { - continue; - } - let abs_page = abs_base.wrapping_add(sp); - let mut key = abs_page << shift; - for bi in 0..256u32 { - let mixed = key.rotate_right(15).wrapping_add(bi); - key = mixed.wrapping_add(bi); - if bi == 0 { - continue; - } - let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize; - if tidx < 0x1000 { - let mutated = data[pg_off + tidx] ^ (key as u8); - if mutated == 0xCC { - hits += 1; - } - } - } - } - hits -} - -// --------------------------------------------------------------------------- -// Tests -// --------------------------------------------------------------------------- - -#[cfg(test)] -mod tests { - use super::*; - - #[test] - fn aes_ks_variant_matches_single_buffer() { - // Random-ish key schedule at ko and data block; both variants must - // produce identical output. - let ko: usize = 0x40; - let mut d = vec![0u8; 0x400]; - let mut x: u32 = 0x12345678; - for b in d.iter_mut() { - x = x.wrapping_mul(1664525).wrapping_add(1013904223); - *b = (x >> 24) as u8; - } - d[ko + 2] = 10; // round count = 10 - d[ko + 3] = 0; - let snap = aes_schedule_snapshot(&d, ko as u32).expect("snapshot"); - - let mut a = d.clone(); - aes_decrypt(&mut a, 0x100, 0x80, ko as u32); - let mut b = d.clone(); - aes_decrypt_ks(&snap, &mut b, 0x100, 0x80); - if a != b { - let idx = (0..a.len()).find(|&i| a[i] != b[i]).unwrap(); - panic!( - "first diff at {idx:#x}: a={:02x} b={:02x}\n a[..]: {:02x?}\n b[..]: {:02x?}", - a[idx], - b[idx], - &a[idx..idx + 16], - &b[idx..idx + 16] - ); - } - } - - #[test] - fn dtbl_variant_matches_single_buffer() { - // Real table + real compressed block lifted from an actual unpack is - // covered by the golden suite; here we just check a trivial stream: - // build a table where every byte is a literal (mode 0, 8 bits), then - // a source stream of N bytes should expand to N identical bytes. - let ko: usize = 0x100; - let mut d = vec![0u8; 0x1000]; - for e in 0..256usize { - let off = ko + e * 3; - let sym = 0x8000u16 | (e as u16 & 0xFF); // terminal, mode 0, payload=e - d[off] = (sym & 0xFF) as u8; - d[off + 1] = (sym >> 8) as u8; - d[off + 2] = 8; // 8 bits per symbol - } - // Source: 16 bytes 0x00..0x0F at src. - let src = 0x600u32; - for i in 0..16u32 { - d[(src + i) as usize] = i as u8; - } - let snap = huffman_table_snapshot(&d, ko as u32).expect("table snapshot"); - - let mut a = vec![0u8; 0x1000]; - a[..d.len()].copy_from_slice(&d); - assert!(decompress(&mut a, src, 0x800, ko as u32, 16, 16)); - let mut b = d.clone(); - assert!(decompress_tbl(&snap, &mut b, src, 0x800, 16, 16)); - assert_eq!(&a[0x800..0x810], &b[0x800..0x810]); - assert_eq!(&b[0x800..0x810], &(0u8..16).collect::>()[..]); - } - - /// Task 4.1 regression: build a synthetic buffer whose valid bytecode block - /// sits PAST `len` but within `len*2`. Assert that the smaller window misses - /// it and the doubled window finds it. - #[test] - fn bytecode_locate_double_window_retry() { - // We place the block at offset (base + len + 16) which is inside - // the len*2 window but outside the len window. - let base: u32 = 0; - let len: u32 = 256; - // Block sits at base + len + 16 = 272, aligned to 16. - let block_pos: usize = (base + len + 16) as usize; // 272 - - // The buffer must be large enough for the block (block_pos + 96 bytes). - let buf_len = block_pos + 256; - let mut buf = vec![0u8; buf_len]; - - // Build a valid plaintext op stream: - // [4, 0, 4, 0, 4, 0, 4, 0, 195] (4 ADD-AL ops then RET) - // Padded to 10 bytes total; count >= 8. - let count: usize = 10; - let mut plain = [0u8; 256]; - plain[0] = 4; - plain[1] = 0; - plain[2] = 4; - plain[3] = 0; - plain[4] = 4; - plain[5] = 0; - plain[6] = 4; - plain[7] = 0; - plain[8] = 195; // ret - - // Compute the LFSR keystream and XOR the first `count` bytes to get the - // encrypted representation that the scanner would decrypt back. - let mut ks = [0u8; 256]; - lfsr_keystream(&mut ks); - for i in 0..count { - buf[block_pos + i] = plain[i] ^ ks[i]; - } - // Raw count byte at block_pos+95 (outside the XOR range since count=10 < 95). - buf[block_pos + 95] = count as u8; - - // Verify our construction: find_bytecode_offset with len should NOT find it. - assert_eq!( - find_bytecode_offset(&buf, base, len), - None, - "smaller window should not find the block" - ); - - // The doubled window should find it at block_pos. - assert_eq!( - find_bytecode_offset(&buf, base, len.saturating_mul(2)), - Some(block_pos as u32), - "doubled window should locate the block" - ); - } - - /// Review regression: a run-fill token with a unit width other than 1/2/4 - /// comes from a corrupt stream and must report failure — previously it - /// wrote nothing yet still counted the bytes as written, leaving stale - /// holes that later stages treated as plaintext. - #[test] - fn decompress_rejects_unknown_run_fill_width() { - // Huffman table at key_offset 0, entry 0: terminal symbol with - // mode 0x200 (run-fill), payload 3 (invalid width), code length 8. - let mut d = vec![0u8; 0x100]; - let sym: u16 = 0x8000 | 0x203; - d[0..2].copy_from_slice(&sym.to_le_bytes()); - d[2] = 8; - // All-zero source -> symbol index 0 -> the invalid run-fill. - assert!(!decompress(&mut d, 0x40, 0x80, 0, 4, 3)); - } - - /// Control for the above: a width-1 run-fill is legal and succeeds. - #[test] - fn decompress_accepts_width1_run_fill() { - let mut d = vec![0u8; 0x100]; - d[0x7F] = 0x5A; // unit to replicate - let sym: u16 = 0x8000 | 0x201; - d[0..2].copy_from_slice(&sym.to_le_bytes()); - d[2] = 8; - assert!(decompress(&mut d, 0x40, 0x80, 0, 4, 3)); - assert_eq!(&d[0x80..0x83], &[0x5A, 0x5A, 0x5A]); - } - - /// Seed the first `count` dd8-targeted positions of each sampled page with - /// the byte that decodes to `0xCC` under the `page+1` formula — i.e. an - /// encrypted `.text` whose plaintext is int3 padding. Positions whose key - /// byte would make the *ciphertext* itself `0xCC` are skipped so the - /// fixture contains no `0xCC` at all and every post-dd8 `0xCC` is a genuine - /// gain over a zero baseline. - fn seed_dd8_int3(data: &mut [u8], text_off: u32, pages: &[u32], count: u32) { - for &sp in pages { - let pg_off = (text_off + sp * 0x1000) as usize; - let mut k = sp.wrapping_add(1); - k = k.rotate_right(15); - let mut planted = 0u32; - for bi in 1..256u32 { - let ri = k.rotate_right(15).wrapping_add(bi); - k = ri.wrapping_add(bi); - if planted >= count { - continue; - } - let ct = 0xCCu8 ^ (k as u8); - if ct == 0xCC { - continue; - } - let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize; - data[pg_off + tidx] = ct; - planted += 1; - } - } - } - - /// Review regression: a near-plaintext `.text` must NOT be dd8-decrypted. - /// dd8 XORs 255 positions per page with pseudo-random bytes, so it - /// manufactures a few `0xCC` for free — under the old bare - /// `best > baseline` test any positive gain was enough to "apply" dd8 and - /// scramble ~1 byte per 16 of a native DLL's already-plaintext code, - /// silently (nothing downstream, including the integrity check, notices). - /// Here the gain is real but small; the floor must still reject it. - #[test] - fn pe32_dd8_skips_text_whose_gain_is_only_noise_sized() { - let text_off: u32 = 0x1000; - let text_size: u32 = 8 * 0x1000; - let mut data = vec![0u8; (text_off + text_size) as usize]; - seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 5); - assert!( - !data.contains(&0xCC), - "fixture must have a zero 0xCC baseline" - ); - assert_eq!( - select_dd8_formula_pe32(&data, text_off, text_size), - None, - "a gain this small is indistinguishable from dd8's own noise" - ); - } - - /// Control for the above: a `.text` whose dd8 pass restores a large amount - /// of int3 padding clears the floor and is decrypted. Same fixture shape, - /// only the amount of restored padding differs. - #[test] - fn pe32_dd8_applies_when_padding_is_restored() { - let text_off: u32 = 0x1000; - let text_size: u32 = 8 * 0x1000; - let mut data = vec![0u8; (text_off + text_size) as usize]; - seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 255); - assert_eq!( - select_dd8_formula_pe32(&data, text_off, text_size), - Some(false), - "encrypted .text must be decrypted with the page+1 formula" - ); - } - - /// Review regression: a zero last-section VA (corrupt section table) must - /// bail instead of building .kmiat at RVA 0 — the old code zeroed - /// `[0, 0x7000)`, wiping the DOS/PE headers, and returned the broken image - /// as a success. A near-2 GiB VA must likewise refuse to grow the image - /// past [`super::MAX_IMAGE_SIZE`]. - #[test] - fn kmiat_bogus_section_va_bails_without_wiping_headers() { - for last_sec_va in [0u32, 0x5000_0000] { - let pe: u32 = 0x80; - let mut data = vec![0xAAu8; 0x8000]; - // COFF header: 1 section, optional header size 0xE0 (PE32). - write_u16(&mut data, pe + 6, 1); - write_u16(&mut data, pe + 20, 0xE0); - // Import directory at pe+0x80: one descriptor + null terminator. - write_u32(&mut data, pe + 0x80, 0x1100); - write_u32(&mut data, pe + 0x84, 0x28); - write_u32(&mut data, 0x1100, 0x1200); // OFT rva - write_u32(&mut data, 0x1100 + 12, 0x1300); // name rva - write_u32(&mut data, 0x1100 + 16, 0x1400); // IAT rva - for b in &mut data[0x1100 + 20..0x1100 + 40] { - *b = 0; // null terminator descriptor - } - data[0x1300..0x1300 + 13].copy_from_slice(b"KERNEL32.dll\0"); - write_u32(&mut data, 0x1200, 0x1500); // thunk -> hint/name - write_u32(&mut data, 0x1204, 0); // thunk terminator - data[0x1500..0x1502].copy_from_slice(&0u16.to_le_bytes()); - data[0x1502..0x1502 + 12].copy_from_slice(b"ExitProcess\0"); - // Section table at pe+24+0xE0 = 0x178; VA field at +12. - write_u32(&mut data, 0x178 + 12, last_sec_va); - - let head_before: Vec = data[..0x400].to_vec(); - let len_before = data.len(); - move_pe32_imports_to_kmiat(&mut data, pe); - assert_eq!( - data.len(), - len_before, - "VA 0x{last_sec_va:08X}: image must not grow" - ); - assert_eq!( - &data[..0x400], - &head_before[..], - "VA 0x{last_sec_va:08X}: headers must be untouched" - ); - } - } -} diff --git a/tests/common/mod.rs b/tests/common/mod.rs deleted file mode 100644 index b598c3f..0000000 --- a/tests/common/mod.rs +++ /dev/null @@ -1,10 +0,0 @@ -//! Shared test fixtures. -#![allow(dead_code)] - -use std::path::PathBuf; - -/// Path to `senbei/samples` — the user-managed corpus dropped in by hand. -/// Git-ignored except its README; tests here run against whatever is present. -pub fn samples_dir() -> PathBuf { - PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("samples") -} diff --git a/web/Cargo.lock b/web/Cargo.lock index 5d0237d..d03a581 100644 --- a/web/Cargo.lock +++ b/web/Cargo.lock @@ -50,21 +50,21 @@ checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0" [[package]] name = "futures-core" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" +checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e" [[package]] name = "futures-task" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" +checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd" [[package]] name = "futures-util" -version = "0.3.33" +version = "0.3.34" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" +checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc" dependencies = [ "futures-core", "futures-task", @@ -87,9 +87,9 @@ dependencies = [ [[package]] name = "js-sys" -version = "0.3.103" +version = "0.3.104" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" +checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a" dependencies = [ "cfg-if", "futures-util", @@ -110,9 +110,9 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50" [[package]] name = "owo-colors" -version = "4.3.0" +version = "4.4.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "d211803b9b6b570f68772237e415a029d5a50c65d382910b879fb19d3271f94d" +checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8" [[package]] name = "pin-project-lite" @@ -122,9 +122,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd" [[package]] name = "portable-atomic" -version = "1.14.0" +version = "1.15.0" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3" +checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85" [[package]] name = "proc-macro2" @@ -160,24 +160,46 @@ dependencies = [ ] [[package]] -name = "senbei" -version = "1.0.0" +name = "senbei-crypto" +version = "1.1.0" +dependencies = [ + "thiserror", +] + +[[package]] +name = "senbei-io" +version = "1.1.0" dependencies = [ "anyhow", "indicatif", "libc", "owo-colors", - "thiserror", + "senbei-metadata", + "senbei-pe", "walkdir", "windows", ] +[[package]] +name = "senbei-metadata" +version = "1.1.0" + +[[package]] +name = "senbei-pe" +version = "1.1.0" +dependencies = [ + "senbei-crypto", + "thiserror", +] + [[package]] name = "senbei-web" -version = "1.0.0" +version = "1.1.0" dependencies = [ "console_error_panic_hook", - "senbei", + "senbei-io", + "senbei-metadata", + "senbei-pe", "wasm-bindgen", ] @@ -200,9 +222,9 @@ dependencies = [ [[package]] name = "syn" -version = "3.0.3" +version = "3.0.4" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" +checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f" dependencies = [ "proc-macro2", "quote", @@ -211,22 +233,22 @@ dependencies = [ [[package]] name = "thiserror" -version = "2.0.19" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" +checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f" dependencies = [ "thiserror-impl", ] [[package]] name = "thiserror-impl" -version = "2.0.19" +version = "2.0.20" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" +checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af" dependencies = [ "proc-macro2", "quote", - "syn 3.0.3", + "syn 3.0.4", ] [[package]] @@ -259,9 +281,9 @@ dependencies = [ [[package]] name = "wasm-bindgen" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" +checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70" dependencies = [ "cfg-if", "once_cell", @@ -272,9 +294,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" +checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1" dependencies = [ "quote", "wasm-bindgen-macro-support", @@ -282,9 +304,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-macro-support" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" +checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284" dependencies = [ "bumpalo", "proc-macro2", @@ -295,9 +317,9 @@ dependencies = [ [[package]] name = "wasm-bindgen-shared" -version = "0.2.126" +version = "0.2.127" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" +checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf" dependencies = [ "unicode-ident", ] diff --git a/web/Cargo.toml b/web/Cargo.toml index 298af82..9cd8ba8 100644 --- a/web/Cargo.toml +++ b/web/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "senbei-web" -version = "1.0.1" +version = "1.1.0" edition = "2024" description = "WebAssembly browser frontend for senbei" license = "AGPL-3.0-only" @@ -9,7 +9,9 @@ license = "AGPL-3.0-only" crate-type = ["cdylib"] [dependencies] -senbei = { path = ".." } +senbei-io = { path = "../senbei-io" } +senbei-metadata = { path = "../senbei-metadata" } +senbei-pe = { path = "../senbei-pe" } wasm-bindgen = "0.2" console_error_panic_hook = "0.1" diff --git a/web/src/lib.rs b/web/src/lib.rs index 976f4c1..e135477 100644 --- a/web/src/lib.rs +++ b/web/src/lib.rs @@ -101,12 +101,12 @@ impl MetadataResult { } } -fn kind_str(kind: senbei::unpacker::Kind) -> &'static str { +fn kind_str(kind: senbei_pe::Kind) -> &'static str { match kind { - senbei::unpacker::Kind::NativeExe => "native-exe", - senbei::unpacker::Kind::ManagedExe => "managed-exe", - senbei::unpacker::Kind::NativeDll => "native-dll", - senbei::unpacker::Kind::ManagedDll => "managed-dll", + senbei_pe::Kind::NativeExe => "native-exe", + senbei_pe::Kind::ManagedExe => "managed-exe", + senbei_pe::Kind::NativeDll => "native-dll", + senbei_pe::Kind::ManagedDll => "managed-dll", } } @@ -117,10 +117,10 @@ fn kind_str(kind: senbei::unpacker::Kind) -> &'static str { /// anything unrecognized. #[wasm_bindgen] pub fn detect(input: &[u8]) -> Option { - if senbei::metadata::is_metadata(input) { + if senbei_metadata::is_metadata(input) { return Some("metadata".to_string()); } - senbei::unpacker::detect(input).map(|d| kind_str(d.kind).to_string()) + senbei_pe::detect(input).map(|d| kind_str(d.kind).to_string()) } /// Unpack a protected module. @@ -134,7 +134,7 @@ pub fn unpack_file( input: &[u8], companion: Option>, ) -> Result { - let r = senbei::job::unpack_bytes(input, companion.as_deref()) + let r = senbei_io::job::unpack_bytes(input, companion.as_deref()) .map_err(|e| JsError::new(&e.to_string()))?; Ok(UnpackResult { kind: kind_str(r.kind).to_string(), @@ -153,7 +153,7 @@ pub fn unpack_file( #[wasm_bindgen] pub fn deobfuscate_metadata(data: &[u8]) -> Result { let (bytes, report) = - senbei::metadata::deobfuscate(data).map_err(|e| JsError::new(&e.to_string()))?; + senbei_metadata::deobfuscate(data).map_err(|e| JsError::new(&e.to_string()))?; Ok(MetadataResult { bytes, version: report.version, @@ -164,14 +164,14 @@ pub fn deobfuscate_metadata(data: &[u8]) -> Result { } /// Unpack a protected module, forcing the EXE pipeline (no DLL-pipeline -/// probe). See [`senbei::job::unpack_bytes_force_exe`] for why the web app +/// probe). See [`senbei_io::job::unpack_bytes_force_exe`] for why the web app /// needs this recovery path. #[wasm_bindgen] pub fn unpack_file_force_exe( input: &[u8], companion: Option>, ) -> Result { - let r = senbei::job::unpack_bytes_force_exe(input, companion.as_deref()) + let r = senbei_io::job::unpack_bytes_force_exe(input, companion.as_deref()) .map_err(|e| JsError::new(&e.to_string()))?; Ok(UnpackResult { kind: kind_str(r.kind).to_string(),