mirror of
https://github.com/Momoko-Ayase/Senbei.git
synced 2026-09-19 03:57:59 -04:00
Merge bfloat16-senbei workspace restructure, bump to 1.1.0
Adopts the fork's workspace split (senbei-cli / senbei-crypto / senbei-io / senbei-metadata / senbei-pe), its structured error taxonomy, entry-transform and layout validation, PE32 dd8 key-formula selection with a skip floor, the CRT entry-stub dd8 oracle, and the extensionless-file scan skip. Kept from senbei on top of the restructure: - ManagedExe detection/routing and the CLR (COR20 + BSJB) metadata restore in the EXE pipeline. - The RET+int3 padding fingerprint as the primary dd8 padding signal, ahead of the mutated-position 0xCC fallback. - docs/, .github/, samples/, tests/ (moved to senbei-cli/tests), and the web/ wasm frontend (rewired to the split crates), all of which the fork had dropped. - The fork's README compatibility matrix is not taken: it names real games, which the public-repo hygiene rules forbid. - The wasm32 localtime fallback in logfile and unpack_bytes_force_exe (the web app's trap-recovery entry point), both lost in the restructure. Golden corpus: 35/35 byte-identical. clippy -D warnings clean; wasm32 check clean for the full workspace.
This commit is contained in:
@@ -19,7 +19,7 @@ jobs:
|
||||
runs-on: windows-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- run: cargo clippy --all-targets -- -D warnings
|
||||
- run: cargo clippy --workspace --all-targets -- -D warnings
|
||||
|
||||
test:
|
||||
# The test suite exercises Windows path semantics, so it runs on Windows.
|
||||
@@ -28,7 +28,7 @@ jobs:
|
||||
runs-on: windows-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- run: cargo test --release
|
||||
- run: cargo test --release --workspace
|
||||
|
||||
check-portable:
|
||||
# Build-only portability gate: non-Windows host and the wasm target the
|
||||
@@ -39,8 +39,8 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- run: cargo clippy --all-targets -- -D warnings
|
||||
- run: cargo check --target wasm32-unknown-unknown
|
||||
- run: cargo clippy --workspace --all-targets -- -D warnings
|
||||
- run: cargo check --workspace --target wasm32-unknown-unknown
|
||||
|
||||
cli:
|
||||
strategy:
|
||||
|
||||
@@ -4,17 +4,19 @@ Guidance for AI coding agents (and human contributors) working in this repo.
|
||||
|
||||
## Project
|
||||
|
||||
Senbei is a static unpacker for Crackproof-protected PE files: a pure,
|
||||
panic-free, no-I/O unpacker core (`src/unpacker/`) plus a thin CLI shell
|
||||
(`src/`), an il2cpp metadata de-obfuscator (`src/metadata.rs`), and a
|
||||
WebAssembly browser frontend (`web/`). Read `docs/design.md` first.
|
||||
Senbei is a static unpacker for Crackproof-protected PE files: a Cargo
|
||||
workspace with a pure, panic-free, no-I/O unpacker core (`senbei-pe/`, built
|
||||
on `senbei-crypto/`), an il2cpp metadata de-obfuscator (`senbei-metadata/`),
|
||||
filesystem/CLI orchestration (`senbei-io/`), the `senbei` binary
|
||||
(`senbei-cli/`), and a WebAssembly browser frontend (`web/`, outside the
|
||||
workspace). Read `docs/design.md` first.
|
||||
|
||||
## Commands
|
||||
|
||||
```cmd
|
||||
cargo build --release :: CLI
|
||||
cargo test --release :: full suite (golden corpus: samples/, git-ignored)
|
||||
cargo clippy --all-targets -- -D warnings
|
||||
cargo build --release :: CLI (default member: senbei-cli)
|
||||
cargo test --release --workspace :: full suite (golden corpus: samples/, git-ignored)
|
||||
cargo clippy --workspace --all-targets -- -D warnings
|
||||
cargo fmt --all
|
||||
cd web && wasm-pack build --target web --release :: browser build
|
||||
```
|
||||
|
||||
Generated
+59
-30
@@ -62,21 +62,21 @@ checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223"
|
||||
|
||||
[[package]]
|
||||
name = "futures-core"
|
||||
version = "0.3.33"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7"
|
||||
checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e"
|
||||
|
||||
[[package]]
|
||||
name = "futures-task"
|
||||
version = "0.3.33"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109"
|
||||
checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd"
|
||||
|
||||
[[package]]
|
||||
name = "futures-util"
|
||||
version = "0.3.33"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa"
|
||||
checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-task",
|
||||
@@ -110,9 +110,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "js-sys"
|
||||
version = "0.3.103"
|
||||
version = "0.3.104"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102"
|
||||
checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"futures-util",
|
||||
@@ -139,9 +139,9 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
||||
|
||||
[[package]]
|
||||
name = "owo-colors"
|
||||
version = "4.3.0"
|
||||
version = "4.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d211803b9b6b570f68772237e415a029d5a50c65d382910b879fb19d3271f94d"
|
||||
checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8"
|
||||
|
||||
[[package]]
|
||||
name = "pin-project-lite"
|
||||
@@ -151,9 +151,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd"
|
||||
|
||||
[[package]]
|
||||
name = "portable-atomic"
|
||||
version = "1.14.0"
|
||||
version = "1.15.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3"
|
||||
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
@@ -208,19 +208,48 @@ dependencies = [
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei"
|
||||
version = "1.0.1"
|
||||
name = "senbei-cli"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"senbei-io",
|
||||
"senbei-metadata",
|
||||
"tempfile",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei-crypto"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"thiserror",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei-io"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"indicatif",
|
||||
"libc",
|
||||
"owo-colors",
|
||||
"senbei-metadata",
|
||||
"senbei-pe",
|
||||
"tempfile",
|
||||
"thiserror",
|
||||
"walkdir",
|
||||
"windows",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei-metadata"
|
||||
version = "1.1.0"
|
||||
|
||||
[[package]]
|
||||
name = "senbei-pe"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"senbei-crypto",
|
||||
"thiserror",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "slab"
|
||||
version = "0.4.12"
|
||||
@@ -240,9 +269,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "3.0.3"
|
||||
version = "3.0.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3"
|
||||
checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -264,22 +293,22 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "thiserror"
|
||||
version = "2.0.19"
|
||||
version = "2.0.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9"
|
||||
checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f"
|
||||
dependencies = [
|
||||
"thiserror-impl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "thiserror-impl"
|
||||
version = "2.0.19"
|
||||
version = "2.0.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd"
|
||||
checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 3.0.3",
|
||||
"syn 3.0.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -312,9 +341,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4"
|
||||
checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"once_cell",
|
||||
@@ -325,9 +354,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1"
|
||||
checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1"
|
||||
dependencies = [
|
||||
"quote",
|
||||
"wasm-bindgen-macro-support",
|
||||
@@ -335,9 +364,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro-support"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e"
|
||||
checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284"
|
||||
dependencies = [
|
||||
"bumpalo",
|
||||
"proc-macro2",
|
||||
@@ -348,9 +377,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-shared"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24"
|
||||
checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf"
|
||||
dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
+25
-25
@@ -1,39 +1,39 @@
|
||||
[package]
|
||||
name = "senbei"
|
||||
version = "1.0.1"
|
||||
[workspace]
|
||||
members = [
|
||||
"senbei-cli",
|
||||
"senbei-crypto",
|
||||
"senbei-io",
|
||||
"senbei-metadata",
|
||||
"senbei-pe",
|
||||
]
|
||||
default-members = ["senbei-cli"]
|
||||
# The wasm frontend is its own crate (own Cargo.lock, cdylib) and stays outside
|
||||
# the workspace.
|
||||
exclude = ["web"]
|
||||
resolver = "2"
|
||||
|
||||
[workspace.package]
|
||||
version = "1.1.0"
|
||||
edition = "2024"
|
||||
description = "Static unpacker for Crackproof-protected PE files"
|
||||
license = "AGPL-3.0-only"
|
||||
keywords = ["unpacker", "reverse-engineering", "pe", "security-research"]
|
||||
categories = ["command-line-utilities"]
|
||||
|
||||
[lib]
|
||||
name = "senbei"
|
||||
path = "src/lib.rs"
|
||||
|
||||
[[bin]]
|
||||
name = "senbei"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
[workspace.dependencies]
|
||||
anyhow = "1"
|
||||
indicatif = "0.18"
|
||||
libc = "0.2"
|
||||
owo-colors = "4"
|
||||
tempfile = "3"
|
||||
thiserror = "2"
|
||||
walkdir = "2"
|
||||
indicatif = "0.18"
|
||||
owo-colors = "4"
|
||||
|
||||
[target.'cfg(windows)'.dependencies]
|
||||
windows = { version = "0.62", features = [
|
||||
"Win32_Foundation",
|
||||
"Win32_System_Console",
|
||||
"Win32_System_SystemInformation",
|
||||
] }
|
||||
|
||||
[target.'cfg(all(not(windows), not(target_arch = "wasm32")))'.dependencies]
|
||||
libc = "0.2"
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
senbei-crypto = { path = "senbei-crypto" }
|
||||
senbei-io = { path = "senbei-io" }
|
||||
senbei-metadata = { path = "senbei-metadata" }
|
||||
senbei-pe = { path = "senbei-pe" }
|
||||
|
||||
[profile.release]
|
||||
opt-level = 3
|
||||
|
||||
+35
-19
@@ -7,41 +7,57 @@ no driver or proxy DLL is involved.
|
||||
|
||||
## Crate layout
|
||||
|
||||
The crate is split into a pure core and a thin CLI shell:
|
||||
Senbei is a Cargo workspace split into a pure core and thin shells around it:
|
||||
|
||||
- **`src/unpacker/`** — the core. Pure functions over byte slices: no file
|
||||
I/O, no environment access (beyond a few debugging overrides, see
|
||||
- **`senbei-pe/`** — the core. Pure functions over byte slices: no file I/O,
|
||||
no environment access (beyond a few debugging overrides, see
|
||||
[development.md](development.md)), panic-free at the public boundary (all
|
||||
internal panics are trapped and converted to `UnpackError::Corrupt`). This
|
||||
is what the WebAssembly build embeds.
|
||||
- **`src/` (top level)** — the CLI shell: argument parsing, recursive folder
|
||||
scanning, per-run log file, progress bar, Explorer-friendly exit pause, and
|
||||
the single-file/folder orchestration in `job.rs`.
|
||||
- **`src/metadata.rs`** — il2cpp `global-metadata.dat` method-token
|
||||
- **`senbei-crypto/`** — cryptographic, checksum, compression, and bytecode
|
||||
primitives the core is built from. Same purity rules as `senbei-pe`.
|
||||
- **`senbei-metadata/`** — il2cpp `global-metadata.dat` method-token
|
||||
de-obfuscation (format version 31; other versions are left untouched).
|
||||
- **`senbei-io/`** — filesystem and orchestration: recursive folder scanning,
|
||||
per-run log file, progress bar, Explorer-friendly exit pause, and the
|
||||
single-file/folder orchestration in `job.rs` (incl. the wasm-safe in-memory
|
||||
byte API used by the web frontend).
|
||||
- **`senbei-cli/`** — the `senbei` binary: argument parsing + dispatch. The
|
||||
integration test suite (incl. the golden corpus test) lives in
|
||||
`senbei-cli/tests/`.
|
||||
|
||||
```
|
||||
src/
|
||||
├── main.rs argument parsing + dispatch
|
||||
├── lib.rs module roots
|
||||
senbei-cli/
|
||||
└── src/main.rs argument parsing + dispatch
|
||||
senbei-io/src/
|
||||
├── job.rs single-file + folder orchestration, out-naming,
|
||||
│ companion splice, stub overlay/TLS restore,
|
||||
│ pipeline routing (incl. the wasm-safe byte API)
|
||||
├── scan.rs recursive Crackproof + metadata discovery
|
||||
├── metadata.rs il2cpp global-metadata.dat de-obfuscation
|
||||
├── logfile.rs per-run timestamped log
|
||||
├── ui.rs progress bar + status lines
|
||||
├── pause.rs Explorer-friendly exit pause
|
||||
└── unpacker/ pure, panic-free, no-I/O core
|
||||
├── mod.rs detection + unpack_auto dispatch
|
||||
├── exe.rs EXE pipeline (PE32+ and PE32)
|
||||
├── dll.rs native + managed DLL pipeline
|
||||
├── integrity.rs static post-unpack sanity check
|
||||
├── primitives.rs decrypt_data* steps, key/shift selection
|
||||
└── pause.rs Explorer-friendly exit pause
|
||||
senbei-metadata/src/
|
||||
└── metadata.rs il2cpp global-metadata.dat de-obfuscation
|
||||
senbei-crypto/src/
|
||||
├── primitives.rs decrypt_data* steps, key derivation
|
||||
├── bytecode.rs bytecode VM
|
||||
├── parallel.rs deterministic block-parallel fan-out
|
||||
├── tables.rs constant tables
|
||||
└── crc32.rs checksum
|
||||
senbei-pe/src/engine/ pure, panic-free, no-I/O core
|
||||
├── mod.rs detection + unpack_auto dispatch
|
||||
├── error.rs structured error taxonomy
|
||||
├── integrity.rs static post-unpack sanity check
|
||||
├── parallel.rs deterministic block-parallel fan-out
|
||||
├── layout/ layout discovery + validation
|
||||
│ ├── dd8.rs .text dd8 key-formula + shift selection
|
||||
│ ├── discovery.rs layout candidate discovery (trial-and-validate)
|
||||
│ └── image.rs PE image reconstruction helpers
|
||||
├── exe/
|
||||
│ ├── pipeline.rs EXE pipeline (PE32+ and PE32 orchestration)
|
||||
│ └── pipeline/pe32.rs PE32-specific EXE restore
|
||||
└── dll/
|
||||
└── pipeline.rs native + managed DLL pipeline
|
||||
```
|
||||
|
||||
## Detection and routing
|
||||
|
||||
+3
-3
@@ -60,9 +60,9 @@ since binaries are not committed).
|
||||
|
||||
## Conventions
|
||||
|
||||
- The `src/unpacker/` core is pure: no file I/O, no panics across the public
|
||||
boundary, no `unsafe`. Keep it that way — it is what the WebAssembly build
|
||||
embeds.
|
||||
- The `senbei-pe/` core (and its `senbei-crypto/` base) is pure: no file I/O,
|
||||
no panics across the public boundary, no `unsafe`. Keep it that way — it is
|
||||
what the WebAssembly build embeds.
|
||||
- Layout heuristics must **trial-and-validate**: never pick a candidate offset
|
||||
on shape alone and trust it; validate by decryption/checksum and fall
|
||||
through to the next candidate on failure. A silent wrong offset produces a
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
[package]
|
||||
name = "senbei-cli"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
description = "Command-line entry point for Senbei"
|
||||
license.workspace = true
|
||||
keywords = ["unpacker", "reverse-engineering", "pe", "security-research"]
|
||||
categories = ["command-line-utilities"]
|
||||
|
||||
[[bin]]
|
||||
name = "senbei"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
senbei-io.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
senbei-io.workspace = true
|
||||
senbei-metadata.workspace = true
|
||||
tempfile.workspace = true
|
||||
@@ -1,4 +1,4 @@
|
||||
use senbei::{job, pause};
|
||||
use senbei_io::{job, pause, scan};
|
||||
use std::path::Path;
|
||||
|
||||
fn main() -> std::process::ExitCode {
|
||||
@@ -27,9 +27,6 @@ fn main() -> std::process::ExitCode {
|
||||
"--no-log" => no_log = true,
|
||||
"--scan-all" => scan_all = true,
|
||||
"--out" => match args.next() {
|
||||
// Reject a missing value (and a following flag swallowed as the
|
||||
// value): previously `--out` at end of argv silently fell back
|
||||
// to the default output directory.
|
||||
Some(v) if !v.starts_with('-') => out = Some(v),
|
||||
_ => {
|
||||
eprintln!("error: --out requires a directory argument");
|
||||
@@ -42,7 +39,6 @@ fn main() -> std::process::ExitCode {
|
||||
return std::process::ExitCode::from(2);
|
||||
}
|
||||
other => {
|
||||
// Previously the last positional silently won.
|
||||
if let Some(prev) = &path {
|
||||
eprintln!("error: multiple input paths given ('{prev}' and '{other}')");
|
||||
return std::process::ExitCode::from(2);
|
||||
@@ -63,33 +59,36 @@ fn main() -> std::process::ExitCode {
|
||||
}
|
||||
let p = Path::new(&p);
|
||||
let out_path = out.as_deref().map(Path::new);
|
||||
let r = if p.is_dir() {
|
||||
let result = if p.is_dir() {
|
||||
job::run_folder_opts(
|
||||
p,
|
||||
out_path,
|
||||
quiet,
|
||||
verbose,
|
||||
no_log,
|
||||
scan_all || senbei::scan::scan_all_env(),
|
||||
scan_all || scan::scan_all_env(),
|
||||
)
|
||||
} else {
|
||||
job::run_file_v(p, out_path, quiet, verbose, no_log)
|
||||
};
|
||||
match r {
|
||||
Ok(s) => {
|
||||
match result {
|
||||
Ok(summary) => {
|
||||
if quiet < 2 {
|
||||
println!(
|
||||
"{} unpacked · {} skipped · {} errors · {} suspect · {} metadata",
|
||||
s.unpacked, s.skipped, s.errors, s.suspect, s.metadata
|
||||
summary.unpacked,
|
||||
summary.skipped,
|
||||
summary.errors,
|
||||
summary.suspect,
|
||||
summary.metadata
|
||||
);
|
||||
println!("done in {} ms", s.duration_ms);
|
||||
println!("done in {} ms", summary.duration_ms);
|
||||
}
|
||||
if s.errors > 0 { 1 } else { 0 }
|
||||
if summary.errors > 0 { 1 } else { 0 }
|
||||
}
|
||||
Err(e) => {
|
||||
// Fatal: out-dir/log create, etc.
|
||||
Err(error) => {
|
||||
if quiet < 2 {
|
||||
eprintln!("error: {e:#}");
|
||||
eprintln!("error: {error:#}");
|
||||
}
|
||||
1
|
||||
}
|
||||
@@ -107,7 +106,7 @@ fn print_help() {
|
||||
);
|
||||
println!(
|
||||
" --scan-all probe every file in a folder, including ones the scan\n\
|
||||
\x20 pre-filter skips (under 4128 bytes, or a bulk-asset\n\
|
||||
\x20 extension like .ab/.xml/.acb). Much slower on game trees."
|
||||
\x20 pre-filter skips (under 4128 bytes, extensionless,\n\
|
||||
\x20 or a bulk-asset extension). Much slower on large trees."
|
||||
);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
//! Shared test fixtures.
|
||||
#![allow(dead_code)]
|
||||
|
||||
use std::path::PathBuf;
|
||||
|
||||
/// Path to the workspace-root `samples/` — the user-managed corpus dropped in
|
||||
/// by hand. Git-ignored except its README; tests here run against whatever is
|
||||
/// present. `CARGO_MANIFEST_DIR` is `senbei-cli/`, so go one level up.
|
||||
pub fn samples_dir() -> PathBuf {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../samples")
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
use senbei::job::{default_out_root_for_file, out_name};
|
||||
use senbei_io::job::{default_out_root_for_file, out_name};
|
||||
use std::path::Path;
|
||||
|
||||
#[test]
|
||||
@@ -1,4 +1,4 @@
|
||||
use senbei::logfile::{Log, local_stamp_compact, local_stamp_display};
|
||||
use senbei_io::logfile::{Log, local_stamp_compact, local_stamp_display};
|
||||
|
||||
#[test]
|
||||
fn local_stamp_compact_matches_shape() {
|
||||
@@ -1,4 +1,4 @@
|
||||
use senbei::job;
|
||||
use senbei_io::job;
|
||||
use std::path::Path;
|
||||
|
||||
fn list_logs(dir: &Path) -> Vec<std::path::PathBuf> {
|
||||
@@ -8,7 +8,7 @@
|
||||
//! - golden present, bytes differ -> FAIL (the test fails)
|
||||
//! - no golden -> WARNING (printed; needs a manual check)
|
||||
//!
|
||||
//! Inputs go through [`senbei::job::unpack_bytes`], the same routing the CLI
|
||||
//! Inputs go through [`senbei_io::job::unpack_bytes`], the same routing the CLI
|
||||
//! uses, **not** `unpack_auto` directly. That matters: `unpack_auto` alone
|
||||
//! cannot reach the external-companion layout, whose stub is meaningless
|
||||
//! without its `<name>._` payload — a corpus wired to `unpack_auto` silently
|
||||
@@ -17,7 +17,7 @@
|
||||
//! samples folder is picked up automatically, exactly as it is on disk.
|
||||
//!
|
||||
//! An input whose bytes carry the il2cpp metadata magic is routed through
|
||||
//! [`senbei::metadata::deobfuscate`] instead, giving the method-token remap
|
||||
//! [`senbei_metadata::deobfuscate`] instead, giving the method-token remap
|
||||
//! real-world coverage (its unit tests only build synthetic layouts).
|
||||
//!
|
||||
//! The folder is git-ignored (see `senbei/samples/README.md`), so the set of
|
||||
@@ -116,10 +116,10 @@ fn samples_unpack_against_goldens() {
|
||||
}
|
||||
};
|
||||
|
||||
let got = if senbei::metadata::is_metadata(&bytes) {
|
||||
let got = if senbei_metadata::is_metadata(&bytes) {
|
||||
// il2cpp metadata: method-token de-obfuscation, no PE pipeline and
|
||||
// no integrity check (the output is not a PE image).
|
||||
match senbei::metadata::deobfuscate(&bytes) {
|
||||
match senbei_metadata::deobfuscate(&bytes) {
|
||||
Ok((out, _report)) => out,
|
||||
Err(e) => {
|
||||
failures.push(format!("{name}: de-obfuscation failed: {e}"));
|
||||
@@ -140,7 +140,7 @@ fn samples_unpack_against_goldens() {
|
||||
},
|
||||
None => None,
|
||||
};
|
||||
let image = match senbei::job::unpack_bytes(&bytes, companion.as_deref()) {
|
||||
let image = match senbei_io::job::unpack_bytes(&bytes, companion.as_deref()) {
|
||||
Ok(img) => img,
|
||||
Err(e) => {
|
||||
failures.push(format!("{name}: unpack failed: {e:?}"));
|
||||
@@ -0,0 +1,9 @@
|
||||
[package]
|
||||
name = "senbei-crypto"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
license.workspace = true
|
||||
description = "Cryptographic and compression primitives for Senbei"
|
||||
|
||||
[dependencies]
|
||||
thiserror.workspace = true
|
||||
@@ -61,8 +61,8 @@ impl OpsLut {
|
||||
pub fn generate(data: &[u8], offset: u32) -> Option<Vec<Op>> {
|
||||
// Bounds-checked cursor: a corrupt `data_offset` (bad decrypt_data6 / the
|
||||
// alignment fallback) must yield `None`, not an out-of-bounds panic — the
|
||||
// panic path would surface as a misleading `UnpackError::Corrupt` instead
|
||||
// of the precise `BytecodeGenFailed`, and any future caller without a
|
||||
// panic path would surface as a misleading `UnpackError::InternalPanic` instead
|
||||
// of the precise `BytecodeGenerationFailed`, and any future caller without a
|
||||
// `catch_unwind` wrapper would abort outright.
|
||||
let mut pos = offset as usize;
|
||||
let mut next = move || {
|
||||
@@ -0,0 +1,77 @@
|
||||
//! Cryptographic, checksum, compression, and bytecode primitives.
|
||||
|
||||
pub mod bytecode;
|
||||
pub mod crc32;
|
||||
pub mod primitives;
|
||||
mod tables;
|
||||
|
||||
/// Maximum buffer size accepted by allocation-sensitive transforms.
|
||||
pub const MAX_IMAGE_SIZE: u64 = 1 << 30;
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum BufferOperation {
|
||||
Read,
|
||||
CopySource,
|
||||
CopyDestination,
|
||||
ZeroFill,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for BufferOperation {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str(match self {
|
||||
Self::Read => "read",
|
||||
Self::CopySource => "copy source",
|
||||
Self::CopyDestination => "copy destination",
|
||||
Self::ZeroFill => "zero-fill",
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
|
||||
pub enum Error {
|
||||
#[error(
|
||||
"{operation} range out of bounds (offset {offset}, size {size}, buffer length {buffer_len})"
|
||||
)]
|
||||
BufferRangeOutOfBounds {
|
||||
operation: BufferOperation,
|
||||
offset: usize,
|
||||
size: usize,
|
||||
buffer_len: usize,
|
||||
},
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum DecompressionFailure {
|
||||
#[error("compressed source size {size} exceeds limit {max}")]
|
||||
SourceTooLarge { size: u32, max: u64 },
|
||||
#[error("Huffman code length {bits} is invalid")]
|
||||
InvalidCodeLength { bits: u8 },
|
||||
#[error("Huffman tree traversal exceeded 64 levels")]
|
||||
HuffmanTraversalLimit,
|
||||
#[error("pending length accumulator overflowed at {pending}")]
|
||||
PendingLengthOverflow { pending: u32 },
|
||||
#[error("output step {step} at byte {written} exceeds expected size {expected}")]
|
||||
OutputOverflow {
|
||||
written: u32,
|
||||
step: u32,
|
||||
expected: u32,
|
||||
},
|
||||
#[error("run-fill width {width} reads before output offset 0x{destination:08X}")]
|
||||
RunFillBeforeOutput { width: u32, destination: u32 },
|
||||
#[error("run-fill width {width} is unsupported")]
|
||||
InvalidRunFillWidth { width: u32 },
|
||||
#[error("back-reference distance {distance} exceeds {written} written bytes")]
|
||||
InvalidBackReference { distance: u32, written: u32 },
|
||||
#[error("Huffman symbol consumed no input and produced no output")]
|
||||
NoProgress,
|
||||
#[error(
|
||||
"output size mismatch (wrote {written}/{expected} bytes after consuming {consumed}/{source_size})"
|
||||
)]
|
||||
OutputSizeMismatch {
|
||||
written: u32,
|
||||
expected: u32,
|
||||
consumed: u32,
|
||||
source_size: u32,
|
||||
},
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -146,11 +146,11 @@ mod tests {
|
||||
#[test]
|
||||
fn generated_tables_match_committed_bytes() {
|
||||
assert_eq!(COLUMMIX1.len(), 1024);
|
||||
assert_eq!(super::super::crc32::compute(&COLUMMIX1), 0x7e8d_5d5f);
|
||||
assert_eq!(super::super::crc32::compute(&COLUMMIX2), 0xfcc4_acfc);
|
||||
assert_eq!(super::super::crc32::compute(&COLUMMIX3), 0x637a_f0cd);
|
||||
assert_eq!(super::super::crc32::compute(&COLUMMIX4), 0x1e7b_c381);
|
||||
assert_eq!(super::super::crc32::compute(&SBOX), 0x10fd_6dc1);
|
||||
assert_eq!(crate::crc32::compute(&COLUMMIX1), 0x7e8d_5d5f);
|
||||
assert_eq!(crate::crc32::compute(&COLUMMIX2), 0xfcc4_acfc);
|
||||
assert_eq!(crate::crc32::compute(&COLUMMIX3), 0x637a_f0cd);
|
||||
assert_eq!(crate::crc32::compute(&COLUMMIX4), 0x1e7b_c381);
|
||||
assert_eq!(crate::crc32::compute(&SBOX), 0x10fd_6dc1);
|
||||
// Spot-check the first dword of each (matches the original first row).
|
||||
assert_eq!(&COLUMMIX1[..4], &[0x50, 0xa7, 0xf4, 0x51]);
|
||||
assert_eq!(&COLUMMIX2[..4], &[0xa7, 0xf4, 0x51, 0x50]);
|
||||
@@ -0,0 +1,23 @@
|
||||
[package]
|
||||
name = "senbei-io"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
license.workspace = true
|
||||
description = "Filesystem, scanning, logging, and CLI orchestration for Senbei"
|
||||
|
||||
[dependencies]
|
||||
anyhow.workspace = true
|
||||
indicatif.workspace = true
|
||||
owo-colors.workspace = true
|
||||
senbei-metadata.workspace = true
|
||||
senbei-pe.workspace = true
|
||||
walkdir.workspace = true
|
||||
|
||||
[target.'cfg(windows)'.dependencies]
|
||||
windows.workspace = true
|
||||
|
||||
[target.'cfg(all(not(windows), not(target_arch = "wasm32")))'.dependencies]
|
||||
libc.workspace = true
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile.workspace = true
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::unpacker;
|
||||
use senbei_pe as unpacker;
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
/// Crackproof header key table lives at this fixed file offset. For the
|
||||
@@ -387,9 +387,9 @@ pub fn run_folder_v(
|
||||
///
|
||||
/// When `scan_all` is true every regular file under `root` is opened and
|
||||
/// content-probed, instead of skipping ones the free directory metadata already
|
||||
/// rules out (too small to hold a Crackproof key table, or a bulk-asset
|
||||
/// extension). See [`crate::scan::find_targets_opts`] — exhaustive scanning is
|
||||
/// dramatically slower on asset-heavy game trees and finds the same targets.
|
||||
/// rules out (extensionless, too small to hold a Crackproof key table, or a
|
||||
/// bulk-asset extension). See [`crate::scan::find_targets_opts`] — exhaustive
|
||||
/// scanning is dramatically slower on asset-heavy trees.
|
||||
pub fn run_folder_opts(
|
||||
root: &Path,
|
||||
out_dir: Option<&Path>,
|
||||
@@ -518,7 +518,7 @@ pub fn run_folder_opts(
|
||||
// il2cpp metadata pass. Crackproof's `-GMD` option obfuscates the method
|
||||
// tokens in `global-metadata.dat`; de-obfuscate any we find so the unpacked
|
||||
// il2cpp game assembly resolves methods instead of indexing its per-module
|
||||
// tables out of bounds (see [`crate::metadata`]). This is additive to the
|
||||
// tables out of bounds (see [`senbei_metadata`]). This is additive to the
|
||||
// Crackproof module unpack above — the metadata blob is not itself a
|
||||
// Crackproof file.
|
||||
for meta in metas {
|
||||
@@ -660,7 +660,7 @@ pub fn run_file_v(
|
||||
let mut buf = [0u8; 4];
|
||||
std::fs::File::open(input)
|
||||
.and_then(|mut f| f.read_exact(&mut buf))
|
||||
.map(|_| crate::metadata::is_metadata(&buf))
|
||||
.map(|_| senbei_metadata::is_metadata(&buf))
|
||||
.unwrap_or(false)
|
||||
};
|
||||
|
||||
@@ -801,13 +801,13 @@ fn panic_payload(panic: &(dyn std::any::Any + Send)) -> String {
|
||||
}
|
||||
}
|
||||
|
||||
/// If `e`'s chain contains [`crate::metadata::Error::UnsupportedVersion`],
|
||||
/// If `e`'s chain contains [`senbei_metadata::Error::UnsupportedVersion`],
|
||||
/// return the version. Used to apply the folder-mode "leave untouched, don't
|
||||
/// fail the run" policy to metadata versions this build can't de-obfuscate.
|
||||
fn unsupported_version(e: &anyhow::Error) -> Option<u32> {
|
||||
for cause in e.chain() {
|
||||
if let Some(crate::metadata::Error::UnsupportedVersion(v)) =
|
||||
cause.downcast_ref::<crate::metadata::Error>()
|
||||
if let Some(senbei_metadata::Error::UnsupportedVersion(v)) =
|
||||
cause.downcast_ref::<senbei_metadata::Error>()
|
||||
{
|
||||
return Some(*v);
|
||||
}
|
||||
@@ -831,19 +831,14 @@ fn write_atomic(dest: &Path, bytes: &[u8]) -> std::io::Result<()> {
|
||||
r
|
||||
}
|
||||
|
||||
/// Detect `bytes` and run the right pipeline. The EXE pipeline is invoked
|
||||
/// directly (no DLL-pipeline probe) when the input was spliced from an
|
||||
/// external companion (`spliced`) or when the caller forces it (`force_exe`
|
||||
/// — the web app's recovery path after a DLL-probe trap; see
|
||||
/// [`unpack_bytes_force_exe`]).
|
||||
/// Detect `bytes` and run the right pipeline. Spliced external companions use
|
||||
/// the EXE pipeline directly because that layout is definitionally EXE-style.
|
||||
///
|
||||
/// Routing spliced inputs straight to the EXE pipeline is safe: the
|
||||
/// companion layout is definitionally the EXE-style shell (the runtime
|
||||
/// loader maps the companion and runs the standard shell unpack), so the DLL
|
||||
/// pipeline probe can never be right for it — and probing is not a no-op on
|
||||
/// targets without unwinding (wasm), where the probe's caught panic becomes
|
||||
/// a fatal trap. Output bytes are identical to the dll-first + exe-fallback
|
||||
/// route for every input that route handles.
|
||||
/// pipeline probe can never be right for it. Output bytes are identical to the
|
||||
/// DLL-first + EXE-fallback route for every input that route handles.
|
||||
fn unpack_spliced_or_auto(
|
||||
bytes: &[u8],
|
||||
spliced: bool,
|
||||
@@ -880,10 +875,9 @@ pub struct UnpackedImage {
|
||||
/// Unpack in-memory `input` bytes, optionally paired with an external
|
||||
/// companion payload `companion` (the `<input>._` file's contents).
|
||||
///
|
||||
/// This is the I/O-free counterpart of [`unpack_one_v`], used by the
|
||||
/// WebAssembly build: splice (when the companion's first 32 bytes match the
|
||||
/// stub header), unpack, overlay the export table and TLS directory from the
|
||||
/// stub, then run the static integrity check.
|
||||
/// This is the in-memory counterpart of [`unpack_one_v`]: splice a matching
|
||||
/// companion, unpack, overlay the export table and TLS directory from the stub,
|
||||
/// then run the static integrity check.
|
||||
pub fn unpack_bytes(
|
||||
input: &[u8],
|
||||
companion: Option<&[u8]>,
|
||||
@@ -960,7 +954,7 @@ pub fn unpack_one_v(
|
||||
/// into a sparse, original-metadata-style value; il2cpp expects the contiguous
|
||||
/// per-module index it indexes its codegen tables with, so a statically-unpacked
|
||||
/// il2cpp game assembly reads garbage and crashes during init. This rewrites
|
||||
/// the tokens back to their canonical form (see [`crate::metadata::deobfuscate`]).
|
||||
/// the tokens back to their canonical form (see [`senbei_metadata::deobfuscate`]).
|
||||
///
|
||||
/// The output is written only when something actually changed
|
||||
/// (`report.remapped > 0`); an already-clean metadata is left untouched and no
|
||||
@@ -970,11 +964,11 @@ pub fn deobfuscate_metadata_to(
|
||||
input: &Path,
|
||||
dest: &Path,
|
||||
verbose: bool,
|
||||
) -> anyhow::Result<crate::metadata::Report> {
|
||||
) -> anyhow::Result<senbei_metadata::Report> {
|
||||
let data = std::fs::read(input)?;
|
||||
// Preserve the metadata::Error in the chain (rather than stringifying it)
|
||||
// so the folder driver can apply its unsupported-version policy.
|
||||
let (out, report) = crate::metadata::deobfuscate(&data)
|
||||
let (out, report) = senbei_metadata::deobfuscate(&data)
|
||||
.map_err(|e| anyhow::Error::new(e).context(format!("{input:?}")))?;
|
||||
if report.remapped > 0 {
|
||||
if let Some(parent) = dest.parent() {
|
||||
@@ -1,7 +1,7 @@
|
||||
//! Filesystem and command-line orchestration.
|
||||
|
||||
pub mod job;
|
||||
pub mod logfile;
|
||||
pub mod metadata;
|
||||
pub mod pause;
|
||||
pub mod scan;
|
||||
pub mod ui;
|
||||
pub mod unpacker;
|
||||
@@ -1,4 +1,4 @@
|
||||
use crate::unpacker::detect;
|
||||
use senbei_pe::detect;
|
||||
use std::io::Read;
|
||||
use std::path::{Path, PathBuf};
|
||||
use walkdir::WalkDir;
|
||||
@@ -16,24 +16,23 @@ const DETECT_PREFIX: u64 = 8 * 1024;
|
||||
/// Smallest file that can possibly be a target, so anything shorter is skipped
|
||||
/// without ever being opened.
|
||||
///
|
||||
/// A Crackproof module needs ≥ 4128 bytes for [`crate::unpacker::detect`]'s key
|
||||
/// A Crackproof module needs ≥ 4128 bytes for [`senbei_pe::detect`]'s key
|
||||
/// table (it reads the dword at 4124), so the bound is exact for the unpack
|
||||
/// path. An il2cpp `global-metadata.dat` only needs 4 bytes to match its magic,
|
||||
/// but its header alone runs to offset 0xB0 and the images/types/methods tables
|
||||
/// it indexes make every real one megabytes long — a sub-4 KiB "metadata" could
|
||||
/// only ever fail [`crate::metadata::deobfuscate`] with `Malformed`, so nothing
|
||||
/// only ever fail [`senbei_metadata::deobfuscate`] with `Malformed`, so nothing
|
||||
/// processable is lost.
|
||||
const MIN_SIZE: u64 = 4128;
|
||||
|
||||
/// File extensions that are bulk data by construction and can never be a PE
|
||||
/// image or an il2cpp metadata blob.
|
||||
///
|
||||
/// This is deliberately a **deny**-list, not an allow-list: the default is to
|
||||
/// probe, so anything unrecognised is still opened. Targets are recognised by
|
||||
/// content, not extension, and can carry arbitrary names — there is no closed
|
||||
/// set of target extensions an allow-list of `exe`/`dll` could enumerate.
|
||||
/// Only extensions that are bulk asset or text formats by construction appear
|
||||
/// here.
|
||||
/// This is deliberately a **deny**-list, not an executable allow-list: unknown
|
||||
/// extensions are still probed. Extensionless files are handled separately by
|
||||
/// [`denied_name`] because asset stores commonly contain tens of thousands of
|
||||
/// extensionless chunks; exhaustive probing remains available through
|
||||
/// `--scan-all`.
|
||||
///
|
||||
/// Set `SENBEI_SCAN_ALL=1` (or pass `--scan-all`) to probe every file regardless.
|
||||
const DENY_EXT: &[&str] = &[
|
||||
@@ -90,11 +89,12 @@ const DENY_EXT: &[&str] = &[
|
||||
"sr",
|
||||
];
|
||||
|
||||
/// Whether `path`'s extension is on [`DENY_EXT`]. Extensionless files are never
|
||||
/// denied (they could be anything).
|
||||
fn denied_ext(path: &Path) -> bool {
|
||||
/// Whether `path` can be skipped from its name alone. Extensionless files and
|
||||
/// files whose extension is on [`DENY_EXT`] are not opened during a default
|
||||
/// scan. `--scan-all` remains available when exhaustive probing is required.
|
||||
fn denied_name(path: &Path) -> bool {
|
||||
let Some(ext) = path.extension() else {
|
||||
return false;
|
||||
return true;
|
||||
};
|
||||
let Some(ext) = ext.to_str() else {
|
||||
return false;
|
||||
@@ -206,12 +206,17 @@ pub fn find_targets_opts(root: &Path, scan_all: bool) -> (Vec<PathBuf>, Vec<Path
|
||||
continue;
|
||||
}
|
||||
if !scan_all {
|
||||
// Name checks come first so extensionless asset chunks never
|
||||
// trigger even an explicit metadata query.
|
||||
if denied_name(entry.path()) {
|
||||
continue;
|
||||
}
|
||||
// Skip on directory metadata alone — never open these.
|
||||
let too_small = entry
|
||||
.metadata()
|
||||
.map(|m| m.len() < MIN_SIZE)
|
||||
.unwrap_or(false);
|
||||
if too_small || denied_ext(entry.path()) {
|
||||
if too_small {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
@@ -223,7 +228,7 @@ pub fn find_targets_opts(root: &Path, scan_all: bool) -> (Vec<PathBuf>, Vec<Path
|
||||
// `Some(Class::None)` means "probed, matched neither detector".
|
||||
let n = paths.len();
|
||||
let mut class: Vec<Option<Class>> = vec![Some(Class::None); n];
|
||||
let workers = crate::unpacker::parallel::thread_cap().clamp(1, n.max(1));
|
||||
let workers = senbei_pe::thread_cap().clamp(1, n.max(1));
|
||||
if workers <= 1 {
|
||||
for (p, c) in paths.iter().zip(class.iter_mut()) {
|
||||
*c = classify(p);
|
||||
@@ -304,7 +309,7 @@ fn classify(path: &Path) -> Option<Class> {
|
||||
let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
if detect(&head).is_some() {
|
||||
Class::Crackproof
|
||||
} else if crate::metadata::is_metadata(&head) {
|
||||
} else if senbei_metadata::is_metadata(&head) {
|
||||
Class::Metadata
|
||||
} else {
|
||||
Class::None
|
||||
@@ -329,29 +334,49 @@ mod tests {
|
||||
#[test]
|
||||
fn denies_bulk_asset_extensions_case_insensitively() {
|
||||
for p in ["a.ab", "a.XML", "a.Acb", "a.ma2", "a.manifest", "a.PNG"] {
|
||||
assert!(denied_ext(Path::new(p)), "{p} should be denied");
|
||||
assert!(denied_name(Path::new(p)), "{p} should be denied");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn denies_extensionless_files() {
|
||||
for p in ["asset", "level0", "0123456789abcdef"] {
|
||||
assert!(denied_name(Path::new(p)), "{p} should be denied");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn never_denies_what_a_target_can_be_named() {
|
||||
// Targets are recognised by content, not name — a protected module
|
||||
// can carry any extension, or none — so names like these must always
|
||||
// be probed. An allow-list would have skipped them.
|
||||
// Unknown extensions must still be probed. This keeps the filter a
|
||||
// narrow deny-list rather than an executable-extension allow-list.
|
||||
for p in [
|
||||
"app.exe.bak",
|
||||
"managed.dll.bak",
|
||||
"daemon.exe",
|
||||
"GameLib.dll",
|
||||
"global-metadata.dat",
|
||||
"noextension",
|
||||
"a.so",
|
||||
"a.bin",
|
||||
] {
|
||||
assert!(!denied_ext(Path::new(p)), "{p} must still be probed");
|
||||
assert!(!denied_name(Path::new(p)), "{p} must still be probed");
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn extensionless_targets_require_exhaustive_scan() {
|
||||
let td = tempfile::tempdir().unwrap();
|
||||
let root = td.path();
|
||||
let mut blob = vec![0u8; MIN_SIZE as usize + 1];
|
||||
blob[..4].copy_from_slice(&0xFAB1_1BAFu32.to_le_bytes());
|
||||
std::fs::write(root.join("metadata"), &blob).unwrap();
|
||||
|
||||
let (_, filtered, _) = find_targets_opts(root, false);
|
||||
assert!(filtered.is_empty());
|
||||
|
||||
let (_, exhaustive, _) = find_targets_opts(root, true);
|
||||
assert_eq!(exhaustive.len(), 1);
|
||||
}
|
||||
|
||||
/// A file below the Crackproof key-table bound is skipped without being
|
||||
/// opened, but a large non-asset file is still probed.
|
||||
#[test]
|
||||
@@ -1,6 +1,6 @@
|
||||
use crate::unpacker::{IntegrityReport, Kind};
|
||||
use indicatif::{ProgressBar, ProgressStyle};
|
||||
use owo_colors::OwoColorize;
|
||||
use senbei_pe::{IntegrityReport, Kind};
|
||||
use std::path::Path;
|
||||
|
||||
/// Create a progress bar for `n` items. Hidden when `quiet` is true.
|
||||
@@ -0,0 +1,6 @@
|
||||
[package]
|
||||
name = "senbei-metadata"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
license.workspace = true
|
||||
description = "Unity il2cpp metadata de-obfuscation for Senbei"
|
||||
@@ -0,0 +1,5 @@
|
||||
//! Unity il2cpp metadata de-obfuscation.
|
||||
|
||||
mod metadata;
|
||||
|
||||
pub use metadata::*;
|
||||
@@ -0,0 +1,10 @@
|
||||
[package]
|
||||
name = "senbei-pe"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
license.workspace = true
|
||||
description = "PE detection, unpacking, and validation for Senbei"
|
||||
|
||||
[dependencies]
|
||||
senbei-crypto.workspace = true
|
||||
thiserror.workspace = true
|
||||
@@ -0,0 +1,3 @@
|
||||
mod pipeline;
|
||||
|
||||
pub use pipeline::*;
|
||||
@@ -13,9 +13,12 @@
|
||||
//! CalculateChecksumWithSizeXor -> primitives::calculate_checksum
|
||||
//! CalculateCrc32 -> crc32::compute (via above)
|
||||
|
||||
use super::UnpackError;
|
||||
use super::bytecode::{Op, OpsLut, generate};
|
||||
use super::primitives::{self, *};
|
||||
use super::super::{
|
||||
BufferOperation, BytecodeStage, DecompressionStage, DescriptorTable, SectionPipeline,
|
||||
UnpackError,
|
||||
};
|
||||
use senbei_crypto::bytecode::{Op, OpsLut, generate};
|
||||
use senbei_crypto::primitives::{self, *};
|
||||
|
||||
/// Read a signed 32-bit little-endian value.
|
||||
fn get_i32(d: &[u8], offset: i32) -> i32 {
|
||||
@@ -68,6 +71,7 @@ fn decrypt_data4(
|
||||
key: i32,
|
||||
decomp_params: &[i32; 4],
|
||||
transform: Option<&[Op]>,
|
||||
stage: DecompressionStage,
|
||||
) -> Result<(), UnpackError> {
|
||||
let addr = get_i32(d, offset);
|
||||
let size = get_i32(d, offset + 4);
|
||||
@@ -84,19 +88,17 @@ fn decrypt_data4(
|
||||
OpsLut::new(ops).map_region(d, addr as usize, size as usize);
|
||||
}
|
||||
|
||||
if size != decompressed_size {
|
||||
// decompress reports corruption (after partial writes) via its bool;
|
||||
// surface it instead of shipping a garbage block.
|
||||
if !decompress(
|
||||
if size != decompressed_size
|
||||
&& let Err(reason) = primitives::decompress_detailed(
|
||||
d,
|
||||
addr as u32,
|
||||
compressed_addr as u32,
|
||||
decomp_params[1] as u32,
|
||||
size as u32,
|
||||
decompressed_size as u32,
|
||||
) {
|
||||
return Err(UnpackError::DecompressFailed);
|
||||
}
|
||||
)
|
||||
{
|
||||
return Err(UnpackError::StageDecompressionFailed { stage, reason });
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -218,7 +220,11 @@ fn decrypt_and_decompress_data(
|
||||
// Guard: need 16 bytes at section_data_offset in `d`
|
||||
let off = section_data_offset as usize;
|
||||
if off.saturating_add(16) > d.len() {
|
||||
return Err(UnpackError::OutOfBounds(off));
|
||||
return Err(UnpackError::DescriptorOutOfBounds {
|
||||
table: DescriptorTable::DllSectionBlocks,
|
||||
offset: off,
|
||||
image_len: d.len(),
|
||||
});
|
||||
}
|
||||
decrypt_data6_shift6(d, section_data_offset, 16);
|
||||
let dest_offset = get_i32(d, section_data_offset);
|
||||
@@ -246,10 +252,10 @@ fn decrypt_and_decompress_data(
|
||||
let lut = OpsLut::new(decrypt_func);
|
||||
let ko0 = decomp_params[0];
|
||||
let ko2 = decomp_params[2];
|
||||
let ks_snap =
|
||||
primitives::aes_schedule_snapshot(d, ko2 as u32).ok_or(UnpackError::Corrupt)?;
|
||||
let ks_snap = primitives::aes_schedule_snapshot(d, ko2 as u32)
|
||||
.ok_or(UnpackError::InvalidAesKeySchedule { offset: ko2 as u32 })?;
|
||||
let tab_snap = primitives::huffman_table_snapshot(d, ko0 as u32)
|
||||
.ok_or(UnpackError::DecompressFailed)?;
|
||||
.ok_or(UnpackError::InvalidHuffmanTable { offset: ko0 as u32 })?;
|
||||
let spans: Vec<(usize, usize)> = blocks
|
||||
.iter()
|
||||
.map(|b| {
|
||||
@@ -277,12 +283,15 @@ fn decrypt_and_decompress_data(
|
||||
b.size as u32,
|
||||
b.expected_crc as u32,
|
||||
) {
|
||||
return Err(UnpackError::DecompressFailed);
|
||||
return Err(UnpackError::SectionDecompressionFailed {
|
||||
pipeline: SectionPipeline::Dll,
|
||||
block: i,
|
||||
});
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
};
|
||||
super::parallel::parallel_for(d, &spans, 1, do_block)?;
|
||||
super::super::parallel::parallel_for(d, &spans, 1, do_block)?;
|
||||
}
|
||||
|
||||
// Zero-fill loop.
|
||||
@@ -292,7 +301,11 @@ fn decrypt_and_decompress_data(
|
||||
// decrypts 16 too, so guard 16 (an 8-byte guard would let
|
||||
// decrypt_data6_shift6 index past the end of a truncated descriptor).
|
||||
if off.saturating_add(16) > d.len() {
|
||||
return Err(UnpackError::OutOfBounds(off));
|
||||
return Err(UnpackError::DescriptorOutOfBounds {
|
||||
table: DescriptorTable::DllZeroFill,
|
||||
offset: off,
|
||||
image_len: d.len(),
|
||||
});
|
||||
}
|
||||
decrypt_data6_shift6(d, section_data_offset, 16);
|
||||
let zero_offset = get_i32(d, section_data_offset);
|
||||
@@ -305,7 +318,12 @@ fn decrypt_and_decompress_data(
|
||||
for i in 0..zero_size {
|
||||
let idx = (zero_offset + i) as usize;
|
||||
if idx >= d.len() {
|
||||
return Err(UnpackError::OutOfBounds(idx));
|
||||
return Err(UnpackError::BufferRangeOutOfBounds {
|
||||
operation: BufferOperation::ZeroFill,
|
||||
offset: idx,
|
||||
size: 1,
|
||||
buffer_len: d.len(),
|
||||
});
|
||||
}
|
||||
d[idx] = 0;
|
||||
}
|
||||
@@ -324,12 +342,16 @@ pub fn unpack_dll(input: &[u8]) -> Result<Vec<u8>, UnpackError> {
|
||||
pub fn unpack_dll_v(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError> {
|
||||
// Trap any out-of-bounds panic from a truncated/garbled file and report it
|
||||
// as a clean error so the public API stays panic-free.
|
||||
super::catch_unpack(move || unpack_dll_inner(input, verbose))
|
||||
super::super::catch_unpack(move || unpack_dll_inner(input, verbose))
|
||||
}
|
||||
|
||||
fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError> {
|
||||
if input.len() < 4096 {
|
||||
return Err(UnpackError::InputTooShort(input.len()));
|
||||
const HEADER_LEN: usize = 4128;
|
||||
if input.len() < HEADER_LEN {
|
||||
return Err(UnpackError::InputTooShort {
|
||||
actual: input.len(),
|
||||
required: HEADER_LEN,
|
||||
});
|
||||
}
|
||||
|
||||
// `file_data` and `original_file_data` both borrow the same protected input.
|
||||
@@ -347,15 +369,18 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
println!(" keys[6] anchor = 0x{:08X}", keys[6] as u32);
|
||||
}
|
||||
|
||||
if !super::is_supported_magic(keys[1] as u32) {
|
||||
return Err(UnpackError::DllUnpack(
|
||||
"Not a Crackproof protected file (KONN magic mismatch)".into(),
|
||||
));
|
||||
if !super::super::is_supported_magic(keys[1] as u32) {
|
||||
return Err(UnpackError::HeaderMagicMismatch {
|
||||
found: keys[1] as u32,
|
||||
});
|
||||
}
|
||||
|
||||
let pe_offset = get_i32(file_data, 60);
|
||||
if pe_offset < 0 || (pe_offset as usize).saturating_add(84) > file_data.len() {
|
||||
return Err(UnpackError::DllUnpack("implausible PE offset".into()));
|
||||
return Err(UnpackError::InvalidPeOffset {
|
||||
offset: i64::from(pe_offset),
|
||||
input_len: file_data.len(),
|
||||
});
|
||||
}
|
||||
// This pipeline is PE32+-only: its header fixups write the data
|
||||
// directories at PE32+ offsets (pe+144..180, pe+136 for the DD blob). On a
|
||||
@@ -363,14 +388,18 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
// structurally plausible but unloadable file. Reject early with a clear
|
||||
// error so `unpack_auto`'s EXE-pipeline fallback handles PE32 DLLs (that
|
||||
// path is PE32-aware — see run_pe32), instead of us mangling them here.
|
||||
if get_i32(file_data, pe_offset + 24) & 0xFFFF != 0x20B {
|
||||
return Err(UnpackError::DllUnpack(
|
||||
"not a PE32+ image (the DLL pipeline handles 64-bit only)".into(),
|
||||
));
|
||||
let optional_magic = get_u16(file_data, (pe_offset + 24) as u32);
|
||||
if optional_magic != 0x20B {
|
||||
return Err(UnpackError::UnsupportedDllPeMagic {
|
||||
found: optional_magic,
|
||||
});
|
||||
}
|
||||
let size_of_image = get_i32(file_data, pe_offset + 80);
|
||||
if size_of_image <= 0 || size_of_image as u64 > super::MAX_IMAGE_SIZE {
|
||||
return Err(UnpackError::DllUnpack("implausible SizeOfImage".into()));
|
||||
if size_of_image <= 0 || size_of_image as u64 > super::super::MAX_IMAGE_SIZE {
|
||||
return Err(UnpackError::InvalidImageSize {
|
||||
size: i64::from(size_of_image),
|
||||
max: super::super::MAX_IMAGE_SIZE,
|
||||
});
|
||||
}
|
||||
let mut out = vec![0u8; size_of_image as usize];
|
||||
let base_offset = keys[6] - keys[3] + 0x2000;
|
||||
@@ -438,6 +467,16 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
println!(" checksum1 = 0x{:08X}", checksum1 as u32);
|
||||
println!(" decrypted_addr1 = 0x{:08X}", decrypted_addr1 as u32);
|
||||
}
|
||||
let primary_end = decrypted_addr1.checked_add(3856);
|
||||
if decrypted_addr1 < keys[3]
|
||||
|| primary_end.is_none_or(|end| end < 0 || end as usize > out.len())
|
||||
{
|
||||
return Err(UnpackError::InvalidDllPrimaryDescriptor {
|
||||
address: decrypted_addr1 as u32,
|
||||
minimum: keys[3] as u32,
|
||||
image_len: out.len(),
|
||||
});
|
||||
}
|
||||
let import_offset = get_i32(&out, decrypted_addr1 + 3444);
|
||||
let decrypted_addr2_size = get_i32(&out, decrypted_addr1 + 3632);
|
||||
decrypt_data3(
|
||||
@@ -512,6 +551,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
table_val ^ checksum2 ^ (xor_accumulator as i32),
|
||||
&decomp_params,
|
||||
None,
|
||||
DecompressionStage::DllCodeBlock1,
|
||||
)?;
|
||||
|
||||
let addr3b = get_i32(&out, decrypted_addr1 + 3728);
|
||||
@@ -527,7 +567,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
let crc_val = {
|
||||
let a = crc_data_addr as usize;
|
||||
let n = crc_data_size as usize;
|
||||
super::crc32::compute(&out[a..a + n]) as i32
|
||||
senbei_crypto::crc32::compute(&out[a..a + n]) as i32
|
||||
};
|
||||
let crc_xored = crc_data_size ^ crc_val;
|
||||
let trailing_val = get_i32(&out, crc_data_addr + crc_data_size - 4);
|
||||
@@ -537,6 +577,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
crc_xored ^ (xor_accumulator as i32) ^ trailing_val,
|
||||
&decomp_params,
|
||||
None,
|
||||
DecompressionStage::DllCodeBlock2,
|
||||
)?;
|
||||
|
||||
let checksum3 = calculate_checksum(&out, (decrypted_addr1 + 3480) as u32) as i32;
|
||||
@@ -549,6 +590,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
(not_val ^ (xor_key as u32)) as i32,
|
||||
&decomp_params,
|
||||
None,
|
||||
DecompressionStage::DllCodeBlock3,
|
||||
)?;
|
||||
|
||||
let addr4 = get_i32(&out, addr4_offset);
|
||||
@@ -574,8 +616,9 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
lfsr_seed_val = lfsr_seed_val.wrapping_add(k);
|
||||
}
|
||||
|
||||
let decrypt_func = generate(&out, lfsr as u32)
|
||||
.ok_or_else(|| UnpackError::DllUnpack("Failed to build decryption expression".into()))?;
|
||||
let decrypt_func = generate(&out, lfsr as u32).ok_or(UnpackError::BytecodeGenerationFailed(
|
||||
BytecodeStage::DllPrimaryDecryptor,
|
||||
))?;
|
||||
|
||||
let addr5_offset = decrypted_addr1 + 3840;
|
||||
let addr5 = get_i32(&out, addr5_offset);
|
||||
@@ -585,6 +628,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
lfsr_seed_val ^ xor_key ^ checksum4,
|
||||
&decomp_params,
|
||||
Some(&decrypt_func),
|
||||
DecompressionStage::DllCodeBlock4,
|
||||
)?;
|
||||
if verbose {
|
||||
println!("[7/9] Decrypting code block 4 (addr5)...");
|
||||
@@ -603,9 +647,9 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
|
||||
let lfsr2 = metadata_offset + 88;
|
||||
decrypt_data6(&mut out, lfsr2 as u32);
|
||||
|
||||
let decrypt_func2 = generate(&out, lfsr2 as u32).ok_or_else(|| {
|
||||
UnpackError::DllUnpack("Failed to build second decryption expression".into())
|
||||
})?;
|
||||
let decrypt_func2 = generate(&out, lfsr2 as u32).ok_or(
|
||||
UnpackError::BytecodeGenerationFailed(BytecodeStage::DllSectionDecryptor),
|
||||
)?;
|
||||
|
||||
let section_image_base = 4095 - get_i32(original_file_data, 4224);
|
||||
let section_data_offset = get_i32(&out, addr5 + 11976);
|
||||
@@ -0,0 +1,253 @@
|
||||
pub use senbei_crypto::{BufferOperation, DecompressionFailure};
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum DecompressionStage {
|
||||
ExeStage3,
|
||||
ExeStage3Secondary,
|
||||
ExeStage4,
|
||||
ExeStage5,
|
||||
Pe32FourthStage,
|
||||
Pe32FifthStage,
|
||||
Pe32SeventhStage,
|
||||
DllCodeBlock1,
|
||||
DllCodeBlock2,
|
||||
DllCodeBlock3,
|
||||
DllCodeBlock4,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for DecompressionStage {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str(match self {
|
||||
Self::ExeStage3 => "EXE stage3",
|
||||
Self::ExeStage3Secondary => "EXE secondary stage3",
|
||||
Self::ExeStage4 => "EXE stage4",
|
||||
Self::ExeStage5 => "EXE stage5",
|
||||
Self::Pe32FourthStage => "PE32 fourth stage",
|
||||
Self::Pe32FifthStage => "PE32 fifth stage",
|
||||
Self::Pe32SeventhStage => "PE32 seventh stage",
|
||||
Self::DllCodeBlock1 => "DLL code block 1",
|
||||
Self::DllCodeBlock2 => "DLL code block 2",
|
||||
Self::DllCodeBlock3 => "DLL code block 3",
|
||||
Self::DllCodeBlock4 => "DLL code block 4",
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum BytecodeStage {
|
||||
ExeStage4,
|
||||
ExeStage5,
|
||||
Pe32CustomDecryptor,
|
||||
Pe32FileDecryptor,
|
||||
DllPrimaryDecryptor,
|
||||
DllSectionDecryptor,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for BytecodeStage {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str(match self {
|
||||
Self::ExeStage4 => "EXE stage4",
|
||||
Self::ExeStage5 => "EXE stage5",
|
||||
Self::Pe32CustomDecryptor => "PE32 custom decryptor",
|
||||
Self::Pe32FileDecryptor => "PE32 file decryptor",
|
||||
Self::DllPrimaryDecryptor => "DLL primary decryptor",
|
||||
Self::DllSectionDecryptor => "DLL section decryptor",
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum SectionPipeline {
|
||||
ExePe32Plus,
|
||||
ExePe32,
|
||||
Dll,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for SectionPipeline {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str(match self {
|
||||
Self::ExePe32Plus => "PE32+ EXE",
|
||||
Self::ExePe32 => "PE32 EXE",
|
||||
Self::Dll => "DLL",
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum DescriptorTable {
|
||||
DllSectionBlocks,
|
||||
DllZeroFill,
|
||||
}
|
||||
|
||||
impl std::fmt::Display for DescriptorTable {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str(match self {
|
||||
Self::DllSectionBlocks => "DLL section-block",
|
||||
Self::DllZeroFill => "DLL zero-fill",
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
|
||||
#[non_exhaustive]
|
||||
pub enum UnpackError {
|
||||
#[error("input too short (need at least {required} bytes, got {actual})")]
|
||||
InputTooShort { actual: usize, required: usize },
|
||||
|
||||
#[error("decrypted header magic mismatch (got 0x{found:08X})")]
|
||||
HeaderMagicMismatch { found: u32 },
|
||||
|
||||
#[error("anchor field not found — corrupt data or wrong offset")]
|
||||
AnchorNotFound,
|
||||
|
||||
#[error("stage1 descriptor not found near anchor 0x{anchor:08X}")]
|
||||
Stage1DescriptorNotFound { anchor: u32 },
|
||||
|
||||
#[error("stage2 field not found — corrupt data or wrong offset")]
|
||||
Stage2NotFound,
|
||||
|
||||
#[error("chk_src_start not found — corrupt data or wrong offset")]
|
||||
ChkSrcStartNotFound,
|
||||
|
||||
#[error("table_start not found — corrupt data or wrong offset")]
|
||||
TableStartNotFound,
|
||||
|
||||
#[error("{0} bytecode generation failed — corrupt data or wrong offset")]
|
||||
BytecodeGenerationFailed(BytecodeStage),
|
||||
|
||||
#[error("stage5 marker not found — this build's layout is not supported by this unpacker")]
|
||||
Stage5MarkerNotFound,
|
||||
|
||||
#[error("not a Crackproof-protected file")]
|
||||
NotCrackproof,
|
||||
|
||||
#[error("invalid PE header offset {offset} for {input_len}-byte input")]
|
||||
InvalidPeOffset { offset: i64, input_len: usize },
|
||||
|
||||
#[error("DLL pipeline requires PE32+ optional-header magic, got 0x{found:04X}")]
|
||||
UnsupportedDllPeMagic { found: u16 },
|
||||
|
||||
#[error(
|
||||
"DLL primary descriptor address 0x{address:08X} is below layout base 0x{minimum:08X} or outside {image_len}-byte image"
|
||||
)]
|
||||
InvalidDllPrimaryDescriptor {
|
||||
address: u32,
|
||||
minimum: u32,
|
||||
image_len: usize,
|
||||
},
|
||||
|
||||
#[error("invalid SizeOfImage {size}; expected 1..={max}")]
|
||||
InvalidImageSize { size: i64, max: u64 },
|
||||
|
||||
#[error(
|
||||
"{operation} range out of bounds (offset {offset}, size {size}, buffer length {buffer_len})"
|
||||
)]
|
||||
BufferRangeOutOfBounds {
|
||||
operation: BufferOperation,
|
||||
offset: usize,
|
||||
size: usize,
|
||||
buffer_len: usize,
|
||||
},
|
||||
|
||||
#[error(
|
||||
"EXE checksum descriptor at 0x{descriptor:08X} points outside input (offset {offset}, size {size}, input length {image_len})"
|
||||
)]
|
||||
ExeChecksumRangeOutOfBounds {
|
||||
descriptor: u32,
|
||||
offset: usize,
|
||||
size: usize,
|
||||
image_len: usize,
|
||||
},
|
||||
|
||||
#[error(
|
||||
"{table} descriptor out of bounds (offset {offset}, size 16, image length {image_len})"
|
||||
)]
|
||||
DescriptorOutOfBounds {
|
||||
table: DescriptorTable,
|
||||
offset: usize,
|
||||
image_len: usize,
|
||||
},
|
||||
|
||||
#[error("PE32 tbl not found — corrupt data or wrong offset")]
|
||||
Pe32TblNotFound,
|
||||
|
||||
#[error("PE32 thirdStage decrypt failed — corrupt data or wrong offset")]
|
||||
Pe32ThirdStageFailed,
|
||||
|
||||
#[error("PE32 customDecryptor not found in sevenStage")]
|
||||
Pe32CustomDecryptorNotFound,
|
||||
|
||||
#[error("PE32 eighthStageKey not found")]
|
||||
Pe32EighthKeyNotFound,
|
||||
|
||||
#[error("PE32 file LFSR not found in eighthStage")]
|
||||
Pe32FileLfsrNotFound,
|
||||
|
||||
#[error("{stage} decompression failed: {reason}")]
|
||||
StageDecompressionFailed {
|
||||
stage: DecompressionStage,
|
||||
reason: DecompressionFailure,
|
||||
},
|
||||
|
||||
#[error("{pipeline} section block {block} decompression failed")]
|
||||
SectionDecompressionFailed {
|
||||
pipeline: SectionPipeline,
|
||||
block: usize,
|
||||
},
|
||||
|
||||
#[error("AES key schedule is outside the image at offset {offset}")]
|
||||
InvalidAesKeySchedule { offset: u32 },
|
||||
|
||||
#[error("Huffman table is outside the image at offset {offset}")]
|
||||
InvalidHuffmanTable { offset: u32 },
|
||||
|
||||
#[error("DLL pipeline failed: {dll}; EXE fallback failed: {exe}")]
|
||||
PipelineFallbackFailed {
|
||||
dll: Box<UnpackError>,
|
||||
exe: Box<UnpackError>,
|
||||
},
|
||||
|
||||
#[error(
|
||||
"PE32 second-stage range is invalid (offset {offset}, size {size}, image length {image_len})"
|
||||
)]
|
||||
Pe32SecondStageRangeInvalid {
|
||||
offset: u32,
|
||||
size: u32,
|
||||
image_len: usize,
|
||||
},
|
||||
|
||||
#[error("PE32 relocation-data descriptor not found")]
|
||||
Pe32RelocationDataNotFound,
|
||||
|
||||
#[error("file decryptor candidate failed structural validation")]
|
||||
FileDecryptorValidationFailed,
|
||||
|
||||
#[error("PE32 memory image could not be rebuilt as a file-layout PE")]
|
||||
Pe32OutputLayoutInvalid,
|
||||
|
||||
#[error("internal panic at {file}:{line}:{column}: {message}")]
|
||||
InternalPanic {
|
||||
message: String,
|
||||
file: String,
|
||||
line: u32,
|
||||
column: u32,
|
||||
},
|
||||
}
|
||||
|
||||
impl From<senbei_crypto::Error> for UnpackError {
|
||||
fn from(error: senbei_crypto::Error) -> Self {
|
||||
match error {
|
||||
senbei_crypto::Error::BufferRangeOutOfBounds {
|
||||
operation,
|
||||
offset,
|
||||
size,
|
||||
buffer_len,
|
||||
} => Self::BufferRangeOutOfBounds {
|
||||
operation,
|
||||
offset,
|
||||
size,
|
||||
buffer_len,
|
||||
},
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
mod pipeline;
|
||||
|
||||
pub use pipeline::*;
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -77,6 +77,49 @@ fn rva_to_off(secs: &[Section], file_len: usize, rva: u32, need: u32) -> Option<
|
||||
None
|
||||
}
|
||||
|
||||
fn is_executable_rva(secs: &[Section], rva: u32) -> bool {
|
||||
secs.iter().any(|section| {
|
||||
let span = section.vsize.max(section.raw_size);
|
||||
rva >= section.va
|
||||
&& rva < section.va.wrapping_add(span)
|
||||
&& (section.chars & 0x2000_0000) != 0
|
||||
})
|
||||
}
|
||||
|
||||
fn check_common_entry_branches(
|
||||
stub: &[u8],
|
||||
ep: u32,
|
||||
secs: &[Section],
|
||||
report: &mut IntegrityReport,
|
||||
) {
|
||||
if stub.len() < 18
|
||||
|| stub[0..3] != [0x48, 0x83, 0xEC]
|
||||
|| stub[4] != 0xE8
|
||||
|| stub[9..12] != [0x48, 0x83, 0xC4]
|
||||
|| stub[12] != stub[3]
|
||||
|| stub[13] != 0xE9
|
||||
{
|
||||
return;
|
||||
}
|
||||
for (name, rel_off, instruction_len) in [("call", 5usize, 9i64), ("jump", 14usize, 18i64)] {
|
||||
let rel = i32::from_le_bytes([
|
||||
stub[rel_off],
|
||||
stub[rel_off + 1],
|
||||
stub[rel_off + 2],
|
||||
stub[rel_off + 3],
|
||||
]) as i64;
|
||||
let target = i64::from(ep) + instruction_len + rel;
|
||||
let valid = u32::try_from(target)
|
||||
.ok()
|
||||
.is_some_and(|rva| is_executable_rva(secs, rva));
|
||||
if !valid {
|
||||
report.issues.push(format!(
|
||||
"entry point {name} target 0x{target:X} is outside executable sections (DD8 selection is likely wrong)"
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Inspect an unpacked PE image and report any defect that would make the OS
|
||||
/// loader fault at runtime. `out` is the bytes the unpacker produced.
|
||||
pub fn check(out: &[u8]) -> IntegrityReport {
|
||||
@@ -253,6 +296,9 @@ pub fn check(out: &[u8]) -> IntegrityReport {
|
||||
"entry point RVA 0x{ep:X} is not in an executable section"
|
||||
));
|
||||
}
|
||||
if let Some(entry_stub) = out.get(off as usize..off as usize + 18) {
|
||||
check_common_entry_branches(entry_stub, ep, &secs, &mut r);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -363,3 +409,40 @@ fn looks_like_dll_name(d: &[u8], off: u32) -> bool {
|
||||
}
|
||||
d[start..end].iter().all(|&b| (0x20..0x7F).contains(&b))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn executable_text() -> Vec<Section> {
|
||||
vec![Section {
|
||||
va: 0x1000,
|
||||
vsize: 0x4000,
|
||||
raw_ptr: 0x1000,
|
||||
raw_size: 0x4000,
|
||||
chars: 0x6000_0020,
|
||||
}]
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn common_entry_stub_rejects_out_of_image_branches() {
|
||||
let stub = [
|
||||
0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x41, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9,
|
||||
0x7A, 0xFE, 0x54, 0xFF,
|
||||
];
|
||||
let mut report = IntegrityReport::default();
|
||||
check_common_entry_branches(&stub, 0x1264, &executable_text(), &mut report);
|
||||
assert_eq!(report.issues.len(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn common_entry_stub_accepts_executable_branches() {
|
||||
let stub = [
|
||||
0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x00, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9,
|
||||
0x7A, 0xFE, 0xFF, 0xFF,
|
||||
];
|
||||
let mut report = IntegrityReport::default();
|
||||
check_common_entry_branches(&stub, 0x1264, &executable_text(), &mut report);
|
||||
assert!(report.ok());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
//! Internal PE layout discovery and image reconstruction.
|
||||
|
||||
mod dd8;
|
||||
mod discovery;
|
||||
mod image;
|
||||
|
||||
pub(super) use dd8::{select_dd8_formula_pe32, select_dd8_shift};
|
||||
pub(super) use discovery::{
|
||||
discover_eighth_slots, find_bytecode_offset, find_lfsr_block, find_str_pos, find_tbl_pe32,
|
||||
find_v_after_pad, find_v4_offset, get_string_to_null, section_name, trial_decrypt5_u32,
|
||||
};
|
||||
pub(super) use image::{
|
||||
compact_memory_image_to_pe, move_pe32_imports_to_kmiat, pe32_imports_already_match_idata_layout,
|
||||
};
|
||||
@@ -0,0 +1,609 @@
|
||||
//! Validation-driven selection for per-page text transforms.
|
||||
|
||||
use super::discovery::trial_decrypt5_u32;
|
||||
|
||||
/// PE32 `.text` dd8 key-formula selection with a skip decision. The packer keys
|
||||
/// the per-page XOR either with `page+1` or `0x8000*(page+1)`; the formula is
|
||||
/// not recorded. Replays the dd8 page pass on a scratch copy of sample pages
|
||||
/// (25/50/75% of `.text`) under each formula and counts how many positions
|
||||
/// decode to `0xCC` (int3 padding).
|
||||
///
|
||||
/// Returns `Some(true)` for the `0x8000*(page+1)` formula, `Some(false)` for
|
||||
/// `page+1`, or `None` when `.text` must NOT be dd8-decrypted at all. The packer
|
||||
/// dd8-encrypts `.text` on EXEs (so unpacking must replay it) but leaves a native
|
||||
/// DLL's `.text` plaintext; replaying dd8 there scrambles ~1 byte per 16-byte
|
||||
/// block. The decision: dd8 only *restores* int3 padding when `.text` was
|
||||
/// genuinely encrypted, so apply it only when the chosen formula's whole-page
|
||||
/// 0xCC count rises *clearly* above the no-dd8 baseline; otherwise skip.
|
||||
///
|
||||
/// "Clearly" matters: dd8 XORs 255 positions per page with pseudo-random bytes,
|
||||
/// so on an already-plaintext `.text` it manufactures ~1 spurious `0xCC` per
|
||||
/// sampled page for free (255/256 expected). A bare `best > baseline` test is
|
||||
/// therefore biased towards *applying* dd8 on exactly the inputs that must skip
|
||||
/// it — and a wrongly-applied dd8 is silent: it scrambles ~1 byte per 16 with no
|
||||
/// error and nothing downstream (not even `integrity::check`, which only reads
|
||||
/// 16 bytes at the entry point) notices. The [`MIN_DD8_NET_GAIN`] floor below is
|
||||
/// the PE32 counterpart of the margin+floor `select_dd8_shift` already applies
|
||||
/// on PE32+ for the same failure mode.
|
||||
pub fn select_dd8_formula_pe32(data: &[u8], text_off: u32, text_size: u32) -> Option<bool> {
|
||||
let num_pages_total = text_size / 0x1000;
|
||||
let mut sample_pages: Vec<u32> = Vec::new();
|
||||
for frac in [0.25f64, 0.5, 0.75] {
|
||||
let pg = (num_pages_total as f64 * frac) as u32;
|
||||
if pg > 0 && pg < num_pages_total {
|
||||
sample_pages.push(pg);
|
||||
}
|
||||
}
|
||||
if sample_pages.is_empty() && num_pages_total > 1 {
|
||||
sample_pages.push(num_pages_total / 2);
|
||||
}
|
||||
let score = |big: bool| -> i64 {
|
||||
let mut total = 0i64;
|
||||
for &sp in &sample_pages {
|
||||
let pg_off = (text_off + sp * 0x1000) as usize;
|
||||
if pg_off + 0x1000 > data.len() {
|
||||
continue;
|
||||
}
|
||||
let mut buf = [0u8; 0x1000];
|
||||
buf.copy_from_slice(&data[pg_off..pg_off + 0x1000]);
|
||||
let pk = if big {
|
||||
0x8000u32.wrapping_mul(sp.wrapping_add(1))
|
||||
} else {
|
||||
sp.wrapping_add(1)
|
||||
};
|
||||
let mut k = pk;
|
||||
let rk = k.rotate_right(15);
|
||||
k = rk;
|
||||
for bi in 1..256u32 {
|
||||
let rk = k.rotate_right(15);
|
||||
let ri = rk.wrapping_add(bi);
|
||||
k = ri.wrapping_add(bi);
|
||||
let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize;
|
||||
if tidx < buf.len() {
|
||||
buf[tidx] ^= k as u8;
|
||||
}
|
||||
}
|
||||
total += buf.iter().filter(|&&b| b == 0xCC).count() as i64;
|
||||
}
|
||||
total
|
||||
};
|
||||
let s_small = score(false);
|
||||
let s_big = score(true);
|
||||
// Baseline: whole-page 0xCC over the same sample pages with NO dd8. dd8 only
|
||||
// rewrites 255 bytes per page, so comparing the chosen formula's whole-page
|
||||
// 0xCC against this baseline reveals whether dd8 *restores* int3 padding
|
||||
// (count rises -> .text was packer-encrypted, apply) or merely scrambles
|
||||
// already-plaintext code (count falls -> native-DLL .text left intact, skip).
|
||||
let mut baseline: i64 = 0;
|
||||
for &sp in &sample_pages {
|
||||
let pg_off = (text_off + sp * 0x1000) as usize;
|
||||
if pg_off + 0x1000 > data.len() {
|
||||
continue;
|
||||
}
|
||||
baseline += data[pg_off..pg_off + 0x1000]
|
||||
.iter()
|
||||
.filter(|&&b| b == 0xCC)
|
||||
.count() as i64;
|
||||
}
|
||||
let big = s_big > s_small;
|
||||
let best = s_small.max(s_big);
|
||||
// Minimum net 0xCC gain over the baseline before dd8 is applied. Noise on an
|
||||
// already-plaintext `.text` is ~1 manufactured 0xCC per sampled page (3 pages
|
||||
// -> ~3); every corpus build that genuinely needs dd8 gains +154 or more
|
||||
// (observed +154 and +312), and the one native DLL that must skip scores -18.
|
||||
// A floor of 32 sits ~10x above the noise and ~5x below the smallest true
|
||||
// positive, so it changes no existing decision.
|
||||
const MIN_DD8_NET_GAIN: i64 = 32;
|
||||
let apply = best.saturating_sub(baseline) >= MIN_DD8_NET_GAIN;
|
||||
if std::env::var("SEL_DIAG").is_ok() {
|
||||
eprintln!(
|
||||
"SEL pe32 dd8 s_small={} s_big={} baseline={} gain={} big={} apply={}",
|
||||
s_small,
|
||||
s_big,
|
||||
baseline,
|
||||
best - baseline,
|
||||
big,
|
||||
apply
|
||||
);
|
||||
}
|
||||
// When no interior pages could be sampled (tiny .text) we cannot measure the
|
||||
// effect; preserve the historical behavior of applying dd8.
|
||||
if sample_pages.is_empty() || apply {
|
||||
Some(big)
|
||||
} else {
|
||||
None
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// dd8 page-XOR shift selection.
|
||||
//
|
||||
// The packer scrambles ~1 byte per 16-byte block of .text via decrypt_data8,
|
||||
// keyed by `page_idx << shift` (absolute page index = text_va >> 12). Observed
|
||||
// shifts are 0 and 15. The shift is NOT stored in any header/config field, so
|
||||
// the decision must be validated against the resulting .text content.
|
||||
//
|
||||
// A recognised CRT entry stub is the strongest oracle: decode skip/0/15 and
|
||||
// require both of its direct rel32 branches to land in executable .text. This
|
||||
// includes the call/jump displacement bytes themselves; an older entry oracle
|
||||
// wildcarded those bytes and could accept a stub whose opcodes looked right but
|
||||
// whose branch targets were outside the image.
|
||||
//
|
||||
// Other entry shapes fall back to padding statistics over a few sample pages
|
||||
// (head/tail margin skipped: entry/exit regions have atypical padding density).
|
||||
// The primary signal is a *structural* fingerprint: the MSVC function-end
|
||||
// padding pattern, a 0xC3 RET opcode followed by a run of >= 4 0xCC int3 bytes.
|
||||
// dd8 XORs one pseudo-random byte per 16-byte block, so an already-plaintext
|
||||
// page keeps its padding runs only under "no dd8", while a packer-encrypted
|
||||
// page restores them only under the correct shift — a wrong candidate destroys
|
||||
// every run it touches and essentially never manufactures a RET followed by a
|
||||
// long int3 run by chance. This separates the states far more cleanly than a
|
||||
// bare 0xCC count, which a wrong candidate inflates for free (~255 coincidences
|
||||
// per page at p=1/256).
|
||||
//
|
||||
// When no candidate produces any RET-anchored padding (sampled pages with
|
||||
// dense code and no padded epilogues), the fingerprint is silent, so the
|
||||
// decision falls back to the older mutated-position 0xCC count. Both signals
|
||||
// use the same decision rule: a candidate must beat the no-dd8 baseline by a
|
||||
// clear margin AND an absolute floor, otherwise dd8 is skipped — a wrongly
|
||||
// applied dd8 scrambles ~1 byte per 16 with no error surfaced downstream.
|
||||
// ---------------------------------------------------------------------------
|
||||
pub fn select_dd8_shift(data: &[u8], text_va: u32, text_size: u32, info3: u32) -> u32 {
|
||||
if let Some((shift, scores)) = select_dd8_by_entry_stub(data, text_va, text_size, info3) {
|
||||
if std::env::var("SEL_DIAG").is_ok() {
|
||||
eprintln!(
|
||||
"SEL dd8 entry best_shift={} none={} s0={} s15={}",
|
||||
shift, scores[0], scores[1], scores[2]
|
||||
);
|
||||
}
|
||||
return shift;
|
||||
}
|
||||
let num_pages_total = text_size >> 12;
|
||||
// Fewer than two pages: nothing meaningful to sample; preserve the
|
||||
// historical behavior (shift 0 — the dd8 loop is empty or single-page).
|
||||
if num_pages_total < 2 {
|
||||
return 0;
|
||||
}
|
||||
let text_off = text_va as usize;
|
||||
|
||||
// Sample up to 4 pages, skipping a head/tail margin (entry/exit regions
|
||||
// have atypical padding density). Small .text: sample every page.
|
||||
let mut sample_pages: Vec<u32> = Vec::new();
|
||||
if num_pages_total <= 4 {
|
||||
sample_pages.extend(0..num_pages_total);
|
||||
} else {
|
||||
let margin = (num_pages_total / 8).max(1);
|
||||
let lo = margin;
|
||||
let hi = num_pages_total - margin;
|
||||
if hi <= lo {
|
||||
sample_pages.extend(0..num_pages_total);
|
||||
} else {
|
||||
let step = ((hi - lo) / 4).max(1);
|
||||
let mut i = 0;
|
||||
while i < 4 {
|
||||
let p = lo + i * step;
|
||||
if p < num_pages_total {
|
||||
sample_pages.push(p);
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
if sample_pages.is_empty() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
let abs_base = text_va >> 12;
|
||||
// Require a clear 2x margin over the already-plaintext baseline AND an
|
||||
// absolute floor. The 2x test alone trips on noise when the counts are
|
||||
// tiny: an external-companion DLL whose .text is already plaintext scores
|
||||
// s15=4 vs none=1 — a spurious 4x — and gets dd8 wrongly applied,
|
||||
// corrupting ~1 byte per 16. The floor rejects that noise while sitting
|
||||
// far below every genuinely-encrypted build's score.
|
||||
const MIN_DD8_HITS: u32 = 8;
|
||||
let margin_pick = |none: u32, s0: u32, s15: u32| -> u32 {
|
||||
let mut best_score = none;
|
||||
let mut best_shift = 99u32; // 99 == skip dd8
|
||||
for (shift, hits) in [(0u32, s0), (15u32, s15)] {
|
||||
if hits > best_score {
|
||||
best_score = hits;
|
||||
best_shift = shift;
|
||||
}
|
||||
}
|
||||
if best_shift != 99 && (best_score < none * 2 || best_score < MIN_DD8_HITS) {
|
||||
best_shift = 99;
|
||||
}
|
||||
best_shift
|
||||
};
|
||||
|
||||
// Primary: RET+int3 padding fingerprint. The fingerprint is diluted across
|
||||
// the whole page (dd8 touches only 255 of 4096 bytes, so even an encrypted
|
||||
// page keeps most of its padding runs), so instead of the fallback's 2x
|
||||
// margin the gate is a *positive delta* over the no-dd8 baseline: on an
|
||||
// already-plaintext .text each wrong shift destroys runs (scores below the
|
||||
// baseline), while the correct shift on an encrypted page restores them
|
||||
// (scores above it). The floor on the delta rejects noise-level gains.
|
||||
let r_none = fingerprint_score(data, text_off, abs_base, &sample_pages, None);
|
||||
let r0 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(0));
|
||||
let r15 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(15));
|
||||
// Fallback: mutated-position 0xCC count, for pages whose code has no
|
||||
// RET-anchored padding at all (the fingerprint is silent there).
|
||||
let (none_hits, s0, s15);
|
||||
let best_shift = if r_none != 0 || r0 != 0 || r15 != 0 {
|
||||
none_hits = 0;
|
||||
s0 = 0;
|
||||
s15 = 0;
|
||||
let mut best_score = r_none;
|
||||
let mut shift = 99u32;
|
||||
for (s, score) in [(0u32, r0), (15u32, r15)] {
|
||||
if score > best_score {
|
||||
best_score = score;
|
||||
shift = s;
|
||||
}
|
||||
}
|
||||
if shift != 99 && best_score.saturating_sub(r_none) < MIN_DD8_HITS {
|
||||
shift = 99;
|
||||
}
|
||||
shift
|
||||
} else {
|
||||
none_hits = score_dd8_baseline(data, text_off, &sample_pages);
|
||||
s0 = score_dd8_shift(data, text_off, text_va, &sample_pages, 0);
|
||||
s15 = score_dd8_shift(data, text_off, text_va, &sample_pages, 15);
|
||||
margin_pick(none_hits, s0, s15)
|
||||
};
|
||||
if std::env::var("SEL_DIAG").is_ok() {
|
||||
eprintln!(
|
||||
"SEL dd8 best_shift={} fp=({},{},{}) cc=({},{},{}) samples={:?}",
|
||||
best_shift, r_none, r0, r15, none_hits, s0, s15, sample_pages
|
||||
);
|
||||
}
|
||||
best_shift
|
||||
}
|
||||
|
||||
/// Minimum 0xCC run length after a RET for the run to count as MSVC
|
||||
/// function-end padding.
|
||||
const MIN_CC_RUN: u32 = 4;
|
||||
|
||||
/// Total length of MSVC function-end padding runs in a page: each 0xC3 byte
|
||||
/// followed by >= [`MIN_CC_RUN`] 0xCC bytes contributes the run length.
|
||||
fn ret_int3_score(page: &[u8]) -> u32 {
|
||||
let mut total = 0u32;
|
||||
let mut i = 0;
|
||||
while i < page.len() {
|
||||
if page[i] == 0xC3 {
|
||||
let mut j = i + 1;
|
||||
while j < page.len() && page[j] == 0xCC {
|
||||
j += 1;
|
||||
}
|
||||
let run = (j - i - 1) as u32;
|
||||
if run >= MIN_CC_RUN {
|
||||
total += run;
|
||||
}
|
||||
i = j;
|
||||
} else {
|
||||
i += 1;
|
||||
}
|
||||
}
|
||||
total
|
||||
}
|
||||
|
||||
/// Replay the dd8 page-XOR in place on one sample page.
|
||||
fn dd8_apply(buf: &mut [u8; 0x1000], abs_page: u32, shift: u32) {
|
||||
let mut key = abs_page << shift;
|
||||
for bi in 0..256u32 {
|
||||
let mixed = key.rotate_right(15).wrapping_add(bi);
|
||||
key = mixed.wrapping_add(bi);
|
||||
// The packer's dd8 loop does not XOR block i=0 (see decrypt_data8).
|
||||
if bi == 0 {
|
||||
continue;
|
||||
}
|
||||
let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize;
|
||||
buf[tidx] ^= key as u8;
|
||||
}
|
||||
}
|
||||
|
||||
/// Sum the RET+int3 fingerprint over the sample pages for one candidate
|
||||
/// (`None` = the no-dd8 baseline, page as-is).
|
||||
fn fingerprint_score(
|
||||
data: &[u8],
|
||||
text_off: usize,
|
||||
abs_base: u32,
|
||||
sample_pages: &[u32],
|
||||
shift: Option<u32>,
|
||||
) -> u32 {
|
||||
let mut total = 0u32;
|
||||
for &sp in sample_pages {
|
||||
let pg_off = text_off + (sp as usize) * 0x1000;
|
||||
if pg_off + 0x1000 > data.len() {
|
||||
continue;
|
||||
}
|
||||
let mut page = [0u8; 0x1000];
|
||||
page.copy_from_slice(&data[pg_off..pg_off + 0x1000]);
|
||||
if let Some(sh) = shift {
|
||||
dd8_apply(&mut page, abs_base.wrapping_add(sp), sh);
|
||||
}
|
||||
total += ret_int3_score(&page);
|
||||
}
|
||||
total
|
||||
}
|
||||
|
||||
/// Select DD8 from the common CRT entry stub when its direct call and jump
|
||||
/// provide a stronger oracle than sparse padding statistics. The candidate is
|
||||
/// accepted only when it is the sole one whose two branch targets stay inside
|
||||
/// `.text`; unrecognised entry code falls through to the padding selector.
|
||||
fn select_dd8_by_entry_stub(
|
||||
data: &[u8],
|
||||
text_va: u32,
|
||||
text_size: u32,
|
||||
info3: u32,
|
||||
) -> Option<(u32, [u8; 3])> {
|
||||
for entry in entry_candidates(data, text_va, text_size, info3) {
|
||||
let [Some(none), Some(s0), Some(s15)] = [None, Some(0), Some(15)]
|
||||
.map(|shift| entry_stub_branch_score(data, text_va, text_size, entry, shift))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
let scores = [none, s0, s15];
|
||||
let best = scores.iter().copied().max()?;
|
||||
if best == 2 && scores.iter().filter(|&&score| score == best).count() == 1 {
|
||||
let index = scores.iter().position(|&score| score == best)?;
|
||||
return Some(([99, 0, 15][index], scores));
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
fn entry_candidates(data: &[u8], text_va: u32, text_size: u32, info3: u32) -> Vec<u32> {
|
||||
let text_end = text_va.saturating_add(text_size);
|
||||
let mut entries = Vec::with_capacity(3);
|
||||
if let Some(pe) = read_u32(data, 0x3C)
|
||||
&& let Some(entry) = pe.checked_add(40).and_then(|offset| read_u32(data, offset))
|
||||
&& (text_va..text_end).contains(&entry)
|
||||
{
|
||||
entries.push(entry);
|
||||
}
|
||||
for metadata_off in [32u32, 64] {
|
||||
let Some(end) = info3
|
||||
.checked_add(metadata_off)
|
||||
.and_then(|offset| offset.checked_add(8))
|
||||
else {
|
||||
continue;
|
||||
};
|
||||
if end as usize > data.len() {
|
||||
continue;
|
||||
}
|
||||
let entry = trial_decrypt5_u32(data, info3 + metadata_off);
|
||||
let image_base = trial_decrypt5_u32(data, info3 + metadata_off + 4);
|
||||
if image_base == info3 && (text_va..text_end).contains(&entry) && !entries.contains(&entry)
|
||||
{
|
||||
entries.push(entry);
|
||||
}
|
||||
}
|
||||
entries
|
||||
}
|
||||
|
||||
fn entry_stub_branch_score(
|
||||
data: &[u8],
|
||||
text_va: u32,
|
||||
text_size: u32,
|
||||
entry: u32,
|
||||
shift: Option<u32>,
|
||||
) -> Option<u8> {
|
||||
let text_end = text_va.checked_add(text_size)?;
|
||||
if entry < text_va || entry.checked_add(18)? > text_end {
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut stub = [0u8; 18];
|
||||
for (offset, byte) in stub.iter_mut().enumerate() {
|
||||
*byte = dd8_candidate_byte(data, entry + offset as u32, shift)?;
|
||||
}
|
||||
if stub[0..3] != [0x48, 0x83, 0xEC]
|
||||
|| stub[4] != 0xE8
|
||||
|| stub[9..12] != [0x48, 0x83, 0xC4]
|
||||
|| stub[12] != stub[3]
|
||||
|| stub[13] != 0xE9
|
||||
{
|
||||
return None;
|
||||
}
|
||||
|
||||
let call_rel = i32::from_le_bytes(stub[5..9].try_into().ok()?) as i64;
|
||||
let jump_rel = i32::from_le_bytes(stub[14..18].try_into().ok()?) as i64;
|
||||
let call_target = i64::from(entry) + 9 + call_rel;
|
||||
let jump_target = i64::from(entry) + 18 + jump_rel;
|
||||
let in_text = |target: i64| target >= i64::from(text_va) && target < i64::from(text_end);
|
||||
Some(u8::from(in_text(call_target)) + u8::from(in_text(jump_target)))
|
||||
}
|
||||
|
||||
fn dd8_candidate_byte(data: &[u8], rva: u32, shift: Option<u32>) -> Option<u8> {
|
||||
let mut byte = *data.get(rva as usize)?;
|
||||
let Some(shift) = shift else {
|
||||
return Some(byte);
|
||||
};
|
||||
let page = rva >> 12;
|
||||
let block = (rva & 0xFFF) >> 4;
|
||||
let mut key = page << shift;
|
||||
for index in 0..=block {
|
||||
let mixed = key.rotate_right(15).wrapping_add(index);
|
||||
key = mixed.wrapping_add(index);
|
||||
if index != 0 {
|
||||
let target = (page << 12)
|
||||
.wrapping_add(index << 4)
|
||||
.wrapping_add(mixed & 0xF);
|
||||
if target == rva {
|
||||
byte ^= key as u8;
|
||||
}
|
||||
}
|
||||
}
|
||||
Some(byte)
|
||||
}
|
||||
|
||||
fn read_u32(data: &[u8], offset: u32) -> Option<u32> {
|
||||
let start = offset as usize;
|
||||
let bytes = data.get(start..start.checked_add(4)?)?;
|
||||
Some(u32::from_le_bytes(bytes.try_into().ok()?))
|
||||
}
|
||||
|
||||
// Baseline: count int3 pads already present at the first byte of each 16-byte
|
||||
// block, i.e. the positions dd8 would target if its in-block offset were 0.
|
||||
fn score_dd8_baseline(data: &[u8], text_off: usize, sample_pages: &[u32]) -> u32 {
|
||||
let mut hits = 0u32;
|
||||
for &sp in sample_pages {
|
||||
let pg_off = text_off + (sp as usize) * 0x1000;
|
||||
if pg_off + 0x1000 > data.len() {
|
||||
continue;
|
||||
}
|
||||
for bi in 1..256usize {
|
||||
if data[pg_off + bi * 16] == 0xCC {
|
||||
hits += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
hits
|
||||
}
|
||||
|
||||
// Replay decrypt_data8 on each sample page under `shift` and count how many of
|
||||
// the 255 mutated positions decode to 0xCC.
|
||||
fn score_dd8_shift(
|
||||
data: &[u8],
|
||||
text_off: usize,
|
||||
text_va: u32,
|
||||
sample_pages: &[u32],
|
||||
shift: u32,
|
||||
) -> u32 {
|
||||
let abs_base = text_va >> 12;
|
||||
let mut hits = 0u32;
|
||||
for &sp in sample_pages {
|
||||
let pg_off = text_off + (sp as usize) * 0x1000;
|
||||
if pg_off + 0x1000 > data.len() {
|
||||
continue;
|
||||
}
|
||||
let abs_page = abs_base.wrapping_add(sp);
|
||||
let mut key = abs_page << shift;
|
||||
for bi in 0..256u32 {
|
||||
let mixed = key.rotate_right(15).wrapping_add(bi);
|
||||
key = mixed.wrapping_add(bi);
|
||||
if bi == 0 {
|
||||
continue;
|
||||
}
|
||||
let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize;
|
||||
if tidx < 0x1000 {
|
||||
let mutated = data[pg_off + tidx] ^ (key as u8);
|
||||
if mutated == 0xCC {
|
||||
hits += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
hits
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn entry_stub_fixture() -> Vec<u8> {
|
||||
let mut data = vec![0u8; 0x5000];
|
||||
data[0x3C..0x40].copy_from_slice(&0x100u32.to_le_bytes());
|
||||
data[0x128..0x12C].copy_from_slice(&0x1264u32.to_le_bytes());
|
||||
data[0x1264..0x1276].copy_from_slice(&[
|
||||
0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x00, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9,
|
||||
0x7A, 0xFE, 0xFF, 0xFF,
|
||||
]);
|
||||
data
|
||||
}
|
||||
|
||||
fn apply_dd8_page(data: &mut [u8], page_rva: u32, shift: u32) {
|
||||
let mut key = (page_rva >> 12) << shift;
|
||||
for index in 0..256u32 {
|
||||
let mixed = key.rotate_right(15).wrapping_add(index);
|
||||
key = mixed.wrapping_add(index);
|
||||
if index == 0 {
|
||||
continue;
|
||||
}
|
||||
let target = page_rva.wrapping_add(index << 4).wrapping_add(mixed & 0xF) as usize;
|
||||
data[target] ^= key as u8;
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn entry_stub_selects_plaintext_and_both_dd8_shifts() {
|
||||
let plain = entry_stub_fixture();
|
||||
assert_eq!(select_dd8_shift(&plain, 0x1000, 0x4000, 0), 99);
|
||||
|
||||
for expected in [0u32, 15] {
|
||||
let mut encrypted = plain.clone();
|
||||
apply_dd8_page(&mut encrypted, 0x1000, expected);
|
||||
assert_eq!(select_dd8_shift(&encrypted, 0x1000, 0x4000, 0), expected);
|
||||
}
|
||||
}
|
||||
|
||||
/// Seed the first `count` dd8-targeted positions of each sampled page with
|
||||
/// the byte that decodes to `0xCC` under the `page+1` formula — i.e. an
|
||||
/// encrypted `.text` whose plaintext is int3 padding. Positions whose key
|
||||
/// byte would make the *ciphertext* itself `0xCC` are skipped so the
|
||||
/// fixture contains no `0xCC` at all and every post-dd8 `0xCC` is a genuine
|
||||
/// gain over a zero baseline.
|
||||
fn seed_dd8_int3(data: &mut [u8], text_off: u32, pages: &[u32], count: u32) {
|
||||
for &sp in pages {
|
||||
let pg_off = (text_off + sp * 0x1000) as usize;
|
||||
let mut k = sp.wrapping_add(1);
|
||||
k = k.rotate_right(15);
|
||||
let mut planted = 0u32;
|
||||
for bi in 1..256u32 {
|
||||
let ri = k.rotate_right(15).wrapping_add(bi);
|
||||
k = ri.wrapping_add(bi);
|
||||
if planted >= count {
|
||||
continue;
|
||||
}
|
||||
let ct = 0xCCu8 ^ (k as u8);
|
||||
if ct == 0xCC {
|
||||
continue;
|
||||
}
|
||||
let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize;
|
||||
data[pg_off + tidx] = ct;
|
||||
planted += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Review regression: a near-plaintext `.text` must NOT be dd8-decrypted.
|
||||
/// dd8 XORs 255 positions per page with pseudo-random bytes, so it
|
||||
/// manufactures a few `0xCC` for free — under the old bare
|
||||
/// `best > baseline` test any positive gain was enough to "apply" dd8 and
|
||||
/// scramble ~1 byte per 16 of a native DLL's already-plaintext code,
|
||||
/// silently (nothing downstream, including the integrity check, notices).
|
||||
/// Here the gain is real but small; the floor must still reject it.
|
||||
#[test]
|
||||
fn pe32_dd8_skips_text_whose_gain_is_only_noise_sized() {
|
||||
let text_off: u32 = 0x1000;
|
||||
let text_size: u32 = 8 * 0x1000;
|
||||
let mut data = vec![0u8; (text_off + text_size) as usize];
|
||||
seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 5);
|
||||
assert!(
|
||||
!data.contains(&0xCC),
|
||||
"fixture must have a zero 0xCC baseline"
|
||||
);
|
||||
assert_eq!(
|
||||
select_dd8_formula_pe32(&data, text_off, text_size),
|
||||
None,
|
||||
"a gain this small is indistinguishable from dd8's own noise"
|
||||
);
|
||||
}
|
||||
|
||||
/// Control for the above: a `.text` whose dd8 pass restores a large amount
|
||||
/// of int3 padding clears the floor and is decrypted. Same fixture shape,
|
||||
/// only the amount of restored padding differs.
|
||||
#[test]
|
||||
fn pe32_dd8_applies_when_padding_is_restored() {
|
||||
let text_off: u32 = 0x1000;
|
||||
let text_size: u32 = 8 * 0x1000;
|
||||
let mut data = vec![0u8; (text_off + text_size) as usize];
|
||||
seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 255);
|
||||
assert_eq!(
|
||||
select_dd8_formula_pe32(&data, text_off, text_size),
|
||||
Some(false),
|
||||
"encrypted .text must be decrypted with the page+1 formula"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,507 @@
|
||||
//! Structural locators for protected PE stages.
|
||||
|
||||
use senbei_crypto::primitives::{get_u32, lfsr_keystream};
|
||||
|
||||
/// Find the 4-byte v_val that follows the LAST occurrence of `48 EB 01 B9`
|
||||
/// (REX.W jmp+1; mov ecx,imm32) plus any 0xCC padding. Used to locate
|
||||
/// stage4's accum2 seed. Works across builds even when API-name anchors are
|
||||
/// absent.
|
||||
pub fn find_v_after_pad(data: &[u8], base: u32, len: u32) -> Option<u32> {
|
||||
let start = base as usize;
|
||||
let end = (base.saturating_add(len)) as usize;
|
||||
if end > data.len() {
|
||||
return None;
|
||||
}
|
||||
let sig = [0x48u8, 0xEB, 0x01, 0xB9];
|
||||
let slice = &data[start..end];
|
||||
// last occurrence
|
||||
let mut last = None;
|
||||
let mut i = 0usize;
|
||||
while i + sig.len() <= slice.len() {
|
||||
if slice[i..i + sig.len()] == sig {
|
||||
last = Some(i);
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
let pos = last?;
|
||||
// skip CCs after the `48 EB 01 B9`
|
||||
let mut after = pos + sig.len();
|
||||
while after < slice.len() && slice[after] == 0xCC {
|
||||
after += 1;
|
||||
}
|
||||
if after + 4 > slice.len() {
|
||||
return None;
|
||||
}
|
||||
Some((start + after) as u32)
|
||||
}
|
||||
|
||||
/// Predict the 4 bytes that DecryptData5(va, size) would produce at va+0..va+4
|
||||
/// without mutating the buffer. The cipher's per-byte transform depends only
|
||||
/// on the byte itself and the low 8 bits of (va+i), with no cross-byte state,
|
||||
/// so each byte can be decrypted in isolation. Used to detect the EP/DD layout
|
||||
/// offset before committing to the actual call.
|
||||
pub fn trial_decrypt5_u32(data: &[u8], va: u32) -> u32 {
|
||||
let mut out = [0u8; 4];
|
||||
for i in 0..4u32 {
|
||||
let b3 = data[(va + i) as usize];
|
||||
let b = (va + i) as u8;
|
||||
let b2 = b.wrapping_add(1);
|
||||
let b4 = b3.rotate_left(2) ^ b2;
|
||||
let b5 = b4.rotate_left(2) ^ b;
|
||||
out[i as usize] = b5.rotate_left(2);
|
||||
}
|
||||
u32::from_le_bytes(out)
|
||||
}
|
||||
|
||||
/// Scan stage4/stage5 for the encrypted custom-decryptor bytecode block. The
|
||||
/// raw byte at p+95 is used by decrypt_data6 as the iteration count. We trial-
|
||||
/// decrypt that many bytes with the LFSR keystream and accept the first
|
||||
/// position where the byte stream parses as a valid opcode sequence ending in
|
||||
/// 195 (ret).
|
||||
pub fn find_bytecode_offset(data: &[u8], base: u32, len: u32) -> Option<u32> {
|
||||
let start = base as usize;
|
||||
let end = (base.saturating_add(len)) as usize;
|
||||
if end > data.len() {
|
||||
return None;
|
||||
}
|
||||
let mut ks = [0u8; 256];
|
||||
lfsr_keystream(&mut ks);
|
||||
// Scan forward from `start+16` on 16-byte boundaries relative to `start`.
|
||||
// The bytecode block is positioned a fixed offset into stage4/stage5; the
|
||||
// lowest parseable candidate is the real one (later ones are coincidental
|
||||
// parses of trailing filler bytes that happen to map to valid opcodes).
|
||||
// The enclosing buffer isn't necessarily 16-aligned to its absolute
|
||||
// address in newer builds, so we anchor the stride to `start`.
|
||||
let mut p = start + 16;
|
||||
while p + 96 <= end {
|
||||
let count = data[p + 95] as usize;
|
||||
if count >= 8 && p + count <= end {
|
||||
let mut buf = [0u8; 256];
|
||||
let take = count.min(256);
|
||||
for i in 0..take {
|
||||
buf[i] = data[p + i] ^ ks[i];
|
||||
}
|
||||
if let Some(nops) = parse_bytecode_check(&buf[..take])
|
||||
&& nops >= 4
|
||||
{
|
||||
return Some(p as u32);
|
||||
}
|
||||
}
|
||||
p += 16;
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Validate bytecode structure without allocating a `Vec` of ops. Returns
|
||||
/// `Some(non_nop_op_count)` if the byte stream parses successfully as a valid
|
||||
/// opcode sequence ending in 195 (ret), `None` otherwise. Allows non-trivial
|
||||
/// bytecode filtering by op count.
|
||||
pub fn parse_bytecode_check(buf: &[u8]) -> Option<usize> {
|
||||
let mut i = 0usize;
|
||||
let mut nops: usize = 0;
|
||||
while i < buf.len() {
|
||||
let b = buf[i];
|
||||
i += 1;
|
||||
match b {
|
||||
4 | 44 | 52 => {
|
||||
if i >= buf.len() {
|
||||
return None;
|
||||
}
|
||||
i += 1;
|
||||
nops += 1;
|
||||
}
|
||||
144 => {}
|
||||
192 | 254 => {
|
||||
if i >= buf.len() {
|
||||
return None;
|
||||
}
|
||||
let mb = buf[i];
|
||||
i += 1;
|
||||
let rm = mb & 7;
|
||||
let mod_ = (mb >> 6) & 3;
|
||||
let reg = (mb >> 3) & 7;
|
||||
if mod_ != 3 || rm != 0 {
|
||||
return None;
|
||||
}
|
||||
if reg > 1 {
|
||||
return None;
|
||||
}
|
||||
if b == 192 {
|
||||
if i >= buf.len() {
|
||||
return None;
|
||||
}
|
||||
i += 1;
|
||||
}
|
||||
nops += 1;
|
||||
}
|
||||
195 => return Some(nops),
|
||||
_ => return None,
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Locate stage3's v4_val: the last non-zero dword in the buffer, anchored
|
||||
/// by the `C3 CC CC CC` (ret + 3 int3) immediately before it.
|
||||
pub fn find_v4_offset(data: &[u8], base: u32, len: u32) -> Option<u32> {
|
||||
let start = base as usize;
|
||||
let end = (base.saturating_add(len)) as usize;
|
||||
if end > data.len() || end < start + 4 {
|
||||
return None;
|
||||
}
|
||||
// walk backwards looking for the first non-zero byte
|
||||
let mut i = end;
|
||||
while i > start && data[i - 1] == 0 {
|
||||
i -= 1;
|
||||
}
|
||||
if i < start + 4 {
|
||||
return None;
|
||||
}
|
||||
// v_val occupies the 4 bytes ending at i (rounded up to dword boundary)
|
||||
let v_end = i;
|
||||
let v_start = ((v_end + 3) & !3).saturating_sub(4);
|
||||
// require that the 4 bytes preceding v_val match `C3 CC CC CC`
|
||||
if v_start < start + 4 || data[v_start - 4..v_start] != [0xC3, 0xCC, 0xCC, 0xCC] {
|
||||
return None;
|
||||
}
|
||||
Some(v_start as u32)
|
||||
}
|
||||
|
||||
/// Scan a sub-buffer for an ASCII needle; return its absolute position.
|
||||
pub fn find_str_pos(data: &[u8], base: u32, len: u32, needle: &[u8]) -> Option<u32> {
|
||||
let start = base as usize;
|
||||
let end = (base.saturating_add(len)) as usize;
|
||||
if end > data.len() || needle.is_empty() {
|
||||
return None;
|
||||
}
|
||||
data[start..end]
|
||||
.windows(needle.len())
|
||||
.position(|w| w == needle)
|
||||
.map(|rel| (start + rel) as u32)
|
||||
}
|
||||
|
||||
pub fn get_string_to_null(data: &[u8], offset: u32) -> String {
|
||||
let start = offset as usize;
|
||||
if start >= data.len() {
|
||||
return String::new();
|
||||
}
|
||||
// Bounded: an unterminated run must never walk off the end of the buffer
|
||||
// (panic) or scan unboundedly into unrelated data.
|
||||
let limit = start.saturating_add(4096).min(data.len());
|
||||
let mut i = start;
|
||||
while i < limit && data[i] != 0 {
|
||||
i += 1;
|
||||
}
|
||||
String::from_utf8_lossy(&data[start..i]).into_owned()
|
||||
}
|
||||
|
||||
/// Read a PE section-name field: exactly 8 bytes, NOT necessarily
|
||||
/// NUL-terminated (a full-width name like `.textbss` has no NUL at all).
|
||||
/// Returns the name with trailing NULs stripped. Using `get_string_to_null`
|
||||
/// here would run past the field into the VirtualSize/VirtualAddress dwords.
|
||||
pub fn section_name(data: &[u8], offset: u32) -> String {
|
||||
let start = offset as usize;
|
||||
let Some(field) = data.get(start..start + 8) else {
|
||||
return String::new();
|
||||
};
|
||||
let end = field.iter().position(|&b| b == 0).unwrap_or(8);
|
||||
String::from_utf8_lossy(&field[..end]).into_owned()
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// PE32 (32-bit) helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/// PE32 shell-table locator. Walks the shell region (`info[6]`) for a dword
|
||||
/// equal to `info[6]` followed by a plausible shell size, returning the table
|
||||
/// base (`candidate = off - 0x88`) when `candidate+0x58` holds a valid pointer.
|
||||
pub fn find_tbl_pe32(data: &[u8], info: &[u32; 8]) -> Option<u32> {
|
||||
let shell = info[6];
|
||||
if (data.len() as u64) < 0x100 {
|
||||
return None;
|
||||
}
|
||||
let hi = (shell as u64)
|
||||
.saturating_add(0x3000)
|
||||
.min(data.len() as u64 - 0x100) as u32;
|
||||
let mut off = shell;
|
||||
while off < hi {
|
||||
if off as usize + 8 <= data.len() {
|
||||
let candidate = off.wrapping_sub(0x88);
|
||||
if candidate >= shell && get_u32(data, off) == info[6] {
|
||||
let shell_size_val = get_u32(data, off.wrapping_add(4));
|
||||
if shell_size_val > 0x1000 && shell_size_val < 0x100000 {
|
||||
let v58_off = candidate.wrapping_add(0x58);
|
||||
if (v58_off as usize + 4) <= data.len() {
|
||||
let v58 = get_u32(data, v58_off);
|
||||
if v58 > 0 && (v58 as usize) < data.len() {
|
||||
return Some(candidate);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
off = off.wrapping_add(4);
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Locate an LFSR-encrypted bytecode block (decrypt_data6 form) in a region.
|
||||
/// `start_off` is the byte offset to begin scanning at, `scan_backward`
|
||||
/// controls direction. Returns the relative offset of the block. Includes full
|
||||
/// opcode-walk validation of candidate blocks.
|
||||
pub fn find_lfsr_block(
|
||||
data: &[u8],
|
||||
base: u32,
|
||||
size: u32,
|
||||
start_off: u32,
|
||||
scan_backward: bool,
|
||||
) -> Option<u32> {
|
||||
if size < 96 {
|
||||
return None;
|
||||
}
|
||||
let mut ks = [0u8; 128];
|
||||
lfsr_keystream(&mut ks);
|
||||
let check = |scan_off: u32| -> bool {
|
||||
let abs_off = base.wrapping_add(scan_off) as usize;
|
||||
if abs_off + 96 > data.len() {
|
||||
return false;
|
||||
}
|
||||
let sz = data[abs_off + 95] as usize;
|
||||
if !(10..=95).contains(&sz) {
|
||||
return false;
|
||||
}
|
||||
let mut decoded = [0u8; 95];
|
||||
for bi in 0..sz {
|
||||
decoded[bi] = data[abs_off + bi] ^ ks[bi];
|
||||
}
|
||||
// Full bytecode validation (shared with the stage4/5 locator): every
|
||||
// opcode must decode with a valid ModR/M and the stream must REACH a
|
||||
// RET (0xC3) as an opcode. The previous check only required a 0xC3
|
||||
// byte *anywhere* in the window and accepted a walk that ran off the
|
||||
// end without hitting RET — a `0x04 0xC3` (ADD 0xC3) tail passed, so
|
||||
// coincidental LFSR-shaped garbage was accepted as a decryptor block.
|
||||
parse_bytecode_check(&decoded[..sz]).is_some()
|
||||
};
|
||||
if scan_backward {
|
||||
let hi = size - 96;
|
||||
if hi >= start_off {
|
||||
let mut scan_off = hi;
|
||||
loop {
|
||||
if check(scan_off) {
|
||||
return Some(scan_off);
|
||||
}
|
||||
if scan_off == start_off {
|
||||
break;
|
||||
}
|
||||
scan_off -= 1;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
let hi = size - 95;
|
||||
let mut scan_off = start_off;
|
||||
while scan_off < hi {
|
||||
if check(scan_off) {
|
||||
return Some(scan_off);
|
||||
}
|
||||
scan_off += 1;
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
/// Slots discovered in the eighthStage for the marker-less layout.
|
||||
pub struct EighthSlots {
|
||||
/// Absolute address of the file-data decryptor LFSR bytecode block. The
|
||||
/// fileCS chain pointer is derived downstream as `file_lfsr - 0x58`.
|
||||
pub file_lfsr: u32,
|
||||
/// Absolute address of the compressedInfo (ptr,size) table pointer slot.
|
||||
pub compressed_info_ptr: u32,
|
||||
}
|
||||
|
||||
/// Marker-independent eighthStage slot discovery (PE32+ branch).
|
||||
///
|
||||
/// Newer Crackproof builds (e.g. some native/managed DLLs) omit the
|
||||
/// `pm\0\0cm\0\0` and `00 00 00 40 01 00 00 00` markers that the older layout's
|
||||
/// walk3/walk4/walk5 slot derivation relies on. Instead this discovers the
|
||||
/// slots structurally:
|
||||
/// * Scan the eighthStage for every LFSR (decrypt_data6) bytecode block.
|
||||
/// * The file decryptor is the LFSR block whose `fileCS = lfsr - 0x58` holds
|
||||
/// a pointer sitting just past `info[3]` (smallest positive distance).
|
||||
/// * `compressedInfo` is the pointer slot whose 16-byte target, after a
|
||||
/// trial `decrypt_data5`, parses as a plausible (src,sSize,dst,dSize)
|
||||
/// descriptor.
|
||||
///
|
||||
/// Returns `None` if no plausible file LFSR is found. `eighth_start`/`eighth_dsz`
|
||||
/// bound the search region; `info3` is `info[3]`; `compress_data_offset` is
|
||||
/// `(!u32(file_data,0x1080)) + 0x1000`; `file_data_len` is the protected file
|
||||
/// length.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn discover_eighth_slots(
|
||||
data: &[u8],
|
||||
eighth_start: u32,
|
||||
eighth_dsz: u32,
|
||||
info3: u32,
|
||||
compress_data_offset: u32,
|
||||
file_data_len: u32,
|
||||
) -> Option<EighthSlots> {
|
||||
// Collect all LFSR candidates (forward scan).
|
||||
//
|
||||
// Advance by 1 after each hit, NOT by 96. A false-positive LFSR match can sit
|
||||
// just before the real file-decryptor block (observed on an il2cpp game
|
||||
// assembly build, 2026-07-13: junk at rel=0x31C1, real block at 0x3210).
|
||||
// Stepping by the LFSR body size then skips the real block and discovery
|
||||
// fails. Byte-stepping is cheap: eighthStage is only a few KB.
|
||||
let mut all_lfsrs: Vec<u32> = Vec::new();
|
||||
let mut scan_off: u32 = 0;
|
||||
while scan_off + 95 < eighth_dsz {
|
||||
match find_lfsr_block(data, eighth_start, eighth_dsz, scan_off, false) {
|
||||
Some(found) => {
|
||||
all_lfsrs.push(found);
|
||||
scan_off = found + 1;
|
||||
}
|
||||
None => break,
|
||||
}
|
||||
}
|
||||
|
||||
// Pick the file LFSR: prefer the candidate whose fileCS pointer sits the
|
||||
// smallest positive distance past info[3].
|
||||
let mut off_file_lfsr: Option<u32> = None;
|
||||
let mut best_dist: Option<u32> = None;
|
||||
for &lfsr_off in &all_lfsrs {
|
||||
if lfsr_off < 0x58 {
|
||||
continue;
|
||||
}
|
||||
let cs_off = lfsr_off - 0x58;
|
||||
let cs_val = get_u32(data, eighth_start.wrapping_add(cs_off));
|
||||
if !(0x1000 < cs_val && (cs_val as usize) < data.len()) {
|
||||
continue;
|
||||
}
|
||||
if cs_val < info3 {
|
||||
continue;
|
||||
}
|
||||
let dist = cs_val - info3;
|
||||
if best_dist.is_none_or(|b| dist < b) {
|
||||
best_dist = Some(dist);
|
||||
off_file_lfsr = Some(lfsr_off);
|
||||
}
|
||||
}
|
||||
// Fallback: last LFSR with any in-image fileCS pointer.
|
||||
if off_file_lfsr.is_none() {
|
||||
for &lfsr_off in all_lfsrs.iter().rev() {
|
||||
if lfsr_off < 0x58 {
|
||||
continue;
|
||||
}
|
||||
let cs_val = get_u32(data, eighth_start.wrapping_add(lfsr_off - 0x58));
|
||||
if 0x1000 < cs_val && (cs_val as usize) < data.len() {
|
||||
off_file_lfsr = Some(lfsr_off);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
let off_file_lfsr = off_file_lfsr?;
|
||||
let off_file_cs = off_file_lfsr - 0x58;
|
||||
|
||||
// Trial-decrypt to find compressedInfo: the pointer slot in the data area
|
||||
// (between fileCS region start and the LFSR) whose target parses as a valid
|
||||
// (src,sSize,dst,dSize) descriptor after a transient decrypt_data5.
|
||||
let scan_from = off_file_lfsr.saturating_sub(0x400);
|
||||
let mut off_compressed_info: Option<u32> = None;
|
||||
let mut doff = scan_from;
|
||||
while doff < off_file_lfsr {
|
||||
if doff == off_file_cs {
|
||||
doff += 4;
|
||||
continue;
|
||||
}
|
||||
let ptr_val = get_u32(data, eighth_start.wrapping_add(doff));
|
||||
if !(0x1000 < ptr_val && (ptr_val as usize) < data.len().saturating_sub(16)) {
|
||||
doff += 4;
|
||||
continue;
|
||||
}
|
||||
// Predict decrypt_data5(ptr_val, 16) without mutating: each dword is
|
||||
// position-keyed and independent, so trial_decrypt5_u32 per dword.
|
||||
let src2 = trial_decrypt5_u32(data, ptr_val);
|
||||
let s_sz2 = trial_decrypt5_u32(data, ptr_val + 4);
|
||||
let dst2 = trial_decrypt5_u32(data, ptr_val + 8);
|
||||
let d_sz2 = trial_decrypt5_u32(data, ptr_val + 12);
|
||||
let src_file_off = src2.wrapping_add(compress_data_offset);
|
||||
let valid = s_sz2 > 0
|
||||
&& s_sz2 < 0x200000
|
||||
&& (src_file_off as u64 + s_sz2 as u64) <= file_data_len as u64
|
||||
&& dst2 >= 0x1000
|
||||
&& (dst2 as u64 + d_sz2 as u64) <= data.len() as u64
|
||||
&& d_sz2 >= s_sz2
|
||||
&& d_sz2 < 0x200000;
|
||||
if valid {
|
||||
off_compressed_info = Some(doff);
|
||||
break;
|
||||
}
|
||||
doff += 4;
|
||||
}
|
||||
let off_compressed_info = off_compressed_info?;
|
||||
|
||||
Some(EighthSlots {
|
||||
file_lfsr: eighth_start.wrapping_add(off_file_lfsr),
|
||||
compressed_info_ptr: eighth_start.wrapping_add(off_compressed_info),
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Task 4.1 regression: build a synthetic buffer whose valid bytecode block
|
||||
/// sits PAST `len` but within `len*2`. Assert that the smaller window misses
|
||||
/// it and the doubled window finds it.
|
||||
#[test]
|
||||
fn bytecode_locate_double_window_retry() {
|
||||
// We place the block at offset (base + len + 16) which is inside
|
||||
// the len*2 window but outside the len window.
|
||||
let base: u32 = 0;
|
||||
let len: u32 = 256;
|
||||
// Block sits at base + len + 16 = 272, aligned to 16.
|
||||
let block_pos: usize = (base + len + 16) as usize; // 272
|
||||
|
||||
// The buffer must be large enough for the block (block_pos + 96 bytes).
|
||||
let buf_len = block_pos + 256;
|
||||
let mut buf = vec![0u8; buf_len];
|
||||
|
||||
// Build a valid plaintext op stream:
|
||||
// [4, 0, 4, 0, 4, 0, 4, 0, 195] (4 ADD-AL ops then RET)
|
||||
// Padded to 10 bytes total; count >= 8.
|
||||
let count: usize = 10;
|
||||
let mut plain = [0u8; 256];
|
||||
plain[0] = 4;
|
||||
plain[1] = 0;
|
||||
plain[2] = 4;
|
||||
plain[3] = 0;
|
||||
plain[4] = 4;
|
||||
plain[5] = 0;
|
||||
plain[6] = 4;
|
||||
plain[7] = 0;
|
||||
plain[8] = 195; // ret
|
||||
|
||||
// Compute the LFSR keystream and XOR the first `count` bytes to get the
|
||||
// encrypted representation that the scanner would decrypt back.
|
||||
let mut ks = [0u8; 256];
|
||||
lfsr_keystream(&mut ks);
|
||||
for i in 0..count {
|
||||
buf[block_pos + i] = plain[i] ^ ks[i];
|
||||
}
|
||||
// Raw count byte at block_pos+95 (outside the XOR range since count=10 < 95).
|
||||
buf[block_pos + 95] = count as u8;
|
||||
|
||||
// Verify our construction: find_bytecode_offset with len should NOT find it.
|
||||
assert_eq!(
|
||||
find_bytecode_offset(&buf, base, len),
|
||||
None,
|
||||
"smaller window should not find the block"
|
||||
);
|
||||
|
||||
// The doubled window should find it at block_pos.
|
||||
assert_eq!(
|
||||
find_bytecode_offset(&buf, base, len.saturating_mul(2)),
|
||||
Some(block_pos as u32),
|
||||
"doubled window should locate the block"
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,481 @@
|
||||
//! PE import reconstruction and memory-image compaction.
|
||||
|
||||
use senbei_crypto::primitives::{get_u16, get_u32, write_u16, write_u32};
|
||||
|
||||
use super::super::MAX_IMAGE_SIZE;
|
||||
|
||||
/// Read a NUL-terminated byte string starting at `off`, bounded to 512 bytes.
|
||||
/// Returns the raw bytes up to the terminator (excluding it).
|
||||
fn read_cstr_bounded(data: &[u8], off: u32) -> Vec<u8> {
|
||||
let start = off as usize;
|
||||
if start >= data.len() {
|
||||
return Vec::new();
|
||||
}
|
||||
let limit = (start + 512).min(data.len());
|
||||
let mut end = start;
|
||||
while end < limit && data[end] != 0 {
|
||||
end += 1;
|
||||
}
|
||||
data[start..end].to_vec()
|
||||
}
|
||||
|
||||
fn align_up_u32(value: u32, alignment: u32) -> u32 {
|
||||
((value.wrapping_add(alignment - 1)) / alignment).wrapping_mul(alignment)
|
||||
}
|
||||
|
||||
fn align_up_u64(value: u64, alignment: u64) -> u64 {
|
||||
value.div_ceil(alignment) * alignment
|
||||
}
|
||||
|
||||
#[derive(Clone)]
|
||||
enum ImportFunc {
|
||||
Ordinal(u32),
|
||||
Name(u16, Vec<u8>),
|
||||
}
|
||||
|
||||
struct ImportDesc {
|
||||
time_date: u32,
|
||||
fwd_chain: u32,
|
||||
dll_name: Vec<u8>,
|
||||
iat_rva: u32,
|
||||
functions: Vec<ImportFunc>,
|
||||
}
|
||||
|
||||
/// Return true when PE32 imports already sit in the original `.idata` layout
|
||||
/// (so no relocation to `.kmiat` is needed). May write the IAT data directory
|
||||
/// (pe+0xD8).
|
||||
pub fn pe32_imports_already_match_idata_layout(data: &mut [u8], pe_header: u32) -> bool {
|
||||
let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32;
|
||||
let sec_table = pe_header.wrapping_add(24).wrapping_add(opt_hdr_size);
|
||||
let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32;
|
||||
let import_rva = get_u32(data, pe_header.wrapping_add(0x80));
|
||||
let import_size = get_u32(data, pe_header.wrapping_add(0x84));
|
||||
let len = data.len() as u32;
|
||||
if !(import_rva > 0 && import_size > 0) {
|
||||
return false;
|
||||
}
|
||||
for idx in 0..num_sections {
|
||||
let sec_off = sec_table.wrapping_add(idx * 40);
|
||||
if (sec_off as usize + 40) > data.len() {
|
||||
return false;
|
||||
}
|
||||
if &data[sec_off as usize..sec_off as usize + 6] != b".idata" {
|
||||
continue;
|
||||
}
|
||||
let sec_va = get_u32(data, sec_off.wrapping_add(12));
|
||||
let sec_size =
|
||||
get_u32(data, sec_off.wrapping_add(8)).max(get_u32(data, sec_off.wrapping_add(16)));
|
||||
let sec_end = sec_va.wrapping_add(sec_size);
|
||||
if !(sec_va <= import_rva
|
||||
&& import_rva < sec_end
|
||||
&& import_rva.wrapping_add(import_size) <= sec_end)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let first_oft = get_u32(data, import_rva);
|
||||
let first_name = get_u32(data, import_rva.wrapping_add(12));
|
||||
let first_iat = get_u32(data, import_rva.wrapping_add(16));
|
||||
if !(sec_va <= first_oft
|
||||
&& first_oft < sec_end
|
||||
&& sec_va <= first_iat
|
||||
&& first_iat < sec_end)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
if !(0x1000 < first_name && first_name < len) {
|
||||
return false;
|
||||
}
|
||||
let dll_name = read_cstr_bounded(data, first_name);
|
||||
let lower: Vec<u8> = dll_name.iter().map(|b| b.to_ascii_lowercase()).collect();
|
||||
if !lower.ends_with(b".dll") {
|
||||
return false;
|
||||
}
|
||||
let mut iat_min = first_iat;
|
||||
let mut iat_max = first_iat;
|
||||
let mut idt_pos = import_rva;
|
||||
while idt_pos.wrapping_add(20) <= len {
|
||||
let oft_rva = get_u32(data, idt_pos);
|
||||
let name_rva = get_u32(data, idt_pos.wrapping_add(12));
|
||||
let iat_rva = get_u32(data, idt_pos.wrapping_add(16));
|
||||
if oft_rva == 0 && name_rva == 0 && iat_rva == 0 {
|
||||
break;
|
||||
}
|
||||
if !(sec_va <= oft_rva && oft_rva < sec_end && sec_va <= iat_rva && iat_rva < sec_end) {
|
||||
return false;
|
||||
}
|
||||
let mut thunk = iat_rva;
|
||||
while thunk.wrapping_add(4) <= sec_end {
|
||||
let tv = get_u32(data, thunk);
|
||||
thunk = thunk.wrapping_add(4);
|
||||
if tv == 0 {
|
||||
break;
|
||||
}
|
||||
}
|
||||
iat_min = iat_min.min(iat_rva);
|
||||
iat_max = iat_max.max(thunk);
|
||||
idt_pos = idt_pos.wrapping_add(20);
|
||||
}
|
||||
if iat_max > iat_min {
|
||||
write_u32(data, pe_header.wrapping_add(0xD8), iat_min);
|
||||
write_u32(data, pe_header.wrapping_add(0xDC), iat_max - iat_min);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Rebuild PE32 import metadata (descriptors, lookup tables, names) into the
|
||||
/// last section as `.kmiat`, leaving the loader-written IAT in place. Mutates
|
||||
/// `data` (may grow it).
|
||||
pub fn move_pe32_imports_to_kmiat(data: &mut Vec<u8>, pe_header: u32) {
|
||||
const SECTION_SIZE: u32 = 0x7000;
|
||||
let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32;
|
||||
let opt_hdr = pe_header.wrapping_add(24);
|
||||
let sec_table = opt_hdr.wrapping_add(opt_hdr_size);
|
||||
let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32;
|
||||
if num_sections == 0 {
|
||||
return;
|
||||
}
|
||||
let import_rva = get_u32(data, pe_header.wrapping_add(0x80));
|
||||
let import_size = get_u32(data, pe_header.wrapping_add(0x84));
|
||||
let len = data.len() as u32;
|
||||
if !(0x1000 < import_rva && import_rva < len && import_size > 0 && import_size < SECTION_SIZE) {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut descriptors: Vec<ImportDesc> = Vec::new();
|
||||
let mut idt_pos = import_rva;
|
||||
while idt_pos.wrapping_add(20) <= len {
|
||||
let oft_rva = get_u32(data, idt_pos);
|
||||
let time_date = get_u32(data, idt_pos.wrapping_add(4));
|
||||
let fwd_chain = get_u32(data, idt_pos.wrapping_add(8));
|
||||
let name_rva = get_u32(data, idt_pos.wrapping_add(12));
|
||||
let iat_rva = get_u32(data, idt_pos.wrapping_add(16));
|
||||
if oft_rva == 0 && name_rva == 0 && iat_rva == 0 {
|
||||
break;
|
||||
}
|
||||
if !(0x1000 < name_rva && name_rva < len) {
|
||||
break;
|
||||
}
|
||||
let dll_name = read_cstr_bounded(data, name_rva);
|
||||
let thunk_rva = if 0x1000 < oft_rva && oft_rva < len {
|
||||
oft_rva
|
||||
} else {
|
||||
iat_rva
|
||||
};
|
||||
let mut functions: Vec<ImportFunc> = Vec::new();
|
||||
let mut thunk_pos = thunk_rva;
|
||||
while 0x1000 < thunk_pos.wrapping_add(4) && thunk_pos.wrapping_add(4) <= len {
|
||||
let thunk_val = get_u32(data, thunk_pos);
|
||||
if thunk_val == 0 {
|
||||
break;
|
||||
}
|
||||
if thunk_val & 0x8000_0000 != 0 {
|
||||
functions.push(ImportFunc::Ordinal(thunk_val & 0xFFFF));
|
||||
} else {
|
||||
let hint = if thunk_val.wrapping_add(2) <= len {
|
||||
get_u16(data, thunk_val)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let func_name = if thunk_val.wrapping_add(2) < len {
|
||||
read_cstr_bounded(data, thunk_val.wrapping_add(2))
|
||||
} else {
|
||||
Vec::new()
|
||||
};
|
||||
functions.push(ImportFunc::Name(hint, func_name));
|
||||
}
|
||||
thunk_pos = thunk_pos.wrapping_add(4);
|
||||
}
|
||||
descriptors.push(ImportDesc {
|
||||
time_date,
|
||||
fwd_chain,
|
||||
dll_name,
|
||||
iat_rva,
|
||||
functions,
|
||||
});
|
||||
idt_pos = idt_pos.wrapping_add(20);
|
||||
}
|
||||
if descriptors.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
for desc in &mut descriptors {
|
||||
let lower: Vec<u8> = desc
|
||||
.dll_name
|
||||
.iter()
|
||||
.map(|b| b.to_ascii_lowercase())
|
||||
.collect();
|
||||
if lower.starts_with(b"api-ms-win-crt-") {
|
||||
desc.dll_name = b"ucrtbase.dll".to_vec();
|
||||
} else {
|
||||
desc.dll_name = lower;
|
||||
}
|
||||
}
|
||||
descriptors.sort_by_key(|d| d.iat_rva);
|
||||
|
||||
let last_sec = sec_table.wrapping_add((num_sections - 1) * 40);
|
||||
let kmiat_rva = get_u32(data, last_sec.wrapping_add(12));
|
||||
// A zero last-section VA means a corrupt section table: building .kmiat at
|
||||
// RVA 0 would zero the DOS/PE headers and emit a structurally broken image
|
||||
// with no error. Bail and keep the original import table.
|
||||
if kmiat_rva == 0 {
|
||||
return;
|
||||
}
|
||||
// Grow the image when .kmiat overruns it, but cap the growth: a corrupt VA
|
||||
// could otherwise request a multi-gigabyte allocation, which aborts the
|
||||
// process (uncatchable). Use u64 math so a near-u32::MAX VA cannot wrap the
|
||||
// end calculation the way the previous wrapping/plain-add mix could.
|
||||
let kmiat_end = kmiat_rva as u64 + SECTION_SIZE as u64;
|
||||
if kmiat_end > MAX_IMAGE_SIZE {
|
||||
return;
|
||||
}
|
||||
if kmiat_end > data.len() as u64 {
|
||||
data.resize(kmiat_end as usize, 0);
|
||||
}
|
||||
// Zero the .kmiat region.
|
||||
for b in &mut data[kmiat_rva as usize..kmiat_end as usize] {
|
||||
*b = 0;
|
||||
}
|
||||
|
||||
let idt_size = (descriptors.len() as u32 + 1) * 20;
|
||||
let oft_start = kmiat_rva;
|
||||
let mut idt_rva = oft_start;
|
||||
for desc in &descriptors {
|
||||
idt_rva = idt_rva.wrapping_add((desc.functions.len() as u32 + 1) * 4);
|
||||
}
|
||||
idt_rva = align_up_u32(idt_rva.wrapping_add(0x2C), 4);
|
||||
|
||||
// Size check: compute the final name_pos and bail if it overruns .kmiat.
|
||||
let mut name_pos_check = idt_rva.wrapping_add(idt_size);
|
||||
for desc in &descriptors {
|
||||
name_pos_check = name_pos_check.wrapping_add(desc.dll_name.len() as u32 + 1);
|
||||
for func in &desc.functions {
|
||||
if let ImportFunc::Name(_, fname) = func {
|
||||
name_pos_check = name_pos_check.wrapping_add(2 + fname.len() as u32 + 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
if name_pos_check > kmiat_rva.wrapping_add(SECTION_SIZE) {
|
||||
// Section too small; keep existing import table untouched.
|
||||
return;
|
||||
}
|
||||
|
||||
let mut oft_pos = oft_start;
|
||||
let mut name_pos = idt_rva.wrapping_add(idt_size);
|
||||
for (idx, desc) in descriptors.iter().enumerate() {
|
||||
let idt_entry = idt_rva.wrapping_add(idx as u32 * 20);
|
||||
let current_oft = oft_pos;
|
||||
write_u32(data, idt_entry, current_oft);
|
||||
write_u32(data, idt_entry.wrapping_add(4), desc.time_date);
|
||||
write_u32(data, idt_entry.wrapping_add(8), desc.fwd_chain);
|
||||
let dll_name_pos = name_pos;
|
||||
write_u32(data, idt_entry.wrapping_add(12), dll_name_pos);
|
||||
write_u32(data, idt_entry.wrapping_add(16), desc.iat_rva);
|
||||
|
||||
let dnp = dll_name_pos as usize;
|
||||
data[dnp..dnp + desc.dll_name.len()].copy_from_slice(&desc.dll_name);
|
||||
data[dnp + desc.dll_name.len()] = 0;
|
||||
name_pos = name_pos.wrapping_add(desc.dll_name.len() as u32 + 1);
|
||||
|
||||
for func in &desc.functions {
|
||||
match func {
|
||||
ImportFunc::Ordinal(ord) => {
|
||||
write_u32(data, oft_pos, 0x8000_0000 | ord);
|
||||
}
|
||||
ImportFunc::Name(hint, fname) => {
|
||||
let hint_name_rva = name_pos;
|
||||
write_u32(data, oft_pos, hint_name_rva);
|
||||
write_u16(data, hint_name_rva, *hint as u32);
|
||||
let fp = (hint_name_rva + 2) as usize;
|
||||
data[fp..fp + fname.len()].copy_from_slice(fname);
|
||||
data[fp + fname.len()] = 0;
|
||||
name_pos = name_pos.wrapping_add(2 + fname.len() as u32 + 1);
|
||||
}
|
||||
}
|
||||
oft_pos = oft_pos.wrapping_add(4);
|
||||
}
|
||||
write_u32(data, oft_pos, 0);
|
||||
oft_pos = oft_pos.wrapping_add(4);
|
||||
}
|
||||
// Null-terminator IDT entry (20 zero bytes) after the last descriptor.
|
||||
let term = idt_rva.wrapping_add(descriptors.len() as u32 * 20) as usize;
|
||||
for b in &mut data[term..term + 20] {
|
||||
*b = 0;
|
||||
}
|
||||
|
||||
let ls = last_sec as usize;
|
||||
data[ls..ls + 8].copy_from_slice(b".kmiat\x00\x00");
|
||||
write_u32(data, last_sec.wrapping_add(8), SECTION_SIZE);
|
||||
write_u32(data, last_sec.wrapping_add(16), SECTION_SIZE);
|
||||
write_u32(data, last_sec.wrapping_add(36), 0xE000_0060);
|
||||
write_u32(data, pe_header.wrapping_add(0x80), idt_rva);
|
||||
write_u32(data, pe_header.wrapping_add(0x84), idt_size);
|
||||
write_u32(
|
||||
data,
|
||||
pe_header.wrapping_add(80),
|
||||
kmiat_rva.wrapping_add(SECTION_SIZE),
|
||||
);
|
||||
}
|
||||
|
||||
/// Convert the unpacked RVA-addressed image back to a compact PE file layout
|
||||
/// (headers at 0x400, sections packed consecutively, FileAlignment 0x200).
|
||||
/// Returns `None` if the accumulated output size wraps or exceeds
|
||||
/// [`MAX_IMAGE_SIZE`]: the final allocation is sized from header-derived
|
||||
/// section data, and an uncapped `vec![0; n]` from a corrupt header would abort
|
||||
/// the process (which `catch_unpack` cannot trap).
|
||||
pub fn compact_memory_image_to_pe(data: &[u8], pe_header: u32) -> Option<Vec<u8>> {
|
||||
const FILE_ALIGNMENT: u32 = 0x200;
|
||||
const HEADER_SIZE: u32 = 0x400;
|
||||
let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32;
|
||||
let opt_hdr = pe_header.wrapping_add(24);
|
||||
let sec_table = opt_hdr.wrapping_add(opt_hdr_size);
|
||||
let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32;
|
||||
|
||||
struct SecLayout {
|
||||
sec_off: u32,
|
||||
va: u32,
|
||||
vsize: u32,
|
||||
raw_ptr: u32,
|
||||
raw_size: u32,
|
||||
}
|
||||
|
||||
let mut raw_cursor: u64 = HEADER_SIZE as u64;
|
||||
let mut raw_layout: Vec<SecLayout> = Vec::new();
|
||||
for idx in 0..num_sections {
|
||||
let sec_off = sec_table.wrapping_add(idx * 40);
|
||||
let vsize = get_u32(data, sec_off.wrapping_add(8));
|
||||
let va = get_u32(data, sec_off.wrapping_add(12));
|
||||
let sd_start = va as usize;
|
||||
let sd_end = if (va.wrapping_add(vsize) as usize) <= data.len() {
|
||||
va.wrapping_add(vsize) as usize
|
||||
} else {
|
||||
data.len()
|
||||
};
|
||||
let section_data: &[u8] = if sd_start <= sd_end && sd_start <= data.len() {
|
||||
&data[sd_start..sd_end]
|
||||
} else {
|
||||
&[]
|
||||
};
|
||||
|
||||
let mut last_nonzero: i64 = -1;
|
||||
for pos in (0..section_data.len()).rev() {
|
||||
if section_data[pos] != 0 {
|
||||
last_nonzero = pos as i64;
|
||||
break;
|
||||
}
|
||||
}
|
||||
let meaningful = if last_nonzero >= 0 {
|
||||
(last_nonzero + 1) as u32
|
||||
} else {
|
||||
0
|
||||
};
|
||||
let mut raw_size = if meaningful != 0 {
|
||||
align_up_u32(meaningful, FILE_ALIGNMENT)
|
||||
} else {
|
||||
0
|
||||
};
|
||||
if vsize != 0 && raw_size == 0 {
|
||||
raw_size = FILE_ALIGNMENT;
|
||||
}
|
||||
raw_size = raw_size.min(align_up_u32(section_data.len() as u32, FILE_ALIGNMENT));
|
||||
|
||||
let raw_ptr = if raw_size != 0 { raw_cursor as u32 } else { 0 };
|
||||
raw_layout.push(SecLayout {
|
||||
sec_off,
|
||||
va,
|
||||
vsize,
|
||||
raw_ptr,
|
||||
raw_size,
|
||||
});
|
||||
if raw_size != 0 {
|
||||
// Accumulate in u64 and cap: section sizes are header-derived, and
|
||||
// a corrupt table could otherwise wrap raw_cursor (small alloc,
|
||||
// huge recorded raw_ptrs → OOB panic) or request an abort-sized
|
||||
// allocation.
|
||||
raw_cursor = align_up_u64(raw_cursor + raw_size as u64, FILE_ALIGNMENT as u64);
|
||||
if raw_cursor > MAX_IMAGE_SIZE {
|
||||
return None;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let mut compact = vec![0u8; raw_cursor as usize];
|
||||
let hdr_copy = (HEADER_SIZE as usize).min(data.len());
|
||||
compact[..hdr_copy].copy_from_slice(&data[..hdr_copy]);
|
||||
write_u32(&mut compact, opt_hdr.wrapping_add(36), FILE_ALIGNMENT);
|
||||
write_u32(&mut compact, opt_hdr.wrapping_add(60), HEADER_SIZE);
|
||||
|
||||
for sl in &raw_layout {
|
||||
write_u32(&mut compact, sl.sec_off.wrapping_add(16), sl.raw_size);
|
||||
write_u32(&mut compact, sl.sec_off.wrapping_add(20), sl.raw_ptr);
|
||||
if sl.raw_size != 0 {
|
||||
let sd_start = sl.va as usize;
|
||||
let sd_end = if (sl.va.wrapping_add(sl.vsize) as usize) <= data.len() {
|
||||
sl.va.wrapping_add(sl.vsize) as usize
|
||||
} else {
|
||||
data.len()
|
||||
};
|
||||
let section_data: &[u8] = if sd_start <= sd_end {
|
||||
&data[sd_start..sd_end]
|
||||
} else {
|
||||
&[]
|
||||
};
|
||||
let copy_size = (sl.raw_size as usize).min(section_data.len());
|
||||
let rp = sl.raw_ptr as usize;
|
||||
compact[rp..rp + copy_size].copy_from_slice(§ion_data[..copy_size]);
|
||||
}
|
||||
}
|
||||
Some(compact)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Review regression: a zero last-section VA (corrupt section table) must
|
||||
/// bail instead of building .kmiat at RVA 0 — the old code zeroed
|
||||
/// `[0, 0x7000)`, wiping the DOS/PE headers, and returned the broken image
|
||||
/// as a success. A near-2 GiB VA must likewise refuse to grow the image
|
||||
/// past [`MAX_IMAGE_SIZE`].
|
||||
#[test]
|
||||
fn kmiat_bogus_section_va_bails_without_wiping_headers() {
|
||||
for last_sec_va in [0u32, 0x5000_0000] {
|
||||
let pe: u32 = 0x80;
|
||||
let mut data = vec![0xAAu8; 0x8000];
|
||||
// COFF header: 1 section, optional header size 0xE0 (PE32).
|
||||
write_u16(&mut data, pe + 6, 1);
|
||||
write_u16(&mut data, pe + 20, 0xE0);
|
||||
// Import directory at pe+0x80: one descriptor + null terminator.
|
||||
write_u32(&mut data, pe + 0x80, 0x1100);
|
||||
write_u32(&mut data, pe + 0x84, 0x28);
|
||||
write_u32(&mut data, 0x1100, 0x1200); // OFT rva
|
||||
write_u32(&mut data, 0x1100 + 12, 0x1300); // name rva
|
||||
write_u32(&mut data, 0x1100 + 16, 0x1400); // IAT rva
|
||||
for b in &mut data[0x1100 + 20..0x1100 + 40] {
|
||||
*b = 0; // null terminator descriptor
|
||||
}
|
||||
data[0x1300..0x1300 + 13].copy_from_slice(b"KERNEL32.dll\0");
|
||||
write_u32(&mut data, 0x1200, 0x1500); // thunk -> hint/name
|
||||
write_u32(&mut data, 0x1204, 0); // thunk terminator
|
||||
data[0x1500..0x1502].copy_from_slice(&0u16.to_le_bytes());
|
||||
data[0x1502..0x1502 + 12].copy_from_slice(b"ExitProcess\0");
|
||||
// Section table at pe+24+0xE0 = 0x178; VA field at +12.
|
||||
write_u32(&mut data, 0x178 + 12, last_sec_va);
|
||||
|
||||
let head_before: Vec<u8> = data[..0x400].to_vec();
|
||||
let len_before = data.len();
|
||||
move_pe32_imports_to_kmiat(&mut data, pe);
|
||||
assert_eq!(
|
||||
data.len(),
|
||||
len_before,
|
||||
"VA 0x{last_sec_va:08X}: image must not grow"
|
||||
);
|
||||
assert_eq!(
|
||||
&data[..0x400],
|
||||
&head_before[..],
|
||||
"VA 0x{last_sec_va:08X}: headers must be untouched"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,30 +1,157 @@
|
||||
//! Pure, panic-free Crackproof unpacker core. No file I/O lives here.
|
||||
//! PE detection, unpacking, and structural validation.
|
||||
|
||||
mod bytecode;
|
||||
mod crc32;
|
||||
pub mod dll;
|
||||
mod error;
|
||||
pub mod exe;
|
||||
pub mod integrity;
|
||||
mod layout;
|
||||
pub(crate) mod parallel;
|
||||
pub(crate) mod primitives;
|
||||
mod tables;
|
||||
|
||||
use senbei_crypto::primitives;
|
||||
use std::cell::RefCell;
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
pub use dll::{unpack_dll, unpack_dll_v};
|
||||
pub use exe::{UnpackError, unpack as unpack_exe, unpack_v as unpack_exe_v};
|
||||
pub use error::*;
|
||||
pub use exe::{unpack as unpack_exe, unpack_v as unpack_exe_v};
|
||||
pub use integrity::{IntegrityReport, check as check_integrity};
|
||||
pub use parallel::thread_cap;
|
||||
|
||||
/// Maximum plausible PE `SizeOfImage` we are willing to allocate a zero buffer
|
||||
/// for. Guards against a corrupt/crafted header requesting a multi-gigabyte
|
||||
/// (or, as a sign-extended negative `i32`, multi-exabyte) allocation, which
|
||||
/// would abort the process — an abort that `catch_unpack` below cannot trap.
|
||||
/// Real protected binaries are far below this.
|
||||
pub(crate) const MAX_IMAGE_SIZE: u64 = 1 << 30; // 1 GiB
|
||||
pub(crate) const MAX_IMAGE_SIZE: u64 = senbei_crypto::MAX_IMAGE_SIZE;
|
||||
|
||||
#[derive(Clone)]
|
||||
pub(crate) struct PanicCapture(Arc<Mutex<Option<PanicDetails>>>);
|
||||
|
||||
#[derive(Clone)]
|
||||
struct PanicDetails {
|
||||
message: String,
|
||||
file: String,
|
||||
line: u32,
|
||||
column: u32,
|
||||
}
|
||||
|
||||
thread_local! {
|
||||
static ACTIVE_PANIC_CAPTURE: RefCell<Option<PanicCapture>> = const { RefCell::new(None) };
|
||||
}
|
||||
|
||||
struct PanicCaptureGuard(Option<PanicCapture>);
|
||||
|
||||
impl Drop for PanicCaptureGuard {
|
||||
fn drop(&mut self) {
|
||||
ACTIVE_PANIC_CAPTURE.with(|slot| {
|
||||
slot.replace(self.0.take());
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
impl PanicCapture {
|
||||
fn new() -> Self {
|
||||
Self(Arc::new(Mutex::new(None)))
|
||||
}
|
||||
|
||||
fn record(&self, info: &std::panic::PanicHookInfo<'_>) {
|
||||
let location = info.location();
|
||||
let details = PanicDetails {
|
||||
message: panic_message(info.payload()),
|
||||
file: location
|
||||
.map(|value| value.file().to_owned())
|
||||
.unwrap_or_else(|| "<unknown>".to_owned()),
|
||||
line: location.map_or(0, std::panic::Location::line),
|
||||
column: location.map_or(0, std::panic::Location::column),
|
||||
};
|
||||
let mut captured = self
|
||||
.0
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
if captured.is_none() {
|
||||
*captured = Some(details);
|
||||
}
|
||||
}
|
||||
|
||||
fn into_error(self, payload: &(dyn std::any::Any + Send)) -> UnpackError {
|
||||
let details = self
|
||||
.0
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner)
|
||||
.clone()
|
||||
.unwrap_or_else(|| PanicDetails {
|
||||
message: panic_message(payload),
|
||||
file: "<unknown>".to_owned(),
|
||||
line: 0,
|
||||
column: 0,
|
||||
});
|
||||
UnpackError::InternalPanic {
|
||||
message: details.message,
|
||||
file: details.file,
|
||||
line: details.line,
|
||||
column: details.column,
|
||||
}
|
||||
}
|
||||
|
||||
fn merge_from(&self, other: &Self) {
|
||||
let details = other
|
||||
.0
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner)
|
||||
.clone();
|
||||
let Some(details) = details else { return };
|
||||
let mut captured = self
|
||||
.0
|
||||
.lock()
|
||||
.unwrap_or_else(std::sync::PoisonError::into_inner);
|
||||
if captured.is_none() {
|
||||
*captured = Some(details);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn panic_message(payload: &(dyn std::any::Any + Send)) -> String {
|
||||
if let Some(message) = payload.downcast_ref::<&str>() {
|
||||
(*message).to_owned()
|
||||
} else if let Some(message) = payload.downcast_ref::<String>() {
|
||||
message.clone()
|
||||
} else {
|
||||
"non-string panic payload".to_owned()
|
||||
}
|
||||
}
|
||||
|
||||
fn install_panic_capture_hook() {
|
||||
static INSTALL: std::sync::Once = std::sync::Once::new();
|
||||
INSTALL.call_once(|| {
|
||||
let previous = std::panic::take_hook();
|
||||
std::panic::set_hook(Box::new(move |info| {
|
||||
let capture = ACTIVE_PANIC_CAPTURE
|
||||
.try_with(|slot| slot.borrow().clone())
|
||||
.ok()
|
||||
.flatten();
|
||||
if let Some(capture) = capture {
|
||||
capture.record(info);
|
||||
} else {
|
||||
previous(info);
|
||||
}
|
||||
}));
|
||||
});
|
||||
}
|
||||
|
||||
pub(crate) fn current_panic_capture() -> Option<PanicCapture> {
|
||||
ACTIVE_PANIC_CAPTURE.with(|slot| slot.borrow().clone())
|
||||
}
|
||||
|
||||
pub(crate) fn with_panic_capture<R>(capture: Option<PanicCapture>, f: impl FnOnce() -> R) -> R {
|
||||
let previous = ACTIVE_PANIC_CAPTURE.with(|slot| slot.replace(capture));
|
||||
let _guard = PanicCaptureGuard(previous);
|
||||
f()
|
||||
}
|
||||
|
||||
/// Run an unpack pipeline, converting any internal panic into a clean
|
||||
/// [`UnpackError::Corrupt`] so the public API stays panic-free on any input
|
||||
/// (truncated/garbled files chase offsets out of bounds). The default panic
|
||||
/// hook is suppressed transiently so a trapped panic does not spill a
|
||||
/// backtrace to stderr.
|
||||
/// [`UnpackError::InternalPanic`] so the public API stays panic-free on any input
|
||||
/// (truncated/garbled files chase offsets out of bounds). The panic location and
|
||||
/// payload are captured for diagnostics without printing a backtrace to stderr.
|
||||
///
|
||||
/// Note: allocation *failures* abort the process and are NOT caught here; size
|
||||
/// requests are bounds-checked against [`MAX_IMAGE_SIZE`] before allocating.
|
||||
@@ -32,17 +159,15 @@ pub(crate) fn catch_unpack<F>(f: F) -> Result<Vec<u8>, UnpackError>
|
||||
where
|
||||
F: FnOnce() -> Result<Vec<u8>, UnpackError>,
|
||||
{
|
||||
// Hook suppression is skipped on wasm: the prebuilt std cannot unwind
|
||||
// there, so a panic traps immediately — and the suppressed hook would
|
||||
// hide the panic message, leaving a bare `unreachable` with no clue.
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
let prev = std::panic::take_hook();
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
std::panic::set_hook(Box::new(|_| {}));
|
||||
let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(f));
|
||||
#[cfg(not(target_arch = "wasm32"))]
|
||||
std::panic::set_hook(prev);
|
||||
r.unwrap_or(Err(UnpackError::Corrupt))
|
||||
install_panic_capture_hook();
|
||||
let capture = PanicCapture::new();
|
||||
let r = with_panic_capture(Some(capture.clone()), || {
|
||||
std::panic::catch_unwind(std::panic::AssertUnwindSafe(f))
|
||||
});
|
||||
match r {
|
||||
Ok(result) => result,
|
||||
Err(payload) => Err(capture.into_error(payload.as_ref())),
|
||||
}
|
||||
}
|
||||
|
||||
/// Crackproof header magic stored in `keys[1]`/`info[1]`.
|
||||
@@ -80,11 +205,8 @@ fn key_table(input: &[u8]) -> Option<[u32; 8]> {
|
||||
if input.len() < 4128 {
|
||||
return None;
|
||||
}
|
||||
// Validate PE signature. `checked_add`, not `+`: `usize` is 32-bit on
|
||||
// wasm32, where an `e_lfanew` of 0xFFFF_FFFC..=0xFFFF_FFFF wraps the bound
|
||||
// check, and the slice below then panics with start > end. `detect` runs on
|
||||
// the folder-scan threads and (in the web app) on the main thread outside
|
||||
// the disposable-worker isolation, so it must not panic on any input.
|
||||
// Validate the PE signature with checked arithmetic so a crafted offset
|
||||
// cannot wrap the bounds check on a narrower target.
|
||||
let e_lfanew = primitives::get_u32(input, 0x3C);
|
||||
let pe_start = e_lfanew as usize;
|
||||
if pe_start.checked_add(4).is_none_or(|end| end > input.len()) {
|
||||
@@ -194,13 +316,76 @@ pub fn unpack_auto_v(input: &[u8], verbose: bool) -> Result<(Kind, Vec<u8>), Unp
|
||||
Ok(out) => out,
|
||||
Err(dll_err) => match exe::unpack_v(input, verbose) {
|
||||
Ok(out) => out,
|
||||
// Surface the DLL-pipeline error, not the EXE one: for a
|
||||
// genuinely corrupt DLL the DLL error is the more relevant
|
||||
// diagnostic, and the EXE fallback is best-effort.
|
||||
Err(_) => return Err(dll_err),
|
||||
Err(exe_err) => {
|
||||
return Err(UnpackError::PipelineFallbackFailed {
|
||||
dll: Box::new(dll_err),
|
||||
exe: Box::new(exe_err),
|
||||
});
|
||||
}
|
||||
},
|
||||
}
|
||||
}
|
||||
};
|
||||
Ok((detected.kind, out))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn caught_panic_reports_location_and_message() {
|
||||
let error = catch_unpack(|| -> Result<Vec<u8>, UnpackError> {
|
||||
panic!("test panic");
|
||||
})
|
||||
.expect_err("panic must become an error");
|
||||
let UnpackError::InternalPanic {
|
||||
message,
|
||||
file,
|
||||
line,
|
||||
column,
|
||||
} = error
|
||||
else {
|
||||
panic!("unexpected error: {error}");
|
||||
};
|
||||
assert_eq!(message, "test panic");
|
||||
assert!(
|
||||
file.ends_with("senbei-pe/src/engine/mod.rs")
|
||||
|| file.ends_with("senbei-pe\\src\\engine\\mod.rs")
|
||||
);
|
||||
assert!(line > 0);
|
||||
assert!(column > 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn worker_panic_keeps_the_worker_source_location() {
|
||||
let error = catch_unpack(|| -> Result<Vec<u8>, UnpackError> {
|
||||
let capture = current_panic_capture();
|
||||
let result = std::thread::spawn(move || {
|
||||
with_panic_capture(capture, || panic!("worker panic"));
|
||||
})
|
||||
.join();
|
||||
if let Err(payload) = result {
|
||||
std::panic::resume_unwind(payload);
|
||||
}
|
||||
Ok(Vec::new())
|
||||
})
|
||||
.expect_err("worker panic must become an error");
|
||||
let UnpackError::InternalPanic {
|
||||
message,
|
||||
file,
|
||||
line,
|
||||
column,
|
||||
} = error
|
||||
else {
|
||||
panic!("unexpected error: {error}");
|
||||
};
|
||||
assert_eq!(message, "worker panic");
|
||||
assert!(
|
||||
file.ends_with("senbei-pe/src/engine/mod.rs")
|
||||
|| file.ends_with("senbei-pe\\src\\engine\\mod.rs")
|
||||
);
|
||||
assert!(line > 0);
|
||||
assert!(column > 0);
|
||||
}
|
||||
}
|
||||
@@ -21,7 +21,7 @@ use std::sync::atomic::{AtomicBool, Ordering};
|
||||
|
||||
/// Worker-thread cap. `SENBEI_THREADS` overrides it (`1` forces the sequential
|
||||
/// path); otherwise the host's available parallelism; otherwise 1.
|
||||
pub(crate) fn thread_cap() -> usize {
|
||||
pub fn thread_cap() -> usize {
|
||||
if let Ok(v) = std::env::var("SENBEI_THREADS")
|
||||
&& let Ok(n) = v.trim().parse::<usize>()
|
||||
&& n >= 1
|
||||
@@ -46,7 +46,7 @@ pub(crate) fn thread_cap() -> usize {
|
||||
///
|
||||
/// Returns the first `Err` any block produces; re-raises the first block panic
|
||||
/// on the calling thread (so the pipeline's existing `catch_unpack` still
|
||||
/// converts it to `UnpackError::Corrupt`).
|
||||
/// converts it to `UnpackError::InternalPanic`).
|
||||
pub(crate) fn parallel_for<E, F>(
|
||||
buf: &mut [u8],
|
||||
spans: &[(usize, usize)],
|
||||
@@ -125,6 +125,7 @@ where
|
||||
let stop = AtomicBool::new(false);
|
||||
let first_err: Mutex<Option<E>> = Mutex::new(None);
|
||||
let first_panic: Mutex<Option<Box<dyn std::any::Any + Send>>> = Mutex::new(None);
|
||||
let panic_capture = super::current_panic_capture();
|
||||
|
||||
std::thread::scope(|scope| {
|
||||
for _ in 0..workers {
|
||||
@@ -133,6 +134,7 @@ where
|
||||
let first_err = &first_err;
|
||||
let first_panic = &first_panic;
|
||||
let f = &f;
|
||||
let panic_capture = panic_capture.clone();
|
||||
scope.spawn(move || {
|
||||
loop {
|
||||
if stop.load(Ordering::Relaxed) {
|
||||
@@ -141,9 +143,15 @@ where
|
||||
let next = iter.lock().unwrap().next();
|
||||
let Some((i, piece)) = next else { break };
|
||||
let span = piece.unwrap();
|
||||
let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
// Keep details local until this panic wins `first_panic`;
|
||||
// otherwise simultaneous workers could pair one worker's
|
||||
// location with another worker's propagated payload.
|
||||
let block_capture = panic_capture.as_ref().map(|_| super::PanicCapture::new());
|
||||
let r = super::with_panic_capture(block_capture.clone(), || {
|
||||
std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
|
||||
f(i, spans[i].0, span)
|
||||
}));
|
||||
}))
|
||||
});
|
||||
match r {
|
||||
Ok(Ok(())) => {}
|
||||
Ok(Err(e)) => {
|
||||
@@ -157,6 +165,11 @@ where
|
||||
Err(panic) => {
|
||||
let mut slot = first_panic.lock().unwrap();
|
||||
if slot.is_none() {
|
||||
if let (Some(parent), Some(block)) =
|
||||
(&panic_capture, &block_capture)
|
||||
{
|
||||
parent.merge_from(block);
|
||||
}
|
||||
*slot = Some(panic);
|
||||
}
|
||||
stop.store(true, Ordering::Relaxed);
|
||||
@@ -0,0 +1,5 @@
|
||||
//! PE detection, unpacking, and structural validation.
|
||||
|
||||
mod engine;
|
||||
|
||||
pub use engine::*;
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,10 +0,0 @@
|
||||
//! Shared test fixtures.
|
||||
#![allow(dead_code)]
|
||||
|
||||
use std::path::PathBuf;
|
||||
|
||||
/// Path to `senbei/samples` — the user-managed corpus dropped in by hand.
|
||||
/// Git-ignored except its README; tests here run against whatever is present.
|
||||
pub fn samples_dir() -> PathBuf {
|
||||
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("samples")
|
||||
}
|
||||
Generated
+54
-32
@@ -50,21 +50,21 @@ checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0"
|
||||
|
||||
[[package]]
|
||||
name = "futures-core"
|
||||
version = "0.3.33"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7"
|
||||
checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e"
|
||||
|
||||
[[package]]
|
||||
name = "futures-task"
|
||||
version = "0.3.33"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109"
|
||||
checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd"
|
||||
|
||||
[[package]]
|
||||
name = "futures-util"
|
||||
version = "0.3.33"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa"
|
||||
checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-task",
|
||||
@@ -87,9 +87,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "js-sys"
|
||||
version = "0.3.103"
|
||||
version = "0.3.104"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102"
|
||||
checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"futures-util",
|
||||
@@ -110,9 +110,9 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
||||
|
||||
[[package]]
|
||||
name = "owo-colors"
|
||||
version = "4.3.0"
|
||||
version = "4.4.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d211803b9b6b570f68772237e415a029d5a50c65d382910b879fb19d3271f94d"
|
||||
checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8"
|
||||
|
||||
[[package]]
|
||||
name = "pin-project-lite"
|
||||
@@ -122,9 +122,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd"
|
||||
|
||||
[[package]]
|
||||
name = "portable-atomic"
|
||||
version = "1.14.0"
|
||||
version = "1.15.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3"
|
||||
checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
|
||||
|
||||
[[package]]
|
||||
name = "proc-macro2"
|
||||
@@ -160,24 +160,46 @@ dependencies = [
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei"
|
||||
version = "1.0.0"
|
||||
name = "senbei-crypto"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"thiserror",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei-io"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"indicatif",
|
||||
"libc",
|
||||
"owo-colors",
|
||||
"thiserror",
|
||||
"senbei-metadata",
|
||||
"senbei-pe",
|
||||
"walkdir",
|
||||
"windows",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei-metadata"
|
||||
version = "1.1.0"
|
||||
|
||||
[[package]]
|
||||
name = "senbei-pe"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"senbei-crypto",
|
||||
"thiserror",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "senbei-web"
|
||||
version = "1.0.0"
|
||||
version = "1.1.0"
|
||||
dependencies = [
|
||||
"console_error_panic_hook",
|
||||
"senbei",
|
||||
"senbei-io",
|
||||
"senbei-metadata",
|
||||
"senbei-pe",
|
||||
"wasm-bindgen",
|
||||
]
|
||||
|
||||
@@ -200,9 +222,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "syn"
|
||||
version = "3.0.3"
|
||||
version = "3.0.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3"
|
||||
checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
@@ -211,22 +233,22 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "thiserror"
|
||||
version = "2.0.19"
|
||||
version = "2.0.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9"
|
||||
checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f"
|
||||
dependencies = [
|
||||
"thiserror-impl",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "thiserror-impl"
|
||||
version = "2.0.19"
|
||||
version = "2.0.20"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd"
|
||||
checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 3.0.3",
|
||||
"syn 3.0.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -259,9 +281,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4"
|
||||
checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"once_cell",
|
||||
@@ -272,9 +294,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1"
|
||||
checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1"
|
||||
dependencies = [
|
||||
"quote",
|
||||
"wasm-bindgen-macro-support",
|
||||
@@ -282,9 +304,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-macro-support"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e"
|
||||
checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284"
|
||||
dependencies = [
|
||||
"bumpalo",
|
||||
"proc-macro2",
|
||||
@@ -295,9 +317,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "wasm-bindgen-shared"
|
||||
version = "0.2.126"
|
||||
version = "0.2.127"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24"
|
||||
checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf"
|
||||
dependencies = [
|
||||
"unicode-ident",
|
||||
]
|
||||
|
||||
+4
-2
@@ -1,6 +1,6 @@
|
||||
[package]
|
||||
name = "senbei-web"
|
||||
version = "1.0.1"
|
||||
version = "1.1.0"
|
||||
edition = "2024"
|
||||
description = "WebAssembly browser frontend for senbei"
|
||||
license = "AGPL-3.0-only"
|
||||
@@ -9,7 +9,9 @@ license = "AGPL-3.0-only"
|
||||
crate-type = ["cdylib"]
|
||||
|
||||
[dependencies]
|
||||
senbei = { path = ".." }
|
||||
senbei-io = { path = "../senbei-io" }
|
||||
senbei-metadata = { path = "../senbei-metadata" }
|
||||
senbei-pe = { path = "../senbei-pe" }
|
||||
wasm-bindgen = "0.2"
|
||||
console_error_panic_hook = "0.1"
|
||||
|
||||
|
||||
+11
-11
@@ -101,12 +101,12 @@ impl MetadataResult {
|
||||
}
|
||||
}
|
||||
|
||||
fn kind_str(kind: senbei::unpacker::Kind) -> &'static str {
|
||||
fn kind_str(kind: senbei_pe::Kind) -> &'static str {
|
||||
match kind {
|
||||
senbei::unpacker::Kind::NativeExe => "native-exe",
|
||||
senbei::unpacker::Kind::ManagedExe => "managed-exe",
|
||||
senbei::unpacker::Kind::NativeDll => "native-dll",
|
||||
senbei::unpacker::Kind::ManagedDll => "managed-dll",
|
||||
senbei_pe::Kind::NativeExe => "native-exe",
|
||||
senbei_pe::Kind::ManagedExe => "managed-exe",
|
||||
senbei_pe::Kind::NativeDll => "native-dll",
|
||||
senbei_pe::Kind::ManagedDll => "managed-dll",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -117,10 +117,10 @@ fn kind_str(kind: senbei::unpacker::Kind) -> &'static str {
|
||||
/// anything unrecognized.
|
||||
#[wasm_bindgen]
|
||||
pub fn detect(input: &[u8]) -> Option<String> {
|
||||
if senbei::metadata::is_metadata(input) {
|
||||
if senbei_metadata::is_metadata(input) {
|
||||
return Some("metadata".to_string());
|
||||
}
|
||||
senbei::unpacker::detect(input).map(|d| kind_str(d.kind).to_string())
|
||||
senbei_pe::detect(input).map(|d| kind_str(d.kind).to_string())
|
||||
}
|
||||
|
||||
/// Unpack a protected module.
|
||||
@@ -134,7 +134,7 @@ pub fn unpack_file(
|
||||
input: &[u8],
|
||||
companion: Option<Vec<u8>>,
|
||||
) -> Result<UnpackResult, JsError> {
|
||||
let r = senbei::job::unpack_bytes(input, companion.as_deref())
|
||||
let r = senbei_io::job::unpack_bytes(input, companion.as_deref())
|
||||
.map_err(|e| JsError::new(&e.to_string()))?;
|
||||
Ok(UnpackResult {
|
||||
kind: kind_str(r.kind).to_string(),
|
||||
@@ -153,7 +153,7 @@ pub fn unpack_file(
|
||||
#[wasm_bindgen]
|
||||
pub fn deobfuscate_metadata(data: &[u8]) -> Result<MetadataResult, JsError> {
|
||||
let (bytes, report) =
|
||||
senbei::metadata::deobfuscate(data).map_err(|e| JsError::new(&e.to_string()))?;
|
||||
senbei_metadata::deobfuscate(data).map_err(|e| JsError::new(&e.to_string()))?;
|
||||
Ok(MetadataResult {
|
||||
bytes,
|
||||
version: report.version,
|
||||
@@ -164,14 +164,14 @@ pub fn deobfuscate_metadata(data: &[u8]) -> Result<MetadataResult, JsError> {
|
||||
}
|
||||
|
||||
/// Unpack a protected module, forcing the EXE pipeline (no DLL-pipeline
|
||||
/// probe). See [`senbei::job::unpack_bytes_force_exe`] for why the web app
|
||||
/// probe). See [`senbei_io::job::unpack_bytes_force_exe`] for why the web app
|
||||
/// needs this recovery path.
|
||||
#[wasm_bindgen]
|
||||
pub fn unpack_file_force_exe(
|
||||
input: &[u8],
|
||||
companion: Option<Vec<u8>>,
|
||||
) -> Result<UnpackResult, JsError> {
|
||||
let r = senbei::job::unpack_bytes_force_exe(input, companion.as_deref())
|
||||
let r = senbei_io::job::unpack_bytes_force_exe(input, companion.as_deref())
|
||||
.map_err(|e| JsError::new(&e.to_string()))?;
|
||||
Ok(UnpackResult {
|
||||
kind: kind_str(r.kind).to_string(),
|
||||
|
||||
Reference in New Issue
Block a user