Merge bfloat16-senbei workspace restructure, bump to 1.1.0

Adopts the fork's workspace split (senbei-cli / senbei-crypto / senbei-io /
senbei-metadata / senbei-pe), its structured error taxonomy, entry-transform
and layout validation, PE32 dd8 key-formula selection with a skip floor, the
CRT entry-stub dd8 oracle, and the extensionless-file scan skip.

Kept from senbei on top of the restructure:
- ManagedExe detection/routing and the CLR (COR20 + BSJB) metadata restore
  in the EXE pipeline.
- The RET+int3 padding fingerprint as the primary dd8 padding signal, ahead
  of the mutated-position 0xCC fallback.
- docs/, .github/, samples/, tests/ (moved to senbei-cli/tests), and the
  web/ wasm frontend (rewired to the split crates), all of which the fork
  had dropped.
- The fork's README compatibility matrix is not taken: it names real games,
  which the public-repo hygiene rules forbid.
- The wasm32 localtime fallback in logfile and unpack_bytes_force_exe (the
  web app's trap-recovery entry point), both lost in the restructure.

Golden corpus: 35/35 byte-identical. clippy -D warnings clean; wasm32 check
clean for the full workspace.
This commit is contained in:
2026-08-30 22:31:21 +08:00
49 changed files with 5259 additions and 3856 deletions
+4 -4
View File
@@ -19,7 +19,7 @@ jobs:
runs-on: windows-latest runs-on: windows-latest
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- run: cargo clippy --all-targets -- -D warnings - run: cargo clippy --workspace --all-targets -- -D warnings
test: test:
# The test suite exercises Windows path semantics, so it runs on Windows. # The test suite exercises Windows path semantics, so it runs on Windows.
@@ -28,7 +28,7 @@ jobs:
runs-on: windows-latest runs-on: windows-latest
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- run: cargo test --release - run: cargo test --release --workspace
check-portable: check-portable:
# Build-only portability gate: non-Windows host and the wasm target the # Build-only portability gate: non-Windows host and the wasm target the
@@ -39,8 +39,8 @@ jobs:
runs-on: ubuntu-latest runs-on: ubuntu-latest
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- run: cargo clippy --all-targets -- -D warnings - run: cargo clippy --workspace --all-targets -- -D warnings
- run: cargo check --target wasm32-unknown-unknown - run: cargo check --workspace --target wasm32-unknown-unknown
cli: cli:
strategy: strategy:
+9 -7
View File
@@ -4,17 +4,19 @@ Guidance for AI coding agents (and human contributors) working in this repo.
## Project ## Project
Senbei is a static unpacker for Crackproof-protected PE files: a pure, Senbei is a static unpacker for Crackproof-protected PE files: a Cargo
panic-free, no-I/O unpacker core (`src/unpacker/`) plus a thin CLI shell workspace with a pure, panic-free, no-I/O unpacker core (`senbei-pe/`, built
(`src/`), an il2cpp metadata de-obfuscator (`src/metadata.rs`), and a on `senbei-crypto/`), an il2cpp metadata de-obfuscator (`senbei-metadata/`),
WebAssembly browser frontend (`web/`). Read `docs/design.md` first. filesystem/CLI orchestration (`senbei-io/`), the `senbei` binary
(`senbei-cli/`), and a WebAssembly browser frontend (`web/`, outside the
workspace). Read `docs/design.md` first.
## Commands ## Commands
```cmd ```cmd
cargo build --release :: CLI cargo build --release :: CLI (default member: senbei-cli)
cargo test --release :: full suite (golden corpus: samples/, git-ignored) cargo test --release --workspace :: full suite (golden corpus: samples/, git-ignored)
cargo clippy --all-targets -- -D warnings cargo clippy --workspace --all-targets -- -D warnings
cargo fmt --all cargo fmt --all
cd web && wasm-pack build --target web --release :: browser build cd web && wasm-pack build --target web --release :: browser build
``` ```
Generated
+59 -30
View File
@@ -62,21 +62,21 @@ checksum = "da7c62ceae207dd37ea5b845da6a0696c799f85e97da1ab5b7910be3c1c80223"
[[package]] [[package]]
name = "futures-core" name = "futures-core"
version = "0.3.33" version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e"
[[package]] [[package]]
name = "futures-task" name = "futures-task"
version = "0.3.33" version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd"
[[package]] [[package]]
name = "futures-util" name = "futures-util"
version = "0.3.33" version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc"
dependencies = [ dependencies = [
"futures-core", "futures-core",
"futures-task", "futures-task",
@@ -110,9 +110,9 @@ dependencies = [
[[package]] [[package]]
name = "js-sys" name = "js-sys"
version = "0.3.103" version = "0.3.104"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a"
dependencies = [ dependencies = [
"cfg-if", "cfg-if",
"futures-util", "futures-util",
@@ -139,9 +139,9 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
[[package]] [[package]]
name = "owo-colors" name = "owo-colors"
version = "4.3.0" version = "4.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d211803b9b6b570f68772237e415a029d5a50c65d382910b879fb19d3271f94d" checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8"
[[package]] [[package]]
name = "pin-project-lite" name = "pin-project-lite"
@@ -151,9 +151,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd"
[[package]] [[package]]
name = "portable-atomic" name = "portable-atomic"
version = "1.14.0" version = "1.15.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3" checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
[[package]] [[package]]
name = "proc-macro2" name = "proc-macro2"
@@ -208,19 +208,48 @@ dependencies = [
] ]
[[package]] [[package]]
name = "senbei" name = "senbei-cli"
version = "1.0.1" version = "1.1.0"
dependencies = [
"senbei-io",
"senbei-metadata",
"tempfile",
]
[[package]]
name = "senbei-crypto"
version = "1.1.0"
dependencies = [
"thiserror",
]
[[package]]
name = "senbei-io"
version = "1.1.0"
dependencies = [ dependencies = [
"anyhow", "anyhow",
"indicatif", "indicatif",
"libc", "libc",
"owo-colors", "owo-colors",
"senbei-metadata",
"senbei-pe",
"tempfile", "tempfile",
"thiserror",
"walkdir", "walkdir",
"windows", "windows",
] ]
[[package]]
name = "senbei-metadata"
version = "1.1.0"
[[package]]
name = "senbei-pe"
version = "1.1.0"
dependencies = [
"senbei-crypto",
"thiserror",
]
[[package]] [[package]]
name = "slab" name = "slab"
version = "0.4.12" version = "0.4.12"
@@ -240,9 +269,9 @@ dependencies = [
[[package]] [[package]]
name = "syn" name = "syn"
version = "3.0.3" version = "3.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
@@ -264,22 +293,22 @@ dependencies = [
[[package]] [[package]]
name = "thiserror" name = "thiserror"
version = "2.0.19" version = "2.0.20"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f"
dependencies = [ dependencies = [
"thiserror-impl", "thiserror-impl",
] ]
[[package]] [[package]]
name = "thiserror-impl" name = "thiserror-impl"
version = "2.0.19" version = "2.0.20"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
"syn 3.0.3", "syn 3.0.4",
] ]
[[package]] [[package]]
@@ -312,9 +341,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen" name = "wasm-bindgen"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70"
dependencies = [ dependencies = [
"cfg-if", "cfg-if",
"once_cell", "once_cell",
@@ -325,9 +354,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen-macro" name = "wasm-bindgen-macro"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1"
dependencies = [ dependencies = [
"quote", "quote",
"wasm-bindgen-macro-support", "wasm-bindgen-macro-support",
@@ -335,9 +364,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen-macro-support" name = "wasm-bindgen-macro-support"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284"
dependencies = [ dependencies = [
"bumpalo", "bumpalo",
"proc-macro2", "proc-macro2",
@@ -348,9 +377,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen-shared" name = "wasm-bindgen-shared"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf"
dependencies = [ dependencies = [
"unicode-ident", "unicode-ident",
] ]
+25 -25
View File
@@ -1,39 +1,39 @@
[package] [workspace]
name = "senbei" members = [
version = "1.0.1" "senbei-cli",
"senbei-crypto",
"senbei-io",
"senbei-metadata",
"senbei-pe",
]
default-members = ["senbei-cli"]
# The wasm frontend is its own crate (own Cargo.lock, cdylib) and stays outside
# the workspace.
exclude = ["web"]
resolver = "2"
[workspace.package]
version = "1.1.0"
edition = "2024" edition = "2024"
description = "Static unpacker for Crackproof-protected PE files"
license = "AGPL-3.0-only" license = "AGPL-3.0-only"
keywords = ["unpacker", "reverse-engineering", "pe", "security-research"]
categories = ["command-line-utilities"]
[lib] [workspace.dependencies]
name = "senbei"
path = "src/lib.rs"
[[bin]]
name = "senbei"
path = "src/main.rs"
[dependencies]
anyhow = "1" anyhow = "1"
indicatif = "0.18"
libc = "0.2"
owo-colors = "4"
tempfile = "3"
thiserror = "2" thiserror = "2"
walkdir = "2" walkdir = "2"
indicatif = "0.18"
owo-colors = "4"
[target.'cfg(windows)'.dependencies]
windows = { version = "0.62", features = [ windows = { version = "0.62", features = [
"Win32_Foundation", "Win32_Foundation",
"Win32_System_Console", "Win32_System_Console",
"Win32_System_SystemInformation", "Win32_System_SystemInformation",
] } ] }
senbei-crypto = { path = "senbei-crypto" }
[target.'cfg(all(not(windows), not(target_arch = "wasm32")))'.dependencies] senbei-io = { path = "senbei-io" }
libc = "0.2" senbei-metadata = { path = "senbei-metadata" }
senbei-pe = { path = "senbei-pe" }
[dev-dependencies]
tempfile = "3"
[profile.release] [profile.release]
opt-level = 3 opt-level = 3
+38 -22
View File
@@ -7,41 +7,57 @@ no driver or proxy DLL is involved.
## Crate layout ## Crate layout
The crate is split into a pure core and a thin CLI shell: Senbei is a Cargo workspace split into a pure core and thin shells around it:
- **`src/unpacker/`** — the core. Pure functions over byte slices: no file - **`senbei-pe/`** — the core. Pure functions over byte slices: no file I/O,
I/O, no environment access (beyond a few debugging overrides, see no environment access (beyond a few debugging overrides, see
[development.md](development.md)), panic-free at the public boundary (all [development.md](development.md)), panic-free at the public boundary (all
internal panics are trapped and converted to `UnpackError::Corrupt`). This internal panics are trapped and converted to `UnpackError::Corrupt`). This
is what the WebAssembly build embeds. is what the WebAssembly build embeds.
- **`src/` (top level)** — the CLI shell: argument parsing, recursive folder - **`senbei-crypto/`** — cryptographic, checksum, compression, and bytecode
scanning, per-run log file, progress bar, Explorer-friendly exit pause, and primitives the core is built from. Same purity rules as `senbei-pe`.
the single-file/folder orchestration in `job.rs`. - **`senbei-metadata/`** — il2cpp `global-metadata.dat` method-token
- **`src/metadata.rs`** — il2cpp `global-metadata.dat` method-token
de-obfuscation (format version 31; other versions are left untouched). de-obfuscation (format version 31; other versions are left untouched).
- **`senbei-io/`** — filesystem and orchestration: recursive folder scanning,
per-run log file, progress bar, Explorer-friendly exit pause, and the
single-file/folder orchestration in `job.rs` (incl. the wasm-safe in-memory
byte API used by the web frontend).
- **`senbei-cli/`** — the `senbei` binary: argument parsing + dispatch. The
integration test suite (incl. the golden corpus test) lives in
`senbei-cli/tests/`.
``` ```
src/ senbei-cli/
── main.rs argument parsing + dispatch ── src/main.rs argument parsing + dispatch
├── lib.rs module roots senbei-io/src/
├── job.rs single-file + folder orchestration, out-naming, ├── job.rs single-file + folder orchestration, out-naming,
│ companion splice, stub overlay/TLS restore, │ companion splice, stub overlay/TLS restore,
│ pipeline routing (incl. the wasm-safe byte API) │ pipeline routing (incl. the wasm-safe byte API)
├── scan.rs recursive Crackproof + metadata discovery ├── scan.rs recursive Crackproof + metadata discovery
├── metadata.rs il2cpp global-metadata.dat de-obfuscation
├── logfile.rs per-run timestamped log ├── logfile.rs per-run timestamped log
├── ui.rs progress bar + status lines ├── ui.rs progress bar + status lines
── pause.rs Explorer-friendly exit pause ── pause.rs Explorer-friendly exit pause
└── unpacker/ pure, panic-free, no-I/O core senbei-metadata/src/
├── mod.rs detection + unpack_auto dispatch └── metadata.rs il2cpp global-metadata.dat de-obfuscation
├── exe.rs EXE pipeline (PE32+ and PE32) senbei-crypto/src/
├── dll.rs native + managed DLL pipeline ├── primitives.rs decrypt_data* steps, key derivation
├── integrity.rs static post-unpack sanity check ├── bytecode.rs bytecode VM
├── primitives.rs decrypt_data* steps, key/shift selection ├── tables.rs constant tables
├── bytecode.rs bytecode VM └── crc32.rs checksum
├── parallel.rs deterministic block-parallel fan-out senbei-pe/src/engine/ pure, panic-free, no-I/O core
├── tables.rs constant tables ├── mod.rs detection + unpack_auto dispatch
└── crc32.rs checksum ├── error.rs structured error taxonomy
├── integrity.rs static post-unpack sanity check
├── parallel.rs deterministic block-parallel fan-out
├── layout/ layout discovery + validation
│ ├── dd8.rs .text dd8 key-formula + shift selection
│ ├── discovery.rs layout candidate discovery (trial-and-validate)
│ └── image.rs PE image reconstruction helpers
├── exe/
│ ├── pipeline.rs EXE pipeline (PE32+ and PE32 orchestration)
│ └── pipeline/pe32.rs PE32-specific EXE restore
└── dll/
└── pipeline.rs native + managed DLL pipeline
``` ```
## Detection and routing ## Detection and routing
+3 -3
View File
@@ -60,9 +60,9 @@ since binaries are not committed).
## Conventions ## Conventions
- The `src/unpacker/` core is pure: no file I/O, no panics across the public - The `senbei-pe/` core (and its `senbei-crypto/` base) is pure: no file I/O,
boundary, no `unsafe`. Keep it that way — it is what the WebAssembly build no panics across the public boundary, no `unsafe`. Keep it that way — it is
embeds. what the WebAssembly build embeds.
- Layout heuristics must **trial-and-validate**: never pick a candidate offset - Layout heuristics must **trial-and-validate**: never pick a candidate offset
on shape alone and trust it; validate by decryption/checksum and fall on shape alone and trust it; validate by decryption/checksum and fall
through to the next candidate on failure. A silent wrong offset produces a through to the next candidate on failure. A silent wrong offset produces a
+20
View File
@@ -0,0 +1,20 @@
[package]
name = "senbei-cli"
version.workspace = true
edition.workspace = true
description = "Command-line entry point for Senbei"
license.workspace = true
keywords = ["unpacker", "reverse-engineering", "pe", "security-research"]
categories = ["command-line-utilities"]
[[bin]]
name = "senbei"
path = "src/main.rs"
[dependencies]
senbei-io.workspace = true
[dev-dependencies]
senbei-io.workspace = true
senbei-metadata.workspace = true
tempfile.workspace = true
+16 -17
View File
@@ -1,4 +1,4 @@
use senbei::{job, pause}; use senbei_io::{job, pause, scan};
use std::path::Path; use std::path::Path;
fn main() -> std::process::ExitCode { fn main() -> std::process::ExitCode {
@@ -27,9 +27,6 @@ fn main() -> std::process::ExitCode {
"--no-log" => no_log = true, "--no-log" => no_log = true,
"--scan-all" => scan_all = true, "--scan-all" => scan_all = true,
"--out" => match args.next() { "--out" => match args.next() {
// Reject a missing value (and a following flag swallowed as the
// value): previously `--out` at end of argv silently fell back
// to the default output directory.
Some(v) if !v.starts_with('-') => out = Some(v), Some(v) if !v.starts_with('-') => out = Some(v),
_ => { _ => {
eprintln!("error: --out requires a directory argument"); eprintln!("error: --out requires a directory argument");
@@ -42,7 +39,6 @@ fn main() -> std::process::ExitCode {
return std::process::ExitCode::from(2); return std::process::ExitCode::from(2);
} }
other => { other => {
// Previously the last positional silently won.
if let Some(prev) = &path { if let Some(prev) = &path {
eprintln!("error: multiple input paths given ('{prev}' and '{other}')"); eprintln!("error: multiple input paths given ('{prev}' and '{other}')");
return std::process::ExitCode::from(2); return std::process::ExitCode::from(2);
@@ -63,33 +59,36 @@ fn main() -> std::process::ExitCode {
} }
let p = Path::new(&p); let p = Path::new(&p);
let out_path = out.as_deref().map(Path::new); let out_path = out.as_deref().map(Path::new);
let r = if p.is_dir() { let result = if p.is_dir() {
job::run_folder_opts( job::run_folder_opts(
p, p,
out_path, out_path,
quiet, quiet,
verbose, verbose,
no_log, no_log,
scan_all || senbei::scan::scan_all_env(), scan_all || scan::scan_all_env(),
) )
} else { } else {
job::run_file_v(p, out_path, quiet, verbose, no_log) job::run_file_v(p, out_path, quiet, verbose, no_log)
}; };
match r { match result {
Ok(s) => { Ok(summary) => {
if quiet < 2 { if quiet < 2 {
println!( println!(
"{} unpacked · {} skipped · {} errors · {} suspect · {} metadata", "{} unpacked · {} skipped · {} errors · {} suspect · {} metadata",
s.unpacked, s.skipped, s.errors, s.suspect, s.metadata summary.unpacked,
summary.skipped,
summary.errors,
summary.suspect,
summary.metadata
); );
println!("done in {} ms", s.duration_ms); println!("done in {} ms", summary.duration_ms);
} }
if s.errors > 0 { 1 } else { 0 } if summary.errors > 0 { 1 } else { 0 }
} }
Err(e) => { Err(error) => {
// Fatal: out-dir/log create, etc.
if quiet < 2 { if quiet < 2 {
eprintln!("error: {e:#}"); eprintln!("error: {error:#}");
} }
1 1
} }
@@ -107,7 +106,7 @@ fn print_help() {
); );
println!( println!(
" --scan-all probe every file in a folder, including ones the scan\n\ " --scan-all probe every file in a folder, including ones the scan\n\
\x20 pre-filter skips (under 4128 bytes, or a bulk-asset\n\ \x20 pre-filter skips (under 4128 bytes, extensionless,\n\
\x20 extension like .ab/.xml/.acb). Much slower on game trees." \x20 or a bulk-asset extension). Much slower on large trees."
); );
} }
+11
View File
@@ -0,0 +1,11 @@
//! Shared test fixtures.
#![allow(dead_code)]
use std::path::PathBuf;
/// Path to the workspace-root `samples/` — the user-managed corpus dropped in
/// by hand. Git-ignored except its README; tests here run against whatever is
/// present. `CARGO_MANIFEST_DIR` is `senbei-cli/`, so go one level up.
pub fn samples_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("../samples")
}
+1 -1
View File
@@ -1,4 +1,4 @@
use senbei::job::{default_out_root_for_file, out_name}; use senbei_io::job::{default_out_root_for_file, out_name};
use std::path::Path; use std::path::Path;
#[test] #[test]
@@ -1,4 +1,4 @@
use senbei::logfile::{Log, local_stamp_compact, local_stamp_display}; use senbei_io::logfile::{Log, local_stamp_compact, local_stamp_display};
#[test] #[test]
fn local_stamp_compact_matches_shape() { fn local_stamp_compact_matches_shape() {
@@ -1,4 +1,4 @@
use senbei::job; use senbei_io::job;
use std::path::Path; use std::path::Path;
fn list_logs(dir: &Path) -> Vec<std::path::PathBuf> { fn list_logs(dir: &Path) -> Vec<std::path::PathBuf> {
@@ -8,7 +8,7 @@
//! - golden present, bytes differ -> FAIL (the test fails) //! - golden present, bytes differ -> FAIL (the test fails)
//! - no golden -> WARNING (printed; needs a manual check) //! - no golden -> WARNING (printed; needs a manual check)
//! //!
//! Inputs go through [`senbei::job::unpack_bytes`], the same routing the CLI //! Inputs go through [`senbei_io::job::unpack_bytes`], the same routing the CLI
//! uses, **not** `unpack_auto` directly. That matters: `unpack_auto` alone //! uses, **not** `unpack_auto` directly. That matters: `unpack_auto` alone
//! cannot reach the external-companion layout, whose stub is meaningless //! cannot reach the external-companion layout, whose stub is meaningless
//! without its `<name>._` payload — a corpus wired to `unpack_auto` silently //! without its `<name>._` payload — a corpus wired to `unpack_auto` silently
@@ -17,7 +17,7 @@
//! samples folder is picked up automatically, exactly as it is on disk. //! samples folder is picked up automatically, exactly as it is on disk.
//! //!
//! An input whose bytes carry the il2cpp metadata magic is routed through //! An input whose bytes carry the il2cpp metadata magic is routed through
//! [`senbei::metadata::deobfuscate`] instead, giving the method-token remap //! [`senbei_metadata::deobfuscate`] instead, giving the method-token remap
//! real-world coverage (its unit tests only build synthetic layouts). //! real-world coverage (its unit tests only build synthetic layouts).
//! //!
//! The folder is git-ignored (see `senbei/samples/README.md`), so the set of //! The folder is git-ignored (see `senbei/samples/README.md`), so the set of
@@ -116,10 +116,10 @@ fn samples_unpack_against_goldens() {
} }
}; };
let got = if senbei::metadata::is_metadata(&bytes) { let got = if senbei_metadata::is_metadata(&bytes) {
// il2cpp metadata: method-token de-obfuscation, no PE pipeline and // il2cpp metadata: method-token de-obfuscation, no PE pipeline and
// no integrity check (the output is not a PE image). // no integrity check (the output is not a PE image).
match senbei::metadata::deobfuscate(&bytes) { match senbei_metadata::deobfuscate(&bytes) {
Ok((out, _report)) => out, Ok((out, _report)) => out,
Err(e) => { Err(e) => {
failures.push(format!("{name}: de-obfuscation failed: {e}")); failures.push(format!("{name}: de-obfuscation failed: {e}"));
@@ -140,7 +140,7 @@ fn samples_unpack_against_goldens() {
}, },
None => None, None => None,
}; };
let image = match senbei::job::unpack_bytes(&bytes, companion.as_deref()) { let image = match senbei_io::job::unpack_bytes(&bytes, companion.as_deref()) {
Ok(img) => img, Ok(img) => img,
Err(e) => { Err(e) => {
failures.push(format!("{name}: unpack failed: {e:?}")); failures.push(format!("{name}: unpack failed: {e:?}"));
+9
View File
@@ -0,0 +1,9 @@
[package]
name = "senbei-crypto"
version.workspace = true
edition.workspace = true
license.workspace = true
description = "Cryptographic and compression primitives for Senbei"
[dependencies]
thiserror.workspace = true
@@ -61,8 +61,8 @@ impl OpsLut {
pub fn generate(data: &[u8], offset: u32) -> Option<Vec<Op>> { pub fn generate(data: &[u8], offset: u32) -> Option<Vec<Op>> {
// Bounds-checked cursor: a corrupt `data_offset` (bad decrypt_data6 / the // Bounds-checked cursor: a corrupt `data_offset` (bad decrypt_data6 / the
// alignment fallback) must yield `None`, not an out-of-bounds panic — the // alignment fallback) must yield `None`, not an out-of-bounds panic — the
// panic path would surface as a misleading `UnpackError::Corrupt` instead // panic path would surface as a misleading `UnpackError::InternalPanic` instead
// of the precise `BytecodeGenFailed`, and any future caller without a // of the precise `BytecodeGenerationFailed`, and any future caller without a
// `catch_unwind` wrapper would abort outright. // `catch_unwind` wrapper would abort outright.
let mut pos = offset as usize; let mut pos = offset as usize;
let mut next = move || { let mut next = move || {
+77
View File
@@ -0,0 +1,77 @@
//! Cryptographic, checksum, compression, and bytecode primitives.
pub mod bytecode;
pub mod crc32;
pub mod primitives;
mod tables;
/// Maximum buffer size accepted by allocation-sensitive transforms.
pub const MAX_IMAGE_SIZE: u64 = 1 << 30;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum BufferOperation {
Read,
CopySource,
CopyDestination,
ZeroFill,
}
impl std::fmt::Display for BufferOperation {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(match self {
Self::Read => "read",
Self::CopySource => "copy source",
Self::CopyDestination => "copy destination",
Self::ZeroFill => "zero-fill",
})
}
}
#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
pub enum Error {
#[error(
"{operation} range out of bounds (offset {offset}, size {size}, buffer length {buffer_len})"
)]
BufferRangeOutOfBounds {
operation: BufferOperation,
offset: usize,
size: usize,
buffer_len: usize,
},
}
#[derive(Debug, Clone, Copy, PartialEq, Eq, thiserror::Error)]
#[non_exhaustive]
pub enum DecompressionFailure {
#[error("compressed source size {size} exceeds limit {max}")]
SourceTooLarge { size: u32, max: u64 },
#[error("Huffman code length {bits} is invalid")]
InvalidCodeLength { bits: u8 },
#[error("Huffman tree traversal exceeded 64 levels")]
HuffmanTraversalLimit,
#[error("pending length accumulator overflowed at {pending}")]
PendingLengthOverflow { pending: u32 },
#[error("output step {step} at byte {written} exceeds expected size {expected}")]
OutputOverflow {
written: u32,
step: u32,
expected: u32,
},
#[error("run-fill width {width} reads before output offset 0x{destination:08X}")]
RunFillBeforeOutput { width: u32, destination: u32 },
#[error("run-fill width {width} is unsupported")]
InvalidRunFillWidth { width: u32 },
#[error("back-reference distance {distance} exceeds {written} written bytes")]
InvalidBackReference { distance: u32, written: u32 },
#[error("Huffman symbol consumed no input and produced no output")]
NoProgress,
#[error(
"output size mismatch (wrote {written}/{expected} bytes after consuming {consumed}/{source_size})"
)]
OutputSizeMismatch {
written: u32,
expected: u32,
consumed: u32,
source_size: u32,
},
}
File diff suppressed because it is too large Load Diff
@@ -146,11 +146,11 @@ mod tests {
#[test] #[test]
fn generated_tables_match_committed_bytes() { fn generated_tables_match_committed_bytes() {
assert_eq!(COLUMMIX1.len(), 1024); assert_eq!(COLUMMIX1.len(), 1024);
assert_eq!(super::super::crc32::compute(&COLUMMIX1), 0x7e8d_5d5f); assert_eq!(crate::crc32::compute(&COLUMMIX1), 0x7e8d_5d5f);
assert_eq!(super::super::crc32::compute(&COLUMMIX2), 0xfcc4_acfc); assert_eq!(crate::crc32::compute(&COLUMMIX2), 0xfcc4_acfc);
assert_eq!(super::super::crc32::compute(&COLUMMIX3), 0x637a_f0cd); assert_eq!(crate::crc32::compute(&COLUMMIX3), 0x637a_f0cd);
assert_eq!(super::super::crc32::compute(&COLUMMIX4), 0x1e7b_c381); assert_eq!(crate::crc32::compute(&COLUMMIX4), 0x1e7b_c381);
assert_eq!(super::super::crc32::compute(&SBOX), 0x10fd_6dc1); assert_eq!(crate::crc32::compute(&SBOX), 0x10fd_6dc1);
// Spot-check the first dword of each (matches the original first row). // Spot-check the first dword of each (matches the original first row).
assert_eq!(&COLUMMIX1[..4], &[0x50, 0xa7, 0xf4, 0x51]); assert_eq!(&COLUMMIX1[..4], &[0x50, 0xa7, 0xf4, 0x51]);
assert_eq!(&COLUMMIX2[..4], &[0xa7, 0xf4, 0x51, 0x50]); assert_eq!(&COLUMMIX2[..4], &[0xa7, 0xf4, 0x51, 0x50]);
+23
View File
@@ -0,0 +1,23 @@
[package]
name = "senbei-io"
version.workspace = true
edition.workspace = true
license.workspace = true
description = "Filesystem, scanning, logging, and CLI orchestration for Senbei"
[dependencies]
anyhow.workspace = true
indicatif.workspace = true
owo-colors.workspace = true
senbei-metadata.workspace = true
senbei-pe.workspace = true
walkdir.workspace = true
[target.'cfg(windows)'.dependencies]
windows.workspace = true
[target.'cfg(all(not(windows), not(target_arch = "wasm32")))'.dependencies]
libc.workspace = true
[dev-dependencies]
tempfile.workspace = true
+19 -25
View File
@@ -1,4 +1,4 @@
use crate::unpacker; use senbei_pe as unpacker;
use std::path::{Path, PathBuf}; use std::path::{Path, PathBuf};
/// Crackproof header key table lives at this fixed file offset. For the /// Crackproof header key table lives at this fixed file offset. For the
@@ -387,9 +387,9 @@ pub fn run_folder_v(
/// ///
/// When `scan_all` is true every regular file under `root` is opened and /// When `scan_all` is true every regular file under `root` is opened and
/// content-probed, instead of skipping ones the free directory metadata already /// content-probed, instead of skipping ones the free directory metadata already
/// rules out (too small to hold a Crackproof key table, or a bulk-asset /// rules out (extensionless, too small to hold a Crackproof key table, or a
/// extension). See [`crate::scan::find_targets_opts`] — exhaustive scanning is /// bulk-asset extension). See [`crate::scan::find_targets_opts`] — exhaustive
/// dramatically slower on asset-heavy game trees and finds the same targets. /// scanning is dramatically slower on asset-heavy trees.
pub fn run_folder_opts( pub fn run_folder_opts(
root: &Path, root: &Path,
out_dir: Option<&Path>, out_dir: Option<&Path>,
@@ -518,7 +518,7 @@ pub fn run_folder_opts(
// il2cpp metadata pass. Crackproof's `-GMD` option obfuscates the method // il2cpp metadata pass. Crackproof's `-GMD` option obfuscates the method
// tokens in `global-metadata.dat`; de-obfuscate any we find so the unpacked // tokens in `global-metadata.dat`; de-obfuscate any we find so the unpacked
// il2cpp game assembly resolves methods instead of indexing its per-module // il2cpp game assembly resolves methods instead of indexing its per-module
// tables out of bounds (see [`crate::metadata`]). This is additive to the // tables out of bounds (see [`senbei_metadata`]). This is additive to the
// Crackproof module unpack above — the metadata blob is not itself a // Crackproof module unpack above — the metadata blob is not itself a
// Crackproof file. // Crackproof file.
for meta in metas { for meta in metas {
@@ -660,7 +660,7 @@ pub fn run_file_v(
let mut buf = [0u8; 4]; let mut buf = [0u8; 4];
std::fs::File::open(input) std::fs::File::open(input)
.and_then(|mut f| f.read_exact(&mut buf)) .and_then(|mut f| f.read_exact(&mut buf))
.map(|_| crate::metadata::is_metadata(&buf)) .map(|_| senbei_metadata::is_metadata(&buf))
.unwrap_or(false) .unwrap_or(false)
}; };
@@ -801,13 +801,13 @@ fn panic_payload(panic: &(dyn std::any::Any + Send)) -> String {
} }
} }
/// If `e`'s chain contains [`crate::metadata::Error::UnsupportedVersion`], /// If `e`'s chain contains [`senbei_metadata::Error::UnsupportedVersion`],
/// return the version. Used to apply the folder-mode "leave untouched, don't /// return the version. Used to apply the folder-mode "leave untouched, don't
/// fail the run" policy to metadata versions this build can't de-obfuscate. /// fail the run" policy to metadata versions this build can't de-obfuscate.
fn unsupported_version(e: &anyhow::Error) -> Option<u32> { fn unsupported_version(e: &anyhow::Error) -> Option<u32> {
for cause in e.chain() { for cause in e.chain() {
if let Some(crate::metadata::Error::UnsupportedVersion(v)) = if let Some(senbei_metadata::Error::UnsupportedVersion(v)) =
cause.downcast_ref::<crate::metadata::Error>() cause.downcast_ref::<senbei_metadata::Error>()
{ {
return Some(*v); return Some(*v);
} }
@@ -831,19 +831,14 @@ fn write_atomic(dest: &Path, bytes: &[u8]) -> std::io::Result<()> {
r r
} }
/// Detect `bytes` and run the right pipeline. The EXE pipeline is invoked /// Detect `bytes` and run the right pipeline. Spliced external companions use
/// directly (no DLL-pipeline probe) when the input was spliced from an /// the EXE pipeline directly because that layout is definitionally EXE-style.
/// external companion (`spliced`) or when the caller forces it (`force_exe`
/// — the web app's recovery path after a DLL-probe trap; see
/// [`unpack_bytes_force_exe`]).
/// ///
/// Routing spliced inputs straight to the EXE pipeline is safe: the /// Routing spliced inputs straight to the EXE pipeline is safe: the
/// companion layout is definitionally the EXE-style shell (the runtime /// companion layout is definitionally the EXE-style shell (the runtime
/// loader maps the companion and runs the standard shell unpack), so the DLL /// loader maps the companion and runs the standard shell unpack), so the DLL
/// pipeline probe can never be right for it — and probing is not a no-op on /// pipeline probe can never be right for it. Output bytes are identical to the
/// targets without unwinding (wasm), where the probe's caught panic becomes /// DLL-first + EXE-fallback route for every input that route handles.
/// a fatal trap. Output bytes are identical to the dll-first + exe-fallback
/// route for every input that route handles.
fn unpack_spliced_or_auto( fn unpack_spliced_or_auto(
bytes: &[u8], bytes: &[u8],
spliced: bool, spliced: bool,
@@ -880,10 +875,9 @@ pub struct UnpackedImage {
/// Unpack in-memory `input` bytes, optionally paired with an external /// Unpack in-memory `input` bytes, optionally paired with an external
/// companion payload `companion` (the `<input>._` file's contents). /// companion payload `companion` (the `<input>._` file's contents).
/// ///
/// This is the I/O-free counterpart of [`unpack_one_v`], used by the /// This is the in-memory counterpart of [`unpack_one_v`]: splice a matching
/// WebAssembly build: splice (when the companion's first 32 bytes match the /// companion, unpack, overlay the export table and TLS directory from the stub,
/// stub header), unpack, overlay the export table and TLS directory from the /// then run the static integrity check.
/// stub, then run the static integrity check.
pub fn unpack_bytes( pub fn unpack_bytes(
input: &[u8], input: &[u8],
companion: Option<&[u8]>, companion: Option<&[u8]>,
@@ -960,7 +954,7 @@ pub fn unpack_one_v(
/// into a sparse, original-metadata-style value; il2cpp expects the contiguous /// into a sparse, original-metadata-style value; il2cpp expects the contiguous
/// per-module index it indexes its codegen tables with, so a statically-unpacked /// per-module index it indexes its codegen tables with, so a statically-unpacked
/// il2cpp game assembly reads garbage and crashes during init. This rewrites /// il2cpp game assembly reads garbage and crashes during init. This rewrites
/// the tokens back to their canonical form (see [`crate::metadata::deobfuscate`]). /// the tokens back to their canonical form (see [`senbei_metadata::deobfuscate`]).
/// ///
/// The output is written only when something actually changed /// The output is written only when something actually changed
/// (`report.remapped > 0`); an already-clean metadata is left untouched and no /// (`report.remapped > 0`); an already-clean metadata is left untouched and no
@@ -970,11 +964,11 @@ pub fn deobfuscate_metadata_to(
input: &Path, input: &Path,
dest: &Path, dest: &Path,
verbose: bool, verbose: bool,
) -> anyhow::Result<crate::metadata::Report> { ) -> anyhow::Result<senbei_metadata::Report> {
let data = std::fs::read(input)?; let data = std::fs::read(input)?;
// Preserve the metadata::Error in the chain (rather than stringifying it) // Preserve the metadata::Error in the chain (rather than stringifying it)
// so the folder driver can apply its unsupported-version policy. // so the folder driver can apply its unsupported-version policy.
let (out, report) = crate::metadata::deobfuscate(&data) let (out, report) = senbei_metadata::deobfuscate(&data)
.map_err(|e| anyhow::Error::new(e).context(format!("{input:?}")))?; .map_err(|e| anyhow::Error::new(e).context(format!("{input:?}")))?;
if report.remapped > 0 { if report.remapped > 0 {
if let Some(parent) = dest.parent() { if let Some(parent) = dest.parent() {
+2 -2
View File
@@ -1,7 +1,7 @@
//! Filesystem and command-line orchestration.
pub mod job; pub mod job;
pub mod logfile; pub mod logfile;
pub mod metadata;
pub mod pause; pub mod pause;
pub mod scan; pub mod scan;
pub mod ui; pub mod ui;
pub mod unpacker;
+47 -22
View File
@@ -1,4 +1,4 @@
use crate::unpacker::detect; use senbei_pe::detect;
use std::io::Read; use std::io::Read;
use std::path::{Path, PathBuf}; use std::path::{Path, PathBuf};
use walkdir::WalkDir; use walkdir::WalkDir;
@@ -16,24 +16,23 @@ const DETECT_PREFIX: u64 = 8 * 1024;
/// Smallest file that can possibly be a target, so anything shorter is skipped /// Smallest file that can possibly be a target, so anything shorter is skipped
/// without ever being opened. /// without ever being opened.
/// ///
/// A Crackproof module needs ≥ 4128 bytes for [`crate::unpacker::detect`]'s key /// A Crackproof module needs ≥ 4128 bytes for [`senbei_pe::detect`]'s key
/// table (it reads the dword at 4124), so the bound is exact for the unpack /// table (it reads the dword at 4124), so the bound is exact for the unpack
/// path. An il2cpp `global-metadata.dat` only needs 4 bytes to match its magic, /// path. An il2cpp `global-metadata.dat` only needs 4 bytes to match its magic,
/// but its header alone runs to offset 0xB0 and the images/types/methods tables /// but its header alone runs to offset 0xB0 and the images/types/methods tables
/// it indexes make every real one megabytes long — a sub-4 KiB "metadata" could /// it indexes make every real one megabytes long — a sub-4 KiB "metadata" could
/// only ever fail [`crate::metadata::deobfuscate`] with `Malformed`, so nothing /// only ever fail [`senbei_metadata::deobfuscate`] with `Malformed`, so nothing
/// processable is lost. /// processable is lost.
const MIN_SIZE: u64 = 4128; const MIN_SIZE: u64 = 4128;
/// File extensions that are bulk data by construction and can never be a PE /// File extensions that are bulk data by construction and can never be a PE
/// image or an il2cpp metadata blob. /// image or an il2cpp metadata blob.
/// ///
/// This is deliberately a **deny**-list, not an allow-list: the default is to /// This is deliberately a **deny**-list, not an executable allow-list: unknown
/// probe, so anything unrecognised is still opened. Targets are recognised by /// extensions are still probed. Extensionless files are handled separately by
/// content, not extension, and can carry arbitrary names — there is no closed /// [`denied_name`] because asset stores commonly contain tens of thousands of
/// set of target extensions an allow-list of `exe`/`dll` could enumerate. /// extensionless chunks; exhaustive probing remains available through
/// Only extensions that are bulk asset or text formats by construction appear /// `--scan-all`.
/// here.
/// ///
/// Set `SENBEI_SCAN_ALL=1` (or pass `--scan-all`) to probe every file regardless. /// Set `SENBEI_SCAN_ALL=1` (or pass `--scan-all`) to probe every file regardless.
const DENY_EXT: &[&str] = &[ const DENY_EXT: &[&str] = &[
@@ -90,11 +89,12 @@ const DENY_EXT: &[&str] = &[
"sr", "sr",
]; ];
/// Whether `path`'s extension is on [`DENY_EXT`]. Extensionless files are never /// Whether `path` can be skipped from its name alone. Extensionless files and
/// denied (they could be anything). /// files whose extension is on [`DENY_EXT`] are not opened during a default
fn denied_ext(path: &Path) -> bool { /// scan. `--scan-all` remains available when exhaustive probing is required.
fn denied_name(path: &Path) -> bool {
let Some(ext) = path.extension() else { let Some(ext) = path.extension() else {
return false; return true;
}; };
let Some(ext) = ext.to_str() else { let Some(ext) = ext.to_str() else {
return false; return false;
@@ -206,12 +206,17 @@ pub fn find_targets_opts(root: &Path, scan_all: bool) -> (Vec<PathBuf>, Vec<Path
continue; continue;
} }
if !scan_all { if !scan_all {
// Name checks come first so extensionless asset chunks never
// trigger even an explicit metadata query.
if denied_name(entry.path()) {
continue;
}
// Skip on directory metadata alone — never open these. // Skip on directory metadata alone — never open these.
let too_small = entry let too_small = entry
.metadata() .metadata()
.map(|m| m.len() < MIN_SIZE) .map(|m| m.len() < MIN_SIZE)
.unwrap_or(false); .unwrap_or(false);
if too_small || denied_ext(entry.path()) { if too_small {
continue; continue;
} }
} }
@@ -223,7 +228,7 @@ pub fn find_targets_opts(root: &Path, scan_all: bool) -> (Vec<PathBuf>, Vec<Path
// `Some(Class::None)` means "probed, matched neither detector". // `Some(Class::None)` means "probed, matched neither detector".
let n = paths.len(); let n = paths.len();
let mut class: Vec<Option<Class>> = vec![Some(Class::None); n]; let mut class: Vec<Option<Class>> = vec![Some(Class::None); n];
let workers = crate::unpacker::parallel::thread_cap().clamp(1, n.max(1)); let workers = senbei_pe::thread_cap().clamp(1, n.max(1));
if workers <= 1 { if workers <= 1 {
for (p, c) in paths.iter().zip(class.iter_mut()) { for (p, c) in paths.iter().zip(class.iter_mut()) {
*c = classify(p); *c = classify(p);
@@ -304,7 +309,7 @@ fn classify(path: &Path) -> Option<Class> {
let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
if detect(&head).is_some() { if detect(&head).is_some() {
Class::Crackproof Class::Crackproof
} else if crate::metadata::is_metadata(&head) { } else if senbei_metadata::is_metadata(&head) {
Class::Metadata Class::Metadata
} else { } else {
Class::None Class::None
@@ -329,29 +334,49 @@ mod tests {
#[test] #[test]
fn denies_bulk_asset_extensions_case_insensitively() { fn denies_bulk_asset_extensions_case_insensitively() {
for p in ["a.ab", "a.XML", "a.Acb", "a.ma2", "a.manifest", "a.PNG"] { for p in ["a.ab", "a.XML", "a.Acb", "a.ma2", "a.manifest", "a.PNG"] {
assert!(denied_ext(Path::new(p)), "{p} should be denied"); assert!(denied_name(Path::new(p)), "{p} should be denied");
}
}
#[test]
fn denies_extensionless_files() {
for p in ["asset", "level0", "0123456789abcdef"] {
assert!(denied_name(Path::new(p)), "{p} should be denied");
} }
} }
#[test] #[test]
fn never_denies_what_a_target_can_be_named() { fn never_denies_what_a_target_can_be_named() {
// Targets are recognised by content, not name — a protected module // Unknown extensions must still be probed. This keeps the filter a
// can carry any extension, or none — so names like these must always // narrow deny-list rather than an executable-extension allow-list.
// be probed. An allow-list would have skipped them.
for p in [ for p in [
"app.exe.bak", "app.exe.bak",
"managed.dll.bak", "managed.dll.bak",
"daemon.exe", "daemon.exe",
"GameLib.dll", "GameLib.dll",
"global-metadata.dat", "global-metadata.dat",
"noextension",
"a.so", "a.so",
"a.bin", "a.bin",
] { ] {
assert!(!denied_ext(Path::new(p)), "{p} must still be probed"); assert!(!denied_name(Path::new(p)), "{p} must still be probed");
} }
} }
#[test]
fn extensionless_targets_require_exhaustive_scan() {
let td = tempfile::tempdir().unwrap();
let root = td.path();
let mut blob = vec![0u8; MIN_SIZE as usize + 1];
blob[..4].copy_from_slice(&0xFAB1_1BAFu32.to_le_bytes());
std::fs::write(root.join("metadata"), &blob).unwrap();
let (_, filtered, _) = find_targets_opts(root, false);
assert!(filtered.is_empty());
let (_, exhaustive, _) = find_targets_opts(root, true);
assert_eq!(exhaustive.len(), 1);
}
/// A file below the Crackproof key-table bound is skipped without being /// A file below the Crackproof key-table bound is skipped without being
/// opened, but a large non-asset file is still probed. /// opened, but a large non-asset file is still probed.
#[test] #[test]
+1 -1
View File
@@ -1,6 +1,6 @@
use crate::unpacker::{IntegrityReport, Kind};
use indicatif::{ProgressBar, ProgressStyle}; use indicatif::{ProgressBar, ProgressStyle};
use owo_colors::OwoColorize; use owo_colors::OwoColorize;
use senbei_pe::{IntegrityReport, Kind};
use std::path::Path; use std::path::Path;
/// Create a progress bar for `n` items. Hidden when `quiet` is true. /// Create a progress bar for `n` items. Hidden when `quiet` is true.
+6
View File
@@ -0,0 +1,6 @@
[package]
name = "senbei-metadata"
version.workspace = true
edition.workspace = true
license.workspace = true
description = "Unity il2cpp metadata de-obfuscation for Senbei"
+5
View File
@@ -0,0 +1,5 @@
//! Unity il2cpp metadata de-obfuscation.
mod metadata;
pub use metadata::*;
+10
View File
@@ -0,0 +1,10 @@
[package]
name = "senbei-pe"
version.workspace = true
edition.workspace = true
license.workspace = true
description = "PE detection, unpacking, and validation for Senbei"
[dependencies]
senbei-crypto.workspace = true
thiserror.workspace = true
+3
View File
@@ -0,0 +1,3 @@
mod pipeline;
pub use pipeline::*;
@@ -13,9 +13,12 @@
//! CalculateChecksumWithSizeXor -> primitives::calculate_checksum //! CalculateChecksumWithSizeXor -> primitives::calculate_checksum
//! CalculateCrc32 -> crc32::compute (via above) //! CalculateCrc32 -> crc32::compute (via above)
use super::UnpackError; use super::super::{
use super::bytecode::{Op, OpsLut, generate}; BufferOperation, BytecodeStage, DecompressionStage, DescriptorTable, SectionPipeline,
use super::primitives::{self, *}; UnpackError,
};
use senbei_crypto::bytecode::{Op, OpsLut, generate};
use senbei_crypto::primitives::{self, *};
/// Read a signed 32-bit little-endian value. /// Read a signed 32-bit little-endian value.
fn get_i32(d: &[u8], offset: i32) -> i32 { fn get_i32(d: &[u8], offset: i32) -> i32 {
@@ -68,6 +71,7 @@ fn decrypt_data4(
key: i32, key: i32,
decomp_params: &[i32; 4], decomp_params: &[i32; 4],
transform: Option<&[Op]>, transform: Option<&[Op]>,
stage: DecompressionStage,
) -> Result<(), UnpackError> { ) -> Result<(), UnpackError> {
let addr = get_i32(d, offset); let addr = get_i32(d, offset);
let size = get_i32(d, offset + 4); let size = get_i32(d, offset + 4);
@@ -84,19 +88,17 @@ fn decrypt_data4(
OpsLut::new(ops).map_region(d, addr as usize, size as usize); OpsLut::new(ops).map_region(d, addr as usize, size as usize);
} }
if size != decompressed_size { if size != decompressed_size
// decompress reports corruption (after partial writes) via its bool; && let Err(reason) = primitives::decompress_detailed(
// surface it instead of shipping a garbage block.
if !decompress(
d, d,
addr as u32, addr as u32,
compressed_addr as u32, compressed_addr as u32,
decomp_params[1] as u32, decomp_params[1] as u32,
size as u32, size as u32,
decompressed_size as u32, decompressed_size as u32,
) { )
return Err(UnpackError::DecompressFailed); {
} return Err(UnpackError::StageDecompressionFailed { stage, reason });
} }
Ok(()) Ok(())
} }
@@ -218,7 +220,11 @@ fn decrypt_and_decompress_data(
// Guard: need 16 bytes at section_data_offset in `d` // Guard: need 16 bytes at section_data_offset in `d`
let off = section_data_offset as usize; let off = section_data_offset as usize;
if off.saturating_add(16) > d.len() { if off.saturating_add(16) > d.len() {
return Err(UnpackError::OutOfBounds(off)); return Err(UnpackError::DescriptorOutOfBounds {
table: DescriptorTable::DllSectionBlocks,
offset: off,
image_len: d.len(),
});
} }
decrypt_data6_shift6(d, section_data_offset, 16); decrypt_data6_shift6(d, section_data_offset, 16);
let dest_offset = get_i32(d, section_data_offset); let dest_offset = get_i32(d, section_data_offset);
@@ -246,10 +252,10 @@ fn decrypt_and_decompress_data(
let lut = OpsLut::new(decrypt_func); let lut = OpsLut::new(decrypt_func);
let ko0 = decomp_params[0]; let ko0 = decomp_params[0];
let ko2 = decomp_params[2]; let ko2 = decomp_params[2];
let ks_snap = let ks_snap = primitives::aes_schedule_snapshot(d, ko2 as u32)
primitives::aes_schedule_snapshot(d, ko2 as u32).ok_or(UnpackError::Corrupt)?; .ok_or(UnpackError::InvalidAesKeySchedule { offset: ko2 as u32 })?;
let tab_snap = primitives::huffman_table_snapshot(d, ko0 as u32) let tab_snap = primitives::huffman_table_snapshot(d, ko0 as u32)
.ok_or(UnpackError::DecompressFailed)?; .ok_or(UnpackError::InvalidHuffmanTable { offset: ko0 as u32 })?;
let spans: Vec<(usize, usize)> = blocks let spans: Vec<(usize, usize)> = blocks
.iter() .iter()
.map(|b| { .map(|b| {
@@ -277,12 +283,15 @@ fn decrypt_and_decompress_data(
b.size as u32, b.size as u32,
b.expected_crc as u32, b.expected_crc as u32,
) { ) {
return Err(UnpackError::DecompressFailed); return Err(UnpackError::SectionDecompressionFailed {
pipeline: SectionPipeline::Dll,
block: i,
});
} }
} }
Ok(()) Ok(())
}; };
super::parallel::parallel_for(d, &spans, 1, do_block)?; super::super::parallel::parallel_for(d, &spans, 1, do_block)?;
} }
// Zero-fill loop. // Zero-fill loop.
@@ -292,7 +301,11 @@ fn decrypt_and_decompress_data(
// decrypts 16 too, so guard 16 (an 8-byte guard would let // decrypts 16 too, so guard 16 (an 8-byte guard would let
// decrypt_data6_shift6 index past the end of a truncated descriptor). // decrypt_data6_shift6 index past the end of a truncated descriptor).
if off.saturating_add(16) > d.len() { if off.saturating_add(16) > d.len() {
return Err(UnpackError::OutOfBounds(off)); return Err(UnpackError::DescriptorOutOfBounds {
table: DescriptorTable::DllZeroFill,
offset: off,
image_len: d.len(),
});
} }
decrypt_data6_shift6(d, section_data_offset, 16); decrypt_data6_shift6(d, section_data_offset, 16);
let zero_offset = get_i32(d, section_data_offset); let zero_offset = get_i32(d, section_data_offset);
@@ -305,7 +318,12 @@ fn decrypt_and_decompress_data(
for i in 0..zero_size { for i in 0..zero_size {
let idx = (zero_offset + i) as usize; let idx = (zero_offset + i) as usize;
if idx >= d.len() { if idx >= d.len() {
return Err(UnpackError::OutOfBounds(idx)); return Err(UnpackError::BufferRangeOutOfBounds {
operation: BufferOperation::ZeroFill,
offset: idx,
size: 1,
buffer_len: d.len(),
});
} }
d[idx] = 0; d[idx] = 0;
} }
@@ -324,12 +342,16 @@ pub fn unpack_dll(input: &[u8]) -> Result<Vec<u8>, UnpackError> {
pub fn unpack_dll_v(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError> { pub fn unpack_dll_v(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError> {
// Trap any out-of-bounds panic from a truncated/garbled file and report it // Trap any out-of-bounds panic from a truncated/garbled file and report it
// as a clean error so the public API stays panic-free. // as a clean error so the public API stays panic-free.
super::catch_unpack(move || unpack_dll_inner(input, verbose)) super::super::catch_unpack(move || unpack_dll_inner(input, verbose))
} }
fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError> { fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError> {
if input.len() < 4096 { const HEADER_LEN: usize = 4128;
return Err(UnpackError::InputTooShort(input.len())); if input.len() < HEADER_LEN {
return Err(UnpackError::InputTooShort {
actual: input.len(),
required: HEADER_LEN,
});
} }
// `file_data` and `original_file_data` both borrow the same protected input. // `file_data` and `original_file_data` both borrow the same protected input.
@@ -347,15 +369,18 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
println!(" keys[6] anchor = 0x{:08X}", keys[6] as u32); println!(" keys[6] anchor = 0x{:08X}", keys[6] as u32);
} }
if !super::is_supported_magic(keys[1] as u32) { if !super::super::is_supported_magic(keys[1] as u32) {
return Err(UnpackError::DllUnpack( return Err(UnpackError::HeaderMagicMismatch {
"Not a Crackproof protected file (KONN magic mismatch)".into(), found: keys[1] as u32,
)); });
} }
let pe_offset = get_i32(file_data, 60); let pe_offset = get_i32(file_data, 60);
if pe_offset < 0 || (pe_offset as usize).saturating_add(84) > file_data.len() { if pe_offset < 0 || (pe_offset as usize).saturating_add(84) > file_data.len() {
return Err(UnpackError::DllUnpack("implausible PE offset".into())); return Err(UnpackError::InvalidPeOffset {
offset: i64::from(pe_offset),
input_len: file_data.len(),
});
} }
// This pipeline is PE32+-only: its header fixups write the data // This pipeline is PE32+-only: its header fixups write the data
// directories at PE32+ offsets (pe+144..180, pe+136 for the DD blob). On a // directories at PE32+ offsets (pe+144..180, pe+136 for the DD blob). On a
@@ -363,14 +388,18 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
// structurally plausible but unloadable file. Reject early with a clear // structurally plausible but unloadable file. Reject early with a clear
// error so `unpack_auto`'s EXE-pipeline fallback handles PE32 DLLs (that // error so `unpack_auto`'s EXE-pipeline fallback handles PE32 DLLs (that
// path is PE32-aware — see run_pe32), instead of us mangling them here. // path is PE32-aware — see run_pe32), instead of us mangling them here.
if get_i32(file_data, pe_offset + 24) & 0xFFFF != 0x20B { let optional_magic = get_u16(file_data, (pe_offset + 24) as u32);
return Err(UnpackError::DllUnpack( if optional_magic != 0x20B {
"not a PE32+ image (the DLL pipeline handles 64-bit only)".into(), return Err(UnpackError::UnsupportedDllPeMagic {
)); found: optional_magic,
});
} }
let size_of_image = get_i32(file_data, pe_offset + 80); let size_of_image = get_i32(file_data, pe_offset + 80);
if size_of_image <= 0 || size_of_image as u64 > super::MAX_IMAGE_SIZE { if size_of_image <= 0 || size_of_image as u64 > super::super::MAX_IMAGE_SIZE {
return Err(UnpackError::DllUnpack("implausible SizeOfImage".into())); return Err(UnpackError::InvalidImageSize {
size: i64::from(size_of_image),
max: super::super::MAX_IMAGE_SIZE,
});
} }
let mut out = vec![0u8; size_of_image as usize]; let mut out = vec![0u8; size_of_image as usize];
let base_offset = keys[6] - keys[3] + 0x2000; let base_offset = keys[6] - keys[3] + 0x2000;
@@ -438,6 +467,16 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
println!(" checksum1 = 0x{:08X}", checksum1 as u32); println!(" checksum1 = 0x{:08X}", checksum1 as u32);
println!(" decrypted_addr1 = 0x{:08X}", decrypted_addr1 as u32); println!(" decrypted_addr1 = 0x{:08X}", decrypted_addr1 as u32);
} }
let primary_end = decrypted_addr1.checked_add(3856);
if decrypted_addr1 < keys[3]
|| primary_end.is_none_or(|end| end < 0 || end as usize > out.len())
{
return Err(UnpackError::InvalidDllPrimaryDescriptor {
address: decrypted_addr1 as u32,
minimum: keys[3] as u32,
image_len: out.len(),
});
}
let import_offset = get_i32(&out, decrypted_addr1 + 3444); let import_offset = get_i32(&out, decrypted_addr1 + 3444);
let decrypted_addr2_size = get_i32(&out, decrypted_addr1 + 3632); let decrypted_addr2_size = get_i32(&out, decrypted_addr1 + 3632);
decrypt_data3( decrypt_data3(
@@ -512,6 +551,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
table_val ^ checksum2 ^ (xor_accumulator as i32), table_val ^ checksum2 ^ (xor_accumulator as i32),
&decomp_params, &decomp_params,
None, None,
DecompressionStage::DllCodeBlock1,
)?; )?;
let addr3b = get_i32(&out, decrypted_addr1 + 3728); let addr3b = get_i32(&out, decrypted_addr1 + 3728);
@@ -527,7 +567,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
let crc_val = { let crc_val = {
let a = crc_data_addr as usize; let a = crc_data_addr as usize;
let n = crc_data_size as usize; let n = crc_data_size as usize;
super::crc32::compute(&out[a..a + n]) as i32 senbei_crypto::crc32::compute(&out[a..a + n]) as i32
}; };
let crc_xored = crc_data_size ^ crc_val; let crc_xored = crc_data_size ^ crc_val;
let trailing_val = get_i32(&out, crc_data_addr + crc_data_size - 4); let trailing_val = get_i32(&out, crc_data_addr + crc_data_size - 4);
@@ -537,6 +577,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
crc_xored ^ (xor_accumulator as i32) ^ trailing_val, crc_xored ^ (xor_accumulator as i32) ^ trailing_val,
&decomp_params, &decomp_params,
None, None,
DecompressionStage::DllCodeBlock2,
)?; )?;
let checksum3 = calculate_checksum(&out, (decrypted_addr1 + 3480) as u32) as i32; let checksum3 = calculate_checksum(&out, (decrypted_addr1 + 3480) as u32) as i32;
@@ -549,6 +590,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
(not_val ^ (xor_key as u32)) as i32, (not_val ^ (xor_key as u32)) as i32,
&decomp_params, &decomp_params,
None, None,
DecompressionStage::DllCodeBlock3,
)?; )?;
let addr4 = get_i32(&out, addr4_offset); let addr4 = get_i32(&out, addr4_offset);
@@ -574,8 +616,9 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
lfsr_seed_val = lfsr_seed_val.wrapping_add(k); lfsr_seed_val = lfsr_seed_val.wrapping_add(k);
} }
let decrypt_func = generate(&out, lfsr as u32) let decrypt_func = generate(&out, lfsr as u32).ok_or(UnpackError::BytecodeGenerationFailed(
.ok_or_else(|| UnpackError::DllUnpack("Failed to build decryption expression".into()))?; BytecodeStage::DllPrimaryDecryptor,
))?;
let addr5_offset = decrypted_addr1 + 3840; let addr5_offset = decrypted_addr1 + 3840;
let addr5 = get_i32(&out, addr5_offset); let addr5 = get_i32(&out, addr5_offset);
@@ -585,6 +628,7 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
lfsr_seed_val ^ xor_key ^ checksum4, lfsr_seed_val ^ xor_key ^ checksum4,
&decomp_params, &decomp_params,
Some(&decrypt_func), Some(&decrypt_func),
DecompressionStage::DllCodeBlock4,
)?; )?;
if verbose { if verbose {
println!("[7/9] Decrypting code block 4 (addr5)..."); println!("[7/9] Decrypting code block 4 (addr5)...");
@@ -603,9 +647,9 @@ fn unpack_dll_inner(input: &[u8], verbose: bool) -> Result<Vec<u8>, UnpackError>
let lfsr2 = metadata_offset + 88; let lfsr2 = metadata_offset + 88;
decrypt_data6(&mut out, lfsr2 as u32); decrypt_data6(&mut out, lfsr2 as u32);
let decrypt_func2 = generate(&out, lfsr2 as u32).ok_or_else(|| { let decrypt_func2 = generate(&out, lfsr2 as u32).ok_or(
UnpackError::DllUnpack("Failed to build second decryption expression".into()) UnpackError::BytecodeGenerationFailed(BytecodeStage::DllSectionDecryptor),
})?; )?;
let section_image_base = 4095 - get_i32(original_file_data, 4224); let section_image_base = 4095 - get_i32(original_file_data, 4224);
let section_data_offset = get_i32(&out, addr5 + 11976); let section_data_offset = get_i32(&out, addr5 + 11976);
+253
View File
@@ -0,0 +1,253 @@
pub use senbei_crypto::{BufferOperation, DecompressionFailure};
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum DecompressionStage {
ExeStage3,
ExeStage3Secondary,
ExeStage4,
ExeStage5,
Pe32FourthStage,
Pe32FifthStage,
Pe32SeventhStage,
DllCodeBlock1,
DllCodeBlock2,
DllCodeBlock3,
DllCodeBlock4,
}
impl std::fmt::Display for DecompressionStage {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(match self {
Self::ExeStage3 => "EXE stage3",
Self::ExeStage3Secondary => "EXE secondary stage3",
Self::ExeStage4 => "EXE stage4",
Self::ExeStage5 => "EXE stage5",
Self::Pe32FourthStage => "PE32 fourth stage",
Self::Pe32FifthStage => "PE32 fifth stage",
Self::Pe32SeventhStage => "PE32 seventh stage",
Self::DllCodeBlock1 => "DLL code block 1",
Self::DllCodeBlock2 => "DLL code block 2",
Self::DllCodeBlock3 => "DLL code block 3",
Self::DllCodeBlock4 => "DLL code block 4",
})
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum BytecodeStage {
ExeStage4,
ExeStage5,
Pe32CustomDecryptor,
Pe32FileDecryptor,
DllPrimaryDecryptor,
DllSectionDecryptor,
}
impl std::fmt::Display for BytecodeStage {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(match self {
Self::ExeStage4 => "EXE stage4",
Self::ExeStage5 => "EXE stage5",
Self::Pe32CustomDecryptor => "PE32 custom decryptor",
Self::Pe32FileDecryptor => "PE32 file decryptor",
Self::DllPrimaryDecryptor => "DLL primary decryptor",
Self::DllSectionDecryptor => "DLL section decryptor",
})
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum SectionPipeline {
ExePe32Plus,
ExePe32,
Dll,
}
impl std::fmt::Display for SectionPipeline {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(match self {
Self::ExePe32Plus => "PE32+ EXE",
Self::ExePe32 => "PE32 EXE",
Self::Dll => "DLL",
})
}
}
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum DescriptorTable {
DllSectionBlocks,
DllZeroFill,
}
impl std::fmt::Display for DescriptorTable {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
f.write_str(match self {
Self::DllSectionBlocks => "DLL section-block",
Self::DllZeroFill => "DLL zero-fill",
})
}
}
#[derive(Debug, Clone, PartialEq, Eq, thiserror::Error)]
#[non_exhaustive]
pub enum UnpackError {
#[error("input too short (need at least {required} bytes, got {actual})")]
InputTooShort { actual: usize, required: usize },
#[error("decrypted header magic mismatch (got 0x{found:08X})")]
HeaderMagicMismatch { found: u32 },
#[error("anchor field not found — corrupt data or wrong offset")]
AnchorNotFound,
#[error("stage1 descriptor not found near anchor 0x{anchor:08X}")]
Stage1DescriptorNotFound { anchor: u32 },
#[error("stage2 field not found — corrupt data or wrong offset")]
Stage2NotFound,
#[error("chk_src_start not found — corrupt data or wrong offset")]
ChkSrcStartNotFound,
#[error("table_start not found — corrupt data or wrong offset")]
TableStartNotFound,
#[error("{0} bytecode generation failed — corrupt data or wrong offset")]
BytecodeGenerationFailed(BytecodeStage),
#[error("stage5 marker not found — this build's layout is not supported by this unpacker")]
Stage5MarkerNotFound,
#[error("not a Crackproof-protected file")]
NotCrackproof,
#[error("invalid PE header offset {offset} for {input_len}-byte input")]
InvalidPeOffset { offset: i64, input_len: usize },
#[error("DLL pipeline requires PE32+ optional-header magic, got 0x{found:04X}")]
UnsupportedDllPeMagic { found: u16 },
#[error(
"DLL primary descriptor address 0x{address:08X} is below layout base 0x{minimum:08X} or outside {image_len}-byte image"
)]
InvalidDllPrimaryDescriptor {
address: u32,
minimum: u32,
image_len: usize,
},
#[error("invalid SizeOfImage {size}; expected 1..={max}")]
InvalidImageSize { size: i64, max: u64 },
#[error(
"{operation} range out of bounds (offset {offset}, size {size}, buffer length {buffer_len})"
)]
BufferRangeOutOfBounds {
operation: BufferOperation,
offset: usize,
size: usize,
buffer_len: usize,
},
#[error(
"EXE checksum descriptor at 0x{descriptor:08X} points outside input (offset {offset}, size {size}, input length {image_len})"
)]
ExeChecksumRangeOutOfBounds {
descriptor: u32,
offset: usize,
size: usize,
image_len: usize,
},
#[error(
"{table} descriptor out of bounds (offset {offset}, size 16, image length {image_len})"
)]
DescriptorOutOfBounds {
table: DescriptorTable,
offset: usize,
image_len: usize,
},
#[error("PE32 tbl not found — corrupt data or wrong offset")]
Pe32TblNotFound,
#[error("PE32 thirdStage decrypt failed — corrupt data or wrong offset")]
Pe32ThirdStageFailed,
#[error("PE32 customDecryptor not found in sevenStage")]
Pe32CustomDecryptorNotFound,
#[error("PE32 eighthStageKey not found")]
Pe32EighthKeyNotFound,
#[error("PE32 file LFSR not found in eighthStage")]
Pe32FileLfsrNotFound,
#[error("{stage} decompression failed: {reason}")]
StageDecompressionFailed {
stage: DecompressionStage,
reason: DecompressionFailure,
},
#[error("{pipeline} section block {block} decompression failed")]
SectionDecompressionFailed {
pipeline: SectionPipeline,
block: usize,
},
#[error("AES key schedule is outside the image at offset {offset}")]
InvalidAesKeySchedule { offset: u32 },
#[error("Huffman table is outside the image at offset {offset}")]
InvalidHuffmanTable { offset: u32 },
#[error("DLL pipeline failed: {dll}; EXE fallback failed: {exe}")]
PipelineFallbackFailed {
dll: Box<UnpackError>,
exe: Box<UnpackError>,
},
#[error(
"PE32 second-stage range is invalid (offset {offset}, size {size}, image length {image_len})"
)]
Pe32SecondStageRangeInvalid {
offset: u32,
size: u32,
image_len: usize,
},
#[error("PE32 relocation-data descriptor not found")]
Pe32RelocationDataNotFound,
#[error("file decryptor candidate failed structural validation")]
FileDecryptorValidationFailed,
#[error("PE32 memory image could not be rebuilt as a file-layout PE")]
Pe32OutputLayoutInvalid,
#[error("internal panic at {file}:{line}:{column}: {message}")]
InternalPanic {
message: String,
file: String,
line: u32,
column: u32,
},
}
impl From<senbei_crypto::Error> for UnpackError {
fn from(error: senbei_crypto::Error) -> Self {
match error {
senbei_crypto::Error::BufferRangeOutOfBounds {
operation,
offset,
size,
buffer_len,
} => Self::BufferRangeOutOfBounds {
operation,
offset,
size,
buffer_len,
},
}
}
}
+3
View File
@@ -0,0 +1,3 @@
mod pipeline;
pub use pipeline::*;
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -77,6 +77,49 @@ fn rva_to_off(secs: &[Section], file_len: usize, rva: u32, need: u32) -> Option<
None None
} }
fn is_executable_rva(secs: &[Section], rva: u32) -> bool {
secs.iter().any(|section| {
let span = section.vsize.max(section.raw_size);
rva >= section.va
&& rva < section.va.wrapping_add(span)
&& (section.chars & 0x2000_0000) != 0
})
}
fn check_common_entry_branches(
stub: &[u8],
ep: u32,
secs: &[Section],
report: &mut IntegrityReport,
) {
if stub.len() < 18
|| stub[0..3] != [0x48, 0x83, 0xEC]
|| stub[4] != 0xE8
|| stub[9..12] != [0x48, 0x83, 0xC4]
|| stub[12] != stub[3]
|| stub[13] != 0xE9
{
return;
}
for (name, rel_off, instruction_len) in [("call", 5usize, 9i64), ("jump", 14usize, 18i64)] {
let rel = i32::from_le_bytes([
stub[rel_off],
stub[rel_off + 1],
stub[rel_off + 2],
stub[rel_off + 3],
]) as i64;
let target = i64::from(ep) + instruction_len + rel;
let valid = u32::try_from(target)
.ok()
.is_some_and(|rva| is_executable_rva(secs, rva));
if !valid {
report.issues.push(format!(
"entry point {name} target 0x{target:X} is outside executable sections (DD8 selection is likely wrong)"
));
}
}
}
/// Inspect an unpacked PE image and report any defect that would make the OS /// Inspect an unpacked PE image and report any defect that would make the OS
/// loader fault at runtime. `out` is the bytes the unpacker produced. /// loader fault at runtime. `out` is the bytes the unpacker produced.
pub fn check(out: &[u8]) -> IntegrityReport { pub fn check(out: &[u8]) -> IntegrityReport {
@@ -253,6 +296,9 @@ pub fn check(out: &[u8]) -> IntegrityReport {
"entry point RVA 0x{ep:X} is not in an executable section" "entry point RVA 0x{ep:X} is not in an executable section"
)); ));
} }
if let Some(entry_stub) = out.get(off as usize..off as usize + 18) {
check_common_entry_branches(entry_stub, ep, &secs, &mut r);
}
} }
} }
} }
@@ -363,3 +409,40 @@ fn looks_like_dll_name(d: &[u8], off: u32) -> bool {
} }
d[start..end].iter().all(|&b| (0x20..0x7F).contains(&b)) d[start..end].iter().all(|&b| (0x20..0x7F).contains(&b))
} }
#[cfg(test)]
mod tests {
use super::*;
fn executable_text() -> Vec<Section> {
vec![Section {
va: 0x1000,
vsize: 0x4000,
raw_ptr: 0x1000,
raw_size: 0x4000,
chars: 0x6000_0020,
}]
}
#[test]
fn common_entry_stub_rejects_out_of_image_branches() {
let stub = [
0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x41, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9,
0x7A, 0xFE, 0x54, 0xFF,
];
let mut report = IntegrityReport::default();
check_common_entry_branches(&stub, 0x1264, &executable_text(), &mut report);
assert_eq!(report.issues.len(), 2);
}
#[test]
fn common_entry_stub_accepts_executable_branches() {
let stub = [
0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x00, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9,
0x7A, 0xFE, 0xFF, 0xFF,
];
let mut report = IntegrityReport::default();
check_common_entry_branches(&stub, 0x1264, &executable_text(), &mut report);
assert!(report.ok());
}
}
+14
View File
@@ -0,0 +1,14 @@
//! Internal PE layout discovery and image reconstruction.
mod dd8;
mod discovery;
mod image;
pub(super) use dd8::{select_dd8_formula_pe32, select_dd8_shift};
pub(super) use discovery::{
discover_eighth_slots, find_bytecode_offset, find_lfsr_block, find_str_pos, find_tbl_pe32,
find_v_after_pad, find_v4_offset, get_string_to_null, section_name, trial_decrypt5_u32,
};
pub(super) use image::{
compact_memory_image_to_pe, move_pe32_imports_to_kmiat, pe32_imports_already_match_idata_layout,
};
+609
View File
@@ -0,0 +1,609 @@
//! Validation-driven selection for per-page text transforms.
use super::discovery::trial_decrypt5_u32;
/// PE32 `.text` dd8 key-formula selection with a skip decision. The packer keys
/// the per-page XOR either with `page+1` or `0x8000*(page+1)`; the formula is
/// not recorded. Replays the dd8 page pass on a scratch copy of sample pages
/// (25/50/75% of `.text`) under each formula and counts how many positions
/// decode to `0xCC` (int3 padding).
///
/// Returns `Some(true)` for the `0x8000*(page+1)` formula, `Some(false)` for
/// `page+1`, or `None` when `.text` must NOT be dd8-decrypted at all. The packer
/// dd8-encrypts `.text` on EXEs (so unpacking must replay it) but leaves a native
/// DLL's `.text` plaintext; replaying dd8 there scrambles ~1 byte per 16-byte
/// block. The decision: dd8 only *restores* int3 padding when `.text` was
/// genuinely encrypted, so apply it only when the chosen formula's whole-page
/// 0xCC count rises *clearly* above the no-dd8 baseline; otherwise skip.
///
/// "Clearly" matters: dd8 XORs 255 positions per page with pseudo-random bytes,
/// so on an already-plaintext `.text` it manufactures ~1 spurious `0xCC` per
/// sampled page for free (255/256 expected). A bare `best > baseline` test is
/// therefore biased towards *applying* dd8 on exactly the inputs that must skip
/// it — and a wrongly-applied dd8 is silent: it scrambles ~1 byte per 16 with no
/// error and nothing downstream (not even `integrity::check`, which only reads
/// 16 bytes at the entry point) notices. The [`MIN_DD8_NET_GAIN`] floor below is
/// the PE32 counterpart of the margin+floor `select_dd8_shift` already applies
/// on PE32+ for the same failure mode.
pub fn select_dd8_formula_pe32(data: &[u8], text_off: u32, text_size: u32) -> Option<bool> {
let num_pages_total = text_size / 0x1000;
let mut sample_pages: Vec<u32> = Vec::new();
for frac in [0.25f64, 0.5, 0.75] {
let pg = (num_pages_total as f64 * frac) as u32;
if pg > 0 && pg < num_pages_total {
sample_pages.push(pg);
}
}
if sample_pages.is_empty() && num_pages_total > 1 {
sample_pages.push(num_pages_total / 2);
}
let score = |big: bool| -> i64 {
let mut total = 0i64;
for &sp in &sample_pages {
let pg_off = (text_off + sp * 0x1000) as usize;
if pg_off + 0x1000 > data.len() {
continue;
}
let mut buf = [0u8; 0x1000];
buf.copy_from_slice(&data[pg_off..pg_off + 0x1000]);
let pk = if big {
0x8000u32.wrapping_mul(sp.wrapping_add(1))
} else {
sp.wrapping_add(1)
};
let mut k = pk;
let rk = k.rotate_right(15);
k = rk;
for bi in 1..256u32 {
let rk = k.rotate_right(15);
let ri = rk.wrapping_add(bi);
k = ri.wrapping_add(bi);
let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize;
if tidx < buf.len() {
buf[tidx] ^= k as u8;
}
}
total += buf.iter().filter(|&&b| b == 0xCC).count() as i64;
}
total
};
let s_small = score(false);
let s_big = score(true);
// Baseline: whole-page 0xCC over the same sample pages with NO dd8. dd8 only
// rewrites 255 bytes per page, so comparing the chosen formula's whole-page
// 0xCC against this baseline reveals whether dd8 *restores* int3 padding
// (count rises -> .text was packer-encrypted, apply) or merely scrambles
// already-plaintext code (count falls -> native-DLL .text left intact, skip).
let mut baseline: i64 = 0;
for &sp in &sample_pages {
let pg_off = (text_off + sp * 0x1000) as usize;
if pg_off + 0x1000 > data.len() {
continue;
}
baseline += data[pg_off..pg_off + 0x1000]
.iter()
.filter(|&&b| b == 0xCC)
.count() as i64;
}
let big = s_big > s_small;
let best = s_small.max(s_big);
// Minimum net 0xCC gain over the baseline before dd8 is applied. Noise on an
// already-plaintext `.text` is ~1 manufactured 0xCC per sampled page (3 pages
// -> ~3); every corpus build that genuinely needs dd8 gains +154 or more
// (observed +154 and +312), and the one native DLL that must skip scores -18.
// A floor of 32 sits ~10x above the noise and ~5x below the smallest true
// positive, so it changes no existing decision.
const MIN_DD8_NET_GAIN: i64 = 32;
let apply = best.saturating_sub(baseline) >= MIN_DD8_NET_GAIN;
if std::env::var("SEL_DIAG").is_ok() {
eprintln!(
"SEL pe32 dd8 s_small={} s_big={} baseline={} gain={} big={} apply={}",
s_small,
s_big,
baseline,
best - baseline,
big,
apply
);
}
// When no interior pages could be sampled (tiny .text) we cannot measure the
// effect; preserve the historical behavior of applying dd8.
if sample_pages.is_empty() || apply {
Some(big)
} else {
None
}
}
// ---------------------------------------------------------------------------
// dd8 page-XOR shift selection.
//
// The packer scrambles ~1 byte per 16-byte block of .text via decrypt_data8,
// keyed by `page_idx << shift` (absolute page index = text_va >> 12). Observed
// shifts are 0 and 15. The shift is NOT stored in any header/config field, so
// the decision must be validated against the resulting .text content.
//
// A recognised CRT entry stub is the strongest oracle: decode skip/0/15 and
// require both of its direct rel32 branches to land in executable .text. This
// includes the call/jump displacement bytes themselves; an older entry oracle
// wildcarded those bytes and could accept a stub whose opcodes looked right but
// whose branch targets were outside the image.
//
// Other entry shapes fall back to padding statistics over a few sample pages
// (head/tail margin skipped: entry/exit regions have atypical padding density).
// The primary signal is a *structural* fingerprint: the MSVC function-end
// padding pattern, a 0xC3 RET opcode followed by a run of >= 4 0xCC int3 bytes.
// dd8 XORs one pseudo-random byte per 16-byte block, so an already-plaintext
// page keeps its padding runs only under "no dd8", while a packer-encrypted
// page restores them only under the correct shift — a wrong candidate destroys
// every run it touches and essentially never manufactures a RET followed by a
// long int3 run by chance. This separates the states far more cleanly than a
// bare 0xCC count, which a wrong candidate inflates for free (~255 coincidences
// per page at p=1/256).
//
// When no candidate produces any RET-anchored padding (sampled pages with
// dense code and no padded epilogues), the fingerprint is silent, so the
// decision falls back to the older mutated-position 0xCC count. Both signals
// use the same decision rule: a candidate must beat the no-dd8 baseline by a
// clear margin AND an absolute floor, otherwise dd8 is skipped — a wrongly
// applied dd8 scrambles ~1 byte per 16 with no error surfaced downstream.
// ---------------------------------------------------------------------------
pub fn select_dd8_shift(data: &[u8], text_va: u32, text_size: u32, info3: u32) -> u32 {
if let Some((shift, scores)) = select_dd8_by_entry_stub(data, text_va, text_size, info3) {
if std::env::var("SEL_DIAG").is_ok() {
eprintln!(
"SEL dd8 entry best_shift={} none={} s0={} s15={}",
shift, scores[0], scores[1], scores[2]
);
}
return shift;
}
let num_pages_total = text_size >> 12;
// Fewer than two pages: nothing meaningful to sample; preserve the
// historical behavior (shift 0 — the dd8 loop is empty or single-page).
if num_pages_total < 2 {
return 0;
}
let text_off = text_va as usize;
// Sample up to 4 pages, skipping a head/tail margin (entry/exit regions
// have atypical padding density). Small .text: sample every page.
let mut sample_pages: Vec<u32> = Vec::new();
if num_pages_total <= 4 {
sample_pages.extend(0..num_pages_total);
} else {
let margin = (num_pages_total / 8).max(1);
let lo = margin;
let hi = num_pages_total - margin;
if hi <= lo {
sample_pages.extend(0..num_pages_total);
} else {
let step = ((hi - lo) / 4).max(1);
let mut i = 0;
while i < 4 {
let p = lo + i * step;
if p < num_pages_total {
sample_pages.push(p);
}
i += 1;
}
}
}
if sample_pages.is_empty() {
return 0;
}
let abs_base = text_va >> 12;
// Require a clear 2x margin over the already-plaintext baseline AND an
// absolute floor. The 2x test alone trips on noise when the counts are
// tiny: an external-companion DLL whose .text is already plaintext scores
// s15=4 vs none=1 — a spurious 4x — and gets dd8 wrongly applied,
// corrupting ~1 byte per 16. The floor rejects that noise while sitting
// far below every genuinely-encrypted build's score.
const MIN_DD8_HITS: u32 = 8;
let margin_pick = |none: u32, s0: u32, s15: u32| -> u32 {
let mut best_score = none;
let mut best_shift = 99u32; // 99 == skip dd8
for (shift, hits) in [(0u32, s0), (15u32, s15)] {
if hits > best_score {
best_score = hits;
best_shift = shift;
}
}
if best_shift != 99 && (best_score < none * 2 || best_score < MIN_DD8_HITS) {
best_shift = 99;
}
best_shift
};
// Primary: RET+int3 padding fingerprint. The fingerprint is diluted across
// the whole page (dd8 touches only 255 of 4096 bytes, so even an encrypted
// page keeps most of its padding runs), so instead of the fallback's 2x
// margin the gate is a *positive delta* over the no-dd8 baseline: on an
// already-plaintext .text each wrong shift destroys runs (scores below the
// baseline), while the correct shift on an encrypted page restores them
// (scores above it). The floor on the delta rejects noise-level gains.
let r_none = fingerprint_score(data, text_off, abs_base, &sample_pages, None);
let r0 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(0));
let r15 = fingerprint_score(data, text_off, abs_base, &sample_pages, Some(15));
// Fallback: mutated-position 0xCC count, for pages whose code has no
// RET-anchored padding at all (the fingerprint is silent there).
let (none_hits, s0, s15);
let best_shift = if r_none != 0 || r0 != 0 || r15 != 0 {
none_hits = 0;
s0 = 0;
s15 = 0;
let mut best_score = r_none;
let mut shift = 99u32;
for (s, score) in [(0u32, r0), (15u32, r15)] {
if score > best_score {
best_score = score;
shift = s;
}
}
if shift != 99 && best_score.saturating_sub(r_none) < MIN_DD8_HITS {
shift = 99;
}
shift
} else {
none_hits = score_dd8_baseline(data, text_off, &sample_pages);
s0 = score_dd8_shift(data, text_off, text_va, &sample_pages, 0);
s15 = score_dd8_shift(data, text_off, text_va, &sample_pages, 15);
margin_pick(none_hits, s0, s15)
};
if std::env::var("SEL_DIAG").is_ok() {
eprintln!(
"SEL dd8 best_shift={} fp=({},{},{}) cc=({},{},{}) samples={:?}",
best_shift, r_none, r0, r15, none_hits, s0, s15, sample_pages
);
}
best_shift
}
/// Minimum 0xCC run length after a RET for the run to count as MSVC
/// function-end padding.
const MIN_CC_RUN: u32 = 4;
/// Total length of MSVC function-end padding runs in a page: each 0xC3 byte
/// followed by >= [`MIN_CC_RUN`] 0xCC bytes contributes the run length.
fn ret_int3_score(page: &[u8]) -> u32 {
let mut total = 0u32;
let mut i = 0;
while i < page.len() {
if page[i] == 0xC3 {
let mut j = i + 1;
while j < page.len() && page[j] == 0xCC {
j += 1;
}
let run = (j - i - 1) as u32;
if run >= MIN_CC_RUN {
total += run;
}
i = j;
} else {
i += 1;
}
}
total
}
/// Replay the dd8 page-XOR in place on one sample page.
fn dd8_apply(buf: &mut [u8; 0x1000], abs_page: u32, shift: u32) {
let mut key = abs_page << shift;
for bi in 0..256u32 {
let mixed = key.rotate_right(15).wrapping_add(bi);
key = mixed.wrapping_add(bi);
// The packer's dd8 loop does not XOR block i=0 (see decrypt_data8).
if bi == 0 {
continue;
}
let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize;
buf[tidx] ^= key as u8;
}
}
/// Sum the RET+int3 fingerprint over the sample pages for one candidate
/// (`None` = the no-dd8 baseline, page as-is).
fn fingerprint_score(
data: &[u8],
text_off: usize,
abs_base: u32,
sample_pages: &[u32],
shift: Option<u32>,
) -> u32 {
let mut total = 0u32;
for &sp in sample_pages {
let pg_off = text_off + (sp as usize) * 0x1000;
if pg_off + 0x1000 > data.len() {
continue;
}
let mut page = [0u8; 0x1000];
page.copy_from_slice(&data[pg_off..pg_off + 0x1000]);
if let Some(sh) = shift {
dd8_apply(&mut page, abs_base.wrapping_add(sp), sh);
}
total += ret_int3_score(&page);
}
total
}
/// Select DD8 from the common CRT entry stub when its direct call and jump
/// provide a stronger oracle than sparse padding statistics. The candidate is
/// accepted only when it is the sole one whose two branch targets stay inside
/// `.text`; unrecognised entry code falls through to the padding selector.
fn select_dd8_by_entry_stub(
data: &[u8],
text_va: u32,
text_size: u32,
info3: u32,
) -> Option<(u32, [u8; 3])> {
for entry in entry_candidates(data, text_va, text_size, info3) {
let [Some(none), Some(s0), Some(s15)] = [None, Some(0), Some(15)]
.map(|shift| entry_stub_branch_score(data, text_va, text_size, entry, shift))
else {
continue;
};
let scores = [none, s0, s15];
let best = scores.iter().copied().max()?;
if best == 2 && scores.iter().filter(|&&score| score == best).count() == 1 {
let index = scores.iter().position(|&score| score == best)?;
return Some(([99, 0, 15][index], scores));
}
}
None
}
fn entry_candidates(data: &[u8], text_va: u32, text_size: u32, info3: u32) -> Vec<u32> {
let text_end = text_va.saturating_add(text_size);
let mut entries = Vec::with_capacity(3);
if let Some(pe) = read_u32(data, 0x3C)
&& let Some(entry) = pe.checked_add(40).and_then(|offset| read_u32(data, offset))
&& (text_va..text_end).contains(&entry)
{
entries.push(entry);
}
for metadata_off in [32u32, 64] {
let Some(end) = info3
.checked_add(metadata_off)
.and_then(|offset| offset.checked_add(8))
else {
continue;
};
if end as usize > data.len() {
continue;
}
let entry = trial_decrypt5_u32(data, info3 + metadata_off);
let image_base = trial_decrypt5_u32(data, info3 + metadata_off + 4);
if image_base == info3 && (text_va..text_end).contains(&entry) && !entries.contains(&entry)
{
entries.push(entry);
}
}
entries
}
fn entry_stub_branch_score(
data: &[u8],
text_va: u32,
text_size: u32,
entry: u32,
shift: Option<u32>,
) -> Option<u8> {
let text_end = text_va.checked_add(text_size)?;
if entry < text_va || entry.checked_add(18)? > text_end {
return None;
}
let mut stub = [0u8; 18];
for (offset, byte) in stub.iter_mut().enumerate() {
*byte = dd8_candidate_byte(data, entry + offset as u32, shift)?;
}
if stub[0..3] != [0x48, 0x83, 0xEC]
|| stub[4] != 0xE8
|| stub[9..12] != [0x48, 0x83, 0xC4]
|| stub[12] != stub[3]
|| stub[13] != 0xE9
{
return None;
}
let call_rel = i32::from_le_bytes(stub[5..9].try_into().ok()?) as i64;
let jump_rel = i32::from_le_bytes(stub[14..18].try_into().ok()?) as i64;
let call_target = i64::from(entry) + 9 + call_rel;
let jump_target = i64::from(entry) + 18 + jump_rel;
let in_text = |target: i64| target >= i64::from(text_va) && target < i64::from(text_end);
Some(u8::from(in_text(call_target)) + u8::from(in_text(jump_target)))
}
fn dd8_candidate_byte(data: &[u8], rva: u32, shift: Option<u32>) -> Option<u8> {
let mut byte = *data.get(rva as usize)?;
let Some(shift) = shift else {
return Some(byte);
};
let page = rva >> 12;
let block = (rva & 0xFFF) >> 4;
let mut key = page << shift;
for index in 0..=block {
let mixed = key.rotate_right(15).wrapping_add(index);
key = mixed.wrapping_add(index);
if index != 0 {
let target = (page << 12)
.wrapping_add(index << 4)
.wrapping_add(mixed & 0xF);
if target == rva {
byte ^= key as u8;
}
}
}
Some(byte)
}
fn read_u32(data: &[u8], offset: u32) -> Option<u32> {
let start = offset as usize;
let bytes = data.get(start..start.checked_add(4)?)?;
Some(u32::from_le_bytes(bytes.try_into().ok()?))
}
// Baseline: count int3 pads already present at the first byte of each 16-byte
// block, i.e. the positions dd8 would target if its in-block offset were 0.
fn score_dd8_baseline(data: &[u8], text_off: usize, sample_pages: &[u32]) -> u32 {
let mut hits = 0u32;
for &sp in sample_pages {
let pg_off = text_off + (sp as usize) * 0x1000;
if pg_off + 0x1000 > data.len() {
continue;
}
for bi in 1..256usize {
if data[pg_off + bi * 16] == 0xCC {
hits += 1;
}
}
}
hits
}
// Replay decrypt_data8 on each sample page under `shift` and count how many of
// the 255 mutated positions decode to 0xCC.
fn score_dd8_shift(
data: &[u8],
text_off: usize,
text_va: u32,
sample_pages: &[u32],
shift: u32,
) -> u32 {
let abs_base = text_va >> 12;
let mut hits = 0u32;
for &sp in sample_pages {
let pg_off = text_off + (sp as usize) * 0x1000;
if pg_off + 0x1000 > data.len() {
continue;
}
let abs_page = abs_base.wrapping_add(sp);
let mut key = abs_page << shift;
for bi in 0..256u32 {
let mixed = key.rotate_right(15).wrapping_add(bi);
key = mixed.wrapping_add(bi);
if bi == 0 {
continue;
}
let tidx = (bi.wrapping_mul(16).wrapping_add(mixed & 0xF)) as usize;
if tidx < 0x1000 {
let mutated = data[pg_off + tidx] ^ (key as u8);
if mutated == 0xCC {
hits += 1;
}
}
}
}
hits
}
#[cfg(test)]
mod tests {
use super::*;
fn entry_stub_fixture() -> Vec<u8> {
let mut data = vec![0u8; 0x5000];
data[0x3C..0x40].copy_from_slice(&0x100u32.to_le_bytes());
data[0x128..0x12C].copy_from_slice(&0x1264u32.to_le_bytes());
data[0x1264..0x1276].copy_from_slice(&[
0x48, 0x83, 0xEC, 0x28, 0xE8, 0x5B, 0x02, 0x00, 0x00, 0x48, 0x83, 0xC4, 0x28, 0xE9,
0x7A, 0xFE, 0xFF, 0xFF,
]);
data
}
fn apply_dd8_page(data: &mut [u8], page_rva: u32, shift: u32) {
let mut key = (page_rva >> 12) << shift;
for index in 0..256u32 {
let mixed = key.rotate_right(15).wrapping_add(index);
key = mixed.wrapping_add(index);
if index == 0 {
continue;
}
let target = page_rva.wrapping_add(index << 4).wrapping_add(mixed & 0xF) as usize;
data[target] ^= key as u8;
}
}
#[test]
fn entry_stub_selects_plaintext_and_both_dd8_shifts() {
let plain = entry_stub_fixture();
assert_eq!(select_dd8_shift(&plain, 0x1000, 0x4000, 0), 99);
for expected in [0u32, 15] {
let mut encrypted = plain.clone();
apply_dd8_page(&mut encrypted, 0x1000, expected);
assert_eq!(select_dd8_shift(&encrypted, 0x1000, 0x4000, 0), expected);
}
}
/// Seed the first `count` dd8-targeted positions of each sampled page with
/// the byte that decodes to `0xCC` under the `page+1` formula — i.e. an
/// encrypted `.text` whose plaintext is int3 padding. Positions whose key
/// byte would make the *ciphertext* itself `0xCC` are skipped so the
/// fixture contains no `0xCC` at all and every post-dd8 `0xCC` is a genuine
/// gain over a zero baseline.
fn seed_dd8_int3(data: &mut [u8], text_off: u32, pages: &[u32], count: u32) {
for &sp in pages {
let pg_off = (text_off + sp * 0x1000) as usize;
let mut k = sp.wrapping_add(1);
k = k.rotate_right(15);
let mut planted = 0u32;
for bi in 1..256u32 {
let ri = k.rotate_right(15).wrapping_add(bi);
k = ri.wrapping_add(bi);
if planted >= count {
continue;
}
let ct = 0xCCu8 ^ (k as u8);
if ct == 0xCC {
continue;
}
let tidx = (bi.wrapping_mul(16).wrapping_add(ri & 0xF)) as usize;
data[pg_off + tidx] = ct;
planted += 1;
}
}
}
/// Review regression: a near-plaintext `.text` must NOT be dd8-decrypted.
/// dd8 XORs 255 positions per page with pseudo-random bytes, so it
/// manufactures a few `0xCC` for free — under the old bare
/// `best > baseline` test any positive gain was enough to "apply" dd8 and
/// scramble ~1 byte per 16 of a native DLL's already-plaintext code,
/// silently (nothing downstream, including the integrity check, notices).
/// Here the gain is real but small; the floor must still reject it.
#[test]
fn pe32_dd8_skips_text_whose_gain_is_only_noise_sized() {
let text_off: u32 = 0x1000;
let text_size: u32 = 8 * 0x1000;
let mut data = vec![0u8; (text_off + text_size) as usize];
seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 5);
assert!(
!data.contains(&0xCC),
"fixture must have a zero 0xCC baseline"
);
assert_eq!(
select_dd8_formula_pe32(&data, text_off, text_size),
None,
"a gain this small is indistinguishable from dd8's own noise"
);
}
/// Control for the above: a `.text` whose dd8 pass restores a large amount
/// of int3 padding clears the floor and is decrypted. Same fixture shape,
/// only the amount of restored padding differs.
#[test]
fn pe32_dd8_applies_when_padding_is_restored() {
let text_off: u32 = 0x1000;
let text_size: u32 = 8 * 0x1000;
let mut data = vec![0u8; (text_off + text_size) as usize];
seed_dd8_int3(&mut data, text_off, &[2, 4, 6], 255);
assert_eq!(
select_dd8_formula_pe32(&data, text_off, text_size),
Some(false),
"encrypted .text must be decrypted with the page+1 formula"
);
}
}
+507
View File
@@ -0,0 +1,507 @@
//! Structural locators for protected PE stages.
use senbei_crypto::primitives::{get_u32, lfsr_keystream};
/// Find the 4-byte v_val that follows the LAST occurrence of `48 EB 01 B9`
/// (REX.W jmp+1; mov ecx,imm32) plus any 0xCC padding. Used to locate
/// stage4's accum2 seed. Works across builds even when API-name anchors are
/// absent.
pub fn find_v_after_pad(data: &[u8], base: u32, len: u32) -> Option<u32> {
let start = base as usize;
let end = (base.saturating_add(len)) as usize;
if end > data.len() {
return None;
}
let sig = [0x48u8, 0xEB, 0x01, 0xB9];
let slice = &data[start..end];
// last occurrence
let mut last = None;
let mut i = 0usize;
while i + sig.len() <= slice.len() {
if slice[i..i + sig.len()] == sig {
last = Some(i);
}
i += 1;
}
let pos = last?;
// skip CCs after the `48 EB 01 B9`
let mut after = pos + sig.len();
while after < slice.len() && slice[after] == 0xCC {
after += 1;
}
if after + 4 > slice.len() {
return None;
}
Some((start + after) as u32)
}
/// Predict the 4 bytes that DecryptData5(va, size) would produce at va+0..va+4
/// without mutating the buffer. The cipher's per-byte transform depends only
/// on the byte itself and the low 8 bits of (va+i), with no cross-byte state,
/// so each byte can be decrypted in isolation. Used to detect the EP/DD layout
/// offset before committing to the actual call.
pub fn trial_decrypt5_u32(data: &[u8], va: u32) -> u32 {
let mut out = [0u8; 4];
for i in 0..4u32 {
let b3 = data[(va + i) as usize];
let b = (va + i) as u8;
let b2 = b.wrapping_add(1);
let b4 = b3.rotate_left(2) ^ b2;
let b5 = b4.rotate_left(2) ^ b;
out[i as usize] = b5.rotate_left(2);
}
u32::from_le_bytes(out)
}
/// Scan stage4/stage5 for the encrypted custom-decryptor bytecode block. The
/// raw byte at p+95 is used by decrypt_data6 as the iteration count. We trial-
/// decrypt that many bytes with the LFSR keystream and accept the first
/// position where the byte stream parses as a valid opcode sequence ending in
/// 195 (ret).
pub fn find_bytecode_offset(data: &[u8], base: u32, len: u32) -> Option<u32> {
let start = base as usize;
let end = (base.saturating_add(len)) as usize;
if end > data.len() {
return None;
}
let mut ks = [0u8; 256];
lfsr_keystream(&mut ks);
// Scan forward from `start+16` on 16-byte boundaries relative to `start`.
// The bytecode block is positioned a fixed offset into stage4/stage5; the
// lowest parseable candidate is the real one (later ones are coincidental
// parses of trailing filler bytes that happen to map to valid opcodes).
// The enclosing buffer isn't necessarily 16-aligned to its absolute
// address in newer builds, so we anchor the stride to `start`.
let mut p = start + 16;
while p + 96 <= end {
let count = data[p + 95] as usize;
if count >= 8 && p + count <= end {
let mut buf = [0u8; 256];
let take = count.min(256);
for i in 0..take {
buf[i] = data[p + i] ^ ks[i];
}
if let Some(nops) = parse_bytecode_check(&buf[..take])
&& nops >= 4
{
return Some(p as u32);
}
}
p += 16;
}
None
}
/// Validate bytecode structure without allocating a `Vec` of ops. Returns
/// `Some(non_nop_op_count)` if the byte stream parses successfully as a valid
/// opcode sequence ending in 195 (ret), `None` otherwise. Allows non-trivial
/// bytecode filtering by op count.
pub fn parse_bytecode_check(buf: &[u8]) -> Option<usize> {
let mut i = 0usize;
let mut nops: usize = 0;
while i < buf.len() {
let b = buf[i];
i += 1;
match b {
4 | 44 | 52 => {
if i >= buf.len() {
return None;
}
i += 1;
nops += 1;
}
144 => {}
192 | 254 => {
if i >= buf.len() {
return None;
}
let mb = buf[i];
i += 1;
let rm = mb & 7;
let mod_ = (mb >> 6) & 3;
let reg = (mb >> 3) & 7;
if mod_ != 3 || rm != 0 {
return None;
}
if reg > 1 {
return None;
}
if b == 192 {
if i >= buf.len() {
return None;
}
i += 1;
}
nops += 1;
}
195 => return Some(nops),
_ => return None,
}
}
None
}
/// Locate stage3's v4_val: the last non-zero dword in the buffer, anchored
/// by the `C3 CC CC CC` (ret + 3 int3) immediately before it.
pub fn find_v4_offset(data: &[u8], base: u32, len: u32) -> Option<u32> {
let start = base as usize;
let end = (base.saturating_add(len)) as usize;
if end > data.len() || end < start + 4 {
return None;
}
// walk backwards looking for the first non-zero byte
let mut i = end;
while i > start && data[i - 1] == 0 {
i -= 1;
}
if i < start + 4 {
return None;
}
// v_val occupies the 4 bytes ending at i (rounded up to dword boundary)
let v_end = i;
let v_start = ((v_end + 3) & !3).saturating_sub(4);
// require that the 4 bytes preceding v_val match `C3 CC CC CC`
if v_start < start + 4 || data[v_start - 4..v_start] != [0xC3, 0xCC, 0xCC, 0xCC] {
return None;
}
Some(v_start as u32)
}
/// Scan a sub-buffer for an ASCII needle; return its absolute position.
pub fn find_str_pos(data: &[u8], base: u32, len: u32, needle: &[u8]) -> Option<u32> {
let start = base as usize;
let end = (base.saturating_add(len)) as usize;
if end > data.len() || needle.is_empty() {
return None;
}
data[start..end]
.windows(needle.len())
.position(|w| w == needle)
.map(|rel| (start + rel) as u32)
}
pub fn get_string_to_null(data: &[u8], offset: u32) -> String {
let start = offset as usize;
if start >= data.len() {
return String::new();
}
// Bounded: an unterminated run must never walk off the end of the buffer
// (panic) or scan unboundedly into unrelated data.
let limit = start.saturating_add(4096).min(data.len());
let mut i = start;
while i < limit && data[i] != 0 {
i += 1;
}
String::from_utf8_lossy(&data[start..i]).into_owned()
}
/// Read a PE section-name field: exactly 8 bytes, NOT necessarily
/// NUL-terminated (a full-width name like `.textbss` has no NUL at all).
/// Returns the name with trailing NULs stripped. Using `get_string_to_null`
/// here would run past the field into the VirtualSize/VirtualAddress dwords.
pub fn section_name(data: &[u8], offset: u32) -> String {
let start = offset as usize;
let Some(field) = data.get(start..start + 8) else {
return String::new();
};
let end = field.iter().position(|&b| b == 0).unwrap_or(8);
String::from_utf8_lossy(&field[..end]).into_owned()
}
// ---------------------------------------------------------------------------
// PE32 (32-bit) helpers
// ---------------------------------------------------------------------------
/// PE32 shell-table locator. Walks the shell region (`info[6]`) for a dword
/// equal to `info[6]` followed by a plausible shell size, returning the table
/// base (`candidate = off - 0x88`) when `candidate+0x58` holds a valid pointer.
pub fn find_tbl_pe32(data: &[u8], info: &[u32; 8]) -> Option<u32> {
let shell = info[6];
if (data.len() as u64) < 0x100 {
return None;
}
let hi = (shell as u64)
.saturating_add(0x3000)
.min(data.len() as u64 - 0x100) as u32;
let mut off = shell;
while off < hi {
if off as usize + 8 <= data.len() {
let candidate = off.wrapping_sub(0x88);
if candidate >= shell && get_u32(data, off) == info[6] {
let shell_size_val = get_u32(data, off.wrapping_add(4));
if shell_size_val > 0x1000 && shell_size_val < 0x100000 {
let v58_off = candidate.wrapping_add(0x58);
if (v58_off as usize + 4) <= data.len() {
let v58 = get_u32(data, v58_off);
if v58 > 0 && (v58 as usize) < data.len() {
return Some(candidate);
}
}
}
}
}
off = off.wrapping_add(4);
}
None
}
/// Locate an LFSR-encrypted bytecode block (decrypt_data6 form) in a region.
/// `start_off` is the byte offset to begin scanning at, `scan_backward`
/// controls direction. Returns the relative offset of the block. Includes full
/// opcode-walk validation of candidate blocks.
pub fn find_lfsr_block(
data: &[u8],
base: u32,
size: u32,
start_off: u32,
scan_backward: bool,
) -> Option<u32> {
if size < 96 {
return None;
}
let mut ks = [0u8; 128];
lfsr_keystream(&mut ks);
let check = |scan_off: u32| -> bool {
let abs_off = base.wrapping_add(scan_off) as usize;
if abs_off + 96 > data.len() {
return false;
}
let sz = data[abs_off + 95] as usize;
if !(10..=95).contains(&sz) {
return false;
}
let mut decoded = [0u8; 95];
for bi in 0..sz {
decoded[bi] = data[abs_off + bi] ^ ks[bi];
}
// Full bytecode validation (shared with the stage4/5 locator): every
// opcode must decode with a valid ModR/M and the stream must REACH a
// RET (0xC3) as an opcode. The previous check only required a 0xC3
// byte *anywhere* in the window and accepted a walk that ran off the
// end without hitting RET — a `0x04 0xC3` (ADD 0xC3) tail passed, so
// coincidental LFSR-shaped garbage was accepted as a decryptor block.
parse_bytecode_check(&decoded[..sz]).is_some()
};
if scan_backward {
let hi = size - 96;
if hi >= start_off {
let mut scan_off = hi;
loop {
if check(scan_off) {
return Some(scan_off);
}
if scan_off == start_off {
break;
}
scan_off -= 1;
}
}
} else {
let hi = size - 95;
let mut scan_off = start_off;
while scan_off < hi {
if check(scan_off) {
return Some(scan_off);
}
scan_off += 1;
}
}
None
}
/// Slots discovered in the eighthStage for the marker-less layout.
pub struct EighthSlots {
/// Absolute address of the file-data decryptor LFSR bytecode block. The
/// fileCS chain pointer is derived downstream as `file_lfsr - 0x58`.
pub file_lfsr: u32,
/// Absolute address of the compressedInfo (ptr,size) table pointer slot.
pub compressed_info_ptr: u32,
}
/// Marker-independent eighthStage slot discovery (PE32+ branch).
///
/// Newer Crackproof builds (e.g. some native/managed DLLs) omit the
/// `pm\0\0cm\0\0` and `00 00 00 40 01 00 00 00` markers that the older layout's
/// walk3/walk4/walk5 slot derivation relies on. Instead this discovers the
/// slots structurally:
/// * Scan the eighthStage for every LFSR (decrypt_data6) bytecode block.
/// * The file decryptor is the LFSR block whose `fileCS = lfsr - 0x58` holds
/// a pointer sitting just past `info[3]` (smallest positive distance).
/// * `compressedInfo` is the pointer slot whose 16-byte target, after a
/// trial `decrypt_data5`, parses as a plausible (src,sSize,dst,dSize)
/// descriptor.
///
/// Returns `None` if no plausible file LFSR is found. `eighth_start`/`eighth_dsz`
/// bound the search region; `info3` is `info[3]`; `compress_data_offset` is
/// `(!u32(file_data,0x1080)) + 0x1000`; `file_data_len` is the protected file
/// length.
#[allow(clippy::too_many_arguments)]
pub fn discover_eighth_slots(
data: &[u8],
eighth_start: u32,
eighth_dsz: u32,
info3: u32,
compress_data_offset: u32,
file_data_len: u32,
) -> Option<EighthSlots> {
// Collect all LFSR candidates (forward scan).
//
// Advance by 1 after each hit, NOT by 96. A false-positive LFSR match can sit
// just before the real file-decryptor block (observed on an il2cpp game
// assembly build, 2026-07-13: junk at rel=0x31C1, real block at 0x3210).
// Stepping by the LFSR body size then skips the real block and discovery
// fails. Byte-stepping is cheap: eighthStage is only a few KB.
let mut all_lfsrs: Vec<u32> = Vec::new();
let mut scan_off: u32 = 0;
while scan_off + 95 < eighth_dsz {
match find_lfsr_block(data, eighth_start, eighth_dsz, scan_off, false) {
Some(found) => {
all_lfsrs.push(found);
scan_off = found + 1;
}
None => break,
}
}
// Pick the file LFSR: prefer the candidate whose fileCS pointer sits the
// smallest positive distance past info[3].
let mut off_file_lfsr: Option<u32> = None;
let mut best_dist: Option<u32> = None;
for &lfsr_off in &all_lfsrs {
if lfsr_off < 0x58 {
continue;
}
let cs_off = lfsr_off - 0x58;
let cs_val = get_u32(data, eighth_start.wrapping_add(cs_off));
if !(0x1000 < cs_val && (cs_val as usize) < data.len()) {
continue;
}
if cs_val < info3 {
continue;
}
let dist = cs_val - info3;
if best_dist.is_none_or(|b| dist < b) {
best_dist = Some(dist);
off_file_lfsr = Some(lfsr_off);
}
}
// Fallback: last LFSR with any in-image fileCS pointer.
if off_file_lfsr.is_none() {
for &lfsr_off in all_lfsrs.iter().rev() {
if lfsr_off < 0x58 {
continue;
}
let cs_val = get_u32(data, eighth_start.wrapping_add(lfsr_off - 0x58));
if 0x1000 < cs_val && (cs_val as usize) < data.len() {
off_file_lfsr = Some(lfsr_off);
break;
}
}
}
let off_file_lfsr = off_file_lfsr?;
let off_file_cs = off_file_lfsr - 0x58;
// Trial-decrypt to find compressedInfo: the pointer slot in the data area
// (between fileCS region start and the LFSR) whose target parses as a valid
// (src,sSize,dst,dSize) descriptor after a transient decrypt_data5.
let scan_from = off_file_lfsr.saturating_sub(0x400);
let mut off_compressed_info: Option<u32> = None;
let mut doff = scan_from;
while doff < off_file_lfsr {
if doff == off_file_cs {
doff += 4;
continue;
}
let ptr_val = get_u32(data, eighth_start.wrapping_add(doff));
if !(0x1000 < ptr_val && (ptr_val as usize) < data.len().saturating_sub(16)) {
doff += 4;
continue;
}
// Predict decrypt_data5(ptr_val, 16) without mutating: each dword is
// position-keyed and independent, so trial_decrypt5_u32 per dword.
let src2 = trial_decrypt5_u32(data, ptr_val);
let s_sz2 = trial_decrypt5_u32(data, ptr_val + 4);
let dst2 = trial_decrypt5_u32(data, ptr_val + 8);
let d_sz2 = trial_decrypt5_u32(data, ptr_val + 12);
let src_file_off = src2.wrapping_add(compress_data_offset);
let valid = s_sz2 > 0
&& s_sz2 < 0x200000
&& (src_file_off as u64 + s_sz2 as u64) <= file_data_len as u64
&& dst2 >= 0x1000
&& (dst2 as u64 + d_sz2 as u64) <= data.len() as u64
&& d_sz2 >= s_sz2
&& d_sz2 < 0x200000;
if valid {
off_compressed_info = Some(doff);
break;
}
doff += 4;
}
let off_compressed_info = off_compressed_info?;
Some(EighthSlots {
file_lfsr: eighth_start.wrapping_add(off_file_lfsr),
compressed_info_ptr: eighth_start.wrapping_add(off_compressed_info),
})
}
#[cfg(test)]
mod tests {
use super::*;
/// Task 4.1 regression: build a synthetic buffer whose valid bytecode block
/// sits PAST `len` but within `len*2`. Assert that the smaller window misses
/// it and the doubled window finds it.
#[test]
fn bytecode_locate_double_window_retry() {
// We place the block at offset (base + len + 16) which is inside
// the len*2 window but outside the len window.
let base: u32 = 0;
let len: u32 = 256;
// Block sits at base + len + 16 = 272, aligned to 16.
let block_pos: usize = (base + len + 16) as usize; // 272
// The buffer must be large enough for the block (block_pos + 96 bytes).
let buf_len = block_pos + 256;
let mut buf = vec![0u8; buf_len];
// Build a valid plaintext op stream:
// [4, 0, 4, 0, 4, 0, 4, 0, 195] (4 ADD-AL ops then RET)
// Padded to 10 bytes total; count >= 8.
let count: usize = 10;
let mut plain = [0u8; 256];
plain[0] = 4;
plain[1] = 0;
plain[2] = 4;
plain[3] = 0;
plain[4] = 4;
plain[5] = 0;
plain[6] = 4;
plain[7] = 0;
plain[8] = 195; // ret
// Compute the LFSR keystream and XOR the first `count` bytes to get the
// encrypted representation that the scanner would decrypt back.
let mut ks = [0u8; 256];
lfsr_keystream(&mut ks);
for i in 0..count {
buf[block_pos + i] = plain[i] ^ ks[i];
}
// Raw count byte at block_pos+95 (outside the XOR range since count=10 < 95).
buf[block_pos + 95] = count as u8;
// Verify our construction: find_bytecode_offset with len should NOT find it.
assert_eq!(
find_bytecode_offset(&buf, base, len),
None,
"smaller window should not find the block"
);
// The doubled window should find it at block_pos.
assert_eq!(
find_bytecode_offset(&buf, base, len.saturating_mul(2)),
Some(block_pos as u32),
"doubled window should locate the block"
);
}
}
+481
View File
@@ -0,0 +1,481 @@
//! PE import reconstruction and memory-image compaction.
use senbei_crypto::primitives::{get_u16, get_u32, write_u16, write_u32};
use super::super::MAX_IMAGE_SIZE;
/// Read a NUL-terminated byte string starting at `off`, bounded to 512 bytes.
/// Returns the raw bytes up to the terminator (excluding it).
fn read_cstr_bounded(data: &[u8], off: u32) -> Vec<u8> {
let start = off as usize;
if start >= data.len() {
return Vec::new();
}
let limit = (start + 512).min(data.len());
let mut end = start;
while end < limit && data[end] != 0 {
end += 1;
}
data[start..end].to_vec()
}
fn align_up_u32(value: u32, alignment: u32) -> u32 {
((value.wrapping_add(alignment - 1)) / alignment).wrapping_mul(alignment)
}
fn align_up_u64(value: u64, alignment: u64) -> u64 {
value.div_ceil(alignment) * alignment
}
#[derive(Clone)]
enum ImportFunc {
Ordinal(u32),
Name(u16, Vec<u8>),
}
struct ImportDesc {
time_date: u32,
fwd_chain: u32,
dll_name: Vec<u8>,
iat_rva: u32,
functions: Vec<ImportFunc>,
}
/// Return true when PE32 imports already sit in the original `.idata` layout
/// (so no relocation to `.kmiat` is needed). May write the IAT data directory
/// (pe+0xD8).
pub fn pe32_imports_already_match_idata_layout(data: &mut [u8], pe_header: u32) -> bool {
let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32;
let sec_table = pe_header.wrapping_add(24).wrapping_add(opt_hdr_size);
let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32;
let import_rva = get_u32(data, pe_header.wrapping_add(0x80));
let import_size = get_u32(data, pe_header.wrapping_add(0x84));
let len = data.len() as u32;
if !(import_rva > 0 && import_size > 0) {
return false;
}
for idx in 0..num_sections {
let sec_off = sec_table.wrapping_add(idx * 40);
if (sec_off as usize + 40) > data.len() {
return false;
}
if &data[sec_off as usize..sec_off as usize + 6] != b".idata" {
continue;
}
let sec_va = get_u32(data, sec_off.wrapping_add(12));
let sec_size =
get_u32(data, sec_off.wrapping_add(8)).max(get_u32(data, sec_off.wrapping_add(16)));
let sec_end = sec_va.wrapping_add(sec_size);
if !(sec_va <= import_rva
&& import_rva < sec_end
&& import_rva.wrapping_add(import_size) <= sec_end)
{
continue;
}
let first_oft = get_u32(data, import_rva);
let first_name = get_u32(data, import_rva.wrapping_add(12));
let first_iat = get_u32(data, import_rva.wrapping_add(16));
if !(sec_va <= first_oft
&& first_oft < sec_end
&& sec_va <= first_iat
&& first_iat < sec_end)
{
return false;
}
if !(0x1000 < first_name && first_name < len) {
return false;
}
let dll_name = read_cstr_bounded(data, first_name);
let lower: Vec<u8> = dll_name.iter().map(|b| b.to_ascii_lowercase()).collect();
if !lower.ends_with(b".dll") {
return false;
}
let mut iat_min = first_iat;
let mut iat_max = first_iat;
let mut idt_pos = import_rva;
while idt_pos.wrapping_add(20) <= len {
let oft_rva = get_u32(data, idt_pos);
let name_rva = get_u32(data, idt_pos.wrapping_add(12));
let iat_rva = get_u32(data, idt_pos.wrapping_add(16));
if oft_rva == 0 && name_rva == 0 && iat_rva == 0 {
break;
}
if !(sec_va <= oft_rva && oft_rva < sec_end && sec_va <= iat_rva && iat_rva < sec_end) {
return false;
}
let mut thunk = iat_rva;
while thunk.wrapping_add(4) <= sec_end {
let tv = get_u32(data, thunk);
thunk = thunk.wrapping_add(4);
if tv == 0 {
break;
}
}
iat_min = iat_min.min(iat_rva);
iat_max = iat_max.max(thunk);
idt_pos = idt_pos.wrapping_add(20);
}
if iat_max > iat_min {
write_u32(data, pe_header.wrapping_add(0xD8), iat_min);
write_u32(data, pe_header.wrapping_add(0xDC), iat_max - iat_min);
}
return true;
}
false
}
/// Rebuild PE32 import metadata (descriptors, lookup tables, names) into the
/// last section as `.kmiat`, leaving the loader-written IAT in place. Mutates
/// `data` (may grow it).
pub fn move_pe32_imports_to_kmiat(data: &mut Vec<u8>, pe_header: u32) {
const SECTION_SIZE: u32 = 0x7000;
let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32;
let opt_hdr = pe_header.wrapping_add(24);
let sec_table = opt_hdr.wrapping_add(opt_hdr_size);
let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32;
if num_sections == 0 {
return;
}
let import_rva = get_u32(data, pe_header.wrapping_add(0x80));
let import_size = get_u32(data, pe_header.wrapping_add(0x84));
let len = data.len() as u32;
if !(0x1000 < import_rva && import_rva < len && import_size > 0 && import_size < SECTION_SIZE) {
return;
}
let mut descriptors: Vec<ImportDesc> = Vec::new();
let mut idt_pos = import_rva;
while idt_pos.wrapping_add(20) <= len {
let oft_rva = get_u32(data, idt_pos);
let time_date = get_u32(data, idt_pos.wrapping_add(4));
let fwd_chain = get_u32(data, idt_pos.wrapping_add(8));
let name_rva = get_u32(data, idt_pos.wrapping_add(12));
let iat_rva = get_u32(data, idt_pos.wrapping_add(16));
if oft_rva == 0 && name_rva == 0 && iat_rva == 0 {
break;
}
if !(0x1000 < name_rva && name_rva < len) {
break;
}
let dll_name = read_cstr_bounded(data, name_rva);
let thunk_rva = if 0x1000 < oft_rva && oft_rva < len {
oft_rva
} else {
iat_rva
};
let mut functions: Vec<ImportFunc> = Vec::new();
let mut thunk_pos = thunk_rva;
while 0x1000 < thunk_pos.wrapping_add(4) && thunk_pos.wrapping_add(4) <= len {
let thunk_val = get_u32(data, thunk_pos);
if thunk_val == 0 {
break;
}
if thunk_val & 0x8000_0000 != 0 {
functions.push(ImportFunc::Ordinal(thunk_val & 0xFFFF));
} else {
let hint = if thunk_val.wrapping_add(2) <= len {
get_u16(data, thunk_val)
} else {
0
};
let func_name = if thunk_val.wrapping_add(2) < len {
read_cstr_bounded(data, thunk_val.wrapping_add(2))
} else {
Vec::new()
};
functions.push(ImportFunc::Name(hint, func_name));
}
thunk_pos = thunk_pos.wrapping_add(4);
}
descriptors.push(ImportDesc {
time_date,
fwd_chain,
dll_name,
iat_rva,
functions,
});
idt_pos = idt_pos.wrapping_add(20);
}
if descriptors.is_empty() {
return;
}
for desc in &mut descriptors {
let lower: Vec<u8> = desc
.dll_name
.iter()
.map(|b| b.to_ascii_lowercase())
.collect();
if lower.starts_with(b"api-ms-win-crt-") {
desc.dll_name = b"ucrtbase.dll".to_vec();
} else {
desc.dll_name = lower;
}
}
descriptors.sort_by_key(|d| d.iat_rva);
let last_sec = sec_table.wrapping_add((num_sections - 1) * 40);
let kmiat_rva = get_u32(data, last_sec.wrapping_add(12));
// A zero last-section VA means a corrupt section table: building .kmiat at
// RVA 0 would zero the DOS/PE headers and emit a structurally broken image
// with no error. Bail and keep the original import table.
if kmiat_rva == 0 {
return;
}
// Grow the image when .kmiat overruns it, but cap the growth: a corrupt VA
// could otherwise request a multi-gigabyte allocation, which aborts the
// process (uncatchable). Use u64 math so a near-u32::MAX VA cannot wrap the
// end calculation the way the previous wrapping/plain-add mix could.
let kmiat_end = kmiat_rva as u64 + SECTION_SIZE as u64;
if kmiat_end > MAX_IMAGE_SIZE {
return;
}
if kmiat_end > data.len() as u64 {
data.resize(kmiat_end as usize, 0);
}
// Zero the .kmiat region.
for b in &mut data[kmiat_rva as usize..kmiat_end as usize] {
*b = 0;
}
let idt_size = (descriptors.len() as u32 + 1) * 20;
let oft_start = kmiat_rva;
let mut idt_rva = oft_start;
for desc in &descriptors {
idt_rva = idt_rva.wrapping_add((desc.functions.len() as u32 + 1) * 4);
}
idt_rva = align_up_u32(idt_rva.wrapping_add(0x2C), 4);
// Size check: compute the final name_pos and bail if it overruns .kmiat.
let mut name_pos_check = idt_rva.wrapping_add(idt_size);
for desc in &descriptors {
name_pos_check = name_pos_check.wrapping_add(desc.dll_name.len() as u32 + 1);
for func in &desc.functions {
if let ImportFunc::Name(_, fname) = func {
name_pos_check = name_pos_check.wrapping_add(2 + fname.len() as u32 + 1);
}
}
}
if name_pos_check > kmiat_rva.wrapping_add(SECTION_SIZE) {
// Section too small; keep existing import table untouched.
return;
}
let mut oft_pos = oft_start;
let mut name_pos = idt_rva.wrapping_add(idt_size);
for (idx, desc) in descriptors.iter().enumerate() {
let idt_entry = idt_rva.wrapping_add(idx as u32 * 20);
let current_oft = oft_pos;
write_u32(data, idt_entry, current_oft);
write_u32(data, idt_entry.wrapping_add(4), desc.time_date);
write_u32(data, idt_entry.wrapping_add(8), desc.fwd_chain);
let dll_name_pos = name_pos;
write_u32(data, idt_entry.wrapping_add(12), dll_name_pos);
write_u32(data, idt_entry.wrapping_add(16), desc.iat_rva);
let dnp = dll_name_pos as usize;
data[dnp..dnp + desc.dll_name.len()].copy_from_slice(&desc.dll_name);
data[dnp + desc.dll_name.len()] = 0;
name_pos = name_pos.wrapping_add(desc.dll_name.len() as u32 + 1);
for func in &desc.functions {
match func {
ImportFunc::Ordinal(ord) => {
write_u32(data, oft_pos, 0x8000_0000 | ord);
}
ImportFunc::Name(hint, fname) => {
let hint_name_rva = name_pos;
write_u32(data, oft_pos, hint_name_rva);
write_u16(data, hint_name_rva, *hint as u32);
let fp = (hint_name_rva + 2) as usize;
data[fp..fp + fname.len()].copy_from_slice(fname);
data[fp + fname.len()] = 0;
name_pos = name_pos.wrapping_add(2 + fname.len() as u32 + 1);
}
}
oft_pos = oft_pos.wrapping_add(4);
}
write_u32(data, oft_pos, 0);
oft_pos = oft_pos.wrapping_add(4);
}
// Null-terminator IDT entry (20 zero bytes) after the last descriptor.
let term = idt_rva.wrapping_add(descriptors.len() as u32 * 20) as usize;
for b in &mut data[term..term + 20] {
*b = 0;
}
let ls = last_sec as usize;
data[ls..ls + 8].copy_from_slice(b".kmiat\x00\x00");
write_u32(data, last_sec.wrapping_add(8), SECTION_SIZE);
write_u32(data, last_sec.wrapping_add(16), SECTION_SIZE);
write_u32(data, last_sec.wrapping_add(36), 0xE000_0060);
write_u32(data, pe_header.wrapping_add(0x80), idt_rva);
write_u32(data, pe_header.wrapping_add(0x84), idt_size);
write_u32(
data,
pe_header.wrapping_add(80),
kmiat_rva.wrapping_add(SECTION_SIZE),
);
}
/// Convert the unpacked RVA-addressed image back to a compact PE file layout
/// (headers at 0x400, sections packed consecutively, FileAlignment 0x200).
/// Returns `None` if the accumulated output size wraps or exceeds
/// [`MAX_IMAGE_SIZE`]: the final allocation is sized from header-derived
/// section data, and an uncapped `vec![0; n]` from a corrupt header would abort
/// the process (which `catch_unpack` cannot trap).
pub fn compact_memory_image_to_pe(data: &[u8], pe_header: u32) -> Option<Vec<u8>> {
const FILE_ALIGNMENT: u32 = 0x200;
const HEADER_SIZE: u32 = 0x400;
let opt_hdr_size = get_u16(data, pe_header.wrapping_add(20)) as u32;
let opt_hdr = pe_header.wrapping_add(24);
let sec_table = opt_hdr.wrapping_add(opt_hdr_size);
let num_sections = get_u16(data, pe_header.wrapping_add(6)) as u32;
struct SecLayout {
sec_off: u32,
va: u32,
vsize: u32,
raw_ptr: u32,
raw_size: u32,
}
let mut raw_cursor: u64 = HEADER_SIZE as u64;
let mut raw_layout: Vec<SecLayout> = Vec::new();
for idx in 0..num_sections {
let sec_off = sec_table.wrapping_add(idx * 40);
let vsize = get_u32(data, sec_off.wrapping_add(8));
let va = get_u32(data, sec_off.wrapping_add(12));
let sd_start = va as usize;
let sd_end = if (va.wrapping_add(vsize) as usize) <= data.len() {
va.wrapping_add(vsize) as usize
} else {
data.len()
};
let section_data: &[u8] = if sd_start <= sd_end && sd_start <= data.len() {
&data[sd_start..sd_end]
} else {
&[]
};
let mut last_nonzero: i64 = -1;
for pos in (0..section_data.len()).rev() {
if section_data[pos] != 0 {
last_nonzero = pos as i64;
break;
}
}
let meaningful = if last_nonzero >= 0 {
(last_nonzero + 1) as u32
} else {
0
};
let mut raw_size = if meaningful != 0 {
align_up_u32(meaningful, FILE_ALIGNMENT)
} else {
0
};
if vsize != 0 && raw_size == 0 {
raw_size = FILE_ALIGNMENT;
}
raw_size = raw_size.min(align_up_u32(section_data.len() as u32, FILE_ALIGNMENT));
let raw_ptr = if raw_size != 0 { raw_cursor as u32 } else { 0 };
raw_layout.push(SecLayout {
sec_off,
va,
vsize,
raw_ptr,
raw_size,
});
if raw_size != 0 {
// Accumulate in u64 and cap: section sizes are header-derived, and
// a corrupt table could otherwise wrap raw_cursor (small alloc,
// huge recorded raw_ptrs → OOB panic) or request an abort-sized
// allocation.
raw_cursor = align_up_u64(raw_cursor + raw_size as u64, FILE_ALIGNMENT as u64);
if raw_cursor > MAX_IMAGE_SIZE {
return None;
}
}
}
let mut compact = vec![0u8; raw_cursor as usize];
let hdr_copy = (HEADER_SIZE as usize).min(data.len());
compact[..hdr_copy].copy_from_slice(&data[..hdr_copy]);
write_u32(&mut compact, opt_hdr.wrapping_add(36), FILE_ALIGNMENT);
write_u32(&mut compact, opt_hdr.wrapping_add(60), HEADER_SIZE);
for sl in &raw_layout {
write_u32(&mut compact, sl.sec_off.wrapping_add(16), sl.raw_size);
write_u32(&mut compact, sl.sec_off.wrapping_add(20), sl.raw_ptr);
if sl.raw_size != 0 {
let sd_start = sl.va as usize;
let sd_end = if (sl.va.wrapping_add(sl.vsize) as usize) <= data.len() {
sl.va.wrapping_add(sl.vsize) as usize
} else {
data.len()
};
let section_data: &[u8] = if sd_start <= sd_end {
&data[sd_start..sd_end]
} else {
&[]
};
let copy_size = (sl.raw_size as usize).min(section_data.len());
let rp = sl.raw_ptr as usize;
compact[rp..rp + copy_size].copy_from_slice(&section_data[..copy_size]);
}
}
Some(compact)
}
#[cfg(test)]
mod tests {
use super::*;
/// Review regression: a zero last-section VA (corrupt section table) must
/// bail instead of building .kmiat at RVA 0 — the old code zeroed
/// `[0, 0x7000)`, wiping the DOS/PE headers, and returned the broken image
/// as a success. A near-2 GiB VA must likewise refuse to grow the image
/// past [`MAX_IMAGE_SIZE`].
#[test]
fn kmiat_bogus_section_va_bails_without_wiping_headers() {
for last_sec_va in [0u32, 0x5000_0000] {
let pe: u32 = 0x80;
let mut data = vec![0xAAu8; 0x8000];
// COFF header: 1 section, optional header size 0xE0 (PE32).
write_u16(&mut data, pe + 6, 1);
write_u16(&mut data, pe + 20, 0xE0);
// Import directory at pe+0x80: one descriptor + null terminator.
write_u32(&mut data, pe + 0x80, 0x1100);
write_u32(&mut data, pe + 0x84, 0x28);
write_u32(&mut data, 0x1100, 0x1200); // OFT rva
write_u32(&mut data, 0x1100 + 12, 0x1300); // name rva
write_u32(&mut data, 0x1100 + 16, 0x1400); // IAT rva
for b in &mut data[0x1100 + 20..0x1100 + 40] {
*b = 0; // null terminator descriptor
}
data[0x1300..0x1300 + 13].copy_from_slice(b"KERNEL32.dll\0");
write_u32(&mut data, 0x1200, 0x1500); // thunk -> hint/name
write_u32(&mut data, 0x1204, 0); // thunk terminator
data[0x1500..0x1502].copy_from_slice(&0u16.to_le_bytes());
data[0x1502..0x1502 + 12].copy_from_slice(b"ExitProcess\0");
// Section table at pe+24+0xE0 = 0x178; VA field at +12.
write_u32(&mut data, 0x178 + 12, last_sec_va);
let head_before: Vec<u8> = data[..0x400].to_vec();
let len_before = data.len();
move_pe32_imports_to_kmiat(&mut data, pe);
assert_eq!(
data.len(),
len_before,
"VA 0x{last_sec_va:08X}: image must not grow"
);
assert_eq!(
&data[..0x400],
&head_before[..],
"VA 0x{last_sec_va:08X}: headers must be untouched"
);
}
}
}
@@ -1,30 +1,157 @@
//! Pure, panic-free Crackproof unpacker core. No file I/O lives here. //! PE detection, unpacking, and structural validation.
mod bytecode;
mod crc32;
pub mod dll; pub mod dll;
mod error;
pub mod exe; pub mod exe;
pub mod integrity; pub mod integrity;
mod layout;
pub(crate) mod parallel; pub(crate) mod parallel;
pub(crate) mod primitives;
mod tables; use senbei_crypto::primitives;
use std::cell::RefCell;
use std::sync::{Arc, Mutex};
pub use dll::{unpack_dll, unpack_dll_v}; pub use dll::{unpack_dll, unpack_dll_v};
pub use exe::{UnpackError, unpack as unpack_exe, unpack_v as unpack_exe_v}; pub use error::*;
pub use exe::{unpack as unpack_exe, unpack_v as unpack_exe_v};
pub use integrity::{IntegrityReport, check as check_integrity}; pub use integrity::{IntegrityReport, check as check_integrity};
pub use parallel::thread_cap;
/// Maximum plausible PE `SizeOfImage` we are willing to allocate a zero buffer /// Maximum plausible PE `SizeOfImage` we are willing to allocate a zero buffer
/// for. Guards against a corrupt/crafted header requesting a multi-gigabyte /// for. Guards against a corrupt/crafted header requesting a multi-gigabyte
/// (or, as a sign-extended negative `i32`, multi-exabyte) allocation, which /// (or, as a sign-extended negative `i32`, multi-exabyte) allocation, which
/// would abort the process — an abort that `catch_unpack` below cannot trap. /// would abort the process — an abort that `catch_unpack` below cannot trap.
/// Real protected binaries are far below this. /// Real protected binaries are far below this.
pub(crate) const MAX_IMAGE_SIZE: u64 = 1 << 30; // 1 GiB pub(crate) const MAX_IMAGE_SIZE: u64 = senbei_crypto::MAX_IMAGE_SIZE;
#[derive(Clone)]
pub(crate) struct PanicCapture(Arc<Mutex<Option<PanicDetails>>>);
#[derive(Clone)]
struct PanicDetails {
message: String,
file: String,
line: u32,
column: u32,
}
thread_local! {
static ACTIVE_PANIC_CAPTURE: RefCell<Option<PanicCapture>> = const { RefCell::new(None) };
}
struct PanicCaptureGuard(Option<PanicCapture>);
impl Drop for PanicCaptureGuard {
fn drop(&mut self) {
ACTIVE_PANIC_CAPTURE.with(|slot| {
slot.replace(self.0.take());
});
}
}
impl PanicCapture {
fn new() -> Self {
Self(Arc::new(Mutex::new(None)))
}
fn record(&self, info: &std::panic::PanicHookInfo<'_>) {
let location = info.location();
let details = PanicDetails {
message: panic_message(info.payload()),
file: location
.map(|value| value.file().to_owned())
.unwrap_or_else(|| "<unknown>".to_owned()),
line: location.map_or(0, std::panic::Location::line),
column: location.map_or(0, std::panic::Location::column),
};
let mut captured = self
.0
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
if captured.is_none() {
*captured = Some(details);
}
}
fn into_error(self, payload: &(dyn std::any::Any + Send)) -> UnpackError {
let details = self
.0
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner)
.clone()
.unwrap_or_else(|| PanicDetails {
message: panic_message(payload),
file: "<unknown>".to_owned(),
line: 0,
column: 0,
});
UnpackError::InternalPanic {
message: details.message,
file: details.file,
line: details.line,
column: details.column,
}
}
fn merge_from(&self, other: &Self) {
let details = other
.0
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner)
.clone();
let Some(details) = details else { return };
let mut captured = self
.0
.lock()
.unwrap_or_else(std::sync::PoisonError::into_inner);
if captured.is_none() {
*captured = Some(details);
}
}
}
fn panic_message(payload: &(dyn std::any::Any + Send)) -> String {
if let Some(message) = payload.downcast_ref::<&str>() {
(*message).to_owned()
} else if let Some(message) = payload.downcast_ref::<String>() {
message.clone()
} else {
"non-string panic payload".to_owned()
}
}
fn install_panic_capture_hook() {
static INSTALL: std::sync::Once = std::sync::Once::new();
INSTALL.call_once(|| {
let previous = std::panic::take_hook();
std::panic::set_hook(Box::new(move |info| {
let capture = ACTIVE_PANIC_CAPTURE
.try_with(|slot| slot.borrow().clone())
.ok()
.flatten();
if let Some(capture) = capture {
capture.record(info);
} else {
previous(info);
}
}));
});
}
pub(crate) fn current_panic_capture() -> Option<PanicCapture> {
ACTIVE_PANIC_CAPTURE.with(|slot| slot.borrow().clone())
}
pub(crate) fn with_panic_capture<R>(capture: Option<PanicCapture>, f: impl FnOnce() -> R) -> R {
let previous = ACTIVE_PANIC_CAPTURE.with(|slot| slot.replace(capture));
let _guard = PanicCaptureGuard(previous);
f()
}
/// Run an unpack pipeline, converting any internal panic into a clean /// Run an unpack pipeline, converting any internal panic into a clean
/// [`UnpackError::Corrupt`] so the public API stays panic-free on any input /// [`UnpackError::InternalPanic`] so the public API stays panic-free on any input
/// (truncated/garbled files chase offsets out of bounds). The default panic /// (truncated/garbled files chase offsets out of bounds). The panic location and
/// hook is suppressed transiently so a trapped panic does not spill a /// payload are captured for diagnostics without printing a backtrace to stderr.
/// backtrace to stderr.
/// ///
/// Note: allocation *failures* abort the process and are NOT caught here; size /// Note: allocation *failures* abort the process and are NOT caught here; size
/// requests are bounds-checked against [`MAX_IMAGE_SIZE`] before allocating. /// requests are bounds-checked against [`MAX_IMAGE_SIZE`] before allocating.
@@ -32,17 +159,15 @@ pub(crate) fn catch_unpack<F>(f: F) -> Result<Vec<u8>, UnpackError>
where where
F: FnOnce() -> Result<Vec<u8>, UnpackError>, F: FnOnce() -> Result<Vec<u8>, UnpackError>,
{ {
// Hook suppression is skipped on wasm: the prebuilt std cannot unwind install_panic_capture_hook();
// there, so a panic traps immediately — and the suppressed hook would let capture = PanicCapture::new();
// hide the panic message, leaving a bare `unreachable` with no clue. let r = with_panic_capture(Some(capture.clone()), || {
#[cfg(not(target_arch = "wasm32"))] std::panic::catch_unwind(std::panic::AssertUnwindSafe(f))
let prev = std::panic::take_hook(); });
#[cfg(not(target_arch = "wasm32"))] match r {
std::panic::set_hook(Box::new(|_| {})); Ok(result) => result,
let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(f)); Err(payload) => Err(capture.into_error(payload.as_ref())),
#[cfg(not(target_arch = "wasm32"))] }
std::panic::set_hook(prev);
r.unwrap_or(Err(UnpackError::Corrupt))
} }
/// Crackproof header magic stored in `keys[1]`/`info[1]`. /// Crackproof header magic stored in `keys[1]`/`info[1]`.
@@ -80,11 +205,8 @@ fn key_table(input: &[u8]) -> Option<[u32; 8]> {
if input.len() < 4128 { if input.len() < 4128 {
return None; return None;
} }
// Validate PE signature. `checked_add`, not `+`: `usize` is 32-bit on // Validate the PE signature with checked arithmetic so a crafted offset
// wasm32, where an `e_lfanew` of 0xFFFF_FFFC..=0xFFFF_FFFF wraps the bound // cannot wrap the bounds check on a narrower target.
// check, and the slice below then panics with start > end. `detect` runs on
// the folder-scan threads and (in the web app) on the main thread outside
// the disposable-worker isolation, so it must not panic on any input.
let e_lfanew = primitives::get_u32(input, 0x3C); let e_lfanew = primitives::get_u32(input, 0x3C);
let pe_start = e_lfanew as usize; let pe_start = e_lfanew as usize;
if pe_start.checked_add(4).is_none_or(|end| end > input.len()) { if pe_start.checked_add(4).is_none_or(|end| end > input.len()) {
@@ -194,13 +316,76 @@ pub fn unpack_auto_v(input: &[u8], verbose: bool) -> Result<(Kind, Vec<u8>), Unp
Ok(out) => out, Ok(out) => out,
Err(dll_err) => match exe::unpack_v(input, verbose) { Err(dll_err) => match exe::unpack_v(input, verbose) {
Ok(out) => out, Ok(out) => out,
// Surface the DLL-pipeline error, not the EXE one: for a Err(exe_err) => {
// genuinely corrupt DLL the DLL error is the more relevant return Err(UnpackError::PipelineFallbackFailed {
// diagnostic, and the EXE fallback is best-effort. dll: Box::new(dll_err),
Err(_) => return Err(dll_err), exe: Box::new(exe_err),
});
}
}, },
} }
} }
}; };
Ok((detected.kind, out)) Ok((detected.kind, out))
} }
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn caught_panic_reports_location_and_message() {
let error = catch_unpack(|| -> Result<Vec<u8>, UnpackError> {
panic!("test panic");
})
.expect_err("panic must become an error");
let UnpackError::InternalPanic {
message,
file,
line,
column,
} = error
else {
panic!("unexpected error: {error}");
};
assert_eq!(message, "test panic");
assert!(
file.ends_with("senbei-pe/src/engine/mod.rs")
|| file.ends_with("senbei-pe\\src\\engine\\mod.rs")
);
assert!(line > 0);
assert!(column > 0);
}
#[test]
fn worker_panic_keeps_the_worker_source_location() {
let error = catch_unpack(|| -> Result<Vec<u8>, UnpackError> {
let capture = current_panic_capture();
let result = std::thread::spawn(move || {
with_panic_capture(capture, || panic!("worker panic"));
})
.join();
if let Err(payload) = result {
std::panic::resume_unwind(payload);
}
Ok(Vec::new())
})
.expect_err("worker panic must become an error");
let UnpackError::InternalPanic {
message,
file,
line,
column,
} = error
else {
panic!("unexpected error: {error}");
};
assert_eq!(message, "worker panic");
assert!(
file.ends_with("senbei-pe/src/engine/mod.rs")
|| file.ends_with("senbei-pe\\src\\engine\\mod.rs")
);
assert!(line > 0);
assert!(column > 0);
}
}
@@ -21,7 +21,7 @@ use std::sync::atomic::{AtomicBool, Ordering};
/// Worker-thread cap. `SENBEI_THREADS` overrides it (`1` forces the sequential /// Worker-thread cap. `SENBEI_THREADS` overrides it (`1` forces the sequential
/// path); otherwise the host's available parallelism; otherwise 1. /// path); otherwise the host's available parallelism; otherwise 1.
pub(crate) fn thread_cap() -> usize { pub fn thread_cap() -> usize {
if let Ok(v) = std::env::var("SENBEI_THREADS") if let Ok(v) = std::env::var("SENBEI_THREADS")
&& let Ok(n) = v.trim().parse::<usize>() && let Ok(n) = v.trim().parse::<usize>()
&& n >= 1 && n >= 1
@@ -46,7 +46,7 @@ pub(crate) fn thread_cap() -> usize {
/// ///
/// Returns the first `Err` any block produces; re-raises the first block panic /// Returns the first `Err` any block produces; re-raises the first block panic
/// on the calling thread (so the pipeline's existing `catch_unpack` still /// on the calling thread (so the pipeline's existing `catch_unpack` still
/// converts it to `UnpackError::Corrupt`). /// converts it to `UnpackError::InternalPanic`).
pub(crate) fn parallel_for<E, F>( pub(crate) fn parallel_for<E, F>(
buf: &mut [u8], buf: &mut [u8],
spans: &[(usize, usize)], spans: &[(usize, usize)],
@@ -125,6 +125,7 @@ where
let stop = AtomicBool::new(false); let stop = AtomicBool::new(false);
let first_err: Mutex<Option<E>> = Mutex::new(None); let first_err: Mutex<Option<E>> = Mutex::new(None);
let first_panic: Mutex<Option<Box<dyn std::any::Any + Send>>> = Mutex::new(None); let first_panic: Mutex<Option<Box<dyn std::any::Any + Send>>> = Mutex::new(None);
let panic_capture = super::current_panic_capture();
std::thread::scope(|scope| { std::thread::scope(|scope| {
for _ in 0..workers { for _ in 0..workers {
@@ -133,6 +134,7 @@ where
let first_err = &first_err; let first_err = &first_err;
let first_panic = &first_panic; let first_panic = &first_panic;
let f = &f; let f = &f;
let panic_capture = panic_capture.clone();
scope.spawn(move || { scope.spawn(move || {
loop { loop {
if stop.load(Ordering::Relaxed) { if stop.load(Ordering::Relaxed) {
@@ -141,9 +143,15 @@ where
let next = iter.lock().unwrap().next(); let next = iter.lock().unwrap().next();
let Some((i, piece)) = next else { break }; let Some((i, piece)) = next else { break };
let span = piece.unwrap(); let span = piece.unwrap();
let r = std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| { // Keep details local until this panic wins `first_panic`;
// otherwise simultaneous workers could pair one worker's
// location with another worker's propagated payload.
let block_capture = panic_capture.as_ref().map(|_| super::PanicCapture::new());
let r = super::with_panic_capture(block_capture.clone(), || {
std::panic::catch_unwind(std::panic::AssertUnwindSafe(|| {
f(i, spans[i].0, span) f(i, spans[i].0, span)
})); }))
});
match r { match r {
Ok(Ok(())) => {} Ok(Ok(())) => {}
Ok(Err(e)) => { Ok(Err(e)) => {
@@ -157,6 +165,11 @@ where
Err(panic) => { Err(panic) => {
let mut slot = first_panic.lock().unwrap(); let mut slot = first_panic.lock().unwrap();
if slot.is_none() { if slot.is_none() {
if let (Some(parent), Some(block)) =
(&panic_capture, &block_capture)
{
parent.merge_from(block);
}
*slot = Some(panic); *slot = Some(panic);
} }
stop.store(true, Ordering::Relaxed); stop.store(true, Ordering::Relaxed);
+5
View File
@@ -0,0 +1,5 @@
//! PE detection, unpacking, and structural validation.
mod engine;
pub use engine::*;
File diff suppressed because it is too large Load Diff
-10
View File
@@ -1,10 +0,0 @@
//! Shared test fixtures.
#![allow(dead_code)]
use std::path::PathBuf;
/// Path to `senbei/samples` — the user-managed corpus dropped in by hand.
/// Git-ignored except its README; tests here run against whatever is present.
pub fn samples_dir() -> PathBuf {
PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("samples")
}
+54 -32
View File
@@ -50,21 +50,21 @@ checksum = "34aa73646ffb006b8f5147f3dc182bd4bcb190227ce861fc4a4844bf8e3cb2c0"
[[package]] [[package]]
name = "futures-core" name = "futures-core"
version = "0.3.33" version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2cd50c473c80f6d7c3670a752354b8e569b1a7cbfdc0419ec88e5edad85e0dc7" checksum = "92d699e522242e69e3003b94ecc1f960f3a5e015aa7c5d7486e65ad01dd94f5e"
[[package]] [[package]]
name = "futures-task" name = "futures-task"
version = "0.3.33" version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b231ed28831efb4a61a08580c4bc233ec56bc009f4cd8f52da2c3cb97df0c109" checksum = "cd417de3d1d015fc3bfd2b1ea46dfc7bab72ef86f1cc7cc9c78e728b34a6d1fd"
[[package]] [[package]]
name = "futures-util" name = "futures-util"
version = "0.3.33" version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a77a90a256fce34da66415271e30f94ee91c57b04b8a2c042d9cf3220179deaa" checksum = "0d50a92467f8ba5dd6e3ee5d4bd04d73ab2e4e1c44474a0674821dfce14b79bc"
dependencies = [ dependencies = [
"futures-core", "futures-core",
"futures-task", "futures-task",
@@ -87,9 +87,9 @@ dependencies = [
[[package]] [[package]]
name = "js-sys" name = "js-sys"
version = "0.3.103" version = "0.3.104"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "53b44bfcdb3f8d5837a46dae1ca9660a837176eee74a28b229bc626816589102" checksum = "0e0c1080212aad755ea003d18543e8768dd432c48819efd73a7bf1e39b7a5a3a"
dependencies = [ dependencies = [
"cfg-if", "cfg-if",
"futures-util", "futures-util",
@@ -110,9 +110,9 @@ checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
[[package]] [[package]]
name = "owo-colors" name = "owo-colors"
version = "4.3.0" version = "4.4.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d211803b9b6b570f68772237e415a029d5a50c65d382910b879fb19d3271f94d" checksum = "13c45bb4a6ae1280ec0803b1ef9d3455eb50f01efbbe1447ab020f1d54fba9d8"
[[package]] [[package]]
name = "pin-project-lite" name = "pin-project-lite"
@@ -122,9 +122,9 @@ checksum = "a89322df9ebe1c1578d689c92318e070967d1042b512afbe49518723f4e6d5cd"
[[package]] [[package]]
name = "portable-atomic" name = "portable-atomic"
version = "1.14.0" version = "1.15.0"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3d20d5497ef88037a52ff98267d066e7f11fcc5e99bbfbd58a42336193aacec3" checksum = "05c8b63e8d9609db387f0324918f81d68fe27748f084ef092fb35954d0539a85"
[[package]] [[package]]
name = "proc-macro2" name = "proc-macro2"
@@ -160,24 +160,46 @@ dependencies = [
] ]
[[package]] [[package]]
name = "senbei" name = "senbei-crypto"
version = "1.0.0" version = "1.1.0"
dependencies = [
"thiserror",
]
[[package]]
name = "senbei-io"
version = "1.1.0"
dependencies = [ dependencies = [
"anyhow", "anyhow",
"indicatif", "indicatif",
"libc", "libc",
"owo-colors", "owo-colors",
"thiserror", "senbei-metadata",
"senbei-pe",
"walkdir", "walkdir",
"windows", "windows",
] ]
[[package]]
name = "senbei-metadata"
version = "1.1.0"
[[package]]
name = "senbei-pe"
version = "1.1.0"
dependencies = [
"senbei-crypto",
"thiserror",
]
[[package]] [[package]]
name = "senbei-web" name = "senbei-web"
version = "1.0.0" version = "1.1.0"
dependencies = [ dependencies = [
"console_error_panic_hook", "console_error_panic_hook",
"senbei", "senbei-io",
"senbei-metadata",
"senbei-pe",
"wasm-bindgen", "wasm-bindgen",
] ]
@@ -200,9 +222,9 @@ dependencies = [
[[package]] [[package]]
name = "syn" name = "syn"
version = "3.0.3" version = "3.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "53e9bae58849f64dfa4f5d5ae372c8341f7305f82a3868709269343628b659a3" checksum = "e6275cddf4610d1775e6d1fe9469b2e77d0f39fd98fb7450901b821e0c53649f"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
@@ -211,22 +233,22 @@ dependencies = [
[[package]] [[package]]
name = "thiserror" name = "thiserror"
version = "2.0.19" version = "2.0.20"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "09a43598840e33d5b0331f38c5e30d13bb11c11210a4b58f0d9b18a5a5eefcd9" checksum = "ec86235f5fcc2a73650310756d2ac5b138a5780bbbdfae3eeccec992c435ba4f"
dependencies = [ dependencies = [
"thiserror-impl", "thiserror-impl",
] ]
[[package]] [[package]]
name = "thiserror-impl" name = "thiserror-impl"
version = "2.0.19" version = "2.0.20"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "43cbfe0cf76104d42a574802844187e84a305e531ed54455f11fbde0f10541cd" checksum = "bc04cd3e1236dd4a98afca4569f2deb3f120e5422a4023be2cb683f8486292af"
dependencies = [ dependencies = [
"proc-macro2", "proc-macro2",
"quote", "quote",
"syn 3.0.3", "syn 3.0.4",
] ]
[[package]] [[package]]
@@ -259,9 +281,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen" name = "wasm-bindgen"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "4b067c0c11094aef6b7a801c1e34a26affafdf3d051dba08456b868789aaf9a4" checksum = "1b70935747edd64d89de3efa29d73789b806c15798f8e7dca4d8ac356b50ce70"
dependencies = [ dependencies = [
"cfg-if", "cfg-if",
"once_cell", "once_cell",
@@ -272,9 +294,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen-macro" name = "wasm-bindgen-macro"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "167ce5e579f6bcf889c4f7175a8a5a585de84e8ff93976ce393efa5f2837aab1" checksum = "77775f8f3f7217702089053b94958f8f54061a3f663417df76e19cbdcca29bc1"
dependencies = [ dependencies = [
"quote", "quote",
"wasm-bindgen-macro-support", "wasm-bindgen-macro-support",
@@ -282,9 +304,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen-macro-support" name = "wasm-bindgen-macro-support"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f3997c7839262f4ef12cf90b818d6340c18e80f263f1a94bf157d0ec4420380e" checksum = "e11d33f857dc2fb11b8bc75aee111aa9cbeb12cd9f25efd3d4c2a3dd4e235284"
dependencies = [ dependencies = [
"bumpalo", "bumpalo",
"proc-macro2", "proc-macro2",
@@ -295,9 +317,9 @@ dependencies = [
[[package]] [[package]]
name = "wasm-bindgen-shared" name = "wasm-bindgen-shared"
version = "0.2.126" version = "0.2.127"
source = "registry+https://github.com/rust-lang/crates.io-index" source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "dc1b4cb0cc549fcf58d7dfc081778139b3d283a081644e833e84682ad71cea24" checksum = "7ef64dbcc55df09c7e5a46182d181c2cfa3e925f3da937ea764728b4bbb9dcbf"
dependencies = [ dependencies = [
"unicode-ident", "unicode-ident",
] ]
+4 -2
View File
@@ -1,6 +1,6 @@
[package] [package]
name = "senbei-web" name = "senbei-web"
version = "1.0.1" version = "1.1.0"
edition = "2024" edition = "2024"
description = "WebAssembly browser frontend for senbei" description = "WebAssembly browser frontend for senbei"
license = "AGPL-3.0-only" license = "AGPL-3.0-only"
@@ -9,7 +9,9 @@ license = "AGPL-3.0-only"
crate-type = ["cdylib"] crate-type = ["cdylib"]
[dependencies] [dependencies]
senbei = { path = ".." } senbei-io = { path = "../senbei-io" }
senbei-metadata = { path = "../senbei-metadata" }
senbei-pe = { path = "../senbei-pe" }
wasm-bindgen = "0.2" wasm-bindgen = "0.2"
console_error_panic_hook = "0.1" console_error_panic_hook = "0.1"
+11 -11
View File
@@ -101,12 +101,12 @@ impl MetadataResult {
} }
} }
fn kind_str(kind: senbei::unpacker::Kind) -> &'static str { fn kind_str(kind: senbei_pe::Kind) -> &'static str {
match kind { match kind {
senbei::unpacker::Kind::NativeExe => "native-exe", senbei_pe::Kind::NativeExe => "native-exe",
senbei::unpacker::Kind::ManagedExe => "managed-exe", senbei_pe::Kind::ManagedExe => "managed-exe",
senbei::unpacker::Kind::NativeDll => "native-dll", senbei_pe::Kind::NativeDll => "native-dll",
senbei::unpacker::Kind::ManagedDll => "managed-dll", senbei_pe::Kind::ManagedDll => "managed-dll",
} }
} }
@@ -117,10 +117,10 @@ fn kind_str(kind: senbei::unpacker::Kind) -> &'static str {
/// anything unrecognized. /// anything unrecognized.
#[wasm_bindgen] #[wasm_bindgen]
pub fn detect(input: &[u8]) -> Option<String> { pub fn detect(input: &[u8]) -> Option<String> {
if senbei::metadata::is_metadata(input) { if senbei_metadata::is_metadata(input) {
return Some("metadata".to_string()); return Some("metadata".to_string());
} }
senbei::unpacker::detect(input).map(|d| kind_str(d.kind).to_string()) senbei_pe::detect(input).map(|d| kind_str(d.kind).to_string())
} }
/// Unpack a protected module. /// Unpack a protected module.
@@ -134,7 +134,7 @@ pub fn unpack_file(
input: &[u8], input: &[u8],
companion: Option<Vec<u8>>, companion: Option<Vec<u8>>,
) -> Result<UnpackResult, JsError> { ) -> Result<UnpackResult, JsError> {
let r = senbei::job::unpack_bytes(input, companion.as_deref()) let r = senbei_io::job::unpack_bytes(input, companion.as_deref())
.map_err(|e| JsError::new(&e.to_string()))?; .map_err(|e| JsError::new(&e.to_string()))?;
Ok(UnpackResult { Ok(UnpackResult {
kind: kind_str(r.kind).to_string(), kind: kind_str(r.kind).to_string(),
@@ -153,7 +153,7 @@ pub fn unpack_file(
#[wasm_bindgen] #[wasm_bindgen]
pub fn deobfuscate_metadata(data: &[u8]) -> Result<MetadataResult, JsError> { pub fn deobfuscate_metadata(data: &[u8]) -> Result<MetadataResult, JsError> {
let (bytes, report) = let (bytes, report) =
senbei::metadata::deobfuscate(data).map_err(|e| JsError::new(&e.to_string()))?; senbei_metadata::deobfuscate(data).map_err(|e| JsError::new(&e.to_string()))?;
Ok(MetadataResult { Ok(MetadataResult {
bytes, bytes,
version: report.version, version: report.version,
@@ -164,14 +164,14 @@ pub fn deobfuscate_metadata(data: &[u8]) -> Result<MetadataResult, JsError> {
} }
/// Unpack a protected module, forcing the EXE pipeline (no DLL-pipeline /// Unpack a protected module, forcing the EXE pipeline (no DLL-pipeline
/// probe). See [`senbei::job::unpack_bytes_force_exe`] for why the web app /// probe). See [`senbei_io::job::unpack_bytes_force_exe`] for why the web app
/// needs this recovery path. /// needs this recovery path.
#[wasm_bindgen] #[wasm_bindgen]
pub fn unpack_file_force_exe( pub fn unpack_file_force_exe(
input: &[u8], input: &[u8],
companion: Option<Vec<u8>>, companion: Option<Vec<u8>>,
) -> Result<UnpackResult, JsError> { ) -> Result<UnpackResult, JsError> {
let r = senbei::job::unpack_bytes_force_exe(input, companion.as_deref()) let r = senbei_io::job::unpack_bytes_force_exe(input, companion.as_deref())
.map_err(|e| JsError::new(&e.to_string()))?; .map_err(|e| JsError::new(&e.to_string()))?;
Ok(UnpackResult { Ok(UnpackResult {
kind: kind_str(r.kind).to_string(), kind: kind_str(r.kind).to_string(),