Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
7a63bac709 | ||
|
|
14eafa0692 | ||
|
|
9560248c49 | ||
|
|
14408b75a0 | ||
|
|
e7b6f727d9 | ||
|
|
040bc8b14d | ||
|
|
7abcfaab72 | ||
|
|
341e079e1e | ||
|
|
25893f1be4 | ||
|
|
efb69e3ba5 | ||
|
|
4cd9b7baa1 | ||
|
|
02e7bc605d | ||
|
|
0563b58f2e | ||
|
|
313460c97f | ||
|
|
68a1a55958 | ||
|
|
0955730045 | ||
|
|
d3103453f1 | ||
|
|
b64a96042e | ||
|
|
074b1ee829 | ||
|
|
8f9bde9b9a | ||
|
|
2d1563c63a | ||
|
|
cbb127a175 | ||
|
|
9c6b7baf83 | ||
|
|
3d738af58f | ||
|
|
3e400edd4f | ||
|
|
9c9de0e095 | ||
|
|
539dd0131b | ||
|
|
5677f42c69 | ||
|
|
f434b9cf2c | ||
|
|
b17761a24e | ||
|
|
f2dd1d2e34 | ||
|
|
671c3c7c8c | ||
|
|
c6be942d1a | ||
|
|
90d20304cd | ||
|
|
50dfe877b9 | ||
|
|
17622a1b59 | ||
|
|
32824fba5b | ||
|
|
109afcdcf7 | ||
|
|
dd749132d5 | ||
|
|
3ff2abbff5 | ||
|
|
03c6d20447 | ||
|
|
237794a6ea | ||
|
|
835e97ce71 | ||
|
|
70e1807e61 | ||
|
|
418abfe79e | ||
|
|
42c62d46b6 | ||
|
|
06d0f9ef8b | ||
|
|
b3887be90f | ||
|
|
3ecebf5b40 | ||
|
|
282651186c | ||
|
|
c3c37380ae | ||
|
|
8ec71834dd | ||
|
|
991977f297 | ||
|
|
d4a32d3a26 | ||
|
|
813edd0965 | ||
|
|
b16eacd6b4 | ||
|
|
776a4fd6eb | ||
|
|
a0f76a4f18 | ||
|
|
afa0a213d6 | ||
|
|
e5f92e591a | ||
|
|
79dbb2cf84 | ||
|
|
b9ca75f471 | ||
|
|
adc8ee37be | ||
|
|
c3e82f28b5 | ||
|
|
6b67c52249 | ||
|
|
dd9e92ed52 | ||
|
|
65ccbcbd92 | ||
|
|
aa149099ec | ||
|
|
7deacf1761 | ||
|
|
47789f71bc | ||
|
|
c31d9fc88e | ||
|
|
0e8c31a9c3 | ||
|
|
03d088abfc | ||
|
|
84e0ba9fa8 | ||
|
|
bc5e3453b4 | ||
|
|
6d9791affc | ||
|
|
d1551eb588 | ||
|
|
c875df49e3 | ||
|
|
93ad1f4854 | ||
|
|
d0393fb629 | ||
|
|
c64bc311ff | ||
|
|
a018e1adc4 | ||
|
|
1d35dcf7c6 | ||
|
|
77ad147563 | ||
|
|
5c64662213 | ||
|
|
1f70398774 | ||
|
|
35c5eedc20 | ||
|
|
f4b95b3dea | ||
|
|
0f61be00b7 | ||
|
|
f4fb5c65e0 | ||
|
|
c8fafec393 | ||
|
|
764535bb7d | ||
|
|
60d9cc1bac | ||
|
|
cbb3517afe | ||
|
|
3980aa8976 | ||
|
|
ffbc1d8399 | ||
|
|
9247e7da2f | ||
|
|
6c92370013 | ||
|
|
3090314717 | ||
|
|
ce246a1c87 | ||
|
|
617532e6ff | ||
|
|
dfd2f023d0 | ||
|
|
bd2ba08bb7 | ||
|
|
dc7c3a7db5 | ||
|
|
cfce270186 | ||
|
|
4ab8303a65 | ||
|
|
80564e0470 | ||
|
|
bab566da40 | ||
|
|
4fe05ba4c0 | ||
|
|
7199ee497a | ||
|
|
9ccb3c8444 | ||
|
|
4959b48386 | ||
|
|
37e056f070 | ||
|
|
1df36c78b2 | ||
|
|
6d2ff4d1fc | ||
|
|
94377c75fd | ||
|
|
0d4aab99df | ||
|
|
b93d10082d | ||
|
|
17bb13b077 | ||
|
|
b68765fe84 | ||
|
|
abffa4235a | ||
|
|
c94e9f4fb7 | ||
|
|
28d5897b86 | ||
|
|
3ed8630535 | ||
|
|
d0d8e2c9bf | ||
|
|
01d4a1ba00 | ||
|
|
5c8b4dc7c5 | ||
|
|
841aa1a1c6 | ||
|
|
14049bb477 | ||
|
|
8ffce6b621 | ||
|
|
9fda690d00 | ||
|
|
f53abfe0d4 | ||
|
|
80be34bff5 | ||
|
|
4447bd60ce | ||
|
|
c054893540 | ||
|
|
6758370f8d | ||
|
|
b2a274782a | ||
|
|
cf7ee69fd5 | ||
|
|
e008e71a17 | ||
|
|
c59e1e3342 | ||
|
|
39d9714ae7 | ||
|
|
dd940583c7 | ||
|
|
4b7e4ddbb3 | ||
|
|
5222458411 | ||
|
|
dd5118ee9c | ||
|
|
a8db2435cd | ||
|
|
ff18d4c3c8 | ||
|
|
f8ed0b99f4 | ||
|
|
8189da1b0c | ||
|
|
698ba36ae4 | ||
|
|
1eb8bdc9c7 | ||
|
|
90a7fe2ff1 | ||
|
|
4e70d9a5c5 | ||
|
|
1b95d346bb | ||
|
|
48663c6a2f | ||
|
|
a05f1d4498 | ||
|
|
3ecb2510e8 | ||
|
|
048f125879 | ||
|
|
65dbcb1ca6 | ||
|
|
b002da4221 | ||
|
|
51d2b14d03 | ||
|
|
c4ad4184ec | ||
|
|
528a6b7345 | ||
|
|
fb321f51eb | ||
|
|
e0ff0cfeb4 | ||
|
|
71686f1407 | ||
|
|
d50a7173ad | ||
|
|
e9811a1e01 | ||
|
|
f3841c8aca | ||
|
|
7d48d820e5 | ||
|
|
42591c77fc | ||
|
|
2efe1425d6 | ||
|
|
b86f7aef17 | ||
|
|
5559987325 | ||
|
|
72bcc371fb | ||
|
|
54d038e478 | ||
|
|
6868b93b7e | ||
|
|
8d39ccc613 | ||
|
|
8c0de5711e | ||
|
|
9f25a4c454 | ||
|
|
30bea12392 | ||
|
|
b2b611fa3b | ||
|
|
4f4b1ed222 | ||
|
|
944e6a8b09 | ||
|
|
5360f8d309 | ||
|
|
8b8bcff106 | ||
|
|
d5a9e70700 | ||
|
|
0bc8d7af9c | ||
|
|
e99b634635 | ||
|
|
f9d081ed45 | ||
|
|
3e13a155fa | ||
|
|
b2e1982051 | ||
|
|
84a77f6a0e | ||
|
|
9de88969ca | ||
|
|
170fd0c064 | ||
|
|
55b97ac576 | ||
|
|
c610285910 | ||
|
|
e4b1e5b19e | ||
|
|
8d4a6d54a4 | ||
|
|
93e1436fc0 | ||
|
|
18f8b285c4 | ||
|
|
c63dafcf1a | ||
|
|
46eb88c51f | ||
|
|
dea968f32b | ||
|
|
079c9b1327 | ||
|
|
327087c70e | ||
|
|
b8fa5e74dc | ||
|
|
3f7d7af472 | ||
|
|
fdd473d7e9 | ||
|
|
05fed1b0e0 | ||
|
|
d444afbdfc | ||
|
|
921404d135 | ||
|
|
399c3d2769 | ||
|
|
dc5b67ed46 | ||
|
|
5c6a6d0785 | ||
|
|
f5e169efb3 | ||
|
|
0bbceed985 | ||
|
|
4fcd28b487 | ||
|
|
9527bc1e13 | ||
|
|
58bdb42f8e | ||
|
|
62450e19bd | ||
|
|
013881ac06 | ||
|
|
5f8dc392c0 | ||
|
|
a32373ff40 | ||
|
|
3efa6211f3 | ||
|
|
9ad68dd092 | ||
|
|
13897e14f0 | ||
|
|
b4bf0daa82 | ||
|
|
ef36b452ad | ||
|
|
f338552969 | ||
|
|
38aa895038 | ||
|
|
e3676e7cdf | ||
|
|
807eb053ca | ||
|
|
7322f4dd8a | ||
|
|
4ed245868e | ||
|
|
50f37462db | ||
|
|
22a3e3fd01 | ||
|
|
99c5fd3500 | ||
|
|
d4c913e0d3 | ||
|
|
0e23a6b291 | ||
|
|
bcf47cc4ca | ||
|
|
a9dc3d7244 | ||
|
|
94a876664b | ||
|
|
7030de4ec9 | ||
|
|
c0434e87de | ||
|
|
ec5cd31ae1 | ||
|
|
a1304f9e78 | ||
|
|
a39045adf1 | ||
|
|
ea72e6df5f | ||
|
|
b2c7490b5b | ||
|
|
3db4106253 | ||
|
|
d34979ac57 | ||
|
|
bf2d15f39c | ||
|
|
d09ed76e07 | ||
|
|
f76688a0dc | ||
|
|
840cb9aef0 | ||
|
|
f3e80c8499 | ||
|
|
c812a32f3d | ||
|
|
eedd27e352 | ||
|
|
d8e5b97c86 | ||
|
|
0151e199ef | ||
|
|
ea99c82e32 | ||
|
|
8822c29905 | ||
|
|
0ee8341aea | ||
|
|
d22d09c898 | ||
|
|
43564d5752 | ||
|
|
508a2c3376 | ||
|
|
45a16991ff | ||
|
|
ca0daaea07 | ||
|
|
a02bbbdd24 | ||
|
|
ba8114f29b | ||
|
|
8421c227cd | ||
|
|
b79ff71b43 | ||
|
|
9b3e281f4d | ||
|
|
e0456c72da | ||
|
|
1c7ccd5a85 | ||
|
|
3bd2fd23b0 | ||
|
|
c00384d4df | ||
|
|
e8c151792d | ||
|
|
20b36229a5 | ||
|
|
71aad385c7 | ||
|
|
ec5b10f83a | ||
|
|
5181f6ce19 | ||
|
|
a711ee1d00 | ||
|
|
f2b7cc9bdd | ||
|
|
6f53767e8b | ||
|
|
7ed798e386 | ||
|
|
b9568242df | ||
|
|
8ac18fa631 | ||
|
|
197489fb7c | ||
|
|
dc7dfc6041 | ||
|
|
0acb326079 | ||
|
|
e632874665 | ||
|
|
88e58bfc95 | ||
|
|
279ba0dd7c | ||
|
|
1eb6910bdb | ||
|
|
e380e3b7c8 | ||
|
|
ff349fa61b | ||
|
|
52fd0f733a | ||
|
|
5b03fd8ebc | ||
|
|
c635190b0d | ||
|
|
34c5293704 | ||
|
|
bb59166e48 | ||
|
|
ea047b57d6 | ||
|
|
909fe48628 | ||
|
|
da19280950 | ||
|
|
2274423a6f | ||
|
|
c1f1593003 | ||
|
|
6718c9cdb2 | ||
|
|
9fdd5edb65 | ||
|
|
8a5a26f2a5 | ||
|
|
97ce0b7fab | ||
|
|
e194ef1585 | ||
|
|
4d1b922232 | ||
|
|
3841ae2250 | ||
|
|
2ccb5c9d01 | ||
|
|
b2bd5f8b3e | ||
|
|
6f055394c7 | ||
|
|
98f3dc513f | ||
|
|
5ecfe7c69a | ||
|
|
f255361683 | ||
|
|
c6e6bb9f4b | ||
|
|
f7edd4e6a9 | ||
|
|
a947439171 | ||
|
|
8e6114cd2a | ||
|
|
8f55cb78d2 | ||
|
|
65e14fe3b7 | ||
|
|
1e3610fd75 | ||
|
|
0c9d375548 | ||
|
|
8aff7fe708 | ||
|
|
489545c865 | ||
|
|
281d8baed6 | ||
|
|
6a4ac97a33 | ||
|
|
a8563e9fa3 | ||
|
|
63f6909ff0 | ||
|
|
9f33306a0a | ||
|
|
43cdc1351d | ||
|
|
5728a7c577 | ||
|
|
f85d91a17a | ||
|
|
a688e2c642 | ||
|
|
05fb632d7c | ||
|
|
3cb0a8f41c | ||
|
|
9dbfb70f7e | ||
|
|
9af3f7da7a | ||
|
|
2638c3075e | ||
|
|
3661942bdb | ||
|
|
43c1f9bda0 | ||
|
|
37832ac2dd | ||
|
|
2263d2cc4e | ||
|
|
3546648faa | ||
|
|
98000869b2 | ||
|
|
e308c5b825 | ||
|
|
5e1f880f6e | ||
|
|
ffe8ee8684 |
@@ -0,0 +1,63 @@
|
||||
version: 2
|
||||
|
||||
# Dependency updates land on `dev`, never on `main`.
|
||||
#
|
||||
# `main` here is a RELEASE POINTER that release.sh moves to each tag. A bot
|
||||
# commit on it would put work there that no tag contains, which is exactly the
|
||||
# state that aborted the 1.6.2 cascade at the last step -- so pointing
|
||||
# Dependabot at main would recreate that failure on a schedule.
|
||||
updates:
|
||||
- package-ecosystem: cargo
|
||||
directory: /
|
||||
target-branch: dev
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
# One PR per week for the routine bumps instead of one per crate. Eight
|
||||
# repos times a handful of crates is a volume nobody reads, and an
|
||||
# unread PR queue is indistinguishable from no updates at all.
|
||||
groups:
|
||||
minor-and-patch:
|
||||
update-types:
|
||||
- minor
|
||||
- patch
|
||||
ignore:
|
||||
# The freemkv crates depend on each other by GIT TAG, re-pinned by
|
||||
# release.sh as part of the release commit. Dependabot cannot see that
|
||||
# cascade, so a PR bumping one of these would fight the release process
|
||||
# and could pin a version whose tag does not exist yet.
|
||||
- dependency-name: freemkv-unlock
|
||||
- dependency-name: libfreemkv
|
||||
- dependency-name: freemkv-keysources
|
||||
- dependency-name: freemkv-i18n
|
||||
- dependency-name: freemkv-engine
|
||||
|
||||
# The workflows are now real infrastructure -- the release cascade, the
|
||||
# cross-platform hash matrix, the disc gate -- so their actions need the same
|
||||
# attention as the crates.
|
||||
- package-ecosystem: github-actions
|
||||
directory: /
|
||||
target-branch: dev
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
groups:
|
||||
actions:
|
||||
update-types:
|
||||
- minor
|
||||
- patch
|
||||
ignore:
|
||||
# NOT a dependency: `dtolnay/rust-toolchain` is versioned by the RUST
|
||||
# release it installs, and the tag we pin is the toolchain CI is pinned
|
||||
# to on purpose -- precommit.sh runs the same one locally so a lint that
|
||||
# passes on a developer's newer default cannot pass CI by accident.
|
||||
#
|
||||
# Dependabot reads those tags as semver and proposed 1.97.0 -> 1.100.0,
|
||||
# a Rust version that does not exist. Every such PR 404s on toolchain
|
||||
# download across all eight repos, and they regenerate weekly -- eight
|
||||
# permanently-red PRs that promote.yml then has to special-case when it
|
||||
# decides whether dev is green.
|
||||
#
|
||||
# Bumping the toolchain is a deliberate, all-eight-repos change, made by
|
||||
# hand together with precommit.sh. There is nothing here for a bot.
|
||||
- dependency-name: dtolnay/rust-toolchain
|
||||
+188
-10
@@ -2,50 +2,228 @@ name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
# dev -> qa -> main. `dev` is where work lands and is meant to be pushed
|
||||
# to often: these are the FAST checks, so a mistake surfaces in minutes.
|
||||
# `qa` is the release candidate — it runs these too, plus the expensive
|
||||
# suite in qa.yml. `main` only ever moves at release time, to a tagged
|
||||
# commit that was already green on qa.
|
||||
branches: [main, dev, qa]
|
||||
pull_request:
|
||||
|
||||
jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: dtolnay/rust-toolchain@1.86.0
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
path: libfreemkv
|
||||
# libfreemkv path-deps ../freemkv-unlock on the BRANCH tip (release.sh
|
||||
# swaps it to a git tag only inside the tagged commit, then restores the
|
||||
# path dep). CI checks out one repo, so the branch tip has never been
|
||||
# buildable here — every green run you have ever seen was a tag build,
|
||||
# and Windows/Linux were first compiled at release time.
|
||||
#
|
||||
# Both repos go into subdirectories because actions/checkout refuses a
|
||||
# `path:` outside $GITHUB_WORKSPACE, and `../freemkv-unlock` is outside.
|
||||
# With this layout the path dep resolves exactly as it does locally.
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
repository: freemkv/freemkv-unlock
|
||||
ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}"
|
||||
path: freemkv-unlock
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
with:
|
||||
components: clippy, rustfmt
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: libfreemkv
|
||||
- run: cargo fmt --check
|
||||
working-directory: libfreemkv
|
||||
# libfreemkv is a library — Cargo.lock is gitignored. --locked
|
||||
# would always fail on a fresh runner because there's no committed
|
||||
# lockfile to lock against. The binary crates (freemkv, autorip,
|
||||
# bdemu) track Cargo.lock and DO use --locked.
|
||||
- run: cargo clippy -- -D warnings
|
||||
# --all-targets so TEST code is linted too. Without it this crate — the
|
||||
# reference implementation for the other seven — was the only one whose
|
||||
# tests had never been linted at all, and it was hiding 74 findings.
|
||||
- run: cargo clippy --all-targets -- -D warnings
|
||||
working-directory: libfreemkv
|
||||
|
||||
test:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: dtolnay/rust-toolchain@1.86.0
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
path: libfreemkv
|
||||
# libfreemkv path-deps ../freemkv-unlock on the BRANCH tip (release.sh
|
||||
# swaps it to a git tag only inside the tagged commit, then restores the
|
||||
# path dep). CI checks out one repo, so the branch tip has never been
|
||||
# buildable here — every green run you have ever seen was a tag build,
|
||||
# and Windows/Linux were first compiled at release time.
|
||||
#
|
||||
# Both repos go into subdirectories because actions/checkout refuses a
|
||||
# `path:` outside $GITHUB_WORKSPACE, and `../freemkv-unlock` is outside.
|
||||
# With this layout the path dep resolves exactly as it does locally.
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
repository: freemkv/freemkv-unlock
|
||||
ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}"
|
||||
path: freemkv-unlock
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: libfreemkv
|
||||
- run: cargo test --tests
|
||||
working-directory: libfreemkv
|
||||
|
||||
check-macos:
|
||||
# dev is the FAST lane: this job still runs, but on the release-candidate
|
||||
# branches rather than on every push to dev. Nothing is deleted and no
|
||||
# platform stops being checked before a release -- qa.yml independently
|
||||
# covers macOS and Windows, and the jobs unique to this file (the Intel
|
||||
# macOS build, the Windows release build) run here on qa and main. A push
|
||||
# to dev is meant to be cheap and frequent; waiting on three runner pools
|
||||
# to agree is what a release candidate is for.
|
||||
#
|
||||
# `if` SKIPS the job (it does not queue). A queued job would be far worse
|
||||
# than a slow one: release.sh's CI gate refuses while any run for the
|
||||
# commit is still in progress, so a never-scheduled job blocks releases
|
||||
# silently -- see the note on real-media in qa.yml.
|
||||
if: github.ref_name == 'qa' || github.ref_name == 'main'
|
||||
runs-on: macos-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: dtolnay/rust-toolchain@1.86.0
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
path: libfreemkv
|
||||
# libfreemkv path-deps ../freemkv-unlock on the BRANCH tip (release.sh
|
||||
# swaps it to a git tag only inside the tagged commit, then restores the
|
||||
# path dep). CI checks out one repo, so the branch tip has never been
|
||||
# buildable here — every green run you have ever seen was a tag build,
|
||||
# and Windows/Linux were first compiled at release time.
|
||||
#
|
||||
# Both repos go into subdirectories because actions/checkout refuses a
|
||||
# `path:` outside $GITHUB_WORKSPACE, and `../freemkv-unlock` is outside.
|
||||
# With this layout the path dep resolves exactly as it does locally.
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
repository: freemkv/freemkv-unlock
|
||||
ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}"
|
||||
path: freemkv-unlock
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: libfreemkv
|
||||
- run: cargo check
|
||||
working-directory: libfreemkv
|
||||
|
||||
check-windows:
|
||||
# dev is the FAST lane: this job still runs, but on the release-candidate
|
||||
# branches rather than on every push to dev. Nothing is deleted and no
|
||||
# platform stops being checked before a release -- qa.yml independently
|
||||
# covers macOS and Windows, and the jobs unique to this file (the Intel
|
||||
# macOS build, the Windows release build) run here on qa and main. A push
|
||||
# to dev is meant to be cheap and frequent; waiting on three runner pools
|
||||
# to agree is what a release candidate is for.
|
||||
#
|
||||
# `if` SKIPS the job (it does not queue). A queued job would be far worse
|
||||
# than a slow one: release.sh's CI gate refuses while any run for the
|
||||
# commit is still in progress, so a never-scheduled job blocks releases
|
||||
# silently -- see the note on real-media in qa.yml.
|
||||
if: github.ref_name == 'qa' || github.ref_name == 'main'
|
||||
runs-on: windows-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: dtolnay/rust-toolchain@1.86.0
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
path: libfreemkv
|
||||
# libfreemkv path-deps ../freemkv-unlock on the BRANCH tip (release.sh
|
||||
# swaps it to a git tag only inside the tagged commit, then restores the
|
||||
# path dep). CI checks out one repo, so the branch tip has never been
|
||||
# buildable here — every green run you have ever seen was a tag build,
|
||||
# and Windows/Linux were first compiled at release time.
|
||||
#
|
||||
# Both repos go into subdirectories because actions/checkout refuses a
|
||||
# `path:` outside $GITHUB_WORKSPACE, and `../freemkv-unlock` is outside.
|
||||
# With this layout the path dep resolves exactly as it does locally.
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
repository: freemkv/freemkv-unlock
|
||||
ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}"
|
||||
path: freemkv-unlock
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: libfreemkv
|
||||
# Build the tests (not just `cargo check`): catches errors in test
|
||||
# code and forces full codegen of the Windows-only SPTI transport
|
||||
# (src/scsi/windows.rs), which never compiles on the Linux/macOS dev
|
||||
# hosts. We don't `cargo test` here — the suite needs no drive but the
|
||||
# extra build is the value; running tests is covered by the Linux job.
|
||||
- run: cargo build --tests
|
||||
working-directory: libfreemkv
|
||||
|
||||
# ── Did this change break anything downstream? ──────────────────────────────
|
||||
#
|
||||
# Every job above proves libfreemkv builds. None proved its DEPENDENTS do,
|
||||
# and that gap is real: an engine signature change broke autorip today and
|
||||
# went unnoticed because consumer CI only fires on a push to that consumer.
|
||||
# libfreemkv sits below all five of them, so a break here is worth strictly
|
||||
# more than a break anywhere else in the project.
|
||||
#
|
||||
# `cargo check --all-targets` only — each dependent owns its own behaviour
|
||||
# and has its own suite. The question here is just "does everything built on
|
||||
# me still compile against this commit".
|
||||
consumers:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with: { path: libfreemkv }
|
||||
- uses: actions/checkout@v7
|
||||
with: { repository: freemkv/freemkv-unlock, ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}", path: freemkv-unlock }
|
||||
- uses: actions/checkout@v7
|
||||
with: { repository: freemkv/freemkv-keysources, ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}", path: freemkv-keysources }
|
||||
- uses: actions/checkout@v7
|
||||
with: { repository: freemkv/freemkv-engine, ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}", path: freemkv-engine }
|
||||
- uses: actions/checkout@v7
|
||||
with: { repository: freemkv/freemkv-i18n, ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}", path: freemkv-i18n }
|
||||
- uses: actions/checkout@v7
|
||||
with: { repository: freemkv/freemkv, ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}", path: freemkv }
|
||||
- uses: actions/checkout@v7
|
||||
with: { repository: freemkv/autorip, ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}", path: autorip }
|
||||
- uses: actions/checkout@v7
|
||||
with: { repository: freemkv/bdemu, ref: "${{ github.ref_name == 'qa' && 'qa' || 'dev' }}", path: bdemu }
|
||||
- name: Point every dependent at THIS libfreemkv commit
|
||||
shell: bash
|
||||
run: |
|
||||
for c in freemkv-keysources freemkv-engine freemkv autorip bdemu; do
|
||||
mkdir -p "$c/.cargo"
|
||||
cat > "$c/.cargo/config.toml" <<'EOF'
|
||||
[patch.crates-io]
|
||||
libfreemkv = { path = "../libfreemkv" }
|
||||
freemkv-keysources = { path = "../freemkv-keysources" }
|
||||
freemkv-engine = { path = "../freemkv-engine" }
|
||||
freemkv-i18n = { path = "../freemkv-i18n" }
|
||||
EOF
|
||||
done
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: |
|
||||
freemkv-keysources
|
||||
freemkv-engine
|
||||
freemkv
|
||||
autorip
|
||||
bdemu
|
||||
# `cargo check` alone only proves the dependents still COMPILE against
|
||||
# this commit. It cannot see a behavioural change — the library keeps its
|
||||
# signatures and a dependent's tests start failing. That is the shape of
|
||||
# every defect worth catching here, so run their suites too.
|
||||
- run: cargo test --tests
|
||||
working-directory: freemkv-keysources
|
||||
- run: cargo test --tests
|
||||
working-directory: freemkv-engine
|
||||
- run: cargo test --tests
|
||||
working-directory: freemkv
|
||||
- run: cargo test --tests
|
||||
working-directory: autorip
|
||||
- run: cargo test --tests
|
||||
working-directory: bdemu
|
||||
|
||||
@@ -11,7 +11,7 @@ jobs:
|
||||
leak-guard:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Compute commit range
|
||||
|
||||
@@ -0,0 +1,133 @@
|
||||
name: qa
|
||||
|
||||
# ── The qa gate: "is this production worth?" ────────────────────────────────
|
||||
#
|
||||
# dev -> qa -> main.
|
||||
#
|
||||
# `dev` is for committing often. ci.yml answers "is it green" in minutes with
|
||||
# fmt, clippy and the unit suite, so a mistake surfaces while it is still cheap
|
||||
# to fix. `qa` is the release-candidate branch, and THIS workflow is the claim
|
||||
# that a commit is production worth: everything expensive that can run without
|
||||
# physical media. `main` only ever receives a qa that went green here.
|
||||
#
|
||||
# Sibling repos are checked out at `qa`, NOT `dev`. A qa run that resolved its
|
||||
# dependencies from dev tips would be validating a combination that is not the
|
||||
# one being released, which is the exact failure this branch exists to prevent.
|
||||
#
|
||||
# What this gate CANNOT cover: `disc://` and real `iso://` need physical media,
|
||||
# and no hosted runner has an optical drive or the image hoard. Those run on a
|
||||
# self-hosted runner (see the media job at the end) and are the one leg that
|
||||
# stays on hardware.
|
||||
on:
|
||||
push:
|
||||
branches: [qa]
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
# ── Name the candidate ────────────────────────────────────────────────
|
||||
#
|
||||
# Every push to `qa` is a release candidate, so every push gets a tag:
|
||||
# v<version>-rc<N>, N incrementing. That is the answer to "which build is on
|
||||
# qa right now, and is it the one I tested?" — a question that otherwise gets
|
||||
# answered from memory.
|
||||
#
|
||||
# This runs FIRST and does not depend on the gates, deliberately. A red
|
||||
# candidate needs a name more than a green one does: "rc3 failed
|
||||
# release-tests on windows" is a sentence you can act on; "qa is red" is not.
|
||||
# Red on qa is a working gate, not an incident — it is the branch saying this
|
||||
# is not production worth yet. Fix on dev, get dev green, push qa again.
|
||||
#
|
||||
# release.yml excludes v*-rc* so a candidate never publishes a release.
|
||||
rc-tag:
|
||||
runs-on: ubuntu-latest
|
||||
permissions:
|
||||
contents: write
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Stamp the next rc
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
v=$(sed -n 's/^version = "\(.*\)"/\1/p' Cargo.toml | head -1)
|
||||
[ -n "$v" ] || { echo "no version in Cargo.toml" >&2; exit 1; }
|
||||
# Numeric sort on the rc ordinal: -rc10 must beat -rc9, and a plain
|
||||
# lexical sort gets that backwards from the tenth candidate on.
|
||||
n=$(git tag -l "v$v-rc*" | sed "s|^v$v-rc||" | sort -n | tail -1)
|
||||
tag="v$v-rc$(( ${n:-0} + 1 ))"
|
||||
git tag "$tag"
|
||||
git push origin "$tag"
|
||||
echo "### Candidate \`$tag\`" >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
# The debug suite runs on every dev push. Release is a DIFFERENT build:
|
||||
# overflow checks are off, debug_assert! is compiled out, and inlining
|
||||
# changes what the optimiser can prove. A test that only passes in debug is
|
||||
# a test that never guarded the binary anyone actually ships.
|
||||
release-tests:
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: ['ubuntu-latest', 'macos-latest']
|
||||
runs-on: ${{ matrix.os }}
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
path: libfreemkv
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
repository: freemkv/freemkv-unlock
|
||||
ref: qa
|
||||
path: freemkv-unlock
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: libfreemkv
|
||||
- run: cargo test --release --tests
|
||||
working-directory: libfreemkv
|
||||
|
||||
# clippy's output is target-dependent: cfg-gated code only gets linted on
|
||||
# the target it compiles for. Linting solely on the dev machine's host
|
||||
# target is how a lint that CI rejects reaches a push.
|
||||
cross-lint:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
path: libfreemkv
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
repository: freemkv/freemkv-unlock
|
||||
ref: qa
|
||||
path: freemkv-unlock
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
with:
|
||||
components: clippy
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: libfreemkv
|
||||
- run: rustup target add x86_64-unknown-linux-gnu
|
||||
- run: cargo clippy --all-targets --target x86_64-unknown-linux-gnu -- -D warnings
|
||||
working-directory: libfreemkv
|
||||
|
||||
# Windows compiles the tests but does not run them, matching the policy
|
||||
# ci.yml already set. The value here is codegen: the #[cfg(windows)] halves
|
||||
# of the SCSI transport and platform layers compile on no other runner, so
|
||||
# without this they are first built at release time.
|
||||
windows-build:
|
||||
runs-on: windows-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
path: libfreemkv
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
repository: freemkv/freemkv-unlock
|
||||
ref: qa
|
||||
path: freemkv-unlock
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
with:
|
||||
workspaces: libfreemkv
|
||||
- run: cargo build --release --tests
|
||||
working-directory: libfreemkv
|
||||
@@ -4,6 +4,11 @@ on:
|
||||
push:
|
||||
tags:
|
||||
- 'v*'
|
||||
# NOT the release-candidate tags. Every push to `qa` stamps a
|
||||
# v<version>-rc<N> so a run can be named, and 'v*' matches those too —
|
||||
# which would have this workflow build and PUBLISH a GitHub release for
|
||||
# every candidate, including the red ones.
|
||||
- '!v*-rc*'
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
@@ -12,7 +17,7 @@ jobs:
|
||||
verify:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: actions/checkout@v7
|
||||
- name: Verify Cargo.toml version matches tag
|
||||
run: |
|
||||
CARGO_VER="v$(grep '^version' Cargo.toml | head -1 | sed 's/.*"\(.*\)"/\1/')"
|
||||
@@ -24,7 +29,7 @@ jobs:
|
||||
|
||||
# Tests run as a PARALLEL TRIPWIRE: they fail the run if they fail, but the
|
||||
# publish/release jobs do NOT `needs:` this job. The tag decision was already
|
||||
# gated by the local precommit (same Rust 1.86, same commit). Binary consumers
|
||||
# gated by the local precommit (same Rust 1.97, same commit). Binary consumers
|
||||
# (freemkv/autorip/bdemu) git-tag-pin libfreemkv and therefore start building
|
||||
# the instant this tag exists — so this test job and the crates.io publish
|
||||
# below must NOT sit on their critical path.
|
||||
@@ -32,8 +37,8 @@ jobs:
|
||||
needs: verify
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: dtolnay/rust-toolchain@1.86.0
|
||||
- uses: actions/checkout@v7
|
||||
- uses: dtolnay/rust-toolchain@1.97.0
|
||||
- uses: Swatinem/rust-cache@v2
|
||||
# libfreemkv is a library — Cargo.lock isn't tracked, so --locked
|
||||
# would always fail (no lockfile to lock against on a fresh runner).
|
||||
@@ -51,8 +56,8 @@ jobs:
|
||||
needs: verify
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: actions/checkout@v7
|
||||
- name: Create GitHub Release
|
||||
uses: softprops/action-gh-release@v2
|
||||
uses: softprops/action-gh-release@v3
|
||||
with:
|
||||
generate_release_notes: true
|
||||
|
||||
@@ -1,36 +0,0 @@
|
||||
name: Update README version
|
||||
|
||||
on:
|
||||
release:
|
||||
types: [published]
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
jobs:
|
||||
update:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
with:
|
||||
ref: main
|
||||
token: ${{ secrets.ORG_DISPATCH_TOKEN }}
|
||||
|
||||
- name: Update version in README
|
||||
run: |
|
||||
VERSION="${{ github.event.release.tag_name }}"
|
||||
FILE="README.md"
|
||||
|
||||
# Update cargo dependency version (e.g. "0.2" -> "0.3")
|
||||
MAJOR_MINOR="${VERSION#v}"
|
||||
MAJOR_MINOR="${MAJOR_MINOR%.*}"
|
||||
sed -i "s|libfreemkv = \"[0-9]*\.[0-9]*\"|libfreemkv = \"${MAJOR_MINOR}\"|" "$FILE"
|
||||
|
||||
- name: Commit and push
|
||||
run: |
|
||||
VERSION="${{ github.event.release.tag_name }}"
|
||||
git config user.name "github-actions[bot]"
|
||||
git config user.email "github-actions[bot]@users.noreply.github.com"
|
||||
git add README.md
|
||||
git diff --cached --quiet || git commit -m "Update to libfreemkv ${VERSION}"
|
||||
git push
|
||||
@@ -14,3 +14,12 @@ scratch/
|
||||
# internal agent context — never publish (path AND dir; leak-guard blocks both)
|
||||
CLAUDE.md
|
||||
.claude/
|
||||
|
||||
# Nightly harness output. Written into the repo it audits, and it embeds
|
||||
# absolute paths from the machine that ran it — which must never reach a public
|
||||
# repo. Ignored rather than relocated so a run from any working copy is safe.
|
||||
.nightly/
|
||||
|
||||
# cargo-mutants working output: large, machine-specific, never committed
|
||||
mutants.out/
|
||||
mutants.out.old/
|
||||
|
||||
+792
-728
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -36,7 +36,7 @@ This Code of Conduct applies within all community spaces, and also applies when
|
||||
|
||||
## Enforcement
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be reported to the community leaders responsible for enforcement at matthew@pq.io. All complaints will be reviewed and investigated promptly and fairly.
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be reported privately to the project maintainer via GitHub at https://github.com/MattJackson. All complaints will be reviewed and investigated promptly and fairly.
|
||||
|
||||
All community leaders are obligated to respect the privacy and security of the reporter of any incident.
|
||||
|
||||
|
||||
+9
-14
@@ -1,8 +1,8 @@
|
||||
[package]
|
||||
name = "libfreemkv"
|
||||
version = "1.4.4"
|
||||
version = "1.6.6"
|
||||
edition = "2024"
|
||||
rust-version = "1.86"
|
||||
rust-version = "1.97"
|
||||
license = "MIT"
|
||||
description = "Open source raw disc access library for optical drives"
|
||||
repository = "https://github.com/freemkv/libfreemkv"
|
||||
@@ -22,27 +22,22 @@ codegen-units = 1
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
sha1 = "0.10"
|
||||
sha2 = "0.10"
|
||||
aes = "0.8"
|
||||
cbc = "0.1"
|
||||
aes = "0.9"
|
||||
# Interim path dep for local cross-repo dev; the release script re-pins this to
|
||||
# `{ git = ".../freemkv-unlock", tag = "vX.Y.Z" }` before tagging libfreemkv (so
|
||||
# the released tag resolves freemkv-unlock from git, not a sibling path).
|
||||
freemkv-unlock = { git = "https://github.com/freemkv/freemkv-unlock", tag = "v1.4.4" }
|
||||
num-bigint = "0.4"
|
||||
num-traits = "0.2"
|
||||
num-integer = "0.1"
|
||||
rand = "0.8"
|
||||
cmac = "0.7"
|
||||
zip = { version = "2", default-features = false, features = ["deflate"] }
|
||||
base64 = "0.22.1"
|
||||
freemkv-unlock = { git = "https://github.com/freemkv/freemkv-unlock", tag = "v1.6.5" }
|
||||
rand = "0.10"
|
||||
zip = { version = "8", default-features = false, features = ["deflate"] }
|
||||
base64 = "0.23"
|
||||
# Read-only XML DOM parser (pure Rust, forbid(unsafe_code), entity-expansion
|
||||
# bounded). Parses the HD-DVD Advanced-Content playlist `ADV_OBJ/VPLST000.XPL`
|
||||
# — untrusted disc bytes — into authoritative titles/clips/chapters. A real
|
||||
# parser, not a hand-rolled scanner: the XPL is genuine XML (comments, varied
|
||||
# attribute order, self-closing tags).
|
||||
roxmltree = "0.20"
|
||||
# Trace-level instrumentation for Disc::copy + SgIoTransport::execute. Permitted
|
||||
# Trace-level instrumentation for the read/transport path (SgIoTransport::execute
|
||||
# and friends). Permitted
|
||||
# under CLAUDE.md ("Acceptable strings: debug/trace logging"). Consumers (autorip)
|
||||
# wire a tracing subscriber and pipe events into the JSONL debug log.
|
||||
tracing = "0.1"
|
||||
|
||||
@@ -55,48 +55,15 @@ output.finish()?;
|
||||
|
||||
### Multi-pass recovery rip
|
||||
|
||||
For damaged discs the library exposes two flat verbs — `Disc::sweep` for the
|
||||
forward Pass 1 and `Disc::patch` for retrying bad ranges. The library never
|
||||
loops; the multipass policy is the caller's job. See
|
||||
[`docs/rip-recovery.md`](docs/rip-recovery.md).
|
||||
Recovery moved OUT of this crate in 1.6.0. The sweep/patch strategy, the
|
||||
ddrescue mapfile, damage classification and the multipass loop now live in the
|
||||
`freemkv-engine` crate as `freemkv_engine::recovery::{copy, sweep, patch}`.
|
||||
|
||||
```rust
|
||||
use libfreemkv::{SweepOptions, PatchOptions};
|
||||
use libfreemkv::disc::{mapfile, mapfile_path_for};
|
||||
use std::path::Path;
|
||||
|
||||
let iso = Path::new("disc.iso");
|
||||
|
||||
// Pass 1: disc → ISO. Skip-on-error, zero-fill, write the sidecar mapfile.
|
||||
disc.sweep(&mut drive, iso, &SweepOptions {
|
||||
decrypt: true,
|
||||
resume: false,
|
||||
batch_sectors: None,
|
||||
skip_on_error: true,
|
||||
progress: None,
|
||||
halt: None,
|
||||
})?;
|
||||
|
||||
// Pass 2..N: retry every non-finished range. Idempotent.
|
||||
loop {
|
||||
let map = mapfile::Mapfile::load(&mapfile_path_for(iso))?;
|
||||
let stats = map.stats();
|
||||
if stats.bytes_pending + stats.bytes_unreadable == 0 { break; }
|
||||
|
||||
let outcome = disc.patch(&mut drive, iso, &PatchOptions {
|
||||
decrypt: true,
|
||||
block_sectors: None,
|
||||
full_recovery: true,
|
||||
reverse: true,
|
||||
wedged_threshold: 50,
|
||||
progress: None,
|
||||
halt: None,
|
||||
})?;
|
||||
if outcome.bytes_recovered_this_pass == 0 { break; }
|
||||
}
|
||||
|
||||
// Mux from the ISO via the normal stream pipeline (no drive involvement).
|
||||
```
|
||||
libfreemkv keeps the layers underneath: the raw single-shot read
|
||||
(`Drive::read`) and the SCSI-fact translation (`SenseFamily`) that the engine's
|
||||
strategy is built on. The dependency runs engine → libfreemkv, so this crate
|
||||
cannot call into it; front-ends get recovery from the engine directly. See
|
||||
[`docs/rip-recovery.md`](docs/rip-recovery.md) for what stayed here.
|
||||
|
||||
## What It Does
|
||||
|
||||
@@ -114,7 +81,7 @@ loop {
|
||||
| Stream | Input | Output | Transport |
|
||||
|--------|-------|--------|-----------|
|
||||
| DiscStream | Yes | -- | Optical drive via SCSI |
|
||||
| IsoStream | Yes | -- | Blu-ray ISO image file (read via stream pipeline; written via `Disc::sweep()`) |
|
||||
| IsoStream | Yes | -- | Blu-ray ISO image file (read via stream pipeline; written by `freemkv_engine::recovery`) |
|
||||
| MkvStream | Yes | Yes | Matroska container |
|
||||
| M2tsStream | Yes | Yes | BD transport stream with FMKV metadata header |
|
||||
| NetworkStream | Yes (listen) | Yes (connect) | TCP with FMKV metadata header |
|
||||
|
||||
+22
@@ -0,0 +1,22 @@
|
||||
# Security Policy
|
||||
|
||||
## Supported versions
|
||||
|
||||
| Version | Supported |
|
||||
| ------- | --------- |
|
||||
| 1.6.x | Yes |
|
||||
| < 1.6 | No |
|
||||
|
||||
Only the current 1.6.x line receives security fixes.
|
||||
|
||||
## Reporting a vulnerability
|
||||
|
||||
Report vulnerabilities privately through GitHub Security Advisories:
|
||||
https://github.com/freemkv/libfreemkv/security/advisories/new
|
||||
|
||||
Do not open a public issue for a security report. Include the affected
|
||||
version, steps to reproduce, and the impact you believe the issue has.
|
||||
|
||||
## Response time
|
||||
|
||||
You will get an initial response within 7 days.
|
||||
+3
-3
@@ -141,8 +141,8 @@ If your machine has a free SATA port, use it.
|
||||
|
||||
freemkv uses a three-layer recovery model. See [`docs/rip-recovery.md`](docs/rip-recovery.md) for full details.
|
||||
|
||||
- **Pass 1 (Disc::copy):** Fast sweep with 64 KB reads. On failure, zero-fills the block and skips forward. Writes a ddrescue-format mapfile for later retry.
|
||||
- **Pass 2+ (Disc::patch):** Targeted re-reads of bad ranges with a long 30-second timeout per CDB. The drive firmware performs its own ECC and laser power retries within that window.
|
||||
- **Pass 1 (`freemkv_engine::recovery::sweep`):** Fast sweep with 64 KB reads. On failure, zero-fills the block and skips forward. Writes a ddrescue-format mapfile for later retry.
|
||||
- **Pass 2+ (`freemkv_engine::recovery::patch`):** Targeted re-reads of bad ranges with a long 60-second timeout per CDB. The drive firmware performs its own ECC and laser power retries within that window.
|
||||
- **In-stream (DiscStream):** Adaptive batch halving -- reduces request size on failure to isolate bad sectors within a larger block.
|
||||
|
||||
This means a disc with some bad sectors will still produce a usable ISO. The damaged areas are zero-filled in pass 1 and retried in subsequent passes. Structure-protected sectors (deliberate unreadable regions from copy protection) will never yield, which is expected.
|
||||
@@ -302,7 +302,7 @@ Many external drive enclosures (Vantec NexStar, Sabrent, OWC, etc.) do not adver
|
||||
When something goes wrong during a rip, gather this information before filing an issue:
|
||||
|
||||
1. **freemkv version:** `freemkv --version` or the crate version in `Cargo.toml`.
|
||||
2. **Drive model:** from the drive label, or from `freemkv info`.
|
||||
2. **Drive model:** from the drive label, or from `freemkv info disc://`.
|
||||
3. **Connection type:** USB (with bridge chipset if known) or direct SATA.
|
||||
4. **Operating system and kernel:** `uname -a`.
|
||||
5. **Kernel messages during the failure:** `dmesg | tail -50` immediately after the crash.
|
||||
|
||||
@@ -22,7 +22,7 @@ fn main() {
|
||||
&target_arch // x86_64 → x86_64
|
||||
};
|
||||
|
||||
std::process::Command::new("cc")
|
||||
let cc_status = std::process::Command::new("cc")
|
||||
.args([
|
||||
"-arch",
|
||||
clang_arch,
|
||||
@@ -38,12 +38,19 @@ fn main() {
|
||||
"-O2",
|
||||
])
|
||||
.status()
|
||||
.expect("failed to compile macos_shim.c");
|
||||
.expect("failed to spawn cc for macos_shim.c");
|
||||
// `.status()` succeeding only means the process RAN. A real compile error
|
||||
// exits non-zero, and ignoring that left no object file, which surfaced
|
||||
// much later as an unexplained link failure against a missing symbol. The
|
||||
// shim is macOS-only and is neither linted nor compiled on the other two
|
||||
// platforms, so a mistake in it has exactly one chance to be noticed.
|
||||
assert!(cc_status.success(), "cc failed to compile macos_shim.c");
|
||||
|
||||
std::process::Command::new("ar")
|
||||
let ar_status = std::process::Command::new("ar")
|
||||
.args(["rcs", &lib, &obj])
|
||||
.status()
|
||||
.expect("failed to create static lib");
|
||||
.expect("failed to spawn ar");
|
||||
assert!(ar_status.success(), "ar failed to create the static lib");
|
||||
|
||||
println!("cargo:rustc-link-search=native={out_dir}");
|
||||
println!("cargo:rustc-link-lib=static=macos_scsi");
|
||||
@@ -77,10 +84,10 @@ fn emit_git_suffix() {
|
||||
// Re-run when HEAD (or the branch it points at) moves so the stamp stays
|
||||
// current without a clean rebuild.
|
||||
println!("cargo:rerun-if-changed=.git/HEAD");
|
||||
if let Ok(head) = std::fs::read_to_string(".git/HEAD") {
|
||||
if let Some(ref_path) = head.strip_prefix("ref: ") {
|
||||
println!("cargo:rerun-if-changed=.git/{}", ref_path.trim());
|
||||
}
|
||||
if let Ok(head) = std::fs::read_to_string(".git/HEAD")
|
||||
&& let Some(ref_path) = head.strip_prefix("ref: ")
|
||||
{
|
||||
println!("cargo:rerun-if-changed=.git/{}", ref_path.trim());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
+1
-1
@@ -12,7 +12,7 @@ Technical documentation for [libfreemkv](https://github.com/freemkv/libfreemkv),
|
||||
|----------|---------------|
|
||||
| [Architecture](architecture.md) | Module map, design principles, error codes, platform support |
|
||||
| [Drive Access](drive-access.md) | Drive, SCSI transport, profiles, unlock, why raw mode is needed |
|
||||
| [Rip Recovery](rip-recovery.md) | Three-layer recovery model: Disc::patch, single-shot Drive::read, DiscStream batch halving |
|
||||
| [Rip Recovery](rip-recovery.md) | What this crate owns of the recovery model: single-shot Drive::read, SenseFamily, DiscStream batch halving (the strategy itself moved to freemkv-engine in 1.6.0) |
|
||||
| [AACS Encryption](aacs.md) | Key resolution (4 paths), content decryption, bus encryption, SCSI handshake |
|
||||
| [UDF Filesystem](udf.md) | UDF 2.50 with metadata partitions, pointer chain, how files are read from disc |
|
||||
| [MPLS Playlists](mpls.md) | Playlist format, play items, STN stream table, coding types |
|
||||
|
||||
+27
-16
@@ -65,27 +65,38 @@ if disc.encrypted {
|
||||
}
|
||||
}
|
||||
|
||||
// Read content -- decryption is automatic
|
||||
let mut reader = disc.open_title(&mut session, 0).unwrap();
|
||||
while let Some(unit) = reader.read_unit().unwrap() {
|
||||
// decrypted content
|
||||
// Read content -- decryption is applied on read by the DiscStream decorator.
|
||||
// Live disc does NOT go through the URL resolver: `input("disc://...")` returns
|
||||
// Error::DiscUrlNotDirect by design.
|
||||
let keys = disc.decrypt_keys();
|
||||
let mut stream = DiscStream::new(
|
||||
Box::new(drive),
|
||||
disc.titles[0].clone(),
|
||||
keys,
|
||||
batch_sectors,
|
||||
disc.titles[0].content_format,
|
||||
false, // raw: false → decrypt on read
|
||||
None, // halt
|
||||
)?;
|
||||
while let Ok(Some(frame)) = stream.read() {
|
||||
// decrypted PES frames
|
||||
}
|
||||
```
|
||||
|
||||
The application never touches keys, never calls decryption functions, and never
|
||||
manages handshakes. All of that is internal to `Disc::scan()` and the content
|
||||
reader.
|
||||
The application never calls decryption functions and never manages the
|
||||
drive-level handshake. It DOES own key resolution — see below.
|
||||
|
||||
### KEYDB Location
|
||||
### Key resolution is the caller's job
|
||||
|
||||
`ScanOptions` controls where the keydb is loaded from. If no explicit path is
|
||||
set, the library checks the standard config locations. To specify an explicit
|
||||
path:
|
||||
`libfreemkv` is **lookup-free: it resolves no keys and reads no keydb.** There is
|
||||
no `ScanOptions::with_keydb`, and `ScanOptions` has no keydb field — its only
|
||||
scan input is the optional drive credentials for the live-drive authenticated
|
||||
handshake.
|
||||
|
||||
```rust
|
||||
let opts = ScanOptions::with_keydb("/path/to/keydb.cfg");
|
||||
let disc = Disc::scan(&mut session, &opts).unwrap();
|
||||
```
|
||||
The caller resolves a key out-of-band through a key source and applies it with
|
||||
[`Disc::decrypt_with`]. `freemkv-keysources` is the crate that implements the
|
||||
keydb and key-server sources; `ScanOptions::key_sources` takes them as
|
||||
`Box<dyn KeySource>`.
|
||||
|
||||
### AacsState
|
||||
|
||||
@@ -97,7 +108,7 @@ After a successful scan, `disc.aacs` contains an `AacsState`:
|
||||
| `bus_encryption` | `bool` | Whether bus encryption is active |
|
||||
| `mkb_version` | `Option<u32>` | MKB version from disc |
|
||||
| `disc_hash` | `String` | Identifier for the disc's key-input files |
|
||||
| `key_source` | `KeySource` | How the disc's key was resolved |
|
||||
| `key_source` | `KeyOrigin` | How the disc's key was resolved |
|
||||
|
||||
## keydb.cfg
|
||||
|
||||
|
||||
+10
-7
@@ -170,17 +170,20 @@ libfreemkv/src/
|
||||
│ ├── writeback_file.rs WritebackFile (was crate::io::Writer)
|
||||
│ └── writeback.rs sync_file_range pipeline
|
||||
├── drive/ Drive (open, init, single-shot read)
|
||||
│ ├── mod.rs Drive struct, init, read (single-shot), reset, eject
|
||||
│ ├── mod.rs Drive struct, init, read (single-shot), eject
|
||||
│ ├── capture.rs Raw drive SCSI capture (INQUIRY/GET_CONFIG) for contribution
|
||||
│ ├── linux.rs Linux drive discovery
|
||||
│ ├── macos.rs macOS drive discovery
|
||||
│ └── windows.rs Windows drive discovery
|
||||
├── disc/ Disc (scan, titles, AACS setup, sweep, patch)
|
||||
│ ├── mod.rs Disc struct, scan, titles, formats; Disc::copy + Disc::sweep (Pass 1)
|
||||
│ ├── sweep.rs Pass 1 internal helpers (pub(super))
|
||||
│ ├── patch.rs Disc::patch (Pass N retry over mapfile)
|
||||
│ ├── mapfile.rs ddrescue-format mapfile
|
||||
│ └── read_error.rs ReadCtx / ReadAction state machine
|
||||
├── disc/ Disc (scan, titles, AACS setup, per-format parsing)
|
||||
│ ├── mod.rs Disc struct, scan, titles, formats
|
||||
│ ├── bluray.rs Blu-ray / UHD scanning (MPLS/CLPI-driven)
|
||||
│ ├── dvd.rs DVD-Video scanning (IFO-driven)
|
||||
│ ├── hddvd.rs HD-DVD scanning
|
||||
│ ├── extract.rs Per-extent content extraction
|
||||
│ ├── encrypt.rs Encrypted-range mapping for content reads
|
||||
│ ├── dvd_audio_probe.rs DVD audio-stream probing
|
||||
│ └── pgs_forced_probe.rs PGS forced-subtitle probing
|
||||
├── scsi/ SCSI transport (Linux SG_IO, macOS IOKit, Windows SPTI)
|
||||
├── unlock.rs Unlocker trait + registry (pluggable unlock seam)
|
||||
├── aacs/ AACS decryption (handshake, keys, keydb, decrypt)
|
||||
|
||||
@@ -96,12 +96,13 @@ After open:
|
||||
- `init()` -- routes to the matching registered unlocker (if any); otherwise
|
||||
a no-op and the cert handshake carries the disc
|
||||
- `probe_disc()` -- probe disc surface for optimal speeds
|
||||
- `read(lba, count, buf, recovery)` -- single-shot read; `recovery` only selects the per-CDB timeout (1.5 s vs. 30 s)
|
||||
- `read(lba, count, buf, recovery)` -- single-shot read; `recovery` only selects the per-CDB timeout (`READ_TIMEOUT_MS` 10 s vs. `READ_RECOVERY_TIMEOUT_MS` 60 s)
|
||||
- `wait_ready()` -- wait for disc insertion
|
||||
- `eject()` -- eject tray
|
||||
|
||||
Recovery is layered above `Drive::read`, not inside it. Layer 1
|
||||
(`Disc::patch`) handles bad-range retry by replaying the ddrescue mapfile.
|
||||
(`freemkv_engine::recovery::patch`, in the engine crate) handles bad-range
|
||||
retry by replaying the ddrescue mapfile.
|
||||
Layer 3 (`DiscStream::fill_extents` adaptive batch sizer) handles in-loop
|
||||
request-size adaptation. Inline recovery (gentle retry → SCSI reset → retry)
|
||||
was removed in 0.13.6 — see [`rip-recovery.md`](rip-recovery.md) and
|
||||
|
||||
+15
-5
@@ -69,13 +69,23 @@ Each stream PID entry header (14 bytes):
|
||||
```
|
||||
Offset Size Field
|
||||
------ ---- -----
|
||||
0 2 Stream PID
|
||||
2 2 Reserved + EP stream type
|
||||
4 2 Number of coarse entries
|
||||
6 4 Number of fine entries (note: 32-bit, can be large)
|
||||
10 4 EP map start offset (relative to EP map start)
|
||||
2 2 stream_PID (byte-aligned)
|
||||
4 10 Bit-packed block, 80 bits total (see below)
|
||||
```
|
||||
|
||||
The stream PID entry is **not** byte-aligned past `stream_PID`. Bytes 4..14 are one
|
||||
80-bit packed field, read as a `u64` plus a trailing `u16`:
|
||||
|
||||
Bits Width Field
|
||||
---- ----- -----
|
||||
0-9 10 reserved
|
||||
10-13 4 EP_stream_type
|
||||
14-29 16 num_EP_coarse
|
||||
30-47 18 num_EP_fine
|
||||
48-79 32 EP_map_start_address (relative to the EP map start)
|
||||
|
||||
Note `num_EP_fine` is **18 bits**, not 32. See `parse_cpi` in `src/clpi.rs`.
|
||||
|
||||
libfreemkv parses only the first stream (primary video), which is sufficient for sector-level seeking.
|
||||
|
||||
### Two-Level Index
|
||||
|
||||
+2
-2
@@ -79,7 +79,7 @@ Insert disc
|
||||
│ Or: read sectors → decrypt → raw bytes (for ISO output)
|
||||
│ Drive::read() is single-shot. DiscStream::fill_extents adapts the
|
||||
│ batch size on failure (halve / probe-up). Bad-range retry is layer
|
||||
│ 1 above this — Disc::patch re-runs against the mapfile.
|
||||
│ 1 above this — freemkv_engine::recovery::patch re-runs against the mapfile.
|
||||
│
|
||||
▼
|
||||
PES frames → output stream (MKV, M2TS, network, etc.)
|
||||
@@ -123,7 +123,7 @@ output.finish()?;
|
||||
| aacs/ | [aacs.md](aacs.md) | Key resolution + content decrypt + bus handshake |
|
||||
| css/ | -- | DVD CSS cipher |
|
||||
| decrypt.rs | -- | Unified decrypt dispatcher (AACS/CSS/None) |
|
||||
| disc/ | [rip-recovery.md](rip-recovery.md) | Disc::scan + Disc::sweep + Disc::patch + mapfile |
|
||||
| disc/ | [rip-recovery.md](rip-recovery.md) | Disc::scan (sweep/patch/mapfile moved to freemkv-engine in 1.6.0) |
|
||||
| labels/ | -- | BD-J stream labels (5 format parsers) |
|
||||
| mux/ | -- | Stream implementations (7 stream types) |
|
||||
| pes.rs | -- | PES frame types + FrameSource / FrameSink traits |
|
||||
|
||||
@@ -58,15 +58,16 @@ selects the per-CDB timeout:
|
||||
|
||||
| `recovery` | Timeout | Used by |
|
||||
|------------|----------|------------------------------------------|
|
||||
| `false` | 1.5 s | `Disc::sweep` fast skip-forward pass, `DiscStream::fill_extents` |
|
||||
| `true` | 30 s | `Disc::patch` retry pass over the mapfile |
|
||||
| `false` | 10 s | `freemkv_engine::recovery::sweep` fast skip-forward pass, `DiscStream::fill_extents` |
|
||||
| `true` | 60 s | `freemkv_engine::recovery::patch` retry pass over the mapfile |
|
||||
|
||||
On any SCSI failure or timeout, `read` returns `Err(DiscRead)` immediately.
|
||||
There are no inline retries, no SCSI reset, no Phase 1/2/3 escalation.
|
||||
|
||||
Recovery is layered above `Drive::read`:
|
||||
|
||||
- **Layer 1 — `Disc::patch`** loops over the ddrescue mapfile and re-issues
|
||||
- **Layer 1 — `freemkv_engine::recovery::patch`** (in the engine crate, not
|
||||
here) loops over the ddrescue mapfile and re-issues
|
||||
`read(.., recovery=true)` against each non-`+` range.
|
||||
- **Layer 3 — `DiscStream::fill_extents`** halves the request size on
|
||||
failure, retries at the same LBA, and probes back up on a clean-read
|
||||
|
||||
+64
-167
@@ -1,135 +1,38 @@
|
||||
# Rip recovery — three-layer architecture
|
||||
# Rip recovery — what libfreemkv owns
|
||||
|
||||
`libfreemkv` supports a multi-stage rip model for damaged or protection-bearing
|
||||
discs: a fast forward sweep that tolerates read failures, in-loop request-size
|
||||
adaptation that survives transient drive trouble without bailing, and targeted
|
||||
retry passes against a persistent bad-range map. The stream pipeline
|
||||
(`DiscStream` + `input`/`output`) operates against the resulting ISO image, so
|
||||
the mux stage never touches the drive.
|
||||
**Recovery strategy moved OUT of this crate in 1.6.0.** The forward sweep, the
|
||||
targeted retry pass, the ddrescue mapfile, damage classification and the
|
||||
multipass loop now live in the **`freemkv-engine`** crate as
|
||||
`freemkv_engine::recovery::{copy, sweep, patch}`. The dependency runs
|
||||
engine → libfreemkv, so this crate cannot call into the engine; front-ends
|
||||
(`freemkv` CLI, autorip) get recovery from the engine directly.
|
||||
|
||||
Recovery is layered cleanly. Each layer has one responsibility and does not
|
||||
reach into the others.
|
||||
What stayed here are the two layers underneath the strategy: the single-shot
|
||||
read primitive, and the in-stream request-size adaptation that sits in front of
|
||||
it. This document covers those, plus the design constraints they encode — the
|
||||
constraints are the reason the strategy above them looks the way it does, so
|
||||
they belong with the code that enforces them.
|
||||
|
||||
For the strategy itself — damage-jump thresholds, pass ordering, mapfile status
|
||||
state machine, wedge detection — read `freemkv-engine/src/recovery/`.
|
||||
|
||||
| Layer | Where it lives | What it does |
|
||||
|-------|---------------|--------------|
|
||||
| 1 — Bad-range retry | `Disc::patch` (one pass over the mapfile per call) | Re-reads non-`+` ranges with the long timeout. Idempotent; caller invokes N times. |
|
||||
| 1 — Bad-range retry | **`freemkv-engine`** (`recovery::patch`) | Re-reads non-`+` ranges with the long timeout. Idempotent; caller invokes N times. |
|
||||
| 2 — Single-shot primitive | `Drive::read` in `src/drive/mod.rs` | One CDB, one timeout, one result. No inline retries, no SCSI reset. |
|
||||
| 3 — In-loop request adaptation | `DiscStream::fill_extents` adaptive batch sizer | Halves the batch on failure, retries at the same LBA, walks back up on a clean-read streak. |
|
||||
| 3 — In-loop request adaptation | `DiscStream::fill_extents` in `src/mux/disc.rs` | Halves the batch on failure, retries at the same LBA, walks back up on a clean-read streak. |
|
||||
|
||||
The library exposes flat verbs; the caller drives the multipass loop. Autorip
|
||||
runs `Disc::sweep` once, then loops `Disc::patch` until either the mapfile is
|
||||
clean or the configured retry budget is exhausted, then hands the ISO off to
|
||||
the mux pipeline. The `freemkv` CLI does the same shape with a
|
||||
terminal-output progress sink. Layer 3 runs inside any consumer of
|
||||
`DiscStream` (direct PES pipeline, ISO playback, etc.) without caller
|
||||
involvement.
|
||||
Layer 2 also translates drive facts: [`SenseFamily`](../src/scsi/mod.rs)
|
||||
classifies SCSI sense data into the categories the engine's strategy routes on
|
||||
(marginal vs. hardware vs. not-ready). Getting that classification wrong
|
||||
silently misroutes recovery, which is why it lives next to the transport rather
|
||||
than in the strategy.
|
||||
|
||||
Three primitives compose the disc-side flow:
|
||||
Layer 3 runs inside any consumer of `DiscStream` — direct PES pipeline, ISO
|
||||
playback — without caller involvement, and applies whether or not the engine's
|
||||
recovery is in play.
|
||||
|
||||
| Primitive | What it does |
|
||||
|---------------------------|-----------------------------------------------------------------------|
|
||||
| `Disc::sweep` | disc → ISO, one forward pass. Writes a sidecar `.mapfile`. Opt-in skip-on-error. |
|
||||
| `Disc::patch` | Re-reads bad ranges from the drive. One pass per call; caller invokes N times. |
|
||||
| `DiscStream` (ISO source) | Reads sectors from the ISO, feeds decrypt → demux → codec → mux. |
|
||||
|
||||
## Data model
|
||||
|
||||
### Mapfile
|
||||
|
||||
Format: [ddrescue](https://www.gnu.org/software/ddrescue/manual/ddrescue_manual.html)-compatible
|
||||
plain text, greppable, tool-interoperable. Flushed to disk on every `record()`
|
||||
so a crashed rip loses at most one block.
|
||||
|
||||
```
|
||||
# Rescue Logfile. Created by libfreemkv v0.13.6
|
||||
# Current pos / status / pass / pass_time
|
||||
0x000000000 ? 1 0
|
||||
# pos size status
|
||||
0x000000000 0x12a35d000 +
|
||||
0x12a35d000 0x000003000 -
|
||||
0x12a360000 0x009c4a000 +
|
||||
0x12d00a000 0x000064000 *
|
||||
```
|
||||
|
||||
Status characters match ddrescue:
|
||||
|
||||
| Char | Meaning |
|
||||
|------|----------------------------------------------------|
|
||||
| `?` | Not yet attempted |
|
||||
| `*` | Fast-pass failed; needs edge-trim |
|
||||
| `/` | Trimmed; interior needs sector scrape |
|
||||
| `-` | Unreadable this session |
|
||||
| `+` | Finished (good) |
|
||||
|
||||
Position and size are hex byte offsets into the ISO.
|
||||
|
||||
### `SweepOptions` and `PatchOptions`
|
||||
|
||||
The library no longer dispatches between sweep and patch internally — the
|
||||
caller picks the verb explicitly per pass. The two option structs are flat
|
||||
and have no overlap:
|
||||
|
||||
```rust
|
||||
SweepOptions {
|
||||
decrypt: true,
|
||||
resume: false,
|
||||
batch_sectors: None,
|
||||
skip_on_error: true, // damage-jump + zero-fill on read failure
|
||||
progress: Some(&reporter),
|
||||
halt: Some(flag),
|
||||
}
|
||||
|
||||
PatchOptions {
|
||||
decrypt: true,
|
||||
block_sectors: None,
|
||||
full_recovery: true,
|
||||
reverse: true, // walk bad ranges high → low LBA
|
||||
wedged_threshold: 50,
|
||||
progress: Some(&reporter),
|
||||
halt: Some(flag),
|
||||
}
|
||||
```
|
||||
|
||||
Caller-orchestrated dispatch (the policy `Disc::copy` used to embed):
|
||||
|
||||
- No mapfile → `sweep` (fresh Pass 1).
|
||||
- Mapfile with `?` ranges → `sweep` with `resume: true`.
|
||||
- Mapfile covers full disc, only `*` / `/` / `-` ranges → `patch`.
|
||||
- Mapfile clean → done; no further pass needed.
|
||||
|
||||
Each consumer (autorip, `freemkv` CLI) implements the loop in roughly five
|
||||
lines of `Mapfile::stats()` checks.
|
||||
|
||||
## Algorithm
|
||||
|
||||
### Pass 1 — fast sweep (`Disc::sweep`)
|
||||
|
||||
1. Read one ECC block (32 sectors for UHD, 16 for BD/DVD) at the current LBA.
|
||||
2. On success: write data to ISO, mark `+`, advance.
|
||||
3. On failure (with `multipass`): zero-fill, mark `*`, advance.
|
||||
4. Track a sliding window of the last 16 ECC block results. When ≥12% are failures
|
||||
→ **damage-jump**: skip ahead by `1024×batch×multiplier` sectors (64 MB base for
|
||||
UHD). Double the multiplier on each jump (64→128→256→512 MB...). Zero-fill the gap as `*`.
|
||||
5. On 16 consecutive good reads: reset jump multiplier to 1, restore max read speed.
|
||||
6. Speed control: damage zone entry → minimum speed, exit → maximum speed.
|
||||
7. Only transport failures (USB bridge crash) abort the pass.
|
||||
|
||||
Pass 1 completes when every byte has been visited (either `+` or `*`).
|
||||
|
||||
### Pass 2+ — patch (`Disc::patch`)
|
||||
|
||||
`Disc::patch` reads the mapfile and iterates every non-`+` range. Default: **reverse** mode
|
||||
(walks ranges from highest LBA to lowest, within each range from end to start).
|
||||
|
||||
1. Issue a single-sector read with 60 s timeout (`recovery=true`). Drive firmware
|
||||
does its own ECC recovery inside that window.
|
||||
2. On success: write the good bytes into the ISO, mark `+`.
|
||||
3. On failure with non-marginal SCSI sense: bail immediately (drive won't produce data).
|
||||
4. On failure with marginal sense: mark `-`, continue.
|
||||
5. Update the mapfile after every block — crash-safe resume.
|
||||
6. Wedged-drive exit: 50 consecutive failures with zero recovery → bail this pass.
|
||||
|
||||
### In-stream — adaptive batch halving (`DiscStream::fill_extents`)
|
||||
## In-stream — adaptive batch halving (`DiscStream::fill_extents`)
|
||||
|
||||
When a consumer reads a `DiscStream` directly (no ISO intermediate),
|
||||
`fill_extents` runs an adaptive sizer in front of `Drive::read`:
|
||||
@@ -143,59 +46,53 @@ When a consumer reads a `DiscStream` directly (no ISO intermediate),
|
||||
`EventKind::SectorSkipped`) when `skip_errors` is set, otherwise return
|
||||
`Err(DiscRead)`.
|
||||
|
||||
This is layer 3. It exists so a transient single-sector glitch in a 32-sector
|
||||
batch can be isolated and read individually without the caller needing to
|
||||
implement retry logic.
|
||||
This exists so a transient single-sector glitch inside a 32-sector batch can be
|
||||
isolated and read individually without the caller implementing retry logic. See
|
||||
[`src/event.rs`](../src/event.rs) for the emitted events.
|
||||
|
||||
## Design choices
|
||||
|
||||
**`Drive::read` is single-shot.** No inline retry phases, no SCSI reset,
|
||||
no eject cycle. The `recovery` flag controls only the per-CDB timeout
|
||||
(1.5 s vs. 30 s); on any failure it returns `Err(DiscRead)` immediately.
|
||||
Inline recovery (5× gentle retry → close + SCSI reset + reopen → 5× more)
|
||||
was removed in 0.13.6. See the stop-wedge postmortem (2026-04-25) for rationale:
|
||||
the inline reset on the LG BU40N (Initio USB-SATA bridge)
|
||||
wedged drive firmware below the bridge without ever recovering a sector,
|
||||
and the gentle-retry phase produced long stretches of 0 KB/s with no
|
||||
recoveries to show for it. Recovery responsibility is now layered: layer 1
|
||||
handles ranges, layer 3 handles request size, neither touches the
|
||||
wedge-prone reset path.
|
||||
These are constraints on the read path, enforced here and relied on by the
|
||||
engine's strategy.
|
||||
|
||||
**No `MODE SELECT` to disable drive retries.** Neither ddrescue
|
||||
nor any consumer ripper does this. Drive firmware has access to raw analog signal, laser
|
||||
power control, and drive-specific ECC tuning that userspace can't replicate —
|
||||
disabling it throws away recovery headroom on marginal sectors. We fail fast
|
||||
via short SG_IO timeouts in pass 1 and let the firmware work the long timeout
|
||||
in pass 2 / patch.
|
||||
**`Drive::read` is single-shot.** No inline retry phases, no SCSI reset, no
|
||||
eject cycle. The `recovery` flag controls only the per-CDB timeout (10 s vs.
|
||||
60 s); on any failure it returns `Err(DiscRead)` immediately. Inline recovery
|
||||
(5× gentle retry → close + SCSI reset + reopen → 5× more) was removed in
|
||||
0.13.6. See the stop-wedge postmortem (2026-04-25) for rationale: the inline
|
||||
reset on the LG BU40N (Initio USB-SATA bridge) wedged drive firmware below the
|
||||
bridge without ever recovering a sector, and the gentle-retry phase produced
|
||||
long stretches of 0 KB/s with nothing to show for it. Recovery responsibility
|
||||
is layered instead: layer 1 handles ranges, layer 3 handles request size,
|
||||
neither touches the wedge-prone reset path.
|
||||
|
||||
**No SCSI reset from any retry path.** `SgIoTransport::reset` (Linux) is
|
||||
trimmed to a kernel SG_IO state flush plus ALLOW MEDIUM REMOVAL — the
|
||||
`SG_SCSI_RESET` ioctl and STOP/START UNIT escalation were removed in 0.13.6.
|
||||
The macOS reset (which had been a no-op) was removed entirely. The top-level
|
||||
`scsi::reset()` / `reset_with_timeout()` / `reset_blocking()` wrappers were
|
||||
also removed (no callers). The remaining `Drive::reset()` is only invoked
|
||||
explicitly by callers that need an eject-cycle escape hatch — it is never
|
||||
reached from a read path.
|
||||
**No `MODE SELECT` to disable drive retries.** Neither ddrescue nor any
|
||||
consumer ripper does this. Drive firmware has access to raw analog signal,
|
||||
laser power control and drive-specific ECC tuning that userspace cannot
|
||||
replicate — disabling it throws away recovery headroom on marginal sectors. The
|
||||
fast pass fails quickly via short SG_IO timeouts and lets the firmware work the
|
||||
long timeout during retry.
|
||||
|
||||
**ISO intermediate, even for single-pass.** Pass 1 always writes an ISO. The
|
||||
mux stage reads the ISO via `FileSectorSource`. For single-pass (no retries),
|
||||
this adds ~2-3 min (local disk mux) but gains resumability across crashes,
|
||||
**No SCSI reset from any read path.** There is no reset escape hatch on
|
||||
`Drive` at all: the `SG_SCSI_RESET` ioctl and STOP/START UNIT escalation went in
|
||||
0.13.6, the macOS reset (always a no-op) was removed entirely, and the
|
||||
top-level `scsi::reset()` wrappers went with their last callers. The only
|
||||
remaining reset is a Windows-specific device-level helper in
|
||||
[`src/scsi/windows.rs`](../src/scsi/windows.rs), never reached from a read.
|
||||
|
||||
**ISO intermediate, even for single-pass.** The engine's Pass 1 always writes
|
||||
an ISO, and the mux stage reads it back via `FileSectorSource`. For a
|
||||
no-retry rip this costs a few minutes but buys resumability across crashes,
|
||||
re-muxability without re-ripping, and a persistent forensic artifact. Callers
|
||||
who need pure speed can bypass and use `DiscStream::new(Box::new(drive), …)`
|
||||
directly — the lib doesn't forbid it, and layer 3 (adaptive batch halving)
|
||||
still applies there.
|
||||
|
||||
**Mapfile in ddrescue format.** Plain text so users can `less` it, `diff` it,
|
||||
or feed it to ddrescue's own tooling. Crash-safe (flush-per-record). Entries
|
||||
coalesce on adjacent same-status ranges so files stay small.
|
||||
|
||||
**Patches target `-`, `*`, `/`, and `?` alike.** The status state machine is
|
||||
ddrescue's but `patch` collapses the distinction — it just tries every
|
||||
non-finished range with the long timeout. Future work can specialize (trim vs.
|
||||
scrape vs. retry with direction reversal) if there's measured benefit.
|
||||
who need pure speed can bypass it with `DiscStream::new(Box::new(drive), …)` —
|
||||
nothing forbids it, and layer 3 still applies there.
|
||||
|
||||
## References
|
||||
|
||||
- [ddrescue manual, Algorithm chapter](https://www.gnu.org/software/ddrescue/manual/ddrescue_manual.html)
|
||||
- [ddrescue optical media notes](https://www.electric-spoon.com/doc/gddrescue/html/Optical-media.html)
|
||||
- Source: [`src/disc/mapfile.rs`](../src/disc/mapfile.rs), [`src/disc/mod.rs`](../src/disc/mod.rs) (`Disc::sweep`), [`src/disc/patch.rs`](../src/disc/patch.rs) (`Disc::patch`), [`src/drive/mod.rs`](../src/drive/mod.rs) (`Drive::read`), [`src/mux/disc.rs`](../src/mux/disc.rs) (`DiscStream::fill_extents`).
|
||||
- Recovery strategy and mapfile: `freemkv-engine/src/recovery/`
|
||||
- In this crate: [`src/drive/mod.rs`](../src/drive/mod.rs) (`Drive::read`),
|
||||
[`src/scsi/mod.rs`](../src/scsi/mod.rs) (`SenseFamily`),
|
||||
[`src/mux/disc.rs`](../src/mux/disc.rs) (`DiscStream::fill_extents`),
|
||||
[`src/event.rs`](../src/event.rs) (progress events).
|
||||
|
||||
+1
-1
@@ -114,7 +114,7 @@ The `read_filesystem()` function in `src/udf.rs` follows the pointer chain above
|
||||
2. Scans sectors 32-63 for the Partition Descriptor and Logical Volume Descriptor.
|
||||
3. If two partition maps exist and the second is Type 2, reads the metadata file ICB at partition_start to find metadata_start.
|
||||
4. Reads the FSD at metadata_start, extracts the root directory ICB LBA.
|
||||
5. Calls `read_directory()` recursively (max depth 3) to build the full file tree.
|
||||
5. Calls `read_directory()` recursively (max depth `MAX_DIR_DEPTH` = 8) to build the full file tree.
|
||||
|
||||
Each directory read involves two sector reads: one for the ICB, then one or more for the directory data. File sizes are read from info_length in each file's ICB.
|
||||
|
||||
|
||||
@@ -1,83 +0,0 @@
|
||||
// Minimal ISO dumper — find exact stall point
|
||||
use libfreemkv::Drive;
|
||||
use std::io::{BufWriter, Write};
|
||||
use std::path::Path;
|
||||
use std::time::Instant;
|
||||
|
||||
fn main() {
|
||||
let args: Vec<String> = std::env::args().collect();
|
||||
if args.len() < 3 {
|
||||
eprintln!("Usage: iso_dump <device> <output>");
|
||||
std::process::exit(1);
|
||||
}
|
||||
|
||||
let mut drive = Drive::open(Path::new(&args[1])).unwrap();
|
||||
drive.wait_ready().unwrap();
|
||||
let _ = drive.init();
|
||||
let _ = drive.probe_disc();
|
||||
|
||||
// AACS handshake — required to read past the protected area
|
||||
eprint!("Scanning disc... ");
|
||||
let _ = libfreemkv::Disc::scan(&mut drive, &libfreemkv::ScanOptions::default());
|
||||
eprintln!("OK");
|
||||
|
||||
let cap = drive.read_capacity().unwrap();
|
||||
let batch = libfreemkv::disc::detect_max_batch_sectors(drive.device_path());
|
||||
|
||||
eprintln!("Device: {} | {} sectors | batch {}", args[1], cap, batch);
|
||||
|
||||
let file = std::fs::File::create(&args[2]).unwrap();
|
||||
let mut w = BufWriter::with_capacity(4 * 1024 * 1024, file);
|
||||
let mut buf = vec![0u8; batch as usize * 2048];
|
||||
let mut lba: u32 = 0;
|
||||
let start = Instant::now();
|
||||
let mut last = Instant::now();
|
||||
let mut bytes: u64 = 0;
|
||||
let mut last_bytes: u64 = 0;
|
||||
|
||||
while lba < cap {
|
||||
let count = ((cap - lba) as u16).min(batch);
|
||||
let n = count as usize * 2048;
|
||||
|
||||
// Tiny yield between reads — test if pacing prevents firmware throttle
|
||||
std::thread::yield_now();
|
||||
let t0 = Instant::now();
|
||||
let ok = drive.read(lba, count, &mut buf[..n], true).is_ok();
|
||||
let read_ms = t0.elapsed().as_millis();
|
||||
|
||||
// Flag slow reads
|
||||
if read_ms > 2000 {
|
||||
eprintln!("\n SLOW READ: LBA {} took {}ms (ok={})", lba, read_ms, ok);
|
||||
}
|
||||
|
||||
if !ok {
|
||||
buf[..n].fill(0);
|
||||
}
|
||||
w.write_all(&buf[..n]).unwrap();
|
||||
lba += count as u32;
|
||||
bytes += n as u64;
|
||||
|
||||
if last.elapsed().as_millis() >= 1000 {
|
||||
let delta = bytes - last_bytes;
|
||||
let speed = delta as f64 / last.elapsed().as_secs_f64() / 1_048_576.0;
|
||||
let avg = bytes as f64 / start.elapsed().as_secs_f64() / 1_048_576.0;
|
||||
let pct = bytes as f64 / (cap as f64 * 2048.0) * 100.0;
|
||||
eprint!(
|
||||
"\r {:.1}% LBA {} | {:.0} MB/s (avg {:.0}) | {:.1} GB ",
|
||||
pct,
|
||||
lba,
|
||||
speed,
|
||||
avg,
|
||||
bytes as f64 / 1e9
|
||||
);
|
||||
last_bytes = bytes;
|
||||
last = Instant::now();
|
||||
}
|
||||
}
|
||||
w.flush().unwrap();
|
||||
eprintln!(
|
||||
"\nDone: {:.1} GB in {:.0}s",
|
||||
bytes as f64 / 1e9,
|
||||
start.elapsed().as_secs_f64()
|
||||
);
|
||||
}
|
||||
+402
-67
@@ -4,11 +4,11 @@
|
||||
#[cfg(test)]
|
||||
use aes::Aes128;
|
||||
#[cfg(test)]
|
||||
use aes::cipher::{KeyInit, generic_array::GenericArray};
|
||||
use aes::cipher::{Array, KeyInit};
|
||||
|
||||
use super::crypto::{aes_cbc_decrypt, aes_ecb_encrypt};
|
||||
// Available at module scope for this module's test fixtures (they reference
|
||||
// `super::AACS_IV` when building CBC ciphertext directly); test-only.
|
||||
use super::crypto::{aes_cbc_decrypt, aes_cbc_encrypt, aes_ecb_encrypt};
|
||||
// Only this module's test fixtures build CBC ciphertext by hand now — the
|
||||
// production paths get the IV from `aes_cbc_encrypt` / `aes_cbc_decrypt`.
|
||||
#[cfg(test)]
|
||||
use super::crypto::AACS_IV;
|
||||
|
||||
@@ -41,7 +41,8 @@ pub const ALIGNED_UNIT_SECTORS: u32 = (ALIGNED_UNIT_LEN / SECTOR_BYTES) as u32;
|
||||
/// underflow wraps to ~2^32 and, because `2^32 ≡ 1 (mod 3)`, mis-reports the
|
||||
/// alignment (e.g. `lba == unit_base - 1` would falsely read as aligned).
|
||||
pub fn is_unit_aligned(lba: u32, unit_base: u32) -> bool {
|
||||
lba.saturating_sub(unit_base) % ALIGNED_UNIT_SECTORS == 0
|
||||
lba.saturating_sub(unit_base)
|
||||
.is_multiple_of(ALIGNED_UNIT_SECTORS)
|
||||
}
|
||||
|
||||
use crate::consts::SECTOR_BYTES;
|
||||
@@ -106,6 +107,23 @@ pub fn aacs_unit_encrypted(unit: &[u8], format: crate::disc::ContentFormat) -> b
|
||||
}
|
||||
}
|
||||
|
||||
/// The AACS encrypted flag from an aligned unit's CLEAR seed, readable even on a
|
||||
/// trailing PARTIAL unit (unlike [`aacs_unit_encrypted`], which requires a whole
|
||||
/// 6144-byte unit). The flag lives at a fixed low offset in the clear header, so a
|
||||
/// fragment that still contains that byte can be classified. Used to catch an
|
||||
/// encrypted unit truncated across a buffer/extent boundary — a fragment we cannot
|
||||
/// CBC-decrypt and must not emit as clear. `false` for a slice too short to hold
|
||||
/// the flag byte. Same clip-anchored-read caveat as [`aacs_unit_encrypted`].
|
||||
pub fn aacs_unit_seed_encrypted(unit: &[u8], format: crate::disc::ContentFormat) -> bool {
|
||||
use crate::disc::ContentFormat;
|
||||
match format {
|
||||
ContentFormat::BdTs => unit.first().is_some_and(|b| b & 0xC0 != 0),
|
||||
ContentFormat::MpegPs => unit
|
||||
.get(PS_SCRAMBLE_OFF)
|
||||
.is_some_and(|b| b & PS_SCRAMBLE_MASK != 0),
|
||||
}
|
||||
}
|
||||
|
||||
/// True when an aligned unit is flagged encrypted AND still looks scrambled
|
||||
/// (structure not yet restored) — i.e. genuine encrypted content NOT yet decrypted.
|
||||
///
|
||||
@@ -129,8 +147,14 @@ pub fn aacs_unit_needs_decrypt(unit: &[u8], format: crate::disc::ContentFormat)
|
||||
}
|
||||
|
||||
/// Minimum synced content packets that PROVE a key opened a unit. Four `0x47`
|
||||
/// syncs ≈ 32 bits of MPEG-TS structure ≈ 1-in-4-billion that a wrong key (uniform
|
||||
/// AES noise, `0x47` at 1/256 per packet) fakes it. It is an ABSOLUTE proof floor,
|
||||
/// syncs are 32 bits of MPEG-TS structure, but the per-UNIT false-pass risk is
|
||||
/// NOT 2^-32: `is_clean_ts` accepts ANY four of the ~31 encrypted packets in a
|
||||
/// 6144-byte aligned unit, so for a wrong key (uniform AES noise, `0x47` at 1/256
|
||||
/// per packet) it is ≈ C(31,4)·256^-4 ≈ 7e-6, i.e. ~1e-5 — the figure
|
||||
/// [`is_clean_ts`]'s own doc below states. 1-in-4-billion is the probability for
|
||||
/// four SPECIFIC packets and overstates the margin by ~4000x; at
|
||||
/// `KEY_PROOF_PACKETS = 3` the per-unit rate is ≈ C(31,3)·256^-3 ≈ 2.6e-4, so do
|
||||
/// NOT lower it on the strength of slack that is not there. It is an ABSOLUTE proof floor,
|
||||
/// NOT a proportion — a unit the key opened but whose content is bad-encoded
|
||||
/// (many non-conforming packets) is proven by ANY four good packets, not rejected
|
||||
/// for the bad ones.
|
||||
@@ -154,8 +178,16 @@ pub fn is_clean(unit: &[u8], format: crate::disc::ContentFormat) -> bool {
|
||||
/// Structural "does this unit carry enough valid MPEG-TS to prove a key opened
|
||||
/// it?" — the Transport-Stream arm of [`is_clean`]. It is
|
||||
/// NOT a decryption verdict: [`decrypt_unit`] applies a key (that is
|
||||
/// "decrypt"); whether the plaintext is clean TS is this SEPARATE question. The
|
||||
/// mux never calls this — TS validity is a muxer concern, never a decrypt result.
|
||||
/// "decrypt"); whether the plaintext is clean TS is this SEPARATE question.
|
||||
///
|
||||
/// The mux is its PRINCIPAL consumer for `BdTs` discs, reaching it through
|
||||
/// [`is_clean`]: `mux::resolve`'s multi-CPS `pick` closure selects a unit key by
|
||||
/// it, `probe_index_phase` reports each FMTS index's interleave parity by it, and
|
||||
/// `decrypt::decrypt_sectors_mapped` uses it as the forensic-range verify net.
|
||||
/// (The doc used to say "the mux never calls this", which invited a maintainer to
|
||||
/// tighten or loosen the proof rule below believing only whole-disc read
|
||||
/// verification was affected — while it in fact changes which unit key a
|
||||
/// multi-CPS disc muxes with and which phase an FMTS index is muxed at.)
|
||||
///
|
||||
/// Rule — evidence is ABSOLUTE, scaled to the packets that exist. Over the
|
||||
/// ENCRYPTED packets (skip packet 0: its `0x47` sits in the clear 16-byte seed, so
|
||||
@@ -309,16 +341,74 @@ pub fn decrypt_unit(unit: &mut [u8], unit_key: &[u8; 16]) {
|
||||
}
|
||||
}
|
||||
|
||||
/// Encrypt one AACS aligned unit (6144 bytes) IN PLACE — the exact inverse of
|
||||
/// [`decrypt_unit`], for authoring an encrypted disc image (and for building
|
||||
/// genuinely-encrypted read-path fixtures).
|
||||
///
|
||||
/// PURE, on the same terms as `decrypt_unit`: it applies the key and nothing else.
|
||||
/// It does NOT set the encrypted flag, because where that flag lives is
|
||||
/// container-specific (CPI bits in byte 0 for BD-TS, elsewhere for HD-DVD-PS) and
|
||||
/// keeping it out is what lets this stay container-agnostic. The only guard is the
|
||||
/// length check, since the crypto is defined only over a whole 6144-byte unit.
|
||||
///
|
||||
/// **Set the encrypted flag BEFORE calling, never after.** Bytes 0..16 are the key
|
||||
/// seed and are left in plaintext, so the Block Key derives from them: mutating any
|
||||
/// header byte after encrypting changes the seed a decryptor will derive from and
|
||||
/// silently yields garbage. The caller's order must be flag, then encrypt.
|
||||
///
|
||||
/// Block Key = AES-128E(Kcu, seed) ⊕ seed, then AES-128-CBC **encrypt** bytes
|
||||
/// 16..6144 under the AACS IV — the forward direction of the same construction
|
||||
/// `decrypt_unit` documents, sharing its module-scope primitives so the two cannot
|
||||
/// drift apart.
|
||||
///
|
||||
/// Note one deliberate asymmetry: `decrypt_unit` restores all-zero-on-disc source
|
||||
/// padding packets to zero. This does not, and need not — an all-zero plaintext
|
||||
/// packet encrypts to ciphertext that is not all-zero, so it is not mistaken for
|
||||
/// padding on the way back and the round trip is still exact. Authoring that wants
|
||||
/// true source-zero padding leaves those packets unencrypted instead.
|
||||
///
|
||||
/// Returns `false` — encrypting nothing — when `unit` is shorter than
|
||||
/// [`ALIGNED_UNIT_LEN`]. That case MUST be checked: the caller has already set the
|
||||
/// container's encrypted flag by then (this function's contract requires it), so
|
||||
/// ignoring the result leaves a unit marked encrypted while still carrying
|
||||
/// plaintext, which is the worst possible outcome for an authoring tool.
|
||||
#[must_use = "returns false when the slice is too short to encrypt, leaving \
|
||||
plaintext behind a flag that already says 'encrypted'"]
|
||||
pub fn encrypt_unit(unit: &mut [u8], unit_key: &[u8; 16]) -> bool {
|
||||
if unit.len() < ALIGNED_UNIT_LEN {
|
||||
return false;
|
||||
}
|
||||
let mut header = [0u8; 16];
|
||||
header.copy_from_slice(&unit[..16]);
|
||||
let derived = aes_ecb_encrypt(unit_key, &header);
|
||||
let mut k = [0u8; 16];
|
||||
for i in 0..16 {
|
||||
k[i] = derived[i] ^ header[i];
|
||||
}
|
||||
// CBC-encrypt bytes 16.. under the fixed AACS IV — the exact forward of the
|
||||
// `aes_cbc_decrypt` call in `decrypt_unit`, and one key expansion for the whole
|
||||
// unit rather than one per 16-byte block.
|
||||
aes_cbc_encrypt(&k, &mut unit[16..ALIGNED_UNIT_LEN]);
|
||||
true
|
||||
}
|
||||
|
||||
/// Remove bus encryption from an aligned unit (AACS 2.0 / UHD).
|
||||
/// Bus encryption uses read_data_key, decrypting bytes 16..2048 of each 2048-byte sector.
|
||||
pub(crate) fn decrypt_bus(unit: &mut [u8], read_data_key: &[u8; 16]) {
|
||||
// Expand the key schedule ONCE for the whole unit. `read_data_key` is
|
||||
// loop-invariant here (and constant for the entire disc), but calling
|
||||
// `aes_cbc_decrypt` per sector rebuilt the AES-128 schedule per sector — three
|
||||
// expansions per 6144-byte aligned unit, i.e. ~29 million redundant expansions
|
||||
// over a 90 GB read on a stock drive, on the per-unit decrypt hot path.
|
||||
// Measured by `decrypt_bus_expands_the_read_data_key_once_per_unit`.
|
||||
let cipher = crate::aacs::crypto::new_cipher_for(read_data_key);
|
||||
for sector_start in (0..ALIGNED_UNIT_LEN).step_by(SECTOR_BYTES) {
|
||||
if sector_start + SECTOR_BYTES > unit.len() {
|
||||
break;
|
||||
}
|
||||
// First 16 bytes of each sector are plaintext
|
||||
aes_cbc_decrypt(
|
||||
read_data_key,
|
||||
crate::aacs::crypto::cbc_decrypt_blocks(
|
||||
&cipher,
|
||||
&mut unit[sector_start + 16..sector_start + SECTOR_BYTES],
|
||||
);
|
||||
}
|
||||
@@ -328,7 +418,105 @@ pub(crate) fn decrypt_bus(unit: &mut [u8], read_data_key: &[u8; 16]) {
|
||||
mod tests {
|
||||
use super::super::crypto::aes_ecb_decrypt;
|
||||
use super::*;
|
||||
use aes::cipher::BlockEncrypt; // test fixtures build ciphertext directly
|
||||
use aes::cipher::BlockCipherEncrypt; // test fixtures build ciphertext directly
|
||||
|
||||
/// [`encrypt_unit`] is the exact inverse of [`decrypt_unit`]: whatever an
|
||||
/// authoring caller encrypts, the read path must recover byte-for-byte.
|
||||
///
|
||||
/// Mutation: drop the trailing `⊕ header` from either function's Block Key
|
||||
/// derivation, or swap `AACS_IV` for zeroes in one of them, and the two stop
|
||||
/// agreeing -> this fails.
|
||||
#[test]
|
||||
fn encrypt_unit_is_the_exact_inverse_of_decrypt_unit() {
|
||||
let key = [0x3Cu8; 16];
|
||||
// Content with no all-zero packets: every byte position exercised.
|
||||
let mut clear: Vec<u8> = (0..ALIGNED_UNIT_LEN)
|
||||
.map(|i| (i * 7 % 251 + 1) as u8)
|
||||
.collect();
|
||||
// The encrypted flag belongs to the caller and must be set BEFORE the
|
||||
// crypto, since bytes 0..16 are the key seed.
|
||||
clear[0] |= 0xC0;
|
||||
|
||||
let mut unit = clear.clone();
|
||||
assert!(
|
||||
encrypt_unit(&mut unit, &key),
|
||||
"a full-length unit must encrypt"
|
||||
);
|
||||
assert_ne!(
|
||||
unit[16..],
|
||||
clear[16..],
|
||||
"the payload must actually be enciphered"
|
||||
);
|
||||
assert_eq!(
|
||||
unit[..16],
|
||||
clear[..16],
|
||||
"the 16-byte seed stays plaintext on disc"
|
||||
);
|
||||
|
||||
decrypt_unit(&mut unit, &key);
|
||||
assert_eq!(unit, clear, "round trip must be byte-exact");
|
||||
}
|
||||
|
||||
/// The documented padding asymmetry actually holds: `decrypt_unit` restores
|
||||
/// all-zero-ON-DISC packets to zero, but an all-zero PLAINTEXT packet enciphers
|
||||
/// to non-zero bytes, so it is not mistaken for padding and still round-trips.
|
||||
/// This is the one place the two functions are deliberately not symmetric, so
|
||||
/// the claim is worth pinning rather than asserting in prose alone.
|
||||
/// A slice too short to encrypt must SAY so. The caller has already set the
|
||||
/// container's encrypted flag by the time it calls this (the contract requires
|
||||
/// flag-before-crypto, since the header is the key seed), so a silent no-op
|
||||
/// leaves a unit advertised as encrypted while still carrying plaintext.
|
||||
#[test]
|
||||
fn encrypt_unit_reports_a_slice_too_short_to_encrypt() {
|
||||
let key = [0x11u8; 16];
|
||||
let mut short = vec![0u8; ALIGNED_UNIT_LEN - 1];
|
||||
short[0] |= 0xC0; // the caller already flagged it encrypted
|
||||
let before = short.clone();
|
||||
assert!(
|
||||
!encrypt_unit(&mut short, &key),
|
||||
"a short slice must report false, not silently succeed"
|
||||
);
|
||||
assert_eq!(
|
||||
short, before,
|
||||
"a refused encrypt must leave the buffer untouched"
|
||||
);
|
||||
|
||||
// Exactly ALIGNED_UNIT_LEN is the boundary and must succeed.
|
||||
let mut exact = vec![0u8; ALIGNED_UNIT_LEN];
|
||||
exact[0] |= 0xC0;
|
||||
assert!(encrypt_unit(&mut exact, &key), "a full unit must encrypt");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn encrypt_unit_round_trips_all_zero_plaintext_packets() {
|
||||
let key = [0xA5u8; 16];
|
||||
let mut clear = vec![0u8; ALIGNED_UNIT_LEN];
|
||||
clear[0] |= 0xC0; // flag before crypto
|
||||
// Give packet 0 some content; leave every later packet entirely zero.
|
||||
for (i, b) in clear[16..192].iter_mut().enumerate() {
|
||||
*b = (i % 255 + 1) as u8;
|
||||
}
|
||||
|
||||
let mut unit = clear.clone();
|
||||
assert!(
|
||||
encrypt_unit(&mut unit, &key),
|
||||
"a full-length unit must encrypt"
|
||||
);
|
||||
// No later packet may encipher to all-zero, or decrypt would treat it as
|
||||
// source padding and the asymmetry would bite.
|
||||
for p in 1..ALIGNED_UNIT_LEN / BD_SOURCE_PACKET_BYTES {
|
||||
let off = p * BD_SOURCE_PACKET_BYTES;
|
||||
assert!(
|
||||
unit[off..off + BD_SOURCE_PACKET_BYTES]
|
||||
.iter()
|
||||
.any(|&b| b != 0),
|
||||
"packet {p} enciphered to all-zero, which decrypt reads as padding"
|
||||
);
|
||||
}
|
||||
|
||||
decrypt_unit(&mut unit, &key);
|
||||
assert_eq!(unit, clear, "zero-payload packets must round trip exactly");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_aes_ecb_roundtrip() {
|
||||
@@ -498,23 +686,11 @@ mod tests {
|
||||
let original = vec![0x42u8; 128]; // 8 blocks
|
||||
let mut data = original.clone();
|
||||
|
||||
// Encrypt with CBC manually (forward direction)
|
||||
fn aes_cbc_encrypt(key: &[u8; 16], data: &mut [u8]) {
|
||||
let cipher = Aes128::new(GenericArray::from_slice(key));
|
||||
let mut prev = super::AACS_IV;
|
||||
let num_blocks = data.len() / 16;
|
||||
for i in 0..num_blocks {
|
||||
let offset = i * 16;
|
||||
for j in 0..16 {
|
||||
data[offset + j] ^= prev[j];
|
||||
}
|
||||
let mut block = GenericArray::clone_from_slice(&data[offset..offset + 16]);
|
||||
cipher.encrypt_block(&mut block);
|
||||
data[offset..offset + 16].copy_from_slice(&block);
|
||||
prev.copy_from_slice(&data[offset..offset + 16]);
|
||||
}
|
||||
}
|
||||
|
||||
// Encrypt with the REAL production primitive. This test previously
|
||||
// defined a local `fn aes_cbc_encrypt` that SHADOWED it, so it round-
|
||||
// tripped a copy of the algorithm against itself and never exercised
|
||||
// `crypto::aes_cbc_encrypt` at all — a mutation to the shipped function
|
||||
// could not fail it.
|
||||
aes_cbc_encrypt(&key, &mut data);
|
||||
assert_ne!(data, original); // should be different after encrypt
|
||||
|
||||
@@ -549,7 +725,7 @@ mod tests {
|
||||
}
|
||||
|
||||
// CBC encrypt bytes 16..6143
|
||||
let cipher = Aes128::new(GenericArray::from_slice(&encrypt_key));
|
||||
let cipher = Aes128::new(&encrypt_key.into());
|
||||
let mut prev = AACS_IV;
|
||||
let num_blocks = (ALIGNED_UNIT_LEN - 16) / 16;
|
||||
for i in 0..num_blocks {
|
||||
@@ -557,7 +733,9 @@ mod tests {
|
||||
for j in 0..16 {
|
||||
plain[off + j] ^= prev[j];
|
||||
}
|
||||
let mut block = GenericArray::clone_from_slice(&plain[off..off + 16]);
|
||||
let mut chunk = [0u8; 16];
|
||||
chunk.copy_from_slice(&plain[off..off + 16]);
|
||||
let mut block: Array<u8, _> = chunk.into();
|
||||
cipher.encrypt_block(&mut block);
|
||||
plain[off..off + 16].copy_from_slice(&block);
|
||||
prev.copy_from_slice(&plain[off..off + 16]);
|
||||
@@ -598,29 +776,13 @@ mod tests {
|
||||
/// header) XOR header`, then CBC-encrypt bytes 16..6144 under the
|
||||
/// fixed AACS IV.
|
||||
fn aacs_encrypt_unit(unit: &mut [u8], unit_key: &[u8; 16]) {
|
||||
// Set the CPI bits (top 2 of byte 0) so the unit reads as encrypted under
|
||||
// `aacs_unit_encrypted` — done BEFORE key derivation so the plaintext
|
||||
// header the real decrypt recovers matches what we encrypt under.
|
||||
// Delegate to the module-scope `pub(crate)` helper (the single encrypt
|
||||
// implementation, shared with the mux `driver.rs` decrypt test).
|
||||
unit[0] |= 0xC0;
|
||||
let header: [u8; 16] = unit[..16].try_into().unwrap();
|
||||
let derived = aes_ecb_encrypt(unit_key, &header);
|
||||
let mut k = [0u8; 16];
|
||||
for i in 0..16 {
|
||||
k[i] = derived[i] ^ header[i];
|
||||
}
|
||||
let cipher = Aes128::new(GenericArray::from_slice(&k));
|
||||
let mut prev = AACS_IV;
|
||||
let num_blocks = (ALIGNED_UNIT_LEN - 16) / 16;
|
||||
for i in 0..num_blocks {
|
||||
let off = 16 + i * 16;
|
||||
for j in 0..16 {
|
||||
unit[off + j] ^= prev[j];
|
||||
}
|
||||
let mut block = GenericArray::clone_from_slice(&unit[off..off + 16]);
|
||||
cipher.encrypt_block(&mut block);
|
||||
unit[off..off + 16].copy_from_slice(&block);
|
||||
prev.copy_from_slice(&unit[off..off + 16]);
|
||||
}
|
||||
assert!(
|
||||
super::encrypt_unit(unit, unit_key),
|
||||
"a full-length unit must encrypt"
|
||||
);
|
||||
}
|
||||
|
||||
/// Build a clear aligned unit with TS sync bytes at offset 4 + k*192.
|
||||
@@ -715,7 +877,7 @@ mod tests {
|
||||
|
||||
// ── Defect-tolerant "did a key OPEN this unit?" verdict ─────────────────
|
||||
//
|
||||
// The Bourne-UHD bug: a commercial disc carries the odd authored-bad TS
|
||||
// The authored-bad-packet bug: a commercial disc carries the odd bad TS
|
||||
// packet (a pressing/encoding defect, or an AACS 2.1 forensic-variant frame)
|
||||
// — one non-conforming packet inside an otherwise perfectly-decrypted 6144
|
||||
// unit. The OLD strict per-packet acceptance rejected the WHOLE unit over
|
||||
@@ -1043,21 +1205,100 @@ mod tests {
|
||||
assert_eq!(aes_ecb_decrypt(&key, &expected), pt);
|
||||
}
|
||||
|
||||
// ── decrypt_bus: one key schedule per unit, not one per sector ─────────
|
||||
|
||||
/// MEASURED, not reasoned: `decrypt_bus` called `aes_cbc_decrypt` once per
|
||||
/// 2048-byte sector, and each call built its own AES-128 key schedule, so a
|
||||
/// 6144-byte aligned unit performed THREE key expansions under the same
|
||||
/// loop-invariant `read_data_key`. On a 90 GB UHD read on a stock (non-
|
||||
/// LibreDrive) drive — ~14.6 million aligned units — that is ~29 million
|
||||
/// redundant expansions on the per-unit decrypt hot path, for a key that is
|
||||
/// constant for the whole disc. The counter is incremented inside
|
||||
/// `crypto::new_cipher`, the single construction site.
|
||||
#[test]
|
||||
fn decrypt_bus_expands_the_read_data_key_once_per_unit() {
|
||||
use crate::aacs::crypto::KEY_EXPANSIONS;
|
||||
let mut unit = clear_unit();
|
||||
let rdk = [0x4Eu8; 16];
|
||||
KEY_EXPANSIONS.with(|c| c.set(0));
|
||||
decrypt_bus(&mut unit, &rdk);
|
||||
let n = KEY_EXPANSIONS.with(|c| c.get());
|
||||
assert_eq!(
|
||||
n, 1,
|
||||
"one aligned unit under one read_data_key must expand the schedule \
|
||||
exactly once, not once per 2048-byte sector"
|
||||
);
|
||||
}
|
||||
|
||||
/// The single-expansion refactor must be byte-identical: bus encryption
|
||||
/// ([C] §4.2) covers bytes 16..2048 of every 2048-byte sector, so a
|
||||
/// three-sector aligned unit round-trips through the forward direction
|
||||
/// sector by sector and `decrypt_bus` must recover it exactly.
|
||||
#[test]
|
||||
fn decrypt_bus_roundtrips_every_sector_region() {
|
||||
let rdk = [0x91u8; 16];
|
||||
let original = clear_unit();
|
||||
let mut unit = original.clone();
|
||||
// Forward direction, region by region — the inverse of decrypt_bus.
|
||||
for start in (0..ALIGNED_UNIT_LEN).step_by(SECTOR_BYTES) {
|
||||
crate::aacs::crypto::aes_cbc_encrypt(&rdk, &mut unit[start + 16..start + SECTOR_BYTES]);
|
||||
}
|
||||
assert_ne!(
|
||||
&unit[16..64],
|
||||
&original[16..64],
|
||||
"the forward direction must have changed the bytes"
|
||||
);
|
||||
decrypt_bus(&mut unit, &rdk);
|
||||
assert_eq!(
|
||||
unit.as_slice(),
|
||||
original.as_slice(),
|
||||
"decrypt_bus must invert the per-sector bus encryption exactly"
|
||||
);
|
||||
}
|
||||
|
||||
// ── CBC decrypt: first-block uses fixed AACS IV ────────────────────────
|
||||
|
||||
/// The published `iv0` bytes, INDEPENDENT of the production constant.
|
||||
///
|
||||
/// [C] §2.1.2 fixes one default CBC IV for every AACS AES-CBC operation.
|
||||
/// Both IV tests below used to compute their expected value from
|
||||
/// `crypto::AACS_IV` itself, so the constant was asserted against itself and
|
||||
/// NOTHING in the suite pinned its bytes: swapping `AACS_IV` for `[0u8; 16]`
|
||||
/// left both tests passing (one builds its ciphertext with the same value and
|
||||
/// the other cancels the change in a triple XOR) while every real AACS disc
|
||||
/// decrypted to noise — block 0 of every 6128-byte aligned unit and of every
|
||||
/// bus-encrypted sector XORed with the wrong IV. This literal is the
|
||||
/// independent witness the tests assert against.
|
||||
const IV0_PUBLISHED: [u8; 16] = [
|
||||
0x0B, 0xA0, 0xF8, 0xDD, 0xFE, 0xA6, 0x1F, 0xB3, 0xD8, 0xDF, 0x9F, 0x56, 0x6A, 0x05, 0x0F,
|
||||
0x78,
|
||||
];
|
||||
|
||||
/// Pins the fixed AACS CBC IV ([C] §2.1.2 `iv0`) against a literal, so a
|
||||
/// change to `crypto::AACS_IV` fails HERE rather than silently shipping.
|
||||
#[test]
|
||||
fn aacs_iv_matches_published_iv0() {
|
||||
assert_eq!(
|
||||
AACS_IV, IV0_PUBLISHED,
|
||||
"the fixed AACS CBC IV must be the published iv0"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn cbc_decrypt_first_block_xors_aacs_iv() {
|
||||
// CBC: P[0] = AES-D(K, C[0]) XOR IV, and the IV is the fixed AACS
|
||||
// constant (not zero). Encrypt a single block forward with IV, then
|
||||
// confirm aes_cbc_decrypt recovers it — proving the IV used on block
|
||||
// 0 is exactly AACS_IV. A mutation that swaps AACS_IV for [0u8;16]
|
||||
// makes the recovered block wrong.
|
||||
// constant (not zero). Encrypt a single block forward with the PUBLISHED
|
||||
// iv0 literal, then confirm aes_cbc_decrypt recovers it — proving the IV
|
||||
// the production code uses on block 0 is exactly that value. Building the
|
||||
// fixture from `IV0_PUBLISHED` rather than from `AACS_IV` is what makes
|
||||
// the claim real: a mutation that swaps AACS_IV for [0u8;16] now makes the
|
||||
// recovered block wrong.
|
||||
let key = [0x24u8; 16];
|
||||
let plain = [0x5Au8; 16];
|
||||
// Forward CBC for one block: C = AES-E(K, P XOR IV).
|
||||
let mut x = plain;
|
||||
for j in 0..16 {
|
||||
x[j] ^= AACS_IV[j];
|
||||
x[j] ^= IV0_PUBLISHED[j];
|
||||
}
|
||||
let ct = aes_ecb_encrypt(&key, &x);
|
||||
let mut buf = ct;
|
||||
@@ -1086,10 +1327,13 @@ mod tests {
|
||||
// * Blocks 1..=3 are independent of the IV — they MUST equal the NIST
|
||||
// plaintext byte-for-byte (P[i] = AES-D(K, C[i]) XOR C[i-1]). This
|
||||
// pins the real reverse-order CBC chaining against a published KAT.
|
||||
// * Block 0 = AES-D(K, C[0]) XOR AACS_IV = NIST_PT[0] XOR NIST_IV
|
||||
// XOR AACS_IV — the documented IV substitution. Asserting this exact
|
||||
// relation pins both the AES decrypt of C[0] AND that block 0 uses
|
||||
// AACS_IV (a swap to [0u8;16] or a chaining bug fails it).
|
||||
// * Block 0 = AES-D(K, C[0]) XOR iv0 = NIST_PT[0] XOR NIST_IV XOR iv0 —
|
||||
// the documented IV substitution. Asserting this exact relation pins
|
||||
// both the AES decrypt of C[0] AND that block 0 uses iv0. The expected
|
||||
// value is built from the `IV0_PUBLISHED` literal, NOT from
|
||||
// `crypto::AACS_IV`: computing it from the production constant made
|
||||
// the change cancel out of the triple XOR, so a swap to [0u8;16] still
|
||||
// passed. It now fails.
|
||||
let key = [
|
||||
0x2B, 0x7E, 0x15, 0x16, 0x28, 0xAE, 0xD2, 0xA6, 0xAB, 0xF7, 0x15, 0x88, 0x09, 0xCF,
|
||||
0x4F, 0x3C,
|
||||
@@ -1129,7 +1373,7 @@ mod tests {
|
||||
// Block 0: NIST_PT[0] XOR NIST_IV XOR AACS_IV (the fixed-IV substitution).
|
||||
let mut expected_block0 = [0u8; 16];
|
||||
for i in 0..16 {
|
||||
expected_block0[i] = nist_plaintext[i] ^ nist_iv[i] ^ AACS_IV[i];
|
||||
expected_block0[i] = nist_plaintext[i] ^ nist_iv[i] ^ IV0_PUBLISHED[i];
|
||||
}
|
||||
assert_eq!(
|
||||
&buf[0..16],
|
||||
@@ -1254,7 +1498,7 @@ mod tests {
|
||||
let plain = unit.clone();
|
||||
|
||||
// Forward: CBC-encrypt unit[s+16 .. s+2048] per sector under AACS IV.
|
||||
let cipher = Aes128::new(GenericArray::from_slice(&rdk));
|
||||
let cipher = Aes128::new(&rdk.into());
|
||||
for s in (0..ALIGNED_UNIT_LEN).step_by(SECTOR_BYTES) {
|
||||
let mut prev = AACS_IV;
|
||||
let body = s + 16;
|
||||
@@ -1265,7 +1509,9 @@ mod tests {
|
||||
for j in 0..16 {
|
||||
unit[off + j] ^= prev[j];
|
||||
}
|
||||
let mut blk = GenericArray::clone_from_slice(&unit[off..off + 16]);
|
||||
let mut chunk = [0u8; 16];
|
||||
chunk.copy_from_slice(&unit[off..off + 16]);
|
||||
let mut blk: Array<u8, _> = chunk.into();
|
||||
cipher.encrypt_block(&mut blk);
|
||||
unit[off..off + 16].copy_from_slice(&blk);
|
||||
prev.copy_from_slice(&unit[off..off + 16]);
|
||||
@@ -1284,7 +1530,7 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
// ── ts_sync_destroyed / ts_sync_count edge cases ───────────────────────
|
||||
// ── is_clean / ts_sync_count edge cases ────────────────────────────────
|
||||
|
||||
#[test]
|
||||
fn ts_sync_destroyed_false_for_sub_unit_length() {
|
||||
@@ -1322,6 +1568,95 @@ mod tests {
|
||||
assert_eq!(ts_sync_count(&unit), 1);
|
||||
}
|
||||
|
||||
// ── the encrypted-flag readers ────────────────────────────────────────
|
||||
|
||||
/// `aacs_unit_seed_encrypted` is the flag reader for a PARTIAL unit — the
|
||||
/// guard that stops a truncated encrypted fragment from being emitted as
|
||||
/// clear content. It reads ONLY the two Copy Permission Indicator bits
|
||||
/// ([BD] §3.10.2, byte 0 bits 6-7); the remaining six bits are
|
||||
/// `TP_extra_header` arrival-timestamp bits and carry no encryption
|
||||
/// meaning.
|
||||
///
|
||||
/// Both failure directions are damaging and silent: a reader that answers
|
||||
/// "encrypted" for a clear fragment discards good content, and one that
|
||||
/// answers "clear" for an encrypted fragment writes ciphertext into the
|
||||
/// output as if it were video.
|
||||
#[test]
|
||||
fn aacs_unit_seed_encrypted_reads_only_the_two_cpi_bits() {
|
||||
use crate::disc::ContentFormat::BdTs;
|
||||
|
||||
// CPI bits clear → NOT encrypted, whatever the ATS bits say.
|
||||
for ats in 0u8..=0x3F {
|
||||
assert!(
|
||||
!aacs_unit_seed_encrypted(&[ats], BdTs),
|
||||
"byte0={ats:#04x} has both CPI bits clear → not encrypted"
|
||||
);
|
||||
}
|
||||
|
||||
// Either CPI bit set → encrypted, whatever the ATS bits say.
|
||||
for &cpi in &[0x40u8, 0x80, 0xC0] {
|
||||
assert!(
|
||||
aacs_unit_seed_encrypted(&[cpi], BdTs),
|
||||
"byte0={cpi:#04x} has a CPI bit set → encrypted"
|
||||
);
|
||||
assert!(
|
||||
aacs_unit_seed_encrypted(&[cpi | 0x3F], BdTs),
|
||||
"ATS bits must not change the answer"
|
||||
);
|
||||
}
|
||||
|
||||
// Too short to hold the flag → false rather than a panic.
|
||||
assert!(!aacs_unit_seed_encrypted(&[], BdTs));
|
||||
}
|
||||
|
||||
/// The MpegPs (HD-DVD `.evo`) side reads `PES_scrambling_control` at its own
|
||||
/// fixed offset, and a fragment shorter than that offset must be reported
|
||||
/// clear rather than panic.
|
||||
#[test]
|
||||
fn aacs_unit_seed_encrypted_reads_the_ps_scramble_flag_or_says_clear() {
|
||||
use crate::disc::ContentFormat::MpegPs;
|
||||
|
||||
let mut frag = vec![0u8; PS_SCRAMBLE_OFF + 1];
|
||||
assert!(!aacs_unit_seed_encrypted(&frag, MpegPs), "flag byte zero");
|
||||
frag[PS_SCRAMBLE_OFF] = PS_SCRAMBLE_MASK;
|
||||
assert!(aacs_unit_seed_encrypted(&frag, MpegPs), "flag byte set");
|
||||
// Bits outside the mask are not the scrambling control.
|
||||
frag[PS_SCRAMBLE_OFF] = !PS_SCRAMBLE_MASK;
|
||||
assert!(!aacs_unit_seed_encrypted(&frag, MpegPs), "outside the mask");
|
||||
// A fragment that stops short of the flag byte is not classifiable.
|
||||
assert!(!aacs_unit_seed_encrypted(&frag[..PS_SCRAMBLE_OFF], MpegPs));
|
||||
}
|
||||
|
||||
/// `aacs_unit_encrypted` is the AUTHORITATIVE gate and requires a WHOLE
|
||||
/// 6144-byte aligned unit: on anything shorter the flag byte is not
|
||||
/// guaranteed to be the unit's, so it must answer `false` and leave the
|
||||
/// partial-unit case to `aacs_unit_seed_encrypted`. A reversed length guard
|
||||
/// would both classify fragments off arbitrary mid-stream bytes and, on an
|
||||
/// empty slice, index out of bounds.
|
||||
#[test]
|
||||
fn aacs_unit_encrypted_requires_a_whole_aligned_unit() {
|
||||
use crate::disc::ContentFormat::BdTs;
|
||||
|
||||
// A short buffer whose byte 0 has the CPI bits set is still NOT a unit.
|
||||
let mut short = vec![0u8; ALIGNED_UNIT_LEN - 1];
|
||||
short[0] = 0xC0;
|
||||
assert!(
|
||||
!aacs_unit_encrypted(&short, BdTs),
|
||||
"a sub-unit buffer must not be classified"
|
||||
);
|
||||
assert!(!aacs_unit_encrypted(&[], BdTs), "empty must not index");
|
||||
|
||||
// Exactly one aligned unit IS classified.
|
||||
let mut unit = vec![0u8; ALIGNED_UNIT_LEN];
|
||||
unit[0] = 0xC0;
|
||||
assert!(
|
||||
aacs_unit_encrypted(&unit, BdTs),
|
||||
"a full unit with CPI set is encrypted"
|
||||
);
|
||||
unit[0] = 0x00;
|
||||
assert!(!aacs_unit_encrypted(&unit, BdTs), "CPI clear is not");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ts_packet_total_for_various_lengths() {
|
||||
// total = len / 192 (BD-TS packet size). Pin a few lengths.
|
||||
|
||||
+175
-9
@@ -8,17 +8,44 @@
|
||||
//! content / keys / variant modules.
|
||||
|
||||
use aes::Aes128;
|
||||
use aes::cipher::{BlockDecrypt, BlockEncrypt, KeyInit, generic_array::GenericArray};
|
||||
use aes::cipher::{Array, BlockCipherDecrypt, BlockCipherEncrypt, KeyInit};
|
||||
|
||||
/// Fixed IV used by AACS for all AES-CBC operations. [C] §2.1.2 (default CBC IV, `iv0`).
|
||||
pub(crate) const AACS_IV: [u8; 16] = [
|
||||
0x0B, 0xA0, 0xF8, 0xDD, 0xFE, 0xA6, 0x1F, 0xB3, 0xD8, 0xDF, 0x9F, 0x56, 0x6A, 0x05, 0x0F, 0x78,
|
||||
];
|
||||
|
||||
// Per-thread count of AES-128 key schedules built through `new_cipher`.
|
||||
// Test-only instrumentation: an AES-128 key expansion is 10 round-key
|
||||
// derivations, and the CBC helpers here run on the per-aligned-unit decrypt hot
|
||||
// path of a whole disc read, so "how many times was the schedule built for one
|
||||
// loop-invariant key" is a property worth asserting rather than reasoning about.
|
||||
// THREAD-LOCAL, not a global atomic: `cargo test` runs tests concurrently, so a
|
||||
// shared counter would see every other test's expansions. See
|
||||
// `content::tests::decrypt_bus_expands_the_read_data_key_once_per_unit`.
|
||||
#[cfg(test)]
|
||||
thread_local! {
|
||||
pub(crate) static KEY_EXPANSIONS: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
|
||||
}
|
||||
|
||||
/// Build an AES-128 key schedule for a caller that will drive
|
||||
/// [`cbc_decrypt_blocks`] over several regions under one key.
|
||||
pub(crate) fn new_cipher_for(key: &[u8; 16]) -> Aes128 {
|
||||
new_cipher(key)
|
||||
}
|
||||
|
||||
/// Build an AES-128 key schedule. The single construction site for the CBC
|
||||
/// helpers, so [`KEY_EXPANSIONS`] can count them under test.
|
||||
fn new_cipher(key: &[u8; 16]) -> Aes128 {
|
||||
#[cfg(test)]
|
||||
KEY_EXPANSIONS.with(|c| c.set(c.get() + 1));
|
||||
Aes128::new(&(*key).into())
|
||||
}
|
||||
|
||||
/// AES-128-ECB encrypt a single 16-byte block. [C] §2.1.1 (`AES-128E`).
|
||||
pub(crate) fn aes_ecb_encrypt(key: &[u8; 16], data: &[u8; 16]) -> [u8; 16] {
|
||||
let cipher = Aes128::new(GenericArray::from_slice(key));
|
||||
let mut block = GenericArray::clone_from_slice(data);
|
||||
let cipher = Aes128::new(&(*key).into());
|
||||
let mut block: Array<u8, _> = (*data).into();
|
||||
cipher.encrypt_block(&mut block);
|
||||
let mut out = [0u8; 16];
|
||||
out.copy_from_slice(&block);
|
||||
@@ -27,25 +54,81 @@ pub(crate) fn aes_ecb_encrypt(key: &[u8; 16], data: &[u8; 16]) -> [u8; 16] {
|
||||
|
||||
/// AES-128-ECB decrypt a single 16-byte block. [C] §2.1.1 (`AES-128D`).
|
||||
pub(crate) fn aes_ecb_decrypt(key: &[u8; 16], data: &[u8; 16]) -> [u8; 16] {
|
||||
let cipher = Aes128::new(GenericArray::from_slice(key));
|
||||
let mut block = GenericArray::clone_from_slice(data);
|
||||
let cipher = Aes128::new(&(*key).into());
|
||||
let mut block: Array<u8, _> = (*data).into();
|
||||
cipher.decrypt_block(&mut block);
|
||||
let mut out = [0u8; 16];
|
||||
out.copy_from_slice(&block);
|
||||
out
|
||||
}
|
||||
|
||||
/// AES-128-CBC decrypt in-place with the fixed AACS IV. [C] §2.1.2 (`AES-128CBCD`).
|
||||
/// AES-128-CBC ENCRYPT in place under the fixed [`AACS_IV`] — the forward
|
||||
/// direction of [`aes_cbc_decrypt`], and its exact inverse. [C] §2.1.2
|
||||
/// (`AES-128CBCE`).
|
||||
///
|
||||
/// Precondition: `data.len()` is a multiple of 16; the assert
|
||||
/// documents/enforces that contract.
|
||||
///
|
||||
/// Constructs the cipher ONCE for the whole slice. Driving this from the
|
||||
/// single-block [`aes_ecb_encrypt`] instead rebuilds the AES key schedule per
|
||||
/// 16-byte block, which for a 6144-byte aligned unit is 383 redundant key
|
||||
/// expansions.
|
||||
pub(crate) fn aes_cbc_encrypt(key: &[u8; 16], data: &mut [u8]) {
|
||||
debug_assert!(
|
||||
data.len().is_multiple_of(16),
|
||||
"aes_cbc_encrypt requires a block-aligned slice"
|
||||
);
|
||||
let cipher = new_cipher(key);
|
||||
let num_blocks = data.len() / 16;
|
||||
let mut prev = AACS_IV;
|
||||
// Forward order: each block is XORed with the PRECEDING ciphertext block.
|
||||
for i in 0..num_blocks {
|
||||
let offset = i * 16;
|
||||
let mut block = [0u8; 16];
|
||||
for j in 0..16 {
|
||||
block[j] = data[offset + j] ^ prev[j];
|
||||
}
|
||||
let mut ga: Array<u8, _> = block.into();
|
||||
cipher.encrypt_block(&mut ga);
|
||||
data[offset..offset + 16].copy_from_slice(&ga);
|
||||
prev.copy_from_slice(&ga);
|
||||
}
|
||||
}
|
||||
|
||||
/// AES-128-CBC DECRYPT in-place with the fixed AACS IV. [C] §2.1.2
|
||||
/// (`AES-128CBCD`).
|
||||
///
|
||||
/// Precondition: `data.len()` is a multiple of 16. Any trailing partial
|
||||
/// block is silently ignored; all callers pass aligned regions (6128 and
|
||||
/// 2032 bytes), and the assert documents/enforces that contract.
|
||||
///
|
||||
/// (This doc block was orphaned onto `aes_cbc_encrypt` above when that function
|
||||
/// was inserted directly after it with no separating blank line, so rustdoc
|
||||
/// rendered the crate's only forward-direction AACS primitive as "decrypt" and
|
||||
/// cited the spec's DECRYPT clause for it, while this function had no doc at
|
||||
/// all. `encrypt_unit_is_the_exact_inverse_of_decrypt_unit` in `content.rs` pins
|
||||
/// the directions behaviourally so a maintainer 'fixing' the contradiction by
|
||||
/// swapping the two bodies fails the suite instead of shipping a second
|
||||
/// decryptor behind an already-set encrypted flag.)
|
||||
pub(crate) fn aes_cbc_decrypt(key: &[u8; 16], data: &mut [u8]) {
|
||||
debug_assert!(
|
||||
data.len() % 16 == 0,
|
||||
data.len().is_multiple_of(16),
|
||||
"aes_cbc_decrypt requires a block-aligned slice"
|
||||
);
|
||||
let cipher = Aes128::new(GenericArray::from_slice(key));
|
||||
cbc_decrypt_blocks(&new_cipher(key), data);
|
||||
}
|
||||
|
||||
/// AES-128-CBC decrypt in place under the fixed [`AACS_IV`] with an ALREADY
|
||||
/// EXPANDED key schedule.
|
||||
///
|
||||
/// Split out of [`aes_cbc_decrypt`] so a caller that decrypts several regions
|
||||
/// under one loop-invariant key expands the schedule once. `decrypt_bus`
|
||||
/// ([`super::content::decrypt_bus`]) is that caller: bus encryption
|
||||
/// ([C] §4.2 / the AACS 2.0 Read Data Key) covers bytes 16..2048 of EVERY
|
||||
/// 2048-byte sector, so a 6144-byte aligned unit is three regions under one
|
||||
/// `read_data_key` — three key schedules where one suffices, on the per-unit
|
||||
/// decrypt hot path of a whole 90 GB read.
|
||||
pub(crate) fn cbc_decrypt_blocks(cipher: &Aes128, data: &mut [u8]) {
|
||||
let num_blocks = data.len() / 16;
|
||||
// Process blocks in reverse to avoid clobbering ciphertext needed for XOR
|
||||
for i in (0..num_blocks).rev() {
|
||||
@@ -57,7 +140,9 @@ pub(crate) fn aes_cbc_decrypt(key: &[u8; 16], data: &mut [u8]) {
|
||||
p.copy_from_slice(&data[(i - 1) * 16..i * 16]);
|
||||
p
|
||||
};
|
||||
let mut block = GenericArray::clone_from_slice(&data[offset..offset + 16]);
|
||||
let mut chunk = [0u8; 16];
|
||||
chunk.copy_from_slice(&data[offset..offset + 16]);
|
||||
let mut block: Array<u8, _> = chunk.into();
|
||||
cipher.decrypt_block(&mut block);
|
||||
for j in 0..16 {
|
||||
data[offset + j] = block[j] ^ prev[j];
|
||||
@@ -100,3 +185,84 @@ pub(crate) fn aesg3(key: &[u8; 16], inc: u8) -> [u8; 16] {
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The AACS-G3 seed `s0`, transcribed independently from [C] §3.2.2 rather
|
||||
/// than read from [`AESG3_SEED`] — a test that sourced the seed from the
|
||||
/// production constant would assert that constant against itself and would
|
||||
/// still pass if it were edited.
|
||||
const S0: [u8; 16] = [
|
||||
0x7B, 0x10, 0x3C, 0x5D, 0xCB, 0x08, 0xC4, 0xE5, 0x1A, 0x27, 0xB0, 0x17, 0x99, 0x05, 0x3B,
|
||||
0xD9,
|
||||
];
|
||||
|
||||
/// An arbitrary non-degenerate key. Nothing about it is secret or special;
|
||||
/// the AES-G3 relation holds for every key, and a constant-returning body
|
||||
/// cannot satisfy it for any.
|
||||
const K: [u8; 16] = [
|
||||
0x0F, 0x1E, 0x2D, 0x3C, 0x4B, 0x5A, 0x69, 0x78, 0x87, 0x96, 0xA5, 0xB4, 0xC3, 0xD2, 0xE1,
|
||||
0xF0,
|
||||
];
|
||||
|
||||
/// `aesg3` is the node function of the AACS subset-difference tree: every
|
||||
/// Processing Key the DK walk produces (`aesg3(node_key, 1)`) and every
|
||||
/// descent step (`aesg3(., 0)` / `aesg3(., 2)`) is one call. A body that
|
||||
/// returned a fixed block would make every device key in the crate derive
|
||||
/// the SAME Processing Key, and a `^` that became `|` or `&` would derive a
|
||||
/// wrong-but-plausible one — in both cases the MKB walk simply stops
|
||||
/// finding Media Keys, with no error to say why.
|
||||
///
|
||||
/// Pinned through the spec relation rather than a re-implementation:
|
||||
/// [C] §3.2.2 defines `AES-G3` as `AES-128D(k, s) XOR s` for
|
||||
/// `s = s0 + inc` (added into the last seed byte), so applying the
|
||||
/// FORWARD primitive [`aes_ecb_encrypt`] — a different function from the
|
||||
/// one under test — to `aesg3(k, inc) XOR s` must reproduce `s` exactly.
|
||||
#[test]
|
||||
fn aesg3_inverts_to_the_spec_seed_under_aes_encrypt() {
|
||||
for inc in 0u8..=2 {
|
||||
let mut seed = S0;
|
||||
seed[15] = seed[15].wrapping_add(inc);
|
||||
|
||||
let out = aesg3(&K, inc);
|
||||
|
||||
// out == AES-128D(K, seed) XOR seed, so out XOR seed is the raw
|
||||
// decryption and re-encrypting it must land back on the seed.
|
||||
let mut pre = [0u8; 16];
|
||||
for i in 0..16 {
|
||||
pre[i] = out[i] ^ seed[i];
|
||||
}
|
||||
assert_eq!(
|
||||
aes_ecb_encrypt(&K, &pre),
|
||||
seed,
|
||||
"AES-G3 inc={inc} must satisfy out = AES-128D(k, s0+inc) XOR (s0+inc)"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The Triple Generator's three outputs ([C] §3.2.2: left = inc 0, the
|
||||
/// Processing Key = inc 1, right = inc 2) are the two child node keys and
|
||||
/// the Processing Key of ONE tree node. They must be three different keys —
|
||||
/// if `inc` were ignored, a descent would revisit its own parent and the
|
||||
/// walk would derive the same key at every level of the tree.
|
||||
#[test]
|
||||
fn aesg3_yields_three_distinct_subkeys_for_the_three_increments() {
|
||||
let left = aesg3(&K, 0);
|
||||
let pk = aesg3(&K, 1);
|
||||
let right = aesg3(&K, 2);
|
||||
assert_ne!(left, pk, "left child and Processing Key must differ");
|
||||
assert_ne!(pk, right, "Processing Key and right child must differ");
|
||||
assert_ne!(left, right, "left and right children must differ");
|
||||
}
|
||||
|
||||
/// Distinct parent keys must yield distinct subkeys — the tree would
|
||||
/// collapse otherwise.
|
||||
#[test]
|
||||
fn aesg3_separates_distinct_parent_keys() {
|
||||
let mut other = K;
|
||||
other[0] ^= 0x01;
|
||||
assert_ne!(aesg3(&K, 1), aesg3(&other, 1));
|
||||
}
|
||||
}
|
||||
|
||||
+1082
-1
File diff suppressed because it is too large
Load Diff
@@ -63,11 +63,17 @@ pub fn unit_disposition(
|
||||
None => UnitDisposition::Default,
|
||||
// In a forensic segment → decide by whether it is our index.
|
||||
Some(seg) => {
|
||||
let seg_index = seg.index as u8;
|
||||
// `seg.index` is an untrusted u16 from IndividualSegment.tbl; a real
|
||||
// forensic index is 1..=32. Compare in u16 space so a corrupt/crafted
|
||||
// index above 255 can't truncate into a valid u8 and alias our index.
|
||||
// The disposition carries a u8 for diagnostics (saturated — an
|
||||
// out-of-range index is never ours anyway).
|
||||
let seg_index = seg.index;
|
||||
let diag = seg_index.min(u8::MAX as u16) as u8;
|
||||
match disc_index {
|
||||
Some(v) if v == seg_index => UnitDisposition::Index(v),
|
||||
Some(_) => UnitDisposition::DropForeignIndex(seg_index),
|
||||
None => UnitDisposition::ForensicNoKey(seg_index),
|
||||
Some(v) if u16::from(v) == seg_index => UnitDisposition::Index(v),
|
||||
Some(_) => UnitDisposition::DropForeignIndex(diag),
|
||||
None => UnitDisposition::ForensicNoKey(diag),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -161,6 +167,20 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn out_of_range_index_does_not_truncate_into_ours() {
|
||||
// A crafted/corrupt segment index of 288 (0x0120) truncates to 32 in a
|
||||
// u8. With our disc index resolved as 32, the old `seg.index as u8`
|
||||
// compare would alias it to OUR index and decrypt with the wrong key.
|
||||
// The u16 compare must instead classify it as foreign.
|
||||
let segs = tbl(&[(288, 100, 200)]);
|
||||
let off = 120u64 * SOURCE_PACKET_LEN;
|
||||
assert_eq!(
|
||||
unit_disposition(off, &segs, Some(32)),
|
||||
UnitDisposition::DropForeignIndex(255)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn straddling_unit_still_classified_as_its_segment() {
|
||||
// A unit whose 32-packet span only tails into the segment still routes
|
||||
@@ -169,7 +189,7 @@ mod tests {
|
||||
let unit_packets = (ALIGNED_UNIT_LEN as u64 / SOURCE_PACKET_LEN) as u32; // 32
|
||||
// Start so the unit covers [80, 80+31] = [80, 111]: overlaps at 100.
|
||||
let off = 80u64 * SOURCE_PACKET_LEN;
|
||||
assert!(80 + unit_packets - 1 >= 100, "sanity: unit tails into seg");
|
||||
assert!(80 + unit_packets > 100, "sanity: unit tails into seg");
|
||||
assert_eq!(
|
||||
unit_disposition(off, &segs, Some(5)),
|
||||
UnitDisposition::Index(5)
|
||||
|
||||
+474
-50
@@ -5,7 +5,6 @@
|
||||
use super::mkb::*;
|
||||
|
||||
/// Parsed Unit_Key_RO.inf file.
|
||||
#[derive(Debug)]
|
||||
pub struct UnitKeyFile {
|
||||
/// Disc hash (SHA1 of the entire file) — used as KEYDB lookup key
|
||||
pub disc_hash: [u8; 20],
|
||||
@@ -23,6 +22,29 @@ pub struct UnitKeyFile {
|
||||
pub title_cps_unit: Vec<u16>,
|
||||
}
|
||||
|
||||
/// Redacting `Debug`, per the policy `aacs::types` documents: this struct holds
|
||||
/// the disc's ENCRYPTED CPS unit keys — exactly the material a keydb entry stores
|
||||
/// — plus the disc hash they are looked up by. A derived `Debug` printed every key
|
||||
/// byte verbatim, so any `{:?}` (a downstream crate, an `assert_eq!` failure
|
||||
/// message, a future `tracing::debug!` in this module) leaked them. Only
|
||||
/// non-secret shape is printed. Guarded by `unit_key_file_debug_is_redacted`.
|
||||
impl std::fmt::Debug for UnitKeyFile {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("UnitKeyFile")
|
||||
// The disc hash is the public keydb lookup key, printed as hex the
|
||||
// same way `DiscEntry` prints its own — never as raw bytes.
|
||||
.field("disc_hash", &disc_hash_hex(&self.disc_hash))
|
||||
.field("app_type", &self.app_type)
|
||||
.field("num_bdmv_dir", &self.num_bdmv_dir)
|
||||
.field("use_skb_mkb", &self.use_skb_mkb)
|
||||
.field("version", &self.version)
|
||||
.field("encrypted_keys", &"<redacted>")
|
||||
.field("encrypted_keys_len", &self.encrypted_keys.len())
|
||||
.field("title_cps_unit", &self.title_cps_unit)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
/// Compute disc hash (SHA1 of Unit_Key_RO.inf content).
|
||||
pub fn disc_hash(data: &[u8]) -> [u8; 20] {
|
||||
use sha1::{Digest, Sha1};
|
||||
@@ -161,33 +183,52 @@ pub fn parse_unit_key_ro(data: &[u8], version: AacsVersion) -> Option<UnitKeyFil
|
||||
})
|
||||
}
|
||||
|
||||
/// HD DVD Video Title Key File (`VTKF000.AACS`) magic — "DVD HD Video TKF".
|
||||
/// HD DVD Video Title Key File (`VTKF%%%.AACS`) magic — "DVD_HD_V_TKF".
|
||||
pub const VTKF_MAGIC: &[u8; 12] = b"DVD_HD_V_TKF";
|
||||
/// Fixed header length before the first title-key entry.
|
||||
/// Fixed header length before the first Title Key Entry (AACS HD DVD Book,
|
||||
/// Table 3-8).
|
||||
const VTKF_HEADER_LEN: usize = 0x80;
|
||||
/// Each title-key entry: BE32 flag + 16-byte encrypted key + 12-byte 0xFF pad.
|
||||
const VTKF_ENTRY_LEN: usize = 0x20;
|
||||
/// Title Key Entry stride (Table 3-8): 1-byte `BIFO` + 3 reserved + 16-byte
|
||||
/// encrypted title key + 16-byte binding MAC = 36 bytes.
|
||||
const VTKF_ENTRY_LEN: usize = 0x24;
|
||||
/// Byte offset of the encrypted title key within an entry (after `BIFO` + 3
|
||||
/// reserved).
|
||||
const VTKF_KEY_OFF: usize = 4;
|
||||
/// Number of Title Key Entry slots in a VTKF (Table 3-8): a fixed 64.
|
||||
const VTKF_MAX_ENTRIES: usize = 64;
|
||||
/// `BIFO` bit 7 (`AV_FLG`): set = this slot carries an available title key.
|
||||
const VTKF_AV_FLG: u8 = 0x80;
|
||||
|
||||
/// Parse an HD DVD `VTKF000.AACS` into the SAME [`UnitKeyFile`] a BD/UHD
|
||||
/// Parse an HD DVD `VTKF%%%.AACS` into the SAME [`UnitKeyFile`] a BD/UHD
|
||||
/// `Unit_Key_RO.inf` yields — so the shared AACS crypto (`derive_unit_keys` →
|
||||
/// `decrypt_unit_key(vuk, …)`) unwraps HD DVD title keys with no change. Only
|
||||
/// the on-disc CONTAINER differs between BD and HD DVD; the title-key unwrap is
|
||||
/// the identical AES-128 VUK step (`Kt = AES-128D(Kvu, Kte)`).
|
||||
///
|
||||
/// Layout (grounded in real discs — Shaun of the Dead, Anchorman, Harry Potter):
|
||||
/// Layout — AACS "HD DVD and DVD Pre-recorded Book" Table 3-8, a fixed
|
||||
/// 2480-byte file, verified byte-exact against real discs (Freedom `VTKF090`,
|
||||
/// Dukes of Hazzard `VTKF000`):
|
||||
/// ```text
|
||||
/// [0x00..0x0C] magic "DVD_HD_V_TKF"
|
||||
/// [0x0C..0x10] BE32 total file length
|
||||
/// [0x10..0x1C] associated playlist name ("VPLST000.XPL")
|
||||
/// [0x1C..0x80] reserved (zero)
|
||||
/// [0x80..] 32-byte entries: BE32 flag | 16-byte ENCRYPTED title key | 12-byte 0xFF pad
|
||||
/// flag bit 31 (0x8000_0000) set = present; a cleared flag ends the table
|
||||
/// [tail] 16-byte signature/MAC (never a key — the cleared-flag stop guards it)
|
||||
/// [0x0C..0x10] BE32 HD_VTKF_SIZE (2480)
|
||||
/// [0x10..0x1C] associated playlist name ("VPLST%%%.XPL")
|
||||
/// [0x1C..0x80] reserved
|
||||
/// [0x80..] 64 entries × 36 bytes:
|
||||
/// BIFO (1) | reserved (3) | ENCRYPTED title key (16) | binding MAC (16)
|
||||
/// BIFO bit 7 (AV_FLG) set = this slot holds a title key
|
||||
/// (pre-recorded discs fill the binding MAC with 0xFF)
|
||||
/// [0x9A0..2480] 16-byte TKF MAC (CMAC keyed by Kvu — NOT a key)
|
||||
/// ```
|
||||
/// Entries number 1..=N as CPS units, matching `Unit_Key_RO`'s 1-based CPS
|
||||
/// numbering, so a title's CPS unit indexes this list identically. The
|
||||
/// title→CPS mapping itself is playlist-driven (`VPLST000.XPL`) and owned by the
|
||||
/// HD DVD enumerator, so `title_cps_unit` is left empty here.
|
||||
/// The slot index (1-based) is the CPS unit number, so an absent slot is
|
||||
/// SKIPPED (not a terminator) — collapsing gaps would renumber later keys and
|
||||
/// hand the wrong title key to CPS unit N+1. The title→CPS mapping is
|
||||
/// playlist-driven (`VPLST%%%.XPL`) and owned by the HD DVD enumerator, so
|
||||
/// `title_cps_unit` is left empty here.
|
||||
///
|
||||
/// The prior parser used a 32-byte stride (a 12-byte pad instead of the 16-byte
|
||||
/// binding MAC). That reads entry #1 correctly but drifts +4 bytes per entry
|
||||
/// after it, so it only decrypted single-CPS-unit discs; every multi-key VTKF
|
||||
/// (Freedom, Harry Potter) yielded garbage keys for CPS unit ≥2.
|
||||
pub fn parse_vtkf(data: &[u8]) -> Option<UnitKeyFile> {
|
||||
if data.len() < VTKF_HEADER_LEN || &data[..12] != VTKF_MAGIC {
|
||||
return None;
|
||||
@@ -198,20 +239,19 @@ pub fn parse_vtkf(data: &[u8]) -> Option<UnitKeyFile> {
|
||||
let hash = disc_hash(data);
|
||||
|
||||
let mut encrypted_keys = Vec::new();
|
||||
let mut pos = VTKF_HEADER_LEN;
|
||||
let mut cps: u32 = 1;
|
||||
while pos + VTKF_ENTRY_LEN <= data.len() {
|
||||
let flag = u32::from_be_bytes([data[pos], data[pos + 1], data[pos + 2], data[pos + 3]]);
|
||||
// A cleared present-bit terminates the key table. The file's trailing
|
||||
// 16-byte signature then follows and must NOT be read as a key.
|
||||
if flag & 0x8000_0000 == 0 {
|
||||
for n in 0..VTKF_MAX_ENTRIES {
|
||||
let pos = VTKF_HEADER_LEN + n * VTKF_ENTRY_LEN;
|
||||
if pos + VTKF_ENTRY_LEN > data.len() {
|
||||
break;
|
||||
}
|
||||
// AV_FLG clear = empty slot: skip it, but keep the slot index as the CPS
|
||||
// number (do NOT break — a gap must not renumber the keys that follow).
|
||||
if data[pos] & VTKF_AV_FLG == 0 {
|
||||
continue;
|
||||
}
|
||||
let mut key = [0u8; 16];
|
||||
key.copy_from_slice(&data[pos + 4..pos + 20]);
|
||||
encrypted_keys.push((cps, key));
|
||||
cps += 1;
|
||||
pos += VTKF_ENTRY_LEN;
|
||||
key.copy_from_slice(&data[pos + VTKF_KEY_OFF..pos + VTKF_KEY_OFF + 16]);
|
||||
encrypted_keys.push((n as u32 + 1, key));
|
||||
}
|
||||
if encrypted_keys.is_empty() {
|
||||
return None;
|
||||
@@ -369,42 +409,51 @@ pub fn parse_content_cert(data: &[u8]) -> Option<ContentCert> {
|
||||
mod vtkf_tests {
|
||||
use super::*;
|
||||
|
||||
/// Build a synthetic `VTKF000.AACS` matching the real on-disc layout
|
||||
/// (Shaun of the Dead / Anchorman): magic, BE32 size, playlist name,
|
||||
/// reserved to 0x80, then 32-byte present-flagged entries, a cleared-flag
|
||||
/// terminator, and a 16-byte trailer.
|
||||
/// Build a synthetic `VTKF%%%.AACS` matching the real on-disc layout (AACS
|
||||
/// HD DVD Book Table 3-8, verified against Freedom `VTKF090` and Dukes
|
||||
/// `VTKF000`): magic, BE32 size, playlist name, reserved to 0x80, then 64
|
||||
/// entry slots of 36 bytes (the first `keys.len()` present with `AV_FLG`
|
||||
/// set, the rest empty), a reserved gap, and the 16-byte trailing TKF MAC.
|
||||
fn synth_vtkf(keys: &[[u8; 16]]) -> Vec<u8> {
|
||||
const FILE_LEN: usize = 2480;
|
||||
let mut v = Vec::new();
|
||||
v.extend_from_slice(VTKF_MAGIC); // 0x00
|
||||
v.extend_from_slice(&0u32.to_be_bytes()); // 0x0C size (patched below)
|
||||
v.extend_from_slice(b"VPLST000.XPL"); // 0x10
|
||||
v.resize(0x80, 0); // reserve to first entry
|
||||
for k in keys {
|
||||
v.extend_from_slice(&0x8000_0000u32.to_be_bytes()); // present flag
|
||||
v.extend_from_slice(k); // 16-byte encrypted title key
|
||||
v.extend_from_slice(&[0xFFu8; 12]); // 0xFF pad → 32-byte entry
|
||||
v.extend_from_slice(&(FILE_LEN as u32).to_be_bytes()); // 0x0C HD_VTKF_SIZE
|
||||
v.extend_from_slice(b"VPLST000.XPL"); // 0x10 playlist name
|
||||
v.resize(VTKF_HEADER_LEN, 0); // reserve to first entry (0x80)
|
||||
for n in 0..VTKF_MAX_ENTRIES {
|
||||
if let Some(k) = keys.get(n) {
|
||||
v.push(VTKF_AV_FLG); // BIFO: AV_FLG set (present)
|
||||
v.extend_from_slice(&[0, 0, 0]); // reserved
|
||||
v.extend_from_slice(k); // 16-byte encrypted title key
|
||||
v.extend_from_slice(&[0xFFu8; 16]); // binding MAC (0xFF, pre-recorded)
|
||||
} else {
|
||||
v.extend_from_slice(&[0u8; VTKF_ENTRY_LEN]); // empty slot (AV_FLG clear)
|
||||
}
|
||||
}
|
||||
// Cleared-flag terminator entry (must NOT be read as a key).
|
||||
v.extend_from_slice(&[0u8; VTKF_ENTRY_LEN]);
|
||||
// 16-byte trailing signature (must NOT be read as a key).
|
||||
v.extend_from_slice(&[0xABu8; 16]);
|
||||
let len = v.len() as u32;
|
||||
v[0x0C..0x10].copy_from_slice(&len.to_be_bytes());
|
||||
v.resize(FILE_LEN - 16, 0); // reserved gap before the trailer
|
||||
v.extend_from_slice(&[0xABu8; 16]); // TKF MAC (must NOT be read as a key)
|
||||
v
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vtkf_extracts_present_entries_and_stops_at_terminator() {
|
||||
fn parse_vtkf_reads_present_entries_skips_empty_ignores_mac() {
|
||||
let k1 = [0x11u8; 16];
|
||||
let k2 = [0x22u8; 16];
|
||||
let k3 = [0x33u8; 16];
|
||||
let data = synth_vtkf(&[k1, k2, k3]);
|
||||
|
||||
let ukf = parse_vtkf(&data).expect("valid VTKF must parse");
|
||||
// Exactly the three present entries — the cleared-flag terminator and
|
||||
// the 16-byte trailer are NOT mistaken for keys.
|
||||
assert_eq!(ukf.encrypted_keys.len(), 3, "must stop at the cleared flag");
|
||||
assert_eq!(ukf.encrypted_keys[0], (1, k1), "CPS units number 1..=N");
|
||||
// Exactly the three present entries — the empty slots and the trailing
|
||||
// 16-byte TKF MAC are NOT mistaken for keys. Critically, k2/k3 are read
|
||||
// at the 36-byte stride (offsets 0xA4, 0xC8); the old 32-byte stride
|
||||
// misread them from inside the previous entry's binding MAC.
|
||||
assert_eq!(ukf.encrypted_keys.len(), 3);
|
||||
assert_eq!(
|
||||
ukf.encrypted_keys[0],
|
||||
(1, k1),
|
||||
"CPS units = 1-based slot index"
|
||||
);
|
||||
assert_eq!(ukf.encrypted_keys[1], (2, k2));
|
||||
assert_eq!(ukf.encrypted_keys[2], (3, k3));
|
||||
assert_eq!(ukf.version, AacsVersion::V10, "HD DVD is AACS 1.0");
|
||||
@@ -412,6 +461,22 @@ mod vtkf_tests {
|
||||
assert_eq!(ukf.disc_hash, disc_hash(&data));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vtkf_reads_a_full_64_entry_file() {
|
||||
// Real discs (Freedom, Dukes) carry all 64 slots present. Every key must
|
||||
// come back, none dropped and none drifted — the regression the 32-byte
|
||||
// stride failed.
|
||||
let keys: Vec<[u8; 16]> = (0..VTKF_MAX_ENTRIES).map(|n| [n as u8; 16]).collect();
|
||||
let ukf = parse_vtkf(&synth_vtkf(&keys)).expect("64-entry VTKF");
|
||||
assert_eq!(ukf.encrypted_keys.len(), 64);
|
||||
assert_eq!(
|
||||
ukf.encrypted_keys[63],
|
||||
(64, [63u8; 16]),
|
||||
"entry 64 at 0x{:x}",
|
||||
VTKF_HEADER_LEN + 63 * VTKF_ENTRY_LEN
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn parse_vtkf_rejects_non_magic() {
|
||||
let mut data = synth_vtkf(&[[0x11u8; 16]]);
|
||||
@@ -454,4 +519,363 @@ mod vtkf_tests {
|
||||
// Same as applying the shared unwrap directly to the stored enc key.
|
||||
assert_eq!(derived, super::super::derive::decrypt_unit_key(&vuk, &enc));
|
||||
}
|
||||
|
||||
/// `UnitKeyFile` holds the disc's ENCRYPTED CPS unit keys. A derived `Debug`
|
||||
/// printed every byte; the hand-written impl must not. Sentinel key byte
|
||||
/// 0xD5 = decimal 213 (a derived `Debug` renders `[u8; 16]` in decimal), the
|
||||
/// same probe `aacs::types::redaction_tests` uses. Mutation guard: putting
|
||||
/// `#[derive(Debug)]` back fails this.
|
||||
#[test]
|
||||
fn unit_key_file_debug_is_redacted() {
|
||||
let f = UnitKeyFile {
|
||||
disc_hash: [0xD5; 20],
|
||||
app_type: 1,
|
||||
num_bdmv_dir: 1,
|
||||
use_skb_mkb: false,
|
||||
version: AacsVersion::V20,
|
||||
encrypted_keys: vec![(0, [0xD5; 16]), (1, [0xD5; 16])],
|
||||
title_cps_unit: vec![0, 1],
|
||||
};
|
||||
let dbg = format!("{f:?}");
|
||||
assert!(
|
||||
!dbg.contains("213"),
|
||||
"UnitKeyFile Debug leaked key bytes (decimal 213): {dbg}"
|
||||
);
|
||||
assert!(
|
||||
dbg.contains("redacted"),
|
||||
"UnitKeyFile Debug missing redaction marker: {dbg}"
|
||||
);
|
||||
// Non-secret shape is still useful for diagnostics.
|
||||
assert!(dbg.contains("encrypted_keys_len: 2"), "{dbg}");
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod read_mkb_tests {
|
||||
use super::*;
|
||||
use crate::scsi::{DataDirection, SCSI_READ_DISC_STRUCTURE, ScsiResult, ScsiTransport};
|
||||
|
||||
/// A drive that answers READ DISC STRUCTURE format 0x83 from a scripted set
|
||||
/// of packs and records every CDB it was handed.
|
||||
struct MkbDrive {
|
||||
/// One entry per pack: the pack's MKB payload bytes.
|
||||
packs: Vec<Vec<u8>>,
|
||||
cdbs: Vec<Vec<u8>>,
|
||||
}
|
||||
|
||||
impl ScsiTransport for MkbDrive {
|
||||
fn execute(
|
||||
&mut self,
|
||||
cdb: &[u8],
|
||||
_direction: DataDirection,
|
||||
data: &mut [u8],
|
||||
_timeout_ms: u32,
|
||||
) -> crate::error::Result<ScsiResult> {
|
||||
self.cdbs.push(cdb.to_vec());
|
||||
// Pack number is carried in the CDB address field (bytes 2..6),
|
||||
// MMC-6 READ DISC STRUCTURE.
|
||||
let pack = u32::from_be_bytes([cdb[2], cdb[3], cdb[4], cdb[5]]) as usize;
|
||||
let body = self.packs.get(pack).cloned().unwrap_or_default();
|
||||
// Header: BE16 data length (counts the 2 header bytes that follow
|
||||
// it plus the payload), reserved byte, pack count, then payload.
|
||||
let data_len = body.len() + 2;
|
||||
data[0..2].copy_from_slice(&(data_len as u16).to_be_bytes());
|
||||
data[2] = 0x00;
|
||||
data[3] = self.packs.len() as u8;
|
||||
data[4..4 + body.len()].copy_from_slice(&body);
|
||||
Ok(ScsiResult {
|
||||
status: 0,
|
||||
bytes_transferred: 4 + body.len(),
|
||||
sense: [0u8; 32],
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// `read_mkb_from_drive` is the in-drive MKB source: every AACS derivation
|
||||
/// downstream (`mkb_find_mk_dv`, the subset-difference walk, the whole
|
||||
/// Media Key ladder) consumes exactly what it returns. An empty return is
|
||||
/// not a benign "no MKB" — it is a total read failure reported as success,
|
||||
/// and every derivation then fails with a key-not-found code that points
|
||||
/// the operator at their keydb rather than at the drive.
|
||||
///
|
||||
/// This pins the CONTENT: the concatenated payload of all packs, in pack
|
||||
/// order, byte for byte.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_returns_the_concatenated_pack_payload() {
|
||||
let pack0: Vec<u8> = (0..600u32).map(|i| (i % 251) as u8).collect();
|
||||
let pack1: Vec<u8> = (0..300u32).map(|i| (i % 253) as u8 ^ 0xA5).collect();
|
||||
let mut drive = MkbDrive {
|
||||
packs: vec![pack0.clone(), pack1.clone()],
|
||||
cdbs: Vec::new(),
|
||||
};
|
||||
|
||||
let mkb = read_mkb_from_drive(&mut drive).expect("scripted drive answers");
|
||||
|
||||
let mut expected = pack0.clone();
|
||||
expected.extend_from_slice(&pack1);
|
||||
assert_eq!(
|
||||
mkb.len(),
|
||||
expected.len(),
|
||||
"every pack's payload must be concatenated, none dropped"
|
||||
);
|
||||
assert!(
|
||||
mkb == expected,
|
||||
"MKB bytes must be the drive's payload in pack order; first \
|
||||
mismatch at {:?}",
|
||||
(0..expected.len()).find(|&i| mkb[i] != expected[i])
|
||||
);
|
||||
|
||||
// MMC-6 READ DISC STRUCTURE with the AACS MKB format code, one command
|
||||
// per pack, pack number in the address field.
|
||||
assert_eq!(drive.cdbs.len(), 2, "one command per declared pack");
|
||||
for (i, cdb) in drive.cdbs.iter().enumerate() {
|
||||
assert_eq!(cdb[0], SCSI_READ_DISC_STRUCTURE, "opcode");
|
||||
assert_eq!(cdb[7], 0x83, "AACS MKB disc-structure format code");
|
||||
assert_eq!(
|
||||
u32::from_be_bytes([cdb[2], cdb[3], cdb[4], cdb[5]]),
|
||||
i as u32,
|
||||
"pack {i} must be requested by number"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The CDB is what the drive actually acts on, and every byte of it is
|
||||
/// load-bearing: a wrong format code returns a different disc structure
|
||||
/// entirely, and a wrong allocation length truncates the pack. The existing
|
||||
/// test above pins the opcode, the format code and the pack number; this
|
||||
/// pins the WHOLE 12-byte CDB, so no field can drift unnoticed.
|
||||
///
|
||||
/// Expected layout (MMC-6 READ DISC STRUCTURE, AACS MKB format):
|
||||
/// `[0]` opcode, `[1]` media type 0x01, `[2..6]` address = pack number
|
||||
/// (BE32), `[6]` layer 0, `[7]` format 0x83, `[8..10]` allocation length
|
||||
/// BE16 = 32772 = `0x80 0x04`, `[10..12]` reserved/control.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_issues_the_exact_mmc_cdb_for_each_pack() {
|
||||
let mut drive = MkbDrive {
|
||||
packs: vec![vec![0x11u8; 64], vec![0x22u8; 64], vec![0x33u8; 64]],
|
||||
cdbs: Vec::new(),
|
||||
};
|
||||
read_mkb_from_drive(&mut drive).expect("scripted drive answers");
|
||||
|
||||
assert_eq!(drive.cdbs.len(), 3, "one command per declared pack");
|
||||
for (pack, cdb) in drive.cdbs.iter().enumerate() {
|
||||
let p = pack as u32;
|
||||
let expected: [u8; 12] = [
|
||||
SCSI_READ_DISC_STRUCTURE,
|
||||
0x01,
|
||||
(p >> 24) as u8,
|
||||
(p >> 16) as u8,
|
||||
(p >> 8) as u8,
|
||||
p as u8,
|
||||
0x00,
|
||||
0x83, // AACS MKB disc-structure format
|
||||
0x80, // allocation length 32772 = 0x8004, high byte
|
||||
0x04, // …low byte
|
||||
0x00,
|
||||
0x00,
|
||||
];
|
||||
assert_eq!(
|
||||
cdb.as_slice(),
|
||||
&expected[..],
|
||||
"CDB for pack {pack} must match the MMC-6 READ DISC STRUCTURE layout"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// A pack payload filling the FULL 32768-byte window must come back whole.
|
||||
/// The `len > 0 && len <= 32768` bound is what stands between a maximal
|
||||
/// pack and a silently dropped one, and the small payloads used elsewhere
|
||||
/// in this module never reach it.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_accepts_a_full_size_pack() {
|
||||
let full: Vec<u8> = (0..32768u32).map(|i| (i % 251) as u8).collect();
|
||||
let other: Vec<u8> = (0..32768u32).map(|i| (i % 241) as u8 ^ 0x5A).collect();
|
||||
// TWO maximal packs: the first-pack read and the per-pack loop carry
|
||||
// separate bounds, so both must accept a full-window payload.
|
||||
let mut drive = MkbDrive {
|
||||
packs: vec![full.clone(), other.clone()],
|
||||
cdbs: Vec::new(),
|
||||
};
|
||||
let mkb = read_mkb_from_drive(&mut drive).expect("scripted drive answers");
|
||||
assert_eq!(
|
||||
mkb.len(),
|
||||
65536,
|
||||
"neither maximal pack may be dropped at the size bound"
|
||||
);
|
||||
let mut expected = full.clone();
|
||||
expected.extend_from_slice(&other);
|
||||
assert!(mkb == expected, "both maximal packs' bytes must be intact");
|
||||
}
|
||||
|
||||
/// A pack that declares only the 2-byte header and NO payload contributes
|
||||
/// nothing, and must not push a phantom byte into the MKB — an off-by-one
|
||||
/// at the zero-length boundary corrupts every following pack's alignment.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_zero_length_pack_contributes_nothing() {
|
||||
let mut drive = MkbDrive {
|
||||
packs: vec![Vec::new(), vec![0xABu8; 32]],
|
||||
cdbs: Vec::new(),
|
||||
};
|
||||
let mkb = read_mkb_from_drive(&mut drive).expect("scripted drive answers");
|
||||
assert_eq!(
|
||||
mkb.len(),
|
||||
32,
|
||||
"an empty pack adds no bytes; only pack 1's payload is present"
|
||||
);
|
||||
assert!(mkb == vec![0xABu8; 32], "and the bytes are pack 1's");
|
||||
}
|
||||
|
||||
/// A drive that DECLARES more payload than it returned must not be
|
||||
/// believed. The BE16 length in the response header is drive-supplied data:
|
||||
/// a firmware bug, a short transfer, or a hostile device can put a value in
|
||||
/// it that runs past the 32772-byte buffer. Copying `len` bytes on that word
|
||||
/// alone panics the rip thread mid-scan.
|
||||
///
|
||||
/// Both the first-pack read and the per-pack loop carry the same bound, so
|
||||
/// both are exercised here: the over-declared pack contributes nothing and
|
||||
/// the honest pack still comes through.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_ignores_a_pack_declaring_more_than_the_buffer_holds() {
|
||||
/// Pack 0 is honest; pack 1 declares a 60000-byte payload it never sent.
|
||||
struct LyingDrive {
|
||||
honest: Vec<u8>,
|
||||
}
|
||||
impl ScsiTransport for LyingDrive {
|
||||
fn execute(
|
||||
&mut self,
|
||||
cdb: &[u8],
|
||||
_direction: DataDirection,
|
||||
data: &mut [u8],
|
||||
_timeout_ms: u32,
|
||||
) -> crate::error::Result<ScsiResult> {
|
||||
let pack = u32::from_be_bytes([cdb[2], cdb[3], cdb[4], cdb[5]]);
|
||||
data[3] = 2; // two packs declared
|
||||
if pack == 0 {
|
||||
let dl = self.honest.len() + 2;
|
||||
data[0..2].copy_from_slice(&(dl as u16).to_be_bytes());
|
||||
data[4..4 + self.honest.len()].copy_from_slice(&self.honest);
|
||||
} else {
|
||||
// A length far beyond the 32772-byte response buffer.
|
||||
data[0..2].copy_from_slice(&60_000u16.to_be_bytes());
|
||||
}
|
||||
Ok(ScsiResult {
|
||||
status: 0,
|
||||
bytes_transferred: 4,
|
||||
sense: [0u8; 32],
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
let honest = vec![0xC7u8; 256];
|
||||
let mut drive = LyingDrive {
|
||||
honest: honest.clone(),
|
||||
};
|
||||
let mkb = read_mkb_from_drive(&mut drive).expect("an over-declared pack is not an error");
|
||||
assert_eq!(
|
||||
mkb.len(),
|
||||
honest.len(),
|
||||
"only the honest pack's bytes may be taken; the over-declared pack \
|
||||
contributes nothing and must not be read past the buffer"
|
||||
);
|
||||
assert!(mkb == honest, "and those bytes are pack 0's");
|
||||
}
|
||||
|
||||
/// The same over-declaration on the FIRST pack, which uses a separate bound
|
||||
/// from the loop's.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_ignores_a_first_pack_declaring_more_than_the_buffer() {
|
||||
struct LyingFirst;
|
||||
impl ScsiTransport for LyingFirst {
|
||||
fn execute(
|
||||
&mut self,
|
||||
_cdb: &[u8],
|
||||
_direction: DataDirection,
|
||||
data: &mut [u8],
|
||||
_timeout_ms: u32,
|
||||
) -> crate::error::Result<ScsiResult> {
|
||||
data[0..2].copy_from_slice(&60_000u16.to_be_bytes());
|
||||
data[3] = 1;
|
||||
Ok(ScsiResult {
|
||||
status: 0,
|
||||
bytes_transferred: 4,
|
||||
sense: [0u8; 32],
|
||||
})
|
||||
}
|
||||
}
|
||||
let mkb = read_mkb_from_drive(&mut LyingFirst).expect("not an error");
|
||||
assert!(
|
||||
mkb.is_empty(),
|
||||
"a first pack declaring more than the buffer holds yields no bytes"
|
||||
);
|
||||
}
|
||||
|
||||
/// A single-pack disc still yields that pack's bytes — the common case, and
|
||||
/// the one where a body returning an empty vector looks most plausible.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_returns_a_single_packs_payload() {
|
||||
let pack: Vec<u8> = (0..1024u32).map(|i| (i * 7 % 256) as u8).collect();
|
||||
let mut drive = MkbDrive {
|
||||
packs: vec![pack.clone()],
|
||||
cdbs: Vec::new(),
|
||||
};
|
||||
let mkb = read_mkb_from_drive(&mut drive).expect("scripted drive answers");
|
||||
assert_eq!(mkb.len(), pack.len(), "single pack payload length");
|
||||
assert!(mkb == pack, "single pack payload bytes");
|
||||
}
|
||||
|
||||
/// A drive that reports a header-only response (`data_len < 2`) has no MKB
|
||||
/// to give. That must be an EMPTY vec, not a partial one — the distinction
|
||||
/// matters because the AACS paths treat a non-empty MKB as parseable.
|
||||
#[test]
|
||||
fn read_mkb_from_drive_empty_response_is_empty() {
|
||||
struct NoMkb;
|
||||
impl ScsiTransport for NoMkb {
|
||||
fn execute(
|
||||
&mut self,
|
||||
_cdb: &[u8],
|
||||
_direction: DataDirection,
|
||||
data: &mut [u8],
|
||||
_timeout_ms: u32,
|
||||
) -> crate::error::Result<ScsiResult> {
|
||||
data[0..2].copy_from_slice(&0u16.to_be_bytes());
|
||||
Ok(ScsiResult {
|
||||
status: 0,
|
||||
bytes_transferred: 4,
|
||||
sense: [0u8; 32],
|
||||
})
|
||||
}
|
||||
}
|
||||
let mkb = read_mkb_from_drive(&mut NoMkb).expect("no-MKB drive still returns Ok");
|
||||
assert!(
|
||||
mkb.is_empty(),
|
||||
"a header-only response carries no MKB bytes"
|
||||
);
|
||||
}
|
||||
|
||||
/// A transport failure on the FIRST pack must propagate as an error — the
|
||||
/// MKB is the root of the whole AACS ladder, so an unreadable one cannot be
|
||||
/// downgraded to "an MKB with no records".
|
||||
#[test]
|
||||
fn read_mkb_from_drive_propagates_the_first_pack_failure() {
|
||||
struct DeadDrive;
|
||||
impl ScsiTransport for DeadDrive {
|
||||
fn execute(
|
||||
&mut self,
|
||||
_cdb: &[u8],
|
||||
_direction: DataDirection,
|
||||
_data: &mut [u8],
|
||||
_timeout_ms: u32,
|
||||
) -> crate::error::Result<ScsiResult> {
|
||||
Err(crate::error::Error::ScsiError {
|
||||
opcode: SCSI_READ_DISC_STRUCTURE,
|
||||
status: 0x02,
|
||||
sense: None,
|
||||
})
|
||||
}
|
||||
}
|
||||
assert!(
|
||||
read_mkb_from_drive(&mut DeadDrive).is_err(),
|
||||
"an unreadable MKB must surface as an error, not an empty MKB"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+105
@@ -432,4 +432,109 @@ mod tests {
|
||||
"trim keeps the framed records, dropping the end marker and padding"
|
||||
);
|
||||
}
|
||||
|
||||
// ── BE24 length field: all THREE bytes ────────────────────────────────
|
||||
|
||||
/// The record length is a big-endian **24-bit** field, so the high byte
|
||||
/// carries lengths of 64 KiB and up. The MKB records that matter most are
|
||||
/// exactly that size — a real UHD cvalue table is `46_101 * 16` bytes and a
|
||||
/// `0x2d` variant record is ~92 KiB — so a walker that dropped the high
|
||||
/// byte would mis-frame every record of a real MKB from the first big one
|
||||
/// onward, and every downstream key lookup would read the wrong bytes.
|
||||
///
|
||||
/// (The pre-existing high-byte test used total length `0x0110`, whose high
|
||||
/// byte is ZERO — it exercised the middle byte only. This one puts a
|
||||
/// non-zero value in the high byte.)
|
||||
#[test]
|
||||
fn mkb_records_honors_the_high_byte_of_the_be24_length() {
|
||||
const TOTAL: usize = 0x0001_0004; // 65_540 — high byte 0x01
|
||||
let mut mkb = vec![REC_VKD_TABLE, 0x01, 0x00, 0x04];
|
||||
mkb.resize(TOTAL, 0xAB);
|
||||
// A second record follows, so a walker that mis-read the length would
|
||||
// frame a different number of records rather than merely a short one.
|
||||
mkb.extend(rec(REC_TYPE_AND_VERSION, &[0x11; 8]));
|
||||
|
||||
let recs = walk_mkb(&mkb);
|
||||
assert_eq!(recs.len(), 2, "the big record must be framed as ONE record");
|
||||
assert_eq!(
|
||||
recs[0].rec_len, TOTAL,
|
||||
"rec_len must include the high BE24 byte"
|
||||
);
|
||||
assert_eq!(recs[0].body.len(), TOTAL - 4);
|
||||
assert_eq!(
|
||||
recs[1].rec_type, REC_TYPE_AND_VERSION,
|
||||
"the following record must start where the big one ends"
|
||||
);
|
||||
}
|
||||
|
||||
// ── Header-only records and the exact end marker ──────────────────────
|
||||
|
||||
/// `rec_len == 4` is a well-formed HEADER-ONLY record (the minimum the
|
||||
/// walker accepts), including one sitting at the very end of the buffer
|
||||
/// with no bytes after it. Rejecting either — the `pos + 4` bound or the
|
||||
/// `rec_len < 4` floor being off by one — silently drops the MKB's last
|
||||
/// record, and "the record isn't there" is indistinguishable from "the disc
|
||||
/// doesn't carry it".
|
||||
#[test]
|
||||
fn mkb_records_yields_a_header_only_record_at_the_buffer_end() {
|
||||
let mut mkb = rec(REC_TYPE_AND_VERSION, &[0xAA, 0xBB]);
|
||||
mkb.extend([REC_VKD_TABLE, 0x00, 0x00, 0x04]); // 4-byte, empty body, at EOF
|
||||
assert_eq!(
|
||||
mkb.len(),
|
||||
10,
|
||||
"sanity: the last record ends at the buffer end"
|
||||
);
|
||||
|
||||
let recs = walk_mkb(&mkb);
|
||||
assert_eq!(recs.len(), 2, "the trailing header-only record is a record");
|
||||
assert_eq!(recs[1].rec_type, REC_VKD_TABLE);
|
||||
assert_eq!(recs[1].rec_len, 4);
|
||||
assert!(recs[1].body.is_empty());
|
||||
}
|
||||
|
||||
/// ONLY the exact `00 00 00 00` marker ends the walk. A record whose TYPE
|
||||
/// happens to be `0x00` but which declares a real length is a record, not
|
||||
/// the end of the MKB — stopping there would truncate everything after it,
|
||||
/// including the cvalue and verify records the key derivation needs.
|
||||
#[test]
|
||||
fn mkb_records_stops_only_on_the_all_zero_end_marker() {
|
||||
// A type-0 record of length 8, then a normal record, then the marker.
|
||||
let mut mkb = vec![0x00, 0x00, 0x00, 0x08, 1, 2, 3, 4];
|
||||
mkb.extend(rec(REC_VKD_TABLE, &[0x55; 16]));
|
||||
mkb.extend([0x00, 0x00, 0x00, 0x00]); // the real end marker
|
||||
mkb.extend(rec(0x99, &[0xFF; 4])); // past the marker: not walked
|
||||
|
||||
let recs = walk_mkb(&mkb);
|
||||
assert_eq!(
|
||||
recs.len(),
|
||||
2,
|
||||
"a type-0 record with a non-zero length is a record, not the end"
|
||||
);
|
||||
assert_eq!(recs[0].rec_type, 0x00);
|
||||
assert_eq!(recs[0].rec_len, 8);
|
||||
assert_eq!(recs[1].rec_type, REC_VKD_TABLE);
|
||||
assert_eq!(recs[1].body, vec![0x55; 16]);
|
||||
}
|
||||
|
||||
/// `mkb_type_raw` reports the 32-bit MKBType field verbatim ([C] §3.2.5.1.1
|
||||
/// Table 3-2), including a value this build does not recognise — the caller
|
||||
/// uses it to tell "unknown MKB generation" from "no Type record at all".
|
||||
/// All four bytes must come from the record body; reading any of them from
|
||||
/// the wrong offset yields a type that silently classifies as a different
|
||||
/// AACS generation.
|
||||
///
|
||||
/// The recognised constants all share bytes with the `0x10` record-type
|
||||
/// header byte (e.g. `MKB_21_CATEGORY_C` is `48 15 10 03`), so this uses a
|
||||
/// value with four distinct bytes, none of them `0x10`.
|
||||
#[test]
|
||||
fn mkb_type_raw_reads_all_four_body_bytes() {
|
||||
const RAW: u32 = 0xDEAD_BEEF;
|
||||
let mkb = type_and_version(RAW, 7);
|
||||
assert_eq!(
|
||||
mkb_type_raw(&mkb),
|
||||
Some(RAW),
|
||||
"every byte of the MKBType field must come from the record body"
|
||||
);
|
||||
assert_eq!(mkb_version(&mkb), Some(7));
|
||||
}
|
||||
}
|
||||
|
||||
+277
-36
@@ -40,17 +40,24 @@ pub mod trace;
|
||||
pub mod types;
|
||||
pub mod variant;
|
||||
|
||||
/// On-disc UDF paths to the AACS key-input files.
|
||||
/// On-disc UDF paths to the AACS key-input files, plus HD DVD AACS-directory
|
||||
/// discovery.
|
||||
///
|
||||
/// BD and UHD keep their key material under `/AACS/…`; HD DVD keeps the
|
||||
/// equivalents under `/ANY!/…` with different names (`VTKF000.AACS` is the
|
||||
/// title-key file — magic `DVD_HD_V_TKF`; `MKBROM.AACS` is the MKB). The
|
||||
/// container difference is expressed here purely as DATA: each ROLE
|
||||
/// ([`UNIT_KEY_RO_PATHS`], [`MKB_PATHS`], [`CONTENT_CERT_PATHS`]) is an ordered
|
||||
/// candidate list, and every reader walks it with [`read_first`] taking the
|
||||
/// first that reads. No reader ever branches on disc type — a BD/UHD disc has
|
||||
/// the `/AACS/` files so those win; an HD DVD has neither, so it falls through
|
||||
/// to the `/ANY!/` entry. Centralised so `resolve_vid_only`, `read_aacs_inputs`,
|
||||
/// BD and UHD keep their key material under a fixed `/AACS/…` tree, so those
|
||||
/// paths are constants. HD DVD keeps the equivalents in a reserved root
|
||||
/// directory whose NAME is authoring-house-specific — observed `ANY!` (Dukes
|
||||
/// of Hazzard) and `AAC!` (Freedom / Memory-Tech), each with a `<name>!_BAK`
|
||||
/// mirror — and whose title-key file is NOT always `VTKF000.AACS` (Freedom
|
||||
/// ships `VTKF090.AACS` + `VTKF100.AACS`). So the HD DVD files are DISCOVERED
|
||||
/// from the parsed UDF tree ([`find_hddvd_aacs_dir`] + [`role_paths`]), never
|
||||
/// hardcoded.
|
||||
///
|
||||
/// Each key ROLE ([`AacsRole`]) resolves to an ordered candidate list — the
|
||||
/// BD/UHD constants first, then whatever the HD DVD directory actually holds —
|
||||
/// which every reader walks with [`read_first`], first-that-reads. No reader
|
||||
/// ever branches on disc type: a BD/UHD disc has the `/AACS/` files so those
|
||||
/// win; an HD DVD has none of them, so it falls through to the discovered
|
||||
/// entries. Centralised so `resolve_vid_only`, `read_aacs_inputs`,
|
||||
/// `read_mkb_content`, and `read_aacs_version` can never silently diverge the
|
||||
/// disc_hash / MKB / VID that another reader feeds a key service.
|
||||
pub const PATH_UNIT_KEY_RO: &str = "/AACS/Unit_Key_RO.inf";
|
||||
@@ -59,41 +66,107 @@ pub const PATH_MKB_RO: &str = "/AACS/MKB_RO.inf";
|
||||
pub const PATH_MKB_RW: &str = "/AACS/MKB_RW.inf";
|
||||
pub const PATH_CONTENT_CERT: &str = "/AACS/Content000.cer";
|
||||
pub const PATH_CONTENT_CERT_ALT: &str = "/AACS/Content001.cer";
|
||||
/// HD DVD title-key file (`/ANY!/`), forwarded as `inf_b64`; the key service
|
||||
/// recognises it by its `DVD_HD_V_TKF` magic.
|
||||
pub const PATH_VTKF_HDDVD: &str = "/ANY!/VTKF000.AACS";
|
||||
/// HD DVD Media Key Block (`/ANY!/`), forwarded as `mkb_b64`.
|
||||
pub const PATH_MKBROM_HDDVD: &str = "/ANY!/MKBROM.AACS";
|
||||
/// HD DVD content certificate (`/ANY!/`); byte 0 gives the AACS major (0x00 → V10).
|
||||
pub const PATH_CONTENT_CERT_HDDVD: &str = "/ANY!/CONTENT_CERT.AACS";
|
||||
|
||||
/// Title-key / `Unit_Key_RO.inf` role, in resolution order (BD/UHD, then HD DVD).
|
||||
pub const UNIT_KEY_RO_PATHS: &[&str] = &[
|
||||
PATH_UNIT_KEY_RO,
|
||||
PATH_UNIT_KEY_RO_DUPLICATE,
|
||||
PATH_VTKF_HDDVD,
|
||||
];
|
||||
/// MKB role, in resolution order (BD/UHD RO then RW, then HD DVD).
|
||||
pub const MKB_PATHS: &[&str] = &[PATH_MKB_RO, PATH_MKB_RW, PATH_MKBROM_HDDVD];
|
||||
/// Content-certificate role, in resolution order (BD/UHD, then HD DVD).
|
||||
pub const CONTENT_CERT_PATHS: &[&str] = &[
|
||||
PATH_CONTENT_CERT,
|
||||
PATH_CONTENT_CERT_ALT,
|
||||
PATH_CONTENT_CERT_HDDVD,
|
||||
];
|
||||
/// An AACS key-input role. [`role_paths`] maps it to an ordered candidate path
|
||||
/// list (BD/UHD constants, then the discovered HD DVD files).
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub enum AacsRole {
|
||||
/// Title-key file: BD/UHD `Unit_Key_RO.inf`, HD DVD `VTKF*.AACS`
|
||||
/// (magic `DVD_HD_V_TKF`). The disc_hash is `SHA1` of this file.
|
||||
UnitKey,
|
||||
/// Media Key Block: BD/UHD `MKB_RO/RW.inf`, HD DVD `MKBROM.AACS`.
|
||||
Mkb,
|
||||
/// Content certificate: BD/UHD `Content000/001.cer`, HD DVD
|
||||
/// `CONTENT_CERT.AACS` (byte 0 gives the AACS major).
|
||||
ContentCert,
|
||||
}
|
||||
|
||||
/// Walk an AACS role's candidate paths and return the first that reads.
|
||||
/// The HD DVD AACS directory in a parsed UDF tree, if present.
|
||||
///
|
||||
/// Identified structurally, NOT by a hardcoded name: the root child directory
|
||||
/// whose name ends in `!` (so the `<name>!_BAK` backup mirror, which also ends
|
||||
/// in a non-`!` char, is not mistaken for it) and which contains `MKBROM.AACS`.
|
||||
/// Observed real names: `ANY!` (Dukes of Hazzard), `AAC!` (Freedom). A BD/UHD
|
||||
/// disc has no such directory → `None`.
|
||||
pub(crate) fn find_hddvd_aacs_dir(udf: &crate::udf::UdfFs) -> Option<&crate::udf::DirEntry> {
|
||||
udf.root.entries.iter().find(|e| {
|
||||
e.is_dir
|
||||
&& e.name.ends_with('!')
|
||||
&& e.entries
|
||||
.iter()
|
||||
.any(|c| !c.is_dir && c.name.eq_ignore_ascii_case("MKBROM.AACS"))
|
||||
})
|
||||
}
|
||||
|
||||
/// Ordered candidate paths for an AACS key [`AacsRole`]: the fixed BD/UHD
|
||||
/// `/AACS/…` paths first, then the actual HD DVD files discovered in the disc's
|
||||
/// AACS directory (see [`find_hddvd_aacs_dir`]). A disc has only one family, so
|
||||
/// the other family's entries simply never read.
|
||||
///
|
||||
/// For [`AacsRole::UnitKey`] every `VTKF*.AACS` in the directory is appended in
|
||||
/// sorted name order — a disc may carry more than one variant (Freedom:
|
||||
/// `VTKF090` + `VTKF100`), not just `VTKF000`.
|
||||
pub(crate) fn role_paths(udf: &crate::udf::UdfFs, role: AacsRole) -> Vec<String> {
|
||||
let mut v: Vec<String> = match role {
|
||||
AacsRole::UnitKey => vec![PATH_UNIT_KEY_RO, PATH_UNIT_KEY_RO_DUPLICATE],
|
||||
AacsRole::Mkb => vec![PATH_MKB_RO, PATH_MKB_RW],
|
||||
AacsRole::ContentCert => vec![PATH_CONTENT_CERT, PATH_CONTENT_CERT_ALT],
|
||||
}
|
||||
.into_iter()
|
||||
.map(String::from)
|
||||
.collect();
|
||||
|
||||
if let Some(dir) = find_hddvd_aacs_dir(udf) {
|
||||
let d = &dir.name;
|
||||
match role {
|
||||
AacsRole::Mkb => v.push(format!("/{d}/MKBROM.AACS")),
|
||||
AacsRole::ContentCert => v.push(format!("/{d}/CONTENT_CERT.AACS")),
|
||||
AacsRole::UnitKey => {
|
||||
// Glob VTKF*.AACS — the title-key filename is not fixed at
|
||||
// VTKF000 (Freedom ships VTKF090 + VTKF100). Sorted for a
|
||||
// deterministic try order.
|
||||
//
|
||||
// Each VTKF%%%.AACS is bound to ONE playlist (VPLST%%%.XPL): the
|
||||
// TKF's 12-byte PLAYLIST_NAME field (bytes 0x10..0x1C) names the
|
||||
// playlist whose Title Keys it carries, and keys from a TKF whose
|
||||
// name does not match the title's playlist must not be used. The
|
||||
// caller resolves this by trying candidates in sorted order and
|
||||
// decrypting with the one whose keys verify — correct for a
|
||||
// single-playlist disc; a name-matched selection keyed on the
|
||||
// active playlist is the precise form for multi-playlist discs.
|
||||
let mut names: Vec<&str> = dir
|
||||
.entries
|
||||
.iter()
|
||||
.filter(|e| !e.is_dir)
|
||||
.filter(|e| {
|
||||
let u = e.name.to_ascii_uppercase();
|
||||
u.starts_with("VTKF") && u.ends_with(".AACS")
|
||||
})
|
||||
.map(|e| e.name.as_str())
|
||||
.collect();
|
||||
names.sort_unstable();
|
||||
v.extend(names.into_iter().map(|n| format!("/{d}/{n}")));
|
||||
}
|
||||
}
|
||||
}
|
||||
v
|
||||
}
|
||||
|
||||
/// Walk an AACS role's candidate paths (from [`role_paths`]) and return the
|
||||
/// first that reads.
|
||||
///
|
||||
/// `read` performs the actual per-path read (full file or bounded prefix), so
|
||||
/// callers share the same first-present walk regardless of read style. Returns
|
||||
/// [`Error::AacsNoKeys`] if no candidate is present. This is the single place
|
||||
/// the `/AACS/` (BD/UHD) vs `/ANY!/` (HD DVD) layout difference is resolved.
|
||||
pub(crate) fn read_first<F>(candidates: &[&str], mut read: F) -> crate::error::Result<Vec<u8>>
|
||||
/// [`Error::AacsNoKeys`] if no candidate is present. Generic over the path
|
||||
/// element (`&str` or owned `String`) so it accepts the `Vec<String>` that
|
||||
/// [`role_paths`] builds from the discovered HD DVD directory.
|
||||
pub(crate) fn read_first<S, F>(candidates: &[S], mut read: F) -> crate::error::Result<Vec<u8>>
|
||||
where
|
||||
S: AsRef<str>,
|
||||
F: FnMut(&str) -> crate::error::Result<Vec<u8>>,
|
||||
{
|
||||
for path in candidates {
|
||||
if let Ok(buf) = read(path) {
|
||||
if let Ok(buf) = read(path.as_ref()) {
|
||||
return Ok(buf);
|
||||
}
|
||||
}
|
||||
@@ -161,4 +234,172 @@ mod tests {
|
||||
None,
|
||||
);
|
||||
}
|
||||
|
||||
// ── HD DVD AACS directory / filename discovery ────────────────────────
|
||||
//
|
||||
// The HD DVD AACS dir name and title-key filename are authoring-specific
|
||||
// and were previously hardcoded to `/ANY!/VTKF000.AACS`. These verify the
|
||||
// discovery replacement against both real-disc shapes: Freedom (`AAC!` +
|
||||
// `VTKF090`/`VTKF100`) and a BD/UHD disc (no HD DVD dir).
|
||||
|
||||
#[test]
|
||||
fn role_paths_discovers_hddvd_dir_and_globs_all_vtkf_variants() {
|
||||
use crate::udf::fixture::*;
|
||||
// Freedom-shaped: an `AAC!` dir (NOT `ANY!`) holding MKBROM + two VTKF
|
||||
// variants (090/100, NOT 000) + a VTUF usage file (must be excluded),
|
||||
// plus the `AAC!_BAK` mirror (must NOT be picked as the AACS dir).
|
||||
let mut disc = MemDisc::new();
|
||||
let aacs_files = vec![
|
||||
file("MKBROM.AACS", 100, 5000, 4096, true),
|
||||
file("CONTENT_CERT.AACS", 101, 5100, 2048, true),
|
||||
file("VTKF100.AACS", 102, 5200, 2048, true),
|
||||
file("VTKF090.AACS", 103, 5300, 2048, true),
|
||||
file("VTUF090.AACS", 104, 5400, 2048, true),
|
||||
];
|
||||
let bak_files = vec![file("MKBROM.AACS", 110, 6000, 4096, true)];
|
||||
let root = DirSpec {
|
||||
name: String::new(),
|
||||
icb_lba: 10,
|
||||
dir_data_lba: 11,
|
||||
files: Vec::new(),
|
||||
subdirs: vec![
|
||||
DirSpec {
|
||||
name: "AAC!".to_string(),
|
||||
icb_lba: 20,
|
||||
dir_data_lba: 21,
|
||||
files: aacs_files,
|
||||
subdirs: vec![],
|
||||
},
|
||||
DirSpec {
|
||||
name: "AAC!_BAK".to_string(),
|
||||
icb_lba: 30,
|
||||
dir_data_lba: 31,
|
||||
files: bak_files,
|
||||
subdirs: vec![],
|
||||
},
|
||||
],
|
||||
};
|
||||
build_udf_skeleton(&mut disc, 10);
|
||||
lay_dir(&mut disc, &root);
|
||||
let udf = crate::udf::read_filesystem(&mut disc).expect("fs");
|
||||
|
||||
// Discovered structurally (ends in '!', holds MKBROM.AACS) — the real
|
||||
// AACS dir, never the `_BAK` mirror.
|
||||
let dir = super::find_hddvd_aacs_dir(&udf).expect("aacs dir");
|
||||
assert_eq!(dir.name, "AAC!");
|
||||
|
||||
// UnitKey: BD/UHD paths first, then EVERY VTKF*.AACS in sorted order
|
||||
// (090 before 100) — NOT hardcoded VTKF000; VTUF (usage) excluded.
|
||||
assert_eq!(
|
||||
super::role_paths(&udf, super::AacsRole::UnitKey),
|
||||
vec![
|
||||
super::PATH_UNIT_KEY_RO.to_string(),
|
||||
super::PATH_UNIT_KEY_RO_DUPLICATE.to_string(),
|
||||
"/AAC!/VTKF090.AACS".to_string(),
|
||||
"/AAC!/VTKF100.AACS".to_string(),
|
||||
]
|
||||
);
|
||||
assert_eq!(
|
||||
super::role_paths(&udf, super::AacsRole::Mkb)
|
||||
.last()
|
||||
.unwrap(),
|
||||
"/AAC!/MKBROM.AACS"
|
||||
);
|
||||
assert_eq!(
|
||||
super::role_paths(&udf, super::AacsRole::ContentCert)
|
||||
.last()
|
||||
.unwrap(),
|
||||
"/AAC!/CONTENT_CERT.AACS"
|
||||
);
|
||||
}
|
||||
|
||||
/// The `!`-suffix and the `MKBROM.AACS` presence test are BOTH required —
|
||||
/// the discovery is a conjunction, not a disjunction.
|
||||
///
|
||||
/// The existing fixtures only ever present a directory that satisfies both
|
||||
/// (`AAC!` with `MKBROM.AACS`) alongside one that satisfies neither
|
||||
/// (`AAC!_BAK` — which contains `MKBROM.AACS` but is ALSO reached only after
|
||||
/// the real dir), so either half of the conjunction could be dropped and the
|
||||
/// same directory would still be found. Here a directory satisfies the name
|
||||
/// half and NOT the contents half: it must not be picked.
|
||||
///
|
||||
/// If it were, the HD DVD path would resolve `MKBROM.AACS`,
|
||||
/// `CONTENT_CERT.AACS` and the title-key file under a directory that holds
|
||||
/// none of them — the disc reports "no AACS key files" and never rips.
|
||||
#[test]
|
||||
fn a_bang_suffixed_directory_without_mkbrom_is_not_the_aacs_directory() {
|
||||
use crate::udf::fixture::*;
|
||||
let mut disc = MemDisc::new();
|
||||
let root = DirSpec {
|
||||
name: String::new(),
|
||||
icb_lba: 10,
|
||||
dir_data_lba: 11,
|
||||
files: Vec::new(),
|
||||
subdirs: vec![DirSpec {
|
||||
// Ends in '!' — but carries no MKBROM.AACS, so it is not the
|
||||
// HD DVD AACS directory.
|
||||
name: "AAC!".to_string(),
|
||||
icb_lba: 20,
|
||||
dir_data_lba: 21,
|
||||
files: vec![
|
||||
file("VTKF090.AACS", 102, 5200, 2048, true),
|
||||
file("CONTENT_CERT.AACS", 103, 5300, 2048, true),
|
||||
],
|
||||
subdirs: vec![],
|
||||
}],
|
||||
};
|
||||
build_udf_skeleton(&mut disc, 10);
|
||||
lay_dir(&mut disc, &root);
|
||||
let udf = crate::udf::read_filesystem(&mut disc).expect("fs");
|
||||
|
||||
assert!(
|
||||
super::find_hddvd_aacs_dir(&udf).is_none(),
|
||||
"a '!' directory without MKBROM.AACS is not the AACS directory"
|
||||
);
|
||||
assert_eq!(
|
||||
super::role_paths(&udf, super::AacsRole::UnitKey),
|
||||
vec![
|
||||
super::PATH_UNIT_KEY_RO.to_string(),
|
||||
super::PATH_UNIT_KEY_RO_DUPLICATE.to_string(),
|
||||
],
|
||||
"no HD DVD candidates may be appended from a directory that was \
|
||||
never identified as the AACS directory"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn role_paths_bd_uhd_disc_yields_no_hddvd_candidates() {
|
||||
use crate::udf::fixture::*;
|
||||
// A `/AACS/` tree (BD/UHD) has no '!' directory → discovery finds none
|
||||
// and the candidate list is exactly the static BD/UHD paths.
|
||||
let mut disc = MemDisc::new();
|
||||
let root = DirSpec {
|
||||
name: String::new(),
|
||||
icb_lba: 10,
|
||||
dir_data_lba: 11,
|
||||
files: Vec::new(),
|
||||
subdirs: vec![DirSpec {
|
||||
name: "AACS".to_string(),
|
||||
icb_lba: 20,
|
||||
dir_data_lba: 21,
|
||||
files: vec![
|
||||
file("Unit_Key_RO.inf", 100, 5000, 2048, true),
|
||||
file("MKB_RO.inf", 101, 5100, 2048, true),
|
||||
],
|
||||
subdirs: vec![],
|
||||
}],
|
||||
};
|
||||
build_udf_skeleton(&mut disc, 10);
|
||||
lay_dir(&mut disc, &root);
|
||||
let udf = crate::udf::read_filesystem(&mut disc).expect("fs");
|
||||
|
||||
assert!(super::find_hddvd_aacs_dir(&udf).is_none());
|
||||
assert_eq!(
|
||||
super::role_paths(&udf, super::AacsRole::UnitKey),
|
||||
vec![
|
||||
super::PATH_UNIT_KEY_RO.to_string(),
|
||||
super::PATH_UNIT_KEY_RO_DUPLICATE.to_string(),
|
||||
]
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -197,6 +197,17 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// A host cert whose (non-secret) certificate body and private key are both
|
||||
/// filled with `byte`, so a cert is identifiable in an aggregated list.
|
||||
fn cert(byte: u8) -> HostCert {
|
||||
HostCert {
|
||||
private_key: [byte; 20],
|
||||
certificate: vec![byte; 92],
|
||||
private_key_v2: None,
|
||||
certificate_v2: None,
|
||||
}
|
||||
}
|
||||
|
||||
fn dk(byte: u8, node: u16) -> DeviceKey {
|
||||
DeviceKey {
|
||||
key: [byte; 16],
|
||||
@@ -357,6 +368,42 @@ mod tests {
|
||||
assert_eq!(got.disc_hash, "vid-a");
|
||||
}
|
||||
|
||||
/// `Providers::host_certs` is the union across the provider array. It is not
|
||||
/// wired into the handshake today (see the module docs), so nothing else in
|
||||
/// the crate would notice a body that dropped every cert on the floor — and
|
||||
/// the day it IS wired in, a silently-empty cert list means the drive AACS
|
||||
/// authentication finds no host certificate to present and every disc fails
|
||||
/// to open, with no indication that the caller's certs were discarded.
|
||||
///
|
||||
/// Unlike the bulk key unions this one does NOT dedup (HostCert is not
|
||||
/// Ord/Hash), so the assertion is on the full concatenation in array order.
|
||||
#[test]
|
||||
fn providers_host_certs_unions_every_providers_certs_in_array_order() {
|
||||
struct Certs(Vec<HostCert>);
|
||||
impl KeyProvider for Certs {
|
||||
fn host_certs(&self) -> Vec<HostCert> {
|
||||
self.0.clone()
|
||||
}
|
||||
}
|
||||
// Distinguish certs by their (non-secret) certificate body, so the
|
||||
// assertion lands on WHICH certs came back, not merely how many.
|
||||
let a = Certs(vec![cert(0xA1), cert(0xA2)]);
|
||||
let b = Certs(vec![cert(0xB1)]);
|
||||
let arr: &[&dyn KeyProvider] = &[&a, &b];
|
||||
|
||||
let got = Providers(arr).host_certs();
|
||||
let bodies: Vec<Vec<u8>> = got.iter().map(|c| c.certificate.clone()).collect();
|
||||
assert_eq!(
|
||||
bodies,
|
||||
vec![vec![0xA1u8; 92], vec![0xA2u8; 92], vec![0xB1u8; 92]],
|
||||
"every provider's certs must survive the union, in array order"
|
||||
);
|
||||
// The private key travels with the cert — a union that returned default
|
||||
// certs would still have the right count.
|
||||
assert_eq!(got[0].private_key, [0xA1u8; 20]);
|
||||
assert_eq!(got[2].private_key, [0xB1u8; 20]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn providers_empty_array_yields_nothing() {
|
||||
let arr: &[&dyn KeyProvider] = &[];
|
||||
|
||||
+209
-17
@@ -13,7 +13,6 @@ use super::mkb::*;
|
||||
// ── Full VUK resolution chain ───────────────────────────────────────────────
|
||||
|
||||
/// Result of resolving a disc's VUK.
|
||||
#[derive(Debug)]
|
||||
pub struct ResolvedKeys {
|
||||
/// Disc hash (SHA1 of Unit_Key_RO.inf)
|
||||
pub disc_hash: [u8; 20],
|
||||
@@ -34,6 +33,23 @@ pub struct ResolvedKeys {
|
||||
pub key_source: u8,
|
||||
}
|
||||
|
||||
// Redacting `Debug`: `vuk` and `unit_keys` are raw key bytes, never printed.
|
||||
// `disc_hash` is the public per-disc identifier (SHA-1 of the .inf), not secret.
|
||||
// Guarded by `resolved_keys_debug_is_redacted`.
|
||||
impl std::fmt::Debug for ResolvedKeys {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("ResolvedKeys")
|
||||
.field("disc_hash", &self.disc_hash)
|
||||
.field("vuk", &self.vuk.map(|_| "<redacted>"))
|
||||
.field("unit_keys_len", &self.unit_keys.len())
|
||||
.field("title_cps_unit", &self.title_cps_unit)
|
||||
.field("version", &self.version)
|
||||
.field("bus_encryption", &self.bus_encryption)
|
||||
.field("key_source", &self.key_source)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
/// Inputs shared by every classical-path resolver. References only —
|
||||
/// callers retain ownership of all buffers.
|
||||
pub struct ResolveContext<'a> {
|
||||
@@ -363,26 +379,18 @@ fn resolve_keys_classical(ctx: &ResolveContext<'_>, version: AacsVersion) -> Opt
|
||||
// One AES-D + magic check per candidate (cheap). mk_dv is hoisted
|
||||
// out of the loop so the MKB is not re-walked per candidate.
|
||||
let mks = providers.media_keys();
|
||||
let mut mk_hits: Vec<[u8; 16]> = Vec::new();
|
||||
if let Some(mk_dv) = mkb_find_mk_dv(mkb) {
|
||||
for mk in &mks {
|
||||
let verifies = aes_ecb_decrypt(mk, &mk_dv)[..8]
|
||||
== [0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF];
|
||||
if verifies && !mk_hits.contains(mk) {
|
||||
mk_hits.push(*mk);
|
||||
if mk_hits.len() > 1 {
|
||||
break; // ambiguous — bail to avoid a wrong key
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if mk_hits.len() == 1 {
|
||||
let vuk = derive_vuk(&mk_hits[0], ctx.volume_id);
|
||||
let chosen_mk = mkb_find_mk_dv(mkb).and_then(|mk_dv| {
|
||||
unique_verifying_mk(&mks, |mk| {
|
||||
aes_ecb_decrypt(mk, &mk_dv)[..8] == MK_VERIFY_MAGIC
|
||||
})
|
||||
});
|
||||
if let Some(mk) = chosen_mk {
|
||||
let vuk = derive_vuk(&mk, ctx.volume_id);
|
||||
tracing::debug!(target: "freemkv::disc", phase = "resolve_keys_path2_5_hit", mk_pool = mks.len(), "media key from keydb MK-pool brute (km_verifies)");
|
||||
// Same class as path 3 (KEYDB MK → derived VUK).
|
||||
return Some(build(Some(vuk), derive_uks(&vuk), 3));
|
||||
}
|
||||
tracing::debug!(target: "freemkv::disc", phase = "resolve_keys_path2_5_miss", mk_pool = mks.len(), mk_hits = mk_hits.len(), "MK-pool brute: no unique verifying MK");
|
||||
tracing::debug!(target: "freemkv::disc", phase = "resolve_keys_path2_5_miss", mk_pool = mks.len(), "MK-pool brute: no unique verifying MK");
|
||||
} else {
|
||||
tracing::debug!(target: "freemkv::disc", phase = "resolve_keys_no_mkb", "no MKB; paths 1/2 skipped");
|
||||
}
|
||||
@@ -434,6 +442,42 @@ fn resolve_keys_classical(ctx: &ResolveContext<'_>, version: AacsVersion) -> Opt
|
||||
None
|
||||
}
|
||||
|
||||
/// First 8 bytes of the plaintext behind an MKB Verify Media Key record — the
|
||||
/// AACS "this is the right Km" sentinel (`0123456789ABCDEF`). A candidate MK
|
||||
/// verifies when AES-128-ECB-D(mk, mk_dv) starts with it.
|
||||
const MK_VERIFY_MAGIC: [u8; 8] = [0x01, 0x23, 0x45, 0x67, 0x89, 0xAB, 0xCD, 0xEF];
|
||||
|
||||
/// The MK-pool selection rule of path 2.5, split out of [`resolve_keys_v1`] so
|
||||
/// the ambiguity guard has a reachable test.
|
||||
///
|
||||
/// `verifies` is the MKB check — in production
|
||||
/// `AES-D(mk, mk_dv)[..8] == MK_VERIFY_MAGIC`. Returns a Media Key only when
|
||||
/// EXACTLY ONE DISTINCT candidate passes. Duplicates of the same key are one
|
||||
/// candidate (a pool aggregated across providers routinely repeats a key), but
|
||||
/// two DIFFERENT keys that both verify mean the pool cannot say which is this
|
||||
/// disc's Km: picking either derives a wrong VUK, and a wrong VUK decrypts to
|
||||
/// plausible-looking garbage rather than failing loudly. Bail and let the
|
||||
/// later hash/VID paths answer instead.
|
||||
///
|
||||
/// The predicate is a parameter rather than the inlined AES check because a
|
||||
/// genuine two-key multi-hit cannot be synthesised: it needs one ciphertext
|
||||
/// that decrypts under two distinct AES-128 keys to plaintexts sharing a
|
||||
/// 64-bit prefix — a 2^64 search. Injecting the verifier is the only way the
|
||||
/// ambiguity branch is reachable from a test at all.
|
||||
fn unique_verifying_mk(mks: &[[u8; 16]], verifies: impl Fn(&[u8; 16]) -> bool) -> Option<[u8; 16]> {
|
||||
let mut hits: Vec<[u8; 16]> = Vec::new();
|
||||
for mk in mks {
|
||||
if verifies(mk) && !hits.contains(mk) {
|
||||
hits.push(*mk);
|
||||
if hits.len() > 1 {
|
||||
// Ambiguous — bail rather than pick a Media Key.
|
||||
return None;
|
||||
}
|
||||
}
|
||||
}
|
||||
hits.first().copied()
|
||||
}
|
||||
|
||||
/// For path 5: cross-reference the disc's `Unit_Key_RO.inf` CPS-unit
|
||||
/// numbering against the KEYDB entry's pre-decrypted unit keys. Every
|
||||
/// CPS unit the disc declares must have a matching entry in KEYDB;
|
||||
@@ -467,6 +511,27 @@ mod tests {
|
||||
use super::super::types::*;
|
||||
use super::*;
|
||||
|
||||
/// `ResolvedKeys` carries the disc's VUK and unit keys raw; `Debug` must not
|
||||
/// leak them. Sentinel 213 (0xD5); non-secret fields are not 213.
|
||||
#[test]
|
||||
fn resolved_keys_debug_is_redacted() {
|
||||
let rk = ResolvedKeys {
|
||||
disc_hash: [0u8; 20],
|
||||
vuk: Some([0xD5; 16]),
|
||||
unit_keys: vec![(1, [0xD5; 16])],
|
||||
title_cps_unit: vec![0],
|
||||
version: AacsVersion::V21,
|
||||
bus_encryption: true,
|
||||
key_source: 1,
|
||||
};
|
||||
let dbg = format!("{rk:?}");
|
||||
assert!(!dbg.contains("213"), "ResolvedKeys leaked keys: {dbg}");
|
||||
assert!(
|
||||
dbg.contains("redacted"),
|
||||
"ResolvedKeys missing marker: {dbg}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Audit #5: the `major` / `from_major` mapping is load-bearing for the
|
||||
/// Unit_Key_RO stride, so pin it as a table. V10 ↔ BD; V20/V21 → UHD; any
|
||||
/// non-BD major selects the V20/V21 64-byte stride (V10 is the only 48-byte).
|
||||
@@ -1263,6 +1328,62 @@ mod tests {
|
||||
"VUK must derive from the verified Km + this disc's VID"
|
||||
);
|
||||
}
|
||||
|
||||
/// Path 2.5's ambiguity guard: when MORE THAN ONE DISTINCT pooled Media Key
|
||||
/// verifies against the MKB, the resolver must return no key at all rather
|
||||
/// than pick one. A wrong Km derives a wrong VUK, and a wrong VUK does not
|
||||
/// fail loudly — it decrypts the title to garbage that muxes and plays as a
|
||||
/// corrupt rip.
|
||||
///
|
||||
/// The real MKB check cannot be forced into a multi-hit: two distinct
|
||||
/// AES-128 keys decrypting one `mk_dv` to plaintexts that share the 64-bit
|
||||
/// verify magic is a 2^64 search, not a fixture. So the rule is tested
|
||||
/// through `unique_verifying_mk`, whose verifier is a parameter — the same
|
||||
/// function `resolve_keys_v1` calls, with the same pool semantics.
|
||||
#[test]
|
||||
fn mk_pool_ambiguity_bails_rather_than_picking_a_media_key() {
|
||||
let a = [0xAAu8; 16];
|
||||
let b = [0xBBu8; 16];
|
||||
let c = [0xCCu8; 16];
|
||||
|
||||
// One verifying candidate → that key.
|
||||
assert_eq!(
|
||||
unique_verifying_mk(&[a, b, c], |mk| *mk == b),
|
||||
Some(b),
|
||||
"a single verifying MK resolves"
|
||||
);
|
||||
|
||||
// The SAME key repeated across providers is one candidate, not an
|
||||
// ambiguity — the dedup (`!hits.contains`) must keep this resolvable.
|
||||
assert_eq!(
|
||||
unique_verifying_mk(&[b, b, b], |mk| *mk == b),
|
||||
Some(b),
|
||||
"duplicates of one key are not ambiguity"
|
||||
);
|
||||
|
||||
// TWO DISTINCT verifying candidates → bail, no key.
|
||||
assert_eq!(
|
||||
unique_verifying_mk(&[a, b], |mk| *mk == a || *mk == b),
|
||||
None,
|
||||
"two distinct verifying MKs must yield NO key, not the first one"
|
||||
);
|
||||
|
||||
// Ambiguity must still be detected when the second hit is last in the
|
||||
// pool, i.e. the scan may not stop at the first hit.
|
||||
assert_eq!(
|
||||
unique_verifying_mk(&[a, c, [0u8; 16], b], |mk| *mk == a || *mk == b),
|
||||
None,
|
||||
"a late second hit is still ambiguous"
|
||||
);
|
||||
|
||||
// Every candidate verifying is the degenerate ambiguous case.
|
||||
assert_eq!(unique_verifying_mk(&[a, b, c], |_| true), None);
|
||||
|
||||
// No candidate verifies → no key (and no panic on an empty pool).
|
||||
assert_eq!(unique_verifying_mk(&[a, b, c], |_| false), None);
|
||||
assert_eq!(unique_verifying_mk(&[], |_| true), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_content_cert_parse() {
|
||||
// AACS 1.0 cert, bus encryption OFF. Content-cert layout: flag in
|
||||
@@ -1785,6 +1906,77 @@ mod tests {
|
||||
assert_eq!(r.vuk, Some(derive_vuk(&mk, &vid)));
|
||||
}
|
||||
|
||||
/// `resolve_keys_v21` gates paths 1 and 3 on `has_vid`, and an all-zero
|
||||
/// Volume ID is the crate's "the VID was never read" sentinel — the SCSI
|
||||
/// handshake leaves the buffer zeroed when it does not run or fails.
|
||||
///
|
||||
/// Both directions matter and both fail silently:
|
||||
/// - treating the zero sentinel as a real VID runs path 3 and derives
|
||||
/// `Kvu = AES-G(Km, 0…0)`, a perfectly well-formed but WRONG VUK. It
|
||||
/// unwraps the title keys to garbage, and nothing downstream errors —
|
||||
/// the rip just decodes to noise.
|
||||
/// - treating a real VID as absent skips paths 1 and 3 entirely, so a
|
||||
/// disc that could have been resolved from its Media Key reports no key.
|
||||
///
|
||||
/// Asserted through the final VUK, not through the flag.
|
||||
#[test]
|
||||
fn resolve_keys_v21_treats_the_all_zero_volume_id_as_no_vid() {
|
||||
let uk_ro = minimal_unit_key_ro();
|
||||
let vid = [0x42u8; 16];
|
||||
let mk = [0x24u8; 16];
|
||||
// A VID-keyed entry carrying an MK and nothing else: no VUK and no unit
|
||||
// keys, so paths 4 and 5 cannot fire and ONLY the VID-gated path 3 can
|
||||
// produce a result.
|
||||
let entry = DiscEntry {
|
||||
disc_hash: "not-this-disc".to_string(),
|
||||
title: "sibling".to_string(),
|
||||
media_key: Some(mk),
|
||||
disc_id: Some(vid),
|
||||
vuk: None,
|
||||
unit_keys: Vec::new(),
|
||||
};
|
||||
let keydb = SuppliedKey {
|
||||
device_keys: Vec::new(),
|
||||
processing_keys: Vec::new(),
|
||||
media_keys: Vec::new(),
|
||||
disc_entry: Some(entry),
|
||||
};
|
||||
let providers: &[&dyn super::super::provider::KeyProvider] = &[&keydb];
|
||||
|
||||
// A real VID → path 3 fires and the VUK derives from Km + THIS VID.
|
||||
let with_vid = ResolveContext {
|
||||
unit_key_ro: &uk_ro,
|
||||
content_cert: None,
|
||||
volume_id: &vid,
|
||||
providers,
|
||||
mkb: None,
|
||||
};
|
||||
let r = resolve_keys_v21(&with_vid).expect("a real VID must reach path 3");
|
||||
assert_eq!(r.key_source, 3);
|
||||
assert_eq!(
|
||||
r.vuk,
|
||||
Some(derive_vuk(&mk, &vid)),
|
||||
"VUK must derive from the Media Key and the disc's own VID"
|
||||
);
|
||||
|
||||
// The all-zero sentinel → paths 1 and 3 are skipped entirely; with no
|
||||
// VUK and no unit keys on the entry, nothing resolves.
|
||||
let no_vid = ResolveContext {
|
||||
unit_key_ro: &uk_ro,
|
||||
content_cert: None,
|
||||
volume_id: &[0u8; 16],
|
||||
providers,
|
||||
mkb: None,
|
||||
};
|
||||
let got = resolve_keys_v21(&no_vid);
|
||||
assert!(
|
||||
got.is_none(),
|
||||
"a zero VID must not be used to derive a VUK; got key_source {:?} vuk {:?}",
|
||||
got.as_ref().map(|r| r.key_source),
|
||||
got.as_ref().map(|r| r.vuk.is_some())
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn resolve_keys_returns_none_when_no_provider_has_anything() {
|
||||
// Empty provider array + VID present + no MKB → all paths miss → None.
|
||||
|
||||
+41
-25
@@ -25,7 +25,7 @@
|
||||
//! u32 start_spn | u32 end_spn (source-packet numbers, inclusive)
|
||||
//! ```
|
||||
//! `index` is the 1..32 forensic index tag, NOT a sequential segment id: measured
|
||||
//! on a retail 2.1 disc (Zombieland) it cycles 1,2,…,32,1,2,… across records in
|
||||
//! on a retail 2.1 disc it cycles 1,2,…,32,1,2,… across records in
|
||||
//! file order — 24 full cycles of 32 plus a final partial cycle of 24 = 792
|
||||
//! records. Source-packet numbers are the 192-byte BDAV packet index: byte offset
|
||||
//! = `spn * 192`. Each segment is ~2560 packets (~480 KB) = 80 aligned units,
|
||||
@@ -41,21 +41,6 @@ pub const SEGMENT_RECORD_LEN: usize = 16;
|
||||
/// Bytes per BDAV source packet (188-byte TS + 4-byte arrival-time header).
|
||||
pub const SOURCE_PACKET_LEN: u64 = 192;
|
||||
|
||||
/// Whether a 2.1 (FMTS) disc may rip WITHOUT the forensic index keys.
|
||||
///
|
||||
/// `true` (today): the forensic segments are skipped as expected loss
|
||||
/// and the bulk of the title decodes with the unit key, so a 2.1 disc rips
|
||||
/// mostly-complete. A unit key (VUK) is still required, exactly as for any AACS
|
||||
/// disc. `false`: the absence of a segment-key source is a hard, UPFRONT failure
|
||||
/// ([`Error::FmtsKeyMissing`]) — the same policy as a missing unit key, so a
|
||||
/// forensic-holed rip is refused rather than produced. No segment-key source
|
||||
/// exists yet, so `true` is the only value under which a 2.1 disc rips at all;
|
||||
/// flip to `false` once segment keys can be sourced and a partial rip should be
|
||||
/// refused. Hardcoded on purpose — not a user setting.
|
||||
///
|
||||
/// [`Error::FmtsKeyMissing`]: crate::error::Error::FmtsKeyMissing
|
||||
pub const BYPASS_FMTS_KEY: bool = false;
|
||||
|
||||
/// One forensic segment: the inclusive source-packet range it occupies in the
|
||||
/// FMTS clip.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
@@ -208,6 +193,11 @@ pub fn fmts_key_ranges(
|
||||
) -> Vec<(u32, u32, usize)> {
|
||||
let mut ranges = Vec::new();
|
||||
for s in segments {
|
||||
// SPNs are untrusted (from IndividualSegment.tbl); an inverted record
|
||||
// (start_spn > end_spn) would underflow `end_byte - 1 - start_byte` below.
|
||||
if s.start_spn > s.end_spn {
|
||||
continue;
|
||||
}
|
||||
let start_byte = s.start_spn as u64 * SOURCE_PACKET_LEN;
|
||||
let end_byte = (s.end_spn as u64 + 1) * SOURCE_PACKET_LEN; // exclusive
|
||||
// A segment is unit-aligned and contiguous in clip bytes; map its first
|
||||
@@ -284,18 +274,44 @@ mod tests {
|
||||
// sectors 937..=946 → LBA 1937..1947, key index 7.
|
||||
assert_eq!(ranges[1], (1937, 1947, 7));
|
||||
|
||||
// The ranges drive an AacsKeyMap with the Unit Key (index 0) as default.
|
||||
let map = crate::decrypt::AacsKeyMap::from_ranges(ranges, 0);
|
||||
assert_eq!(map.key_idx_for(500), 0, "outside any segment → Unit Key");
|
||||
assert_eq!(map.key_idx_for(1012), 5, "inside index-5 segment → key 5");
|
||||
assert_eq!(map.key_idx_for(1940), 7, "inside index-7 segment → key 7");
|
||||
// The ranges drive a positive AacsKeyMap: an LBA in no range has no key.
|
||||
let map = crate::decrypt::AacsKeyMap::from_ranges(ranges);
|
||||
assert_eq!(map.key_idx_for(500), None, "outside any segment → no key");
|
||||
assert_eq!(
|
||||
map.key_idx_for(1012),
|
||||
Some(5),
|
||||
"inside index-5 segment → key 5"
|
||||
);
|
||||
assert_eq!(
|
||||
map.key_idx_for(1940),
|
||||
Some(7),
|
||||
"inside index-7 segment → key 7"
|
||||
);
|
||||
assert_eq!(
|
||||
map.key_idx_for(1019),
|
||||
0,
|
||||
"segment end is exclusive → Unit Key"
|
||||
None,
|
||||
"segment end is exclusive → no key"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fmts_key_ranges_skips_inverted_segment_without_underflow() {
|
||||
use crate::disc::Extent;
|
||||
let extents = vec![Extent {
|
||||
start_lba: 1000,
|
||||
sector_count: 1_000_000,
|
||||
}];
|
||||
// start_spn == end_spn + 1: `end_byte - 1 - start_byte` would underflow.
|
||||
// The record must be skipped rather than panic (debug) / wrap (release).
|
||||
let segs = vec![Segment {
|
||||
index: 5,
|
||||
start_spn: 200,
|
||||
end_spn: 199,
|
||||
}];
|
||||
let ranges = fmts_key_ranges(&segs, &extents, &|v| v as usize);
|
||||
assert!(ranges.is_empty(), "inverted segment yields no range");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn clip_byte_to_lba_walks_extents() {
|
||||
use crate::disc::Extent;
|
||||
@@ -318,7 +334,7 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn parses_real_disc_layout() {
|
||||
// First three records observed on retail 2.1 (Zombieland): the variant
|
||||
// First three records observed on retail 2.1: the variant
|
||||
// field counts 1,2,3,… (it wraps at 32 further into the table — see
|
||||
// `index_field_cycles_one_to_thirty_two`), segments are 2560 packets.
|
||||
let tbl = build_tbl(&[
|
||||
@@ -380,7 +396,7 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn index_field_cycles_one_to_thirty_two() {
|
||||
// Reality on Zombieland: field@4 is the index, cycling 1..=32 in file
|
||||
// Reality on a retail 2.1 disc: field@4 is the index, cycling 1..=32 in file
|
||||
// order (NOT a sequential segment id). Reproduce one-and-a-bit cycles.
|
||||
let mut recs = Vec::new();
|
||||
let mut spn = 1000u32;
|
||||
|
||||
+234
-8
@@ -6,7 +6,7 @@
|
||||
//! owns only the crypto and these value types that flow through it.
|
||||
|
||||
/// A device key for MKB subset-difference tree processing.
|
||||
#[derive(Debug, Clone)]
|
||||
#[derive(Clone)]
|
||||
pub struct DeviceKey {
|
||||
pub key: [u8; 16],
|
||||
pub node: u16,
|
||||
@@ -15,7 +15,7 @@ pub struct DeviceKey {
|
||||
}
|
||||
|
||||
/// Host certificate + private key for AACS SCSI authentication.
|
||||
#[derive(Debug, Clone)]
|
||||
#[derive(Clone)]
|
||||
pub struct HostCert {
|
||||
/// AACS 1.0: 20 bytes. AACS 2.0: 32 bytes.
|
||||
pub private_key: [u8; 20],
|
||||
@@ -28,22 +28,22 @@ pub struct HostCert {
|
||||
}
|
||||
|
||||
/// Volume ID (16 bytes) — read from the disc via the SCSI handshake / OEM path.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Vid(pub [u8; 16]);
|
||||
|
||||
/// Media Key (Km, 16 bytes) — the MKB-scoped key derived from device keys.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
pub struct MediaKey(pub [u8; 16]);
|
||||
|
||||
/// Volume Unique Key (VUK / Kvu, 16 bytes) — derived from `MediaKey` + `Vid`,
|
||||
/// decrypts the per-disc encrypted title keys in `Unit_Key_RO.inf`.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
pub struct Vuk(pub [u8; 16]);
|
||||
|
||||
/// Processing Key (Kp, 16 bytes) — an MKB Subset-Difference key that yields the
|
||||
/// Media Key. A leaked/precomputed PK in the keydb, or the intermediate PK a
|
||||
/// device-key walk derives at its matching SD node.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
pub struct ProcessingKey(pub [u8; 16]);
|
||||
|
||||
/// One decrypted per-CPS-unit AACS title key.
|
||||
@@ -53,7 +53,7 @@ pub struct ProcessingKey(pub [u8; 16]);
|
||||
/// area). The CPS-unit *number* association is a higher-level concern owned by
|
||||
/// [`super::inf::parse_unit_key_ro`], which pairs each positional key with its
|
||||
/// declared CPS unit; this primitive only does the AES, so it surfaces position.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
#[derive(Clone, Copy, PartialEq, Eq)]
|
||||
pub struct UnitKey {
|
||||
pub idx: u32,
|
||||
pub key: [u8; 16],
|
||||
@@ -95,7 +95,7 @@ impl UnitKey {
|
||||
}
|
||||
|
||||
/// A per-disc entry from the key database.
|
||||
#[derive(Debug, Clone)]
|
||||
#[derive(Clone)]
|
||||
pub struct DiscEntry {
|
||||
/// Disc hash (20 bytes, hex)
|
||||
pub disc_hash: String,
|
||||
@@ -110,3 +110,229 @@ pub struct DiscEntry {
|
||||
/// Unit keys (title keys) indexed by CPS unit number
|
||||
pub unit_keys: Vec<(u32, [u8; 16])>,
|
||||
}
|
||||
|
||||
// ── Redacting `Debug` impls ──────────────────────────────────────────────────
|
||||
//
|
||||
// Every type above carries AACS secret material (device keys, host PRIVATE keys,
|
||||
// media/volume/processing/unit keys). `#[derive(Debug)]` would print those bytes
|
||||
// verbatim, so a stray `debug!("{:?}", …)` or a panic message would leak the
|
||||
// keys. These hand-written impls print only NON-secret shape (presence, lengths,
|
||||
// tree coordinates, indices) — never key bytes. `decrypt::DecryptKeys` follows
|
||||
// the same policy by omitting `Debug` entirely; here we keep `Debug` because
|
||||
// these are `PartialEq`/`Eq` value types used in `assert_eq!` and nested inside
|
||||
// other `#[derive(Debug)]` structs, so the trait must exist — just not leak.
|
||||
// Guarded by `redaction_tests` below.
|
||||
|
||||
impl std::fmt::Debug for DeviceKey {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("DeviceKey")
|
||||
.field("key", &"<redacted>")
|
||||
.field("node", &self.node)
|
||||
.field("uv", &self.uv)
|
||||
.field("u_mask_shift", &self.u_mask_shift)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for HostCert {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("HostCert")
|
||||
.field("private_key", &"<redacted>")
|
||||
.field("certificate_len", &self.certificate.len())
|
||||
.field("private_key_v2", &self.private_key_v2.map(|_| "<redacted>"))
|
||||
.field(
|
||||
"certificate_v2_len",
|
||||
&self.certificate_v2.as_ref().map(|c| c.len()),
|
||||
)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for Vid {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str("Vid(<redacted>)")
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for MediaKey {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str("MediaKey(<redacted>)")
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for Vuk {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str("Vuk(<redacted>)")
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for ProcessingKey {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.write_str("ProcessingKey(<redacted>)")
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for UnitKey {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("UnitKey")
|
||||
.field("idx", &self.idx)
|
||||
.field("key", &"<redacted>")
|
||||
.field("index_number", &self.index_number)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for DiscEntry {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("DiscEntry")
|
||||
.field("disc_hash", &self.disc_hash)
|
||||
.field("title", &self.title)
|
||||
.field("media_key", &self.media_key.map(|_| "<redacted>"))
|
||||
.field("disc_id", &self.disc_id.map(|_| "<redacted>"))
|
||||
.field("vuk", &self.vuk.map(|_| "<redacted>"))
|
||||
.field("unit_keys_len", &self.unit_keys.len())
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod unit_key_tests {
|
||||
use super::*;
|
||||
|
||||
/// `is_default_index` is the public predicate that separates ordinary
|
||||
/// (index-0) content keys from FMTS forensic index keys ([`UnitKey`] docs;
|
||||
/// AACS 2.1 `IndividualSegment.tbl` tagging). A body answering `true` for
|
||||
/// everything would present a forensic index key as an ordinary content
|
||||
/// key — the caller would decrypt the bulk of the title with a key that
|
||||
/// only opens 1/32nd of it; answering `false` for everything would hide
|
||||
/// every ordinary key.
|
||||
///
|
||||
/// Pinned against the two NAMED constructors, which are the contract:
|
||||
/// [`UnitKey::new`] builds the ordinary key, [`UnitKey::forensic`] builds
|
||||
/// an index key for `1..=32`.
|
||||
#[test]
|
||||
fn is_default_index_separates_the_two_constructors() {
|
||||
let ordinary = UnitKey::new(0, [0xAA; 16]);
|
||||
assert!(
|
||||
ordinary.is_default_index(),
|
||||
"UnitKey::new builds the ordinary (index-0) key"
|
||||
);
|
||||
|
||||
// Every forensic index the spec allows must be reported as NOT default.
|
||||
for n in 1u8..=32 {
|
||||
let k = UnitKey::forensic(0, [0xAA; 16], n);
|
||||
assert!(
|
||||
!k.is_default_index(),
|
||||
"UnitKey::forensic({n}) is an index key, not the default key"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The predicate must agree with the one consumer of `index_number` in the
|
||||
/// crate: [`crate::aacs::index_select::resolve_disc_index`] resolves the
|
||||
/// disc's forensic index from exactly the keys that are NOT default. If
|
||||
/// the two disagree, a disc resolves an index whose key the rest of the
|
||||
/// pipeline treats as ordinary (or vice versa).
|
||||
#[test]
|
||||
fn is_default_index_agrees_with_the_forensic_index_resolver() {
|
||||
use crate::aacs::index_select::resolve_disc_index;
|
||||
|
||||
let keys = [
|
||||
UnitKey::new(0, [0x11; 16]),
|
||||
UnitKey::forensic(1, [0x22; 16], 7),
|
||||
];
|
||||
assert_eq!(
|
||||
resolve_disc_index(&keys),
|
||||
Some(7),
|
||||
"sanity: the resolver picks the forensic key's index"
|
||||
);
|
||||
|
||||
let non_default: Vec<u8> = keys
|
||||
.iter()
|
||||
.filter(|k| !k.is_default_index())
|
||||
.map(|k| k.index_number)
|
||||
.collect();
|
||||
assert_eq!(
|
||||
non_default,
|
||||
vec![7],
|
||||
"exactly the key the resolver picked must be non-default"
|
||||
);
|
||||
|
||||
// An all-ordinary key set resolves no index, and every key must report
|
||||
// itself default.
|
||||
let plain = [UnitKey::new(0, [0x11; 16]), UnitKey::new(1, [0x22; 16])];
|
||||
assert_eq!(resolve_disc_index(&plain), None);
|
||||
assert!(plain.iter().all(|k| k.is_default_index()));
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod redaction_tests {
|
||||
use super::*;
|
||||
|
||||
// Sentinel key byte 0xD5 = decimal 213. A derived `Debug` prints `[u8;N]`
|
||||
// as decimal, so a leaked key surfaces the substring "213"; the redacting
|
||||
// impls must not. No non-secret field below is 213, so "213" appearing means
|
||||
// key bytes leaked. Each type must also carry a "redacted" marker (or omit
|
||||
// the secret entirely) so re-adding `#[derive(Debug)]` fails this test.
|
||||
const S: u8 = 0xD5;
|
||||
|
||||
fn assert_redacted(what: &str, dbg: &str) {
|
||||
assert!(
|
||||
!dbg.contains("213"),
|
||||
"{what}: Debug leaked key bytes (found decimal 213): {dbg}"
|
||||
);
|
||||
assert!(
|
||||
dbg.contains("redacted"),
|
||||
"{what}: Debug missing redaction marker: {dbg}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn device_key_debug_is_redacted() {
|
||||
let d = DeviceKey {
|
||||
key: [S; 16],
|
||||
node: 1,
|
||||
uv: 2,
|
||||
u_mask_shift: 3,
|
||||
};
|
||||
assert_redacted("DeviceKey", &format!("{d:?}"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn host_cert_debug_is_redacted() {
|
||||
let h = HostCert {
|
||||
private_key: [S; 20],
|
||||
certificate: vec![0u8; 92],
|
||||
private_key_v2: Some([S; 32]),
|
||||
certificate_v2: None,
|
||||
};
|
||||
assert_redacted("HostCert", &format!("{h:?}"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn newtype_keys_debug_is_redacted() {
|
||||
assert_redacted("Vid", &format!("{:?}", Vid([S; 16])));
|
||||
assert_redacted("MediaKey", &format!("{:?}", MediaKey([S; 16])));
|
||||
assert_redacted("Vuk", &format!("{:?}", Vuk([S; 16])));
|
||||
assert_redacted("ProcessingKey", &format!("{:?}", ProcessingKey([S; 16])));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unit_key_debug_is_redacted() {
|
||||
assert_redacted("UnitKey", &format!("{:?}", UnitKey::new(0, [S; 16])));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn disc_entry_debug_is_redacted() {
|
||||
let e = DiscEntry {
|
||||
disc_hash: "0xAA".into(),
|
||||
title: "T".into(),
|
||||
media_key: Some([S; 16]),
|
||||
disc_id: Some([S; 16]),
|
||||
vuk: Some([S; 16]),
|
||||
unit_keys: vec![(1, [S; 16])],
|
||||
};
|
||||
assert_redacted("DiscEntry", &format!("{e:?}"));
|
||||
}
|
||||
}
|
||||
|
||||
+973
-9
File diff suppressed because it is too large
Load Diff
+101
-840
File diff suppressed because it is too large
Load Diff
+9
-1
@@ -99,7 +99,11 @@ pub mod coding_type {
|
||||
|
||||
/// Secondary Dolby Digital Plus audio (BD-ROM convention).
|
||||
pub const AC3_PLUS_SECONDARY: u8 = 0xA1;
|
||||
/// Secondary DTS-HD audio (lossless MA, not lossy HR) (BD-ROM convention).
|
||||
/// Secondary DTS-HD audio — DTS Express / DTS-HD LBR, a LOSSY low-bitrate
|
||||
/// stream for picture-in-picture and BD-J mixing (BD-ROM Part 3
|
||||
/// `stream_coding_type` table). The lossless primary is [`DTS_HD_MA`]
|
||||
/// (0x86); this code is its lossy secondary counterpart, parallel to
|
||||
/// [`AC3_PLUS_SECONDARY`] (0xA1) on the Dolby side.
|
||||
pub const DTS_HD_SECONDARY: u8 = 0xA2;
|
||||
}
|
||||
|
||||
@@ -110,6 +114,10 @@ pub mod coding_type {
|
||||
pub mod pes_stream_id {
|
||||
/// Video stream (`110x xxxx`; freemkv emits the base id `0xE0`).
|
||||
pub const VIDEO: u8 = 0xE0;
|
||||
/// system_header start code — the MPEG-PS `00 00 01 BB` structural header
|
||||
/// (rate/bound bounds), never an elementary stream. On a DVD NAV pack it
|
||||
/// follows the pack header, so it lands at sector offset 0x11.
|
||||
pub const SYSTEM_HEADER: u8 = 0xBB;
|
||||
/// private_stream_1 — AC-3 / DTS / LPCM / PGS subtitle payloads.
|
||||
pub const PRIVATE_STREAM_1: u8 = 0xBD;
|
||||
/// padding_stream — stuffing bytes only, no payload to demux.
|
||||
|
||||
@@ -443,6 +443,49 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
/// The length guard is a FLOOR, not a ceiling: `descramble_sector` is a
|
||||
/// no-op below one sector, and processes the FIRST sector of anything at
|
||||
/// least that long (the loop is `.take(2048)`). `css::descramble_sector` is
|
||||
/// a public entry taking `&mut [u8]` of any length, so a caller handing it a
|
||||
/// multi-sector buffer must get its first sector descrambled — a guard that
|
||||
/// rejected over-long buffers would hand that caller its ciphertext back
|
||||
/// unchanged, with the scramble flag cleared as if it had worked.
|
||||
#[test]
|
||||
fn descramble_processes_the_first_sector_of_an_over_long_buffer() {
|
||||
let title_key = [0x42, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0xDE, 0xAD, 0xBE, 0xEF, 0x42];
|
||||
|
||||
// Two sectors' worth of buffer; only the first is a sector.
|
||||
let mut buf = vec![0xAAu8; 4096];
|
||||
buf[0x14] = 0x30;
|
||||
buf[0x54..0x59].copy_from_slice(&seed);
|
||||
let original = buf.clone();
|
||||
|
||||
descramble_sector(&title_key, &mut buf);
|
||||
|
||||
assert_ne!(
|
||||
&buf[0x80..0x800],
|
||||
&original[0x80..0x800],
|
||||
"the first sector's body must be descrambled"
|
||||
);
|
||||
assert_eq!(buf[0x14] & 0x30, 0x00, "and its scramble flag cleared");
|
||||
assert_eq!(
|
||||
&buf[2048..4096],
|
||||
&original[2048..4096],
|
||||
"bytes past the first sector must be left untouched"
|
||||
);
|
||||
|
||||
// The result must equal what a caller gets by passing exactly one
|
||||
// sector — the same transform, not a length-dependent one.
|
||||
let mut one = original[..2048].to_vec();
|
||||
descramble_sector(&title_key, &mut one);
|
||||
assert_eq!(
|
||||
&buf[..2048],
|
||||
&one[..],
|
||||
"the first sector must descramble identically either way"
|
||||
);
|
||||
}
|
||||
|
||||
/// Descramble is keyed by `title_key XOR seed`: two different title keys
|
||||
/// produce two different bodies for the same scrambled input. A cipher that
|
||||
/// ignored the title key (or mixed it in wrongly) would yield identical
|
||||
|
||||
+982
-91
File diff suppressed because it is too large
Load Diff
@@ -450,6 +450,64 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// `descramble_matches` is the ONLY gate between the LFSR search and a key
|
||||
/// handed back to the caller: both [`recover_title_key`] and the crib-driven
|
||||
/// `crack_title_key_inner` return a candidate only if this says the key
|
||||
/// really descrambles the sector to the known plaintext. A body that always
|
||||
/// answered `true` would let the first spurious LFSR-seed match through as
|
||||
/// the title key — the ripper would then descramble the whole title with a
|
||||
/// key that opens nothing, producing garbage rather than a "no key" error.
|
||||
///
|
||||
/// Pinned both directions: the genuine key is accepted, and EVERY key one
|
||||
/// bit away from it is rejected. The one-bit neighbours are the strongest
|
||||
/// form of wrong key — a gate that only rejects wildly different keys would
|
||||
/// still pass a near-miss out of the 2^16 seed search.
|
||||
#[test]
|
||||
fn descramble_matches_accepts_only_the_key_the_sector_was_scrambled_with() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
let (sector, _body) = synth_sector(&title_key, &seed, &PES);
|
||||
|
||||
assert!(
|
||||
descramble_matches(§or, &title_key, &PES),
|
||||
"the key the sector was scrambled with must be accepted"
|
||||
);
|
||||
|
||||
for byte in 0..5usize {
|
||||
for bit in 0..8u32 {
|
||||
let mut wrong = title_key;
|
||||
wrong[byte] ^= 1u8 << bit;
|
||||
assert!(
|
||||
!descramble_matches(§or, &wrong, &PES),
|
||||
"key differing only in byte {byte} bit {bit} must be rejected"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The gate is applied to a COPY: verifying a candidate must not modify the
|
||||
/// caller's sector. `recover_title_key` runs the gate and then hands the
|
||||
/// sector on to be descrambled for real — if verification descrambled in
|
||||
/// place, that second descramble would run over already-transformed bytes
|
||||
/// (and, worse, a rejected candidate would leave the sector corrupted).
|
||||
#[test]
|
||||
fn descramble_matches_does_not_disturb_the_caller_s_sector() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
let (sector, _body) = synth_sector(&title_key, &seed, &PES);
|
||||
let before = sector.clone();
|
||||
|
||||
assert!(descramble_matches(§or, &title_key, &PES));
|
||||
let mut wrong = title_key;
|
||||
wrong[0] ^= 0x01;
|
||||
assert!(!descramble_matches(§or, &wrong, &PES));
|
||||
|
||||
assert_eq!(
|
||||
sector, before,
|
||||
"verification must leave the sector byte-for-byte unchanged"
|
||||
);
|
||||
}
|
||||
|
||||
/// MANDATORY (Task C.1): the crib-based entry point crack_title_key —
|
||||
/// no plaintext supplied — recovers a round-tripping key when the
|
||||
/// cleartext ends in a periodic run that continues into 0x80.
|
||||
@@ -571,4 +629,512 @@ mod tests {
|
||||
let _ = crack_title_key(§or);
|
||||
}
|
||||
}
|
||||
|
||||
// ── entry-point guards on caller- and disc-supplied lengths ────────────
|
||||
|
||||
/// A sector buffer that ENDS inside the encrypted region must be refused,
|
||||
/// not sliced.
|
||||
///
|
||||
/// `recover_title_key` slices `sector[0x80..0x8A]` unconditionally after its
|
||||
/// length guard. The existing short-sector test uses `SECTOR_BYTES - 1`,
|
||||
/// which is still long enough for that slice to succeed — so the guard was
|
||||
/// never the thing producing the `None`, and dropping it (or weakening the
|
||||
/// `||` to `&&`, which a full-length crib satisfies) changed nothing
|
||||
/// observable. On a real short read this is an out-of-bounds panic on the
|
||||
/// rip thread.
|
||||
#[test]
|
||||
fn recover_rejects_a_sector_that_ends_inside_the_encrypted_region() {
|
||||
for len in [0x81usize, 0x85, 0x89] {
|
||||
let mut sector = vec![0x11u8; len];
|
||||
sector[FLAG_BYTE] = 0x30; // scrambled, so no other guard fires first
|
||||
assert!(
|
||||
recover_title_key(§or, &PES).is_none(),
|
||||
"a {len}-byte buffer cannot supply ten ciphertext bytes at 0x80"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// A buffer LONGER than one sector is still one sector: both entry points
|
||||
/// read the first `SECTOR_BYTES` and must recover the key from it.
|
||||
///
|
||||
/// Callers read DVD data in multi-sector blocks, so an over-long slice is
|
||||
/// the normal case, not an exotic one. A length guard that rejected it
|
||||
/// (`len > SECTOR_BYTES` instead of `<`) would make every block-read caller
|
||||
/// silently unable to crack anything.
|
||||
#[test]
|
||||
fn a_buffer_longer_than_one_sector_still_yields_its_key() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
|
||||
let (sector, _) = synth_sector(&title_key, &seed, &PES);
|
||||
let mut padded = sector.clone();
|
||||
padded.extend_from_slice(&[0xA7u8; 512]);
|
||||
assert_eq!(
|
||||
recover_title_key(&padded, &PES),
|
||||
Some(title_key),
|
||||
"a two-and-a-bit-sector buffer must still recover the first sector's key"
|
||||
);
|
||||
|
||||
let (periodic, _) = synth_periodic_sector(&title_key, &seed, 5);
|
||||
let mut padded = periodic.clone();
|
||||
padded.extend_from_slice(&[0xA7u8; 512]);
|
||||
assert_eq!(
|
||||
crack_title_key(&padded),
|
||||
crack_title_key(&periodic),
|
||||
"padding past the sector must not change the crack result"
|
||||
);
|
||||
assert!(crack_title_key(&padded).is_some());
|
||||
}
|
||||
|
||||
/// `recover_title_key` accepts MORE than ten bytes of known plaintext, and
|
||||
/// uses all of it: the extra bytes tighten the `descramble_matches` gate.
|
||||
/// The ten-byte figure is a MINIMUM (the cipher is iterated ten times), not
|
||||
/// an exact requirement — a guard reading it as an upper bound would reject
|
||||
/// every caller that knows a longer crib.
|
||||
#[test]
|
||||
fn recover_accepts_more_than_ten_bytes_of_known_plaintext() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
let long_plain: Vec<u8> = (0..64u8)
|
||||
.map(|k| k.wrapping_mul(37).wrapping_add(5))
|
||||
.collect();
|
||||
let (sector, _) = synth_sector(&title_key, &seed, &long_plain);
|
||||
|
||||
assert_eq!(
|
||||
recover_title_key(§or, &long_plain),
|
||||
Some(title_key),
|
||||
"64 bytes of known plaintext must be accepted, not rejected as \
|
||||
'more than ten'"
|
||||
);
|
||||
}
|
||||
|
||||
/// The scramble-flag gate on a sector whose BODY really is ciphertext.
|
||||
///
|
||||
/// Both entry points refuse a sector with `sector[0x14] & 0x30 == 0`: an
|
||||
/// unscrambled sector has no title key to recover, and its bytes at 0x80
|
||||
/// are already plaintext. Every prior test of this gate used an all-zero or
|
||||
/// all-`0x11` sector, where the recovery would have found nothing anyway —
|
||||
/// so widening the mask test (`&` to `|`, which makes it true for EVERY
|
||||
/// flag byte) produced the same `None` and went unseen.
|
||||
///
|
||||
/// Here the sector is genuinely scrambled and its key IS recoverable; only
|
||||
/// the cleared flag stands in the way. If the gate stops working, both
|
||||
/// functions start returning keys for sectors the disc says are in the
|
||||
/// clear.
|
||||
#[test]
|
||||
fn a_recoverable_sector_with_the_scramble_bits_cleared_is_still_refused() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
|
||||
let (mut sector, _) = synth_sector(&title_key, &seed, &PES);
|
||||
assert_eq!(
|
||||
recover_title_key(§or, &PES),
|
||||
Some(title_key),
|
||||
"fixture check: with the flag set this sector's key IS recoverable"
|
||||
);
|
||||
sector[FLAG_BYTE] = 0x00;
|
||||
assert_eq!(
|
||||
recover_title_key(§or, &PES),
|
||||
None,
|
||||
"scramble bits clear → no title key, even though one could be found"
|
||||
);
|
||||
|
||||
let (mut periodic, _) = synth_periodic_sector(&title_key, &seed, 5);
|
||||
assert!(
|
||||
crack_title_key(&periodic).is_some(),
|
||||
"fixture check: with the flag set this sector cracks"
|
||||
);
|
||||
assert!(
|
||||
attack_crib(&periodic).is_some(),
|
||||
"fixture check: with the flag set this sector has a usable crib"
|
||||
);
|
||||
periodic[FLAG_BYTE] = 0x00;
|
||||
assert_eq!(
|
||||
crack_title_key(&periodic),
|
||||
None,
|
||||
"scramble bits clear → no crack, even though one would succeed"
|
||||
);
|
||||
// `attack_crib` carries its own copy of the same gate, and it is the one
|
||||
// that actually stops the crack (`crack_title_key`'s is defensive
|
||||
// duplication). The crib doubles as the decrypt path's cached-key
|
||||
// oracle, so a widened mask there would hand that path a "predicted
|
||||
// plaintext" for sectors that were never scrambled.
|
||||
assert_eq!(
|
||||
attack_crib(&periodic),
|
||||
None,
|
||||
"an unscrambled sector has no predicted plaintext to offer"
|
||||
);
|
||||
}
|
||||
|
||||
// ── descramble_matches: the verification gate's own mechanics ──────────
|
||||
|
||||
/// The gate must verify a candidate against the sector's CIPHERTEXT
|
||||
/// regardless of what the sector's own flag byte says.
|
||||
///
|
||||
/// `descramble_matches` forces `0x10` on its copy precisely because
|
||||
/// [`super::lfsr::descramble_sector`] is a no-op when the scramble bits are
|
||||
/// clear — without that, verifying a scrambled-but-unflagged sector
|
||||
/// compares raw ciphertext against the crib, and every candidate key is
|
||||
/// rejected. Nothing exercised it: every fixture already had the flag set,
|
||||
/// where forcing the bit is a no-op.
|
||||
#[test]
|
||||
fn descramble_matches_forces_the_scramble_flag_on_its_own_copy() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
let (mut sector, _) = synth_sector(&title_key, &seed, &PES);
|
||||
sector[FLAG_BYTE] = 0x00;
|
||||
|
||||
assert!(
|
||||
descramble_matches(§or, &title_key, &PES),
|
||||
"the body is ciphertext and the key is right — the gate must \
|
||||
descramble it even though the flag byte says otherwise"
|
||||
);
|
||||
let mut wrong = title_key;
|
||||
wrong[0] ^= 0x01;
|
||||
assert!(!descramble_matches(§or, &wrong, &PES));
|
||||
}
|
||||
|
||||
/// The gate compares the WHOLE supplied plaintext, clamped to the encrypted
|
||||
/// region.
|
||||
///
|
||||
/// Two properties in one, because they are the two halves of
|
||||
/// `plain.len().min(SECTOR_BYTES - ENCRYPTED_START)`:
|
||||
///
|
||||
/// - it must compare beyond the first sixteen bytes, or a key that opens
|
||||
/// only the head of the crib is accepted; and
|
||||
/// - it must never compare past the end of the sector — a caller that
|
||||
/// knows more plaintext than the 1920-byte encrypted region holds
|
||||
/// otherwise indexes off the end of the buffer and panics.
|
||||
#[test]
|
||||
fn descramble_matches_compares_all_of_the_plaintext_and_no_more_than_the_sector() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
let body: Vec<u8> = (0..64u8)
|
||||
.map(|k| k.wrapping_mul(29).wrapping_add(3))
|
||||
.collect();
|
||||
let (sector, _) = synth_sector(&title_key, &seed, &body);
|
||||
|
||||
assert!(descramble_matches(§or, &title_key, &body));
|
||||
|
||||
// A crib agreeing for the first 16 bytes and diverging after must be
|
||||
// rejected: the comparison window is the crib's length, not a fixed
|
||||
// prefix.
|
||||
let mut tail_wrong = body.clone();
|
||||
tail_wrong[40] ^= 0xFF;
|
||||
assert!(
|
||||
!descramble_matches(§or, &title_key, &tail_wrong),
|
||||
"a crib that diverges at byte 40 must not match"
|
||||
);
|
||||
assert_eq!(
|
||||
tail_wrong[..16],
|
||||
body[..16],
|
||||
"fixture check: the first 16 bytes are identical, so only a \
|
||||
comparison that runs past them can tell these apart"
|
||||
);
|
||||
|
||||
// A crib LONGER than the encrypted region: the comparison is clamped to
|
||||
// the sector, not run off the end of it.
|
||||
let plain_len = SECTOR_BYTES - ENCRYPTED_START;
|
||||
let mut over_long = vec![0u8; plain_len + 10];
|
||||
let (full_sector, full_body) = synth_sector(&title_key, &seed, &[0x00u8; 10]);
|
||||
over_long[..plain_len].copy_from_slice(&full_body[ENCRYPTED_START..]);
|
||||
assert!(
|
||||
descramble_matches(&full_sector, &title_key, &over_long),
|
||||
"a crib longer than the encrypted region must be clamped, not \
|
||||
compared past the end of the sector"
|
||||
);
|
||||
}
|
||||
|
||||
// ── attack_crib: known-answer vectors ──────────────────────────────────
|
||||
//
|
||||
// `attack_crib` is BOTH the cracker's known plaintext and the decrypt
|
||||
// path's "did the cached key descramble correctly?" oracle. Until now it
|
||||
// was only ever exercised end-to-end through `crack_title_key`, on a
|
||||
// fixture whose periodic run covered 39 bytes (0x59..0x80) — long enough
|
||||
// that the run start, the cycle count and the `i % best_p` wrap were all
|
||||
// slack. A crib that silently drifts costs a rip its title key.
|
||||
|
||||
/// Build a sector whose clear header ends in a `period`-length repeating
|
||||
/// run of exactly `run_len` bytes immediately before 0x80.
|
||||
///
|
||||
/// The run is anchored to ABSOLUTE sector offset (`sec[x] = pat[x % period]`),
|
||||
/// which is what makes "the run continues past 0x80" a statement independent
|
||||
/// of the code under test: the byte at `0x80 + i` of the underlying
|
||||
/// plaintext is `pat[(0x80 + i) % period]`.
|
||||
///
|
||||
/// Everything before the run is `0x00` (the pattern bytes are all >= 0xD0,
|
||||
/// so the run cannot be extended backwards by accident), and the encrypted
|
||||
/// region is filled with `0xFF` — so a crib that reads past 0x80 into
|
||||
/// "ciphertext" is immediately visible.
|
||||
fn sector_with_trailing_run(period: usize, run_len: usize) -> Vec<u8> {
|
||||
assert!(
|
||||
run_len < ENCRYPTED_START,
|
||||
"the run lives in the clear header"
|
||||
);
|
||||
let mut sector = vec![0u8; SECTOR_BYTES];
|
||||
sector[FLAG_BYTE] = 0x10;
|
||||
for b in sector[ENCRYPTED_START..].iter_mut() {
|
||||
*b = 0xFF;
|
||||
}
|
||||
let pat: Vec<u8> = (0..period).map(|k| 0xD0u8 + k as u8).collect();
|
||||
for x in (ENCRYPTED_START - run_len)..ENCRYPTED_START {
|
||||
sector[x] = pat[x % period];
|
||||
}
|
||||
sector
|
||||
}
|
||||
|
||||
/// The crib the run PREDICTS: the periodic pattern continued past 0x80.
|
||||
fn expected_crib(period: usize) -> [u8; 10] {
|
||||
let pat: Vec<u8> = (0..period).map(|k| 0xD0u8 + k as u8).collect();
|
||||
let mut out = [0u8; 10];
|
||||
for (i, o) in out.iter_mut().enumerate() {
|
||||
*o = pat[(ENCRYPTED_START + i) % period];
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
/// KNOWN ANSWER: for a run of `run_len` bytes with period 5 ending exactly
|
||||
/// at 0x80, the crib is the run continued forward — the same ten bytes for
|
||||
/// every run length, because the prediction depends only on the pattern and
|
||||
/// the phase, never on how many cycles happened to be visible.
|
||||
///
|
||||
/// The short lengths are the load-bearing ones: at `run_len = 11` the crib
|
||||
/// window starts at 0x76 and is only 10 bytes from the end of the header, so
|
||||
/// any drift in `plain_start`, in `cycles * best_p`, or in the `i % best_p`
|
||||
/// wrap reads the 0xFF "ciphertext" instead of the run.
|
||||
#[test]
|
||||
fn attack_crib_predicts_the_periodic_run_continuing_past_0x80() {
|
||||
for &run_len in &[11usize, 12, 13, 14, 15, 16, 20, 31] {
|
||||
let sector = sector_with_trailing_run(5, run_len);
|
||||
assert_eq!(
|
||||
attack_crib(§or),
|
||||
Some(expected_crib(5)),
|
||||
"period-5 run of {run_len} bytes must predict the run continuing"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The same known answer across several periods, including a period that
|
||||
/// does NOT divide 0x80 (so the crib's phase is non-zero and a body that
|
||||
/// restarted the pattern at index 0 gives a different answer).
|
||||
#[test]
|
||||
fn attack_crib_recovers_the_run_period_and_phase() {
|
||||
// 0x80 % period: 3 for 5, 2 for 6, 2 for 7, 8 for 0x18 — all non-zero,
|
||||
// so the predicted first byte is NOT pat[0] in any of these cases.
|
||||
for &period in &[5usize, 6, 7, 0x18] {
|
||||
let sector = sector_with_trailing_run(period, 3 * period + 1);
|
||||
let crib =
|
||||
attack_crib(§or).unwrap_or_else(|| panic!("no crib for period {period}"));
|
||||
assert_eq!(crib, expected_crib(period), "period {period}");
|
||||
assert_ne!(
|
||||
crib[0], 0xD0,
|
||||
"period {period} does not divide 0x80, so the crib must not \
|
||||
start at pattern index 0"
|
||||
);
|
||||
assert!(
|
||||
crib.iter().all(|&b| b != 0xFF),
|
||||
"period {period}: the crib must never contain a byte read from \
|
||||
the encrypted region"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// A run of exactly ONE cycle (plus the trivial tail the detector counts) is
|
||||
/// not enough to predict forward: [`attack_crib`] requires at least two full
|
||||
/// cycles. Weakening that guard would let a one-off byte sequence be
|
||||
/// declared periodic and produce a confidently wrong crib — which the
|
||||
/// decrypt path uses as its "is my cached key still right?" oracle.
|
||||
#[test]
|
||||
fn attack_crib_refuses_a_run_shorter_than_two_cycles() {
|
||||
// period 8, run of 9 bytes: best_plen = 8, 8 / 8 == 1 cycle.
|
||||
assert_eq!(attack_crib(§or_with_trailing_run(8, 9)), None);
|
||||
// period 0x18, run of 0x19 bytes: one cycle.
|
||||
assert_eq!(attack_crib(§or_with_trailing_run(0x18, 0x19)), None);
|
||||
// ...and one more byte of run does not conjure a second cycle either.
|
||||
assert_eq!(attack_crib(§or_with_trailing_run(8, 10)), None);
|
||||
}
|
||||
|
||||
/// A header with no repeating tail at all yields no crib. Asserted on a
|
||||
/// header whose bytes are pairwise distinct right up to 0x80, so no cycle
|
||||
/// length in 2..0x2F can match even one byte.
|
||||
#[test]
|
||||
fn attack_crib_refuses_a_header_with_no_periodic_tail() {
|
||||
let mut sector = vec![0u8; SECTOR_BYTES];
|
||||
sector[FLAG_BYTE] = 0x10;
|
||||
// 0x00..0x80 strictly increasing: sec[a] == sec[b] iff a == b, so the
|
||||
// detector's `sec[0x7f - (j % i)] == sec[0x7f - j]` needs j % i == j,
|
||||
// which the scan's starting `j = i + 1` already excludes.
|
||||
for (x, b) in sector[..ENCRYPTED_START].iter_mut().enumerate() {
|
||||
*b = x as u8;
|
||||
}
|
||||
assert_eq!(attack_crib(§or), None);
|
||||
// And the cracker built on it reports no key rather than guessing.
|
||||
assert_eq!(crack_title_key(§or), None);
|
||||
}
|
||||
|
||||
/// `attack_crib` indexes `sector[0x7f - j]` with no per-access bound, so its
|
||||
/// own length guard is the only thing between a short buffer and an
|
||||
/// out-of-bounds read. Nothing reached it: every caller-level test used a
|
||||
/// full sector, and the entry points' guards fire first.
|
||||
#[test]
|
||||
fn attack_crib_refuses_a_buffer_shorter_than_a_sector() {
|
||||
for len in [0x15usize, 0x40, 0x7F, SECTOR_BYTES - 1] {
|
||||
let mut sector = vec![0x11u8; len];
|
||||
sector[FLAG_BYTE] = 0x30; // scrambled, so the flag half cannot fire
|
||||
assert_eq!(
|
||||
attack_crib(§or),
|
||||
None,
|
||||
"a {len}-byte buffer is not a sector"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// A header that is periodic ALL THE WAY to offset 0 must not walk the
|
||||
/// backward scan off the front of the sector.
|
||||
///
|
||||
/// The detector counts backwards from 0x7f while `j < 0x80`. On a fully
|
||||
/// periodic header the run never breaks, so `j` reaches 0x7f and the bound
|
||||
/// is the ONLY thing that stops it — one step further and `0x7f - j`
|
||||
/// underflows a `usize` and panics. A constant or fully-patterned 128-byte
|
||||
/// header is ordinary DVD data (padding, a run of zeros), not a crafted
|
||||
/// input, and every existing fixture had a filler/run boundary well before
|
||||
/// offset 0 that stopped the scan early.
|
||||
#[test]
|
||||
fn attack_crib_survives_a_header_that_is_periodic_to_offset_zero() {
|
||||
let mut sector = vec![0u8; SECTOR_BYTES];
|
||||
sector[FLAG_BYTE] = 0x30;
|
||||
let period = 5usize;
|
||||
let pat: Vec<u8> = (0..period).map(|k| 0xD0u8 + k as u8).collect();
|
||||
for (x, b) in sector[..ENCRYPTED_START].iter_mut().enumerate() {
|
||||
*b = pat[x % period];
|
||||
}
|
||||
for b in sector[ENCRYPTED_START..].iter_mut() {
|
||||
*b = 0xFF;
|
||||
}
|
||||
// The FLAG byte sits inside the header at 0x14, so it interrupts the
|
||||
// pattern there; re-lay it and accept that 0x14 breaks the run — the
|
||||
// scan still reaches offset 0x15 - 1 = 0x14 going backwards, i.e.
|
||||
// j = 0x7f - 0x14 = 0x6b, well short of the bound. Instead put the
|
||||
// scramble flag bits into a byte value that IS the pattern's.
|
||||
sector[FLAG_BYTE] = pat[FLAG_BYTE % period];
|
||||
assert_ne!(
|
||||
sector[FLAG_BYTE] & 0x30,
|
||||
0,
|
||||
"fixture check: the pattern byte at 0x14 must itself carry \
|
||||
scramble bits, so the header stays unbroken"
|
||||
);
|
||||
|
||||
assert_eq!(
|
||||
attack_crib(§or),
|
||||
Some(expected_crib(period)),
|
||||
"a fully periodic header must predict its own continuation, and \
|
||||
the backward scan must stop at offset 0"
|
||||
);
|
||||
}
|
||||
|
||||
/// The crib is read from the CLEAR header only. A run that reaches 0x80 must
|
||||
/// predict from the header bytes, never from the encrypted region — the
|
||||
/// previously-fixed bug this function's doc comment records. Pinned by
|
||||
/// rewriting the encrypted region and requiring the crib not to move.
|
||||
#[test]
|
||||
fn attack_crib_is_independent_of_the_encrypted_region() {
|
||||
let base = sector_with_trailing_run(5, 11);
|
||||
let crib = attack_crib(&base).expect("crib");
|
||||
for fill in [0x00u8, 0x5A, 0xD1, 0xFF] {
|
||||
let mut s = base.clone();
|
||||
for b in s[ENCRYPTED_START..].iter_mut() {
|
||||
*b = fill;
|
||||
}
|
||||
assert_eq!(
|
||||
attack_crib(&s),
|
||||
Some(crib),
|
||||
"the crib must not depend on the encrypted region (fill {fill:#04x})"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// ── recover_title_key_from_plain: input-length guard ───────────────────
|
||||
|
||||
/// `recover_title_key_from_plain` unconditionally builds a 10-byte keystream
|
||||
/// buffer from `crypted[0..10]` and `decrypted[0..10]`, so its length guard
|
||||
/// is the only thing standing between a short slice and an
|
||||
/// index-out-of-bounds PANIC.
|
||||
///
|
||||
/// Nothing reached that guard before: `recover_title_key` rejects
|
||||
/// `plain.len() < 10` at its own door and always hands on exactly ten
|
||||
/// ciphertext bytes, and `crack_title_key_inner` always passes a fixed
|
||||
/// `[u8; 10]` crib. The guard is a live contract for any future caller and
|
||||
/// was executed by no test at either boundary.
|
||||
#[test]
|
||||
fn recover_title_key_from_plain_refuses_fewer_than_ten_bytes_of_either_input() {
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
let full = [0xA5u8; 10];
|
||||
for n in 0..10usize {
|
||||
assert_eq!(
|
||||
recover_title_key_from_plain(&full[..n], &full, &seed),
|
||||
None,
|
||||
"{n} ciphertext bytes is fewer than the ten the cipher iterates"
|
||||
);
|
||||
assert_eq!(
|
||||
recover_title_key_from_plain(&full, &full[..n], &seed),
|
||||
None,
|
||||
"{n} plaintext bytes is fewer than the ten the cipher iterates"
|
||||
);
|
||||
}
|
||||
// Exactly ten of each is ACCEPTED as far as the search — the boundary is
|
||||
// `< 10`, not `<= 10`. (Whether this particular keystream has a seed is
|
||||
// immaterial; what must not happen is an early `None` from the guard.)
|
||||
// Proven through the round-trip fixture, whose inputs are exactly ten
|
||||
// bytes and which does recover its key.
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let (sector, _) = synth_sector(&title_key, &seed, &PES);
|
||||
assert_eq!(
|
||||
recover_title_key_from_plain(
|
||||
§or[ENCRYPTED_START..ENCRYPTED_START + 10],
|
||||
&PES,
|
||||
&seed
|
||||
),
|
||||
Some(title_key),
|
||||
"exactly ten bytes of each input must run the search, not trip the guard"
|
||||
);
|
||||
}
|
||||
|
||||
/// The seed XOR-back ([`recover_title_key_from_plain`]'s last step) is what
|
||||
/// turns the recovered LFSR key into the TITLE key: `key ^= sector_seed`.
|
||||
/// Pinned as a known answer across seeds that differ only in one byte — the
|
||||
/// same ciphertext/plaintext pair therefore must yield title keys differing
|
||||
/// in exactly that byte.
|
||||
///
|
||||
/// Without this, a body that ORed the seed in (or dropped the step) still
|
||||
/// round-trips on any fixture whose seed is zero, and on the non-zero ones
|
||||
/// the failure looks like "no key found" rather than a wrong step.
|
||||
#[test]
|
||||
fn recover_title_key_from_plain_xors_the_sector_seed_back_out() {
|
||||
let title_key = [0x42u8, 0x13, 0x37, 0xBE, 0xEF];
|
||||
let seed = [0x11u8, 0x22, 0x33, 0x44, 0x55];
|
||||
let (sector, _) = synth_sector(&title_key, &seed, &PES);
|
||||
let crypted = §or[ENCRYPTED_START..ENCRYPTED_START + 10];
|
||||
|
||||
// The cipher is seeded from `title_key XOR seed`, so re-running the SAME
|
||||
// ciphertext/plaintext against a seed differing in one byte must return
|
||||
// a title key differing in exactly that byte — the XOR is a bijection.
|
||||
assert_eq!(
|
||||
recover_title_key_from_plain(crypted, &PES, &seed),
|
||||
Some(title_key)
|
||||
);
|
||||
for byte in 0..5usize {
|
||||
for bit in [0u32, 3, 7] {
|
||||
let mut alt_seed = seed;
|
||||
alt_seed[byte] ^= 1u8 << bit;
|
||||
let mut expected = title_key;
|
||||
expected[byte] ^= 1u8 << bit;
|
||||
assert_eq!(
|
||||
recover_title_key_from_plain(crypted, &PES, &alt_seed),
|
||||
Some(expected),
|
||||
"seed byte {byte} bit {bit} must XOR straight through to the \
|
||||
title key"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1115
-886
File diff suppressed because it is too large
Load Diff
+105
-47
@@ -22,10 +22,7 @@
|
||||
//! `Disc`-level dump ([`dump_disc`]) covers everything that survives
|
||||
//! lowering: titles, streams, the picked main feature, and AACS state.
|
||||
|
||||
use crate::disc::{
|
||||
AudioChannels, ColorSpace, Disc, DiscTitle, FrameRate, HdrFormat, Resolution, SampleRate,
|
||||
Stream,
|
||||
};
|
||||
use crate::disc::{ColorSpace, Disc, DiscTitle, FrameRate, HdrFormat, Resolution, Stream};
|
||||
use crate::ifo::{CellCategory, DvdTitle};
|
||||
|
||||
const DIAG: &str = "freemkv::diag";
|
||||
@@ -95,36 +92,12 @@ pub fn hdr_str(h: HdrFormat) -> &'static str {
|
||||
}
|
||||
}
|
||||
|
||||
/// Channel count from an [`AudioChannels`] layout (what lands in the MKV
|
||||
/// `Channels` element).
|
||||
pub fn channel_count(ch: AudioChannels) -> u8 {
|
||||
match ch {
|
||||
AudioChannels::Mono => 1,
|
||||
AudioChannels::Stereo => 2,
|
||||
AudioChannels::Stereo21 => 3,
|
||||
AudioChannels::Quad => 4,
|
||||
AudioChannels::Surround50 => 5,
|
||||
AudioChannels::Surround51 => 6,
|
||||
AudioChannels::Surround61 => 7,
|
||||
AudioChannels::Surround71 => 8,
|
||||
AudioChannels::Unknown => 0,
|
||||
}
|
||||
}
|
||||
|
||||
/// Sample-rate in Hz for a [`SampleRate`].
|
||||
pub fn sample_rate_hz(s: SampleRate) -> u32 {
|
||||
match s {
|
||||
SampleRate::S44_1 => 44100,
|
||||
SampleRate::S48 => 48000,
|
||||
SampleRate::S88_2 => 88200,
|
||||
SampleRate::S96 => 96000,
|
||||
SampleRate::S176_4 => 176400,
|
||||
SampleRate::S192 => 192000,
|
||||
SampleRate::S48_96 => 96000,
|
||||
SampleRate::S48_192 => 192000,
|
||||
SampleRate::Unknown => 0,
|
||||
}
|
||||
}
|
||||
// `channel_count` and `sample_rate_hz` lived here as a third copy of the
|
||||
// AudioChannels/SampleRate mappings. They were the only HONEST copy — returning
|
||||
// 0 for Unknown where the canonical accessors fabricated 6 channels at 48 kHz —
|
||||
// and their only caller was the trace line below, in this same file. The
|
||||
// canonical accessors are honest now, so the duplicates are gone rather than
|
||||
// left to drift a fourth time.
|
||||
|
||||
// ── DVD cell-category dump (from the IFO scan, pre-lowering) ─────────────────
|
||||
|
||||
@@ -336,7 +309,7 @@ fn frame_record(track_idx: usize, pts_ns: i64, keyframe: bool, data: &[u8]) -> V
|
||||
/// `track_number` is the 1-based MKV track number; `track` is the built
|
||||
/// [`crate::mux::mkv::MkvTrack`] whose fields map one-to-one onto the emitted
|
||||
/// elements (see `MkvMuxer::new`). No-op unless the diag target is on.
|
||||
pub fn dump_mkv_track(track_number: u64, track: &crate::mux::mkv::MkvTrack) {
|
||||
pub(crate) fn dump_mkv_track(track_number: u64, track: &crate::mux::mkv::MkvTrack) {
|
||||
if !diag_enabled() {
|
||||
return;
|
||||
}
|
||||
@@ -499,15 +472,31 @@ pub fn dump_disc(disc: &Disc) {
|
||||
tracing::debug!(
|
||||
target: DIAG,
|
||||
"tag=decision pick=main_feature title_idx=0 playlist={:?} dur={:.1}s \
|
||||
size={}B clips={} reason=canonical_title_order(fits-disc, fewest-clips, longest, richest-audio)",
|
||||
size={}B clips={} reason={}",
|
||||
main.playlist,
|
||||
main.duration_secs,
|
||||
main.size_bytes,
|
||||
main.clips.len(),
|
||||
main_feature_reason(),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The `reason=` token on the main-feature decision row.
|
||||
///
|
||||
/// DERIVED from [`Disc::CANONICAL_TITLE_ORDER_KEYS`], which lives beside the
|
||||
/// comparator that actually implements them — never restated here. The previous
|
||||
/// hand-written copy drifted (it advertised a `fewest-clips` key the comparator
|
||||
/// had replaced with largest-physical-size), which made the self-diagnosing log
|
||||
/// explain the pick with a rule the code does not apply. A diagnostic that
|
||||
/// disagrees with the decision it documents is worse than no diagnostic.
|
||||
fn main_feature_reason() -> String {
|
||||
format!(
|
||||
"canonical_title_order({})",
|
||||
Disc::CANONICAL_TITLE_ORDER_KEYS.join(", ")
|
||||
)
|
||||
}
|
||||
|
||||
fn dump_aacs(disc: &Disc) {
|
||||
let Some(a) = disc.aacs.as_ref() else {
|
||||
if disc.css.is_some() {
|
||||
@@ -530,7 +519,7 @@ fn dump_aacs(disc: &Disc) {
|
||||
a.bus_encryption,
|
||||
a.mkb_version,
|
||||
a.disc_hash,
|
||||
a.key_source.name(),
|
||||
a.key_source,
|
||||
a.vuk.is_some(),
|
||||
a.unit_keys.len(),
|
||||
a.uk_ro.len(),
|
||||
@@ -610,8 +599,8 @@ fn dump_title(ti: usize, title: &DiscTitle) {
|
||||
a.pid,
|
||||
a.codec,
|
||||
a.channels,
|
||||
channel_count(a.channels),
|
||||
sample_rate_hz(a.sample_rate),
|
||||
a.channels.count(),
|
||||
a.sample_rate.hz(),
|
||||
a.language,
|
||||
a.secondary,
|
||||
),
|
||||
@@ -631,6 +620,60 @@ fn dump_title(ti: usize, title: &DiscTitle) {
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
// Needed only by the tests: the production code in this file no longer names
|
||||
// these types directly, since the local channel/sample-rate duplicates were
|
||||
// deleted in favour of the canonical accessors.
|
||||
use crate::disc::{AudioChannels, SampleRate};
|
||||
|
||||
/// The main-feature decision row must NAME `canonical_title_order`'s sort
|
||||
/// keys, not restate them from memory. The restated copy had drifted: it
|
||||
/// still advertised a "fewest-clips" key long after the comparator replaced
|
||||
/// clip-count with largest-physical-size, so a bug report read at
|
||||
/// `--log-level 3` explained the pick with a rule the code does not apply.
|
||||
///
|
||||
/// The behavioural half is asserted first — against the comparator itself,
|
||||
/// with literals — so the key names are checked against what the code
|
||||
/// actually does, not against the string that names them.
|
||||
#[test]
|
||||
fn main_feature_reason_names_the_comparators_real_keys() {
|
||||
use crate::disc::{Clip, Disc, DiscTitle};
|
||||
|
||||
let sized = |size_bytes: u64, n_clips: usize| DiscTitle {
|
||||
size_bytes,
|
||||
clips: (0..n_clips)
|
||||
.map(|i| Clip {
|
||||
feed_span: None,
|
||||
clip_id: format!("{i:05}"),
|
||||
in_time: 0,
|
||||
out_time: 0,
|
||||
duration_secs: 0.0,
|
||||
source_packets: 0,
|
||||
})
|
||||
.collect(),
|
||||
..DiscTitle::empty()
|
||||
};
|
||||
// A 40-clip 8 GB title beats a 1-clip 1 GB title: the comparator's
|
||||
// primary key among disc-fitting titles is LARGEST SIZE. "fewest clips"
|
||||
// would predict the opposite, so the drifted string described a rule
|
||||
// the comparator does not implement.
|
||||
let many_clips_big = sized(8_000_000_000, 40);
|
||||
let one_clip_small = sized(1_000_000_000, 1);
|
||||
assert_eq!(
|
||||
Disc::canonical_title_order(&many_clips_big, &one_clip_small, 25_000_000_000),
|
||||
std::cmp::Ordering::Less,
|
||||
"largest size wins regardless of clip count"
|
||||
);
|
||||
|
||||
let reason = main_feature_reason();
|
||||
assert!(
|
||||
!reason.contains("clips"),
|
||||
"the reason must not advertise a clip-count key the comparator dropped: {reason}"
|
||||
);
|
||||
assert_eq!(
|
||||
reason, "canonical_title_order(fits-disc, largest-size, longest, richest-audio)",
|
||||
"the reason must name the comparator's four keys in priority order"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn res_str_keeps_interlace_marker() {
|
||||
@@ -656,18 +699,33 @@ mod tests {
|
||||
assert_eq!(hdr_str(HdrFormat::Sdr), "SDR");
|
||||
}
|
||||
|
||||
/// Moved from the deleted local duplicates onto the canonical accessors,
|
||||
/// with the Unknown case added — which is the whole point of the change.
|
||||
#[test]
|
||||
fn channel_count_matches_layout() {
|
||||
assert_eq!(channel_count(AudioChannels::Mono), 1);
|
||||
assert_eq!(channel_count(AudioChannels::Stereo), 2);
|
||||
assert_eq!(channel_count(AudioChannels::Surround51), 6);
|
||||
assert_eq!(channel_count(AudioChannels::Surround71), 8);
|
||||
fn channel_count_matches_layout_and_is_zero_when_unknown() {
|
||||
assert_eq!(AudioChannels::Mono.count(), 1);
|
||||
assert_eq!(AudioChannels::Stereo.count(), 2);
|
||||
assert_eq!(AudioChannels::Surround51.count(), 6);
|
||||
assert_eq!(AudioChannels::Surround71.count(), 8);
|
||||
// The one that matters. This used to return 6, which is indistinguishable
|
||||
// from a real 5.1 track and left every caller responsible for checking
|
||||
// the variant first.
|
||||
assert_eq!(
|
||||
AudioChannels::Unknown.count(),
|
||||
0,
|
||||
"an unknown layout must not report a plausible channel count"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sample_rate_hz_values() {
|
||||
assert_eq!(sample_rate_hz(SampleRate::S48), 48000);
|
||||
assert_eq!(sample_rate_hz(SampleRate::S96), 96000);
|
||||
fn sample_rate_hz_values_and_zero_when_unknown() {
|
||||
assert_eq!(SampleRate::S48.hz(), 48000.0);
|
||||
assert_eq!(SampleRate::S96.hz(), 96000.0);
|
||||
assert_eq!(
|
||||
SampleRate::Unknown.hz(),
|
||||
0.0,
|
||||
"an unknown sample rate must not report a plausible 48 kHz"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -0,0 +1,743 @@
|
||||
//! ECMA-167 / UDF 1.02 descriptor encoder.
|
||||
//!
|
||||
//! Turns a [`Layout`](super::layout::Layout) — a directory tree with every
|
||||
//! ICB, directory-data and file-data block already assigned — into the set of
|
||||
//! metadata sectors a real UDF volume would carry. Nothing here touches the
|
||||
//! filesystem: it is a pure function from layout to sectors, which is what
|
||||
//! makes it testable against the production parser in `udf.rs`.
|
||||
//!
|
||||
//! What is emitted, in volume order:
|
||||
//!
|
||||
//! | sector | descriptor |
|
||||
//! |---|---|
|
||||
//! | 16, 17, 18 | Volume Recognition Sequence — `BEA01`, `NSR02`, `TEA01` (ECMA-167 2/9.1) |
|
||||
//! | 32… | Main Volume Descriptor Sequence — PVD, IUVD, PD, LVD, USD, TD |
|
||||
//! | 48… | Reserve VDS (byte-identical but for the tag locations) |
|
||||
//! | 64, 65 | Logical Volume Integrity Sequence — LVID, TD |
|
||||
//! | 256 | Anchor Volume Descriptor Pointer |
|
||||
//! | `part_start` + 0, +1 | File Set Descriptor, TD |
|
||||
//! | `part_start` + … | File Entries (ICBs) and directory data (FIDs) |
|
||||
//! | last sector | Anchor Volume Descriptor Pointer (copy) |
|
||||
//!
|
||||
//! UDF revision 1.02 with a single Type-1 partition map is deliberate: it is
|
||||
//! the DVD-Video profile, it is the shape `read_filesystem` takes when
|
||||
//! `num_partition_maps < 2`, and it avoids the UDF 2.50 Metadata Partition
|
||||
//! entirely. That also means a synthetic image never exercises the Metadata
|
||||
//! Partition path in `udf.rs` (`:946-991`) — see the module docs on `dirimage`.
|
||||
|
||||
use super::layout::{DirNode, Layout};
|
||||
use crate::error::{Error, Result};
|
||||
use std::collections::BTreeMap;
|
||||
|
||||
/// Logical block / sector size. Fixed for every optical profile this crate
|
||||
/// reads, and the same quantity as [`crate::consts::SECTOR_BYTES`] — aliased
|
||||
/// rather than re-declared so the two cannot drift apart. The short name is
|
||||
/// kept because it appears in ~25 extent and offset expressions across
|
||||
/// `dirimage`, where the longer one would bury the arithmetic.
|
||||
pub(super) use crate::consts::SECTOR_BYTES as SECTOR;
|
||||
|
||||
/// Descriptor version recorded in every tag. 2 = ECMA-167 2nd edition, which
|
||||
/// is what UDF revisions up to and including 2.00 require.
|
||||
const DESC_VERSION: u16 = 2;
|
||||
|
||||
/// UDF revision recorded in the domain EntityID suffix (1.02, BCD-ish u16).
|
||||
const UDF_REVISION: u16 = 0x0102;
|
||||
|
||||
/// A fixed recording timestamp, so an image synthesized from the same folder
|
||||
/// twice is byte-identical. Real mtimes would make every test golden-file
|
||||
/// comparison and every `dir:// -> iso://` re-run differ for no benefit.
|
||||
const FIXED_TIME: Timestamp = Timestamp {
|
||||
year: 2000,
|
||||
month: 1,
|
||||
day: 1,
|
||||
};
|
||||
|
||||
struct Timestamp {
|
||||
year: i16,
|
||||
month: u8,
|
||||
day: u8,
|
||||
}
|
||||
|
||||
/// The synthesized metadata: absolute LBA → sector contents. Data sectors are
|
||||
/// NOT here; they are served from the backing files.
|
||||
pub(super) type MetaSectors = BTreeMap<u32, Box<[u8; SECTOR]>>;
|
||||
|
||||
/// The descriptor-tag CRC of ECMA-167 7.2.4: polynomial 0x1021, initial value
|
||||
/// ZERO, no reflection, no final XOR — the variant catalogued as CRC-16/XMODEM
|
||||
/// (check value 0x31C3), NOT CCITT-FALSE, which seeds at 0xFFFF and would make
|
||||
/// every descriptor this crate writes fail a conformant driver's validation.
|
||||
fn crc16(data: &[u8]) -> u16 {
|
||||
let mut crc: u16 = 0;
|
||||
for &b in data {
|
||||
crc ^= (b as u16) << 8;
|
||||
for _ in 0..8 {
|
||||
crc = if crc & 0x8000 != 0 {
|
||||
(crc << 1) ^ 0x1021
|
||||
} else {
|
||||
crc << 1
|
||||
};
|
||||
}
|
||||
}
|
||||
crc
|
||||
}
|
||||
|
||||
/// Write an ECMA-167 3/7.2 descriptor tag over `buf[0..16]`.
|
||||
///
|
||||
/// `tag_loc` is the block number of the sector holding the descriptor —
|
||||
/// ABSOLUTE for the volume-space descriptors (AVDP, VDS, LVID) and
|
||||
/// PARTITION-RELATIVE for everything inside the partition (FSD, File Entries).
|
||||
/// Getting that wrong is the classic reason a hand-built volume mounts nowhere:
|
||||
/// a driver that validates the tag location rejects the descriptor outright.
|
||||
///
|
||||
/// `desc_len` is the descriptor's total length including the tag; the CRC
|
||||
/// covers `buf[16..desc_len]`.
|
||||
fn finish_tag(buf: &mut [u8], tag_id: u16, tag_loc: u32, desc_len: usize) {
|
||||
buf[0..2].copy_from_slice(&tag_id.to_le_bytes());
|
||||
buf[2..4].copy_from_slice(&DESC_VERSION.to_le_bytes());
|
||||
buf[4] = 0; // checksum, filled below
|
||||
buf[5] = 0; // reserved
|
||||
buf[6..8].copy_from_slice(&0u16.to_le_bytes()); // tag serial number
|
||||
let crc_len = desc_len - 16;
|
||||
let crc = crc16(&buf[16..desc_len]);
|
||||
buf[8..10].copy_from_slice(&crc.to_le_bytes());
|
||||
buf[10..12].copy_from_slice(&(crc_len as u16).to_le_bytes());
|
||||
buf[12..16].copy_from_slice(&tag_loc.to_le_bytes());
|
||||
// ECMA-167 3/7.2.3: sum of bytes 0..16 EXCLUDING byte 4, modulo 256.
|
||||
let sum: u32 = buf[0..16]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(i, _)| *i != 4)
|
||||
.map(|(_, b)| *b as u32)
|
||||
.sum();
|
||||
buf[4] = (sum % 256) as u8;
|
||||
}
|
||||
|
||||
/// ECMA-167 1/7.2.1 charspec: type 0 (CS0) + "OSTA Compressed Unicode".
|
||||
fn put_charspec(buf: &mut [u8]) {
|
||||
buf[0] = 0;
|
||||
let id = b"OSTA Compressed Unicode";
|
||||
buf[1..1 + id.len()].copy_from_slice(id);
|
||||
}
|
||||
|
||||
/// ECMA-167 1/7.4 EntityID: flags byte, 23 identifier bytes, 8 suffix bytes.
|
||||
fn put_entity_id(buf: &mut [u8], id: &[u8], suffix: &[u8]) {
|
||||
buf[0] = 0;
|
||||
let n = id.len().min(23);
|
||||
buf[1..1 + n].copy_from_slice(&id[..n]);
|
||||
let m = suffix.len().min(8);
|
||||
buf[24..24 + m].copy_from_slice(&suffix[..m]);
|
||||
}
|
||||
|
||||
/// The `*OSTA UDF Compliant` domain EntityID suffix: UDF revision, domain
|
||||
/// flags (0 = neither hard nor soft write-protected), reserved.
|
||||
fn domain_suffix() -> [u8; 8] {
|
||||
let mut s = [0u8; 8];
|
||||
s[0..2].copy_from_slice(&UDF_REVISION.to_le_bytes());
|
||||
s
|
||||
}
|
||||
|
||||
/// This crate's implementation EntityID suffix: OS class / OS identifier
|
||||
/// (0 = undefined, deliberately — the image is not OS-specific) + 6 free bytes.
|
||||
fn impl_suffix() -> [u8; 8] {
|
||||
[0u8; 8]
|
||||
}
|
||||
|
||||
fn put_impl_id(buf: &mut [u8]) {
|
||||
put_entity_id(buf, b"*freemkv", &impl_suffix());
|
||||
}
|
||||
|
||||
fn put_domain_id(buf: &mut [u8]) {
|
||||
put_entity_id(buf, b"*OSTA UDF Compliant", &domain_suffix());
|
||||
}
|
||||
|
||||
/// OSTA CS0 d-string: a compression-ID byte, the characters, then the used
|
||||
/// length in the FIELD'S LAST byte (ECMA-167 1/7.2.12 + UDF 2.1.3). An
|
||||
/// all-zero field is the empty string.
|
||||
fn put_dstring(buf: &mut [u8], s: &str) {
|
||||
if s.is_empty() {
|
||||
return;
|
||||
}
|
||||
let encoded = encode_cs0(s);
|
||||
// Leave room for the trailing length byte.
|
||||
let room = buf.len() - 1;
|
||||
let n = encoded.len().min(room);
|
||||
buf[..n].copy_from_slice(&encoded[..n]);
|
||||
buf[buf.len() - 1] = n as u8;
|
||||
}
|
||||
|
||||
/// OSTA CS0: compression ID 8 (one byte per character) when every character
|
||||
/// is ASCII, otherwise compression ID 16 (UTF-16BE).
|
||||
///
|
||||
/// ASCII rather than Latin-1 for the 8-bit form on purpose: `parse_udf_name`
|
||||
/// (`udf.rs:1467`) decodes a compression-8 name with `from_utf8_lossy`, so a
|
||||
/// 0x80-0xFF byte — legal CS0 — would come back as U+FFFD. Every character
|
||||
/// above 0x7F therefore takes the 16-bit form, which that parser decodes
|
||||
/// correctly.
|
||||
pub(super) fn encode_cs0(s: &str) -> Vec<u8> {
|
||||
if s.is_ascii() {
|
||||
let mut v = Vec::with_capacity(1 + s.len());
|
||||
v.push(8u8);
|
||||
v.extend_from_slice(s.as_bytes());
|
||||
v
|
||||
} else {
|
||||
let mut v = vec![16u8];
|
||||
for u in s.encode_utf16() {
|
||||
v.extend_from_slice(&u.to_be_bytes());
|
||||
}
|
||||
v
|
||||
}
|
||||
}
|
||||
|
||||
/// ECMA-167 1/7.3 timestamp, 12 bytes. Type 1 (local time) with a zero
|
||||
/// offset, i.e. UTC.
|
||||
fn put_timestamp(buf: &mut [u8]) {
|
||||
buf[0..2].copy_from_slice(&0x1000u16.to_le_bytes());
|
||||
buf[2..4].copy_from_slice(&FIXED_TIME.year.to_le_bytes());
|
||||
buf[4] = FIXED_TIME.month;
|
||||
buf[5] = FIXED_TIME.day;
|
||||
}
|
||||
|
||||
/// ECMA-167 3/7.1 extent_ad: length in BYTES, then location.
|
||||
fn put_extent_ad(buf: &mut [u8], len_bytes: u32, lba: u32) {
|
||||
buf[0..4].copy_from_slice(&len_bytes.to_le_bytes());
|
||||
buf[4..8].copy_from_slice(&lba.to_le_bytes());
|
||||
}
|
||||
|
||||
/// ECMA-167 4/14.14.2 long_ad: length+type, then lb_addr (block, partition
|
||||
/// reference), then 6 implementation-use bytes.
|
||||
fn put_long_ad(buf: &mut [u8], len_bytes: u32, lba: u32) {
|
||||
buf[0..4].copy_from_slice(&len_bytes.to_le_bytes());
|
||||
buf[4..8].copy_from_slice(&lba.to_le_bytes());
|
||||
buf[8..10].copy_from_slice(&0u16.to_le_bytes()); // partition reference 0
|
||||
}
|
||||
|
||||
/// ECMA-167 4/14.14.1 short_ad. The top two bits of the length word are the
|
||||
/// extent TYPE (0 = recorded and allocated), which is exactly why `udf.rs`
|
||||
/// masks with `0x3FFF_FFFF` when it reads one back — the mask is the field
|
||||
/// boundary, not a truncation bug.
|
||||
fn put_short_ad(buf: &mut [u8], len_bytes: u32, lba: u32) {
|
||||
debug_assert!(len_bytes <= 0x3FFF_FFFF, "AD length must fit 30 bits");
|
||||
buf[0..4].copy_from_slice(&len_bytes.to_le_bytes());
|
||||
buf[4..8].copy_from_slice(&lba.to_le_bytes());
|
||||
}
|
||||
|
||||
fn blank() -> Box<[u8; SECTOR]> {
|
||||
Box::new([0u8; SECTOR])
|
||||
}
|
||||
|
||||
// ── Volume-space descriptors ────────────────────────────────────────────────
|
||||
|
||||
/// ECMA-167 2/9.1 Volume Structure Descriptor: the three-sector recognition
|
||||
/// sequence an OS looks for before it will even consider the volume UDF.
|
||||
fn volume_recognition(id: &[u8; 5]) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
s[0] = 0; // structure type
|
||||
s[1..6].copy_from_slice(id);
|
||||
s[6] = 1; // structure version
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.1 Primary Volume Descriptor.
|
||||
fn primary_volume(volume_id: &str, lba: u32, seq: u32) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
s[16..20].copy_from_slice(&seq.to_le_bytes());
|
||||
s[20..24].copy_from_slice(&0u32.to_le_bytes()); // PVD number
|
||||
put_dstring(&mut s[24..56], volume_id);
|
||||
s[56..58].copy_from_slice(&1u16.to_le_bytes()); // volume sequence number
|
||||
s[58..60].copy_from_slice(&1u16.to_le_bytes()); // max volume sequence number
|
||||
s[60..62].copy_from_slice(&2u16.to_le_bytes()); // interchange level
|
||||
s[62..64].copy_from_slice(&2u16.to_le_bytes()); // max interchange level
|
||||
s[64..68].copy_from_slice(&1u32.to_le_bytes()); // character set list
|
||||
s[68..72].copy_from_slice(&1u32.to_le_bytes()); // max character set list
|
||||
// UDF 2.2.2.5: the first 8 characters of the volume set identifier must be
|
||||
// unique. A fixed hex prefix plus the volume id is sufficient here — the
|
||||
// image is single-volume and never joins a real volume set.
|
||||
put_dstring(&mut s[72..200], &format!("46524D4B{volume_id}"));
|
||||
put_charspec(&mut s[200..264]); // descriptor character set
|
||||
put_charspec(&mut s[264..328]); // explanatory character set
|
||||
put_timestamp(&mut s[376..388]);
|
||||
put_impl_id(&mut s[388..420]);
|
||||
finish_tag(&mut s[..], 1, lba, 512);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.4 + UDF 2.2.7 Implementation Use Volume Descriptor
|
||||
/// (`*UDF LV Info`). Not read by `udf.rs`, required by the spec.
|
||||
fn impl_use_volume(volume_id: &str, lba: u32, seq: u32) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
s[16..20].copy_from_slice(&seq.to_le_bytes());
|
||||
put_entity_id(&mut s[20..52], b"*UDF LV Info", &domain_suffix());
|
||||
put_charspec(&mut s[52..116]); // LVI charset
|
||||
put_dstring(&mut s[116..244], volume_id); // logical volume identifier
|
||||
put_impl_id(&mut s[352..384]);
|
||||
finish_tag(&mut s[..], 4, lba, 512);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.5 Partition Descriptor — the descriptor `read_filesystem`
|
||||
/// takes `partition_start` from (offset 188).
|
||||
fn partition(part_start: u32, part_sectors: u32, lba: u32, seq: u32) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
s[16..20].copy_from_slice(&seq.to_le_bytes());
|
||||
s[20..22].copy_from_slice(&1u16.to_le_bytes()); // partition flags: allocated
|
||||
s[22..24].copy_from_slice(&0u16.to_le_bytes()); // partition number
|
||||
put_entity_id(&mut s[24..56], b"+NSR02", &[]);
|
||||
// s[56..184] partition contents use = Partition Header Descriptor. All
|
||||
// zero: a read-only partition records no unallocated/freed space tables.
|
||||
s[184..188].copy_from_slice(&1u32.to_le_bytes()); // access type: read only
|
||||
s[188..192].copy_from_slice(&part_start.to_le_bytes());
|
||||
s[192..196].copy_from_slice(&part_sectors.to_le_bytes());
|
||||
put_impl_id(&mut s[196..228]);
|
||||
finish_tag(&mut s[..], 5, lba, 512);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.6 Logical Volume Descriptor. Carries the FSD long_ad and the
|
||||
/// partition map table; `read_filesystem` reads `num_partition_maps` at 268
|
||||
/// and takes the single-partition path when it is 1.
|
||||
fn logical_volume(
|
||||
volume_id: &str,
|
||||
fsd_lba: u32,
|
||||
integrity_lba: u32,
|
||||
integrity_sectors: u32,
|
||||
lba: u32,
|
||||
seq: u32,
|
||||
) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
s[16..20].copy_from_slice(&seq.to_le_bytes());
|
||||
put_charspec(&mut s[20..84]);
|
||||
put_dstring(&mut s[84..212], volume_id);
|
||||
s[212..216].copy_from_slice(&(SECTOR as u32).to_le_bytes()); // logical block size
|
||||
put_domain_id(&mut s[216..248]);
|
||||
// Logical volume contents use = long_ad of the File Set Descriptor,
|
||||
// partition-relative. One sector.
|
||||
put_long_ad(&mut s[248..264], SECTOR as u32, fsd_lba);
|
||||
s[264..268].copy_from_slice(&6u32.to_le_bytes()); // map table length
|
||||
s[268..272].copy_from_slice(&1u32.to_le_bytes()); // number of partition maps
|
||||
put_impl_id(&mut s[272..304]);
|
||||
put_extent_ad(
|
||||
&mut s[432..440],
|
||||
integrity_sectors * SECTOR as u32,
|
||||
integrity_lba,
|
||||
);
|
||||
// ECMA-167 3/10.7.2 Type 1 partition map.
|
||||
s[440] = 1; // map type
|
||||
s[441] = 6; // map length
|
||||
s[442..444].copy_from_slice(&1u16.to_le_bytes()); // volume sequence number
|
||||
s[444..446].copy_from_slice(&0u16.to_le_bytes()); // partition number
|
||||
finish_tag(&mut s[..], 6, lba, 446);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.8 Unallocated Space Descriptor with zero extents — the whole
|
||||
/// volume is accounted for by the partition.
|
||||
fn unallocated_space(lba: u32, seq: u32) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
s[16..20].copy_from_slice(&seq.to_le_bytes());
|
||||
s[20..24].copy_from_slice(&0u32.to_le_bytes());
|
||||
finish_tag(&mut s[..], 7, lba, 24);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.9 Terminating Descriptor.
|
||||
fn terminating(lba: u32) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
finish_tag(&mut s[..], 8, lba, 512);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.10 + UDF 2.2.6 Logical Volume Integrity Descriptor, closed.
|
||||
fn integrity(
|
||||
part_sectors: u32,
|
||||
files: u32,
|
||||
dirs: u32,
|
||||
next_uid: u64,
|
||||
lba: u32,
|
||||
) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
put_timestamp(&mut s[16..28]);
|
||||
s[28..32].copy_from_slice(&1u32.to_le_bytes()); // integrity type: close
|
||||
// s[32..40] next integrity extent: none.
|
||||
s[40..48].copy_from_slice(&next_uid.to_le_bytes()); // logical volume contents use: next unique id
|
||||
s[72..76].copy_from_slice(&1u32.to_le_bytes()); // number of partitions
|
||||
s[76..80].copy_from_slice(&46u32.to_le_bytes()); // length of implementation use
|
||||
s[80..84].copy_from_slice(&0u32.to_le_bytes()); // free space: none (read-only)
|
||||
s[84..88].copy_from_slice(&part_sectors.to_le_bytes()); // size table
|
||||
put_impl_id(&mut s[88..120]);
|
||||
s[120..124].copy_from_slice(&files.to_le_bytes());
|
||||
s[124..128].copy_from_slice(&dirs.to_le_bytes());
|
||||
s[128..130].copy_from_slice(&UDF_REVISION.to_le_bytes()); // min read revision
|
||||
s[130..132].copy_from_slice(&UDF_REVISION.to_le_bytes()); // min write revision
|
||||
s[132..134].copy_from_slice(&UDF_REVISION.to_le_bytes()); // max write revision
|
||||
finish_tag(&mut s[..], 9, lba, 134);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 3/10.2 Anchor Volume Descriptor Pointer. `read_filesystem` reads
|
||||
/// the main VDS extent from offsets 16..24 and sweeps it.
|
||||
fn anchor(main_lba: u32, reserve_lba: u32, vds_sectors: u32, lba: u32) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
put_extent_ad(&mut s[16..24], vds_sectors * SECTOR as u32, main_lba);
|
||||
put_extent_ad(&mut s[24..32], vds_sectors * SECTOR as u32, reserve_lba);
|
||||
finish_tag(&mut s[..], 2, lba, 512);
|
||||
s
|
||||
}
|
||||
|
||||
/// ECMA-167 4/14.1 File Set Descriptor. `read_filesystem` requires tag 256 at
|
||||
/// the first block of the (metadata =) partition and reads the root ICB block
|
||||
/// from offset 404.
|
||||
fn file_set(volume_id: &str, root_icb: u32, lba: u32) -> Box<[u8; SECTOR]> {
|
||||
let mut s = blank();
|
||||
put_timestamp(&mut s[16..28]);
|
||||
s[28..30].copy_from_slice(&3u16.to_le_bytes()); // interchange level
|
||||
s[30..32].copy_from_slice(&3u16.to_le_bytes()); // max interchange level
|
||||
s[32..36].copy_from_slice(&1u32.to_le_bytes()); // character set list
|
||||
s[36..40].copy_from_slice(&1u32.to_le_bytes()); // max character set list
|
||||
s[40..44].copy_from_slice(&0u32.to_le_bytes()); // file set number
|
||||
s[44..48].copy_from_slice(&0u32.to_le_bytes()); // file set descriptor number
|
||||
put_charspec(&mut s[48..112]);
|
||||
put_dstring(&mut s[112..240], volume_id);
|
||||
put_charspec(&mut s[240..304]);
|
||||
put_dstring(&mut s[304..336], volume_id);
|
||||
put_long_ad(&mut s[400..416], SECTOR as u32, root_icb);
|
||||
put_domain_id(&mut s[416..448]);
|
||||
finish_tag(&mut s[..], 256, lba, 512);
|
||||
s
|
||||
}
|
||||
|
||||
// ── Partition-space descriptors ─────────────────────────────────────────────
|
||||
|
||||
/// UDF permission word: read + execute for owner, group and other. No write
|
||||
/// bit anywhere — the volume is read-only.
|
||||
const PERM_R_X: u32 = 0x0000_1000 | 0x0000_0400 | 0x0000_0080 | 0x0000_0020 | 0x4 | 0x1;
|
||||
|
||||
/// ECMA-167 4/14.9 File Entry (tag 261).
|
||||
///
|
||||
/// Tag 261 rather than the Extended File Entry (266) real BD-ROMs use: an EFE
|
||||
/// requires UDF 2.00+, and this image declares 1.02. `udf.rs` reads both — the
|
||||
/// 261 field offsets it uses (l_ea 168, l_ad 172, ADs at 176 + l_ea) are the
|
||||
/// ones written here.
|
||||
///
|
||||
/// `extents` are partition-relative (block, byte-length) pairs, already split
|
||||
/// so no single one exceeds the 30-bit AD length field.
|
||||
fn file_entry(
|
||||
is_dir: bool,
|
||||
info_len: u64,
|
||||
extents: &[(u32, u32)],
|
||||
link_count: u16,
|
||||
unique_id: u64,
|
||||
lba: u32,
|
||||
) -> Result<Box<[u8; SECTOR]>> {
|
||||
let mut s = blank();
|
||||
// ICB tag (ECMA-167 4/14.6) at offset 16.
|
||||
s[16..20].copy_from_slice(&0u32.to_le_bytes()); // prior recorded direct entries
|
||||
s[20..22].copy_from_slice(&4u16.to_le_bytes()); // strategy type 4
|
||||
s[24..26].copy_from_slice(&1u16.to_le_bytes()); // max number of entries
|
||||
s[27] = if is_dir { 4 } else { 5 }; // file type: directory / byte sequence
|
||||
// s[28..34] parent ICB location: not recorded (permitted).
|
||||
// s[34..36] ICB flags: 0 => short allocation descriptors. `udf.rs:601`
|
||||
// reads exactly this word to pick its AD stride.
|
||||
s[34..36].copy_from_slice(&0u16.to_le_bytes());
|
||||
// UDF's sentinel for "not specified" is 0xFFFFFFFF, not 0 — 0 is a real
|
||||
// uid/gid (root). A synthesized image has no meaningful owner, and a driver
|
||||
// that maps these through would otherwise report every file as root-owned.
|
||||
s[36..40].copy_from_slice(&u32::MAX.to_le_bytes()); // uid: not specified
|
||||
s[40..44].copy_from_slice(&u32::MAX.to_le_bytes()); // gid: not specified
|
||||
s[44..48].copy_from_slice(&PERM_R_X.to_le_bytes());
|
||||
s[48..50].copy_from_slice(&link_count.to_le_bytes());
|
||||
s[56..64].copy_from_slice(&info_len.to_le_bytes());
|
||||
let blocks: u64 = extents
|
||||
.iter()
|
||||
.map(|(_, len)| (*len as u64).div_ceil(SECTOR as u64))
|
||||
.sum();
|
||||
s[64..72].copy_from_slice(&blocks.to_le_bytes()); // logical blocks recorded
|
||||
put_timestamp(&mut s[72..84]); // access
|
||||
put_timestamp(&mut s[84..96]); // modification
|
||||
put_timestamp(&mut s[96..108]); // attribute
|
||||
s[108..112].copy_from_slice(&1u32.to_le_bytes()); // checkpoint
|
||||
put_impl_id(&mut s[128..160]);
|
||||
s[160..168].copy_from_slice(&unique_id.to_le_bytes());
|
||||
s[168..172].copy_from_slice(&0u32.to_le_bytes()); // length of EAs
|
||||
let l_ad = extents.len() * 8;
|
||||
// A short AD is 8 bytes and the entry has 2048 - 176 = 1872 bytes for
|
||||
// them, i.e. 234 extents — over 200 GiB at the per-AD ceiling. Beyond
|
||||
// that an Allocation Extent Descriptor chain would be required; refuse
|
||||
// rather than write a truncated list.
|
||||
if 176 + l_ad > SECTOR {
|
||||
return Err(Error::DirImageTooLarge);
|
||||
}
|
||||
s[172..176].copy_from_slice(&(l_ad as u32).to_le_bytes());
|
||||
for (i, (elba, len)) in extents.iter().enumerate() {
|
||||
let off = 176 + i * 8;
|
||||
put_short_ad(&mut s[off..off + 8], *len, *elba);
|
||||
}
|
||||
finish_tag(&mut s[..], 261, lba, 176 + l_ad);
|
||||
Ok(s)
|
||||
}
|
||||
|
||||
/// ECMA-167 4/14.4 File Identifier Descriptor, appended to `buf`.
|
||||
///
|
||||
/// FIDs are packed with no inter-descriptor padding beyond the 4-byte
|
||||
/// alignment the spec mandates, and they are allowed to span logical blocks —
|
||||
/// which is also what `read_directory` (`udf.rs:1312`) assumes: it walks the
|
||||
/// directory extent as one flat byte run and STOPS at the first non-257 tag,
|
||||
/// so any block-alignment gap would truncate the directory.
|
||||
fn push_fid(buf: &mut Vec<u8>, name: &str, icb_lba: u32, is_dir: bool, is_parent: bool) {
|
||||
let start = buf.len();
|
||||
let name_field: Vec<u8> = if is_parent {
|
||||
Vec::new()
|
||||
} else {
|
||||
encode_cs0(name)
|
||||
};
|
||||
let l_fi = name_field.len();
|
||||
let mut fid = vec![0u8; 38];
|
||||
fid[16..18].copy_from_slice(&1u16.to_le_bytes()); // file version number
|
||||
let mut chars = 0u8;
|
||||
if is_dir {
|
||||
chars |= 0x02;
|
||||
}
|
||||
if is_parent {
|
||||
chars |= 0x08;
|
||||
}
|
||||
fid[18] = chars;
|
||||
// The planner refuses any name whose encoding exceeds what this byte can
|
||||
// hold (`layout::MAX_CS0_NAME_BYTES`), so this cannot wrap in practice. The
|
||||
// assert states the invariant where it is relied on rather than trusting a
|
||||
// check three files away; a wrap here would desynchronise the directory.
|
||||
debug_assert!(
|
||||
l_fi <= u8::MAX as usize,
|
||||
"FID name length must fit one byte"
|
||||
);
|
||||
fid[19] = l_fi as u8;
|
||||
put_long_ad(&mut fid[20..36], SECTOR as u32, icb_lba);
|
||||
fid[36..38].copy_from_slice(&0u16.to_le_bytes()); // length of implementation use
|
||||
buf.extend_from_slice(&fid);
|
||||
buf.extend_from_slice(&name_field);
|
||||
let unpadded = buf.len() - start;
|
||||
let padded = unpadded.div_ceil(4) * 4;
|
||||
buf.resize(start + padded, 0);
|
||||
// The tag is written last: its CRC covers the descriptor body, which the
|
||||
// padding is not part of (ECMA-167 4/14.4.9 counts padding outside the
|
||||
// CRC'd length).
|
||||
let tag_loc_placeholder = 0;
|
||||
finish_tag(
|
||||
&mut buf[start..start + unpadded],
|
||||
257,
|
||||
tag_loc_placeholder,
|
||||
unpadded,
|
||||
);
|
||||
}
|
||||
|
||||
/// Serialize one directory's FID list (parent entry first, then children).
|
||||
pub(super) fn dir_fids(dir: &DirNode) -> Vec<u8> {
|
||||
let mut buf = Vec::new();
|
||||
push_fid(&mut buf, "", dir.parent_icb_lba, true, true);
|
||||
for sub in &dir.dirs {
|
||||
push_fid(&mut buf, &sub.name, sub.icb_lba, true, false);
|
||||
}
|
||||
for f in &dir.files {
|
||||
push_fid(&mut buf, &f.name, f.icb_lba, false, false);
|
||||
}
|
||||
buf
|
||||
}
|
||||
|
||||
/// Patch every FID's tag location to the block it actually lands in. ECMA-167
|
||||
/// 3/7.2.2 makes the tag location the block of the descriptor, and a FID that
|
||||
/// spans two blocks records the block it STARTS in.
|
||||
fn fix_fid_tag_locations(buf: &mut [u8], first_block: u32) {
|
||||
let mut pos = 0usize;
|
||||
while pos + 38 <= buf.len() {
|
||||
let l_fi = buf[pos + 19] as usize;
|
||||
let l_iu = u16::from_le_bytes([buf[pos + 36], buf[pos + 37]]) as usize;
|
||||
let unpadded = 38 + l_iu + l_fi;
|
||||
if pos + unpadded > buf.len() {
|
||||
break;
|
||||
}
|
||||
let block = first_block + (pos / SECTOR) as u32;
|
||||
finish_tag(&mut buf[pos..pos + unpadded], 257, block, unpadded);
|
||||
pos += unpadded.div_ceil(4) * 4;
|
||||
}
|
||||
}
|
||||
|
||||
// ── Whole-image assembly ────────────────────────────────────────────────────
|
||||
|
||||
/// Volume-space block of the Volume Recognition Sequence.
|
||||
const VRS_START: u32 = 16;
|
||||
/// Volume-space block of the Main Volume Descriptor Sequence.
|
||||
pub(super) const MAIN_VDS_START: u32 = 32;
|
||||
/// Volume-space block of the Reserve Volume Descriptor Sequence.
|
||||
pub(super) const RESERVE_VDS_START: u32 = 48;
|
||||
/// Sectors reserved for each VDS. ECMA-167 3/10.2.1 requires an anchor to
|
||||
/// record at least 16.
|
||||
pub(super) const VDS_SECTORS: u32 = 16;
|
||||
/// Volume-space block of the Logical Volume Integrity Sequence.
|
||||
pub(super) const LVID_START: u32 = 64;
|
||||
/// Sectors reserved for the integrity sequence (LVID + TD).
|
||||
pub(super) const LVID_SECTORS: u32 = 2;
|
||||
/// The mandatory anchor block (ECMA-167 3/10.2).
|
||||
pub(super) const ANCHOR_LBA: u32 = 256;
|
||||
/// First block a partition may start at. Everything above is volume space.
|
||||
pub(super) const MIN_PART_START: u32 = 320;
|
||||
|
||||
/// Emit the six-descriptor Volume Descriptor Sequence at `start`.
|
||||
fn write_vds(out: &mut MetaSectors, layout: &Layout, start: u32) {
|
||||
let vid = &layout.volume_id;
|
||||
out.insert(start, primary_volume(vid, start, 1));
|
||||
out.insert(start + 1, impl_use_volume(vid, start + 1, 2));
|
||||
out.insert(
|
||||
start + 2,
|
||||
partition(layout.part_start, layout.part_sectors, start + 2, 3),
|
||||
);
|
||||
out.insert(
|
||||
start + 3,
|
||||
logical_volume(vid, 0, LVID_START, LVID_SECTORS, start + 3, 4),
|
||||
);
|
||||
out.insert(start + 4, unallocated_space(start + 4, 5));
|
||||
out.insert(start + 5, terminating(start + 5));
|
||||
}
|
||||
|
||||
/// Recursively emit one directory's File Entry and FID list, then its
|
||||
/// children's.
|
||||
fn write_dir(out: &mut MetaSectors, layout: &Layout, dir: &DirNode) -> Result<()> {
|
||||
let mut fids = dir_fids(dir);
|
||||
fix_fid_tag_locations(&mut fids, dir.data_lba);
|
||||
debug_assert_eq!(fids.len(), dir.data_bytes as usize);
|
||||
|
||||
// A directory's link count is 1 (its own FID in the parent) plus one for
|
||||
// each child directory's parent FID pointing back at it.
|
||||
// The planner caps subdirectory fan-out (`layout::MAX_SUBDIRS`) so this
|
||||
// cannot overflow; saturating rather than wrapping keeps a future change to
|
||||
// that cap from silently producing a wrong count.
|
||||
let link_count = (dir.dirs.len() as u16).saturating_add(1);
|
||||
let fe = file_entry(
|
||||
true,
|
||||
fids.len() as u64,
|
||||
&[(dir.data_lba, fids.len() as u32)],
|
||||
link_count,
|
||||
dir.unique_id,
|
||||
dir.icb_lba,
|
||||
)?;
|
||||
out.insert(layout.part_start + dir.icb_lba, fe);
|
||||
|
||||
for (i, chunk) in fids.chunks(SECTOR).enumerate() {
|
||||
let mut s = blank();
|
||||
s[..chunk.len()].copy_from_slice(chunk);
|
||||
out.insert(layout.part_start + dir.data_lba + i as u32, s);
|
||||
}
|
||||
|
||||
for f in &dir.files {
|
||||
let extents: Vec<(u32, u32)> = f.extents.iter().map(|e| (e.lba, e.bytes)).collect();
|
||||
let fe = file_entry(false, f.size, &extents, 1, f.unique_id, f.icb_lba)?;
|
||||
out.insert(layout.part_start + f.icb_lba, fe);
|
||||
}
|
||||
|
||||
for sub in &dir.dirs {
|
||||
write_dir(out, layout, sub)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Build every metadata sector of the synthesized volume.
|
||||
pub(super) fn encode(layout: &Layout) -> Result<MetaSectors> {
|
||||
let mut out = MetaSectors::new();
|
||||
|
||||
out.insert(VRS_START, volume_recognition(b"BEA01"));
|
||||
out.insert(VRS_START + 1, volume_recognition(b"NSR02"));
|
||||
out.insert(VRS_START + 2, volume_recognition(b"TEA01"));
|
||||
|
||||
write_vds(&mut out, layout, MAIN_VDS_START);
|
||||
write_vds(&mut out, layout, RESERVE_VDS_START);
|
||||
|
||||
out.insert(
|
||||
LVID_START,
|
||||
integrity(
|
||||
layout.part_sectors,
|
||||
layout.file_count,
|
||||
layout.dir_count,
|
||||
layout.next_unique_id,
|
||||
LVID_START,
|
||||
),
|
||||
);
|
||||
out.insert(LVID_START + 1, terminating(LVID_START + 1));
|
||||
|
||||
let avdp = anchor(MAIN_VDS_START, RESERVE_VDS_START, VDS_SECTORS, ANCHOR_LBA);
|
||||
out.insert(ANCHOR_LBA, avdp);
|
||||
let last = layout.total_sectors - 1;
|
||||
out.insert(
|
||||
last,
|
||||
anchor(MAIN_VDS_START, RESERVE_VDS_START, VDS_SECTORS, last),
|
||||
);
|
||||
|
||||
// Partition block 0 must hold the File Set Descriptor: `read_filesystem`
|
||||
// reads exactly `metadata_start` (== partition start on a single-partition
|
||||
// volume) and rejects the volume outright if the tag there is not 256.
|
||||
out.insert(
|
||||
layout.part_start,
|
||||
file_set(&layout.volume_id, layout.root.icb_lba, 0),
|
||||
);
|
||||
out.insert(layout.part_start + 1, terminating(1));
|
||||
|
||||
write_dir(&mut out, layout, &layout.root)?;
|
||||
Ok(out)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// The reference check value for CRC-16/XMODEM — poly 0x1021 seeded at 0,
|
||||
/// which is what ECMA-167 7.2.4 specifies: "123456789" → 0x31C3. Seeding
|
||||
/// at 0xFFFF instead (CCITT-FALSE) yields 0x29B1, and that mutant is
|
||||
/// invisible to `udf.rs`, which never verifies a tag CRC — it would only
|
||||
/// show up as a volume no operating system will mount.
|
||||
#[test]
|
||||
fn crc16_matches_the_ecma167_check_value() {
|
||||
assert_eq!(crc16(b"123456789"), 0x31C3);
|
||||
assert_ne!(crc16(b"123456789"), 0x29B1, "not the 0xFFFF-seeded variant");
|
||||
}
|
||||
|
||||
/// ECMA-167 3/7.2.3: the checksum is the sum of the tag's first 16 bytes
|
||||
/// EXCLUDING the checksum byte itself, modulo 256.
|
||||
#[test]
|
||||
fn tag_checksum_excludes_its_own_byte() {
|
||||
let mut buf = [0u8; 512];
|
||||
buf[16..24].copy_from_slice(&[1, 2, 3, 4, 5, 6, 7, 8]);
|
||||
finish_tag(&mut buf, 261, 0x1234, 512);
|
||||
let sum: u32 = buf[0..16]
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(i, _)| *i != 4)
|
||||
.map(|(_, b)| *b as u32)
|
||||
.sum();
|
||||
assert_eq!(buf[4] as u32, sum % 256);
|
||||
// And the recorded CRC covers the body, not the tag.
|
||||
let crc = u16::from_le_bytes([buf[8], buf[9]]);
|
||||
assert_eq!(crc, crc16(&buf[16..512]));
|
||||
assert_eq!(u16::from_le_bytes([buf[10], buf[11]]), 496);
|
||||
assert_eq!(
|
||||
u32::from_le_bytes([buf[12], buf[13], buf[14], buf[15]]),
|
||||
0x1234
|
||||
);
|
||||
}
|
||||
|
||||
/// ASCII takes compression ID 8; anything above takes 16 (UTF-16BE),
|
||||
/// because `parse_udf_name` decodes compression-8 bytes as UTF-8.
|
||||
#[test]
|
||||
fn cs0_picks_the_encoding_the_parser_can_decode() {
|
||||
assert_eq!(encode_cs0("AB"), vec![8, b'A', b'B']);
|
||||
let e = encode_cs0("Ä");
|
||||
assert_eq!(e[0], 16);
|
||||
assert_eq!(&e[1..], &[0x00, 0xC4]);
|
||||
assert_eq!(crate::udf::parse_udf_name(&e), "Ä");
|
||||
}
|
||||
|
||||
/// A d-string records its used length in the field's LAST byte, and the
|
||||
/// production parser must read the same string back.
|
||||
#[test]
|
||||
fn dstring_round_trips_through_the_production_parser() {
|
||||
let mut field = [0u8; 32];
|
||||
put_dstring(&mut field, "FREEMKV");
|
||||
assert_eq!(field[31], 8, "compid byte + 7 characters");
|
||||
assert_eq!(crate::udf::parse_dstring_for_test(&field), "FREEMKV");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,362 @@
|
||||
//! `dir://` as an image-level SOURCE: a synthetic UDF volume over a folder.
|
||||
//!
|
||||
//! A user's extracted disc — a DVD `VIDEO_TS/` or a Blu-ray `BDMV/`, typically
|
||||
//! a MakeMKV-style backup — has files but no sectors, and everything above the
|
||||
//! sector layer in this crate wants sectors: `Disc::scan_image`, `UdfFs`,
|
||||
//! `ifo.rs`, `mpls.rs`, `clpi.rs` and the mux all read through a
|
||||
//! [`SectorSource`]. [`DirImage`] supplies one.
|
||||
//!
|
||||
//! The trick is that nothing is emulated. A real, minimal, valid UDF 1.02
|
||||
//! volume is synthesized over the folder:
|
||||
//!
|
||||
//! * **Metadata sectors** (anchors, the volume descriptor sequences, the File
|
||||
//! Set Descriptor, every File Entry, every directory's FID list) are encoded
|
||||
//! into RAM by [`encode`] — a few MiB even for a large Blu-ray.
|
||||
//! * **Data sectors** are not materialized at all. Each one maps to a byte
|
||||
//! range of a real file, read on demand.
|
||||
//!
|
||||
//! So `udf::read_filesystem` parses this image by exactly the same code path it
|
||||
//! parses a real disc with, and every consumer above it is unchanged. The cost
|
||||
//! is that a single-partition synthetic volume never exercises the UDF 2.50
|
||||
//! Metadata Partition path (`udf.rs:946-991`) that every real BD-ROM uses —
|
||||
//! this module's tests do not cover that block and must not be read as if they
|
||||
//! did.
|
||||
//!
|
||||
//! What this module deliberately does NOT do:
|
||||
//!
|
||||
//! * **3D / SSIF** — rejected up front ([`Error::DirImageSsifUnsupported`]).
|
||||
//! An SSIF aliases the same sectors as its base and dependent `.m2ts`; the
|
||||
//! planner allocates disjoint extents, so a 3D folder would produce silently
|
||||
//! wrong output.
|
||||
//! * **HD-DVD `HVDVD_TS/`** — no title enumerator constraint is modelled.
|
||||
//! * **Encrypted folders** — a folder whose content is still AACS-scrambled is
|
||||
//! rejected by the caller-side probe, not decrypted here.
|
||||
|
||||
mod encode;
|
||||
mod layout;
|
||||
|
||||
use crate::error::{Error, Result};
|
||||
#[cfg(target_os = "linux")]
|
||||
use crate::io::file_sector_source::linux::drop_window;
|
||||
#[cfg(target_os = "macos")]
|
||||
use crate::io::file_sector_source::macos::drop_window;
|
||||
#[cfg(not(any(target_os = "linux", target_os = "macos", target_os = "windows")))]
|
||||
use crate::io::file_sector_source::other::drop_window;
|
||||
#[cfg(target_os = "windows")]
|
||||
use crate::io::file_sector_source::windows::drop_window;
|
||||
use crate::sector::SectorSource;
|
||||
use encode::{MetaSectors, SECTOR};
|
||||
use std::fs::File;
|
||||
use std::io::{Read, Seek, SeekFrom};
|
||||
use std::path::{Path, PathBuf};
|
||||
|
||||
/// How many host files may be held open at once.
|
||||
///
|
||||
/// A Blu-ray `BDMV/` can exceed a thousand files while macOS `RLIMIT_NOFILE`
|
||||
/// defaults to 256, so "open every file up front" is not available. Reads are
|
||||
/// overwhelmingly sequential through one large stream file at a time, so a
|
||||
/// small LRU keeps the hit rate near 1 while bounding descriptors.
|
||||
const HANDLE_CACHE: usize = 16;
|
||||
|
||||
/// One file's bytes at one place in the image.
|
||||
#[derive(Debug, Clone)]
|
||||
struct DataRange {
|
||||
/// Absolute first block.
|
||||
start_lba: u32,
|
||||
/// Blocks covered (the last one may be partially used, and is zero-padded).
|
||||
sectors: u32,
|
||||
/// Index into [`DirImage::files`].
|
||||
file: usize,
|
||||
/// Byte offset within the file at which this range's bytes begin.
|
||||
offset: u64,
|
||||
/// Byte length of the range.
|
||||
bytes: u64,
|
||||
}
|
||||
|
||||
/// A file the image reads through.
|
||||
#[derive(Debug)]
|
||||
struct FileRef {
|
||||
host: PathBuf,
|
||||
disc_path: String,
|
||||
size: u64,
|
||||
/// Host mtime at plan time — see `layout::FileNode::mtime` for why size
|
||||
/// alone is not enough.
|
||||
mtime: Option<std::time::SystemTime>,
|
||||
}
|
||||
|
||||
/// A synthesized UDF disc image over a host directory.
|
||||
///
|
||||
/// Owns everything it reads through (`PathBuf`s and its own file handles), so
|
||||
/// it is `Send + 'static` and can be moved into `build_iso_pipeline`, which
|
||||
/// hands it to `PrefetchedSectorSource`'s producer thread.
|
||||
pub struct DirImage {
|
||||
meta: MetaSectors,
|
||||
/// Sorted by `start_lba`, non-overlapping.
|
||||
ranges: Vec<DataRange>,
|
||||
files: Vec<FileRef>,
|
||||
open: Vec<(usize, File)>,
|
||||
total_sectors: u32,
|
||||
volume_id: String,
|
||||
data_bytes: u64,
|
||||
}
|
||||
|
||||
impl std::fmt::Debug for DirImage {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("DirImage")
|
||||
.field("volume_id", &self.volume_id)
|
||||
.field("total_sectors", &self.total_sectors)
|
||||
.field("files", &self.files.len())
|
||||
.field("meta_sectors", &self.meta.len())
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl DirImage {
|
||||
/// Plan and encode an image over `root`.
|
||||
///
|
||||
/// Every error is decided here, at plan time, where it can name the file
|
||||
/// responsible — the read path is deliberately left with nothing to decide
|
||||
/// except "this file changed underneath me".
|
||||
pub fn open(root: &Path) -> Result<Self> {
|
||||
let plan = layout::plan(root)?;
|
||||
let meta = encode::encode(&plan)?;
|
||||
|
||||
let mut nodes = Vec::new();
|
||||
layout::flatten(&plan.root, &mut nodes);
|
||||
|
||||
let mut files = Vec::with_capacity(nodes.len());
|
||||
let mut ranges = Vec::new();
|
||||
for (idx, node) in nodes.iter().enumerate() {
|
||||
// Carry the plan-time mtime ONLY for files whose CONTENT the plan
|
||||
// read — the DVD IFOs, whose bytes 0xC0/0xC4 decide where every VOB
|
||||
// is placed (`layout::place_video_ts` -> `read_head`).
|
||||
//
|
||||
// For every other file the plan depends on the SIZE alone, and size
|
||||
// is already checked. Comparing mtime on those buys nothing and
|
||||
// costs real false positives: disc backups commonly live on
|
||||
// exFAT/FAT32, which stores local time, so a long rip spanning a
|
||||
// DST transition sees a whole-hour shift on a file nobody touched
|
||||
// and would abort hours in, blaming a change that did not happen.
|
||||
// The multi-gigabyte VOBs are exactly the files a long rip re-opens
|
||||
// after the handle cache evicts them.
|
||||
let content_sensitive = node
|
||||
.disc_path
|
||||
.rsplit('.')
|
||||
.next()
|
||||
.is_some_and(|e| e.eq_ignore_ascii_case("IFO"));
|
||||
files.push(FileRef {
|
||||
host: node.host.clone(),
|
||||
disc_path: node.disc_path.clone(),
|
||||
size: node.size,
|
||||
mtime: content_sensitive.then_some(node.mtime).flatten(),
|
||||
});
|
||||
let mut offset = 0u64;
|
||||
for e in &node.extents {
|
||||
ranges.push(DataRange {
|
||||
start_lba: plan.part_start + e.lba,
|
||||
sectors: (e.bytes as u64).div_ceil(SECTOR as u64) as u32,
|
||||
file: idx,
|
||||
offset,
|
||||
bytes: e.bytes as u64,
|
||||
});
|
||||
offset += e.bytes as u64;
|
||||
}
|
||||
}
|
||||
ranges.sort_by_key(|r| r.start_lba);
|
||||
debug_assert!(
|
||||
ranges
|
||||
.windows(2)
|
||||
.all(|w| w[0].start_lba + w[0].sectors <= w[1].start_lba),
|
||||
"planned data ranges must not overlap"
|
||||
);
|
||||
|
||||
let data_bytes = layout::total_data_bytes(&plan.root);
|
||||
tracing::info!(
|
||||
target: "freemkv::dirimage",
|
||||
volume_id = %plan.volume_id,
|
||||
files = files.len(),
|
||||
dirs = plan.dir_count,
|
||||
meta_blocks = layout::metadata_block_count(&plan.root),
|
||||
total_sectors = plan.total_sectors,
|
||||
"synthesized UDF image over directory"
|
||||
);
|
||||
|
||||
Ok(Self {
|
||||
meta,
|
||||
ranges,
|
||||
files,
|
||||
open: Vec::new(),
|
||||
total_sectors: plan.total_sectors,
|
||||
volume_id: plan.volume_id,
|
||||
data_bytes,
|
||||
})
|
||||
}
|
||||
|
||||
/// UDF volume identifier the image declares (the folder's own name).
|
||||
pub fn volume_id(&self) -> &str {
|
||||
&self.volume_id
|
||||
}
|
||||
|
||||
/// Total bytes of real file content the image carries — the folder's size,
|
||||
/// not the image's (which also counts metadata and inter-file gaps).
|
||||
pub fn data_bytes(&self) -> u64 {
|
||||
self.data_bytes
|
||||
}
|
||||
|
||||
/// The range covering `lba`, if any.
|
||||
fn range_at(&self, lba: u32) -> Option<&DataRange> {
|
||||
let i = self.ranges.partition_point(|r| r.start_lba <= lba);
|
||||
let r = self.ranges.get(i.checked_sub(1)?)?;
|
||||
(lba < r.start_lba + r.sectors).then_some(r)
|
||||
}
|
||||
|
||||
/// Borrow an open handle for `file`, opening it (and evicting the
|
||||
/// least-recently-used handle) if necessary.
|
||||
///
|
||||
/// Opening is also where the plan is revalidated. A folder is not a disc:
|
||||
/// a file can be shortened or replaced between planning and reading, and
|
||||
/// zero-filling the difference would turn "the user deleted something"
|
||||
/// into corrupt output at exit 0. The size is re-checked here, and a
|
||||
/// truncation that happens while the handle is already open is caught by
|
||||
/// the short read in [`Self::fill`].
|
||||
fn handle(&mut self, file: usize) -> Result<&mut File> {
|
||||
if let Some(pos) = self.open.iter().position(|(i, _)| *i == file) {
|
||||
// `open` is ordered most-recently-used first.
|
||||
let entry = self.open.remove(pos);
|
||||
self.open.insert(0, entry);
|
||||
return Ok(&mut self.open[0].1);
|
||||
}
|
||||
let f = File::open(&self.files[file].host).map_err(Error::from)?;
|
||||
let md = f.metadata().map_err(Error::from)?;
|
||||
// Size AND mtime. Size alone is content-blind, and this plan depends on
|
||||
// content: a DVD's VOB placement comes from bytes 0xC0/0xC4 of its IFO,
|
||||
// and an IFO rewritten in place keeps its length because IFOs occupy a
|
||||
// whole number of sectors. The size check would pass while every title
|
||||
// extent pointed at the wrong sectors — corrupt video behind an intact
|
||||
// structure, reported complete at exit 0.
|
||||
//
|
||||
// Only compared when both sides have a timestamp; a platform or
|
||||
// filesystem that reports none simply falls back to the size check
|
||||
// rather than failing every read.
|
||||
let changed_size = md.len() != self.files[file].size;
|
||||
let changed_mtime = match (self.files[file].mtime, md.modified().ok()) {
|
||||
(Some(planned), Some(live)) => planned != live,
|
||||
_ => false,
|
||||
};
|
||||
if changed_size || changed_mtime {
|
||||
return Err(Error::DirImageFileChanged {
|
||||
path: self.files[file].disc_path.clone(),
|
||||
});
|
||||
}
|
||||
if self.open.len() >= HANDLE_CACHE {
|
||||
self.open.pop();
|
||||
}
|
||||
self.open.insert(0, (file, f));
|
||||
Ok(&mut self.open[0].1)
|
||||
}
|
||||
|
||||
/// Fill `out` (a whole number of sectors) from one data range, starting at
|
||||
/// `lba`. `out` is already zeroed, so a file's tail sector comes back
|
||||
/// zero-padded — which is exactly what `file_extents`' `div_ceil(2048)`
|
||||
/// (`udf.rs:816`) makes every consumer expect.
|
||||
fn fill(&mut self, r: &DataRange, lba: u32, out: &mut [u8]) -> Result<()> {
|
||||
let within = (lba - r.start_lba) as u64 * SECTOR as u64;
|
||||
let want = (r.bytes.saturating_sub(within)).min(out.len() as u64) as usize;
|
||||
if want == 0 {
|
||||
return Ok(());
|
||||
}
|
||||
let at = r.offset + within;
|
||||
let file = r.file;
|
||||
let h = self.handle(file)?;
|
||||
h.seek(SeekFrom::Start(at)).map_err(Error::from)?;
|
||||
let res = h.read_exact(&mut out[..want]);
|
||||
if res.is_ok() {
|
||||
// Release the window just read, every time.
|
||||
//
|
||||
// The ISO source accumulates and drops in chunks because it reads
|
||||
// one file linearly, so a running start offset always names the
|
||||
// bytes it has consumed. Reads here jump between files, so there is
|
||||
// no single cursor to accumulate against — an accumulated byte
|
||||
// count paired with one read's offset names 1/Nth of what was
|
||||
// actually consumed and leaves the rest pinned, which is how the
|
||||
// first version of this got it wrong.
|
||||
//
|
||||
// Dropping per read costs one advisory syscall per batch (4-16 MiB),
|
||||
// which is nothing against the read itself, and it is correct
|
||||
// regardless of how reads interleave across files.
|
||||
if let Some((_, fh)) = self.open.iter().find(|(i, _)| *i == file) {
|
||||
drop_window(fh, at, want as u64);
|
||||
}
|
||||
}
|
||||
match res {
|
||||
Ok(()) => Ok(()),
|
||||
// The file shrank while the handle was open. Same verdict as the
|
||||
// size check in `handle`, reached the other way.
|
||||
Err(e) if e.kind() == std::io::ErrorKind::UnexpectedEof => {
|
||||
Err(Error::DirImageFileChanged {
|
||||
path: self.files[file].disc_path.clone(),
|
||||
})
|
||||
}
|
||||
Err(e) => Err(Error::from(e)),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl SectorSource for DirImage {
|
||||
fn capacity_sectors(&self) -> u32 {
|
||||
self.total_sectors
|
||||
}
|
||||
|
||||
fn read_sectors(
|
||||
&mut self,
|
||||
lba: u32,
|
||||
count: u16,
|
||||
buf: &mut [u8],
|
||||
_recovery: bool,
|
||||
) -> Result<usize> {
|
||||
let need = count as usize * SECTOR;
|
||||
if buf.len() < need {
|
||||
return Err(Error::UdfBufferTooSmall);
|
||||
}
|
||||
buf[..need].fill(0);
|
||||
// Walk the request in RUNS, not sector by sector. A mux batch is 8192
|
||||
// sectors and almost always lands entirely inside one stream file's
|
||||
// extent; per-sector seek+read would issue 8192 syscalls for what is
|
||||
// one 16 MiB sequential read.
|
||||
let mut i = 0u32;
|
||||
while i < count as u32 {
|
||||
// Checked: callers saturate their LBAs (`disc/dvd.rs` builds a cell
|
||||
// start as `vob_start_sector.saturating_add(cell.first_sector)`, and
|
||||
// the prefetcher adds an offset the same way), so a crafted IFO can
|
||||
// present a request at the very top of the address space. Wrapping
|
||||
// here would fold `at` back to a LOW sector and hand the muxer a
|
||||
// different file's bytes with nothing reported.
|
||||
let Some(at) = lba.checked_add(i) else {
|
||||
break;
|
||||
};
|
||||
let off = i as usize * SECTOR;
|
||||
if let Some(s) = self.meta.get(&at) {
|
||||
buf[off..off + SECTOR].copy_from_slice(&s[..]);
|
||||
i += 1;
|
||||
continue;
|
||||
}
|
||||
// Metadata blocks all sit below the data floor, so a data range is
|
||||
// never interrupted by one.
|
||||
match self.range_at(at).cloned() {
|
||||
Some(r) => {
|
||||
let run = (r.start_lba + r.sectors - at).min(count as u32 - i);
|
||||
let end = off + run as usize * SECTOR;
|
||||
self.fill(&r, at, &mut buf[off..end])?;
|
||||
i += run;
|
||||
}
|
||||
// A gap between planned extents. Reads as zeros, exactly as an
|
||||
// unrecorded sector of a real image does.
|
||||
None => i += 1,
|
||||
}
|
||||
}
|
||||
Ok(need)
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests;
|
||||
File diff suppressed because it is too large
Load Diff
+1377
-121
File diff suppressed because it is too large
Load Diff
+214
-22
@@ -7,13 +7,32 @@ use crate::udf;
|
||||
|
||||
impl Disc {
|
||||
/// Scan DVD titles from IFO files (VIDEO_TS.IFO + VTS_XX_0.IFO).
|
||||
///
|
||||
/// Cancellation: `halt` is polled before the IFO tree is read, and an IFO
|
||||
/// read that fails with [`Error::Halted`] — how a live drive reports a
|
||||
/// Stop, since `Drive::checked_exec` fails EVERY command once its flag is
|
||||
/// set — is propagated rather than swallowed. Every other IFO failure
|
||||
/// keeps its best-effort `Ok(vec![])`.
|
||||
///
|
||||
/// It has to be an error and not an empty title list, for the same reason
|
||||
/// spelled out on [`Disc::scan_hddvd_titles`]: a cancelled enumeration
|
||||
/// that returned `Ok` would be indistinguishable from a disc that
|
||||
/// genuinely holds fewer titles. This one was the worst of the three
|
||||
/// enumerators — a bare `Err(_) => return Vec::new()` turned an operator
|
||||
/// Stop into ZERO titles at rc=0, a disc reported as carrying no video at
|
||||
/// all.
|
||||
pub(super) fn scan_dvd_titles(
|
||||
reader: &mut dyn SectorSource,
|
||||
udf_fs: &udf::UdfFs,
|
||||
) -> Vec<DiscTitle> {
|
||||
halt: Option<&crate::halt::Halt>,
|
||||
) -> Result<Vec<DiscTitle>> {
|
||||
if halt.is_some_and(|h| h.is_cancelled()) {
|
||||
return Err(Error::Halted);
|
||||
}
|
||||
let dvd_info = match ifo::parse_vmg(reader, udf_fs) {
|
||||
Ok(info) => info,
|
||||
Err(_) => return Vec::new(),
|
||||
Err(Error::Halted) => return Err(Error::Halted),
|
||||
Err(_) => return Ok(Vec::new()),
|
||||
};
|
||||
|
||||
let mut titles = Vec::new();
|
||||
@@ -179,7 +198,10 @@ impl Disc {
|
||||
// the coded video frame the subpicture was authored against
|
||||
// (720x480 NTSC / 720x576 PAL) so players place and scale the
|
||||
// bitmap correctly.
|
||||
let (vid_w, vid_h) = ts.video.resolution.pixels();
|
||||
// format_palette guards on (0, 0) and omits its `size:` line,
|
||||
// so an unresolved resolution degrades to a palette-only .idx
|
||||
// rather than one claiming a 0x0 frame.
|
||||
let (vid_w, vid_h) = ts.video.resolution.pixels().unwrap_or((0, 0));
|
||||
let codec_data = dvd_title
|
||||
.palette
|
||||
.as_ref()
|
||||
@@ -240,7 +262,14 @@ impl Disc {
|
||||
}
|
||||
}
|
||||
|
||||
titles
|
||||
// Polled again AFTER the loop: a cancel raised during the IFO reads
|
||||
// that `parse_vmg` performs per title set has nothing left to poll,
|
||||
// so without this a partially enumerated disc could still be handed
|
||||
// back as success.
|
||||
if halt.is_some_and(|h| h.is_cancelled()) {
|
||||
return Err(Error::Halted);
|
||||
}
|
||||
Ok(titles)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -525,6 +554,118 @@ mod tests {
|
||||
// Tests
|
||||
// ---------------------------------------------------------------
|
||||
|
||||
/// A `SectorSource` that fails every read at or above `halt_at` with
|
||||
/// [`Error::Halted`] — how a LIVE DRIVE behaves once the operator presses
|
||||
/// Stop: `Drive::checked_exec` fails every SCSI command with `Halted` from
|
||||
/// then on, and `Drive::read` deliberately preserves the variant. Reads
|
||||
/// below the threshold still succeed, so the scan gets far enough to have
|
||||
/// something to truncate.
|
||||
struct HaltingReader<'a> {
|
||||
inner: &'a mut MemDisc,
|
||||
halt_at: u32,
|
||||
}
|
||||
impl SectorSource for HaltingReader<'_> {
|
||||
fn read_sectors(
|
||||
&mut self,
|
||||
lba: u32,
|
||||
count: u16,
|
||||
buf: &mut [u8],
|
||||
recovery: bool,
|
||||
) -> crate::error::Result<usize> {
|
||||
if lba >= self.halt_at {
|
||||
return Err(crate::error::Error::Halted);
|
||||
}
|
||||
self.inner.read_sectors(lba, count, buf, recovery)
|
||||
}
|
||||
}
|
||||
|
||||
/// A Stop on a LIVE DRIVE never touches `ScanOptions::halt`: `Drive` has
|
||||
/// its own flag and `checked_exec` fails every SCSI command with
|
||||
/// [`Error::Halted`] once it is set. The DVD enumerator must not swallow
|
||||
/// that into a successful scan.
|
||||
///
|
||||
/// This was the worst of the three enumerators. RED BEFORE GREEN, two
|
||||
/// distinct swallows, both measured with the fix reverted:
|
||||
/// * `ifo::parse_vmg` treats a failed title set as a placeholder entry
|
||||
/// and continues, so a cancel landing on VTS_02's IFO returned
|
||||
/// `Ok([VTS_01_1.VOB])` — one title from a two-title disc.
|
||||
/// * `scan_dvd_titles`'s `Err(_) => return Vec::new()` turned a cancel
|
||||
/// landing on VIDEO_TS.IFO itself into ZERO titles at rc=0 — a disc
|
||||
/// reported as holding no video at all.
|
||||
///
|
||||
/// Both are indistinguishable from a real disc, and both are now
|
||||
/// `Err(Error::Halted)`.
|
||||
#[test]
|
||||
fn halted_ifo_read_is_not_reported_as_a_shorter_disc() {
|
||||
// Two title sets: VTS_01's IFO data at PART_START+6000, VTS_02's at
|
||||
// PART_START+7000. Both ICBs sit far below, so the filesystem
|
||||
// metadata resolves and only the second title set's CONTENT is
|
||||
// cancelled — the truncation case.
|
||||
let mut disc = MemDisc::new();
|
||||
let vmg = build_vmg(&[(1, 1, 1), (1, 2, 1)]);
|
||||
let vts1 = build_vts(100, 0x00, &[], &[], &[(0, 9)], false);
|
||||
let vts2 = build_vts(200, 0x00, &[], &[], &[(0, 19)], false);
|
||||
let udf = build_video_ts_fs(
|
||||
&mut disc,
|
||||
&[
|
||||
FileSpec {
|
||||
name: "VIDEO_TS.IFO".into(),
|
||||
icb_lba: 60,
|
||||
data_lba: 5000,
|
||||
contents: vmg,
|
||||
},
|
||||
FileSpec {
|
||||
name: "VTS_01_0.IFO".into(),
|
||||
icb_lba: 62,
|
||||
data_lba: 6000,
|
||||
contents: vts1,
|
||||
},
|
||||
FileSpec {
|
||||
name: "VTS_02_0.IFO".into(),
|
||||
icb_lba: 64,
|
||||
data_lba: 7000,
|
||||
contents: vts2,
|
||||
},
|
||||
],
|
||||
);
|
||||
// Sanity: both title sets enumerate when nothing is cancelled, so a
|
||||
// short list below can only be the cancel.
|
||||
assert_eq!(
|
||||
Disc::scan_dvd_titles(&mut disc, &udf, None)
|
||||
.expect("scan")
|
||||
.len(),
|
||||
2,
|
||||
"fixture must offer two title sets"
|
||||
);
|
||||
|
||||
let mut reader = HaltingReader {
|
||||
inner: &mut disc,
|
||||
halt_at: PART_START + 7000, // VTS_02_0.IFO's data extent
|
||||
};
|
||||
let res = Disc::scan_dvd_titles(&mut reader, &udf, None);
|
||||
assert!(
|
||||
matches!(res, Err(crate::error::Error::Halted)),
|
||||
"a cancelled title-set read must surface as a cancelled scan, not \
|
||||
as a disc with fewer titles; got {:?}",
|
||||
res.map(|ts| ts.iter().map(|t| t.playlist.clone()).collect::<Vec<_>>())
|
||||
);
|
||||
|
||||
// The same cancel one level up: VIDEO_TS.IFO itself. This is the
|
||||
// `Err(_) => Vec::new()` path — a cancel that used to report a DVD as
|
||||
// carrying no titles whatsoever.
|
||||
let mut reader = HaltingReader {
|
||||
inner: &mut disc,
|
||||
halt_at: PART_START + 5000,
|
||||
};
|
||||
let res = Disc::scan_dvd_titles(&mut reader, &udf, None);
|
||||
assert!(
|
||||
matches!(res, Err(crate::error::Error::Halted)),
|
||||
"a cancelled VMG read must surface as a cancelled scan, not as an \
|
||||
empty disc; got {:?}",
|
||||
res.map(|ts| ts.len())
|
||||
);
|
||||
}
|
||||
|
||||
/// scan_dvd_titles returns empty when VIDEO_TS.IFO can't be parsed
|
||||
/// (dvd.rs: `parse_vmg(...) Err → return Vec::new()`). Never panics.
|
||||
#[test]
|
||||
@@ -532,7 +673,11 @@ mod tests {
|
||||
let mut disc = MemDisc::new();
|
||||
// VIDEO_TS exists but VIDEO_TS.IFO is missing.
|
||||
let udf = build_video_ts_fs(&mut disc, &[]);
|
||||
assert!(Disc::scan_dvd_titles(&mut disc, &udf).is_empty());
|
||||
assert!(
|
||||
Disc::scan_dvd_titles(&mut disc, &udf, None)
|
||||
.expect("scan")
|
||||
.is_empty()
|
||||
);
|
||||
}
|
||||
|
||||
/// Single VTS, single title, one cell. Extent absolute LBA =
|
||||
@@ -568,7 +713,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let titles = Disc::scan_dvd_titles(&mut disc, &udf);
|
||||
let titles = Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan");
|
||||
assert_eq!(titles.len(), 1);
|
||||
let t = &titles[0];
|
||||
assert_eq!(t.extents.len(), 1);
|
||||
@@ -624,7 +769,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let titles = Disc::scan_dvd_titles(&mut disc, &udf);
|
||||
let titles = Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan");
|
||||
assert_eq!(titles.len(), 1);
|
||||
let t = &titles[0];
|
||||
assert_eq!(t.extents.len(), 1);
|
||||
@@ -682,7 +827,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
assert_eq!(t.extents.len(), 1);
|
||||
let got = t.extents[0].start_lba;
|
||||
// The one correct answer: all three terms summed (9000 + 700 + 33).
|
||||
@@ -740,7 +885,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
assert_eq!(t.extents.len(), 2);
|
||||
assert_eq!(t.extents[0].start_lba, 9500); // ifo_lba(9000) + 500 + 0
|
||||
assert_eq!(t.extents[0].sector_count, 100);
|
||||
@@ -781,7 +926,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
let v = t
|
||||
.streams
|
||||
.iter()
|
||||
@@ -836,7 +981,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
let v = t
|
||||
.streams
|
||||
.iter()
|
||||
@@ -897,7 +1042,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
let audios: Vec<_> = t
|
||||
.streams
|
||||
.iter()
|
||||
@@ -908,7 +1053,7 @@ mod tests {
|
||||
.collect();
|
||||
assert_eq!(audios.len(), 2);
|
||||
assert_eq!(audios[0].codec, Codec::Ac3);
|
||||
assert_eq!(audios[0].language, "en");
|
||||
assert_eq!(audios[0].language, "eng");
|
||||
assert_eq!(audios[1].codec, Codec::Dts);
|
||||
// Real channel layouts survive the scan (not a 1ch placeholder): the
|
||||
// AC-3 is 5.1 (6ch), the DTS is 2.0 (2ch).
|
||||
@@ -967,7 +1112,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
let audios: Vec<_> = t
|
||||
.streams
|
||||
.iter()
|
||||
@@ -1024,7 +1169,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
let subs: Vec<_> = t
|
||||
.streams
|
||||
.iter()
|
||||
@@ -1037,7 +1182,7 @@ mod tests {
|
||||
// Languages preserved in order.
|
||||
assert_eq!(
|
||||
subs.iter().map(|s| s.language.as_str()).collect::<Vec<_>>(),
|
||||
vec!["en", "fr", "de"]
|
||||
vec!["eng", "fra", "deu"]
|
||||
);
|
||||
// PIDs are 0x20 + ordinal, all distinct.
|
||||
let pids: Vec<u16> = subs.iter().map(|s| s.pid).collect();
|
||||
@@ -1084,7 +1229,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
let sub = t
|
||||
.streams
|
||||
.iter()
|
||||
@@ -1094,7 +1239,7 @@ mod tests {
|
||||
})
|
||||
.expect("subtitle stream");
|
||||
assert_eq!(sub.codec, Codec::DvdSub);
|
||||
assert_eq!(sub.language, "en");
|
||||
assert_eq!(sub.language, "eng");
|
||||
assert!(
|
||||
sub.codec_data.is_some(),
|
||||
"non-zero palette must yield codec_data"
|
||||
@@ -1138,7 +1283,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let titles = Disc::scan_dvd_titles(&mut disc, &udf);
|
||||
let titles = Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan");
|
||||
assert_eq!(titles.len(), 2);
|
||||
// title_number is a running counter across all title sets.
|
||||
assert_eq!(titles[0].playlist_id, 1);
|
||||
@@ -1175,7 +1320,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
// One program in the program map → one chapter time (0.0 for the
|
||||
// first program). Name is the ordinal from chapter_name(0).
|
||||
assert_eq!(t.chapters.len(), 1);
|
||||
@@ -1260,7 +1405,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
// The leading 0x90 cell is dropped: 2 feature extents, not 3.
|
||||
assert_eq!(t.extents.len(), 2, "leading angle sub-block cell dropped");
|
||||
// First extent starts at the feature cell (vob 1000 + 100), not at 1000+0.
|
||||
@@ -1312,7 +1457,7 @@ mod tests {
|
||||
},
|
||||
],
|
||||
);
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf)[0];
|
||||
let t = &Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan")[0];
|
||||
// Nothing dropped: both cells become extents, starting at the very head.
|
||||
assert_eq!(t.extents.len(), 2);
|
||||
assert_eq!(t.extents[0].start_lba, 9000 + 1000); // ifo_lba + vtstt + 0, head intact
|
||||
@@ -1320,4 +1465,51 @@ mod tests {
|
||||
// Chapter 0 stays at 0.0 (no shift).
|
||||
assert!((t.chapters[0].time_secs - 0.0).abs() < 0.01);
|
||||
}
|
||||
|
||||
/// Audio PID fallback (dvd.rs `Disc::scan_dvd_titles`): when an audio
|
||||
/// stream has no on-wire private_stream_1 sub-stream id — MP1/MP2 audio,
|
||||
/// per `ifo::assign_audio_sub_stream_ids` — the PID falls back to
|
||||
/// `0xBD00 + i` where `i` is the stream's positional index in the IFO
|
||||
/// audio-attribute table. Two MPEG-audio (coding_mode 2) streams must
|
||||
/// land on two DISTINCT, correctly-offset PIDs: 0xBD00 and 0xBD01. This
|
||||
/// pins the `+` (not `-`/`*`) so the second stream doesn't collide with,
|
||||
/// or wrap under, the first.
|
||||
#[test]
|
||||
fn scan_dvd_titles_mp2_audio_pid_fallback_is_additive() {
|
||||
let mut disc = MemDisc::new();
|
||||
let vmg = build_vmg(&[(1, 1, 1)]);
|
||||
// coding_mode bits are b0>>5 & 0x7; mode 2 = MPEG-1 Layer II (Mp2),
|
||||
// which `assign_audio_sub_stream_ids` leaves at `sub_stream_id: None`.
|
||||
// b0 = 0b010_00000 = 0x40. b1 = 0 (mono, sample rate 48k).
|
||||
let audio = [(0x40u8, 0x00u8, [0u8, 0u8]), (0x40u8, 0x00u8, [0u8, 0u8])];
|
||||
let vts = build_vts(1000, 0x00, &audio, &[], &[(10, 109)], false);
|
||||
let udf = build_video_ts_fs(
|
||||
&mut disc,
|
||||
&[
|
||||
FileSpec {
|
||||
name: "VIDEO_TS.IFO".into(),
|
||||
icb_lba: 60,
|
||||
data_lba: 5000,
|
||||
contents: vmg,
|
||||
},
|
||||
FileSpec {
|
||||
name: "VTS_01_0.IFO".into(),
|
||||
icb_lba: 62,
|
||||
data_lba: 6000,
|
||||
contents: vts,
|
||||
},
|
||||
],
|
||||
);
|
||||
let titles = Disc::scan_dvd_titles(&mut disc, &udf, None).expect("scan");
|
||||
let t = &titles[0];
|
||||
let audio_pids: Vec<u16> = t
|
||||
.streams
|
||||
.iter()
|
||||
.filter_map(|s| match s {
|
||||
Stream::Audio(a) => Some(a.pid),
|
||||
_ => None,
|
||||
})
|
||||
.collect();
|
||||
assert_eq!(audio_pids, vec![0xBD00u16, 0xBD01u16]);
|
||||
}
|
||||
}
|
||||
|
||||
+173
-5
@@ -104,10 +104,10 @@ fn max_substream_channels(data: &[u8]) -> Option<u8> {
|
||||
};
|
||||
let start = pos + rel;
|
||||
let frame = &data[start..];
|
||||
if let Some(ch) = ac3::acmod_channels(frame) {
|
||||
if ch > 0 {
|
||||
best = Some(best.map_or(ch, |b| b.max(ch)));
|
||||
}
|
||||
if let Some(ch) = ac3::acmod_channels(frame)
|
||||
&& ch > 0
|
||||
{
|
||||
best = Some(best.map_or(ch, |b| b.max(ch)));
|
||||
}
|
||||
// Advance past this frame by its declared size when that is mappable;
|
||||
// otherwise step 2 bytes past the sync and re-scan for the next one.
|
||||
@@ -242,7 +242,11 @@ pub fn probe_and_remap<S: SectorSource + ?Sized>(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::disc::{AudioChannels, AudioStream, Codec, LabelPurpose, SampleRate};
|
||||
use crate::disc::{
|
||||
AudioChannels, AudioStream, Codec, ContentFormat, DiscTitle, Extent, LabelPurpose,
|
||||
SampleRate,
|
||||
};
|
||||
use crate::sector::SectorSource;
|
||||
|
||||
/// Build a single, correctly-SIZED AC-3 frame whose `acmod`/`lfeon` encode a
|
||||
/// known channel count. `byte4` is `fscod=0 | frmsizecod=0`, so
|
||||
@@ -463,4 +467,168 @@ mod tests {
|
||||
};
|
||||
assert_eq!(a.pid, 0xBD80, "no probe data → keep ordinal");
|
||||
}
|
||||
|
||||
/// `max_substream_channels` must locate the sync at its true ABSOLUTE
|
||||
/// position (`pos + rel`) when it is preceded by non-sync bytes, not just
|
||||
/// when the sync sits at offset 0. Regression guard for a hand-checked
|
||||
/// mutation (`+` → `-` at the `pos + rel` offset computation): with `pos`
|
||||
/// starting at 0 and the first sync found 3 bytes in, `pos - rel` would
|
||||
/// underflow a `usize` and panic, or (if it somehow didn't) index the
|
||||
/// wrong start entirely. `pos + rel` is the only computation that is
|
||||
/// always in-bounds, since `rel` is itself bounded by the length of the
|
||||
/// slice searched from `pos`.
|
||||
#[test]
|
||||
fn max_substream_channels_locates_sync_after_leading_non_sync_bytes() {
|
||||
let mut data = vec![0xAA, 0xAA, 0xAA]; // no 0x0B77 pattern in here
|
||||
data.extend(ac3_frame(2, false)); // real 2.0 frame, sync at absolute offset 3
|
||||
assert_eq!(
|
||||
max_substream_channels(&data),
|
||||
Some(2),
|
||||
"must find and decode the frame whose sync is NOT at offset 0"
|
||||
);
|
||||
}
|
||||
|
||||
/// When an AC-3 header's `fscod`/`frmsizecod` is unmappable (reserved
|
||||
/// `fscod == 3`), `max_substream_channels` must fall back to stepping
|
||||
/// `start + 2` bytes past the sync to re-lock onto the next genuine sync,
|
||||
/// and must keep making forward progress doing so (never revisit the same
|
||||
/// sync, which would loop forever, and never jump so far that it skips
|
||||
/// the very next real frame). This lays a bogus-sized header at absolute
|
||||
/// offset 4 (so `start == 4`, `start + 2 == 6`) immediately followed, at
|
||||
/// offset 6, by a real, fully decodable 2.0 frame — the position the
|
||||
/// `+ 2` fallback must land on exactly.
|
||||
#[test]
|
||||
fn max_substream_channels_unmappable_size_steps_forward_by_two() {
|
||||
let mut real = ac3_frame(2, false);
|
||||
// Overwrite the (unchecked) CRC bytes of the real frame — these double
|
||||
// as byte4/byte5 of the bogus header 2 bytes earlier, at absolute
|
||||
// offset 4: byte4 = 0xC0 (fscod=3 reserved -> ac3_frame_size == 0,
|
||||
// unmappable), byte5 = 0xF8 (bsid=31 >= 11 -> acmod_channels == None,
|
||||
// so the bogus header itself never contributes a spurious channel
|
||||
// count).
|
||||
real[2] = 0xC0;
|
||||
real[3] = 0xF8;
|
||||
let mut data = vec![0xAA, 0xAA, 0xAA, 0xAA]; // offsets 0..4, no sync
|
||||
data.push(0x0B); // offset 4: bogus header sync byte 0
|
||||
data.push(0x77); // offset 5: bogus header sync byte 1
|
||||
data.extend(real); // offset 6..: the real frame (also serves as the
|
||||
// bogus header's byte4/byte5 at offsets 8/9)
|
||||
assert_eq!(
|
||||
max_substream_channels(&data),
|
||||
Some(2),
|
||||
"must recover the real frame 2 bytes after the unmappable-size sync, not lose it"
|
||||
);
|
||||
}
|
||||
|
||||
/// Same fallback as above, but with the unmappable-size sync at absolute
|
||||
/// offset 0 (`start == 0`) so that stepping backward instead of forward
|
||||
/// (`start - 2`) would underflow rather than merely land on the wrong
|
||||
/// byte. Also proves the real frame is still found 6 bytes further in,
|
||||
/// confirming forward progress past the bogus header.
|
||||
#[test]
|
||||
fn max_substream_channels_unmappable_size_at_start_steps_forward_not_back() {
|
||||
let mut data = vec![0x0B, 0x77, 0x00, 0x00, 0xC0, 0xF8]; // bogus header, offsets 0..6
|
||||
data.extend(ac3_frame(2, false)); // real 2.0 frame at offset 6
|
||||
assert_eq!(
|
||||
max_substream_channels(&data),
|
||||
Some(2),
|
||||
"must step forward past the bogus header at offset 0 and find the real frame at offset 6"
|
||||
);
|
||||
}
|
||||
|
||||
/// `remap_audio_pids` must read a stream's CURRENT physical sub-stream id
|
||||
/// from the low byte of its PID via `pid & 0x00FF` — not `|` or `^` with
|
||||
/// `0x00FF`, both of which force the low byte to `0xFF` regardless of the
|
||||
/// real PID and so always miss the "already matches" shortcut. That
|
||||
/// matters observably when TWO physical sub-streams share the same probed
|
||||
/// channel count: with a correct read, a stream already sitting on a
|
||||
/// matching sub-stream is left alone (conservative, per the module's
|
||||
/// documented behaviour); with the low byte forced to `0xFF`,
|
||||
/// `probed.get(&0xFF)` is always `None`, so the code falls through to the
|
||||
/// "find any unclaimed match" path and picks the FIRST (lowest-keyed,
|
||||
/// BTreeMap-ordered) matching physical sub-stream instead — which here is
|
||||
/// a *different* sub-stream (0x80) than the one the PID already correctly
|
||||
/// names (0x81), producing a spurious PID change.
|
||||
#[test]
|
||||
fn remap_reads_current_substream_via_and_not_or_or_xor() {
|
||||
let mut probed = BTreeMap::new();
|
||||
probed.insert(0x80u8, 6u8);
|
||||
probed.insert(0x81u8, 6u8); // ambiguous: two physical 6ch sub-streams
|
||||
let mut streams = vec![ac3_stream(0xBD81, AudioChannels::Surround51)];
|
||||
let changed = remap_audio_pids(&mut streams, &probed);
|
||||
assert_eq!(
|
||||
changed, 0,
|
||||
"already sitting on a matching physical sub-stream (0x81) must be left alone"
|
||||
);
|
||||
let Stream::Audio(a) = &streams[0] else {
|
||||
panic!()
|
||||
};
|
||||
assert_eq!(
|
||||
a.pid, 0xBD81,
|
||||
"must not be bumped to the other matching sub-stream (0x80)"
|
||||
);
|
||||
}
|
||||
|
||||
/// A `SectorSource` stub that hands back fixed bytes regardless of the
|
||||
/// requested LBA/count, for exercising `probe_and_remap`'s end-to-end
|
||||
/// wiring (format/AC-3/extent/count guards -> read -> probe -> remap).
|
||||
struct FixedSource {
|
||||
data: Vec<u8>,
|
||||
}
|
||||
|
||||
impl SectorSource for FixedSource {
|
||||
fn read_sectors(
|
||||
&mut self,
|
||||
_lba: u32,
|
||||
_count: u16,
|
||||
buf: &mut [u8],
|
||||
_recovery: bool,
|
||||
) -> crate::error::Result<usize> {
|
||||
let n = self.data.len().min(buf.len());
|
||||
buf[..n].copy_from_slice(&self.data[..n]);
|
||||
Ok(n)
|
||||
}
|
||||
}
|
||||
|
||||
/// End-to-end `probe_and_remap`: a Silence-of-the-Lambs-shaped MpegPs
|
||||
/// title (one declared 5.1 AC-3 stream ordinally assigned 0x80) whose
|
||||
/// physical VOB bytes carry the 2.0 down-mix on 0x80 and the real 5.1 on
|
||||
/// 0x81. This must reach the `remap_audio_pids` call and re-route the
|
||||
/// stream to 0xBD81. It also, by construction, proves each of the guards
|
||||
/// along the way lets a real, positive case through: the content-format
|
||||
/// check must NOT bail on `MpegPs` (only on non-`MpegPs`), the AC-3
|
||||
/// presence check must NOT bail when AC-3 IS present, and the
|
||||
/// sector-count check must NOT bail when the count is nonzero — any one
|
||||
/// of those inverted would skip the probe entirely and leave the PID at
|
||||
/// its untouched ordinal value (0xBD80), which the assertion below would
|
||||
/// catch.
|
||||
#[test]
|
||||
fn probe_and_remap_reroutes_silence_of_the_lambs_scenario_end_to_end() {
|
||||
let mut bytes = ps_ac3(0x80, 2, false); // physical 0x80 = 2.0 down-mix
|
||||
bytes.extend(ps_ac3(0x81, 7, true)); // physical 0x81 = 5.1 main mix
|
||||
let mut title = DiscTitle {
|
||||
playlist: "00001.ifo".into(),
|
||||
playlist_id: 1,
|
||||
duration_secs: 60.0,
|
||||
size_bytes: bytes.len() as u64,
|
||||
clips: Vec::new(),
|
||||
streams: vec![ac3_stream(0xBD80, AudioChannels::Surround51)],
|
||||
chapters: Vec::new(),
|
||||
extents: vec![Extent {
|
||||
start_lba: 0,
|
||||
sector_count: 2,
|
||||
}],
|
||||
content_format: ContentFormat::MpegPs,
|
||||
codec_privates: vec![None],
|
||||
};
|
||||
let mut source = FixedSource { data: bytes };
|
||||
probe_and_remap(&mut source, &mut title);
|
||||
let Stream::Audio(a) = &title.streams[0] else {
|
||||
panic!("audio")
|
||||
};
|
||||
assert_eq!(
|
||||
a.pid, 0xBD81,
|
||||
"declared 5.1 stream must be re-routed to the physical 5.1 sub-stream 0x81"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+110
-41
@@ -7,7 +7,6 @@ use crate::udf;
|
||||
|
||||
/// Result of SCSI AACS handshake (ECDH authentication).
|
||||
/// Only available when scanning from a real drive, not ISO images.
|
||||
#[derive(Debug)]
|
||||
pub(super) struct HandshakeResult {
|
||||
pub volume_id: [u8; 16],
|
||||
pub read_data_key: Option<[u8; 16]>,
|
||||
@@ -29,6 +28,19 @@ pub(super) struct HandshakeResult {
|
||||
pub drive_unlocked: bool,
|
||||
}
|
||||
|
||||
// Redacting `Debug`: `volume_id` and `read_data_key` (the AACS 2.0 bus key) are
|
||||
// secret; print only shape. Guarded by `handshake_result_debug_is_redacted`.
|
||||
impl std::fmt::Debug for HandshakeResult {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("HandshakeResult")
|
||||
.field("volume_id", &"<redacted>")
|
||||
.field("read_data_key", &self.read_data_key.map(|_| "<redacted>"))
|
||||
.field("read_data_key_err", &self.read_data_key_err)
|
||||
.field("drive_unlocked", &self.drive_unlocked)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
/// Single source of truth for "is AACS bus encryption gone for this scan?". The
|
||||
/// gate asks ONLY this — `if !removed { error }` — never enumerating cases. Bus
|
||||
/// encryption is gone when ANY of these holds:
|
||||
@@ -169,6 +181,24 @@ fn cert_unlock_outcome(e: &CertUnlockFailure) -> crate::aacs::trace::UnlockOutco
|
||||
}
|
||||
}
|
||||
|
||||
/// Did the cert handshake actually carry a Volume ID?
|
||||
///
|
||||
/// Extracted so it can be tested as a VALUE. It only ever reaches an operator
|
||||
/// as the `has_volume_id` field of the `bus_key_unavailable` warn, and
|
||||
/// asserting on a `tracing` field means installing a capturing subscriber —
|
||||
/// which is thread-local, while `tracing`'s callsite-interest cache is global.
|
||||
/// Those two facts race: the test failed roughly one run in ten under the full
|
||||
/// parallel suite while passing every time in isolation, and serialising the
|
||||
/// captures was not enough because the cache can be re-evaluated against the
|
||||
/// process default rather than the thread-local dispatch.
|
||||
///
|
||||
/// A predicate this small does not need a subscriber to verify. The polarity is
|
||||
/// the whole point: an `==` here would tell an operator a VID was absent on
|
||||
/// exactly the discs where one was present.
|
||||
fn handshake_has_volume_id(h: &HandshakeResult) -> bool {
|
||||
h.volume_id != [0u8; 16]
|
||||
}
|
||||
|
||||
impl Disc {
|
||||
/// SCSI handshake — drives the VID-acquisition flow and returns
|
||||
/// a structured `HandshakeResult` for downstream key resolution.
|
||||
@@ -312,13 +342,18 @@ impl Disc {
|
||||
use crate::aacs;
|
||||
|
||||
let uk_ro_data =
|
||||
aacs::read_first(aacs::UNIT_KEY_RO_PATHS, |p| udf_fs.read_file(reader, p))?;
|
||||
aacs::read_first(&aacs::role_paths(udf_fs, aacs::AacsRole::UnitKey), |p| {
|
||||
udf_fs.read_file(reader, p)
|
||||
})?;
|
||||
let dh = aacs::inf::disc_hash(&uk_ro_data);
|
||||
|
||||
let cc = aacs::read_first(aacs::CONTENT_CERT_PATHS, |p| udf_fs.read_file(reader, p))
|
||||
.ok()
|
||||
.as_deref()
|
||||
.and_then(aacs::inf::parse_content_cert);
|
||||
let cc = aacs::read_first(
|
||||
&aacs::role_paths(udf_fs, aacs::AacsRole::ContentCert),
|
||||
|p| udf_fs.read_file(reader, p),
|
||||
)
|
||||
.ok()
|
||||
.as_deref()
|
||||
.and_then(aacs::inf::parse_content_cert);
|
||||
let bus_encryption = cc.as_ref().map(|c| c.bus_encryption).unwrap_or(false);
|
||||
// No-cert default = UHD (V20 stride), matching `read_aacs_version` so the
|
||||
// scanned `AacsState.version` and the out-of-band fetch agree. A wrong
|
||||
@@ -352,7 +387,7 @@ impl Disc {
|
||||
// file/ISO, drive unlock, cert bus key). The gate enumerates nothing.
|
||||
if !bus_encryption_removed(bus_encryption, handshake) {
|
||||
let (rdk_err, has_vid) = handshake
|
||||
.map(|h| (h.read_data_key_err, h.volume_id != [0u8; 16]))
|
||||
.map(|h| (h.read_data_key_err, handshake_has_volume_id(h)))
|
||||
.unwrap_or((None, false));
|
||||
tracing::warn!(
|
||||
target: "freemkv::disc",
|
||||
@@ -428,6 +463,27 @@ mod tests {
|
||||
use crate::sector::SectorSource;
|
||||
use std::collections::HashMap;
|
||||
|
||||
/// `HandshakeResult` carries the Volume ID and the AACS 2.0 bus (read-data)
|
||||
/// key; `Debug` must redact both. Sentinel 213 (0xD5).
|
||||
#[test]
|
||||
fn handshake_result_debug_is_redacted() {
|
||||
let hs = HandshakeResult {
|
||||
volume_id: [0xD5; 16],
|
||||
read_data_key: Some([0xD5; 16]),
|
||||
read_data_key_err: None,
|
||||
drive_unlocked: false,
|
||||
};
|
||||
let d = format!("{hs:?}");
|
||||
assert!(
|
||||
!d.contains("213"),
|
||||
"HandshakeResult leaked VID/bus key: {d}"
|
||||
);
|
||||
assert!(
|
||||
d.contains("redacted"),
|
||||
"HandshakeResult missing marker: {d}"
|
||||
);
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------
|
||||
// In-memory disc + minimal UDF image with a single physical
|
||||
// partition (metadata_start == partition_start). Offsets cited
|
||||
@@ -911,41 +967,54 @@ mod tests {
|
||||
|
||||
/// Unit_Key_RO.inf is read from /AACS/DUPLICATE when the primary copy
|
||||
/// is absent (encrypt.rs `.or_else(|_| read_file(DUPLICATE/...))`).
|
||||
/// This is the damaged-primary recovery path real discs rely on.
|
||||
#[test]
|
||||
fn resolve_vid_only_falls_back_to_duplicate_unit_key_ro() {
|
||||
let mut disc = MemDisc::new();
|
||||
// Build AACS dir with a DUPLICATE subdir holding Unit_Key_RO.inf.
|
||||
let uk = vec![0x55u8; 48];
|
||||
let mut dup_fids = Vec::new();
|
||||
push_fid(&mut dup_fids, "", 70, true, true);
|
||||
push_fid(&mut dup_fids, "Unit_Key_RO.inf", 72, false, false);
|
||||
disc.put(PART_START + 72, build_file_icb(uk.len() as u32, 9000));
|
||||
disc.put_bytes(PART_START + 9000, &uk);
|
||||
disc.put(PART_START + 70, build_file_icb(dup_fids.len() as u32, 71));
|
||||
disc.put_bytes(PART_START + 71, &dup_fids);
|
||||
// AACS dir: only a DUPLICATE subdir (no primary Unit_Key_RO.inf).
|
||||
let mut aacs_fids = Vec::new();
|
||||
push_fid(&mut aacs_fids, "", 50, true, true);
|
||||
push_fid(&mut aacs_fids, "DUPLICATE", 70, true, false);
|
||||
disc.put(PART_START + 50, build_file_icb(aacs_fids.len() as u32, 51));
|
||||
disc.put_bytes(PART_START + 51, &aacs_fids);
|
||||
let mut root_fids = Vec::new();
|
||||
push_fid(&mut root_fids, "", 10, true, true);
|
||||
push_fid(&mut root_fids, "AACS", 50, true, false);
|
||||
disc.put(PART_START + 10, build_file_icb(root_fids.len() as u32, 11));
|
||||
disc.put_bytes(PART_START + 11, &root_fids);
|
||||
build_udf_skeleton(&mut disc, 10);
|
||||
let udf = udf::read_filesystem(&mut disc).expect("fs");
|
||||
|
||||
let st = Disc::resolve_vid_only(&udf, &mut disc, None).expect("DUPLICATE fallback");
|
||||
// disc_hash must be computed over the DUPLICATE bytes.
|
||||
assert_eq!(
|
||||
st.disc_hash,
|
||||
aacs::inf::disc_hash_hex(&aacs::inf::disc_hash(&uk)),
|
||||
"fallback must hash the DUPLICATE Unit_Key_RO.inf"
|
||||
fn handshake_has_volume_id_reports_presence_not_absence() {
|
||||
let with_vid = HandshakeResult {
|
||||
volume_id: [0x11u8; 16],
|
||||
read_data_key: None,
|
||||
read_data_key_err: None,
|
||||
drive_unlocked: false,
|
||||
};
|
||||
assert!(
|
||||
super::handshake_has_volume_id(&with_vid),
|
||||
"a non-zero Volume ID must report as PRESENT"
|
||||
);
|
||||
assert_eq!(st.uk_ro, uk);
|
||||
|
||||
let without = HandshakeResult {
|
||||
volume_id: [0u8; 16],
|
||||
..with_vid
|
||||
};
|
||||
assert!(
|
||||
!super::handshake_has_volume_id(&without),
|
||||
"an all-zero Volume ID is the absent case"
|
||||
);
|
||||
|
||||
// One bit of difference is still a VID: the check is != all-zero, not a
|
||||
// heuristic about how much of it looks populated.
|
||||
let mut barely = [0u8; 16];
|
||||
barely[15] = 1;
|
||||
assert!(
|
||||
super::handshake_has_volume_id(&HandshakeResult {
|
||||
volume_id: barely,
|
||||
..with_vid
|
||||
}),
|
||||
"any non-zero byte makes a Volume ID present"
|
||||
);
|
||||
}
|
||||
|
||||
/// The gate itself still hard-errors — the property the log line annotates.
|
||||
#[test]
|
||||
fn resolve_vid_only_bus_key_gate_hard_errors_without_a_read_data_key() {
|
||||
let (mut disc, udf) = disc_with_cert(0x01, true);
|
||||
let hs = HandshakeResult {
|
||||
volume_id: [0x11u8; 16],
|
||||
read_data_key: None,
|
||||
read_data_key_err: None,
|
||||
drive_unlocked: false,
|
||||
};
|
||||
let err = Disc::resolve_vid_only(&udf, &mut disc, Some(&hs))
|
||||
.expect_err("bus-encrypted, no read_data_key must still hard-error");
|
||||
assert!(matches!(err, Error::AacsBusKeyUnavailable));
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------
|
||||
@@ -976,7 +1045,7 @@ mod tests {
|
||||
/// A minimal in-test KeySource that yields no keys but a fixed cert list.
|
||||
struct CertSource(Vec<aacs::types::HostCert>);
|
||||
impl crate::KeySource for CertSource {
|
||||
fn get_uk(
|
||||
fn get_unit_keys(
|
||||
&self,
|
||||
_ctx: &dyn crate::keysource::ResolveCtx,
|
||||
) -> Result<Vec<crate::aacs::types::UnitKey>> {
|
||||
|
||||
+1004
-83
File diff suppressed because it is too large
Load Diff
+2306
-62
File diff suppressed because it is too large
Load Diff
-1670
File diff suppressed because it is too large
Load Diff
+3488
-2777
File diff suppressed because it is too large
Load Diff
-1657
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,218 +0,0 @@
|
||||
//! `Disc::sweep`'s consumer-side `Sink<WorkItem>`.
|
||||
//!
|
||||
//! Background: the original sweep loop runs strictly serialised —
|
||||
//! SCSI read → decrypt → seek + write → mapfile.record → next iter.
|
||||
//! On a healthy disc the SCSI read costs ~5-12 ms per 64 KB batch and
|
||||
//! the post-read work (decrypt 1-3 ms + file write + mapfile fsync
|
||||
//! 5-15 ms) adds another batch's worth of latency. The drive idles
|
||||
//! during the post-read work; throughput tops out at the *sum* of
|
||||
//! both costs.
|
||||
//!
|
||||
//! A producer/consumer split overlaps the two stages on the generic
|
||||
//! [`crate::io::Pipeline`] + [`crate::io::Sink`] primitive. This module
|
||||
//! is the sweep-specific `Sink` impl; the producer-side state machine
|
||||
//! (read_error context, decrypt, set_speed, halt) stays in
|
||||
//! `Disc::sweep` in `disc/mod.rs`.
|
||||
//!
|
||||
//! Correctness invariants preserved:
|
||||
//! - Mapfile is single-writer (consumer-only). No locking.
|
||||
//! - All `read_error::ReadCtx` state stays on the producer thread.
|
||||
//! - `set_speed` calls happen on the producer thread (same thread that
|
||||
//! owns the `SectorSource`). No new SCSI concurrency.
|
||||
//! - Per-iteration ordering of file-write → mapfile-record is kept
|
||||
//! intact in the consumer (write before record), so the on-disk
|
||||
//! invariant "mapfile only marks Finished what the file has
|
||||
//! received" survives a crash mid-pass.
|
||||
//! - Only one SCSI command is in flight at a time; error-path timing
|
||||
//! is identical and no new retry logic is introduced.
|
||||
|
||||
use std::io::{Seek, SeekFrom, Write};
|
||||
use std::sync::mpsc::{Receiver, SyncSender, sync_channel};
|
||||
|
||||
use crate::error::Error;
|
||||
use crate::io::{Flow, Sink};
|
||||
|
||||
use super::mapfile::{MapStats, Mapfile, SectorStatus};
|
||||
|
||||
/// Reusable zero buffer for SkipFill / GapFill / BisectBad. 64 KB
|
||||
/// matches the existing zero_gap chunk size used by the pre-split
|
||||
/// sweep loop.
|
||||
const ZERO_CHUNK: usize = 64 * 1024;
|
||||
|
||||
/// Producer → Consumer messages. The consumer applies these in FIFO
|
||||
/// order; ordering of file writes and mapfile records across items is
|
||||
/// preserved.
|
||||
pub(super) enum WorkItem {
|
||||
/// Successful batch read. Producer has already decrypted `buf` if
|
||||
/// `opts.decrypt` was set. Consumer writes `buf` at `pos` and
|
||||
/// records the range as `Finished`.
|
||||
Good { pos: u64, buf: Vec<u8> },
|
||||
|
||||
/// Bisect inner-loop good single sector (already decrypted by the
|
||||
/// producer). 2048 bytes.
|
||||
BisectGood { pos: u64, buf: Box<[u8; 2048]> },
|
||||
|
||||
/// Bisect inner-loop bad single sector. Consumer writes 2048
|
||||
/// zeros at `pos` and records the sector as `NonTrimmed`.
|
||||
BisectBad { pos: u64 },
|
||||
|
||||
/// Whole-batch zero-fill (failed batch on `SkipBlock`, or the
|
||||
/// failed batch portion of `JumpAhead`). Consumer streams zeros
|
||||
/// across `[pos, pos+len)` and records the range as `NonTrimmed`.
|
||||
SkipFill { pos: u64, len: u64 },
|
||||
|
||||
/// Gap fill following a `JumpAhead`. Same effect as `SkipFill`;
|
||||
/// distinguished only so future logging / instrumentation can
|
||||
/// tell them apart without parsing a flag.
|
||||
GapFill { pos: u64, len: u64 },
|
||||
|
||||
/// Producer wants the latest mapfile stats for the progress
|
||||
/// callback. Consumer responds on `prog_tx` with a fresh
|
||||
/// [`ProgressSnapshot`]. Best-effort: if the producer hasn't
|
||||
/// drained the previous snapshot, the new one is silently
|
||||
/// dropped — the producer's local cache stays current enough.
|
||||
StatsRequest,
|
||||
}
|
||||
|
||||
/// Snapshot the consumer sends back to the producer for the progress
|
||||
/// callback.
|
||||
pub(super) struct ProgressSnapshot {
|
||||
pub stats: MapStats,
|
||||
pub bad_ranges: Vec<(u64, u64)>,
|
||||
}
|
||||
|
||||
/// Final summary returned by the consumer thread on shutdown — what
|
||||
/// `SweepSink::close` produces, surfaced to the producer via
|
||||
/// `Pipeline::finish`.
|
||||
pub(super) struct ConsumerSummary {
|
||||
pub stats: MapStats,
|
||||
}
|
||||
|
||||
/// Drain any pending progress snapshots from the consumer. Returns
|
||||
/// the most recent one, if any. The producer caches it and uses it
|
||||
/// for subsequent progress callbacks until a fresh one arrives.
|
||||
pub(super) fn try_recv_progress(rx: &Receiver<ProgressSnapshot>) -> Option<ProgressSnapshot> {
|
||||
let mut latest = None;
|
||||
while let Ok(snap) = rx.try_recv() {
|
||||
latest = Some(snap);
|
||||
}
|
||||
latest
|
||||
}
|
||||
|
||||
/// `Sink<WorkItem>` for sweep. Owns the writeback file + mapfile +
|
||||
/// progress back-channel. `apply` carries the file-write +
|
||||
/// mapfile.record per item; `close` drains the writeback pipeline,
|
||||
/// fsyncs the ISO, and flushes the mapfile.
|
||||
pub(super) struct SweepSink {
|
||||
file: crate::io::WritebackFile,
|
||||
map: Mapfile,
|
||||
/// `sync_all`-on-failure-is-an-error iff the output is a regular
|
||||
/// file. `/dev/null` and pipes always fail `sync_all`; that's not
|
||||
/// a real error.
|
||||
is_regular: bool,
|
||||
/// Back-channel for `StatsRequest` responses. The producer caches
|
||||
/// the latest snapshot and uses it for the progress callback;
|
||||
/// dropped sends on a full channel are by design.
|
||||
prog_tx: SyncSender<ProgressSnapshot>,
|
||||
/// Reusable zero buffer for SkipFill / GapFill / BisectBad. Held
|
||||
/// in the sink so each apply call doesn't reallocate.
|
||||
zero: Box<[u8; ZERO_CHUNK]>,
|
||||
}
|
||||
|
||||
impl SweepSink {
|
||||
/// Construct a new `SweepSink` plus the matching progress
|
||||
/// receiver. Channel depth on the back-channel is `1` — the
|
||||
/// producer's cache is the source of truth between snapshots.
|
||||
pub(super) fn new(
|
||||
file: crate::io::WritebackFile,
|
||||
map: Mapfile,
|
||||
is_regular: bool,
|
||||
) -> (Self, Receiver<ProgressSnapshot>) {
|
||||
let (prog_tx, prog_rx) = sync_channel::<ProgressSnapshot>(1);
|
||||
let sink = SweepSink {
|
||||
file,
|
||||
map,
|
||||
is_regular,
|
||||
prog_tx,
|
||||
zero: Box::new([0u8; ZERO_CHUNK]),
|
||||
};
|
||||
(sink, prog_rx)
|
||||
}
|
||||
}
|
||||
|
||||
impl Sink<WorkItem> for SweepSink {
|
||||
type Output = ConsumerSummary;
|
||||
|
||||
fn apply(&mut self, item: WorkItem) -> Result<Flow, Error> {
|
||||
match item {
|
||||
WorkItem::Good { pos, buf } => {
|
||||
// Decrypt is on the producer; consumer assumes plaintext.
|
||||
let len = buf.len() as u64;
|
||||
self.file.seek(SeekFrom::Start(pos))?;
|
||||
self.file.write_all(&buf)?;
|
||||
self.map.record(pos, len, SectorStatus::Finished)?;
|
||||
}
|
||||
WorkItem::BisectGood { pos, buf } => {
|
||||
self.file.seek(SeekFrom::Start(pos))?;
|
||||
self.file.write_all(&buf[..])?;
|
||||
self.map.record(pos, 2048, SectorStatus::Finished)?;
|
||||
}
|
||||
WorkItem::BisectBad { pos } => {
|
||||
self.file.seek(SeekFrom::Start(pos))?;
|
||||
self.file.write_all(&self.zero[..2048])?;
|
||||
self.map.record(pos, 2048, SectorStatus::NonTrimmed)?;
|
||||
}
|
||||
WorkItem::SkipFill { pos, len } | WorkItem::GapFill { pos, len } => {
|
||||
self.file.seek(SeekFrom::Start(pos))?;
|
||||
// Subsequent writes are sequential; `WritebackFile`'s
|
||||
// seek-elision keeps them on the writeback pipeline path.
|
||||
let mut filled = 0u64;
|
||||
while filled < len {
|
||||
let chunk = (len - filled).min(self.zero.len() as u64) as usize;
|
||||
self.file.write_all(&self.zero[..chunk])?;
|
||||
filled += chunk as u64;
|
||||
}
|
||||
self.map.record(pos, len, SectorStatus::NonTrimmed)?;
|
||||
}
|
||||
WorkItem::StatsRequest => {
|
||||
let stats = self.map.stats();
|
||||
// DAMAGE only — NOT NonTried. NonTried is the unread remainder
|
||||
// ahead of the sweep head, not damage; including it made the live
|
||||
// located drilldown (at-risk movie time + range count) treat the
|
||||
// whole unread disc as confirmed damage, so at sweep start it
|
||||
// showed ~full-movie at-risk and melted to 0 as the sweep
|
||||
// progressed. Matches the one-shot progress path, which already
|
||||
// excludes NonTried.
|
||||
let bad_ranges = self.map.ranges_with(&[
|
||||
SectorStatus::NonTrimmed,
|
||||
SectorStatus::Unreadable,
|
||||
SectorStatus::NonScraped,
|
||||
]);
|
||||
// Best-effort: drop on backpressure; producer's cache
|
||||
// stays current enough.
|
||||
let _ = self
|
||||
.prog_tx
|
||||
.try_send(ProgressSnapshot { stats, bad_ranges });
|
||||
}
|
||||
}
|
||||
Ok(Flow::Continue)
|
||||
}
|
||||
|
||||
fn close(mut self) -> Result<Self::Output, Error> {
|
||||
// Drain the writeback pipeline + fsync the ISO, then persist
|
||||
// any pending mapfile state. Same finalisation order as the
|
||||
// pre-Pipeline consumer loop.
|
||||
if let Err(e) = self.file.sync_all() {
|
||||
if self.is_regular {
|
||||
return Err(Error::IoError { source: e });
|
||||
}
|
||||
// Non-regular outputs (/dev/null, pipes) always fail
|
||||
// sync_all; that's not a real error.
|
||||
}
|
||||
self.map.flush()?;
|
||||
|
||||
Ok(ConsumerSummary {
|
||||
stats: self.map.stats(),
|
||||
})
|
||||
}
|
||||
}
|
||||
+6
-8
@@ -23,14 +23,12 @@ pub fn find_drives() -> Vec<(String, DriveId)> {
|
||||
if !std::path::Path::new(&path).exists() {
|
||||
continue;
|
||||
}
|
||||
if let Ok(mut transport) = crate::scsi::open(std::path::Path::new(&path)) {
|
||||
if let Ok(id) = DriveId::from_drive(transport.as_mut()) {
|
||||
if !id.raw_inquiry.is_empty()
|
||||
&& (id.raw_inquiry[0] & 0x1F) == SCSI_PERIPHERAL_TYPE_OPTICAL
|
||||
{
|
||||
drives.push((path, id));
|
||||
}
|
||||
}
|
||||
if let Ok(mut transport) = crate::scsi::open(std::path::Path::new(&path))
|
||||
&& let Ok(id) = DriveId::from_drive(transport.as_mut())
|
||||
&& !id.raw_inquiry.is_empty()
|
||||
&& (id.raw_inquiry[0] & 0x1F) == SCSI_PERIPHERAL_TYPE_OPTICAL
|
||||
{
|
||||
drives.push((path, id));
|
||||
}
|
||||
}
|
||||
drives
|
||||
|
||||
+35
-6
@@ -25,12 +25,11 @@ pub fn find_drives() -> Vec<(String, DriveId)> {
|
||||
let path = std::path::Path::new(&info.path);
|
||||
match crate::scsi::open(path) {
|
||||
Ok(mut transport) => {
|
||||
if let Ok(id) = DriveId::from_drive(transport.as_mut()) {
|
||||
if !id.raw_inquiry.is_empty()
|
||||
&& (id.raw_inquiry[0] & 0x1F) == SCSI_PERIPHERAL_TYPE_OPTICAL
|
||||
{
|
||||
drives.push((info.path.clone(), id));
|
||||
}
|
||||
if let Ok(id) = DriveId::from_drive(transport.as_mut())
|
||||
&& !id.raw_inquiry.is_empty()
|
||||
&& (id.raw_inquiry[0] & 0x1F) == SCSI_PERIPHERAL_TYPE_OPTICAL
|
||||
{
|
||||
drives.push((info.path.clone(), id));
|
||||
}
|
||||
}
|
||||
Err(_) => {
|
||||
@@ -53,3 +52,33 @@ pub fn resolve_device(path: &str) -> Result<(String, DeviceResolution)> {
|
||||
}
|
||||
Ok((path.to_string(), DeviceResolution::Direct))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod resolve_device_tests {
|
||||
use super::*;
|
||||
|
||||
/// An existing path resolves unchanged as `Direct` — macOS has no
|
||||
/// `sr`->`sg` substitution, so the returned path must be byte-identical
|
||||
/// to the input, not some canonicalised/mutated form.
|
||||
#[test]
|
||||
fn existing_path_resolves_direct_unchanged() {
|
||||
// Use the test binary's own executable path: guaranteed to exist,
|
||||
// no fixture file needed.
|
||||
let exe = std::env::current_exe().unwrap();
|
||||
let path = exe.to_str().unwrap();
|
||||
let (resolved, kind) = resolve_device(path).expect("existing path must resolve");
|
||||
assert_eq!(resolved, path, "path must be returned unchanged");
|
||||
assert_eq!(kind, DeviceResolution::Direct);
|
||||
}
|
||||
|
||||
/// A path that does not exist must error with `DeviceNotFound` carrying
|
||||
/// the original path, never silently succeed.
|
||||
#[test]
|
||||
fn missing_path_is_device_not_found() {
|
||||
let path = "/dev/freemkv-definitely-does-not-exist-0xdead";
|
||||
match resolve_device(path) {
|
||||
Err(Error::DeviceNotFound { path: p }) => assert_eq!(p, path),
|
||||
other => panic!("expected DeviceNotFound, got {other:?}"),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+1259
-134
File diff suppressed because it is too large
Load Diff
+896
-108
File diff suppressed because it is too large
Load Diff
+267
@@ -0,0 +1,267 @@
|
||||
//! Seeded robustness harness for the untrusted-input parsers.
|
||||
//!
|
||||
//! Every parser reached from here takes bytes that came off a disc, and this
|
||||
//! crate's primary boundary is that the disc is untrusted: a malformed, damaged
|
||||
//! or hostile image must never crash the library. These tests assert exactly
|
||||
//! that one property — **the parser returns `Ok` or `Err`, and never panics.**
|
||||
//!
|
||||
//! # Why this exists rather than `cargo-fuzz`
|
||||
//!
|
||||
//! `cargo-fuzz` needs a nightly toolchain (`-Zsanitizer` plus SanitizerCoverage
|
||||
//! for libFuzzer's coverage feedback) and this project pins stable. So the
|
||||
//! generator lives here instead. It gives up coverage-guided mutation — the real
|
||||
//! loss — and keeps everything else: millions of cases, structure-aware input,
|
||||
//! and a crash corpus. It also gains determinism, which a fuzzer does not have:
|
||||
//! the same seed replays the same cases on any machine.
|
||||
//!
|
||||
//! # Why no `proptest` or `arbitrary`
|
||||
//!
|
||||
//! This crate has exactly one dev-dependency. That is a deliberate posture, and
|
||||
//! a randomness crate is not worth ten transitive dependencies when the parsers
|
||||
//! take plain `&[u8]` and a good enough generator is forty lines.
|
||||
//!
|
||||
//! # Budget
|
||||
//!
|
||||
//! `FREEMKV_HARNESS_CASES` sets cases per generator per target (default 256, low
|
||||
//! enough that the per-commit gate stays under a second). The overnight run sets
|
||||
//! it to millions. `FREEMKV_HARNESS_SEED` overrides the seed; the default is
|
||||
//! fixed so a failure in CI reproduces locally verbatim.
|
||||
//!
|
||||
//! # On failure
|
||||
//!
|
||||
//! The panic message carries the seed, generator and case index. Re-run with
|
||||
//! `FREEMKV_HARNESS_SEED=<seed>` to reproduce, then write the offending bytes
|
||||
//! into `tests/corpus/` as a permanent regression fixture — discovery happens
|
||||
//! here, defence happens there.
|
||||
|
||||
#![cfg(test)]
|
||||
|
||||
/// Marsaglia xorshift64. Not cryptographic and does not need to be: the job is
|
||||
/// a reproducible spread of bytes, and a named algorithm beats an ad-hoc LCG
|
||||
/// whose period nobody has checked.
|
||||
struct Rng(u64);
|
||||
|
||||
impl Rng {
|
||||
fn new(seed: u64) -> Self {
|
||||
// A zero seed is a fixed point of xorshift — it would emit zeros forever
|
||||
// and every generated case would be identical.
|
||||
Self(if seed == 0 {
|
||||
0x2545_F491_4F6C_DD1D
|
||||
} else {
|
||||
seed
|
||||
})
|
||||
}
|
||||
|
||||
fn next(&mut self) -> u64 {
|
||||
self.0 ^= self.0 << 13;
|
||||
self.0 ^= self.0 >> 7;
|
||||
self.0 ^= self.0 << 17;
|
||||
self.0
|
||||
}
|
||||
|
||||
fn byte(&mut self) -> u8 {
|
||||
(self.next() >> 24) as u8
|
||||
}
|
||||
|
||||
/// Uniform-ish in `0..n`. The modulo bias is irrelevant at these magnitudes.
|
||||
fn below(&mut self, n: usize) -> usize {
|
||||
if n == 0 {
|
||||
0
|
||||
} else {
|
||||
(self.next() % n as u64) as usize
|
||||
}
|
||||
}
|
||||
|
||||
fn fill(&mut self, len: usize) -> Vec<u8> {
|
||||
(0..len).map(|_| self.byte()).collect()
|
||||
}
|
||||
}
|
||||
|
||||
/// Budget per generator per target.
|
||||
fn cases() -> usize {
|
||||
std::env::var("FREEMKV_HARNESS_CASES")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(256)
|
||||
}
|
||||
|
||||
fn seed() -> u64 {
|
||||
std::env::var("FREEMKV_HARNESS_SEED")
|
||||
.ok()
|
||||
.and_then(|v| v.parse().ok())
|
||||
.unwrap_or(0x5EED_1234_ABCD_0001)
|
||||
}
|
||||
|
||||
/// Largest generated input. Big enough to carry a plausible header plus a body,
|
||||
/// small enough that millions of cases stay quick.
|
||||
const MAX_LEN: usize = 4096;
|
||||
|
||||
/// Drive `f` over three generators and report which case broke it.
|
||||
///
|
||||
/// A panic inside `f` fails the test on its own — nothing is caught here,
|
||||
/// because catching would risk reporting a pass on an input that aborted. The
|
||||
/// wrapper exists to make the failing case *identifiable*: the harness prints
|
||||
/// the seed, generator and index before each call, so the last line before a
|
||||
/// panic names the exact case to reproduce.
|
||||
fn sweep<F: FnMut(&[u8])>(target: &str, magic: &[u8], f: F) {
|
||||
sweep_n(target, magic, cases(), f)
|
||||
}
|
||||
|
||||
/// `sweep` with an explicit budget. The budget is a PARAMETER rather than read
|
||||
/// from the environment inside the loop: the meta-tests below need a small,
|
||||
/// fixed count, and `std::env::set_var` is unsound once the test harness runs
|
||||
/// tests in parallel — two tests setting the same variable race, which is
|
||||
/// exactly what happened on the first run of this file.
|
||||
fn sweep_n<F: FnMut(&[u8])>(target: &str, magic: &[u8], n: usize, mut f: F) {
|
||||
let s = seed();
|
||||
|
||||
// 1. Pure random bytes. Cheap, and almost always rejected at the magic
|
||||
// number — it exercises the entry guards and little else. Kept because
|
||||
// the entry guards are themselves worth exercising.
|
||||
let mut rng = Rng::new(s);
|
||||
for i in 0..n {
|
||||
let len = rng.below(MAX_LEN);
|
||||
let buf = rng.fill(len);
|
||||
run(target, "random", s, i, &buf, &mut f);
|
||||
}
|
||||
|
||||
// 2. Valid magic, random body. THE generator that matters: pure random
|
||||
// input dies at the magic check and never reaches the parser body, so
|
||||
// without this the sweep only ever tests the first few lines.
|
||||
let mut rng = Rng::new(s ^ 0xA5A5_A5A5_A5A5_A5A5);
|
||||
for i in 0..n {
|
||||
let mut buf = magic.to_vec();
|
||||
let tail = rng.below(MAX_LEN.saturating_sub(magic.len()));
|
||||
buf.extend(rng.fill(tail));
|
||||
run(target, "magic+noise", s, i, &buf, &mut f);
|
||||
}
|
||||
|
||||
// 3. Structured mutation of a plausible record: a valid magic, then mostly
|
||||
// zeroes, with a handful of bytes corrupted and a truncation. Length and
|
||||
// offset fields live in those early bytes, so this is what reaches the
|
||||
// arithmetic — the offsets, counts and sizes a hostile image would lie
|
||||
// about.
|
||||
let mut rng = Rng::new(s ^ 0x1234_5678_9ABC_DEF0);
|
||||
for i in 0..n {
|
||||
let mut buf = vec![0u8; 512];
|
||||
buf[..magic.len().min(512)].copy_from_slice(&magic[..magic.len().min(512)]);
|
||||
for _ in 0..rng.below(24) + 1 {
|
||||
let at = rng.below(buf.len());
|
||||
buf[at] = rng.byte();
|
||||
}
|
||||
buf.truncate(rng.below(buf.len()) + 1);
|
||||
run(target, "mutate", s, i, &buf, &mut f);
|
||||
}
|
||||
}
|
||||
|
||||
fn run<F: FnMut(&[u8])>(target: &str, generator: &str, seed: u64, i: usize, buf: &[u8], f: &mut F) {
|
||||
// Printed, not asserted: `cargo test` swallows stdout for passing tests and
|
||||
// shows it for failing ones, so this line is invisible until it is the last
|
||||
// thing before a panic — at which point it is exactly what is needed.
|
||||
println!(
|
||||
"harness {target}/{generator} seed={seed:#x} case={i} len={} :: \
|
||||
FREEMKV_HARNESS_SEED={seed} to reproduce",
|
||||
buf.len()
|
||||
);
|
||||
f(buf);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mpls_parse_never_panics() {
|
||||
sweep("mpls", b"MPLS", |b| {
|
||||
let _ = crate::mpls::parse(b);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn clpi_parse_never_panics() {
|
||||
sweep("clpi", b"HDMV", |b| {
|
||||
let _ = crate::clpi::parse(b);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn udf_name_parse_never_panics() {
|
||||
// No magic: the compression ID is the first byte and every value is legal
|
||||
// input to reject, so the "magic" is a byte the sweep will mutate anyway.
|
||||
sweep("udf_name", &[8], |b| {
|
||||
let _ = crate::udf::parse_udf_name(b);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ps_demuxer_feed_never_panics() {
|
||||
// Stateful, unlike the others: the demuxer carries a buffer across feeds, so
|
||||
// each case is fed to a FRESH demuxer and then a shared one. The shared pass
|
||||
// is what exercises cross-feed state — a start code split over a boundary,
|
||||
// a held PES completed by later bytes, the carry-over cap.
|
||||
let mut shared = crate::mux::ps::PsDemuxer::new();
|
||||
sweep("ps_demux", &[0x00, 0x00, 0x01, 0xBA], |b| {
|
||||
let mut fresh = crate::mux::ps::PsDemuxer::new();
|
||||
let _ = fresh.feed(b);
|
||||
let _ = shared.feed(b);
|
||||
});
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn mkv_lacing_split_never_panics() {
|
||||
// All four lacing modes, including the reserved bit pattern. A degenerate
|
||||
// fixed lace was a real defect found by audit round 5.
|
||||
sweep("mkv_lacing", &[0x00], |b| {
|
||||
for lacing in 0u8..=3 {
|
||||
let _ = crate::mux::mkvstream::split_lacing(lacing, b);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/// The generators must actually differ, or the sweep is one generator run three
|
||||
/// times and the coverage claim is false.
|
||||
#[test]
|
||||
fn the_three_generators_produce_different_inputs() {
|
||||
let mut seen: Vec<Vec<u8>> = Vec::new();
|
||||
sweep_n("probe", b"MPLS", 1, |b| seen.push(b.to_vec()));
|
||||
assert_eq!(seen.len(), 3, "one case per generator");
|
||||
assert_ne!(seen[0], seen[1], "random and magic+noise must differ");
|
||||
assert_ne!(seen[1], seen[2], "magic+noise and mutate must differ");
|
||||
assert!(
|
||||
seen[1].starts_with(b"MPLS"),
|
||||
"the magic+noise generator must actually carry the magic, or it never \
|
||||
reaches the parser body"
|
||||
);
|
||||
}
|
||||
|
||||
/// The same seed must replay the same bytes, or a reported failure cannot be
|
||||
/// reproduced and the harness is worthless as a regression tool.
|
||||
#[test]
|
||||
fn a_seed_replays_identically() {
|
||||
let mut a = Vec::new();
|
||||
let mut b = Vec::new();
|
||||
sweep_n("probe", b"MPLS", 4, |x| a.push(x.to_vec()));
|
||||
sweep_n("probe", b"MPLS", 4, |x| b.push(x.to_vec()));
|
||||
assert_eq!(a, b, "the same seed must produce the same cases");
|
||||
}
|
||||
|
||||
/// The harness is worthless if its cases die at the entry guards, so this
|
||||
/// MEASURES how deep they actually reach instead of assuming. A generator that
|
||||
/// never gets past a length or magic check exercises the first ten lines and
|
||||
/// nothing else — the fuzzing equivalent of a test that cannot fail.
|
||||
#[test]
|
||||
fn the_generators_actually_reach_the_parser_bodies() {
|
||||
// mpls::parse rejects at: len < 40, bad magic, then playlist_start + 10 >
|
||||
// len. Anything that returns Ok got all the way through the play-item loop.
|
||||
let mut ok = 0usize;
|
||||
let mut total = 0usize;
|
||||
sweep_n("reach", b"MPLS", 20000, |b| {
|
||||
total += 1;
|
||||
if crate::mpls::parse(b).is_ok() {
|
||||
ok += 1;
|
||||
}
|
||||
});
|
||||
assert!(
|
||||
ok > 0,
|
||||
"not one of {total} generated cases parsed successfully — the generators \
|
||||
are all being rejected at the entry guards, so this harness is testing \
|
||||
the guards and nothing behind them"
|
||||
);
|
||||
println!("mpls reach: {ok}/{total} cases parsed to completion");
|
||||
}
|
||||
+47
-5
@@ -16,11 +16,11 @@
|
||||
/// (case-insensitive), then requires an even run of ASCII hex digits. Any
|
||||
/// non-hex byte, or an odd length, yields `None`.
|
||||
pub fn parse_hex_bytes(s: &str) -> Option<Vec<u8>> {
|
||||
let body = strip_prefix(s.trim());
|
||||
let body = strip_hex_prefix(s.trim());
|
||||
let bytes = body.as_bytes();
|
||||
// Empty → empty Vec (a legitimately-empty variable-length field); odd length
|
||||
// is malformed. (`parse_hex_fixed` enforces a concrete length separately.)
|
||||
if bytes.len() % 2 != 0 {
|
||||
if !bytes.len().is_multiple_of(2) {
|
||||
return None;
|
||||
}
|
||||
let mut out = Vec::with_capacity(bytes.len() / 2);
|
||||
@@ -34,7 +34,7 @@ pub fn parse_hex_bytes(s: &str) -> Option<Vec<u8>> {
|
||||
/// prefix; requires EXACTLY `2*N` ASCII hex digits after it. `None` on any
|
||||
/// non-hex byte or a length mismatch.
|
||||
pub fn parse_hex_fixed<const N: usize>(s: &str) -> Option<[u8; N]> {
|
||||
let body = strip_prefix(s.trim());
|
||||
let body = strip_hex_prefix(s.trim());
|
||||
let bytes = body.as_bytes();
|
||||
if bytes.len() != 2 * N {
|
||||
return None;
|
||||
@@ -46,8 +46,33 @@ pub fn parse_hex_fixed<const N: usize>(s: &str) -> Option<[u8; N]> {
|
||||
Some(out)
|
||||
}
|
||||
|
||||
/// Strip a single leading `0x` / `0X` if present (case-insensitive).
|
||||
fn strip_prefix(s: &str) -> &str {
|
||||
/// Parse a hex string into a `u16`. Accepts an optional `0x`/`0X` prefix
|
||||
/// (case-insensitive) via the same [`strip_hex_prefix`] the byte parsers use.
|
||||
/// `None` on any non-hex content or overflow.
|
||||
///
|
||||
/// Exists so callers never hand-roll `from_str_radix(s.trim_start_matches("0x"), 16)`
|
||||
/// — a **case-sensitive** strip that silently dropped an uppercase-`0X` value.
|
||||
/// (That reintroduced-in-keydb bug is exactly what this module was built to kill;
|
||||
/// the integer fields now share the one prefix rule.)
|
||||
pub fn parse_hex_u16(s: &str) -> Option<u16> {
|
||||
u16::from_str_radix(strip_hex_prefix(s.trim()), 16).ok()
|
||||
}
|
||||
|
||||
/// Parse a hex string into a `u32`. See [`parse_hex_u16`].
|
||||
pub fn parse_hex_u32(s: &str) -> Option<u32> {
|
||||
u32::from_str_radix(strip_hex_prefix(s.trim()), 16).ok()
|
||||
}
|
||||
|
||||
/// Parse a hex string into a `u8`. See [`parse_hex_u16`].
|
||||
pub fn parse_hex_u8(s: &str) -> Option<u8> {
|
||||
u8::from_str_radix(strip_hex_prefix(s.trim()), 16).ok()
|
||||
}
|
||||
|
||||
/// Strip a single leading `0x` / `0X` if present (case-insensitive). Public so
|
||||
/// callers that only need the prefix rule (e.g. normalizing a disc hash) reuse
|
||||
/// the one definition instead of hand-rolling a case-sensitive
|
||||
/// `trim_start_matches("0x")`.
|
||||
pub fn strip_hex_prefix(s: &str) -> &str {
|
||||
s.strip_prefix("0x")
|
||||
.or_else(|| s.strip_prefix("0X"))
|
||||
.unwrap_or(s)
|
||||
@@ -95,6 +120,23 @@ mod tests {
|
||||
assert_eq!(parse_hex_fixed::<16>(&s), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn hex_ints_accept_both_prefix_cases_and_bare() {
|
||||
// The regression the keydb device-key bug hit: uppercase `0X` must parse
|
||||
// identically to `0x` and to a bare value.
|
||||
assert_eq!(parse_hex_u16("0x0001"), Some(1));
|
||||
assert_eq!(parse_hex_u16("0X0001"), Some(1));
|
||||
assert_eq!(parse_hex_u16("0001"), Some(1));
|
||||
assert_eq!(parse_hex_u16(" 0XABCD "), Some(0xABCD));
|
||||
assert_eq!(parse_hex_u32("0X00000002"), Some(2));
|
||||
assert_eq!(parse_hex_u32("deadbeef"), Some(0xDEAD_BEEF));
|
||||
assert_eq!(parse_hex_u8("0X03"), Some(3));
|
||||
assert_eq!(parse_hex_u8("ff"), Some(0xFF));
|
||||
// Overflow / non-hex → None.
|
||||
assert_eq!(parse_hex_u8("0x1FF"), None);
|
||||
assert_eq!(parse_hex_u16("0xzz"), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn bytes_variable_length_and_odd_rejected() {
|
||||
assert_eq!(parse_hex_bytes("0xAABBCC"), Some(vec![0xAA, 0xBB, 0xCC]));
|
||||
|
||||
+193
-1
@@ -49,13 +49,30 @@ pub struct DriveId {
|
||||
pub raw_gc_010c: Vec<u8>,
|
||||
}
|
||||
|
||||
/// SPC-4 standard INQUIRY data: 36 bytes through `product_revision`. Anything
|
||||
/// shorter cannot populate the identity fields this type promises.
|
||||
const INQUIRY_STANDARD_LEN: usize = 36;
|
||||
|
||||
impl DriveId {
|
||||
/// Probe a real drive via SCSI and build its identity.
|
||||
pub fn from_drive(transport: &mut dyn ScsiTransport) -> Result<Self> {
|
||||
// INQUIRY — SPC-4 §6.4
|
||||
let mut inquiry = vec![0u8; 96];
|
||||
let cdb_inq = [0x12, 0x00, 0x00, 0x00, 0x60, 0x00];
|
||||
transport.execute(&cdb_inq, DataDirection::FromDevice, &mut inquiry, 5000)?;
|
||||
let inq = transport.execute(&cdb_inq, DataDirection::FromDevice, &mut inquiry, 5000)?;
|
||||
// `bytes_transferred` is device-reported and untrusted — the same rule
|
||||
// the two GET CONFIGURATION calls below already apply. It was ignored
|
||||
// here, and the buffer is pre-zeroed, so a drive answering GOOD with a
|
||||
// short or empty data phase (a USB-SATA bridge mid-wedge does exactly
|
||||
// this) decoded to blank identity strings and a byte 0 of 0x00. Every
|
||||
// platform enumerator gates on `raw_inquiry[0] & 0x1F == OPTICAL`, so
|
||||
// 0x00 reads as DIRECT ACCESS and the drive silently disappears from
|
||||
// the device list instead of reporting a failed probe.
|
||||
if inq.bytes_transferred < INQUIRY_STANDARD_LEN {
|
||||
return Err(crate::error::Error::DriveInquiryShort);
|
||||
}
|
||||
// Never decode past what the drive actually sent.
|
||||
inquiry.truncate(inq.bytes_transferred.min(inquiry.len()));
|
||||
|
||||
// GET CONFIGURATION Feature 010Ch — MMC-6 §6.6.
|
||||
// Best-effort: 010Ch (Firmware Information) is an optional feature.
|
||||
@@ -262,6 +279,54 @@ mod tests {
|
||||
/// Spec: SPC-4 §6.4.2 — bytes[8:16] are vendor ID; a truncated buffer
|
||||
/// (e.g. a device that reports fewer than 8 bytes) must not panic.
|
||||
/// Mutation: removing the `data.len() > start` guard makes it panic on short inputs.
|
||||
/// A drive that answers INQUIRY with GOOD status but a short or empty
|
||||
/// data phase must fail the probe, not present as a blank drive.
|
||||
///
|
||||
/// The buffer is pre-zeroed, so decoding it unconditionally yielded empty
|
||||
/// vendor/product/revision strings and a byte 0 of 0x00. Every platform
|
||||
/// enumerator gates on `raw_inquiry[0] & 0x1F == SCSI_PERIPHERAL_TYPE_OPTICAL`,
|
||||
/// and 0x00 is DIRECT ACCESS — so the drive silently vanished from the
|
||||
/// device list rather than reporting that its identity probe failed. A
|
||||
/// USB-SATA bridge mid-wedge does exactly this.
|
||||
///
|
||||
/// The two GET CONFIGURATION calls in the same function already clamped on
|
||||
/// `bytes_transferred`, with a comment calling it untrusted; INQUIRY, three
|
||||
/// lines above them, discarded it.
|
||||
#[test]
|
||||
fn inquiry_with_a_short_data_phase_fails_instead_of_reporting_a_blank_drive() {
|
||||
/// GOOD status, no sense, and only `n` bytes written.
|
||||
struct ShortInquiry(usize);
|
||||
impl ScsiTransport for ShortInquiry {
|
||||
fn execute(
|
||||
&mut self,
|
||||
_cdb: &[u8],
|
||||
_dir: DataDirection,
|
||||
_buf: &mut [u8],
|
||||
_timeout_ms: u32,
|
||||
) -> Result<ScsiResult> {
|
||||
Ok(ScsiResult {
|
||||
status: 0,
|
||||
sense: [0u8; 32],
|
||||
bytes_transferred: self.0,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Empty data phase — the case that made a real drive disappear.
|
||||
assert!(matches!(
|
||||
DriveId::from_drive(&mut ShortInquiry(0)),
|
||||
Err(crate::error::Error::DriveInquiryShort)
|
||||
));
|
||||
// One byte short of the SPC-4 standard 36-byte header.
|
||||
assert!(matches!(
|
||||
DriveId::from_drive(&mut ShortInquiry(35)),
|
||||
Err(crate::error::Error::DriveInquiryShort)
|
||||
));
|
||||
// Exactly the standard length is acceptable: the optional
|
||||
// vendor-specific tail past byte 36 is allowed to be absent.
|
||||
assert!(DriveId::from_drive(&mut ShortInquiry(36)).is_ok());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ascii_field_short_buffer_returns_empty() {
|
||||
// Buffer of length 5: start=8 is beyond the end → empty string.
|
||||
@@ -339,9 +404,136 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
/// `ascii_field`'s guard is `data.len() > start` (strictly greater), not
|
||||
/// `>=`: a buffer whose length is exactly `start` has NO byte at that
|
||||
/// offset, so it must still yield empty, not attempt to slice.
|
||||
/// Mutation: `>` -> `>=` would try to slice `data[start..]` when
|
||||
/// `data.len() == start`, which panics (empty range at the very end is
|
||||
/// fine, but the guard's job is the `< start` case below it — pinning the
|
||||
/// exact boundary catches an off-by-one either direction).
|
||||
#[test]
|
||||
fn ascii_field_boundary_len_equals_start_is_empty() {
|
||||
let buf = vec![0u8; 8];
|
||||
assert_eq!(ascii_field(&buf, 8, 16), "");
|
||||
}
|
||||
|
||||
/// One byte past the boundary: `data.len() == start + 1` must extract
|
||||
/// that single byte (clamped to `end`), proving the guard is `>` and not
|
||||
/// off by one in the other direction.
|
||||
#[test]
|
||||
fn ascii_field_boundary_len_one_past_start_extracts_one_byte() {
|
||||
let mut buf = vec![0u8; 9];
|
||||
buf[8] = b'X';
|
||||
assert_eq!(ascii_field(&buf, 8, 16), "X");
|
||||
}
|
||||
|
||||
/// `Display` renders the four trimmed identity fields space-separated —
|
||||
/// the human-readable counterpart of `match_key`'s pipe-separated form.
|
||||
/// Not exercised anywhere else in this test module.
|
||||
/// Mutation: replacing the `fmt` body with `Ok(Default::default())`
|
||||
/// writes nothing at all, so formatting any `DriveId` yields "".
|
||||
#[test]
|
||||
fn display_formats_trimmed_fields_space_separated() {
|
||||
let mut inquiry = vec![0u8; 96];
|
||||
inquiry[8..16].copy_from_slice(b"PIONEER ");
|
||||
inquiry[16..32].copy_from_slice(b"BD-RW BDR-S09 ");
|
||||
inquiry[32..36].copy_from_slice(b"1.34");
|
||||
inquiry[36..43].copy_from_slice(b" 16/04/");
|
||||
let id = DriveId::from_inquiry(&inquiry, "201604250000");
|
||||
assert_eq!(id.to_string(), "PIONEER BD-RW BDR-S09 1.34 16/04/");
|
||||
}
|
||||
|
||||
/// GET CONFIGURATION failure (transport error) must not abort the
|
||||
/// identity probe — firmware_date is empty, raw_gc_010c is empty.
|
||||
/// Mutation: propagating the GET_CONFIGURATION error with `?` aborts from_drive.
|
||||
/// Transport whose GET CONFIGURATION responses report an exact,
|
||||
/// caller-chosen `bytes_transferred` for each of the two GC features
|
||||
/// (010Ch firmware date / 0108h serial), so the `end > 12` / `> 12`
|
||||
/// boundary guards can be pinned precisely. INQUIRY always succeeds.
|
||||
struct FixedGcCountTransport {
|
||||
firmware_bytes: usize,
|
||||
serial_bytes: usize,
|
||||
}
|
||||
|
||||
impl ScsiTransport for FixedGcCountTransport {
|
||||
fn execute(
|
||||
&mut self,
|
||||
cdb: &[u8],
|
||||
_dir: DataDirection,
|
||||
buf: &mut [u8],
|
||||
_timeout_ms: u32,
|
||||
) -> Result<ScsiResult> {
|
||||
for b in buf.iter_mut() {
|
||||
*b = b'Z';
|
||||
}
|
||||
let bytes_transferred = match cdb.first() {
|
||||
Some(&0x12) => buf.len(),
|
||||
Some(&0x46) if cdb[3] == 0x0C => self.firmware_bytes,
|
||||
Some(&0x46) if cdb[3] == 0x08 => self.serial_bytes,
|
||||
_ => buf.len(),
|
||||
};
|
||||
Ok(ScsiResult {
|
||||
status: 0,
|
||||
bytes_transferred,
|
||||
sense: [0u8; 32],
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
/// `end > 12` in the firmware-date branch (`from_drive`) is a strict
|
||||
/// inequality: `bytes_transferred == 12` reports the field absent
|
||||
/// (offset 12 is the first byte of the 12-char date; a count of exactly
|
||||
/// 12 covers bytes 0..12, none of which is the date), so `firmware_date`
|
||||
/// must be empty, not the mutant's off-by-one read.
|
||||
/// Mutation: `>` -> `>=` would try `gc[12..12]` at the boundary — an
|
||||
/// empty but non-panicking slice — silently reporting "present" data
|
||||
/// that is actually all outside the transferred count.
|
||||
#[test]
|
||||
fn from_drive_firmware_date_boundary_exactly_12_is_empty() {
|
||||
let mut t = FixedGcCountTransport {
|
||||
firmware_bytes: 12,
|
||||
serial_bytes: 0,
|
||||
};
|
||||
let id = DriveId::from_drive(&mut t).unwrap();
|
||||
assert_eq!(id.firmware_date, "");
|
||||
}
|
||||
|
||||
/// One byte past the boundary (`bytes_transferred == 13`) must extract
|
||||
/// exactly the one available date byte (offset 12), proving the guard
|
||||
/// is `>` and the slice end is clamped to `end`, not always to 24.
|
||||
#[test]
|
||||
fn from_drive_firmware_date_boundary_13_extracts_one_byte() {
|
||||
let mut t = FixedGcCountTransport {
|
||||
firmware_bytes: 13,
|
||||
serial_bytes: 0,
|
||||
};
|
||||
let id = DriveId::from_drive(&mut t).unwrap();
|
||||
assert_eq!(id.firmware_date, "Z");
|
||||
}
|
||||
|
||||
/// Same `> 12` boundary for the serial-number branch: exactly 12
|
||||
/// transferred bytes must yield an empty serial.
|
||||
#[test]
|
||||
fn from_drive_serial_boundary_exactly_12_is_empty() {
|
||||
let mut t = FixedGcCountTransport {
|
||||
firmware_bytes: 0,
|
||||
serial_bytes: 12,
|
||||
};
|
||||
let id = DriveId::from_drive(&mut t).unwrap();
|
||||
assert_eq!(id.serial_number, "");
|
||||
}
|
||||
|
||||
/// One byte past the serial boundary extracts exactly that byte.
|
||||
#[test]
|
||||
fn from_drive_serial_boundary_13_extracts_one_byte() {
|
||||
let mut t = FixedGcCountTransport {
|
||||
firmware_bytes: 0,
|
||||
serial_bytes: 13,
|
||||
};
|
||||
let id = DriveId::from_drive(&mut t).unwrap();
|
||||
assert_eq!(id.serial_number, "Z");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn from_drive_gc_failure_yields_empty_firmware_date() {
|
||||
struct GcFailTransport;
|
||||
|
||||
+1168
-107
File diff suppressed because it is too large
Load Diff
+4
-4
@@ -133,10 +133,10 @@ where
|
||||
match rx.recv_timeout(slice) {
|
||||
Ok(v) => return Ok(v),
|
||||
Err(RecvTimeoutError::Timeout) => {
|
||||
if let Some(h) = halt {
|
||||
if h.is_cancelled() {
|
||||
return Err(BoundedError::Halted);
|
||||
}
|
||||
if let Some(h) = halt
|
||||
&& h.is_cancelled()
|
||||
{
|
||||
return Err(BoundedError::Halted);
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return Err(BoundedError::Timeout);
|
||||
|
||||
@@ -249,6 +249,27 @@ impl Drop for BytePrefetcher {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// `RECYCLE_DEPTH` must be one MORE than `FORWARD_DEPTH` per its own
|
||||
/// doc comment: the producer needs at least one buffer to fill while
|
||||
/// the consumer holds the other `FORWARD_DEPTH`-worth in flight. A
|
||||
/// `+` -> `*`/`-` mutation on `FORWARD_DEPTH + 1` would under-size the
|
||||
/// recycle channel (e.g. `FORWARD_DEPTH * 1 == FORWARD_DEPTH`, one
|
||||
/// short), which starves the producer of a spare buffer.
|
||||
#[test]
|
||||
fn recycle_depth_is_forward_depth_plus_one() {
|
||||
assert_eq!(RECYCLE_DEPTH, FORWARD_DEPTH + 1);
|
||||
assert_eq!(RECYCLE_DEPTH, 3, "FORWARD_DEPTH is 2, so recycle must be 3");
|
||||
}
|
||||
|
||||
/// `DEFAULT_CHUNK_BYTES` is documented as 16 MiB. Pins the literal so a
|
||||
/// `*` -> `+`/`/` mutation on either factor (16 * 1024 * 1024) is
|
||||
/// caught by a concrete, spec-derived expected value rather than by
|
||||
/// recomputing the same expression.
|
||||
#[test]
|
||||
fn default_chunk_bytes_is_16_mib() {
|
||||
assert_eq!(DEFAULT_CHUNK_BYTES, 16_777_216, "documented as 16 MiB");
|
||||
}
|
||||
|
||||
/// Endless reader: every `read` fills the whole buffer and never
|
||||
/// hits EOF, so the producer keeps trying to push batches forward
|
||||
/// until the forward channel disconnects. Exactly the shape that
|
||||
|
||||
@@ -23,7 +23,7 @@
|
||||
use std::fs::File;
|
||||
use std::os::unix::io::AsRawFd;
|
||||
|
||||
pub(super) fn hint_sequential(file: &File, _len_bytes: u64) {
|
||||
pub(crate) fn hint_sequential(file: &File, _len_bytes: u64) {
|
||||
// Best-effort: return value ignored. A fadvise failure has no
|
||||
// user-observable consequence.
|
||||
unsafe {
|
||||
@@ -34,7 +34,7 @@ pub(super) fn hint_sequential(file: &File, _len_bytes: u64) {
|
||||
/// Drop pages in the half-open byte range `[start, start+len)` from
|
||||
/// the page cache. Called periodically by `read_sectors` to bound the
|
||||
/// read-side page cache pressure.
|
||||
pub(super) fn drop_window(file: &File, start: u64, len: u64) {
|
||||
pub(crate) fn drop_window(file: &File, start: u64, len: u64) {
|
||||
unsafe {
|
||||
libc::posix_fadvise(
|
||||
file.as_raw_fd(),
|
||||
@@ -58,7 +58,7 @@ pub(super) fn drop_window(file: &File, start: u64, len: u64) {
|
||||
/// can only pre-stage a tiny slice of the next batch. An explicit
|
||||
/// `readahead()` of the same size as the current batch tells the
|
||||
/// kernel to queue the full next-batch read now.
|
||||
pub(super) fn prefetch(file: &File, offset: u64, len: u64) {
|
||||
pub(crate) fn prefetch(file: &File, offset: u64, len: u64) {
|
||||
unsafe {
|
||||
libc::readahead(file.as_raw_fd(), offset as i64, len as usize);
|
||||
}
|
||||
|
||||
@@ -15,7 +15,7 @@ use std::os::unix::io::AsRawFd;
|
||||
/// pipeline depth.
|
||||
const RDADVISE_MAX_BYTES: i64 = 64 * 1024 * 1024;
|
||||
|
||||
pub(super) fn hint_sequential(file: &File, len_bytes: u64) {
|
||||
pub(crate) fn hint_sequential(file: &File, len_bytes: u64) {
|
||||
let bytes = (len_bytes as i64).min(RDADVISE_MAX_BYTES);
|
||||
let mut ra = libc::radvisory {
|
||||
ra_offset: 0,
|
||||
@@ -33,14 +33,14 @@ pub(super) fn hint_sequential(file: &File, len_bytes: u64) {
|
||||
/// approximation: no-op. macOS's unified buffer cache is generally
|
||||
/// less prone to the pin-everything pathology that triggers the
|
||||
/// regression on Linux NFS clients.
|
||||
pub(super) fn drop_window(_file: &File, _start: u64, _len: u64) {}
|
||||
pub(crate) fn drop_window(_file: &File, _start: u64, _len: u64) {}
|
||||
|
||||
/// Async-prefetch the byte range `[offset, offset+len)`. macOS uses
|
||||
/// the same `fcntl(F_RDADVISE, &radvisory)` primitive as the open-
|
||||
/// time sequential hint, just targeted at a moving window instead of
|
||||
/// the whole file. The kernel queues I/O for the requested range and
|
||||
/// returns immediately.
|
||||
pub(super) fn prefetch(file: &File, offset: u64, len: u64) {
|
||||
pub(crate) fn prefetch(file: &File, offset: u64, len: u64) {
|
||||
let bytes = (len as i64).min(RDADVISE_MAX_BYTES);
|
||||
let mut ra = libc::radvisory {
|
||||
ra_offset: offset as libc::off_t,
|
||||
|
||||
@@ -48,22 +48,25 @@
|
||||
//! far smaller than our 16 MiB app-level batch.
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
mod linux;
|
||||
pub(crate) mod linux;
|
||||
#[cfg(target_os = "macos")]
|
||||
mod macos;
|
||||
pub(crate) mod macos;
|
||||
#[cfg(not(any(target_os = "linux", target_os = "macos", target_os = "windows")))]
|
||||
mod other;
|
||||
pub(crate) mod other;
|
||||
#[cfg(target_os = "windows")]
|
||||
mod windows;
|
||||
pub(crate) mod windows;
|
||||
|
||||
// The page-cache hints are shared with any other file-backed sector source:
|
||||
// `dirimage` reads host files the same way and needs the same eviction, or a
|
||||
// large rip pins every byte it has read (see this module's DONTNEED note).
|
||||
#[cfg(target_os = "linux")]
|
||||
use linux as platform;
|
||||
pub(crate) use linux as platform;
|
||||
#[cfg(target_os = "macos")]
|
||||
use macos as platform;
|
||||
pub(crate) use macos as platform;
|
||||
#[cfg(not(any(target_os = "linux", target_os = "macos", target_os = "windows")))]
|
||||
use other as platform;
|
||||
pub(crate) use other as platform;
|
||||
#[cfg(target_os = "windows")]
|
||||
use windows as platform;
|
||||
pub(crate) use windows as platform;
|
||||
|
||||
use std::fs::File;
|
||||
use std::io::{Read, Seek, SeekFrom};
|
||||
@@ -85,11 +88,27 @@ use crate::consts::{SECTOR_BYTES, SECTOR_BYTES_U64};
|
||||
/// pressure concurrent writes. Override via `FREEMKV_READ_DROP_CHUNK_MIB`.
|
||||
const READ_DROP_CHUNK_BYTES_DEFAULT: u64 = 32 * 1024 * 1024;
|
||||
|
||||
/// Upper bound (in MiB) accepted from `FREEMKV_READ_DROP_CHUNK_MIB`. 64 GiB —
|
||||
/// generous for any real medium, and small enough that `n * 1024 * 1024` cannot
|
||||
/// wrap `u64`. Mirrors `WRITEBACK_CHUNK_MIB_MAX`, whose identical multiply is
|
||||
/// bounded for exactly this reason: without the bound, a value above 2^44
|
||||
/// overflows — a panic on the first ISO open in an overflow-checked build, and in
|
||||
/// release a wrap to a near-zero window that fires `drop_window` on every read.
|
||||
/// Out-of-range values fall back to the default.
|
||||
const READ_DROP_CHUNK_MIB_MAX: u64 = 64 * 1024;
|
||||
|
||||
fn read_drop_chunk_bytes() -> u64 {
|
||||
std::env::var("FREEMKV_READ_DROP_CHUNK_MIB")
|
||||
.ok()
|
||||
.and_then(|v| v.parse::<u64>().ok())
|
||||
.filter(|&n| n > 0)
|
||||
resolve_read_drop_chunk(
|
||||
std::env::var("FREEMKV_READ_DROP_CHUNK_MIB")
|
||||
.ok()
|
||||
.and_then(|v| v.parse::<u64>().ok()),
|
||||
)
|
||||
}
|
||||
|
||||
/// The pure part of [`read_drop_chunk_bytes`], split out so the bound is
|
||||
/// testable without mutating process environment.
|
||||
fn resolve_read_drop_chunk(mib: Option<u64>) -> u64 {
|
||||
mib.filter(|&n| n > 0 && n <= READ_DROP_CHUNK_MIB_MAX)
|
||||
.map(|n| n * 1024 * 1024)
|
||||
.unwrap_or(READ_DROP_CHUNK_BYTES_DEFAULT)
|
||||
}
|
||||
@@ -171,12 +190,20 @@ impl SectorSource for FileSectorSource {
|
||||
) -> Result<usize> {
|
||||
let count = count as u32;
|
||||
let bytes = count as usize * SECTOR_BYTES;
|
||||
debug_assert!(
|
||||
out.len() >= bytes,
|
||||
"FileSectorSource::read_sectors: out len {} < requested {}",
|
||||
out.len(),
|
||||
bytes
|
||||
);
|
||||
// A real check, not a debug_assert: this is a public `SectorSource` impl,
|
||||
// so an undersized `out` is caller input, and `out[..bytes]` below would
|
||||
// panic with 'range end index out of range' in release where the assert is
|
||||
// compiled away. `Drive::read_fua` already carries exactly this guard, with
|
||||
// a comment recording the same panic being fixed there — this impl was
|
||||
// simply never given it, and `PrefetchedSectorSource` has a regression test
|
||||
// for the case that this one lacked.
|
||||
if out.len() < bytes {
|
||||
return Err(Error::DiscRead {
|
||||
sector: lba as u64,
|
||||
status: None,
|
||||
sense: None,
|
||||
});
|
||||
}
|
||||
if count == 0 {
|
||||
return Ok(0);
|
||||
}
|
||||
@@ -219,6 +246,40 @@ mod tests {
|
||||
use std::io::Write;
|
||||
use tempfile::tempdir;
|
||||
|
||||
/// An undersized output buffer must return an error, never panic. This is a
|
||||
/// public `SectorSource` impl, so buffer length is caller input, and the guard
|
||||
/// used to be a `debug_assert!` — compiled out in release, where the
|
||||
/// `out[..bytes]` slice then panicked with 'range end index out of range'.
|
||||
///
|
||||
/// `Drive::read_fua` already carries this exact guard with a comment recording
|
||||
/// the same panic being fixed there, and `PrefetchedSectorSource` has
|
||||
/// `direct_read_too_small_buffer_errors` for the same case; this impl had
|
||||
/// neither.
|
||||
#[test]
|
||||
fn read_sectors_with_an_undersized_buffer_errors_rather_than_panicking() {
|
||||
let dir = tempdir().unwrap();
|
||||
let iso = dir.path().join("t.iso");
|
||||
make_iso(&iso, 8);
|
||||
let mut src = FileSectorSource::open(&iso).expect("iso opens");
|
||||
|
||||
// Ask for four sectors but supply room for barely more than one.
|
||||
let mut out = vec![0u8; SECTOR_BYTES + 1];
|
||||
let err = src
|
||||
.read_sectors(0, 4, &mut out, false)
|
||||
.expect_err("an undersized buffer must be an error, not a panic");
|
||||
assert!(
|
||||
matches!(err, Error::DiscRead { .. }),
|
||||
"expected DiscRead, got {err:?}"
|
||||
);
|
||||
|
||||
// Exactly-sized still works, so the guard is not off by one.
|
||||
let mut out = vec![0u8; 4 * SECTOR_BYTES];
|
||||
assert_eq!(
|
||||
src.read_sectors(0, 4, &mut out, false).unwrap(),
|
||||
4 * SECTOR_BYTES
|
||||
);
|
||||
}
|
||||
|
||||
/// Build a deterministic ISO of `sectors` sectors where sector `n`
|
||||
/// is filled with the byte pattern `((n & 0xff) as u8)`. Lets us
|
||||
/// verify any sector by content alone.
|
||||
@@ -376,19 +437,36 @@ mod tests {
|
||||
// Additional coverage.
|
||||
// ---------------------------------------------------------------
|
||||
|
||||
/// `count == 0` must short-circuit to Ok(0) WITHOUT seeking or
|
||||
/// reading, even at an out-of-range LBA — the early-return guard
|
||||
/// runs before any I/O. Grounding: `if count == 0 { return Ok(0) }`.
|
||||
/// `count == 0` must short-circuit to Ok(0) WITHOUT seeking or reading,
|
||||
/// even at an out-of-range LBA. Grounding: `if count == 0 { return Ok(0) }`.
|
||||
///
|
||||
/// The `Ok(0)` return alone proves nothing: with the guard deleted, a seek
|
||||
/// past EOF succeeds (POSIX permits seeking beyond the end of a file) and a
|
||||
/// zero-length `read_exact` returns `Ok(())` immediately, so the call still
|
||||
/// returns `Ok(0)`. The observable difference is the file's cursor — the
|
||||
/// seek MOVES it to `lba * 2048`. Assert on that, so the guard is what the
|
||||
/// test is actually measuring.
|
||||
#[test]
|
||||
fn zero_count_returns_zero_no_io() {
|
||||
let dir = tempdir().unwrap();
|
||||
let path = dir.path().join("zc.iso");
|
||||
make_iso(&path, 4);
|
||||
let mut src = FileSectorSource::open(&path).unwrap();
|
||||
let before = src.file.stream_position().expect("cursor readable");
|
||||
assert_eq!(before, 0, "a freshly opened file starts at offset 0");
|
||||
// LBA far past EOF — must not matter because count==0 returns early.
|
||||
let mut buf = [0u8; 1];
|
||||
let n = src.read_sectors(1_000_000, 0, &mut buf, false).unwrap();
|
||||
assert_eq!(n, 0);
|
||||
assert_eq!(
|
||||
src.file.stream_position().expect("cursor readable"),
|
||||
before,
|
||||
"count == 0 must return before the seek — an unmoved cursor is the \
|
||||
only observable proof that no I/O was issued"
|
||||
);
|
||||
// And the drop-window accounting must not have advanced either.
|
||||
assert_eq!(src.bytes_read_since_drop, 0);
|
||||
assert_eq!(src.drop_window_start, 0);
|
||||
}
|
||||
|
||||
/// Reading past EOF must ERROR (read_exact's UnexpectedEof), never
|
||||
@@ -518,4 +596,36 @@ mod tests {
|
||||
lba += batch as u32;
|
||||
}
|
||||
}
|
||||
|
||||
/// `FREEMKV_READ_DROP_CHUNK_MIB` must be BOUNDED before the MiB→byte
|
||||
/// multiply, exactly as its writeback twin bounds the identical multiply.
|
||||
/// Unbounded, any value above 2^44 overflowed `n * 1024 * 1024`: a panic on
|
||||
/// the first ISO open in an overflow-checked build, and in release a wrap to
|
||||
/// a near-zero window that fires `drop_window` on essentially every read.
|
||||
#[test]
|
||||
fn read_drop_chunk_env_is_bounded_before_the_multiply() {
|
||||
// Default when unset / zero / out of range.
|
||||
assert_eq!(resolve_read_drop_chunk(None), READ_DROP_CHUNK_BYTES_DEFAULT);
|
||||
assert_eq!(
|
||||
resolve_read_drop_chunk(Some(0)),
|
||||
READ_DROP_CHUNK_BYTES_DEFAULT
|
||||
);
|
||||
// The overflow value: `u64::MAX * 1024 * 1024` panicked here.
|
||||
assert_eq!(
|
||||
resolve_read_drop_chunk(Some(u64::MAX)),
|
||||
READ_DROP_CHUNK_BYTES_DEFAULT
|
||||
);
|
||||
assert_eq!(
|
||||
resolve_read_drop_chunk(Some(READ_DROP_CHUNK_MIB_MAX + 1)),
|
||||
READ_DROP_CHUNK_BYTES_DEFAULT
|
||||
);
|
||||
// In-range values convert MiB→bytes. Mutation: `* 1024` breaks this.
|
||||
assert_eq!(resolve_read_drop_chunk(Some(1)), 1024 * 1024);
|
||||
assert_eq!(
|
||||
resolve_read_drop_chunk(Some(READ_DROP_CHUNK_MIB_MAX)),
|
||||
READ_DROP_CHUNK_MIB_MAX * 1024 * 1024
|
||||
);
|
||||
// And the bound itself keeps the multiply inside u64.
|
||||
assert!((READ_DROP_CHUNK_MIB_MAX as u128) * 1024 * 1024 <= u64::MAX as u128);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,8 +4,8 @@
|
||||
|
||||
use std::fs::File;
|
||||
|
||||
pub(super) fn hint_sequential(_file: &File, _len_bytes: u64) {}
|
||||
pub(crate) fn hint_sequential(_file: &File, _len_bytes: u64) {}
|
||||
|
||||
pub(super) fn drop_window(_file: &File, _start: u64, _len: u64) {}
|
||||
pub(crate) fn drop_window(_file: &File, _start: u64, _len: u64) {}
|
||||
|
||||
pub(super) fn prefetch(_file: &File, _offset: u64, _len: u64) {}
|
||||
pub(crate) fn prefetch(_file: &File, _offset: u64, _len: u64) {}
|
||||
|
||||
@@ -9,7 +9,7 @@ use std::fs::File;
|
||||
/// No-op stub. `FILE_FLAG_SEQUENTIAL_SCAN` can only be set at
|
||||
/// `CreateFile` open time, which the plain `File::open` path does not
|
||||
/// do, so there is no post-open hint to issue here.
|
||||
pub(super) fn hint_sequential(_file: &File, _len_bytes: u64) {
|
||||
pub(crate) fn hint_sequential(_file: &File, _len_bytes: u64) {
|
||||
tracing::debug!(
|
||||
target: "mux",
|
||||
"FileSectorSource hint_sequential: windows no-op stub"
|
||||
@@ -19,10 +19,10 @@ pub(super) fn hint_sequential(_file: &File, _len_bytes: u64) {
|
||||
/// Windows page-cache eviction is not exposed via a posix_fadvise
|
||||
/// equivalent. The kernel does its own working-set management. No-op
|
||||
/// for now.
|
||||
pub(super) fn drop_window(_file: &File, _start: u64, _len: u64) {}
|
||||
pub(crate) fn drop_window(_file: &File, _start: u64, _len: u64) {}
|
||||
|
||||
/// Windows async-prefetch hint. With FILE_FLAG_SEQUENTIAL_SCAN at
|
||||
/// open the kernel already prefetches aggressively, so there's no
|
||||
/// per-range hint we'd add on top. No-op stub for parity with the
|
||||
/// posix platforms.
|
||||
pub(super) fn prefetch(_file: &File, _offset: u64, _len: u64) {}
|
||||
pub(crate) fn prefetch(_file: &File, _offset: u64, _len: u64) {}
|
||||
|
||||
@@ -0,0 +1,283 @@
|
||||
//! `write_image` — write an image-level source out as a sector image.
|
||||
//!
|
||||
//! This is the plain image writer: sectors in from any [`SectorSource`], bytes
|
||||
//! out to a file, in order, once. It is what an `iso://` DESTINATION means when
|
||||
//! the source is not a physical drive.
|
||||
//!
|
||||
//! # Why this is not `freemkv_engine::copy`
|
||||
//!
|
||||
//! The engine's `copy` is the RECOVERY path — mapfile sidecar, `--multipass`
|
||||
//! sweep/patch, damage-jump, ECC-aware batching, auto-resume. Every one of those
|
||||
//! exists because an optical drive returns read errors on marginal media. A
|
||||
//! file-backed or synthesized source has no marginal media: a read either
|
||||
//! succeeds or the underlying file is broken, and retrying it is pointless.
|
||||
//!
|
||||
//! Routing a non-drive source through the recovery path is not merely wasteful,
|
||||
//! it is wrong. Its mapfile identity check compares AACS unit keys and the VID,
|
||||
//! both of which are empty for an already-decrypted source, so identity passes
|
||||
//! for ANY such source: a second run with a different input to the same output
|
||||
//! path would resume over the previous image and produce wrong content at exit
|
||||
//! zero. Keeping the two paths separate makes that unrepresentable.
|
||||
//!
|
||||
//! So: drive sources get `freemkv_engine::copy`. Everything else gets this.
|
||||
|
||||
use crate::consts::SECTOR_BYTES;
|
||||
use crate::error::{Error, Result};
|
||||
use crate::halt::Halt;
|
||||
use crate::sector::SectorSource;
|
||||
use std::fs::File;
|
||||
use std::io::{BufWriter, Write};
|
||||
use std::path::Path;
|
||||
|
||||
/// Sectors per read/write batch. 4 MiB — large enough that per-call overhead
|
||||
/// disappears against a file-backed source, small enough that the buffer is not
|
||||
/// a notable allocation and cancellation stays responsive.
|
||||
const BATCH_SECTORS: u32 = 2048;
|
||||
|
||||
/// Write `total_sectors` sectors from `reader` to `dest`.
|
||||
///
|
||||
/// Reads sequentially from LBA 0 and writes in order, so the output is a faithful
|
||||
/// image of whatever the source presents — decrypted if the caller wrapped the
|
||||
/// source in a [`DecryptingSectorSource`](crate::sector::decrypting::DecryptingSectorSource),
|
||||
/// ciphertext if it did not. This function performs no decryption itself and makes
|
||||
/// no decryption decision; that belongs to the caller, which knows whether the run
|
||||
/// is `--raw`.
|
||||
///
|
||||
/// `on_progress` is called after each batch with the cumulative byte count, for
|
||||
/// front-end progress reporting. It must not block.
|
||||
///
|
||||
/// `halt` is checked once per batch. On cancellation the partial file is left in
|
||||
/// place — the caller decides whether a partial image is worth keeping, and
|
||||
/// deleting a multi-gigabyte file the user may want to inspect is not this
|
||||
/// function's call to make.
|
||||
///
|
||||
/// Returns the number of bytes written.
|
||||
///
|
||||
/// # Errors
|
||||
///
|
||||
/// - [`Error::Halted`] if `halt` was cancelled.
|
||||
/// - [`Error::IoError`] if the destination cannot be created or written.
|
||||
/// - Whatever the source's `read_sectors` returns. A short read is an error, not
|
||||
/// a zero-fill: silently padding a truncated source produces an image that
|
||||
/// looks complete and is not.
|
||||
pub fn write_image(
|
||||
reader: &mut dyn SectorSource,
|
||||
dest: &Path,
|
||||
total_sectors: u32,
|
||||
halt: &Halt,
|
||||
mut on_progress: impl FnMut(u64),
|
||||
) -> Result<u64> {
|
||||
if total_sectors == 0 {
|
||||
return Err(Error::EmptyImage);
|
||||
}
|
||||
|
||||
let file = File::create(dest).map_err(|source| Error::IoError { source })?;
|
||||
let mut out = BufWriter::with_capacity(BATCH_SECTORS as usize * SECTOR_BYTES, file);
|
||||
|
||||
let mut buf = vec![0u8; BATCH_SECTORS as usize * SECTOR_BYTES];
|
||||
let mut written: u64 = 0;
|
||||
let mut lba: u32 = 0;
|
||||
|
||||
while lba < total_sectors {
|
||||
if halt.is_cancelled() {
|
||||
return Err(Error::Halted);
|
||||
}
|
||||
let count = BATCH_SECTORS.min(total_sectors - lba);
|
||||
let want = count as usize * SECTOR_BYTES;
|
||||
// `recovery = false`: a file-backed source ignores the flag, and a
|
||||
// retry loop over a local file would only re-read the same bytes.
|
||||
let got = reader.read_sectors(lba, count as u16, &mut buf[..want], false)?;
|
||||
if got != want {
|
||||
return Err(Error::ShortImageRead {
|
||||
lba,
|
||||
expected: want as u32,
|
||||
got: got as u32,
|
||||
});
|
||||
}
|
||||
out.write_all(&buf[..want])
|
||||
.map_err(|source| Error::IoError { source })?;
|
||||
written += want as u64;
|
||||
lba += count;
|
||||
on_progress(written);
|
||||
}
|
||||
|
||||
// flush() only pushes the BufWriter's bytes into the kernel via write(2).
|
||||
// It makes no durability promise at all, so returning Ok here would report
|
||||
// a finished image while up to several gigabytes of it still sit in the
|
||||
// page cache. A crash, a power loss, or yanking the removable/network
|
||||
// volume the image was written to then leaves a truncated or empty file
|
||||
// that the caller was told was complete.
|
||||
//
|
||||
// For a 6-90 GB image that is exactly the failure this crate treats as
|
||||
// worst: success reported over wrong output. `into_inner` is used rather
|
||||
// than `flush` so a buffered-write error is surfaced instead of being
|
||||
// dropped on the floor by BufWriter's Drop.
|
||||
let file = out.into_inner().map_err(|e| Error::IoError {
|
||||
source: e.into_error(),
|
||||
})?;
|
||||
file.sync_all()
|
||||
.map_err(|source| Error::IoError { source })?;
|
||||
Ok(written)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::error::Result as FmResult;
|
||||
|
||||
/// A source that yields a deterministic byte per sector, so the written
|
||||
/// image can be checked positionally rather than just by length.
|
||||
struct PatternSource {
|
||||
sectors: u32,
|
||||
/// Sectors after which `read_sectors` reports a short read.
|
||||
short_after: Option<u32>,
|
||||
}
|
||||
|
||||
impl SectorSource for PatternSource {
|
||||
fn capacity_sectors(&self) -> u32 {
|
||||
self.sectors
|
||||
}
|
||||
fn read_sectors(
|
||||
&mut self,
|
||||
lba: u32,
|
||||
count: u16,
|
||||
buf: &mut [u8],
|
||||
_recovery: bool,
|
||||
) -> FmResult<usize> {
|
||||
let want = count as usize * SECTOR_BYTES;
|
||||
if self.short_after.is_some_and(|after| lba >= after) {
|
||||
return Ok(want - 1);
|
||||
}
|
||||
for s in 0..count as usize {
|
||||
let byte = ((lba as usize + s) % 251) as u8;
|
||||
buf[s * SECTOR_BYTES..(s + 1) * SECTOR_BYTES].fill(byte);
|
||||
}
|
||||
Ok(want)
|
||||
}
|
||||
}
|
||||
|
||||
fn tmp(name: &str) -> std::path::PathBuf {
|
||||
let mut p = std::env::temp_dir();
|
||||
p.push(format!("fmkv-image-writer-{name}-{}", std::process::id()));
|
||||
p
|
||||
}
|
||||
|
||||
/// The written image is byte-for-byte what the source presented, at the
|
||||
/// right offsets — not merely the right length.
|
||||
#[test]
|
||||
fn writes_every_sector_in_order() {
|
||||
let dest = tmp("order");
|
||||
let mut src = PatternSource {
|
||||
sectors: 5000,
|
||||
short_after: None,
|
||||
};
|
||||
let n = write_image(&mut src, &dest, 5000, &Halt::new(), |_| {}).expect("write");
|
||||
assert_eq!(n, 5000 * SECTOR_BYTES as u64);
|
||||
|
||||
let data = std::fs::read(&dest).expect("read back");
|
||||
assert_eq!(data.len(), 5000 * SECTOR_BYTES);
|
||||
// Spot-check across batch boundaries (BATCH_SECTORS = 2048): the last
|
||||
// sector of batch 0, the first of batch 1, and the final sector.
|
||||
for lba in [0usize, 2047, 2048, 4095, 4096, 4999] {
|
||||
let want = (lba % 251) as u8;
|
||||
assert_eq!(
|
||||
data[lba * SECTOR_BYTES],
|
||||
want,
|
||||
"sector {lba} head byte wrong — batching lost or duplicated a sector"
|
||||
);
|
||||
assert_eq!(
|
||||
data[(lba + 1) * SECTOR_BYTES - 1],
|
||||
want,
|
||||
"sector {lba} tail"
|
||||
);
|
||||
}
|
||||
let _ = std::fs::remove_file(&dest);
|
||||
}
|
||||
|
||||
/// A tail shorter than a full batch must still be written whole — the
|
||||
/// classic off-by-one when `total_sectors` is not a batch multiple.
|
||||
#[test]
|
||||
fn writes_a_partial_final_batch() {
|
||||
let dest = tmp("tail");
|
||||
let mut src = PatternSource {
|
||||
sectors: 2049,
|
||||
short_after: None,
|
||||
};
|
||||
let n = write_image(&mut src, &dest, 2049, &Halt::new(), |_| {}).expect("write");
|
||||
assert_eq!(n, 2049 * SECTOR_BYTES as u64);
|
||||
assert_eq!(
|
||||
std::fs::metadata(&dest).expect("stat").len(),
|
||||
2049 * SECTOR_BYTES as u64
|
||||
);
|
||||
let _ = std::fs::remove_file(&dest);
|
||||
}
|
||||
|
||||
/// A short read is an error. Zero-filling would yield an image that looks
|
||||
/// complete and is not — the single worst outcome for an archival copy.
|
||||
#[test]
|
||||
fn short_read_is_an_error_not_a_zero_fill() {
|
||||
let dest = tmp("short");
|
||||
let mut src = PatternSource {
|
||||
sectors: 4096,
|
||||
short_after: Some(2048),
|
||||
};
|
||||
let err = write_image(&mut src, &dest, 4096, &Halt::new(), |_| {}).expect_err("must fail");
|
||||
assert!(
|
||||
matches!(err, Error::ShortImageRead { lba: 2048, .. }),
|
||||
"got {err:?}"
|
||||
);
|
||||
let _ = std::fs::remove_file(&dest);
|
||||
}
|
||||
|
||||
/// Cancellation stops the run and reports it, rather than finishing quietly
|
||||
/// or reporting success on a partial image.
|
||||
#[test]
|
||||
fn cancellation_halts_and_reports() {
|
||||
let dest = tmp("halt");
|
||||
let mut src = PatternSource {
|
||||
sectors: 100_000,
|
||||
short_after: None,
|
||||
};
|
||||
let halt = Halt::new();
|
||||
halt.cancel();
|
||||
let err = write_image(&mut src, &dest, 100_000, &halt, |_| {}).expect_err("must halt");
|
||||
assert!(matches!(err, Error::Halted), "got {err:?}");
|
||||
let _ = std::fs::remove_file(&dest);
|
||||
}
|
||||
|
||||
/// Progress is cumulative and monotonic, and its final value equals the
|
||||
/// returned byte count — a front-end that trusts the callback must not end
|
||||
/// up disagreeing with the return value.
|
||||
#[test]
|
||||
fn progress_is_cumulative_and_ends_at_the_total() {
|
||||
let dest = tmp("progress");
|
||||
let mut src = PatternSource {
|
||||
sectors: 5000,
|
||||
short_after: None,
|
||||
};
|
||||
let mut seen: Vec<u64> = Vec::new();
|
||||
let n = write_image(&mut src, &dest, 5000, &Halt::new(), |b| seen.push(b)).expect("write");
|
||||
assert!(
|
||||
seen.windows(2).all(|w| w[1] > w[0]),
|
||||
"not monotonic: {seen:?}"
|
||||
);
|
||||
assert_eq!(*seen.last().expect("at least one callback"), n);
|
||||
let _ = std::fs::remove_file(&dest);
|
||||
}
|
||||
|
||||
/// A zero-sector source is a caller error, not a zero-byte image: an empty
|
||||
/// ISO is never what anyone wanted, and failing here names the problem.
|
||||
#[test]
|
||||
fn zero_sectors_is_an_error() {
|
||||
let dest = tmp("empty");
|
||||
let mut src = PatternSource {
|
||||
sectors: 0,
|
||||
short_after: None,
|
||||
};
|
||||
let err = write_image(&mut src, &dest, 0, &Halt::new(), |_| {}).expect_err("must fail");
|
||||
assert!(matches!(err, Error::EmptyImage), "got {err:?}");
|
||||
// The destination must not have been created — a failed run leaves no
|
||||
// stub for a later run to mistake for output.
|
||||
assert!(!dest.exists(), "empty run created a file");
|
||||
}
|
||||
}
|
||||
+3
-3
@@ -31,6 +31,7 @@ pub(crate) mod bounded;
|
||||
pub mod byte_prefetcher;
|
||||
pub mod file_sector_source;
|
||||
pub mod fsync;
|
||||
pub mod image_writer;
|
||||
pub mod sink;
|
||||
mod writeback;
|
||||
mod writeback_file;
|
||||
@@ -40,9 +41,8 @@ pub(crate) mod platform_macos;
|
||||
|
||||
pub mod pipeline;
|
||||
|
||||
pub(crate) use writeback_file::WritebackFile;
|
||||
pub use writeback_file::WritebackFile;
|
||||
|
||||
pub use pipeline::{
|
||||
DEFAULT_PIPELINE_DEPTH, Flow, Pipeline, READ_PIPELINE_DEPTH, Sink, WRITE_PIPELINE_DEPTH,
|
||||
WRITE_THROUGH_DEPTH,
|
||||
DEFAULT_PIPELINE_DEPTH, Flow, Pipeline, Sink, WRITE_PIPELINE_DEPTH, WRITE_THROUGH_DEPTH,
|
||||
};
|
||||
|
||||
+352
-65
@@ -33,7 +33,7 @@
|
||||
//! consumer lag detection). This is critical for diagnosing stalls.
|
||||
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, Ordering};
|
||||
use std::sync::atomic::{AtomicBool, AtomicU8, Ordering};
|
||||
use std::thread::{self, JoinHandle};
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
@@ -117,8 +117,32 @@ fn consumer_panicked(payload: Box<dyn std::any::Any + Send>) -> Error {
|
||||
Error::PipelineConsumerPanicked
|
||||
}
|
||||
|
||||
/// Consumer lifecycle state, shared between the caller and the consumer thread.
|
||||
///
|
||||
/// A plain `AtomicBool` could not make "the caller abandons" and "the consumer
|
||||
/// commits to finalising" mutually exclusive: the consumer loaded the flag, the
|
||||
/// caller stored it, and the consumer then finalised the container anyway — the
|
||||
/// caller reporting the rip as interrupted while a fully finalised MKV (Cues
|
||||
/// written, Segment size patched) landed on disk, indistinguishable from a
|
||||
/// complete one. The two transitions are therefore a single compare-exchange each,
|
||||
/// out of [`state::RUNNING`]: whoever wins decides, and the loser observes the
|
||||
/// winner. (`ST_RUNNING` does not exist anywhere in the crate — the constants are
|
||||
/// `state::RUNNING` / `state::ABANDONED` / `state::CLOSING` below, and both
|
||||
/// compare-exchange sites that must stay in step with this argument name them.)
|
||||
mod state {
|
||||
/// Consumer is running; neither side has committed yet.
|
||||
pub const RUNNING: u8 = 0;
|
||||
/// The caller gave up on the consumer and will report failure — the consumer
|
||||
/// must NOT finalise the output.
|
||||
pub const ABANDONED: u8 = 1;
|
||||
/// The consumer has committed to `close()` (finalising the output). The caller
|
||||
/// can no longer abandon it; it must wait for the result it is about to
|
||||
/// produce.
|
||||
pub const CLOSING: u8 = 2;
|
||||
}
|
||||
|
||||
/// After a halt or deadline fires, spin-poll `handle.is_finished()` for
|
||||
/// [`FINISH_GRACE_SECS`] before accepting the thread leak. This converts
|
||||
/// `grace` before accepting the thread leak. This converts
|
||||
/// the common "nearly-done" consumer (whose own bounded_syscall just
|
||||
/// returned and is about to drop its output file) into a clean join,
|
||||
/// releasing the file handle without waiting the full grace period.
|
||||
@@ -132,11 +156,12 @@ fn consumer_panicked(payload: Box<dyn std::any::Any + Send>) -> Error {
|
||||
/// syscall itself; that still returns on its own (or at process exit).
|
||||
fn finish_with_grace<R: Send + 'static>(
|
||||
handle: thread::JoinHandle<Result<R, Error>>,
|
||||
abandoned: &Arc<AtomicBool>,
|
||||
state: &Arc<AtomicU8>,
|
||||
grace: Duration,
|
||||
leak_err: Error,
|
||||
) -> Result<R, Error> {
|
||||
let grace = Instant::now() + Duration::from_secs(FINISH_GRACE_SECS);
|
||||
while Instant::now() < grace {
|
||||
let deadline = Instant::now() + grace;
|
||||
while Instant::now() < deadline {
|
||||
if handle.is_finished() {
|
||||
return match handle.join() {
|
||||
Ok(result) => result,
|
||||
@@ -145,16 +170,50 @@ fn finish_with_grace<R: Send + 'static>(
|
||||
}
|
||||
thread::sleep(POLL_INTERVAL);
|
||||
}
|
||||
// Grace expired. Signal abandonment, then log and leak. Setting the
|
||||
// flag BEFORE dropping the handle guarantees the leaked consumer
|
||||
// observes it the moment its wedged syscall returns: it then skips
|
||||
// any further `apply` and skips `close()`, rather than running on to
|
||||
// finalise the abandoned output file.
|
||||
// `Release` here pairs with the `Acquire` loads in the consumer loop so
|
||||
// the leaked consumer reliably observes the flag the moment its wedged
|
||||
// syscall returns, even on weak memory models (ARM64/POWER) where
|
||||
// `Relaxed` gives no cross-thread visibility guarantee.
|
||||
abandoned.store(true, Ordering::Release);
|
||||
// Grace expired. CLAIM abandonment, then log and leak. Claiming BEFORE
|
||||
// dropping the handle guarantees the leaked consumer observes it the moment
|
||||
// its wedged syscall returns: it then skips any further `apply` and skips
|
||||
// `close()`, rather than running on to finalise the abandoned output file.
|
||||
//
|
||||
// A compare-exchange, not a store, because the consumer may have committed to
|
||||
// `close()` in the instant between our last `is_finished()` poll and now. It
|
||||
// then cannot be stopped — the finalise IS happening — so abandoning it would
|
||||
// report the rip as interrupted while a valid, fully finalised container
|
||||
// lands on disk. Losing the race means waiting for the result the consumer is
|
||||
// already producing instead. `AcqRel` pairs with the consumer's own
|
||||
// compare-exchange and with the `Acquire` loads in its drain loop, so the flag
|
||||
// is reliably observed even on weak memory models (ARM64/POWER).
|
||||
if state
|
||||
.compare_exchange(
|
||||
state::RUNNING,
|
||||
state::ABANDONED,
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_err()
|
||||
{
|
||||
tracing::warn!(
|
||||
target: "freemkv::pipeline",
|
||||
phase = "finish_with_halt_close_in_flight",
|
||||
"pipeline consumer had already committed to finalising the output; \
|
||||
waiting for it rather than reporting an unfinalised output"
|
||||
);
|
||||
let close_deadline = Instant::now() + grace;
|
||||
while Instant::now() < close_deadline {
|
||||
if handle.is_finished() {
|
||||
return match handle.join() {
|
||||
Ok(result) => result,
|
||||
Err(payload) => Err(consumer_panicked(payload)),
|
||||
};
|
||||
}
|
||||
thread::sleep(POLL_INTERVAL);
|
||||
}
|
||||
// Still finalising after a second grace window: leak and report the wedge.
|
||||
// The output may end up finalised by the leaked thread — but that is now a
|
||||
// wedged-`close()` case, not the check-then-finalise race.
|
||||
drop(handle);
|
||||
return Err(leak_err);
|
||||
}
|
||||
tracing::warn!(
|
||||
target: "freemkv::pipeline",
|
||||
phase = "finish_with_halt_grace_expired",
|
||||
@@ -173,14 +232,9 @@ fn finish_with_grace<R: Send + 'static>(
|
||||
|
||||
/// Default channel depth for callers without a specific reason to
|
||||
/// pick another value. Kept conservative (4) — most callers should
|
||||
/// use READ_PIPELINE_DEPTH or WRITE_PIPELINE_DEPTH instead.
|
||||
/// use WRITE_PIPELINE_DEPTH instead.
|
||||
pub const DEFAULT_PIPELINE_DEPTH: usize = 4;
|
||||
|
||||
/// Read pipeline depth. Larger buffer compensates for drive variability
|
||||
/// and NFS sync_file_range stalls; keeps ISO reader thread fed even when
|
||||
/// consumer blocks on write.
|
||||
pub const READ_PIPELINE_DEPTH: usize = 32;
|
||||
|
||||
/// Write pipeline depth. Smaller buffer reduces backpressure risk when
|
||||
/// sync_file_range blocks; prevents producer from accumulating too much
|
||||
/// work while consumer waits for NFS to drain.
|
||||
@@ -189,7 +243,8 @@ pub const WRITE_PIPELINE_DEPTH: usize = 16;
|
||||
/// Channel depth for write-through pipelines. Each `send` fully
|
||||
/// drains before the next can enqueue. Use this when the producer
|
||||
/// must observe consumer side-effects (e.g. mapfile state) before
|
||||
/// emitting the next item. Currently used by `disc::patch`.
|
||||
/// emitting the next item. Used by `freemkv_engine::recovery::patch` — the
|
||||
/// recovery strategy moved to that crate in 1.6.0, so there is no `patch` here.
|
||||
pub const WRITE_THROUGH_DEPTH: usize = 1;
|
||||
|
||||
/// Outcome of [`Sink::apply`]: either keep feeding items
|
||||
@@ -247,7 +302,21 @@ pub struct Pipeline<I: Send + 'static, R: Send + 'static> {
|
||||
/// a syscall the consumer is currently wedged in, but it does bound
|
||||
/// the damage to "whatever write is already in flight" once that
|
||||
/// syscall returns, instead of running on to a clean finalise.
|
||||
abandoned: Arc<AtomicBool>,
|
||||
///
|
||||
/// One of [`state::RUNNING`] / [`state::ABANDONED`] / [`state::CLOSING`];
|
||||
/// both transitions are compare-exchanges so abandoning and finalising are
|
||||
/// mutually exclusive rather than racing.
|
||||
state: Arc<AtomicU8>,
|
||||
/// Set by the consumer the moment an `apply` returns `Err`. The consumer keeps
|
||||
/// draining the channel after that (so the producer never blocks on a dead
|
||||
/// receiver) — which means a producer watching only `send`'s return value
|
||||
/// cannot tell the difference between "being consumed" and "being discarded
|
||||
/// after a fatal write error", and would go on reading the whole remaining
|
||||
/// disc before `finish()` finally surfaced the error. This flag is that
|
||||
/// missing edge: [`Pipeline::send_with_halt`] fails fast on it, and
|
||||
/// [`Pipeline::consumer_failed`] exposes it to producers that use plain
|
||||
/// [`Pipeline::send`].
|
||||
failed: Arc<AtomicBool>,
|
||||
}
|
||||
|
||||
impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
@@ -256,17 +325,21 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
///
|
||||
/// The thread is named `freemkv-pipeline-consumer` so it shows up
|
||||
/// distinctly in stack traces and `top -H`. Callers that want a
|
||||
/// more specific name (e.g. `freemkv-sweep-consumer`) should use
|
||||
/// [`Pipeline::spawn_named`] instead. Returns an `Error::IoError`
|
||||
/// if the OS refuses the thread spawn (resource exhaustion);
|
||||
/// callers already operate in fallible context, so this is
|
||||
/// propagated rather than panicked.
|
||||
/// more specific name should use [`Pipeline::spawn_named`] instead.
|
||||
/// Returns an `Error::IoError` if the OS refuses the thread spawn
|
||||
/// (resource exhaustion); callers already operate in fallible context, so
|
||||
/// this is propagated rather than panicked.
|
||||
///
|
||||
/// Sweep uses [`Pipeline::spawn_named`] directly so the consumer
|
||||
/// thread shows up as `freemkv-sweep-consumer`; mux uses
|
||||
/// `freemkv-mux-consumer`. `Pipeline::spawn` (this function, with
|
||||
/// the default name) is used by `disc::patch` and by the unit
|
||||
/// tests in this module.
|
||||
/// Inside this crate the only [`Pipeline::spawn_named`] caller is the mux
|
||||
/// driver, which names its thread `freemkv-mux-consumer`. `Pipeline::spawn`
|
||||
/// (this function, with the default name) is used only by the unit tests in
|
||||
/// this module.
|
||||
///
|
||||
/// This paragraph twice named a caller that had left the crate: first
|
||||
/// `disc::patch`, then Sweep and its `freemkv-sweep-consumer` thread. Both
|
||||
/// went to freemkv-engine with the recovery passes in 1.6.0, and each in
|
||||
/// turn sent readers hunting a component that is not here. Name callers
|
||||
/// that live in THIS crate, or none.
|
||||
pub fn spawn<S: Sink<I, Output = R>>(depth: usize, sink: S) -> Result<Self, Error> {
|
||||
Self::spawn_named("freemkv-pipeline-consumer", depth, sink)
|
||||
}
|
||||
@@ -274,15 +347,17 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
/// Like [`Pipeline::spawn`] but lets the caller supply the
|
||||
/// consumer thread's name. Useful when several pipelines run in
|
||||
/// the same process and stack traces / `top -H` need to tell them
|
||||
/// apart (e.g. `freemkv-sweep-consumer`, `freemkv-mux-consumer`).
|
||||
/// apart (e.g. `freemkv-mux-consumer`).
|
||||
pub fn spawn_named<S: Sink<I, Output = R>>(
|
||||
name: &str,
|
||||
depth: usize,
|
||||
sink: S,
|
||||
) -> Result<Self, Error> {
|
||||
let (tx, rx) = bounded::<I>(depth);
|
||||
let abandoned = Arc::new(AtomicBool::new(false));
|
||||
let abandoned_consumer = abandoned.clone();
|
||||
let state = Arc::new(AtomicU8::new(state::RUNNING));
|
||||
let state_consumer = state.clone();
|
||||
let failed = Arc::new(AtomicBool::new(false));
|
||||
let failed_consumer = failed.clone();
|
||||
let handle = thread::Builder::new()
|
||||
.name(name.into())
|
||||
.spawn(move || -> Result<R, Error> {
|
||||
@@ -313,7 +388,7 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
// dead receiver, but we touch the output no further. The
|
||||
// final post-loop abandonment check returns the error
|
||||
// and skips `close()`.
|
||||
if abandoned_consumer.load(Ordering::Acquire) {
|
||||
if state_consumer.load(Ordering::Acquire) == state::ABANDONED {
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -342,6 +417,12 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
tracing::debug!("Pipeline: apply error, stopping, err={:?}", e);
|
||||
}
|
||||
first_err = Some(e);
|
||||
// Publish the failure so the producer can stop
|
||||
// FEEDING a dead write side instead of only learning
|
||||
// about it at `finish()` — by which time it has read
|
||||
// the rest of the disc. `Release` pairs with the
|
||||
// `Acquire` load in `send_with_halt`.
|
||||
failed_consumer.store(true, Ordering::Release);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -402,13 +483,38 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
// MKV Cues + patching the segment header) on a file the
|
||||
// caller already reported as failed is exactly the
|
||||
// write race we must not run.
|
||||
if abandoned_consumer.load(Ordering::Acquire) {
|
||||
return Err(Error::Halted);
|
||||
}
|
||||
|
||||
match first_err {
|
||||
Some(e) => Err(e),
|
||||
None => sink.close(),
|
||||
// No `close()` on this path, so there is nothing to claim —
|
||||
// just report, unless the caller has already given up on us.
|
||||
Some(e) => {
|
||||
if state_consumer.load(Ordering::Acquire) == state::ABANDONED {
|
||||
Err(Error::Halted)
|
||||
} else {
|
||||
Err(e)
|
||||
}
|
||||
}
|
||||
// CLAIM the finalise. A plain load here left a window in which
|
||||
// the caller stored `abandoned` AFTER we read it as clear, so
|
||||
// `close()` ran anyway and finalised (Cues + Segment-size
|
||||
// patch) an output the caller had already reported as
|
||||
// interrupted — a truncated rip indistinguishable from a
|
||||
// complete one. The compare-exchange closes that window: if the
|
||||
// caller got there first we skip `close()`, and if we get there
|
||||
// first the caller waits for us instead of abandoning.
|
||||
None => {
|
||||
if state_consumer
|
||||
.compare_exchange(
|
||||
state::RUNNING,
|
||||
state::CLOSING,
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire,
|
||||
)
|
||||
.is_err()
|
||||
{
|
||||
return Err(Error::Halted);
|
||||
}
|
||||
sink.close()
|
||||
}
|
||||
}
|
||||
})
|
||||
.map_err(|e| Error::IoError { source: e })?;
|
||||
@@ -416,10 +522,24 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
Ok(Pipeline {
|
||||
tx,
|
||||
handle,
|
||||
abandoned,
|
||||
state,
|
||||
failed,
|
||||
})
|
||||
}
|
||||
|
||||
/// Whether the consumer's `apply` has already failed fatally.
|
||||
///
|
||||
/// The consumer keeps draining the channel after an `apply` error (so the
|
||||
/// producer never blocks on a dead receiver), which means `send` keeps
|
||||
/// succeeding and a producer has no other way to tell that everything it feeds
|
||||
/// is being discarded. A long-running producer — the mux frame pump reading a
|
||||
/// 60 GB title off an optical drive — should check this and unwind instead of
|
||||
/// reading the rest of the disc for a write that has already failed.
|
||||
/// [`Pipeline::send_with_halt`] checks it automatically.
|
||||
pub fn consumer_failed(&self) -> bool {
|
||||
self.failed.load(Ordering::Acquire)
|
||||
}
|
||||
|
||||
/// Push one item. Blocks if the channel is full — that's the
|
||||
/// back-pressure the whole primitive exists to provide. Returns
|
||||
/// the item back if the consumer thread is gone (panicked or
|
||||
@@ -512,11 +632,34 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
/// wedged inside an unkillable syscall, the producer can still
|
||||
/// observe `/api/stop` and unwind within
|
||||
/// [`SEND_HALT_CHECK_INTERVAL`].
|
||||
/// NOT a `foo_with_X` variant of [`Pipeline::send`], despite the name.
|
||||
/// The two encode OPPOSITE policies on the same event, each with its own
|
||||
/// test: after the consumer's `apply` has failed, `send` still succeeds
|
||||
/// (the consumer keeps draining, so the channel accepts the item), while
|
||||
/// this one hands the item straight back — so a producer does not read an
|
||||
/// hour of disc for a write that died on the first frame. Collapsing them
|
||||
/// into one Option-parameterised method deletes one of those behaviours;
|
||||
/// it was tried and `apply_error_drains_then_propagates` caught it.
|
||||
pub fn send_with_halt(&self, item: I, halt: &Halt, deadline: Duration) -> Result<(), I> {
|
||||
use crossbeam_channel::SendTimeoutError;
|
||||
let end = Instant::now() + deadline;
|
||||
let mut pending = item;
|
||||
loop {
|
||||
// The consumer's `apply` has failed fatally: everything sent from here
|
||||
// is drained and discarded, so hand the item back at once. Without this
|
||||
// the producer saw every send succeed (the channel is always being
|
||||
// drained) and went on reading the whole remaining title — an hour of
|
||||
// drive time on a UHD — for a write that died on the first frame, only
|
||||
// learning about it at `finish()`.
|
||||
if self.consumer_failed() {
|
||||
if debug_enabled() {
|
||||
tracing::debug!(
|
||||
"Pipeline send_with_halt: consumer apply failed, returning item={}",
|
||||
std::any::type_name::<I>()
|
||||
);
|
||||
}
|
||||
return Err(pending);
|
||||
}
|
||||
// Pre-check the cheap exit conditions before parking.
|
||||
if halt.is_cancelled() {
|
||||
if debug_enabled() {
|
||||
@@ -571,7 +714,8 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
let Pipeline {
|
||||
tx,
|
||||
handle,
|
||||
abandoned: _,
|
||||
state: _,
|
||||
failed: _,
|
||||
} = self;
|
||||
// Explicit drop, although the destructure already drops `tx`
|
||||
// at end-of-scope. Being explicit keeps the intent obvious.
|
||||
@@ -607,11 +751,18 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
/// Plain [`Pipeline::finish`] is preserved for callers without a
|
||||
/// halt-token plumbed through; that path still blocks indefinitely
|
||||
/// on `join()`, matching pre-0.20.8 behaviour.
|
||||
/// Also not a `foo_with_X` variant: [`Pipeline::finish`] joins and waits
|
||||
/// however long the consumer needs, while this one gives up after
|
||||
/// `JOIN_TIMEOUT_SECS` and reports halted. Which is right depends on
|
||||
/// whether the caller has a user waiting to cancel — the mux driver does
|
||||
/// and uses this; the unit tests do not and use the plain join. Merging
|
||||
/// them means picking one of those policies for both.
|
||||
pub fn finish_with_halt(self, halt: Option<&Halt>) -> Result<R, Error> {
|
||||
let Pipeline {
|
||||
tx,
|
||||
handle,
|
||||
abandoned,
|
||||
state,
|
||||
failed: _,
|
||||
} = self;
|
||||
drop(tx);
|
||||
let deadline = Instant::now() + Duration::from_secs(JOIN_TIMEOUT_SECS);
|
||||
@@ -622,13 +773,23 @@ impl<I: Send + 'static, R: Send + 'static> Pipeline<I, R> {
|
||||
Err(payload) => Err(consumer_panicked(payload)),
|
||||
};
|
||||
}
|
||||
if let Some(h) = halt {
|
||||
if h.is_cancelled() {
|
||||
return finish_with_grace(handle, &abandoned, Error::Halted);
|
||||
}
|
||||
if let Some(h) = halt
|
||||
&& h.is_cancelled()
|
||||
{
|
||||
return finish_with_grace(
|
||||
handle,
|
||||
&state,
|
||||
Duration::from_secs(FINISH_GRACE_SECS),
|
||||
Error::Halted,
|
||||
);
|
||||
}
|
||||
if Instant::now() >= deadline {
|
||||
return finish_with_grace(handle, &abandoned, Error::PipelineJoinTimeout);
|
||||
return finish_with_grace(
|
||||
handle,
|
||||
&state,
|
||||
Duration::from_secs(FINISH_GRACE_SECS),
|
||||
Error::PipelineJoinTimeout,
|
||||
);
|
||||
}
|
||||
thread::sleep(POLL_INTERVAL);
|
||||
}
|
||||
@@ -1080,20 +1241,6 @@ mod tests {
|
||||
/// A sink that records the exact order of items it receives, so we
|
||||
/// can prove the channel is FIFO (no reordering). `close` returns
|
||||
/// the recorded vector.
|
||||
struct OrderSink {
|
||||
seen: Vec<u64>,
|
||||
}
|
||||
impl Sink<u64> for OrderSink {
|
||||
type Output = Vec<u64>;
|
||||
fn apply(&mut self, item: u64) -> Result<Flow, Error> {
|
||||
self.seen.push(item);
|
||||
Ok(Flow::Continue)
|
||||
}
|
||||
fn close(self) -> Result<Vec<u64>, Error> {
|
||||
Ok(self.seen)
|
||||
}
|
||||
}
|
||||
|
||||
/// Zero items sent: closing the pipeline immediately must still
|
||||
/// call `close()` exactly once and return its Output. The consumer
|
||||
/// loop's `while let Ok = rx.recv()` exits on the dropped tx with
|
||||
@@ -1598,4 +1745,144 @@ mod tests {
|
||||
let res = pipe.finish_with_halt(None);
|
||||
assert!(matches!(res, Ok(190)), "expected Ok(190), got {res:?}");
|
||||
}
|
||||
|
||||
/// A fatal `apply` error must become visible to the PRODUCER, not only to
|
||||
/// `finish()`. The consumer keeps draining after the error (so the producer
|
||||
/// never blocks on a dead receiver), which meant every `send_with_halt`
|
||||
/// returned `Ok` for the rest of the run: on a 60 GB mkv:// mux that hit
|
||||
/// ENOSPC on the first frame, the mux driver read the entire remaining title —
|
||||
/// an hour of optical-drive time — before learning the write had died.
|
||||
#[test]
|
||||
fn send_with_halt_fails_fast_once_apply_has_failed() {
|
||||
struct FailFirst {
|
||||
failed: Arc<AtomicUsize>,
|
||||
}
|
||||
impl Sink<u64> for FailFirst {
|
||||
type Output = ();
|
||||
fn apply(&mut self, _item: u64) -> Result<Flow, Error> {
|
||||
self.failed.fetch_add(1, Ordering::SeqCst);
|
||||
Err(Error::DecryptFailed)
|
||||
}
|
||||
fn close(self) -> Result<(), Error> {
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
let applied = Arc::new(AtomicUsize::new(0));
|
||||
let pipe = Pipeline::spawn(
|
||||
DEFAULT_PIPELINE_DEPTH,
|
||||
FailFirst {
|
||||
failed: applied.clone(),
|
||||
},
|
||||
)
|
||||
.expect("spawn");
|
||||
let halt = crate::halt::Halt::new();
|
||||
let deadline = Duration::from_secs(5);
|
||||
|
||||
// Feed one item and wait until the consumer has actually applied (and
|
||||
// failed on) it, so the check below is deterministic rather than racy.
|
||||
pipe.send_with_halt(0u64, &halt, deadline)
|
||||
.expect("the first send lands");
|
||||
let until = Instant::now() + Duration::from_secs(2);
|
||||
while Instant::now() < until && applied.load(Ordering::SeqCst) == 0 {
|
||||
std::thread::sleep(Duration::from_millis(5));
|
||||
}
|
||||
assert_eq!(applied.load(Ordering::SeqCst), 1, "apply ran and failed");
|
||||
|
||||
assert!(pipe.consumer_failed(), "the failure must be observable");
|
||||
// The very next send must hand the item straight back — the producer's
|
||||
// signal to stop reading the disc.
|
||||
assert_eq!(
|
||||
pipe.send_with_halt(1u64, &halt, deadline),
|
||||
Err(1u64),
|
||||
"send_with_halt must fail fast once the consumer's apply has failed"
|
||||
);
|
||||
// The halt was never fired, so this is not a cancellation: the real error
|
||||
// still comes out of finish().
|
||||
assert!(matches!(pipe.finish(), Err(Error::DecryptFailed)));
|
||||
assert_eq!(
|
||||
applied.load(Ordering::SeqCst),
|
||||
1,
|
||||
"no further item was applied"
|
||||
);
|
||||
}
|
||||
|
||||
/// The abandon/finalise race. A consumer that has ALREADY committed to
|
||||
/// `close()` when the grace period expires cannot be stopped — the finalise is
|
||||
/// happening — so the caller must wait for its result instead of reporting the
|
||||
/// output as un-finalised. With a plain flag the consumer read it as clear, the
|
||||
/// caller then stored it, and the caller returned `Err(Halted)`
|
||||
/// (`completed = false`) while a fully finalised MKV (Cues written, Segment
|
||||
/// size patched) landed on disk — a truncated rip indistinguishable from a
|
||||
/// complete one.
|
||||
#[test]
|
||||
fn abandon_loses_to_a_close_already_committed() {
|
||||
let state = Arc::new(AtomicU8::new(state::RUNNING));
|
||||
let release = Arc::new(AtomicBool::new(false));
|
||||
let in_close = Arc::new(AtomicBool::new(false));
|
||||
|
||||
let (st, rel, inc) = (state.clone(), release.clone(), in_close.clone());
|
||||
let handle = thread::Builder::new()
|
||||
.name("test-consumer".into())
|
||||
.spawn(move || -> Result<u64, Error> {
|
||||
// Exactly what the consumer does before finalising: claim the
|
||||
// right to close.
|
||||
assert!(
|
||||
st.compare_exchange(
|
||||
state::RUNNING,
|
||||
state::CLOSING,
|
||||
Ordering::AcqRel,
|
||||
Ordering::Acquire
|
||||
)
|
||||
.is_ok(),
|
||||
"the consumer claims the finalise first"
|
||||
);
|
||||
inc.store(true, Ordering::SeqCst);
|
||||
// Inside `close()`, finalising the container.
|
||||
while !rel.load(Ordering::SeqCst) {
|
||||
thread::sleep(Duration::from_millis(5));
|
||||
}
|
||||
Ok(42)
|
||||
})
|
||||
.expect("spawn");
|
||||
|
||||
let until = Instant::now() + Duration::from_secs(2);
|
||||
while Instant::now() < until && !in_close.load(Ordering::SeqCst) {
|
||||
thread::sleep(Duration::from_millis(5));
|
||||
}
|
||||
assert!(in_close.load(Ordering::SeqCst), "consumer reached close()");
|
||||
|
||||
// Finish the close only AFTER the first grace window has expired, so the
|
||||
// caller genuinely reaches the abandon decision with a close in flight.
|
||||
let rel = release.clone();
|
||||
thread::spawn(move || {
|
||||
// Past the first grace window (and past the 250 ms poll cadence that
|
||||
// bounds when the window is actually observed), inside the second.
|
||||
//
|
||||
// These intervals used to be 600 ms against a 300 ms grace, which
|
||||
// left NO margin: two 300 ms windows end at 600 ms, and the 250 ms
|
||||
// poll cadence can push the observation later still, so on a loaded
|
||||
// runner the second window expired first and the caller abandoned —
|
||||
// failing with Err(Halted) against a race, not a defect.
|
||||
//
|
||||
// Scaled up so the jitter is small relative to the intervals: the
|
||||
// first window ends at ~1.0-1.25 s and the second at ~2.0-2.25 s,
|
||||
// so releasing at 1.6 s sits well inside the second with roughly
|
||||
// 350 ms of slack on either side. The ordering under test is
|
||||
// unchanged; only the margin is.
|
||||
thread::sleep(Duration::from_millis(1600));
|
||||
rel.store(true, Ordering::SeqCst);
|
||||
});
|
||||
|
||||
let grace = Duration::from_secs(1);
|
||||
let res = finish_with_grace(handle, &state, grace, Error::Halted);
|
||||
assert!(
|
||||
matches!(res, Ok(42)),
|
||||
"a finalise already in flight must be waited for, not abandoned: {res:?}"
|
||||
);
|
||||
assert_eq!(
|
||||
state.load(Ordering::SeqCst),
|
||||
state::CLOSING,
|
||||
"the caller must not have overwritten the consumer's claim"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -272,7 +272,7 @@ impl WritebackPipeline {
|
||||
self.chunk_bytes,
|
||||
self.skip_wait(),
|
||||
);
|
||||
if self.chunk_count % SIZE_LOG_INTERVAL == 0 {
|
||||
if self.chunk_count.is_multiple_of(SIZE_LOG_INTERVAL) {
|
||||
tracing::debug!(
|
||||
target: "mux",
|
||||
"WritebackPipeline chunk_bytes={} after {} chunks is_nfs={} degraded={}",
|
||||
|
||||
@@ -30,14 +30,15 @@ pub(super) fn preallocate(file: &File, size_bytes: u64) {
|
||||
);
|
||||
}
|
||||
|
||||
/// Run `fsync` on `file` with a 60 s deadline. On timeout — and
|
||||
/// likewise on halt or a lost worker — we log and return `Ok(())`: the
|
||||
/// kernel will still flush on close, so the data is best-effort durable.
|
||||
/// The alternative (trap the thread for the rest of the rip, or return
|
||||
/// an error that aborts an otherwise-complete mux) is worse, so all
|
||||
/// three fallbacks return `Ok(())`. `Ok(())` from these paths is NOT a
|
||||
/// durability barrier — the durable flush did not complete; only the
|
||||
/// hang is bounded.
|
||||
/// Run `fsync` on `file` with a 60 s deadline. On timeout, halt or a lost
|
||||
/// worker we log and return `Err` — matching macOS. POSIX gives `fsync`
|
||||
/// exactly one way to say "the data is on stable storage" and that is a zero
|
||||
/// return; a call that never reached the device has not earned it, so `Ok(())`
|
||||
/// from here means the flush completed and nothing else.
|
||||
///
|
||||
/// The kernel will still flush on close, so the data is usually durable
|
||||
/// anyway — but that is a probability, not a barrier, and a caller that needs
|
||||
/// crash-consistency has to be able to tell the difference.
|
||||
///
|
||||
/// ## fd-reuse safety
|
||||
///
|
||||
@@ -78,27 +79,7 @@ pub(super) fn durable_sync(file: &File) -> io::Result<()> {
|
||||
},
|
||||
) {
|
||||
Ok(inner) => inner,
|
||||
Err(crate::io::bounded::BoundedError::Timeout) => {
|
||||
tracing::error!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all fsync timed out after 60s; kernel will flush on close (best-effort)"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
Err(crate::io::bounded::BoundedError::Halted) => {
|
||||
tracing::warn!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all fsync skipped (halt requested); data not durably flushed, kernel will flush on close"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
Err(crate::io::bounded::BoundedError::WorkerLost) => {
|
||||
tracing::error!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all fsync worker lost before completion; data not durably flushed, kernel will flush on close"
|
||||
);
|
||||
Ok(())
|
||||
}
|
||||
Err(e) => bounded_failure_to_result(e),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -143,3 +124,77 @@ mod tests {
|
||||
durable_sync(f.as_file()).expect("durable_sync must return Ok on a local tempfile");
|
||||
}
|
||||
}
|
||||
|
||||
/// Map a [`crate::io::bounded::BoundedError`] from the bounded `fsync` onto the
|
||||
/// `io::Error` `durable_sync` returns.
|
||||
///
|
||||
/// Every arm means the same thing: **no sync observably ran**. All three used to
|
||||
/// return `Ok(())`, so `WritebackFile::sync_all` reported success for a
|
||||
/// durability barrier that never happened. POSIX gives `fsync` one way to say
|
||||
/// "the data is on stable storage" — a zero return — and a call that never
|
||||
/// reached the device has not earned it.
|
||||
///
|
||||
/// This mirrors the macOS `F_FULLFSYNC` mapping exactly. The two were found
|
||||
/// carrying the identical defect, and a platform disagreeing with its sibling
|
||||
/// about whether a failed sync is an error is the "works on my platform" class
|
||||
/// this crate has been bitten by before — most recently an over-length SCSI CDB
|
||||
/// that macOS rejected and the other two silently truncated.
|
||||
///
|
||||
/// No message text (this crate ships no user-facing English): the kind, and
|
||||
/// `EIO` for the worker-lost case, are the signal; `tracing` carries the detail.
|
||||
fn bounded_failure_to_result(e: crate::io::bounded::BoundedError) -> io::Result<()> {
|
||||
match e {
|
||||
crate::io::bounded::BoundedError::Timeout => {
|
||||
tracing::error!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all fsync timed out after 60s; data NOT durably flushed, kernel will flush on close"
|
||||
);
|
||||
Err(crate::error::Error::SyncTimeout.into())
|
||||
}
|
||||
crate::io::bounded::BoundedError::Halted => {
|
||||
tracing::warn!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all fsync skipped (halt requested); data NOT durably flushed, kernel will flush on close"
|
||||
);
|
||||
Err(crate::error::Error::Halted.into())
|
||||
}
|
||||
crate::io::bounded::BoundedError::WorkerLost => {
|
||||
tracing::error!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all fsync worker lost before completion; data NOT durably flushed, kernel will flush on close"
|
||||
);
|
||||
// EIO, matching the macOS sibling: a consumer distinguishing these
|
||||
// three failures does so on the same value on every platform.
|
||||
// ErrorKind::Other carries nothing a caller can branch on.
|
||||
Err(crate::error::Error::SyncWorkerLost.into())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod bounded_failure_tests {
|
||||
use super::*;
|
||||
use crate::io::bounded::BoundedError;
|
||||
|
||||
/// Every bounded-fsync failure must be an error. Asserted per variant rather
|
||||
/// than as a loop so a new variant defaulting to Ok cannot slip through.
|
||||
#[test]
|
||||
fn no_bounded_fsync_failure_maps_to_ok() {
|
||||
assert_eq!(
|
||||
bounded_failure_to_result(BoundedError::Timeout)
|
||||
.expect_err("a timed-out fsync must be an error")
|
||||
.kind(),
|
||||
io::ErrorKind::TimedOut
|
||||
);
|
||||
assert_eq!(
|
||||
bounded_failure_to_result(BoundedError::Halted)
|
||||
.expect_err("a halted fsync must be an error")
|
||||
.kind(),
|
||||
io::ErrorKind::Interrupted
|
||||
);
|
||||
assert!(
|
||||
bounded_failure_to_result(BoundedError::WorkerLost).is_err(),
|
||||
"a lost fsync worker must be an error"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -105,15 +105,46 @@ pub(super) fn durable_sync(file: &File) -> io::Result<()> {
|
||||
},
|
||||
) {
|
||||
Ok(inner) => inner,
|
||||
Err(crate::io::bounded::BoundedError::Timeout) => {
|
||||
Err(e) => bounded_failure_to_result(e),
|
||||
}
|
||||
}
|
||||
|
||||
/// Map a [`crate::io::bounded::BoundedError`] from the bounded `F_FULLFSYNC`
|
||||
/// onto the `io::Error` `durable_sync` returns.
|
||||
///
|
||||
/// Every arm here means the same thing: **no sync observably ran**. All three
|
||||
/// previously returned `Ok(())`, so `WritebackFile::sync_all` reported success
|
||||
/// for a durability barrier that never happened — a total failure exiting 0,
|
||||
/// with only a log line to distinguish it. POSIX gives `fsync` exactly one way
|
||||
/// to say "the data is on stable storage" and that is a zero return; a call
|
||||
/// that never reached the device has not earned it.
|
||||
///
|
||||
/// The errors carry no message text (this crate ships no user-facing English):
|
||||
/// the kind, and `EIO` for the worker-lost case, are the whole signal, and the
|
||||
/// `tracing` lines above/below carry the operator detail.
|
||||
fn bounded_failure_to_result(e: crate::io::bounded::BoundedError) -> io::Result<()> {
|
||||
match e {
|
||||
crate::io::bounded::BoundedError::Timeout => {
|
||||
tracing::error!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all F_FULLFSYNC timed out after 60s; kernel will flush on close (best-effort)"
|
||||
"WritebackFile::sync_all F_FULLFSYNC timed out after 60s; data NOT durably flushed, kernel will flush on close"
|
||||
);
|
||||
Ok(())
|
||||
Err(crate::error::Error::SyncTimeout.into())
|
||||
}
|
||||
crate::io::bounded::BoundedError::Halted => {
|
||||
tracing::warn!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all F_FULLFSYNC skipped (halt requested); data NOT durably flushed, kernel will flush on close"
|
||||
);
|
||||
Err(crate::error::Error::Halted.into())
|
||||
}
|
||||
crate::io::bounded::BoundedError::WorkerLost => {
|
||||
tracing::error!(
|
||||
target: "mux",
|
||||
"WritebackFile::sync_all F_FULLFSYNC worker lost before completion; data NOT durably flushed, kernel will flush on close"
|
||||
);
|
||||
Err(crate::error::Error::SyncWorkerLost.into())
|
||||
}
|
||||
Err(crate::io::bounded::BoundedError::Halted) => Ok(()),
|
||||
Err(crate::io::bounded::BoundedError::WorkerLost) => Ok(()),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -156,4 +187,75 @@ mod tests {
|
||||
// durable_sync must complete without error on the local tempfile.
|
||||
durable_sync(f.as_file()).expect("durable_sync must return Ok on a local tempfile");
|
||||
}
|
||||
|
||||
/// Every `BoundedError` arm of the bounded `F_FULLFSYNC` means no sync
|
||||
/// observably ran. All three returned `Ok(())`, so `sync_all` reported a
|
||||
/// durability barrier that never happened — the caller could not tell a
|
||||
/// completed flush from a skipped one by any means except reading a log.
|
||||
///
|
||||
/// Asserted on the concrete `ErrorKind` / `errno` each arm must produce,
|
||||
/// so a future arm that quietly reverts to `Ok(())` fails here.
|
||||
#[test]
|
||||
fn every_bounded_failure_is_reported_as_an_error() {
|
||||
use crate::io::bounded::BoundedError;
|
||||
|
||||
let timeout = bounded_failure_to_result(BoundedError::Timeout)
|
||||
.expect_err("a timed-out F_FULLFSYNC must be an error");
|
||||
assert_eq!(
|
||||
timeout.kind(),
|
||||
io::ErrorKind::TimedOut,
|
||||
"a timed-out F_FULLFSYNC must not be reported as a completed sync"
|
||||
);
|
||||
|
||||
let halted = bounded_failure_to_result(BoundedError::Halted)
|
||||
.expect_err("a halted F_FULLFSYNC must be an error");
|
||||
assert_eq!(
|
||||
halted.kind(),
|
||||
io::ErrorKind::Interrupted,
|
||||
"a halted F_FULLFSYNC must not be reported as a completed sync"
|
||||
);
|
||||
|
||||
// The three arms must be DISTINGUISHABLE, not merely non-Ok. Each
|
||||
// carries its own numeric code through the "E<code>" prefix that
|
||||
// `From<Error> for io::Error` mints — the only shape `error_code`
|
||||
// recognises. A bare `ErrorKind` cannot be classified, which is how a
|
||||
// user cancel here used to read as a hard I/O failure.
|
||||
let lost = bounded_failure_to_result(BoundedError::WorkerLost)
|
||||
.expect_err("a lost F_FULLFSYNC worker must be an error");
|
||||
assert!(
|
||||
lost.to_string()
|
||||
.starts_with(&format!("E{}", crate::error::E_SYNC_WORKER_LOST)),
|
||||
"a lost worker must be identifiable, got {lost}"
|
||||
);
|
||||
assert!(
|
||||
timeout
|
||||
.to_string()
|
||||
.starts_with(&format!("E{}", crate::error::E_SYNC_TIMEOUT)),
|
||||
"a timeout must be distinguishable from a lost worker, got {timeout}"
|
||||
);
|
||||
assert!(
|
||||
crate::error::is_halt(&halted),
|
||||
"a halt must satisfy the crate's own is_halt(), or the CLI reports a \
|
||||
user cancel as a failure; got {halted}"
|
||||
);
|
||||
}
|
||||
|
||||
/// The failure path must be reachable through the public surface: a
|
||||
/// `WritebackFile::sync_all` that hits any of these arms must surface an
|
||||
/// `Err`, not a silent `Ok`. Pinned at the mapping boundary because the
|
||||
/// timeout itself is not deterministically inducible in a unit test.
|
||||
#[test]
|
||||
fn bounded_failures_are_never_mapped_to_ok() {
|
||||
use crate::io::bounded::BoundedError;
|
||||
for e in [
|
||||
BoundedError::Timeout,
|
||||
BoundedError::Halted,
|
||||
BoundedError::WorkerLost,
|
||||
] {
|
||||
assert!(
|
||||
bounded_failure_to_result(e).is_err(),
|
||||
"a bounded F_FULLFSYNC failure must never map to Ok"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -97,7 +97,7 @@ fn writeback_chunk_bytes() -> u64 {
|
||||
.unwrap_or(WRITEBACK_CHUNK_BYTES_DEFAULT)
|
||||
}
|
||||
|
||||
pub(crate) struct WritebackFile {
|
||||
pub struct WritebackFile {
|
||||
file: File,
|
||||
pipeline: WritebackPipeline,
|
||||
pos: u64,
|
||||
@@ -115,7 +115,7 @@ impl WritebackFile {
|
||||
/// once so the pipeline starts tracking from wherever the file
|
||||
/// already is (typically 0 for fresh files; non-zero for resumed
|
||||
/// or appended files).
|
||||
pub(crate) fn new(mut file: File) -> io::Result<Self> {
|
||||
pub fn new(mut file: File) -> io::Result<Self> {
|
||||
let pos = file.stream_position()?;
|
||||
let pipeline = WritebackPipeline::new(&file, pos, writeback_chunk_bytes());
|
||||
Ok(Self {
|
||||
@@ -136,7 +136,7 @@ impl WritebackFile {
|
||||
/// [`Self::create_with_size_hint`] so the kernel can pre-reserve
|
||||
/// extents.
|
||||
#[allow(dead_code)]
|
||||
pub(crate) fn create(path: &Path) -> io::Result<Self> {
|
||||
pub fn create(path: &Path) -> io::Result<Self> {
|
||||
let file = File::create(path)?;
|
||||
Self::new(file)
|
||||
}
|
||||
@@ -153,7 +153,7 @@ impl WritebackFile {
|
||||
/// On platforms without an extent-preallocation primitive this is
|
||||
/// equivalent to `create` — the size hint is dropped after a debug
|
||||
/// log.
|
||||
pub(crate) fn create_with_size_hint(path: &Path, size_bytes: u64) -> io::Result<Self> {
|
||||
pub fn create_with_size_hint(path: &Path, size_bytes: u64) -> io::Result<Self> {
|
||||
let file = File::create(path)?;
|
||||
platform::preallocate(&file, size_bytes);
|
||||
Self::new(file)
|
||||
@@ -163,7 +163,7 @@ impl WritebackFile {
|
||||
/// wrap it. Mirrors `File::open` semantics for the writable case
|
||||
/// — used by patch / resume paths that mutate an existing ISO in
|
||||
/// place.
|
||||
pub(crate) fn open(path: &Path) -> io::Result<Self> {
|
||||
pub fn open(path: &Path) -> io::Result<Self> {
|
||||
let file = OpenOptions::new().write(true).open(path)?;
|
||||
Self::new(file)
|
||||
}
|
||||
@@ -178,13 +178,20 @@ impl WritebackFile {
|
||||
/// is left to the kernel's normal flush-on-close path — best
|
||||
/// effort, but bounded.
|
||||
///
|
||||
/// IMPORTANT: on Linux/macOS a successful `Ok(())` does NOT
|
||||
/// guarantee the data is durable if the bounded fsync timed out or
|
||||
/// was halted — only the hang is bounded, the fsync may not have
|
||||
/// completed. Callers needing crash-consistency (e.g. mux-finish
|
||||
/// then external commit/DB update) must not treat `Ok(())` as a
|
||||
/// durability barrier.
|
||||
pub(crate) fn sync_all(&mut self) -> io::Result<()> {
|
||||
/// A bounded-fsync failure is returned as an `Err` on BOTH platforms, so
|
||||
/// `Ok(())` means the flush completed and a caller needing
|
||||
/// crash-consistency can treat it as a durability barrier.
|
||||
///
|
||||
/// The three causes are DISTINGUISHABLE by numeric code, because a caller
|
||||
/// should not retry a lost worker the way it retries a timeout, and must
|
||||
/// not report a user cancel as a failure:
|
||||
///
|
||||
/// * [`E_SYNC_TIMEOUT`](crate::error::E_SYNC_TIMEOUT) — deadline expired
|
||||
/// * [`E_HALTED`](crate::error::E_HALTED) — cancelled;
|
||||
/// [`is_halt`](crate::error::is_halt) recognises it
|
||||
/// * [`E_SYNC_WORKER_LOST`](crate::error::E_SYNC_WORKER_LOST) — the worker
|
||||
/// thread died before reporting
|
||||
pub fn sync_all(&mut self) -> io::Result<()> {
|
||||
if self.seek_count > 0 {
|
||||
tracing::debug!(
|
||||
target: "mux",
|
||||
@@ -256,9 +263,8 @@ impl super::sink::SequentialSink for WritebackFile {
|
||||
/// the same work [`Self::sync_all`] does. Implemented explicitly (no
|
||||
/// blanket impl) so a `dyn SequentialSink` / `dyn RandomAccessSink`
|
||||
/// `finish()` actually finalises + fsyncs instead of hitting a no-op
|
||||
/// default. Note the bounded-fsync caveat from [`Self::sync_all`]
|
||||
/// applies: `Ok(())` is not a durability barrier if the fsync timed
|
||||
/// out or was halted.
|
||||
/// default. A bounded-fsync failure surfaces as an `Err` here, on every
|
||||
/// platform, exactly as it does from [`Self::sync_all`].
|
||||
fn finish(&mut self) -> io::Result<()> {
|
||||
self.sync_all()
|
||||
}
|
||||
|
||||
+584
-86
@@ -53,9 +53,30 @@ pub const MIN_SAMPLE_UNITS: usize = 8;
|
||||
/// units it yields); the *requested* count is a caller-side compile-time constant that
|
||||
/// callers pin to `MIN_SAMPLE_UNITS` (see e.g. autorip's `SAMPLE_UNITS`). Together the
|
||||
/// two make under-sampling unrepresentable at the request boundary.
|
||||
#[derive(Debug, Clone)]
|
||||
///
|
||||
/// The wrapped samples are on-disc AACS ciphertext — the same bytes the sibling
|
||||
/// [`DiscInputs::samples`] redacts as key MATERIAL — so [`Debug`] is hand-written
|
||||
/// and redacting; see the impl below.
|
||||
#[derive(Clone)]
|
||||
pub struct DecodeSampleSet(Vec<Vec<u8>>);
|
||||
|
||||
impl std::fmt::Debug for DecodeSampleSet {
|
||||
/// Prints the SHAPE only. A derived `Debug` dumped every wrapped sample
|
||||
/// verbatim: a `DecodeSampleSet` carries at least [`MIN_SAMPLE_UNITS`]
|
||||
/// 6144-byte aligned units (≥ 49 KiB, in practice multi-MB) of AACS
|
||||
/// ciphertext plus each unit's clear 16-byte derivation seed, so one
|
||||
/// `tracing::debug!("{set:?}")` on a failed `/decode` request — or an
|
||||
/// `assert_eq!` whose panic message formats it — wrote all of it to the log
|
||||
/// that gets attached to a bug report. Same policy and same shape as
|
||||
/// [`DiscInputs`]'s impl below.
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("DecodeSampleSet")
|
||||
.field("units", &"<redacted>")
|
||||
.field("units_len", &self.0.len())
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
impl DecodeSampleSet {
|
||||
/// Wrap `units` iff it carries at least [`MIN_SAMPLE_UNITS`] samples; `None`
|
||||
/// otherwise (the caller then skips the online source rather than sending an
|
||||
@@ -82,9 +103,12 @@ impl DecodeSampleSet {
|
||||
}
|
||||
|
||||
/// The public AACS inputs a key source needs to look a disc up. Captured at
|
||||
/// scan; contains no secrets — only the disc identity and the on-disc AACS
|
||||
/// structures a source or key server may key on.
|
||||
#[derive(Debug, Clone)]
|
||||
/// scan; carries no DERIVED secrets (no media key, VUK or plaintext unit key) —
|
||||
/// only the disc identity and the on-disc AACS structures a source or key server
|
||||
/// may key on. The on-disc structures are nonetheless key MATERIAL (the encrypted
|
||||
/// title keys live in `unit_key_ro`), so [`Debug`] is hand-written and redacting;
|
||||
/// see the impl below.
|
||||
#[derive(Clone)]
|
||||
pub struct DiscInputs {
|
||||
/// SHA-1 of `Unit_Key_RO.inf`, `0x`-prefixed hex. The value a keydb keys
|
||||
/// its per-disc entries by, and a key server identifies the disc with.
|
||||
@@ -115,7 +139,31 @@ pub struct DiscInputs {
|
||||
pub volume_label: Option<String>,
|
||||
}
|
||||
|
||||
/// A lazy view of a disc's AACS material, handed to [`KeySource::get_uk`] so a
|
||||
/// Redacting `Debug`, per the policy `aacs::types` documents (and which
|
||||
/// `aacs::types::Vid` already applies to this very Volume ID). `DiscInputs` is
|
||||
/// public and returned by [`crate::Disc::inputs`], so a consumer's
|
||||
/// `tracing::debug!("{inputs:?}")` used to print the Volume ID, the whole
|
||||
/// `Unit_Key_RO.inf` (the encrypted title keys), the entire MKB and every
|
||||
/// ciphertext sample verbatim into a log that ends up attached to a bug report.
|
||||
/// Only non-secret identity and shape (presence, lengths) is printed.
|
||||
impl std::fmt::Debug for DiscInputs {
|
||||
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
|
||||
f.debug_struct("DiscInputs")
|
||||
.field("disc_hash", &self.disc_hash)
|
||||
.field("volume_id", &"<redacted>")
|
||||
.field("version", &self.version)
|
||||
.field("mkb", &"<redacted>")
|
||||
.field("mkb_len", &self.mkb.len())
|
||||
.field("unit_key_ro", &"<redacted>")
|
||||
.field("unit_key_ro_len", &self.unit_key_ro.len())
|
||||
.field("samples", &"<redacted>")
|
||||
.field("samples_len", &self.samples.len())
|
||||
.field("volume_label", &self.volume_label)
|
||||
.finish()
|
||||
}
|
||||
}
|
||||
|
||||
/// A lazy view of a disc's AACS material, handed to [`KeySource::get_unit_keys`] so a
|
||||
/// source can drive the derivation chain without holding the disc reader.
|
||||
///
|
||||
/// "Lazy" by contract: each accessor returns only what the source asks for, so a
|
||||
@@ -233,10 +281,34 @@ impl ResolveCtx for DiscInputsCtx<'_> {
|
||||
/// ([`resolve_and_apply`]) tries each source in order and validates the returned
|
||||
/// keys against real ciphertext before committing them, so a wrong key from one
|
||||
/// source transparently falls through to the next.
|
||||
///
|
||||
/// Two explicit resolve operations, one per key kind — never one overloaded call
|
||||
/// whose meaning depends on how many keys came back:
|
||||
/// * [`get_unit_keys`](Self::get_unit_keys) — the disc's base per-CPS-unit Unit
|
||||
/// Keys (index space = CPS-unit number). The common path for every disc.
|
||||
/// * [`get_fmts_indexes`](Self::get_fmts_indexes) — the AACS 2.1 forensic index
|
||||
/// keys (index space = forensic index 1..N). Defaults to empty: a source with
|
||||
/// no forensic material opts out, and only an FMTS disc ever asks.
|
||||
///
|
||||
/// What each source must do to answer is the source's own business: a keydb keys
|
||||
/// on `disc_hash` and reads no samples; the online source submits the ctx's
|
||||
/// content samples (a base batch for `get_unit_keys`, an index-1 anchor batch for
|
||||
/// `get_fmts_indexes`) to the key service.
|
||||
pub trait KeySource {
|
||||
/// Resolve this disc's terminal Unit Keys from this source. An empty `Vec`
|
||||
/// is a genuine "no key here"; `Err` is a source failure.
|
||||
fn get_uk(&self, ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error>;
|
||||
/// Resolve this disc's base per-CPS-unit Unit Keys from this source. An empty
|
||||
/// `Vec` is a genuine "no key here"; `Err` is a source failure.
|
||||
fn get_unit_keys(&self, ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error>;
|
||||
|
||||
/// Resolve this disc's AACS 2.1 forensic index keys — the per-index keys the
|
||||
/// base Unit Key cannot open (see [`crate::aacs::segment`]) — ordered by
|
||||
/// forensic index (element `i` carries `UnitKey.idx == i`, forensic index
|
||||
/// `i + 1`). The source hands back the COMPLETE set it holds; the caller
|
||||
/// trusts any non-empty result as all of them and never assumes a fixed count.
|
||||
/// Defaults to empty: a source with no forensic material (a plain keydb, the
|
||||
/// mapfile) opts out, and only an FMTS disc's mux ever calls this.
|
||||
fn get_fmts_indexes(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
|
||||
/// The AACS host certificate(s) this source can supply for the live-drive
|
||||
/// SCSI mutual-auth handshake (the OEM/AACS baseline route). `mkb` is the
|
||||
@@ -273,7 +345,7 @@ pub fn resolve_and_apply(
|
||||
/// [`crate::aacs::trace::ResolutionTrace`] recording, per source, what happened — for
|
||||
/// applications to render. ZERO English; the trace is typed enums only.
|
||||
///
|
||||
/// One-shot per source: each source's [`KeySource::get_uk`] is called exactly
|
||||
/// One-shot per source: each source's [`KeySource::get_unit_keys`] is called exactly
|
||||
/// once with a [`DiscInputsCtx`] over `inputs`. Non-empty Unit Keys are mapped
|
||||
/// to terminal [`Key::Unit`]s and applied via [`crate::Disc::decrypt_with`],
|
||||
/// which validates them against `inputs.samples` and only mutates the disc on
|
||||
@@ -283,8 +355,23 @@ pub fn resolve_and_apply(
|
||||
/// from [`crate::aacs::derive::decrypt_unit_key`]; the library's canonical CPS-unit number is
|
||||
/// `position + 1` (matching [`crate::aacs::inf::parse_unit_key_ro`]'s `(i + 1)`), so
|
||||
/// the committed `AacsState.unit_keys` is byte-identical to the library-resolved
|
||||
/// path. The number is cosmetic for descramble (the decrypt path strips it and
|
||||
/// tries every key) but is kept faithful to the resolver's convention.
|
||||
/// path.
|
||||
///
|
||||
/// The NUMBER itself is not what descramble indexes by — but the ORDER is
|
||||
/// load-bearing, so a source must return its keys in CPS-unit order. Trial
|
||||
/// decrypt-and-check was deliberately deleted (see
|
||||
/// [`crate::decrypt::AacsKeyMap`]: decryption is driven by the disc's CPS-unit /
|
||||
/// FMTS-segment structure, "never by trial-decrypt-and-check per unit"), and
|
||||
/// `decrypt_sectors_mapped` indexes the committed pool POSITIONALLY —
|
||||
/// `unit_keys[key_idx].1`, where `key_idx` is a POSITION in the Vec a source
|
||||
/// returned, recorded by `resolve_mux_key_map_cached` / `resolve_fmts_key_map`.
|
||||
/// Return the same keys in a different order and every `AacsKeyMap` points at the
|
||||
/// wrong key: the whole title decrypts under a neighbour's key, or a forensic
|
||||
/// range trips the `is_clean` net into `DecryptFailed`. (The doc used to say the
|
||||
/// number "is cosmetic for descramble (the decrypt path strips it and tries every
|
||||
/// key)", which is what the DELETED trial-decrypt path did; the only place that
|
||||
/// still tries every key is `Disc::decrypt_with`'s sample VALIDATION, which does
|
||||
/// not descramble content.)
|
||||
pub fn resolve_and_apply_traced(
|
||||
sources: &[Box<dyn KeySource>],
|
||||
inputs: &DiscInputs,
|
||||
@@ -294,6 +381,14 @@ pub fn resolve_and_apply_traced(
|
||||
|
||||
let mut trace = crate::aacs::trace::ResolutionTrace::new();
|
||||
|
||||
// The FIRST source failure seen, if any. A source that returns `Err` did not
|
||||
// answer "no key for this disc" — it could not answer at all — and that
|
||||
// reason is stamped onto `disc.aacs_error` below so the decrypt gate reports
|
||||
// THAT instead of the generic `NoDiscKey`. First-wins (not last) so the
|
||||
// ordered sources' most-preferred failure is the one the operator is told
|
||||
// about, matching the first-valid-wins rule for successes.
|
||||
let mut source_failure: Option<crate::error::Error> = None;
|
||||
|
||||
// The ctx parses Unit_Key_RO.inf at the stride for `inputs.version` (the
|
||||
// disc's own AACS major), so the stride is the disc's single source of truth.
|
||||
let ctx = DiscInputsCtx::new(inputs);
|
||||
@@ -301,7 +396,7 @@ pub fn resolve_and_apply_traced(
|
||||
for source in sources {
|
||||
// `who` is the source's own stable identifier — no enum to map back to.
|
||||
let who = source.label().to_string();
|
||||
match source.get_uk(&ctx) {
|
||||
match source.get_unit_keys(&ctx) {
|
||||
Ok(uks) if !uks.is_empty() => {
|
||||
// Positional index → canonical CPS-unit number (position + 1).
|
||||
let unit_keys: Vec<(u32, [u8; 16])> = uks
|
||||
@@ -326,17 +421,56 @@ pub fn resolve_and_apply_traced(
|
||||
outcome: KeyOutcome::NoKey,
|
||||
});
|
||||
}
|
||||
// Empty (no key here) or a source failure — both are "no key from
|
||||
// this source"; move on to the next.
|
||||
Ok(_) | Err(_) => {
|
||||
// The source ANSWERED and holds nothing for this disc. This — and
|
||||
// only this — is `NoEntry`: the claim "I looked, it is not there".
|
||||
Ok(_) => {
|
||||
trace.keys.push(KeyStep {
|
||||
who,
|
||||
path: vec![KeyNode::NoEntry],
|
||||
outcome: KeyOutcome::NoKey,
|
||||
});
|
||||
}
|
||||
// The source could NOT answer — it was unreachable, it errored, or it
|
||||
// refused. Nothing is known about whether a key exists, so the path
|
||||
// is EMPTY: recording `NoEntry` here is exactly the conflation that
|
||||
// made a seven-hour run of HTTP 502s render as
|
||||
// `key: online > no entry > NO KEY` + `E7022 No key source has a
|
||||
// decryption key for this disc`, and sent operators hunting for a VUK
|
||||
// that was never missing.
|
||||
//
|
||||
// The reason itself rides out on `disc.aacs_error` (below), the
|
||||
// channel `Disc::ensure_decryptable_keys` already reads for the
|
||||
// E7017-vs-E7022 split — so the decrypt gate raises the SOURCE's code
|
||||
// (`KeyServiceUnavailable` / `KeyServiceUnauthorized` /
|
||||
// `KeyServiceRateLimited`) instead of the generic `NoDiscKey`.
|
||||
//
|
||||
// `KeyOutcome` deliberately gains no variant: it is matched
|
||||
// exhaustively by every front-end's trace renderer (freemkv's
|
||||
// `pipe::render_resolution_trace`, autorip's
|
||||
// `keysource::render_resolution_trace`), and this fix must not turn
|
||||
// into a breaking change across four repos to say something the error
|
||||
// code already says precisely.
|
||||
Err(e) => {
|
||||
if source_failure.is_none() {
|
||||
source_failure = Some(e);
|
||||
}
|
||||
trace.keys.push(KeyStep {
|
||||
who,
|
||||
path: Vec::new(),
|
||||
outcome: KeyOutcome::NoKey,
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
// Nothing resolved. If a source FAILED rather than answered, stamp that
|
||||
// reason onto the disc so the decrypt gate can report it — but never clobber
|
||||
// a reason the scan already captured (e.g. `AacsVidUnavailable`), which is
|
||||
// closer to the disc itself than a source outage is.
|
||||
if let Some(e) = source_failure
|
||||
&& disc.aacs_error.is_none()
|
||||
{
|
||||
disc.aacs_error = Some(e);
|
||||
}
|
||||
(false, trace)
|
||||
}
|
||||
|
||||
@@ -351,71 +485,156 @@ pub fn resolve_and_apply_traced(
|
||||
/// [`resolve_and_apply`] this does not validate/commit to a disc — the read's
|
||||
/// decorator re-decrypts with the returned keys, which is the validation.
|
||||
pub fn fetch_unit_keys(sources: &[Box<dyn KeySource>], ctx: &dyn ResolveCtx) -> Vec<UnitKey> {
|
||||
for source in sources {
|
||||
if let Ok(uks) = source.get_uk(ctx) {
|
||||
if !uks.is_empty() {
|
||||
return uks;
|
||||
}
|
||||
}
|
||||
}
|
||||
Vec::new()
|
||||
drive_unit_keys(sources, ctx).keys
|
||||
}
|
||||
|
||||
/// Build the read-time key-fetch closure from the disc's public AACS inputs and
|
||||
/// a way to (re)build the application's key sources. The decorator calls it with
|
||||
/// the still-scrambled unit ciphertext when no held key opens that unit; it runs
|
||||
/// [`fetch_unit_keys`] with those bytes as `samples` and returns any keys.
|
||||
/// Whether a driver run resolved keys, and — when it did NOT — whether the miss
|
||||
/// was a genuine "no source holds this key" (`errored == false`) or at least one
|
||||
/// source FAILED (`errored == true`, e.g. a network source was unreachable). The
|
||||
/// distinction gates negative-result memoization: an empty-because-absent result
|
||||
/// is safe to cache, an empty-because-a-source-was-down result is transient and
|
||||
/// must NOT be cached (the key may resolve once the source recovers).
|
||||
struct FetchOutcome {
|
||||
keys: Vec<UnitKey>,
|
||||
errored: bool,
|
||||
}
|
||||
|
||||
/// [`fetch_unit_keys`] plus the error signal: drive `sources` in order, return the
|
||||
/// first source's non-empty Unit Keys, and flag whether any source that failed to
|
||||
/// answer did so with an `Err` (a source failure) rather than an empty `Ok`
|
||||
/// (genuine absence — see [`KeySource::get_unit_keys`]).
|
||||
fn drive_unit_keys(sources: &[Box<dyn KeySource>], ctx: &dyn ResolveCtx) -> FetchOutcome {
|
||||
let mut errored = false;
|
||||
for source in sources {
|
||||
match source.get_unit_keys(ctx) {
|
||||
Ok(uks) if !uks.is_empty() => {
|
||||
return FetchOutcome {
|
||||
keys: uks,
|
||||
errored: false,
|
||||
};
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(_) => errored = true,
|
||||
}
|
||||
}
|
||||
FetchOutcome {
|
||||
keys: Vec::new(),
|
||||
errored,
|
||||
}
|
||||
}
|
||||
|
||||
/// The forensic counterpart to [`fetch_unit_keys`]: drive `sources` in order and
|
||||
/// return the first source's non-empty AACS 2.1 forensic index set. `ctx` carries
|
||||
/// the index-1 anchor batch (the mux, which owns disc geometry, gathers it and
|
||||
/// injects it as the ctx's samples); a source that needs no samples (a keydb
|
||||
/// keying on `disc_hash`) ignores them. Whatever the winning source returns —
|
||||
/// ≥ 1 key — is trusted as the COMPLETE ordered set; no fixed count is assumed.
|
||||
pub fn fetch_fmts_indexes(sources: &[Box<dyn KeySource>], ctx: &dyn ResolveCtx) -> Vec<UnitKey> {
|
||||
drive_fmts_indexes(sources, ctx).keys
|
||||
}
|
||||
|
||||
/// [`fetch_fmts_indexes`] plus the error signal (see [`drive_unit_keys`]): the
|
||||
/// forensic counterpart that flags whether any source `Err`ed during the miss.
|
||||
fn drive_fmts_indexes(sources: &[Box<dyn KeySource>], ctx: &dyn ResolveCtx) -> FetchOutcome {
|
||||
let mut errored = false;
|
||||
for source in sources {
|
||||
match source.get_fmts_indexes(ctx) {
|
||||
Ok(uks) if !uks.is_empty() => {
|
||||
return FetchOutcome {
|
||||
keys: uks,
|
||||
errored: false,
|
||||
};
|
||||
}
|
||||
Ok(_) => {}
|
||||
Err(_) => errored = true,
|
||||
}
|
||||
}
|
||||
FetchOutcome {
|
||||
keys: Vec::new(),
|
||||
errored,
|
||||
}
|
||||
}
|
||||
|
||||
/// Build the read-time [`crate::sector::KeyFetch`] from the disc's public AACS
|
||||
/// inputs and a way to (re)build the application's key sources. The returned
|
||||
/// resolver has the two explicit operations the mux and recovery decorator call:
|
||||
/// [`unit_keys`](crate::sector::KeyFetch::unit_keys) drives [`fetch_unit_keys`]
|
||||
/// (base per-CPS-unit keys), [`fmts_indexes`](crate::sector::KeyFetch::fmts_indexes)
|
||||
/// drives [`fetch_fmts_indexes`] (the AACS 2.1 forensic set). Each is handed the
|
||||
/// caller's sample batch as the ctx's `samples`, so a source pulls whatever
|
||||
/// material it needs.
|
||||
///
|
||||
/// One builder, used by every read path (sweep / patch / mux) and by every
|
||||
/// consumer (CLI, autorip) — neither application contains the fetch logic, only
|
||||
/// its key-source config. Returns a **shared, stateless** [`crate::sector::KeyFetch`]
|
||||
/// (`Arc<Fn>`): build it once, clone it into each read path. `make_sources` is
|
||||
/// invoked per fetch (the cold path, ~once per CPS unit) so the closure stays
|
||||
/// One builder, used by every read path (sweep / patch / mux) and every consumer
|
||||
/// (CLI, autorip) — neither application contains the fetch logic, only its
|
||||
/// key-source config. Cheap to clone; build once, clone into each read path.
|
||||
/// `make_sources` is invoked per fetch (the cold path) so the resolver stays
|
||||
/// `Send + Sync` without requiring `KeySource: Send`.
|
||||
pub fn key_fetch(
|
||||
inputs: DiscInputs,
|
||||
make_sources: std::sync::Arc<dyn Fn() -> Vec<Box<dyn KeySource>> + Send + Sync>,
|
||||
) -> crate::sector::KeyFetch {
|
||||
// Memoize by the fingerprint of the sample batch. The resolved keys are
|
||||
// disc-level (the same clip's index / CPS keys are identical for every title
|
||||
// that references it), and this one closure is shared across every title's mux
|
||||
// — so the first title resolves a given batch over the network and every later
|
||||
// title (or repeated batch) is answered from the cache with no request. Empty
|
||||
// replies are cached too: a key the service does not have for a batch will not
|
||||
// appear on a re-ask, so re-hitting the network buys nothing.
|
||||
let cache: std::sync::Arc<std::sync::Mutex<std::collections::HashMap<u64, Vec<[u8; 16]>>>> =
|
||||
std::sync::Arc::new(std::sync::Mutex::new(std::collections::HashMap::new()));
|
||||
std::sync::Arc::new(move |samples: &[Vec<u8>]| -> Vec<[u8; 16]> {
|
||||
let fp = {
|
||||
use std::hash::{Hash, Hasher};
|
||||
let mut h = std::collections::hash_map::DefaultHasher::new();
|
||||
samples.len().hash(&mut h);
|
||||
for s in samples {
|
||||
s.hash(&mut h);
|
||||
// One driver behind both operations: rebuild the sources, inject `samples`
|
||||
// as the ctx's content samples, run `drive` (the per-kind fetch), map the
|
||||
// resolved UnitKeys to raw keys. Memoized by the fingerprint of the sample
|
||||
// batch: the resolved keys are disc-level (a clip's index / CPS keys are
|
||||
// identical for every title that references it), so the first batch resolves
|
||||
// over the network and every repeat is answered from the cache with no
|
||||
// request. A GENUINELY-empty reply (every source ran and none held the key)
|
||||
// is cached too — the key the service lacks for a batch won't appear on a
|
||||
// re-ask, so re-hitting the network buys nothing. But an empty reply caused
|
||||
// by a source FAILURE (network down, source unreachable) is NOT cached: that
|
||||
// is a transient miss, and caching it would permanently drop a unit that
|
||||
// could be recovered once the source recovers — the `errored` flag on
|
||||
// `FetchOutcome` draws exactly that line. Each operation gets its OWN cache:
|
||||
// a base batch and a forensic anchor never collide, and the same bytes could
|
||||
// legitimately resolve differently per op.
|
||||
// The per-kind driver: `drive_unit_keys` or `drive_fmts_indexes`.
|
||||
type FetchDriver = fn(&[Box<dyn KeySource>], &dyn ResolveCtx) -> FetchOutcome;
|
||||
fn make_op(
|
||||
inputs: DiscInputs,
|
||||
make_sources: std::sync::Arc<dyn Fn() -> Vec<Box<dyn KeySource>> + Send + Sync>,
|
||||
drive: FetchDriver,
|
||||
) -> crate::sector::KeyFetchFn {
|
||||
let cache: std::sync::Arc<std::sync::Mutex<std::collections::HashMap<u64, Vec<[u8; 16]>>>> =
|
||||
std::sync::Arc::new(std::sync::Mutex::new(std::collections::HashMap::new()));
|
||||
std::sync::Arc::new(move |samples: &[Vec<u8>]| -> Vec<[u8; 16]> {
|
||||
let fp = {
|
||||
use std::hash::{Hash, Hasher};
|
||||
let mut h = std::collections::hash_map::DefaultHasher::new();
|
||||
samples.len().hash(&mut h);
|
||||
for s in samples {
|
||||
s.hash(&mut h);
|
||||
}
|
||||
h.finish()
|
||||
};
|
||||
if let Some(hit) = cache.lock().unwrap_or_else(|e| e.into_inner()).get(&fp) {
|
||||
return hit.clone();
|
||||
}
|
||||
h.finish()
|
||||
};
|
||||
if let Some(hit) = cache.lock().unwrap_or_else(|e| e.into_inner()).get(&fp) {
|
||||
return hit.clone();
|
||||
}
|
||||
let sources = make_sources();
|
||||
let mut di = inputs.clone();
|
||||
di.samples = samples.to_vec();
|
||||
// Parse Unit_Key_RO.inf at the disc's OWN stride (carried on `inputs`):
|
||||
// an online /decode reply that returns a VUK (not a terminal UK) then
|
||||
// derives unit keys from `enc_title_keys`, which a V10 disc parses at the
|
||||
// 48-byte stride — hardcoding the V20 stride here corrupted them.
|
||||
let ctx = DiscInputsCtx::new(&di);
|
||||
let keys: Vec<[u8; 16]> = fetch_unit_keys(&sources, &ctx)
|
||||
.into_iter()
|
||||
.map(|u| u.key)
|
||||
.collect();
|
||||
cache
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.insert(fp, keys.clone());
|
||||
keys
|
||||
})
|
||||
let sources = make_sources();
|
||||
let mut di = inputs.clone();
|
||||
di.samples = samples.to_vec();
|
||||
// Parse Unit_Key_RO.inf at the disc's OWN stride (carried on `inputs`):
|
||||
// an online /decode reply that returns a VUK (not a terminal UK) then
|
||||
// derives unit keys from `enc_title_keys`, which a V10 disc parses at
|
||||
// the 48-byte stride — hardcoding the V20 stride here corrupted them.
|
||||
let ctx = DiscInputsCtx::new(&di);
|
||||
let outcome = drive(&sources, &ctx);
|
||||
let keys: Vec<[u8; 16]> = outcome.keys.into_iter().map(|u| u.key).collect();
|
||||
// Memoize a positive result always; memoize a NEGATIVE (empty) result
|
||||
// only when it is a genuine absence, never when a source errored — a
|
||||
// transient outage must not permanently poison this fingerprint.
|
||||
if !keys.is_empty() || !outcome.errored {
|
||||
cache
|
||||
.lock()
|
||||
.unwrap_or_else(|e| e.into_inner())
|
||||
.insert(fp, keys.clone());
|
||||
}
|
||||
keys
|
||||
})
|
||||
}
|
||||
let unit = make_op(inputs.clone(), make_sources.clone(), drive_unit_keys);
|
||||
let fmts = make_op(inputs, make_sources, drive_fmts_indexes);
|
||||
crate::sector::KeyFetch::new(unit, fmts)
|
||||
}
|
||||
|
||||
/// Read up to `n` ENCRYPTED 6144-byte aligned units from `title`'s body, raw (no
|
||||
@@ -430,9 +649,9 @@ pub fn key_fetch(
|
||||
///
|
||||
/// "Encrypted" is decided by [`crate::aacs::content::aacs_unit_encrypted`] — the
|
||||
/// AACS Copy Permission Indicator (CPI) in the top 2 bits of byte 0, the
|
||||
/// spec-correct signal (`buf[0] & 0xc0`). NOT the `ts_sync_destroyed`
|
||||
/// sync heuristic: destroyed TS syncs do not imply encryption (an FMTS variant
|
||||
/// frame or an odd clear unit can lack syncs yet be unencrypted), and a clear
|
||||
/// spec-correct signal (`buf[0] & 0xc0`). NOT the `is_clean` TS-sync
|
||||
/// heuristic: a unit lacking clean TS syncs does not imply encryption (an FMTS
|
||||
/// variant frame or an odd clear unit can lack syncs yet be unencrypted), and a clear
|
||||
/// unit sent to a key server yields nothing to validate against — the "0
|
||||
/// encrypted units" rejection. A clip opens with clear navigation units (PAT/PMT,
|
||||
/// menus) whose CPI is clear; only CPI-flagged content units are collected —
|
||||
@@ -560,7 +779,7 @@ mod tests {
|
||||
fn key_source_host_certs_defaults_to_empty() {
|
||||
struct MinimalSource;
|
||||
impl KeySource for MinimalSource {
|
||||
fn get_uk(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
}
|
||||
@@ -617,7 +836,7 @@ mod tests {
|
||||
fn trace_who_is_the_source_label_verbatim() {
|
||||
struct LabeledSource(&'static str);
|
||||
impl KeySource for LabeledSource {
|
||||
fn get_uk(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
fn label(&self) -> &'static str {
|
||||
@@ -674,19 +893,19 @@ mod tests {
|
||||
|
||||
struct EmptySource;
|
||||
impl KeySource for EmptySource {
|
||||
fn get_uk(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
}
|
||||
struct ErroringSource;
|
||||
impl KeySource for ErroringSource {
|
||||
fn get_uk(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Err(Error::AacsNoKeys)
|
||||
}
|
||||
}
|
||||
struct HasKey([u8; 16]);
|
||||
impl KeySource for HasKey {
|
||||
fn get_uk(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(vec![UnitKey::new(0, self.0)])
|
||||
}
|
||||
}
|
||||
@@ -729,7 +948,7 @@ mod tests {
|
||||
seen: Arc<Mutex<Vec<Vec<u8>>>>,
|
||||
}
|
||||
impl KeySource for Probe {
|
||||
fn get_uk(&self, ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
fn get_unit_keys(&self, ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
if let Ok(s) = ctx.samples(8) {
|
||||
self.seen.lock().unwrap().extend(s);
|
||||
}
|
||||
@@ -749,7 +968,7 @@ mod tests {
|
||||
|
||||
let cb = key_fetch(empty_inputs(), make);
|
||||
let samples = vec![vec![0xEEu8; crate::aacs::content::ALIGNED_UNIT_LEN]];
|
||||
let got = cb(&samples);
|
||||
let got = cb.unit_keys(&samples);
|
||||
assert_eq!(
|
||||
got,
|
||||
vec![key],
|
||||
@@ -763,6 +982,225 @@ mod tests {
|
||||
assert_eq!(*builds.lock().unwrap(), 1, "make_sources invoked per fetch");
|
||||
}
|
||||
|
||||
/// `key_fetch` memoizes each operation by the fingerprint of the sample batch:
|
||||
/// identical samples reuse the cached keys (no rebuild), different samples miss,
|
||||
/// the two operations keep independent caches, and even an empty reply is cached.
|
||||
#[test]
|
||||
fn key_fetch_memoizes_per_op_by_sample_fingerprint() {
|
||||
let builds = Arc::new(Mutex::new(0usize));
|
||||
let builds_c = Arc::clone(&builds);
|
||||
let key = [0x11u8; 16];
|
||||
let make: Arc<dyn Fn() -> Vec<Box<dyn KeySource>> + Send + Sync> = Arc::new(move || {
|
||||
*builds_c.lock().unwrap() += 1;
|
||||
vec![Box::new(HasKey(key)) as Box<dyn KeySource>]
|
||||
});
|
||||
let cb = key_fetch(empty_inputs(), make);
|
||||
let a = vec![vec![0xAAu8; 8]];
|
||||
let b = vec![vec![0xBBu8; 8]];
|
||||
|
||||
// First resolve for `a` builds sources; the identical repeat is cached.
|
||||
assert_eq!(cb.unit_keys(&a), vec![key]);
|
||||
assert_eq!(cb.unit_keys(&a), vec![key]);
|
||||
assert_eq!(
|
||||
*builds.lock().unwrap(),
|
||||
1,
|
||||
"identical samples reuse the cache"
|
||||
);
|
||||
|
||||
// A different sample batch is a cache miss → one more build.
|
||||
assert_eq!(cb.unit_keys(&b), vec![key]);
|
||||
assert_eq!(
|
||||
*builds.lock().unwrap(),
|
||||
2,
|
||||
"different samples miss the cache"
|
||||
);
|
||||
|
||||
// The forensic op has its OWN cache (HasKey has no forensic keys → empty),
|
||||
// so `a` builds once more here; its empty reply is then cached too.
|
||||
assert!(cb.fmts_indexes(&a).is_empty());
|
||||
assert_eq!(
|
||||
*builds.lock().unwrap(),
|
||||
3,
|
||||
"unit/fmts caches are independent"
|
||||
);
|
||||
assert!(cb.fmts_indexes(&a).is_empty());
|
||||
assert_eq!(
|
||||
*builds.lock().unwrap(),
|
||||
3,
|
||||
"an empty reply is cached, not re-asked"
|
||||
);
|
||||
}
|
||||
|
||||
/// A transient source outage must NOT be memoized as a permanent "no key":
|
||||
/// a fingerprint whose first fetch failed because the source errored must be
|
||||
/// re-asked, and once the source recovers the key resolves. Regression guard
|
||||
/// for the negative-result memoization fix — caching the errored empty would
|
||||
/// permanently drop a recoverable unit for the rest of the op.
|
||||
#[test]
|
||||
fn errored_empty_is_not_cached_and_retries_when_source_recovers() {
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
|
||||
let key = [0x77u8; 16];
|
||||
// Shared across every `make_sources()` rebuild: call 0 errors (source
|
||||
// down), every later call succeeds (source recovered).
|
||||
let calls = Arc::new(AtomicUsize::new(0));
|
||||
|
||||
struct Flaky {
|
||||
calls: Arc<AtomicUsize>,
|
||||
key: [u8; 16],
|
||||
}
|
||||
impl KeySource for Flaky {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
if self.calls.fetch_add(1, Ordering::SeqCst) == 0 {
|
||||
Err(Error::AacsNoKeys) // first attempt: source unreachable
|
||||
} else {
|
||||
Ok(vec![UnitKey::new(0, self.key)])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
let calls_c = Arc::clone(&calls);
|
||||
let make: Arc<dyn Fn() -> Vec<Box<dyn KeySource>> + Send + Sync> = Arc::new(move || {
|
||||
vec![Box::new(Flaky {
|
||||
calls: Arc::clone(&calls_c),
|
||||
key,
|
||||
}) as Box<dyn KeySource>]
|
||||
});
|
||||
|
||||
let cb = key_fetch(empty_inputs(), make);
|
||||
let samples = vec![vec![0xCDu8; 8]];
|
||||
|
||||
// First fetch: the source errors → empty, but the miss must NOT be cached.
|
||||
assert!(
|
||||
cb.unit_keys(&samples).is_empty(),
|
||||
"source down → empty this time"
|
||||
);
|
||||
// Second fetch, SAME samples: not blocked by a cached empty → the now-
|
||||
// recovered source resolves the key.
|
||||
assert_eq!(
|
||||
cb.unit_keys(&samples),
|
||||
vec![key],
|
||||
"recovered source resolves — errored empty was not memoized"
|
||||
);
|
||||
}
|
||||
|
||||
/// A GENUINE absence (a source that runs and returns an empty `Ok`) is still
|
||||
/// memoized — the benefit the fix preserves. A source counting its calls must
|
||||
/// be asked exactly once for a fingerprint whose first (clean) reply was empty.
|
||||
#[test]
|
||||
fn genuine_empty_is_still_memoized() {
|
||||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||||
|
||||
let calls = Arc::new(AtomicUsize::new(0));
|
||||
|
||||
struct AlwaysEmpty {
|
||||
calls: Arc<AtomicUsize>,
|
||||
}
|
||||
impl KeySource for AlwaysEmpty {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
self.calls.fetch_add(1, Ordering::SeqCst);
|
||||
Ok(Vec::new()) // ran fine, genuinely holds no key
|
||||
}
|
||||
}
|
||||
|
||||
let calls_c = Arc::clone(&calls);
|
||||
let make: Arc<dyn Fn() -> Vec<Box<dyn KeySource>> + Send + Sync> = Arc::new(move || {
|
||||
vec![Box::new(AlwaysEmpty {
|
||||
calls: Arc::clone(&calls_c),
|
||||
}) as Box<dyn KeySource>]
|
||||
});
|
||||
|
||||
let cb = key_fetch(empty_inputs(), make);
|
||||
let samples = vec![vec![0xEFu8; 8]];
|
||||
|
||||
assert!(cb.unit_keys(&samples).is_empty());
|
||||
assert!(cb.unit_keys(&samples).is_empty());
|
||||
assert_eq!(
|
||||
calls.load(Ordering::SeqCst),
|
||||
1,
|
||||
"a clean empty reply is cached — the source is asked only once"
|
||||
);
|
||||
}
|
||||
|
||||
/// The two `KeyFetch` operations route to the two DISTINCT trait methods:
|
||||
/// `unit_keys` drives `get_unit_keys`, `fmts_indexes` drives
|
||||
/// `get_fmts_indexes`. A source that returns different keys per method proves
|
||||
/// the seam no longer collapses "1 base key" and "the forensic set" into one
|
||||
/// overloaded call — the operation, not the return length, decides which.
|
||||
#[test]
|
||||
fn key_fetch_routes_unit_and_fmts_to_distinct_source_methods() {
|
||||
const BASE: [u8; 16] = [0xB0; 16];
|
||||
const F1: [u8; 16] = [0xF1; 16];
|
||||
const F2: [u8; 16] = [0xF2; 16];
|
||||
|
||||
struct TwoOp;
|
||||
impl KeySource for TwoOp {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(vec![UnitKey::new(0, BASE)])
|
||||
}
|
||||
fn get_fmts_indexes(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(vec![UnitKey::new(0, F1), UnitKey::new(1, F2)])
|
||||
}
|
||||
}
|
||||
|
||||
let make: Arc<dyn Fn() -> Vec<Box<dyn KeySource>> + Send + Sync> =
|
||||
Arc::new(|| vec![Box::new(TwoOp) as Box<dyn KeySource>]);
|
||||
let cb = key_fetch(empty_inputs(), make);
|
||||
let samples = vec![vec![0x01u8; 4]];
|
||||
|
||||
assert_eq!(
|
||||
cb.unit_keys(&samples),
|
||||
vec![BASE],
|
||||
"unit_keys resolves the base Unit Key via get_unit_keys"
|
||||
);
|
||||
assert_eq!(
|
||||
cb.fmts_indexes(&samples),
|
||||
vec![F1, F2],
|
||||
"fmts_indexes resolves the forensic set (any length) via get_fmts_indexes"
|
||||
);
|
||||
}
|
||||
|
||||
/// `KeyFetch::unit_only` serves base keys but NEVER a forensic set — the
|
||||
/// contract the sweep/patch recovery decorator relies on (it resolves CPS
|
||||
/// units only). Its `fmts_indexes` is unconditionally empty.
|
||||
#[test]
|
||||
fn key_fetch_unit_only_never_serves_forensic() {
|
||||
let f = crate::sector::KeyFetch::unit_only(std::sync::Arc::new(|_| vec![[0xAA; 16]]));
|
||||
assert_eq!(f.unit_keys(&[vec![0u8; 4]]), vec![[0xAA; 16]]);
|
||||
assert!(
|
||||
f.fmts_indexes(&[vec![0u8; 4]]).is_empty(),
|
||||
"unit_only resolver yields no forensic keys"
|
||||
);
|
||||
}
|
||||
|
||||
/// `get_fmts_indexes` defaults to empty, so a base-only source (a keydb) opts
|
||||
/// out of the forensic path without implementing it. `fetch_fmts_indexes` then
|
||||
/// falls through to the next source, exactly like the unit-key driver.
|
||||
#[test]
|
||||
fn fetch_fmts_indexes_skips_default_optout_source() {
|
||||
struct BaseOnly; // uses the default (empty) get_fmts_indexes
|
||||
impl KeySource for BaseOnly {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(vec![UnitKey::new(0, [0x11; 16])])
|
||||
}
|
||||
}
|
||||
struct Forensic;
|
||||
impl KeySource for Forensic {
|
||||
fn get_unit_keys(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(Vec::new())
|
||||
}
|
||||
fn get_fmts_indexes(&self, _ctx: &dyn ResolveCtx) -> Result<Vec<UnitKey>, Error> {
|
||||
Ok(vec![UnitKey::new(0, [0x77; 16])])
|
||||
}
|
||||
}
|
||||
let inputs = empty_inputs();
|
||||
let ctx = DiscInputsCtx::new(&inputs);
|
||||
let sources: Vec<Box<dyn KeySource>> = vec![Box::new(BaseOnly), Box::new(Forensic)];
|
||||
let got = fetch_fmts_indexes(&sources, &ctx);
|
||||
assert_eq!(got.len(), 1);
|
||||
assert_eq!(got[0].key, [0x77; 16], "the base-only source is skipped");
|
||||
}
|
||||
|
||||
/// #4 regression: encrypted content NOT at the extent midpoint (a late-
|
||||
/// starting feature, or a midpoint landing in clear nav) must still be
|
||||
/// sampled — empty samples make `decrypt_with` skip wrong-key validation.
|
||||
@@ -848,12 +1286,12 @@ mod tests {
|
||||
}
|
||||
|
||||
/// DISCRIMINATING: selection is by the AACS CPI (byte 0), NOT the
|
||||
/// `ts_sync_destroyed` heuristic. Half the units are sync-destroyed but
|
||||
/// TS-sync clarity heuristic. Half the units lack TS syncs but are
|
||||
/// CPI-CLEAR (`byte0 & 0xC0 == 0`) — genuinely UNencrypted units that merely
|
||||
/// lack TS syncs; the old sampler collected these and the key server rejected
|
||||
/// the POST as "0 encrypted units". `read_encrypted_units` must skip them and
|
||||
/// return ONLY CPI-flagged units. A regression to `ts_sync_destroyed` would
|
||||
/// collect the CPI-clear units too and fail the `& 0xC0` assertion.
|
||||
/// return ONLY CPI-flagged units. A regression to selecting by TS-sync clarity
|
||||
/// would collect the CPI-clear units too and fail the `& 0xC0` assertion.
|
||||
#[test]
|
||||
fn read_encrypted_units_selects_by_cpi_not_ts_sync() {
|
||||
use crate::aacs::content::{ALIGNED_UNIT_LEN, ALIGNED_UNIT_SECTORS, aacs_unit_encrypted};
|
||||
@@ -862,7 +1300,8 @@ mod tests {
|
||||
|
||||
// Even units: CPI-clear (byte0 & 0xC0 == 0) AND sync-destroyed (no 0x47).
|
||||
// Odd units: CPI-set (byte0 = 0xC0) with a scrambled body.
|
||||
// `ts_sync_destroyed` is TRUE for BOTH; `aacs_unit_encrypted` only odd.
|
||||
// Neither has clean TS syncs, so `is_clean` is FALSE for BOTH;
|
||||
// `aacs_unit_encrypted` flags only the odd units.
|
||||
struct MixSource {
|
||||
ext_start: u32,
|
||||
total_units: u32,
|
||||
@@ -884,7 +1323,7 @@ mod tests {
|
||||
break;
|
||||
}
|
||||
let abs = (lba - self.ext_start) / ALIGNED_UNIT_SECTORS + i as u32;
|
||||
if abs % 2 == 0 {
|
||||
if abs.is_multiple_of(2) {
|
||||
chunk.fill(0x11); // CPI-clear (0x11 & 0xC0 == 0), no TS sync
|
||||
} else {
|
||||
chunk.fill(0xAB); // scrambled body (no TS sync)
|
||||
@@ -982,4 +1421,63 @@ mod tests {
|
||||
assert_eq!(k10[1], [0x10; 16], "V10 reads the 2nd key at +48");
|
||||
assert_ne!(k20[1], k10[1], "the parse stride follows inputs.version");
|
||||
}
|
||||
|
||||
/// `DiscInputs` is public and returned by `Disc::inputs`, so any consumer's
|
||||
/// `tracing::debug!("{inputs:?}")` prints it. A derived `Debug` printed the
|
||||
/// Volume ID (the value `aacs::types::Vid` deliberately renders as
|
||||
/// `Vid(<redacted>)`), the whole `Unit_Key_RO.inf` (the encrypted title keys),
|
||||
/// the entire MKB and every ciphertext sample verbatim. Sentinel byte
|
||||
/// 0xD5 = decimal 213, matching `aacs::types::redaction_tests`. Mutation
|
||||
/// guard: restoring `#[derive(Debug)]` fails this.
|
||||
#[test]
|
||||
fn disc_inputs_debug_is_redacted() {
|
||||
let inputs = DiscInputs {
|
||||
disc_hash: "0xAA".into(),
|
||||
volume_id: [0xD5; 16],
|
||||
version: 2,
|
||||
mkb: vec![0xD5; 64],
|
||||
unit_key_ro: vec![0xD5; 48],
|
||||
samples: vec![vec![0xD5; 6144]],
|
||||
volume_label: Some("TITLE_2024".into()),
|
||||
};
|
||||
let dbg = format!("{inputs:?}");
|
||||
assert!(
|
||||
!dbg.contains("213"),
|
||||
"DiscInputs Debug leaked key material (decimal 213): {dbg}"
|
||||
);
|
||||
assert!(
|
||||
dbg.contains("redacted"),
|
||||
"DiscInputs Debug missing redaction marker: {dbg}"
|
||||
);
|
||||
// Non-secret identity and shape stay printable for diagnostics.
|
||||
assert!(dbg.contains("0xAA"), "{dbg}");
|
||||
assert!(dbg.contains("mkb_len: 64"), "{dbg}");
|
||||
assert!(dbg.contains("unit_key_ro_len: 48"), "{dbg}");
|
||||
assert!(dbg.contains("samples_len: 1"), "{dbg}");
|
||||
assert!(dbg.contains("TITLE_2024"), "{dbg}");
|
||||
}
|
||||
|
||||
/// `DecodeSampleSet` is public and wraps the SAME on-disc ciphertext the
|
||||
/// sibling `DiscInputs` redacts, so a derived `Debug` dumped ≥ MIN_SAMPLE_UNITS
|
||||
/// × 6144 bytes of verbatim AACS ciphertext (plus every unit's clear 16-byte
|
||||
/// derivation seed) into any log that formatted it. Sentinel byte 0xD5 =
|
||||
/// decimal 213, matching `aacs::types::redaction_tests` and the
|
||||
/// `DiscInputs` test above. Mutation guard: restoring `#[derive(Debug)]`
|
||||
/// fails this.
|
||||
#[test]
|
||||
fn decode_sample_set_debug_is_redacted() {
|
||||
let set = DecodeSampleSet::new(vec![vec![0xD5; 6144]; MIN_SAMPLE_UNITS])
|
||||
.expect("MIN_SAMPLE_UNITS units is a valid set");
|
||||
let dbg = format!("{set:?}");
|
||||
assert!(
|
||||
!dbg.contains("213"),
|
||||
"DecodeSampleSet Debug leaked ciphertext (decimal 213): {dbg}"
|
||||
);
|
||||
assert!(
|
||||
dbg.contains("redacted"),
|
||||
"DecodeSampleSet Debug missing redaction marker: {dbg}"
|
||||
);
|
||||
// Non-secret shape stays printable for diagnostics.
|
||||
assert!(dbg.contains("units_len: 8"), "{dbg}");
|
||||
}
|
||||
}
|
||||
|
||||
+45
-16
@@ -100,10 +100,10 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<DiscMetadata>
|
||||
// Disc-set position is disc-global; first one we successfully
|
||||
// read wins. (All bdmt_*.xml on a given disc carry the same
|
||||
// value in practice.)
|
||||
if out.disc_number.is_none() {
|
||||
if let Some(ds) = disc_set {
|
||||
out.disc_number = Some(ds);
|
||||
}
|
||||
if out.disc_number.is_none()
|
||||
&& let Some(ds) = disc_set
|
||||
{
|
||||
out.disc_number = Some(ds);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -184,20 +184,20 @@ fn extract_title(xml_text: &str) -> Option<String> {
|
||||
// xml::text already trims its result, so an empty string after
|
||||
// extraction means a genuinely empty element.
|
||||
for tag in ["name", "title"] {
|
||||
if let Some(s) = xml::text(xml_text, tag) {
|
||||
if !s.is_empty() {
|
||||
return Some(s);
|
||||
}
|
||||
if let Some(s) = xml::text(xml_text, tag)
|
||||
&& !s.is_empty()
|
||||
{
|
||||
return Some(s);
|
||||
}
|
||||
}
|
||||
// tableOfContents/titleName: search inside the toc block so we
|
||||
// don't accidentally pick a stray <titleName> from elsewhere.
|
||||
if let Some((s, e)) = xml::find_element(xml_text, "tableOfContents", 0) {
|
||||
let block = &xml_text[s..e];
|
||||
if let Some(t) = xml::text(block, "titleName") {
|
||||
if !t.is_empty() {
|
||||
return Some(t);
|
||||
}
|
||||
if let Some(t) = xml::text(block, "titleName")
|
||||
&& !t.is_empty()
|
||||
{
|
||||
return Some(t);
|
||||
}
|
||||
}
|
||||
None
|
||||
@@ -370,10 +370,10 @@ mod tests {
|
||||
if let Some(d) = desc {
|
||||
meta.descriptions.insert(lang.to_string(), d);
|
||||
}
|
||||
if meta.disc_number.is_none() {
|
||||
if let Some(d) = ds {
|
||||
meta.disc_number = Some(d);
|
||||
}
|
||||
if meta.disc_number.is_none()
|
||||
&& let Some(d) = ds
|
||||
{
|
||||
meta.disc_number = Some(d);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -566,4 +566,33 @@ mod tests {
|
||||
let (title, _, _) = parse_bdmt_xml(xml).unwrap();
|
||||
assert_eq!(title, "Real Title");
|
||||
}
|
||||
|
||||
/// `is_bdmt_filename` must recognize the `bdmt_<lang>.xml` convention
|
||||
/// and reject everything else — it drives `detect`'s directory scan.
|
||||
/// Mutation: stub the return to a constant `true`/`false` → every
|
||||
/// directory listing (or none) would match regardless of filename.
|
||||
#[test]
|
||||
fn is_bdmt_filename_matches_convention_only() {
|
||||
assert!(is_bdmt_filename("bdmt_eng.xml"));
|
||||
assert!(is_bdmt_filename("BDMT_FRA.XML"));
|
||||
assert!(!is_bdmt_filename("bdmt_engl.xml"));
|
||||
assert!(!is_bdmt_filename("index.bdmv"));
|
||||
assert!(!is_bdmt_filename("foo.xml"));
|
||||
}
|
||||
|
||||
/// Spec: "Disc 1 of 1" (a single-disc release whose bdmt XML still
|
||||
/// carries `<di:numSets>1</di:numSets>`) is a valid, non-nonsensical
|
||||
/// pair — `total < 1` must reject only `total == 0`, not `total == 1`.
|
||||
/// Mutation: `total < 1` -> `total == 1` or `total <= 1` would reject
|
||||
/// this legitimate (1, 1) pair as if it were malformed.
|
||||
#[test]
|
||||
fn disc_set_allows_single_disc_release() {
|
||||
let xml = r#"<discInfo xmlns:di="urn:BDA:bdmv;disclibmeta">
|
||||
<di:name>Film</di:name>
|
||||
<di:discNumber>1</di:discNumber>
|
||||
<di:numSets>1</di:numSets>
|
||||
</discInfo>"#;
|
||||
let (_, _, set) = parse_bdmt_xml(xml).unwrap();
|
||||
assert_eq!(set, Some((1, 1)));
|
||||
}
|
||||
}
|
||||
|
||||
+387
-1
@@ -989,7 +989,18 @@ impl<'a> Reader<'a> {
|
||||
}
|
||||
|
||||
fn slice(&mut self, n: usize, needed: &'static str) -> Result<&'a [u8]> {
|
||||
if self.pos + n > self.data.len() {
|
||||
// `n` is attacker-supplied: it comes from a JVMS `u4` attribute_length
|
||||
// / code_length (§4.7, §4.7.3) or a `u2` Utf8 length (§4.4.7). Unlike
|
||||
// the fixed-width readers above, whose `self.pos + k` cannot leave the
|
||||
// buffer's own address range, `self.pos + n` can wrap — on a 32-bit
|
||||
// target a `u4` length near 0xFFFF_FFFF plus a non-zero `pos` panics
|
||||
// in debug and in release wraps to a SMALL end offset that passes the
|
||||
// bounds check, after which the slice index itself panics. Checked, so
|
||||
// an out-of-range length is the EOF error it always should have been.
|
||||
let Some(end) = self.pos.checked_add(n) else {
|
||||
return Err(Error::UnexpectedEof { needed });
|
||||
};
|
||||
if end > self.data.len() {
|
||||
return Err(Error::UnexpectedEof { needed });
|
||||
}
|
||||
let s = &self.data[self.pos..self.pos + n];
|
||||
@@ -1006,6 +1017,46 @@ impl<'a> Reader<'a> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// `Reader::slice` takes an attacker-supplied length: a JVMS `u4`
|
||||
/// `attribute_length` / `code_length` (§4.7, §4.7.3) or a `u2` Utf8
|
||||
/// length (§4.4.7). Adding it to `pos` without a wrap check panics on
|
||||
/// overflow in debug and, in release, wraps to a small end offset that
|
||||
/// slips past the bounds check and then panics inside the slice index.
|
||||
/// Both are panics escaping a parser whose whole input is untrusted disc
|
||||
/// bytes; the contract is an EOF error.
|
||||
#[test]
|
||||
fn slice_rejects_a_length_that_would_wrap_pos() {
|
||||
let data = [0u8; 16];
|
||||
let mut r = Reader::new(&data);
|
||||
r.u64("advance pos").expect("8 bytes available");
|
||||
// pos is now 8; usize::MAX would wrap the end offset to 7.
|
||||
match r.slice(usize::MAX, "wrapping length") {
|
||||
Err(Error::UnexpectedEof { .. }) => {}
|
||||
Err(other) => panic!("expected UnexpectedEof, got {other:?}"),
|
||||
Ok(s) => panic!("expected UnexpectedEof, got a {}-byte slice", s.len()),
|
||||
}
|
||||
// The reader must not have consumed anything.
|
||||
match r.slice(8, "remaining bytes") {
|
||||
Ok(s) => assert_eq!(s.len(), 8, "pos moved on the rejected slice"),
|
||||
Err(e) => panic!("the remaining 8 bytes must still be readable: {e:?}"),
|
||||
}
|
||||
}
|
||||
|
||||
/// The ordinary out-of-range case (no wrap) must keep returning EOF, and
|
||||
/// an exactly-fitting length must still succeed — the check is `>`, not
|
||||
/// `>=`.
|
||||
#[test]
|
||||
fn slice_boundary_is_inclusive_of_the_final_byte() {
|
||||
let data = [0u8; 16];
|
||||
let mut r = Reader::new(&data);
|
||||
assert_eq!(r.slice(16, "whole buffer").expect("exact fit").len(), 16);
|
||||
let mut r = Reader::new(&data);
|
||||
assert!(matches!(
|
||||
r.slice(17, "one past"),
|
||||
Err(Error::UnexpectedEof { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_non_class_bytes() {
|
||||
match ClassFile::parse(b"\x00\x01\x02\x03DEAD") {
|
||||
@@ -1337,4 +1388,339 @@ mod tests {
|
||||
let _ = decode_modified_utf8(&buf);
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// ConstantPool / ClassFile accessor correctness
|
||||
//
|
||||
// These exercise plain data accessors on an already-parsed pool
|
||||
// (built via the test-only `from_entries` constructor) — not the
|
||||
// untrusted-bytes parsing path, just "does the right variant map to
|
||||
// the right Option value."
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
fn sample_pool() -> ConstantPool {
|
||||
// index: 0=Empty (reserved), 1=Utf8("Hello"), 2=Integer(42),
|
||||
// 3=String{string_index:1}, 4=Class{name_index:1}, 5=Float(1.5),
|
||||
// 6=Long(9), 7=Empty (2-slot tail), 8=Double(2.5), 9=Empty (tail).
|
||||
ConstantPool::from_entries(vec![
|
||||
CpInfo::Empty,
|
||||
CpInfo::Utf8("Hello".to_string()),
|
||||
CpInfo::Integer(42),
|
||||
CpInfo::String { string_index: 1 },
|
||||
CpInfo::Class { name_index: 1 },
|
||||
CpInfo::Float(1.5),
|
||||
CpInfo::Long(9),
|
||||
CpInfo::Empty,
|
||||
CpInfo::Double(2.5),
|
||||
CpInfo::Empty,
|
||||
])
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_pool_string_resolves_through_string_index() {
|
||||
let pool = sample_pool();
|
||||
// index 3 is CpInfo::String{string_index: 1} -> utf8(1) = "Hello".
|
||||
assert_eq!(pool.string(3), Some("Hello"));
|
||||
// Wrong variant (Integer at index 2) must not resolve as a string.
|
||||
assert_eq!(pool.string(2), None);
|
||||
// Out of range index.
|
||||
assert_eq!(pool.string(999), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_pool_integer_resolves_only_integer_entries() {
|
||||
let pool = sample_pool();
|
||||
assert_eq!(pool.integer(2), Some(42));
|
||||
// Wrong variant (Utf8 at index 1) must not resolve as an integer.
|
||||
assert_eq!(pool.integer(1), None);
|
||||
assert_eq!(pool.integer(999), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_pool_load_constant_display_covers_ldc_operand_kinds() {
|
||||
let pool = sample_pool();
|
||||
assert_eq!(
|
||||
pool.load_constant_display(1),
|
||||
Some("utf8:\"Hello\"".to_string())
|
||||
);
|
||||
assert_eq!(pool.load_constant_display(2), Some("int:42".to_string()));
|
||||
assert_eq!(
|
||||
pool.load_constant_display(3),
|
||||
Some("str:\"Hello\"".to_string())
|
||||
);
|
||||
assert_eq!(
|
||||
pool.load_constant_display(4),
|
||||
Some("class:\"Hello\"".to_string())
|
||||
);
|
||||
assert_eq!(pool.load_constant_display(5), Some("float:1.5".to_string()));
|
||||
assert_eq!(pool.load_constant_display(6), Some("long:9".to_string()));
|
||||
assert_eq!(
|
||||
pool.load_constant_display(8),
|
||||
Some("double:2.5".to_string())
|
||||
);
|
||||
// A variant with no display arm (e.g. reserved Empty slot) -> None.
|
||||
assert_eq!(pool.load_constant_display(0), None);
|
||||
assert_eq!(pool.load_constant_display(999), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_pool_len_and_is_empty() {
|
||||
let pool = sample_pool();
|
||||
assert_eq!(pool.len(), 10);
|
||||
assert!(!pool.is_empty());
|
||||
|
||||
let empty = ConstantPool::from_entries(vec![]);
|
||||
assert_eq!(empty.len(), 0);
|
||||
assert!(empty.is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn constant_pool_iter_yields_index_and_entry_pairs() {
|
||||
let pool = ConstantPool::from_entries(vec![
|
||||
CpInfo::Empty,
|
||||
CpInfo::Utf8("A".to_string()),
|
||||
CpInfo::Integer(7),
|
||||
]);
|
||||
let indices: Vec<u16> = pool.iter().map(|(i, _)| i).collect();
|
||||
assert_eq!(indices, vec![0, 1, 2]);
|
||||
// Confirm the entries themselves come through, not an empty iterator.
|
||||
let utf8_at_1 = pool.iter().find(|(i, _)| *i == 1).map(|(_, e)| match e {
|
||||
CpInfo::Utf8(s) => s.as_str(),
|
||||
_ => "?",
|
||||
});
|
||||
assert_eq!(utf8_at_1, Some("A"));
|
||||
}
|
||||
|
||||
fn class_file_with(this_class: u16, super_class: u16, pool: ConstantPool) -> ClassFile {
|
||||
ClassFile {
|
||||
minor_version: 0,
|
||||
major_version: 0,
|
||||
constant_pool: pool,
|
||||
access_flags: 0,
|
||||
this_class,
|
||||
super_class,
|
||||
interfaces: Vec::new(),
|
||||
fields: Vec::new(),
|
||||
methods: Vec::new(),
|
||||
attributes: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn this_class_name_and_super_class_name_resolve_distinct_indices() {
|
||||
let pool = ConstantPool::from_entries(vec![
|
||||
CpInfo::Empty,
|
||||
CpInfo::Utf8("com/example/Foo".to_string()),
|
||||
CpInfo::Utf8("com/example/Bar".to_string()),
|
||||
CpInfo::Class { name_index: 1 },
|
||||
CpInfo::Class { name_index: 2 },
|
||||
]);
|
||||
let cf = class_file_with(3, 4, pool);
|
||||
assert_eq!(cf.this_class_name(), Some("com/example/Foo"));
|
||||
assert_eq!(cf.super_class_name(), Some("com/example/Bar"));
|
||||
|
||||
// this_class index pointing at a non-Class entry must not resolve.
|
||||
let pool2 = ConstantPool::from_entries(vec![
|
||||
CpInfo::Empty,
|
||||
CpInfo::Utf8("not a class ref".to_string()),
|
||||
]);
|
||||
let cf2 = class_file_with(1, 1, pool2);
|
||||
assert_eq!(cf2.this_class_name(), None);
|
||||
assert_eq!(cf2.super_class_name(), None);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn member_descriptor_resolves_the_descriptor_not_the_name() {
|
||||
let pool = ConstantPool::from_entries(vec![
|
||||
CpInfo::Empty,
|
||||
CpInfo::Utf8("doStuff".to_string()), // index 1: name
|
||||
CpInfo::Utf8("()V".to_string()), // index 2: descriptor
|
||||
]);
|
||||
let cf = class_file_with(0, 0, pool);
|
||||
let m = Member {
|
||||
access_flags: 0,
|
||||
name_index: 1,
|
||||
descriptor_index: 2,
|
||||
attributes: Vec::new(),
|
||||
};
|
||||
assert_eq!(cf.member_descriptor(&m), Some("()V"));
|
||||
assert_ne!(cf.member_descriptor(&m), Some("doStuff"));
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// Reader::u16/u32/u64 boundary + value correctness
|
||||
//
|
||||
// Mirrors `slice_boundary_is_inclusive_of_the_final_byte`: an
|
||||
// exact-fit read must succeed, one byte short must fail. Plus
|
||||
// positive-value tests so a scrambled byte assembly (not just an
|
||||
// out-of-bounds read) would be caught.
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn u16_boundary_is_inclusive_of_the_final_byte() {
|
||||
let data = [0xAB, 0xCD];
|
||||
let mut r = Reader::new(&data);
|
||||
assert_eq!(r.u16("exact fit").expect("2 bytes available"), 0xABCD);
|
||||
|
||||
let data = [0xAB];
|
||||
let mut r = Reader::new(&data);
|
||||
assert!(matches!(
|
||||
r.u16("one byte short"),
|
||||
Err(Error::UnexpectedEof { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn u16_decodes_big_endian_value() {
|
||||
let data = [0x01, 0x02];
|
||||
let mut r = Reader::new(&data);
|
||||
assert_eq!(r.u16("value").unwrap(), 0x0102);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn u32_boundary_is_inclusive_of_the_final_byte() {
|
||||
let data = [0x00, 0x00, 0x00, 0x2A];
|
||||
let mut r = Reader::new(&data);
|
||||
assert_eq!(r.u32("exact fit").expect("4 bytes available"), 42);
|
||||
|
||||
let data = [0x00, 0x00, 0x00];
|
||||
let mut r = Reader::new(&data);
|
||||
assert!(matches!(
|
||||
r.u32("one byte short"),
|
||||
Err(Error::UnexpectedEof { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn u32_decodes_big_endian_value() {
|
||||
let data = [0x00, 0x00, 0x05, 0x39]; // 1337
|
||||
let mut r = Reader::new(&data);
|
||||
assert_eq!(r.u32("value").unwrap(), 1337);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn u64_boundary_is_inclusive_of_the_final_byte() {
|
||||
// pos == 0, buffer exactly 8 bytes: must succeed.
|
||||
let data = [0, 0, 0, 0, 0, 0, 0, 0x7B]; // 123
|
||||
let mut r = Reader::new(&data);
|
||||
assert_eq!(r.u64("exact fit").expect("8 bytes available"), 123);
|
||||
|
||||
// pos == 0, buffer one byte short of 8: must fail cleanly, not
|
||||
// panic on the internal self.data[self.pos + 7] index.
|
||||
let data = [0u8; 7];
|
||||
let mut r = Reader::new(&data);
|
||||
assert!(matches!(
|
||||
r.u64("one byte short"),
|
||||
Err(Error::UnexpectedEof { .. })
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn u64_decodes_big_endian_value() {
|
||||
let data = [0, 0, 0, 0, 0, 0, 0x05, 0x39]; // 1337
|
||||
let mut r = Reader::new(&data);
|
||||
assert_eq!(r.u64("value").unwrap(), 1337);
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// decode_modified_utf8: 3-byte (BMP) decode path
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn modified_utf8_three_byte_cjk() {
|
||||
// U+3042 (hiragana あ) in modified UTF-8: 1110xxxx 10xxxxxx 10xxxxxx
|
||||
// = 0xE3 0x81 0x82.
|
||||
let s = decode_modified_utf8(&[0xE3, 0x81, 0x82]).unwrap();
|
||||
assert_eq!(s, "\u{3042}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn modified_utf8_three_byte_rejects_bad_first_continuation() {
|
||||
// Valid lead byte (0xE3), but the first continuation byte is not
|
||||
// 10xxxxxx (0x01 instead) — must be rejected, proving the first
|
||||
// `& 0xC0 != 0x80` check is live.
|
||||
assert!(decode_modified_utf8(&[0xE3, 0x01, 0x82]).is_err());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn modified_utf8_three_byte_rejects_bad_second_continuation() {
|
||||
// Valid lead + first continuation, but the second continuation
|
||||
// byte is not 10xxxxxx — proves the second check is independently
|
||||
// live (not short-circuited by the first).
|
||||
assert!(decode_modified_utf8(&[0xE3, 0x81, 0x01]).is_err());
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// read_constant_pool: Long/Double two-slot skip, real byte parsing
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn constant_pool_long_entry_occupies_two_slots_via_real_parse() {
|
||||
// Real class-file bytes (not the `from_entries` synthetic ctor):
|
||||
// magic + minor/major + cp_count=4 + tag=5 (Long, 8-byte payload
|
||||
// at index 1, reserved slot at index 2) + tag=1 (Utf8 at index 3)
|
||||
// + empty access_flags/this/super/interfaces/fields/methods/attrs.
|
||||
let mut buf = vec![
|
||||
0xCA, 0xFE, 0xBA, 0xBE, // magic
|
||||
0x00, 0x00, // minor
|
||||
0x00, 0x34, // major
|
||||
0x00, 0x04, // cp_count = 4 (0=Empty,1=Long,2=Empty tail,3=Utf8)
|
||||
5, // Long tag
|
||||
];
|
||||
buf.extend_from_slice(&0x1122_3344_5566_7788u64.to_be_bytes()); // 8-byte payload
|
||||
buf.push(1); // Utf8 tag
|
||||
let name = b"marker";
|
||||
buf.extend_from_slice(&(name.len() as u16).to_be_bytes());
|
||||
buf.extend_from_slice(name);
|
||||
// access_flags, this_class, super_class, interfaces_count
|
||||
buf.extend_from_slice(&[0, 0, 0, 0, 0, 0, 0, 0]);
|
||||
// fields_count, methods_count, attributes_count
|
||||
buf.extend_from_slice(&[0, 0, 0, 0, 0, 0]);
|
||||
|
||||
let cf = ClassFile::parse(&buf).expect("well-formed synthetic class file");
|
||||
assert_eq!(cf.constant_pool.len(), 4);
|
||||
// The Long occupies indices 1 AND 2 (its reserved tail slot).
|
||||
// The Utf8 must resolve at index 3 = long_index(1) + 2, NOT +1.
|
||||
assert_eq!(cf.constant_pool.utf8(3), Some("marker"));
|
||||
// Index 2 is the reserved tail slot: not a Utf8, must not
|
||||
// resolve as one (guards against the Utf8 landing one slot early).
|
||||
assert_eq!(cf.constant_pool.utf8(2), None);
|
||||
match cf.constant_pool.get(1) {
|
||||
Some(CpInfo::Long(v)) => assert_eq!(*v, 0x1122_3344_5566_7788u64 as i64),
|
||||
other => panic!("expected Long at index 1, got {:?}", other),
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// instruction_size: tableswitch/lookupswitch with non-degenerate
|
||||
// low/high/npairs (the existing tests only cover low==high==0 and
|
||||
// npairs==0, which can't distinguish `-` from `+` in the entry-count
|
||||
// arithmetic).
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn instruction_size_tableswitch_non_degenerate_range() {
|
||||
// low=1, high=4 -> 4 entries (high-low+1 = 4). A `-`->`+` mutation
|
||||
// on that arithmetic would instead compute high+low+1 = 6.
|
||||
let mut code = vec![TABLESWITCH];
|
||||
code.extend_from_slice(&[0, 0, 0]); // padding
|
||||
code.extend_from_slice(&[0, 0, 0, 0]); // default offset
|
||||
code.extend_from_slice(&1i32.to_be_bytes()); // low = 1
|
||||
code.extend_from_slice(&4i32.to_be_bytes()); // high = 4
|
||||
code.extend_from_slice(&[0; 16]); // 4 jump entries * 4 bytes
|
||||
// total = 1 (opcode) + 3 (pad) + 12 (default/low/high) + 16 (entries) = 32
|
||||
assert_eq!(instruction_size(&code, 0), Some(32));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn instruction_size_lookupswitch_non_degenerate_npairs() {
|
||||
// npairs = 3 -> 3 * 8 = 24 bytes of pairs.
|
||||
let mut code = vec![LOOKUPSWITCH];
|
||||
code.extend_from_slice(&[0, 0, 0]); // padding
|
||||
code.extend_from_slice(&[0, 0, 0, 0]); // default
|
||||
code.extend_from_slice(&3i32.to_be_bytes()); // npairs = 3
|
||||
code.extend_from_slice(&[0; 24]); // 3 pairs
|
||||
// total = 1 + 3 + 8 (default/npairs) + 24 = 36
|
||||
assert_eq!(instruction_size(&code, 0), Some(36));
|
||||
}
|
||||
}
|
||||
|
||||
+208
-36
@@ -36,17 +36,18 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
|
||||
// Stream number mapping from playbackconfig.xml
|
||||
let mut stream_map: HashMap<String, u16> = HashMap::new();
|
||||
if let Some(pc_data) = super::read_jar_file(reader, udf, "playbackconfig.xml") {
|
||||
if let Ok(pc_text) = std::str::from_utf8(&pc_data) {
|
||||
parse_playback_config(pc_text, &mut stream_map);
|
||||
}
|
||||
if let Some(pc_data) = super::read_jar_file(reader, udf, "playbackconfig.xml")
|
||||
&& let Ok(pc_text) = std::str::from_utf8(&pc_data)
|
||||
{
|
||||
parse_playback_config(pc_text, &mut stream_map);
|
||||
}
|
||||
|
||||
let stream_nums = assign_stream_numbers(&stream_infos, &stream_map);
|
||||
let stream_nums = assign_stream_numbers(&stream_infos, &stream_map)?;
|
||||
|
||||
let mut labels = Vec::new();
|
||||
for (info, &stream_num) in stream_infos.iter().zip(stream_nums.iter()) {
|
||||
labels.push(StreamLabel {
|
||||
stream_id: None,
|
||||
stream_number: stream_num,
|
||||
stream_type: info.stream_type,
|
||||
language: info.language.clone(),
|
||||
@@ -76,7 +77,29 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
/// map-assigned one. (Both numbering domains are 1-based per type, and
|
||||
/// `apply_labels` matches on `(type, stream_number)`, so a collision
|
||||
/// would mislabel tracks.)
|
||||
fn assign_stream_numbers(infos: &[StreamInfo], stream_map: &HashMap<String, u16>) -> Vec<u16> {
|
||||
///
|
||||
/// Returns `None` when the 1-based stream-number space is exhausted — every
|
||||
/// number in `1..=u16::MAX` for that type is either already claimed by the map
|
||||
/// or already synthesized. That is unreachable on real media: the BD STN_table
|
||||
/// carries at most 32 primary audio and 32 PG streams per playlist, so the
|
||||
/// 65535-wide space leaves >2000x headroom. It IS reachable from a crafted
|
||||
/// `streamproperties.xml` listing >65535 stream entries, and the only correct
|
||||
/// answers there are "fail the parse" or "emit colliding numbers"; we fail.
|
||||
///
|
||||
/// The skip search is bounded by the numbering space itself: a `u16`
|
||||
/// `saturating_add` here parked the counter at `u16::MAX` forever whenever the
|
||||
/// map also claimed `u16::MAX`, turning an overflow guard into a hang that
|
||||
/// `apply()`'s `catch_unwind` cannot interrupt. The counters are therefore
|
||||
/// widened to `u32` so the skip loop strictly increases toward a fixed ceiling
|
||||
/// (guaranteeing termination) and exhaustion is reported rather than absorbed.
|
||||
fn assign_stream_numbers(
|
||||
infos: &[StreamInfo],
|
||||
stream_map: &HashMap<String, u16>,
|
||||
) -> Option<Vec<u16>> {
|
||||
/// One past the last assignable stream number, as a `u32` so the
|
||||
/// counters can step off the end of the `u16` domain without wrapping.
|
||||
const NUMBER_SPACE_END: u32 = u16::MAX as u32 + 1;
|
||||
|
||||
// Numbers already claimed by the map, per type. A map value of 0 is NOT a
|
||||
// claim: apply_labels binds on 1-based stream numbers, so 0 is unmatchable.
|
||||
// Treat 0 as "unmapped" here (defense in depth — parse_playback_config also
|
||||
@@ -96,8 +119,8 @@ fn assign_stream_numbers(infos: &[StreamInfo], stream_map: &HashMap<String, u16>
|
||||
}
|
||||
}
|
||||
|
||||
let mut audio_idx: u16 = 1;
|
||||
let mut sub_idx: u16 = 1;
|
||||
let mut audio_idx: u32 = 1;
|
||||
let mut sub_idx: u32 = 1;
|
||||
let mut out = Vec::with_capacity(infos.len());
|
||||
for info in infos {
|
||||
let n = match stream_map.get(&info.id).copied() {
|
||||
@@ -107,21 +130,31 @@ fn assign_stream_numbers(infos: &[StreamInfo], stream_map: &HashMap<String, u16>
|
||||
StreamLabelType::Audio => (&mut audio_idx, &taken_audio),
|
||||
StreamLabelType::Subtitle => (&mut sub_idx, &taken_sub),
|
||||
};
|
||||
// Advance past any number already claimed via the map.
|
||||
// saturating: a crafted XML with >65k stream entries must
|
||||
// not overflow (panic in debug, wrap-to-0 in release) on
|
||||
// untrusted disc bytes.
|
||||
while taken.contains(idx) {
|
||||
*idx = idx.saturating_add(1);
|
||||
// Advance past any number already claimed via the map. The
|
||||
// counter strictly increases and NUMBER_SPACE_END is fixed, so
|
||||
// this terminates in at most 65535 steps for any input.
|
||||
while *idx < NUMBER_SPACE_END && taken.contains(&(*idx as u16)) {
|
||||
*idx += 1;
|
||||
}
|
||||
let n = *idx;
|
||||
*idx = idx.saturating_add(1);
|
||||
if *idx >= NUMBER_SPACE_END {
|
||||
// Numbering space exhausted. Emitting anything here would
|
||||
// either wrap to 0 (unmatchable) or duplicate a number
|
||||
// already bound to a different stream, so the parse fails.
|
||||
tracing::warn!(
|
||||
streams = infos.len(),
|
||||
"criterion: 1-based u16 stream-number space exhausted; \
|
||||
refusing to synthesize a colliding stream number"
|
||||
);
|
||||
return None;
|
||||
}
|
||||
let n = *idx as u16;
|
||||
*idx += 1;
|
||||
n
|
||||
}
|
||||
};
|
||||
out.push(n);
|
||||
}
|
||||
out
|
||||
Some(out)
|
||||
}
|
||||
|
||||
struct StreamInfo {
|
||||
@@ -189,14 +222,13 @@ fn parse_playback_config(text: &str, map: &mut HashMap<String, u16>) {
|
||||
if let (Some(stream_id_str), Some(info_id)) = (
|
||||
xml::text(block, "StreamID"),
|
||||
xml::text(block, "StreamInfo_ID"),
|
||||
) {
|
||||
if let Ok(stream_num) = stream_id_str.parse::<u16>() {
|
||||
// Stream numbers are 1-based per the apply_labels
|
||||
// contract; a mapped 0 is unmatchable and silently
|
||||
// drops the label. Skip it rather than store it.
|
||||
if stream_num != 0 {
|
||||
map.insert(info_id, stream_num);
|
||||
}
|
||||
) && let Ok(stream_num) = stream_id_str.parse::<u16>()
|
||||
{
|
||||
// Stream numbers are 1-based per the apply_labels
|
||||
// contract; a mapped 0 is unmatchable and silently
|
||||
// drops the label. Skip it rather than store it.
|
||||
if stream_num != 0 {
|
||||
map.insert(info_id, stream_num);
|
||||
}
|
||||
}
|
||||
from = end;
|
||||
@@ -226,11 +258,89 @@ mod tests {
|
||||
info("a1", StreamLabelType::Audio),
|
||||
info("s0", StreamLabelType::Subtitle),
|
||||
];
|
||||
let nums = assign_stream_numbers(&infos, &HashMap::new());
|
||||
let nums =
|
||||
assign_stream_numbers(&infos, &HashMap::new()).expect("numbering space not exhausted");
|
||||
// Per-type 1-based: audio 1,2 ; subtitle 1.
|
||||
assert_eq!(nums, vec![1, 2, 1]);
|
||||
}
|
||||
|
||||
/// Immunity pin. `parse_stream_infos` emits one `StreamInfo` per
|
||||
/// `*StreamInfos` element unconditionally — no filter, no `continue` — so
|
||||
/// an element whose fields are missing or unrecognized still occupies its
|
||||
/// position, and `assign_stream_numbers` still spends a number on it.
|
||||
///
|
||||
/// That is the property that keeps this parser out of the failure mode
|
||||
/// where a skipped entry pulls every later label one stream forward. It
|
||||
/// is load-bearing for the fallback path specifically: with no
|
||||
/// `playbackconfig.xml` the numbers come purely from position in this
|
||||
/// list, so dropping an element there would shift the rest.
|
||||
///
|
||||
/// Mutation: skip elements with an empty `ID`/`LangInfoID` → the two
|
||||
/// real audio streams renumber to 1 and 2.
|
||||
/// Immunity pin, section-boundary half. Each stream here is one closed XML
|
||||
/// element, and every field is read out of `&text[start..end]` — the range
|
||||
/// `xml::find_element` returned — so one element can never absorb the next
|
||||
/// one's fields, however the document is malformed around it. Contrast the
|
||||
/// flat-string walk in pixelogic, where a section whose end marker is
|
||||
/// missing keeps consuming entries as STN slots.
|
||||
///
|
||||
/// The missing-boundary case fails closed. An element with no close tag of
|
||||
/// its own ends at the NEXT close tag, so it absorbs the element behind it
|
||||
/// — the list comes back SHORTER. It cannot come back longer: nothing
|
||||
/// outside a returned range is ever read as a stream, and `find_element`
|
||||
/// yields `None` rather than a range running to EOF when no close tag
|
||||
/// exists at all. A malformed document can cost this parser a slot; it can
|
||||
/// never invent one.
|
||||
///
|
||||
/// Mutation: read fields from the document rather than the element's
|
||||
/// range, or let a close-less element run to EOF → the trailing elements
|
||||
/// re-enter the list as extra streams.
|
||||
#[test]
|
||||
fn an_unterminated_stream_element_shortens_the_list_it_cannot_extend_it() {
|
||||
let sp = concat!(
|
||||
"<AudioStreamInfos><ID>a0</ID><LangInfoID>ENG</LangInfoID></AudioStreamInfos>",
|
||||
// No `</AudioStreamInfos>` for this one.
|
||||
"<AudioStreamInfos><ID>a1</ID><LangInfoID>FRA</LangInfoID>",
|
||||
"<AudioStreamInfos><ID>a2</ID><LangInfoID>DEU</LangInfoID></AudioStreamInfos>",
|
||||
);
|
||||
let infos = parse_stream_infos(sp);
|
||||
assert_eq!(
|
||||
infos.iter().map(|i| i.id.as_str()).collect::<Vec<_>>(),
|
||||
vec!["a0", "a1"],
|
||||
"the close-less element absorbs the one behind it — two slots, not \
|
||||
three, and never four"
|
||||
);
|
||||
assert_eq!(infos[1].language, "fra", "and keeps its own leading fields");
|
||||
|
||||
// With no close tag anywhere behind it, the element is not returned at
|
||||
// all and the walk ends — the tail of the document never becomes a
|
||||
// stream list.
|
||||
let no_close = "<AudioStreamInfos><ID>a0</ID><LangInfoID>ENG</LangInfoID>";
|
||||
assert!(parse_stream_infos(no_close).is_empty());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn unusable_stream_element_still_occupies_its_position() {
|
||||
let sp = r#"
|
||||
<AudioStreamInfos><ID>a0</ID><LangInfoID>ENG_US</LangInfoID></AudioStreamInfos>
|
||||
<AudioStreamInfos></AudioStreamInfos>
|
||||
<AudioStreamInfos><ID>a2</ID><LangInfoID>FRA</LangInfoID><Content>COMMENTARY</Content></AudioStreamInfos>
|
||||
<SubtitleStreamInfos><ID>s0</ID><LangInfoID></LangInfoID><Qualifier>WAT</Qualifier></SubtitleStreamInfos>
|
||||
<SubtitleStreamInfos><ID>s1</ID><LangInfoID>ENG</LangInfoID><Qualifier>SDH</Qualifier></SubtitleStreamInfos>
|
||||
"#;
|
||||
let infos = parse_stream_infos(sp);
|
||||
assert_eq!(infos.len(), 5, "every element yields a StreamInfo");
|
||||
let nums =
|
||||
assign_stream_numbers(&infos, &HashMap::new()).expect("numbering space not exhausted");
|
||||
assert_eq!(
|
||||
nums,
|
||||
vec![1, 2, 3, 1, 2],
|
||||
"the blank element owns audio slot 2, so the commentary is slot 3"
|
||||
);
|
||||
assert_eq!(infos[2].purpose, LabelPurpose::Commentary);
|
||||
assert_eq!(infos[4].qualifier, LabelQualifier::Sdh);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn fallback_does_not_collide_with_partial_map() {
|
||||
// Map claims audio "a1" -> 1. The unmapped audio "a0" must NOT
|
||||
@@ -242,7 +352,7 @@ mod tests {
|
||||
info("a1", StreamLabelType::Audio), // mapped → 1
|
||||
info("a2", StreamLabelType::Audio), // unmapped → fallback
|
||||
];
|
||||
let nums = assign_stream_numbers(&infos, &map);
|
||||
let nums = assign_stream_numbers(&infos, &map).expect("numbering space not exhausted");
|
||||
// a0 skips the taken 1 → 2; a1 keeps 1; a2 → 3. All distinct.
|
||||
assert_eq!(nums, vec![2, 1, 3]);
|
||||
let mut sorted = nums.clone();
|
||||
@@ -260,7 +370,10 @@ mod tests {
|
||||
info("a0", StreamLabelType::Audio),
|
||||
info("a1", StreamLabelType::Audio),
|
||||
];
|
||||
assert_eq!(assign_stream_numbers(&infos, &map), vec![5, 9]);
|
||||
assert_eq!(
|
||||
assign_stream_numbers(&infos, &map).expect("numbering space not exhausted"),
|
||||
vec![5, 9]
|
||||
);
|
||||
}
|
||||
|
||||
// ── Additional hardening tests ─────────────────────────────────────────
|
||||
@@ -276,7 +389,8 @@ mod tests {
|
||||
info("a1", StreamLabelType::Audio),
|
||||
info("s1", StreamLabelType::Subtitle),
|
||||
];
|
||||
let nums = assign_stream_numbers(&infos, &HashMap::new());
|
||||
let nums =
|
||||
assign_stream_numbers(&infos, &HashMap::new()).expect("numbering space not exhausted");
|
||||
// Audio: 1, 2; Subtitle: 1, 2 — each counter resets at 1 per type.
|
||||
assert_eq!(nums[0], 1); // audio 1
|
||||
assert_eq!(nums[1], 1); // subtitle 1
|
||||
@@ -293,7 +407,7 @@ mod tests {
|
||||
let mut map = HashMap::new();
|
||||
map.insert("a0".to_string(), 0u16); // 0 must not be treated as a claim
|
||||
let infos = vec![info("a0", StreamLabelType::Audio)];
|
||||
let nums = assign_stream_numbers(&infos, &map);
|
||||
let nums = assign_stream_numbers(&infos, &map).expect("numbering space not exhausted");
|
||||
// 0 is treated as unmapped → the fallback counter assigns 1.
|
||||
assert_eq!(nums[0], 1);
|
||||
}
|
||||
@@ -309,7 +423,7 @@ mod tests {
|
||||
info("real", StreamLabelType::Audio),
|
||||
info("bad", StreamLabelType::Audio),
|
||||
];
|
||||
let nums = assign_stream_numbers(&infos, &map);
|
||||
let nums = assign_stream_numbers(&infos, &map).expect("numbering space not exhausted");
|
||||
assert_eq!(nums[0], 1); // the genuinely-mapped stream keeps 1
|
||||
assert_eq!(nums[1], 2); // the 0-stream is synthesized to the next free slot
|
||||
}
|
||||
@@ -326,16 +440,74 @@ mod tests {
|
||||
info("a0", StreamLabelType::Audio), // fallback
|
||||
info("s0", StreamLabelType::Subtitle), // mapped → 2
|
||||
];
|
||||
let nums = assign_stream_numbers(&infos, &map);
|
||||
let nums = assign_stream_numbers(&infos, &map).expect("numbering space not exhausted");
|
||||
// Audio fallback for a0 → 1 (subtitle's taken-2 doesn't block it).
|
||||
assert_eq!(nums[0], 1);
|
||||
assert_eq!(nums[1], 2);
|
||||
}
|
||||
|
||||
/// Spec: saturating_add prevents overflow when many streams are listed.
|
||||
/// Mutation: use wrapping_add → counter wraps to 0 and collides.
|
||||
/// A crafted `streamproperties.xml` can drive the fallback counter to the
|
||||
/// top of the 1-based u16 stream-number space and then present one more
|
||||
/// unmapped stream whose successor number is also claimed by the map.
|
||||
///
|
||||
/// This must TERMINATE. The bound is the numbering space itself, so the
|
||||
/// assertion is on the spec-derived exhaustion behaviour (`None`), not on
|
||||
/// any tunable constant. Run on a worker thread with a deadline so a
|
||||
/// non-terminating loop fails the test in 20 s instead of hanging CI.
|
||||
#[test]
|
||||
fn assign_stream_numbers_saturation_on_overflow() {
|
||||
fn exhausted_numbering_terminates_instead_of_looping() {
|
||||
let (tx, rx) = std::sync::mpsc::channel();
|
||||
let worker = std::thread::spawn(move || {
|
||||
// One mapped audio stream claims the last number in the space.
|
||||
let mut map = HashMap::new();
|
||||
map.insert("claims_max".to_string(), u16::MAX);
|
||||
let mut infos = vec![info("claims_max", StreamLabelType::Audio)];
|
||||
// Enough unmapped audio streams to walk the counter to the top.
|
||||
for i in 0..=(u16::MAX as u32) {
|
||||
infos.push(info(&format!("u{i}"), StreamLabelType::Audio));
|
||||
}
|
||||
let _ = tx.send(assign_stream_numbers(&infos, &map));
|
||||
});
|
||||
match rx.recv_timeout(std::time::Duration::from_secs(20)) {
|
||||
Ok(result) => {
|
||||
worker.join().expect("worker panicked");
|
||||
assert!(
|
||||
result.is_none(),
|
||||
"an exhausted 1-based u16 numbering space must fail the parse, \
|
||||
not emit colliding or wrapped stream numbers"
|
||||
);
|
||||
}
|
||||
Err(_) => panic!(
|
||||
"assign_stream_numbers did not terminate within 20s — \
|
||||
non-terminating skip loop on crafted stream_map"
|
||||
),
|
||||
}
|
||||
}
|
||||
|
||||
/// The whole 1-based u16 space must remain usable: 65535 unmapped audio
|
||||
/// streams get 65535 distinct numbers with no panic and no wrap. The
|
||||
/// literals here are the JVMS-independent, spec-derived size of a u16
|
||||
/// 1-based numbering domain, not a tunable cap.
|
||||
#[test]
|
||||
fn full_u16_numbering_space_is_usable_and_unique() {
|
||||
let infos: Vec<StreamInfo> = (0..65_535u32)
|
||||
.map(|i| info(&format!("a{i}"), StreamLabelType::Audio))
|
||||
.collect();
|
||||
let nums = assign_stream_numbers(&infos, &HashMap::new()).expect("space is not exhausted");
|
||||
assert_eq!(nums.len(), 65_535);
|
||||
assert_eq!(nums[0], 1);
|
||||
assert_eq!(nums[65_534], 65_535);
|
||||
let mut sorted = nums.clone();
|
||||
sorted.sort_unstable();
|
||||
sorted.dedup();
|
||||
assert_eq!(sorted.len(), 65_535, "stream numbers must all be distinct");
|
||||
}
|
||||
|
||||
/// Spec: a partially-mapped playlist with many claimed numbers must still
|
||||
/// synthesize past every claim without panicking or colliding.
|
||||
/// Mutation: drop the skip loop → the fallback reuses a claimed number.
|
||||
#[test]
|
||||
fn fallback_skips_a_dense_block_of_claimed_numbers() {
|
||||
// Force the counter past u16::MAX by pre-taking all values 1..=u16::MAX.
|
||||
// Doing that for real would be slow; instead inject u16::MAX into taken.
|
||||
let mut map = HashMap::new();
|
||||
@@ -362,7 +534,7 @@ mod tests {
|
||||
qualifier: LabelQualifier::None,
|
||||
});
|
||||
// This must not panic.
|
||||
let nums = assign_stream_numbers(&infos, &map);
|
||||
let nums = assign_stream_numbers(&infos, &map).expect("numbering space not exhausted");
|
||||
assert_eq!(nums.len(), 501);
|
||||
// The last (unmapped) entry's number must be > 500 (skipped all taken).
|
||||
assert!(nums[500] > 500);
|
||||
|
||||
+276
-119
@@ -50,10 +50,10 @@ fn merge(ls: Vec<StreamLabel>, mb: Vec<StreamLabel>) -> Vec<StreamLabel> {
|
||||
if let Some(mb_match) = mb
|
||||
.iter()
|
||||
.find(|m| m.stream_type == label.stream_type && m.stream_number == label.stream_number)
|
||||
&& label.name.is_empty()
|
||||
&& !mb_match.name.is_empty()
|
||||
{
|
||||
if label.name.is_empty() && !mb_match.name.is_empty() {
|
||||
label.name = mb_match.name.clone();
|
||||
}
|
||||
label.name = mb_match.name.clone();
|
||||
}
|
||||
}
|
||||
// Append any menu_base-only stream (present in mb but not in ls by
|
||||
@@ -87,7 +87,21 @@ fn prefix_is_commentary(prefix: &str) -> bool {
|
||||
fn parse_language_streams(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<Vec<StreamLabel>> {
|
||||
let data = super::read_jar_file(reader, udf, "language_streams.txt")?;
|
||||
let text = std::str::from_utf8(&data).ok()?;
|
||||
let labels = parse_language_streams_text(text);
|
||||
if labels.is_empty() {
|
||||
return None;
|
||||
}
|
||||
Some(labels)
|
||||
}
|
||||
|
||||
/// Parse the body of a `language_streams.txt` file into stream labels.
|
||||
///
|
||||
/// This is the shipping parser: [`parse_language_streams`] does the UDF read
|
||||
/// and UTF-8 decode and then delegates here. It is split out — rather than
|
||||
/// duplicated under `#[cfg(test)]`, which is what it used to be — so the unit
|
||||
/// tests below exercise production code. A test that re-implements the
|
||||
/// function it guards cannot fail when the real function breaks.
|
||||
fn parse_language_streams_text(text: &str) -> Vec<StreamLabel> {
|
||||
let mut labels = Vec::new();
|
||||
|
||||
for line in text.lines() {
|
||||
@@ -193,122 +207,7 @@ fn parse_language_streams(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<
|
||||
}
|
||||
|
||||
labels.push(StreamLabel {
|
||||
stream_number: stream_num,
|
||||
stream_type,
|
||||
language,
|
||||
name: String::new(),
|
||||
purpose: final_purpose,
|
||||
qualifier,
|
||||
codec_hint,
|
||||
variant: variant_code,
|
||||
});
|
||||
}
|
||||
|
||||
if labels.is_empty() {
|
||||
return None;
|
||||
}
|
||||
Some(labels)
|
||||
}
|
||||
|
||||
/// Parse the body of a `language_streams.txt` file into stream labels. Split
|
||||
/// out from [`parse_language_streams`] so unit tests exercise the real parsing
|
||||
/// logic without needing a SectorSource / UdfFs.
|
||||
#[cfg(test)]
|
||||
fn parse_language_streams_text(text: &str) -> Vec<StreamLabel> {
|
||||
let mut labels = Vec::new();
|
||||
|
||||
for line in text.lines() {
|
||||
let line = line.trim();
|
||||
if line.is_empty() || line.starts_with('#') {
|
||||
continue;
|
||||
}
|
||||
|
||||
let parts: Vec<&str> = line.split(',').map(|s| s.trim()).collect();
|
||||
if parts.len() < 4 {
|
||||
continue;
|
||||
}
|
||||
|
||||
let type_str = parts[1];
|
||||
let stream_num: u16 = match parts[2].parse() {
|
||||
Ok(n) if n > 0 => n,
|
||||
_ => continue,
|
||||
};
|
||||
let language = parts[3].to_string();
|
||||
let variant = if parts.len() > 4 {
|
||||
parts[4].to_string()
|
||||
} else {
|
||||
String::new()
|
||||
};
|
||||
|
||||
let (stream_type, purpose, qualifier) = match type_str {
|
||||
"audio_production" => (
|
||||
StreamLabelType::Audio,
|
||||
LabelPurpose::Normal,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"audio_commentary" => (
|
||||
StreamLabelType::Audio,
|
||||
LabelPurpose::Commentary,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"audio_ime" => (
|
||||
StreamLabelType::Audio,
|
||||
LabelPurpose::Ime,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"subtitle_production" => (
|
||||
StreamLabelType::Subtitle,
|
||||
LabelPurpose::Normal,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"subtitle_commentary" => (
|
||||
StreamLabelType::Subtitle,
|
||||
LabelPurpose::Commentary,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"subtitle_narrative" => (
|
||||
StreamLabelType::Subtitle,
|
||||
LabelPurpose::Normal,
|
||||
LabelQualifier::Forced,
|
||||
),
|
||||
"subtitle_dual" => (
|
||||
StreamLabelType::Subtitle,
|
||||
LabelPurpose::Normal,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"subtitle_bonus" => (
|
||||
StreamLabelType::Subtitle,
|
||||
LabelPurpose::Normal,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"subtitle_ime" => (
|
||||
StreamLabelType::Subtitle,
|
||||
LabelPurpose::Ime,
|
||||
LabelQualifier::None,
|
||||
),
|
||||
"subtitle_ime_narrative" => (
|
||||
StreamLabelType::Subtitle,
|
||||
LabelPurpose::Ime,
|
||||
LabelQualifier::Forced,
|
||||
),
|
||||
_ => continue,
|
||||
};
|
||||
|
||||
let mut codec_hint = String::new();
|
||||
let mut variant_code = String::new();
|
||||
let mut final_purpose = purpose;
|
||||
|
||||
if !variant.is_empty() {
|
||||
match variant.as_str() {
|
||||
"eda" => final_purpose = LabelPurpose::Descriptive,
|
||||
"csp" | "cs" | "lsp" | "ls" | "cf" | "pf" | "bp" | "pp" => {
|
||||
variant_code = variant.clone();
|
||||
}
|
||||
_ => codec_hint = vocab::codec(&variant).to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
labels.push(StreamLabel {
|
||||
stream_id: None,
|
||||
stream_number: stream_num,
|
||||
stream_type,
|
||||
language,
|
||||
@@ -426,6 +325,68 @@ mod tests {
|
||||
assert_eq!(labels[0].qualifier, LabelQualifier::None);
|
||||
}
|
||||
|
||||
/// Spec: `menu_base.prop` lines are skipped when `is_empty() ||
|
||||
/// starts_with('#')` — either alone is sufficient. A commented-out
|
||||
/// key=value line must never be parsed into an entry.
|
||||
/// Mutation: `||` -> `&&` requires both, which a non-empty comment
|
||||
/// line can't satisfy, so it falls through to `line.find('=')` and
|
||||
/// gets parsed as a real property.
|
||||
#[test]
|
||||
fn menu_base_comment_line_with_equals_is_still_skipped() {
|
||||
let labels = parse_props(
|
||||
"#audio_1.class=AudioButton\n\
|
||||
#audio_1.streamNumber=9\n\
|
||||
#audio_1.name=Should Not Appear\n\
|
||||
audio_2.class=AudioButton\n\
|
||||
audio_2.streamNumber=1\n\
|
||||
audio_2.name=Real Track\n",
|
||||
);
|
||||
assert_eq!(labels.len(), 1, "commented-out entry must not be parsed");
|
||||
assert_eq!(labels[0].name, "Real Track");
|
||||
}
|
||||
|
||||
/// Spec: `menu_base.prop` streamNumber (or audioStream/subtitleStream)
|
||||
/// must be strictly positive — `0` means "no STN entry" and must be
|
||||
/// skipped, matching the `n > 0` guard on the language_streams side.
|
||||
/// Mutation: `n > 0` -> `n >= 0` (or the guard deleted) would let a
|
||||
/// stream_num of 0 through, emitting a dead label apply_labels can
|
||||
/// never match (its counter starts at 1).
|
||||
#[test]
|
||||
fn menu_base_zero_stream_number_skipped() {
|
||||
let labels = parse_props(
|
||||
"audio_1.class=AudioButton\n\
|
||||
audio_1.streamNumber=0\n\
|
||||
audio_1.name=Disabled Slot\n",
|
||||
);
|
||||
assert!(
|
||||
labels.is_empty(),
|
||||
"streamNumber=0 must be skipped, got {labels:?}"
|
||||
);
|
||||
}
|
||||
|
||||
/// Spec: `is_subtitle` is `class.contains("SubtitleButton") ||
|
||||
/// prefix.starts_with("subtitle_")` — EITHER signal alone is
|
||||
/// sufficient to classify (and keep) a subtitle entry whose prefix
|
||||
/// doesn't follow the `subtitle_` naming convention.
|
||||
/// Mutation: `||` -> `&&` would require BOTH signals; an entry whose
|
||||
/// class says SubtitleButton but whose prefix is something else
|
||||
/// (e.g. a vendor-specific button id) would then satisfy neither
|
||||
/// `is_audio` nor `is_subtitle` and get dropped entirely.
|
||||
#[test]
|
||||
fn menu_base_subtitle_class_alone_is_sufficient() {
|
||||
let labels = parse_props(
|
||||
"menuBtn7.class=SubtitleButton\n\
|
||||
menuBtn7.streamNumber=1\n\
|
||||
menuBtn7.name=English SDH\n",
|
||||
);
|
||||
assert_eq!(
|
||||
labels.len(),
|
||||
1,
|
||||
"class=SubtitleButton alone must classify as subtitle, not be dropped"
|
||||
);
|
||||
assert_eq!(labels[0].stream_type, StreamLabelType::Subtitle);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn prefix_commentary_segment_match_not_substring() {
|
||||
// Genuine commentary group segments match.
|
||||
@@ -441,6 +402,7 @@ mod tests {
|
||||
|
||||
fn lbl(t: StreamLabelType, n: u16, name: &str) -> StreamLabel {
|
||||
StreamLabel {
|
||||
stream_id: None,
|
||||
stream_number: n,
|
||||
stream_type: t,
|
||||
language: String::new(),
|
||||
@@ -452,6 +414,39 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// Spec: `merge`'s `mb.iter().find(...)` must match an mb entry by
|
||||
/// (stream_type AND stream_number) TOGETHER — either alone is not a
|
||||
/// unique key (there can be an audio #1 and a subtitle #1, or two
|
||||
/// different audio streams).
|
||||
/// Mutation: `&&` -> `||` inside the closure would match on type OR
|
||||
/// number alone, so `.find` (which returns the FIRST match) can pick
|
||||
/// an mb entry with the right type but the WRONG stream number.
|
||||
#[test]
|
||||
fn merge_matches_mb_entry_by_type_and_number_together() {
|
||||
// ls wants audio #2 (empty name, so it will borrow from mb).
|
||||
let ls = vec![lbl(StreamLabelType::Audio, 2, "")];
|
||||
// mb's FIRST audio entry is #1 (wrong number); its #2 entry (the
|
||||
// real match) comes second.
|
||||
let mb = vec![
|
||||
lbl(StreamLabelType::Audio, 1, "Wrong Number Match"),
|
||||
lbl(StreamLabelType::Audio, 2, "Correct Match"),
|
||||
];
|
||||
let merged = merge(ls, mb);
|
||||
assert_eq!(
|
||||
merged.len(),
|
||||
2,
|
||||
"mb's own audio #1 must also survive as its own entry"
|
||||
);
|
||||
let a2 = merged
|
||||
.iter()
|
||||
.find(|l| l.stream_type == StreamLabelType::Audio && l.stream_number == 2)
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
a2.name, "Correct Match",
|
||||
"must match mb by (type AND number), not type or number alone"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn merge_preserves_menu_base_only_streams() {
|
||||
// language_streams covers audio 1; menu_base has audio 1 (name)
|
||||
@@ -521,6 +516,47 @@ mod tests {
|
||||
assert_eq!(labels[0].qualifier, LabelQualifier::Forced);
|
||||
}
|
||||
|
||||
/// Immunity pin against the defect measured in the `paramount` parser,
|
||||
/// where a vendor `forced_sub` cell hung off a FULL dialogue track's own
|
||||
/// slot to say "this track also contains forced signs", and reading that
|
||||
/// cell as "this track is forced" flagged 30 MB dialogue tracks forced.
|
||||
///
|
||||
/// This format cannot express that. The forced signal is not a flag beside
|
||||
/// a track's entry — it IS the entry's stream-kind token, drawn from a
|
||||
/// closed vocabulary in which `subtitle_production` (the full dialogue
|
||||
/// track) and `subtitle_narrative` (the forced-narrative track) are
|
||||
/// mutually exclusive alternatives in the same position. A row is one or
|
||||
/// the other; there is no cell a full track can carry to acquire the
|
||||
/// qualifier, so the paramount failure mode has no encoding here.
|
||||
///
|
||||
/// Mutation: give `subtitle_production` a `Forced` qualifier, or add a
|
||||
/// forced side-flag that both kinds may carry.
|
||||
#[test]
|
||||
fn a_full_subtitle_track_kind_can_never_carry_the_forced_qualifier() {
|
||||
// Every subtitle kind in the vocabulary, one row each.
|
||||
let text = "id1,subtitle_production,1,eng\n\
|
||||
id2,subtitle_commentary,2,eng\n\
|
||||
id3,subtitle_dual,3,eng\n\
|
||||
id4,subtitle_bonus,4,eng\n\
|
||||
id5,subtitle_ime,5,kor\n\
|
||||
id6,subtitle_narrative,6,eng\n\
|
||||
id7,subtitle_ime_narrative,7,kor\n";
|
||||
let labels = parse_language_streams_text(text);
|
||||
let forced: Vec<&str> = labels
|
||||
.iter()
|
||||
.filter(|l| l.qualifier == LabelQualifier::Forced)
|
||||
.map(|l| l.language.as_str())
|
||||
.collect();
|
||||
assert_eq!(
|
||||
forced.len(),
|
||||
2,
|
||||
"only the two narrative kinds are forced, got {forced:?}"
|
||||
);
|
||||
// The full dialogue kind specifically.
|
||||
let production = parse_language_streams_text("id,subtitle_production,1,eng\n");
|
||||
assert_eq!(production[0].qualifier, LabelQualifier::None);
|
||||
}
|
||||
|
||||
/// Spec: `subtitle_commentary` → Subtitle / Commentary.
|
||||
/// Mutation: treat as Normal → subtitle commentary not flagged.
|
||||
#[test]
|
||||
@@ -564,6 +600,72 @@ mod tests {
|
||||
assert!(labels.is_empty());
|
||||
}
|
||||
|
||||
/// Immunity pin. `language_streams.txt` states each stream's number in
|
||||
/// field 3, so a row the parser cannot use is simply dropped — it can
|
||||
/// never renumber the rows behind it. This is the property that keeps
|
||||
/// this parser out of the STN-slot-shifting failure mode that bites
|
||||
/// parsers which count positionally: there, a skipped entry silently
|
||||
/// pulls every later label one stream forward.
|
||||
///
|
||||
/// Mutation: replace `parts[2]` with a running per-type counter → the
|
||||
/// three unusable rows here collapse the survivors onto 1/2 and 1.
|
||||
#[test]
|
||||
fn ls_stream_numbers_come_from_the_row_not_a_counter() {
|
||||
let labels = parse_language_streams_text(
|
||||
"id,audio_production,4,eng\n\
|
||||
id,audio_bonus_extended,5,eng\n\
|
||||
id,audio_production,0,fra\n\
|
||||
id,audio_production,7,fra\n\
|
||||
id,subtitle_production\n\
|
||||
id,subtitle_narrative,9,deu\n",
|
||||
);
|
||||
let nums: Vec<(StreamLabelType, u16)> = labels
|
||||
.iter()
|
||||
.map(|l| (l.stream_type, l.stream_number))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
nums,
|
||||
vec![
|
||||
(StreamLabelType::Audio, 4),
|
||||
(StreamLabelType::Audio, 7),
|
||||
(StreamLabelType::Subtitle, 9),
|
||||
],
|
||||
"an unusable row drops out without shifting the numbering"
|
||||
);
|
||||
assert_eq!(labels[2].qualifier, LabelQualifier::Forced);
|
||||
}
|
||||
|
||||
/// Immunity pin, `menu_base.prop` side: the number comes from the
|
||||
/// entry's own `streamNumber` property, so a skipped entry (commented
|
||||
/// out, `streamNumber=0`, neither audio nor subtitle) leaves the
|
||||
/// surviving entries on their authored slots.
|
||||
///
|
||||
/// Mutation: number by iteration order → the survivors collapse to 1/2.
|
||||
#[test]
|
||||
fn menu_base_stream_numbers_come_from_the_entry_not_a_counter() {
|
||||
let labels = parse_props(
|
||||
"#audio_0.class=AudioButton\n\
|
||||
#audio_0.streamNumber=1\n\
|
||||
audio_1.class=AudioButton\n\
|
||||
audio_1.streamNumber=0\n\
|
||||
audio_2.class=AudioButton\n\
|
||||
audio_2.streamNumber=6\n\
|
||||
other_1.class=SomeOtherButton\n\
|
||||
other_1.streamNumber=2\n\
|
||||
subtitle_1.class=SubtitleButton\n\
|
||||
subtitle_1.streamNumber=11\n",
|
||||
);
|
||||
let nums: Vec<(StreamLabelType, u16)> = labels
|
||||
.iter()
|
||||
.map(|l| (l.stream_type, l.stream_number))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
nums,
|
||||
vec![(StreamLabelType::Audio, 6), (StreamLabelType::Subtitle, 11),],
|
||||
"skipped entries must not renumber the ones that survive"
|
||||
);
|
||||
}
|
||||
|
||||
/// Spec: `eda` variant → `Descriptive` purpose.
|
||||
/// Mutation: miss the `eda` branch → purpose stays Normal.
|
||||
#[test]
|
||||
@@ -608,6 +710,60 @@ mod tests {
|
||||
assert_eq!(labels[0].language, "eng");
|
||||
}
|
||||
|
||||
/// Spec: the skip test is `is_empty() || starts_with('#')` — EITHER
|
||||
/// condition alone must skip the line. A commented-out line that
|
||||
/// happens to look like valid CSV (a real authoring pattern for
|
||||
/// disabling a stream entry) must never produce a label.
|
||||
/// Mutation: `||` -> `&&` requires BOTH conditions, which a non-empty
|
||||
/// comment line can never satisfy, so it would fall through to the
|
||||
/// CSV parser and (since it has >= 4 comma fields) emit a spurious
|
||||
/// label instead of being skipped.
|
||||
#[test]
|
||||
fn ls_comment_line_with_csv_shape_is_still_skipped() {
|
||||
let labels =
|
||||
parse_language_streams_text("#id,audio_production,1,eng\nid2,audio_production,2,fra\n");
|
||||
assert_eq!(
|
||||
labels.len(),
|
||||
1,
|
||||
"the commented-out CSV-shaped line must not parse"
|
||||
);
|
||||
assert_eq!(labels[0].language, "fra");
|
||||
}
|
||||
|
||||
/// Spec: `subtitle_dual` is a recognized subtitle type (Normal/no
|
||||
/// qualifier). Mutation: delete this match arm → falls to the
|
||||
/// catch-all `_ => continue`, silently dropping the stream.
|
||||
#[test]
|
||||
fn ls_subtitle_dual_parsed() {
|
||||
let labels = parse_language_streams_text("id,subtitle_dual,1,eng\n");
|
||||
assert_eq!(labels.len(), 1, "subtitle_dual must produce a label");
|
||||
assert_eq!(labels[0].stream_type, StreamLabelType::Subtitle);
|
||||
assert_eq!(labels[0].purpose, LabelPurpose::Normal);
|
||||
assert_eq!(labels[0].qualifier, LabelQualifier::None);
|
||||
}
|
||||
|
||||
/// Spec: `subtitle_bonus` is a recognized subtitle type (Normal/no
|
||||
/// qualifier). Mutation: delete this match arm → dropped as unknown.
|
||||
#[test]
|
||||
fn ls_subtitle_bonus_parsed() {
|
||||
let labels = parse_language_streams_text("id,subtitle_bonus,2,eng\n");
|
||||
assert_eq!(labels.len(), 1, "subtitle_bonus must produce a label");
|
||||
assert_eq!(labels[0].stream_type, StreamLabelType::Subtitle);
|
||||
assert_eq!(labels[0].purpose, LabelPurpose::Normal);
|
||||
}
|
||||
|
||||
/// Spec: `subtitle_ime` maps to Subtitle/Ime (no Forced qualifier,
|
||||
/// unlike `subtitle_ime_narrative`).
|
||||
/// Mutation: delete this match arm → dropped as unknown.
|
||||
#[test]
|
||||
fn ls_subtitle_ime_parsed() {
|
||||
let labels = parse_language_streams_text("id,subtitle_ime,3,jpn\n");
|
||||
assert_eq!(labels.len(), 1, "subtitle_ime must produce a label");
|
||||
assert_eq!(labels[0].stream_type, StreamLabelType::Subtitle);
|
||||
assert_eq!(labels[0].purpose, LabelPurpose::Ime);
|
||||
assert_eq!(labels[0].qualifier, LabelQualifier::None);
|
||||
}
|
||||
|
||||
/// Spec: multiple valid lines produce multiple labels.
|
||||
/// Mutation: stop after first label → only 1 label returned.
|
||||
#[test]
|
||||
@@ -756,6 +912,7 @@ fn parse_menu_base_text(text: &str) -> Vec<StreamLabel> {
|
||||
.unwrap_or_default();
|
||||
|
||||
labels.push(StreamLabel {
|
||||
stream_id: None,
|
||||
stream_number: stream_num,
|
||||
stream_type,
|
||||
language,
|
||||
|
||||
+265
-8
@@ -93,6 +93,45 @@ fn scan_jar(archive: &mut jar::Jar) -> Vec<StreamLabel> {
|
||||
out
|
||||
}
|
||||
|
||||
/// Cap on the bytes retained for one stream label.
|
||||
///
|
||||
/// The label is an owned copy of a slice of a `CONSTANT_Utf8_info` entry,
|
||||
/// whose `length` field is a `u16` (JVMS §4.4.7) — so a single crafted
|
||||
/// constant contributes up to 65535 bytes, and the `u16` stream-number
|
||||
/// keyspace admits 65536 of them per type.
|
||||
///
|
||||
/// Headroom: real dbp menu labels are short display names — "English Dolby
|
||||
/// Atmos" (19 bytes), "Spanish 5.1 Dolby Digital" (25). The longest plausible
|
||||
/// retail string ("Portuguese (Brazilian) 5.1 Dolby Digital Plus") is 45
|
||||
/// bytes. 256 leaves >5x headroom over that, and any string past it is menu
|
||||
/// geometry or padding, never a language name — `vocab::lang` would not
|
||||
/// resolve it anyway.
|
||||
const MAX_LABEL_BYTES: usize = 256;
|
||||
|
||||
/// Cap on retained stream slots per type.
|
||||
///
|
||||
/// The keys come from `parse::<u16>()` on disc bytes, so all 65536 slots per
|
||||
/// type are reachable; paired with [`MAX_LABEL_BYTES`] this bounds the whole
|
||||
/// scan at 2 x 512 x 256 bytes.
|
||||
///
|
||||
/// Headroom: the BD STN_table admits at most 32 primary audio and 32 PG
|
||||
/// streams per playlist, and dbp emits one menu TextField per stream. 512
|
||||
/// leaves 16x headroom over the spec maximum.
|
||||
const MAX_LABELS_PER_TYPE: usize = 512;
|
||||
|
||||
/// Record `label` for stream `n`, honouring the retention caps. Existing
|
||||
/// slots are still overwritten at the cap so the documented last-write-wins
|
||||
/// behaviour is preserved; only NEW slots are refused.
|
||||
fn retain_label(map: &mut BTreeMap<u16, String>, n: u16, label: &str) {
|
||||
if label.len() > MAX_LABEL_BYTES {
|
||||
return;
|
||||
}
|
||||
if map.len() >= MAX_LABELS_PER_TYPE && !map.contains_key(&n) {
|
||||
return;
|
||||
}
|
||||
map.insert(n, label.to_string());
|
||||
}
|
||||
|
||||
fn collect_textfield(
|
||||
s: &str,
|
||||
audios: &mut BTreeMap<u16, String>,
|
||||
@@ -112,15 +151,15 @@ fn collect_textfield(
|
||||
}
|
||||
if let Some(rest) = kind_n.strip_prefix("Audio") {
|
||||
if let Ok(n) = rest.parse::<u16>() {
|
||||
audios.insert(n, label.to_string());
|
||||
retain_label(audios, n, label);
|
||||
}
|
||||
} else if let Some(rest) = kind_n.strip_prefix("Subtitle") {
|
||||
if let Ok(n) = rest.parse::<u16>() {
|
||||
// Subtitle0 is conventionally the "None / Off" disable
|
||||
// button, not an actual subtitle stream.
|
||||
if n > 0 {
|
||||
subs.insert(n, label.to_string());
|
||||
}
|
||||
} else if let Some(rest) = kind_n.strip_prefix("Subtitle")
|
||||
&& let Ok(n) = rest.parse::<u16>()
|
||||
{
|
||||
// Subtitle0 is conventionally the "None / Off" disable
|
||||
// button, not an actual subtitle stream.
|
||||
if n > 0 {
|
||||
retain_label(subs, n, label);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -132,6 +171,7 @@ fn make_label(num: u16, label: String, stream_type: StreamLabelType) -> StreamLa
|
||||
let qualifier = vocab::qualifier(&label);
|
||||
let purpose = vocab::purpose(&label);
|
||||
StreamLabel {
|
||||
stream_id: None,
|
||||
stream_number: num,
|
||||
stream_type,
|
||||
language,
|
||||
@@ -147,6 +187,223 @@ fn make_label(num: u16, label: String, stream_type: StreamLabelType) -> StreamLa
|
||||
mod tests {
|
||||
use super::super::{LabelPurpose, LabelQualifier};
|
||||
use super::*;
|
||||
use std::io::{Cursor, Write as _};
|
||||
|
||||
/// Build a minimal, structurally valid `.class` file (JVMS §4.1) whose
|
||||
/// constant pool holds exactly the given `Utf8` strings (indices 1..=N,
|
||||
/// no long/double slot padding needed for plain strings). No fields,
|
||||
/// methods, interfaces, or attributes — `scan_jar`'s only interest is
|
||||
/// the constant pool.
|
||||
fn build_class(utf8_entries: &[&str]) -> Vec<u8> {
|
||||
let mut out = Vec::new();
|
||||
out.extend_from_slice(&0xCAFEBABEu32.to_be_bytes()); // magic
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // minor_version
|
||||
out.extend_from_slice(&52u16.to_be_bytes()); // major_version (Java 8)
|
||||
out.extend_from_slice(&((utf8_entries.len() + 1) as u16).to_be_bytes()); // cp_count
|
||||
for s in utf8_entries {
|
||||
out.push(1); // CONSTANT_Utf8 tag
|
||||
out.extend_from_slice(&(s.len() as u16).to_be_bytes());
|
||||
out.extend_from_slice(s.as_bytes());
|
||||
}
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // access_flags
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // this_class
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // super_class
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // interfaces_count
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // fields_count
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // methods_count
|
||||
out.extend_from_slice(&0u16.to_be_bytes()); // attributes_count
|
||||
out
|
||||
}
|
||||
|
||||
/// Zip `entries` (name -> bytes) into an in-memory, Stored (uncompressed)
|
||||
/// `jar::Jar` via the `zip` crate's own writer — a real archive, not a
|
||||
/// hand-rolled central directory.
|
||||
fn build_jar(entries: &[(&str, Vec<u8>)]) -> jar::Jar {
|
||||
let mut buf = Vec::new();
|
||||
{
|
||||
let mut writer = zip::ZipWriter::new(Cursor::new(&mut buf));
|
||||
let opts = zip::write::SimpleFileOptions::default()
|
||||
.compression_method(zip::CompressionMethod::Stored);
|
||||
for (name, data) in entries {
|
||||
writer.start_file(*name, opts).expect("start_file");
|
||||
writer.write_all(&data[..]).expect("write class bytes");
|
||||
}
|
||||
writer.finish().expect("finish zip");
|
||||
}
|
||||
zip::ZipArchive::new(Cursor::new(buf)).expect("valid zip")
|
||||
}
|
||||
|
||||
/// `scan_jar` wires together `for_each_class`, constant-pool iteration,
|
||||
/// `collect_textfield`, and `make_label` into the actual per-jar scan
|
||||
/// used by `parse`. The pure `collect_textfield`/`make_label` unit
|
||||
/// tests above don't exercise this wiring at all.
|
||||
///
|
||||
/// Mutation: replace the whole function body with `vec![]` — every
|
||||
/// dbp disc would silently lose all its stream labels regardless of
|
||||
/// what's in the jar.
|
||||
#[test]
|
||||
fn scan_jar_extracts_labels_from_real_class_entries() {
|
||||
let class_bytes = build_class(&[
|
||||
"com/dbp/Whatever", // unrelated string — must be ignored
|
||||
"LTextField,Audio1,English Dolby Atmos,Fontstrip_Composite,296,763",
|
||||
"HTextField,Subtitle1,English SDH,Fontstrip_Composite,1312,763",
|
||||
"ATextField,Subtitle0,None,Fontstrip_Composite,1312,843", // disable button, skipped
|
||||
]);
|
||||
let mut archive = build_jar(&[("com/dbp/Menu.class", class_bytes)]);
|
||||
|
||||
let labels = scan_jar(&mut archive);
|
||||
|
||||
assert_eq!(
|
||||
labels.len(),
|
||||
2,
|
||||
"expected one audio + one real subtitle label"
|
||||
);
|
||||
let audio = labels
|
||||
.iter()
|
||||
.find(|l| l.stream_type == StreamLabelType::Audio)
|
||||
.expect("audio label present");
|
||||
assert_eq!(audio.stream_number, 1);
|
||||
assert_eq!(audio.language, "eng");
|
||||
|
||||
let sub = labels
|
||||
.iter()
|
||||
.find(|l| l.stream_type == StreamLabelType::Subtitle)
|
||||
.expect("subtitle label present");
|
||||
assert_eq!(sub.stream_number, 1);
|
||||
assert_eq!(sub.qualifier, LabelQualifier::Sdh);
|
||||
}
|
||||
|
||||
/// Immunity pin. Every dbp label states its own slot in the `AudioN` /
|
||||
/// `SubtitleN` token, so the numbering survives gaps and skipped entries
|
||||
/// intact. Nothing here counts positionally, which is what keeps this
|
||||
/// parser out of the failure mode where a skipped entry pulls every later
|
||||
/// label one stream forward.
|
||||
///
|
||||
/// Mutation: number by iteration order → `Audio4` becomes 2 and
|
||||
/// `Subtitle3` becomes 1, silently rebinding both to other streams.
|
||||
#[test]
|
||||
fn stream_numbers_come_from_the_token_not_iteration_order() {
|
||||
let class_bytes = build_class(&[
|
||||
"LTextField,Audio1,English Dolby Atmos,Fontstrip_Composite,296,763",
|
||||
// Slots 2 and 3 have no menu TextField authored.
|
||||
"LTextField,Audio4,French 5.1 Dolby Digital,Fontstrip_Composite,296,803",
|
||||
// Not a stream: the disable-subtitles button.
|
||||
"ATextField,Subtitle0,None,Fontstrip_Composite,1312,843",
|
||||
// Unparseable slot token — dropped, and must shift nothing.
|
||||
"HTextField,SubtitleX,German,Fontstrip_Composite,1312,883",
|
||||
"HTextField,Subtitle3,English SDH,Fontstrip_Composite,1312,763",
|
||||
]);
|
||||
let mut archive = build_jar(&[("com/dbp/Menu.class", class_bytes)]);
|
||||
|
||||
let labels = scan_jar(&mut archive);
|
||||
let nums: Vec<(StreamLabelType, u16)> = labels
|
||||
.iter()
|
||||
.map(|l| (l.stream_type, l.stream_number))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
nums,
|
||||
vec![
|
||||
(StreamLabelType::Audio, 1),
|
||||
(StreamLabelType::Audio, 4),
|
||||
(StreamLabelType::Subtitle, 3),
|
||||
],
|
||||
"unlabelled and unusable slots leave the authored numbers alone"
|
||||
);
|
||||
}
|
||||
|
||||
/// A `CONSTANT_Utf8_info` carries a `u16` length (JVMS §4.4.7), so one
|
||||
/// crafted constant contributes up to 65535 bytes and the `u16` stream
|
||||
/// keyspace admits 65536 slots per type — ~4 GiB of retained `String` per
|
||||
/// map from a jar that is orders of magnitude smaller.
|
||||
///
|
||||
/// Boundary literals, not the constant: a 256-byte label is kept, 257 and
|
||||
/// the JVMS maximum 65535 are refused.
|
||||
#[test]
|
||||
fn oversized_labels_are_not_retained() {
|
||||
let mut audios = BTreeMap::new();
|
||||
let mut subs = BTreeMap::new();
|
||||
|
||||
collect_textfield(
|
||||
&format!("XTextField,Audio1,{},rest", "A".repeat(256)),
|
||||
&mut audios,
|
||||
&mut subs,
|
||||
);
|
||||
assert_eq!(
|
||||
audios.get(&1).map(String::len),
|
||||
Some(256),
|
||||
"a 256-byte label must still be retained"
|
||||
);
|
||||
|
||||
collect_textfield(
|
||||
&format!("XTextField,Audio2,{},rest", "A".repeat(257)),
|
||||
&mut audios,
|
||||
&mut subs,
|
||||
);
|
||||
assert!(!audios.contains_key(&2), "a 257-byte label must be refused");
|
||||
|
||||
collect_textfield(
|
||||
&format!("XTextField,Subtitle1,{},rest", "B".repeat(65_535)),
|
||||
&mut audios,
|
||||
&mut subs,
|
||||
);
|
||||
assert!(
|
||||
!subs.contains_key(&1),
|
||||
"a JVMS-maximum 65535-byte Utf8 label must be refused"
|
||||
);
|
||||
}
|
||||
|
||||
/// The stream-slot keyspace is the full `u16` on both maps. Offer 600
|
||||
/// distinct audio slots; exactly 512 are retained.
|
||||
#[test]
|
||||
fn retained_stream_slots_are_capped_per_type() {
|
||||
let mut audios = BTreeMap::new();
|
||||
let mut subs = BTreeMap::new();
|
||||
for n in 1..=600u16 {
|
||||
collect_textfield(
|
||||
&format!("XTextField,Audio{n},English,rest"),
|
||||
&mut audios,
|
||||
&mut subs,
|
||||
);
|
||||
}
|
||||
assert_eq!(
|
||||
audios.len(),
|
||||
512,
|
||||
"600 audio slots offered, {} retained — the slot count is unbounded",
|
||||
audios.len()
|
||||
);
|
||||
}
|
||||
|
||||
/// Reaching the slot cap must not break the documented last-write-wins
|
||||
/// behaviour for slots already held.
|
||||
#[test]
|
||||
fn existing_slot_is_still_overwritten_at_the_cap() {
|
||||
let mut audios = BTreeMap::new();
|
||||
let mut subs = BTreeMap::new();
|
||||
for n in 1..=600u16 {
|
||||
collect_textfield(
|
||||
&format!("XTextField,Audio{n},English,rest"),
|
||||
&mut audios,
|
||||
&mut subs,
|
||||
);
|
||||
}
|
||||
collect_textfield("XTextField,Audio1,Spanish,rest", &mut audios, &mut subs);
|
||||
assert_eq!(audios.get(&1).map(String::as_str), Some("Spanish"));
|
||||
}
|
||||
|
||||
/// Headroom: the longest plausible retail label must survive untouched.
|
||||
#[test]
|
||||
fn longest_realistic_label_survives_the_cap() {
|
||||
let mut audios = BTreeMap::new();
|
||||
let mut subs = BTreeMap::new();
|
||||
let real = "Portuguese (Brazilian) 5.1 Dolby Digital Plus";
|
||||
assert_eq!(real.len(), 45, "fixture length changed");
|
||||
collect_textfield(
|
||||
&format!("XTextField,Audio1,{real},Fontstrip_Composite,296,763"),
|
||||
&mut audios,
|
||||
&mut subs,
|
||||
);
|
||||
assert_eq!(audios.get(&1).map(String::as_str), Some(real));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn collect_extracts_audio_and_subtitle_indices() {
|
||||
|
||||
+1632
-39
File diff suppressed because it is too large
Load Diff
@@ -216,6 +216,27 @@ mod tests {
|
||||
ZipArchive::new(Cursor::new(bytes)).expect("valid zip")
|
||||
}
|
||||
|
||||
/// The doc comment states the cap is 64 MiB. Pin the exact numeric
|
||||
/// value (not derived from the same `64 * 1024 * 1024` expression
|
||||
/// under test — a hardcoded literal) so a mutation of the arithmetic
|
||||
/// (e.g. `*` -> `+`) is caught even though no test builds an actual
|
||||
/// 64 MiB buffer.
|
||||
#[test]
|
||||
fn max_class_bytes_is_64_mebibytes() {
|
||||
assert_eq!(MAX_CLASS_BYTES, 67_108_864);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn has_path_prefix_matches_only_declared_prefix() {
|
||||
let jar = open(build_stored_zip(
|
||||
"com/dbp/Loader.class",
|
||||
MINIMAL_CLASS,
|
||||
MINIMAL_CLASS.len() as u32,
|
||||
));
|
||||
assert!(has_path_prefix(&jar, "com/dbp/"));
|
||||
assert!(!has_path_prefix(&jar, "com/bydeluxe/"));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn try_each_class_reads_minimal_class() {
|
||||
let mut jar = open(build_stored_zip(
|
||||
|
||||
+1928
-224
File diff suppressed because it is too large
Load Diff
+245
-119
@@ -61,22 +61,7 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
return None;
|
||||
}
|
||||
|
||||
let mut labels: Vec<StreamLabel> = Vec::new();
|
||||
// (stream_type_tag, language, codec_hint, pid) — PID is the
|
||||
// canonical "same physical stream" key; type+lang+codec round
|
||||
// out the rare case where two distinct logical streams happen
|
||||
// to share a PID across playlists with different metadata.
|
||||
let mut seen: Vec<(StreamLabelType, String, String, u16)> = Vec::new();
|
||||
|
||||
// Global 1-based counters keyed by StreamLabelType. Incremented
|
||||
// only when an entry survives dedup, so stream_numbers are dense
|
||||
// (1, 2, 3, ...) per type across the whole disc — not reset per
|
||||
// playlist. A disc with 2 MPLS files that each list the same
|
||||
// 8 audio streams ends up with audio_1..audio_8, not audio_1..
|
||||
// audio_16 or audio_1..audio_8 with audio_1 duplicated.
|
||||
let mut audio_idx: u16 = 0;
|
||||
let mut sub_idx: u16 = 0;
|
||||
|
||||
let mut playlists: Vec<crate::mpls::Playlist> = Vec::new();
|
||||
for name in &mpls_names {
|
||||
let path = format!("/BDMV/PLAYLIST/{}", name);
|
||||
let Ok(data) = udf.read_file(reader, &path) else {
|
||||
@@ -85,28 +70,66 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
let Ok(playlist) = crate::mpls::parse(&data) else {
|
||||
continue;
|
||||
};
|
||||
playlists.push(playlist);
|
||||
}
|
||||
|
||||
let labels = build_labels(&playlists);
|
||||
if labels.is_empty() {
|
||||
return None;
|
||||
}
|
||||
|
||||
// MPLS gives language + codec but never editorial info (no
|
||||
// commentary/SDH/director's cut). Low confidence means framework
|
||||
// parsers (paramount, criterion, pixelogic, ctrm, dbp, deluxe) always
|
||||
// win when they match. MPLS only gets chosen as the parser when
|
||||
// nothing else fired — exactly the universal-fallback role we want.
|
||||
Some(ParseResult::low(labels))
|
||||
}
|
||||
|
||||
/// Convert every stream entry across `playlists` into one [`StreamLabel`] per
|
||||
/// physical stream. Factored out of [`parse`] so unit tests can drive the
|
||||
/// actual conversion logic (stream-type mapping, identity, slot numbering)
|
||||
/// directly from already-parsed [`crate::mpls::Playlist`] values, without
|
||||
/// needing a synthetic on-disc UDF image.
|
||||
///
|
||||
/// Identity is `(clip, PID)` — what the STN entry states — and it is both the
|
||||
/// dedup key and the label's [`StreamId`]. A stream twenty playlists list is
|
||||
/// one label; two clips that both open their first audio at 0x1100 are two.
|
||||
/// This replaced a disc-global dense counter that numbered surviving entries
|
||||
/// 1, 2, 3, … in playlist-directory order: that number was not an STN slot in
|
||||
/// anything, but it was handed to a binder that reads `stream_number` as one.
|
||||
fn build_labels(playlists: &[crate::mpls::Playlist]) -> Vec<StreamLabel> {
|
||||
use std::collections::HashSet;
|
||||
let mut labels: Vec<StreamLabel> = Vec::new();
|
||||
let mut seen: HashSet<super::StreamId> = HashSet::new();
|
||||
|
||||
for playlist in playlists {
|
||||
// `Playlist::streams` is the FIRST play item's STN table, so every
|
||||
// entry here is a stream of that play item's clip — the same clip
|
||||
// `disc::bluray` records as the title's `clips[0]`. That pairing is
|
||||
// what makes the PID an identity rather than a 16-bit number.
|
||||
//
|
||||
// Streams cannot be non-empty without a play item to have read them
|
||||
// from, so the empty case is unreachable on a real disc; entries we
|
||||
// cannot identify are skipped rather than emitted as unbindable
|
||||
// labels.
|
||||
let Some(clip_id) = playlist.play_items.first().map(|pi| pi.clip_id.clone()) else {
|
||||
continue;
|
||||
};
|
||||
|
||||
// 1-based STN slot within THIS playlist's table, per type — the
|
||||
// `stream_number` field's documented meaning, counted the same way
|
||||
// `disc::bluray` counts the stream list it builds from these entries.
|
||||
// Nothing binds through it (these labels bind by id); it is stated
|
||||
// truthfully rather than invented so that a reader of the label list
|
||||
// sees where on its own playlist each stream sits.
|
||||
let mut audio_idx: u16 = 0;
|
||||
let mut sub_idx: u16 = 0;
|
||||
|
||||
for entry in &playlist.streams {
|
||||
let label_type = match entry.stream_type {
|
||||
2 | 5 => StreamLabelType::Audio, // primary + secondary audio
|
||||
3 => StreamLabelType::Subtitle, // PG subtitle
|
||||
// 1 = primary video, 6 = secondary video, 7 = DV EL
|
||||
// → no StreamLabelType variant for video, skip.
|
||||
// 4 = IG (interactive graphics) — not a user-facing
|
||||
// stream, skip.
|
||||
_ => continue,
|
||||
};
|
||||
|
||||
let language = normalize_language(&entry.language);
|
||||
let name = language_display_name(&language);
|
||||
let codec_hint = build_codec_hint(label_type, entry);
|
||||
|
||||
let key = (label_type, language.clone(), codec_hint.clone(), entry.pid);
|
||||
if seen.contains(&key) {
|
||||
let Some(label_type) = label_type_for(entry) else {
|
||||
continue;
|
||||
}
|
||||
seen.push(key);
|
||||
|
||||
};
|
||||
let stream_number = match label_type {
|
||||
StreamLabelType::Audio => {
|
||||
audio_idx += 1;
|
||||
@@ -118,7 +141,20 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
}
|
||||
};
|
||||
|
||||
let stream_id = super::StreamId {
|
||||
clip_id: clip_id.clone(),
|
||||
pid: entry.pid,
|
||||
};
|
||||
if !seen.insert(stream_id.clone()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
let language = normalize_language(&entry.language);
|
||||
let name = language_display_name(&language);
|
||||
let codec_hint = build_codec_hint(label_type, entry);
|
||||
|
||||
labels.push(StreamLabel {
|
||||
stream_id: Some(stream_id),
|
||||
stream_number,
|
||||
stream_type: label_type,
|
||||
language,
|
||||
@@ -130,17 +166,38 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
});
|
||||
}
|
||||
}
|
||||
labels
|
||||
}
|
||||
|
||||
if labels.is_empty() {
|
||||
/// Which per-type numbering list an STN entry belongs to, or `None` when it
|
||||
/// is not a labellable stream at all.
|
||||
///
|
||||
/// This MUST agree with the stream list `disc::bluray` builds from the same
|
||||
/// entries, because that list is what `labels::apply_labels` counts against
|
||||
/// when it binds `stream_number`. The two counters run over the same STN
|
||||
/// entries in the same order, so any entry one side keeps and the other drops
|
||||
/// — or files under a different type — shifts every later label of that type
|
||||
/// onto the wrong stream. Three rules, all mirroring `disc::bluray`:
|
||||
///
|
||||
/// * `coding_type == 0` is the STN table's empty/padding slot. Not a
|
||||
/// stream on either side.
|
||||
/// * a PG coding_type in an audio STN slot is a subtitle, not audio.
|
||||
/// `mpls::parse_stream_entry` has a dedicated arm for this layout, so it
|
||||
/// is an authored shape rather than a corruption.
|
||||
/// * video (1 / 6 / 7 = primary, secondary, Dolby Vision EL) and IG (4)
|
||||
/// have no `StreamLabelType`; they are numbered in their own STN lists
|
||||
/// and never interleave with the audio or PG lists.
|
||||
fn label_type_for(entry: &crate::mpls::StreamEntry) -> Option<StreamLabelType> {
|
||||
use crate::consts::coding_type as c;
|
||||
if entry.coding_type == 0 {
|
||||
return None;
|
||||
}
|
||||
|
||||
// MPLS gives language + codec but never editorial info (no
|
||||
// commentary/SDH/director's cut). Low confidence means framework
|
||||
// parsers (paramount, criterion, pixelogic, ctrm, dbp, deluxe) always
|
||||
// win when they match. MPLS only gets chosen as the parser when
|
||||
// nothing else fired — exactly the universal-fallback role we want.
|
||||
Some(ParseResult::low(labels))
|
||||
match entry.stream_type {
|
||||
2 | 5 if entry.coding_type == c::PG => Some(StreamLabelType::Subtitle),
|
||||
2 | 5 => Some(StreamLabelType::Audio),
|
||||
3 => Some(StreamLabelType::Subtitle),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
fn has_mpls_extension(name: &str) -> bool {
|
||||
@@ -327,69 +384,37 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// A playlist over clip "00001". `Playlist::streams` is read out of the
|
||||
/// first play item's STN table, so a playlist that has streams always has
|
||||
/// a play item to have read them from — the fixture carries one so tests
|
||||
/// exercise the shape production sees, and so each label gets the
|
||||
/// `(clip, PID)` identity it is bound by.
|
||||
fn playlist_with(streams: Vec<StreamEntry>) -> Playlist {
|
||||
playlist_on("00001", streams)
|
||||
}
|
||||
|
||||
fn playlist_on(clip_id: &str, streams: Vec<StreamEntry>) -> Playlist {
|
||||
Playlist {
|
||||
version: "0200".to_string(),
|
||||
play_items: Vec::new(),
|
||||
play_items: vec![crate::mpls::PlayItem {
|
||||
clip_id: clip_id.to_string(),
|
||||
in_time: 0,
|
||||
out_time: 0,
|
||||
connection_condition: 1,
|
||||
}],
|
||||
streams,
|
||||
marks: Vec::new(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Drive the same conversion logic that `parse()` runs on real
|
||||
/// disc data, but starting from already-parsed Playlists so we
|
||||
/// don't have to synthesize valid MPLS bytes.
|
||||
/// Drive the actual production conversion logic (`build_labels`, the
|
||||
/// function `parse()` calls) starting from already-parsed Playlists,
|
||||
/// so tests don't have to synthesize valid on-disc MPLS/UDF bytes.
|
||||
/// This calls the *real* code under test rather than a hand-written
|
||||
/// re-implementation, so mutations inside `build_labels` (stream-type
|
||||
/// mapping, dedup key, counters) are actually caught here.
|
||||
fn labels_from_playlists(playlists: &[Playlist]) -> Vec<StreamLabel> {
|
||||
let mut labels: Vec<StreamLabel> = Vec::new();
|
||||
let mut seen: Vec<(StreamLabelType, String, String, u16)> = Vec::new();
|
||||
|
||||
// Global counters hoisted OUT of the playlist loop to match
|
||||
// production `parse()` (lines 77-78): stream_numbers are dense
|
||||
// per type across the whole disc, not reset per playlist.
|
||||
let mut audio_idx: u16 = 0;
|
||||
let mut sub_idx: u16 = 0;
|
||||
|
||||
for playlist in playlists {
|
||||
for entry in &playlist.streams {
|
||||
let label_type = match entry.stream_type {
|
||||
2 | 5 => StreamLabelType::Audio,
|
||||
3 => StreamLabelType::Subtitle,
|
||||
_ => continue,
|
||||
};
|
||||
// Dedup BEFORE consuming a counter value, matching prod
|
||||
// parse() ordering so a deduped duplicate does not burn a
|
||||
// stream number.
|
||||
let language = normalize_language(&entry.language);
|
||||
let name = language_display_name(&language);
|
||||
let codec_hint = build_codec_hint(label_type, entry);
|
||||
let key = (label_type, language.clone(), codec_hint.clone(), entry.pid);
|
||||
if seen.contains(&key) {
|
||||
continue;
|
||||
}
|
||||
seen.push(key);
|
||||
let stream_number = match label_type {
|
||||
StreamLabelType::Audio => {
|
||||
audio_idx += 1;
|
||||
audio_idx
|
||||
}
|
||||
StreamLabelType::Subtitle => {
|
||||
sub_idx += 1;
|
||||
sub_idx
|
||||
}
|
||||
};
|
||||
labels.push(StreamLabel {
|
||||
stream_number,
|
||||
stream_type: label_type,
|
||||
language,
|
||||
name,
|
||||
purpose: LabelPurpose::Normal,
|
||||
qualifier: LabelQualifier::None,
|
||||
codec_hint,
|
||||
variant: String::new(),
|
||||
});
|
||||
}
|
||||
}
|
||||
labels
|
||||
build_labels(playlists)
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -445,31 +470,114 @@ mod tests {
|
||||
assert_eq!(labels[2].language, "fra");
|
||||
}
|
||||
|
||||
/// `stream_number` is bound by `labels::apply_labels` against the title's
|
||||
/// own stream list, which `disc::bluray` builds from these same STN
|
||||
/// entries. That builder DROPS an entry whose `coding_type` is 0 — the
|
||||
/// STN table's empty/padding slot — so it must not be counted here
|
||||
/// either. Counting it advances the audio counter past a stream that
|
||||
/// never materializes, and every label behind it binds one stream late.
|
||||
#[test]
|
||||
fn dedup_streams_across_playlists() {
|
||||
// Two playlists, same English TrueHD 7.1 PID 0x1100 in both.
|
||||
// Expect one Audio label, not two.
|
||||
let pl1 = playlist_with(vec![
|
||||
fn padding_stn_entry_does_not_consume_a_label_slot() {
|
||||
let pl = playlist_with(vec![
|
||||
audio_entry(0x1100, 0x83, 12, 1, "eng"),
|
||||
audio_entry(0x1101, 0x81, 6, 1, "fra"),
|
||||
// coding_type 0: STN padding. Not a stream.
|
||||
audio_entry(0x1101, 0x00, 0, 0, ""),
|
||||
audio_entry(0x1102, 0x81, 6, 1, "fra"),
|
||||
]);
|
||||
let pl2 = playlist_with(vec![
|
||||
audio_entry(0x1100, 0x83, 12, 1, "eng"), // duplicate
|
||||
audio_entry(0x1102, 0x82, 6, 1, "deu"), // new
|
||||
]);
|
||||
let labels = labels_from_playlists(&[pl1, pl2]);
|
||||
// Expected: eng@0x1100, fra@0x1101, deu@0x1102 — three uniques.
|
||||
assert_eq!(labels.len(), 3);
|
||||
// PID isn't stored on StreamLabel, so assert on the surviving
|
||||
// language set instead.
|
||||
let mut langs: Vec<String> = labels.iter().map(|l| l.language.clone()).collect();
|
||||
langs.sort();
|
||||
assert_eq!(langs, vec!["deu", "eng", "fra"]);
|
||||
let labels = labels_from_playlists(&[pl]);
|
||||
assert_eq!(labels.len(), 2, "the padding slot yields no label");
|
||||
assert_eq!(labels[0].language, "eng");
|
||||
assert_eq!(labels[0].stream_number, 1);
|
||||
assert_eq!(labels[1].language, "fra");
|
||||
assert_eq!(
|
||||
labels[1].stream_number, 2,
|
||||
"padding is absent from the title's stream list, so `fra` is \
|
||||
audio stream 2"
|
||||
);
|
||||
}
|
||||
|
||||
// Stream numbers must be DENSE and GLOBAL across playlists, not
|
||||
// reset per playlist. eng (pl1) = 1, fra (pl1) = 2, the duplicate
|
||||
// eng in pl2 is deduped (no number consumed), and deu (pl2) = 3.
|
||||
// Regression guard for the per-playlist counter-reset divergence.
|
||||
/// A PG coding_type sitting in an audio STN slot is a real, documented
|
||||
/// shape — `mpls::parse_stream_entry` has an explicit arm for it, and
|
||||
/// `disc::bluray` builds it as a Subtitle stream, not an Audio one. This
|
||||
/// module must classify it the same way, or the audio counter runs one
|
||||
/// ahead and the subtitle counter one behind for every later stream.
|
||||
#[test]
|
||||
fn pg_coding_type_in_an_audio_slot_counts_as_a_subtitle() {
|
||||
let mut misplaced = audio_entry(0x1200, 0x90, 0, 0, "spa");
|
||||
misplaced.stream_type = 2;
|
||||
let pl = playlist_with(vec![
|
||||
audio_entry(0x1100, 0x83, 12, 1, "eng"),
|
||||
misplaced,
|
||||
audio_entry(0x1101, 0x81, 6, 1, "fra"),
|
||||
pg_entry(0x1201, "deu"),
|
||||
]);
|
||||
let labels = labels_from_playlists(&[pl]);
|
||||
|
||||
let audio: Vec<_> = labels
|
||||
.iter()
|
||||
.filter(|l| l.stream_type == StreamLabelType::Audio)
|
||||
.map(|l| (l.language.as_str(), l.stream_number))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
audio,
|
||||
vec![("eng", 1), ("fra", 2)],
|
||||
"the PG entry is not an audio stream and must not number one"
|
||||
);
|
||||
|
||||
let sub: Vec<_> = labels
|
||||
.iter()
|
||||
.filter(|l| l.stream_type == StreamLabelType::Subtitle)
|
||||
.map(|l| (l.language.as_str(), l.stream_number))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
sub,
|
||||
vec![("spa", 1), ("deu", 2)],
|
||||
"it is subtitle stream 1, ahead of the PG-slot entry"
|
||||
);
|
||||
}
|
||||
|
||||
/// Two playlists over the SAME clip that both list PID 0x1100: one
|
||||
/// physical stream, so one label. Identity is `(clip, PID)`, and each
|
||||
/// label states the STN slot it holds in its own playlist.
|
||||
#[test]
|
||||
fn one_label_per_stream_across_playlists_on_the_same_clip() {
|
||||
let pl1 = playlist_on(
|
||||
"00001",
|
||||
vec![
|
||||
audio_entry(0x1100, 0x83, 12, 1, "eng"),
|
||||
audio_entry(0x1101, 0x81, 6, 1, "fra"),
|
||||
],
|
||||
);
|
||||
let pl2 = playlist_on(
|
||||
"00001",
|
||||
vec![
|
||||
audio_entry(0x1100, 0x83, 12, 1, "eng"), // same stream
|
||||
audio_entry(0x1102, 0x82, 6, 1, "deu"), // new
|
||||
],
|
||||
);
|
||||
let labels = labels_from_playlists(&[pl1, pl2]);
|
||||
assert_eq!(
|
||||
labels.len(),
|
||||
3,
|
||||
"eng/fra/deu — the duplicate eng is one stream"
|
||||
);
|
||||
|
||||
let id = |lang: &str| {
|
||||
labels
|
||||
.iter()
|
||||
.find(|l| l.language == lang)
|
||||
.and_then(|l| l.stream_id.clone())
|
||||
.map(|i| (i.clip_id, i.pid))
|
||||
};
|
||||
assert_eq!(id("eng"), Some(("00001".into(), 0x1100)));
|
||||
assert_eq!(id("fra"), Some(("00001".into(), 0x1101)));
|
||||
assert_eq!(id("deu"), Some(("00001".into(), 0x1102)));
|
||||
|
||||
// `stream_number` is the entry's slot in ITS OWN playlist's STN table
|
||||
// — deu is pl2's second audio, so 2, not "the third distinct stream
|
||||
// seen while scanning the disc". It used to be the latter: a dense
|
||||
// disc-global counter that named no table anyone could count against,
|
||||
// handed to a binder that reads the field as an STN slot.
|
||||
let num = |lang: &str| {
|
||||
labels
|
||||
.iter()
|
||||
@@ -478,7 +586,25 @@ mod tests {
|
||||
};
|
||||
assert_eq!(num("eng"), Some(1));
|
||||
assert_eq!(num("fra"), Some(2));
|
||||
assert_eq!(num("deu"), Some(3));
|
||||
assert_eq!(num("deu"), Some(2), "pl2's second audio slot");
|
||||
}
|
||||
|
||||
/// The same PID in two DIFFERENT clips is two different streams — a PID is
|
||||
/// only unique within one clip. Deduping on the PID alone (as the old
|
||||
/// key's `(type, language, codec_hint, pid)` did across clips) collapses
|
||||
/// them into one label, and the second clip's stream is then described by
|
||||
/// the first clip's.
|
||||
#[test]
|
||||
fn same_pid_in_two_clips_is_two_streams() {
|
||||
let pl1 = playlist_on("00001", vec![audio_entry(0x1100, 0x83, 12, 1, "eng")]);
|
||||
let pl2 = playlist_on("00002", vec![audio_entry(0x1100, 0x83, 12, 1, "eng")]);
|
||||
let labels = labels_from_playlists(&[pl1, pl2]);
|
||||
assert_eq!(labels.len(), 2, "different clips: two distinct streams");
|
||||
let clips: Vec<String> = labels
|
||||
.iter()
|
||||
.filter_map(|l| l.stream_id.as_ref().map(|i| i.clip_id.clone()))
|
||||
.collect();
|
||||
assert_eq!(clips, vec!["00001", "00002"]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
+563
-59
@@ -3,18 +3,28 @@
|
||||
//! Richest structured format. Complete language lists with forced flags
|
||||
//! and commentary indices per playlist, all in XML attributes.
|
||||
//!
|
||||
//! NOT A SPECIFICATION. `/BDMV/JAR/` is application-defined space, so this
|
||||
//! file is one authoring house's internal metadata that happens to ship on
|
||||
//! the pressing. There is nothing to look up: every field meaning here was
|
||||
//! derived by measuring real discs and cross-checking against per-display-set
|
||||
//! content. Treat an unfamiliar value as unknown rather than guessing — the
|
||||
//! disc's own `forced_on_flag` is the only authoritative forced signal.
|
||||
//!
|
||||
//! ```xml
|
||||
//! <playlist name="Feature" id="00222"
|
||||
//! aud="eng,deu,spa,spa,fra"
|
||||
//! sub="eng,eng,zho,ces,dan"
|
||||
//! forced_sub="0,0,0,1,0"
|
||||
//! forced_sub="0,0,0,1,3"
|
||||
//! aud_com1_idx="10"
|
||||
//! sub_com1_idx="23,24,25" />
|
||||
//! ```
|
||||
//!
|
||||
//! `forced_sub` is an ENUMERATION, not a boolean — see [`ForcedSub`].
|
||||
|
||||
use super::{LabelPurpose, LabelQualifier, ParseResult, StreamLabel, StreamLabelType, xml};
|
||||
use crate::sector::SectorSource;
|
||||
use crate::udf::UdfFs;
|
||||
use std::collections::HashSet;
|
||||
|
||||
pub fn detect(_reader: &mut dyn SectorSource, udf: &UdfFs) -> bool {
|
||||
super::jar_file_exists(udf, "playlists.xml")
|
||||
@@ -32,11 +42,138 @@ pub fn parse(reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
if labels.is_empty() {
|
||||
return None;
|
||||
}
|
||||
// High confidence: paramount's playlists.xml is fully structured
|
||||
// and we extract every documented field.
|
||||
// High confidence: this format is fully structured and we extract
|
||||
// every field whose meaning the corpus establishes. "Documented" would
|
||||
// be the wrong word — see the module note; nothing about it is.
|
||||
Some(ParseResult::high(labels))
|
||||
}
|
||||
|
||||
/// One cell of the `forced_sub` CSV.
|
||||
///
|
||||
/// The attribute reads like a boolean and was parsed as one (`cell == "1"` →
|
||||
/// forced). It is not. Every image in the corpus carrying this vendor's
|
||||
/// `playlists.xml` — seven distinct discs — uses four values, and decoding
|
||||
/// three of those discs' feature subtitle tracks and counting every PGS
|
||||
/// display set separates them into two populations two orders of magnitude
|
||||
/// apart:
|
||||
///
|
||||
/// * `0` — a subtitle track with no forced-narrative content. On the two
|
||||
/// discs measured that use the flag at all, not one `0` track carried a
|
||||
/// single `forced_on_flag` display set.
|
||||
/// * `1` — a FULL DIALOGUE track that additionally contains some
|
||||
/// forced-narrative signs. On one measured disc, all nine `1` cells are
|
||||
/// full tracks of 949-1411 display sets, eight of them carrying 5-14
|
||||
/// flagged sets and the ninth none; that disc has no dedicated forced
|
||||
/// track at all. On another, all seven `1` cells are full tracks of
|
||||
/// 1602-1651 display sets carrying 0-31 flagged sets. Reading `1` as
|
||||
/// forced is what made one language present as two identical full
|
||||
/// subtitle tracks with one of them flagged forced.
|
||||
/// * `2` and `3` — a DEDICATED forced-narrative track. These take their own
|
||||
/// trailing STN slots, one per localized language, duplicating a language
|
||||
/// that already holds a full track earlier in the list. Measured: the two
|
||||
/// `2` slots on one disc are 15 and 10 display sets, EVERY one flagged
|
||||
/// forced, against ~1600 on that disc's full tracks; the four `3` slots on
|
||||
/// another are 7, 14, 23 and 59 display sets against 1216-2655. What
|
||||
/// distinguishes `2` from `3` the corpus does not reveal — both sit in the
|
||||
/// same trailing position, both measure the same shape, and one disc uses
|
||||
/// each for a different language — so both map alike.
|
||||
///
|
||||
/// So the old reading was wrong in BOTH directions: it flagged full dialogue
|
||||
/// tracks forced, and it discarded the cells that name the real forced tracks.
|
||||
///
|
||||
/// The `1` case is deliberately NOT carried through as a weaker "contains
|
||||
/// forced segments" hint. There is no qualifier for that, and the asymmetry
|
||||
/// argues against inventing one here: a wrong forced flag on a 30 MB dialogue
|
||||
/// track is the user-visible defect, while a missing hint costs nothing.
|
||||
///
|
||||
/// An unrecognised cell maps to [`ForcedSub::None`] — the conservative
|
||||
/// direction, since asserting forced is the expensive mistake.
|
||||
#[derive(Clone, Copy, PartialEq, Eq, Debug)]
|
||||
enum ForcedSub {
|
||||
/// No forced-narrative content, or an unrecognised cell.
|
||||
None,
|
||||
/// A full dialogue track that also carries forced-narrative segments.
|
||||
ContainsForcedSegments,
|
||||
/// A dedicated forced-narrative track.
|
||||
ForcedNarrative,
|
||||
}
|
||||
|
||||
/// The one number behind every cap in this parser: the highest CSV cell
|
||||
/// position that can ever be addressed.
|
||||
///
|
||||
/// The labelling loops number cells 1-based into a `u16` and `break` at
|
||||
/// `u16::try_from(i + 1)`, so cell `MAX_COM_INDICES` and everything past it is
|
||||
/// never visited. Two different things are measured against that, and they are
|
||||
/// not the same bound:
|
||||
///
|
||||
/// - **A VALUE at or beyond it cannot match any cell.** This is what caps the
|
||||
/// set: values are filtered before insertion, so at most `MAX_COM_INDICES`
|
||||
/// distinct entries can ever be stored, however long the attribute is. The
|
||||
/// `HashSet` that replaced a linear scan fixed the LOOKUP cost; this is what
|
||||
/// fixes the ALLOCATION, and a disc declaring half a billion indices no
|
||||
/// longer costs half a billion entries.
|
||||
/// - **A POSITION at or beyond it describes nothing new.** This caps the WORK,
|
||||
/// not the memory — the value filter already made the set small, but without
|
||||
/// it every one of those half a billion cells is still split and parsed. It
|
||||
/// is an early exit at the first position whose contents provably cannot
|
||||
/// matter, and it is why `forced_sub` — which holds no set at all, and so
|
||||
/// gets no protection from the value rule — is bounded too.
|
||||
///
|
||||
/// Real authoring is nowhere near either limit: the BD STN table admits at
|
||||
/// most 32 streams per playlist, so nothing legitimate is lost.
|
||||
const MAX_COM_INDICES: usize = u16::MAX as usize;
|
||||
|
||||
/// Parse a `*_com1_idx` attribute into the set the labelling loops query.
|
||||
///
|
||||
/// Extracted so the BOUND is observable. Asserting it through
|
||||
/// `labels_from_feature` is not possible: a `HashSet` collapses repeated
|
||||
/// values, and an out-of-range index changes no label either way, so such a
|
||||
/// test passes whether or not the cap exists — an assertion that cannot fail.
|
||||
/// Returning the set lets a test hand in tens of thousands of DISTINCT
|
||||
/// unaddressable indices and see them refused.
|
||||
fn com_indices(attr: Option<String>) -> HashSet<usize> {
|
||||
attr.map(|s| {
|
||||
s.split(',')
|
||||
.take(MAX_COM_INDICES)
|
||||
.filter_map(|i| i.trim().parse().ok())
|
||||
.filter(|&i| i < MAX_COM_INDICES)
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
/// Parse the `forced_sub` attribute into the cell list the subtitle loop
|
||||
/// queries — the third attacker-controlled CSV in this file, and the last one
|
||||
/// that was still unbounded.
|
||||
///
|
||||
/// Bounded by POSITION, and it has no other choice: a `*_com1_idx` list holds
|
||||
/// values that can be filtered, and that filter is what caps its set, but a
|
||||
/// `forced_sub` cell is a classification of the position it sits at, so there
|
||||
/// is nothing to filter and nothing else would ever cap this. The Vec is read
|
||||
/// only as `forced.get(i)` from a loop that stops at `MAX_COM_INDICES`, so
|
||||
/// every cell past that is unreachable by construction.
|
||||
///
|
||||
/// Extracted, like [`com_indices`], so the bound is OBSERVABLE. Through
|
||||
/// `labels_from_feature` it is not: the subtitle loop cannot reach those cells
|
||||
/// either, so a label-level assertion passes whether or not the cap exists.
|
||||
fn forced_subs(attr: Option<String>) -> Vec<ForcedSub> {
|
||||
attr.map(|s| {
|
||||
s.split(',')
|
||||
.take(MAX_COM_INDICES)
|
||||
.map(forced_sub_cell)
|
||||
.collect()
|
||||
})
|
||||
.unwrap_or_default()
|
||||
}
|
||||
|
||||
fn forced_sub_cell(cell: &str) -> ForcedSub {
|
||||
match cell.trim() {
|
||||
"1" => ForcedSub::ContainsForcedSegments,
|
||||
"2" | "3" => ForcedSub::ForcedNarrative,
|
||||
_ => ForcedSub::None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Build the stream labels from a single `<playlist .../>` feature
|
||||
/// element. Split out from `parse` so the per-type numbering and
|
||||
/// commentary/forced-index logic is unit-testable without a
|
||||
@@ -49,17 +186,33 @@ fn labels_from_feature(feature: &str) -> Vec<StreamLabel> {
|
||||
// aud_com1_idx is a trimmed, comma-separated list of CSV positions
|
||||
// (some authoring tools emit whitespace, and multiple commentary
|
||||
// tracks are possible) — symmetric with sub_com1_idx below.
|
||||
let com_indices: Vec<usize> = xml::attr(feature, "aud_com1_idx")
|
||||
.map(|s| s.split(',').filter_map(|i| i.trim().parse().ok()).collect())
|
||||
.unwrap_or_default();
|
||||
// A HashSet, not a Vec: `com_indices` is parsed straight out of an
|
||||
// attacker-controlled attribute with no length bound and was scanned
|
||||
// linearly once per stream, so `aud="..."` and `aud_com1_idx="..."`
|
||||
// both grown large make this quadratic in the size of one XML file.
|
||||
// Membership is the only operation performed on it.
|
||||
let com_indices = com_indices(xml::attr(feature, "aud_com1_idx"));
|
||||
|
||||
// stream_number must match apply_labels' monotonic 1-based
|
||||
// per-type counter, which increments once per *real* stream — so
|
||||
// it counts only non-empty slots, not the raw CSV index. The
|
||||
// commentary index comparison stays on the raw CSV index `i`,
|
||||
// since aud_com1_idx is positional against the original CSV.
|
||||
let mut audio_num: u16 = 0;
|
||||
// The CSV *is* the STN list: one cell per stream, in stream order,
|
||||
// and `aud_com1_idx` is a 0-based index into those same cells. So
|
||||
// `stream_number` is the cell's own 1-based position — NOT a counter
|
||||
// that only advances on cells carrying a language.
|
||||
//
|
||||
// A cell with an empty language still occupies its STN slot; it just
|
||||
// has nothing to label. Renumbering the surviving cells 1..N shifts
|
||||
// every label behind an empty cell one slot forward, which is how a
|
||||
// marker authored for one stream ends up written onto the stream in
|
||||
// front of it (see the subtitle side, where the marker is `forced`).
|
||||
//
|
||||
// `u16::try_from` rather than `saturating_add`: past the 1-based u16
|
||||
// numbering space every cell would collapse onto `u16::MAX`, binding
|
||||
// several streams to one label. Stop emitting instead. Unreachable on
|
||||
// real media — the BD STN_table admits at most 32 primary audio
|
||||
// streams per playlist.
|
||||
for (i, lang) in aud.split(',').enumerate() {
|
||||
let Ok(stream_number) = u16::try_from(i + 1) else {
|
||||
break;
|
||||
};
|
||||
let lang = lang.trim();
|
||||
if lang.is_empty() {
|
||||
continue;
|
||||
@@ -69,9 +222,9 @@ fn labels_from_feature(feature: &str) -> Vec<StreamLabel> {
|
||||
} else {
|
||||
LabelPurpose::Normal
|
||||
};
|
||||
audio_num = audio_num.saturating_add(1);
|
||||
labels.push(StreamLabel {
|
||||
stream_number: audio_num,
|
||||
stream_id: None,
|
||||
stream_number,
|
||||
stream_type: StreamLabelType::Audio,
|
||||
language: lang.to_string(),
|
||||
name: String::new(),
|
||||
@@ -85,18 +238,21 @@ fn labels_from_feature(feature: &str) -> Vec<StreamLabel> {
|
||||
|
||||
// Parse subtitle streams
|
||||
if let Some(sub) = xml::attr(feature, "sub") {
|
||||
let forced: Vec<bool> = xml::attr(feature, "forced_sub")
|
||||
.map(|s| s.split(',').map(|f| f.trim() == "1").collect())
|
||||
.unwrap_or_default();
|
||||
let forced = forced_subs(xml::attr(feature, "forced_sub"));
|
||||
|
||||
let com_indices: Vec<usize> = xml::attr(feature, "sub_com1_idx")
|
||||
.map(|s| s.split(',').filter_map(|i| i.trim().parse().ok()).collect())
|
||||
.unwrap_or_default();
|
||||
// HashSet for the same reason as the audio side above: unbounded
|
||||
// parsed input, membership-only use, linear scan once per stream.
|
||||
let com_indices = com_indices(xml::attr(feature, "sub_com1_idx"));
|
||||
|
||||
// As with audio: count only non-empty slots for stream_number,
|
||||
// but keep com/forced lookups on the raw CSV index `i`.
|
||||
let mut sub_num: u16 = 0;
|
||||
// As with audio: the cell position IS the STN slot. `forced_sub` and
|
||||
// `sub_com1_idx` are indexed against those same cells, so an empty
|
||||
// cell must not renumber the cells behind it — a forced marker
|
||||
// authored for one PG slot would otherwise be written onto an
|
||||
// earlier, full-dialogue subtitle track.
|
||||
for (i, lang) in sub.split(',').enumerate() {
|
||||
let Ok(stream_number) = u16::try_from(i + 1) else {
|
||||
break;
|
||||
};
|
||||
let lang = lang.trim();
|
||||
if lang.is_empty() {
|
||||
continue;
|
||||
@@ -108,15 +264,17 @@ fn labels_from_feature(feature: &str) -> Vec<StreamLabel> {
|
||||
LabelPurpose::Normal
|
||||
};
|
||||
|
||||
let qualifier = if forced.get(i).copied().unwrap_or(false) {
|
||||
LabelQualifier::Forced
|
||||
} else {
|
||||
LabelQualifier::None
|
||||
// Only a DEDICATED forced-narrative slot earns the forced flag.
|
||||
// A cell marking a full track as merely containing forced segments
|
||||
// is dropped, not weakened into a forced label (see [`ForcedSub`]).
|
||||
let qualifier = match forced.get(i).copied().unwrap_or(ForcedSub::None) {
|
||||
ForcedSub::ForcedNarrative => LabelQualifier::Forced,
|
||||
ForcedSub::ContainsForcedSegments | ForcedSub::None => LabelQualifier::None,
|
||||
};
|
||||
|
||||
sub_num = sub_num.saturating_add(1);
|
||||
labels.push(StreamLabel {
|
||||
stream_number: sub_num,
|
||||
stream_id: None,
|
||||
stream_number,
|
||||
stream_type: StreamLabelType::Subtitle,
|
||||
language: lang.to_string(),
|
||||
name: String::new(),
|
||||
@@ -142,10 +300,10 @@ fn find_feature_playlist(text: &str) -> Option<String> {
|
||||
let element = &text[start..end];
|
||||
|
||||
// Prefer name="Feature" explicitly.
|
||||
if let Some(name) = xml::attr(element, "name") {
|
||||
if name.eq_ignore_ascii_case("Feature") {
|
||||
return Some(element.to_string());
|
||||
}
|
||||
if let Some(name) = xml::attr(element, "name")
|
||||
&& name.eq_ignore_ascii_case("Feature")
|
||||
{
|
||||
return Some(element.to_string());
|
||||
}
|
||||
|
||||
// Otherwise pick the one with the most audio streams. Count only
|
||||
@@ -168,6 +326,206 @@ fn find_feature_playlist(text: &str) -> Option<String> {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// Immunity pin, section-boundary half. The pixelogic parser walks a flat
|
||||
/// string sequence and recognises its feature section's END by marker
|
||||
/// alone, so a section with no marker behind it runs off into whatever
|
||||
/// follows and counts it as more STN slots. Nothing here can do that: the
|
||||
/// stream list is one attribute of one XML element, so its length is the
|
||||
/// CSV's own cell count and its scope is the element's byte range that
|
||||
/// `xml::find_element` returns. Text after the element — including the
|
||||
/// next playlist's own `aud` — is not reachable from it.
|
||||
///
|
||||
/// And when the boundary is MISSING the failure is closed, not open:
|
||||
/// `xml::find_element` needs a matching close tag and yields `None`
|
||||
/// without one, so an unterminated element ends the walk rather than
|
||||
/// swallowing the rest of the document.
|
||||
///
|
||||
/// Mutation: hand `labels_from_feature` the document instead of the
|
||||
/// element, or let an unterminated element run to EOF → the bonus
|
||||
/// playlist's languages join the feature's stream list.
|
||||
#[test]
|
||||
fn a_playlists_stream_list_cannot_run_into_the_next_playlist() {
|
||||
let doc = r#"
|
||||
<playlist name="Feature" aud="eng,fra" sub="eng,spa" forced_sub="0,1"/>
|
||||
<playlist name="Bonus" aud="deu,ita,jpn" sub="deu,ita,jpn"/>
|
||||
"#;
|
||||
let feature = find_feature_playlist(doc).expect("feature playlist found");
|
||||
let labels = labels_from_feature(&feature);
|
||||
let got: Vec<(StreamLabelType, u16, &str)> = labels
|
||||
.iter()
|
||||
.map(|l| (l.stream_type, l.stream_number, l.language.as_str()))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
got,
|
||||
vec![
|
||||
(StreamLabelType::Audio, 1, "eng"),
|
||||
(StreamLabelType::Audio, 2, "fra"),
|
||||
(StreamLabelType::Subtitle, 1, "eng"),
|
||||
(StreamLabelType::Subtitle, 2, "spa"),
|
||||
],
|
||||
"the CSV's own cells are the whole stream list"
|
||||
);
|
||||
|
||||
// Same document with the feature element left unterminated.
|
||||
let unterminated = r#"
|
||||
<playlist name="Feature" aud="eng,fra">
|
||||
<playlist name="Bonus" aud="deu,ita,jpn"/>
|
||||
"#;
|
||||
assert!(
|
||||
find_feature_playlist(unterminated).is_none(),
|
||||
"a missing element boundary truncates the walk, never extends it"
|
||||
);
|
||||
}
|
||||
|
||||
/// `sub_com1_idx` is parsed straight out of the disc's `playlists.xml`,
|
||||
/// which is attacker-controlled and has no length bound of its own.
|
||||
///
|
||||
/// This replaces a WALL-CLOCK test. That one built a 200 000 x 1 000 001
|
||||
/// fixture and failed if it took over 10 s, to prove the membership test
|
||||
/// was a set rather than a linear scan. Measured on the machine that
|
||||
/// wrote this: 1.62 s alone, and OVER 10 s — a real failure — when the
|
||||
/// suite's other 3 347 tests were running concurrently. A 6x margin
|
||||
/// against a shared CPU is not a margin; it is a CI failure that looks
|
||||
/// like a flake and gets re-run until it passes.
|
||||
///
|
||||
/// It also measured the wrong thing. Making the lookup O(1) bounded the
|
||||
/// QUERY, not the PARSE: the set was still built from every entry the
|
||||
/// disc declared, so a hostile playlist could still force an unbounded
|
||||
/// allocation before any lookup happened. `MAX_COM_INDICES` bounds that.
|
||||
///
|
||||
/// What THIS test guards is that bounding did not change what a
|
||||
/// legitimate playlist MEANS: it goes red if the bound is set too LOW
|
||||
/// (verified at 2 — the real indices `0,2,4` stop resolving and the
|
||||
/// purposes change). It does NOT go red if the bound is deleted
|
||||
/// entirely, because the out-of-range filler is unobservable at the
|
||||
/// label level and a `HashSet` collapses the repeats. Enforcement is
|
||||
/// proven separately, by
|
||||
/// `distinct_unaddressable_indices_are_refused_not_stored`, which reads
|
||||
/// the set itself. Two tests, two properties; neither pretends to the
|
||||
/// other's job.
|
||||
#[test]
|
||||
fn bounding_the_parse_does_not_change_a_legitimate_playlist() {
|
||||
// Three real indices, then far more entries than can address a cell.
|
||||
const OVERSIZED: usize = MAX_COM_INDICES + 10_000;
|
||||
let mut feature = String::from(r#"<playlist name="Feature" sub=""#);
|
||||
feature.push_str(&"eng,".repeat(8));
|
||||
feature.pop();
|
||||
feature.push_str(r#"" sub_com1_idx="0,2,4,"#);
|
||||
feature.push_str(&"9999999,".repeat(OVERSIZED));
|
||||
feature.pop();
|
||||
feature.push_str(r#"" />"#);
|
||||
|
||||
let labels = labels_from_feature(&feature);
|
||||
|
||||
// The fixture's real indices still decide the purposes: bounding the
|
||||
// parse must not change what a legitimate playlist means.
|
||||
assert_eq!(labels.len(), 8);
|
||||
assert_eq!(labels[0].purpose, LabelPurpose::Commentary);
|
||||
assert_eq!(labels[1].purpose, LabelPurpose::Normal);
|
||||
assert_eq!(labels[2].purpose, LabelPurpose::Commentary);
|
||||
assert_eq!(labels[3].purpose, LabelPurpose::Normal);
|
||||
assert_eq!(labels[4].purpose, LabelPurpose::Commentary);
|
||||
}
|
||||
|
||||
/// The set REFUSES unaddressable indices, so a hostile playlist cannot
|
||||
/// inflate it. DISTINCT values on purpose: a `HashSet` collapses repeats,
|
||||
/// so a million copies of one index costs one entry and would prove
|
||||
/// nothing. Fifty thousand distinct out-of-range indices cost fifty
|
||||
/// thousand entries without the filter, and none with it — so this test
|
||||
/// goes red if the bound is removed, which the label-level assertions
|
||||
/// below cannot do.
|
||||
#[test]
|
||||
fn distinct_unaddressable_indices_are_refused_not_stored() {
|
||||
let hostile: String = (MAX_COM_INDICES..MAX_COM_INDICES + 50_000)
|
||||
.map(|i| i.to_string())
|
||||
.collect::<Vec<_>>()
|
||||
.join(",");
|
||||
let set = com_indices(Some(hostile));
|
||||
assert!(
|
||||
set.is_empty(),
|
||||
"kept {} unaddressable indices — the parse is still unbounded",
|
||||
set.len()
|
||||
);
|
||||
// The addressable ones are still kept.
|
||||
assert_eq!(com_indices(Some("0,2,4".to_string())).len(), 3);
|
||||
}
|
||||
|
||||
/// `forced_sub` is bounded too — the third CSV in the same function, and
|
||||
/// the one that had no value filter to hide behind.
|
||||
///
|
||||
/// Read through `forced_subs` rather than through the labels for the same
|
||||
/// reason the two tests above read the set: the subtitle loop stops at
|
||||
/// `MAX_COM_INDICES`, so a label-level assertion cannot tell a bounded
|
||||
/// parse from an unbounded one.
|
||||
#[test]
|
||||
fn forced_sub_cells_past_the_last_addressable_one_are_not_parsed() {
|
||||
let hostile = "0,".repeat(MAX_COM_INDICES + 50_000);
|
||||
let cells = forced_subs(Some(hostile));
|
||||
assert_eq!(
|
||||
cells.len(),
|
||||
MAX_COM_INDICES,
|
||||
"parsed {} cells — the forced_sub parse is still unbounded",
|
||||
cells.len()
|
||||
);
|
||||
}
|
||||
|
||||
/// Bounding it must not change what a legitimate playlist means: the
|
||||
/// cells that CAN address a stream still classify exactly as before.
|
||||
#[test]
|
||||
fn bounding_forced_sub_leaves_the_addressable_cells_alone() {
|
||||
let cells = forced_subs(Some("0,1,3,2".to_string()));
|
||||
assert_eq!(
|
||||
cells,
|
||||
vec![
|
||||
ForcedSub::None,
|
||||
ForcedSub::ContainsForcedSegments,
|
||||
ForcedSub::ForcedNarrative,
|
||||
ForcedSub::ForcedNarrative,
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
/// An index that cannot address any cell is dropped rather than STORED.
|
||||
///
|
||||
/// Asserted through `com_indices`, not through the labels: the labelling
|
||||
/// loop never queries a cell position that high, so at the label level
|
||||
/// retaining the value is unobservable and the assertion could not fail.
|
||||
/// Reading the set is what makes the claim checkable.
|
||||
#[test]
|
||||
fn an_index_that_cannot_address_any_cell_is_not_retained() {
|
||||
let set = com_indices(Some(format!(
|
||||
"1,{},{}",
|
||||
MAX_COM_INDICES,
|
||||
MAX_COM_INDICES + 1
|
||||
)));
|
||||
assert_eq!(
|
||||
set.len(),
|
||||
1,
|
||||
"only the addressable index belongs in the set, got {set:?}"
|
||||
);
|
||||
assert!(set.contains(&1));
|
||||
}
|
||||
|
||||
/// Headroom: the BD STN_table admits at most 32 PG streams per playlist,
|
||||
/// and a real `sub_com1_idx` lists a handful of commentary tracks. The set
|
||||
/// must behave identically to the old scan on real-shaped input.
|
||||
#[test]
|
||||
fn commentary_indices_still_match_on_real_shaped_input() {
|
||||
let feature = r#"<playlist name="Feature" sub="eng,eng,zho,ces,dan" sub_com1_idx="1,3" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let purposes: Vec<LabelPurpose> = labels.iter().map(|l| l.purpose).collect();
|
||||
assert_eq!(
|
||||
purposes,
|
||||
vec![
|
||||
LabelPurpose::Normal,
|
||||
LabelPurpose::Commentary,
|
||||
LabelPurpose::Normal,
|
||||
LabelPurpose::Commentary,
|
||||
LabelPurpose::Normal,
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
fn audio(labels: &[StreamLabel]) -> Vec<&StreamLabel> {
|
||||
labels
|
||||
.iter()
|
||||
@@ -182,11 +540,56 @@ mod tests {
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// The `aud` / `sub` CSVs are the vendor's STN-ordered stream lists: one
|
||||
/// slot per stream, and `aud_com1_idx` / `forced_sub` are indexed against
|
||||
/// those same slot positions. A slot whose language cell is empty carries
|
||||
/// nothing to label but still OCCUPIES its slot, so it must not renumber
|
||||
/// the slots behind it.
|
||||
///
|
||||
/// Numbering only the slots that carry a language collapsed every later
|
||||
/// label one position forward per empty cell, which is how a forced
|
||||
/// marker authored for one STN slot lands on the full-subtitle track in
|
||||
/// front of it.
|
||||
#[test]
|
||||
fn empty_middle_slot_does_not_inflate_stream_number() {
|
||||
// aud="eng,,fra": the empty middle slot is skipped, and the
|
||||
// second real stream (fra) must be numbered 2, matching
|
||||
// apply_labels' monotonic counter — not 3 (its raw CSV index).
|
||||
fn empty_csv_slot_still_occupies_its_stn_slot() {
|
||||
// Audio: slot 2 is empty; `fra` is STN slot 3 and is the commentary
|
||||
// the vendor pointed at with the 0-based CSV index 2.
|
||||
let feature = r#"<playlist name="Feature" aud="eng,,fra" aud_com1_idx="2" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let a = audio(&labels);
|
||||
assert_eq!(a.len(), 2, "the empty slot carries no label");
|
||||
assert_eq!(a[0].language, "eng");
|
||||
assert_eq!(a[0].stream_number, 1);
|
||||
assert_eq!(a[1].language, "fra");
|
||||
assert_eq!(
|
||||
a[1].stream_number, 3,
|
||||
"an empty CSV cell occupies STN slot 2, so `fra` is slot 3"
|
||||
);
|
||||
assert_eq!(a[1].purpose, LabelPurpose::Commentary);
|
||||
|
||||
// Subtitles: same shape, and the consequence is a misplaced forced
|
||||
// flag. `forced_sub` index 2 is the forced-narrative track; with the
|
||||
// empty slot renumbered away it would be written onto STN slot 2.
|
||||
let feature = r#"<playlist name="Feature" sub="eng,,fra" forced_sub="0,0,3" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let s = subs(&labels);
|
||||
assert_eq!(s.len(), 2);
|
||||
assert_eq!(s[0].language, "eng");
|
||||
assert_eq!(s[0].stream_number, 1);
|
||||
assert_eq!(s[0].qualifier, LabelQualifier::None);
|
||||
assert_eq!(s[1].language, "fra");
|
||||
assert_eq!(
|
||||
s[1].stream_number, 3,
|
||||
"the forced marker belongs to STN slot 3, not slot 2"
|
||||
);
|
||||
assert_eq!(s[1].qualifier, LabelQualifier::Forced);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn empty_middle_slot_carries_no_label_but_keeps_its_slot() {
|
||||
// aud="eng,,fra": the empty middle cell yields no label — there is
|
||||
// nothing to label — but it still owns STN slot 2, so `fra` is slot
|
||||
// 3. (This test previously asserted 2, pinning the renumbering bug.)
|
||||
let feature = r#"<playlist name="Feature" aud="eng,,fra" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let a = audio(&labels);
|
||||
@@ -194,7 +597,7 @@ mod tests {
|
||||
assert_eq!(a[0].language, "eng");
|
||||
assert_eq!(a[0].stream_number, 1);
|
||||
assert_eq!(a[1].language, "fra");
|
||||
assert_eq!(a[1].stream_number, 2);
|
||||
assert_eq!(a[1].stream_number, 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -202,7 +605,7 @@ mod tests {
|
||||
// Whitespace around the index, and a multi-value list, must both
|
||||
// resolve. com index is positional against the raw CSV, so with
|
||||
// an empty slot at position 1, " 2 " marks the 'fra' track
|
||||
// (CSV index 2) as commentary.
|
||||
// (CSV index 2, STN slot 3) as commentary.
|
||||
let feature = r#"<playlist aud="eng,,fra" aud_com1_idx=" 2 " />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let a = audio(&labels);
|
||||
@@ -214,10 +617,10 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn forced_sub_aligns_with_raw_csv_index() {
|
||||
// sub="eng,eng,zho,ces" forced_sub="0,0,0,1": the forced flag is
|
||||
// sub="eng,eng,zho,ces" forced_sub="0,0,0,3": the forced marker is
|
||||
// positional on the raw CSV, so 'ces' (index 3) is forced; its
|
||||
// stream_number is its non-empty position (4 here, no gaps).
|
||||
let feature = r#"<playlist sub="eng,eng,zho,ces" forced_sub="0,0,0,1" />"#;
|
||||
// stream_number is its 1-based cell position, 4.
|
||||
let feature = r#"<playlist sub="eng,eng,zho,ces" forced_sub="0,0,0,3" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let s = subs(&labels);
|
||||
assert_eq!(s.len(), 4);
|
||||
@@ -262,10 +665,12 @@ mod tests {
|
||||
assert!(feature.contains(r#"name="MainMovie""#));
|
||||
}
|
||||
|
||||
/// Spec: stream_number for audio is 1-based and increments only on non-empty slots.
|
||||
/// Mutation: increment for empty slots too → stream numbers inflate.
|
||||
/// Spec: stream_number for audio is the cell's own 1-based CSV position,
|
||||
/// because the CSV is the STN list and empty cells are slots too.
|
||||
/// Mutation: count only non-empty cells → every label behind an empty
|
||||
/// cell shifts one slot forward.
|
||||
#[test]
|
||||
fn audio_stream_numbering_skips_empty_slots() {
|
||||
fn audio_stream_numbering_uses_raw_csv_slot_position() {
|
||||
let feature = r#"<playlist name="Feature" aud="eng,,fra,,spa" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let a = audio(&labels);
|
||||
@@ -273,9 +678,9 @@ mod tests {
|
||||
assert_eq!(a[0].language, "eng");
|
||||
assert_eq!(a[0].stream_number, 1);
|
||||
assert_eq!(a[1].language, "fra");
|
||||
assert_eq!(a[1].stream_number, 2);
|
||||
assert_eq!(a[1].stream_number, 3);
|
||||
assert_eq!(a[2].language, "spa");
|
||||
assert_eq!(a[2].stream_number, 3);
|
||||
assert_eq!(a[2].stream_number, 5);
|
||||
}
|
||||
|
||||
/// Spec: forced subtitle at the last position with gaps in between.
|
||||
@@ -283,9 +688,9 @@ mod tests {
|
||||
/// Mutation: use stream_number (dense) instead of raw index → wrong subtitle forced.
|
||||
#[test]
|
||||
fn forced_sub_uses_raw_csv_index_with_gaps() {
|
||||
// sub="eng,,fra,,spa" forced_sub="0,0,0,0,1"
|
||||
// raw CSV index 4 = "spa"; stream_number for spa = 3 (3rd non-empty).
|
||||
let feature = r#"<playlist name="Feature" sub="eng,,fra,,spa" forced_sub="0,0,0,0,1" />"#;
|
||||
// sub="eng,,fra,,spa" forced_sub="0,0,0,0,3"
|
||||
// raw CSV index 4 = "spa", i.e. STN slot 5.
|
||||
let feature = r#"<playlist name="Feature" sub="eng,,fra,,spa" forced_sub="0,0,0,0,3" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let s = subs(&labels);
|
||||
assert_eq!(s.len(), 3);
|
||||
@@ -295,6 +700,7 @@ mod tests {
|
||||
assert_eq!(s[1].qualifier, LabelQualifier::None);
|
||||
assert_eq!(s[2].language, "spa");
|
||||
assert_eq!(s[2].qualifier, LabelQualifier::Forced);
|
||||
assert_eq!(s[2].stream_number, 5);
|
||||
}
|
||||
|
||||
/// Spec: aud_com1_idx is positional against the raw CSV.
|
||||
@@ -303,13 +709,14 @@ mod tests {
|
||||
/// Mutation: use stream_number instead of raw CSV index → wrong stream is commentary.
|
||||
#[test]
|
||||
fn audio_commentary_index_raw_csv_position() {
|
||||
// aud="eng,,fra,spa" aud_com1_idx="2" → CSV index 2 = "fra".
|
||||
// "fra" is stream_number 2 (second non-empty slot, skipping the empty).
|
||||
// aud="eng,,fra,spa" aud_com1_idx="2" → CSV index 2 = "fra",
|
||||
// which is STN slot 3.
|
||||
let feature = r#"<playlist name="Feature" aud="eng,,fra,spa" aud_com1_idx="2" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let a = audio(&labels);
|
||||
assert_eq!(a.len(), 3);
|
||||
assert_eq!(a[1].language, "fra");
|
||||
assert_eq!(a[1].stream_number, 3);
|
||||
assert_eq!(a[1].purpose, LabelPurpose::Commentary);
|
||||
assert_eq!(a[0].purpose, LabelPurpose::Normal);
|
||||
assert_eq!(a[2].purpose, LabelPurpose::Normal);
|
||||
@@ -352,10 +759,11 @@ mod tests {
|
||||
assert!(s.is_empty(), "no subtitle labels when sub is absent");
|
||||
}
|
||||
|
||||
/// Spec: audio stream_number uses saturating_add on overflow (per u16 cap).
|
||||
/// Mutation: use wrapping_add → stream numbers wrap to 0, skipping apply.
|
||||
/// Spec: audio stream_number is the cell's 1-based position and never
|
||||
/// wraps; past the u16 space the parser stops emitting.
|
||||
/// Mutation: cast `i + 1` to u16 → stream numbers wrap to 0, skipping apply.
|
||||
#[test]
|
||||
fn audio_stream_number_saturates_not_wraps() {
|
||||
fn audio_stream_number_never_wraps() {
|
||||
// 65535 audio tracks is impossible on a real disc but the parser must
|
||||
// not panic or produce 0. Build a comma-separated list of 65535 "eng"s.
|
||||
// We only run the number-assignment logic via labels_from_feature.
|
||||
@@ -375,15 +783,92 @@ mod tests {
|
||||
assert_eq!(last, 300);
|
||||
}
|
||||
|
||||
/// Spec: forced_sub with whitespace around "1" must still parse as true.
|
||||
/// Mutation: use `== "1"` instead of `trim() == "1"` → " 1 " fails.
|
||||
/// Spec: a `forced_sub` cell with surrounding whitespace still classifies.
|
||||
/// Mutation: drop the `trim()` → " 3 " falls through to the unrecognised
|
||||
/// arm and the disc's forced-narrative track loses its label.
|
||||
#[test]
|
||||
fn forced_sub_whitespace_around_one() {
|
||||
let feature = r#"<playlist name="Feature" sub="eng,fra" forced_sub="0, 1" />"#;
|
||||
fn forced_sub_cells_are_trimmed_before_classification() {
|
||||
let feature = r#"<playlist name="Feature" sub="eng,fra,spa" forced_sub="0, 3 , 1 " />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let s = subs(&labels);
|
||||
assert_eq!(s[0].qualifier, LabelQualifier::None);
|
||||
assert_eq!(s[1].qualifier, LabelQualifier::Forced);
|
||||
assert_eq!(s[2].qualifier, LabelQualifier::None);
|
||||
}
|
||||
|
||||
/// `forced_sub` is an enumeration, and `1` is its "full dialogue track that
|
||||
/// also carries forced signs" value — NOT "this track is forced".
|
||||
///
|
||||
/// Measured on a disc whose feature declares nine `1` cells among 32
|
||||
/// subtitle slots: all nine are full dialogue tracks of 949-1411 display
|
||||
/// sets, and the disc has no dedicated forced track at all. Reading `1` as
|
||||
/// forced is what produced two identical full subtitle tracks for one
|
||||
/// language with one of them flagged forced.
|
||||
///
|
||||
/// Nothing downstream can undo this on the discs that need it most:
|
||||
/// `mux::codec::pgs::demotable` may only clear a vendor forced label where
|
||||
/// some track on the disc demonstrably sets `forced_on_flag`, and measured
|
||||
/// discs using this label format never set it.
|
||||
///
|
||||
/// Mutation: `"1" => ForcedNarrative` (the old reading) → red.
|
||||
#[test]
|
||||
fn a_contains_forced_segments_cell_is_not_a_forced_track() {
|
||||
let feature = r#"<playlist name="Feature" sub="eng,ces,deu" forced_sub="0,1,1" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let s = subs(&labels);
|
||||
assert_eq!(s.len(), 3);
|
||||
assert!(
|
||||
s.iter().all(|l| l.qualifier == LabelQualifier::None),
|
||||
"a `1` marks a full track containing forced signs, not a forced track"
|
||||
);
|
||||
}
|
||||
|
||||
/// `2` and `3` are the cells that DO name a dedicated forced-narrative
|
||||
/// track, and the old boolean reading discarded both.
|
||||
///
|
||||
/// Measured: these cells occupy their own trailing STN slots, one per
|
||||
/// localized language, duplicating a language that already holds a full
|
||||
/// track earlier in the list. On one measured disc the four `3` slots carry
|
||||
/// 7, 14, 23 and 59 display sets against 1216-2655 on the full tracks they
|
||||
/// duplicate — and not one display set anywhere on that disc carries
|
||||
/// `forced_on_flag`, so neither the scan probe nor the muxer can promote
|
||||
/// them from content. The vendor cell is the only evidence there is.
|
||||
///
|
||||
/// Mutation: drop either arm of the `"2" | "3"` match → red.
|
||||
#[test]
|
||||
fn a_dedicated_forced_narrative_cell_is_a_forced_track() {
|
||||
// The measured shape: full tracks first, their forced companions in
|
||||
// trailing slots of the same languages.
|
||||
let feature =
|
||||
r#"<playlist name="Feature" sub="eng,cat,jpn,cat,jpn" forced_sub="0,0,0,2,3" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let s = subs(&labels);
|
||||
assert_eq!(s.len(), 5);
|
||||
assert_eq!(s[1].qualifier, LabelQualifier::None, "the full cat track");
|
||||
assert_eq!(s[2].qualifier, LabelQualifier::None, "the full jpn track");
|
||||
assert_eq!(s[3].qualifier, LabelQualifier::Forced, "cat forced slot");
|
||||
assert_eq!(s[3].stream_number, 4);
|
||||
assert_eq!(s[4].qualifier, LabelQualifier::Forced, "jpn forced slot");
|
||||
assert_eq!(s[4].stream_number, 5);
|
||||
}
|
||||
|
||||
/// An unrecognised cell must fall to NOT forced. Asserting forced is the
|
||||
/// expensive mistake (a full dialogue track a player then burns on screen),
|
||||
/// so an unknown value from a future authoring revision must not be able to
|
||||
/// make that claim.
|
||||
///
|
||||
/// Mutation: `_ => ForcedNarrative`, or treating "any non-zero" as forced.
|
||||
#[test]
|
||||
fn an_unrecognised_forced_sub_cell_is_not_forced() {
|
||||
let feature = r#"<playlist name="Feature" sub="eng,fra,spa,ita" forced_sub="4,x,,-1" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
let s = subs(&labels);
|
||||
assert_eq!(s.len(), 4);
|
||||
assert!(s.iter().all(|l| l.qualifier == LabelQualifier::None));
|
||||
// ...and so must a cell the CSV simply does not reach.
|
||||
let feature = r#"<playlist name="Feature" sub="eng,fra" forced_sub="0" />"#;
|
||||
let labels = labels_from_feature(feature);
|
||||
assert_eq!(subs(&labels)[1].qualifier, LabelQualifier::None);
|
||||
}
|
||||
|
||||
/// Spec: `find_feature_playlist` returns None when XML has no `<playlist>` elements.
|
||||
@@ -393,4 +878,23 @@ mod tests {
|
||||
assert!(find_feature_playlist("").is_none());
|
||||
assert!(find_feature_playlist("<root />").is_none());
|
||||
}
|
||||
|
||||
/// Spec: on a tie in audio-slot count, the FIRST playlist encountered
|
||||
/// wins (consistent with `select_result`'s first-wins tiebreak
|
||||
/// elsewhere in the registry) — later playlists only displace the
|
||||
/// current best on a STRICTLY greater count.
|
||||
/// Mutation: `count > best_aud_count` -> `count >= best_aud_count`
|
||||
/// would let a later tied playlist silently displace the first.
|
||||
#[test]
|
||||
fn find_feature_first_wins_on_audio_count_tie() {
|
||||
let xml = r#"
|
||||
<playlist name="A" aud="eng,fra" />
|
||||
<playlist name="B" aud="deu,spa" />
|
||||
"#;
|
||||
let feature = find_feature_playlist(xml).expect("a feature is found");
|
||||
assert!(
|
||||
feature.contains(r#"name="A""#),
|
||||
"first playlist must win a tie, got: {feature}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
+785
-64
File diff suppressed because it is too large
Load Diff
+19
-12
@@ -1,8 +1,8 @@
|
||||
//! Menu-graphic filename language hints.
|
||||
//!
|
||||
//! Some BD-J discs encode per-language menu artwork with the language in the
|
||||
//! filename, e.g. `Dune_UHD01_Eng_Composite1.png`,
|
||||
//! `VForVendetta_UHD01_FRE_Composite2.png`. The `_UHD01_{LANG}_Composite`
|
||||
//! filename, e.g. `Feature_UHD01_Eng_Composite1.png`,
|
||||
//! `AltFeature_UHD01_FRE_Composite2.png`. The `_UHD01_{LANG}_Composite`
|
||||
//! marker is authored deliberately, so the set of `{LANG}` tokens is the set
|
||||
//! of menu languages the disc ships.
|
||||
//!
|
||||
@@ -41,15 +41,16 @@ pub fn parse(_reader: &mut dyn SectorSource, udf: &UdfFs) -> Option<ParseResult>
|
||||
fn labels_from_filenames(names: &[String]) -> Vec<StreamLabel> {
|
||||
let mut seen: Vec<&'static str> = Vec::new();
|
||||
for name in names {
|
||||
if let Some(code) = filename_lang(name) {
|
||||
if !seen.contains(&code) {
|
||||
seen.push(code);
|
||||
}
|
||||
if let Some(code) = filename_lang(name)
|
||||
&& !seen.contains(&code)
|
||||
{
|
||||
seen.push(code);
|
||||
}
|
||||
}
|
||||
seen.into_iter()
|
||||
.enumerate()
|
||||
.map(|(i, code)| StreamLabel {
|
||||
stream_id: None,
|
||||
stream_number: (i as u16).saturating_add(1),
|
||||
stream_type: StreamLabelType::Audio,
|
||||
language: code.to_string(),
|
||||
@@ -91,10 +92,16 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn extracts_confirmed_samples() {
|
||||
assert_eq!(filename_lang("Dune_UHD01_Eng_Composite1.png"), Some("eng"));
|
||||
assert_eq!(filename_lang("Dune_UHD01_Ger_Composite2.png"), Some("deu"));
|
||||
assert_eq!(
|
||||
filename_lang("VForVendetta_UHD01_FRE_Composite2.png"),
|
||||
filename_lang("Feature_UHD01_Eng_Composite1.png"),
|
||||
Some("eng")
|
||||
);
|
||||
assert_eq!(
|
||||
filename_lang("Feature_UHD01_Ger_Composite2.png"),
|
||||
Some("deu")
|
||||
);
|
||||
assert_eq!(
|
||||
filename_lang("AltFeature_UHD01_FRE_Composite2.png"),
|
||||
Some("fra")
|
||||
);
|
||||
}
|
||||
@@ -120,9 +127,9 @@ mod tests {
|
||||
#[test]
|
||||
fn dedups_and_numbers_distinct_languages() {
|
||||
let names = vec![
|
||||
"Dune_UHD01_Eng_Composite1.png".to_string(),
|
||||
"Dune_UHD01_Eng_Composite2.png".to_string(),
|
||||
"Dune_UHD01_Ger_Composite1.png".to_string(),
|
||||
"Feature_UHD01_Eng_Composite1.png".to_string(),
|
||||
"Feature_UHD01_Eng_Composite2.png".to_string(),
|
||||
"Feature_UHD01_Ger_Composite1.png".to_string(),
|
||||
"LoadingComposite1.png".to_string(),
|
||||
];
|
||||
let labels = labels_from_filenames(&names);
|
||||
|
||||
+1
-1
@@ -169,7 +169,7 @@ mod tests {
|
||||
/// Mutation: skip the final `if !current.is_empty()` emit → trailing run lost.
|
||||
#[test]
|
||||
fn large_buffer_trailing_run_emitted() {
|
||||
let buf: Vec<u8> = (0..1000u32).map(|i| (0x41u8 + (i % 26) as u8)).collect();
|
||||
let buf: Vec<u8> = (0..1000u32).map(|i| 0x41u8 + (i % 26) as u8).collect();
|
||||
let got = extract_ascii_strings(&buf, 1);
|
||||
// All printable, so one big run at the end.
|
||||
assert!(!got.is_empty());
|
||||
|
||||
+431
-1
@@ -177,7 +177,7 @@ const BARE_LANGS: &[(&str, &str)] = &[
|
||||
];
|
||||
|
||||
/// Map a short menu-graphic language token (as embedded in authoring
|
||||
/// filenames like `Dune_UHD01_Eng_Composite1.png`) to an ISO-639-2/T code.
|
||||
/// filenames like `Feature_UHD01_Eng_Composite1.png`) to an ISO-639-2/T code.
|
||||
///
|
||||
/// These filename tokens are compact 2/3-letter abbreviations, NOT the full
|
||||
/// language names [`lang`] handles, so they get their own certain table.
|
||||
@@ -219,6 +219,237 @@ pub fn menu_lang(token: &str) -> Option<&'static str> {
|
||||
Some(code)
|
||||
}
|
||||
|
||||
// ── ISO 639-1 → ISO 639-2 ────────────────────────────────────────────────────
|
||||
|
||||
/// The complete ISO 639-1 set, paired with its ISO 639-2/**T** (terminological)
|
||||
/// code. Every two-letter code ISO 639-1 defines appears exactly once.
|
||||
///
|
||||
/// /T is the variant the rest of this crate uses — [`lang`] and [`menu_lang`]
|
||||
/// both normalize to it (`deu` not `ger`, `fra` not `fre`, `zho` not `chi`,
|
||||
/// `ces`, `nld`, `ell`, `ron`, `slk`, `isl`, `eus`, `hrv`) — so the three
|
||||
/// tables cannot disagree. `iso639_1_agrees_with_menu_lang` pins that.
|
||||
///
|
||||
/// For the 165 codes where 639-2/B and /T are identical this distinction does
|
||||
/// not arise; it only matters for the 20-odd languages with a distinct
|
||||
/// bibliographic code.
|
||||
const ISO_639_1_TO_2: &[(&str, &str)] = &[
|
||||
("aa", "aar"),
|
||||
("ab", "abk"),
|
||||
("ae", "ave"),
|
||||
("af", "afr"),
|
||||
("ak", "aka"),
|
||||
("am", "amh"),
|
||||
("an", "arg"),
|
||||
("ar", "ara"),
|
||||
("as", "asm"),
|
||||
("av", "ava"),
|
||||
("ay", "aym"),
|
||||
("az", "aze"),
|
||||
("ba", "bak"),
|
||||
("be", "bel"),
|
||||
("bg", "bul"),
|
||||
("bh", "bih"),
|
||||
("bi", "bis"),
|
||||
("bm", "bam"),
|
||||
("bn", "ben"),
|
||||
("bo", "bod"),
|
||||
("br", "bre"),
|
||||
("bs", "bos"),
|
||||
("ca", "cat"),
|
||||
("ce", "che"),
|
||||
("ch", "cha"),
|
||||
("co", "cos"),
|
||||
("cr", "cre"),
|
||||
("cs", "ces"),
|
||||
("cu", "chu"),
|
||||
("cv", "chv"),
|
||||
("cy", "cym"),
|
||||
("da", "dan"),
|
||||
("de", "deu"),
|
||||
("dv", "div"),
|
||||
("dz", "dzo"),
|
||||
("ee", "ewe"),
|
||||
("el", "ell"),
|
||||
("en", "eng"),
|
||||
("eo", "epo"),
|
||||
("es", "spa"),
|
||||
("et", "est"),
|
||||
("eu", "eus"),
|
||||
("fa", "fas"),
|
||||
("ff", "ful"),
|
||||
("fi", "fin"),
|
||||
("fj", "fij"),
|
||||
("fo", "fao"),
|
||||
("fr", "fra"),
|
||||
("fy", "fry"),
|
||||
("ga", "gle"),
|
||||
("gd", "gla"),
|
||||
("gl", "glg"),
|
||||
("gn", "grn"),
|
||||
("gu", "guj"),
|
||||
("gv", "glv"),
|
||||
("ha", "hau"),
|
||||
("he", "heb"),
|
||||
("hi", "hin"),
|
||||
("ho", "hmo"),
|
||||
("hr", "hrv"),
|
||||
("ht", "hat"),
|
||||
("hu", "hun"),
|
||||
("hy", "hye"),
|
||||
("hz", "her"),
|
||||
("ia", "ina"),
|
||||
("id", "ind"),
|
||||
("ie", "ile"),
|
||||
("ig", "ibo"),
|
||||
("ii", "iii"),
|
||||
("ik", "ipk"),
|
||||
("io", "ido"),
|
||||
("is", "isl"),
|
||||
("it", "ita"),
|
||||
("iu", "iku"),
|
||||
("ja", "jpn"),
|
||||
("jv", "jav"),
|
||||
("ka", "kat"),
|
||||
("kg", "kon"),
|
||||
("ki", "kik"),
|
||||
("kj", "kua"),
|
||||
("kk", "kaz"),
|
||||
("kl", "kal"),
|
||||
("km", "khm"),
|
||||
("kn", "kan"),
|
||||
("ko", "kor"),
|
||||
("kr", "kau"),
|
||||
("ks", "kas"),
|
||||
("ku", "kur"),
|
||||
("kv", "kom"),
|
||||
("kw", "cor"),
|
||||
("ky", "kir"),
|
||||
("la", "lat"),
|
||||
("lb", "ltz"),
|
||||
("lg", "lug"),
|
||||
("li", "lim"),
|
||||
("ln", "lin"),
|
||||
("lo", "lao"),
|
||||
("lt", "lit"),
|
||||
("lu", "lub"),
|
||||
("lv", "lav"),
|
||||
("mg", "mlg"),
|
||||
("mh", "mah"),
|
||||
("mi", "mri"),
|
||||
("mk", "mkd"),
|
||||
("ml", "mal"),
|
||||
("mn", "mon"),
|
||||
("mr", "mar"),
|
||||
("ms", "msa"),
|
||||
("mt", "mlt"),
|
||||
("my", "mya"),
|
||||
("na", "nau"),
|
||||
("nb", "nob"),
|
||||
("nd", "nde"),
|
||||
("ne", "nep"),
|
||||
("ng", "ndo"),
|
||||
("nl", "nld"),
|
||||
("nn", "nno"),
|
||||
("no", "nor"),
|
||||
("nr", "nbl"),
|
||||
("nv", "nav"),
|
||||
("ny", "nya"),
|
||||
("oc", "oci"),
|
||||
("oj", "oji"),
|
||||
("om", "orm"),
|
||||
("or", "ori"),
|
||||
("os", "oss"),
|
||||
("pa", "pan"),
|
||||
("pi", "pli"),
|
||||
("pl", "pol"),
|
||||
("ps", "pus"),
|
||||
("pt", "por"),
|
||||
("qu", "que"),
|
||||
("rm", "roh"),
|
||||
("rn", "run"),
|
||||
("ro", "ron"),
|
||||
("ru", "rus"),
|
||||
("rw", "kin"),
|
||||
("sa", "san"),
|
||||
("sc", "srd"),
|
||||
("sd", "snd"),
|
||||
("se", "sme"),
|
||||
("sg", "sag"),
|
||||
("si", "sin"),
|
||||
("sk", "slk"),
|
||||
("sl", "slv"),
|
||||
("sm", "smo"),
|
||||
("sn", "sna"),
|
||||
("so", "som"),
|
||||
("sq", "sqi"),
|
||||
("sr", "srp"),
|
||||
("ss", "ssw"),
|
||||
("st", "sot"),
|
||||
("su", "sun"),
|
||||
("sv", "swe"),
|
||||
("sw", "swa"),
|
||||
("ta", "tam"),
|
||||
("te", "tel"),
|
||||
("tg", "tgk"),
|
||||
("th", "tha"),
|
||||
("ti", "tir"),
|
||||
("tk", "tuk"),
|
||||
("tl", "tgl"),
|
||||
("tn", "tsn"),
|
||||
("to", "ton"),
|
||||
("tr", "tur"),
|
||||
("ts", "tso"),
|
||||
("tt", "tat"),
|
||||
("tw", "twi"),
|
||||
("ty", "tah"),
|
||||
("ug", "uig"),
|
||||
("uk", "ukr"),
|
||||
("ur", "urd"),
|
||||
("uz", "uzb"),
|
||||
("ve", "ven"),
|
||||
("vi", "vie"),
|
||||
("vo", "vol"),
|
||||
("wa", "wln"),
|
||||
("wo", "wol"),
|
||||
("xh", "xho"),
|
||||
("yi", "yid"),
|
||||
("yo", "yor"),
|
||||
("za", "zha"),
|
||||
("zh", "zho"),
|
||||
("zu", "zul"),
|
||||
];
|
||||
|
||||
/// The three two-letter codes ISO 639-1 has since withdrawn, mapped to their
|
||||
/// replacements. DVD-Video froze its language list on the 1988 edition, so
|
||||
/// discs authored to the spec carry these spellings and no other table sees
|
||||
/// them: `iw` Hebrew (now `he`), `in` Indonesian (now `id`), `ji` Yiddish
|
||||
/// (now `yi`).
|
||||
const ISO_639_1_DEPRECATED: &[(&str, &str)] = &[("iw", "he"), ("in", "id"), ("ji", "yi")];
|
||||
|
||||
/// Map an ISO 639-1 two-letter language code to its ISO 639-2/T three-letter
|
||||
/// code, accepting the withdrawn DVD-era spellings (`iw`, `in`, `ji`) as
|
||||
/// aliases for their replacements.
|
||||
///
|
||||
/// Covers the WHOLE of ISO 639-1, unlike [`menu_lang`], whose table only spans
|
||||
/// the languages that show up in Blu-ray menu-graphic filenames. Callers that
|
||||
/// convert a spec field — a DVD IFO attribute block, say — need the whole set:
|
||||
/// narrowing it to the menu vocabulary would fold every other language onto
|
||||
/// one value and make a disc's tracks indistinguishable from each other.
|
||||
///
|
||||
/// Case-insensitive and trimmed. Returns `None` for anything that is not an
|
||||
/// ISO 639-1 code, so callers decide the fallback rather than getting a guess.
|
||||
pub fn iso639_1_to_iso639_2(code: &str) -> Option<&'static str> {
|
||||
let c = code.trim().to_ascii_lowercase();
|
||||
let c = ISO_639_1_DEPRECATED
|
||||
.iter()
|
||||
.find(|(old, _)| *old == c)
|
||||
.map_or(c.as_str(), |(_, new)| new);
|
||||
ISO_639_1_TO_2
|
||||
.iter()
|
||||
.find(|(two, _)| *two == c)
|
||||
.map(|(_, three)| *three)
|
||||
}
|
||||
|
||||
// ── Purpose ──────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Classify a free-form English label string into a [`LabelPurpose`].
|
||||
@@ -685,4 +916,203 @@ mod tests {
|
||||
fn codec_empty_passes_through() {
|
||||
assert_eq!(codec(""), "");
|
||||
}
|
||||
|
||||
/// `purpose()`'s multi-word-compound fast path ORs two independent
|
||||
/// phrase checks ("audio description" / "descriptive service"). Each
|
||||
/// phrase, when it appears as a *word*-bounded match, is independently
|
||||
/// caught by the has_word fallback further down — so the OR only
|
||||
/// matters when a phrase appears as a *substring inside a larger word*
|
||||
/// (no boundary), which .contains() still catches but has_word() would
|
||||
/// reject.
|
||||
///
|
||||
/// Mutation: replace `||` with `&&` at line 238 → since "audio
|
||||
/// description" is absent here, the AND fails, the fast path doesn't
|
||||
/// fire, and the fallback has_word("descriptive") also fails (no word
|
||||
/// boundary before "descriptive" in "nondescriptive"), so purpose()
|
||||
/// wrongly returns Normal instead of Descriptive.
|
||||
#[test]
|
||||
fn purpose_descriptive_service_substring_without_word_boundary() {
|
||||
assert_eq!(
|
||||
purpose("nondescriptive service track"),
|
||||
LabelPurpose::Descriptive
|
||||
);
|
||||
}
|
||||
|
||||
/// `menu_lang()` maps every authoring-filename token in its table
|
||||
/// (ISO-639-2/B and /T spellings, plus ISO-639-1) to the canonical
|
||||
/// /T code used by the rest of the pipeline. Exhaustive per-arm check:
|
||||
/// deleting any single match arm makes that arm's tokens return None
|
||||
/// instead of the documented code.
|
||||
#[test]
|
||||
fn menu_lang_covers_every_table_entry() {
|
||||
let cases: &[(&str, &str)] = &[
|
||||
("eng", "eng"),
|
||||
("en", "eng"),
|
||||
("ger", "deu"),
|
||||
("deu", "deu"),
|
||||
("de", "deu"),
|
||||
("fre", "fra"),
|
||||
("fra", "fra"),
|
||||
("fr", "fra"),
|
||||
("spa", "spa"),
|
||||
("es", "spa"),
|
||||
("ita", "ita"),
|
||||
("it", "ita"),
|
||||
("por", "por"),
|
||||
("pt", "por"),
|
||||
("jpn", "jpn"),
|
||||
("jap", "jpn"),
|
||||
("ja", "jpn"),
|
||||
("kor", "kor"),
|
||||
("ko", "kor"),
|
||||
("chi", "zho"),
|
||||
("zho", "zho"),
|
||||
("zh", "zho"),
|
||||
("rus", "rus"),
|
||||
("ru", "rus"),
|
||||
("dut", "nld"),
|
||||
("nld", "nld"),
|
||||
("nl", "nld"),
|
||||
("pol", "pol"),
|
||||
("pl", "pol"),
|
||||
("cze", "ces"),
|
||||
("ces", "ces"),
|
||||
("cs", "ces"),
|
||||
("dan", "dan"),
|
||||
("da", "dan"),
|
||||
("fin", "fin"),
|
||||
("fi", "fin"),
|
||||
("nor", "nor"),
|
||||
("no", "nor"),
|
||||
("swe", "swe"),
|
||||
("sv", "swe"),
|
||||
("hun", "hun"),
|
||||
("hu", "hun"),
|
||||
("gre", "ell"),
|
||||
("ell", "ell"),
|
||||
("el", "ell"),
|
||||
("tur", "tur"),
|
||||
("tr", "tur"),
|
||||
("ara", "ara"),
|
||||
("ar", "ara"),
|
||||
("hin", "hin"),
|
||||
("hi", "hin"),
|
||||
("tha", "tha"),
|
||||
("th", "tha"),
|
||||
("ukr", "ukr"),
|
||||
("uk", "ukr"),
|
||||
("cat", "cat"),
|
||||
("ca", "cat"),
|
||||
];
|
||||
for (token, expected) in cases {
|
||||
assert_eq!(
|
||||
menu_lang(token),
|
||||
Some(*expected),
|
||||
"menu_lang({:?}) should map to {:?}",
|
||||
token,
|
||||
expected
|
||||
);
|
||||
}
|
||||
// Case-insensitive and trimmed.
|
||||
assert_eq!(menu_lang("ENG"), Some("eng"));
|
||||
assert_eq!(menu_lang(" Eng "), Some("eng"));
|
||||
// Unrecognized token -> None, never a guess.
|
||||
assert_eq!(menu_lang("xyz"), None);
|
||||
assert_eq!(menu_lang(""), None);
|
||||
}
|
||||
|
||||
/// Structural invariants of `ISO_639_1_TO_2`: it must hold the complete
|
||||
/// ISO 639-1 set (184 codes), every key a distinct pair of lowercase
|
||||
/// letters and every value three lowercase letters. A typo'd or duplicated
|
||||
/// row fails here rather than silently mislabelling a track.
|
||||
#[test]
|
||||
fn iso639_1_table_is_complete_and_well_formed() {
|
||||
assert_eq!(
|
||||
ISO_639_1_TO_2.len(),
|
||||
184,
|
||||
"ISO 639-1 defines 184 two-letter codes; the table must hold all \
|
||||
of them"
|
||||
);
|
||||
let mut keys: Vec<&str> = ISO_639_1_TO_2.iter().map(|(two, _)| *two).collect();
|
||||
keys.sort_unstable();
|
||||
let unique = keys.len();
|
||||
keys.dedup();
|
||||
assert_eq!(unique, keys.len(), "no ISO 639-1 code may appear twice");
|
||||
for (two, three) in ISO_639_1_TO_2 {
|
||||
assert!(
|
||||
two.len() == 2 && two.bytes().all(|b| b.is_ascii_lowercase()),
|
||||
"{two:?} is not a two-letter lowercase ISO 639-1 code"
|
||||
);
|
||||
assert!(
|
||||
three.len() == 3 && three.bytes().all(|b| b.is_ascii_lowercase()),
|
||||
"{three:?} is not a three-letter lowercase ISO 639-2 code"
|
||||
);
|
||||
}
|
||||
// The withdrawn DVD-era spellings resolve, and are not themselves
|
||||
// rows in the main table (they are aliases, not codes).
|
||||
for (old, new) in ISO_639_1_DEPRECATED {
|
||||
assert!(
|
||||
!ISO_639_1_TO_2.iter().any(|(two, _)| two == old),
|
||||
"withdrawn code {old:?} must not be a table row"
|
||||
);
|
||||
assert_eq!(
|
||||
iso639_1_to_iso639_2(old),
|
||||
iso639_1_to_iso639_2(new),
|
||||
"withdrawn code {old:?} must resolve exactly as {new:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The two tables must not disagree. Every two-letter token `menu_lang`
|
||||
/// accepts has to yield the same ISO 639-2/T code through
|
||||
/// `iso639_1_to_iso639_2`, so a DVD-sourced language and a Blu-ray
|
||||
/// menu-label language for the same tongue never produce different
|
||||
/// `Language` elements.
|
||||
#[test]
|
||||
fn iso639_1_agrees_with_menu_lang() {
|
||||
for (two, three) in ISO_639_1_TO_2 {
|
||||
if let Some(via_menu) = menu_lang(two) {
|
||||
assert_eq!(
|
||||
via_menu, *three,
|
||||
"menu_lang({two:?}) = {via_menu:?} disagrees with the ISO \
|
||||
639-1 table's {three:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
// Spot-check the /T choice itself, on the languages where /B differs.
|
||||
for (two, t_code) in [
|
||||
("de", "deu"),
|
||||
("fr", "fra"),
|
||||
("zh", "zho"),
|
||||
("cs", "ces"),
|
||||
("nl", "nld"),
|
||||
("el", "ell"),
|
||||
("ro", "ron"),
|
||||
("sk", "slk"),
|
||||
("is", "isl"),
|
||||
("hy", "hye"),
|
||||
("ka", "kat"),
|
||||
("fa", "fas"),
|
||||
] {
|
||||
assert_eq!(
|
||||
iso639_1_to_iso639_2(two),
|
||||
Some(t_code),
|
||||
"the crate standardises on ISO 639-2/T, so {two:?} is \
|
||||
{t_code:?} and never the bibliographic form"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// Trimming, case-insensitivity, and the no-guess contract.
|
||||
#[test]
|
||||
fn iso639_1_normalizes_input_and_never_guesses() {
|
||||
assert_eq!(iso639_1_to_iso639_2("RO"), Some("ron"));
|
||||
assert_eq!(iso639_1_to_iso639_2(" Ro "), Some("ron"));
|
||||
assert_eq!(iso639_1_to_iso639_2("IW"), Some("heb"));
|
||||
assert_eq!(iso639_1_to_iso639_2("zz"), None);
|
||||
assert_eq!(iso639_1_to_iso639_2(""), None);
|
||||
assert_eq!(iso639_1_to_iso639_2("e"), None);
|
||||
// A three-letter code is not ISO 639-1 input — that is menu_lang's job.
|
||||
assert_eq!(iso639_1_to_iso639_2("eng"), None);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -634,4 +634,136 @@ mod tests {
|
||||
let (s, e) = find_element(xml, "name", 0).unwrap();
|
||||
assert_eq!(&xml[s..e], "<di:name>Title</di:name>");
|
||||
}
|
||||
|
||||
// ── Malformed / truncated input (untrusted on-disc XML) ────────────────
|
||||
//
|
||||
// These scrapers run on XML lifted out of BD-J jar entries, which is
|
||||
// attacker-controllable. Every scan in this module must terminate and
|
||||
// stay in bounds on truncated or unbalanced input rather than panic.
|
||||
// XML 1.0 §2.3 defines the Name production these boundary rules model.
|
||||
|
||||
/// A quoted attribute value that is never closed must terminate the
|
||||
/// scan at EOF rather than reading past the end of the buffer.
|
||||
#[test]
|
||||
fn attr_unterminated_quoted_value_scan_stops_at_eof() {
|
||||
// The scanner enters the `y="` value and runs off the end looking
|
||||
// for the closing quote; `name` is never found.
|
||||
assert_eq!(attr(r#"<x y="oops"#, "name"), None);
|
||||
assert_eq!(attr("<x y='oops", "name"), None);
|
||||
// The truncated attribute itself has no terminated value either.
|
||||
assert_eq!(attr(r#"<x y="oops"#, "y"), None);
|
||||
}
|
||||
|
||||
/// An attribute name at EOF followed only by whitespace (no `=`) must
|
||||
/// return None, not read past the buffer while skipping that whitespace.
|
||||
#[test]
|
||||
fn attr_name_with_trailing_whitespace_and_no_equals_returns_none() {
|
||||
assert_eq!(attr("<x name ", "name"), None);
|
||||
}
|
||||
|
||||
/// `name=` followed only by whitespace to EOF has no value to return.
|
||||
#[test]
|
||||
fn attr_equals_with_trailing_whitespace_and_no_value_returns_none() {
|
||||
assert_eq!(attr("<x name= ", "name"), None);
|
||||
}
|
||||
|
||||
/// A quoted attribute value is opaque: a `name="..."` pair that appears
|
||||
/// *inside* another attribute's value must never be reported, even when
|
||||
/// it is preceded by whitespace so it would otherwise clear the
|
||||
/// word-boundary check.
|
||||
#[test]
|
||||
fn attr_decoy_name_after_space_inside_quoted_value_is_skipped() {
|
||||
assert_eq!(attr(r#"<x y=" name='decoy'" />"#, "name"), None);
|
||||
// The real attribute after the decoy still resolves.
|
||||
assert_eq!(
|
||||
attr(r#"<x y=" name='decoy'" name="real" />"#, "name"),
|
||||
Some("real".into())
|
||||
);
|
||||
}
|
||||
|
||||
/// XML 1.0 §2.3 NameChar includes `-`, `_` and `.`, so `q-a`, `q_a` and
|
||||
/// `q.a` are each a single attribute name distinct from `a`. Searching
|
||||
/// for `a` must not match the tail of any of them.
|
||||
#[test]
|
||||
fn attr_name_char_boundary_covers_hyphen_underscore_and_dot() {
|
||||
assert_eq!(
|
||||
attr(r#"<x q-a="decoy" a="real" />"#, "a"),
|
||||
Some("real".into())
|
||||
);
|
||||
assert_eq!(
|
||||
attr(r#"<x q_a="decoy" a="real" />"#, "a"),
|
||||
Some("real".into())
|
||||
);
|
||||
assert_eq!(
|
||||
attr(r#"<x q.a="decoy" a="real" />"#, "a"),
|
||||
Some("real".into())
|
||||
);
|
||||
}
|
||||
|
||||
/// An open tag truncated mid-attribute never terminates, so no element
|
||||
/// can be returned — and the attribute walk must not read past EOF.
|
||||
#[test]
|
||||
fn find_element_unterminated_open_tag_returns_none() {
|
||||
assert_eq!(find_element("<x attr=", "x", 0), None);
|
||||
}
|
||||
|
||||
/// A `/` as the final byte of the buffer is not a self-closing marker;
|
||||
/// probing for the `>` that would follow it must stay in bounds.
|
||||
#[test]
|
||||
fn find_element_trailing_slash_at_eof_returns_none() {
|
||||
assert_eq!(find_element("<a /", "a", 0), None);
|
||||
}
|
||||
|
||||
/// `/>` inside a quoted attribute value does not close the element.
|
||||
#[test]
|
||||
fn find_element_quoted_self_close_marker_does_not_end_element() {
|
||||
let xml = r#"<x a="/>"/>"#;
|
||||
let (s, e) = find_element(xml, "x", 0).unwrap();
|
||||
assert_eq!(&xml[s..e], r#"<x a="/>"/>"#);
|
||||
}
|
||||
|
||||
/// An attribute value whose quote is never closed leaves the open tag
|
||||
/// unterminated; the scan must end at EOF and report no element.
|
||||
#[test]
|
||||
fn find_element_unterminated_quoted_attr_returns_none() {
|
||||
assert_eq!(find_element(r#"<x a="oops"#, "x", 0), None);
|
||||
}
|
||||
|
||||
/// A `/` in the middle of an unquoted attribute value is not a
|
||||
/// self-closing marker — only `/>` is.
|
||||
#[test]
|
||||
fn find_element_unquoted_slash_is_not_self_closing() {
|
||||
let xml = "<a href=x/y>body</a>";
|
||||
let (s, e) = find_element(xml, "a", 0).unwrap();
|
||||
assert_eq!(&xml[s..e], "<a href=x/y>body</a>");
|
||||
}
|
||||
|
||||
/// `text` must locate the real end of the open tag: a bare `/` inside
|
||||
/// an unquoted attribute value must not be treated as `/>`, which would
|
||||
/// shift the body start and leak tag bytes into the returned text.
|
||||
#[test]
|
||||
fn text_unquoted_slash_in_attr_does_not_truncate_body() {
|
||||
assert_eq!(text("<x a=b/c>hello</x>", "x"), Some("hello".into()));
|
||||
}
|
||||
|
||||
/// A `>` inside a quoted attribute value must not be mistaken for the
|
||||
/// end of the open tag when `text` computes the body start.
|
||||
#[test]
|
||||
fn text_quoted_gt_in_attr_does_not_truncate_body() {
|
||||
assert_eq!(text(r#"<x a="b>c">hello</x>"#, "x"), Some("hello".into()));
|
||||
}
|
||||
|
||||
/// A close tag truncated mid-name (`</x` with no `>`) is not a close
|
||||
/// tag; matching it must stay in bounds and report no text.
|
||||
#[test]
|
||||
fn text_truncated_close_tag_returns_none() {
|
||||
assert_eq!(text("<x>body</x", "x"), None);
|
||||
}
|
||||
|
||||
/// A `/` in element content is only a close tag when preceded by `<`.
|
||||
/// Body text containing `a/x>` must not be mistaken for `</x>`.
|
||||
#[test]
|
||||
fn text_slash_in_body_is_not_a_close_tag() {
|
||||
assert_eq!(text("<x>a/x> </x>", "x"), Some("a/x>".into()));
|
||||
}
|
||||
}
|
||||
|
||||
+57
-26
@@ -32,7 +32,7 @@
|
||||
//! let opts = libfreemkv::InputOptions::default();
|
||||
//! let mut input = libfreemkv::input("iso://disc.iso", &opts)?;
|
||||
//! let title = input.info().clone();
|
||||
//! let mut output = libfreemkv::output("mkv://Movie.mkv", &title)?;
|
||||
//! let mut output = libfreemkv::output("mkv://Movie.mkv", &title, None)?;
|
||||
//! // Propagate read errors instead of silently stopping on the first one.
|
||||
//! while let Some(frame) = input.read()? {
|
||||
//! output.write(&frame)?;
|
||||
@@ -98,7 +98,7 @@ pub const VERSION_LABEL: &str = concat!(env!("FREEMKV_VERSION"), env!("GIT_SUFFI
|
||||
|
||||
/// The muxing/writing-application string written into MKV output
|
||||
/// (`"freemkv <version> (g<hash>)"`).
|
||||
pub const MUX_APP: &str = concat!("freemkv ", env!("FREEMKV_VERSION"), env!("GIT_SUFFIX"));
|
||||
pub(crate) const MUX_APP: &str = concat!("freemkv ", env!("FREEMKV_VERSION"), env!("GIT_SUFFIX"));
|
||||
|
||||
pub mod aacs;
|
||||
pub(crate) mod clpi;
|
||||
@@ -106,12 +106,15 @@ pub mod consts;
|
||||
pub mod css;
|
||||
pub mod decrypt;
|
||||
pub mod diag;
|
||||
pub mod dirimage;
|
||||
pub mod disc;
|
||||
pub mod drive;
|
||||
pub mod dvdnav;
|
||||
pub mod error;
|
||||
pub mod event;
|
||||
pub mod halt;
|
||||
#[cfg(test)]
|
||||
mod harness;
|
||||
pub mod hex;
|
||||
pub(crate) mod identity;
|
||||
pub(crate) mod ifo;
|
||||
@@ -125,7 +128,9 @@ pub(crate) mod platform;
|
||||
pub mod progress;
|
||||
pub mod scsi;
|
||||
pub mod sector;
|
||||
pub(crate) mod speed;
|
||||
pub mod session;
|
||||
#[cfg(test)]
|
||||
pub(crate) mod testlog;
|
||||
pub(crate) mod udf;
|
||||
pub(crate) mod unlock_bridge;
|
||||
|
||||
@@ -137,38 +142,65 @@ pub(crate) mod unlock_bridge;
|
||||
pub use drive::capture::{
|
||||
CapturedFeature, DriveCapture, capture_drive_data, mask_bytes, mask_string,
|
||||
};
|
||||
pub use drive::{Drive, DriveStatus, find_drive};
|
||||
pub use drive::{Drive, DriveStatus, extract_scsi_context, find_drive};
|
||||
|
||||
// ─── Disc session (drive open + SCSI bring-up hoist) ─────────────────────────
|
||||
//
|
||||
// One entry point that opens a drive and brings the transport up, so consumers
|
||||
// stop hand-rolling `open → wait_ready → init → probe_disc → identify → scan`.
|
||||
// Owns the `Drive` by value; forwards consumer-built key material into
|
||||
// `ScanOptions` (the library derives no certs — see `KeySpec`).
|
||||
pub use session::{
|
||||
DeviceTarget, DiscSession, KeySourceFactory, KeySpec, ResolvedKeys, resolve_keys_for, scan_dir,
|
||||
scan_iso,
|
||||
};
|
||||
|
||||
// ─── Errors ─────────────────────────────────────────────────────────────────
|
||||
//
|
||||
// All fallible APIs return `Result<T, Error>`. `Error` is a typed enum with a
|
||||
// numeric `code()`; **no English text in the library** — applications map
|
||||
// codes to localized messages. See `error.rs` for the full taxonomy.
|
||||
pub use error::{Error, Result};
|
||||
pub use error::{
|
||||
Error, Result, error_code, is_disc_level_no_key, is_halt, is_skippable_title_stub,
|
||||
};
|
||||
|
||||
// ─── Cooperative cancellation ───────────────────────────────────────────────
|
||||
//
|
||||
// One-bit cooperative cancellation token, shared by every long-running loop
|
||||
// in libfreemkv (sweep, patch, mux). Clone it cheaply; pass it by value into
|
||||
// each component; poll `is_cancelled()` inside the loop body.
|
||||
// One-bit cooperative cancellation token, shared by every long-running loop —
|
||||
// libfreemkv's mux, and the recovery passes (sweep/patch) that now live in the
|
||||
// freemkv-engine crate. Clone it cheaply; pass it by value into each component;
|
||||
// poll `is_cancelled()` inside the loop body.
|
||||
pub use halt::Halt;
|
||||
|
||||
// Generic bounded producer/consumer primitive used by sweep, patch, and
|
||||
// mux to overlap reads with writes via a dedicated consumer thread.
|
||||
// Generic bounded producer/consumer primitive used by the mux pipeline (and,
|
||||
// via this re-export, by the engine's sweep/patch recovery passes) to overlap
|
||||
// reads with writes via a dedicated consumer thread.
|
||||
// `Pipeline::spawn(name, depth, sink)` spawns a named consumer; `pipe.send(item)`
|
||||
// pushes one item with back-pressure; `pipe.finish()` joins the
|
||||
// consumer and surfaces its `close()` output. Callers implement `Sink`
|
||||
// to define per-item behaviour and end-of-stream finalisation.
|
||||
//
|
||||
// `DEFAULT_PIPELINE_DEPTH` (=4) is for callers without specific needs;
|
||||
// most should use READ_PIPELINE_DEPTH or WRITE_PIPELINE_DEPTH instead.
|
||||
// most should use WRITE_PIPELINE_DEPTH instead.
|
||||
// Patch uses `WRITE_THROUGH_DEPTH` (=1). Returning `Flow::Stop` from
|
||||
// `apply` ends the consumer cleanly (still calls `close()`).
|
||||
pub use io::pipeline::{
|
||||
DEFAULT_PIPELINE_DEPTH, Flow, Pipeline, READ_PIPELINE_DEPTH, Sink, WRITE_PIPELINE_DEPTH,
|
||||
WRITE_THROUGH_DEPTH,
|
||||
DEFAULT_PIPELINE_DEPTH, Flow, Pipeline, Sink, WRITE_PIPELINE_DEPTH, WRITE_THROUGH_DEPTH,
|
||||
};
|
||||
|
||||
// ─── Bounded-cache buffered file writer ─────────────────────────────────────
|
||||
//
|
||||
// Drop-in `std::fs::File` replacement used everywhere the lib writes large
|
||||
// sequential output (mux, extract, sweep, patch) — drains dirty pages
|
||||
// continuously instead of bursting. General I/O infra, not recovery policy;
|
||||
// promoted to `pub` so freemkv-engine's relocated sweep/patch can use it too.
|
||||
pub use io::WritebackFile;
|
||||
/// Write an image-level source out as a sector image — what an `iso://`
|
||||
/// DESTINATION means for any source that is not a physical drive. Drive sources
|
||||
/// go through `freemkv_engine::copy`, which is the recovery path; see
|
||||
/// [`io::image_writer`] for why the two are deliberately separate.
|
||||
pub use io::image_writer::write_image;
|
||||
|
||||
// ─── Drive events (low-level callbacks) ─────────────────────────────────────
|
||||
pub use event::{BatchSizeReason, Event, EventKind};
|
||||
pub use identity::DriveId;
|
||||
@@ -187,10 +219,7 @@ pub use identity::DriveId;
|
||||
// don't touch `DecryptKeys` directly — `DiscStream::new(reader, title, keys, …)`
|
||||
// accepts whatever `Disc::decrypt_keys()` returned. `decrypt_sectors()` is
|
||||
// for callers that operate on raw sector buffers (e.g. ISO patching).
|
||||
pub use decrypt::{
|
||||
AacsKeyMap, DecryptKeys, decrypt_sectors, decrypt_sectors_mapped, decrypt_threads,
|
||||
set_decrypt_threads,
|
||||
};
|
||||
pub use decrypt::{AacsKeyMap, DecryptKeys, decrypt_sectors, decrypt_threads, set_decrypt_threads};
|
||||
|
||||
// ─── Disc structure ─────────────────────────────────────────────────────────
|
||||
//
|
||||
@@ -203,12 +232,12 @@ pub use decrypt::{
|
||||
// — not the `pes::Stream` trait re-exported below as `PesStream`. Two
|
||||
// different concepts, the same short name; the trait gets the `Pes`
|
||||
// prefix at the crate root to keep both addressable.
|
||||
pub use dirimage::DirImage;
|
||||
pub use disc::{
|
||||
AacsState, AudioChannels, AudioStream, Clip, Codec, ColorSpace, ContentFormat, DamageSeverity,
|
||||
Disc, DiscFormat, DiscId, DiscTitle, DriveCredentials, Extent, ExtractOptions, ExtractResult,
|
||||
FileResult, FrameRate, HdrFormat, Key, KeyOrigin, LabelPurpose, LabelQualifier, PatchOptions,
|
||||
PatchOutcome, Resolution, SampleRate, ScanOptions, Stream, SubtitleStream, SweepOptions,
|
||||
VideoStream, classify_damage,
|
||||
AacsState, AudioChannels, AudioStream, Clip, Codec, ColorSpace, ContentFormat, Disc,
|
||||
DiscFormat, DiscId, DiscTitle, DriveCredentials, Extent, ExtractOptions, ExtractResult,
|
||||
FileResult, FrameRate, HdrFormat, Key, KeyOrigin, LabelPurpose, LabelQualifier, Resolution,
|
||||
SampleRate, ScanOptions, Stream, SubtitleStream, VideoStream,
|
||||
};
|
||||
pub use keysource::{DiscInputs, KeySource, read_encrypted_units, resolve_and_apply};
|
||||
|
||||
@@ -241,6 +270,8 @@ pub use mux::NullStream;
|
||||
pub use mux::StdioStream;
|
||||
pub use mux::WriteSeek;
|
||||
pub use mux::{InputOptions, StreamUrl, input, output, parse_url};
|
||||
pub use mux::{Medium, SourceInfo};
|
||||
pub use mux::{Mp4FitReport, Mp4Sink, Mp4SkipReason, mp4_fit_report};
|
||||
|
||||
// ─── Lower-level surfaces ───────────────────────────────────────────────────
|
||||
//
|
||||
@@ -252,10 +283,10 @@ pub use mux::{InputOptions, StreamUrl, input, output, parse_url};
|
||||
// `SectorSource` to get plaintext sectors out.
|
||||
pub use mux::build_iso_pipeline;
|
||||
pub use mux::resolve_mux_key_map;
|
||||
pub use scsi::{DriveInfo, ScsiSense, ScsiTransport, drive_has_disc, list_drives};
|
||||
pub use mux::select::{PidFilter, StreamSelection};
|
||||
pub use mux::{MuxEvents, MuxInput, MuxOptions, MuxOutcome, mux_stream};
|
||||
pub use scsi::{DriveInfo, ScsiSense, ScsiTransport, SenseFamily, drive_has_disc, list_drives};
|
||||
pub use sector::{
|
||||
DecryptingSectorSource, FileSectorSink, FileSectorSource, KeyFetch, PrefetchedSectorSource,
|
||||
SectorSink, SectorSource,
|
||||
DecryptingSectorSource, FileSectorSource, KeyFetch, PrefetchedSectorSource, SectorSource,
|
||||
};
|
||||
pub use speed::DriveSpeed;
|
||||
pub use udf::{UdfFs, read_filesystem};
|
||||
|
||||
+316
-3
@@ -39,6 +39,18 @@ pub(crate) struct PlaylistMark {
|
||||
pub timestamp: u32,
|
||||
}
|
||||
|
||||
impl PlaylistMark {
|
||||
/// Is this mark a chapter entry point?
|
||||
///
|
||||
/// Only `mark_type == 1` counts. Type 0 is reserved and type 2 is a link
|
||||
/// point, and neither is a chapter. Every chapter filter in the crate goes
|
||||
/// through here: two hand-rolled copies had already drifted, one testing
|
||||
/// `<= 1` and silently counting reserved marks as chapters.
|
||||
pub(crate) fn is_chapter_mark(&self) -> bool {
|
||||
self.mark_type == 1
|
||||
}
|
||||
}
|
||||
|
||||
/// A play item — one clip reference with in/out times.
|
||||
#[derive(Debug)]
|
||||
pub(crate) struct PlayItem {
|
||||
@@ -432,14 +444,13 @@ fn parse_stream_entry(item: &[u8], pos: usize, stream_type: u8) -> Option<(Strea
|
||||
}
|
||||
}
|
||||
}
|
||||
STREAM_CATEGORY_PG_SUBTITLE => {
|
||||
STREAM_CATEGORY_PG_SUBTITLE
|
||||
// PG: coding_type(1) + language(3).
|
||||
// IG is parsed only to advance spos and is then discarded by the
|
||||
// caller, so it deliberately has no arm here.
|
||||
if sa.len() >= 4 {
|
||||
if sa.len() >= 4 => {
|
||||
language = String::from_utf8_lossy(&sa[1..4]).to_string();
|
||||
}
|
||||
}
|
||||
_ => {}
|
||||
}
|
||||
|
||||
@@ -1417,4 +1428,306 @@ mod tests {
|
||||
data[8..12].copy_from_slice(&40u32.to_be_bytes()); // playlist_start = 40 = len
|
||||
assert!(parse(&data).is_err());
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// Added: STN-table block alignment and section-boundary hardening.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Build an MPLS from raw PlayItem bodies, with no PlayListMark section
|
||||
/// (mark_start = 0). Lets a test control item_length exactly.
|
||||
fn build_mpls_raw_items(items: &[Vec<u8>]) -> Vec<u8> {
|
||||
let playlist_start: u32 = 40;
|
||||
let mut buf = Vec::new();
|
||||
buf.extend_from_slice(b"MPLS0200");
|
||||
buf.extend_from_slice(&playlist_start.to_be_bytes());
|
||||
buf.extend_from_slice(&[0u8; 28]); // mark_start = 0, then padding
|
||||
let pl_start = buf.len();
|
||||
buf.extend_from_slice(&[0u8; 4]); // PlayList length placeholder
|
||||
buf.extend_from_slice(&[0u8; 2]); // reserved
|
||||
buf.extend_from_slice(&(items.len() as u16).to_be_bytes());
|
||||
buf.extend_from_slice(&[0u8; 2]); // num_sub_paths
|
||||
for it in items {
|
||||
buf.extend_from_slice(&(it.len() as u16).to_be_bytes());
|
||||
buf.extend_from_slice(it);
|
||||
}
|
||||
let pl_len = (buf.len() - pl_start - 4) as u32;
|
||||
buf[pl_start..pl_start + 4].copy_from_slice(&pl_len.to_be_bytes());
|
||||
buf
|
||||
}
|
||||
|
||||
/// The 20 bytes a PlayItem needs for clip_id(5) + codec_id(4) +
|
||||
/// connection_condition(1) + reserved(2) + IN_time(4) + OUT_time(4).
|
||||
fn play_item_20(clip: &[u8; 5], cc: u8, in_t: u32, out_t: u32) -> Vec<u8> {
|
||||
let mut it = Vec::new();
|
||||
it.extend_from_slice(clip);
|
||||
it.extend_from_slice(b"M2TS");
|
||||
it.push(cc);
|
||||
it.extend_from_slice(&[0u8; 2]);
|
||||
it.extend_from_slice(&in_t.to_be_bytes());
|
||||
it.extend_from_slice(&out_t.to_be_bytes());
|
||||
assert_eq!(it.len(), 20);
|
||||
it
|
||||
}
|
||||
|
||||
/// A PlayItem body of exactly 20 bytes carries every field the parser
|
||||
/// reads (the last is OUT_time at [16..20]), so it must be RECORDED,
|
||||
/// not skipped — and it has no STN table, which starts at byte 32.
|
||||
#[test]
|
||||
fn play_item_of_exactly_20_bytes_is_recorded_without_stn() {
|
||||
let data = build_mpls_raw_items(&[play_item_20(b"00007", 5, 90_000, 180_000)]);
|
||||
let pl = parse(&data).expect("a 20-byte PlayItem must parse");
|
||||
assert_eq!(pl.play_items.len(), 1);
|
||||
assert_eq!(pl.play_items[0].clip_id, "00007");
|
||||
assert_eq!(pl.play_items[0].in_time, 90_000);
|
||||
assert_eq!(pl.play_items[0].out_time, 180_000);
|
||||
assert_eq!(pl.play_items[0].connection_condition, 5);
|
||||
assert!(pl.streams.is_empty(), "no STN table exists below byte 32");
|
||||
}
|
||||
|
||||
/// A 40-byte MPLS whose PlayList section is exactly its 10-byte header
|
||||
/// (length(4)+reserved(2)+num_play_items(2)+num_sub_paths(2)) ending at
|
||||
/// EOF is structurally complete, not truncated: nothing the parser reads
|
||||
/// lies past the buffer, so it must parse to an empty playlist.
|
||||
#[test]
|
||||
fn minimum_size_mpls_with_empty_playlist_header_parses() {
|
||||
let mut data = vec![0u8; 40];
|
||||
data[0..4].copy_from_slice(b"MPLS");
|
||||
data[4..8].copy_from_slice(b"0200");
|
||||
data[8..12].copy_from_slice(&30u32.to_be_bytes()); // playlist_start + 10 == 40
|
||||
// mark_start (12..16) stays 0; num_play_items at data[36..38] is 0.
|
||||
let pl = parse(&data).expect("40-byte MPLS with a complete PlayList header must parse");
|
||||
assert!(pl.play_items.is_empty());
|
||||
assert!(pl.streams.is_empty());
|
||||
assert!(pl.marks.is_empty());
|
||||
}
|
||||
|
||||
/// A mark_start of 0 means "no PlayListMark section". The file header
|
||||
/// bytes at offset 0 must not be decoded as one — data[4..6] is the
|
||||
/// version string "02", which as a big-endian num_marks would be 12338.
|
||||
#[test]
|
||||
fn mark_start_zero_does_not_parse_header_as_marks() {
|
||||
let data = build_mpls_raw_items(&[play_item_20(b"00007", 1, 0, 90_000)]);
|
||||
assert_eq!(
|
||||
&data[12..16],
|
||||
&[0, 0, 0, 0],
|
||||
"fixture must have mark_start 0"
|
||||
);
|
||||
let pl = parse(&data).expect("should parse");
|
||||
assert!(
|
||||
pl.marks.is_empty(),
|
||||
"mark_start == 0 must mean absent, got {} marks",
|
||||
pl.marks.len()
|
||||
);
|
||||
}
|
||||
|
||||
/// Full STN table walk with every category populated and DISTINCT
|
||||
/// counts, so no count byte can be read from a neighbour's offset
|
||||
/// without changing the result.
|
||||
///
|
||||
/// Each secondary block is followed by its reference block(s), which
|
||||
/// per the BD STN table are num_refs(1) + reserved(1) + one byte per
|
||||
/// ref + one padding byte when the ref count is odd. Every ref count
|
||||
/// here is 1 — the value that distinguishes `n % 2` (=1) from `n / 2`
|
||||
/// (=0) — so a wrong skip length misaligns the cursor and every
|
||||
/// following stream decodes from the wrong offset. IG entries are
|
||||
/// consumed to keep the cursor aligned but never retained.
|
||||
#[test]
|
||||
fn full_stn_table_block_alignment() {
|
||||
let mut entries: Vec<Vec<u8>> = vec![
|
||||
build_stream_entry_video(0x1011, 0x1B, 6, 1, None),
|
||||
build_stream_entry_audio(0x1100, 0x83, 6, 1, b"eng"),
|
||||
build_stream_entry_audio(0x1101, 0x86, 3, 1, b"fra"),
|
||||
build_stream_entry_pg(0x1200, 0x90, b"eng"),
|
||||
build_stream_entry_pg(0x1201, 0x90, b"fra"),
|
||||
build_stream_entry_pg(0x1202, 0x90, b"deu"),
|
||||
];
|
||||
for i in 0..4u16 {
|
||||
entries.push(build_stream_entry_pg(0x1400 + i, 0x91, b"eng"));
|
||||
}
|
||||
// secondary audio + its secondary-audio ref block (1 ref → 1 pad)
|
||||
let mut sec_audio = build_stream_entry_audio(0x1A00, 0x83, 3, 1, b"spa");
|
||||
sec_audio.extend_from_slice(&[1, 0, 0x55, 0x00]);
|
||||
entries.push(sec_audio);
|
||||
// secondary video + audio-ref block + PiP-PG-ref block
|
||||
let mut sec_video = build_stream_entry_video(0x1B00, 0x1B, 4, 1, None);
|
||||
sec_video.extend_from_slice(&[1, 0, 0x55, 0x00]);
|
||||
sec_video.extend_from_slice(&[1, 0, 0x66, 0x00]);
|
||||
entries.push(sec_video);
|
||||
// PiP PG + its ref block
|
||||
let mut pip_pg = build_stream_entry_pg(0x1C00, 0x90, b"jpn");
|
||||
pip_pg.extend_from_slice(&[1, 0, 0x77, 0x00]);
|
||||
entries.push(pip_pg);
|
||||
// Dolby Vision enhancement layer
|
||||
entries.push(build_stream_entry_video(0x1015, 0x24, 8, 1, Some(0x12)));
|
||||
|
||||
let data = build_mpls(
|
||||
&[(b"00001", 1, 0, 9_000_000)],
|
||||
(1, 2, 3, 4, 1, 1, 1, 1),
|
||||
&entries,
|
||||
);
|
||||
let pl = parse(&data).expect("should parse");
|
||||
|
||||
let got: Vec<(u8, u16, bool)> = pl
|
||||
.streams
|
||||
.iter()
|
||||
.map(|s| (s.stream_type, s.pid, s.secondary))
|
||||
.collect();
|
||||
assert_eq!(
|
||||
got,
|
||||
vec![
|
||||
(1, 0x1011, false), // primary video
|
||||
(2, 0x1100, false), // primary audio ×2
|
||||
(2, 0x1101, false),
|
||||
(3, 0x1200, false), // PG ×3
|
||||
(3, 0x1201, false),
|
||||
(3, 0x1202, false),
|
||||
// the 4 IG entries are consumed and discarded
|
||||
(5, 0x1A00, true), // secondary audio
|
||||
(6, 0x1B00, true), // secondary video
|
||||
(3, 0x1C00, true), // PiP PG
|
||||
(7, 0x1015, true), // Dolby Vision EL
|
||||
]
|
||||
);
|
||||
// Languages prove each entry was decoded at its own offset.
|
||||
assert_eq!(pl.streams[1].language, "eng");
|
||||
assert_eq!(pl.streams[2].language, "fra");
|
||||
assert_eq!(pl.streams[6].language, "spa");
|
||||
assert_eq!(pl.streams[8].language, "jpn");
|
||||
}
|
||||
|
||||
/// A secondary block whose stream entry ends exactly at the end of the
|
||||
/// PlayItem has no reference block at all; the count byte must not be
|
||||
/// read from one-past-the-end. Covers all three secondary blocks that
|
||||
/// carry reference data.
|
||||
#[test]
|
||||
fn secondary_ref_block_at_item_end_is_not_read() {
|
||||
let video = build_stream_entry_video(0x1011, 0x1B, 6, 1, None);
|
||||
|
||||
// Secondary audio is the last entry, with no ref bytes following.
|
||||
let sec_audio = build_stream_entry_audio(0x1A00, 0x83, 3, 1, b"eng");
|
||||
let data = build_mpls(
|
||||
&[(b"00001", 1, 0, 9_000_000)],
|
||||
(1, 0, 0, 0, 1, 0, 0, 0),
|
||||
&[video.clone(), sec_audio],
|
||||
);
|
||||
let pl = parse(&data).expect("secondary audio at item end");
|
||||
assert_eq!(pl.streams.len(), 2);
|
||||
assert_eq!(pl.streams[1].pid, 0x1A00);
|
||||
|
||||
// Secondary video is the last entry, with no ref bytes following.
|
||||
let sec_video = build_stream_entry_video(0x1B00, 0x1B, 4, 1, None);
|
||||
let data = build_mpls(
|
||||
&[(b"00001", 1, 0, 9_000_000)],
|
||||
(1, 0, 0, 0, 0, 1, 0, 0),
|
||||
&[video.clone(), sec_video.clone()],
|
||||
);
|
||||
let pl = parse(&data).expect("secondary video at item end");
|
||||
assert_eq!(pl.streams.len(), 2);
|
||||
assert_eq!(pl.streams[1].pid, 0x1B00);
|
||||
|
||||
// Secondary video whose audio-ref block ends exactly at item end, so
|
||||
// the PiP-PG ref count byte would sit one past it.
|
||||
let mut sec_video_arefs = sec_video;
|
||||
sec_video_arefs.extend_from_slice(&[0, 0]); // n_arefs = 0, reserved
|
||||
let data = build_mpls(
|
||||
&[(b"00001", 1, 0, 9_000_000)],
|
||||
(1, 0, 0, 0, 0, 1, 0, 0),
|
||||
&[video.clone(), sec_video_arefs],
|
||||
);
|
||||
let pl = parse(&data).expect("secondary video aref block at item end");
|
||||
assert_eq!(pl.streams.len(), 2);
|
||||
assert_eq!(pl.streams[1].pid, 0x1B00);
|
||||
|
||||
// PiP PG is the last entry, with no ref bytes following.
|
||||
let pip_pg = build_stream_entry_pg(0x1C00, 0x90, b"jpn");
|
||||
let data = build_mpls(
|
||||
&[(b"00001", 1, 0, 9_000_000)],
|
||||
(1, 0, 0, 0, 0, 0, 1, 0),
|
||||
&[video, pip_pg],
|
||||
);
|
||||
let pl = parse(&data).expect("PiP PG at item end");
|
||||
assert_eq!(pl.streams.len(), 2);
|
||||
assert_eq!(pl.streams[1].pid, 0x1C00);
|
||||
}
|
||||
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
// parse_stream_entry bounds, exercised directly.
|
||||
// ─────────────────────────────────────────────────────────────────────
|
||||
|
||||
/// Fewer than 2 bytes remain for the stream_entry header
|
||||
/// (length(1) + stream_entry_type(1)) → None, without reading either.
|
||||
#[test]
|
||||
fn stream_entry_header_past_end_is_none() {
|
||||
let item = [0u8; 8];
|
||||
for pos in 7..12usize {
|
||||
assert!(
|
||||
parse_stream_entry(&item, pos, STREAM_CATEGORY_VIDEO).is_none(),
|
||||
"pos={pos}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The stream_attributes header (length(1) + coding_type(1)) lies past
|
||||
/// the end of the PlayItem → None, without reading the length byte.
|
||||
#[test]
|
||||
fn stream_attributes_header_past_end_is_none() {
|
||||
// se_len = 3 → se_end = 4 == item.len(); the sa length byte would be
|
||||
// at item[4] and the coding type at item[5].
|
||||
let item = [3u8, STREAM_ENTRY_PLAYITEM_CLIP, 0x10, 0x11];
|
||||
assert!(parse_stream_entry(&item, 0, STREAM_CATEGORY_VIDEO).is_none());
|
||||
}
|
||||
|
||||
/// A declared stream_attributes length of 0 has no coding_type byte and
|
||||
/// must be rejected — even when the (empty) attribute region is itself
|
||||
/// in bounds.
|
||||
#[test]
|
||||
fn zero_length_attributes_in_bounds_is_none() {
|
||||
// se_len = 3 → se_end = 4; sa_len = item[4] = 0 → sa_end = 5 ≤ 6.
|
||||
let item = [3u8, STREAM_ENTRY_PLAYITEM_CLIP, 0x10, 0x11, 0, 0];
|
||||
assert!(parse_stream_entry(&item, 0, STREAM_CATEGORY_VIDEO).is_none());
|
||||
}
|
||||
|
||||
/// stream_attributes of exactly 1 byte carries only the coding_type.
|
||||
/// That is the minimum the parser accepts, so the entry is returned
|
||||
/// with its PID and coding_type and no format-specific fields — for a
|
||||
/// PG stream the 3-byte language must NOT be read past the attributes.
|
||||
#[test]
|
||||
fn one_byte_stream_attributes_yields_bare_entry() {
|
||||
// se_len = 3 → se_end = 4; sa_len = 1 → sa_end = 6 == item.len().
|
||||
let item = [3u8, STREAM_ENTRY_PLAYITEM_CLIP, 0x10, 0x11, 1, 0x1B];
|
||||
let (entry, next) =
|
||||
parse_stream_entry(&item, 0, STREAM_CATEGORY_VIDEO).expect("1-byte attrs are valid");
|
||||
assert_eq!(entry.pid, 0x1011);
|
||||
assert_eq!(entry.coding_type, 0x1B);
|
||||
assert_eq!(entry.video_format, 0);
|
||||
assert_eq!(entry.video_rate, 0);
|
||||
assert_eq!(next, 6);
|
||||
|
||||
let pg = [3u8, STREAM_ENTRY_PLAYITEM_CLIP, 0x12, 0x00, 1, 0x90];
|
||||
let (entry, _) = parse_stream_entry(&pg, 0, STREAM_CATEGORY_PG_SUBTITLE)
|
||||
.expect("1-byte PG attrs are valid");
|
||||
assert_eq!(entry.pid, 0x1200);
|
||||
assert_eq!(entry.coding_type, 0x90);
|
||||
assert_eq!(entry.language, "");
|
||||
}
|
||||
|
||||
/// Type 0 is reserved and type 2 is a link point; neither is a chapter.
|
||||
/// `labels::collect_chapter_summary` used to filter on `mark_type <= 1`,
|
||||
/// which counted reserved marks and inflated the public `chapter_count`
|
||||
/// (and let a playlist whose only marks are reserved pass the
|
||||
/// `chapter_count == 0` skip). Both call sites now share this predicate.
|
||||
#[test]
|
||||
fn only_entry_marks_count_as_chapters() {
|
||||
let mk = |mark_type| PlaylistMark {
|
||||
mark_type,
|
||||
play_item_ref: 0,
|
||||
timestamp: 0,
|
||||
};
|
||||
assert!(
|
||||
!mk(0).is_chapter_mark(),
|
||||
"type 0 is reserved, not a chapter"
|
||||
);
|
||||
assert!(mk(1).is_chapter_mark());
|
||||
assert!(!mk(2).is_chapter_mark(), "type 2 is a link point");
|
||||
}
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user