Compare commits
34
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b3eea75365 | ||
|
|
07f7ed4259 | ||
|
|
dc6ee3dc9b | ||
|
|
7fd95ddede | ||
|
|
350b179c9e | ||
|
|
0cc4c4f3a3 | ||
|
|
c5b2dc0f55 | ||
|
|
8db9a6f52f | ||
|
|
31f249ae04 | ||
|
|
d6e1e3bf86 | ||
|
|
6992f0e783 | ||
|
|
bb920b5bd0 | ||
|
|
802997439d | ||
|
|
495eaa4ab8 | ||
|
|
bc8e67888e | ||
|
|
706eea8709 | ||
|
|
163ce54b77 | ||
|
|
d293646f69 | ||
|
|
5416788274 | ||
|
|
ad2a75d9a0 | ||
|
|
2b921fafe4 | ||
|
|
98c0e4dba2 | ||
|
|
557b3edab4 | ||
|
|
cf540f1ecc | ||
|
|
e0412329ff | ||
|
|
97558c259f | ||
|
|
948d53b67e | ||
|
|
54932e97c6 | ||
|
|
745c4f62ce | ||
|
|
e2f8ef2918 | ||
|
|
393b931fff | ||
|
|
d759ef32f1 | ||
|
|
4d090d9a81 | ||
|
|
39327beea7 |
@@ -184,6 +184,46 @@ out="$(run_report "$root")"
|
||||
assert_contains "same-line annotation removed: counts 2, so it was worth 1" "$out" \
|
||||
" baseline DEVIATION: the tree carries 2 tests marked \`@FailsOnEmulatorApi37\` but the baseline says 3 — update FAILS_ON_EMULATOR_API37_BASELINE"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 4. A run the abort truncated, with fewer failures than the baseline: NOT a deviation.
|
||||
#
|
||||
# `expected` comes from `Starting N tests`, printed before anything can abort, so it still
|
||||
# answers "is the marked set the size the baseline says". `failed` is a tally of what actually
|
||||
# ran, and on a truncated run the tests after the abort never start. Measured on 2026-09-05, two
|
||||
# api37-debug dispatches of the same four marked tests: 4/4/4 and then 4/3/3. Announcing the
|
||||
# second as "one now passes" is the wrong reading, and #120 is the standing lesson about a notice
|
||||
# that is wrong often enough to be skimmed past.
|
||||
# ---------------------------------------------------------------------------
|
||||
root="$(make_root "$FIXTURE_DIR" 3)"
|
||||
cat > "$root/gradle.log" <<'TRUNCATED'
|
||||
> Task :app:connectedDebugAndroidTest
|
||||
Starting 3 tests on test(AVD) - 16
|
||||
There was 2 failure(s).
|
||||
Test run failed to complete. Expected 3 tests, received 2. onError: commandError=false message=INSTRUMENTATION_ABORTED: System has crashed.
|
||||
TRUNCATED
|
||||
out="$(run_report "$root")"
|
||||
assert_contains "truncated run: the truncation is reported" "$out" ' completed cleanly: no'
|
||||
assert_absent "truncated run: the short failure count is not a deviation" "$out" 'tests failed, the baseline is'
|
||||
# And the match line has to say what actually happened rather than repeat the baseline: PR #245's
|
||||
# advisory leg printed `failed: 4` three lines above `matches (5 expected, 5 failed)`.
|
||||
assert_contains "truncated run: the match line does not claim the baseline's failure count" "$out" \
|
||||
' baseline: matches (3 expected; 2 of 3 failed, on a run the abort truncated — not compared)'
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# 5. The same short failure count on a run that finished IS a deviation.
|
||||
#
|
||||
# The pair is the point: case 4 must not have bought its quiet by disabling the check outright.
|
||||
# ---------------------------------------------------------------------------
|
||||
root="$(make_root "$FIXTURE_DIR" 3)"
|
||||
cat > "$root/gradle.log" <<'CLEAN'
|
||||
> Task :app:connectedDebugAndroidTest
|
||||
Starting 3 tests on test(AVD) - 16
|
||||
There was 2 failure(s).
|
||||
CLEAN
|
||||
out="$(run_report "$root")"
|
||||
assert_contains "clean run, short by one: the deviation fires" "$out" \
|
||||
'2 tests failed, the baseline is 3'
|
||||
|
||||
echo
|
||||
if [ "$failures" -eq 0 ]; then
|
||||
echo "e2e-report-shape-test.sh: all checks passed"
|
||||
|
||||
@@ -252,8 +252,22 @@ if [ -n "$baseline" ]; then
|
||||
if [ "$expected" != "unknown" ] && [ "$expected" != "$baseline" ]; then
|
||||
deviations+=("the runner started $expected tests, the baseline is $baseline")
|
||||
fi
|
||||
# `expected` is compared on every run and `failed` only on a run that finished, and the
|
||||
# difference is the truncation this file already records rather than compares. `expected`
|
||||
# comes from `Starting N tests`, which is printed before anything can abort, so it answers
|
||||
# "is the marked set the size the baseline says" whatever happens afterwards. `failed` is a
|
||||
# tally of what actually ran: on a truncated run the tests after the abort never start, so
|
||||
# comparing it to the baseline announces a deviation about the framework dying rather than
|
||||
# about the test list. Measured on 2026-09-05, two api37-debug dispatches of the same four
|
||||
# marked tests: 4/4/4 and then 4/3/3, the second having lost the last test to the abort.
|
||||
# Announcing that as "one now passes" is exactly the wrong reading, and #120 is the standing
|
||||
# lesson about a notice that is wrong often enough to be skimmed past.
|
||||
if [ "$failed" != "unknown" ] && [ "$failed" != "$baseline" ]; then
|
||||
deviations+=("$failed tests failed, the baseline is $baseline — every test carrying the marker is expected to fail on this image, so fewer means one now passes and more means a new one joined")
|
||||
if [ "$completed" = "**no**" ]; then
|
||||
echo "::debug::$failed of $baseline marked tests failed, on a run the abort truncated — not compared"
|
||||
else
|
||||
deviations+=("$failed tests failed, the baseline is $baseline — every test carrying the marker is expected to fail on this image, so fewer means one now passes and more means a new one joined")
|
||||
fi
|
||||
fi
|
||||
fi
|
||||
if [ -n "$marked" ] && [ "$marked" != "$baseline" ]; then
|
||||
@@ -283,7 +297,16 @@ if [ -n "$failed_names" ]; then
|
||||
fi
|
||||
if [ "$advisory" = "yes" ]; then
|
||||
if [ "${#deviations[@]}" -eq 0 ]; then
|
||||
echo " baseline: matches ($baseline expected, $baseline failed)"
|
||||
# Two spellings, because one of them would be a lie half the time. `$baseline expected,
|
||||
# $baseline failed` is only true of a run that finished; on a truncated one `failed` is a
|
||||
# tally of the tests that got to run before the framework died, and printing the baseline in
|
||||
# its place claims a number nobody measured. Seen on PR #245's advisory leg, which reported
|
||||
# `failed: 4` three lines above `matches (5 expected, 5 failed)`.
|
||||
if [ "$failed" != "unknown" ] && [ "$failed" != "$baseline" ]; then
|
||||
echo " baseline: matches ($baseline expected; $failed of $baseline failed, on a run the abort truncated — not compared)"
|
||||
else
|
||||
echo " baseline: matches ($baseline expected, $baseline failed)"
|
||||
fi
|
||||
else
|
||||
printf ' baseline DEVIATION: %s\n' "${deviations[@]}"
|
||||
fi
|
||||
|
||||
+66
-50
@@ -52,40 +52,72 @@ WEDGE_TIMEOUT=1200
|
||||
# the same shape as E2E_EXTRA_GRADLE_ARGS below. The other four E2E legs run byte-identical
|
||||
# commands with it unset.
|
||||
#
|
||||
# WHY IT RUNS HERE, BEFORE THE LOGCAT STREAM: `adb shell stop` ends the `adb logcat` started
|
||||
# below, and nothing restarts it, so a disable performed after that point would cost this leg
|
||||
# its whole diagnostic story for the part of the run that matters. Everything this function
|
||||
# counts comes from `adb logcat -d -b crash`, which is a fresh read each time and independent
|
||||
# of the stream.
|
||||
# WHY IT RUNS HERE, BEFORE THE LOGCAT STREAM: it is a 45-second wait, and the stream below is
|
||||
# meant to cover the suite rather than the wait. Everything this function counts comes from
|
||||
# `adb logcat -d -b crash`, a fresh read each time and independent of the stream. (The original
|
||||
# reason was stronger and no longer applies: `adb shell stop` would have ended the streamed
|
||||
# `adb logcat` and nothing restarts it. There is no `stop` here any more -- see below.)
|
||||
#
|
||||
# WHAT IT IS FOR: the android-37.x images abort surfaceflinger from RegionSamplingThread inside
|
||||
# their own gralloc mapper (docs/api-37-emulator-crash.md). surfaceflinger is a critical service,
|
||||
# so init SIGKILLs zygote with it and the framework restarts under the run -- Gradle then reports
|
||||
# WHAT IT IS FOR -- AND THE NAME IS NOW WRONG, WHICH IS WHY THIS PARAGRAPH IS LONG.
|
||||
# The android-37.x images abort surfaceflinger from RegionSamplingThread inside their own gralloc
|
||||
# mapper (docs/api-37-emulator-crash.md). surfaceflinger is a critical service, so init SIGKILLs
|
||||
# zygote with it and the framework restarts under the run -- Gradle then reports
|
||||
# `cmd: Can't find service: package` and `Starting 0 tests`. RegionSamplingThread exists only
|
||||
# because SystemUI registers a nav-bar luma-sampling listener, so removing the package removes
|
||||
# the whole chain. Measured cadence of those kills: 20-90 s apart, median 60-70 s, three to five
|
||||
# in a four-minute window -- fast enough that install and instrumentation start-up do not fit
|
||||
# inside one gap.
|
||||
# because SystemUI registers a nav-bar luma-sampling listener, so this was written to remove the
|
||||
# package and with it the whole chain. Measured cadence of those kills on `-gpu host`: 20-90 s
|
||||
# apart, median 60-70 s, three to five in a four-minute window.
|
||||
#
|
||||
# **THE DISABLE HALF OF THAT HAS NEVER WORKED, AND THE QUIET WINDOW IS WHAT THE LEG ACTUALLY
|
||||
# GETS.** Measured 2026-09-05, two ways that agree:
|
||||
#
|
||||
# - On CI, in the gating leg of run 34006456986: `pm disable-user` is accepted at 02:28:37.9 and
|
||||
# `com.android.systemui` really is in `pm list packages -d` at 02:29:33 -- and SystemUI is
|
||||
# started anyway at 02:28:39.5 and again at 02:28:52.3, the second of which (pid 4275) is
|
||||
# alive for the whole instrumentation run, logging `WindowManagerShell ...
|
||||
# app=com.android.systemui` minutes after this function prints its final line.
|
||||
# - Locally on android-37.0, with the package verified disabled before AND after a deliberate
|
||||
# `stop; start`: `com.android.systemui` comes up 3 s after `system_server` regardless.
|
||||
#
|
||||
# So `pm disable-user --user 0 com.android.systemui` does not stop SystemUI starting on this
|
||||
# image, whatever else happens. The name `E2E_DISABLE_SYSTEM_UI` and the name of this function are
|
||||
# kept because the matrix row, both workflows and two documents refer to them, and a rename would
|
||||
# touch all of that to no benefit -- read this comment, not the name.
|
||||
#
|
||||
# WHAT IS LEFT IS LOAD-BEARING, so do not delete the function as dead weight. It is the 45-second
|
||||
# window with zero new `hasReadColorBufferDma` aborts. The boot-time aborts land close together --
|
||||
# 02:28:18 and 02:28:43 in that same run -- and the wait is what puts instrumentation (02:32:42)
|
||||
# after them rather than inside one. That is what stops a leg reporting `Starting 0 tests`, and it
|
||||
# is why the three-round retry stays.
|
||||
#
|
||||
# THE `pm disable-user` CALL STAYS TOO, for a narrower reason than it was written for: every green
|
||||
# leg and every measurement quoted anywhere about this row was taken with it applied and SystemUI
|
||||
# running. Removing it would change the configuration the numbers came from, which is not a change
|
||||
# to make while fixing a flake.
|
||||
#
|
||||
# AND THE FRAMEWORK RESTART IS GONE, having been measured to be worse than nothing. It was written
|
||||
# as `adb shell stop; adb shell start`, which are root-only; adbd is not root, so every leg printed
|
||||
# `Must be root` twice and restarted nothing. Adding `adb root` made it real, and api37-debug run
|
||||
# 34010167885 is what that looks like: `pm disable-user` reports success, the stop lands ~2 s later
|
||||
# and kills system_server before PackageManager has flushed its delayed write of package
|
||||
# restrictions, so the state is gone on the way back up -- `NOT DISABLED after the restart`, three
|
||||
# rounds, `final state: SystemUI STILL ENABLED`, and the leg then reported `expected: 0,
|
||||
# received: 0`. A 15 s pause before the stop does make the state survive (bisected locally), and it
|
||||
# still does not help, because of the two measurements above. So the restart is removed rather than
|
||||
# repaired: it cost the leg every test it had, and there is nothing for it to buy.
|
||||
#
|
||||
# NOTHING HERE TRUSTS A COMMAND'S OWN REPORT, and that is not paranoia: of four runs of an
|
||||
# earlier one-shot version, one (32646029143) reported `new state: disabled-user` and then
|
||||
# started SystemUI eight more times, with ten more aborts. `pm disable-user` can be accepted by
|
||||
# a system_server that is SIGKILLed before the state is written, and `pm disable-user` does not
|
||||
# retract SystemUI's existing region-sampling registration either -- by the time boot completes
|
||||
# it has already registered, so only a framework restart brings back a SystemUI-less
|
||||
# surfaceflinger. Hence: disable, take the framework DOWN and confirm system_server is really
|
||||
# gone (an earlier probe asked `service check` 0.3 s after `stop` and got `found` from the
|
||||
# system_server that was still exiting, so its wait was not a wait), bring it back, verify the
|
||||
# package against `pm list packages -d`, and require a 45 s window with zero new aborts.
|
||||
# Three rounds, because one is not reliable and the failure is silent.
|
||||
# started SystemUI eight more times. So this reports what `pm list packages -d` says AND what
|
||||
# `pidof` says, side by side, rather than one line implying both.
|
||||
# ---------------------------------------------------------------------------
|
||||
count_aborts() { adb logcat -d -b crash 2> /dev/null | grep -c 'hasReadColorBufferDma'; }
|
||||
systemui_disabled() { adb shell pm list packages -d 2> /dev/null | grep -q 'com.android.systemui'; }
|
||||
systemui_pid() { adb shell pidof com.android.systemui 2> /dev/null | tr -d '\r\n'; }
|
||||
|
||||
disable_region_sampling() {
|
||||
local round=1 i out before after
|
||||
local round=1 i out pid before after
|
||||
while [ "$round" -le 3 ]; do
|
||||
echo "--- SystemUI disable, round $round ---"
|
||||
echo "--- round $round ---"
|
||||
for i in $(seq 1 10); do
|
||||
out="$(adb shell pm disable-user --user 0 com.android.systemui 2>&1 | tr -d '\r')"
|
||||
echo " pm attempt $i: $out"
|
||||
@@ -93,36 +125,20 @@ disable_region_sampling() {
|
||||
sleep 5
|
||||
done
|
||||
|
||||
echo " restarting the framework"
|
||||
adb shell stop
|
||||
for i in $(seq 1 20); do
|
||||
[ -z "$(adb shell pidof system_server 2> /dev/null | tr -d '\r\n')" ] && break
|
||||
sleep 2
|
||||
done
|
||||
echo " system_server down after ~$((i * 2)) s"
|
||||
adb shell start
|
||||
for i in $(seq 1 30); do
|
||||
if adb shell service check package 2> /dev/null | grep -q ': found' \
|
||||
&& adb shell service check activity 2> /dev/null | grep -q ': found' \
|
||||
&& [ -n "$(adb shell pidof system_server 2> /dev/null | tr -d '\r\n')" ]; then
|
||||
echo " services back after ~$((i * 5)) s"
|
||||
break
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
|
||||
if systemui_disabled; then
|
||||
echo " verified: com.android.systemui is in pm list packages -d"
|
||||
echo " pm list packages -d: com.android.systemui is in it"
|
||||
else
|
||||
echo " NOT DISABLED after the restart -- the package state did not survive"
|
||||
round=$((round + 1))
|
||||
continue
|
||||
echo " pm list packages -d: com.android.systemui is NOT in it"
|
||||
fi
|
||||
# Printed next to the line above precisely because the two disagree on this image, and a
|
||||
# reader who sees only the first will believe something that is not true.
|
||||
pid="$(systemui_pid)"
|
||||
echo " com.android.systemui pid: ${pid:-none} (expected: a pid -- see the header)"
|
||||
|
||||
before="$(count_aborts)"
|
||||
sleep 45
|
||||
after="$(count_aborts)"
|
||||
echo " abort rate, SystemUI disabled: $((after - before)) new in 45 s (total ${after:-0})"
|
||||
echo " aborts: $((after - before)) new in 45 s (total ${after:-0})"
|
||||
[ "$((after - before))" -eq 0 ] && break
|
||||
echo " still aborting after round $round"
|
||||
round=$((round + 1))
|
||||
@@ -131,16 +147,16 @@ disable_region_sampling() {
|
||||
# A warning rather than an exit. If the disable did not take, the run is about to report
|
||||
# `Starting 0 tests` and fail on its own -- and it will do so with the logcat, the crash
|
||||
# buffer and the diagnostics attached, which is more useful than dying here with none of it.
|
||||
if systemui_disabled; then
|
||||
echo " final state: SystemUI disabled"
|
||||
if [ "$((after - before))" -eq 0 ]; then
|
||||
echo " final state: 45 s with no new aborts -- the suite starts here"
|
||||
else
|
||||
echo "::warning::E2E api${LABEL}: SystemUI is still enabled -- expect INSTRUMENTATION_ABORTED"
|
||||
echo "::warning::E2E api${LABEL}: still aborting after three rounds -- expect INSTRUMENTATION_ABORTED"
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
if [ "${E2E_DISABLE_SYSTEM_UI:-}" = "1" ]; then
|
||||
echo "::group::E2E api${LABEL} -- removing the region-sampling listener"
|
||||
echo "::group::E2E api${LABEL} -- waiting out the boot-time gralloc aborts"
|
||||
disable_region_sampling
|
||||
echo "::endgroup::"
|
||||
fi
|
||||
|
||||
@@ -28,6 +28,15 @@ name: API 37 debug
|
||||
# - It does not fork .github/scripts/e2e-run.sh. That script owns the FAILED-vs-WEDGED
|
||||
# split, the SIGQUIT thread dump and the streamed logcat, and it is the copy CI
|
||||
# exercises every day. This calls it, exactly as status_check.yml does.
|
||||
#
|
||||
# The SystemUI disable below is the exception, and it is a real one: this workflow
|
||||
# drives it from its own probe step so `disable_system_ui` can be turned off for a
|
||||
# dispatch, where the real leg gets it through `E2E_DISABLE_SYSTEM_UI`. Two copies of
|
||||
# that logic therefore exist and must be changed together. **This instrument is also
|
||||
# what established that the disable half of it does nothing** -- run 34010167885, in
|
||||
# which making its framework restart real cost the leg every test it had. Read
|
||||
# .github/scripts/e2e-run.sh's header for the measurements; the restart is gone from
|
||||
# both copies and what remains is the 45-second quiet window.
|
||||
# - It does not change status_check.yml. If a configuration here turns out to work,
|
||||
# the change to the real matrix is proposed separately.
|
||||
#
|
||||
@@ -291,45 +300,30 @@ jobs:
|
||||
sleep 5
|
||||
done
|
||||
|
||||
# pm disable-user does not retract SystemUI's existing region-sampling
|
||||
# registration -- by the time boot completes it has already registered. Only a
|
||||
# framework restart brings back a SystemUI-less SurfaceFlinger. See
|
||||
# disable_region_sampling in tools/local-emulator/run-e2e.sh.
|
||||
echo " restarting the framework"
|
||||
adb shell stop
|
||||
for i in $(seq 1 20); do
|
||||
[ -z "$(adb shell pidof system_server 2> /dev/null | tr -d '\r\n')" ] && break
|
||||
sleep 2
|
||||
done
|
||||
echo " system_server down after $((i * 2)) s"
|
||||
adb shell start
|
||||
for i in $(seq 1 30); do
|
||||
if adb shell service check package 2> /dev/null | grep -q ': found' \
|
||||
&& adb shell service check activity 2> /dev/null | grep -q ': found' \
|
||||
&& [ -n "$(adb shell pidof system_server 2> /dev/null | tr -d '\r\n')" ]; then
|
||||
echo " services back after $((i * 5)) s"
|
||||
break
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
|
||||
# NO FRAMEWORK RESTART. There was one here, and making it work (it needed
|
||||
# `adb root`) is what proved the whole disable is ineffective on this image:
|
||||
# SystemUI starts anyway, measured on CI and locally, and the restart itself
|
||||
# loses the package state to PackageManager's delayed write and leaves the leg
|
||||
# reporting `Starting 0 tests`. e2e-run.sh's header carries the measurements.
|
||||
# What is left, and what is load-bearing, is the quiet window below.
|
||||
if systemui_disabled; then
|
||||
echo " verified: com.android.systemui is in pm list packages -d"
|
||||
echo " pm list packages -d: com.android.systemui is in it"
|
||||
else
|
||||
echo " NOT DISABLED after the restart -- the package state did not survive"
|
||||
round=$((round + 1))
|
||||
continue
|
||||
echo " pm list packages -d: com.android.systemui is NOT in it"
|
||||
fi
|
||||
# Beside it, because the two disagree on this image and the first line alone
|
||||
# reads as a claim about the process that is not true.
|
||||
echo " com.android.systemui pid: $(adb shell pidof com.android.systemui 2> /dev/null | tr -d '\r\n')"
|
||||
|
||||
before="$(count_aborts)"
|
||||
sleep 45
|
||||
after="$(count_aborts)"
|
||||
echo "--- abort rate, SystemUI disabled: $((after - before)) new in 45 s (total ${after:-0}) ---"
|
||||
echo "--- aborts: $((after - before)) new in 45 s (total ${after:-0}) ---"
|
||||
[ "$((after - before))" -eq 0 ] && break
|
||||
echo " still aborting after round $round"
|
||||
round=$((round + 1))
|
||||
done
|
||||
systemui_disabled && echo "final state: SystemUI disabled" || echo "final state: SystemUI STILL ENABLED -- expect Starting 0 tests"
|
||||
systemui_disabled && echo "final state: com.android.systemui is disabled in pm (it still runs)" || echo "final state: com.android.systemui is not even disabled in pm"
|
||||
fi
|
||||
|
||||
echo "--- crash buffer (tail 60) ---"
|
||||
|
||||
@@ -254,31 +254,31 @@ jobs:
|
||||
api-level: "36"
|
||||
# API 37, and it is NOT the same device as the four rows above it.
|
||||
#
|
||||
# CAVEAT, read this before trusting a green here: this leg runs with
|
||||
# SystemUI disabled and the framework restarted under it. No other leg
|
||||
# and no Pixel run uses that configuration. It is defensible only because
|
||||
# nothing THIS LEG RUNS touches system UI -- Media3, FFmpeg and
|
||||
# WorkManager tests -- and because the alternative is no CI coverage of
|
||||
# the level this app targets. **Anything that ever does depend on system
|
||||
# UI must not trust this row.** E2E_DISABLE_SYSTEM_UI is what does it;
|
||||
# .github/scripts/e2e-run.sh explains the mechanism and why every step of
|
||||
# it is verified rather than assumed.
|
||||
# THE CAVEAT THAT USED TO BE HERE IS WITHDRAWN, 2026-09-05, and the
|
||||
# withdrawal is good news. It said this leg "runs with SystemUI disabled
|
||||
# and the framework restarted under it", that no other leg or Pixel run
|
||||
# uses that configuration, and that anything depending on system UI must
|
||||
# not trust this row. **None of that was ever true.** Measured: the
|
||||
# framework restart is two root-only adb commands that answered `Must be
|
||||
# root` on every leg ever run, and `pm disable-user` does not stop SystemUI
|
||||
# starting on this image anyway -- in run 34006456986 the package is
|
||||
# verified disabled at 02:29:33 and SystemUI (pid 4275) is up from 02:28:52
|
||||
# for the whole run. So this row's device configuration is the same as the
|
||||
# other four's, and a green here means what a green on 33-36 means.
|
||||
#
|
||||
# "this leg" and not "this suite", since 2026-08-24, and the difference is
|
||||
# now load-bearing: SafPickerRoundTripTest DOES touch system UI. It drives
|
||||
# DocumentsUI and rotates the display, and both reach the gralloc mapper
|
||||
# this image aborts in -- disabling SystemUI removes the IDLE trigger, not
|
||||
# those. Measured per method on android-37.0: the ROTATION test takes the
|
||||
# framework down (INSTRUMENTATION_ABORTED) and carries
|
||||
# @FailsOnEmulatorApi37, so notAnnotation below keeps it off this row; the
|
||||
# PICKER test passes and runs here like anything else. A rotation rebuilds
|
||||
# every surface at once, and starting another app's activity does not.
|
||||
# E2E_DISABLE_SYSTEM_UI still exists and still runs, because what it
|
||||
# actually buys is a 45-second window with no new gralloc aborts before the
|
||||
# suite starts -- the boot-time ones land close together and instrumentation
|
||||
# has to begin after them, not between them. The name is stale and kept:
|
||||
# read .github/scripts/e2e-run.sh's header, which carries the measurements.
|
||||
#
|
||||
# So this row does now run one test that depends on system UI, and the
|
||||
# caveat above still applies to it: a green here is not evidence the picker
|
||||
# works on a device with SystemUI running -- the Pixel release check is.
|
||||
# docs/api-37-emulator-crash.md has the per-method measurements, and the
|
||||
# correction that produced them.
|
||||
# notAnnotation below keeps five tests off this row, and one of
|
||||
# them is new. SafPickerRoundTripTest's PICKER test was measured on
|
||||
# 2026-08-24 as passing here and was left on the leg; four gating logcats
|
||||
# read on 2026-09-05 show it aborting system_server from the task-snapshot
|
||||
# path on every single run, pass or fail, which is what had been failing
|
||||
# unrelated PRs (#108). Both of that class's tests now carry the marker.
|
||||
# docs/api-37-emulator-crash.md has the timings and the correction.
|
||||
#
|
||||
# api-level must be a POINT release. A bare 37 is not an SDK package and
|
||||
# fails during setup, which cost a run to discover. `37.0` is the choice
|
||||
|
||||
@@ -76,17 +76,45 @@ days. Read it as the current answer, and see the git history if you need the old
|
||||
`angle_indirect` and `swangle_indirect` all boot, while `auto`, `off`, `guest` and
|
||||
`swiftshader_indirect` do not. `docs/local-emulator.md` has the evidence and the per-API renderer
|
||||
table.
|
||||
- **CI runs API 37, and it gates.** The matrix is 33/34/35/36/37. **Three** of the 60 instrumented
|
||||
tests cannot pass on that image, for two unrelated reasons: two Media3 hardware transcodes fail
|
||||
inside the emulator's own `c2.goldfish.h264.decoder`, and one SAF test takes the framework down
|
||||
when it rotates the display. All three carry `@FailsOnEmulatorApi37` and run in a separate
|
||||
`continue-on-error` job; the gating leg runs the other 57.
|
||||
- **CI runs API 37, and it gates.** The matrix is 33/34/35/36/37. **Five** of the 69 instrumented
|
||||
tests cannot be *run* on that image, for three unrelated reasons: three Media3 tests fail inside
|
||||
the emulator's own `c2.goldfish.h264.decoder`, one SAF test takes the framework down when it
|
||||
rotates the display, and its sibling — the SAF picker round trip — aborts `system_server` from
|
||||
the task-snapshot path whether it passes or not. All five carry `@FailsOnEmulatorApi37` and run
|
||||
in a separate `continue-on-error` job; the gating leg runs the other 64.
|
||||
|
||||
**These two numbers move with the suite and are derived, not remembered.** `grep -cE
|
||||
'^\s*@Test' ` over `app/src/androidTest` is the first; the second is that minus the marker
|
||||
count `.github/scripts/e2e-report-shape.sh` greps. Cross-check against any run's shape rather
|
||||
than trusting the sentence: a leg below 37 reports the first as `expected`, and the API 37
|
||||
gating leg reports the second.
|
||||
|
||||
**That third reason is why "cannot pass" became "cannot be run" on 2026-09-05.** Four gating
|
||||
runs were read logcat-first — 34006456986, 34001744574, 34001377499 and the green 34002313300 —
|
||||
and each carries exactly two `hasReadColorBufferDma` aborts before the suite (surfaceflinger,
|
||||
during boot and the SystemUI disable) and exactly **one** during it: `system_server`, thread
|
||||
`TaskSnapshotPer`, always inside the picker test's window, and nothing else in the gating set
|
||||
reached the mapper at all. Whether the leg went red was luck — one run passed the test and lost
|
||||
the leg anyway with `failed: 0`, another passed it 0.6 s after the abort and went green. That is
|
||||
#108, it cost roughly a third of the gating legs over the wave-4 landings (#190), and a marker
|
||||
is what it needed. `docs/api-37-emulator-crash.md` has the timings.
|
||||
|
||||
**A second thing came out of those logcats, and it withdraws a caveat rather than adding one.**
|
||||
The API 37 row was documented as the one leg running "with SystemUI disabled and the framework
|
||||
restarted under it", which nothing else does. Neither half was ever happening: `adb shell stop`
|
||||
and `start` are root-only and answered `Must be root` on every leg ever run, and `pm
|
||||
disable-user` does not stop SystemUI starting on this image anyway — measured on CI and locally,
|
||||
with and without a real restart. **So this row's device configuration is the same as the other
|
||||
four's, and a green here means what a green at 33–36 means.** `E2E_DISABLE_SYSTEM_UI` is kept
|
||||
under its now-stale name because what it really buys is a 45-second window with no new gralloc
|
||||
aborts before the suite starts, which is load-bearing; `.github/scripts/e2e-run.sh`'s header is
|
||||
where that is written down.
|
||||
|
||||
That job is still called `E2E API 37 Media3 hardware transcode (advisory)`, which no longer
|
||||
describes everything in it. The name is kept deliberately — it is not a required context and
|
||||
people have learned to look for it — so **read the marker, not the name**, for what it holds.
|
||||
**It is red on every PR, by design**: do not read it as your change breaking something, and do
|
||||
not read a green run as evidence those three tests pass.
|
||||
not read a green run as evidence those five tests pass.
|
||||
`docs/api-37-emulator-crash.md` has the measurements.
|
||||
|
||||
**That instruction is also why nobody looks, so the job now reports its own shape** — expected,
|
||||
@@ -106,7 +134,7 @@ days. Read it as the current answer, and see the git history if you need the old
|
||||
is gradle never returning, so the log it left says nothing about it.
|
||||
|
||||
Still true, and the reason the advisory job is not simply deleted: **API 37 needs a manual check on
|
||||
the Pixel 10 Pro XL before each release.** Those three tests are the one thing CI cannot answer
|
||||
the Pixel 10 Pro XL before each release.** Those five tests are the one thing CI cannot answer
|
||||
for.
|
||||
|
||||
On a device or emulator, build only the ABI it can execute:
|
||||
@@ -332,9 +360,24 @@ install for code that can never run — and on API 37 the full APK does not fit
|
||||
without the pin they would fail loudly; `HardwareFallbackTest`'s are about the *output*, so it
|
||||
passes quietly. **Prefer asserting the path over asserting the artefact** where the two differ.
|
||||
|
||||
The read was a triage, not a test push, and five of its six findings are prose rather than code —
|
||||
The read was a triage, not a test push, and six of its seven findings are prose rather than code —
|
||||
the suite itself is in good shape. What had drifted is its self-description.
|
||||
|
||||
**Working the tickets then found the thing the read could not: one production defect.** #238 —
|
||||
joining files picked through the system picker failed outright on the stream-copy path. The
|
||||
concat demuxer whitelists protocols separately from `-safe 0`, and `ffkitsaf` was not on the
|
||||
list; only `STREAM_COPY` feeds it a list file, and every existing join test passed
|
||||
`Uri.fromFile`, so **the one broken combination was the only one a user could reach**. Not a
|
||||
missed line and not an unasserted value — two covered things no test put together, which is the
|
||||
gap shape a coverage number is worst at.
|
||||
|
||||
**E7 is the other reusable result**, because it re-scoped its own ticket. A real
|
||||
`DocumentsProvider` cannot be reached without the picker: an unprotected one is refused at
|
||||
install, instrumentation runs in the app's uid so the test APK's identity is no help, and shell
|
||||
identity is denied too — each denial naming `ACTION_OPEN_DOCUMENT`. So #226 has no cheap headless
|
||||
half. But the *input* bridge needs no documents provider at all, which is what kept #225 headless
|
||||
and is how #238 surfaced.
|
||||
|
||||
- **Testable code is not done until it is tested.** If a piece is unit testable, it gets unit
|
||||
tests before it counts as done. If it is e2e testable, it gets e2e tests. Both clauses apply —
|
||||
a change that is both needs both.
|
||||
|
||||
@@ -42,6 +42,28 @@
|
||||
<action android:name="android.content.action.DOCUMENTS_PROVIDER" />
|
||||
</intent-filter>
|
||||
</provider>
|
||||
|
||||
<!--
|
||||
A PLAIN provider, for the ffkitsaf bridge on the success path.
|
||||
|
||||
FFmpegKitConfig.getSafParameterForRead is on every real user conversion and was on no
|
||||
passing test: they all pass Uri.fromFile, which takes the other arm. Only its failure
|
||||
side was covered, by UnopenableUriTest naming an authority that does not exist.
|
||||
|
||||
The documents provider above cannot serve this. Any DOCUMENTS_PROVIDER must hold
|
||||
MANAGE_DOCUMENTS or the platform refuses to install it, instrumentation runs in the
|
||||
target app's process and so carries the app's uid, and the resulting denial says what
|
||||
is actually required: access obtained through ACTION_OPEN_DOCUMENT. That means a picker,
|
||||
and the flake it brings. See issue #226.
|
||||
|
||||
The bridge does not need a documents provider. It opens a descriptor through the
|
||||
resolver and hands FFmpeg a saf: path, so any readable content:// URI exercises it, and
|
||||
an ordinary provider is allowed to be exported without a permission.
|
||||
-->
|
||||
<provider
|
||||
android:name="org.libremediaconverter.saf.FixtureContentProvider"
|
||||
android:authorities="org.libremediaconverter.test.content"
|
||||
android:exported="true" />
|
||||
</application>
|
||||
|
||||
</manifest>
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
package org.libremediaconverter
|
||||
|
||||
/**
|
||||
* Marks an instrumented test that does not pass on the `android-37.x` **emulator** system images.
|
||||
* Marks an instrumented test that cannot be run on the `android-37.x` **emulator** system images.
|
||||
*
|
||||
* This is a marker, not a skip. Nothing reads it except CI, and CI reads it twice — once with
|
||||
* `notAnnotation` to build the gating API 37 leg, and once with `annotation` to build the advisory
|
||||
@@ -9,6 +9,15 @@ package org.libremediaconverter
|
||||
* That is the whole reason there is one annotation rather than a pair of test lists: two lists
|
||||
* drift, and the drift is silent in both directions (a test that runs nowhere reads as green).
|
||||
*
|
||||
* **"Cannot be run" covers two things, and it said only the first until 2026-09-05.** Four of the
|
||||
* five carriers simply fail: three Media3 tests die in the image's own `c2.goldfish.h264.decoder`,
|
||||
* and the SAF rotation test takes the framework down with it. The fifth —
|
||||
* `SafPickerRoundTripTest.pickingAFileThroughTheSystemPickerFillsInTheFileCard` — **passes about
|
||||
* half the time and aborts `system_server` every time**, which is worse for a gating leg than an
|
||||
* honest failure: it fails the leg from the teardown, with no failing test to point at (#108).
|
||||
* The wording was widened rather than the test excused; that test's own KDoc has the four-run
|
||||
* measurement.
|
||||
*
|
||||
* It says only what has been measured: **on the emulator, at API 37.** The same tests pass on a
|
||||
* physical Pixel 10 Pro XL at API 37 and at API 33–36 on the same runner under the same renderer,
|
||||
* so this must never be read as "this test is allowed to fail at API 37" — only as "the API 37
|
||||
@@ -17,7 +26,7 @@ package org.libremediaconverter
|
||||
*
|
||||
* Removing it is the goal, and the trigger is written down: a new API 37.x system image, or an
|
||||
* ATD image for 37. Delete the annotation from the tests, and the advisory job goes empty and
|
||||
* the gating one grows by two.
|
||||
* the gating one grows by [FAILS_ON_EMULATOR_API37_BASELINE].
|
||||
*
|
||||
* **How many tests carry it is committed below**, as [FAILS_ON_EMULATOR_API37_BASELINE], and the
|
||||
* advisory job checks the run against it. Adding or removing a marker means changing that number
|
||||
@@ -37,11 +46,28 @@ annotation class FailsOnEmulatorApi37
|
||||
* keep printing with nothing to compare to, so it announces that it could not read the baseline
|
||||
* rather than falling quiet. If you see that notice, this line is what it means.
|
||||
*
|
||||
* **One number, both checks, and that is what the marker means.** A test carrying it cannot pass
|
||||
* **One number, both checks, and that is what the marker means.** A test carrying it cannot be run
|
||||
* on this image, so the count is simultaneously how many the advisory leg runs and how many fail.
|
||||
* A *smaller* failure count is the interesting direction: it means one of them now passes, which
|
||||
* is the trigger the KDoc above names for deleting the annotation.
|
||||
*
|
||||
* **The picker test is the one to read that sentence carefully for.**
|
||||
* `pickingAFileThroughTheSystemPickerFillsInTheFileCard` was marked on 2026-09-05 for aborting
|
||||
* `system_server` rather than for failing (#108), and on the gating leg it passed two runs of
|
||||
* four. It fails on the advisory leg because the rotation test runs before it and takes the
|
||||
* framework down first — measured, `api37-debug.yml` run 34008889182, which reports
|
||||
* `expected: 4, received: 4, failed: 4` with the four in the order Media3, Media3, rotation,
|
||||
* picker. (Those dispatches predate the third Media3 marker landing on `main`, so their totals
|
||||
* are four rather than five; the ordering they establish is what matters here.)
|
||||
*
|
||||
* **But a second dispatch of the identical configuration reported 4/3/3**, having lost the last
|
||||
* test to the abort rather than to anything about the test list, and that is why
|
||||
* `e2e-report-shape.sh` compares `failed` only on a run that finished. `expected` is compared
|
||||
* always — it comes from `Starting N tests`, which is printed before anything can abort, so it is
|
||||
* the field that answers "is the marked set the size this number says". Read a *clean* run
|
||||
* reporting fewer failures than this as one of them now passing; read a truncated one as the
|
||||
* framework having died, which is this job's normal.
|
||||
*
|
||||
* So: adding or removing a [FailsOnEmulatorApi37] means changing this number, in this file, in
|
||||
* the same diff. The report says so on the run itself if you forget — it prints the tree's own
|
||||
* `grep` count beside this one.
|
||||
@@ -52,4 +78,4 @@ annotation class FailsOnEmulatorApi37
|
||||
* `INSTRUMENTATION_ABORTED`, so the count is a number taken from a partial run. The report
|
||||
* records the truncation next to the counts for that reason.
|
||||
*/
|
||||
const val FAILS_ON_EMULATOR_API37_BASELINE = 3
|
||||
const val FAILS_ON_EMULATOR_API37_BASELINE = 5
|
||||
|
||||
@@ -7,6 +7,10 @@ import androidx.media3.common.MimeTypes
|
||||
import androidx.media3.common.util.UnstableApi
|
||||
import androidx.test.ext.junit.runners.AndroidJUnit4
|
||||
import androidx.test.platform.app.InstrumentationRegistry
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.cancelAndJoin
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import org.junit.After
|
||||
@@ -14,6 +18,7 @@ import org.junit.Assert.assertEquals
|
||||
import org.junit.Assert.assertFalse
|
||||
import org.junit.Assert.assertNull
|
||||
import org.junit.Assert.assertTrue
|
||||
import org.junit.Assert.fail
|
||||
import org.junit.Before
|
||||
import org.junit.Test
|
||||
import org.junit.runner.RunWith
|
||||
@@ -262,6 +267,92 @@ class Media3EngineTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancelling a *running* export stops it, completing #224's third engine.
|
||||
*
|
||||
* The two FFmpeg engines were done first (`ad2a75d`, `d293646`); this is
|
||||
* `Media3Engine.transcode`'s `invokeOnCancellation`, which posts `transformer.cancel()` onto the
|
||||
* engine's own `HandlerThread` because `cancel()` has the same single-thread requirement as
|
||||
* `start()`.
|
||||
*
|
||||
* ## Why the assertion is the output file here, and was not for FFmpeg
|
||||
*
|
||||
* The FFmpeg side could not use the file: `invokeOnCancellation` unlinks it, and on POSIX ffmpeg
|
||||
* keeps writing to the unlinked inode, so the path stays gone whether or not the cancel landed.
|
||||
* It asserted the session's return code instead.
|
||||
*
|
||||
* `Media3Engine` deletes nothing — the partial is `ConversionWorker`'s to clean up — so the file
|
||||
* *is* the evidence. An export that was cancelled leaves no moov atom, so `MediaExtractor`
|
||||
* either finds no video track or refuses the file outright with
|
||||
* `IOException: Failed to instantiate extractor` — measured, and both mean interrupted. One
|
||||
* that ran to completion leaves a playable HEVC file, which is the only outcome treated as a
|
||||
* miss. The wait before
|
||||
* reading it is deliberately several times the length of the export, so a *non*-cancelled export
|
||||
* has certainly finished by then: the failure direction is "the file became valid", never "we
|
||||
* did not wait long enough".
|
||||
*
|
||||
* ## Why it retries
|
||||
*
|
||||
* Same reason as the other two, measured there: the committed fixture is 3 s at 320x240 and the
|
||||
* export outruns a naive cancel on a loaded runner. An attempt whose export finished before the
|
||||
* cancel landed has tested nothing, so it is a miss and is retried; only exhausting
|
||||
* [CANCEL_ATTEMPTS] fails. With `transformer.cancel()` removed every attempt produces a playable
|
||||
* file, so the mutation still bites — it just takes five tries to say so.
|
||||
*
|
||||
* Progress having been reported is what proves the export really started, so a miss is
|
||||
* distinguishable from an export that never ran at all — which matters on the API 37 image,
|
||||
* where the decoder is what fails.
|
||||
*/
|
||||
@Test
|
||||
@FailsOnEmulatorApi37
|
||||
fun cancellingARunningExportStopsIt(): Unit = runBlocking {
|
||||
val outcomes = mutableListOf<String>()
|
||||
|
||||
repeat(CANCEL_ATTEMPTS) { attempt ->
|
||||
val partial = File(context.cacheDir, "cancelled_export_$attempt.mp4").apply { delete() }
|
||||
|
||||
val job = launch(Dispatchers.IO) {
|
||||
engine.transcode(
|
||||
input = Uri.fromFile(input),
|
||||
output = partial,
|
||||
request = ConversionRequest(OutputFormat.MP4_H265.spec),
|
||||
)
|
||||
}
|
||||
|
||||
// The muxer creating the file is proof the export really started, and it is the
|
||||
// earliest such proof available -- earlier than the first progress tick.
|
||||
withTimeout(TIMEOUT_MS) {
|
||||
while (!partial.exists() && job.isActive) delay(POLL_MS)
|
||||
}
|
||||
val started = partial.exists()
|
||||
job.cancelAndJoin()
|
||||
|
||||
if (!started) {
|
||||
// The export failed before writing anything. That is not a cancellation result
|
||||
// either way, so it is not allowed to pass as one.
|
||||
outcomes += "attempt $attempt never produced an output file to cancel"
|
||||
return@repeat
|
||||
}
|
||||
|
||||
// Several times the export's own length, so a cancel that did not land has certainly
|
||||
// finished. The failure direction is "the file became playable", never "too soon".
|
||||
delay(SETTLE_MS)
|
||||
|
||||
// A cancelled export reports itself two ways and both mean the same thing: no video
|
||||
// track, or MediaExtractor refusing the file outright with "Failed to instantiate
|
||||
// extractor" because there is no moov atom to read. Only a *playable* file is a miss.
|
||||
val video = runCatching { videoMimeTypeOf(partial) }.getOrNull()
|
||||
partial.delete()
|
||||
if (video == null) return@runBlocking
|
||||
outcomes += "attempt $attempt produced a playable $video"
|
||||
}
|
||||
|
||||
fail(
|
||||
"never interrupted a running export in $CANCEL_ATTEMPTS attempts, so either every " +
|
||||
"export finished first or cancellation does not reach the transformer: $outcomes",
|
||||
)
|
||||
}
|
||||
|
||||
private fun videoMimeTypeOf(file: File): String? {
|
||||
val extractor = MediaExtractor()
|
||||
try {
|
||||
@@ -280,6 +371,19 @@ class Media3EngineTest {
|
||||
private companion object {
|
||||
const val TIMEOUT_SECONDS = 120L
|
||||
|
||||
/** Bounds the wait for the muxer to create the file; a hang here is a defect. */
|
||||
const val TIMEOUT_MS = 30_000L
|
||||
const val POLL_MS = 25L
|
||||
|
||||
/**
|
||||
* How long to let a *failed* cancel finish. Several times the export's own length, so
|
||||
* "the file is not playable" cannot mean "not yet".
|
||||
*/
|
||||
const val SETTLE_MS = 10_000L
|
||||
|
||||
/** See the KDoc: a miss is the loaded-runner case, not a defect. */
|
||||
const val CANCEL_ATTEMPTS = 5
|
||||
|
||||
/**
|
||||
* Short on purpose. Nothing is decoded or encoded on this path — the builder refuses the
|
||||
* input outright — so anything approaching this is a hang, which is what the test is
|
||||
|
||||
@@ -13,6 +13,7 @@ import androidx.work.WorkManager
|
||||
import androidx.work.Worker
|
||||
import androidx.work.WorkerParameters
|
||||
import androidx.work.workDataOf
|
||||
import kotlinx.coroutines.CompletableDeferred
|
||||
import kotlinx.coroutines.flow.first
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
@@ -27,7 +28,10 @@ import org.junit.runner.RunWith
|
||||
import org.libremediaconverter.join.JoinState
|
||||
import org.libremediaconverter.join.JoinViewModel
|
||||
import org.libremediaconverter.model.ConcatStrategy
|
||||
import org.libremediaconverter.model.ConversionRequest
|
||||
import org.libremediaconverter.model.Engine
|
||||
import org.libremediaconverter.model.OutputFormat
|
||||
import org.libremediaconverter.model.QualityTier
|
||||
import org.libremediaconverter.work.ConcatWorker
|
||||
import org.libremediaconverter.work.ConversionWorker
|
||||
import org.libremediaconverter.work.JobTags
|
||||
@@ -64,6 +68,26 @@ class EchoWorker(context: Context, params: WorkerParameters) : Worker(context, p
|
||||
* path, foreground service included — into a synchronous test double, depending on class order.
|
||||
*/
|
||||
@UnstableApi
|
||||
/**
|
||||
* A [SoftwareTranscoder] that holds the worker in [WorkInfo.State.RUNNING] until released.
|
||||
*
|
||||
* Declared here rather than in `FakeFailures` because it is the only test that needs a job to stay
|
||||
* live on demand, and the shape is specific to that: the others fake a *failure*, this fakes
|
||||
* *duration*.
|
||||
*/
|
||||
private class BlockingTranscoder(private val released: CompletableDeferred<Unit>) : SoftwareTranscoder {
|
||||
override suspend fun run(
|
||||
request: ConversionRequest,
|
||||
inputPath: String,
|
||||
output: File,
|
||||
durationMs: Long,
|
||||
onProgress: (Int) -> Unit,
|
||||
) {
|
||||
released.await()
|
||||
output.writeBytes(ByteArray(1_024))
|
||||
}
|
||||
}
|
||||
|
||||
@RunWith(AndroidJUnit4::class)
|
||||
class ReattachOnLaunchTest {
|
||||
|
||||
@@ -75,7 +99,13 @@ class ReattachOnLaunchTest {
|
||||
fun clearTheQueue() = emptyQueueAndStaging()
|
||||
|
||||
@After
|
||||
fun leaveNothingBehind() = emptyQueueAndStaging()
|
||||
fun leaveNothingBehind() {
|
||||
// The suite runs without Android Test Orchestrator, so every class shares one process and
|
||||
// a swapped seam outlives the class that set it. Only one test here swaps one, but a
|
||||
// BlockingTranscoder left in place would hang the next class that converts anything.
|
||||
ConversionDependencies.reset()
|
||||
emptyQueueAndStaging()
|
||||
}
|
||||
|
||||
/**
|
||||
* The claim the whole fix rests on, checked against the production request builder rather
|
||||
@@ -261,6 +291,69 @@ class ReattachOnLaunchTest {
|
||||
return request.id
|
||||
}
|
||||
|
||||
/**
|
||||
* Reattaching to a conversion that is **running right now**, which nothing had ever driven.
|
||||
*
|
||||
* This class covers a job that finished, one whose staged file is gone, an ambiguous pair, one
|
||||
* still queued, and one the user cancelled. [Reattachment.rank] gives
|
||||
* [WorkInfo.State.RUNNING] the **highest** rank of all — "live work outranks a finished result
|
||||
* because a running job is holding a foreground service" — and no test on either source set
|
||||
* ever produced one. `ReattachmentTest` exercises the ranking as a pure function over
|
||||
* fabricated snapshots; what was missing is a ViewModel meeting a real running job.
|
||||
*
|
||||
* It is also the likeliest reattachment there is: the user starts a conversion, leaves, and
|
||||
* comes back while it is still going.
|
||||
*
|
||||
* ## Why the engine is a fake here, and why that is not a weakening
|
||||
*
|
||||
* The job has to still be running when the ViewModel is built, and every real conversion in
|
||||
* this suite finishes in about a second — racing that is what made the cancellation tests flaky
|
||||
* enough to need retries (#224). A [SoftwareTranscoder] that blocks until released removes the
|
||||
* race outright: the job is `RUNNING` for exactly as long as the test wants.
|
||||
*
|
||||
* Nothing about reattachment depends on which engine is transcoding. What is under test is the
|
||||
* tag query, [Reattachment.choose] over live WorkManager state, and `observe` mapping it to
|
||||
* [ConversionState.Converting] — all of which run identically whatever is doing the work.
|
||||
*
|
||||
* ## What this does not do, and cannot (#230)
|
||||
*
|
||||
* It does not kill the process. `docs/defect-audit.md` D3/D13 record that `am kill` refuses a
|
||||
* process holding a foreground service, and there is a more basic obstacle: **instrumentation
|
||||
* runs in the app's own process**, so any route that really killed it would take the test
|
||||
* runner with it and there would be nothing left to assert with. A relaunch-and-observe test
|
||||
* needs two instrumentation runs, which the runner does not provide.
|
||||
*
|
||||
* So process death stays device-manual, and this is the closest observable analogue: a fresh
|
||||
* ViewModel, with no memory of the work, meeting a job that is genuinely mid-flight.
|
||||
*/
|
||||
@Test
|
||||
fun reattachesToAConversionThatIsStillRunning(): Unit = runBlocking {
|
||||
val released = CompletableDeferred<Unit>()
|
||||
ConversionDependencies.software = { BlockingTranscoder(released) }
|
||||
|
||||
val request = ConversionWorker.request(
|
||||
inputUri = Uri.fromFile(stage("running_input.mp3")),
|
||||
displayName = RUNNING_NAME,
|
||||
sizeBytes = RUNNING_SIZE,
|
||||
spec = OutputFormat.MP3.spec,
|
||||
quality = QualityTier.FAST,
|
||||
)
|
||||
workManager.enqueue(request).result.get()
|
||||
|
||||
// Deterministic: the worker cannot finish until this test lets it.
|
||||
withTimeout(TIMEOUT_MS) {
|
||||
workManager.getWorkInfoByIdFlow(request.id).first { it?.state == WorkInfo.State.RUNNING }
|
||||
}
|
||||
|
||||
val reattached = awaitConversion<ConversionState.Converting>()
|
||||
|
||||
assertEquals(RUNNING_NAME, reattached.input.displayName)
|
||||
assertEquals(RUNNING_SIZE, reattached.input.sizeBytes)
|
||||
|
||||
released.complete(Unit)
|
||||
workManager.cancelWorkById(request.id).result.get()
|
||||
}
|
||||
|
||||
/**
|
||||
* Enqueues a job that stays [WorkInfo.State.ENQUEUED]. The delay is what holds it there: it
|
||||
* is long enough that nothing can run it during a test, and it is cancelled either way.
|
||||
@@ -326,5 +419,9 @@ class ReattachOnLaunchTest {
|
||||
* against WorkManager's database, so this is generous rather than tuned.
|
||||
*/
|
||||
const val SETTLE_MS = 5_000L
|
||||
|
||||
/** Read back off the job's tags by the reattaching ViewModel, so both have to survive. */
|
||||
const val RUNNING_NAME = "still_running.mp3"
|
||||
const val RUNNING_SIZE = 4_242L
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,10 +4,20 @@ import android.media.MediaExtractor
|
||||
import android.media.MediaFormat
|
||||
import androidx.test.ext.junit.runners.AndroidJUnit4
|
||||
import androidx.test.platform.app.InstrumentationRegistry
|
||||
import com.arthenica.ffmpegkit.FFmpegKit
|
||||
import com.arthenica.ffmpegkit.FFmpegSession
|
||||
import com.arthenica.ffmpegkit.ReturnCode
|
||||
import com.arthenica.ffmpegkit.SessionState
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.cancelAndJoin
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import org.junit.After
|
||||
import org.junit.Assert.assertEquals
|
||||
import org.junit.Assert.assertTrue
|
||||
import org.junit.Assert.fail
|
||||
import org.junit.Before
|
||||
import org.junit.Test
|
||||
import org.junit.runner.RunWith
|
||||
@@ -139,6 +149,160 @@ class FFmpegEngineTest {
|
||||
assertEquals("OggS", magic)
|
||||
}
|
||||
|
||||
/**
|
||||
* The percentage itself, which every other test in this class computes and none of them reads.
|
||||
*
|
||||
* `FFmpegEngine` derives progress as `stats.time / durationMs * 100`, and the statistics
|
||||
* callback runs on every conversion here — but every call site omits `onProgress`, so until
|
||||
* this test nothing on any source set had ever looked at the number (#229). #196 covered the
|
||||
* *worker's* progress lambda, and did it with a fake engine that reports whatever the test
|
||||
* tells it to; `ProgressNotificationTest` covers throttling the same way. The arithmetic was
|
||||
* the one part with no reader.
|
||||
*
|
||||
* ## Why the duration is deliberately wrong
|
||||
*
|
||||
* `sample_h264.mp4` is exactly 3.000 s, and this passes **30 s** as the duration. So the
|
||||
* conversion still encodes the whole clip, `stats.time` still climbs to about 3000 ms, and the
|
||||
* reported percentage tops out around **10** rather than 100.
|
||||
*
|
||||
* That is what makes the assertion bite. A range check alone is worthless here: replacing
|
||||
* `percent` with a constant `0` satisfies "every value is in 0..100" and "the values never go
|
||||
* backwards", and so does a list of `[0, 100]`. Pinning the *band* rejects every constant, and
|
||||
* — because the band is a tenth of the way up — it also rejects an implementation that ignores
|
||||
* `durationMs`, which would report ~100 for the same run.
|
||||
*
|
||||
* The bound is deliberately loose (5..25 for an expected 10). The last statistics callback can
|
||||
* land slightly before the final frame, so the peak is "about 3000 ms of a claimed 30 000",
|
||||
* not exactly it.
|
||||
*/
|
||||
@Test
|
||||
fun progressIsReportedAsAFractionOfTheDurationItWasGiven() {
|
||||
val seen = mutableListOf<Int>()
|
||||
val out = outputFor("out_progress.mp4")
|
||||
runBlocking {
|
||||
engine.run(
|
||||
request = ConversionRequest(spec = OutputFormat.MP4_H264.spec, quality = QualityTier.BEST),
|
||||
inputPath = input.absolutePath,
|
||||
output = out,
|
||||
// Ten times the fixture's real 3 s. See the KDoc.
|
||||
durationMs = 30_000,
|
||||
onProgress = { percent -> seen += percent },
|
||||
)
|
||||
}
|
||||
|
||||
assertTrue("the statistics callback never reported progress", seen.isNotEmpty())
|
||||
assertTrue("progress out of range: $seen", seen.all { it in 0..100 })
|
||||
assertEquals("progress went backwards: $seen", seen.sorted(), seen)
|
||||
// The band. Rejects any constant, and rejects ignoring durationMs (which would read ~100).
|
||||
val peak = seen.max()
|
||||
assertTrue(
|
||||
"3 s of media against a claimed 30 s should peak near 10%, got $peak from $seen",
|
||||
peak in 5..25,
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancelling a *running* conversion actually stops the native session.
|
||||
*
|
||||
* Nothing on any source set did this before (#224). Every `cancel` in `app/src/androidTest` is
|
||||
* `WorkManager.cancelWorkById` against work that is **queued or already finished** — the two in
|
||||
* `ReattachOnLaunchTest` cancel a job carrying a one-hour initial delay, and one immediately
|
||||
* after enqueue. On the JVM, `WorkerCancellationTest` and `HardwareFallbackTest`'s cancellation
|
||||
* case drive a `SoftwareTranscoder` double that records the call. No test had ever asked a real
|
||||
* native session to stop. This is `docs/defect-audit.md` **D10**'s forcing condition.
|
||||
*
|
||||
* It is the one path where cancelling wrong is silently expensive rather than loudly broken: a
|
||||
* missed `FFmpegKit.cancel` leaves the native process encoding to completion while the UI says
|
||||
* the job is cancelled, and nothing reports the battery and thermal cost.
|
||||
*
|
||||
* ## Why the assertion is the session's return code, not the output file
|
||||
*
|
||||
* The obvious assertion — the partial output is gone — **cannot fail**, so it would have been a
|
||||
* vacuous test. `invokeOnCancellation` deletes the path, and on POSIX unlinking a file ffmpeg
|
||||
* still holds open leaves ffmpeg writing to the unlinked inode; the path stays gone whether or
|
||||
* not the cancel ever reached the session. Deleting `FFmpegKit.cancel` and keeping
|
||||
* `output.delete()` passes that check every time.
|
||||
*
|
||||
* What distinguishes them is the session's own verdict: a cancelled session ends with the
|
||||
* cancel return code, a completed one ends successfully. That is a fact about the session
|
||||
* rather than about timing, so it is read *after* waiting for the session to leave
|
||||
* [SessionState.RUNNING] rather than at a fixed delay.
|
||||
*
|
||||
* ## Why it cancels on RUNNING rather than on the first progress callback
|
||||
*
|
||||
* Measured. Cancelling from the first `onProgress` was tried first and **failed on a local API
|
||||
* 34 emulator with `state=COMPLETED rc=0`** — every committed fixture is 2-3 s at 320x240, and
|
||||
* the encode finishes before the first statistics callback has been delivered and acted on. The
|
||||
* progress callback proves the session is running, but arrives too late to interrupt anything.
|
||||
* `FFmpegKit.listSessions` shows the session [SessionState.RUNNING] far earlier.
|
||||
*
|
||||
* ## Why it retries, which is the part that took two attempts to get right
|
||||
*
|
||||
* Waiting for `RUNNING` is not on its own enough. With `MP4_H265` at [QualityTier.BEST] this
|
||||
* passed four consecutive local runs and all five CI legs, then failed on the API 34 and 35 legs
|
||||
* of the next PR with `state=COMPLETED rc=0`. Nothing had changed: on a loaded runner the thread
|
||||
* that observed `RUNNING` can be descheduled long enough for a short encode to finish before it
|
||||
* calls `cancel`. A longer timeout does not help — the wait already succeeded.
|
||||
*
|
||||
* Two changes together, because neither is sufficient:
|
||||
*
|
||||
* - **A slower encode.** `WEBM_VP9` at `BEST` is the slowest thing this builder emits:
|
||||
* `libvpx-vp9 -crf 31 -b:v 0`, with `-deadline realtime` added **only** on
|
||||
* [QualityTier.FAST]. Probed on an API 34 emulator, that session is still `RUNNING` at 1 s
|
||||
* and finished by 2 s, against well under a second for x265 `-preset medium`.
|
||||
* - **Retrying the attempt.** An attempt whose session finished before the cancel landed has
|
||||
* not tested anything, so it is not a failure — it is a miss, and it is retried. Only
|
||||
* exhausting [CANCEL_ATTEMPTS] is a failure, and its message says which case it hit.
|
||||
*
|
||||
* That keeps the mutation honest: with `FFmpegKit.cancel` removed **every** attempt ends
|
||||
* `COMPLETED`, so the test still fails — it just takes [CANCEL_ATTEMPTS] tries to say so.
|
||||
*
|
||||
* The session is identified by diffing against the ids present before each attempt, because
|
||||
* this class has already produced eight of them by the time this executes.
|
||||
*/
|
||||
@Test
|
||||
fun cancellingARunningConversionCancelsTheNativeSession(): Unit = runBlocking {
|
||||
val outcomes = mutableListOf<String>()
|
||||
|
||||
repeat(CANCEL_ATTEMPTS) { attempt ->
|
||||
val before = FFmpegKit.listSessions().map { it.getSessionId() }.toSet()
|
||||
val out = outputFor("out_cancelled_$attempt.webm")
|
||||
|
||||
val job = launch(Dispatchers.IO) {
|
||||
engine.run(
|
||||
// The slowest target this builder emits -- see the KDoc. Not decoration:
|
||||
// with a faster one this loses the race on a loaded CI runner.
|
||||
request = ConversionRequest(spec = OutputFormat.WEBM_VP9.spec, quality = QualityTier.BEST),
|
||||
inputPath = input.absolutePath,
|
||||
output = out,
|
||||
durationMs = 3_000,
|
||||
)
|
||||
}
|
||||
|
||||
val ours = withTimeout(TIMEOUT_MS) {
|
||||
var found: FFmpegSession? = null
|
||||
while (found == null) {
|
||||
found = FFmpegKit.listSessions().firstOrNull { it.getSessionId() !in before }
|
||||
if (found == null) delay(POLL_MS)
|
||||
}
|
||||
found
|
||||
}
|
||||
job.cancelAndJoin()
|
||||
withTimeout(TIMEOUT_MS) {
|
||||
while (ours.getState() == SessionState.RUNNING) delay(POLL_MS)
|
||||
}
|
||||
|
||||
if (ReturnCode.isCancel(ours.getReturnCode())) return@runBlocking
|
||||
// The encode beat us to it. That attempt proved nothing either way, so try again.
|
||||
outcomes += "state=${ours.getState()} rc=${ours.getReturnCode()}"
|
||||
}
|
||||
|
||||
fail(
|
||||
"never interrupted a running session in $CANCEL_ATTEMPTS attempts, so either every " +
|
||||
"encode finished first or cancellation does not reach it: $outcomes",
|
||||
)
|
||||
}
|
||||
|
||||
// --- the quality tier the GPL licence was taken for --------------------
|
||||
|
||||
@Test
|
||||
@@ -176,4 +340,19 @@ class FFmpegEngineTest {
|
||||
}.exceptionOrNull()
|
||||
assertTrue("expected an FFmpegException, got $failure", failure is FFmpegEngine.FFmpegException)
|
||||
}
|
||||
|
||||
private companion object {
|
||||
/** Generous: it bounds a hang, and every wait here normally settles in well under a second. */
|
||||
const val TIMEOUT_MS = 30_000L
|
||||
const val POLL_MS = 50L
|
||||
|
||||
/**
|
||||
* How many times to try to catch the session mid-encode.
|
||||
*
|
||||
* Each miss costs about the length of one VP9 encode -- a second or two -- and a miss is
|
||||
* the loaded-runner case rather than a defect. Five is enough that exhausting them means
|
||||
* cancellation is not reaching the session, which is what the failure message says.
|
||||
*/
|
||||
const val CANCEL_ATTEMPTS = 5
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,17 +5,29 @@ import android.media.MediaFormat
|
||||
import android.net.Uri
|
||||
import androidx.test.ext.junit.runners.AndroidJUnit4
|
||||
import androidx.test.platform.app.InstrumentationRegistry
|
||||
import com.arthenica.ffmpegkit.FFmpegKit
|
||||
import com.arthenica.ffmpegkit.FFmpegSession
|
||||
import com.arthenica.ffmpegkit.ReturnCode
|
||||
import com.arthenica.ffmpegkit.SessionState
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.cancelAndJoin
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import org.junit.After
|
||||
import org.junit.Assert.assertEquals
|
||||
import org.junit.Assert.assertTrue
|
||||
import org.junit.Assert.fail
|
||||
import org.junit.Before
|
||||
import org.junit.Test
|
||||
import org.junit.runner.RunWith
|
||||
import org.libremediaconverter.convert.MediaProbe
|
||||
import org.libremediaconverter.convert.StagingNames
|
||||
import org.libremediaconverter.ffmpeg.ConcatEngine
|
||||
import org.libremediaconverter.ffmpeg.FFmpegEngine
|
||||
import org.libremediaconverter.model.ConcatStrategy
|
||||
import org.libremediaconverter.work.ConcatWorker
|
||||
import java.io.File
|
||||
|
||||
/**
|
||||
@@ -50,6 +62,82 @@ class ConcatEngineTest {
|
||||
(staged + listOf(clipA, clipB, clipMismatched)).forEach { it.delete() }
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancelling a *running* join actually stops the native session.
|
||||
*
|
||||
* The `FFmpegEngine` half of #224 landed first (PR #236); this is the same gap in
|
||||
* [ConcatEngine]. Before these two, no test on any source set had ever asked a real native
|
||||
* session to stop — every `cancel` in `app/src/androidTest` targets WorkManager entries that
|
||||
* are queued or already finished.
|
||||
*
|
||||
* ## Two things carried over from the conversion side, both measured there
|
||||
*
|
||||
* **The assertion is the session's return code.** A cancelled session ends with the cancel
|
||||
* code, a completed one does not. The alternative — checking the output file — is even less
|
||||
* available here than it was for conversions: [ConcatEngine] does not delete its output on
|
||||
* cancellation at all. Its `invokeOnCancellation` is `FFmpegKit.cancel(...)` and nothing else,
|
||||
* where [org.libremediaconverter.ffmpeg.FFmpegEngine]'s also deletes the partial. Whether that
|
||||
* asymmetry is deliberate is a separate question from this test, which is why this asserts the
|
||||
* thing that is true of both.
|
||||
*
|
||||
* **The cancel is triggered on [SessionState.RUNNING], not on progress.** `ConcatWorker`
|
||||
* publishes no progress at all, so there is no callback to hang it on even in principle — but
|
||||
* the conversion side established the deeper reason: the committed clips are 2 s at 320x240 and
|
||||
* the encode outruns a callback-triggered cancel.
|
||||
*
|
||||
* **And the attempt is retried**, for the reason the conversion side measured the hard way: on
|
||||
* a loaded runner the thread that observed `RUNNING` can be descheduled long enough for a short
|
||||
* encode to finish before it calls `cancel`, which failed two CI legs there. An attempt whose
|
||||
* session finished first has tested nothing, so it is a miss rather than a failure; only
|
||||
* exhausting [CANCEL_ATTEMPTS] fails, and with `FFmpegKit.cancel` removed every attempt misses,
|
||||
* so the mutation still bites.
|
||||
*
|
||||
* The inputs are deliberately the **mismatched** pair, so [ConcatStrategy.REENCODE] is chosen.
|
||||
* A stream copy of two short clips is close to instantaneous and would leave nothing to
|
||||
* interrupt; re-encoding is the case where a user would actually reach for Cancel.
|
||||
*
|
||||
* *Mutation:* drop `FFmpegKit.cancel(session.getSessionId())` from `ConcatEngine`'s
|
||||
* `invokeOnCancellation` — the session runs to completion and this fails.
|
||||
*/
|
||||
@Test
|
||||
fun cancellingARunningJoinCancelsTheNativeSession(): Unit = runBlocking {
|
||||
val outcomes = mutableListOf<String>()
|
||||
|
||||
repeat(CANCEL_ATTEMPTS) { attempt ->
|
||||
val before = FFmpegKit.listSessions().map { it.getSessionId() }.toSet()
|
||||
val out = output("cancelled_join_$attempt.mp4")
|
||||
|
||||
val job = launch(Dispatchers.IO) {
|
||||
engine.join(
|
||||
listOf(Uri.fromFile(clipA), Uri.fromFile(clipMismatched)),
|
||||
out,
|
||||
ConcatWorker.DEFAULT_FORMAT,
|
||||
)
|
||||
}
|
||||
|
||||
val ours = withTimeout(TIMEOUT_MS) {
|
||||
var found: FFmpegSession? = null
|
||||
while (found == null) {
|
||||
found = FFmpegKit.listSessions().firstOrNull { it.getSessionId() !in before }
|
||||
if (found == null) delay(POLL_MS)
|
||||
}
|
||||
found
|
||||
}
|
||||
job.cancelAndJoin()
|
||||
withTimeout(TIMEOUT_MS) {
|
||||
while (ours.getState() == SessionState.RUNNING) delay(POLL_MS)
|
||||
}
|
||||
|
||||
if (ReturnCode.isCancel(ours.getReturnCode())) return@runBlocking
|
||||
outcomes += "state=${ours.getState()} rc=${ours.getReturnCode()}"
|
||||
}
|
||||
|
||||
fail(
|
||||
"never interrupted a running join in $CANCEL_ATTEMPTS attempts, so either every " +
|
||||
"encode finished first or cancellation does not reach it: $outcomes",
|
||||
)
|
||||
}
|
||||
|
||||
private fun copyAsset(name: String): File {
|
||||
val out = File(context.cacheDir, name)
|
||||
InstrumentationRegistry.getInstrumentation().context.assets
|
||||
@@ -149,6 +237,55 @@ class ConcatEngineTest {
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* A failed join tells the user the return code and what FFmpeg said.
|
||||
*
|
||||
* **This is the device half of #203/#217**, whose PR closed by noting the join legs had not
|
||||
* been run. Running them would not have answered it: nothing on either source set drove a real
|
||||
* join *failure*, so the unified message was asserted only against values a JVM test hands to
|
||||
* `sessionOutcome` directly.
|
||||
*
|
||||
* What is device-only here is that the three reads behind that message work against a real
|
||||
* native session at all — `getReturnCode`, `getFailStackTrace` and `getAllLogsAsString`. If
|
||||
* the log tail came back null or empty on a device, the user would get `Joining failed (1): `
|
||||
* with nothing after the colon and every JVM test would still pass.
|
||||
*
|
||||
* **What this deliberately does not pin is the preference between the two detail sources.** On
|
||||
* an ordinary non-zero return code FFmpegKit reports no fail stack trace, so the stack-trace-
|
||||
* first rule and the log-tail-first rule produce the same text and no assertion here can tell
|
||||
* them apart. That ordering is [SessionOutcomeTest][org.libremediaconverter.ffmpeg.SessionOutcomeTest]'s
|
||||
* job, where both sources can be non-blank at once. Asserting it here would be a test whose
|
||||
* KDoc claims more than it checks — the `probeForConcat` mistake wave 3 caught.
|
||||
*
|
||||
* The failure is forced with an input that does not exist, which the concat demuxer rejects
|
||||
* the same way on every FFmpeg build, rather than with malformed media whose handling varies.
|
||||
*/
|
||||
@Test
|
||||
fun aFailedJoinReportsTheReturnCodeAndWhatFFmpegSaid(): Unit = runBlocking {
|
||||
val missing = File(context.cacheDir, "no_such_clip.mp4").also { it.delete() }
|
||||
val out = output("joined_failure.mp4")
|
||||
|
||||
val failure = runCatching {
|
||||
engine.join(listOf(Uri.fromFile(clipA), Uri.fromFile(missing)), out)
|
||||
}.exceptionOrNull()
|
||||
|
||||
assertTrue(
|
||||
"a join over a missing input must fail, got $failure",
|
||||
failure is FFmpegEngine.FFmpegException,
|
||||
)
|
||||
val message = failure?.message.orEmpty()
|
||||
assertTrue(
|
||||
"the message must name the operation and carry the return code, was: '$message'",
|
||||
message.startsWith("Joining failed ("),
|
||||
)
|
||||
// The half a JVM test cannot reach: a real session actually produced detail to show.
|
||||
val detail = message.substringAfter("): ", "")
|
||||
assertTrue(
|
||||
"the message stopped at the return code and told the user nothing, was: '$message'",
|
||||
detail.isNotBlank(),
|
||||
)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun theListFileIsCleanedUpAfterJoining(): Unit = runBlocking {
|
||||
val out = output("joined_cleanup.mp4")
|
||||
@@ -180,4 +317,13 @@ class ConcatEngineTest {
|
||||
a.width != mismatched.width || a.height != mismatched.height,
|
||||
)
|
||||
}
|
||||
|
||||
private companion object {
|
||||
/** Generous: it bounds a hang, and both waits here normally settle in well under a second. */
|
||||
const val TIMEOUT_MS = 30_000L
|
||||
const val POLL_MS = 50L
|
||||
|
||||
/** See the conversion side: a miss is the loaded-runner case, not a defect. */
|
||||
const val CANCEL_ATTEMPTS = 5
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
package org.libremediaconverter.saf
|
||||
|
||||
import androidx.media3.common.util.UnstableApi
|
||||
import androidx.test.ext.junit.runners.AndroidJUnit4
|
||||
import androidx.test.platform.app.InstrumentationRegistry
|
||||
import androidx.work.WorkInfo
|
||||
import androidx.work.WorkManager
|
||||
import kotlinx.coroutines.flow.first
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import org.junit.After
|
||||
import org.junit.Assert.assertEquals
|
||||
import org.junit.Assert.assertTrue
|
||||
import org.junit.Test
|
||||
import org.junit.runner.RunWith
|
||||
import org.libremediaconverter.ffmpeg.ConcatEngine
|
||||
import org.libremediaconverter.model.Engine
|
||||
import org.libremediaconverter.model.OutputFormat
|
||||
import org.libremediaconverter.model.QualityTier
|
||||
import org.libremediaconverter.work.ConversionWorker
|
||||
import java.io.File
|
||||
|
||||
/**
|
||||
* A `content://` input reaching FFmpeg successfully, which nothing had ever driven (#225).
|
||||
*
|
||||
* `FFmpegKitConfig.getSafParameterForRead` stands between a SAF grant and the native process, and
|
||||
* it is on **every real user conversion**. Every passing convert and join test in this suite hands
|
||||
* the worker a `Uri.fromFile(...)`, which takes the `uri.path` arm instead — so the bridge was
|
||||
* exercised only on its failure side, by `UnopenableUriTest` naming an authority that does not
|
||||
* exist. That proves the error message, not the bridge.
|
||||
*
|
||||
* ## Why a plain provider rather than the documents one
|
||||
*
|
||||
* [FixtureDocumentsProvider] cannot be reached from the app, measured three ways on an API 34
|
||||
* emulator (#226): a `DOCUMENTS_PROVIDER` declared without `MANAGE_DOCUMENTS` is refused at install
|
||||
* — *"Provider must be protected by MANAGE_DOCUMENTS"*; instrumentation runs in the **target app's
|
||||
* process**, so `Instrumentation.getContext()` still carries the app's uid and is denied; and
|
||||
* `adoptShellPermissionIdentity(MANAGE_DOCUMENTS)` is denied identically. The denial names the only
|
||||
* way in: *"you obtain access using ACTION_OPEN_DOCUMENT or related APIs"*.
|
||||
*
|
||||
* The bridge does not need one. It opens a descriptor through the resolver and hands FFmpeg a
|
||||
* `saf:` path, so any readable `content://` URI exercises it — and [FixtureContentProvider] is an
|
||||
* ordinary provider, which may be exported without a permission. The whole class is headless: no
|
||||
* DocumentsUI, and none of the flake #190 records.
|
||||
*
|
||||
* ## Why MP3
|
||||
*
|
||||
* The bridge lives on the FFmpeg arm, and MP3 is the format the router sends there unconditionally
|
||||
* — no platform encoder exists at any API level, so `ConversionWorkerTest.routesAnMp3JobToFfmpeg…`
|
||||
* relies on the same fact. Choosing a video target would make the engine depend on the device's
|
||||
* codecs, and #223 is what that costs.
|
||||
*
|
||||
* *Mutation:* make `getSafParameterForRead` return `uri.toString()`. FFmpeg cannot open it and both
|
||||
* tests fail; nothing else in either suite notices.
|
||||
*/
|
||||
@UnstableApi
|
||||
@RunWith(AndroidJUnit4::class)
|
||||
class ContentUriInputTest {
|
||||
|
||||
private val context = InstrumentationRegistry.getInstrumentation().targetContext
|
||||
private val workManager = WorkManager.getInstance(context)
|
||||
|
||||
@After
|
||||
fun tearDown() {
|
||||
File(context.cacheDir, "conversions").listFiles()?.forEach { it.delete() }
|
||||
}
|
||||
|
||||
@Test
|
||||
fun aContentUriInputConvertsThroughTheSafBridge(): Unit = runBlocking {
|
||||
val input = FixtureContentProvider.uriFor(SAMPLE)
|
||||
val request = ConversionWorker.request(
|
||||
inputUri = input,
|
||||
displayName = SAMPLE,
|
||||
sizeBytes = 0L,
|
||||
spec = OutputFormat.MP3.spec,
|
||||
quality = QualityTier.FAST,
|
||||
)
|
||||
workManager.enqueue(request).result.get()
|
||||
|
||||
val terminal = withTimeout(TIMEOUT_MS) {
|
||||
workManager.getWorkInfoByIdFlow(request.id).first { it != null && it.state.isFinished }
|
||||
}
|
||||
|
||||
val error = terminal?.outputData?.getString(ConversionWorker.KEY_ERROR)
|
||||
assertEquals(
|
||||
"a content:// input must convert, but failed with: $error",
|
||||
WorkInfo.State.SUCCEEDED,
|
||||
terminal?.state,
|
||||
)
|
||||
// The bridge is on the FFmpeg arm only, so this is part of the claim rather than colour.
|
||||
assertEquals(Engine.FFMPEG.name, terminal?.outputData?.getString(ConversionWorker.KEY_ENGINE_USED))
|
||||
|
||||
val out = File(terminal!!.outputData.getString(ConversionWorker.KEY_OUTPUT_PATH)!!)
|
||||
assertTrue("no output produced from a content:// input", out.exists() && out.length() > 0)
|
||||
out.delete()
|
||||
}
|
||||
|
||||
/**
|
||||
* The same bridge on the join path, which has its own copy of the call (`ConcatEngine:36`).
|
||||
*
|
||||
* Driven through the engine rather than `ConcatWorker` because the engine is where the branch
|
||||
* is; the worker adds a foreground service and nothing else this is about.
|
||||
*/
|
||||
@Test
|
||||
fun contentUriInputsJoinThroughTheSafBridge(): Unit = runBlocking {
|
||||
val out = File(context.cacheDir, "joined_from_content.mp4").apply { delete() }
|
||||
val result = ConcatEngine(context).join(
|
||||
listOf(FixtureContentProvider.uriFor(CLIP_A), FixtureContentProvider.uriFor(CLIP_B)),
|
||||
out,
|
||||
OutputFormat.MP4_H264,
|
||||
)
|
||||
|
||||
assertTrue("no output produced from content:// inputs", result.output.length() > 0)
|
||||
out.delete()
|
||||
}
|
||||
|
||||
private companion object {
|
||||
const val SAMPLE = "sample_h264.mp4"
|
||||
const val CLIP_A = "clip_a.mp4"
|
||||
const val CLIP_B = "clip_b.mp4"
|
||||
const val TIMEOUT_MS = 300_000L
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,135 @@
|
||||
package org.libremediaconverter.saf;
|
||||
|
||||
import android.content.ContentProvider;
|
||||
import android.content.ContentValues;
|
||||
import android.database.Cursor;
|
||||
import android.database.MatrixCursor;
|
||||
import android.net.Uri;
|
||||
import android.os.ParcelFileDescriptor;
|
||||
import android.provider.OpenableColumns;
|
||||
|
||||
import java.io.File;
|
||||
import java.io.FileNotFoundException;
|
||||
import java.io.FileOutputStream;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.io.OutputStream;
|
||||
|
||||
/**
|
||||
* A plain {@link ContentProvider} serving the committed media fixtures over {@code content://}.
|
||||
*
|
||||
* <p><b>Why this exists alongside {@link FixtureDocumentsProvider}.</b> Every passing convert and
|
||||
* join test hands the worker a {@code Uri.fromFile(...)}, which takes the {@code uri.path} arm and
|
||||
* never touches {@code FFmpegKitConfig.getSafParameterForRead}. That bridge is on 100% of real user
|
||||
* conversions and was on 0% of tested ones; only its failure side was covered, by
|
||||
* {@code UnopenableUriTest} pointing at an authority that does not exist.
|
||||
*
|
||||
* <p><b>Why not the documents provider.</b> It cannot be reached. Measured three ways on an API 34
|
||||
* emulator: a {@code DOCUMENTS_PROVIDER} declared without {@code MANAGE_DOCUMENTS} is refused at
|
||||
* install ("Provider must be protected by MANAGE_DOCUMENTS"); instrumentation runs in the target
|
||||
* app's process, so {@code Instrumentation.getContext()} still carries the app's uid and is denied;
|
||||
* and {@code adoptShellPermissionIdentity(MANAGE_DOCUMENTS)} is denied identically. The denial says
|
||||
* what is required — <i>"you obtain access using ACTION_OPEN_DOCUMENT or related APIs"</i> — so a
|
||||
* documents provider is reachable only through a picker-issued grant. See issue #226.
|
||||
*
|
||||
* <p>The bridge does not need one. {@code getSafParameterForRead} opens a file descriptor through
|
||||
* the resolver and hands FFmpeg a {@code saf:} path; any readable {@code content://} URI exercises
|
||||
* it. An ordinary provider may be exported without a permission, so this one is, and the whole test
|
||||
* stays headless — no DocumentsUI, and none of the flake #190 records.
|
||||
*
|
||||
* <p>Unlike {@link FixtureDocumentsProvider} this may use {@code androidx} and Kotlin freely — it is
|
||||
* loaded into the app process like any other provider, not into the bare test process. It is kept
|
||||
* in Java anyway, next to its sibling, so the two read alike.
|
||||
*/
|
||||
public final class FixtureContentProvider extends ContentProvider {
|
||||
|
||||
/** Authority. Distinct from the documents provider's, and from anything the app declares. */
|
||||
public static final String AUTHORITY = "org.libremediaconverter.test.content";
|
||||
|
||||
/** Builds a URI for one of this source set's committed assets, e.g. {@code sample_h264.mp4}. */
|
||||
public static Uri uriFor(String assetName) {
|
||||
return new Uri.Builder().scheme("content").authority(AUTHORITY).appendPath(assetName).build();
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onCreate() {
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public ParcelFileDescriptor openFile(Uri uri, String mode) throws FileNotFoundException {
|
||||
if (!"r".equals(mode)) {
|
||||
throw new FileNotFoundException("this provider is read-only: " + mode);
|
||||
}
|
||||
return ParcelFileDescriptor.open(unpack(assetOf(uri)), ParcelFileDescriptor.MODE_READ_ONLY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Enough of {@link OpenableColumns} for {@code InputQuery.describe} to name and size the input.
|
||||
*
|
||||
* <p>Without these the app reaches the "Size unknown" screen, which is a different test.
|
||||
*/
|
||||
@Override
|
||||
public Cursor query(Uri uri, String[] projection, String selection, String[] args, String sort) {
|
||||
String asset = assetOf(uri);
|
||||
File file;
|
||||
try {
|
||||
file = unpack(asset);
|
||||
} catch (FileNotFoundException e) {
|
||||
return null;
|
||||
}
|
||||
MatrixCursor cursor = new MatrixCursor(
|
||||
new String[] {OpenableColumns.DISPLAY_NAME, OpenableColumns.SIZE});
|
||||
cursor.newRow().add(OpenableColumns.DISPLAY_NAME, asset).add(OpenableColumns.SIZE, file.length());
|
||||
return cursor;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String getType(Uri uri) {
|
||||
return assetOf(uri).endsWith(".m4a") ? "audio/mp4" : "video/mp4";
|
||||
}
|
||||
|
||||
@Override
|
||||
public Uri insert(Uri uri, ContentValues values) {
|
||||
throw new UnsupportedOperationException("read-only fixture provider");
|
||||
}
|
||||
|
||||
@Override
|
||||
public int delete(Uri uri, String selection, String[] args) {
|
||||
throw new UnsupportedOperationException("read-only fixture provider");
|
||||
}
|
||||
|
||||
@Override
|
||||
public int update(Uri uri, ContentValues values, String selection, String[] args) {
|
||||
throw new UnsupportedOperationException("read-only fixture provider");
|
||||
}
|
||||
|
||||
private static String assetOf(Uri uri) {
|
||||
String asset = uri.getLastPathSegment();
|
||||
return asset == null ? "" : asset;
|
||||
}
|
||||
|
||||
/**
|
||||
* The asset on disk, unpacked the first time anything asks.
|
||||
*
|
||||
* <p>Reported as {@link FileNotFoundException} rather than swallowed: a provider answering with
|
||||
* a zero-byte file would fail the conversion for a reason nothing states.
|
||||
*/
|
||||
private File unpack(String asset) throws FileNotFoundException {
|
||||
File file = new File(getContext().getCacheDir(), "provided_" + asset);
|
||||
if (file.length() > 0L) {
|
||||
return file;
|
||||
}
|
||||
try (InputStream source = getContext().getAssets().open(asset);
|
||||
OutputStream sink = new FileOutputStream(file)) {
|
||||
byte[] buffer = new byte[8192];
|
||||
int read;
|
||||
while ((read = source.read(buffer)) != -1) {
|
||||
sink.write(buffer, 0, read);
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new FileNotFoundException("could not unpack " + asset + ": " + e);
|
||||
}
|
||||
return file;
|
||||
}
|
||||
}
|
||||
@@ -298,7 +298,35 @@ class SafPickerRoundTripTest {
|
||||
device.waitForIdle()
|
||||
}
|
||||
|
||||
/**
|
||||
* **Marked for API 37 because of what it does to the image, not because it fails there.**
|
||||
*
|
||||
* This is the one place the marker's KDoc phrase "cannot pass on this image" does not fit, and
|
||||
* the distinction is worth keeping rather than smoothing over. Across the four gating API 37
|
||||
* runs whose logcats were read on 2026-09-05 — 34006456986, 34001744574, 34001377499 and the
|
||||
* green 34002313300 — the leg carries exactly two `hasReadColorBufferDma` aborts before the
|
||||
* suite starts (both `surfaceflinger`, during boot and the SystemUI disable) and then exactly
|
||||
* **one** during it. Every time, that one is `system_server` on the `TaskSnapshotPer` thread,
|
||||
* and every time it lands inside this test's window. No other test in the gating set reaches
|
||||
* the mapper at all.
|
||||
*
|
||||
* So this test kills the framework on that image whether it passes or not, and whether the leg
|
||||
* goes red is luck: 34001377499 passed it and lost the leg anyway (`failed: 0`, teardown
|
||||
* broken), 34002313300 passed it 0.6 s after the abort and went green. That is #108, and it is
|
||||
* why the leg was failing on unrelated PRs.
|
||||
*
|
||||
* `docs/api-37-emulator-crash.md` measured this test on 2026-08-24, recorded "passes, 4 aborts
|
||||
* in the window", and concluded that a rotation reaches the mapper where starting DocumentsUI
|
||||
* does not. The aborts were seen; what was not drawn out is that they are this test's own and
|
||||
* are not intermittent.
|
||||
*
|
||||
* The marker is what routes it off the gating leg and into the advisory job beside its
|
||||
* rotation sibling. **It is not a statement about the picker**: the same test passes on API
|
||||
* 33–36 on the same runner and on the Pixel 10 Pro XL, which is where API 37's answer comes
|
||||
* from.
|
||||
*/
|
||||
@Test
|
||||
@FailsOnEmulatorApi37
|
||||
fun pickingAFileThroughTheSystemPickerFillsInTheFileCard() {
|
||||
pickTheFixture()
|
||||
|
||||
@@ -604,6 +632,9 @@ class SafPickerRoundTripTest {
|
||||
* It is also why this counts backs rather than pressing a fixed number of them. One back is
|
||||
* enough from Recent and two are needed from inside the root, but a third from Recent would
|
||||
* finish `MainActivity` and take the rest of the test with it.
|
||||
*
|
||||
* **[forceStopThePicker] is the escalation after the presses, and it exists because a back
|
||||
* press is not always deliverable.** See its own KDoc for the measurement.
|
||||
*/
|
||||
private fun dismissThePicker() {
|
||||
repeat(BACK_PRESSES) {
|
||||
@@ -618,15 +649,50 @@ class SafPickerRoundTripTest {
|
||||
// The check after the last press, and not a spare one: `repeat` presses on its final
|
||||
// iteration too, so without this a dismissal that worked on the last press would still be
|
||||
// reported as a failure to close.
|
||||
if (awaitAppFocus()) return
|
||||
forceStopThePicker()
|
||||
if (!awaitAppFocus()) {
|
||||
throw AssertionError(
|
||||
"the system picker would not close: after $BACK_PRESSES back presses the app " +
|
||||
"still does not have the window focus, and ${device.currentPackageName} is " +
|
||||
"in front. What could be seen: " + describeWindows(),
|
||||
"the system picker would not close: after $BACK_PRESSES back presses and a " +
|
||||
"force-stop of $DOCUMENTS_UI_PACKAGE the app still does not have the window " +
|
||||
"focus, and ${device.currentPackageName} is in front. What could be seen: " +
|
||||
describeWindows(),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Kills the picker's process, for when no back press can reach it.
|
||||
*
|
||||
* **The failure this exists for cannot be answered with input, and that is the whole point.**
|
||||
* Measured on the gating API 37 legs of runs 34006456986 and 34001744574, which fail this way
|
||||
* and whose logcats say the same thing in the same order. `UiObject2.click()` on the fixture's
|
||||
* root is injected at the node's centre and the framework discards it —
|
||||
* `InputDispatcher: No new touched window at (539.0, 525.0) in display 0` — because
|
||||
* `PickActivity` has published accessibility nodes but has no touchable window there yet.
|
||||
* `click()` cannot see that and returns normally, so the walk goes on to wait out
|
||||
* [PICKER_TIMEOUT_MS] for a fixture that was never navigated to. By the time this function's
|
||||
* caller starts pressing back, WindowManager is still saying
|
||||
* `no window has focus but ...PickActivity may eventually add a window when it finishes
|
||||
* starting up` — and goes on saying it for another 63 s. Every one of the four presses is
|
||||
* dropped, and DocumentsUI ANRs on `Input dispatching timed out`.
|
||||
*
|
||||
* So the picker is in front, unreachable by key or by touch, and [pickTheFixture]'s whole
|
||||
* point — that a second `PickActivity` rebuilds every window and list in it — is unreachable
|
||||
* with it. `am force-stop` goes around input entirely: `UiAutomation` runs shell commands as
|
||||
* uid 2000, which holds `FORCE_STOP_PACKAGES`, so the picker's process is killed, its
|
||||
* activity leaves the task it was launched into, and `MainActivity` — the activity below it in
|
||||
* that same task — is resumed with the focus.
|
||||
*
|
||||
* **Only on the failure path**, after every back press has been spent, so a picker that closes
|
||||
* the ordinary way never reaches this and is not altered by it. If the framework itself is
|
||||
* gone, this cannot help either, and the caller still reports what it could see.
|
||||
*/
|
||||
private fun forceStopThePicker() {
|
||||
device.executeShellCommand("am force-stop $DOCUMENTS_UI_PACKAGE")
|
||||
device.waitForIdle()
|
||||
}
|
||||
|
||||
/** True once [MainActivity] has the window focus, false if it does not take it in time. */
|
||||
private fun awaitAppFocus(): Boolean = try {
|
||||
composeRule.waitUntil("the app has the window focus back", FOCUS_TIMEOUT_MS) {
|
||||
|
||||
+129
@@ -0,0 +1,129 @@
|
||||
package org.libremediaconverter.work
|
||||
|
||||
import android.net.Uri
|
||||
import androidx.media3.common.util.UnstableApi
|
||||
import androidx.test.ext.junit.runners.AndroidJUnit4
|
||||
import androidx.test.platform.app.InstrumentationRegistry
|
||||
import androidx.work.OneTimeWorkRequestBuilder
|
||||
import androidx.work.WorkInfo
|
||||
import androidx.work.WorkManager
|
||||
import kotlinx.coroutines.flow.first
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import org.junit.After
|
||||
import org.junit.Assert.assertEquals
|
||||
import org.junit.Assert.assertNotNull
|
||||
import org.junit.Before
|
||||
import org.junit.Test
|
||||
import org.junit.runner.RunWith
|
||||
import org.libremediaconverter.model.OutputFormat
|
||||
import org.libremediaconverter.model.QualityTier
|
||||
import java.io.File
|
||||
import java.util.concurrent.TimeUnit
|
||||
|
||||
/**
|
||||
* The Cancel button in the notification shade actually cancels the job.
|
||||
*
|
||||
* `ConversionNotifications.build` attaches one action, wired to
|
||||
* `WorkManager.createCancelPendingIntent(id)`. Before this test `createCancelPendingIntent` had
|
||||
* **no references anywhere outside its own declaration** — no JVM test, no instrumented test
|
||||
* (#227).
|
||||
*
|
||||
* That matters more than an ordinary uncovered line. A conversion runs in a foreground service and
|
||||
* the user is invited to leave the app; once they do, this action is the only way to stop it. If
|
||||
* the `PendingIntent` carries the wrong id, the button does nothing, the notification stays, and
|
||||
* the job runs to completion — with no error, no log, and no screen to look at.
|
||||
*
|
||||
* ## Why this fires the intent rather than reading the shade
|
||||
*
|
||||
* The obvious version asks `NotificationManager.getActiveNotifications()` for id 1001 and taps what
|
||||
* it finds. That was rejected: the instrumented suite grants no runtime permissions, so
|
||||
* `POST_NOTIFICATIONS` is denied throughout, and whether a suppressed foreground-service
|
||||
* notification is returned there is a platform detail that varies — the test would be asserting
|
||||
* something about notification *visibility* rather than about cancellation.
|
||||
*
|
||||
* The `PendingIntent` is the subject; where it is read from is incidental. Building the
|
||||
* notification for a real, live work id and firing its action exercises exactly the thing that can
|
||||
* be wrong — a real `PendingIntent` dispatch reaching real `WorkManager` — and does it the same way
|
||||
* on every API level.
|
||||
*
|
||||
* ## Why the job is delayed rather than running
|
||||
*
|
||||
* A conversion of the committed 3 s fixture finishes in well under a second on an emulator
|
||||
* (`HardwareFallbackTest` completed one in 448 ms), so racing a cancel against a running job would
|
||||
* be flaky in the direction that fails. An initial delay keeps the job reliably `ENQUEUED`, which
|
||||
* is a state `cancelWorkById` acts on identically — what is under test is whether firing the action
|
||||
* reaches WorkManager with the right id, not which state it interrupts.
|
||||
*
|
||||
* *Mutation:* build the `PendingIntent` from `UUID.randomUUID()` instead of the request's id. The
|
||||
* notification looks identical and the job is never cancelled.
|
||||
*/
|
||||
@UnstableApi
|
||||
@RunWith(AndroidJUnit4::class)
|
||||
class NotificationCancelActionTest {
|
||||
|
||||
private val context = InstrumentationRegistry.getInstrumentation().targetContext
|
||||
private val workManager = WorkManager.getInstance(context)
|
||||
private lateinit var input: File
|
||||
|
||||
@Before
|
||||
fun setUp() {
|
||||
input = File(context.cacheDir, "cancel_action_sample.mp4")
|
||||
InstrumentationRegistry.getInstrumentation().context.assets
|
||||
.open("sample_h264.mp4")
|
||||
.use { asset -> input.outputStream().use { asset.copyTo(it) } }
|
||||
}
|
||||
|
||||
@After
|
||||
fun tearDown() {
|
||||
input.delete()
|
||||
File(context.cacheDir, "conversions").listFiles()?.forEach { it.delete() }
|
||||
}
|
||||
|
||||
@Test
|
||||
fun theNotificationsCancelActionCancelsThatJob(): Unit = runBlocking {
|
||||
val request = ConversionWorker.request(
|
||||
inputUri = Uri.fromFile(input),
|
||||
displayName = input.name,
|
||||
sizeBytes = input.length(),
|
||||
spec = OutputFormat.MP4_H264.spec,
|
||||
quality = QualityTier.FAST,
|
||||
).let { base ->
|
||||
// Rebuild with a delay so the job stays ENQUEUED for the whole test. See the KDoc.
|
||||
OneTimeWorkRequestBuilder<ConversionWorker>()
|
||||
.setInputData(base.workSpec.input)
|
||||
.setInitialDelay(1, TimeUnit.HOURS)
|
||||
.build()
|
||||
}
|
||||
workManager.enqueue(request).result.get()
|
||||
|
||||
// The job is queued and waiting, which is the state the cancel has to interrupt.
|
||||
assertEquals(
|
||||
WorkInfo.State.ENQUEUED,
|
||||
withTimeout(TIMEOUT_MS) {
|
||||
workManager.getWorkInfoByIdFlow(request.id).first { it != null }
|
||||
}?.state,
|
||||
)
|
||||
|
||||
val notification = ConversionNotifications(context)
|
||||
.build(request.id, title = input.name, percent = 0, indeterminate = true)
|
||||
val action = notification.actions?.firstOrNull()
|
||||
assertNotNull("the progress notification carries no action to cancel with", action)
|
||||
|
||||
// The whole point: fire it the way the shade would, and see the job stop.
|
||||
action!!.actionIntent.send()
|
||||
|
||||
val terminal = withTimeout(TIMEOUT_MS) {
|
||||
workManager.getWorkInfoByIdFlow(request.id).first { it != null && it.state.isFinished }
|
||||
}
|
||||
assertEquals(
|
||||
"firing the notification's Cancel action must cancel the job it was built for",
|
||||
WorkInfo.State.CANCELLED,
|
||||
terminal?.state,
|
||||
)
|
||||
}
|
||||
|
||||
private companion object {
|
||||
const val TIMEOUT_MS = 30_000L
|
||||
}
|
||||
}
|
||||
@@ -35,6 +35,21 @@ object FFmpegConcatCommand {
|
||||
add("concat")
|
||||
add("-safe")
|
||||
add("0")
|
||||
// And -protocol_whitelist permits the *scheme* those paths carry, which is a
|
||||
// separate gate (#238). Every input the user actually picks is a content:// URI --
|
||||
// JoinScreen uses OpenMultipleDocuments -- so ConcatEngine maps it through
|
||||
// FFmpegKitConfig.getSafParameterForRead and writes an `ffkitsaf:` path into the
|
||||
// list file. The concat demuxer applies its own whitelist, defaulting to
|
||||
// "file,crypto,data", and refused every one of them:
|
||||
//
|
||||
// [ffkitsaf @ ...] Protocol 'ffkitsaf' not on whitelist 'file,crypto,data'!
|
||||
//
|
||||
// This only widens that default. It is on the stream-copy branch alone because it
|
||||
// is the only one that feeds the demuxer a list file -- REENCODE passes each input
|
||||
// with its own -i, where the whitelist does not apply, which is why joining over SAF
|
||||
// worked for mismatched clips and failed for matching ones.
|
||||
add("-protocol_whitelist")
|
||||
add(PROTOCOL_WHITELIST)
|
||||
add("-i")
|
||||
add(listFile.absolutePath)
|
||||
add("-c")
|
||||
@@ -84,4 +99,12 @@ object FFmpegConcatCommand {
|
||||
add(output.absolutePath)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The concat demuxer's protocol whitelist: FFmpeg's own default, plus ffmpeg-kit's SAF scheme.
|
||||
*
|
||||
* Spelled out rather than appended to an unknown default, because the default is FFmpeg's and
|
||||
* could change under us; naming all four keeps the command self-describing. See #238.
|
||||
*/
|
||||
private const val PROTOCOL_WHITELIST = "file,crypto,data,ffkitsaf"
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ import android.net.Uri
|
||||
import androidx.activity.ComponentActivity
|
||||
import androidx.compose.ui.test.assertIsDisplayed
|
||||
import androidx.compose.ui.test.junit4.v2.createAndroidComposeRule
|
||||
import androidx.compose.ui.test.onAllNodesWithTag
|
||||
import androidx.compose.ui.test.onNodeWithTag
|
||||
import androidx.compose.ui.test.performClick
|
||||
import androidx.media3.common.util.UnstableApi
|
||||
@@ -77,6 +78,28 @@ class LauncherWiringTest {
|
||||
* The transposition guard. A picked file has to reach `onInputPicked`, which is observable as
|
||||
* the screen arriving at `Ready` with the file card showing — `save()` from `Idle` returns at
|
||||
* its own guard and leaves nothing behind.
|
||||
*
|
||||
* ## Why this waits rather than asserting straight away (#220)
|
||||
*
|
||||
* `onInputPicked` does not reach `Ready` on the calling thread. It hops twice —
|
||||
* `withContext(pickDispatcher) { InputQuery.describe(...) }` and then the probe — and
|
||||
* `pickDispatcher` defaults to `Dispatchers.IO`, a real background thread that Compose's
|
||||
* idling does not know about. `deliver` therefore returns with the state still `Idle` more
|
||||
* often than not, and asserting immediately was a race the test usually won.
|
||||
*
|
||||
* It lost five times on CI in one day, on PRs whose diffs were instrumented tests and
|
||||
* documentation, which is what #220 was filed for. `waitUntil` polls through
|
||||
* `waitForIdle`, so it drains the main looper each time round and sees the recomposition that
|
||||
* the IO hop eventually posts back.
|
||||
*
|
||||
* **Injecting the dispatcher would be better and is not available here.** `pickDispatcher` is
|
||||
* a constructor parameter precisely so a test can pin it, but this test composes the real
|
||||
* `ConverterScreen`, which resolves its own ViewModel through `viewModel()` — the seam exists
|
||||
* one layer below the thing under test. Pinning it would mean not testing the launcher edge,
|
||||
* which is the whole point of this class.
|
||||
*
|
||||
* The wait does not weaken the assertion: transposing the two callbacks leaves the screen in
|
||||
* `Idle` forever, so it fails on the timeout with the same meaning it failed with before.
|
||||
*/
|
||||
@Test
|
||||
fun `a picked document is loaded as input rather than saved to`() {
|
||||
@@ -85,6 +108,11 @@ class LauncherWiringTest {
|
||||
composeRule.onNodeWithTag(TestTags.Converter.CHOOSE_FILE).performClick()
|
||||
deliver(Uri.parse("content://test/holiday.mkv"))
|
||||
|
||||
composeRule.waitUntil(PICK_TIMEOUT_MS) {
|
||||
composeRule.onAllNodesWithTag(TestTags.Converter.FILE_CARD_NAME)
|
||||
.fetchSemanticsNodes()
|
||||
.isNotEmpty()
|
||||
}
|
||||
composeRule.onNodeWithTag(TestTags.Converter.FILE_CARD_NAME).assertIsDisplayed()
|
||||
}
|
||||
|
||||
@@ -138,4 +166,13 @@ class LauncherWiringTest {
|
||||
)
|
||||
composeRule.waitForIdle()
|
||||
}
|
||||
|
||||
private companion object {
|
||||
/**
|
||||
* Long enough that a slow CI runner is not the reason this fails, short enough that a
|
||||
* genuinely transposed callback does not stall the suite. The pick normally lands in
|
||||
* single-digit milliseconds.
|
||||
*/
|
||||
const val PICK_TIMEOUT_MS = 10_000L
|
||||
}
|
||||
}
|
||||
|
||||
@@ -56,6 +56,33 @@ class FFmpegConcatCommandTest {
|
||||
assertEquals("0", args[args.indexOf("-safe") + 1])
|
||||
}
|
||||
|
||||
/**
|
||||
* The gate that `-safe 0` does not open, and the one every real join needs (#238).
|
||||
*
|
||||
* `-safe 0` permits absolute *paths*; the concat demuxer separately whitelists the *protocol*,
|
||||
* defaulting to `file,crypto,data`. `JoinScreen` picks with `OpenMultipleDocuments`, so real
|
||||
* inputs are `content://` and `ConcatEngine` writes `ffkitsaf:` paths into the list file — which
|
||||
* the demuxer refused outright, failing every stream-copy join a user could actually start.
|
||||
*
|
||||
* The re-encode strategy has no equivalent assertion because it needs none: it passes each
|
||||
* input with its own `-i` and never feeds the demuxer a list file. That asymmetry is exactly
|
||||
* why the defect survived — joining mismatched clips over SAF worked.
|
||||
*/
|
||||
@Test
|
||||
fun `stream copy whitelists the protocol its list file entries actually use`() {
|
||||
val args = FFmpegConcatCommand.build(
|
||||
ConcatStrategy.STREAM_COPY,
|
||||
inputs,
|
||||
listFile,
|
||||
output,
|
||||
OutputFormat.MP4_H264,
|
||||
)
|
||||
val whitelist = args[args.indexOf("-protocol_whitelist") + 1].split(",")
|
||||
assertTrue("ffmpeg-kit's SAF scheme must be permitted, got $whitelist", "ffkitsaf" in whitelist)
|
||||
// The defaults have to survive too: the list file itself is opened over `file`.
|
||||
assertTrue("the demuxer still reads the list file itself, got $whitelist", "file" in whitelist)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun `re-encode passes every input separately and builds a filter graph`() {
|
||||
val args = FFmpegConcatCommand.build(
|
||||
|
||||
+189
-18
@@ -315,25 +315,90 @@ clean zero. Its own post-disable check on the run recorded below printed
|
||||
|
||||
So what is reliably achieved is a **rate collapse** — from roughly one abort every fourteen
|
||||
seconds to one every forty-five — which a 47-second Gradle run survives and a five-minute one
|
||||
might not. The 180-second zero above is one measurement on a device that had been up for twelve
|
||||
minutes and had already cycled its framework several times. The harness prints the quiet-check
|
||||
delta on every run precisely so this is visible rather than assumed.
|
||||
might not.
|
||||
|
||||
One ordering detail cost a whole run and is now encoded in `disable_region_sampling`: by the time
|
||||
`sys.boot_completed` flips, SystemUI has **already registered**, and `pm disable-user` does not
|
||||
retract an existing registration — it only stops the package being started again. Disabling it
|
||||
and proceeding straight to the tests fails exactly as before. The harness therefore does
|
||||
`stop; start` afterwards, so the framework that comes back never starts SystemUI at all.
|
||||
**And that restart has never happened — which is how the disable turned out not to work either.**
|
||||
Corrected 2026-09-05; this replaces the two paragraphs above rather than qualifying them.
|
||||
|
||||
`adb shell stop` and `start` are root-only, adbd is not root on a booted emulator, and all three
|
||||
copies of this logic called them without `adb root`. On CI both printed `Must be root`, between
|
||||
lines that read as if the restart had happened; `run-e2e.sh` sent them to `/dev/null`, so its
|
||||
`Must be root` was never even visible. Neither number in those logs was an observation either —
|
||||
the `pidof` loop breaks when the process is gone and otherwise falls out at its last iteration,
|
||||
and the old code printed the iteration count either way, so `system_server down after ~40 s` is
|
||||
what a stop that did nothing looks like.
|
||||
|
||||
Adding `adb root` made the restart real, and **that is what proved the disable ineffective**.
|
||||
`api37-debug` run 34010167885, `disable_system_ui=true`:
|
||||
|
||||
```
|
||||
--- disable round 1 ---
|
||||
pm attempt 1: Package com.android.systemui new state: disabled-user
|
||||
restarting the framework
|
||||
adbd is running as root
|
||||
system_server down after 2 s
|
||||
services back after 10 s
|
||||
NOT DISABLED after the restart -- the package state did not survive
|
||||
```
|
||||
|
||||
Three rounds of that, then `final state: SystemUI STILL ENABLED`, and the leg reported
|
||||
`expected: 0, received: 0` — `Starting 0 tests`, the exact failure this function exists to
|
||||
prevent.
|
||||
|
||||
Bisected locally on `android-37.0`, which explains the lost state and nothing else:
|
||||
|
||||
| arm | sequence | disabled after the restart? |
|
||||
|---|---|---|
|
||||
| A | `pm disable-user`, then `stop` at once | **no** |
|
||||
| B | `pm disable-user`, wait 15 s, then `stop` | **yes** |
|
||||
|
||||
That is PackageManager's delayed write of package restrictions: the stop kills `system_server`
|
||||
before the settings are flushed, and arm A is what CI did. **Arm B does not help either**, which
|
||||
is the measurement that matters. With the package verified `disabled-user` before *and* after a
|
||||
further clean restart:
|
||||
|
||||
```
|
||||
package still disabled? YES
|
||||
processes:
|
||||
9275 00:17 system_server
|
||||
9695 00:14 com.android.systemui <- started 3 s after system_server
|
||||
```
|
||||
|
||||
CI's own logcat says the same without any restart at all. In the gating leg of run 34006456986,
|
||||
`pm disable-user` is accepted at 02:28:37.9 and the package really is in `pm list packages -d` at
|
||||
02:29:33 — and SystemUI is started at 02:28:39.5 and again at 02:28:52.3, the second of which
|
||||
(pid 4275) is alive for the whole instrumentation run.
|
||||
|
||||
**So `pm disable-user --user 0 com.android.systemui` does not stop SystemUI starting on this
|
||||
image**, with or without a framework restart, on CI or locally. The premise this section was
|
||||
built on — "the framework that comes back never starts SystemUI at all" — is false.
|
||||
|
||||
Two things follow, pointing in opposite directions.
|
||||
|
||||
- **The restart is removed rather than repaired**, in all three copies. It cost a leg every test
|
||||
it had and there is nothing for it to buy. What is kept is the 45-second window with zero new
|
||||
aborts, which was always the part doing the work: in that same run the boot aborts land at
|
||||
02:28:18 and 02:28:43, and the wait is what puts instrumentation at 02:32:42 — after them
|
||||
rather than inside one. The `pm disable-user` call is kept too, for a narrower reason than it
|
||||
was written for: every green leg and every number quoted about this row was measured with it
|
||||
applied, and changing the configuration while fixing a flake is not a trade worth making.
|
||||
- **The rate collapse recorded above is not evidence of what it says.** Both arms of that
|
||||
comparison had SystemUI running. What it measured is a device twelve minutes into its uptime
|
||||
against one that had just booted — a real difference, and a different claim. The quiet gate is
|
||||
still worth having on exactly that reading.
|
||||
|
||||
### The two deviations, stated plainly
|
||||
|
||||
1. **The renderer is ANGLE, not the host GPU.** Shared with nothing else in the matrix — API
|
||||
33–36 run `-gpu host` locally, and CI runs `swiftshader_indirect`.
|
||||
2. **SystemUI is disabled.** The API 37 leg does not run the same device configuration as any
|
||||
other leg or as the Pixel. It was defensible here because nothing in this suite touched
|
||||
system UI — Media3, FFmpeg and WorkManager tests — and because the alternative is no local
|
||||
API 37 coverage at all. **Anything that ever does depend on system UI must not trust this
|
||||
leg.** Something now does; see the section below.
|
||||
2. **SystemUI is asked to be disabled, and runs anyway.** This was written as the deviation that
|
||||
mattered — "anything that ever does depend on system UI must not trust this leg" — and the
|
||||
measurements above say the deviation does not exist: the package is marked `disabled-user` and
|
||||
`com.android.systemui` is up for the whole leg regardless. **The correction is good news
|
||||
rather than bad.** This row is *more* comparable to API 33–36 and to the Pixel than it has
|
||||
been claiming, not less, and the test that depends on system UI (see the section below) was
|
||||
never running in the exotic configuration this bullet describes. What `pm disable-user` leaves
|
||||
behind is a package-manager flag nothing acts on.
|
||||
|
||||
### Something does depend on system UI now, and half of it is excluded
|
||||
|
||||
@@ -341,12 +406,13 @@ Added 2026-08-24, and the first entry on this page that is not a codec.
|
||||
|
||||
`SafPickerRoundTripTest` drives the real system file picker and rotates the display. Both reach
|
||||
the gralloc mapper — DocumentsUI is another app's windows, and a rotation rebuilds every surface
|
||||
on screen — and **disabling SystemUI does not help**, because it removes the *idle* trigger
|
||||
(RegionSamplingThread's nav-bar luma sampling) and not this one.
|
||||
on screen — and **disabling SystemUI does not help**. Two reasons now, and only the first was
|
||||
known when this was written: it removes the *idle* trigger (RegionSamplingThread's nav-bar luma
|
||||
sampling) and not this one, and — see the section above — it does not remove SystemUI either.
|
||||
|
||||
Measured one method per fresh emulator, `android-37.0`, `swangle_indirect`, SystemUI disabled and
|
||||
verified quiet — separately, because inferring the second from the first is the mistake this
|
||||
page's opening correction is about:
|
||||
Measured one method per fresh emulator, `android-37.0`, `swangle_indirect`, with the disable
|
||||
applied and verified quiet — separately, because inferring the second from the first is the
|
||||
mistake this page's opening correction is about:
|
||||
|
||||
| test | result on android-37.0 | `hasReadColorBufferDma` aborts in the window |
|
||||
|---|---|---|
|
||||
@@ -357,6 +423,111 @@ So a rotation, which rebuilds every surface at once, is what the mapper does not
|
||||
starting DocumentsUI is not. Only the rotation test carries `@FailsOnEmulatorApi37`; the picker
|
||||
test runs on the gating leg like anything else.
|
||||
|
||||
#### That last sentence was wrong for twelve days, and the aborts in the table said so
|
||||
|
||||
**Corrected 2026-09-05.** Read the second row again: the picker test passes *and takes four
|
||||
`hasReadColorBufferDma` aborts with it*. This section counted them, put them in the table, and then
|
||||
drew the conclusion from the pass/fail column alone. The right question is not "does the test
|
||||
pass" but "does the image survive it", and the answer had been printed in the right-hand column
|
||||
from the day it was written.
|
||||
|
||||
Four gating API 37 runs read logcat-first — 34006456986, 34001744574, 34001377499, and the **green**
|
||||
34002313300 — say it without ambiguity. Each carries exactly two aborts before the suite starts
|
||||
(both `surfaceflinger`, during boot and the SystemUI disable) and then exactly **one** during it:
|
||||
|
||||
| run | picker test window | the run's only in-suite abort | leg |
|
||||
|---|---|---|---|
|
||||
| 34006456986 | 02:33:04.2 → 02:34:46.9, **failed** | 02:34:46.845 | red, `failed: 1` |
|
||||
| 34001744574 | 00:55:41.4 → 00:57:23.9, **failed** | 00:57:23.794 | red, `failed: 1` |
|
||||
| 34001377499 | 00:35:53.3 → 00:36:00.6, passed | 00:35:59.662 | red, `failed: 0` |
|
||||
| 34002313300 | 00:58:12.7 → 00:58:19.8, passed | 00:58:19.218 | green |
|
||||
|
||||
Every one is `system_server`, thread `TaskSnapshotPer`, and every one lands inside that test's
|
||||
window. Nothing else in the gating set reached the mapper at all. So the picker test is
|
||||
**deterministic** in what it does to the image and a coin flip in what the leg reports: 34001377499
|
||||
passed it and lost the leg from teardown with no failing test to name, and 34002313300 passed it
|
||||
0.6 s after the abort and went green.
|
||||
|
||||
That is #108, which had been filed against this behaviour in August and left open because the
|
||||
trigger was unknown. The trigger is this test. It now carries `@FailsOnEmulatorApi37` too, and the
|
||||
marker's KDoc had to widen from "does not pass on this image" to "cannot be run on this image" to
|
||||
say so honestly.
|
||||
|
||||
The stack, for the record, is a different caller from either of the two above:
|
||||
|
||||
```
|
||||
Cmdline: system_server name: TaskSnapshotPer
|
||||
Abort message: 'Assertion failed: !rcEnc->featureInfo()->hasReadColorBufferDma'
|
||||
|
||||
#04 mapper.ranchu.so GoldfishMapper::readFromHost(cb_handle_t const&) const+543
|
||||
#06 libui.so android::Gralloc5Mapper::lock(...)+63
|
||||
#10 libandroid_runtime.so android::lockImageFromBuffer(...)+374
|
||||
#15 framework.jar android.media.ImageReader$SurfaceImage.getPlanes+50
|
||||
#17 services.jar com.android.server.wm.TaskSnapshotConvertUtil.copyToSwBitmapDirect+56
|
||||
#28 services.jar com.android.server.wm.SnapshotPersistQueue$StoreWriteQueueItem.writeBuffer+66
|
||||
#32 services.jar com.android.server.wm.SnapshotPersistQueue$1.run+186
|
||||
```
|
||||
|
||||
WindowManager writing a task snapshot to disk, which needs the buffer as a software bitmap, which
|
||||
is the non-DMA readback path. `PickActivity` is started **into the app's own task** (`Task #11
|
||||
A=10234:org.libremediaconverter` in the logcat), so the snapshot being persisted is that task's,
|
||||
and the churn at the end of the pick is what schedules it.
|
||||
|
||||
#### There is no shell knob for task snapshots, and that was checked rather than assumed
|
||||
|
||||
#108 asks whether `TaskSnapshotPersister` is suppressible the way the region-sampling listener was.
|
||||
Probed on a local `android-37.0 google_apis x86_64` AVD, 2026-09-05:
|
||||
|
||||
```
|
||||
getprop | grep -i snapshot # nothing but apexd-snapshotde
|
||||
settings list global | grep -iE 'snapshot|recents' # empty
|
||||
device_config list window_manager | grep -i snapshot # empty
|
||||
cmd window help # no snapshot or screenshot command
|
||||
dumpsys window | grep -i snapshot # mSnapshotEnabled=true, for Task and Activity
|
||||
```
|
||||
|
||||
`mSnapshotEnabled` is real state and there is nothing that sets it from outside. The only
|
||||
`device_config` hits anywhere in the tree are aconfig flags — e.g.
|
||||
`windowing_frontend/com.android.window.flags.respect_requested_task_snapshot_resolution` — which
|
||||
tune the snapshot rather than disable it. So the marker is the available answer, not the lazy one.
|
||||
|
||||
#### When the picker test does fail, the abort is the coda and not the cause
|
||||
|
||||
Worth separating, because the failure message points the wrong way. In both runs where the test
|
||||
itself went red, it had been broken for 98 seconds before the abort landed. The discriminator is
|
||||
one line, present in both reds and absent from the green:
|
||||
|
||||
```
|
||||
I/InputDispatcher: No new touched window at (539.0, 525.0) in display 0
|
||||
```
|
||||
|
||||
(539, 525) is the centre of the fixture's root row — the same coordinates the green run clicks.
|
||||
The touch reaches no window and is discarded; `UiObject2.click()` cannot see that and returns
|
||||
normally. DocumentsUI then logs nothing at all, where the green run logs `DocumentStack` and
|
||||
`Creating new directory loader` 40 ms after its click. The walk waits out its timeout twice for a
|
||||
fixture it never navigated to, and by the time the back presses start, WindowManager is still
|
||||
saying `no window has focus but ...PickActivity may eventually add a window when it finishes
|
||||
starting up` — for another 63 s. All four presses are dropped, DocumentsUI ANRs on
|
||||
`Input dispatching timed out`, and only *then* does the abort fire and make the failure message
|
||||
read `no windows at all`.
|
||||
|
||||
`SafPickerRoundTripTest.forceStopThePicker` is the answer to that half: `am force-stop` goes around
|
||||
input entirely, so the picker's process can be removed from a task no key press can reach and
|
||||
`pickTheFixture`'s whole-picker retry — which exists for exactly this — becomes reachable again.
|
||||
That is a fix to the test on every level, not to API 37.
|
||||
|
||||
**It was made to bite before it was believed.** On a local API 36 emulator, with the walk cut short
|
||||
so the picker is left open and in front and with `device.pressBack()` removed, so that nothing but
|
||||
the force-stop can close it:
|
||||
|
||||
| | result |
|
||||
|---|---|
|
||||
| with `forceStopThePicker()` | **passes** — `ActivityManager: Force stopping com.google.android.documentsui ... from pid 5334`, `Killing 5269:com.google.android.documentsui (adj 0)`, a second `PickActivity` opens, the retry completes the pick |
|
||||
| with the one call removed | **fails** — `the system picker would not close: after 4 back presses ... com.google.android.documentsui is in front`, which is the API 37 failure verbatim |
|
||||
|
||||
The unmutated class passes on that emulator either way, which is the point of running the mutation
|
||||
at all: the recovery path is unreachable on a healthy device, so a green suite says nothing about it.
|
||||
|
||||
#### The correction that produced that table
|
||||
|
||||
**The first version of this section said both tests failed, and put the marker on the class.** The
|
||||
|
||||
+66
-11
@@ -1,6 +1,6 @@
|
||||
# E2E-read findings
|
||||
|
||||
**Status:** six findings, none fixed, none urgent — **plus one confirmed vacuous test, which is a
|
||||
**Status:** seven findings; E4 fixed, the rest standing, none urgent — **plus one confirmed vacuous test, which is a
|
||||
ticket rather than an entry here** (see [Not covered here](#not-covered-here)). `E1`–`E6` came from
|
||||
the 2026-09-05 read of the instrumented suite. Every entry here is a *test-suite* observation —
|
||||
something a new test would not fix, because the test already exists and the problem is what it
|
||||
@@ -242,6 +242,49 @@ visible skip or a red test, and that decision is the ticket's.
|
||||
|
||||
---
|
||||
|
||||
## E7 — a real `DocumentsProvider` cannot be reached without the picker, so there is no cheap SAF test
|
||||
|
||||
**Severity: n/a · Confirmed by measurement · this is a platform rule, not a gap**
|
||||
|
||||
Added 2026-09-06, from doing #225 and #226 rather than from reading.
|
||||
|
||||
`OutputPublisher.publish`'s destination side is asserted only against Robolectric fakes —
|
||||
`FakeSafProvider`, registered with `asDocumentsProvider = true`, which is the flag that *makes*
|
||||
`DocumentsContract.isDocumentUri` answer true. #226 split that into a cheap headless half (drive a
|
||||
real `DocumentsProvider` directly) and an expensive picker-driven half.
|
||||
|
||||
**The cheap half does not exist.** Three approaches, all measured on an API 34 emulator:
|
||||
|
||||
| approach | result |
|
||||
|---|---|
|
||||
| a second `DOCUMENTS_PROVIDER` declared **without** `MANAGE_DOCUMENTS` | refused at install: `SecurityException: Provider must be protected by MANAGE_DOCUMENTS` |
|
||||
| create the document as the **test APK**, which owns the provider | denied — instrumentation runs *in the target app's process*, so it carries the app's uid whatever `Context` is asked |
|
||||
| `uiAutomation.adoptShellPermissionIdentity(MANAGE_DOCUMENTS)` | denied identically |
|
||||
|
||||
The denial names the only way in:
|
||||
|
||||
> `Permission Denial: opening provider …FixtureDocumentsProvider from
|
||||
> ProcessRecord{… org.libremediaconverter/u0a192} requires that you obtain access using
|
||||
> ACTION_OPEN_DOCUMENT or related APIs`
|
||||
|
||||
And the intent filter is not optional: without it `isDocumentUri` returns false, which is exactly
|
||||
the branch guarding `deletePartialOutput` — so a provider without the filter tests nothing the
|
||||
ticket is about.
|
||||
|
||||
**So any test of `publish` against a real `DocumentsProvider` must drive DocumentsUI**, and pays
|
||||
#190's flake tax. The work is one item at that cost, not two, and #226 was updated to say so.
|
||||
|
||||
### What this does *not* block, which is the useful half
|
||||
|
||||
`FFmpegKitConfig.getSafParameterForRead` — the bridge on every real conversion and join — needs no
|
||||
documents provider. It opens a descriptor through the resolver, so **any readable `content://` URI
|
||||
exercises it**, and an ordinary `ContentProvider` may be exported without a permission. That is what
|
||||
`FixtureContentProvider` is, and it made #225 headless.
|
||||
|
||||
**That distinction was worth the trouble**: the first test ever to hand the join path a real
|
||||
`content://` input found #238, a defect that broke joining for every user who picks matched files.
|
||||
The expensive gate protects the *destination* side; the *input* side never needed it.
|
||||
|
||||
## Summary
|
||||
|
||||
| ID | Finding | Severity | Evidence | Action |
|
||||
@@ -249,11 +292,12 @@ visible skip or a red test, and that decision is the ticket's.
|
||||
| E1 | `RemuxTest`'s KDoc claims engine assertions three of its tests correctly omit | low | confirmed by inspection; traced through `MEDIA3_CONTAINERS` | **one line of KDoc** — the tests are right |
|
||||
| E2 | Three of the 60 instrumented tests assert nothing; two never run | n/a | confirmed by inspection; `docs/local-emulator.md:305` | **no action** — deliberate; but 60 ≠ 60 |
|
||||
| E3 | `…AndReportsProgress` does not assert progress fired | low | confirmed by inspection; reason inline | **no action** — the name overstates, the KDoc corrects it |
|
||||
| E4 | The API 37 marker's KDoc says "two"; three tests carry it | low | confirmed by inspection; baseline const says 3 | **fix the sentence** |
|
||||
| E4 | The API 37 marker's KDoc says "two"; three tests carry it | low | confirmed by inspection; baseline const says 3 | **fixed** in #243 — it names the constant now |
|
||||
| E5 | `coverage-read-findings.md` F7's "uncovered" half is stale | low | confirmed by inspection; `RemuxTest.kt:111` drives it | **amend F7** — "device-only" stands, "uncovered" does not |
|
||||
| E6 | The device-capability assertion asks the class under test what to expect | low | confirmed by inspection; no third oracle exists on a device | **no action** — read with **#223** |
|
||||
| E7 | A real `DocumentsProvider` is unreachable without the picker, so #226 has no cheap half | n/a | measured three ways on API 34; each denial names `ACTION_OPEN_DOCUMENT` | **no action** — it re-scoped #226 |
|
||||
|
||||
**Five of the six are prose, not code**, and that is the shape of this read. The instrumented suite
|
||||
**Six of the seven are prose, not code**, and that is the shape of this read. The instrumented suite
|
||||
is in good condition: 57 of its 60 tests bite, the fixtures are committed with their generation
|
||||
recipes, and the one class that asserts nothing says so in its first line. What this read found is
|
||||
that **the suite's self-description has drifted from the suite** in five small places and one large
|
||||
@@ -312,14 +356,25 @@ decision, not a detail — see **E6** for why no third option exists — and **#
|
||||
|
||||
| # | Gap |
|
||||
|---|---|
|
||||
| **#223** | `HardwareFallbackTest` never attempts the hardware path on any emulator leg |
|
||||
| **#224** | Cancelling a *running* native session, in any of the three engines |
|
||||
| **#225** | No `content://` input has reached a *successful* conversion — the ffkitsaf bridge |
|
||||
| **#226** | `OutputPublisher.publish` against a real `DocumentsProvider`, and the SAF premise it rests on |
|
||||
| **#227** | The notification's Cancel action has never been fired |
|
||||
| **#228** | `encodesFlacLosslessAudio` and `encodesOpus` pass on any non-empty file |
|
||||
| **#229** | FFmpeg's progress percentage is computed everywhere and asserted nowhere |
|
||||
| **#230** | *(spike)* whether a running conversion's process can be killed under instrumentation |
|
||||
| # | Gap | Outcome |
|
||||
|---|---|---|
|
||||
| **#223** | `HardwareFallbackTest` never attempts the hardware path on any emulator leg | closed — it skips instead of passing vacuously |
|
||||
| **#224** | Cancelling a *running* native session, in any of the three engines | closed — all three engines |
|
||||
| **#225** | No `content://` input has reached a *successful* conversion — the ffkitsaf bridge | closed, and it found **#238** |
|
||||
| **#226** | `OutputPublisher.publish` against a real `DocumentsProvider` | **open** — re-scoped by E7; one picker-driven item, not two |
|
||||
| **#227** | The notification's Cancel action has never been fired | closed |
|
||||
| **#228** | `encodesFlacLosslessAudio` and `encodesOpus` pass on any non-empty file | closed |
|
||||
| **#229** | FFmpeg's progress percentage is computed everywhere and asserted nowhere | closed |
|
||||
| **#230** | *(spike)* whether a running conversion's process can be killed | closed — it cannot; the runner shares the app's process |
|
||||
|
||||
**The read's own result, once the tickets were worked: one production defect.** #238 — joining files
|
||||
picked through the system picker failed outright on the stream-copy path, because the concat demuxer
|
||||
whitelists protocols separately from `-safe 0` and `ffkitsaf` was not on the list. Only `STREAM_COPY`
|
||||
feeds the demuxer a list file, and every existing join test passed `Uri.fromFile`, so the one broken
|
||||
combination was the only one a user could reach.
|
||||
|
||||
That is the argument for this kind of read in one line: the gap was not a missed line or an
|
||||
unasserted value, it was **a combination of two covered things that no test put together**.
|
||||
|
||||
**Nothing here was filed as a coverage delta.** Each names the mutation that has to go red, which is
|
||||
the acceptance criterion wave 4 established and which caught two vacuous tests in that wave before
|
||||
|
||||
@@ -355,12 +355,10 @@ boot_emulator() {
|
||||
# may be in one of its restarts and `pm` is simply not published yet. The first attempt at this
|
||||
# failed exactly that way, with `cmd: Can't find service: package`.
|
||||
#
|
||||
# The framework restart at the end is not optional, and finding that out cost a run. By the
|
||||
# time `sys.boot_completed` flips, SystemUI has already registered its region-sampling listener,
|
||||
# and `pm disable-user` does not retract a registration that already happened -- it only stops
|
||||
# the package being started again. So the first attempt disabled SystemUI, reported success, and
|
||||
# then died exactly as before with `Starting 0 tests` and four more aborts. `stop; start` cycles
|
||||
# zygote deliberately, and the framework that comes back up does not start SystemUI at all.
|
||||
# This used to end with a framework restart, described here as "not optional". It was neither
|
||||
# optional nor happening -- see the block inside the function. What the first attempt's
|
||||
# `Starting 0 tests` and four more aborts actually showed is that a `pm disable-user` on its own
|
||||
# buys nothing, which is still true; what was wrong is the conclusion that a restart would.
|
||||
disable_region_sampling() {
|
||||
local api="$1" out i before after ready
|
||||
case "$api" in 37 | 37.*) ;; *) return 0 ;; esac
|
||||
@@ -383,14 +381,18 @@ disable_region_sampling() {
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo " restarting the framework so the region-sampling listener goes with it"
|
||||
emu_adb shell stop > /dev/null 2>&1
|
||||
emu_adb shell start > /dev/null 2>&1
|
||||
# There is no property worth waiting on here, and an earlier version of this only looked
|
||||
# like it was waiting on one: `stop` does not clear sys.boot_completed, so it still reads
|
||||
# `1` throughout the restart and any loop over it returns at once. The loop below is the
|
||||
# wait -- and it polls the better thing anyway, since `Can't find service: package` is the
|
||||
# failure it exists to prevent.
|
||||
# NO FRAMEWORK RESTART, and the two lines that used to be here are why this comment is long.
|
||||
# They were `emu_adb shell stop` and `emu_adb shell start`, both redirected to /dev/null, and
|
||||
# both root-only -- so what they printed there was `Must be root` and what they did was nothing,
|
||||
# here and in the two CI copies alike. Making them real (2026-09-05) is what established that
|
||||
# the disable never worked in the first place: with the package verified `disabled-user` before
|
||||
# AND after a clean restart on android-37.0, `com.android.systemui` comes up 3 s after
|
||||
# `system_server` regardless, and the same is visible in CI's own logcat. The restart also loses
|
||||
# the package state to PackageManager's delayed write if it lands too soon after the `pm` call,
|
||||
# which cost api37-debug run 34010167885 every test in the leg.
|
||||
#
|
||||
# So the useful part of this function is the quiet window below, not the disable. See
|
||||
# .github/scripts/e2e-run.sh's header, and docs/api-37-emulator-crash.md.
|
||||
ready=0
|
||||
for i in $(seq 1 30); do
|
||||
if emu_adb shell service check package 2> /dev/null | grep -q ': found' \
|
||||
|
||||
Reference in New Issue
Block a user