From b52f8c47e1cc33b59fb33f8c2ac657ec52c86164 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Wed, 2 Sep 2026 11:02:39 +0200 Subject: [PATCH 01/21] Fix Windows VM helper rejecting setgid source directories GNU chmod leaves setuid/setgid on directories for numeric modes of four digits or fewer, so chmod 0700 cannot satisfy the exact-700 mount check when ~/Windows was created with g+s. Harden with a-s,u=rwx,go= and print the observed modes when the check still fails. --- bin/omarchy-windows-vm | 16 +++++++++----- .../shell.d/windows-vm-mount-boundary-test.sh | 12 +++++++++++ test/shell.d/windows-vm-test.sh | 21 +++++++++++++++++++ 3 files changed, 44 insertions(+), 5 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 96a8d575509..af7a8e53bbb 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -554,6 +554,14 @@ bind_mount_leaf() { } } +# Make a directory mode 700, including leftover setuid/setgid. GNU chmod keeps +# those bits on directories for numeric modes of four digits or fewer, so +# `chmod 0700` cannot satisfy the exact-700 checks when ~/Windows was created +# setgid (omacom/omarchy#9698). +chmod_private_dir() { + chmod a-s,u=rwx,go= -- "$@" +} + prepare_caller_mounts() { local storage_fd storage_id shared_fd shared_id storage_mode shared_mode CALLER_MOUNTS_NEW_STORAGE=0 @@ -580,9 +588,7 @@ prepare_caller_mounts() { # Privacy is an explicit preflight step for both already-pinned sources, not # a side effect halfway through the two-mount transaction. Old umask-022 # installs are hardened together before either Docker-facing anchor changes. - # The mode is symbolic because a numeric chmod keeps setuid/setgid on a - # directory, and dockur marks an initially empty /shared setgid (2777). - chmod u=rwx,go=,a-s -- "/proc/$BASHPID/fd/$storage_fd" "/proc/$BASHPID/fd/$shared_fd" || { + chmod_private_dir "/proc/$BASHPID/fd/$storage_fd" "/proc/$BASHPID/fd/$shared_fd" || { exec {storage_fd}<&- exec {shared_fd}<&- return 1 @@ -1018,7 +1024,7 @@ prepare_user_mount_sources() { echo "omarchy-windows-vm: storage and shared must be different directories" >&2 return 1 } - chmod u=rwx,go=,a-s -- "$storage" "$shared" + chmod_private_dir "$storage" "$shared" || return 1 } storage_space_path() { @@ -1061,7 +1067,7 @@ write_credentials() { local username="$1" password="$2" old_umask dir tmp dir=$(dirname -- "$CREDENTIALS_FILE") mkdir -p "$dir" || return 1 - chmod 0700 "$dir" || return 1 + chmod_private_dir "$dir" || return 1 old_umask=$(umask) umask 077 tmp=$(mktemp "$dir/.credentials.XXXXXX") || { umask "$old_umask"; return 1; } diff --git a/test/shell.d/windows-vm-mount-boundary-test.sh b/test/shell.d/windows-vm-mount-boundary-test.sh index a18eede5475..0978b89ea6f 100644 --- a/test/shell.d/windows-vm-mount-boundary-test.sh +++ b/test/shell.d/windows-vm-mount-boundary-test.sh @@ -123,6 +123,18 @@ if setpriv --reuid=1001 --regid=1001 --clear-groups cat "$EXPECTED_SHARED/shared fi pass "cross-filesystem symlink sources bind by identity and migrated 0700 leaves deny another account" +# GNU chmod 0700 leaves directory setgid; leftover g+s used to fail the +# exact-700 check and block every privileged action (omacom/omarchy#9698). +chmod 2700 /home/storage-target +chmod 2777 /home/shared-target +chown 1000:1000 /home/storage-target /home/shared-target +with_vm_lock prepare_caller_mounts || fail "root could not harden setgid VM source directories" +[[ $(command stat -Lc '%a' /home/storage-target) == 700 ]] || fail "storage still had special bits after hardening" +[[ $(command stat -Lc '%a' /home/shared-target) == 700 ]] || fail "shared still had special bits after hardening" +[[ $(command stat -Lc '%u:%a' "$EXPECTED_STORAGE") == 1000:700 && + $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == 1000:700 ]] || fail "setgid hardening did not leave caller-owned 0700 anchors" +pass "setgid VM source directories harden to exactly 700" + # Existing production boundary components are never repaired in place when # their ownership or write permissions are unsafe. Both the preparation path # and the final pre-Docker guard must fail closed without disturbing the binds. diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index e9d200c5d49..9258fbfd38b 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -19,3 +19,24 @@ pass "Windows VM does not restart automatically at boot" rg -q 'title:"?Windows VM - Omarchy"' "$windows_vm_command" || fail "Windows VM launches FreeRDP with its expected title" pass "Windows VM launches FreeRDP with its expected title" + +# User-side source hardening must clear leftover directory setgid. GNU chmod +# 0700 does not, so a pre-existing ~/Windows mode 2700/2777 used to survive +# prepare_user_mount_sources and then fail the privileged exact-700 check. +( + test_home=$(mktemp -d) + trap 'rm -rf "$test_home"' EXIT + mkdir -p "$test_home/.windows" "$test_home/Windows" "$test_home/.config/windows" + chmod 2700 "$test_home/.windows" + chmod 2777 "$test_home/Windows" + chmod 2755 "$test_home/.config/windows" + HOME=$test_home + set -- help + source "$windows_vm_command" >/dev/null + prepare_user_mount_sources || fail "user mount source hardening failed on setgid directories" + [[ $(stat -Lc '%a' "$HOME/.windows") == 700 ]] || fail "storage mode is $(stat -Lc '%a' "$HOME/.windows"), expected 700" + [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "shared mode is $(stat -Lc '%a' "$HOME/Windows"), expected 700" + write_credentials alice secret || fail "write_credentials failed on a setgid config dir" + [[ $(stat -Lc '%a' "$HOME/.config/windows") == 700 ]] || fail "credentials dir mode is $(stat -Lc '%a' "$HOME/.config/windows"), expected 700" +) +pass "user mount sources with leftover setgid harden to exactly 700" From 25be3e9f56ccbf0f083b3559b5ce6f4ed628a5b3 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Wed, 2 Sep 2026 11:28:29 +0200 Subject: [PATCH 02/21] Stop dockur Samba from chmod 2777 on ~/Windows samba.sh treats an empty /shared bind as uninitialized and chmod 2777s it at container start, undoing the host 700 privacy check after every launch. Keep a hidden sentinel in the share and re-harden the directory after docker compose up. --- bin/omarchy-windows-vm | 31 ++++++++++++++++++++++++++++++- test/shell.d/windows-vm-test.sh | 1 + 2 files changed, 31 insertions(+), 1 deletion(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index af7a8e53bbb..50480bbe9b7 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -562,6 +562,24 @@ chmod_private_dir() { chmod a-s,u=rwx,go= -- "$@" } +# dockur samba.sh does `chmod 2777` on an empty /shared bind at container start. +# A hidden sentinel keeps the share non-empty so that path is skipped, and the +# host privacy mode 700 is restored after `docker compose up`. +SHARED_SENTINEL=.omarchy-keep + +ensure_shared_sentinel() { + local shared="$1" sentinel + [[ -d $shared ]] || return 0 + sentinel="$shared/$SHARED_SENTINEL" + [[ -e $sentinel ]] || : >"$sentinel" || return 1 +} + +restore_shared_privacy() { + local shared="${1:-$EXPECTED_SHARED}" + [[ -n $shared && -d $shared ]] || return 0 + chmod_private_dir "$shared" || return 1 +} + prepare_caller_mounts() { local storage_fd storage_id shared_fd shared_id storage_mode shared_mode CALLER_MOUNTS_NEW_STORAGE=0 @@ -601,6 +619,11 @@ prepare_caller_mounts() { echo "omarchy-windows-vm: could not make the VM data directories private: $LEGACY_STORAGE is mode ${storage_mode:-unknown}, $LEGACY_SHARED is mode ${shared_mode:-unknown} (expected 700)" >&2 return 1 fi + ensure_shared_sentinel "/proc/$BASHPID/fd/$shared_fd" || { + exec {storage_fd}<&- + exec {shared_fd}<&- + return 1 + } if bind_mount_leaf "$storage_fd" "$storage_id" "$EXPECTED_STORAGE"; then CALLER_MOUNTS_NEW_STORAGE=$MOUNT_LEAF_NEW @@ -902,7 +925,11 @@ assert_mounts_safe() { } } -__priv_up() { assert_mounts_safe && dc up -d; } +__priv_up() { + assert_mounts_safe || return 1 + dc up -d || return 1 + restore_shared_privacy +} __priv_down() { dc down; } @@ -915,6 +942,7 @@ __priv_up_wait() { if [[ $status != "running" ]]; then dc up -d || return 1 fi + restore_shared_privacy || return 1 # docker logs persists across restarts, so anchor the scan to the current # start time; an empty --since would match a stale "started successfully". @@ -1025,6 +1053,7 @@ prepare_user_mount_sources() { return 1 } chmod_private_dir "$storage" "$shared" || return 1 + ensure_shared_sentinel "$shared" || return 1 } storage_space_path() { diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index 9258fbfd38b..81871bdc041 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -36,6 +36,7 @@ pass "Windows VM launches FreeRDP with its expected title" prepare_user_mount_sources || fail "user mount source hardening failed on setgid directories" [[ $(stat -Lc '%a' "$HOME/.windows") == 700 ]] || fail "storage mode is $(stat -Lc '%a' "$HOME/.windows"), expected 700" [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "shared mode is $(stat -Lc '%a' "$HOME/Windows"), expected 700" + [[ -f $HOME/Windows/.omarchy-keep ]] || fail "shared sentinel was not created" write_credentials alice secret || fail "write_credentials failed on a setgid config dir" [[ $(stat -Lc '%a' "$HOME/.config/windows") == 700 ]] || fail "credentials dir mode is $(stat -Lc '%a' "$HOME/.config/windows"), expected 700" ) From d59c9aafab9c2f5dc57fc86cc52b0dc963e415fe Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Wed, 2 Sep 2026 22:22:52 +0200 Subject: [PATCH 03/21] Drop privileged share sentinel after review Creating ~/.omarchy-keep as root in a caller-owned directory is a symlink-follow write primitive. Restore mode 700 on the pinned directory inodes after the guest reports ready, and never fail a successful start on that chmod. --- bin/omarchy-windows-vm | 40 +++++++++++---------------------- test/shell.d/windows-vm-test.sh | 6 ++++- 2 files changed, 18 insertions(+), 28 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 50480bbe9b7..24d281c0671 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -562,22 +562,16 @@ chmod_private_dir() { chmod a-s,u=rwx,go= -- "$@" } -# dockur samba.sh does `chmod 2777` on an empty /shared bind at container start. -# A hidden sentinel keeps the share non-empty so that path is skipped, and the -# host privacy mode 700 is restored after `docker compose up`. -SHARED_SENTINEL=.omarchy-keep - -ensure_shared_sentinel() { - local shared="$1" sentinel - [[ -d $shared ]] || return 0 - sentinel="$shared/$SHARED_SENTINEL" - [[ -e $sentinel ]] || : >"$sentinel" || return 1 -} - +# dockur samba.sh chmod 2777s an empty /shared at container start. Re-harden +# the already-pinned directory inodes only — never create files in the +# caller-owned share as root (that is a symlink-follow write primitive). +# Best-effort: a failed chmod must not fail a VM that already started. restore_shared_privacy() { - local shared="${1:-$EXPECTED_SHARED}" - [[ -n $shared && -d $shared ]] || return 0 - chmod_private_dir "$shared" || return 1 + local dir + for dir in "$EXPECTED_SHARED" "$LEGACY_SHARED"; do + [[ -n $dir && -d $dir && ! -L $dir ]] || continue + chmod_private_dir "$dir" || true + done } prepare_caller_mounts() { @@ -619,11 +613,6 @@ prepare_caller_mounts() { echo "omarchy-windows-vm: could not make the VM data directories private: $LEGACY_STORAGE is mode ${storage_mode:-unknown}, $LEGACY_SHARED is mode ${shared_mode:-unknown} (expected 700)" >&2 return 1 fi - ensure_shared_sentinel "/proc/$BASHPID/fd/$shared_fd" || { - exec {storage_fd}<&- - exec {shared_fd}<&- - return 1 - } if bind_mount_leaf "$storage_fd" "$storage_id" "$EXPECTED_STORAGE"; then CALLER_MOUNTS_NEW_STORAGE=$MOUNT_LEAF_NEW @@ -925,11 +914,7 @@ assert_mounts_safe() { } } -__priv_up() { - assert_mounts_safe || return 1 - dc up -d || return 1 - restore_shared_privacy -} +__priv_up() { assert_mounts_safe && dc up -d; } __priv_down() { dc down; } @@ -942,7 +927,6 @@ __priv_up_wait() { if [[ $status != "running" ]]; then dc up -d || return 1 fi - restore_shared_privacy || return 1 # docker logs persists across restarts, so anchor the scan to the current # start time; an empty --since would match a stale "started successfully". @@ -950,6 +934,9 @@ __priv_up_wait() { while true; do started_at=$(docker inspect --format='{{.State.StartedAt}}' "$CONTAINER" 2>/dev/null) if [[ -n $started_at ]] && docker logs --since "$started_at" "$CONTAINER" 2>&1 | grep -qi "windows started successfully"; then + # samba.sh has already run by the time the guest reports ready, so this + # chmod lands after the 2777 rather than racing `dc up -d`. + restore_shared_privacy return 0 fi sleep 2 @@ -1053,7 +1040,6 @@ prepare_user_mount_sources() { return 1 } chmod_private_dir "$storage" "$shared" || return 1 - ensure_shared_sentinel "$shared" || return 1 } storage_space_path() { diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index 81871bdc041..aa48ef5a3ab 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -36,8 +36,12 @@ pass "Windows VM launches FreeRDP with its expected title" prepare_user_mount_sources || fail "user mount source hardening failed on setgid directories" [[ $(stat -Lc '%a' "$HOME/.windows") == 700 ]] || fail "storage mode is $(stat -Lc '%a' "$HOME/.windows"), expected 700" [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "shared mode is $(stat -Lc '%a' "$HOME/Windows"), expected 700" - [[ -f $HOME/Windows/.omarchy-keep ]] || fail "shared sentinel was not created" write_credentials alice secret || fail "write_credentials failed on a setgid config dir" [[ $(stat -Lc '%a' "$HOME/.config/windows") == 700 ]] || fail "credentials dir mode is $(stat -Lc '%a' "$HOME/.config/windows"), expected 700" + chmod 2777 "$HOME/Windows" + EXPECTED_SHARED=$HOME/Windows LEGACY_SHARED=$HOME/Windows restore_shared_privacy + [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "restore_shared_privacy left mode $(stat -Lc '%a' "$HOME/Windows")" + mkdir -p "$HOME/missing-parent" + EXPECTED_SHARED=$HOME/missing-parent/nope LEGACY_SHARED="" restore_shared_privacy || fail "restore_shared_privacy failed on a missing path" ) pass "user mount sources with leftover setgid harden to exactly 700" From e4494a5e001864b10c571f4d6633b8a0fbccec70 Mon Sep 17 00:00:00 2001 From: Omabot Date: Thu, 3 Sep 2026 05:22:17 -0700 Subject: [PATCH 04/21] Re-harden the share through the protected anchor only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit restore_shared_privacy also chmodded $LEGACY_SHARED, which is $HOME/Windows: a pathname the unprivileged caller owns. The [[ -d && ! -L ]] test and the chmod are two syscalls, so the caller can swap the directory for a symlink in between and make the root half of __priv_up_wait chmod an arbitrary path to 0700 with the set-ID bits cleared. On a worker a swapper loop won that race on its 260th iteration, taking a root-owned 4755 binary outside the caller's home to root:700. The loop's other element already covers the case. $EXPECTED_SHARED sits in the root-owned 0711 boundary tree the caller cannot write, and assert_mounts_safe has just proved through mounts_ready that it is a bind of the same inode as $LEGACY_SHARED — so chmodding the anchor is what ~/Windows ends up at, measured rather than assumed. Co-Authored-By: Claude Opus 5 (1M context) --- bin/omarchy-windows-vm | 13 ++++++------- test/shell.d/windows-vm-test.sh | 6 ++++++ 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 24d281c0671..28a68152d4a 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -563,15 +563,14 @@ chmod_private_dir() { } # dockur samba.sh chmod 2777s an empty /shared at container start. Re-harden -# the already-pinned directory inodes only — never create files in the -# caller-owned share as root (that is a symlink-follow write primitive). +# only the protected anchor, which sits in the root-owned boundary tree the +# caller cannot write: it is a bind of the same inode as $LEGACY_SHARED, so this +# is what ~/Windows ends up at. Never chmod $LEGACY_SHARED by pathname — the +# caller can swap it for a symlink between the test and the chmod. # Best-effort: a failed chmod must not fail a VM that already started. restore_shared_privacy() { - local dir - for dir in "$EXPECTED_SHARED" "$LEGACY_SHARED"; do - [[ -n $dir && -d $dir && ! -L $dir ]] || continue - chmod_private_dir "$dir" || true - done + [[ -n $EXPECTED_SHARED && -d $EXPECTED_SHARED && ! -L $EXPECTED_SHARED ]] || return 0 + chmod_private_dir "$EXPECTED_SHARED" || true } prepare_caller_mounts() { diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index aa48ef5a3ab..220f0b759e7 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -43,5 +43,11 @@ pass "Windows VM launches FreeRDP with its expected title" [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "restore_shared_privacy left mode $(stat -Lc '%a' "$HOME/Windows")" mkdir -p "$HOME/missing-parent" EXPECTED_SHARED=$HOME/missing-parent/nope LEGACY_SHARED="" restore_shared_privacy || fail "restore_shared_privacy failed on a missing path" + # The home pathname is caller-controlled, so root must never chmod it directly. + mkdir -p "$HOME/legacy-only" + chmod 2777 "$HOME/legacy-only" + EXPECTED_SHARED="" LEGACY_SHARED=$HOME/legacy-only restore_shared_privacy + [[ $(stat -Lc '%a' "$HOME/legacy-only") == 2777 ]] || + fail "restore_shared_privacy chmodded the caller-controlled home pathname" ) pass "user mount sources with leftover setgid harden to exactly 700" From 5cb28aeb6e0134724fb8efdf9c8076bb02556198 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Thu, 3 Sep 2026 17:41:32 +0200 Subject: [PATCH 05/21] Restore share privacy on install, stop, and up_wait timeout install uses priv up, which returned as soon as the container started and never re-hardened ~/Windows after samba.sh chmod 2777. Wait for that 2777 (or the shared-folder log line) before restoring, and also restore after dc down and when the guest-ready wait times out. --- bin/omarchy-windows-vm | 42 +++++++++++++++++++++++++++++++-- test/shell.d/windows-vm-test.sh | 3 +++ 2 files changed, 43 insertions(+), 2 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 28a68152d4a..b5628034e78 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -573,6 +573,33 @@ restore_shared_privacy() { chmod_private_dir "$EXPECTED_SHARED" || true } +# `dc up -d` returns when the container is started, not when samba.sh has +# chmodded /shared. Wait until that has happened (or a short timeout) so the +# restore lands after 2777 rather than before it. Used by `up` (install) which +# cannot wait for the full guest-ready line. +wait_then_restore_shared_privacy() { + local i mode started_at + [[ -n $EXPECTED_SHARED ]] || return 0 + started_at=$(docker inspect --format='{{.State.StartedAt}}' "$CONTAINER" 2>/dev/null) || started_at="" + if [[ -z $started_at ]]; then + restore_shared_privacy + return 0 + fi + for i in {1..60}; do + mode=$(stat -Lc '%a' "$EXPECTED_SHARED" 2>/dev/null) || mode="" + if [[ $mode == 2777 || $mode == 777 ]]; then + restore_shared_privacy + return 0 + fi + if docker logs --since "$started_at" "$CONTAINER" 2>&1 | grep -qiE 'shared folder|samba'; then + restore_shared_privacy + return 0 + fi + sleep 0.25 + done + restore_shared_privacy +} + prepare_caller_mounts() { local storage_fd storage_id shared_fd shared_id storage_mode shared_mode CALLER_MOUNTS_NEW_STORAGE=0 @@ -913,9 +940,19 @@ assert_mounts_safe() { } } -__priv_up() { assert_mounts_safe && dc up -d; } +__priv_up() { + assert_mounts_safe || return 1 + dc up -d || return 1 + wait_then_restore_shared_privacy + return 0 +} -__priv_down() { dc down; } +__priv_down() { + local rc=0 + dc down || rc=$? + resolve_caller && restore_shared_privacy + return "$rc" +} # Bring the VM up and wait until the guest reports it is ready, all under a # single elevation so the readiness poll does not prompt on every iteration. @@ -941,6 +978,7 @@ __priv_up_wait() { sleep 2 ((++count > 60)) && { echo "Timeout: Windows VM did not report ready within 2 minutes" >&2 + restore_shared_privacy return 1 } done diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index 220f0b759e7..bf7816adb57 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -43,6 +43,9 @@ pass "Windows VM launches FreeRDP with its expected title" [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "restore_shared_privacy left mode $(stat -Lc '%a' "$HOME/Windows")" mkdir -p "$HOME/missing-parent" EXPECTED_SHARED=$HOME/missing-parent/nope LEGACY_SHARED="" restore_shared_privacy || fail "restore_shared_privacy failed on a missing path" + chmod 2777 "$HOME/Windows" + CONTAINER=omarchy-windows-does-not-exist EXPECTED_SHARED=$HOME/Windows wait_then_restore_shared_privacy + [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "wait_then_restore_shared_privacy left mode $(stat -Lc '%a' "$HOME/Windows") without a container" # The home pathname is caller-controlled, so root must never chmod it directly. mkdir -p "$HOME/legacy-only" chmod 2777 "$HOME/legacy-only" From fa54de7d93fc4610b001926552180f4019ee8e9a Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Fri, 4 Sep 2026 20:50:32 +0200 Subject: [PATCH 06/21] Close the install-path 2777 window without holding pkexec The 15s wait on priv up expires before dockur finishes the ISO download, so samba.sh still chmod 2777s afterwards. Watch the caller's share in the background after install and restore as the owner. On stop, re-harden every protected per-uid share under the runtime mounts tree instead of resolve_caller (no PKEXEC_UID under direct sudo, and a second user would restore the wrong anchor). --- bin/omarchy-windows-vm | 66 +++++++++++++++++---------------- test/shell.d/windows-vm-test.sh | 13 +++++-- 2 files changed, 45 insertions(+), 34 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index b5628034e78..74608991155 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -573,31 +573,39 @@ restore_shared_privacy() { chmod_private_dir "$EXPECTED_SHARED" || true } -# `dc up -d` returns when the container is started, not when samba.sh has -# chmodded /shared. Wait until that has happened (or a short timeout) so the -# restore lands after 2777 rather than before it. Used by `up` (install) which -# cannot wait for the full guest-ready line. -wait_then_restore_shared_privacy() { - local i mode started_at - [[ -n $EXPECTED_SHARED ]] || return 0 - started_at=$(docker inspect --format='{{.State.StartedAt}}' "$CONTAINER" 2>/dev/null) || started_at="" - if [[ -z $started_at ]]; then - restore_shared_privacy - return 0 - fi - for i in {1..60}; do - mode=$(stat -Lc '%a' "$EXPECTED_SHARED" 2>/dev/null) || mode="" - if [[ $mode == 2777 || $mode == 777 ]]; then - restore_shared_privacy - return 0 - fi - if docker logs --since "$started_at" "$CONTAINER" 2>&1 | grep -qiE 'shared folder|samba'; then - restore_shared_privacy - return 0 - fi - sleep 0.25 +# After `dc down` there is no PKEXEC_UID on a direct `sudo ... stop`, and a +# second user on the box would resolve a different per-uid anchor. Walk the +# root-owned mounts tree instead of calling resolve_caller. +restore_all_shared_privacy() { + local dir canonical prefix + prefix=$(realpath -e -- "$RUNTIME_DIR/mounts/users" 2>/dev/null) || return 0 + for dir in "$prefix"/*/shared; do + [[ -d $dir && ! -L $dir ]] || continue + canonical=$(realpath -e -- "$dir" 2>/dev/null) || continue + [[ $canonical == "$dir" ]] || continue + [[ $canonical == "$prefix/"*"/shared" ]] || continue + chmod_private_dir "$canonical" || true done - restore_shared_privacy +} + +# Unprivileged: samba.sh runs only after dockur's ISO download (10-15 minutes +# on a fresh install). Do not hold a polkit session open for that. Watch the +# caller's own share and chmod it as the owner once it becomes 2777. +schedule_share_privacy_restore() { + local dir="$HOME/Windows" + [[ -e $dir ]] || return 0 + ( + local i mode + for i in {1..1800}; do + mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" + if [[ $mode == 2777 || $mode == 777 ]]; then + chmod_private_dir "$dir" || true + exit 0 + fi + sleep 1 + done + ) >/dev/null 2>&1 & + disown || true } prepare_caller_mounts() { @@ -940,17 +948,12 @@ assert_mounts_safe() { } } -__priv_up() { - assert_mounts_safe || return 1 - dc up -d || return 1 - wait_then_restore_shared_privacy - return 0 -} +__priv_up() { assert_mounts_safe && dc up -d; } __priv_down() { local rc=0 dc down || rc=$? - resolve_caller && restore_shared_privacy + restore_all_shared_privacy return "$rc" } @@ -1395,6 +1398,7 @@ EOF echo " - Port already in use: check if another VM is running" exit 1 fi + schedule_share_privacy_restore echo "" echo "Windows VM is starting up!" diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index bf7816adb57..a658330fa5c 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -43,14 +43,21 @@ pass "Windows VM launches FreeRDP with its expected title" [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "restore_shared_privacy left mode $(stat -Lc '%a' "$HOME/Windows")" mkdir -p "$HOME/missing-parent" EXPECTED_SHARED=$HOME/missing-parent/nope LEGACY_SHARED="" restore_shared_privacy || fail "restore_shared_privacy failed on a missing path" - chmod 2777 "$HOME/Windows" - CONTAINER=omarchy-windows-does-not-exist EXPECTED_SHARED=$HOME/Windows wait_then_restore_shared_privacy - [[ $(stat -Lc '%a' "$HOME/Windows") == 700 ]] || fail "wait_then_restore_shared_privacy left mode $(stat -Lc '%a' "$HOME/Windows") without a container" # The home pathname is caller-controlled, so root must never chmod it directly. mkdir -p "$HOME/legacy-only" chmod 2777 "$HOME/legacy-only" EXPECTED_SHARED="" LEGACY_SHARED=$HOME/legacy-only restore_shared_privacy [[ $(stat -Lc '%a' "$HOME/legacy-only") == 2777 ]] || fail "restore_shared_privacy chmodded the caller-controlled home pathname" + mkdir -p "$test_home/runtime/mounts/users/1000/shared" "$test_home/runtime/mounts/users/1001/shared" + chmod 2777 "$test_home/runtime/mounts/users/1000/shared" "$test_home/runtime/mounts/users/1001/shared" + chmod 2777 "$HOME/Windows" + RUNTIME_DIR=$test_home/runtime restore_all_shared_privacy + [[ $(stat -Lc '%a' "$test_home/runtime/mounts/users/1000/shared") == 700 ]] || + fail "restore_all_shared_privacy left uid 1000 share at $(stat -Lc '%a' "$test_home/runtime/mounts/users/1000/shared")" + [[ $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared") == 700 ]] || + fail "restore_all_shared_privacy left uid 1001 share at $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared")" + [[ $(stat -Lc '%a' "$HOME/Windows") == 2777 ]] || + fail "restore_all_shared_privacy chmodded the caller home share" ) pass "user mount sources with leftover setgid harden to exactly 700" From c2162bd8a51cf4fe7b368a62a71612d83792878a Mon Sep 17 00:00:00 2001 From: Omabot Date: Sat, 5 Sep 2026 05:30:18 -0700 Subject: [PATCH 07/21] Keep the install share watcher alive when its terminal closes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit disown only stops bash from hupping a background job when the shell itself exits. The watcher runs in the install terminal's foreground process group, so the kernel hangs it up when that terminal goes away — and install is launched by omarchy-launch-floating-terminal-with-presentation, which closes as soon as the user dismisses the "Press any key" prompt, minutes before dockur's samba.sh reaches the chmod 2777 the watcher exists to undo. Ignoring SIGHUP in the subshell is what disown was reaching for. An asynchronous command already ignores SIGINT and SIGQUIT when job control is off, so HUP is the only gap. Measured under a pty: without the trap the share is still 2777 four seconds after the terminal exits; with it the watcher restores 700. The new test reproduces that shape with script(1) and fails when the trap is removed. Co-Authored-By: Claude Opus 5 (1M context) --- bin/omarchy-windows-vm | 4 ++++ test/shell.d/windows-vm-test.sh | 25 +++++++++++++++++++++++++ 2 files changed, 29 insertions(+) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 74608991155..048e13bf308 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -595,6 +595,10 @@ schedule_share_privacy_restore() { local dir="$HOME/Windows" [[ -e $dir ]] || return 0 ( + # disown only keeps bash from hupping this on exit. The watcher is in the + # install terminal's foreground process group, so closing that terminal + # still hangs it up from the kernel side, seconds after install returns. + trap '' HUP local i mode for i in {1..1800}; do mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index a658330fa5c..9b1d8edb15f 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -61,3 +61,28 @@ pass "Windows VM launches FreeRDP with its expected title" fail "restore_all_shared_privacy chmodded the caller home share" ) pass "user mount sources with leftover setgid harden to exactly 700" + +# install runs in a floating terminal that closes as soon as it returns, while +# dockur is still ten minutes from the chmod 2777 the watcher exists to undo. +# script(1) reproduces that shape: a pty whose controlling process exits. +( + test_home=$(mktemp -d) + trap 'rm -rf "$test_home"' EXIT + mkdir -p "$test_home/Windows" + chmod 700 "$test_home/Windows" + cat >"$test_home/install.sh" </dev/null +schedule_share_privacy_restore +EOF + script -q -c "bash $test_home/install.sh" /dev/null >/dev/null 2>&1 + chmod 2777 "$test_home/Windows" + for _ in {1..40}; do + [[ $(stat -Lc '%a' "$test_home/Windows") == 700 ]] && break + sleep 0.25 + done + [[ $(stat -Lc '%a' "$test_home/Windows") == 700 ]] || + fail "the install share watcher left the share at $(stat -Lc '%a' "$test_home/Windows")" +) +pass "the install share watcher outlives the terminal install ran in" From 29545b73a528356076ba0c424bc7e49063637bad Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Sat, 5 Sep 2026 18:57:11 +0200 Subject: [PATCH 08/21] Restore share privacy for sudoless-Docker stops and harden the install watcher --- bin/omarchy-windows-vm | 44 ++++++++++++++++--- .../shell.d/windows-vm-mount-boundary-test.sh | 43 ++++++++++++++++++ test/shell.d/windows-vm-test.sh | 15 +++++-- 3 files changed, 92 insertions(+), 10 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 048e13bf308..f7704175fc5 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -573,11 +573,17 @@ restore_shared_privacy() { chmod_private_dir "$EXPECTED_SHARED" || true } -# After `dc down` there is no PKEXEC_UID on a direct `sudo ... stop`, and a -# second user on the box would resolve a different per-uid anchor. Walk the -# root-owned mounts tree instead of calling resolve_caller. +# Root-only: the mounts tree is 0711, so an unprivileged caller cannot list it +# and the glob below would silently match nothing. After `dc down` there is no +# PKEXEC_UID on a direct `sudo ... stop`, and a second user on the box would +# resolve a different per-uid anchor, so root walks the tree instead. A +# sudoless caller restores its own anchor through restore_shared_privacy. restore_all_shared_privacy() { local dir canonical prefix + ((EUID == 0)) || return 0 + # Same refusal prepare_runtime_tree applies: a non-standard privileged + # runtime is supported only for unprivileged tests/development. + [[ $RUNTIME_DIR == /var/lib/omarchy/windows ]] || return 0 prefix=$(realpath -e -- "$RUNTIME_DIR/mounts/users" 2>/dev/null) || return 0 for dir in "$prefix"/*/shared; do [[ -d $dir && ! -L $dir ]] || continue @@ -600,14 +606,25 @@ schedule_share_privacy_restore() { # still hangs it up from the kernel side, seconds after install returns. trap '' HUP local i mode - for i in {1..1800}; do + # 2x the documented 10-15 minute download is a thin margin on a slow link, + # so budget an hour; on expiry the final restore below still closes the + # 2777 window for anything that happened before it. + for i in {1..3600}; do mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" if [[ $mode == 2777 || $mode == 777 ]]; then - chmod_private_dir "$dir" || true - exit 0 + chmod_private_dir "$dir" 2>/dev/null || true + # Do not exit on a failed chmod: a transient failure (a swapped path, + # an unwritable moment) must not end the watcher permanently while the + # share is still exposed. Leave only once the mode is really 700. + [[ $(stat -Lc '%a' "$dir" 2>/dev/null) == 700 ]] && exit 0 fi sleep 1 done + chmod_private_dir "$dir" 2>/dev/null || true + # The subshell output goes to /dev/null, so the only visible trace of an + # expiry is this journal entry. + logger -t omarchy-windows-vm \ + "share privacy watcher timed out; applied a final restore to $dir" 2>/dev/null || true ) >/dev/null 2>&1 & disown || true } @@ -957,7 +974,20 @@ __priv_up() { assert_mounts_safe && dc up -d; } __priv_down() { local rc=0 dc down || rc=$? - restore_all_shared_privacy + if ((EUID == 0)); then + restore_all_shared_privacy + else + # A sudoless-Docker stop runs unelevated, where the mounts tree is not + # listable. restore_shared_privacy still works here: the anchor sits under + # the root-owned boundary tree the caller cannot rename, and while the + # bind exists it is the caller's own inode, so the owner chmod succeeds. + # If the bind is gone (fresh reboot) the chmod fails and the next launch + # re-hardens through prepare_user_mount_sources instead. Best-effort: the + # container is already down, so a failed restore must not fail the stop. + if resolve_caller; then + restore_shared_privacy + fi + fi return "$rc" } diff --git a/test/shell.d/windows-vm-mount-boundary-test.sh b/test/shell.d/windows-vm-mount-boundary-test.sh index 0978b89ea6f..badb1662023 100644 --- a/test/shell.d/windows-vm-mount-boundary-test.sh +++ b/test/shell.d/windows-vm-mount-boundary-test.sh @@ -135,6 +135,49 @@ with_vm_lock prepare_caller_mounts || fail "root could not harden setgid VM sour $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == 1000:700 ]] || fail "setgid hardening did not leave caller-owned 0700 anchors" pass "setgid VM source directories harden to exactly 700" +# Stop-time restore must cover every caller shape. Root walks the mounts tree +# for a direct `sudo ... stop`; a sudoless-Docker stop is unelevated, where the +# 0711 tree is not listable, and must restore the caller's own anchor +# owner-side instead. A second account's anchor is never another user's to +# chmod, and nothing runs through a non-standard privileged runtime path. +mkdir -p "$USERS_DIR/1001/shared" "$test_tmp/mounts/users/1000/shared" +chmod 2777 "$USERS_DIR/1001/shared" "$EXPECTED_SHARED" "$test_tmp/mounts/users/1000/shared" +RUNTIME_DIR=$test_tmp restore_all_shared_privacy +[[ $(command stat -Lc '%a' "$EXPECTED_SHARED") == 2777 && + $(command stat -Lc '%a' "$USERS_DIR/1001/shared") == 2777 && + $(command stat -Lc '%a' "$test_tmp/mounts/users/1000/shared") == 2777 ]] || + fail "root restored anchors through a non-standard runtime path" +install -d -m 0755 /var/vm-test-bin +install -m 0644 "$test_tmp/omarchy-windows-vm" /var/vm-test-bin/omarchy-windows-vm +child_rc=0 +setpriv --reuid=1000 --regid=1000 --clear-groups bash -c ' + # The namespace has no passwd entry for uid 1000, so mirror the stub above. + getent() { + if [[ $1 == passwd && $2 == 1000 ]]; then + printf "alice:x:1000:1000::/home/alice:/bin/bash\n" + return 0 + fi + return 2 + } + set -- help + source /var/vm-test-bin/omarchy-windows-vm >/dev/null 2>&1 + resolve_caller || exit 10 + restore_all_shared_privacy || exit 11 + [[ $(command stat -Lc "%a" "$USERS_DIR/1000/shared") == 2777 ]] || exit 12 + [[ $(command stat -Lc "%a" "$USERS_DIR/1001/shared") == 2777 ]] || exit 13 + restore_shared_privacy || exit 14 + [[ $(command stat -Lc "%a" "$USERS_DIR/1000/shared") == 700 ]] || exit 15 + [[ $(command stat -Lc "%a" "$USERS_DIR/1001/shared") == 2777 ]] || exit 16 +' || child_rc=$? +[[ $child_rc == 0 ]] || fail "sudoless stop restore misbehaved (rc=$child_rc)" +chmod 2777 "$EXPECTED_SHARED" +restore_all_shared_privacy +[[ $(command stat -Lc '%a' "$EXPECTED_SHARED") == 700 && + $(command stat -Lc '%a' "$USERS_DIR/1001/shared") == 700 ]] || + fail "root walk did not restore every anchor to 700" +rm -rf /var/vm-test-bin +pass "stop-time restore walks the tree as root and restores only the caller's anchor unprivileged" + # Existing production boundary components are never repaired in place when # their ownership or write permissions are unsafe. Both the preparation path # and the final pre-Docker guard must fail closed without disturbing the binds. diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index 9b1d8edb15f..c29e84e8860 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -52,11 +52,20 @@ pass "Windows VM launches FreeRDP with its expected title" mkdir -p "$test_home/runtime/mounts/users/1000/shared" "$test_home/runtime/mounts/users/1001/shared" chmod 2777 "$test_home/runtime/mounts/users/1000/shared" "$test_home/runtime/mounts/users/1001/shared" chmod 2777 "$HOME/Windows" + # The mounts tree is root-owned and not listable unprivileged, so the walk is + # root-only: here it must do nothing at all, not look like a restore. RUNTIME_DIR=$test_home/runtime restore_all_shared_privacy + [[ $(stat -Lc '%a' "$test_home/runtime/mounts/users/1000/shared") == 2777 ]] || + fail "unprivileged restore_all_shared_privacy touched uid 1000 share: $(stat -Lc '%a' "$test_home/runtime/mounts/users/1000/shared")" + [[ $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared") == 2777 ]] || + fail "unprivileged restore_all_shared_privacy touched uid 1001 share: $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared")" + # A sudoless-Docker stop instead restores the caller's own anchor owner-side, + # and never another user's. + EXPECTED_SHARED=$test_home/runtime/mounts/users/1000/shared restore_shared_privacy [[ $(stat -Lc '%a' "$test_home/runtime/mounts/users/1000/shared") == 700 ]] || - fail "restore_all_shared_privacy left uid 1000 share at $(stat -Lc '%a' "$test_home/runtime/mounts/users/1000/shared")" - [[ $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared") == 700 ]] || - fail "restore_all_shared_privacy left uid 1001 share at $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared")" + fail "owner-side restore left uid 1000 share at $(stat -Lc '%a' "$test_home/runtime/mounts/users/1000/shared")" + [[ $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared") == 2777 ]] || + fail "owner-side restore touched uid 1001 share: $(stat -Lc '%a' "$test_home/runtime/mounts/users/1001/shared")" [[ $(stat -Lc '%a' "$HOME/Windows") == 2777 ]] || fail "restore_all_shared_privacy chmodded the caller home share" ) From 0d2473445514a42429a6be0f684557ff61569c25 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Sun, 6 Sep 2026 20:04:55 +0200 Subject: [PATCH 09/21] Refuse to elevate a mismatched packaged copy --- bin/omarchy-windows-vm | 22 ++++++++++++++++++++++ test/shell.d/windows-vm-test.sh | 27 +++++++++++++++++++++++++++ 2 files changed, 49 insertions(+) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index f7704175fc5..a429b9c96d9 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -77,6 +77,24 @@ priv_target() { printf '%s\n' "$candidate" } +# pkexec can only run the packaged copy at its fixed root-owned path, which a +# dev-link checkout never shadows: the unprivileged half then runs the checkout +# while the elevated half runs the package. All privileged mount work happens in +# the elevated half, so a stale packaged copy would re-apply whatever chmod +# semantics it shipped with and fail closed with no diagnostic (observed as a +# launch refusing a 2777 share and leaving it at 2700). Refuse unless both +# halves are the same build. +privileged_copy_matches() { + local target="$1" target_hash self_hash + [[ -r ${BASH_SOURCE[0]} && -r $target ]] || return 1 + # The common case runs both halves from the same file; only a dev checkout + # needs the content comparison. + [[ ${BASH_SOURCE[0]} -ef $target ]] && return 0 + target_hash=$(sha256sum -- "$target") || return 1 + self_hash=$(sha256sum -- "${BASH_SOURCE[0]}") || return 1 + [[ ${target_hash%% *} == "${self_hash%% *}" ]] +} + # Run a privileged VM action. write_compose always elevates (the compose is # root-owned); the daemon operations run directly when sudoless Docker is on and # otherwise behind a polkit prompt. The stock org.freedesktop.policykit.exec @@ -104,6 +122,10 @@ priv() { echo "omarchy-windows-vm: refusing to run a non-root-owned command as root" >&2 return 1 } + privileged_copy_matches "$target" || { + echo "omarchy-windows-vm: refusing to elevate: $target is not this command; refresh the installed omarchy package so root runs the same build" >&2 + return 1 + } pkexec "$target" __priv "$action" "$@" } diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index c29e84e8860..c7b9befd33d 100644 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -95,3 +95,30 @@ EOF fail "the install share watcher left the share at $(stat -Lc '%a' "$test_home/Windows")" ) pass "the install share watcher outlives the terminal install ran in" + +# pkexec runs the packaged copy, which a dev link cannot shadow. A stale +# packaged copy used to re-apply the pre-fix chmod semantics with no diagnostic, +# failing a launch and leaving the share at 2700. The skew check must refuse +# before pkexec and say why, and must accept an identical copy. +( + test_home=$(mktemp -d) + trap 'rm -rf "$test_home"' EXIT + set -- help + source "$windows_vm_command" >/dev/null + cp "$windows_vm_command" "$test_home/copy" + privileged_copy_matches "$test_home/copy" || fail "the skew check rejected an identical packaged copy" + printf drift >>"$test_home/copy" + privileged_copy_matches "$test_home/copy" && fail "the skew check accepted a drifted packaged copy" + docker_needs_sudo() { return 0; } + printf '#!/bin/bash\n' >"$test_home/packaged" + chmod 755 "$test_home/packaged" + priv_target() { printf '%s\n' "$test_home/packaged"; } + pkexec() { : >"$test_home/elevated"; } + privileged_copy_matches() { return 1; } + priv status >/dev/null 2>&1 && fail "priv elevated despite a mismatched privileged copy" + [[ ! -e $test_home/elevated ]] || fail "priv reached pkexec with a mismatched privileged copy" + privileged_copy_matches() { return 0; } + priv status >/dev/null 2>&1 || fail "priv refused a matching privileged copy" + [[ -e $test_home/elevated ]] || fail "priv did not reach pkexec with a matching privileged copy" +) +pass "elevation refuses a mismatched privileged copy and accepts an identical one" From 3b54377f21838001dbf1d2abfbe6390e7041e035 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Sun, 6 Sep 2026 20:04:55 +0200 Subject: [PATCH 10/21] Pick boundary-test fixture uids that cannot collide with host VM anchors --- .../shell.d/windows-vm-mount-boundary-test.sh | 84 +++++++++++-------- 1 file changed, 50 insertions(+), 34 deletions(-) diff --git a/test/shell.d/windows-vm-mount-boundary-test.sh b/test/shell.d/windows-vm-mount-boundary-test.sh index badb1662023..6cf736fd20f 100644 --- a/test/shell.d/windows-vm-mount-boundary-test.sh +++ b/test/shell.d/windows-vm-mount-boundary-test.sh @@ -19,6 +19,22 @@ trap 'rm -rf "$test_tmp"' EXIT # mount-safe copy of the helper before the mounts land. cp "$ROOT/bin/omarchy-windows-vm" "$test_tmp/omarchy-windows-vm" +# A host VM leaves its anchors mounted at exactly the production anchor paths +# this test re-creates for its fixture uid. The tmpfs below hides them from the +# filesystem, but the user-ns copy of the mount table keeps them listed in +# /proc/self/mountinfo — and they are MNT_LOCKED, so they cannot be detached — +# while mountpoint(1) matches entries by path: the fresh anchors then look like +# existing mounts and prepare_mount_anchor short-circuits without creating +# them. Pick fixture uids whose anchor paths cannot collide with whatever the +# host VM left behind. +TEST_UID=4242 +TEST_UID_OTHER=$((TEST_UID + 1)) +while grep -qE "mounts/users/(${TEST_UID}|${TEST_UID_OTHER})/(storage|shared) " /proc/self/mountinfo; do + TEST_UID=$((TEST_UID + 2)) + TEST_UID_OTHER=$((TEST_UID + 1)) +done +export TEST_UID TEST_UID_OTHER + # Hide host state before creating the production paths used by the root helper. mount -t tmpfs -o mode=0755,size=8m run-test /run mkdir -p /run/lock @@ -42,8 +58,8 @@ stat() { TEST_PASSWD_HOME=/home/alice getent() { - if [[ $1 == passwd && ${2:-} == 1000 ]]; then - printf 'alice:x:1000:1000::%s:/bin/bash\n' "$TEST_PASSWD_HOME" + if [[ $1 == passwd && ${2:-} == ${TEST_UID} ]]; then + printf "alice:x:${TEST_UID}:${TEST_UID}::%s:/bin/bash\n" "$TEST_PASSWD_HOME" return 0 fi return 2 @@ -63,14 +79,14 @@ assert_no_runtime_mutation "zero PKEXEC_UID" PKEXEC_UID=not-a-number resolve_caller 2>/dev/null && fail "root accepted nonnumeric PKEXEC_UID" assert_no_runtime_mutation "nonnumeric PKEXEC_UID" -PKEXEC_UID=1001 +PKEXEC_UID=${TEST_UID_OTHER} resolve_caller 2>/dev/null && fail "root accepted uid absent from passwd" assert_no_runtime_mutation "missing passwd entry" -PKEXEC_UID=1000 +PKEXEC_UID=${TEST_UID} resolve_caller 2>/dev/null && fail "root accepted a home not owned by caller" assert_no_runtime_mutation "wrong-owned home" -chown 1000:1000 /home/alice +chown ${TEST_UID}:${TEST_UID} /home/alice chmod 0777 /home resolve_caller 2>/dev/null && fail "root accepted writable home parent" @@ -78,7 +94,7 @@ assert_no_runtime_mutation "writable parent" chmod 0755 /home mkdir /home/real-alice -chown 1000:1000 /home/real-alice +chown ${TEST_UID}:${TEST_UID} /home/real-alice ln -s /home/real-alice /home/link-alice TEST_PASSWD_HOME=/home/link-alice resolve_caller 2>/dev/null && fail "root accepted symlinked passwd home" @@ -90,14 +106,14 @@ pass "root dispatch rejects missing/invalid uid, passwd, owner, symlink, and wri # Put each familiar source on its own filesystem. Both start with legacy 0755 # permissions and world-readable payloads to prove migration hardens the leaves. mkdir /home/storage-target /home/shared-target -mount -t tmpfs -o uid=1000,gid=1000,mode=0755,size=3g storage-test /home/storage-target -mount -t tmpfs -o uid=1000,gid=1000,mode=0755,size=64m shared-test /home/shared-target +mount -t tmpfs -o uid=${TEST_UID},gid=${TEST_UID},mode=0755,size=3g storage-test /home/storage-target +mount -t tmpfs -o uid=${TEST_UID},gid=${TEST_UID},mode=0755,size=64m shared-test /home/shared-target ln -s /home/storage-target /home/alice/.windows ln -s /home/shared-target /home/alice/Windows -chown -h 1000:1000 /home/alice/.windows /home/alice/Windows +chown -h ${TEST_UID}:${TEST_UID} /home/alice/.windows /home/alice/Windows printf disk >/home/storage-target/disk.img printf shared >/home/shared-target/shared.txt -chown 1000:1000 /home/storage-target/disk.img /home/shared-target/shared.txt +chown ${TEST_UID}:${TEST_UID} /home/storage-target/disk.img /home/shared-target/shared.txt chmod 0644 /home/storage-target/disk.img /home/shared-target/shared.txt home_dev=$(command stat -Lc '%d' /home/alice) @@ -113,12 +129,12 @@ resolve_caller [[ $(command stat -Lc '%d' "$CALLER_DATA_ROOT") != "$storage_dev" ]] || fail "Docker boundary unexpectedly shares the storage filesystem" [[ $(command stat -Lc '%u:%a' "$MOUNT_ROOT") == 0:711 && $(command stat -Lc '%u:%a' "$CALLER_DATA_ROOT") == 0:711 ]] || fail "production ancestors are not root-owned/private-boundary modes" -[[ $(command stat -Lc '%u:%a' "$EXPECTED_STORAGE") == 1000:700 && - $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == 1000:700 ]] || fail "migrated leaves are not caller-owned 0700" -if setpriv --reuid=1001 --regid=1001 --clear-groups cat "$EXPECTED_STORAGE/disk.img" >/dev/null 2>&1; then +[[ $(command stat -Lc '%u:%a' "$EXPECTED_STORAGE") == ${TEST_UID}:700 && + $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == ${TEST_UID}:700 ]] || fail "migrated leaves are not caller-owned 0700" +if setpriv --reuid=${TEST_UID_OTHER} --regid=${TEST_UID_OTHER} --clear-groups cat "$EXPECTED_STORAGE/disk.img" >/dev/null 2>&1; then fail "another local account read the VM disk through its anchor" fi -if setpriv --reuid=1001 --regid=1001 --clear-groups cat "$EXPECTED_SHARED/shared.txt" >/dev/null 2>&1; then +if setpriv --reuid=${TEST_UID_OTHER} --regid=${TEST_UID_OTHER} --clear-groups cat "$EXPECTED_SHARED/shared.txt" >/dev/null 2>&1; then fail "another local account read shared files through their anchor" fi pass "cross-filesystem symlink sources bind by identity and migrated 0700 leaves deny another account" @@ -127,12 +143,12 @@ pass "cross-filesystem symlink sources bind by identity and migrated 0700 leaves # exact-700 check and block every privileged action (omacom/omarchy#9698). chmod 2700 /home/storage-target chmod 2777 /home/shared-target -chown 1000:1000 /home/storage-target /home/shared-target +chown ${TEST_UID}:${TEST_UID} /home/storage-target /home/shared-target with_vm_lock prepare_caller_mounts || fail "root could not harden setgid VM source directories" [[ $(command stat -Lc '%a' /home/storage-target) == 700 ]] || fail "storage still had special bits after hardening" [[ $(command stat -Lc '%a' /home/shared-target) == 700 ]] || fail "shared still had special bits after hardening" -[[ $(command stat -Lc '%u:%a' "$EXPECTED_STORAGE") == 1000:700 && - $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == 1000:700 ]] || fail "setgid hardening did not leave caller-owned 0700 anchors" +[[ $(command stat -Lc '%u:%a' "$EXPECTED_STORAGE") == ${TEST_UID}:700 && + $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == ${TEST_UID}:700 ]] || fail "setgid hardening did not leave caller-owned 0700 anchors" pass "setgid VM source directories harden to exactly 700" # Stop-time restore must cover every caller shape. Root walks the mounts tree @@ -140,21 +156,21 @@ pass "setgid VM source directories harden to exactly 700" # 0711 tree is not listable, and must restore the caller's own anchor # owner-side instead. A second account's anchor is never another user's to # chmod, and nothing runs through a non-standard privileged runtime path. -mkdir -p "$USERS_DIR/1001/shared" "$test_tmp/mounts/users/1000/shared" -chmod 2777 "$USERS_DIR/1001/shared" "$EXPECTED_SHARED" "$test_tmp/mounts/users/1000/shared" +mkdir -p "$USERS_DIR/${TEST_UID_OTHER}/shared" "$test_tmp/mounts/users/${TEST_UID}/shared" +chmod 2777 "$USERS_DIR/${TEST_UID_OTHER}/shared" "$EXPECTED_SHARED" "$test_tmp/mounts/users/${TEST_UID}/shared" RUNTIME_DIR=$test_tmp restore_all_shared_privacy [[ $(command stat -Lc '%a' "$EXPECTED_SHARED") == 2777 && - $(command stat -Lc '%a' "$USERS_DIR/1001/shared") == 2777 && - $(command stat -Lc '%a' "$test_tmp/mounts/users/1000/shared") == 2777 ]] || + $(command stat -Lc '%a' "$USERS_DIR/${TEST_UID_OTHER}/shared") == 2777 && + $(command stat -Lc '%a' "$test_tmp/mounts/users/${TEST_UID}/shared") == 2777 ]] || fail "root restored anchors through a non-standard runtime path" install -d -m 0755 /var/vm-test-bin install -m 0644 "$test_tmp/omarchy-windows-vm" /var/vm-test-bin/omarchy-windows-vm child_rc=0 -setpriv --reuid=1000 --regid=1000 --clear-groups bash -c ' - # The namespace has no passwd entry for uid 1000, so mirror the stub above. +setpriv --reuid=${TEST_UID} --regid=${TEST_UID} --clear-groups bash -c ' + # The namespace has no passwd entry for uid ${TEST_UID}, so mirror the stub above. getent() { - if [[ $1 == passwd && $2 == 1000 ]]; then - printf "alice:x:1000:1000::/home/alice:/bin/bash\n" + if [[ $1 == passwd && $2 == ${TEST_UID} ]]; then + printf "alice:x:${TEST_UID}:${TEST_UID}::/home/alice:/bin/bash\n" return 0 fi return 2 @@ -163,17 +179,17 @@ setpriv --reuid=1000 --regid=1000 --clear-groups bash -c ' source /var/vm-test-bin/omarchy-windows-vm >/dev/null 2>&1 resolve_caller || exit 10 restore_all_shared_privacy || exit 11 - [[ $(command stat -Lc "%a" "$USERS_DIR/1000/shared") == 2777 ]] || exit 12 - [[ $(command stat -Lc "%a" "$USERS_DIR/1001/shared") == 2777 ]] || exit 13 + [[ $(command stat -Lc "%a" "$USERS_DIR/${TEST_UID}/shared") == 2777 ]] || exit 12 + [[ $(command stat -Lc "%a" "$USERS_DIR/${TEST_UID_OTHER}/shared") == 2777 ]] || exit 13 restore_shared_privacy || exit 14 - [[ $(command stat -Lc "%a" "$USERS_DIR/1000/shared") == 700 ]] || exit 15 - [[ $(command stat -Lc "%a" "$USERS_DIR/1001/shared") == 2777 ]] || exit 16 + [[ $(command stat -Lc "%a" "$USERS_DIR/${TEST_UID}/shared") == 700 ]] || exit 15 + [[ $(command stat -Lc "%a" "$USERS_DIR/${TEST_UID_OTHER}/shared") == 2777 ]] || exit 16 ' || child_rc=$? [[ $child_rc == 0 ]] || fail "sudoless stop restore misbehaved (rc=$child_rc)" chmod 2777 "$EXPECTED_SHARED" restore_all_shared_privacy [[ $(command stat -Lc '%a' "$EXPECTED_SHARED") == 700 && - $(command stat -Lc '%a' "$USERS_DIR/1001/shared") == 700 ]] || + $(command stat -Lc '%a' "$USERS_DIR/${TEST_UID_OTHER}/shared") == 700 ]] || fail "root walk did not restore every anchor to 700" rm -rf /var/vm-test-bin pass "stop-time restore walks the tree as root and restores only the caller's anchor unprivileged" @@ -187,10 +203,10 @@ mounts_ready 2>/dev/null && fail "final guard accepted a group-writable mount bo [[ $(command stat -Lc '%a' "$MOUNT_ROOT") == 731 ]] || fail "rejection unexpectedly changed the writable boundary" chmod 0711 "$MOUNT_ROOT" -chown 1000:1000 "$USERS_DIR" +chown ${TEST_UID}:${TEST_UID} "$USERS_DIR" with_vm_lock prepare_caller_mounts 2>/dev/null && fail "root repaired a caller-owned mount boundary instead of rejecting it" mounts_ready 2>/dev/null && fail "final guard accepted a caller-owned mount boundary" -[[ $(command stat -Lc '%u' "$USERS_DIR") == 1000 ]] || fail "rejection unexpectedly changed the boundary owner" +[[ $(command stat -Lc '%u' "$USERS_DIR") == ${TEST_UID} ]] || fail "rejection unexpectedly changed the boundary owner" chown root:root "$USERS_DIR" [[ $(mount_layer_count "$EXPECTED_STORAGE") == 1 && @@ -264,7 +280,7 @@ umount "$EXPECTED_SHARED" umount "$EXPECTED_STORAGE" rm /home/alice/Windows ln -s / /home/alice/Windows -chown -h 1000:1000 /home/alice/Windows +chown -h ${TEST_UID}:${TEST_UID} /home/alice/Windows with_vm_lock prepare_caller_mounts 2>/dev/null && fail "root accepted a non-caller-owned second source" [[ $(mount_layer_count "$EXPECTED_STORAGE") == 0 && $(mount_layer_count "$EXPECTED_SHARED") == 0 ]] || fail "failed second-source preflight left a partial bind" [[ $(readlink /home/alice/Windows) == / ]] || fail "failed preflight consumed or quarantined symlink" @@ -274,7 +290,7 @@ pass "root preflights both sources before mounting either and preserves rejectio # root-planted anchor symlink to the expected mounted source is rejected. rm /home/alice/Windows ln -s /home/shared-target /home/alice/Windows -chown -h 1000:1000 /home/alice/Windows +chown -h ${TEST_UID}:${TEST_UID} /home/alice/Windows rmdir "$EXPECTED_STORAGE" ln -s /home/storage-target "$EXPECTED_STORAGE" storage_id=$(command stat -Lc '%d:%i' /home/storage-target) From 09e4c157df45a4bebffec7978cc4a7e552d7bef1 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Tue, 8 Sep 2026 09:41:22 +0200 Subject: [PATCH 11/21] Run the install share watcher outside the terminal scope --- bin/omarchy-windows-vm | 73 +++++++++++++++++++++------------ test/shell.d/windows-vm-test.sh | 21 ++++++++-- 2 files changed, 65 insertions(+), 29 deletions(-) mode change 100644 => 100755 test/shell.d/windows-vm-test.sh diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index a429b9c96d9..af1974eb22a 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -619,34 +619,55 @@ restore_all_shared_privacy() { # Unprivileged: samba.sh runs only after dockur's ISO download (10-15 minutes # on a fresh install). Do not hold a polkit session open for that. Watch the # caller's own share and chmod it as the owner once it becomes 2777. +watch_share_privacy() { + local dir="$1" i mode result + [[ -e $dir ]] || return 0 + # 2x the documented 10-15 minute download is a thin margin on a slow link, + # so budget an hour; on expiry restore only if the share is still exposed. + for i in {1..3600}; do + mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" + if [[ $mode == 2777 || $mode == 777 ]]; then + chmod_private_dir "$dir" 2>/dev/null || true + # Do not exit on a failed chmod: a transient failure (a swapped path, + # an unwritable moment) must not end the watcher permanently while the + # share is still exposed. Leave only once the mode is really 700. + [[ $(stat -Lc '%a' "$dir" 2>/dev/null) == 700 ]] && return 0 + fi + sleep 1 + done + mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" + if [[ $mode == 2777 || $mode == 777 ]]; then + chmod_private_dir "$dir" 2>/dev/null || true + fi + result=$(stat -Lc '%a' "$dir" 2>/dev/null) || result="missing" + logger -t omarchy-windows-vm \ + "share privacy watcher timed out; $dir mode is $result" 2>/dev/null || true +} + schedule_share_privacy_restore() { - local dir="$HOME/Windows" + local dir="$HOME/Windows" self [[ -e $dir ]] || return 0 + self=$(readlink -f -- "${BASH_SOURCE[0]}") || self="${BASH_SOURCE[0]}" + # install runs inside omarchy-launch-floating-terminal-with-presentation, + # which is a uwsm-app systemd scope. Dismissing that window SIGTERMs leftover + # cgroup members; trap '' HUP does not cover that. A user unit is outside + # the scope, so the wait survives the terminal. + if command -v systemd-run >/dev/null; then + systemctl --user reset-failed omarchy-windows-share-privacy.service 2>/dev/null || true + systemctl --user stop omarchy-windows-share-privacy.service 2>/dev/null || true + if systemd-run --user --quiet --collect \ + --unit=omarchy-windows-share-privacy \ + --description="Restore ~/Windows mode after dockur samba.sh" \ + /bin/bash -c 'set -- help; source "$1" >/dev/null; watch_share_privacy "$2"' \ + bash "$self" "$dir"; then + return 0 + fi + fi + # No user bus (tests, a stripped session): ignore HUP/TERM so a closing + # terminal cannot kill the wait the way the scope would. ( - # disown only keeps bash from hupping this on exit. The watcher is in the - # install terminal's foreground process group, so closing that terminal - # still hangs it up from the kernel side, seconds after install returns. - trap '' HUP - local i mode - # 2x the documented 10-15 minute download is a thin margin on a slow link, - # so budget an hour; on expiry the final restore below still closes the - # 2777 window for anything that happened before it. - for i in {1..3600}; do - mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" - if [[ $mode == 2777 || $mode == 777 ]]; then - chmod_private_dir "$dir" 2>/dev/null || true - # Do not exit on a failed chmod: a transient failure (a swapped path, - # an unwritable moment) must not end the watcher permanently while the - # share is still exposed. Leave only once the mode is really 700. - [[ $(stat -Lc '%a' "$dir" 2>/dev/null) == 700 ]] && exit 0 - fi - sleep 1 - done - chmod_private_dir "$dir" 2>/dev/null || true - # The subshell output goes to /dev/null, so the only visible trace of an - # expiry is this journal entry. - logger -t omarchy-windows-vm \ - "share privacy watcher timed out; applied a final restore to $dir" 2>/dev/null || true + trap '' HUP TERM + watch_share_privacy "$dir" ) >/dev/null 2>&1 & disown || true } @@ -1004,7 +1025,7 @@ __priv_down() { # the root-owned boundary tree the caller cannot rename, and while the # bind exists it is the caller's own inode, so the owner chmod succeeds. # If the bind is gone (fresh reboot) the chmod fails and the next launch - # re-hardens through prepare_user_mount_sources instead. Best-effort: the + # re-hardens through prepare_caller_mounts instead. Best-effort: the # container is already down, so a failed restore must not fail the stop. if resolve_caller; then restore_shared_privacy diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh old mode 100644 new mode 100755 index c7b9befd33d..169b138273a --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -73,19 +73,34 @@ pass "user mount sources with leftover setgid harden to exactly 700" # install runs in a floating terminal that closes as soon as it returns, while # dockur is still ten minutes from the chmod 2777 the watcher exists to undo. -# script(1) reproduces that shape: a pty whose controlling process exits. +# Production leaves that wait in a user unit so dismissing the uwsm-app scope +# cannot SIGTERM it. script(1) plus a systemd-run stub that still dies with the +# pty proves the fallback also outlives the terminal. ( test_home=$(mktemp -d) trap 'rm -rf "$test_home"' EXIT - mkdir -p "$test_home/Windows" + mkdir -p "$test_home/Windows" "$test_home/bin" chmod 700 "$test_home/Windows" + cat >"$test_home/bin/systemd-run" <<'EOF' +#!/bin/bash +printf 'systemd-run' >>"$TEST_LOG" +printf '\t%s' "$@" >>"$TEST_LOG" +printf '\n' >>"$TEST_LOG" +exit 1 +EOF + chmod +x "$test_home/bin/systemd-run" cat >"$test_home/install.sh" <"\$TEST_LOG" set -- help source "$windows_vm_command" >/dev/null schedule_share_privacy_restore EOF script -q -c "bash $test_home/install.sh" /dev/null >/dev/null 2>&1 + grep -q '^systemd-run' "$test_home/systemd-run.log" || + fail "the install share watcher did not try to leave the terminal scope" chmod 2777 "$test_home/Windows" for _ in {1..40}; do [[ $(stat -Lc '%a' "$test_home/Windows") == 700 ]] && break From 2f83392cb547aa4ffd05fe3c71af889769826724 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Wed, 16 Sep 2026 20:44:38 +0200 Subject: [PATCH 12/21] fix(windows-vm): use the command helper for the systemd-run probe --- bin/omarchy-windows-vm | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index af1974eb22a..1f75d9d76ae 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -652,7 +652,7 @@ schedule_share_privacy_restore() { # which is a uwsm-app systemd scope. Dismissing that window SIGTERMs leftover # cgroup members; trap '' HUP does not cover that. A user unit is outside # the scope, so the wait survives the terminal. - if command -v systemd-run >/dev/null; then + if omarchy-cmd-present systemd-run; then systemctl --user reset-failed omarchy-windows-share-privacy.service 2>/dev/null || true systemctl --user stop omarchy-windows-share-privacy.service 2>/dev/null || true if systemd-run --user --quiet --collect \ From 7f2274cef69d2ea884f91b988fd367812346e98b Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Sun, 4 Oct 2026 21:03:48 +0200 Subject: [PATCH 13/21] Use TEST_UID in merged upstream setgid assertions --- test/shell.d/windows-vm-mount-boundary-test.sh | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/test/shell.d/windows-vm-mount-boundary-test.sh b/test/shell.d/windows-vm-mount-boundary-test.sh index 6cf736fd20f..642ca675a5e 100644 --- a/test/shell.d/windows-vm-mount-boundary-test.sh +++ b/test/shell.d/windows-vm-mount-boundary-test.sh @@ -226,8 +226,8 @@ umount "$EXPECTED_STORAGE" chmod 2777 /home/shared-target chmod 6755 /home/storage-target with_vm_lock prepare_caller_mounts || fail "root could not rebind setgid sources" -[[ $(command stat -Lc '%u:%a' "$EXPECTED_STORAGE") == 1000:700 && - $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == 1000:700 ]] || fail "setgid sources were not hardened to 0700" +[[ $(command stat -Lc '%u:%a' "$EXPECTED_STORAGE") == ${TEST_UID}:700 && + $(command stat -Lc '%u:%a' "$EXPECTED_SHARED") == ${TEST_UID}:700 ]] || fail "setgid sources were not hardened to 0700" mounts_ready || fail "final guard rejected rebound setgid sources" pass "hardening clears the setuid/setgid bits a numeric chmod keeps on directories" From 1ea2e9fe44ebd4390e657c311f6b70993bb333f9 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:29:44 +0200 Subject: [PATCH 14/21] fix(windows-vm): start the install watcher, arm it on failed launch, outlive slow downloads --- bin/omarchy-windows-vm | 44 ++++++++++++++++++++++++++++++++++++------ 1 file changed, 38 insertions(+), 6 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 1f75d9d76ae..7e76ec32090 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -616,15 +616,33 @@ restore_all_shared_privacy() { done } +# Positive evidence first: a successful inspect reporting anything but a live +# container (running, restarting, or created — a flip may still come) means +# no later samba start can flip the share again. A failed probe reads as +# gone too: the anchors assume a local daemon, which a down daemon cannot be +# flipping through — and only a success keeps the wait alive. +container_gone() { + local status + status=$(docker inspect --format='{{.State.Status}}' "$CONTAINER" 2>/dev/null) || return 0 + if [[ $status == "running" || $status == "restarting" || $status == "created" ]]; then + return 1 + else + return 0 + fi +} + # Unprivileged: samba.sh runs only after dockur's ISO download (10-15 minutes # on a fresh install). Do not hold a polkit session open for that. Watch the # caller's own share and chmod it as the owner once it becomes 2777. watch_share_privacy() { local dir="$1" i mode result [[ -e $dir ]] || return 0 - # 2x the documented 10-15 minute download is a thin margin on a slow link, - # so budget an hour; on expiry restore only if the share is still exposed. - for i in {1..3600}; do + # Twice the documented download is a thin margin on a slow link, so budget + # hours rather than minutes — and stop depending on the clock alone. A + # download slower than any fixed budget would outlast the watcher while + # samba can still flip the share later, so leave only once the share is + # fixed, the container is gone, or the budget ends. + for i in {1..21600}; do mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" if [[ $mode == 2777 || $mode == 777 ]]; then chmod_private_dir "$dir" 2>/dev/null || true @@ -633,6 +651,9 @@ watch_share_privacy() { # share is still exposed. Leave only once the mode is really 700. [[ $(stat -Lc '%a' "$dir" 2>/dev/null) == 700 ]] && return 0 fi + if ((i % 60 == 0)) && container_gone; then + break + fi sleep 1 done mode=$(stat -Lc '%a' "$dir" 2>/dev/null) || mode="" @@ -641,7 +662,7 @@ watch_share_privacy() { fi result=$(stat -Lc '%a' "$dir" 2>/dev/null) || result="missing" logger -t omarchy-windows-vm \ - "share privacy watcher timed out; $dir mode is $result" 2>/dev/null || true + "share privacy watcher exiting; $dir mode is $result" 2>/dev/null || true } schedule_share_privacy_restore() { @@ -652,14 +673,20 @@ schedule_share_privacy_restore() { # which is a uwsm-app systemd scope. Dismissing that window SIGTERMs leftover # cgroup members; trap '' HUP does not cover that. A user unit is outside # the scope, so the wait survives the terminal. + # Sourcing this file runs the dispatcher, so take the harmless help branch + # the way the shell tests do when sourcing it. if omarchy-cmd-present systemd-run; then systemctl --user reset-failed omarchy-windows-share-privacy.service 2>/dev/null || true systemctl --user stop omarchy-windows-share-privacy.service 2>/dev/null || true + # Stash the service arguments first: `set -- help` would otherwise + # clobber them, sourcing "help" instead of the helper and starting the + # watcher with an empty directory (silently watching nothing while the + # successful unit launch skips the fallback below). if systemd-run --user --quiet --collect \ --unit=omarchy-windows-share-privacy \ --description="Restore ~/Windows mode after dockur samba.sh" \ - /bin/bash -c 'set -- help; source "$1" >/dev/null; watch_share_privacy "$2"' \ - bash "$self" "$dir"; then + /bin/bash -c 'helper=$1; dir=$2; set -- help; source "$helper" >/dev/null; watch_share_privacy "$dir"' \ + watcher "$self" "$dir"; then return 0 fi fi @@ -1551,6 +1578,11 @@ launch_windows() { echo "Starting Windows VM (this may prompt for authorization)..." if ! priv up_wait; then + # The container may still be coming up in the background (slow download, + # slow guest): the one-shot priv-side restore in up_wait cannot cover a + # samba flip that lands after this failure, so arm the owner-side watcher + # for it before reporting. + schedule_share_privacy_restore echo "❌ Failed to start Windows VM!" echo " Try checking: omarchy-windows-vm status" omarchy-notification-send -u critical "Windows VM" "Failed to start Windows VM" From a17c6290dd80f3450c851dcbeaf9e5908c97f5fa Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:29:44 +0200 Subject: [PATCH 15/21] test(windows-vm): cover watcher start, launch arming, and watcher lifetime --- test/shell.d/windows-vm-test.sh | 134 ++++++++++++++++++++++++++++++++ 1 file changed, 134 insertions(+) diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index 169b138273a..589dd1f3b9e 100755 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -111,6 +111,140 @@ EOF ) pass "the install share watcher outlives the terminal install ran in" +# The user unit used to source "help" instead of the helper (`set -- help` +# clobbered the service arguments), so the watcher silently never started +# while the successful unit launch skipped the fallback entirely. A stub that +# runs the service command the way a successful start would must leave a 2777 +# share at 700 — the broken wiring leaves it exposed. +( + test_home=$(mktemp -d) + trap 'rm -rf "$test_home"' EXIT + mkdir -p "$test_home/Windows" "$test_home/bin" + chmod 2777 "$test_home/Windows" + cat >"$test_home/bin/systemd-run" <<'EOF' +#!/bin/bash +printf 'systemd-run' >>"$TEST_LOG" +printf '\t%s' "$@" >>"$TEST_LOG" +printf '\n' >>"$TEST_LOG" +# Simulate a successful transient unit start by running the service command. +while [[ $# -gt 0 && $1 != /bin/bash ]]; do shift; done +[[ $1 == /bin/bash ]] || exit 1 +exec "$@" +EOF + chmod +x "$test_home/bin/systemd-run" + cat >"$test_home/schedule.sh" <"\$TEST_LOG" +set -- help +source "$windows_vm_command" >/dev/null +schedule_share_privacy_restore +EOF + bash "$test_home/schedule.sh" || fail "the systemd share watcher failed to start" + grep -q 'omarchy-windows-share-privacy' "$test_home/systemd-run.log" || + fail "the share watcher did not start as a user unit" + [[ $(stat -Lc '%a' "$test_home/Windows") == 700 ]] || + fail "the systemd share watcher left the share at $(stat -Lc '%a' "$test_home/Windows")" +) +pass "the systemd share watcher starts and hardens the share" + +# A launch that fails (up_wait timeout, slow guest) leaves the container +# coming up in the background: the one-shot priv-side restore cannot cover a +# samba flip that lands after the failure, so the failure path must arm the +# owner-side watcher before reporting. +( + test_home=$(mktemp -d) + trap 'rm -rf "$test_home"' EXIT + mkdir -p "$test_home/runtime" "$test_home/Windows" + touch "$test_home/runtime/docker-compose.yml" + chmod 700 "$test_home/Windows" + HOME=$test_home + OMARCHY_WINDOWS_DIR=$test_home/runtime + export HOME OMARCHY_WINDOWS_DIR + set -- help + source "$windows_vm_command" >/dev/null + read_credential() { return 1; } + priv() { return 1; } + omarchy-notification-send() { return 0; } + schedule_share_privacy_restore() { : >"$test_home/scheduled"; } + if ( launch_windows "" ); then + fail "launch_windows succeeded with a failing up_wait" + fi + [[ -e $test_home/scheduled ]] || fail "a failed launch did not arm the share watcher" +) +pass "a failed launch arms the share privacy watcher" + +# The watcher must outlive a slow download: a flip that lands after minutes +# of private share still gets fixed while the container runs. +( + test_home=$(mktemp -d) + trap 'rm -rf "$test_home"' EXIT + mkdir -p "$test_home/Windows" + chmod 700 "$test_home/Windows" + HOME=$test_home + export HOME + set -- help + source "$windows_vm_command" >/dev/null + docker() { echo "running"; return 0; } + ( sleep 2; chmod 2777 "$test_home/Windows" ) & + flipper=$! + watch_share_privacy "$test_home/Windows" & watcher=$! + # Bounded wait: the fix lands seconds after the flip; a broken watcher + # would sit on the loop until its hour budget instead. + for _ in {1..40}; do + kill -0 $watcher 2>/dev/null || break + sleep 0.5 + done + if kill -0 $watcher 2>/dev/null; then + kill "$watcher" 2>/dev/null || true + wait "$watcher" 2>/dev/null || true + fail "the watcher did not fix a flip that landed while watching" + fi + wait "$watcher" 2>/dev/null || true + wait "$flipper" 2>/dev/null || true + [[ $(stat -Lc '%a' "$test_home/Windows") == 700 ]] || + fail "a late flip left the share at $(stat -Lc '%a' "$test_home/Windows")" +) +pass "the share watcher fixes a flip that lands while watching" + +# …but it must not watch forever: once the container is gone nothing can flip +# the share again, so the wait ends instead of sitting out its whole budget. +# (Only a successful inspect reporting a live container keeps it alive.) +( + test_home=$(mktemp -d) + trap 'rm -rf "$test_home"' EXIT + mkdir -p "$test_home/Windows" + chmod 700 "$test_home/Windows" + HOME=$test_home + export HOME + set -- help + source "$windows_vm_command" >/dev/null + cat >"$test_home/watch.sh" </dev/null +sleep() { :; } +docker() { return 1; } +watch_share_privacy "\$HOME/Windows" +EOF + timeout 120 bash "$test_home/watch.sh" || + fail "the watcher outlived a gone container" + [[ $(stat -Lc '%a' "$test_home/Windows") == 700 ]] || + fail "the watcher exit left the share at $(stat -Lc '%a' "$test_home/Windows")" + docker() { return 1; } + container_gone || fail "a missing container does not read as gone" + docker() { echo "running"; return 0; } + if container_gone; then + fail "a running container reads as gone" + fi + docker() { echo "restarting"; return 0; } + if container_gone; then + fail "a restarting container reads as gone" + fi +) +pass "the share watcher leaves once the container is gone" + # pkexec runs the packaged copy, which a dev link cannot shadow. A stale # packaged copy used to re-apply the pre-fix chmod semantics with no diagnostic, # failing a launch and leaving the share at 2700. The skew check must refuse From 2e3fbfaa033e5a49db3daaf264b7be95930395a6 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:39:06 +0200 Subject: [PATCH 16/21] fix(windows-vm): keep watching when the daemon cannot be inspected --- bin/omarchy-windows-vm | 22 ++++++++++++++-------- 1 file changed, 14 insertions(+), 8 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 7e76ec32090..3609945e1d9 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -616,15 +616,21 @@ restore_all_shared_privacy() { done } -# Positive evidence first: a successful inspect reporting anything but a live -# container (running, restarting, or created — a flip may still come) means -# no later samba start can flip the share again. A failed probe reads as -# gone too: the anchors assume a local daemon, which a down daemon cannot be -# flipping through — and only a success keeps the wait alive. +# Whether the VM container is gone, so no later samba start can flip the +# share again. Only positive evidence ends the wait: a successful inspect +# reporting anything but a live container (running, restarting, or created — +# a flip may still come), or a probe that positively reports no such +# container. Anything else — a permission-denied inspect on a default +# install whose daemon is root-owned, an unreachable daemon — keeps watching: +# exiting past a flip that may still come would reopen the hole, while a +# lingering watcher is merely untidy. container_gone() { - local status - status=$(docker inspect --format='{{.State.Status}}' "$CONTAINER" 2>/dev/null) || return 0 - if [[ $status == "running" || $status == "restarting" || $status == "created" ]]; then + local probe + probe=$(docker inspect --format='{{.State.Status}}' "$CONTAINER" 2>&1) || { + [[ $probe == *"No such object"* || $probe == *"No such container"* ]] && return 0 + return 1 + } + if [[ $probe == "running" || $probe == "restarting" || $probe == "created" ]]; then return 1 else return 0 From a744d131b92f3c012ca0b8f7af5492d0a7c5f1ad Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:39:06 +0200 Subject: [PATCH 17/21] test(windows-vm): denied probes keep watching, missing containers exit --- test/shell.d/windows-vm-test.sh | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index 589dd1f3b9e..d31aa14db2c 100755 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -225,15 +225,22 @@ export HOME=$test_home set -- help source "$windows_vm_command" >/dev/null sleep() { :; } -docker() { return 1; } +docker() { echo "Error: No such object: omarchy-windows" >&2; return 1; } watch_share_privacy "\$HOME/Windows" EOF timeout 120 bash "$test_home/watch.sh" || fail "the watcher outlived a gone container" [[ $(stat -Lc '%a' "$test_home/Windows") == 700 ]] || fail "the watcher exit left the share at $(stat -Lc '%a' "$test_home/Windows")" - docker() { return 1; } + docker() { echo "Error: No such object: omarchy-windows" >&2; return 1; } container_gone || fail "a missing container does not read as gone" + # A default install cannot inspect the root-owned daemon at all: that + # permission failure must keep watching, not read as gone while samba can + # still flip the share later. + docker() { echo "permission denied while trying to connect to the Docker daemon socket" >&2; return 1; } + if container_gone; then + fail "an uninspectable daemon reads as a gone container" + fi docker() { echo "running"; return 0; } if container_gone; then fail "a running container reads as gone" From 69d694d2b5af39adcb06a03594044364d275929d Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:51:19 +0200 Subject: [PATCH 18/21] fix(windows-vm): keep watching while the container is paused --- bin/omarchy-windows-vm | 13 ++++++------- 1 file changed, 6 insertions(+), 7 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 3609945e1d9..872e488fca4 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -618,19 +618,18 @@ restore_all_shared_privacy() { # Whether the VM container is gone, so no later samba start can flip the # share again. Only positive evidence ends the wait: a successful inspect -# reporting anything but a live container (running, restarting, or created — -# a flip may still come), or a probe that positively reports no such -# container. Anything else — a permission-denied inspect on a default -# install whose daemon is root-owned, an unreachable daemon — keeps watching: -# exiting past a flip that may still come would reopen the hole, while a -# lingering watcher is merely untidy. +# reporting anything but a live container (running, restarting, created, or +# paused — a paused entrypoint resumes, so a flip may still come), or a +# probe that positively reports no such container. Anything else — denied, +# unreachable — keeps watching: exiting past a flip that may still come +# would reopen the hole, while a lingering watcher is merely untidy. container_gone() { local probe probe=$(docker inspect --format='{{.State.Status}}' "$CONTAINER" 2>&1) || { [[ $probe == *"No such object"* || $probe == *"No such container"* ]] && return 0 return 1 } - if [[ $probe == "running" || $probe == "restarting" || $probe == "created" ]]; then + if [[ $probe == "running" || $probe == "restarting" || $probe == "created" || $probe == "paused" ]]; then return 1 else return 0 From 412701fbda765bf50f2eda6713107df8db4096f2 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:51:19 +0200 Subject: [PATCH 19/21] test(windows-vm): paused containers keep watching, stopped ones exit --- test/shell.d/windows-vm-test.sh | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index d31aa14db2c..6b578ede293 100755 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -249,6 +249,14 @@ EOF if container_gone; then fail "a restarting container reads as gone" fi + # A paused entrypoint resumes: samba can still flip the share afterwards, + # so paused keeps the watcher alive rather than ending it. + docker() { echo "paused"; return 0; } + if container_gone; then + fail "a paused container reads as gone" + fi + docker() { echo "exited"; return 0; } + container_gone || fail "a stopped container does not read as gone" ) pass "the share watcher leaves once the container is gone" From c341b07ecbc2585141270a21dd193f261090262b Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:57:43 +0200 Subject: [PATCH 20/21] fix(windows-vm): keep inspect diagnostics out of the container state match --- bin/omarchy-windows-vm | 18 +++++++++++++----- 1 file changed, 13 insertions(+), 5 deletions(-) diff --git a/bin/omarchy-windows-vm b/bin/omarchy-windows-vm index 872e488fca4..0e2a4218b09 100755 --- a/bin/omarchy-windows-vm +++ b/bin/omarchy-windows-vm @@ -622,14 +622,22 @@ restore_all_shared_privacy() { # paused — a paused entrypoint resumes, so a flip may still come), or a # probe that positively reports no such container. Anything else — denied, # unreachable — keeps watching: exiting past a flip that may still come -# would reopen the hole, while a lingering watcher is merely untidy. +# would reopen the hole, while a lingering watcher is merely untidy. Status +# and diagnostics travel separately: a warning on stderr must never poison +# the stdout match and read a live container as gone. container_gone() { - local probe - probe=$(docker inspect --format='{{.State.Status}}' "$CONTAINER" 2>&1) || { - [[ $probe == *"No such object"* || $probe == *"No such container"* ]] && return 0 + local status err_file + err_file=$(mktemp) || return 1 + status=$(docker inspect --format='{{.State.Status}}' "$CONTAINER" 2>"$err_file") || { + if grep -qE "No such object|No such container" "$err_file"; then + rm -f -- "$err_file" + return 0 + fi + rm -f -- "$err_file" return 1 } - if [[ $probe == "running" || $probe == "restarting" || $probe == "created" || $probe == "paused" ]]; then + rm -f -- "$err_file" + if [[ $status == "running" || $status == "restarting" || $status == "created" || $status == "paused" ]]; then return 1 else return 0 From 661748859ce1ccf414471d1fc41e3aaddaef1835 Mon Sep 17 00:00:00 2001 From: Emiel Kollof Date: Mon, 5 Oct 2026 08:57:43 +0200 Subject: [PATCH 21/21] test(windows-vm): stderr warnings do not read a live container as gone --- test/shell.d/windows-vm-test.sh | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/test/shell.d/windows-vm-test.sh b/test/shell.d/windows-vm-test.sh index 6b578ede293..90b37c05025 100755 --- a/test/shell.d/windows-vm-test.sh +++ b/test/shell.d/windows-vm-test.sh @@ -257,6 +257,12 @@ EOF fi docker() { echo "exited"; return 0; } container_gone || fail "a stopped container does not read as gone" + # A successful inspect that also warns on stderr must not poison the + # status match: diagnostics travel separately from the reported state. + docker() { echo "WARNING: API deprecation notice" >&2; echo "running"; return 0; } + if container_gone; then + fail "a warning on stderr reads a live container as gone" + fi ) pass "the share watcher leaves once the container is gone"