fix: preserve real exit code in EXIT trap and fold parallel job state

This commit is contained in:
openhands committed 2026-08-11 20:30:30 +02:00
1 parent ade0f286d3
commit 578a6e943a
1 file changed
+92 -14
+92 -14
View File
@@ -696,6 +696,9 @@ acquire_lock() {
}
release_lock() {
# Parallel job subshells inherit fd 9; releasing it here would drop the
# main process's lock while the other jobs are still running.
[ -n "$PARALLEL_JOB" ] && return 0
flock -u 9 2>/dev/null || true
rm -f "$LOCK_FILE" 2>/dev/null || true
}
@@ -719,7 +722,6 @@ die() {
cursor_show
exit 1
}
cursor_hide() { tput civis 2>/dev/null || true; }
cursor_show() { tput cnorm 2>/dev/null || true; }
clear_line() { printf "\r\033[K"; }
@@ -850,7 +852,12 @@ print(json.dumps(payload))
cleanup_notify_failure() { $NOTIFY_FAILED && return; NOTIFY_FAILED=true; notify "FAILED" "$1"; }
cleanup_on_exit() {
local ec=$?; cursor_show; release_lock
local ec=$?
# Parallel job subshells must never run the full cleanup (rollback,
# lock release, notify); only the main process owns those. Their status
# is captured by parallel_wait instead.
if [ -n "$PARALLEL_JOB" ]; then return $ec; fi
cursor_show; release_lock
if [ $ec -ne 0 ] && [ "$ROLLBACK_NEEDED" = true ]; then
echo -e "\n ${C_BG_RED}${C_WHITE}${C_BOLD} $(_t "ROLLBACK") ${C_RESET}"
echo -e " $(_t "Restoring config back-ups from") $CONFIG_BACKUP_DIR/rollback..."
@@ -904,17 +911,22 @@ detect_best_branch() {
git_update() {
local repo="$1" branch="$2"
local name; name=$(basename "$repo")
local rc=1
(
cd "$repo" || { warn "Cannot access $repo"; exit 1; }
# Only stash when the working tree is dirty, to avoid losing nothing
# and to keep the pull fast-forwardable.
cd "$repo" || { warn "Cannot access $repo"; exit 2; }
# Only stash TRACKED changes (never untracked files — runtime config
# data in public/configuration/ must stay in place) and only when the
# working tree is dirty, to keep the pull fast-forwardable. The stash is
# left in the repo (recoverable via `git stash list`), never dropped.
if ! git diff --quiet 2>/dev/null || ! git diff --cached --quiet 2>/dev/null; then
git stash --include-untracked 2>/dev/null || true
git stash push -m "auto-update $(date +%F-%T)" 2>/dev/null \
&& warn "Local tracked changes in $(basename "$repo") stashed (see 'git stash list')"
fi
local old_head; old_head=$(git rev-parse HEAD 2>/dev/null || echo "")
local rb; rb=$(detect_best_branch "$repo" "$branch")
[ "$rb" != "$branch" ] && info "Branch '$branch' not in $(basename "$repo"), using '$rb'"
git checkout "$rb" 2>/dev/null || { warn "Cannot checkout '$rb' in $(basename "$repo")"; exit 1; }
git checkout "$rb" 2>/dev/null || { warn "Cannot checkout '$rb' in $(basename "$repo")"; exit 2; }
local pulled=1
for attempt in 1 2 3; do
if git pull --ff-only origin "$rb" 2>/dev/null; then
@@ -925,9 +937,38 @@ git_update() {
warn "Pull attempt %s failed in %s, retrying..." "$attempt" "$(basename "$repo")" >&2
sleep 2
done
[ "$pulled" -ne 0 ] && { warn "Pull failed in $(basename "$repo")"; exit 1; }
[ "$(git rev-parse HEAD 2>/dev/null)" != "$old_head" ]
if [ "$pulled" -ne 0 ]; then
warn "Pull failed in $(basename "$repo")"
echo "git error: pull failed" > "$PARALLEL_STATE_DIR/giterror-$name"
exit 2
fi
# Remember the pre-update commit so a later build failure can reset the
# repo back to it; otherwise the next run would see HEAD==origin and
# silently skip the never-built change.
echo "$old_head" > "$PARALLEL_STATE_DIR/oldhead-$name"
[ "$(git rev-parse HEAD 2>/dev/null)" != "$old_head" ] && exit 0
exit 1
)
rc=$?
# Return 0 = new commits fetched, 1 = nothing to do, 2 = real git error.
return $rc
}
# Reverse a successful pull after a failed build, so the change is picked up
# again on the next run instead of being silently skipped (HEAD would otherwise
# equal origin and git_update would report "already up to date").
git_revert_repo() {
local repo="$1"
local name; name=$(basename "$repo")
[ -f "$PARALLEL_STATE_DIR/oldhead-$name" ] || return 0
local prev; prev=$(cat "$PARALLEL_STATE_DIR/oldhead-$name" 2>/dev/null)
[ -n "$prev" ] || return 0
if git -C "$repo" reset --hard "$prev" >/dev/null 2>&1; then
warn "Reverted $(basename "$repo") to $prev (build failed)"
else
warn "Could not revert $(basename "$repo") to $prev"
fi
rm -f "$PARALLEL_STATE_DIR/oldhead-$name" 2>/dev/null || true
}
# yarn install scoped to a per-repo cache folder. Parallel installs (renderer +
@@ -940,7 +981,11 @@ yarn_install_repo() {
local cache_dir="$YARN_CACHE_DIR/$(basename "$repo")"
mkdir -p "$cache_dir" 2>/dev/null || true
cd "$repo" 2>/dev/null || { warn "Cannot cd into %s" "$repo"; return 1; }
yarn install --cache-folder "$cache_dir" "$@" >> "$LOG_FILE" 2>&1
# NODE_ENV may be exported as "production" via .env (the CMS runs in prod),
# which makes yarn 1 skip devDependencies — vite is a devDep, so builds
# would fail with "vite: not found". Force a non-production env so the full
# dependency tree (including dev deps) is always installed.
NODE_ENV=development yarn install --cache-folder "$cache_dir" "$@" >> "$LOG_FILE" 2>&1
local rc=$?
cd "$SCRIPT_DIR" 2>/dev/null || true
return $rc
@@ -974,6 +1019,10 @@ parallel_run() {
local name="$1"; shift
PARALLEL_NAMES+=("$name")
(
# This subshell inherits the parent's EXIT trap and lock fd 9. Mark it
# so cleanup_on_exit/release_lock become no-ops here: the main process
# owns rollback, lock release and notifications.
PARALLEL_JOB=1
# Snapshot inherited state so we only persist the delta back: the
# emulator (run before the parallel phase) may already be in
# UPDATED_REPOS, so emitting the full array would duplicate it.
@@ -1147,7 +1196,15 @@ preflight_check() {
# =============================================================================
update_emulator() {
step 2 8 "Update & Build Emulator"
local grc=1
if git_update "$EMULATOR_REPO" "$NITRO_BRANCH"; then
grc=0
else
grc=$?
fi
if [ $grc -eq 2 ]; then
die "Emulator update failed (git error)"
elif [ $grc -eq 0 ]; then
HAD_UPDATES=true; UPDATED_REPOS+=("Emulator")
cd "$EMULATOR_MODULE"
backup_binaries
@@ -1163,6 +1220,7 @@ update_emulator() {
else
spinner_stop fail
tail -30 "$LOG_FILE" | grep -E '(ERROR|FAILURE|BUILD)' || true
git_revert_repo "$EMULATOR_REPO"
die "Maven build failed (see: %s)" "$LOG_FILE"
fi
local jar=$(find target -maxdepth 1 -name 'Polaris-*-jar-with-dependencies.jar' -printf '%T@ %p\n' 2>/dev/null | sort -rn | sed -n '1s/^[0-9.]* //p' | xargs -r basename 2>/dev/null || echo "")
@@ -1186,14 +1244,24 @@ EOJ
update_renderer() {
step 3 8 "Update Nitro_Render_V3"
local grc=1
if git_update "$NITRO_RENDERER" "$NITRO_BRANCH"; then
grc=0
else
grc=$?
fi
if [ $grc -eq 2 ]; then
# Let the caller fail the parallel group; the state file records it.
warn "Renderer update failed (git error)"
return 1
elif [ $grc -eq 0 ]; then
HAD_UPDATES=true; UPDATED_REPOS+=("Nitro_Render_V3")
cd "$NITRO_RENDERER"
spinner_start "Installing deps..."
if ! yarn_install_repo "$NITRO_RENDERER" --frozen-lockfile --prefer-offline; then
warn "Renderer: yarn install failed — cleaning node_modules and retrying"
yarn_reset_repo "$NITRO_RENDERER"
yarn_install_repo "$NITRO_RENDERER" || { spinner_stop fail; die "yarn install failed (see: %s)" "$LOG_FILE"; }
yarn_install_repo "$NITRO_RENDERER" || { spinner_stop fail; git_revert_repo "$NITRO_RENDERER"; die "yarn install failed (see: %s)" "$LOG_FILE"; }
fi
cd "$NITRO_RENDERER"
if [ ! -d node_modules/pixi.js ]; then
@@ -1201,7 +1269,7 @@ update_renderer() {
# Parallel yarn installs can corrupt a shared cache; resetting the
# repo's own cache folder guarantees a clean fetch from the registry.
yarn_reset_repo "$NITRO_RENDERER"
yarn_install_repo "$NITRO_RENDERER" || { spinner_stop fail; die "yarn install failed (see: %s)" "$LOG_FILE"; }
yarn_install_repo "$NITRO_RENDERER" || { spinner_stop fail; git_revert_repo "$NITRO_RENDERER"; die "yarn install failed (see: %s)" "$LOG_FILE"; }
fi
spinner_stop ok; ok "Renderer dependencies installed"
else
@@ -1213,7 +1281,16 @@ update_renderer() {
update_client() {
step 4 8 "Update & Build Nitro-V3"
local grc=1
if git_update "$NITRO_CLIENT" "$NITRO_BRANCH"; then
grc=0
else
grc=$?
fi
if [ $grc -eq 2 ]; then
warn "Nitro-V3 update failed (git error)"
return 1
elif [ $grc -eq 0 ]; then
HAD_UPDATES=true; NITRO_BUILT=true; UPDATED_REPOS+=("Nitro-V3")
cd "$NITRO_CLIENT"
backup_binaries
@@ -1224,7 +1301,7 @@ update_client() {
if ! yarn_install_repo "$NITRO_CLIENT" --frozen-lockfile --prefer-offline; then
warn "Nitro-V3: yarn install failed — cleaning node_modules and retrying"
yarn_reset_repo "$NITRO_CLIENT"
yarn_install_repo "$NITRO_CLIENT" || { spinner_stop fail; die "yarn install failed (see: %s)" "$LOG_FILE"; }
yarn_install_repo "$NITRO_CLIENT" || { spinner_stop fail; git_revert_repo "$NITRO_CLIENT"; die "yarn install failed (see: %s)" "$LOG_FILE"; }
fi
cd "$NITRO_CLIENT"
if [ ! -x node_modules/.bin/vite ]; then
@@ -1232,7 +1309,7 @@ update_client() {
# Parallel yarn installs can corrupt a shared cache; resetting the
# repo's own cache folder guarantees a clean fetch from the registry.
yarn_reset_repo "$NITRO_CLIENT"
yarn_install_repo "$NITRO_CLIENT" || { spinner_stop fail; die "yarn install failed (see: %s)" "$LOG_FILE"; }
yarn_install_repo "$NITRO_CLIENT" || { spinner_stop fail; git_revert_repo "$NITRO_CLIENT"; die "yarn install failed (see: %s)" "$LOG_FILE"; }
cd "$NITRO_CLIENT"
fi
spinner_stop ok
@@ -1242,6 +1319,7 @@ update_client() {
else
spinner_stop fail
tail -30 "$LOG_FILE" || true
git_revert_repo "$NITRO_CLIENT"
die "yarn build failed (see: %s)" "$LOG_FILE"
fi
else