From c8a63e691fbbb3b37c1667b3730a6313c5c161ce Mon Sep 17 00:00:00 2001
From: Alexey <1556417+alex-solovyev@users.noreply.github.com>
Date: Sat, 10 Oct 2026 22:45:00 +0000
Subject: [PATCH 1/5] fix: explain tag-push release evidence failures
(GH#34286)
---
...ll-loop-helper-state-lifecycle-commands.sh | 26 +++++++++++++++++++
1 file changed, 26 insertions(+)
diff --git a/.agents/scripts/full-loop-helper-state-lifecycle-commands.sh b/.agents/scripts/full-loop-helper-state-lifecycle-commands.sh
index 7a20929952..da5bcbd0d1 100644
--- a/.agents/scripts/full-loop-helper-state-lifecycle-commands.sh
+++ b/.agents/scripts/full-loop-helper-state-lifecycle-commands.sh
@@ -760,6 +760,29 @@ _full_loop_resolve_remote_release_tag_commit() {
return 0
}
+_full_loop_report_release_run_events() {
+ local repo="$1"
+ local workflow_file="$2"
+ local merge_commit="$3"
+ local tag_name="$4"
+ local runs_json=""
+ local events=""
+ # Diagnostics only: never substitute these runs for publication evidence.
+ runs_json=$(gh api "repos/${repo}/actions/workflows/${workflow_file}/runs?head_sha=${merge_commit}&status=success&per_page=100" 2>/dev/null) || return 0
+ events=$(jq -r --arg sha "$merge_commit" '
+ [.workflow_runs[]? | select(.head_sha == $sha and .status == "completed" and .conclusion == "success")
+ | .event | select(type == "string")] | unique | join(", ")
+ ' <<<"$runs_json" 2>/dev/null) || return 0
+ print_error "Successful events of workflow ${workflow_file} for ${merge_commit}: ${events:-none}"
+ if jq -e --arg sha "$merge_commit" --arg tag "$tag_name" '
+ any(.workflow_runs[]?; .head_sha == $sha and .head_branch == $tag and
+ .event == "push" and .status == "completed" and .conclusion == "success")
+ ' <<<"$runs_json" >/dev/null 2>&1; then
+ print_error "A successful tag-push run exists for ${tag_name}; retry with --workflow ${workflow_file} --event push"
+ fi
+ return 0
+}
+
_full_loop_verify_published_release() {
local repo="$1"
local tag_name="$2"
@@ -835,6 +858,9 @@ _full_loop_verify_published_release() {
)] | length > 0
' <<<"$workflow_runs_json" >/dev/null || {
print_error "No successful ${workflow_event} run${workflow_file:+ of workflow ${workflow_file}} found for ${merge_commit}"
+ if [[ -n "$workflow_file" && "$workflow_event" == "$_FULL_LOOP_WORKFLOW_EVENT_RELEASE" ]]; then
+ _full_loop_report_release_run_events "$repo" "$workflow_file" "$merge_commit" "$tag_name"
+ fi
return 1
}
return 0
From 6684634c94ba3f7e42839b66679c3943dc0db4ae Mon Sep 17 00:00:00 2001
From: Alexey <1556417+alex-solovyev@users.noreply.github.com>
Date: Sat, 10 Oct 2026 22:48:09 +0000
Subject: [PATCH 2/5] test: cover tag-push release diagnostic safety (GH#34286)
---
.../test-full-loop-completion-evidence.sh | 32 +++++++++++++++++++
1 file changed, 32 insertions(+)
diff --git a/.agents/scripts/tests/test-full-loop-completion-evidence.sh b/.agents/scripts/tests/test-full-loop-completion-evidence.sh
index 4a0b183f50..3a213631ea 100755
--- a/.agents/scripts/tests/test-full-loop-completion-evidence.sh
+++ b/.agents/scripts/tests/test-full-loop-completion-evidence.sh
@@ -67,6 +67,20 @@ if [[ "$*" == *"/actions/runs?event=release"* || "$*" == *"/actions/workflows/re
esac
exit 0
fi
+if [[ "$*" == *"/actions/workflows/release.yml/runs?event=release&"* ]]; then
+ printf '%s\n' '{"workflow_runs":[]}'
+ exit 0
+fi
+if [[ "$*" == *"/actions/workflows/release.yml/runs?head_sha="* ]]; then
+ case "$release_mode" in
+ diagnostic-unavailable) exit 70 ;;
+ diagnostic-wrong-sha) merge_sha="2222222222222222222222222222222222222222" ;;
+ esac
+ jq -cn --arg sha "$merge_sha" --arg mode "$release_mode" '{workflow_runs:[{event:"push",status:"completed",
+ conclusion:(if $mode == "diagnostic-failed" then "failure" else "success" end),
+ head_branch:(if $mode == "diagnostic-main" then "main" else "v3.0.0" end),head_sha:$sha}]}'
+ exit 0
+fi
if [[ "$*" == *"repos/marcusquinn/aidevops/actions/workflows/release.yml/runs?event=push&status=success&per_page=100"* ]]; then
case "$release_mode" in
push-failed-workflow)
@@ -263,6 +277,24 @@ AIDEVOPS_FULL_LOOP_RECEIPT_DIR="$receipt_dir" AIDEVOPS_FULL_LOOP_CLEANUP_DIR="$c
cmp -s "$published_cleanup_receipt" "${ROOT}/published-release-receipt-before.json"
printf 'PASS verified manual release records published evidence and reconciles cleanup idempotently\n'
+for diagnostic_mode in valid diagnostic-wrong-sha diagnostic-failed diagnostic-main diagnostic-unavailable; do
+ if diagnostic_output=$(COMPLETION_RELEASE_MODE="$diagnostic_mode" AIDEVOPS_FULL_LOOP_RECEIPT_DIR="$receipt_dir" \
+ PATH="${ROOT}/bin:/opt/homebrew/bin:/usr/bin:/bin" \
+ bash "$published_record_runner" 60 v3.0.0 marcusquinn/aidevops --workflow release.yml 2>&1); then
+ printf 'FAIL diagnostic query recorded publication evidence\n'
+ exit 1
+ fi
+ [[ ! -e "${receipt_dir}/marcusquinn_aidevops-60.status" ]]
+ [[ "$diagnostic_output" == *"No successful release run of workflow release.yml"* ]]
+ if [[ "$diagnostic_mode" == "valid" ]]; then
+ [[ "$diagnostic_output" == *"Successful events of workflow release.yml"*": push"* ]]
+ [[ "$diagnostic_output" == *"--event push"* ]]
+ else
+ [[ "$diagnostic_output" != *"--event push"* ]]
+ fi
+done
+printf 'PASS missing release event suggests matching tag-push evidence without accepting mismatched or unavailable diagnostics\n'
+
push_published_worktree="${ROOT}/push-published-release-worktree"
mkdir -p "$push_published_worktree"
push_published_receipt=$(full_loop_write_cleanup_deferred marcusquinn/aidevops 61 "$push_published_worktree" \
From bd9290663524d17ec95b5119ba40c548813f8955 Mon Sep 17 00:00:00 2001
From: alex-solovyev <1556417+alex-solovyev@users.noreply.github.com>
Date: Sat, 10 Oct 2026 19:00:21 -0400
Subject: [PATCH 3/5] GH#34287: fix(pulse): resume interrupted async
housekeeping and reconcile before blocker catch-up (#34291)
* GH#34287: fix(pulse): resume interrupted async housekeeping and reconcile before blocker catch-up
* GH#34287: fix(pulse): initialise housekeeping lock/state locals for unbound-var lint
---
.agents/scripts/pulse-dispatch-engine.sh | 182 ++++++++++++++++--
.../test-pulse-post-dispatch-housekeeping.sh | 138 ++++++++++++-
2 files changed, 308 insertions(+), 12 deletions(-)
diff --git a/.agents/scripts/pulse-dispatch-engine.sh b/.agents/scripts/pulse-dispatch-engine.sh
index 04bbbced97..9fc0863c49 100755
--- a/.agents/scripts/pulse-dispatch-engine.sh
+++ b/.agents/scripts/pulse-dispatch-engine.sh
@@ -1185,6 +1185,145 @@ _pulse_run_blocker_refresh_catchup() {
return 0
}
+#######################################
+# Return the post-dispatch housekeeping stage-progress state file path.
+#
+# Lives next to (not inside) the lock directory so a stale-lock reclaim, which
+# removes the lock directory, keeps the interrupted run's progress (GH#34287).
+#
+# Returns 0 always; emits path on stdout.
+#######################################
+_pulse_post_dispatch_housekeeping_statefile() {
+ printf '%s\n' "${HOME}/.aidevops/logs/pulse-post-dispatch-housekeeping.state"
+ return 0
+}
+
+#######################################
+# Return 0 when a housekeeping stage completed within the resume window.
+#
+# The state file holds `stage=epoch` lines written after each stage finishes
+# and is removed by a clean, complete run. Fresh entries therefore only exist
+# after an interrupted run (deploy reconciliation SIGTERM, crash, reboot).
+# AIDEVOPS_PULSE_HOUSEKEEPING_RESUME_WINDOW_S=0 disables resume.
+#
+# Args:
+# $1 - stage name
+# $2 - state file path
+# Returns 0 if recently completed, 1 otherwise; emits age seconds on stdout.
+#######################################
+_pulse_housekeeping_stage_recently_done() {
+ local stage_name="$1"
+ local state_file="$2"
+ local window="${AIDEVOPS_PULSE_HOUSEKEEPING_RESUME_WINDOW_S:-1800}"
+ [[ "$window" =~ ^[0-9]+$ ]] || window=1800
+ [[ "$window" -gt 0 && -f "$state_file" ]] || return 1
+
+ local line="" done_epoch=""
+ while IFS= read -r line || [[ -n "$line" ]]; do
+ if [[ "$line" == "${stage_name}="* ]]; then
+ done_epoch="${line#*=}"
+ fi
+ done <"$state_file"
+ [[ "$done_epoch" =~ ^[0-9]+$ ]] || return 1
+
+ local now=""
+ now=$(date +%s 2>/dev/null) || return 1
+ [[ "$now" =~ ^[0-9]+$ ]] || return 1
+ local age=$((now - done_epoch))
+ [[ "$age" -ge 0 && "$age" -lt "$window" ]] || return 1
+ printf '%s\n' "$age"
+ return 0
+}
+
+#######################################
+# Record a finished housekeeping stage in the state file (stage=epoch).
+#
+# Args:
+# $1 - stage name
+# $2 - state file path
+# Returns 0 always (progress persistence is best-effort).
+#######################################
+_pulse_housekeeping_mark_stage_done() {
+ local stage_name="$1"
+ local state_file="$2"
+ local now=""
+ now=$(date +%s 2>/dev/null) || return 0
+ [[ "$now" =~ ^[0-9]+$ ]] || return 0
+
+ local tmp_file="${state_file}.tmp.${BASHPID:-$$}"
+ local line=""
+ {
+ if [[ -f "$state_file" ]]; then
+ while IFS= read -r line || [[ -n "$line" ]]; do
+ [[ -n "$line" && "$line" != "${stage_name}="* ]] && printf '%s\n' "$line"
+ done <"$state_file"
+ fi
+ printf '%s=%s\n' "$stage_name" "$now"
+ } >"$tmp_file" 2>/dev/null || {
+ rm -f "$tmp_file" 2>/dev/null || true
+ return 0
+ }
+ mv -f "$tmp_file" "$state_file" 2>/dev/null || rm -f "$tmp_file" 2>/dev/null || true
+ return 0
+}
+
+#######################################
+# Run one resumable housekeeping stage.
+#
+# Skips the stage when an interrupted earlier run already completed it within
+# the resume window; otherwise records it as the current stage (for the
+# signal trap), runs it, and persists its completion.
+#
+# Args:
+# $1 - stage name (state key)
+# $2 - state file path
+# $@ - stage command and arguments
+# Returns 0 always.
+#######################################
+_pulse_run_resumable_housekeeping_stage() {
+ local stage_name="$1"
+ local state_file="$2"
+ shift 2
+ local age=""
+ if age=$(_pulse_housekeeping_stage_recently_done "$stage_name" "$state_file"); then
+ echo "[pulse-wrapper] Async post-dispatch housekeeping: resume — skipping ${stage_name} (completed ${age}s ago by an interrupted run)" >>"$LOGFILE"
+ return 0
+ fi
+ _PULSE_HOUSEKEEPING_CURRENT_STAGE="$stage_name"
+ local stage_rc=0
+ "$@" || stage_rc=$?
+ # A stage child killed by a signal (e.g. the same deploy SIGTERM that is
+ # about to reach this subshell) did not finish: leave it for the resume.
+ if [[ "$stage_rc" -gt 128 ]]; then
+ echo "[pulse-wrapper] Async post-dispatch housekeeping: ${stage_name} ended by signal (rc=${stage_rc}) — not recorded as complete" >>"$LOGFILE"
+ else
+ _pulse_housekeeping_mark_stage_done "$stage_name" "$state_file"
+ fi
+ _PULSE_HOUSEKEEPING_CURRENT_STAGE="between_stages"
+ return 0
+}
+
+#######################################
+# Log and exit when the async housekeeping subshell receives a signal.
+#
+# Setup reconciliation SIGTERMs every Pulse runtime process on deploy
+# (intended: old-bundle code must be replaced). Logging the signal and the
+# active stage makes those deaths attributable from pulse.log (GH#34287).
+#
+# Args:
+# $1 - signal name (TERM or HUP)
+# Exits 143 for TERM, 129 for HUP.
+#######################################
+_pulse_housekeeping_on_signal() {
+ local signal_name="$1"
+ trap - TERM HUP
+ echo "[pulse-wrapper] Async post-dispatch housekeeping: terminated by SIG${signal_name} during ${_PULSE_HOUSEKEEPING_CURRENT_STAGE:-unknown}" >>"$LOGFILE"
+ if [[ "$signal_name" == "HUP" ]]; then
+ exit 129
+ fi
+ exit 143
+}
+
#######################################
# Run non-dispatch post-dispatch housekeeping stages.
#
@@ -1192,6 +1331,11 @@ _pulse_run_blocker_refresh_catchup() {
# immediate worker claim/ledger safety boundary. They can therefore run under a
# separate lock while the main pulse proceeds to prefetch + the next refill.
#
+# GH#34287: deploy reconciliation kills long runs, so per-stage completion is
+# persisted and a following run resumes at the first incomplete stage. The
+# cheap terminal reconciles (preflight_ownership_reconcile) run before the
+# slow GH#33246 blocker-refresh catch-up so they are not starved behind it.
+#
# Args:
# $1 - per-stage timeout seconds
# Returns 0 always.
@@ -1200,22 +1344,34 @@ _pulse_run_post_dispatch_housekeeping_stages() {
local stage_timeout="${1:-${PREFLIGHT_GROUP_TIMEOUT:-${PRE_RUN_STAGE_TIMEOUT:-600}}}"
[[ "$stage_timeout" =~ ^[0-9]+$ ]] || stage_timeout=600
- local lockdir
+ local lockdir="" state_file=""
lockdir="$(_pulse_post_dispatch_housekeeping_lockdir)"
+ state_file="$(_pulse_post_dispatch_housekeeping_statefile)"
if ! _pulse_acquire_post_dispatch_housekeeping_lock "$lockdir"; then
return 0
fi
+ _PULSE_HOUSEKEEPING_CURRENT_STAGE="startup"
echo "[pulse-wrapper] Async post-dispatch housekeeping: started (timeout=${stage_timeout}s)" >>"$LOGFILE"
- _pulse_run_optional_stage_with_timeout "coderabbit_review" "$stage_timeout" run_daily_codebase_review || true
- _pulse_run_optional_stage_with_timeout "post_merge_scanner" "$stage_timeout" _run_post_merge_review_scanner || true
- _pulse_run_optional_stage_with_timeout "pr_review_thread_response" "$stage_timeout" _run_pr_review_thread_response_scanner || true
- _pulse_run_optional_stage_with_timeout "auto_decomposer_scanner" "$stage_timeout" _run_auto_decomposer_scanner || true
- _pulse_run_optional_stage_with_timeout "dedup_cleanup" "$stage_timeout" run_simplification_dedup_cleanup || true
- _pulse_run_optional_stage_with_timeout "fast_fail_prune_expired" "$stage_timeout" fast_fail_prune_expired || true
- _pulse_run_blocker_refresh_catchup "$stage_timeout" || true
- run_stage_with_timeout "preflight_ownership_reconcile" "$stage_timeout" \
- _preflight_ownership_reconcile "$stage_timeout" || true
+ _pulse_run_resumable_housekeeping_stage "coderabbit_review" "$state_file" \
+ _pulse_run_optional_stage_with_timeout "coderabbit_review" "$stage_timeout" run_daily_codebase_review
+ _pulse_run_resumable_housekeeping_stage "post_merge_scanner" "$state_file" \
+ _pulse_run_optional_stage_with_timeout "post_merge_scanner" "$stage_timeout" _run_post_merge_review_scanner
+ _pulse_run_resumable_housekeeping_stage "pr_review_thread_response" "$state_file" \
+ _pulse_run_optional_stage_with_timeout "pr_review_thread_response" "$stage_timeout" _run_pr_review_thread_response_scanner
+ _pulse_run_resumable_housekeeping_stage "auto_decomposer_scanner" "$state_file" \
+ _pulse_run_optional_stage_with_timeout "auto_decomposer_scanner" "$stage_timeout" _run_auto_decomposer_scanner
+ _pulse_run_resumable_housekeeping_stage "dedup_cleanup" "$state_file" \
+ _pulse_run_optional_stage_with_timeout "dedup_cleanup" "$stage_timeout" run_simplification_dedup_cleanup
+ _pulse_run_resumable_housekeeping_stage "fast_fail_prune_expired" "$state_file" \
+ _pulse_run_optional_stage_with_timeout "fast_fail_prune_expired" "$stage_timeout" fast_fail_prune_expired
+ _pulse_run_resumable_housekeeping_stage "preflight_ownership_reconcile" "$state_file" \
+ run_stage_with_timeout "preflight_ownership_reconcile" "$stage_timeout" \
+ _preflight_ownership_reconcile "$stage_timeout"
+ _pulse_run_resumable_housekeeping_stage "blocker_refresh_catchup" "$state_file" \
+ _pulse_run_blocker_refresh_catchup "$stage_timeout"
+ _PULSE_HOUSEKEEPING_CURRENT_STAGE="finishing"
+ rm -f "$state_file" 2>/dev/null || true
echo "[pulse-wrapper] Async post-dispatch housekeeping: complete" >>"$LOGFILE"
_pulse_release_post_dispatch_housekeeping_lock "$lockdir"
@@ -1251,7 +1407,11 @@ _pulse_start_post_dispatch_housekeeping() {
set -m 2>/dev/null || true
(
set +m 2>/dev/null || true
- trap - EXIT INT TERM
+ trap - EXIT INT
+ # GH#34287: attribute deploy-reconciliation kills in pulse.log.
+ _PULSE_HOUSEKEEPING_CURRENT_STAGE="launch"
+ trap '_pulse_housekeeping_on_signal TERM' TERM
+ trap '_pulse_housekeeping_on_signal HUP' HUP
AIDEVOPS_PULSE_STAGE_CYCLE_CLAMP=0
_pulse_run_post_dispatch_housekeeping_stages "$stage_timeout"
) >>"$LOGFILE" 2>&1 &
diff --git a/.agents/scripts/tests/test-pulse-post-dispatch-housekeeping.sh b/.agents/scripts/tests/test-pulse-post-dispatch-housekeeping.sh
index 7ff5c73691..eb3c0fd5c3 100755
--- a/.agents/scripts/tests/test-pulse-post-dispatch-housekeeping.sh
+++ b/.agents/scripts/tests/test-pulse-post-dispatch-housekeeping.sh
@@ -273,7 +273,8 @@ test_housekeeping_blocker_refresh_catchup() {
touch -t 202601010000 "$DEP_GRAPH_CACHE_FILE"
_pulse_run_post_dispatch_housekeeping_stages 7
got=$(_catchup_stages)
- if [[ "$got" != "dep_graph:1 blocked_refresh brief_hold_release ownership_reconcile " ]]; then
+ # GH#34287: the cheap terminal reconcile runs before the slow catch-up.
+ if [[ "$got" != "ownership_reconcile dep_graph:1 blocked_refresh brief_hold_release " ]]; then
failures=$((failures + 1))
failmsg="${failmsg} | stale cache order: ${got}"
fi
@@ -306,11 +307,146 @@ test_housekeeping_blocker_refresh_catchup() {
return 0
}
+_dead_pid() {
+ sleep 0 &
+ local pid=$!
+ wait "$pid" 2>/dev/null || true
+ printf '%s\n' "$pid"
+ return 0
+}
+
+# GH#34287: a run that reclaims a killed run's lock resumes at the first
+# incomplete stage; a clean complete run clears the progress state.
+test_housekeeping_resumes_after_interrupted_run() {
+ setup_test_env
+ local lockdir state_file now failures=0 failmsg=""
+ lockdir="$(_pulse_post_dispatch_housekeeping_lockdir)"
+ state_file="$(_pulse_post_dispatch_housekeeping_statefile)"
+ now=$(date +%s)
+ mkdir -p "$lockdir"
+ _dead_pid >"${lockdir}/pid"
+ printf 'coderabbit_review=%s\npost_merge_scanner=%s\npr_review_thread_response=%s\n' \
+ "$((now - 60))" "$((now - 30))" "$((now - 4000))" >"$state_file"
+
+ _pulse_run_post_dispatch_housekeeping_stages 7
+
+ local skipped
+ for skipped in coderabbit post_merge; do
+ if grep -q "^${skipped}$" "$STAGE_LOG" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | fresh completed stage ${skipped} re-ran"
+ fi
+ done
+ local expected
+ for expected in pr_review_thread_response auto_decomposer dedup_cleanup fast_fail_prune ownership_reconcile; do
+ if ! grep -q "^${expected}$" "$STAGE_LOG" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | incomplete or stale stage ${expected} skipped"
+ fi
+ done
+ if ! grep -q 'reclaiming stale lock' "$LOGFILE" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | stale lock not reclaimed"
+ fi
+ if ! grep -q 'resume — skipping coderabbit_review' "$LOGFILE" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | resume skip log missing"
+ fi
+ if [[ -e "$state_file" ]]; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | complete run left progress state"
+ fi
+
+ # Resume disabled: every stage runs despite fresh state.
+ : >"$STAGE_LOG"
+ printf 'coderabbit_review=%s\n' "$now" >"$state_file"
+ AIDEVOPS_PULSE_HOUSEKEEPING_RESUME_WINDOW_S=0 _pulse_run_post_dispatch_housekeeping_stages 7
+ if ! grep -q '^coderabbit$' "$STAGE_LOG" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | disabled resume skipped coderabbit"
+ fi
+
+ if [[ "$failures" -eq 0 ]]; then
+ print_result "post-dispatch housekeeping resumes at first incomplete stage" 0
+ else
+ print_result "post-dispatch housekeeping resumes at first incomplete stage" 1 "$failmsg"
+ fi
+ teardown_test_env
+ return 0
+}
+
+_stage_rc_143() { return 143; }
+_stage_rc_1() { return 1; }
+
+# GH#34287: a stage killed by a signal is not recorded as complete.
+test_housekeeping_signal_killed_stage_not_recorded() {
+ setup_test_env
+ local state_file failures=0 failmsg=""
+ state_file="$(_pulse_post_dispatch_housekeeping_statefile)"
+ _pulse_run_resumable_housekeeping_stage "killed_stage" "$state_file" _stage_rc_143
+ _pulse_run_resumable_housekeeping_stage "failed_stage" "$state_file" _stage_rc_1
+ if grep -q '^killed_stage=' "$state_file" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | signal-killed stage recorded"
+ fi
+ if ! grep -q '^failed_stage=[0-9][0-9]*$' "$state_file" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | finished (failed) stage not recorded"
+ fi
+ if [[ "$failures" -eq 0 ]]; then
+ print_result "post-dispatch housekeeping does not record signal-killed stages" 0
+ else
+ print_result "post-dispatch housekeeping does not record signal-killed stages" 1 "$failmsg"
+ fi
+ teardown_test_env
+ return 0
+}
+
+# GH#34287: SIGTERM (deploy reconciliation) is attributed in pulse.log.
+test_async_housekeeping_logs_termination() {
+ setup_test_env
+ export AIDEVOPS_PULSE_ASYNC_POST_DISPATCH_HOUSEKEEPING=1
+ export TEST_STAGE_SLEEP_ONCE=2
+ local failures=0 failmsg="" child_pid="" attempts=0
+ _pulse_start_post_dispatch_housekeeping 7
+ child_pid=$(sed -n 's/.*Async post-dispatch housekeeping: launched pid=\([0-9][0-9]*\).*/\1/p' "$LOGFILE" 2>/dev/null | tail -n 1)
+ while [[ "$attempts" -lt 20 ]] && ! grep -q '^optional:coderabbit_review:' "$STAGE_LOG" 2>/dev/null; do
+ sleep 0.25
+ attempts=$((attempts + 1))
+ done
+ if [[ -n "$child_pid" ]]; then
+ kill -TERM "$child_pid" 2>/dev/null || true
+ fi
+ attempts=0
+ while [[ "$attempts" -lt 12 ]] && ! grep -q 'terminated by SIGTERM' "$LOGFILE" 2>/dev/null; do
+ sleep 1
+ attempts=$((attempts + 1))
+ done
+ if ! grep -q 'Async post-dispatch housekeeping: terminated by SIGTERM during coderabbit_review' "$LOGFILE" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | termination log missing: $(grep -o 'terminated by.*' "$LOGFILE" 2>/dev/null | tr '\n' ' ')"
+ fi
+ if grep -q 'Async post-dispatch housekeeping: complete' "$LOGFILE" 2>/dev/null; then
+ failures=$((failures + 1))
+ failmsg="${failmsg} | killed run reported complete"
+ fi
+ if [[ "$failures" -eq 0 ]]; then
+ print_result "post-dispatch housekeeping logs SIGTERM with active stage" 0
+ else
+ print_result "post-dispatch housekeeping logs SIGTERM with active stage" 1 "$failmsg"
+ fi
+ teardown_test_env
+ return 0
+}
+
main() {
test_sync_housekeeping_runs_all_stages
test_async_housekeeping_returns_before_slow_stage
test_housekeeping_lock_skips_live_duplicate
test_housekeeping_blocker_refresh_catchup
+ test_housekeeping_resumes_after_interrupted_run
+ test_housekeeping_signal_killed_stage_not_recorded
+ test_async_housekeeping_logs_termination
printf '\n============================================\n'
printf 'Tests run: %d\n' "$TESTS_RUN"
From b0208154fd428e6404a89bec27565f6a3b461270 Mon Sep 17 00:00:00 2001
From: alex-solovyev <1556417+alex-solovyev@users.noreply.github.com>
Date: Sat, 10 Oct 2026 19:03:50 -0400
Subject: [PATCH 4/5] GH#34289: fix(pulse): defer parent-close near reconcile
budget (GH#34289) (#34292)
* fix(pulse): defer parent-close when reconcile budget is below mutation reserve (GH#34289)
* refactor: extract parent close postcondition helper to stay under complexity limit
---
.../scripts/pulse-issue-reconcile-actions.sh | 52 +++++++++++++++----
.../tests/test-pulse-issue-reconcile.sh | 41 +++++++++++++++
2 files changed, 84 insertions(+), 9 deletions(-)
diff --git a/.agents/scripts/pulse-issue-reconcile-actions.sh b/.agents/scripts/pulse-issue-reconcile-actions.sh
index ab567ba65f..62527b6331 100644
--- a/.agents/scripts/pulse-issue-reconcile-actions.sh
+++ b/.agents/scripts/pulse-issue-reconcile-actions.sh
@@ -800,6 +800,43 @@ _pir_reopen_unstable_parent_close() {
return 0
}
+#######################################
+# Close a parent, verify the post-close snapshot, and compensate by reopening
+# when evidence changed. Args: slug parent_num live_body child_nums child_source
+# Returns: 0=closed and stable, 1=close failed or compensated
+#######################################
+_pir_close_parent_with_postcondition() {
+ local slug="$1" parent_num="$2" live_parent_body="$3" child_nums="$4" child_source="$5"
+ gh issue close "$parent_num" --repo "$slug" >/dev/null 2>&1 || return 1
+ if ! _pir_parent_close_postcondition_is_stable "$slug" "$parent_num" "$live_parent_body" \
+ "$child_nums" "$child_source"; then
+ if [[ "$_PIR_CPT_POST_CLOSE_REOPEN_ALLOWED" -eq 1 ]]; then
+ _pir_reopen_unstable_parent_close "$slug" "$parent_num" || \
+ echo "[pulse-wrapper] Reconcile parent-task: unable to compensate unstable close #${parent_num} in ${slug}" >>"${LOGFILE:-/dev/null}"
+ fi
+ return 1
+ fi
+ return 0
+}
+
+#######################################
+# Budget guard for multi-round-trip mutations (GH#34289). Reads the
+# single-pass locals _t2984_start_ts/_t2984_budget via dynamic scope. A parent
+# close spans several reads, the close, a stability check, an optional reopen
+# and a comment; starting one near the budget risks a mid-action stage kill.
+# Env: RECONCILE_MUTATION_RESERVE_SECS=reserve seconds (default 90)
+# Returns: 0=enough budget (or no budget active), 1=too little remaining
+#######################################
+_pir_budget_allows_mutation() {
+ local reserve="${RECONCILE_MUTATION_RESERVE_SECS:-90}"
+ [[ "$reserve" =~ ^[0-9]+$ ]] || reserve=90
+ local budget="${_t2984_budget:-0}" start="${_t2984_start_ts:-}"
+ [[ "$budget" =~ ^[0-9]+$ && "$start" =~ ^[0-9]+$ ]] || return 0
+ [[ "$budget" -gt 0 ]] || return 0
+ [[ $((budget - (SECONDS - start))) -ge "$reserve" ]] || return 1
+ return 0
+}
+
#######################################
# t2138 / t3544: extract per-parent close logic. Keeps
# reconcile_completed_parent_tasks under the 100-line shell-complexity
@@ -813,6 +850,10 @@ _try_close_parent_tracker() {
local live_child_nums="" live_child_source=""
local live_parent_revision=""
+ if ! _pir_budget_allows_mutation; then
+ echo "[pulse-wrapper] Reconcile parent-task: deferred: insufficient reconcile budget for parent close #${parent_num} in ${slug}" >>"${LOGFILE:-/dev/null}"
+ return 1
+ fi
_pir_verify_parent_children_closed "$slug" "$child_nums" || return 1
child_count="$_PIR_CPT_VERIFIED_CHILD_COUNT"
child_summary="$_PIR_CPT_VERIFIED_CHILD_SUMMARY"
@@ -883,15 +924,8 @@ _try_close_parent_tracker() {
return 1
fi
local close_basis="$_PIR_PARENT_TRUST_BASIS"
- gh issue close "$parent_num" --repo "$slug" >/dev/null 2>&1 || return 1
- if ! _pir_parent_close_postcondition_is_stable "$slug" "$parent_num" "$live_parent_body" \
- "$child_nums" "$child_source"; then
- if [[ "$_PIR_CPT_POST_CLOSE_REOPEN_ALLOWED" -eq 1 ]]; then
- _pir_reopen_unstable_parent_close "$slug" "$parent_num" || \
- echo "[pulse-wrapper] Reconcile parent-task: unable to compensate unstable close #${parent_num} in ${slug}" >>"${LOGFILE:-/dev/null}"
- fi
- return 1
- fi
+ _pir_close_parent_with_postcondition "$slug" "$parent_num" "$live_parent_body" \
+ "$child_nums" "$child_source" || return 1
gh_issue_comment "$parent_num" --repo "$slug" \
--body "## All declared child tasks completed — closing parent tracker
diff --git a/.agents/scripts/tests/test-pulse-issue-reconcile.sh b/.agents/scripts/tests/test-pulse-issue-reconcile.sh
index 6e4656a1fc..6506cde279 100644
--- a/.agents/scripts/tests/test-pulse-issue-reconcile.sh
+++ b/.agents/scripts/tests/test-pulse-issue-reconcile.sh
@@ -1772,6 +1772,46 @@ EOF
return 0
}
+# ---------------------------------------------------------------------------
+# GH#34289: a parent close must not start when the remaining reconcile budget
+# is below the mutation reserve (outer stage kill would land mid-action).
+# ---------------------------------------------------------------------------
+test_gh34289_parent_close_deferred_near_budget() {
+ local actions_sh="${SCRIPT_DIR}/../pulse-issue-reconcile-actions.sh"
+ local logfile=""
+ logfile=$(mktemp)
+ local result=""
+ result=$(ACTIONS_SH="$actions_sh" LOGFILE="$logfile" bash -c '
+ _PIR_SCRIPT_DIR="/nonexistent"
+ source "$ACTIONS_SH"
+ GH_CALLS=0
+ gh() { GH_CALLS=$((GH_CALLS + 1)); return 1; }
+ _t2984_budget=360
+ SECONDS=300
+ _t2984_start_ts=0
+ _try_close_parent_tracker owner/repo 42 43 graph "" && echo "near:ran" || echo "near:deferred"
+ printf "near-gh-calls:%s\n" "$GH_CALLS"
+ _t2984_start_ts=$SECONDS
+ _pir_budget_allows_mutation && echo "fresh:allowed" || echo "fresh:blocked"
+ _t2984_budget=0
+ _pir_budget_allows_mutation && echo "nobudget:allowed" || echo "nobudget:blocked"
+ ' 2>/dev/null)
+
+ local all_ok=1
+ printf '%s\n' "$result" | grep -qx 'near:deferred' || all_ok=0
+ printf '%s\n' "$result" | grep -qx 'near-gh-calls:0' || all_ok=0
+ printf '%s\n' "$result" | grep -qx 'fresh:allowed' || all_ok=0
+ printf '%s\n' "$result" | grep -qx 'nobudget:allowed' || all_ok=0
+ grep -q 'deferred: insufficient reconcile budget for parent close #42' "$logfile" 2>/dev/null || all_ok=0
+ rm -f "$logfile"
+ if [[ "$all_ok" -eq 1 ]]; then
+ _pass "parent close deferred (no gh calls, logged) when reconcile budget is nearly exhausted"
+ else
+ _fail "budget-deferral result mismatch: ${result}"
+ fi
+ return 0
+}
+
# ---------------------------------------------------------------------------
# Run all tests
# ---------------------------------------------------------------------------
@@ -1807,6 +1847,7 @@ test_gh32640_ciw_rsd_recurrent_file_size_debt_gate
test_available_feedback_worker_issue_not_assigned
test_feedback_backfill_uses_label_constants
test_pr_lookup_uncertainty_preserves_issue_state
+test_gh34289_parent_close_deferred_near_budget
echo ""
echo "Results: ${pass} passed, ${fail} failed"
From c49f66173761a8ed1e29509c03acc9db405bb735 Mon Sep 17 00:00:00 2001
From: Marcus Quinn <6428977+marcusquinn@users.noreply.github.com>
Date: Sun, 11 Oct 2026 00:04:14 +0100
Subject: [PATCH 5/5] GH#34293: docs: keep Qlty on organisation repos and skip
its out-of-minutes check
---
.agents/reference/ci-gate-policy.md | 5 ++++-
.agents/tools/code-review/qlty.md | 13 +++++++++++++
.agents/tools/code-review/setup.md | 6 +++++-
.agents/tools/wordpress/wp-plugin-new.md | 2 +-
4 files changed, 23 insertions(+), 3 deletions(-)
diff --git a/.agents/reference/ci-gate-policy.md b/.agents/reference/ci-gate-policy.md
index d91431efba..e3242595c4 100644
--- a/.agents/reference/ci-gate-policy.md
+++ b/.agents/reference/ci-gate-policy.md
@@ -163,7 +163,10 @@ run those directly too, without `-k` (they do not provide unittest filtering).
not rerun it, change code, file an issue or wait for it. Merge when the
other checks pass, and note the skipped check in the PR body. A qlty
failure with real findings still counts, and an out-of-credits failure on
- a public personal-account repository is unexpected: report it.
+ a public personal-account repository is unexpected: report it. The free
+ minutes return monthly, so keep the app installed (it is not a reason to
+ remove Qlty) and never make "qlty check" a required check on a limited
+ repository. Setup guidance: `tools/code-review/qlty.md`.
## Private repositories and public launch
diff --git a/.agents/tools/code-review/qlty.md b/.agents/tools/code-review/qlty.md
index 1efd73666b..f60c8ad1e8 100644
--- a/.agents/tools/code-review/qlty.md
+++ b/.agents/tools/code-review/qlty.md
@@ -28,9 +28,22 @@ tools:
- Commands: `bash .agents/scripts/qlty-cli.sh install|init|check|fmt --all|smells --all [ORG]`
- Storage: `~/.config/aidevops/api-keys` (600 permissions)
- Path: Ensure `~/.qlty/bin` in PATH
+- GitHub App on organisation or private repos: monthly minutes run out; an out-of-minutes "qlty check" is billing, not a finding. Keep the app and skip the check (`reference/ci-gate-policy.md` rule 14; see "GitHub App: monthly minutes")
+## GitHub App: monthly minutes
+
+Qlty's GitHub App posts a "qlty check" on pull requests. Its analysis minutes are limited on private repositories and on every organisation-owned repository, public or private; only public repositories on personal accounts are unlimited. The free minutes return each month, so early-month PRs get real findings and later ones may fail with "Qlty did not run because you are out of minutes".
+
+When setting up or managing repositories:
+
+- Keep the app installed for its free minutes. A red out-of-minutes check is not a reason to uninstall Qlty, remove it from a repository's docs, buy credits or ask the owner, unless the owner asks about cost.
+- Skip the out-of-minutes failure: don't rerun it, change code, file an issue or wait for it. Merge when the other checks pass and note the skipped check in the PR body. The merge helpers do this already (`gh-checks-wait-helper.sh` `apply_qlty_billing_policy`, `pulse-merge-required-checks.sh` `_pmrc_repo_has_qlty_credit_limit`).
+- Never make "qlty check" a required status check or ruleset check on a credit-limited repository: once the minutes run out it would block every merge for the rest of the month.
+- A qlty check with real findings still counts, and an out-of-minutes failure on a public personal-account repository is unexpected: report it.
+- CI jobs that run the qlty CLI in a repository's own Actions (such as the aidevops Qlty Regression Gate and Qlty Smell Threshold) are separate from the app and do not use its minutes; their failures are real.
+
## Credentials
Three types, selected in priority order:
diff --git a/.agents/tools/code-review/setup.md b/.agents/tools/code-review/setup.md
index fb0d5288be..43a57ec534 100644
--- a/.agents/tools/code-review/setup.md
+++ b/.agents/tools/code-review/setup.md
@@ -24,8 +24,11 @@ Setup time: ~5 min per platform via GitHub OAuth. Targets: CodeFactor A+, Codacy
| CodeFactor | → add repo → enable GitHub Checks | A-F grade, cyclomatic complexity, technical debt, trends | — |
| Codacy | → import repo | Security scanning, quality metrics, test coverage, standards | `.codacy.yml` |
| SonarCloud | → create org → import project → add `SONAR_TOKEN` GitHub secret | Security hotspots, bugs, code smells, duplication, quality gate | `sonar-project.properties` |
+| Qlty | Install the Qlty GitHub App for the account or organisation | Multi-linter findings and smells on PRs; monthly minutes limited on organisation and private repos | `.qlty/qlty.toml` (optional) |
-Deep dives: `coderabbit.md`, `codacy.md`, `tools.md`, `.agents/scripts/sonarcloud-cli.sh`
+Deep dives: `coderabbit.md`, `codacy.md`, `qlty.md`, `tools.md`, `.agents/scripts/sonarcloud-cli.sh`
+
+Qlty's out-of-minutes failure is billing: skip it, keep the app, and never make "qlty check" a required check on a credit-limited repository (`reference/ci-gate-policy.md` rule 14).
@@ -48,3 +51,4 @@ Replace `{owner}/{repo}` with your repository slug.
| CodeRabbit not reviewing | Ensure repo is connected, app permissions granted, and PR triggers enabled. |
| CodeFactor not updating | Check repo connection, webhook/GitHub Checks status, and repo authorization. |
| Codacy analysis issues | Check `.codacy.yml`, confirm import succeeded, and verify file types are supported. |
+| "qlty check" fails: "out of minutes" | Billing, not a finding. Skip it and merge on the other checks; keep the app; never make it required (`qlty.md`, `reference/ci-gate-policy.md` rule 14). |
diff --git a/.agents/tools/wordpress/wp-plugin-new.md b/.agents/tools/wordpress/wp-plugin-new.md
index dc3963e7e5..f1dfc09f51 100644
--- a/.agents/tools/wordpress/wp-plugin-new.md
+++ b/.agents/tools/wordpress/wp-plugin-new.md
@@ -64,7 +64,7 @@ For **public repos**:
- **Codacy**: inject `CODACY_API_TOKEN` securely (`aidevops secret set CODACY_API_TOKEN`), then run `quality`. It adds the repository with API v3 `POST /repositories` (`provider: gh`, `repositoryFullPath: OWNER/SLUG`), tolerates an already-added 409, fetches `GET /organizations/gh/OWNER/repositories/SLUG`, and restores the badge from `data.badges.grade`, never a copied project ID. Missing token, access or pending badge is reported; rerun after recovery.
- **SonarCloud**: inject `SONAR_TOKEN`, optionally `SONAR_ORGANIZATION` / `SONAR_PROJECT_KEY` when they differ from the GitHub owner / `OWNER_SLUG`; verify they match this plugin's scanner configuration, not another repo's environment. The helper provisions a public project when missing and only restores its badge after a quality-gate measure exists. **Project creation and measures do not prove GitHub binding**: the current published Web API exposes no GitHub import endpoint. The sole SonarCloud human fallback is: [Import a project](https://sonarcloud.io/projects/create) — an org admin imports this GitHub repo and configures its analysis method, then the AI reruns `quality`. For starter v1.0.7+ with `.github/workflows/sonarcloud.yml`, keep **Automatic Analysis off** and configure the repository `SONAR_TOKEN` securely for the Actions scanner (see the copied `DEVELOPMENT.md` "Services setup"); older copies without the scanner may enable Automatic Analysis. Never run both methods or claim binding from project existence alone.
-- **Apps**: Codacy and CodeFactor must appear on the first PR; also verify CodeRabbit, Qlty and Socket. A check/status proves integration visibility, not a passing result. The helper reports each missing service; the AI diagnoses existing app selection and configuration, and an app admin grants repository access only if necessary. Do not silently accept missing public Codacy/CodeFactor checks. The helper restores CodeFactor's badge only after its public endpoint serves a grade SVG; restore a latest-release badge only after an actual release exists.
+- **Apps**: Codacy and CodeFactor must appear on the first PR; also verify CodeRabbit, Qlty and Socket. A check/status proves integration visibility, not a passing result; a "qlty check" failing with "out of minutes" still proves Qlty is connected, and is skipped, never a reason to remove Qlty (`reference/ci-gate-policy.md` rule 14, `tools/code-review/qlty.md`). The helper reports each missing service; the AI diagnoses existing app selection and configuration, and an app admin grants repository access only if necessary. Do not silently accept missing public Codacy/CodeFactor checks. The helper restores CodeFactor's badge only after its public endpoint serves a grade SVG; restore a latest-release badge only after an actual release exists.
- **Full review**: after the plugin is its own, create/deduplicate the **Code Audit Routines** issue using the signed framework issue wrapper and dashboard pattern in `scripts/stats-quality-sweep-issues.sh` (`_ensure_quality_issue`). Include repo scripts, scope and verification; mention `@coderabbitai` to request a **full codebase review**, not merely the PR diff (`tools/code-review/coderabbit.md`, "Daily Code Quality Review"). Confirm the review request was accepted; report app/plan limitations rather than inventing a review.
For **private repos**, run local Composer/PHPCS/PHPStan, ShellCheck and Docker checks and generate metrics now. Defer hosted onboarding by default: Codacy and CodeFactor free tiers are public-only; the current starter documents limited private SonarCloud free-plan capacity (50,000 organization-wide lines), so verify current organization limits before using an existing entitlement. Do not assume all hosted services are free for private code. CodeRabbit, Qlty and Socket work only when their installed app's repository selection and plan allow private repos; verify actual first-PR checks/statuses, not assumed free access. Do not purchase plans or make the repo public to fix missing checks. At owner-approved public launch, repeat this checklist (`wp-plugin-release.md`).