diff --git a/.github/scripts/monitor_slurm_job.sh b/.github/scripts/monitor_slurm_job.sh index 057cd54cad..83d40570e8 100755 --- a/.github/scripts/monitor_slurm_job.sh +++ b/.github/scripts/monitor_slurm_job.sh @@ -97,7 +97,7 @@ is_terminal_state() { # Optionally bound how long a job may sit un-started in the queue. On the # preemptible Phoenix 'embers' QOS a job routinely stays PENDING for hours and -# needs most of the job-level `timeout-minutes` (480m) window to backfill onto a +# needs most of the job-level `timeout-minutes` (1380m) window to backfill onto a # free node; that job timeout is the real backstop. Default to 0 (wait # indefinitely, up to the job timeout) so ordinary queue pressure does not turn # otherwise-healthy jobs into red CI. Set SLURM_MAX_QUEUE_SECONDS>0 to opt into diff --git a/.github/scripts/run_parallel_benchmarks.sh b/.github/scripts/run_parallel_benchmarks.sh index 2f262692d6..e2565d2983 100755 --- a/.github/scripts/run_parallel_benchmarks.sh +++ b/.github/scripts/run_parallel_benchmarks.sh @@ -86,7 +86,7 @@ else # On Phoenix 'embers' a long benchmark job can be preempted (PreemptMode=CANCEL, # so it is killed rather than requeued). On preemption (run_monitored exit 76) # resubmit a fresh job in the same tree and re-monitor, bounded by - # MAX_PREEMPT_RESUBMITS (the 480m job timeout is the real backstop). Note: a + # MAX_PREEMPT_RESUBMITS (the 1380m job timeout is the real backstop). Note: a # resubmitted job no longer overlaps its counterpart, slightly reducing # same-load fairness -- still preferable to failing the run on an infra preempt. : "${MAX_PREEMPT_RESUBMITS:=10}" diff --git a/.github/scripts/submit-slurm-job.sh b/.github/scripts/submit-slurm-job.sh index fca1ff8a09..1db3bb5306 100755 --- a/.github/scripts/submit-slurm-job.sh +++ b/.github/scripts/submit-slurm-job.sh @@ -266,7 +266,7 @@ EOT # preempted job is killed outright and `--requeue` never restarts it. When the # monitor reports preemption (exit 76), submit a fresh job and monitor again. # Bounded by MAX_PREEMPT_RESUBMITS as a runaway guard; the job-level -# `timeout-minutes` (480m) remains the real backstop. +# `timeout-minutes` (1380m) remains the real backstop. : "${MAX_PREEMPT_RESUBMITS:=10}" # Node faults get a much tighter bound than preemption: preemption is routine on # 'embers' and says nothing about the node, whereas hitting a second unusable diff --git a/.github/workflows/bench.yml b/.github/workflows/bench.yml index 235228c925..649b32684b 100644 --- a/.github/workflows/bench.yml +++ b/.github/workflows/bench.yml @@ -103,7 +103,8 @@ jobs: runs-on: group: ${{ matrix.group }} labels: ${{ matrix.labels }} - timeout-minutes: 480 + # Includes SLURM queue wait (Phoenix embers is priority 0, often hours). Kept under 24h, when GITHUB_TOKEN expires. + timeout-minutes: 1380 steps: - name: Clean stale output files run: rm -f *.out diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 1119bed64b..89645bf3d1 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -377,7 +377,8 @@ jobs: # cpe/25.03 introduced an IPA SIGSEGV in CCE 19.0.0). Allow Frontier to # fail without blocking PR merges; Phoenix remains a hard gate. continue-on-error: ${{ matrix.runner == 'frontier' }} - timeout-minutes: 480 + # Includes SLURM queue wait (Phoenix embers is priority 0, often hours). Kept under 24h, when GITHUB_TOKEN expires. + timeout-minutes: 1380 strategy: matrix: include: @@ -555,7 +556,8 @@ jobs: needs: [lint-gate, file-changes] # Frontier is non-blocking for the same reason as the self job above. continue-on-error: ${{ matrix.runner == 'frontier' }} - timeout-minutes: 480 + # Includes SLURM queue wait (Phoenix embers is priority 0, often hours). Kept under 24h, when GITHUB_TOKEN expires. + timeout-minutes: 1380 strategy: matrix: include: