#!/bin/bash

## Copyright (C) 2026 - 2026 ENCRYPTED SUPPORT LLC <adrelanos@whonix.org>
## See the file COPYING for copying conditions.

## AI-Assisted

## Explain a build step that DIED rather than failed.
##
## THE PROBLEM THIS SOLVES: the build lane has repeatedly ended with the build
## step never concluding, no 'ERROR detected in script' marker anywhere, and the
## log simply stopping mid-trace. That shape says the process was killed, not
## that the build failed -- but nothing recorded WHY, so each occurrence cost a
## full re-run to learn nothing new.
##
## Runs on failure only, and never fails the job itself: its output is evidence,
## and an evidence collector that can itself fail replaces one unexplained
## failure with two. Every command is therefore guarded and the script ends in
## an explicit success.
##
## The decisive signal is docker's own OOMKilled flag on the build container.
## Kernel and cgroup counters are collected as corroboration.
##
## style-ok: no-has -- runner-diagnostic script; does not source derivative-maker
## helper-scripts (same class as ci/free-disk).

set -o errexit
set -o nounset
set -o pipefail
set -o errtrace
shopt -s inherit_errexit
shopt -s shift_verbose
export LC_ALL=C

section() {
   ## No blank-line separator (R-042): the marker line is the separator.
   printf '%s\n' "===== $* ====="
}

section "docker containers (OOMKilled is the decisive field)"
if command -v docker >/dev/null 2>&1; then
   sudo --non-interactive -- docker ps --all \
      --format '{{.Names}} {{.Status}} {{.Image}}' 2>/dev/null || true
   container_ids="$( sudo --non-interactive -- docker ps --all --quiet 2>/dev/null || true )"
   while IFS= read -r container_id; do
      [ -n "${container_id}" ] || continue
      sudo --non-interactive -- docker inspect \
         --format '{{.Name}} OOMKilled={{.State.OOMKilled}} ExitCode={{.State.ExitCode}} Error={{.State.Error}}' \
         -- "${container_id}" 2>/dev/null || true
   done <<< "${container_ids}"
else
   printf '%s\n' "docker not available"
fi

section "kernel OOM / kill records"
## Two sources: the ring buffer and the journal. Either may be empty depending on
## how the runner is configured, so both are tried.
dmesg_oom="$( sudo --non-interactive -- dmesg 2>/dev/null \
   | grep --extended-regexp --ignore-case 'out of memory|oom-kill|killed process' \
   | tail -n 20 || true )"
if [ -n "${dmesg_oom}" ]; then
   printf '%s\n' "${dmesg_oom}"
else
   printf '%s\n' "no OOM records in dmesg"
fi

journal_oom="$( sudo --non-interactive -- journalctl --dmesg --no-pager 2>/dev/null \
   | grep --extended-regexp --ignore-case 'out of memory|oom-kill|killed process' \
   | tail -n 20 || true )"
if [ -n "${journal_oom}" ]; then
   printf '%s\n' "${journal_oom}"
else
   printf '%s\n' "no OOM records in the journal"
fi

section "memory"
free --human 2>/dev/null || true
printf '%s\n' "--- /proc/meminfo (selected) ---"
grep --extended-regexp '^(MemTotal|MemAvailable|SwapTotal|SwapFree|Committed_AS)' \
   -- /proc/meminfo 2>/dev/null || true

section "disk"
df --human-readable -- / /home /var/lib/docker 2>/dev/null || true

section "load"
uptime 2>/dev/null || true

## Always succeed: this is diagnosis, not a gate.
exit 0
