Compare commits

..

1 Commits

Author SHA1 Message Date
CanbiZ (MickLesk)
30f855f852 core: Improve GPU detection and mapping logic
Refactor GPU detection logic to map DRI nodes to their owning GPUs based on vendor IDs. Update messages for detected Intel and AMD GPUs.
2026-07-20 11:05:51 +02:00
22 changed files with 1191 additions and 999 deletions

View File

@@ -511,32 +511,8 @@ Exercise vigilance regarding copycat or coat-tailing sites that seek to exploit
- #### 🐞 Bug Fixes
- RomM: use backup helpers in update / clear folder [@MickLesk](https://github.com/MickLesk) ([#15915](https://github.com/community-scripts/ProxmoxVE/pull/15915))
- fix: vikunja: asset selection [@CrazyWolf13](https://github.com/CrazyWolf13) ([#15929](https://github.com/community-scripts/ProxmoxVE/pull/15929))
- Zammad : bind Elasticsearch to 127.0.0.1 [@MickLesk](https://github.com/MickLesk) ([#15909](https://github.com/community-scripts/ProxmoxVE/pull/15909))
- Omada: fix package version extraction [@MickLesk](https://github.com/MickLesk) ([#15908](https://github.com/community-scripts/ProxmoxVE/pull/15908))
- fix(wanderer): use PocketBase-relative plugin symlink in unprivileged LXC [@michelroegl-brunner](https://github.com/michelroegl-brunner) ([#15911](https://github.com/community-scripts/ProxmoxVE/pull/15911))
- #### ✨ New Features
- AFFiNE: Bump version to v0.27.2 [@MickLesk](https://github.com/MickLesk) ([#15930](https://github.com/community-scripts/ProxmoxVE/pull/15930))
- #### 💥 Breaking Changes
- Gotify: Migration to v3 [@MickLesk](https://github.com/MickLesk) ([#15912](https://github.com/community-scripts/ProxmoxVE/pull/15912))
### 💾 Core
- #### ✨ New Features
- core: refactor to single-reporter telemetry and better error_handling [@MickLesk](https://github.com/MickLesk) ([#15933](https://github.com/community-scripts/ProxmoxVE/pull/15933))
- tools.func: add support for extracting 7z archives [@MickLesk](https://github.com/MickLesk) ([#15919](https://github.com/community-scripts/ProxmoxVE/pull/15919))
- Meilisearch : use dumpless Meilisearch upgrades [@MickLesk](https://github.com/MickLesk) ([#15921](https://github.com/community-scripts/ProxmoxVE/pull/15921))
- #### 🔧 Refactor
- core: Improve GPU detection and mapping logic [@MickLesk](https://github.com/MickLesk) ([#15918](https://github.com/community-scripts/ProxmoxVE/pull/15918))
## 2026-07-19
### 🚀 Updated Scripts

View File

@@ -30,7 +30,7 @@ function update_script() {
exit
fi
RELEASE="v0.27.2"
RELEASE="v0.27.0"
if check_for_gh_release "affine_app" "toeverything/AFFiNE" "${RELEASE}" "each release is tested individually before the version is updated. Please do not open issues for this"; then
msg_info "Stopping Services"
systemctl stop affine-web affine-worker
@@ -38,19 +38,10 @@ function update_script() {
ensure_dependencies cmake
create_backup /opt/affine/.env /root/.affine/config /root/.affine/storage
create_backup /root/.affine/config /root/.affine/storage
CLEAN_INSTALL=1 fetch_and_deploy_gh_release "affine_app" "toeverything/AFFiNE" "tarball" "${RELEASE}" "/opt/affine"
# Restore BEFORE the build: CLEAN_INSTALL wiped /opt/affine including .env,
# and the build below sources it
restore_backup
if [[ ! -f /opt/affine/.env ]]; then
msg_error "/opt/affine/.env is missing (lost by an earlier update). Recreate it before retrying — see the AFFiNE install script for the expected variables."
exit 1
fi
msg_info "Rebuilding Application (Patience ~25 mins, don't close the console!)"
cd /opt/affine
source /root/.profile
@@ -120,6 +111,8 @@ TURBO
set -a && source /opt/affine/.env && set +a
$STD node ./scripts/self-host-predeploy.js
restore_backup
msg_info "Starting Services"
systemctl start affine-web affine-worker
msg_ok "Started Services"

View File

@@ -36,30 +36,6 @@ function update_script() {
fetch_and_deploy_gh_release "gotify" "gotify/server" "prebuild" "latest" "/opt/gotify" "gotify-linux-$(arch_resolve).zip"
chmod +x /opt/gotify/gotify-linux-$(arch_resolve)
if [[ ! -f /opt/gotify/gotify-server.env ]]; then
gotify_old_config=""
for f in /opt/gotify/config.yml /etc/gotify/config.yml; do
[[ -f "$f" ]] && gotify_old_config="$f" && break
done
if [[ -n "$gotify_old_config" ]]; then
msg_info "Migrating ${gotify_old_config} to env format (Gotify 3.x)"
if /opt/gotify/gotify-linux-$(arch_resolve) migrate-config "$gotify_old_config" >/opt/gotify/gotify-server.env 2>/dev/null; then
mv "$gotify_old_config" "${gotify_old_config}.bak"
msg_ok "Migrated config to /opt/gotify/gotify-server.env (backup: ${gotify_old_config}.bak)"
else
rm -f /opt/gotify/gotify-server.env
msg_warn "Config migration failed — left ${gotify_old_config} in place, review manually"
fi
fi
fi
if ! grep -qE '^ExecStart=.* serve' /etc/systemd/system/gotify.service 2>/dev/null; then
msg_info "Migrating service to serve subcommand (Gotify 3.x)"
sed -i -E 's|^(ExecStart=/opt/gotify/.*gotify-linux-[^ ]+)$|\1 serve|' /etc/systemd/system/gotify.service
systemctl daemon-reload
msg_ok "Migrated service to serve subcommand"
fi
msg_info "Starting Service"
systemctl start gotify
msg_ok "Started Service"

View File

@@ -42,9 +42,6 @@ function update_script() {
CLEAN_INSTALL=1 fetch_and_deploy_gh_release "nametag" "mattogodoy/nametag" "tarball" "latest" "/opt/nametag"
# Restore .env BEFORE the build: CLEAN_INSTALL wiped it and the build sources it
cp /opt/nametag.env.bak /opt/nametag/.env
msg_info "Rebuilding Application"
cd /opt/nametag
$STD npm ci

View File

@@ -43,7 +43,7 @@ function update_script() {
grep -o 'https://static\.tp-link\.com/upload/software/[^"]*linux_x64[^"]*\.deb' |
head -n1)
OMADA_PKG=$(basename "${OMADA_URL}")
VERSION=$(sed -n 's/.*_v\([0-9.]*\)_linux.*/\1/p' <<<"${OMADA_PKG}")
VERSION=$(sed -n 's/.*_v\([0-9.]*\)_.*_\([0-9]\{14\}\)\.deb$/\1-\2/p' <<<"${OMADA_PKG}")
CURRENT_VERSION=$(cat $HOME/.omada 2>/dev/null || echo "0")

View File

@@ -41,12 +41,9 @@ function update_script() {
CLEAN_INSTALL=1 fetch_and_deploy_gh_release "postiz" "gitroomhq/postiz-app" "tarball"
# Restore BEFORE the build: CLEAN_INSTALL wiped /opt/postiz including .env,
# and the build below sources it
restore_backup
msg_info "Building Application"
cd /opt/postiz
cp /opt/postiz_env.bak /opt/postiz/.env
set -a && source /opt/postiz/.env && set +a
export NODE_OPTIONS="--max-old-space-size=4096"
$STD pnpm install
@@ -60,6 +57,7 @@ function update_script() {
msg_ok "Ran Database Migrations"
mkdir -p /opt/postiz/uploads
restore_backup
msg_info "Starting Services"
systemctl start postiz-backend postiz-frontend postiz-orchestrator

View File

@@ -37,29 +37,18 @@ function update_script() {
systemctl stop romm-backend romm-worker romm-scheduler romm-watcher
msg_ok "Stopped Services"
create_backup /opt/romm/.env
BACKUP_DIR=/opt/romm-players.backup create_backup \
/opt/romm/frontend/dist/assets/emulatorjs \
/opt/romm/frontend/dist/assets/ruffle
msg_info "Backing up configuration"
cp /opt/romm/.env /opt/romm/.env.backup
msg_ok "Backed up configuration"
CLEAN_INSTALL=1 fetch_and_deploy_gh_release "romm" "rommapp/romm" "tarball" "latest" "/opt/romm"
restore_backup
fetch_and_deploy_gh_release "romm" "rommapp/romm" "tarball" "latest" "/opt/romm"
msg_info "Updating ROMM"
cp /opt/romm/.env.backup /opt/romm/.env
cd /opt/romm
$STD uv sync --all-extras
cd /opt/romm/backend
$STD uv run alembic upgrade head
if [[ -f /opt/romm/backend/utils/rom_patcher/package.json ]]; then
cd /opt/romm/backend/utils/rom_patcher
$STD npm install --ignore-scripts --no-audit --no-fund
if [[ -d node_modules/rom-patcher/rom-patcher-js ]]; then
rm -rf rom-patcher-js
cp -r node_modules/rom-patcher/rom-patcher-js ./rom-patcher-js
fi
rm -rf node_modules
fi
cd /opt/romm/frontend
$STD npm install
$STD npm run build
@@ -84,12 +73,6 @@ function update_script() {
msg_ok "Started Services"
msg_ok "Updated successfully"
fi
if check_for_gh_release "EmulatorJS" "EmulatorJS/EmulatorJS" "v4.2.3"; then
CLEAN_INSTALL=1 fetch_and_deploy_gh_release "EmulatorJS" "EmulatorJS/EmulatorJS" "prebuild" "v4.2.3" "/opt/romm/frontend/dist/assets/emulatorjs" "4.2.3.7z"
systemctl restart romm-backend romm-worker romm-scheduler romm-watcher
msg_ok "Updated EmulatorJS successfully"
fi
exit
}

View File

@@ -65,7 +65,7 @@ function update_script() {
systemctl stop vikunja
msg_ok "Stopped Service"
fetch_and_deploy_gh_release "vikunja" "go-vikunja/vikunja" "binary" "latest" "" "vikunja-*-$(arch_resolve "x86_64" "aarch64").deb"
fetch_and_deploy_gh_release "vikunja" "go-vikunja/vikunja" "binary"
$STD systemctl daemon-reload
msg_info "Starting Service"

View File

@@ -31,7 +31,7 @@ PG_DB_NAME="affine" PG_DB_USER="affine" setup_postgresql_db
NODE_VERSION="22" setup_nodejs
setup_rust
fetch_and_deploy_gh_release "affine_app" "toeverything/AFFiNE" "tarball" "v0.27.2" "/opt/affine"
fetch_and_deploy_gh_release "affine_app" "toeverything/AFFiNE" "tarball" "v0.27.0" "/opt/affine"
msg_info "Setting up Directories"
rm -rf /root/.affine

View File

@@ -27,7 +27,7 @@ After=network.target
Type=simple
User=root
WorkingDirectory=/opt/gotify
ExecStart=/opt/gotify/gotify-linux-$(arch_resolve) serve
ExecStart=/opt/gotify/./gotify-linux-$(arch_resolve)
Restart=always
RestartSec=3

View File

@@ -42,7 +42,7 @@ OMADA_PKG=$(basename "${OMADA_URL}")
curl_download "${OMADA_PKG}" "${OMADA_URL}"
$STD dpkg -i "${OMADA_PKG}"
rm -rf "${OMADA_PKG}"
VERSION=$(sed -n 's/.*_v\([0-9.]*\)_linux.*/\1/p' <<<"${OMADA_PKG}")
VERSION=$(sed -n 's/.*_v\([0-9.]*\)_.*_\([0-9]\{14\}\)\.deb$/\1-\2/p' <<<"${OMADA_PKG}")
echo "${VERSION}" >$HOME/.omada
msg_ok "Installed Omada Controller"

View File

@@ -134,8 +134,6 @@ else
fi
fetch_and_deploy_gh_release "romm" "rommapp/romm" "tarball"
fetch_and_deploy_gh_release "ruffle" "ruffle-rs/ruffle" "prebuild" "latest" "/opt/romm/frontend/dist/assets/ruffle" "ruffle-*-web-selfhosted.zip"
fetch_and_deploy_gh_release "EmulatorJS" "EmulatorJS/EmulatorJS" "prebuild" "v4.2.3" "/opt/romm/frontend/dist/assets/emulatorjs" "4.2.3.7z"
msg_info "Creating environment file"
sed -i 's/^supervised no/supervised systemd/' /etc/redis/redis.conf
@@ -161,9 +159,6 @@ ROMM_AUTH_SECRET_KEY=$AUTH_SECRET_KEY
DISABLE_DOWNLOAD_ENDPOINT_AUTH=false
DISABLE_CSRF_PROTECTION=false
SCREENSCRAPER_DEV_ID=
SCREENSCRAPER_DEV_PASSWORD=
ENABLE_RESCAN_ON_FILESYSTEM_CHANGE=true
RESCAN_ON_FILESYSTEM_CHANGE_DELAY=5
@@ -186,18 +181,6 @@ cd /opt/romm/backend
$STD uv run alembic upgrade head
msg_ok "Set up RomM Backend"
if [[ -f /opt/romm/backend/utils/rom_patcher/package.json ]]; then
msg_info "Building ROM Patcher helper"
cd /opt/romm/backend/utils/rom_patcher
$STD npm install --ignore-scripts --no-audit --no-fund
if [[ -d node_modules/rom-patcher/rom-patcher-js ]]; then
rm -rf rom-patcher-js
cp -r node_modules/rom-patcher/rom-patcher-js ./rom-patcher-js
fi
rm -rf node_modules
msg_ok "Built ROM Patcher helper"
fi
msg_info "Setting up RomM Frontend"
cd /opt/romm/frontend
$STD npm install

View File

@@ -13,7 +13,7 @@ setting_up_container
network_check
update_os
fetch_and_deploy_gh_release "vikunja" "go-vikunja/vikunja" "binary" "latest" "" "vikunja-*-$(arch_resolve "x86_64" "aarch64").deb"
fetch_and_deploy_gh_release "vikunja" "go-vikunja/vikunja" "binary"
msg_info "Setting up Vikunja"
sed -i 's|^# \(service:\)|\1|' /etc/vikunja/config.yml

View File

@@ -31,10 +31,7 @@ $STD apt install -y elasticsearch
sed -i 's/^#\{0,2\} *-Xms[0-9]*g.*/-Xms2g/' /etc/elasticsearch/jvm.options
sed -i 's/^#\{0,2\} *-Xmx[0-9]*g.*/-Xmx2g/' /etc/elasticsearch/jvm.options
cat <<EOF >/etc/elasticsearch/elasticsearch.yml
path.data: /var/lib/elasticsearch
path.logs: /var/log/elasticsearch
discovery.type: single-node
network.host: 127.0.0.1
xpack.security.enabled: false
bootstrap.memory_lock: false
EOF
@@ -43,7 +40,7 @@ systemctl daemon-reload
systemctl enable -q elasticsearch
systemctl restart -q elasticsearch
for i in $(seq 1 30); do
if curl -s http://127.0.0.1:9200 >/dev/null 2>&1; then
if curl -s http://localhost:9200 >/dev/null 2>&1; then
break
fi
sleep 2
@@ -58,7 +55,7 @@ setup_deb822_repo \
"$(get_os_info version_id)" \
"main"
$STD apt install -y zammad
$STD zammad run rails r "Setting.set('es_url', 'http://127.0.0.1:9200')"
$STD zammad run rails r "Setting.set('es_url', 'http://localhost:9200')"
$STD zammad run rake zammad:searchindex:rebuild
msg_ok "Installed Zammad"

View File

@@ -6,10 +6,6 @@
if ! command -v curl >/dev/null 2>&1; then
apk update && apk add curl >/dev/null 2>&1
fi
# Mark container context BEFORE error handling starts: error_handler/on_exit
# must write local failure artifacts instead of talking to the telemetry API
# (the host is the single telemetry reporter).
export TELEMETRY_CONTEXT="container"
source <(curl -fsSL https://raw.githubusercontent.com/community-scripts/ProxmoxVE/main/misc/core.func)
source <(curl -fsSL https://raw.githubusercontent.com/community-scripts/ProxmoxVE/main/misc/error_handler.func)
load_functions
@@ -45,7 +41,7 @@ post_progress_to_api() {
curl -fsS -m 5 -X POST "https://telemetry.community-scripts.org/telemetry" \
-H "Content-Type: application/json" \
-d "{\"random_id\":\"${RANDOM_UUID}\",\"execution_id\":\"${EXECUTION_ID:-${RANDOM_UUID}}\",\"type\":\"lxc\",\"nsapp\":\"${app:-unknown}\",\"status\":\"${progress_status}\",\"platform\":\"${TELEMETRY_PLATFORM:-}\",\"repo_source\":\"${REPO_SOURCE:-}\",\"repo_slug\":\"${REPO_SLUG:-}\"}" &>/dev/null || true
-d "{\"random_id\":\"${RANDOM_UUID}\",\"execution_id\":\"${EXECUTION_ID:-${RANDOM_UUID}}\",\"type\":\"lxc\",\"nsapp\":\"${app:-unknown}\",\"status\":\"${progress_status}\"}" &>/dev/null || true
}
# This function enables IPv6 if it's not disabled and sets verbose mode

File diff suppressed because it is too large Load Diff

View File

@@ -4072,11 +4072,6 @@ build_container() {
export RANDOM_UUID="$RANDOM_UUID"
export EXECUTION_ID="$EXECUTION_ID"
export SESSION_ID="$SESSION_ID"
# Repo attribution + platform for container-side progress pings (the
# container inherits the host's detection instead of re-detecting)
export REPO_SOURCE="${REPO_SOURCE:-}"
export REPO_SLUG="${REPO_SLUG:-}"
export TELEMETRY_PLATFORM="pve"
export CACHER="$APT_CACHER"
export CACHER_IP="$APT_CACHER_IP"
if [[ -n "${HTTP_PROXY:-}" ]]; then
@@ -4900,16 +4895,6 @@ EOF
# Point INSTALL_LOG to combined log so get_full_log() finds it
INSTALL_LOG="$combined_log"
fi
# Pull the structured error capture (.errinfo) from the container.
# It contains EXACTLY the output of the command that failed (written by
# silent()/error_handler inside the container) and is the primary source
# for the telemetry error trace - instead of a generic log tail.
local host_errinfo="/tmp/.errinfo-${SESSION_ID}"
if timeout 8 pct pull "$CTID" "/root/.install-${SESSION_ID}.log.errinfo" "$host_errinfo" 2>/dev/null && [[ -s "$host_errinfo" ]]; then
TELEMETRY_ERRINFO="$host_errinfo"
export TELEMETRY_ERRINFO
fi
fi
# Defense-in-depth: Ensure error handling stays disabled during recovery.
@@ -5185,8 +5170,6 @@ EOF
echo -e " Verbose: ${GN}enabled${CL}"
echo ""
msg_info "Restarting installation..."
# New telemetry execution for the retry (previous one keeps its "failed")
declare -f telemetry_new_attempt &>/dev/null && telemetry_new_attempt
# Re-run build_container
build_container
return $?
@@ -5226,10 +5209,6 @@ EOF
echo ""
msg_info "Re-running installation script..."
# New telemetry execution for the in-place retry
declare -f telemetry_new_attempt &>/dev/null && telemetry_new_attempt
declare -f post_to_api &>/dev/null && post_to_api 2>/dev/null || true
# Re-run install script in existing container (don't destroy/recreate)
set +Eeuo pipefail
trap - ERR
@@ -5293,7 +5272,6 @@ EOF
echo -e " Verbose: ${GN}enabled${CL}"
echo ""
msg_info "Restarting installation..."
declare -f telemetry_new_attempt &>/dev/null && telemetry_new_attempt
build_container
return $?
fi
@@ -5323,7 +5301,6 @@ EOF
echo -e " Verbose: ${GN}enabled${CL}"
echo ""
msg_info "Restarting installation..."
declare -f telemetry_new_attempt &>/dev/null && telemetry_new_attempt
build_container
return $?
fi
@@ -5348,7 +5325,6 @@ EOF
echo -e " Verbose: ${GN}enabled${CL}"
echo ""
msg_info "Restarting installation..."
declare -f telemetry_new_attempt &>/dev/null && telemetry_new_attempt
build_container
return $?
fi
@@ -7023,16 +6999,6 @@ ensure_log_on_host() {
rm -f "$temp_log"
fi
fi
# Also pull the structured error capture (.errinfo) so the telemetry error
# trace shows the failing command's exact output (signal-exit paths reach
# this via on_exit before/instead of the recovery flow)
if [[ -z "${TELEMETRY_ERRINFO:-}" || ! -s "${TELEMETRY_ERRINFO:-}" ]]; then
local host_errinfo="/tmp/.errinfo-${SESSION_ID}"
if timeout 8 pct pull "$CTID" "/root/.install-${SESSION_ID}.log.errinfo" "$host_errinfo" 2>/dev/null && [[ -s "$host_errinfo" ]]; then
TELEMETRY_ERRINFO="$host_errinfo"
export TELEMETRY_ERRINFO
fi
fi
if [[ -s "$combined_log" ]]; then
INSTALL_LOG="$combined_log"
fi

View File

@@ -546,12 +546,6 @@ silent() {
set +Eeuo pipefail
trap - ERR
# Byte offset BEFORE the command runs - everything the log grows by is
# exactly this command's output (used for the .errinfo telemetry capture)
local start_bytes=0
[[ -f "$logfile" ]] && start_bytes=$(stat -c%s "$logfile" 2>/dev/null || echo 0)
[[ ! "$start_bytes" =~ ^[0-9]+$ ]] && start_bytes=0
"$@" >>"$logfile" 2>&1
local rc=$?
@@ -573,43 +567,12 @@ silent() {
export _SILENT_FAILED_LINE="$caller_line"
export _SILENT_FAILED_LOG="$logfile"
# ── Structured error capture (.errinfo) for telemetry ──
# Extract exactly THIS command's output (from the recorded byte offset),
# strip ANSI/progress noise, keep the last 60 lines. The host builds the
# telemetry error trace from this file (pulled from the container on
# failure). Self-contained - containers don't source api.func.
{
local flat_cmd
flat_cmd=$(printf '%s' "$cmd" | tr '\n' ' ' | head -c 300)
echo "EXIT_CODE=${rc}"
echo "LINE=${caller_line}"
echo "COMMAND=${flat_cmd}"
echo "--- OUTPUT ---"
if [[ -s "$logfile" ]]; then
local segment
segment=$(tail -c +"$((start_bytes + 1))" "$logfile" 2>/dev/null |
sed 's/\r$//' |
sed 's/\x1b\[[0-9;]*[a-zA-Z]//g' |
grep -avE '^(Get:|Hit:|Ign:|Fetched |Reading package lists|Reading state information|Building dependency tree|Selecting previously|Preparing to unpack|Unpacking |Processing triggers for|\(Reading database|[0-9]+%[[:space:]]*\[)' |
grep -avE '^[[:space:]]*$' |
tail -n 60)
# If the noise filter swallowed everything, fall back to the raw tail
if [[ -z "$segment" ]]; then
segment=$(tail -c +"$((start_bytes + 1))" "$logfile" 2>/dev/null |
sed 's/\r$//' | sed 's/\x1b\[[0-9;]*[a-zA-Z]//g' | tail -n 60)
fi
printf '%s' "$segment" | head -c 10240
fi
} >"${logfile}.errinfo" 2>/dev/null || true
return "$rc"
fi
# Clear stale flags on success (prevents false positives if a previous
# $STD cmd || true failed and a later non-silent command triggers error_handler)
unset _SILENT_FAILED_RC _SILENT_FAILED_CMD _SILENT_FAILED_LINE _SILENT_FAILED_LOG 2>/dev/null || true
# Also drop a stale .errinfo from a previously tolerated failure ($STD cmd || true)
rm -f "${logfile}.errinfo" 2>/dev/null || true
}
# ------------------------------------------------------------------------------

View File

@@ -7,17 +7,12 @@
# License: MIT | https://github.com/community-scripts/ProxmoxVE/raw/main/LICENSE
# ------------------------------------------------------------------------------
#
# Provides error handling and signal management for all scripts.
#
# TELEMETRY CONTRACT:
# - HOST context: this file reports terminal statuses via post_update_to_api
# (full metadata + focused error trace).
# - CONTAINER context: this file NEVER talks to the telemetry API. It writes
# two local artifacts that the host picks up after lxc-attach returns:
# /root/.install-<SESSION_ID>.failed → exit code (flag file)
# /root/.install-<SESSION_ID>.log.errinfo → structured error capture
# This guarantees the server only ever sees ONE terminal event per
# execution - the host's complete one.
# Provides comprehensive error handling and signal management for all scripts.
# Includes:
# - Exit code explanations (shell, package managers, databases, custom codes)
# - Error handler with detailed logging
# - Signal handlers (EXIT, INT, TERM)
# - Initialization function for trap setup
#
# Usage:
# source <(curl -fsSL .../error_handler.func)
@@ -26,14 +21,15 @@
# ------------------------------------------------------------------------------
# ==============================================================================
# SECTION 1: EXIT CODE EXPLANATIONS (fallback)
# SECTION 1: EXIT CODE EXPLANATIONS
# ==============================================================================
# ------------------------------------------------------------------------------
# explain_exit_code()
#
# - Canonical version lives in api.func (sourced before this file on the host)
# - This fallback covers the container context where api.func is not sourced
# - Canonical version is defined in api.func (sourced before this file)
# - This section only provides a fallback if api.func was not loaded
# - See api.func SECTION 1 for the authoritative exit code mappings
# ------------------------------------------------------------------------------
if ! declare -f explain_exit_code &>/dev/null; then
explain_exit_code() {
@@ -98,6 +94,8 @@ if ! declare -f explain_exit_code &>/dev/null; then
100) echo "APT: Package manager error (broken packages / dependency problems)" ;;
101) echo "APT: Configuration error (bad sources.list, malformed config)" ;;
102) echo "APT: Lock held by another process (dpkg/apt still running)" ;;
# --- Script Validation & Setup (103-123) ---
103) echo "Validation: Shell is not Bash" ;;
104) echo "Validation: Not running as root (or invoked via sudo)" ;;
105) echo "Validation: Proxmox VE version not supported" ;;
@@ -179,8 +177,9 @@ if ! declare -f explain_exit_code &>/dev/null; then
223) echo "Proxmox: Template not available after download" ;;
224) echo "Proxmox: PBS storage is for backups only" ;;
225) echo "Proxmox: No template available for OS/Version" ;;
226) echo "Proxmox: VM disk import or post-creation setup failed" ;;
231) echo "Proxmox: LXC stack upgrade failed" ;;
# --- Tools & Addon Scripts (232-238) ---
232) echo "Tools: Wrong execution environment (run on PVE host, not inside LXC)" ;;
233) echo "Tools: Application not installed (update prerequisite missing)" ;;
234) echo "Tools: No LXC containers found or available" ;;
@@ -188,6 +187,7 @@ if ! declare -f explain_exit_code &>/dev/null; then
236) echo "Tools: Required hardware not detected" ;;
237) echo "Tools: Dependency package installation failed" ;;
238) echo "Tools: OS or distribution not supported for this addon" ;;
239) echo "npm/Node.js: Unexpected runtime error or dependency failure" ;;
243) echo "Node.js: Out of memory (JavaScript heap out of memory)" ;;
245) echo "Node.js: Invalid command-line option" ;;
@@ -195,11 +195,14 @@ if ! declare -f explain_exit_code &>/dev/null; then
247) echo "Node.js: Fatal internal error" ;;
248) echo "Node.js: Invalid C++ addon / N-API failure" ;;
249) echo "npm/pnpm/yarn: Unknown fatal error" ;;
# --- Application Install/Update Errors (250-254) ---
250) echo "App: Download failed or version not determined" ;;
251) echo "App: File extraction failed (corrupt or incomplete archive)" ;;
252) echo "App: Required file or resource not found" ;;
253) echo "App: Data migration required — update aborted" ;;
254) echo "App: User declined prompt or input timed out" ;;
255) echo "DPKG: Fatal internal error" ;;
*) echo "Unknown error" ;;
esac
@@ -207,98 +210,24 @@ if ! declare -f explain_exit_code &>/dev/null; then
fi
# ==============================================================================
# SECTION 2: CONTEXT DETECTION & CONTAINER ARTIFACTS
# ==============================================================================
# ------------------------------------------------------------------------------
# _is_container_context()
#
# - Returns 0 (true) when running INSIDE the LXC container being installed
# - TELEMETRY_CONTEXT can override the heuristic ("host" / "container");
# install.func sets TELEMETRY_CONTEXT=container during bootstrap
# ------------------------------------------------------------------------------
_is_container_context() {
case "${TELEMETRY_CONTEXT:-}" in
container) return 0 ;;
host) return 1 ;;
esac
# Proxmox/Incus tooling exists only on the host
command -v pveversion &>/dev/null && return 1
command -v pct &>/dev/null && return 1
command -v incus &>/dev/null && return 1
# systemd-detect-virt reports lxc inside containers
if command -v systemd-detect-virt &>/dev/null; then
case "$(systemd-detect-virt -c 2>/dev/null)" in
lxc | lxc-libvirt | openvz) return 0 ;;
esac
fi
# PCT_OSTYPE is exported into the install environment by the host
[[ -n "${PCT_OSTYPE:-}" ]] && return 0
return 1
}
# ------------------------------------------------------------------------------
# _container_write_failure()
#
# - Writes the failure artifacts inside the container for the host to pick up:
# * flag file with the exit code
# * .errinfo capture (if silent() has not already written a better one)
# * copy of the install log
# - This REPLACES any direct telemetry send from the container
# - Arguments: $1 = exit_code, $2 = command (optional), $3 = line (optional)
# ------------------------------------------------------------------------------
_container_write_failure() {
local exit_code="${1:-1}"
local command="${2:-}"
local line="${3:-}"
local sid="${SESSION_ID:-error}"
# Flag file with exit code (host reads this after lxc-attach returns)
echo "$exit_code" >"/root/.install-${sid}.failed" 2>/dev/null || true
# Keep the install log where the host expects it
if [[ -n "${INSTALL_LOG:-}" && -f "${INSTALL_LOG}" && "${INSTALL_LOG}" != "/root/.install-${sid}.log" ]]; then
cp "${INSTALL_LOG}" "/root/.install-${sid}.log" 2>/dev/null || true
fi
# Structured error capture (skip if silent() already wrote the exact
# output segment of the failing command - that one is always better).
# Self-contained: api.func is NOT sourced inside containers.
local errinfo="${INSTALL_LOG:-/root/.install-${sid}.log}.errinfo"
if [[ ! -s "$errinfo" ]]; then
local flat_cmd
flat_cmd=$(printf '%s' "${command:-unknown}" | tr '\n' ' ' | head -c 300)
{
echo "EXIT_CODE=${exit_code}"
echo "LINE=${line:-0}"
echo "COMMAND=${flat_cmd}"
echo "--- OUTPUT ---"
if [[ -n "${INSTALL_LOG:-}" && -s "${INSTALL_LOG}" ]]; then
tail -n 80 "${INSTALL_LOG}" 2>/dev/null |
sed 's/\r$//' |
sed 's/\x1b\[[0-9;]*[a-zA-Z]//g' |
grep -avE '^(Get:|Hit:|Ign:|Fetched |Reading package lists|Reading state information|Building dependency tree|Selecting previously|Preparing to unpack|Unpacking |Processing triggers for|\(Reading database|[0-9]+%[[:space:]]*\[)' |
grep -avE '^[[:space:]]*$' |
tail -n 60 | head -c 10240
fi
} >"$errinfo" 2>/dev/null || true
fi
}
# ==============================================================================
# SECTION 3: ERROR HANDLER
# SECTION 2: ERROR HANDLERS
# ==============================================================================
# ------------------------------------------------------------------------------
# error_handler()
#
# - Main error handler triggered by ERR trap
# - Displays error message with line number, exit code, explanation, command
# - Shows last 20 lines of the active log
# - Emits actionable hints for common failure patterns (OOM, APT, network...)
# - HOST: reports "failed" to telemetry (full payload via api.func)
# - CONTAINER: writes failure artifacts, sends nothing
# - Exits with original exit code
# - Arguments: exit_code, command, line_number
# - Behavior:
# * Returns silently if exit_code is 0 (success)
# * Sources explain_exit_code() for detailed error description
# * Displays error message with:
# - Line number where error occurred
# - Exit code with explanation
# - Command that failed
# * Shows last 20 lines of SILENT_LOGFILE if available
# * Copies log to container /root for later inspection
# * Exits with original exit code
# ------------------------------------------------------------------------------
error_handler() {
local exit_code=${1:-$?}
@@ -308,10 +237,12 @@ error_handler() {
command="${command//\$STD/}"
# If error originated from silent(), use its captured metadata
# This provides the actual command and line number instead of "silent ..."
if [[ -n "${_SILENT_FAILED_RC:-}" ]]; then
exit_code="$_SILENT_FAILED_RC"
command="$_SILENT_FAILED_CMD"
line_number="$_SILENT_FAILED_LINE"
# Clear flags to prevent stale data on subsequent errors
unset _SILENT_FAILED_RC _SILENT_FAILED_CMD _SILENT_FAILED_LINE
fi
@@ -319,12 +250,8 @@ error_handler() {
return 0
fi
# Export the failure location so telemetry can include the "where"
FAILED_COMMAND="$command"
FAILED_LINE="$line_number"
export FAILED_COMMAND FAILED_LINE
# Stop spinner and restore cursor FIRST - before any output
# Stop spinner and restore cursor FIRST — before any output
# This prevents spinner text overlapping with error messages
if declare -f stop_spinner >/dev/null 2>&1; then
stop_spinner 2>/dev/null || true
fi
@@ -332,18 +259,18 @@ error_handler() {
local explanation
explanation="$(explain_exit_code "$exit_code")"
if [[ "$explanation" == curl:* && ! "$command" =~ (^|[[:space:]])([^[:space:]]*/)?curl([[:space:]]|$) ]]; then
explanation="Command failed with exit status ${exit_code}"
fi
# ── Telemetry / failure artifacts ──
if _is_container_context; then
_container_write_failure "$exit_code" "$command" "$line_number"
elif declare -f post_update_to_api &>/dev/null; then
# ALWAYS report failure to API immediately - don't wait for container checks
# This ensures we capture failures that occur before/after container exists
if declare -f post_update_to_api &>/dev/null; then
post_update_to_api "failed" "$exit_code" 2>/dev/null || true
else
# Container context: post_update_to_api not available (api.func not sourced)
# Send status directly via curl so container failures are never lost
_send_abort_telemetry "$exit_code" 2>/dev/null || true
fi
# ── Display ──
# Use msg_error if available, fallback to echo
if declare -f msg_error >/dev/null 2>&1; then
msg_error "in line ${line_number}: exit code ${exit_code} (${explanation}): while executing command ${command}"
else
@@ -361,7 +288,8 @@ error_handler() {
} >>"$DEBUG_LOGFILE"
fi
# Get active log file (prefer silent()'s logfile when available)
# Get active log file (BUILD_LOG or INSTALL_LOG)
# Prefer silent()'s logfile when available (contains the actual command output)
local active_log=""
if [[ -n "${_SILENT_FAILED_LOG:-}" && -s "${_SILENT_FAILED_LOG}" ]]; then
active_log="$_SILENT_FAILED_LOG"
@@ -372,17 +300,21 @@ error_handler() {
active_log="$SILENT_LOGFILE"
fi
# If active_log points to a container-internal path that doesn't exist on host,
# fall back to BUILD_LOG (host-side log)
if [[ -n "$active_log" && ! -s "$active_log" && -n "${BUILD_LOG:-}" && -s "${BUILD_LOG}" ]]; then
active_log="$BUILD_LOG"
fi
# Show last log lines if available
if [[ -n "$active_log" && -s "$active_log" ]]; then
echo -e "\n${TAB}--- Last 20 lines of log ---"
tail -n 20 "$active_log"
echo -e "${TAB}-----------------------------------\n"
fi
# ── Node.js heap OOM detection with actionable guidance ──
# Detect probable Node.js heap OOM and print actionable guidance.
# This avoids generic SIGABRT/SIGKILL confusion for frontend build failures.
local node_oom_detected="false"
local node_build_context="false"
if [[ "$command" =~ (npm|pnpm|yarn|node|vite|turbo) ]]; then
@@ -399,6 +331,7 @@ error_handler() {
if [[ "$node_oom_detected" == "true" ]] || { [[ "$node_build_context" == "true" ]] && [[ "$exit_code" =~ ^(134|137)$ ]]; }; then
local heap_hint_mb=""
# If explicitly configured, prefer the current value for troubleshooting output.
if [[ -n "${NODE_OPTIONS:-}" ]] && [[ "${NODE_OPTIONS}" =~ max-old-space-size=([0-9]+) ]]; then
heap_hint_mb="${BASH_REMATCH[1]}"
elif [[ -n "${var_ram:-}" ]] && [[ "${var_ram}" =~ ^[0-9]+$ ]]; then
@@ -425,13 +358,14 @@ error_handler() {
fi
fi
# ── Log-pattern analysis: actionable hints for common failure causes ──
# ── Log-pattern analysis: detect common failure causes and emit actionable hints ──
if [[ -n "$active_log" && -s "$active_log" ]]; then
local _log_tail
_log_tail=$(tail -n 60 "$active_log" 2>/dev/null || true)
# 1. APT/dpkg dependency conflict
if echo "$_log_tail" | grep -qE "Depends:|depends on.*but.*not installed|broken packages|unmet dep|dependency problems"; then
# Check for PostgreSQL-specific version mismatch (most actionable)
local _pg_conflict
_pg_conflict=$(echo "$_log_tail" | grep -oE 'postgresql-[0-9]+ but.*installed' | head -1 || true)
if [[ -n "$_pg_conflict" ]]; then
@@ -452,7 +386,7 @@ error_handler() {
msg_warn "Hint: A repository GPG key may be missing, expired, or the keyring file is not yet present (/usr/share/postgresql-common/pgdg/apt.postgresql.org.asc etc.)."
msg_warn "Hint: Install the 'postgresql-common' package first, or re-add the repository with its correct signing key."
fi
# 3. Network / DNS failure
# 3. Network / DNS failure during apt-get or curl
elif echo "$_log_tail" | grep -qE "Could not resolve|Failed to fetch|Unable to connect|Name or service not known|Network is unreachable|curl.*resolve"; then
if declare -f msg_warn >/dev/null 2>&1; then
msg_warn "Network or DNS failure detected."
@@ -473,12 +407,17 @@ error_handler() {
fi
fi
# ── Context-specific cleanup ──
if _is_container_context; then
# Container: artifacts already written above; nothing more to do
:
# Detect context: Container (INSTALL_LOG set + inside container /root) vs Host
if [[ -n "${INSTALL_LOG:-}" && -f "${INSTALL_LOG:-}" && -d /root ]]; then
# CONTAINER CONTEXT: Copy log and create flag file for host
local container_log="/root/.install-${SESSION_ID:-error}.log"
cp "${INSTALL_LOG}" "$container_log" 2>/dev/null || true
# Create error flag file with exit code for host detection
echo "$exit_code" >"/root/.install-${SESSION_ID:-error}.failed" 2>/dev/null || true
# Log path is shown by host as combined log - no need to show container path
else
# HOST: show log path and offer container cleanup
# HOST CONTEXT: Show local log path and offer container cleanup
if [[ -n "$active_log" && -s "$active_log" ]]; then
if declare -f msg_custom >/dev/null 2>&1; then
msg_custom "📋" "${YW}" "Full log: ${active_log}"
@@ -496,6 +435,7 @@ error_handler() {
echo -en "${YW}Remove broken container ${CTID}? (Y/n) [auto-remove in 60s]: ${CL}"
fi
# Read user response
local response=""
if read -t 60 -r response; then
if [[ -z "$response" || "$response" =~ ^[Yy]$ ]]; then
@@ -536,6 +476,12 @@ error_handler() {
echo -e "${GN}${CL} Container ${CTID} removed"
fi
fi
# Force one final status update attempt after cleanup
# This ensures status is updated even if the first attempt failed (e.g., HTTP 400)
if declare -f post_update_to_api &>/dev/null; then
post_update_to_api "failed" "$exit_code" "force"
fi
fi
fi
@@ -543,33 +489,106 @@ error_handler() {
}
# ==============================================================================
# SECTION 4: TELEMETRY & CLEANUP HELPERS FOR SIGNAL HANDLERS
# SECTION 3: TELEMETRY & CLEANUP HELPERS FOR SIGNAL HANDLERS
# ==============================================================================
# ------------------------------------------------------------------------------
# _send_abort_telemetry() (compatibility name)
# _send_abort_telemetry()
#
# - HOST: reports via post_update_to_api (signals map to "aborted" there)
# - CONTAINER: writes failure artifacts instead of sending anything
# - Sends failure/abort status to telemetry API
# - Works in BOTH host context (post_update_to_api available) and
# container context (only curl available, api.func not sourced)
# - Container context is critical: without this, container-side failures
# and signal exits are never reported, leaving records stuck in
# "installing" or "configuring" forever
# - Arguments: $1 = exit_code
# ------------------------------------------------------------------------------
_send_abort_telemetry() {
local exit_code="${1:-1}"
if _is_container_context; then
_container_write_failure "$exit_code"
return 0
fi
# Try full API function first (host context - api.func sourced)
if declare -f post_update_to_api &>/dev/null; then
post_update_to_api "failed" "$exit_code" 2>/dev/null || true
return
fi
# Fallback: direct curl (container context - api.func NOT sourced)
# This is the ONLY way containers can report failures to telemetry
command -v curl &>/dev/null || return 0
[[ "${DIAGNOSTICS:-no}" == "no" ]] && return 0
[[ -z "${RANDOM_UUID:-}" ]] && return 0
# Collect last 200 log lines for error diagnosis (best-effort)
# Container context has no get_full_log(), so we gather as much as possible
local error_text=""
local logfile=""
if [[ -n "${INSTALL_LOG:-}" && -s "${INSTALL_LOG}" ]]; then
logfile="${INSTALL_LOG}"
elif [[ -n "${SILENT_LOGFILE:-}" && -s "${SILENT_LOGFILE}" ]]; then
logfile="${SILENT_LOGFILE}"
fi
if [[ -n "$logfile" ]]; then
error_text=$(tail -n 200 "$logfile" 2>/dev/null | sed 's/\x1b\[[0-9;]*[a-zA-Z]//g; s/\\/\\\\/g; s/"/\\"/g; s/\r//g' | tr '\n' '|' | sed 's/|$//' | head -c 16384 | tr -d '\000-\010\013\014\016-\037\177') || true
fi
# Prepend exit code explanation header (like build_error_string does on host)
local explanation=""
if declare -f explain_exit_code &>/dev/null; then
explanation=$(explain_exit_code "$exit_code" 2>/dev/null) || true
fi
if [[ -n "$explanation" && -n "$error_text" ]]; then
error_text="exit_code=${exit_code} | ${explanation}|---|${error_text}"
elif [[ -n "$explanation" && -z "$error_text" ]]; then
error_text="exit_code=${exit_code} | ${explanation}"
fi
# Calculate duration if start time is available
local duration=""
if [[ -n "${DIAGNOSTICS_START_TIME:-}" ]]; then
duration=$(($(date +%s) - DIAGNOSTICS_START_TIME))
fi
# Categorize error if function is available (may not be in minimal container context)
local error_category=""
if declare -f categorize_error &>/dev/null; then
error_category=$(categorize_error "$exit_code" 2>/dev/null) || true
fi
# Build JSON payload with error context
local payload
payload="{\"random_id\":\"${RANDOM_UUID}\",\"execution_id\":\"${EXECUTION_ID:-${RANDOM_UUID}}\",\"type\":\"${TELEMETRY_TYPE:-lxc}\",\"nsapp\":\"${NSAPP:-${app:-unknown}}\",\"status\":\"failed\",\"exit_code\":${exit_code}"
[[ -n "$error_text" ]] && payload="${payload},\"error\":\"${error_text}\""
[[ -n "$error_category" ]] && payload="${payload},\"error_category\":\"${error_category}\""
[[ -n "$duration" ]] && payload="${payload},\"duration\":${duration}"
payload="${payload}}"
local api_url="${TELEMETRY_URL:-https://telemetry.community-scripts.org/telemetry}"
# 2 attempts (retry once on failure) — original had no retry
local attempt
for attempt in 1 2; do
if curl -fsS -m 5 -X POST "$api_url" \
-H "Content-Type: application/json" \
-d "$payload" &>/dev/null; then
return 0
fi
[[ $attempt -eq 1 ]] && sleep 1
done
return 0
}
# ------------------------------------------------------------------------------
# _stop_container_if_installing()
#
# - Stops the LXC container if we're in the install phase (host only)
# - Stops the LXC container if we're in the install phase
# - Prevents orphaned container processes when the host exits due to a signal
# (SSH disconnect, Ctrl+C, SIGTERM) — without this, the container keeps
# running and may send "configuring" status AFTER the host already sent
# "failed", leaving records permanently stuck in "configuring"
# - Only acts when:
# * CONTAINER_INSTALLING flag is set (during lxc-attach in build_container)
# * CTID is set (container was created)
# * pct command is available (we're on the Proxmox host, not inside a container)
# - Does NOT destroy the container — just stops it for potential debugging
# ------------------------------------------------------------------------------
_stop_container_if_installing() {
[[ "${CONTAINER_INSTALLING:-}" == "true" ]] || return 0
@@ -579,47 +598,51 @@ _stop_container_if_installing() {
}
# ==============================================================================
# SECTION 5: SIGNAL HANDLERS
# SECTION 4: SIGNAL HANDLERS
# ==============================================================================
# ------------------------------------------------------------------------------
# on_exit()
#
# - EXIT trap handler — runs on EVERY script termination
# - CONTAINER: ensures failure artifacts exist on non-zero exit (this also
# covers silent()'s direct `exit $rc`, which bypasses the ERR trap)
# - HOST: catches executions that never sent a final status:
# * non-zero exit → "failed" (signal codes map to "aborted")
# * zero exit with an "installing" record but no final → "aborted"
# (e.g. user cancelled a whiptail dialog, script exited cleanly)
# - Catches orphaned "installing"/"configuring" records:
# * If post_to_api sent "installing" but post_update_to_api never ran
# * Reports final status to prevent records stuck forever
# - Best-effort log collection for failed installs
# - Stops orphaned container processes on failure
# - Cleans up lock files
# ------------------------------------------------------------------------------
on_exit() {
local exit_code=$?
if _is_container_context; then
if [[ $exit_code -ne 0 ]]; then
_container_write_failure "$exit_code" "${FAILED_COMMAND:-}" "${FAILED_LINE:-}"
fi
else
if [[ "${POST_UPDATE_DONE:-}" != "true" ]] && declare -f post_update_to_api >/dev/null 2>&1; then
# Report orphaned telemetry records
# Two scenarios handled:
# 1. POST_TO_API_DONE=true but POST_UPDATE_DONE=false: Record was created but
# never got a final status update → send abort/done now.
# 2. POST_TO_API_DONE=false but DIAGNOSTICS=yes: Initial post failed (server
# unreachable/timeout), but the server has fallback create-on-update logic,
# so a status update can still create the record. Worth one last try.
if [[ "${POST_UPDATE_DONE:-}" != "true" ]]; then
if [[ "${POST_TO_API_DONE:-}" == "true" || "${DIAGNOSTICS:-no}" == "yes" ]]; then
if [[ $exit_code -ne 0 ]]; then
post_update_to_api "failed" "$exit_code" 2>/dev/null || true
elif [[ "${POST_TO_API_DONE:-}" == "true" ]]; then
# Clean exit but no success was ever reported: the user backed out
# somewhere. Report as aborted so the record doesn't stay "installing".
post_update_to_api "aborted" "0" 2>/dev/null || true
_send_abort_telemetry "$exit_code"
elif [[ "${INSTALL_COMPLETE:-}" == "true" ]] && declare -f post_update_to_api >/dev/null 2>&1; then
# Only report success if the install was explicitly marked complete.
# Without this guard, early bailouts (e.g. user cancelled) with exit 0
# would be falsely reported as successful installations.
post_update_to_api "done" "0" 2>/dev/null || true
fi
fi
fi
# Best-effort log collection on failure
if [[ $exit_code -ne 0 ]] && declare -f ensure_log_on_host >/dev/null 2>&1; then
ensure_log_on_host 2>/dev/null || true
fi
# Best-effort log collection on failure (non-critical, telemetry already sent)
if [[ $exit_code -ne 0 ]] && declare -f ensure_log_on_host >/dev/null 2>&1; then
ensure_log_on_host 2>/dev/null || true
fi
# Stop orphaned container if we're in the install phase
if [[ $exit_code -ne 0 ]]; then
_stop_container_if_installing
fi
# Stop orphaned container if we're in the install phase and exiting with error
if [[ $exit_code -ne 0 ]]; then
_stop_container_if_installing
fi
[[ -n "${lockfile:-}" && -e "$lockfile" ]] && rm -f "$lockfile"
@@ -627,19 +650,21 @@ on_exit() {
}
# ------------------------------------------------------------------------------
# on_interrupt() - SIGINT (Ctrl+C)
# on_interrupt()
#
# - SIGINT (Ctrl+C) trap handler
# - Reports status FIRST (time-critical: container may be dying)
# - Stops orphaned container to prevent "configuring" ghost records
# - Exits with code 130 (128 + SIGINT=2)
# ------------------------------------------------------------------------------
on_interrupt() {
# Stop spinner and restore cursor before any output
if declare -f stop_spinner >/dev/null 2>&1; then
stop_spinner 2>/dev/null || true
fi
printf "\e[?25h" 2>/dev/null || true
if _is_container_context; then
_container_write_failure "130"
elif declare -f post_update_to_api &>/dev/null; then
post_update_to_api "aborted" "130" 2>/dev/null || true
fi
_send_abort_telemetry "130"
_stop_container_if_installing
if declare -f msg_error >/dev/null 2>&1; then
msg_error "Interrupted by user (SIGINT)" 2>/dev/null || true
@@ -650,19 +675,21 @@ on_interrupt() {
}
# ------------------------------------------------------------------------------
# on_terminate() - SIGTERM
# on_terminate()
#
# - SIGTERM trap handler
# - Reports status FIRST (time-critical: process being killed)
# - Stops orphaned container to prevent "configuring" ghost records
# - Exits with code 143 (128 + SIGTERM=15)
# ------------------------------------------------------------------------------
on_terminate() {
# Stop spinner and restore cursor before any output
if declare -f stop_spinner >/dev/null 2>&1; then
stop_spinner 2>/dev/null || true
fi
printf "\e[?25h" 2>/dev/null || true
if _is_container_context; then
_container_write_failure "143"
elif declare -f post_update_to_api &>/dev/null; then
post_update_to_api "aborted" "143" 2>/dev/null || true
fi
_send_abort_telemetry "143"
_stop_container_if_installing
if declare -f msg_error >/dev/null 2>&1; then
msg_error "Terminated by signal (SIGTERM)" 2>/dev/null || true
@@ -673,32 +700,45 @@ on_terminate() {
}
# ------------------------------------------------------------------------------
# on_hangup() - SIGHUP (SSH disconnect, terminal closed)
# on_hangup()
#
# - SIGHUP trap handler (SSH disconnect, terminal closed)
# - CRITICAL: This was previously MISSING from catch_errors(), causing
# container processes to become orphans on SSH disconnect — the #1 cause
# of records stuck in "installing" and "configuring" states
# - Reports status via direct curl (terminal is already closed, no output)
# - Stops orphaned container to prevent ghost records
# - Exits with code 129 (128 + SIGHUP=1)
# ------------------------------------------------------------------------------
on_hangup() {
# Stop spinner (no cursor restore needed — terminal is already gone)
if declare -f stop_spinner >/dev/null 2>&1; then
stop_spinner 2>/dev/null || true
fi
if _is_container_context; then
_container_write_failure "129"
elif declare -f post_update_to_api &>/dev/null; then
post_update_to_api "aborted" "129" 2>/dev/null || true
fi
_send_abort_telemetry "129"
_stop_container_if_installing
exit 129
}
# ==============================================================================
# SECTION 6: INITIALIZATION
# SECTION 5: INITIALIZATION
# ==============================================================================
# ------------------------------------------------------------------------------
# catch_errors()
#
# - Initializes error handling and signal traps
# - set -Ee -o pipefail (+ set -u when STRICT_UNSET=1)
# - Traps: ERR → error_handler, EXIT → on_exit, INT/TERM/HUP → signal handlers
# - Enables strict error handling:
# * set -Ee: Exit on error, inherit ERR trap in functions
# * set -o pipefail: Pipeline fails if any command fails
# * set -u: (optional) Exit on undefined variable (if STRICT_UNSET=1)
# - Sets up traps:
# * ERR → error_handler (script errors)
# * EXIT → on_exit (any termination — cleanup + orphan detection)
# * INT → on_interrupt (Ctrl+C)
# * TERM → on_terminate (kill / systemd stop)
# * HUP → on_hangup (SSH disconnect / terminal closed)
# - Call this function early in every script
# ------------------------------------------------------------------------------
catch_errors() {

View File

@@ -32,11 +32,6 @@ if ! command -v curl >/dev/null 2>&1; then
apt update >/dev/null 2>&1
apt install -y curl >/dev/null 2>&1
fi
# Mark container context BEFORE error handling starts: error_handler/on_exit
# must write local failure artifacts instead of talking to the telemetry API
# (the host is the single telemetry reporter).
export TELEMETRY_CONTEXT="container"
source <(curl -fsSL https://raw.githubusercontent.com/community-scripts/ProxmoxVE/main/misc/core.func)
source <(curl -fsSL https://raw.githubusercontent.com/community-scripts/ProxmoxVE/main/misc/error_handler.func)
load_functions
@@ -70,12 +65,9 @@ post_progress_to_api() {
local progress_status="${1:-configuring}"
# Progress pings are the ONLY telemetry a container sends (terminal statuses
# are reported by the host). Include execution_id + platform + repo
# attribution (exported by the host) so the server can correlate and filter.
curl -fsS -m 5 -X POST "https://telemetry.community-scripts.org/telemetry" \
-H "Content-Type: application/json" \
-d "{\"random_id\":\"${RANDOM_UUID}\",\"execution_id\":\"${EXECUTION_ID:-${RANDOM_UUID}}\",\"type\":\"lxc\",\"nsapp\":\"${app:-unknown}\",\"status\":\"${progress_status}\",\"platform\":\"${TELEMETRY_PLATFORM:-}\",\"repo_source\":\"${REPO_SOURCE:-}\",\"repo_slug\":\"${REPO_SLUG:-}\"}" &>/dev/null || true
-d "{\"random_id\":\"${RANDOM_UUID}\",\"execution_id\":\"${EXECUTION_ID:-${RANDOM_UUID}}\",\"type\":\"lxc\",\"nsapp\":\"${app:-unknown}\",\"status\":\"${progress_status}\"}" &>/dev/null || true
}
# ==============================================================================

View File

@@ -2523,9 +2523,7 @@ _download_source_tarball() {
# directory). Extracts <tarball_path> into <workdir>, then copies the contents
# of that top-level directory into <target>.
#
# - Honors CLEAN_INSTALL=1 (wipes <target> first, dotfiles included — back
# up config dotfiles like .env via create_backup and call restore_backup
# BEFORE any build step that sources them).
# - Honors CLEAN_INSTALL=1 (wipes <target> first, dotfiles included).
# - Does NOT own <workdir>: the caller creates it and is responsible for its
# cleanup (typically via a RETURN trap on its tmpdir).
# - cp failures are non-fatal here, matching the previous inline behavior.
@@ -2563,9 +2561,7 @@ _deploy_source_tarball() {
# a single top-level directory, that directory is stripped (its contents land
# directly in <target>); otherwise the archive contents are copied as-is.
#
# - Honors CLEAN_INSTALL=1 (wipes <target> first, dotfiles included — back
# up config dotfiles like .env via create_backup and call restore_backup
# BEFORE any build step that sources them).
# - Honors CLEAN_INSTALL=1 (wipes <target> first, dotfiles included).
# - Does NOT own <workdir>: the caller creates it and cleans it up.
#
# Returns: 0 on success, 65 on unsupported format, 251 on extraction failure,
@@ -2591,16 +2587,6 @@ _deploy_unpacked_archive() {
msg_error "Failed to extract TAR archive"
return 251
}
elif [[ "$filename" == *.7z ]]; then
if [[ -f /etc/alpine-release ]]; then
ensure_dependencies 7zip
else
ensure_dependencies p7zip-full
fi
7z x -y -o"$workdir" "$archive" >/dev/null 2>&1 || {
msg_error "Failed to extract 7z archive"
return 251
}
else
msg_error "Unsupported archive format: $filename"
return 65
@@ -6995,7 +6981,7 @@ setup_meilisearch() {
CURRENT_VERSION=$(/usr/bin/meilisearch --version 2>/dev/null | grep -oE '[0-9]+\.[0-9]+\.[0-9]+' | head -1) || CURRENT_VERSION="0.0.0"
NEW_VERSION="${CHECK_UPDATE_RELEASE#v}"
# Extract major.minor for comparison (Meilisearch changes its on-disk DB format between minor versions)
# Extract major.minor for comparison (Meilisearch requires dump/restore between minor versions)
local CURRENT_MAJOR_MINOR NEW_MAJOR_MINOR
CURRENT_MAJOR_MINOR=$(echo "$CURRENT_VERSION" | cut -d. -f1,2)
NEW_MAJOR_MINOR=$(echo "$NEW_VERSION" | cut -d. -f1,2)
@@ -7007,72 +6993,140 @@ setup_meilisearch() {
msg_info "MeiliSearch version change detected (${CURRENT_VERSION}${NEW_VERSION}), preparing data migration"
fi
# Read connection + storage paths for the in-place (dumpless) upgrade
local MEILI_HOST MEILI_PORT MEILI_DB_PATH
# Read config values for dump/restore
local MEILI_HOST MEILI_PORT MEILI_MASTER_KEY MEILI_DUMP_DIR
MEILI_HOST="${MEILISEARCH_HOST:-127.0.0.1}"
MEILI_PORT="${MEILISEARCH_PORT:-7700}"
MEILI_DB_PATH=$(grep -E "^db_path\s*=" /etc/meilisearch.toml 2>/dev/null | sed 's/.*=\s*"\(.*\)"/\1/' | tr -d ' ' || true)
MEILI_DB_PATH="${MEILI_DB_PATH:-/var/lib/meilisearch/data}"
MEILI_DUMP_DIR="${MEILISEARCH_DUMP_DIR:-/var/lib/meilisearch/dumps}"
MEILI_MASTER_KEY=$(grep -E "^master_key\s*=" /etc/meilisearch.toml 2>/dev/null | sed 's/.*=\s*"\(.*\)"/\1/' | tr -d ' ' || true)
systemctl stop meilisearch
# Create dump before update if migration is needed
local DUMP_UID=""
if [[ "$NEEDS_MIGRATION" == "true" ]] && [[ -n "$MEILI_MASTER_KEY" ]]; then
msg_info "Creating MeiliSearch data dump before upgrade"
# Safety backup of the data dir before any in-place migration.
# Dumpless upgrade mutates the DB in place and cannot be rolled back on
# failure, so keep a tarball to restore from. (No master_key/API needed.)
local MEILI_BACKUP=""
if [[ "$NEEDS_MIGRATION" == "true" ]]; then
MEILI_BACKUP="/var/lib/meilisearch/pre-upgrade-${CURRENT_VERSION}.tar.gz"
msg_info "Backing up data dir before upgrade → ${MEILI_BACKUP}"
if ! tar -czf "$MEILI_BACKUP" -C "$MEILI_DB_PATH" . 2>/dev/null; then
msg_error "MeiliSearch data backup failed — aborting upgrade to avoid data loss"
systemctl start meilisearch
return 100
# Trigger dump creation
local DUMP_RESPONSE
DUMP_RESPONSE=$(curl -s -X POST "http://${MEILI_HOST}:${MEILI_PORT}/dumps" \
-H "Authorization: Bearer ${MEILI_MASTER_KEY}" \
-H "Content-Type: application/json" 2>/dev/null) || true
# The initial response only contains taskUid, not dumpUid
# dumpUid is only available after the task completes
local TASK_UID
TASK_UID=$(echo "$DUMP_RESPONSE" | grep -oP '"taskUid":\s*\K[0-9]+' || true)
if [[ -n "$TASK_UID" ]]; then
msg_info "Waiting for dump task ${TASK_UID} to complete..."
local MAX_WAIT=120
local WAITED=0
local TASK_RESULT=""
while [[ $WAITED -lt $MAX_WAIT ]]; do
TASK_RESULT=$(curl -s "http://${MEILI_HOST}:${MEILI_PORT}/tasks/${TASK_UID}" \
-H "Authorization: Bearer ${MEILI_MASTER_KEY}" 2>/dev/null) || true
local TASK_STATUS
TASK_STATUS=$(echo "$TASK_RESULT" | grep -oP '"status":\s*"\K[^"]+' || true)
if [[ "$TASK_STATUS" == "succeeded" ]]; then
# Extract dumpUid from the completed task details
DUMP_UID=$(echo "$TASK_RESULT" | grep -oP '"dumpUid":\s*"\K[^"]+' || true)
if [[ -n "$DUMP_UID" ]]; then
msg_ok "MeiliSearch dump created successfully: ${DUMP_UID}"
else
msg_warn "Dump task succeeded but could not extract dumpUid"
fi
break
elif [[ "$TASK_STATUS" == "failed" ]]; then
local ERROR_MSG
ERROR_MSG=$(echo "$TASK_RESULT" | grep -oP '"message":\s*"\K[^"]+' || echo "Unknown error")
msg_warn "MeiliSearch dump failed: ${ERROR_MSG}"
break
fi
sleep 2
WAITED=$((WAITED + 2))
done
if [[ $WAITED -ge $MAX_WAIT ]]; then
msg_warn "MeiliSearch dump timed out after ${MAX_WAIT}s"
fi
else
msg_warn "Could not trigger MeiliSearch dump (no taskUid in response)"
msg_info "Response was: ${DUMP_RESPONSE:-empty}"
fi
msg_ok "Backup created: ${MEILI_BACKUP}"
fi
# Replace the binary
if [[ "$NEEDS_MIGRATION" == "true" ]] && [[ -z "$DUMP_UID" ]]; then
msg_error "MeiliSearch migration requires a successful dump before upgrade"
msg_error "Ensure the service is running and master_key is configured, or set MEILISEARCH_SKIP_MIGRATION=1 to force (data loss risk)"
if [[ "${MEILISEARCH_SKIP_MIGRATION:-}" != "1" ]]; then
return 100
fi
msg_warn "MEILISEARCH_SKIP_MIGRATION=1 — proceeding without dump (manual reindex may be required)"
fi
# Stop service and update binary
systemctl stop meilisearch
if [[ "$(arch_resolve)" == "arm64" ]]; then
fetch_and_deploy_gh_release "meilisearch" "meilisearch/meilisearch" "singlefile" "latest" "/usr/bin" "meilisearch-linux-aarch64"
else
fetch_and_deploy_gh_release "meilisearch" "meilisearch/meilisearch" "binary"
fi
if [[ "$NEEDS_MIGRATION" == "true" ]]; then
# One-time in-place migration via --experimental-dumpless-upgrade.
# Meilisearch migrates the on-disk DB during this startup; subsequent
# normal (systemd) starts run without the flag.
msg_info "Migrating data in place (dumpless upgrade ${CURRENT_VERSION}${NEW_VERSION})"
/usr/bin/meilisearch --config-file-path /etc/meilisearch.toml --experimental-dumpless-upgrade >/dev/null 2>&1 &
local MEILI_PID=$!
# If migration needed and dump was created, remove old data and import dump
if [[ "$NEEDS_MIGRATION" == "true" ]] && [[ -n "$DUMP_UID" ]]; then
local MEILI_DB_PATH
MEILI_DB_PATH=$(grep -E "^db_path\s*=" /etc/meilisearch.toml 2>/dev/null | sed 's/.*=\s*"\(.*\)"/\1/' | tr -d ' ' || true)
MEILI_DB_PATH="${MEILI_DB_PATH:-/var/lib/meilisearch/data}"
local MAX_WAIT=300
local WAITED=0
local MIGRATED=false
while [[ $WAITED -lt $MAX_WAIT ]]; do
if curl -sf "http://${MEILI_HOST}:${MEILI_PORT}/health" &>/dev/null; then
MIGRATED=true
break
msg_info "Removing old MeiliSearch database for migration"
find "${MEILI_DB_PATH:?}" -mindepth 1 -delete
# Import dump using CLI flag (this is the supported method)
local DUMP_FILE="${MEILI_DUMP_DIR}/${DUMP_UID}.dump"
if [[ -f "$DUMP_FILE" ]]; then
msg_info "Importing dump: ${DUMP_FILE}"
# Start meilisearch with --import-dump flag
# This is a one-time import that happens during startup
/usr/bin/meilisearch --config-file-path /etc/meilisearch.toml --import-dump "$DUMP_FILE" >/dev/null 2>&1 &
local MEILI_PID=$!
# Wait for meilisearch to become healthy (import happens during startup)
msg_info "Waiting for MeiliSearch to import and start..."
local MAX_WAIT=300
local WAITED=0
while [[ $WAITED -lt $MAX_WAIT ]]; do
if curl -sf "http://${MEILI_HOST}:${MEILI_PORT}/health" &>/dev/null; then
msg_ok "MeiliSearch is healthy after import"
break
fi
# Check if process is still running
if ! kill -0 $MEILI_PID 2>/dev/null; then
msg_warn "MeiliSearch process exited during import"
break
fi
sleep 3
WAITED=$((WAITED + 3))
done
# Stop the manual process
kill $MEILI_PID 2>/dev/null || true
wait $MEILI_PID 2>/dev/null || true
sleep 2
# Start via systemd for proper management
systemctl start meilisearch
if systemctl is-active --quiet meilisearch; then
msg_ok "MeiliSearch migrated successfully"
else
msg_warn "MeiliSearch failed to start after migration - check logs with: journalctl -u meilisearch"
fi
# Bail early if the migration process died
if ! kill -0 $MEILI_PID 2>/dev/null; then
msg_warn "MeiliSearch process exited during migration"
break
fi
sleep 3
WAITED=$((WAITED + 3))
done
# Stop the one-shot migration process and hand over to systemd
kill $MEILI_PID 2>/dev/null || true
wait $MEILI_PID 2>/dev/null || true
sleep 2
systemctl start meilisearch
if [[ "$MIGRATED" == "true" ]] && systemctl is-active --quiet meilisearch; then
msg_ok "MeiliSearch migrated successfully (backup kept at ${MEILI_BACKUP})"
else
msg_error "MeiliSearch migration failed. Restore with: systemctl stop meilisearch; rm -rf ${MEILI_DB_PATH:?}/*; tar -xzf ${MEILI_BACKUP} -C ${MEILI_DB_PATH}; then reinstall the previous version"
msg_warn "Dump file not found: ${DUMP_FILE}"
systemctl start meilisearch
fi
else
systemctl start meilisearch

View File

@@ -573,10 +573,8 @@ cleanup() {
if [[ $exit_code -ne 0 ]]; then
post_update_to_api "failed" "$exit_code"
else
# Exited cleanly but description()/success was never called: the user
# backed out of a dialog. Report as aborted - NOT "failed 1" (which
# produced meaningless 'General error' records with no error text).
post_update_to_api "aborted" "0"
# Exited cleanly but description()/success was never called — shouldn't happen
post_update_to_api "failed" "1"
fi
fi
fi