You can not select more than 25 topics
Topics must start with a letter or number, can include dashes ('-') and can be up to 35 characters long.
503 lines
27 KiB
503 lines
27 KiB
######################################################################
|
|
# @CCOSTAN - Follow Me on X
|
|
# For more info visit https://www.vcloudinfo.com/click-here
|
|
# Original Repo : https://github.com/CCOSTAN/Home-AssistantConfig
|
|
# -------------------------------------------------------------------
|
|
# Proxmox Host Automations - repairs and Joanna dispatch
|
|
# Update, runtime, and disk health automations.
|
|
# -------------------------------------------------------------------
|
|
# Related Issue: 1584
|
|
# Related Issue: 1798
|
|
# Notes: Proxmox update availability is dashboard-only maintenance state; no Repair is created.
|
|
# Notes: Adds normalized runtime + disk health signals for dashboard/alerts.
|
|
# Notes: Joanna dispatch handles overnight Proxmox updates plus sustained runtime/disk degradations.
|
|
# Notes: Docker14 must be confirmed stopped before either Proxmox node reboots and restored after recovery.
|
|
# Notes: Return migrations are sequential; Docker14 startup waits for expected placement and capacity headroom.
|
|
# Notes: Overnight maintenance uses a four-hour fail-safe alert lease and releases its own lease after verified recovery.
|
|
# Notes: Normalized disk usage sensors expose state_class for long-term trend rollups.
|
|
######################################################################
|
|
template:
|
|
- sensor:
|
|
- name: "Proxmox1 Disk Used Percentage"
|
|
unique_id: proxmox1_disk_used_percentage
|
|
unit_of_measurement: "%"
|
|
state_class: measurement
|
|
icon: mdi:harddisk
|
|
availability: >-
|
|
{% set preferred = states('sensor.node_proxmox1_disk_used_percentage') %}
|
|
{% set used = states('sensor.node_proxmox1_disk') %}
|
|
{% set total = states('sensor.node_proxmox1_max_disk') %}
|
|
{{ preferred not in ['unknown', 'unavailable', 'none', ''] or
|
|
(used not in ['unknown', 'unavailable', 'none', ''] and
|
|
total not in ['unknown', 'unavailable', 'none', ''] and
|
|
(total | float(0)) > 0) }}
|
|
state: >-
|
|
{% set preferred = states('sensor.node_proxmox1_disk_used_percentage') %}
|
|
{% if preferred not in ['unknown', 'unavailable', 'none', ''] %}
|
|
{{ preferred | float(0) | round(1) }}
|
|
{% else %}
|
|
{% set used = states('sensor.node_proxmox1_disk') | float(0) %}
|
|
{% set total = states('sensor.node_proxmox1_max_disk') | float(0) %}
|
|
{% if total > 0 %}
|
|
{{ ((used / total) * 100) | round(1) }}
|
|
{% else %}
|
|
0
|
|
{% endif %}
|
|
{% endif %}
|
|
|
|
- name: "Proxmox02 Disk Used Percentage"
|
|
unique_id: proxmox02_disk_used_percentage
|
|
unit_of_measurement: "%"
|
|
state_class: measurement
|
|
icon: mdi:harddisk
|
|
availability: >-
|
|
{% set preferred = states('sensor.node_proxmox02_disk_used_percentage') %}
|
|
{% set used = states('sensor.node_proxmox02_disk') %}
|
|
{% set total = states('sensor.node_proxmox02_max_disk') %}
|
|
{{ preferred not in ['unknown', 'unavailable', 'none', ''] or
|
|
(used not in ['unknown', 'unavailable', 'none', ''] and
|
|
total not in ['unknown', 'unavailable', 'none', ''] and
|
|
(total | float(0)) > 0) }}
|
|
state: >-
|
|
{% set preferred = states('sensor.node_proxmox02_disk_used_percentage') %}
|
|
{% if preferred not in ['unknown', 'unavailable', 'none', ''] %}
|
|
{{ preferred | float(0) | round(1) }}
|
|
{% else %}
|
|
{% set used = states('sensor.node_proxmox02_disk') | float(0) %}
|
|
{% set total = states('sensor.node_proxmox02_max_disk') | float(0) %}
|
|
{% if total > 0 %}
|
|
{{ ((used / total) * 100) | round(1) }}
|
|
{% else %}
|
|
0
|
|
{% endif %}
|
|
{% endif %}
|
|
|
|
- binary_sensor:
|
|
- name: "Proxmox1 Runtime Healthy"
|
|
unique_id: proxmox1_runtime_healthy
|
|
device_class: running
|
|
state: >-
|
|
{% set state_value = states('binary_sensor.node_proxmox1_status') %}
|
|
{% if state_value in ['on', 'off'] %}
|
|
{{ state_value == 'on' }}
|
|
{% else %}
|
|
{% set status = states('sensor.node_proxmox1_status') | lower %}
|
|
{{ status in ['online', 'running', 'on'] }}
|
|
{% endif %}
|
|
|
|
- name: "Proxmox02 Runtime Healthy"
|
|
unique_id: proxmox02_runtime_healthy
|
|
device_class: running
|
|
state: >-
|
|
{% set state_value = states('binary_sensor.node_proxmox02_status') %}
|
|
{% if state_value in ['on', 'off'] %}
|
|
{{ state_value == 'on' }}
|
|
{% else %}
|
|
{% set status = states('sensor.node_proxmox02_status') | lower %}
|
|
{{ status in ['online', 'running', 'on'] }}
|
|
{% endif %}
|
|
|
|
automation:
|
|
- alias: "Proxmox Updates Joanna Dispatch"
|
|
id: proxmox_updates_joanna_dispatch
|
|
description: "Log when Proxmox updates appear, then dispatch Joanna overnight if updates remain."
|
|
mode: restart
|
|
trigger:
|
|
- platform: state
|
|
id: detected
|
|
entity_id:
|
|
- binary_sensor.node_proxmox1_updates_packages
|
|
- binary_sensor.node_proxmox02_updates_packages
|
|
to: "on"
|
|
- platform: time
|
|
id: overnight
|
|
at: "02:15:00"
|
|
condition:
|
|
- condition: template
|
|
value_template: >-
|
|
{{ is_state('binary_sensor.node_proxmox1_updates_packages', 'on') or
|
|
is_state('binary_sensor.node_proxmox02_updates_packages', 'on') }}
|
|
action:
|
|
- choose:
|
|
- conditions:
|
|
- condition: trigger
|
|
id: detected
|
|
sequence:
|
|
- delay: "00:01:00"
|
|
- condition: template
|
|
value_template: >-
|
|
{{ is_state('binary_sensor.node_proxmox1_updates_packages', 'on') or
|
|
is_state('binary_sensor.node_proxmox02_updates_packages', 'on') }}
|
|
- variables:
|
|
proxmox1_updates_count: "{{ states('sensor.node_proxmox1_total_updates') | int(0) }}"
|
|
proxmox02_updates_count: "{{ states('sensor.node_proxmox02_total_updates') | int(0) }}"
|
|
proxmox1_updates_summary: >-
|
|
{% set updates = state_attr('sensor.node_proxmox1_total_updates', 'updates_list') | default([], true) %}
|
|
{% if updates is sequence and updates is not string and updates | count > 0 %}
|
|
{{ updates | join('; ') }}
|
|
{% else %}
|
|
none reported
|
|
{% endif %}
|
|
proxmox02_updates_summary: >-
|
|
{% set updates = state_attr('sensor.node_proxmox02_total_updates', 'updates_list') | default([], true) %}
|
|
{% if updates is sequence and updates is not string and updates | count > 0 %}
|
|
{{ updates | join('; ') }}
|
|
{% else %}
|
|
none reported
|
|
{% endif %}
|
|
kernel_update_packages: >-
|
|
{% set ns = namespace(items=[]) %}
|
|
{% for entity_id in ['sensor.node_proxmox1_total_updates', 'sensor.node_proxmox02_total_updates'] %}
|
|
{% set updates = state_attr(entity_id, 'updates_list') | default([], true) %}
|
|
{% if updates is sequence and updates is not string %}
|
|
{% for item in updates %}
|
|
{% set haystack = item | string | lower %}
|
|
{% if (haystack | regex_findall('(proxmox-kernel|pve-kernel|linux-image|kernel)') | count) > 0 %}
|
|
{% set ns.items = ns.items + [item] %}
|
|
{% endif %}
|
|
{% endfor %}
|
|
{% endif %}
|
|
{% endfor %}
|
|
{{ ns.items | join('; ') if ns.items | count > 0 else 'none detected' }}
|
|
previous_maintenance_snooze_until: "{{ states('input_datetime.docker_container_alerts_snooze_until') }}"
|
|
previous_maintenance_snooze_timestamp: >-
|
|
{{ state_attr('input_datetime.docker_container_alerts_snooze_until', 'timestamp') | float(0) }}
|
|
requested_maintenance_snooze_timestamp: "{{ (now() + timedelta(hours=4)).timestamp() }}"
|
|
maintenance_snooze_timestamp: >-
|
|
{{ [previous_maintenance_snooze_timestamp | float(0),
|
|
requested_maintenance_snooze_timestamp | float(0)] | max }}
|
|
maintenance_snooze_until: >-
|
|
{{ maintenance_snooze_timestamp | float(0) | timestamp_custom('%Y-%m-%d %H:%M:%S', true) }}
|
|
maintenance_snooze_lease_owned: >-
|
|
{{ requested_maintenance_snooze_timestamp | float(0) > previous_maintenance_snooze_timestamp | float(0) }}
|
|
structured_request: |-
|
|
PROXMOX_HOST_UPGRADE_REQUEST
|
|
tracking_issue=https://github.com/CCOSTAN/Home-AssistantConfig/issues/1798
|
|
triggered_entity={{ trigger.entity_id | default('overnight_time_trigger', true) }}
|
|
dispatch_reason={{ trigger.id }}
|
|
dispatch_window=overnight_02:15_local
|
|
cluster_nodes=ProxMox1,ProxMox02
|
|
proxmox1_updates_sensor=binary_sensor.node_proxmox1_updates_packages
|
|
proxmox1_updates_state={{ states('binary_sensor.node_proxmox1_updates_packages') }}
|
|
proxmox1_updates_count={{ proxmox1_updates_count }}
|
|
proxmox1_updates={{ proxmox1_updates_summary }}
|
|
proxmox02_updates_sensor=binary_sensor.node_proxmox02_updates_packages
|
|
proxmox02_updates_state={{ states('binary_sensor.node_proxmox02_updates_packages') }}
|
|
proxmox02_updates_count={{ proxmox02_updates_count }}
|
|
proxmox02_updates={{ proxmox02_updates_summary }}
|
|
kernel_update_detected={{ (kernel_update_packages | trim) != 'none detected' }}
|
|
kernel_update_packages={{ kernel_update_packages | trim }}
|
|
maintenance_snooze_previous_until={{ previous_maintenance_snooze_until }}
|
|
maintenance_snooze_previous_timestamp={{ previous_maintenance_snooze_timestamp }}
|
|
maintenance_snooze_applied_until={{ maintenance_snooze_until }}
|
|
maintenance_snooze_lease_owned={{ maintenance_snooze_lease_owned }}
|
|
maintenance_alert_policy=The applied deadline is a four-hour fail-safe lease, not a fixed quiet period. Before stopping Docker14, verify binary_sensor.docker_container_alerts_snoozed is on and input_datetime.docker_container_alerts_snooze_until equals maintenance_snooze_applied_until. Keep it active through all return migrations, placement and capacity checks, Docker14 startup, and dependent-service verification.
|
|
maintenance_snooze_cleanup_policy=After successful final verification, read input_datetime.docker_container_alerts_snooze_until again. If maintenance_snooze_lease_owned is true and its live value still exactly equals maintenance_snooze_applied_until, restore maintenance_snooze_previous_until only when that previous deadline is still in the future; otherwise set the helper to the current time so the snooze clears immediately. If the live value differs, is later, or maintenance_snooze_lease_owned is false, preserve it because another operator or automation owns it. Verify binary_sensor.docker_container_alerts_snoozed is off when this task cleared the lease, or explicitly report the preserved owner deadline. If maintenance or verification fails, do not clear the lease; leave the fail-safe deadline in place and report it.
|
|
required_policy=This dispatch is Carlo's authorization to install pending same-release Proxmox package updates after live preflight passes; do not stop after read-only validation. Inspect live Proxmox cluster health, storage, VM placement, node status, and destination memory headroom before changing anything. Patch hosts one node at a time during the overnight window and verify each node remains healthy before moving to the next. Do not perform major-version upgrades, repository migrations, destructive cleanup, or force operations from this request. If the live preflight is unhealthy, pause and report instead of patching. If kernel packages are present or a reboot is required after kernel updates, use the $kernel-refresh skill and follow its ordered node-cycling workflow; if that skill is unavailable, stop before any reboot and report the blocker.
|
|
migration_policy=Capture each running VM's expected final node before evacuation. Migrate exactly one VM at a time and wait for task success plus running/unlocked/HA-started state before starting another. If any return migration fails or any non-Docker14 VM remains on ProxMox02, recover the expected placement and recheck memory headroom before continuing; do not start Docker14 while placement is incomplete.
|
|
docker14_policy=Verify the live placement of Frigate/Docker14 VM 101, gracefully shut it down before rebooting either Proxmox node even when VM 101 is hosted on the other node, confirm Proxmox reports it stopped, and start it only after the final host cycle, sequential return migrations, and capacity checks are complete. Starting the existing stopped VM 101 for this authorized recovery does not require separate confirmation. Never issue a Proxmox reboot while Docker14 is running.
|
|
completion_policy=After each host, verify package versions and the update list. Final success requires expected VM placement, no locks, HA resources started, Docker14 and dependent services healthy, enough host memory headroom, and verified cleanup or preservation of the maintenance snooze lease; report progress, failures, resume state, package versions, snooze disposition, and final verification.
|
|
- choose:
|
|
- conditions:
|
|
- condition: trigger
|
|
id: overnight
|
|
sequence:
|
|
- service: input_datetime.set_datetime
|
|
target:
|
|
entity_id: input_datetime.docker_container_alerts_snooze_until
|
|
data:
|
|
datetime: "{{ maintenance_snooze_until }}"
|
|
- service: script.send_to_logbook
|
|
data:
|
|
topic: "PROXMOX"
|
|
message: >-
|
|
Overnight Proxmox update window reached. Joanna patch orchestration requested.
|
|
Kernel refresh: {{ 'yes' if (kernel_update_packages | trim) != 'none detected' else 'not detected from HA package list' }}.
|
|
- service: script.joanna_dispatch
|
|
data:
|
|
trigger_context: "HA automation proxmox_updates_joanna_dispatch (Proxmox Updates Joanna Dispatch - Overnight)"
|
|
source: "home_assistant_automation.proxmox_updates_joanna_dispatch"
|
|
summary: >-
|
|
Overnight Proxmox patch window reached. Patch hosts one at a time.
|
|
Kernel refresh {{ 'required' if (kernel_update_packages | trim) != 'none detected' else 'not detected from HA package list' }}.
|
|
entity_ids:
|
|
- "binary_sensor.node_proxmox1_updates_packages"
|
|
- "sensor.node_proxmox1_total_updates"
|
|
- "binary_sensor.node_proxmox02_updates_packages"
|
|
- "sensor.node_proxmox02_total_updates"
|
|
- "binary_sensor.proxmox1_runtime_healthy"
|
|
- "binary_sensor.proxmox02_runtime_healthy"
|
|
- "binary_sensor.qemu_docker2_101_status"
|
|
diagnostics: >-
|
|
proxmox1_updates_count={{ proxmox1_updates_count }},
|
|
proxmox02_updates_count={{ proxmox02_updates_count }},
|
|
kernel_update_detected={{ (kernel_update_packages | trim) != 'none detected' }},
|
|
kernel_update_packages={{ kernel_update_packages | trim }},
|
|
container_alerts_snooze_previous_until={{ previous_maintenance_snooze_until }},
|
|
container_alerts_snooze_applied_until={{ maintenance_snooze_until }},
|
|
container_alerts_snooze_lease_owned={{ maintenance_snooze_lease_owned }}
|
|
request: "{{ structured_request }}"
|
|
default:
|
|
- service: script.send_to_logbook
|
|
data:
|
|
topic: "PROXMOX"
|
|
message: >-
|
|
Proxmox updates detected on one or more hosts. Joanna patch dispatch scheduled for 02:15 if updates remain.
|
|
Kernel refresh: {{ 'yes' if (kernel_update_packages | trim) != 'none detected' else 'not detected from HA package list' }}.
|
|
|
|
- alias: "Proxmox Runtime Repair Issues"
|
|
id: proxmox_runtime_repairs
|
|
description: "Create and clear Repairs when Proxmox node runtime becomes unhealthy."
|
|
mode: restart
|
|
trigger:
|
|
- platform: state
|
|
entity_id:
|
|
- binary_sensor.proxmox1_runtime_healthy
|
|
- binary_sensor.proxmox02_runtime_healthy
|
|
variables:
|
|
node_name: >-
|
|
{% if 'proxmox1' in trigger.entity_id %}Proxmox1{% else %}Proxmox02{% endif %}
|
|
issue_id: >-
|
|
{% if 'proxmox1' in trigger.entity_id %}
|
|
proxmox1_runtime_unhealthy
|
|
{% else %}
|
|
proxmox02_runtime_unhealthy
|
|
{% endif %}
|
|
runtime_entity: >-
|
|
{% if 'proxmox1' in trigger.entity_id %}
|
|
binary_sensor.proxmox1_runtime_healthy
|
|
{% else %}
|
|
binary_sensor.proxmox02_runtime_healthy
|
|
{% endif %}
|
|
status_entity: >-
|
|
{% if 'proxmox1' in trigger.entity_id %}
|
|
{% if states('binary_sensor.node_proxmox1_status') not in ['unknown', 'unavailable', 'none', ''] %}
|
|
binary_sensor.node_proxmox1_status
|
|
{% else %}
|
|
sensor.node_proxmox1_status
|
|
{% endif %}
|
|
{% else %}
|
|
{% if states('binary_sensor.node_proxmox02_status') not in ['unknown', 'unavailable', 'none', ''] %}
|
|
binary_sensor.node_proxmox02_status
|
|
{% else %}
|
|
sensor.node_proxmox02_status
|
|
{% endif %}
|
|
{% endif %}
|
|
status_value: "{{ states(status_entity) }}"
|
|
trigger_context: "HA automation proxmox_runtime_repairs (Proxmox Runtime Repair Issues)"
|
|
action:
|
|
- choose:
|
|
- conditions: "{{ trigger.to_state.state == 'off' }}"
|
|
sequence:
|
|
- delay: "00:02:00"
|
|
- condition: template
|
|
value_template: "{{ is_state(trigger.entity_id, 'off') }}"
|
|
- service: repairs.create
|
|
data:
|
|
issue_id: "{{ issue_id }}"
|
|
severity: error
|
|
persistent: true
|
|
title: "{{ node_name }} runtime degraded"
|
|
description: >
|
|
{{ node_name }} has remained offline for over 2 minutes.
|
|
Check node status in Proxmox and restore runtime.
|
|
- service: script.joanna_dispatch
|
|
data:
|
|
trigger_context: "{{ trigger_context }}"
|
|
source: "home_assistant_automation.proxmox_runtime_repairs"
|
|
summary: "{{ node_name }} runtime has remained degraded for over 2 minutes"
|
|
entity_ids:
|
|
- "{{ runtime_entity }}"
|
|
- "{{ status_entity }}"
|
|
diagnostics: >-
|
|
issue_id={{ issue_id }},
|
|
node_name={{ node_name }},
|
|
runtime_entity={{ runtime_entity }},
|
|
status_entity={{ status_entity }},
|
|
status_value={{ status_value }},
|
|
unhealthy_for=2m
|
|
request: >-
|
|
Investigate {{ node_name }} runtime degradation and restore node availability if possible.
|
|
Check host status, cluster connectivity, storage reachability, and recent update activity first.
|
|
Do not reboot the host unless explicitly requested.
|
|
- service: script.send_to_logbook
|
|
data:
|
|
topic: "PROXMOX"
|
|
message: >-
|
|
{{ node_name }} runtime is degraded. Repair {{ issue_id }} opened and Joanna investigation requested.
|
|
default:
|
|
- service: repairs.remove
|
|
continue_on_error: true
|
|
data:
|
|
issue_id: "{{ issue_id }}"
|
|
- service: script.send_to_logbook
|
|
data:
|
|
topic: "PROXMOX"
|
|
message: "{{ node_name }} runtime recovered."
|
|
|
|
- alias: "Proxmox Disk Pressure Repair Issues"
|
|
id: proxmox_disk_pressure_repairs
|
|
description: "Create and clear Repairs when Proxmox node disk usage stays elevated."
|
|
mode: restart
|
|
trigger:
|
|
- platform: numeric_state
|
|
entity_id:
|
|
- sensor.proxmox1_disk_used_percentage
|
|
- sensor.proxmox02_disk_used_percentage
|
|
above: 85
|
|
below: 92
|
|
for: "00:15:00"
|
|
id: warning
|
|
- platform: numeric_state
|
|
entity_id:
|
|
- sensor.proxmox1_disk_used_percentage
|
|
- sensor.proxmox02_disk_used_percentage
|
|
above: 92
|
|
id: critical
|
|
- platform: state
|
|
entity_id:
|
|
- sensor.proxmox1_disk_used_percentage
|
|
- sensor.proxmox02_disk_used_percentage
|
|
id: band_change
|
|
- platform: numeric_state
|
|
entity_id:
|
|
- sensor.proxmox1_disk_used_percentage
|
|
- sensor.proxmox02_disk_used_percentage
|
|
below: 85
|
|
id: recovered
|
|
variables:
|
|
node_name: >-
|
|
{% if 'proxmox1' in trigger.entity_id %}Proxmox1{% else %}Proxmox02{% endif %}
|
|
issue_id: >-
|
|
{% if 'proxmox1' in trigger.entity_id %}
|
|
proxmox1_disk_pressure
|
|
{% else %}
|
|
proxmox02_disk_pressure
|
|
{% endif %}
|
|
disk_entity: "{{ trigger.entity_id }}"
|
|
raw_disk_entity: >-
|
|
{% if 'proxmox1' in trigger.entity_id %}
|
|
sensor.node_proxmox1_disk_used_percentage
|
|
{% else %}
|
|
sensor.node_proxmox02_disk_used_percentage
|
|
{% endif %}
|
|
disk_pct: "{{ states(disk_entity) | float(0) }}"
|
|
previous_disk_pct: >-
|
|
{% if trigger.from_state is not none and trigger.from_state.state not in ['unknown', 'unavailable', 'none', ''] %}
|
|
{{ trigger.from_state.state | float(0) }}
|
|
{% else %}
|
|
0
|
|
{% endif %}
|
|
previous_band: >-
|
|
{% if previous_disk_pct >= 92 %}
|
|
critical
|
|
{% elif previous_disk_pct >= 85 %}
|
|
warning
|
|
{% else %}
|
|
normal
|
|
{% endif %}
|
|
action:
|
|
- choose:
|
|
- conditions:
|
|
- condition: trigger
|
|
id: critical
|
|
sequence:
|
|
- service: repairs.create
|
|
data:
|
|
issue_id: "{{ issue_id }}"
|
|
severity: error
|
|
persistent: true
|
|
title: "{{ node_name }} disk pressure critical ({{ disk_pct | round(1) }}%)"
|
|
description: >
|
|
{{ node_name }} disk usage is critically high.
|
|
Free disk space or expand storage allocation.
|
|
- service: script.joanna_dispatch
|
|
data:
|
|
trigger_context: "HA automation proxmox_disk_pressure_repairs (Proxmox Disk Pressure Repair Issues - Critical)"
|
|
source: "home_assistant_automation.proxmox_disk_pressure_repairs.critical"
|
|
summary: "{{ node_name }} disk pressure is critical at {{ disk_pct | round(1) }}%"
|
|
entity_ids:
|
|
- "{{ disk_entity }}"
|
|
- "{{ raw_disk_entity }}"
|
|
diagnostics: >-
|
|
issue_id={{ issue_id }},
|
|
node_name={{ node_name }},
|
|
disk_entity={{ disk_entity }},
|
|
raw_disk_entity={{ raw_disk_entity }},
|
|
disk_pct={{ disk_pct | round(1) }},
|
|
threshold=92
|
|
request: >-
|
|
Investigate critical disk pressure on {{ node_name }} and recommend safe remediation.
|
|
Check local storage usage, backups, logs, snapshots, and VM or container disk consumers first.
|
|
Do not delete VM disks or reboot the host unless explicitly requested.
|
|
- service: script.send_to_logbook
|
|
data:
|
|
topic: "PROXMOX"
|
|
message: >-
|
|
{{ node_name }} disk usage is critical at {{ disk_pct | round(1) }}%.
|
|
Repair {{ issue_id }} opened and Joanna investigation requested.
|
|
- conditions:
|
|
- condition: trigger
|
|
id: warning
|
|
- condition: template
|
|
value_template: "{{ previous_band != 'critical' }}"
|
|
sequence:
|
|
- service: repairs.create
|
|
data:
|
|
issue_id: "{{ issue_id }}"
|
|
severity: warning
|
|
persistent: true
|
|
title: "{{ node_name }} disk pressure warning ({{ disk_pct | round(1) }}%)"
|
|
description: >
|
|
{{ node_name }} disk usage has stayed above 85% for 15 minutes.
|
|
Plan cleanup before capacity reaches critical levels.
|
|
- service: script.joanna_dispatch
|
|
data:
|
|
trigger_context: "HA automation proxmox_disk_pressure_repairs (Proxmox Disk Pressure Repair Issues - Warning)"
|
|
source: "home_assistant_automation.proxmox_disk_pressure_repairs.warning"
|
|
summary: "{{ node_name }} disk pressure warning at {{ disk_pct | round(1) }}%"
|
|
entity_ids:
|
|
- "{{ disk_entity }}"
|
|
- "{{ raw_disk_entity }}"
|
|
diagnostics: >-
|
|
issue_id={{ issue_id }},
|
|
node_name={{ node_name }},
|
|
disk_entity={{ disk_entity }},
|
|
raw_disk_entity={{ raw_disk_entity }},
|
|
disk_pct={{ disk_pct | round(1) }},
|
|
threshold=85,
|
|
sustained_for=15m
|
|
request: >-
|
|
Investigate elevated disk usage on {{ node_name }} and recommend safe cleanup actions before it becomes critical.
|
|
Check local storage usage, backups, logs, snapshots, and VM or container disk consumers first.
|
|
Do not delete VM disks or reboot the host unless explicitly requested.
|
|
- service: script.send_to_logbook
|
|
data:
|
|
topic: "PROXMOX"
|
|
message: >-
|
|
{{ node_name }} disk usage warning at {{ disk_pct | round(1) }}%.
|
|
Repair {{ issue_id }} opened and Joanna investigation requested.
|
|
- conditions:
|
|
- condition: trigger
|
|
id: band_change
|
|
- condition: template
|
|
value_template: "{{ previous_band == 'critical' and disk_pct >= 85 and disk_pct < 92 }}"
|
|
sequence:
|
|
- service: repairs.create
|
|
data:
|
|
issue_id: "{{ issue_id }}"
|
|
severity: warning
|
|
persistent: true
|
|
title: "{{ node_name }} disk pressure warning ({{ disk_pct | round(1) }}%)"
|
|
description: >
|
|
{{ node_name }} disk usage is elevated but no longer critical.
|
|
Plan cleanup before capacity reaches critical levels again.
|
|
- conditions:
|
|
- condition: trigger
|
|
id: recovered
|
|
sequence:
|
|
- service: repairs.remove
|
|
continue_on_error: true
|
|
data:
|
|
issue_id: "{{ issue_id }}"
|