Refine WAN and infrastructure health signals

pull/1898/head
Carlo Costanzo 2 months ago
parent 160f39fe5c
commit 342e0b49f0

@ -17,6 +17,7 @@
# Notes: Status telemetry polling expects !secret bearclaw_status_url (token header stays !secret bearclaw_token). # Notes: Status telemetry polling expects !secret bearclaw_status_url (token header stays !secret bearclaw_token).
# Notes: Status telemetry includes the latest 12 dispatches in the rolling 24-hour window for the Joanna Activity drill-down. # Notes: Status telemetry includes the latest 12 dispatches in the rolling 24-hour window for the Joanna Activity drill-down.
# Notes: Codex reset-credit expiry telemetry is sourced from the appliance `codexUsage` status field. # Notes: Codex reset-credit expiry telemetry is sourced from the appliance `codexUsage` status field.
# Notes: Reset-credit urgency is dashboard-only: warning within 7 days and critical within 3 days; no Repair is created.
# Notes: Codex reset walkthrough: https://youtu.be/7wKhtrvtyiI # Notes: Codex reset walkthrough: https://youtu.be/7wKhtrvtyiI
# Notes: Codex reset companion post: https://www.vcloudinfo.com/2026/06/track-codex-resets-home-assistant.html # Notes: Codex reset companion post: https://www.vcloudinfo.com/2026/06/track-codex-resets-home-assistant.html
# Notes: TeslaMate road-trip walkthrough: https://youtu.be/9-9T6v17NEw # Notes: TeslaMate road-trip walkthrough: https://youtu.be/9-9T6v17NEw
@ -412,74 +413,22 @@ template:
soonest_expires_at: >- soonest_expires_at: >-
{% set usage = state_attr('sensor.bearclaw_status_telemetry', 'codexUsage') | default({}, true) %} {% set usage = state_attr('sensor.bearclaw_status_telemetry', 'codexUsage') | default({}, true) %}
{{ usage.get('soonestExpiresAt') }} {{ usage.get('soonestExpiresAt') }}
automation:
- id: codex_reset_credit_expiry_repair
alias: Codex Reset Credit Expiry Repair
description: "Create/clear a Repair while available Codex reset credits are within the expiry warning window."
mode: single
trigger:
- platform: state
entity_id: binary_sensor.codex_reset_credit_expiring_soon
- platform: time_pattern
hours: "/6"
variables:
usage: "{{ state_attr('sensor.bearclaw_status_telemetry', 'codexUsage') | default({}, true) }}"
usage_ok: "{{ usage.get('ok', false) in [true, 'true', 'True', 'on', 'yes', 1, '1'] }}"
issue_active: "{{ is_state('binary_sensor.codex_reset_credit_expiring_soon', 'on') }}"
warn_days: "{{ usage.get('warnDays', 7) | int(7) }}"
warning_count: "{{ usage.get('warningCount', 0) | int(0) }}"
min_days: "{{ usage.get('minDaysRemaining') }}"
soonest_expires_at: "{{ usage.get('soonestExpiresAt') | default('unknown', true) }}"
last_check_at: "{{ usage.get('lastCheckAt') | default('unknown', true) }}"
expiring_summary: >-
{% set credits = usage.get('credits', []) %}
{% set ns = namespace(lines=[]) %}
{% for credit in credits %}
{% set status = credit.get('status', 'unknown') %}
{% set days = credit.get('days_remaining') %}
{% if status == 'available' and days is number and days <= warn_days %}
{% set ns.lines = ns.lines + [
'Reset #' ~ credit.get('index', loop.index) ~ ': ' ~
(days | round(1)) ~ ' days remaining; expires ' ~
credit.get('expires_at', 'unknown')
] %}
{% endif %}
{% endfor %}
{{ ns.lines | join('\n') if ns.lines | count > 0 else 'No reset credits are inside the warning window.' }}
action:
- choose:
- conditions: "{{ issue_active }}"
sequence:
- service: repairs.create
data:
issue_id: codex_reset_credit_expiring_soon
title: "Codex reset credit expiring soon"
description: >-
{{ warning_count }} available Codex reset credit(s) expire within {{ warn_days }} days.
Soonest expiration: {{ soonest_expires_at }}
Minimum days remaining: {{ min_days }}
Last successful/attempted check: {{ last_check_at }}
{{ expiring_summary }}
severity: warning
persistent: true
- service: script.send_to_logbook
data:
topic: "CODEX"
message: >-
Codex reset credit warning active: {{ warning_count }} reset(s) within {{ warn_days }} days.
- conditions: "{{ usage_ok }}"
sequence:
- service: repairs.remove
continue_on_error: true
data:
issue_id: codex_reset_credit_expiring_soon
- service: script.send_to_logbook
data:
topic: "CODEX"
message: "Codex reset credit warning cleared after a successful reset snapshot."
- name: Codex Reset Credit Expiring Critical
unique_id: codex_reset_credit_expiring_critical
device_class: problem
icon: mdi:calendar-alert
state: >-
{% set usage = state_attr('sensor.bearclaw_status_telemetry', 'codexUsage') | default({}, true) %}
{% set available = usage.get('availableCount', 0) | int(0) %}
{% set min_days = usage.get('minDaysRemaining') %}
{{ available > 0 and min_days not in [none, 'unknown', 'unavailable', ''] and min_days | float(99) <= 3 }}
attributes:
critical_days: 3
min_days_remaining: >-
{% set usage = state_attr('sensor.bearclaw_status_telemetry', 'codexUsage') | default({}, true) %}
{{ usage.get('minDaysRemaining') }}
automation:
- id: bearclaw_lifecycle_webhook - id: bearclaw_lifecycle_webhook
alias: BearClaw Lifecycle Webhook alias: BearClaw Lifecycle Webhook
description: Receives structured BearClaw lifecycle callbacks for HA-triggered work. description: Receives structured BearClaw lifecycle callbacks for HA-triggered work.

@ -10,7 +10,9 @@
# ------------------------------------------------------------------- # -------------------------------------------------------------------
# Related Issue: 1584 # Related Issue: 1584
# Notes: Home dashboard consumes `infra_*` entities for exceptions-only alerts. # Notes: Home dashboard consumes `infra_*` entities for exceptions-only alerts.
# Notes: WAN outages alert immediately at near-zero throughput or 90% packet loss; performance degradation requires 6 continuous hours below 300/300 Mbps, above 80 ms, above 5% loss, or unavailable telemetry. # Notes: WAN outages require all four independent ICMP/HTTPS probes to fail continuously for 2 minutes.
# Notes: Continuous latency/loss determine WAN quality; Ookla throughput is a separate twice-daily capacity snapshot and never declares an outage.
# Notes: Severe quality turns critical after 30 minutes above 250 ms or 25% loss; moderate quality degradation requires 6 hours above 80 ms or 5% loss.
# Notes: Nightly Duplicati verification runs at 08:00 after the 05:30 Duplicati job and docker_14 reboot window. # Notes: Nightly Duplicati verification runs at 08:00 after the 05:30 Duplicati job and docker_14 reboot window.
# Notes: Duplicati transport/API errors are logged only; repairs are reserved for proven failed or stale backups. # Notes: Duplicati transport/API errors are logged only; repairs are reserved for proven failed or stale backups.
# Notes: Duplicati failure Repairs enable a recovery poll that clears the Repair after a later successful run. # Notes: Duplicati failure Repairs enable a recovery poll that clears the Repair after a later successful run.
@ -82,6 +84,28 @@ command_line:
state_class: measurement state_class: measurement
value_template: "{{ (value | regex_replace('[^0-9.]', '')) or 'unknown' }}" value_template: "{{ (value | regex_replace('[^0-9.]', '')) or 'unknown' }}"
- sensor:
name: Infra WAN Reachability
unique_id: infra_wan_reachability
command: >-
/bin/bash -c 'ok=0; cf=down; google=down; quad9=down; https=down;
if ping -q -c 1 -W 1 1.1.1.1 >/dev/null 2>&1; then cf=up; ok=$((ok+1)); fi;
if ping -q -c 1 -W 1 8.8.8.8 >/dev/null 2>&1; then google=up; ok=$((ok+1)); fi;
if ping -q -c 1 -W 1 9.9.9.9 >/dev/null 2>&1; then quad9=up; ok=$((ok+1)); fi;
if curl -fsS --max-time 4 https://www.google.com/generate_204 >/dev/null 2>&1; then https=up; ok=$((ok+1)); fi;
status=partial; if [ "$ok" -ge 3 ]; then status=online; elif [ "$ok" -eq 0 ]; then status=offline; fi;
printf "{\"status\":\"%s\",\"checks_passed\":%s,\"checks_total\":4,\"cloudflare_icmp\":\"%s\",\"google_icmp\":\"%s\",\"quad9_icmp\":\"%s\",\"google_https\":\"%s\"}\n"
"$status" "$ok" "$cf" "$google" "$quad9" "$https"'
scan_interval: 60
value_template: "{{ value_json.status | default('unknown') }}"
json_attributes:
- checks_passed
- checks_total
- cloudflare_icmp
- google_icmp
- quad9_icmp
- google_https
- sensor: - sensor:
name: Infra External IP Fallback name: Infra External IP Fallback
unique_id: infra_external_ip_fallback unique_id: infra_external_ip_fallback
@ -229,52 +253,96 @@ template:
- name: "Infra WAN Outage" - name: "Infra WAN Outage"
unique_id: infra_wan_outage unique_id: infra_wan_outage
device_class: problem device_class: problem
delay_on: "00:02:00"
delay_off: "00:01:00"
state: "{{ is_state('sensor.infra_wan_reachability', 'offline') }}"
attributes:
evaluation_window: "2 minutes"
probe_policy: "all 4 probes failed"
- name: "Infra WAN Capacity Degraded"
unique_id: infra_wan_capacity_degraded
device_class: problem
availability: >-
{{ states('sensor.ookla_speedtest_download') not in ['unknown', 'unavailable', 'none', ''] and
states('sensor.ookla_speedtest_upload') not in ['unknown', 'unavailable', 'none', ''] }}
state: >- state: >-
{% set loss_raw = states('sensor.infra_wan_packet_loss') %} {{ states('sensor.ookla_speedtest_download') | float(0) < 300 or
{% set download_raw = states('sensor.speedtest_download') %} states('sensor.ookla_speedtest_upload') | float(0) < 300 }}
{% set upload_raw = states('sensor.speedtest_upload') %} attributes:
{% set invalid_values = ['unknown', 'unavailable', 'none', ''] %} source: "Ookla Speedtest CLI"
{% set loss = loss_raw | float(0) %} download_threshold_mbps: 300
{% set download = download_raw | float(0) %} upload_threshold_mbps: 300
{% set upload = upload_raw | float(0) %} last_test: "{{ states('sensor.ookla_speedtest_last_test') }}"
{{ (loss_raw not in invalid_values and loss >= 90) or server: "{{ states('sensor.ookla_speedtest_server') }}"
(download_raw not in invalid_values and download <= 1) or freshness_entity: "binary_sensor.infra_wan_capacity_stale"
(upload_raw not in invalid_values and upload <= 1) }}
- name: "Infra WAN Capacity Stale"
unique_id: infra_wan_capacity_stale
device_class: problem
state: >-
{% set last_raw = states('sensor.ookla_speedtest_last_test') %}
{% if last_raw in ['unknown', 'unavailable', 'none', ''] %}
true
{% else %}
{{ as_timestamp(now()) - as_timestamp(last_raw, 0) > 50400 }}
{% endif %}
attributes:
freshness_limit: "14 hours"
schedule: "02:17 and 14:17 local"
- name: "Infra WAN Sustained Degradation" - name: "Infra WAN Sustained Degradation"
unique_id: infra_wan_sustained_degradation unique_id: infra_wan_sustained_degradation
device_class: problem device_class: problem
delay_on: "06:00:00" delay_on: "06:00:00"
delay_off: "00:30:00" delay_off: "00:05:00"
state: >-
{% set loss_raw = states('sensor.infra_wan_packet_loss') %}
{% set lat_raw = states('sensor.infra_wan_latency_ms') %}
{% set invalid_values = ['unknown', 'unavailable', 'none', ''] %}
{% set valid = loss_raw not in invalid_values and lat_raw not in invalid_values %}
{% set loss = loss_raw | float(0) %}
{% set lat = lat_raw | float(0) %}
{{ valid and (loss > 5 or lat > 80) }}
- name: "Infra WAN Severe Degradation"
unique_id: infra_wan_severe_degradation
device_class: problem
delay_on: "00:30:00"
state: >- state: >-
{% set loss_raw = states('sensor.infra_wan_packet_loss') %} {% set loss_raw = states('sensor.infra_wan_packet_loss') %}
{% set lat_raw = states('sensor.infra_wan_latency_ms') %} {% set lat_raw = states('sensor.infra_wan_latency_ms') %}
{% set download_raw = states('sensor.speedtest_download') %}
{% set upload_raw = states('sensor.speedtest_upload') %}
{% set invalid_values = ['unknown', 'unavailable', 'none', ''] %} {% set invalid_values = ['unknown', 'unavailable', 'none', ''] %}
{% set invalid = loss_raw in invalid_values or {% set valid = loss_raw not in invalid_values and
lat_raw in invalid_values or lat_raw not in invalid_values %}
download_raw in invalid_values or
upload_raw in invalid_values %}
{% set loss = loss_raw | float(0) %} {% set loss = loss_raw | float(0) %}
{% set lat = lat_raw | float(0) %} {% set lat = lat_raw | float(0) %}
{% set download = download_raw | float(0) %} {{ valid and (loss > 25 or lat > 250) }}
{% set upload = upload_raw | float(0) %}
{{ invalid or loss > 5 or lat > 80 or download < 300 or upload < 300 }} - name: "Infra WAN Critical"
unique_id: infra_wan_critical
device_class: problem
state: >-
{{ is_state('binary_sensor.infra_wan_outage', 'on') or
is_state('binary_sensor.infra_wan_severe_degradation', 'on') }}
attributes:
severe_evaluation_window: "30 minutes"
severe_latency_threshold_ms: 250
severe_packet_loss_threshold_percent: 25
- name: "Infra WAN Quality Degraded" - name: "Infra WAN Quality Degraded"
unique_id: infra_wan_quality_degraded unique_id: infra_wan_quality_degraded
device_class: problem device_class: problem
state: >- state: >-
{{ is_state('binary_sensor.infra_wan_outage', 'on') or {{ is_state('binary_sensor.infra_wan_critical', 'on') or
is_state('binary_sensor.infra_wan_sustained_degradation', 'on') }} is_state('binary_sensor.infra_wan_sustained_degradation', 'on') }}
attributes: attributes:
evaluation_window: "6 hours" evaluation_window: "6 hours"
throughput_threshold_mbps: 300
latency_threshold_ms: 80 latency_threshold_ms: 80
packet_loss_threshold_percent: 5 packet_loss_threshold_percent: 5
outage_throughput_mbps: 1 capacity_entity: "binary_sensor.infra_wan_capacity_degraded"
outage_packet_loss_percent: 90 reachability_entity: "sensor.infra_wan_reachability"
severity_scale: "healthy | degraded | critical"
- name: "Infra DNS Pihole Degraded" - name: "Infra DNS Pihole Degraded"
unique_id: infra_dns_pihole_degraded unique_id: infra_dns_pihole_degraded
@ -403,8 +471,6 @@ template:
{{ is_state('binary_sensor.docker_container_telemetry_degraded', 'on') or {{ is_state('binary_sensor.docker_container_telemetry_degraded', 'on') or
states('binary_sensor.proxmox1_runtime_healthy') != 'on' or states('binary_sensor.proxmox1_runtime_healthy') != 'on' or
states('binary_sensor.proxmox02_runtime_healthy') != 'on' or states('binary_sensor.proxmox02_runtime_healthy') != 'on' or
is_state('binary_sensor.node_proxmox1_updates_packages', 'on') or
is_state('binary_sensor.node_proxmox02_updates_packages', 'on') or
states('sensor.proxmox_garage_average_temperature') in ['unknown', 'unavailable', 'none', ''] or states('sensor.proxmox_garage_average_temperature') in ['unknown', 'unavailable', 'none', ''] or
states('sensor.proxmox_garage_average_temperature') | float(0) > 145 }} states('sensor.proxmox_garage_average_temperature') | float(0) > 145 }}

@ -8,10 +8,11 @@
# ------------------------------------------------------------------- # -------------------------------------------------------------------
# Related Issue: 1584 # Related Issue: 1584
# Related Issue: 1798 # Related Issue: 1798
# Notes: Creates HA repair issues when proxmox nodes report updates. # Notes: Proxmox update availability is dashboard-only maintenance state; no Repair is created.
# Notes: Proxmox update activity writes one final HA success notification per patched host.
# Notes: Adds normalized runtime + disk health signals for dashboard/alerts. # Notes: Adds normalized runtime + disk health signals for dashboard/alerts.
# Notes: Joanna dispatch handles overnight Proxmox updates plus sustained runtime/disk degradations. # Notes: Joanna dispatch handles overnight Proxmox updates plus sustained runtime/disk degradations.
# Notes: Docker14 must be confirmed stopped before either Proxmox node reboots and restored after recovery.
# Notes: Return migrations are sequential; Docker14 startup waits for expected placement and capacity headroom.
# Notes: Normalized disk usage sensors expose state_class for long-term trend rollups. # Notes: Normalized disk usage sensors expose state_class for long-term trend rollups.
###################################################################### ######################################################################
template: template:
@ -112,66 +113,6 @@ automation:
topic: "FRIGATE" topic: "FRIGATE"
message: "Frigate server rebooted at 5 AM." message: "Frigate server rebooted at 5 AM."
- alias: "Proxmox Updates Repair Issues"
id: proxmox_updates_repair
description: "Track repair issues when Proxmox hosts report updates, then notify once per host when updates clear."
mode: restart
trigger:
- platform: state
entity_id: binary_sensor.node_proxmox1_updates_packages
to: "on"
- platform: state
entity_id: binary_sensor.node_proxmox1_updates_packages
to: "off"
- platform: state
entity_id: binary_sensor.node_proxmox02_updates_packages
to: "on"
- platform: state
entity_id: binary_sensor.node_proxmox02_updates_packages
to: "off"
variables:
node_name: >
{% if 'proxmox1' in trigger.entity_id %}Proxmox1{% else %}Proxmox02{% endif %}
issue_id: >
{% if 'proxmox1' in trigger.entity_id %}
proxmox1_updates_available
{% else %}
proxmox02_updates_available
{% endif %}
action:
- choose:
- conditions: "{{ trigger.to_state.state == 'on' }}"
sequence:
- service: repairs.create
data:
issue_id: "{{ issue_id }}"
severity: warning
persistent: false
title: "{{ node_name }} has updates available"
description: >
{{ trigger.entity_id }} is ON, indicating pending updates on {{ node_name }}.
Apply updates in Proxmox, then reload this sensor to clear the issue.
default:
- service: repairs.remove
continue_on_error: true
data:
issue_id: "{{ issue_id }}"
- service: persistent_notification.dismiss
continue_on_error: true
data:
notification_id: "proxmox_updates_joanna_dispatch"
- service: persistent_notification.create
data:
notification_id: "{{ issue_id }}"
title: "{{ node_name }} Proxmox updates applied"
message: >
{{ node_name }} updates were successfully applied.
Update telemetry is now {{ trigger.to_state.state }}, and the repair issue has been cleared.
- service: script.send_to_logbook
data:
topic: "PROXMOX"
message: "{{ node_name }} Proxmox updates were successfully applied."
- alias: "Proxmox Updates Joanna Dispatch" - alias: "Proxmox Updates Joanna Dispatch"
id: proxmox_updates_joanna_dispatch id: proxmox_updates_joanna_dispatch
description: "Log when Proxmox updates appear, then dispatch Joanna overnight if updates remain." description: "Log when Proxmox updates appear, then dispatch Joanna overnight if updates remain."
@ -250,7 +191,10 @@ automation:
proxmox02_updates={{ proxmox02_updates_summary }} proxmox02_updates={{ proxmox02_updates_summary }}
kernel_update_detected={{ (kernel_update_packages | trim) != 'none detected' }} kernel_update_detected={{ (kernel_update_packages | trim) != 'none detected' }}
kernel_update_packages={{ kernel_update_packages | trim }} kernel_update_packages={{ kernel_update_packages | trim }}
required_policy=This dispatch is Carlo's authorization to install pending same-release Proxmox package updates after live preflight passes; do not stop after read-only validation. Inspect live Proxmox cluster health, storage, VM placement, and node status before changing anything. Patch hosts one node at a time during the overnight window and verify each node remains healthy before moving to the next. Do not perform major-version upgrades, repository migrations, destructive cleanup, or force operations from this request. If the live preflight is unhealthy, pause and report instead of patching. If kernel packages are present or a reboot is required after kernel updates, use the $kernel-refresh skill and follow its ordered node-cycling workflow. Verify the live placement of Frigate/Docker14 VM 101, stop it before rebooting the host that owns it, and start it only after the host cycle and VM migrations are complete. After each host, verify the package versions/update list, then report progress, failures, resume state, package versions, and final verification. required_policy=This dispatch is Carlo's authorization to install pending same-release Proxmox package updates after live preflight passes; do not stop after read-only validation. Inspect live Proxmox cluster health, storage, VM placement, node status, and destination memory headroom before changing anything. Patch hosts one node at a time during the overnight window and verify each node remains healthy before moving to the next. Do not perform major-version upgrades, repository migrations, destructive cleanup, or force operations from this request. If the live preflight is unhealthy, pause and report instead of patching. If kernel packages are present or a reboot is required after kernel updates, use the $kernel-refresh skill and follow its ordered node-cycling workflow; if that skill is unavailable, stop before any reboot and report the blocker.
migration_policy=Capture each running VM's expected final node before evacuation. Migrate exactly one VM at a time and wait for task success plus running/unlocked/HA-started state before starting another. If any return migration fails or any non-Docker14 VM remains on ProxMox02, recover the expected placement and recheck memory headroom before continuing; do not start Docker14 while placement is incomplete.
docker14_policy=Verify the live placement of Frigate/Docker14 VM 101, gracefully shut it down before rebooting either Proxmox node even when VM 101 is hosted on the other node, confirm Proxmox reports it stopped, and start it only after the final host cycle, sequential return migrations, and capacity checks are complete. Starting the existing stopped VM 101 for this authorized recovery does not require separate confirmation. Never issue a Proxmox reboot while Docker14 is running.
completion_policy=After each host, verify package versions and the update list. Final success requires expected VM placement, no locks, HA resources started, Docker14 and dependent services healthy, and enough host memory headroom; report progress, failures, resume state, package versions, and final verification.
- choose: - choose:
- conditions: - conditions:
- condition: trigger - condition: trigger

@ -3,17 +3,17 @@
# For more info visit https://www.vcloudinfo.com/click-here # For more info visit https://www.vcloudinfo.com/click-here
# Original Repo : https://github.com/CCOSTAN/Home-AssistantConfig # Original Repo : https://github.com/CCOSTAN/Home-AssistantConfig
# ------------------------------------------------------------------- # -------------------------------------------------------------------
# Speedtest Alerts - Log sustained WAN degradation, outages, and recovery # Ookla WAN Alerts - Log corroborated outages, quality degradation, capacity snapshots, and recovery
# Related Issue: 1550 # Related Issue: 1550
# Uses the normalized WAN health sensors plus `script.send_to_logbook`. # Uses continuous reachability/quality sensors plus Ookla CLI capacity snapshots.
# ------------------------------------------------------------------- # -------------------------------------------------------------------
# Notes: Brief Speedtest transitions do not create Activity entries; outages are immediate and performance degradation requires 6 continuous hours. # Notes: Capacity snapshots never declare the WAN down; brief quality transitions do not create Activity entries.
###################################################################### ######################################################################
automation: automation:
- alias: "Internet WAN Degraded (Logbook)" - alias: "Internet WAN Degraded (Logbook)"
id: notify-carlo-slow-internet-speed id: notify-carlo-slow-internet-speed
description: "Logs an Activity entry for an immediate outage or sustained WAN degradation." description: "Logs an Activity entry for a corroborated outage or sustained WAN quality degradation."
trigger: trigger:
- platform: state - platform: state
entity_id: binary_sensor.infra_wan_quality_degraded entity_id: binary_sensor.infra_wan_quality_degraded
@ -25,11 +25,19 @@ automation:
topic: "NETWORK" topic: "NETWORK"
message: >- message: >-
{% set outage = is_state('binary_sensor.infra_wan_outage', 'on') %} {% set outage = is_state('binary_sensor.infra_wan_outage', 'on') %}
Download: {{ states('sensor.speedtest_download') }} Mbps, {% set severe = is_state('binary_sensor.infra_wan_severe_degradation', 'on') %}
upload: {{ states('sensor.speedtest_upload') }} Mbps, Reachability: {{ states('sensor.infra_wan_reachability') }},
latency: {{ states('sensor.infra_wan_latency_ms') }} ms, continuous latency: {{ states('sensor.infra_wan_latency_ms') }} ms,
packet loss: {{ states('sensor.infra_wan_packet_loss') }}%. packet loss: {{ states('sensor.infra_wan_packet_loss') }}%.
{{ 'Immediate WAN outage detected.' if outage else 'WAN performance remained degraded for 6 hours.' }} Latest Ookla capacity: {{ states('sensor.ookla_speedtest_download') }} Mbps down /
{{ states('sensor.ookla_speedtest_upload') }} Mbps up.
{% if outage %}
Corroborated WAN outage detected.
{% elif severe %}
Severe WAN quality remained degraded for 30 minutes.
{% else %}
WAN quality remained degraded for 6 hours.
{% endif %}
mode: single mode: single
- alias: "Internet WAN Restored (Logbook)" - alias: "Internet WAN Restored (Logbook)"
@ -45,9 +53,49 @@ automation:
data: data:
topic: "NETWORK" topic: "NETWORK"
message: >- message: >-
Download: {{ states('sensor.speedtest_download') }} Mbps, Reachability: {{ states('sensor.infra_wan_reachability') }},
Upload: {{ states('sensor.speedtest_upload') }} Mbps, continuous latency: {{ states('sensor.infra_wan_latency_ms') }} ms,
latency: {{ states('sensor.infra_wan_latency_ms') }} ms,
packet loss: {{ states('sensor.infra_wan_packet_loss') }}%. packet loss: {{ states('sensor.infra_wan_packet_loss') }}%.
Latest Ookla capacity: {{ states('sensor.ookla_speedtest_download') }} Mbps down /
{{ states('sensor.ookla_speedtest_upload') }} Mbps up.
WAN alert cleared. WAN alert cleared.
mode: single mode: single
- alias: "Internet WAN Capacity Degraded (Logbook)"
id: log-ookla-wan-capacity-degraded
description: "Logs a slow Ookla capacity snapshot without declaring a WAN outage."
trigger:
- platform: state
entity_id: binary_sensor.infra_wan_capacity_degraded
from: 'off'
to: 'on'
action:
- service: script.send_to_logbook
data:
topic: "NETWORK"
message: >-
Ookla capacity snapshot was below 300/300 Mbps:
{{ states('sensor.ookla_speedtest_download') }} Mbps down /
{{ states('sensor.ookla_speedtest_upload') }} Mbps up,
{{ states('sensor.ookla_speedtest_ping') }} ms idle latency,
server {{ states('sensor.ookla_speedtest_server') }}.
mode: single
- alias: "Internet WAN Capacity Restored (Logbook)"
id: log-ookla-wan-capacity-restored
description: "Logs when a later Ookla capacity snapshot clears the advisory."
trigger:
- platform: state
entity_id: binary_sensor.infra_wan_capacity_degraded
from: 'on'
to: 'off'
action:
- service: script.send_to_logbook
data:
topic: "NETWORK"
message: >-
Ookla capacity snapshot recovered:
{{ states('sensor.ookla_speedtest_download') }} Mbps down /
{{ states('sensor.ookla_speedtest_upload') }} Mbps up,
server {{ states('sensor.ookla_speedtest_server') }}.
mode: single

Loading…
Cancel
Save

Powered by TurnKey Linux.