Release v1.0.0
This commit is contained in:
156
deploy/alerts.yml
Normal file
156
deploy/alerts.yml
Normal file
@@ -0,0 +1,156 @@
|
||||
groups:
|
||||
- name: frameflow-availability
|
||||
rules:
|
||||
- alert: FrameFlowApiDown
|
||||
expr: up{job="frameflow-api"} == 0
|
||||
for: 2m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: FrameFlow API is down
|
||||
description: Prometheus has been unable to scrape the API for two minutes.
|
||||
|
||||
- alert: FrameFlowWorkerDown
|
||||
expr: up{job="frameflow-worker"} == 0
|
||||
for: 2m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: FrameFlow worker is down
|
||||
description: Prometheus has been unable to scrape the generation worker for two minutes.
|
||||
|
||||
- alert: FrameFlowDependencyUnavailable
|
||||
expr: frameflow_application_dependency_available == 0
|
||||
for: 2m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: FrameFlow dependency is unavailable
|
||||
description: "API readiness dependency {{ $labels.dependency }} has been unavailable for two minutes."
|
||||
|
||||
- alert: FrameFlowWorkerHeartbeatStale
|
||||
expr: time() - frameflow_generation_worker_heartbeat_timestamp_seconds > 30
|
||||
for: 1m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: FrameFlow worker heartbeat is stale
|
||||
description: The worker metrics process is reachable but its Redis heartbeat has stopped.
|
||||
|
||||
- alert: FrameFlowInfrastructureExporterDown
|
||||
expr: up{job=~"frameflow-node|frameflow-minio"} == 0
|
||||
for: 5m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: FrameFlow infrastructure metrics are unavailable
|
||||
description: "Prometheus has been unable to scrape {{ $labels.job }} for five minutes."
|
||||
|
||||
- name: frameflow-capacity-and-backup
|
||||
rules:
|
||||
- alert: FrameFlowHostFilesystemLow
|
||||
expr: |
|
||||
(
|
||||
node_filesystem_avail_bytes{fstype!~"tmpfs|devtmpfs|overlay|squashfs|nsfs|tracefs|cgroup2?"}
|
||||
/
|
||||
node_filesystem_size_bytes{fstype!~"tmpfs|devtmpfs|overlay|squashfs|nsfs|tracefs|cgroup2?"}
|
||||
) < 0.15
|
||||
and on(instance, device, mountpoint)
|
||||
node_filesystem_readonly == 0
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: FrameFlow host filesystem has less than 15% free space
|
||||
description: "Filesystem {{ $labels.mountpoint }} on {{ $labels.instance }} is running low on space."
|
||||
|
||||
- alert: FrameFlowMinioCapacityLow
|
||||
expr: |
|
||||
minio_cluster_capacity_usable_free_bytes
|
||||
/
|
||||
minio_cluster_capacity_usable_total_bytes
|
||||
< 0.15
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: FrameFlow object storage has less than 15% usable capacity
|
||||
description: "MinIO cluster {{ $labels.server }} is running low on usable object-storage capacity."
|
||||
|
||||
- alert: FrameFlowBackupStale
|
||||
expr: |
|
||||
(time() - frameflow_backup_last_success_timestamp_seconds > 90000)
|
||||
or absent(frameflow_backup_last_success_timestamp_seconds)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: critical
|
||||
annotations:
|
||||
summary: FrameFlow has no successful backup in the last 25 hours
|
||||
description: Run and verify the scheduled PostgreSQL, Redis, and MinIO backup.
|
||||
|
||||
- alert: FrameFlowRestoreRehearsalStale
|
||||
expr: |
|
||||
(time() - frameflow_restore_rehearsal_last_success_timestamp_seconds > 3024000)
|
||||
or absent(frameflow_restore_rehearsal_last_success_timestamp_seconds)
|
||||
for: 15m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: FrameFlow has no successful restore rehearsal in the last 35 days
|
||||
description: Restore a verified backup into the isolated rehearsal environment and investigate any validation failure.
|
||||
|
||||
- name: frameflow-performance
|
||||
rules:
|
||||
- alert: FrameFlowApiHighErrorRate
|
||||
expr: |
|
||||
(
|
||||
sum(rate(frameflow_api_requests_total{status=~"5.."}[5m]))
|
||||
/
|
||||
clamp_min(sum(rate(frameflow_api_requests_total[5m])), 0.001)
|
||||
) > 0.05
|
||||
and sum(rate(frameflow_api_requests_total[5m])) > 0.05
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: FrameFlow API error rate is high
|
||||
description: More than 5% of API requests have returned 5xx responses for ten minutes.
|
||||
|
||||
- alert: FrameFlowApiLatencyHigh
|
||||
expr: |
|
||||
histogram_quantile(0.95,
|
||||
sum by (le) (rate(frameflow_api_request_duration_seconds_bucket[10m]))
|
||||
) > 2
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: FrameFlow API latency is high
|
||||
description: API p95 latency has exceeded two seconds for ten minutes.
|
||||
|
||||
- alert: FrameFlowGenerationFailureRateHigh
|
||||
expr: |
|
||||
(
|
||||
sum(rate(frameflow_generation_jobs_processed_total{result="failed"}[10m]))
|
||||
/
|
||||
clamp_min(sum(rate(frameflow_generation_jobs_processed_total[10m])), 0.001)
|
||||
) > 0.20
|
||||
and sum(rate(frameflow_generation_jobs_processed_total[10m])) > 0.01
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: FrameFlow generation failure rate is high
|
||||
description: More than 20% of recent generation attempts have failed.
|
||||
|
||||
- alert: FrameFlowQueueLatencyHigh
|
||||
expr: |
|
||||
histogram_quantile(0.95,
|
||||
sum by (le) (rate(frameflow_generation_job_queue_delay_seconds_bucket[10m]))
|
||||
) > 60
|
||||
for: 10m
|
||||
labels:
|
||||
severity: warning
|
||||
annotations:
|
||||
summary: FrameFlow queue latency is high
|
||||
description: Generation job p95 queue delay has exceeded 60 seconds for ten minutes.
|
||||
Reference in New Issue
Block a user