Files
AI_Video_Factory/deploy/alerts.yml
youbin 36f466ba11
Some checks failed
CI / Migrations, tests, build, and audit (push) Failing after 10m7s
CI / Production gate and container images (push) Successful in 20m35s
Release v1.0.0
2026-07-31 14:21:43 +08:00

157 lines
5.8 KiB
YAML

groups:
- name: frameflow-availability
rules:
- alert: FrameFlowApiDown
expr: up{job="frameflow-api"} == 0
for: 2m
labels:
severity: critical
annotations:
summary: FrameFlow API is down
description: Prometheus has been unable to scrape the API for two minutes.
- alert: FrameFlowWorkerDown
expr: up{job="frameflow-worker"} == 0
for: 2m
labels:
severity: critical
annotations:
summary: FrameFlow worker is down
description: Prometheus has been unable to scrape the generation worker for two minutes.
- alert: FrameFlowDependencyUnavailable
expr: frameflow_application_dependency_available == 0
for: 2m
labels:
severity: critical
annotations:
summary: FrameFlow dependency is unavailable
description: "API readiness dependency {{ $labels.dependency }} has been unavailable for two minutes."
- alert: FrameFlowWorkerHeartbeatStale
expr: time() - frameflow_generation_worker_heartbeat_timestamp_seconds > 30
for: 1m
labels:
severity: critical
annotations:
summary: FrameFlow worker heartbeat is stale
description: The worker metrics process is reachable but its Redis heartbeat has stopped.
- alert: FrameFlowInfrastructureExporterDown
expr: up{job=~"frameflow-node|frameflow-minio"} == 0
for: 5m
labels:
severity: critical
annotations:
summary: FrameFlow infrastructure metrics are unavailable
description: "Prometheus has been unable to scrape {{ $labels.job }} for five minutes."
- name: frameflow-capacity-and-backup
rules:
- alert: FrameFlowHostFilesystemLow
expr: |
(
node_filesystem_avail_bytes{fstype!~"tmpfs|devtmpfs|overlay|squashfs|nsfs|tracefs|cgroup2?"}
/
node_filesystem_size_bytes{fstype!~"tmpfs|devtmpfs|overlay|squashfs|nsfs|tracefs|cgroup2?"}
) < 0.15
and on(instance, device, mountpoint)
node_filesystem_readonly == 0
for: 15m
labels:
severity: warning
annotations:
summary: FrameFlow host filesystem has less than 15% free space
description: "Filesystem {{ $labels.mountpoint }} on {{ $labels.instance }} is running low on space."
- alert: FrameFlowMinioCapacityLow
expr: |
minio_cluster_capacity_usable_free_bytes
/
minio_cluster_capacity_usable_total_bytes
< 0.15
for: 15m
labels:
severity: warning
annotations:
summary: FrameFlow object storage has less than 15% usable capacity
description: "MinIO cluster {{ $labels.server }} is running low on usable object-storage capacity."
- alert: FrameFlowBackupStale
expr: |
(time() - frameflow_backup_last_success_timestamp_seconds > 90000)
or absent(frameflow_backup_last_success_timestamp_seconds)
for: 15m
labels:
severity: critical
annotations:
summary: FrameFlow has no successful backup in the last 25 hours
description: Run and verify the scheduled PostgreSQL, Redis, and MinIO backup.
- alert: FrameFlowRestoreRehearsalStale
expr: |
(time() - frameflow_restore_rehearsal_last_success_timestamp_seconds > 3024000)
or absent(frameflow_restore_rehearsal_last_success_timestamp_seconds)
for: 15m
labels:
severity: warning
annotations:
summary: FrameFlow has no successful restore rehearsal in the last 35 days
description: Restore a verified backup into the isolated rehearsal environment and investigate any validation failure.
- name: frameflow-performance
rules:
- alert: FrameFlowApiHighErrorRate
expr: |
(
sum(rate(frameflow_api_requests_total{status=~"5.."}[5m]))
/
clamp_min(sum(rate(frameflow_api_requests_total[5m])), 0.001)
) > 0.05
and sum(rate(frameflow_api_requests_total[5m])) > 0.05
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow API error rate is high
description: More than 5% of API requests have returned 5xx responses for ten minutes.
- alert: FrameFlowApiLatencyHigh
expr: |
histogram_quantile(0.95,
sum by (le) (rate(frameflow_api_request_duration_seconds_bucket[10m]))
) > 2
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow API latency is high
description: API p95 latency has exceeded two seconds for ten minutes.
- alert: FrameFlowGenerationFailureRateHigh
expr: |
(
sum(rate(frameflow_generation_jobs_processed_total{result="failed"}[10m]))
/
clamp_min(sum(rate(frameflow_generation_jobs_processed_total[10m])), 0.001)
) > 0.20
and sum(rate(frameflow_generation_jobs_processed_total[10m])) > 0.01
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow generation failure rate is high
description: More than 20% of recent generation attempts have failed.
- alert: FrameFlowQueueLatencyHigh
expr: |
histogram_quantile(0.95,
sum by (le) (rate(frameflow_generation_job_queue_delay_seconds_bucket[10m]))
) > 60
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow queue latency is high
description: Generation job p95 queue delay has exceeded 60 seconds for ten minutes.