Release v1.0.0
Some checks failed
CI / Migrations, tests, build, and audit (push) Failing after 10m7s
CI / Production gate and container images (push) Successful in 20m35s

This commit is contained in:
2026-07-31 14:21:43 +08:00
commit 36f466ba11
187 changed files with 98907 additions and 0 deletions

View File

@@ -0,0 +1,22 @@
global:
resolve_timeout: 5m
route:
receiver: frameflow-operator-webhook
group_by: [alertname, severity]
group_wait: 30s
group_interval: 5m
repeat_interval: 4h
routes:
- receiver: frameflow-operator-webhook
matchers:
- alertname="FrameFlowDeliveryTest"
group_wait: 1s
group_interval: 5s
repeat_interval: 1h
receivers:
- name: frameflow-operator-webhook
webhook_configs:
- url: "__ALERTMANAGER_WEBHOOK_URL__"
send_resolved: true

156
deploy/alerts.yml Normal file
View File

@@ -0,0 +1,156 @@
groups:
- name: frameflow-availability
rules:
- alert: FrameFlowApiDown
expr: up{job="frameflow-api"} == 0
for: 2m
labels:
severity: critical
annotations:
summary: FrameFlow API is down
description: Prometheus has been unable to scrape the API for two minutes.
- alert: FrameFlowWorkerDown
expr: up{job="frameflow-worker"} == 0
for: 2m
labels:
severity: critical
annotations:
summary: FrameFlow worker is down
description: Prometheus has been unable to scrape the generation worker for two minutes.
- alert: FrameFlowDependencyUnavailable
expr: frameflow_application_dependency_available == 0
for: 2m
labels:
severity: critical
annotations:
summary: FrameFlow dependency is unavailable
description: "API readiness dependency {{ $labels.dependency }} has been unavailable for two minutes."
- alert: FrameFlowWorkerHeartbeatStale
expr: time() - frameflow_generation_worker_heartbeat_timestamp_seconds > 30
for: 1m
labels:
severity: critical
annotations:
summary: FrameFlow worker heartbeat is stale
description: The worker metrics process is reachable but its Redis heartbeat has stopped.
- alert: FrameFlowInfrastructureExporterDown
expr: up{job=~"frameflow-node|frameflow-minio"} == 0
for: 5m
labels:
severity: critical
annotations:
summary: FrameFlow infrastructure metrics are unavailable
description: "Prometheus has been unable to scrape {{ $labels.job }} for five minutes."
- name: frameflow-capacity-and-backup
rules:
- alert: FrameFlowHostFilesystemLow
expr: |
(
node_filesystem_avail_bytes{fstype!~"tmpfs|devtmpfs|overlay|squashfs|nsfs|tracefs|cgroup2?"}
/
node_filesystem_size_bytes{fstype!~"tmpfs|devtmpfs|overlay|squashfs|nsfs|tracefs|cgroup2?"}
) < 0.15
and on(instance, device, mountpoint)
node_filesystem_readonly == 0
for: 15m
labels:
severity: warning
annotations:
summary: FrameFlow host filesystem has less than 15% free space
description: "Filesystem {{ $labels.mountpoint }} on {{ $labels.instance }} is running low on space."
- alert: FrameFlowMinioCapacityLow
expr: |
minio_cluster_capacity_usable_free_bytes
/
minio_cluster_capacity_usable_total_bytes
< 0.15
for: 15m
labels:
severity: warning
annotations:
summary: FrameFlow object storage has less than 15% usable capacity
description: "MinIO cluster {{ $labels.server }} is running low on usable object-storage capacity."
- alert: FrameFlowBackupStale
expr: |
(time() - frameflow_backup_last_success_timestamp_seconds > 90000)
or absent(frameflow_backup_last_success_timestamp_seconds)
for: 15m
labels:
severity: critical
annotations:
summary: FrameFlow has no successful backup in the last 25 hours
description: Run and verify the scheduled PostgreSQL, Redis, and MinIO backup.
- alert: FrameFlowRestoreRehearsalStale
expr: |
(time() - frameflow_restore_rehearsal_last_success_timestamp_seconds > 3024000)
or absent(frameflow_restore_rehearsal_last_success_timestamp_seconds)
for: 15m
labels:
severity: warning
annotations:
summary: FrameFlow has no successful restore rehearsal in the last 35 days
description: Restore a verified backup into the isolated rehearsal environment and investigate any validation failure.
- name: frameflow-performance
rules:
- alert: FrameFlowApiHighErrorRate
expr: |
(
sum(rate(frameflow_api_requests_total{status=~"5.."}[5m]))
/
clamp_min(sum(rate(frameflow_api_requests_total[5m])), 0.001)
) > 0.05
and sum(rate(frameflow_api_requests_total[5m])) > 0.05
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow API error rate is high
description: More than 5% of API requests have returned 5xx responses for ten minutes.
- alert: FrameFlowApiLatencyHigh
expr: |
histogram_quantile(0.95,
sum by (le) (rate(frameflow_api_request_duration_seconds_bucket[10m]))
) > 2
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow API latency is high
description: API p95 latency has exceeded two seconds for ten minutes.
- alert: FrameFlowGenerationFailureRateHigh
expr: |
(
sum(rate(frameflow_generation_jobs_processed_total{result="failed"}[10m]))
/
clamp_min(sum(rate(frameflow_generation_jobs_processed_total[10m])), 0.001)
) > 0.20
and sum(rate(frameflow_generation_jobs_processed_total[10m])) > 0.01
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow generation failure rate is high
description: More than 20% of recent generation attempts have failed.
- alert: FrameFlowQueueLatencyHigh
expr: |
histogram_quantile(0.95,
sum by (le) (rate(frameflow_generation_job_queue_delay_seconds_bucket[10m]))
) > 60
for: 10m
labels:
severity: warning
annotations:
summary: FrameFlow queue latency is high
description: Generation job p95 queue delay has exceeded 60 seconds for ten minutes.

View File

@@ -0,0 +1,83 @@
upstream frameflow_api {
server api:8787;
keepalive 32;
}
server {
listen 80 default_server;
server_name ${APP_HOST};
root /usr/share/nginx/html;
index index.html;
client_max_body_size ${MAX_UPLOAD_SIZE};
add_header X-Content-Type-Options nosniff always;
add_header X-Frame-Options DENY always;
add_header Referrer-Policy strict-origin-when-cross-origin always;
add_header Permissions-Policy "camera=(), geolocation=(), microphone=()" always;
location /api/ {
proxy_pass http://frameflow_api;
proxy_http_version 1.1;
proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_set_header Connection "";
}
location /health/ {
proxy_pass http://frameflow_api;
proxy_http_version 1.1;
proxy_set_header Host $http_host;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
}
location = /metrics {
return 404;
}
location ~* \.(?:css|js|jpg|jpeg|png|webp|avif|svg|ico|woff2?)$ {
try_files $uri =404;
expires 7d;
add_header Cache-Control "public, max-age=604800, immutable";
}
location / {
try_files $uri $uri/ /index.html;
add_header Cache-Control "no-cache";
}
}
server {
listen 80;
server_name ${MEDIA_HOST};
client_max_body_size ${MAX_UPLOAD_SIZE};
location = /minio/v2/metrics {
return 404;
}
location ^~ /minio/v2/metrics/ {
return 404;
}
location = /minio/v3/metrics {
return 404;
}
location ^~ /minio/v3/metrics/ {
return 404;
}
location / {
proxy_pass http://minio:9000;
proxy_http_version 1.1;
proxy_request_buffering off;
proxy_buffering off;
proxy_set_header Host $http_host;
proxy_set_header X-Real-IP $remote_addr;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
}
}

32
deploy/prometheus.yml Normal file
View File

@@ -0,0 +1,32 @@
global:
scrape_interval: 15s
evaluation_interval: 15s
alerting:
alertmanagers:
- static_configs:
- targets: ["alertmanager:9093"]
rule_files:
- /etc/prometheus/alerts.yml
scrape_configs:
- job_name: frameflow-api
metrics_path: /metrics
static_configs:
- targets: ["api:8787"]
- job_name: frameflow-worker
metrics_path: /metrics
static_configs:
- targets: ["worker:9092"]
- job_name: frameflow-node
metrics_path: /metrics
static_configs:
- targets: ["node-exporter:9100"]
- job_name: frameflow-minio
metrics_path: /minio/v2/metrics/cluster
static_configs:
- targets: ["minio:9000"]

View File

@@ -0,0 +1,9 @@
# Copy to /etc/frameflow/backup.env and keep it readable only by root and the
# FrameFlow service group. These paths must match the installed application,
# encrypted backup mount, and Node Exporter textfile collector configuration.
FRAMEFLOW_ENV_FILE=/srv/frameflow/.env.production
FRAMEFLOW_BACKUP_MOUNTPOINT=/mnt/encrypted-backups
FRAMEFLOW_BACKUP_DESTINATION=/mnt/encrypted-backups/frameflow
FRAMEFLOW_BACKUP_METRICS_DIR=/var/lib/frameflow/metrics
FRAMEFLOW_RESTORE_METRICS_DIR=/var/lib/frameflow/metrics
FRAMEFLOW_RESTORE_TIMEOUT_SECONDS=300

View File

@@ -0,0 +1,35 @@
[Unit]
Description=FrameFlow verified PostgreSQL, Redis, and MinIO backup
Documentation=file:///srv/frameflow/docs/deployment.md
Requires=docker.service
After=docker.service network-online.target
Wants=network-online.target
RequiresMountsFor=/mnt/encrypted-backups
AssertPathIsMountPoint=/mnt/encrypted-backups
ConditionPathExists=/srv/frameflow/scripts/scheduled-maintenance.sh
ConditionPathExists=/etc/frameflow/backup.env
[Service]
Type=oneshot
User=frameflow
Group=frameflow
SupplementaryGroups=docker
EnvironmentFile=/etc/frameflow/backup.env
WorkingDirectory=/srv/frameflow
ExecStart=/srv/frameflow/scripts/scheduled-maintenance.sh backup
UMask=0077
TimeoutStartSec=6h
Nice=10
IOSchedulingClass=best-effort
IOSchedulingPriority=6
NoNewPrivileges=true
PrivateTmp=true
ProtectHome=true
ProtectSystem=strict
ProtectControlGroups=true
ProtectKernelModules=true
ProtectKernelTunables=true
RestrictRealtime=true
RestrictSUIDSGID=true
LockPersonality=true
ReadWritePaths=/mnt/encrypted-backups/frameflow /var/lib/frameflow/metrics /var/run/docker.sock

View File

@@ -0,0 +1,12 @@
[Unit]
Description=Run the FrameFlow verified backup every day
[Timer]
OnCalendar=*-*-* 03:15:00
RandomizedDelaySec=15m
AccuracySec=1m
Persistent=true
Unit=frameflow-backup.service
[Install]
WantedBy=timers.target

View File

@@ -0,0 +1,36 @@
[Unit]
Description=FrameFlow isolated restore rehearsal of the latest verified backup
Documentation=file:///srv/frameflow/docs/deployment.md
Requires=docker.service frameflow-backup.service
After=docker.service frameflow-backup.service network-online.target
Wants=network-online.target
RequiresMountsFor=/mnt/encrypted-backups
AssertPathIsMountPoint=/mnt/encrypted-backups
ConditionPathExists=/srv/frameflow/scripts/scheduled-maintenance.sh
ConditionPathExists=/etc/frameflow/backup.env
[Service]
Type=oneshot
User=frameflow
Group=frameflow
SupplementaryGroups=docker
EnvironmentFile=/etc/frameflow/backup.env
WorkingDirectory=/srv/frameflow
ExecStart=/srv/frameflow/scripts/scheduled-maintenance.sh restore-latest
UMask=0077
TimeoutStartSec=6h
Nice=10
IOSchedulingClass=best-effort
IOSchedulingPriority=6
NoNewPrivileges=true
PrivateTmp=true
ProtectHome=true
ProtectSystem=strict
ProtectControlGroups=true
ProtectKernelModules=true
ProtectKernelTunables=true
RestrictRealtime=true
RestrictSUIDSGID=true
LockPersonality=true
ReadOnlyPaths=/mnt/encrypted-backups/frameflow
ReadWritePaths=/var/lib/frameflow/metrics /var/run/docker.sock

View File

@@ -0,0 +1,12 @@
[Unit]
Description=Run a FrameFlow isolated restore rehearsal every month
[Timer]
OnCalendar=*-*-01 05:15:00
RandomizedDelaySec=30m
AccuracySec=1m
Persistent=true
Unit=frameflow-restore-rehearsal.service
[Install]
WantedBy=timers.target