feat: 综合平台services多服务支持+SW网络优先缓存修复+适配层/定时任务完善
This commit is contained in:
@@ -0,0 +1,98 @@
|
||||
"""监控数据自动清理服务
|
||||
|
||||
按配置的保留周期清理 metrics.db 中的历史数据,防止 SQLite 无限膨胀:
|
||||
- MetricPoint: 高频时序数据,默认保留 30 天
|
||||
- SecurityCheck: 安全检查历史,默认保留 90 天
|
||||
- EventLog: 事件日志,默认保留 180 天
|
||||
|
||||
提供两种触发方式:
|
||||
1. 应用启动后由 asyncio 后台任务按 CLEANUP_INTERVAL_HOURS 周期执行
|
||||
2. 通过 API 手动触发(POST /api/monitor/cleanup)
|
||||
"""
|
||||
|
||||
import asyncio
|
||||
import logging
|
||||
from datetime import timedelta
|
||||
|
||||
from sqlmodel import Session, delete, select
|
||||
|
||||
from app.core.config import settings
|
||||
from app.core.timeutils import utcnow
|
||||
from app.database import metrics_engine
|
||||
from app.models.monitor import EventLog, MetricPoint, SecurityCheck
|
||||
|
||||
logger = logging.getLogger("vps-manager.cleanup")
|
||||
|
||||
# 单批删除行数上限:避免大表单条 DELETE 长时间持有写锁,阻塞 Agent 上报
|
||||
_BATCH_SIZE = 1000
|
||||
|
||||
|
||||
def _delete_in_batches(session: Session, model, cutoff, batch_size: int = _BATCH_SIZE) -> int:
|
||||
"""分批删除指定模型的过期数据(ts < cutoff),返回总删除行数
|
||||
|
||||
SQLite 不支持 DELETE ... LIMIT,用子查询 SELECT id ... LIMIT 实现分批。
|
||||
"""
|
||||
total = 0
|
||||
while True:
|
||||
subq = select(model.id).where(model.ts < cutoff).limit(batch_size)
|
||||
ids = [row[0] if isinstance(row, tuple) else row for row in session.exec(subq).all()]
|
||||
if not ids:
|
||||
break
|
||||
session.exec(delete(model).where(model.id.in_(ids)))
|
||||
session.commit() # 每批独立提交,缩短写锁持有时间
|
||||
total += len(ids)
|
||||
if len(ids) < batch_size:
|
||||
break
|
||||
return total
|
||||
|
||||
|
||||
def cleanup_metrics(
|
||||
metrics_days: int | None = None,
|
||||
security_days: int | None = None,
|
||||
event_log_days: int | None = None,
|
||||
) -> dict:
|
||||
"""按保留天数清理过期监控数据,返回各类删除条数"""
|
||||
metrics_days = metrics_days if metrics_days is not None else settings.METRICS_RETENTION_DAYS
|
||||
security_days = security_days if security_days is not None else settings.SECURITY_RETENTION_DAYS
|
||||
event_log_days = event_log_days if event_log_days is not None else settings.EVENT_LOG_RETENTION_DAYS
|
||||
|
||||
now = utcnow()
|
||||
result = {"metric_points": 0, "security_checks": 0, "event_logs": 0}
|
||||
|
||||
with Session(metrics_engine) as session:
|
||||
if metrics_days > 0:
|
||||
cutoff = now - timedelta(days=metrics_days)
|
||||
result["metric_points"] = _delete_in_batches(session, MetricPoint, cutoff)
|
||||
|
||||
if security_days > 0:
|
||||
cutoff = now - timedelta(days=security_days)
|
||||
result["security_checks"] = _delete_in_batches(session, SecurityCheck, cutoff)
|
||||
|
||||
if event_log_days > 0:
|
||||
cutoff = now - timedelta(days=event_log_days)
|
||||
result["event_logs"] = _delete_in_batches(session, EventLog, cutoff)
|
||||
|
||||
logger.info(
|
||||
"监控数据清理完成:metric_points=%s, security_checks=%s, event_logs=%s",
|
||||
result["metric_points"],
|
||||
result["security_checks"],
|
||||
result["event_logs"],
|
||||
)
|
||||
return result
|
||||
|
||||
|
||||
async def cleanup_loop() -> None:
|
||||
"""后台周期清理任务(CLEANUP_INTERVAL_HOURS=0 时不启动)"""
|
||||
interval_hours = settings.CLEANUP_INTERVAL_HOURS
|
||||
if interval_hours <= 0:
|
||||
logger.info("监控数据自动清理已禁用(CLEANUP_INTERVAL_HOURS=0)")
|
||||
return
|
||||
interval_sec = interval_hours * 3600
|
||||
logger.info("监控数据自动清理已启动,间隔 %s 小时", interval_hours)
|
||||
while True:
|
||||
try:
|
||||
# 在线程池执行同步 DB 操作,避免阻塞事件循环
|
||||
await asyncio.to_thread(cleanup_metrics)
|
||||
except Exception: # noqa: BLE001
|
||||
logger.exception("监控数据自动清理执行失败")
|
||||
await asyncio.sleep(interval_sec)
|
||||
Reference in New Issue
Block a user