Files
SycunandParallels 470eb5ead1 feat: add runtime storage cleanup with per-category retention (#310) (#313)
Runtime artifacts (agent workspaces, tool-output spill, C2 payloads, chat
uploads, workflow checkpoints, diagnostic logs) previously accumulated
without bound: most were only removed when a conversation or project was
deleted, and tmp/c2 plus workflow checkpoints were never removed at all.

Add a storage cleaner with named per-category tasks (Gitea-style), a
settings page tab, and a background sweep that is off by default so
upgrading never deletes existing data.

Safety properties, since mis-deleting live task data costs far more than
the disk saved:
- dry-run is the default; a real cleanup requires dry_run=false together
  with confirm=true at the API layer, not just a frontend dialog
- sessions active within active_grace_hours are always skipped, and a
  failed activity lookup skips conservatively (fail closed)
- directories whose conversation/project no longer exists are reclaimed
  as orphans after orphan_grace_days
- scanners never follow symlinks and every candidate path is confined to
  its category root; deletion renames to a .tmp-for-deletion marker first
  so a crash leaves recoverable residue instead of a half-deleted dir
- storage:* permissions are admin-only; without the grantSystemRolePermissions
  skip the default branch would have given operators an irreversible
  file-deletion right

Also fix two confirmed leaks: DeleteConversation left chat_uploads files
on disk (their rows already vanished via ON DELETE CASCADE), and workflow
checkpoints had no deletion path at all.

Co-authored-by: Parallels <parallels@kali-linux-2025-2.localdomain>
2026-09-25 14:46:23 +08:00

157 lines
4.7 KiB
Go

package handler
import (
"errors"
"net/http"
"strconv"
"cyberstrike-ai/internal/audit"
"cyberstrike-ai/internal/config"
"cyberstrike-ai/internal/storage"
"github.com/gin-gonic/gin"
"go.uber.org/zap"
)
// StorageHandler 提供运行空间占用统计与垃圾清理 API。
type StorageHandler struct {
cleaner *storage.Cleaner
cfg *config.Config
audit *audit.Service
logger *zap.Logger
}
// NewStorageHandler 创建存储清理 handler。
func NewStorageHandler(cleaner *storage.Cleaner, cfg *config.Config, logger *zap.Logger) *StorageHandler {
return &StorageHandler{cleaner: cleaner, cfg: cfg, logger: logger}
}
// SetAudit wires platform audit logging.
func (h *StorageHandler) SetAudit(s *audit.Service) {
if h != nil {
h.audit = s
}
}
// storageCleanupRequest 是 POST /api/storage/cleanup 的请求体。
type storageCleanupRequest struct {
// DryRun 省略时按 true 处理:只统计不删除。真正删除必须显式传 false。
DryRun *bool `json:"dry_run"`
// Confirm 为 false 时即使 dry_run=false 也拒绝执行。
// 磁盘删除不可逆,确认必须是 API 层的显式动作,而不只依赖前端弹窗。
Confirm bool `json:"confirm"`
Categories []string `json:"categories"`
}
// Meta GET /api/storage/meta 返回清理策略与各类别元信息。
func (h *StorageHandler) Meta(c *gin.Context) {
st := h.effectiveConfig()
items := make([]gin.H, 0, len(config.StorageCategoryOrder))
for _, info := range storage.DescribeCategories() {
items = append(items, gin.H{
"key": info.Key,
"label": info.Label,
"hint": info.Hint,
"enabled": st.CategoryEnabled(info.Key),
"retention_days": st.CategoryRetentionDays(info.Key),
"default_retention": config.StorageCategoryDefaults[info.Key],
})
}
c.JSON(http.StatusOK, gin.H{
"auto_clean": st.AutoCleanEffective(),
"interval_minutes": st.IntervalMinutesEffective(),
"orphan_grace_days": st.OrphanGraceDaysEffective(),
"active_grace_hours": st.ActiveGraceHoursEffective(),
"categories": items,
})
}
// Status GET /api/storage/status 返回文件系统容量与各类别占用/可回收量。
// ?refresh=1 强制重新遍历目录,否则使用短 TTL 缓存。
func (h *StorageHandler) Status(c *gin.Context) {
if h.cleaner == nil {
c.JSON(http.StatusServiceUnavailable, gin.H{"error": "存储清理未初始化"})
return
}
refresh, _ := strconv.ParseBool(c.Query("refresh"))
rep := h.cleaner.Inspect(refresh)
c.JSON(http.StatusOK, gin.H{
"filesystem": rep.Filesystem,
"categories": rep.Categories,
"totals": rep.Totals,
"scanned_at": rep.StartedAt,
"duration_ms": rep.DurationMS,
})
}
// Cleanup POST /api/storage/cleanup 执行清理(或预览)。
func (h *StorageHandler) Cleanup(c *gin.Context) {
if h.cleaner == nil {
c.JSON(http.StatusServiceUnavailable, gin.H{"error": "存储清理未初始化"})
return
}
var req storageCleanupRequest
if c.Request.ContentLength > 0 {
if err := c.ShouldBindJSON(&req); err != nil {
c.JSON(http.StatusBadRequest, gin.H{"error": "无效的请求参数: " + err.Error()})
return
}
}
dryRun := req.DryRun == nil || *req.DryRun
if !dryRun && !req.Confirm {
c.JSON(http.StatusBadRequest, gin.H{"error": "删除不可逆,执行真实清理必须同时传 dry_run=false 与 confirm=true"})
return
}
rep, err := h.cleaner.Clean(storage.CleanRequest{
DryRun: dryRun,
Categories: req.Categories,
Trigger: "manual",
})
switch {
case errors.Is(err, storage.ErrCleanupInProgress):
c.JSON(http.StatusConflict, gin.H{"error": err.Error()})
return
case errors.Is(err, storage.ErrUnknownCategory):
c.JSON(http.StatusBadRequest, gin.H{"error": err.Error()})
return
case err != nil:
c.JSON(http.StatusInternalServerError, gin.H{"error": err.Error()})
return
}
if !dryRun {
h.recordCleanup(c, rep)
}
c.JSON(http.StatusOK, gin.H{
"dry_run": rep.DryRun,
"filesystem": rep.Filesystem,
"categories": rep.Categories,
"totals": rep.Totals,
"duration_ms": rep.DurationMS,
})
}
func (h *StorageHandler) recordCleanup(c *gin.Context, rep *storage.Report) {
if h.audit == nil {
return
}
detail := map[string]interface{}{
"removed_units": rep.Totals.RemovedUnits,
"freed_bytes": rep.Totals.FreedBytes,
"errors": rep.Totals.Errors,
}
if rep.Totals.Errors > 0 {
h.audit.RecordFail(c, "storage", "cleanup", "运行空间清理完成但存在失败项", detail)
return
}
h.audit.RecordOK(c, "storage", "cleanup", "清理运行空间垃圾", "storage", "", detail)
}
func (h *StorageHandler) effectiveConfig() config.StorageConfig {
if h.cfg == nil {
return config.StorageConfig{}
}
return h.cfg.Storage
}