fix: improve handling of restore operations

- restore operations are split into a new flow
 - added support displaying restore operation percentage and other
   details in tree view
This commit is contained in:
garethgeorge
2024-07-02 20:51:09 -07:00
parent ca6807e81d
commit 8992e4e75f
11 changed files with 142 additions and 122 deletions
+1 -6
View File
@@ -400,12 +400,7 @@ func (s *BackrestHandler) Restore(ctx context.Context, req *connect.Request[v1.R
}
at := time.Now()
flowID, err := tasks.FlowIDForSnapshotID(s.oplog, req.Msg.SnapshotId)
if err != nil {
return nil, fmt.Errorf("failed to get flow ID for snapshot %q: %w", req.Msg.SnapshotId, err)
}
s.orchestrator.ScheduleTask(tasks.NewOneoffRestoreTask(req.Msg.RepoId, req.Msg.PlanId, flowID, at, req.Msg.SnapshotId, req.Msg.Path, req.Msg.Target), tasks.TaskPriorityInteractive+tasks.TaskPriorityDefault)
s.orchestrator.ScheduleTask(tasks.NewOneoffRestoreTask(req.Msg.RepoId, req.Msg.PlanId, 0 /* flowID */, at, req.Msg.SnapshotId, req.Msg.Path, req.Msg.Target), tasks.TaskPriorityInteractive+tasks.TaskPriorityDefault)
return connect.NewResponse(&emptypb.Empty{}), nil
}
@@ -13,16 +13,20 @@ import (
const (
gcStartupDelay = 60 * time.Second
gcInterval = 24 * time.Hour
// keep operations that are eligible for gc for 30 days OR up to a limit of 100 for any one plan.
// an operation is eligible for gc if:
// - it has no snapshot associated with it
// - it has a forgotten snapshot associated with it
gcHistoryAge = 30 * 24 * time.Hour
gcHistoryMaxCount = 1000
// keep stats operations for 1 year (they're small and useful for long term trends)
gcHistoryStatsAge = 365 * 24 * time.Hour
)
// gcAgeForOperation returns the age at which an operation is eligible for garbage collection.
func gcAgeForOperation(op *v1.Operation) time.Duration {
switch op.Op.(type) {
// stats, check, and prune operations are kept for a year
case *v1.Operation_OperationStats, *v1.Operation_OperationCheck, *v1.Operation_OperationPrune:
return 365 * 24 * time.Hour
// all other operations are kept for 30 days
default:
return 30 * 24 * time.Hour
}
}
type CollectGarbageTask struct {
BaseTask
firstRun bool
@@ -83,11 +87,8 @@ func (t *CollectGarbageTask) gcOperations(oplog *oplog.OpLog) error {
forgot, ok := snapshotForgottenForFlow[op.FlowId]
if !ok {
// no snapshot associated with this flow; check if it's old enough to be gc'd
maxAgeForType := gcHistoryAge.Milliseconds()
if _, isStats := op.Op.(*v1.Operation_OperationStats); isStats {
maxAgeForType = gcHistoryStatsAge.Milliseconds()
}
if curTime-op.UnixTimeStartMs > maxAgeForType {
maxAgeForOperation := gcAgeForOperation(op)
if curTime-op.UnixTimeStartMs > maxAgeForOperation.Milliseconds() {
forgetIDs = append(forgetIDs, op.Id)
}
} else if forgot {
@@ -107,17 +108,3 @@ func (t *CollectGarbageTask) gcOperations(oplog *oplog.OpLog) error {
zap.Any("operations_removed", len(forgetIDs)))
return nil
}
func (t *CollectGarbageTask) Cancel(withStatus v1.OperationStatus) error {
return nil
}
func (t *CollectGarbageTask) OperationId() int64 {
return 0
}
type gcOpInfo struct {
id int64 // operation ID
timestamp int64 // unix time milliseconds
isStats bool // true if this is a stats operation
}
+2 -2
View File
@@ -77,7 +77,7 @@ func restoreHelper(ctx context.Context, st ScheduledTask, taskRunner TaskRunner,
zap.S().Infof("restore progress: %v", entry)
restoreOp.Status = entry
restoreOp.LastStatus = entry
sendWg.Add(1)
go func() {
@@ -91,7 +91,7 @@ func restoreHelper(ctx context.Context, st ScheduledTask, taskRunner TaskRunner,
if err != nil {
return fmt.Errorf("restore failed: %w", err)
}
restoreOp.Status = summary
restoreOp.LastStatus = summary
return nil
}