init
This commit is contained in:
@@ -0,0 +1,259 @@
|
||||
package deploy
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/iceBear67/simplepages/api"
|
||||
"github.com/iceBear67/simplepages/internal/store"
|
||||
)
|
||||
|
||||
const (
|
||||
// defaultBlobGrace is how long a blob must have been unreferenced before
|
||||
// its content is removed.
|
||||
//
|
||||
// It is what makes the read path safe without reference counting requests.
|
||||
// A handler resolves a digest from its snapshot and then opens it; between
|
||||
// those two instants the collector could in principle delete the file. An
|
||||
// hour is an enormous margin for a gap that is measured in microseconds,
|
||||
// and it costs only some disk that was going to be reclaimed anyway.
|
||||
defaultBlobGrace = time.Hour
|
||||
|
||||
// failedRetention is how long a failed deployment's row is kept. Its
|
||||
// manifest is already gone, so this is purely so that an operator
|
||||
// investigating a broken CI job can still see that it failed and why.
|
||||
failedRetention = 24 * time.Hour
|
||||
)
|
||||
|
||||
// Collect runs one garbage collection pass: retention first, then the content
|
||||
// nothing references any more.
|
||||
//
|
||||
// Two passes in that order, and not one, because the first is what makes work
|
||||
// for the second. Deleting a deployment drops its manifest rows, the delete
|
||||
// trigger takes each blob's refcount down, and only then can a blob be seen to
|
||||
// be unreferenced. The second pass will not act on those blobs in this same
|
||||
// run — the trigger sets last_ref_at to now and the grace period has not
|
||||
// elapsed — which is deliberate: content that just became unreachable is
|
||||
// exactly the content some in-flight request is most likely to still be
|
||||
// reading.
|
||||
//
|
||||
// A dry run reports the deployments that would be deleted and the blobs that
|
||||
// are collectable right now. It cannot report the blobs the deletions would
|
||||
// free, because nothing has been deleted; the number is a floor, not an
|
||||
// estimate, and it is honest about being one.
|
||||
func (s *Service) Collect(ctx context.Context, dryRun bool) (api.GCStats, error) {
|
||||
stats := api.GCStats{DryRun: dryRun}
|
||||
|
||||
if !dryRun {
|
||||
// Abandoned uploads first, so that whatever only they referenced is
|
||||
// already unreferenced by the time the blob pass looks.
|
||||
n, err := s.DB.ExpireStaleDeployments(ctx, time.Now().Add(-staleUploadAge),
|
||||
"abandoned: no activity for "+staleUploadAge.String())
|
||||
if err != nil {
|
||||
return stats, err
|
||||
}
|
||||
if n > 0 {
|
||||
s.Log.InfoContext(ctx, "expired unfinished deployments", "count", n)
|
||||
}
|
||||
}
|
||||
|
||||
projects, err := s.DB.AllProjects(ctx)
|
||||
if err != nil {
|
||||
return stats, err
|
||||
}
|
||||
for _, p := range projects {
|
||||
n, err := s.collectProject(ctx, p, dryRun)
|
||||
stats.DeploymentsDeleted += n
|
||||
if err != nil {
|
||||
return stats, err
|
||||
}
|
||||
}
|
||||
|
||||
blobs, err := s.DB.UnreferencedBlobs(ctx, time.Now().Add(-s.blobGrace()), 0)
|
||||
if err != nil {
|
||||
return stats, err
|
||||
}
|
||||
for _, b := range blobs {
|
||||
if dryRun {
|
||||
stats.BlobsDeleted++
|
||||
stats.BytesFreed += b.Size
|
||||
continue
|
||||
}
|
||||
deleted, err := s.DB.DeleteBlob(ctx, b.Digest, func() error { return s.CAS.Remove(b.Digest) })
|
||||
if err != nil {
|
||||
return stats, err
|
||||
}
|
||||
// Not deleted means the blob was referenced again between the listing
|
||||
// and the delete — a new deployment naming content that was about to be
|
||||
// collected. Leaving it alone is the whole point of rechecking the
|
||||
// refcount inside the transaction.
|
||||
if deleted {
|
||||
stats.BlobsDeleted++
|
||||
stats.BytesFreed += b.Size
|
||||
}
|
||||
}
|
||||
return stats, nil
|
||||
}
|
||||
|
||||
// collectProject applies one project's retention policy.
|
||||
//
|
||||
// The active deployment is not considered at all: it is excluded by the query,
|
||||
// so no counting mistake here can reach it.
|
||||
func (s *Service) collectProject(ctx context.Context, p *store.Project, dryRun bool) (int, error) {
|
||||
deps, err := s.DB.InactiveDeployments(ctx, p.ID)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
grace := time.Duration(p.RetentionGraceS) * time.Second
|
||||
now := time.Now()
|
||||
|
||||
var deleted, kept int
|
||||
for _, dep := range deps {
|
||||
switch dep.State {
|
||||
case store.StateReady:
|
||||
// Newest first, so the first RetentionCount of them are the ones
|
||||
// worth keeping for a rollback.
|
||||
if kept < p.RetentionCount {
|
||||
kept++
|
||||
continue
|
||||
}
|
||||
if now.Sub(retiredAt(dep)) < grace {
|
||||
continue
|
||||
}
|
||||
case store.StateFailed:
|
||||
if now.Sub(dep.CreatedAt) < failedRetention {
|
||||
continue
|
||||
}
|
||||
case store.StateDeleting:
|
||||
// Already claimed by a sweep that did not finish. No grace applies:
|
||||
// nothing may serve a tree that is half removed.
|
||||
default:
|
||||
// pending or uploading. Someone may still be uploading to it, and
|
||||
// ExpireStaleDeployments is what decides when they are not.
|
||||
continue
|
||||
}
|
||||
if dryRun {
|
||||
deleted++
|
||||
continue
|
||||
}
|
||||
if err := s.claim(ctx, p, dep); err != nil {
|
||||
switch {
|
||||
case errors.Is(err, store.ErrNotFound):
|
||||
// Deleted by someone else between the listing and now.
|
||||
case errors.Is(err, store.ErrConflict):
|
||||
// Activated between the listing and now — a rollback landed on
|
||||
// a deployment retention had picked. Correct outcome: the claim
|
||||
// is refused and the deployment stays.
|
||||
s.Log.InfoContext(ctx, "skipped a deployment that was activated during collection",
|
||||
"project", p.Name, "deployment", dep.PublicID)
|
||||
default:
|
||||
return deleted, err
|
||||
}
|
||||
continue
|
||||
}
|
||||
if err := s.removeDeployment(ctx, dep); err != nil {
|
||||
// The row is in the deleting state and committed, so recovery or
|
||||
// the next sweep will finish it. Nothing serves it in the meantime.
|
||||
s.Log.WarnContext(ctx, "could not finish deleting a deployment",
|
||||
"project", p.Name, "deployment", dep.PublicID, "err", err)
|
||||
continue
|
||||
}
|
||||
deleted++
|
||||
}
|
||||
return deleted, nil
|
||||
}
|
||||
|
||||
// retiredAt is when a deployment stopped being served, or when it was created
|
||||
// if it never was. It is what the retention grace is measured from.
|
||||
func retiredAt(dep *store.Deployment) time.Time {
|
||||
if dep.DeactivatedAt != nil {
|
||||
return *dep.DeactivatedAt
|
||||
}
|
||||
return dep.CreatedAt
|
||||
}
|
||||
|
||||
// claim marks a deployment as being deleted, which is what makes it safe to
|
||||
// start removing files.
|
||||
//
|
||||
// The project lock is held for exactly this statement. Activation takes the
|
||||
// same lock and re-reads the row under it, so the two orderings are the only
|
||||
// possible ones: either activation commits first and this fails on the
|
||||
// active = 0 condition, or this commits first and activation finds a deployment
|
||||
// in the deleting state and refuses it. There is no interleaving in which a
|
||||
// tree is removed from underneath a deployment that has just become active.
|
||||
func (s *Service) claim(ctx context.Context, p *store.Project, dep *store.Deployment) error {
|
||||
unlock, err := s.locks.lock(ctx, p.ID)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer unlock()
|
||||
return s.DB.MarkDeploymentDeleting(ctx, dep.ID)
|
||||
}
|
||||
|
||||
// Delete removes one deployment on request. The active one cannot be deleted:
|
||||
// that is a conflict, not a permission problem, and the client is expected to
|
||||
// activate something else first.
|
||||
func (s *Service) Delete(ctx context.Context, p *store.Project, dep *store.Deployment) error {
|
||||
if err := s.claim(ctx, p, dep); err != nil {
|
||||
switch {
|
||||
case errors.Is(err, store.ErrNotFound):
|
||||
return api.Errorf(api.CodeNotFound, "no such deployment")
|
||||
case errors.Is(err, store.ErrConflict):
|
||||
return api.Errorf(api.CodeDeploymentActive,
|
||||
"this deployment is the one the project is serving; activate another one first")
|
||||
}
|
||||
return err
|
||||
}
|
||||
if err := s.removeDeployment(ctx, dep); err != nil {
|
||||
return err
|
||||
}
|
||||
s.Log.InfoContext(ctx, "deployment deleted", "project", p.Name, "deployment", dep.PublicID)
|
||||
return nil
|
||||
}
|
||||
|
||||
// RemoveProjectTrees deletes what a project left on disk once its rows are
|
||||
// gone. Its blobs are freed by the same cascade and collected on the next pass.
|
||||
func (s *Service) RemoveProjectTrees(projectID int64) error {
|
||||
if s.Dir == "" {
|
||||
return nil
|
||||
}
|
||||
return os.RemoveAll(filepath.Join(s.Dir, strconv.FormatInt(projectID, 10)))
|
||||
}
|
||||
|
||||
// RunCollector collects on a timer until ctx is done. A failed pass is logged
|
||||
// and the next one runs as scheduled: everything the collector does is
|
||||
// idempotent, so there is nothing to unwind and no reason to stop.
|
||||
func (s *Service) RunCollector(ctx context.Context, every time.Duration) {
|
||||
if every <= 0 {
|
||||
return
|
||||
}
|
||||
t := time.NewTicker(every)
|
||||
defer t.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-t.C:
|
||||
stats, err := s.Collect(ctx, false)
|
||||
if err != nil {
|
||||
s.Log.WarnContext(ctx, "garbage collection did not finish", "err", err)
|
||||
}
|
||||
if stats.DeploymentsDeleted > 0 || stats.BlobsDeleted > 0 {
|
||||
s.Log.InfoContext(ctx, "collected",
|
||||
"deployments", stats.DeploymentsDeleted,
|
||||
"blobs", stats.BlobsDeleted, "bytes", stats.BytesFreed)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (s *Service) blobGrace() time.Duration {
|
||||
if s.BlobGrace == 0 {
|
||||
return defaultBlobGrace
|
||||
}
|
||||
return s.BlobGrace
|
||||
}
|
||||
Reference in New Issue
Block a user