init
This commit is contained in:
@@ -0,0 +1,213 @@
|
||||
package deploy
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/iceBear67/simplepages/internal/cas"
|
||||
"github.com/iceBear67/simplepages/internal/store"
|
||||
)
|
||||
|
||||
// staleUploadAge is how long a deployment may sit unfinished before recovery
|
||||
// gives up on it. Generous on purpose: the only thing separating a CI job that
|
||||
// died from one that is uploading a large site over a slow link is how long it
|
||||
// has been, and expiring the second kind turns a slow deploy into a failed one.
|
||||
const staleUploadAge = 24 * time.Hour
|
||||
|
||||
// Recover makes the filesystem and the database agree again after a crash or an
|
||||
// unclean shutdown.
|
||||
//
|
||||
// It runs once at startup, before any listener exists and before the registry is
|
||||
// built, so it can assume it is the only thing touching either. Everything it
|
||||
// does is idempotent: being killed halfway through only means the next start
|
||||
// finds a little more to do.
|
||||
//
|
||||
// Nothing here is allowed to be fatal. A server that refuses to start because
|
||||
// one directory could not be swept is worse than one that starts and logs it —
|
||||
// the deployments themselves are already durable, and every inconsistency this
|
||||
// looks for is one the running system tolerates.
|
||||
func (s *Service) Recover(ctx context.Context) error {
|
||||
// An upload that was in progress is unreferenced by construction: a blob only
|
||||
// becomes reachable by being renamed out of the temp directory.
|
||||
if n, err := s.CAS.PurgeTemp(); err != nil {
|
||||
s.Log.Warn("could not clear interrupted uploads", "err", err)
|
||||
} else if n > 0 {
|
||||
s.Log.Info("cleared interrupted uploads", "count", n)
|
||||
}
|
||||
|
||||
if err := s.recoverBlobs(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Then the deployments that were still being written when the process went
|
||||
// away, so their manifest rows are gone before the blob collector next runs
|
||||
// and can reclaim whatever only they referenced.
|
||||
n, err := s.DB.ExpireStaleDeployments(ctx, time.Now().Add(-staleUploadAge),
|
||||
"abandoned: no activity for "+staleUploadAge.String())
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if n > 0 {
|
||||
s.Log.Info("expired unfinished deployments", "count", n)
|
||||
}
|
||||
|
||||
if err := s.resumeDeletions(ctx); err != nil {
|
||||
return err
|
||||
}
|
||||
return s.sweepTrees(ctx)
|
||||
}
|
||||
|
||||
// recoverBlobs finds content the database believes is on disk and is not.
|
||||
//
|
||||
// This is the "restored the database, lost the disk" case, and also what a
|
||||
// half-finished collector sweep leaves behind. Marking the rows absent is
|
||||
// enough to fix it: the next deploy naming one of these digests is asked to
|
||||
// upload it, and every deployment that referenced it becomes deployable again as
|
||||
// soon as one does.
|
||||
func (s *Service) recoverBlobs(ctx context.Context) error {
|
||||
var absent []cas.Digest
|
||||
err := s.DB.EachPresentBlob(ctx, func(d cas.Digest) error {
|
||||
has, err := s.CAS.Has(d)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if !has {
|
||||
absent = append(absent, d)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(absent) == 0 {
|
||||
return nil
|
||||
}
|
||||
if err := s.DB.MarkBlobsAbsent(ctx, absent); err != nil {
|
||||
return err
|
||||
}
|
||||
// Loud, because on a healthy server this number is zero. Anything else means
|
||||
// the content store lost data, and an operator wants to hear about it before
|
||||
// a deploy fails for a reason that looks like the client's fault.
|
||||
s.Log.Warn("content is missing from the store and was marked for re-upload",
|
||||
"blobs", len(absent))
|
||||
return nil
|
||||
}
|
||||
|
||||
// resumeDeletions finishes what a collector sweep was doing when it stopped.
|
||||
//
|
||||
// A deployment enters the deleting state before anything of it is removed, so a
|
||||
// row still in that state is a tree that may be half gone. Half a tree is
|
||||
// exactly what nothing may serve, which is why the state is committed first: it
|
||||
// makes an interrupted deletion recognisable rather than indistinguishable from
|
||||
// a healthy deployment.
|
||||
func (s *Service) resumeDeletions(ctx context.Context) error {
|
||||
deps, err := s.DB.DeploymentsInState(ctx, store.StateDeleting, 0)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, dep := range deps {
|
||||
if err := s.removeDeployment(ctx, dep); err != nil {
|
||||
s.Log.Warn("could not finish deleting a deployment",
|
||||
"deployment", dep.PublicID, "err", err)
|
||||
continue
|
||||
}
|
||||
s.Log.Info("finished deleting a deployment", "deployment", dep.PublicID)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// removeDeployment takes a claimed deployment the rest of the way: its tree
|
||||
// first, then its rows. Tree before rows, because a row without a tree is a
|
||||
// deployment that simply cannot be activated, whereas a tree without a row is
|
||||
// disk nobody will ever account for again.
|
||||
func (s *Service) removeDeployment(ctx context.Context, dep *store.Deployment) error {
|
||||
if s.Dir != "" {
|
||||
dir := DeploymentDir(s.Dir, dep.ProjectID, dep.PublicID)
|
||||
if err := os.RemoveAll(dir); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := os.RemoveAll(dir + stagingSuffix); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return s.DB.DeleteDeployment(ctx, dep.ID)
|
||||
}
|
||||
|
||||
// sweepTrees removes assembled trees that nothing refers to.
|
||||
//
|
||||
// Two kinds: staging directories, which are by definition a build that never
|
||||
// finished, and directories whose deployment row is gone — the reverse of the
|
||||
// blob audit above, and the residue of a deletion that removed rows before it
|
||||
// removed files, or of a database restored from an older backup than the disk.
|
||||
func (s *Service) sweepTrees(ctx context.Context) error {
|
||||
if s.Dir == "" {
|
||||
return nil
|
||||
}
|
||||
refs, err := s.DB.AllDeploymentRefs(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
known := make(map[store.DeploymentRef]struct{}, len(refs))
|
||||
for _, r := range refs {
|
||||
known[r] = struct{}{}
|
||||
}
|
||||
|
||||
projects, err := os.ReadDir(s.Dir)
|
||||
if err != nil {
|
||||
if errors.Is(err, fs.ErrNotExist) {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
var staging, orphans int
|
||||
for _, pe := range projects {
|
||||
// Only the layout this package writes is ever considered for removal:
|
||||
// one directory per project id, named by the id. Anything else under
|
||||
// $DATA_DIR/deployments was put there by someone else and is left alone.
|
||||
projectID, err := strconv.ParseInt(pe.Name(), 10, 64)
|
||||
if err != nil || !pe.IsDir() {
|
||||
continue
|
||||
}
|
||||
projectDir := filepath.Join(s.Dir, pe.Name())
|
||||
entries, err := os.ReadDir(projectDir)
|
||||
if err != nil {
|
||||
s.Log.Warn("could not read a project's deployment directory", "dir", projectDir, "err", err)
|
||||
continue
|
||||
}
|
||||
for _, e := range entries {
|
||||
if !e.IsDir() {
|
||||
continue
|
||||
}
|
||||
name := e.Name()
|
||||
target := filepath.Join(projectDir, name)
|
||||
switch {
|
||||
case strings.HasSuffix(name, stagingSuffix):
|
||||
if err := os.RemoveAll(target); err != nil {
|
||||
s.Log.Warn("could not remove a staging directory", "dir", target, "err", err)
|
||||
continue
|
||||
}
|
||||
staging++
|
||||
default:
|
||||
if _, ok := known[store.DeploymentRef{ProjectID: projectID, PublicID: name}]; ok {
|
||||
continue
|
||||
}
|
||||
if err := os.RemoveAll(target); err != nil {
|
||||
s.Log.Warn("could not remove an orphaned deployment tree", "dir", target, "err", err)
|
||||
continue
|
||||
}
|
||||
orphans++
|
||||
}
|
||||
}
|
||||
}
|
||||
if staging > 0 || orphans > 0 {
|
||||
s.Log.Info("swept unreferenced deployment trees", "staging", staging, "orphaned", orphans)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
Reference in New Issue
Block a user