package artifacts import ( "bytes" "context" "crypto/sha256" "encoding/hex" "encoding/json" "fmt" "io" "net/http" "strconv" "strings" "time" "github.com/novox/mesh-controller/internal/catalogue" ) // Every archive the store keeps is held by a manifest (novox/hq issue 253, ADR 0189). // // **The store's collector marks only from manifests.** The mesh's store is a stock registry, and // its nightly `registry garbage-collect` walks every manifest in every repository, marks the blobs // those manifests name, and deletes every blob it did not mark. An image is a manifest, so what // the mesh keeps of an image survives. An archive was not: the builder put it in the store as a // bare blob — upload, then `PUT ?digest=` — and nothing in the store names it. To the collector a // bare blob is unreferenced, so the first real collection would have deleted every archive the // mesh holds, kept or not, and every machine pinning a bundle would have found it gone. The // collector runs `--dry-run` until this is true. // // **So each archive gets a holder**: the smallest OCI image manifest that names it — the empty // config, one layer, nothing else — put in the archive's own repository, by digest, untagged. The // collector marks it and so keeps the archive; the sweep lets go of an archive by deleting its // holder first, which is what lets the bytes go at the next collection. // // **Nothing a machine reads changes.** The recorded reference stays // `artifact-store:////blobs/sha256:…`, and machines fetch the blob exactly as // before. The holder is the store's bookkeeping, not a second way to reach anything. // // **Deterministic, so it never needs recording.** The holder is composed from the archive's digest // and size alone, in a fixed field order with no timestamps or annotations, so the sweep can // compute which manifest holds any archive from the reference it already has plus one HEAD for the // size. No schema change, no second record that could disagree with the store. const ( // mediaManifest is the type a holder is put and asked for as. mediaManifest = "application/vnd.oci.image.manifest.v1+json" // mediaEmpty is the OCI empty descriptor's type: a config that says nothing, for a manifest // whose only purpose is to name its layer. mediaEmpty = "application/vnd.oci.empty.v1+json" // emptyDigest is the digest of `{}`, the empty config's content, fixed by the OCI spec. emptyDigest = "sha256:44136fa355b3678a1146ad16f7e8649e94fb4fc21fe77e8310c060f61caaff8a" // mediaArchive is the layer type an archive is held as. Every archive the builder publishes is // `pack`'s gzipped tar, so this is the true type and not a placeholder — and it is a constant, // not read from anywhere, because the holder must be recomputable from the reference alone. mediaArchive = "application/vnd.oci.image.layer.v1.tar+gzip" ) // emptyConfig is the content emptyDigest names. var emptyConfig = []byte("{}") // manifestAccept is what a manifest is asked for as. A registry answers a manifest HEAD only in a // type the caller named, and answers 404 to a bare one for a manifest it holds perfectly well // (measured 2026-09-28; internal/builder/registry.go says how that was found). var manifestAccept = []string{ mediaManifest, "application/vnd.docker.distribution.manifest.v2+json", } type descriptor struct { MediaType string `json:"mediaType"` Digest string `json:"digest"` Size int64 `json:"size"` } type holderManifest struct { SchemaVersion int `json:"schemaVersion"` MediaType string `json:"mediaType"` Config descriptor `json:"config"` Layers []descriptor `json:"layers"` } // Holder is the manifest that holds an archive in the store, and its digest. // // A pure function of the archive's digest and size: the same two in give the same bytes out, // always, because `encoding/json` writes a struct's fields in their declared order and there is // nothing here that varies by when or where it was composed. func Holder(digest string, size int64) (body []byte, holder string) { body, err := json.Marshal(holderManifest{ SchemaVersion: 2, MediaType: mediaManifest, Config: descriptor{MediaType: mediaEmpty, Digest: emptyDigest, Size: int64(len(emptyConfig))}, Layers: []descriptor{{MediaType: mediaArchive, Digest: digest, Size: size}}, }) if err != nil { // Marshalling a struct of strings and integers cannot fail. panic(err) } sum := sha256.Sum256(body) return body, "sha256:" + hex.EncodeToString(sum[:]) } // Hold makes sure the store holds this archive by a manifest, and says whether it had to write one. // // Takes a reference as the mesh records it. An image is its own manifest and needs no holder, so // it answers false and nothing is asked. Idempotent: a holder already there is left alone, which // is what lets the sweep run it over every kept archive on every build and so backfill the bare // blobs published before holders existed (novox/hq issue 253). // // Gone when the store does not hold the archive at all: there is nothing to hold, and that is a // fact the caller reports rather than one this invents a remedy for. func (s Store) Hold(ctx context.Context, reference string) (bool, error) { repository, digest, archive, err := s.archive(reference) if err != nil || !archive { return false, err } size, err := s.blobSize(ctx, repository, digest) if err != nil { return false, err } return s.HoldBlob(ctx, repository, digest, size) } // Held is whether the store holds this archive by its manifest. Asks and changes nothing — the // question an operator needs answered with "none unheld" before the collector is let loose. // // An image answers true: it is its own manifest. An archive the store does not have answers Gone. func (s Store) Held(ctx context.Context, reference string) (bool, error) { repository, digest, archive, err := s.archive(reference) if err != nil { return false, err } if !archive { return true, nil } size, err := s.blobSize(ctx, repository, digest) if err != nil { return false, err } _, holder := Holder(digest, size) return s.has(ctx, s.url(repository, "manifests", holder), manifestAccept...) } // HoldBlob puts the holder for a blob of this digest and size into its repository, unless it is // there already. Answers whether it wrote one. // // The builder calls this with the size it has just uploaded; the sweep, through Hold, with the size // the store reports. Both arrive at the same holder, which is the point of composing it. func (s Store) HoldBlob(ctx context.Context, repository, digest string, size int64) (bool, error) { if s.Address == "" { return false, fmt.Errorf("this mesh has no artifact store on its network to hold %s/%s in", repository, digest) } body, holder := Holder(digest, size) there, err := s.has(ctx, s.url(repository, "manifests", holder), manifestAccept...) if err != nil { return false, err } if there { return false, nil } // The config must be in the repository before a manifest naming it is accepted: a registry // refuses a manifest whose blobs it cannot find there, which is the property that makes a // holder mean something. if err := s.putBlob(ctx, repository, emptyDigest, emptyConfig); err != nil { return false, err } // **By digest, never by tag.** A tag would be one more name to move and one more thing the // collector's `--delete-untagged` would read as meaningful; the mesh names nothing by tag that // it pins by digest, and an untagged manifest is kept by plain collection. request, err := http.NewRequestWithContext(ctx, http.MethodPut, s.url(repository, "manifests", holder), bytes.NewReader(body)) if err != nil { return false, err } request.Header.Set("Content-Type", mediaManifest) response, err := s.client().Do(request) if err != nil { return false, err } defer response.Body.Close() if response.StatusCode != http.StatusCreated { said, _ := io.ReadAll(io.LimitReader(response.Body, 4096)) return false, fmt.Errorf("the artifact store refused to hold %s/%s: %s %s", repository, digest, response.Status, strings.TrimSpace(string(said))) } return true, nil } // letGoOfHolder deletes the manifest holding an archive, before the archive's own link goes. // // **Holder first.** Deleting the blob link alone leaves a manifest still naming the blob, and the // collector would keep its bytes for ever on the strength of it — the sweep would record the // archive collected while the disk said otherwise. Deleting the holder first and failing before // the link goes leaves an unheld archive that the next sweep still offers, which is safe. // // A store that no longer has the blob answers Gone: without its size the holder cannot be named, // and without the blob there is nothing left for a holder to keep. A store that never had a holder // for it — an archive published before holders, never backfilled — answers 404 to the delete, and // that is the outcome wanted. func (s Store) letGoOfHolder(ctx context.Context, repository, digest string) error { size, err := s.blobSize(ctx, repository, digest) if err != nil { return err } _, holder := Holder(digest, size) err = s.remove(ctx, s.url(repository, "manifests", holder), repository+"/manifests/"+holder) if err == Gone { return nil } return err } // archive reads a recorded reference into its repository and digest, and whether it is an archive // at all. Refuses as ErrNotOurs anything the mesh did not put in its own store. func (s Store) archive(reference string) (repository, digest string, archive bool, err error) { path, kept := catalogue.InArtifactStore(reference) if !kept { return "", "", false, fmt.Errorf("%w: %s", ErrNotOurs, reference) } if s.Address == "" { return "", "", false, fmt.Errorf("this mesh has no artifact store on its network to ask about %s", reference) } repository, kind, digest, err := split(path) if err != nil { return "", "", false, err } return repository, digest, kind == "blobs", nil } // blobSize is how large the store says a blob is; Gone when it does not have it. func (s Store) blobSize(ctx context.Context, repository, digest string) (int64, error) { request, err := http.NewRequestWithContext(ctx, http.MethodHead, s.url(repository, "blobs", digest), nil) if err != nil { return 0, err } response, err := s.client().Do(request) if err != nil { return 0, fmt.Errorf("cannot reach the artifact store at %s: %w", s.Address, err) } defer response.Body.Close() switch response.StatusCode { case http.StatusOK: case http.StatusNotFound: return 0, Gone default: return 0, fmt.Errorf("the artifact store answered %s for %s/blobs/%s", response.Status, repository, digest) } // Read from the header rather than ContentLength: a HEAD's ContentLength is what the response // says it would have sent, which Go reports faithfully, but a proxy in between is free to drop // it, and the header is what the registry itself wrote. if length := response.Header.Get("Content-Length"); length != "" { if n, err := strconv.ParseInt(length, 10, 64); err == nil && n >= 0 { return n, nil } } if response.ContentLength >= 0 { return response.ContentLength, nil } return 0, fmt.Errorf("the artifact store holds %s/blobs/%s and will not say how large it is", repository, digest) } // putBlob uploads a small blob unless the repository already has it: ask where, then put it there // naming the digest — the registry's own two steps, the same the builder takes for an archive. func (s Store) putBlob(ctx context.Context, repository, digest string, body []byte) error { if there, err := s.has(ctx, s.url(repository, "blobs", digest)); err != nil { return err } else if there { return nil } start, err := http.NewRequestWithContext(ctx, http.MethodPost, "http://"+s.Address+"/v2/"+repository+"/blobs/uploads/", nil) if err != nil { return err } begun, err := s.client().Do(start) if err != nil { return fmt.Errorf("cannot start an upload to %s: %w", repository, err) } begun.Body.Close() if begun.StatusCode != http.StatusAccepted { return fmt.Errorf("the artifact store answered %s when asked where to put a blob in %s", begun.Status, repository) } where := begun.Header.Get("Location") if where == "" { return fmt.Errorf("the artifact store accepted an upload to %s and said nowhere to put it", repository) } if strings.HasPrefix(where, "/") { where = "http://" + s.Address + where } separator := "?" if strings.Contains(where, "?") { separator = "&" } put, err := http.NewRequestWithContext(ctx, http.MethodPut, where+separator+"digest="+digest, bytes.NewReader(body)) if err != nil { return err } put.Header.Set("Content-Type", "application/octet-stream") done, err := s.client().Do(put) if err != nil { return err } defer done.Body.Close() if done.StatusCode != http.StatusCreated { said, _ := io.ReadAll(io.LimitReader(done.Body, 4096)) return fmt.Errorf("the artifact store refused a blob in %s: %s %s", repository, done.Status, strings.TrimSpace(string(said))) } return nil } // has is whether the store answers 200 for a HEAD at that URL. func (s Store) has(ctx context.Context, url string, accept ...string) (bool, error) { request, err := http.NewRequestWithContext(ctx, http.MethodHead, url, nil) if err != nil { return false, err } for _, media := range accept { request.Header.Add("Accept", media) } response, err := s.client().Do(request) if err != nil { return false, fmt.Errorf("cannot reach the artifact store at %s: %w", s.Address, err) } defer response.Body.Close() switch response.StatusCode { case http.StatusOK: return true, nil case http.StatusNotFound: return false, nil default: return false, fmt.Errorf("the artifact store answered %s for %s", response.Status, url) } } func (s Store) url(repository, kind, digest string) string { return "http://" + s.Address + "/v2/" + repository + "/" + kind + "/" + digest } func (s Store) client() *http.Client { if s.HTTP != nil { return s.HTTP } return &http.Client{Timeout: 30 * time.Second} }