Initial commit

This commit is contained in:
2026-09-12 23:26:07 +02:00
commit a180fe4b52
35 changed files with 4921 additions and 0 deletions
+68
View File
@@ -0,0 +1,68 @@
package store
import (
"crypto/rand"
"encoding/hex"
"errors"
"regexp"
"strings"
)
// vanityRe is deliberately narrow: lowercase alphanumerics plus dot, dash and
// underscore, starting with an alphanumeric, 2-64 characters. Anything that
// could be mistaken for a path element, a dotfile or a traversal is excluded.
var vanityRe = regexp.MustCompile(`^[a-z0-9][a-z0-9._-]{1,63}$`)
// reserved names would shadow a route or a well-known file if they were ever
// allowed into the object namespace.
var reserved = map[string]bool{
"d": true, "i": true, "api": true, "static": true,
"favicon.ico": true, "robots.txt": true, "index.html": true,
"sitemap.xml": true, "tokens.json": true, "objects": true,
}
var ErrBadID = errors.New("invalid name")
// CleanID validates an id arriving from a URL or from a vanity request and
// returns its canonical form. IDs are lowercased so that a case-insensitive
// filesystem cannot be tricked into treating two distinct names as one object.
//
// This is the *only* function permitted to turn caller input into a path
// element; every filesystem path in this package is built from its output.
func CleanID(s string) (string, error) {
s = strings.ToLower(strings.TrimSpace(s))
if !vanityRe.MatchString(s) {
return "", ErrBadID
}
// The regexp permits interior dots; a doubled dot or a trailing dot is
// still refused so no spelling of a traversal survives.
if strings.Contains(s, "..") || strings.HasSuffix(s, ".") {
return "", ErrBadID
}
if reserved[s] {
return "", ErrBadID
}
return s, nil
}
// NewUUID returns a random RFC 4122 version 4 UUID.
func NewUUID() (string, error) {
var b [16]byte
if _, err := rand.Read(b[:]); err != nil {
return "", err
}
b[6] = (b[6] & 0x0f) | 0x40 // version 4
b[8] = (b[8] & 0x3f) | 0x80 // variant 10
h := hex.EncodeToString(b[:])
return h[:8] + "-" + h[8:12] + "-" + h[12:16] + "-" + h[16:20] + "-" + h[20:], nil
}
// NewSecret returns a high-entropy URL-safe secret, used for both API tokens
// and per-object delete tokens.
func NewSecret() (string, error) {
var b [32]byte
if _, err := rand.Read(b[:]); err != nil {
return "", err
}
return hex.EncodeToString(b[:]), nil
}
+76
View File
@@ -0,0 +1,76 @@
package store
import (
"strings"
"time"
"unicode/utf8"
)
// Meta is the flat per-object record stored alongside the blob. Its presence on
// disk is what makes an object visible; an object directory without one is
// either mid-upload or crash debris.
type Meta struct {
ID string `json:"id"`
Filename string `json:"filename"`
Size int64 `json:"size"`
SHA256 string `json:"sha256"`
Created time.Time `json:"created"`
Expires *time.Time `json:"expires"` // nil means never
Owner string `json:"owner"` // "" means anonymous
Vanity bool `json:"vanity"`
// DeleteHash is the SHA-256 of the delete token handed to the uploader.
// The token itself is shown once and never stored.
DeleteHash string `json:"delete_hash"`
}
// Expired reports whether the object's lifetime has run out.
func (m *Meta) Expired(now time.Time) bool {
return m.Expires != nil && !now.Before(*m.Expires)
}
const fallbackFilename = "download.bin"
// maxFilenameBytes matches the common filesystem limit; the name is only ever
// metadata here, but keeping it bounded keeps headers and pages sane.
const maxFilenameBytes = 255
// SanitizeFilename reduces a caller-supplied filename to something safe to put
// in a Content-Disposition header and to show on a page.
//
// The result is never used to build a path - paths come from CleanID alone -
// so this guards against header injection and display confusion rather than
// traversal. Separators are stripped regardless, so that a name surviving to
// some future code path cannot carry a directory with it.
func SanitizeFilename(name string) string {
// Take the last element under either separator convention: browsers on
// Windows have historically sent full paths.
if i := strings.LastIndexAny(name, `/\`); i >= 0 {
name = name[i+1:]
}
if !utf8.ValidString(name) {
name = strings.ToValidUTF8(name, "")
}
name = strings.Map(func(r rune) rune {
switch {
case r < 0x20, r == 0x7f: // control characters, CR and LF included
return -1
case r == '/', r == '\\', r == 0:
return -1
}
return r
}, name)
name = strings.TrimSpace(name)
if len(name) > maxFilenameBytes {
name = name[:maxFilenameBytes]
// Do not leave a partial rune at the end.
for len(name) > 0 && !utf8.ValidString(name) {
name = name[:len(name)-1]
}
}
if name == "" || name == "." || name == ".." {
return fallbackFilename
}
return name
}
+406
View File
@@ -0,0 +1,406 @@
// Package store implements the flat-file object store: one directory per
// object, holding the blob and a JSON metadata sidecar. There is no database;
// an in-memory index is rebuilt from disk at startup and kept in sync.
package store
import (
"crypto/sha256"
"encoding/hex"
"encoding/json"
"errors"
"fmt"
"hash"
"io"
"io/fs"
"os"
"path/filepath"
"sync"
"time"
)
const (
blobName = "blob"
partName = "blob.part"
metaName = "meta.json"
dirPerm fs.FileMode = 0o775
filePerm fs.FileMode = 0o664
// debrisMaxAge is how long an object directory with no metadata is left
// alone before being treated as the remains of a killed upload.
debrisMaxAge = 24 * time.Hour
)
var (
ErrNotFound = errors.New("object not found")
ErrExists = errors.New("name already taken")
ErrTooLarge = errors.New("upload exceeds the size limit")
)
// Store owns the data directory.
type Store struct {
dir string
objects string
// root confines every object file operation to the objects directory.
// The data directory is group-writable by design, so a symlink planted
// there must not be able to redirect a read or a write outside it.
root *os.Root
mu sync.RWMutex
index map[string]*Meta
total int64
}
// Open prepares the data directory and rebuilds the index from it.
func Open(dir string) (*Store, error) {
s := &Store{
dir: dir,
objects: filepath.Join(dir, "objects"),
index: make(map[string]*Meta),
}
if err := os.MkdirAll(s.objects, dirPerm); err != nil {
return nil, err
}
root, err := os.OpenRoot(s.objects)
if err != nil {
return nil, err
}
s.root = root
if err := s.load(); err != nil {
return nil, err
}
return s, nil
}
// DataDir is the directory the store was opened on.
func (s *Store) DataDir() string { return s.dir }
// Close releases the handle on the objects directory.
func (s *Store) Close() error { return s.root.Close() }
// objectDir builds an object's path for display. Actual file operations go
// through s.root instead, which cannot be walked out of.
func (s *Store) objectDir(id string) string { return filepath.Join(s.objects, id) }
// within builds a root-relative path for one of an object's files.
func within(id, name string) string { return id + "/" + name }
func (s *Store) load() error {
entries, err := os.ReadDir(s.objects)
if err != nil {
return err
}
for _, e := range entries {
if !e.IsDir() {
continue
}
id, err := CleanID(e.Name())
if err != nil || id != e.Name() {
// Not a name this service could have created; leave it be.
continue
}
m, err := s.readMeta(id)
if err != nil {
continue // incomplete or unreadable; the debris sweep handles it
}
// A leftover .part in a committed object is always stale at startup.
s.root.Remove(within(id, partName))
s.index[id] = m
s.total += m.Size
}
return nil
}
func (s *Store) readMeta(id string) (*Meta, error) {
b, err := s.root.ReadFile(within(id, metaName))
if err != nil {
return nil, err
}
var m Meta
if err := json.Unmarshal(b, &m); err != nil {
return nil, err
}
if m.ID != id {
return nil, fmt.Errorf("store: metadata for %q claims id %q", id, m.ID)
}
return &m, nil
}
// Exists reports whether a name is currently taken, expired objects included:
// a name stays claimed until its object is actually removed.
func (s *Store) Exists(id string) bool {
_, err := s.root.Lstat(id)
return err == nil
}
// Reserve claims a name by creating its directory. os.Mkdir is atomic, so this
// is the point at which a vanity collision is detected - before any of the
// caller's body has been read.
func (s *Store) Reserve(id string) (*Upload, error) {
if err := s.root.Mkdir(id, dirPerm); err != nil {
if errors.Is(err, fs.ErrExist) {
return nil, ErrExists
}
return nil, err
}
f, err := s.root.OpenFile(within(id, partName), os.O_WRONLY|os.O_CREATE|os.O_EXCL, filePerm)
if err != nil {
s.root.RemoveAll(id)
return nil, err
}
return &Upload{s: s, id: id, f: f, h: sha256.New()}, nil
}
// ReserveRandom claims a fresh UUIDv4 name.
func (s *Store) ReserveRandom() (*Upload, error) {
for range 8 {
id, err := NewUUID()
if err != nil {
return nil, err
}
u, err := s.Reserve(id)
if errors.Is(err, ErrExists) {
continue // astronomically unlikely; retry regardless
}
return u, err
}
return nil, errors.New("store: could not allocate an unused id")
}
// Upload is an in-flight object. It is an io.Writer so callers can stream a
// request body straight to disk; nothing is ever buffered in memory.
type Upload struct {
s *Store
id string
f *os.File
h hash.Hash
n int64
limit int64 // 0 means unlimited
done bool
}
func (u *Upload) ID() string { return u.id }
func (u *Upload) Size() int64 { return u.n }
// SetLimit caps the number of bytes the upload will accept. The cap is applied
// to bytes actually written, never to a declared Content-Length.
func (u *Upload) SetLimit(n int64) { u.limit = n }
func (u *Upload) Write(p []byte) (int, error) {
if u.limit > 0 && u.n+int64(len(p)) > u.limit {
return 0, ErrTooLarge
}
n, err := u.f.Write(p)
u.n += int64(n)
u.h.Write(p[:n])
return n, err
}
// Commit makes the object visible. The ordering matters: the blob is durable
// and in place before the metadata that advertises it is written, and the
// metadata is renamed into place atomically.
func (u *Upload) Commit(m *Meta) error {
if u.done {
return errors.New("store: upload already finished")
}
m.ID = u.id
m.Size = u.n
m.SHA256 = hex.EncodeToString(u.h.Sum(nil))
if err := u.f.Sync(); err != nil {
return err
}
if err := u.f.Close(); err != nil {
return err
}
if err := u.s.root.Rename(within(u.id, partName), within(u.id, blobName)); err != nil {
return err
}
if err := u.s.writeMetaAtomic(u.id, m); err != nil {
return err
}
if err := u.s.syncDir(u.id); err != nil {
return err
}
u.done = true
u.s.mu.Lock()
u.s.index[u.id] = m
u.s.total += m.Size
u.s.mu.Unlock()
return nil
}
// Abort discards an incomplete upload, releasing its name.
func (u *Upload) Abort() {
if u.done {
return
}
u.done = true
u.f.Close()
u.s.root.RemoveAll(u.id)
}
// writeMetaAtomic serialises m to a temporary file in the object's own
// directory, fsyncs it, and renames it into place. Only once this rename lands
// does the object become visible to a reader.
func (s *Store) writeMetaAtomic(id string, m *Meta) error {
b, err := json.MarshalIndent(m, "", " ")
if err != nil {
return err
}
b = append(b, '\n')
suffix, err := NewSecret()
if err != nil {
return err
}
tmpPath := within(id, "."+metaName+"."+suffix[:16])
tmp, err := s.root.OpenFile(tmpPath, os.O_WRONLY|os.O_CREATE|os.O_EXCL, filePerm)
if err != nil {
return err
}
defer s.root.Remove(tmpPath) // no-op once the rename succeeds
if _, err := tmp.Write(b); err != nil {
tmp.Close()
return err
}
if err := tmp.Sync(); err != nil {
tmp.Close()
return err
}
if err := tmp.Close(); err != nil {
return err
}
return s.root.Rename(tmpPath, within(id, metaName))
}
// syncDir flushes a directory entry so a rename survives a power loss.
func (s *Store) syncDir(id string) error {
d, err := s.root.Open(id)
if err != nil {
return err
}
defer d.Close()
return d.Sync()
}
// Get returns an object's metadata, treating an expired object as absent and
// removing it on the spot. Expiry is checked here, on every read, so a stalled
// sweeper can never serve a file past its lifetime.
func (s *Store) Get(id string, now time.Time) (*Meta, error) {
s.mu.RLock()
m, ok := s.index[id]
s.mu.RUnlock()
if !ok {
return nil, ErrNotFound
}
if m.Expired(now) {
s.Delete(id)
return nil, ErrNotFound
}
return m, nil
}
// OpenBlob returns the metadata and an open handle to the object's bytes.
func (s *Store) OpenBlob(id string, now time.Time) (*Meta, *os.File, error) {
m, err := s.Get(id, now)
if err != nil {
return nil, nil, err
}
f, err := s.root.Open(within(id, blobName))
if err != nil {
// Metadata without a blob means the data directory was tampered with.
s.Delete(id)
return nil, nil, ErrNotFound
}
return m, f, nil
}
// Delete removes an object and frees its name.
func (s *Store) Delete(id string) error {
s.mu.Lock()
if m, ok := s.index[id]; ok {
s.total -= m.Size
delete(s.index, id)
}
s.mu.Unlock()
return s.root.RemoveAll(id)
}
// Total reports the number of bytes currently stored.
func (s *Store) Total() int64 {
s.mu.RLock()
defer s.mu.RUnlock()
return s.total
}
// Count reports the number of live objects.
func (s *Store) Count() int {
s.mu.RLock()
defer s.mu.RUnlock()
return len(s.index)
}
// Sweep removes expired objects and long-abandoned upload directories. It
// returns the number of objects removed.
func (s *Store) Sweep(now time.Time) int {
s.mu.RLock()
var expired []string
for id, m := range s.index {
if m.Expired(now) {
expired = append(expired, id)
}
}
s.mu.RUnlock()
for _, id := range expired {
s.Delete(id)
}
s.sweepDebris(now)
return len(expired)
}
// sweepDebris removes object directories that never gained metadata and are
// older than debrisMaxAge - the remains of an upload killed mid-flight.
func (s *Store) sweepDebris(now time.Time) {
entries, err := os.ReadDir(s.objects)
if err != nil {
return
}
for _, e := range entries {
if !e.IsDir() {
continue
}
s.mu.RLock()
_, live := s.index[e.Name()]
s.mu.RUnlock()
if live {
continue
}
info, err := e.Info()
if err != nil || now.Sub(info.ModTime()) < debrisMaxAge {
continue
}
if _, err := s.root.Stat(within(e.Name(), metaName)); err == nil {
continue // has metadata but is not indexed; leave it for a human
}
s.root.RemoveAll(e.Name())
}
}
// List returns every live object's metadata, for administrative use.
func (s *Store) List() []*Meta {
s.mu.RLock()
defer s.mu.RUnlock()
out := make([]*Meta, 0, len(s.index))
for _, m := range s.index {
out = append(out, m)
}
return out
}
var _ io.Writer = (*Upload)(nil)
+294
View File
@@ -0,0 +1,294 @@
package store
import (
"crypto/sha256"
"encoding/hex"
"os"
"path/filepath"
"strings"
"testing"
"time"
)
func TestCleanID(t *testing.T) {
valid := []string{"my-file", "a1", "godot.zip", "a_b.c-d", "ABC"}
for _, in := range valid {
got, err := CleanID(in)
if err != nil {
t.Errorf("CleanID(%q): %v", in, err)
continue
}
if got != strings.ToLower(in) {
t.Errorf("CleanID(%q) = %q, want it lowercased", in, got)
}
}
// Anything that could escape the objects directory, shadow a route, or
// collide on a case-insensitive filesystem must be refused.
invalid := []string{
"", "a", ".", "..", "...", "../etc/passwd", "a/b", `a\b`, "/abs",
".hidden", "a..b", "trailing.", "api", "static", "d", "i",
"robots.txt", "tokens.json", "with space", "emoji-🙂",
strings.Repeat("x", 65), "a\x00b", "a\nb",
}
for _, in := range invalid {
if got, err := CleanID(in); err == nil {
t.Errorf("CleanID(%q) = %q, want an error", in, got)
}
}
}
func TestCleanIDAcceptsGeneratedUUIDs(t *testing.T) {
for range 100 {
id, err := NewUUID()
if err != nil {
t.Fatal(err)
}
if len(id) != 36 || id[14] != '4' {
t.Fatalf("NewUUID() = %q, not a v4 UUID", id)
}
if got, err := CleanID(id); err != nil || got != id {
t.Fatalf("CleanID(%q) = %q, %v", id, got, err)
}
}
}
func TestSanitizeFilename(t *testing.T) {
cases := map[string]string{
"MyGame.zip": "MyGame.zip",
`C:\Users\me\Desktop\thing.exe`: "thing.exe",
"/etc/passwd": "passwd",
"../../escape.txt": "escape.txt",
"": "download.bin",
".": "download.bin",
"..": "download.bin",
" ": "download.bin",
"with\r\nheader: injected": "withheader: injected",
"null\x00byte": "nullbyte",
"naïve fïle.txt": "naïve fïle.txt",
`quo"te.txt`: `quo"te.txt`,
}
for in, want := range cases {
if got := SanitizeFilename(in); got != want {
t.Errorf("SanitizeFilename(%q) = %q, want %q", in, got, want)
}
}
long := SanitizeFilename(strings.Repeat("é", 400))
if len(long) > maxFilenameBytes {
t.Errorf("a long name was not truncated: %d bytes", len(long))
}
}
func TestUploadIsInvisibleUntilCommitted(t *testing.T) {
s, err := Open(t.TempDir())
if err != nil {
t.Fatal(err)
}
now := time.Now()
up, err := s.Reserve("thing")
if err != nil {
t.Fatal(err)
}
up.Write([]byte("partial"))
// The name is claimed, but the object does not exist yet.
if !s.Exists("thing") {
t.Error("the name was not claimed")
}
if _, err := s.Get("thing", now); err != ErrNotFound {
t.Errorf("Get on an uncommitted upload = %v, want ErrNotFound", err)
}
if _, err := s.Reserve("thing"); err != ErrExists {
t.Error("a claimed name was handed out twice")
}
if err := up.Commit(&Meta{Created: now}); err != nil {
t.Fatal(err)
}
m, err := s.Get("thing", now)
if err != nil {
t.Fatalf("Get after Commit: %v", err)
}
if m.Size != 7 {
t.Errorf("size = %d, want 7", m.Size)
}
want := sha256.Sum256([]byte("partial"))
if m.SHA256 != hex.EncodeToString(want[:]) {
t.Errorf("SHA256 = %q, want %x", m.SHA256, want)
}
if s.Total() != 7 {
t.Errorf("Total() = %d, want 7", s.Total())
}
}
func TestAbortReleasesTheName(t *testing.T) {
dir := t.TempDir()
s, err := Open(dir)
if err != nil {
t.Fatal(err)
}
up, err := s.Reserve("thing")
if err != nil {
t.Fatal(err)
}
up.Write([]byte("partial"))
up.Abort()
if s.Exists("thing") {
t.Error("Abort did not release the name")
}
if _, err := os.Stat(filepath.Join(dir, "objects", "thing")); !os.IsNotExist(err) {
t.Error("Abort left the directory behind")
}
if _, err := s.Reserve("thing"); err != nil {
t.Errorf("the name could not be reused: %v", err)
}
}
func TestLimitStopsAtTheCap(t *testing.T) {
s, err := Open(t.TempDir())
if err != nil {
t.Fatal(err)
}
up, err := s.Reserve("thing")
if err != nil {
t.Fatal(err)
}
defer up.Abort()
up.SetLimit(10)
if _, err := up.Write([]byte("0123456789")); err != nil {
t.Fatalf("writing exactly the limit: %v", err)
}
if _, err := up.Write([]byte("x")); err != ErrTooLarge {
t.Errorf("writing past the limit = %v, want ErrTooLarge", err)
}
if up.Size() != 10 {
t.Errorf("Size() = %d, want 10", up.Size())
}
}
func TestIndexIsRebuiltFromDisk(t *testing.T) {
dir := t.TempDir()
now := time.Now()
s, err := Open(dir)
if err != nil {
t.Fatal(err)
}
up, _ := s.Reserve("survivor")
up.Write([]byte("bytes"))
if err := up.Commit(&Meta{Created: now}); err != nil {
t.Fatal(err)
}
// A .part with no metadata is what a killed upload leaves behind.
stale, _ := s.Reserve("stale")
stale.Write([]byte("half"))
// Reopening is what a restart does.
s2, err := Open(dir)
if err != nil {
t.Fatal(err)
}
if _, err := s2.Get("survivor", now); err != nil {
t.Errorf("a committed object did not survive a restart: %v", err)
}
if _, err := s2.Get("stale", now); err != ErrNotFound {
t.Error("an uncommitted object became visible after a restart")
}
if s2.Total() != 5 {
t.Errorf("Total() = %d, want 5", s2.Total())
}
}
func TestSweepRemovesExpired(t *testing.T) {
s, err := Open(t.TempDir())
if err != nil {
t.Fatal(err)
}
now := time.Now()
deadline := now.Add(time.Hour)
up, _ := s.Reserve("temporary")
up.Write([]byte("x"))
up.Commit(&Meta{Created: now, Expires: &deadline})
keep, _ := s.Reserve("permanent")
keep.Write([]byte("x"))
keep.Commit(&Meta{Created: now})
if n := s.Sweep(now); n != 0 {
t.Errorf("swept %d objects before anything expired", n)
}
if n := s.Sweep(now.Add(2 * time.Hour)); n != 1 {
t.Errorf("swept %d objects, want 1", n)
}
if s.Count() != 1 || s.Total() != 1 {
t.Errorf("after sweeping: count = %d, total = %d, want 1 and 1", s.Count(), s.Total())
}
if _, err := s.Get("permanent", now.Add(10*365*24*time.Hour)); err != nil {
t.Error("an object with no expiry was swept")
}
}
// The objects directory is group-writable by design, so a planted symlink is a
// realistic way to try to make the service read or clobber a file elsewhere.
// Every object operation goes through an os.Root, which refuses to follow one
// out of the directory.
func TestSymlinksCannotEscapeTheObjectsDirectory(t *testing.T) {
dir := t.TempDir()
outside := filepath.Join(dir, "outside")
if err := os.WriteFile(filepath.Join(dir, "secret.txt"), []byte("password"), 0o600); err != nil {
t.Fatal(err)
}
if err := os.MkdirAll(outside, 0o755); err != nil {
t.Fatal(err)
}
s, err := Open(dir)
if err != nil {
t.Fatal(err)
}
defer s.Close()
objects := filepath.Join(dir, "objects")
now := time.Now()
// A blob that is a symlink to a file outside the store.
if err := os.Mkdir(filepath.Join(objects, "sneaky"), 0o775); err != nil {
t.Fatal(err)
}
if err := os.Symlink(filepath.Join(dir, "secret.txt"), filepath.Join(objects, "sneaky", "blob")); err != nil {
t.Fatal(err)
}
meta := []byte(`{"id":"sneaky","filename":"x","size":8,"created":"2026-01-01T00:00:00Z","expires":null}`)
if err := os.WriteFile(filepath.Join(objects, "sneaky", "meta.json"), meta, 0o664); err != nil {
t.Fatal(err)
}
// Reopen so the planted object is indexed, as it would be after a restart.
s2, err := Open(dir)
if err != nil {
t.Fatal(err)
}
defer s2.Close()
if _, _, err := s2.OpenBlob("sneaky", now); err == nil {
t.Error("a blob symlinked outside the store was opened")
}
// A whole object directory that is a symlink elsewhere.
if err := os.Symlink(outside, filepath.Join(objects, "elsewhere")); err != nil {
t.Fatal(err)
}
if _, err := s2.Reserve("elsewhere"); err != ErrExists {
t.Errorf("Reserve over a symlink = %v, want ErrExists", err)
}
// Writing through it must not reach the target directory either.
if err := s2.writeMetaAtomic("elsewhere", &Meta{ID: "elsewhere"}); err == nil {
t.Error("metadata was written through a symlinked directory")
}
if entries, _ := os.ReadDir(outside); len(entries) != 0 {
t.Errorf("%d files were created outside the store", len(entries))
}
}