diff --git a/.github/workflows/unit-test.yml b/.github/workflows/unit-test.yml index f79afbe2a..31b1981a3 100644 --- a/.github/workflows/unit-test.yml +++ b/.github/workflows/unit-test.yml @@ -201,6 +201,81 @@ jobs: go test -v -timeout=1h ./... --tags=unittest,azurite,fuse3 fi + - name: Tiered storage mount smoke test + if: matrix.job_name == 'linux-fuse3' + env: + S3_KEY_ID: ${{ env.AWS_ACCESS_KEY_ID }} + S3_SECRET_KEY: ${{ env.AWS_SECRET_ACCESS_KEY }} + S3_ENDPOINT: ${{ env.AWS_ENDPOINT }} + S3_REGION: ${{ env.AWS_REGION }} + S3_BUCKET_NAME: ${{ env.AWS_BUCKET_NAME }} + run: | + set -euo pipefail + MOUNT_DIR=/tmp/cloudfuse-tiered-mount + TIER_DIR=/tmp/cloudfuse-tiered-data + CONFIG=/tmp/cloudfuse-tiered.yaml + SOURCE=/tmp/cloudfuse-tiered-source + PERSISTENT=/tmp/cloudfuse-tiered-persistent + MOUNT_PID= + + cleanup() { + if mountpoint -q "$MOUNT_DIR"; then + fusermount3 -u "$MOUNT_DIR" -z || true + fi + if [[ -n "$MOUNT_PID" ]]; then + kill "$MOUNT_PID" 2>/dev/null || true + fi + rm -rf "$MOUNT_DIR" "$TIER_DIR" "$CONFIG" "$SOURCE" "$PERSISTENT" + } + trap cleanup EXIT + + mkdir -p "$MOUNT_DIR" "$TIER_DIR" + ./cloudfuse gen-test-config \ + --config-file=s3_key_tiered_storage.yaml \ + --temp-path="$TIER_DIR" \ + --output-file="$CONFIG" + + mount_tiered() { + ./cloudfuse mount "$MOUNT_DIR" --config-file="$CONFIG" --foreground=true & + MOUNT_PID=$! + for _ in {1..30}; do + mountpoint -q "$MOUNT_DIR" && return + sleep 1 + done + return 1 + } + + unmount_tiered() { + fusermount3 -u "$MOUNT_DIR" + wait "$MOUNT_PID" + MOUNT_PID= + } + + mount_tiered + head -c 900000 /dev/urandom > "$SOURCE" + cp "$SOURCE" "$MOUNT_DIR/evicted.bin" + for _ in {1..30}; do + aws --endpoint-url "$AWS_ENDPOINT" s3api head-object \ + --bucket "$AWS_BUCKET_NAME" --key evicted.bin >/dev/null 2>&1 && break + sleep 1 + done + aws --endpoint-url "$AWS_ENDPOINT" s3api head-object \ + --bucket "$AWS_BUCKET_NAME" --key evicted.bin >/dev/null + test ! -e "$TIER_DIR/evicted.bin" + cmp "$SOURCE" "$MOUNT_DIR/evicted.bin" + test ! -e "$TIER_DIR/evicted.bin" + + head -c 100000 /dev/urandom > "$PERSISTENT" + cp "$PERSISTENT" "$MOUNT_DIR/persistent.bin" + test -e "$TIER_DIR/persistent.bin" + unmount_tiered + + mount_tiered + test -e "$TIER_DIR/persistent.bin" + cmp "$PERSISTENT" "$MOUNT_DIR/persistent.bin" + rm "$MOUNT_DIR/persistent.bin" + unmount_tiered + test: name: Build and Test on Windows environment: testing diff --git a/README.md b/README.md index dd9eceec9..c4d4ed267 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ Cloudfuse provides the ability to mount a cloud bucket in your local filesystem on Linux and Windows. With Cloudfuse you can easily read and write to the cloud, and connect programs on your computer to the cloud even if they're not cloud-aware. -Cloudfuse uses file caching to provide the performance of local storage, or you can use streaming mode to efficiently access small parts of large files (e.g. video playback). +Cloudfuse can run in two modes: caching, where the cloud is the source of truth and local disk holds a temporary copy for speed, or tiered storage, where local disk is the primary copy and the cloud is overflow for data that no longer fits locally. Cloudfuse is a fork of [blobfuse2](https://github.com/Azure/azure-storage-fuse), and adds S3 support and Windows support. Cloudfuse supports clouds with an S3 or Azure interface. @@ -145,10 +145,10 @@ manually install Cloudfuse. ## Basic Use -The following describes how to use the Cloudfuse CLI. If you would like to use a GUI checkout the Cloudfuse GUI repo at . +The following describes how to use the Cloudfuse CLI. 1. Create a basic configuration file (TUI): - If you would like an easy way to get started with cloudfuse, run the following to launch a TUI to configure cloudfuse. If you prefer to configure manually, checkout how to write a config file: + If you would like an easy way to get started with cloudfuse, run the following to launch a TUI to configure cloudfuse. The TUI does not currently support tiering. If you prefer to configure manually, checkout how to write a config file: ```bash cloudfuse config @@ -314,6 +314,21 @@ Cloudfuse now supports offline access through the `file_cache` component. When c > **Note:** Cloudfuse uses eventual consistency with last-writer-wins semantics. Offline access can extend the consistency window indefinitely and **increases the risk of data conflicts in multi-client setups!** See [component/file_cache/OfflineAccess.md](component/file_cache/OfflineAccess.md) for full details and configuration guidance. +## Tiered Storage + +Use `tiered_storage` for on-prem-first configurations, when you want cloud storage to expand existing local storage. To minimize cloud storage costs, `tiered_storage` only maintains one copy of each file, moving old files to cloud storage once local storage fills up. + +Note: Do not use `tiered_storage` if you use cloud storage to access your data remotely. Only overflow data will be present in the cloud. + +See `sample_configs/sampleTieredStorageConfigS3.yaml` and `setup/baseConfig.yaml` for configuration. + +### Expanding Existing Local Storage Using Tiered Storage + +When using `tiered_storage` to add cloud capacity to an existing local storage location, please ensure all access is done through the Cloudfuse mount path / virtual drive. The `tiered_storage` `path` and all its existing contents will become *internal* local storage. Accessing the `tiered_storage` `path` directly may have unpredictable results. +For example, if you want expand the capacity of your existing local drive, `D:`, you could change that local drive's letter to `E:` first, then set `tiered_storage` `path` to `E:` and mount cloudfuse to `D:`. Then any existing applications or workflows that normally use `D:` would continue as normal, now pointed at the cloudfuse mount location. + +**Cloudfuse may upload and delete local files under `path` as part of normal overflow eviction, so anything still reading or writing there directly can lose data or see missing files.** + ## Limitations ### NOTICE diff --git a/cmd/imports.go b/cmd/imports.go index e253a2a75..b1bbb9314 100644 --- a/cmd/imports.go +++ b/cmd/imports.go @@ -29,11 +29,14 @@ import ( _ "github.com/Seagate/cloudfuse/component/attr_cache" _ "github.com/Seagate/cloudfuse/component/azstorage" _ "github.com/Seagate/cloudfuse/component/block_cache" + _ "github.com/Seagate/cloudfuse/component/custom" + _ "github.com/Seagate/cloudfuse/component/entry_cache" _ "github.com/Seagate/cloudfuse/component/file_cache" _ "github.com/Seagate/cloudfuse/component/libfuse" _ "github.com/Seagate/cloudfuse/component/loopback" _ "github.com/Seagate/cloudfuse/component/s3storage" _ "github.com/Seagate/cloudfuse/component/size_tracker" _ "github.com/Seagate/cloudfuse/component/stream" + _ "github.com/Seagate/cloudfuse/component/tiered_storage" _ "github.com/Seagate/cloudfuse/component/xload" ) diff --git a/cmd/mount_linux_test.go b/cmd/mount_linux_test.go index 3eca73150..18623cebd 100644 --- a/cmd/mount_linux_test.go +++ b/cmd/mount_linux_test.go @@ -804,6 +804,17 @@ func (suite *mountTestSuite) TestCleanUpOnStartFlag() { suite.assert.False(common.IsDirectoryEmpty(cachedirs[(i+1)%3])) suite.assert.False(common.IsDirectoryEmpty(cachedirs[(i+2)%3])) } + + tieredPath := filepath.Join(testDir, "tiered") + err = os.MkdirAll(tieredPath, 0755) + suite.assert.NoError(err) + err = os.WriteFile(filepath.Join(tieredPath, "authoritative"), []byte("data"), 0644) + suite.assert.NoError(err) + config.Set("tiered_storage.path", tieredPath) + config.Set("cleanup-on-start", "true") + options.Components = []string{"tiered_storage"} + suite.assert.NoError(options.tempCacheCleanup()) + suite.assert.False(common.IsDirectoryEmpty(tieredPath)) } // TestValidateMountOptionsInvalidLogLevel tests validation with invalid log level diff --git a/common/lock_map.go b/common/lock_map.go index 98cab2dcf..1709eccc5 100644 --- a/common/lock_map.go +++ b/common/lock_map.go @@ -33,7 +33,7 @@ import ( // Lock item for each file type LockMapItem struct { - handleCount uint32 + handleCount atomic.Uint32 dirtyCount atomic.Uint32 mtx sync.RWMutex downloadTime time.Time @@ -58,7 +58,7 @@ func (l *LockMap) Get(name string) *LockMapItem { if lockIntf, found := l.locks.Load(name); found { return lockIntf.(*LockMapItem) } - lockIntf, _ := l.locks.LoadOrStore(name, &LockMapItem{handleCount: 0}) + lockIntf, _ := l.locks.LoadOrStore(name, &LockMapItem{}) item := lockIntf.(*LockMapItem) return item } @@ -79,6 +79,10 @@ func (l *LockMapItem) Unlock() { l.mtx.Unlock() } +func (l *LockMapItem) TryLock() bool { + return l.mtx.TryLock() +} + func (l *LockMapItem) RLock() { l.mtx.RLock() } @@ -90,17 +94,25 @@ func (l *LockMapItem) RUnlock() { // Increment the handle count func (l *LockMapItem) Inc() { - l.handleCount++ + l.handleCount.Add(1) } // Decrement the handle count func (l *LockMapItem) Dec() { - l.handleCount-- + for { + current := l.handleCount.Load() + if current == 0 { + return + } + if l.handleCount.CompareAndSwap(current, current-1) { + return + } + } } // Get the current handle count func (l *LockMapItem) Count() uint32 { - return l.handleCount + return l.handleCount.Load() } // Increment dirty-handle count. diff --git a/common/util.go b/common/util.go index e951f307d..25b30b9db 100644 --- a/common/util.go +++ b/common/util.go @@ -618,20 +618,14 @@ func ComponentInPipeline(pipeline []string, component string) bool { } func ValidatePipeline(pipeline []string) error { - // file-cache, block-cache and xload are mutually exclusive - if ComponentInPipeline(pipeline, "file_cache") && - ComponentInPipeline(pipeline, "block_cache") { - return fmt.Errorf("mount: file-cache and block-cache cannot be used together") - } - - if ComponentInPipeline(pipeline, "file_cache") && - ComponentInPipeline(pipeline, "xload") { - return fmt.Errorf("mount: file-cache and xload cannot be used together") + var selected []string + for _, component := range []string{"file_cache", "block_cache", "xload", "tiered_storage"} { + if ComponentInPipeline(pipeline, component) { + selected = append(selected, component) + } } - - if ComponentInPipeline(pipeline, "block_cache") && - ComponentInPipeline(pipeline, "xload") { - return fmt.Errorf("mount: block-cache and xload cannot be used together") + if len(selected) > 1 { + return fmt.Errorf("mount: %s cannot be used together", strings.Join(selected, ", ")) } return nil diff --git a/common/util_test.go b/common/util_test.go index 7d2529f9f..dfda113b3 100644 --- a/common/util_test.go +++ b/common/util_test.go @@ -665,6 +665,15 @@ func (suite *utilTestSuite) TestValidatePipeline() { err = ValidatePipeline([]string{"libfuse", "file_cache", "block_cache", "xload", "azstorage"}) suite.Error(err) + err = ValidatePipeline([]string{"libfuse", "file_cache", "tiered_storage", "azstorage"}) + suite.Error(err) + + err = ValidatePipeline([]string{"libfuse", "block_cache", "tiered_storage", "azstorage"}) + suite.Error(err) + + err = ValidatePipeline([]string{"libfuse", "xload", "tiered_storage", "azstorage"}) + suite.Error(err) + err = ValidatePipeline([]string{"libfuse", "file_cache", "azstorage"}) suite.NoError(err) @@ -673,6 +682,9 @@ func (suite *utilTestSuite) TestValidatePipeline() { err = ValidatePipeline([]string{"libfuse", "xload", "attr_cache", "azstorage"}) suite.NoError(err) + + err = ValidatePipeline([]string{"libfuse", "tiered_storage", "attr_cache", "azstorage"}) + suite.NoError(err) } func (suite *utilTestSuite) TestUpdatePipeline() { diff --git a/component/tiered_storage/cache_size.go b/component/tiered_storage/cache_size.go new file mode 100644 index 000000000..667091bad --- /dev/null +++ b/component/tiered_storage/cache_size.go @@ -0,0 +1,114 @@ +/* + Licensed under the MIT License . + + Copyright © 2026 Seagate Technology LLC and/or its Affiliates + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "sync/atomic" + "time" + + "github.com/Seagate/cloudfuse/common" + "github.com/Seagate/cloudfuse/common/log" +) + +// driftWarnBytes is how far the running total may diverge from a measurement +// before the correction is worth a log line. +const driftWarnBytes = 16 * common.MbToBytes + +// cacheSizeTracker tracks how many bytes of local storage the component is +// using. +// +// The total is maintained incrementally, because measuring it costs a du(1) +// subprocess on Linux and a full tree walk on Windows - far too expensive to do +// per write or per eviction tick. Incremental accounting drifts (directory +// entries, sparse files, anything that touches the cache directory behind our +// back), so Reconcile periodically replaces the total with a real measurement. +// +// Sizes are apparent bytes, matching `du -sb` on Linux. Windows measures +// allocated sectors instead, so the two platforms disagree slightly for small +// files; Reconcile is what keeps that from accumulating. +type cacheSizeTracker struct { + path string + used atomic.Int64 + reconcileInterval time.Duration + lastReconcile atomic.Int64 // unix nanoseconds +} + +func newCacheSizeTracker(path string, reconcileInterval time.Duration) *cacheSizeTracker { + t := &cacheSizeTracker{path: path, reconcileInterval: reconcileInterval} + t.lastReconcile.Store(time.Now().UnixNano()) + return t +} + +// Used returns the current cache usage in bytes. +func (t *cacheSizeTracker) Used() int64 { + return t.used.Load() +} + +// Add applies a signed change in bytes. The total is clamped at zero: drift +// could otherwise drive it negative and hide a full cache. +func (t *cacheSizeTracker) Add(delta int64) { + if delta == 0 { + return + } + for { + current := t.used.Load() + next := max(current+delta, 0) + if t.used.CompareAndSwap(current, next) { + return + } + } +} + +// Refresh measures the cache directory and replaces the running total. +// Changes made while the measurement is in flight are lost, which is inherent +// to reconciling against a moving target and is why it is not the hot path. +func (t *cacheSizeTracker) Refresh() { + t.lastReconcile.Store(time.Now().UnixNano()) + + usage, err := common.GetUsage(t.path) + if err != nil { + log.Err("cacheSizeTracker::Refresh : failed to measure %s [%v]", t.path, err) + return + } + + measured := int64(usage) + previous := t.used.Swap(measured) + if drift := measured - previous; drift > driftWarnBytes || drift < -driftWarnBytes { + log.Info( + "cacheSizeTracker::Refresh : corrected usage by %d bytes (was %d, now %d)", + drift, + previous, + measured, + ) + } +} + +// Reconcile refreshes the running total, but no more often than reconcileInterval. +func (t *cacheSizeTracker) Reconcile() { + if time.Since(time.Unix(0, t.lastReconcile.Load())) < t.reconcileInterval { + return + } + t.Refresh() +} diff --git a/component/tiered_storage/lru_policy.go b/component/tiered_storage/lru_policy.go new file mode 100644 index 000000000..66c0dc925 --- /dev/null +++ b/component/tiered_storage/lru_policy.go @@ -0,0 +1,463 @@ +/* + Licensed under the MIT License . + + Copyright © 2026 Seagate Technology LLC and/or its Affiliates + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "fmt" + "os" + "path/filepath" + "sync" + "time" + + "github.com/Seagate/cloudfuse/common" + "github.com/Seagate/cloudfuse/common/log" +) + +type lruNode struct { + prev *lruNode + next *lruNode + name string + + evicting bool +} + +type uploadJob struct { + name string + wg *sync.WaitGroup +} + +// lruQueueConfig holds the fields needed to construct an lruQueue. +type lruQueueConfig struct { + cachePath string + fileLocks *common.LockMap // uses object name (common.JoinUnixFilepath) + size *cacheSizeTracker + + maxCacheSize float64 + // threshold and targetRatio are fractions of maxCacheSize. + threshold float64 + targetRatio float64 + numWorkers int + maxEviction uint32 + pollInterval time.Duration + + // Called with the object's file lock held. + uploadandCleanFn func(name string) error +} + +// lruQueue moves local-only objects to cloud storage. +// File locks may be held before mu, never after it. +type lruQueue struct { + // mu guards the list and its nodes. + mu sync.Mutex + head *lruNode + tail *lruNode + // nodeMap is written under mu, but may be read without it. + nodeMap sync.Map + + // evictMu serializes eviction cycles. + evictMu sync.Mutex + + checkerWg sync.WaitGroup + workerWg sync.WaitGroup + + uploadChan chan uploadJob + doneChan chan struct{} + + lruQueueConfig +} + +func newLRUQueue(cfg lruQueueConfig) *lruQueue { + return &lruQueue{lruQueueConfig: cfg} +} + +func (q *lruQueue) StartPolicy() error { + if q.uploadandCleanFn == nil { + return fmt.Errorf("lruQueue: upload function not set") + } + if q.numWorkers <= 0 { + return fmt.Errorf("lruQueue: numWorkers must be > 0") + } + if q.fileLocks == nil { + return fmt.Errorf("lruQueue: file locks not set") + } + if q.size == nil { + return fmt.Errorf("lruQueue: cache size tracker not set") + } + if q.pollInterval <= 0 { + q.pollInterval = capacityPollInterval + } + if q.maxEviction == 0 { + q.maxEviction = defaultMaxEviction + } + + q.doneChan = make(chan struct{}) + q.uploadChan = make(chan uploadJob) + + q.workerWg.Add(q.numWorkers) + for range q.numWorkers { + go q.worker() + } + + q.checkerWg.Add(1) + go q.capacityChecker() + + return nil +} + +// StopPolicy waits for in-flight uploads but leaves queued files local. +func (q *lruQueue) StopPolicy() error { + if q.doneChan == nil { + return nil + } + close(q.doneChan) + q.checkerWg.Wait() + + // Prevent EvictNow from sending while uploadChan is closed. + q.evictMu.Lock() + close(q.uploadChan) + q.evictMu.Unlock() + + q.workerWg.Wait() + return nil +} + +// Enqueue adds or promotes an object unless a worker owns it. +func (q *lruQueue) Enqueue(name string) { + q.mu.Lock() + defer q.mu.Unlock() + + val, found := q.nodeMap.Load(name) + if !found { + node := &lruNode{name: name} + q.nodeMap.Store(name, node) + q.setHead(node) + return + } + + node := val.(*lruNode) + if node.evicting { + return + } + q.extractNode(node) + q.setHead(node) +} + +// Dequeue removes an object and cancels any in-flight eviction. +func (q *lruQueue) Dequeue(name string) { + log.Trace("lruQueue::Dequeue : %s", name) + + q.mu.Lock() + defer q.mu.Unlock() + + val, found := q.nodeMap.LoadAndDelete(name) + if !found { + return + } + + node := val.(*lruNode) + if !node.evicting { + q.extractNode(node) + } +} + +// Rename moves src's entry to dst, keeping its LRU position. Any dst entry is +// discarded and in-flight evictions of either name are cancelled. +func (q *lruQueue) Rename(src, dst string) { + q.mu.Lock() + defer q.mu.Unlock() + + if val, found := q.nodeMap.LoadAndDelete(dst); found && !val.(*lruNode).evicting { + q.extractNode(val.(*lruNode)) + } + val, found := q.nodeMap.LoadAndDelete(src) + if !found { + return + } + + old := val.(*lruNode) + node := &lruNode{name: dst} + q.nodeMap.Store(dst, node) + if old.evicting { + q.setTail(node) + return + } + + // replace the old src node in the linked list + node.prev, node.next = old.prev, old.next + if old.prev != nil { + old.prev.next = node + } else { + q.head = node + } + if old.next != nil { + old.next.prev = node + } else { + q.tail = node + } + old.prev, old.next = nil, nil +} + +// mu must be held +func (q *lruQueue) setHead(node *lruNode) { + node.prev = nil + node.next = q.head + if q.head != nil { + q.head.prev = node + } + q.head = node + if q.tail == nil { + q.tail = node + } +} + +// mu must be held +func (q *lruQueue) setTail(node *lruNode) { + node.next = nil + node.prev = q.tail + if q.tail != nil { + q.tail.next = node + } + q.tail = node + if q.head == nil { + q.head = node + } +} + +// mu must be held +func (q *lruQueue) extractNode(node *lruNode) { + if node == q.head { + q.head = node.next + } + if node == q.tail { + q.tail = node.prev + } + + if node.next != nil { + node.next.prev = node.prev + } + if node.prev != nil { + node.prev.next = node.next + } + node.prev = nil + node.next = nil +} + +func (q *lruQueue) capacityChecker() { + defer q.checkerWg.Done() + + ticker := time.NewTicker(q.pollInterval) + defer ticker.Stop() + + for { + select { + case <-q.doneChan: + return + case <-ticker.C: + q.size.Reconcile() + if float64(q.size.Used()) <= q.maxCacheSize*q.threshold { + continue + } + q.evictDownTo(int64(q.maxCacheSize * q.targetRatio)) + } + } +} + +// EvictNow synchronously makes room for growBytes. +func (q *lruQueue) EvictNow(growBytes int64) bool { + target := min(int64(q.maxCacheSize*q.targetRatio), int64(q.maxCacheSize)-growBytes) + q.evictDownTo(target) + return float64(q.size.Used()+growBytes) <= q.maxCacheSize +} + +// evictDownTo runs one eviction pass toward targetUsed. +func (q *lruQueue) evictDownTo(targetUsed int64) { + q.evictMu.Lock() + defer q.evictMu.Unlock() + + select { + case <-q.doneChan: + return + default: + } + + used := q.size.Used() + if used <= targetUsed { + return + } + + names := q.nominate(used-targetUsed, int(q.maxEviction)) + if len(names) == 0 { + log.Info("lruQueue::evictDownTo : nothing is eligible for eviction") + return + } + + q.runBatch(names) + if usedAfter := q.size.Used(); float64(usedAfter) > q.maxCacheSize*q.threshold { + log.Err( + "lruQueue::evictDownTo : usage remains above high threshold after eviction (using %d, high threshold %.0f)", + usedAfter, + q.maxCacheSize*q.threshold, + ) + } +} + +func (q *lruQueue) runBatch(names []string) { + var wg sync.WaitGroup + + wg.Add(len(names)) + for i, name := range names { + select { + case q.uploadChan <- uploadJob{name: name, wg: &wg}: + case <-q.doneChan: + for _, undelivered := range names[i:] { + q.requeue(undelivered, false) + wg.Done() + } + wg.Wait() + return + } + } + wg.Wait() +} + +// nominate claims enough least-recently-used files to cover bytesNeeded. +// Open files are promoted instead. +func (q *lruQueue) nominate(bytesNeeded int64, limit int) []string { + q.mu.Lock() + defer q.mu.Unlock() + + var names []string + var busy []*lruNode + var total int64 + + for node := q.tail; node != nil && len(names) < limit && total < bytesNeeded; { + prev := node.prev + + if q.fileLocks.Get(node.name).Count() > 0 { + busy = append(busy, node) + } else { + info, err := os.Stat(filepath.Join(q.cachePath, node.name)) + if err != nil { + log.Warn("lruQueue::nominate : dropping %s [%v]", node.name, err) + q.extractNode(node) + q.nodeMap.Delete(node.name) + } else { + q.extractNode(node) + node.evicting = true + names = append(names, node.name) + total += info.Size() + } + } + + node = prev + } + + // Relink after walking so traversal cannot revisit promoted nodes. + for _, node := range busy { + q.extractNode(node) + q.setHead(node) + } + + return names +} + +func (q *lruQueue) claimed(name string) bool { + q.mu.Lock() + defer q.mu.Unlock() + + val, found := q.nodeMap.Load(name) + return found && val.(*lruNode).evicting +} + +// Failed uploads return to the tail; busy files return to the head. +func (q *lruQueue) requeue(name string, failed bool) { + q.mu.Lock() + defer q.mu.Unlock() + + val, found := q.nodeMap.Load(name) + if !found { + return + } + node := val.(*lruNode) + + node.evicting = false + if failed { + q.setTail(node) + return + } + q.setHead(node) +} + +func (q *lruQueue) drop(name string) { + q.mu.Lock() + defer q.mu.Unlock() + + q.nodeMap.Delete(name) +} + +func (q *lruQueue) worker() { + defer q.workerWg.Done() + for job := range q.uploadChan { + q.evictOne(job) + } +} + +func (q *lruQueue) evictOne(job uploadJob) { + defer job.wg.Done() + + flock := q.fileLocks.Get(job.name) + if !flock.TryLock() { + q.requeue(job.name, false) + return + } + defer flock.Unlock() + + // Dequeue may cancel the job while it waits for a worker. + if !q.claimed(job.name) { + // any entry now under this name belongs to a newer file. + return + } + + if flock.Count() > 0 { + q.requeue(job.name, false) + return + } + + _, err := os.Stat(filepath.Join(q.cachePath, job.name)) + if err != nil { + log.Warn("lruQueue::evictOne : %s has no local copy, dropping it [%v]", job.name, err) + q.drop(job.name) + return + } + + if err := q.uploadandCleanFn(job.name); err != nil { + // TODO: more granular exception handling + log.Err("lruQueue::evictOne : %s upload failed [%v]", job.name, err) + q.requeue(job.name, true) + return + } + + q.drop(job.name) +} diff --git a/component/tiered_storage/lru_policy_test.go b/component/tiered_storage/lru_policy_test.go new file mode 100644 index 000000000..512c7eb4a --- /dev/null +++ b/component/tiered_storage/lru_policy_test.go @@ -0,0 +1,341 @@ +/* + Licensed under the MIT License . + + Copyright © 2023-2026 Seagate Technology LLC and/or its Affiliates + Copyright © 2020-2026 Microsoft Corporation. All rights reserved. + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "fmt" + "os" + "path/filepath" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/Seagate/cloudfuse/common" + "github.com/Seagate/cloudfuse/common/log" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/suite" +) + +type lruPolicyTestSuite struct { + suite.Suite + assert *assert.Assertions + policy *lruQueue +} + +var cache_path = filepath.Join(home_dir, "file_cache"+randomString(8)) + +const lruTestFileSize = 250 * 1024 + +func (suite *lruPolicyTestSuite) SetupTest() { + err := log.SetDefaultLogger("silent", common.LogConfig{Level: common.ELogLevel.LOG_DEBUG()}) + if err != nil { + panic(fmt.Sprintf("Unable to set silent logger as default: %v", err)) + } + suite.assert = assert.New(suite.T()) + + err = os.Mkdir(cache_path, 0777) + suite.assert.NoError(err) + + suite.setupTestHelper(lruQueueConfig{ + cachePath: cache_path, + maxCacheSize: 1 * common.MbToBytes, + threshold: 0.8, + targetRatio: 0.6, + numWorkers: 8, + pollInterval: time.Millisecond, + fileLocks: common.NewLockMap(), + size: newCacheSizeTracker(cache_path, 0), + + uploadandCleanFn: func(name string) error { + return nil + }, + }) +} + +// setupTestHelper creates and starts an lruQueue for testing. +func (suite *lruPolicyTestSuite) setupTestHelper(cfg lruQueueConfig) { + suite.policy = newLRUQueue(cfg) + + err := suite.policy.StartPolicy() + suite.assert.NoError(err) +} + +func (suite *lruPolicyTestSuite) cleanupTest() { + err := suite.policy.StopPolicy() + suite.assert.NoError(err) + + err = os.RemoveAll(cache_path) + suite.assert.NoError(err) +} + +func (suite *lruPolicyTestSuite) createQueuedFiles(names ...string) { + data := make([]byte, lruTestFileSize) + for _, name := range names { + suite.Require().NoError(os.WriteFile(filepath.Join(cache_path, name), data, 0644)) + suite.policy.Enqueue(name) + } +} + +func (suite *lruPolicyTestSuite) TestEnqueue() { + defer suite.cleanupTest() + + suite.policy.Enqueue("file1") + suite.policy.Enqueue("file2") + suite.policy.Enqueue("file3") + suite.assert.Equal("file3", suite.policy.head.name) + suite.assert.Equal("file1", suite.policy.tail.name) + + suite.policy.Enqueue("file1") + suite.assert.Equal("file1", suite.policy.head.name) + suite.assert.Equal("file2", suite.policy.tail.name) +} + +func (suite *lruPolicyTestSuite) TestDequeue() { + defer suite.cleanupTest() + + suite.policy.Enqueue("file1") + suite.policy.Enqueue("file2") + suite.policy.Dequeue("file1") + suite.assert.Equal("file2", suite.policy.head.name) + suite.assert.Equal("file2", suite.policy.tail.name) +} + +func (suite *lruPolicyTestSuite) TestRenameKeepsPosition() { + defer suite.cleanupTest() + + suite.policy.Enqueue("file1") + suite.policy.Enqueue("file2") + suite.policy.Enqueue("file3") + suite.policy.Rename("file2", "file3") + suite.policy.Rename("file1", "renamed") + + suite.assert.Equal("file3", suite.policy.head.name) + suite.assert.Equal("renamed", suite.policy.tail.name) + suite.assert.Equal(suite.policy.tail, suite.policy.head.next) + suite.assert.Nil(suite.policy.tail.next) + _, found := suite.policy.nodeMap.Load("file1") + suite.assert.False(found) + _, found = suite.policy.nodeMap.Load("file2") + suite.assert.False(found) +} + +func (suite *lruPolicyTestSuite) TestRenameCancelsEviction() { + defer suite.cleanupTest() + + suite.policy.Enqueue("file1") + suite.policy.Enqueue("file2") + suite.policy.mu.Lock() + node := suite.policy.tail + suite.policy.extractNode(node) + node.evicting = true + suite.policy.mu.Unlock() + + suite.policy.Rename("file1", "renamed") + suite.assert.False(suite.policy.claimed("file1")) + suite.assert.False(suite.policy.claimed("renamed")) + suite.assert.Equal("renamed", suite.policy.tail.name) +} + +func (suite *lruPolicyTestSuite) TestEvictionRunsOnePass() { + defer suite.cleanupTest() + suite.cleanupTest() // teardown the default policy generated in SetupTest + + cachePath := suite.T().TempDir() + size := newCacheSizeTracker(cachePath, time.Hour) + var uploaded atomic.Int32 + suite.setupTestHelper(lruQueueConfig{ + cachePath: cachePath, + maxCacheSize: common.MbToBytes, + threshold: 0.8, + targetRatio: 0.6, + numWorkers: 1, + maxEviction: 1, + pollInterval: time.Hour, + fileLocks: common.NewLockMap(), + size: size, + + uploadandCleanFn: func(name string) error { + info, err := os.Stat(filepath.Join(cachePath, name)) + if err != nil { + return err + } + if err := os.Remove(filepath.Join(cachePath, name)); err != nil { + return err + } + size.Add(-info.Size()) + uploaded.Add(1) + return nil + }, + }) + + for i := 1; i <= 4; i++ { + name := fmt.Sprintf("file%d", i) + err := os.WriteFile(filepath.Join(cachePath, name), make([]byte, lruTestFileSize), 0644) + suite.Require().NoError(err) + suite.policy.Enqueue(name) + } + size.Refresh() + + suite.policy.evictDownTo(int64(suite.policy.maxCacheSize * suite.policy.targetRatio)) + + suite.assert.EqualValues(1, uploaded.Load()) + suite.assert.LessOrEqual( + float64(size.Used()), + suite.policy.maxCacheSize*suite.policy.threshold, + ) + suite.assert.Greater( + float64(size.Used()), + suite.policy.maxCacheSize*suite.policy.targetRatio, + ) +} + +func (suite *lruPolicyTestSuite) TestCapacityCheckerEviction() { + defer suite.cleanupTest() + + var mu sync.Mutex + + var uploaded []string + suite.policy.uploadandCleanFn = func(name string) error { + mu.Lock() + defer mu.Unlock() + uploaded = append(uploaded, name) + return os.Remove(filepath.Join(cache_path, name)) + } + + suite.createQueuedFiles("file1", "file2", "file3", "file4") + + _, ex1 := suite.policy.nodeMap.Load("file1") + _, ex2 := suite.policy.nodeMap.Load("file2") + _, ex3 := suite.policy.nodeMap.Load("file3") + _, ex4 := suite.policy.nodeMap.Load("file4") + + suite.assert.True(ex1) + suite.assert.True(ex2) + suite.assert.True(ex3) + suite.assert.True(ex4) + + suite.assert.Eventually(func() bool { + mu.Lock() + defer mu.Unlock() + return len(uploaded) >= 2 + }, time.Second, time.Millisecond) + mu.Lock() + snapshot := append([]string(nil), uploaded...) + mu.Unlock() + + suite.assert.Contains(snapshot, "file1") + suite.assert.Contains(snapshot, "file2") + suite.assert.NotContains(snapshot, "file3") + suite.assert.NotContains(snapshot, "file4") + + _, ex1 = suite.policy.nodeMap.Load("file1") + _, ex2 = suite.policy.nodeMap.Load("file2") + _, ex3 = suite.policy.nodeMap.Load("file3") + _, ex4 = suite.policy.nodeMap.Load("file4") + + suite.assert.False(ex1) + suite.assert.False(ex2) + suite.assert.True(ex3) + suite.assert.True(ex4) +} + +func (suite *lruPolicyTestSuite) TestCapacityCheckerEvictionOpenHandle() { + defer suite.cleanupTest() + var mu sync.Mutex + + var uploaded []string + suite.policy.uploadandCleanFn = func(name string) error { + mu.Lock() + defer mu.Unlock() + uploaded = append(uploaded, name) + return os.Remove(filepath.Join(cache_path, name)) + } + + suite.createQueuedFiles("file1") + + flock := suite.policy.fileLocks.Get("file1") + flock.Lock() + flock.Inc() + flock.Unlock() + + suite.createQueuedFiles("file2", "file3", "file4") + + suite.assert.Eventually(func() bool { + mu.Lock() + defer mu.Unlock() + return len(uploaded) >= 2 + }, time.Second, time.Millisecond) + mu.Lock() + snapshot := append([]string(nil), uploaded...) + mu.Unlock() + + suite.policy.mu.Lock() + suite.assert.Equal("file1", suite.policy.head.name) + suite.assert.Equal("file4", suite.policy.tail.name) + suite.policy.mu.Unlock() + + suite.assert.Contains(snapshot, "file2") + suite.assert.Contains(snapshot, "file3") + suite.assert.NotContains(snapshot, "file1") + suite.assert.NotContains(snapshot, "file4") +} + +func (suite *lruPolicyTestSuite) TestStopPolicyMidUpload() { + var mu sync.Mutex + var uploaded []string + started := make(chan struct{}, 1) + suite.policy.uploadandCleanFn = func(name string) error { + select { + case started <- struct{}{}: + default: + } + time.Sleep(20 * time.Millisecond) + mu.Lock() + defer mu.Unlock() + uploaded = append(uploaded, name) + return os.Remove(filepath.Join(cache_path, name)) + } + + suite.createQueuedFiles("file1", "file2", "file3", "file4") + <-started + + err := suite.policy.StopPolicy() + suite.assert.NoError(err) + + mu.Lock() + snapshot := append([]string(nil), uploaded...) + mu.Unlock() + + suite.assert.Contains(snapshot, "file1") + + err = os.RemoveAll(cache_path) + suite.assert.NoError(err) +} + +func TestLRUPolicyTestSuite(t *testing.T) { + suite.Run(t, new(lruPolicyTestSuite)) +} diff --git a/component/tiered_storage/persistence.go b/component/tiered_storage/persistence.go new file mode 100644 index 000000000..ed041cde4 --- /dev/null +++ b/component/tiered_storage/persistence.go @@ -0,0 +1,261 @@ +/* + Licensed under the MIT License . + + Copyright © 2026 Seagate Technology LLC and/or its Affiliates + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "encoding/gob" + "errors" + "fmt" + "io/fs" + "os" + "path/filepath" + "slices" + "strings" + + "github.com/Seagate/cloudfuse/common" +) + +const ( + tieredStorageSnapshotPath = ".tieredStorageSnapshot.gob" + tieredStorageSnapshotVersion = 1 +) + +type persistedFileState struct { + Size int64 + Mtime int64 + CloudBacked bool + Dirty bool + Mode uint32 + ModeDirty bool + Owner int64 + Group int64 + OwnerDirty bool +} + +type tieredStorageSnapshot struct { + Version uint32 + Files map[string]persistedFileState + LRUOrder []string +} + +type recoveredFile struct { + name string + info fs.FileInfo + state persistedFileState +} + +func (c *TieredStorage) writeSnapshot() error { + snapshot := tieredStorageSnapshot{ + Version: tieredStorageSnapshotVersion, + Files: make(map[string]persistedFileState), + } + + c.fileMap.Range(func(key, value any) bool { + name := key.(string) + localPath, err := c.localPath(name) + if err != nil { + return true + } + info, err := os.Stat(localPath) + if err != nil || !info.Mode().IsRegular() { + return true + } + node := value.(*FileNode) + snapshot.Files[name] = persistedFileState{ + Size: info.Size(), + Mtime: info.ModTime().UnixNano(), + CloudBacked: node.cloudBacked.Load(), + Dirty: node.isDirty.Load(), + Mode: node.mode.Load(), + ModeDirty: node.modeDirty.Load(), + Owner: node.owner.Load(), + Group: node.group.Load(), + OwnerDirty: node.ownerDirty.Load(), + } + return true + }) + + c.policy.mu.Lock() + for node := c.policy.head; node != nil; node = node.next { + snapshot.LRUOrder = append(snapshot.LRUOrder, node.name) + } + c.policy.mu.Unlock() + + path := filepath.Join(c.tmpPath, tieredStorageSnapshotPath) + tmpPath := path + ".tmp" + file, err := common.OpenFile(tmpPath, os.O_CREATE|os.O_TRUNC|os.O_WRONLY, 0600) + if err != nil { + return fmt.Errorf("create state snapshot: %w", err) + } + if err = gob.NewEncoder(file).Encode(snapshot); err == nil { + err = file.Sync() + } + if closeErr := file.Close(); err == nil { + err = closeErr + } + if err != nil { + _ = os.Remove(tmpPath) + return fmt.Errorf("write state snapshot: %w", err) + } + if err := replaceFile(tmpPath, path); err != nil { + _ = os.Remove(tmpPath) + return fmt.Errorf("install state snapshot: %w", err) + } + return nil +} + +func (c *TieredStorage) readSnapshot() (*tieredStorageSnapshot, error) { + path := filepath.Join(c.tmpPath, tieredStorageSnapshotPath) + _ = os.Remove(path + ".tmp") + file, err := common.Open(path) + if errors.Is(err, os.ErrNotExist) { + return nil, nil + } + if err != nil { + return nil, err + } + defer os.Remove(path) + defer file.Close() + + var snapshot tieredStorageSnapshot + if err := gob.NewDecoder(file).Decode(&snapshot); err != nil { + return nil, fmt.Errorf("decode state snapshot: %w", err) + } + if snapshot.Version != tieredStorageSnapshotVersion { + return nil, fmt.Errorf("unsupported state snapshot version %d", snapshot.Version) + } + return &snapshot, nil +} + +func (c *TieredStorage) recoverLocalState(snapshot *tieredStorageSnapshot) error { + var recovered []recoveredFile + err := filepath.WalkDir(c.tmpPath, func(path string, entry fs.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + if entry.IsDir() || !entry.Type().IsRegular() { + return nil + } + if strings.HasSuffix(entry.Name(), partialDownloadSuffix) { + info, err := entry.Info() + if err != nil { + return err + } + //nolint:gosec // WalkDir is confined to c.tmpPath. + if err := os.Remove(path); err != nil { + return err + } + c.cacheSize.Add(-info.Size()) + return nil + } + if entry.Name() == tieredStorageSnapshotPath || + entry.Name() == tieredStorageSnapshotPath+".tmp" { + return nil + } + + rel, err := filepath.Rel(c.tmpPath, path) + if err != nil { + return err + } + name := common.NormalizeObjectName(filepath.ToSlash(rel)) + info, err := entry.Info() + if err != nil { + return err + } + + state := persistedFileState{ + Size: info.Size(), + Mtime: info.ModTime().UnixNano(), + Dirty: true, + } + if snapshot != nil { + if saved, found := snapshot.Files[name]; found && + saved.Size == info.Size() && saved.Mtime == info.ModTime().UnixNano() { + state = saved + } + } + + if state.CloudBacked && !state.Dirty { + //nolint:gosec // WalkDir is confined to c.tmpPath. + if err := os.Remove(path); err != nil { + return err + } + c.cacheSize.Add(-info.Size()) + return nil + } + if !state.CloudBacked { + state.Dirty = true + } + recovered = append(recovered, recoveredFile{name: name, info: info, state: state}) + return nil + }) + if err != nil { + return err + } + + byName := make(map[string]recoveredFile, len(recovered)) + for _, file := range recovered { + byName[file.name] = file + node := &FileNode{name: file.name} + node.size.Store(file.info.Size()) + node.cloudBacked.Store(file.state.CloudBacked) + node.isDirty.Store(file.state.Dirty) + node.mode.Store(file.state.Mode) + node.modeDirty.Store(file.state.ModeDirty) + node.owner.Store(file.state.Owner) + node.group.Store(file.state.Group) + node.ownerDirty.Store(file.state.OwnerDirty) + c.fileMap.Store(file.name, node) + } + + var order []string + queued := make(map[string]struct{}, len(recovered)) + if snapshot != nil { + for _, name := range snapshot.LRUOrder { + if _, found := byName[name]; found { + order = append(order, name) + queued[name] = struct{}{} + } + } + } + + var unmatched []recoveredFile + for _, file := range recovered { + if _, found := queued[file.name]; !found { + unmatched = append(unmatched, file) + } + } + slices.SortFunc(unmatched, func(a, b recoveredFile) int { + return b.info.ModTime().Compare(a.info.ModTime()) + }) + for _, file := range unmatched { + order = append(order, file.name) + } + + for _, name := range slices.Backward(order) { + c.policy.Enqueue(name) + } + return nil +} diff --git a/component/tiered_storage/tiered_storage.go b/component/tiered_storage/tiered_storage.go new file mode 100644 index 000000000..413c729e3 --- /dev/null +++ b/component/tiered_storage/tiered_storage.go @@ -0,0 +1,1366 @@ +/* + Licensed under the MIT License . + + Copyright © 2023-2026 Seagate Technology LLC and/or its Affiliates + Copyright © 2020-2026 Microsoft Corporation. All rights reserved. + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "context" + "errors" + "fmt" + "io" + "os" + "path/filepath" + "runtime" + "slices" + "strings" + "sync" + "sync/atomic" + "syscall" + "time" + + "github.com/Seagate/cloudfuse/common" + "github.com/Seagate/cloudfuse/common/config" + "github.com/Seagate/cloudfuse/common/log" + "github.com/Seagate/cloudfuse/internal" + "github.com/Seagate/cloudfuse/internal/handlemap" +) + +// TieredStorage keeps new data local and uses cloud storage as overflow. +// Cloud objects are cached only while open. +// +// Lock order: file lock, evictMu, then lruQueue.mu. Rename takes file locks in +// lexical order. Never acquire a file lock while holding lruQueue.mu. +type TieredStorage struct { + internal.BaseComponent + + fileMap sync.Map // uses object name (common.JoinUnixFilepath) + + policy *lruQueue + + fileLocks *common.LockMap // uses object name (common.JoinUnixFilepath) + tmpPath string // uses os.Separator (filepath.Join) + + cacheSize *cacheSizeTracker + maxCacheSize float64 +} + +// FileNode fields are atomic except name, which requires the file lock. +type FileNode struct { + name string + size atomic.Int64 + cloudBacked atomic.Bool + isDirty atomic.Bool + mode atomic.Uint32 + modeDirty atomic.Bool + owner atomic.Int64 + group atomic.Int64 + ownerDirty atomic.Bool +} + +type TieredStorageOptions struct { + TmpPath string `config:"path" yaml:"path,omitempty"` + MaxSizeMB float64 `config:"max-size-mb" yaml:"max-size-mb,omitempty"` + HighThreshold uint32 `config:"high-threshold" yaml:"high-threshold,omitempty"` + LowThreshold uint32 `config:"low-threshold" yaml:"low-threshold,omitempty"` + MaxEviction uint32 `config:"max-eviction" yaml:"max-eviction,omitempty"` + Parallelism uint32 `config:"parallelism" yaml:"parallelism,omitempty"` + PollIntervalSec uint32 `config:"poll-interval-sec" yaml:"poll-interval-sec,omitempty"` +} + +const ( + compName = "tiered_storage" + defaultHighThreshold = 90 + defaultLowThreshold = 80 + defaultParallelism = 8 + defaultMaxEviction = 5000 + capacityPollInterval = time.Second + reconcileCapacityInterval = 5 * time.Minute + partialDownloadSuffix = ".cloudfuse-partial" + // Cache holds user data and is only accessed by this process (CWE-732). + cacheDirPerm os.FileMode = 0700 +) + +var _ internal.Component = &TieredStorage{} + +func (c *TieredStorage) Name() string { + return compName +} + +func (c *TieredStorage) SetName(name string) { + c.BaseComponent.SetName(name) +} + +func (c *TieredStorage) SetNextComponent(nc internal.Component) { + c.BaseComponent.SetNextComponent(nc) +} + +func (c *TieredStorage) Start(ctx context.Context) error { + log.Trace("TieredStorage::Start : Starting component %s", c.Name()) + + snapshot, err := c.readSnapshot() + if err != nil { + log.Warn("TieredStorage::Start : ignoring invalid state snapshot [%v]", err) + } + + c.cacheSize.Refresh() + if err := c.recoverLocalState(snapshot); err != nil { + return fmt.Errorf("TieredStorage: failed to recover local state: %w", err) + } + + if c.policy != nil { + if err := c.policy.StartPolicy(); err != nil { + log.Err("TieredStorage::Start : failed to start LRU policy [%v]", err) + return err + } + } + + return nil +} + +// Stop leaves local files in place for the next mount. +func (c *TieredStorage) Stop() error { + log.Trace("TieredStorage::Stop : Stopping component %s", c.Name()) + + if c.policy != nil { + if err := c.policy.StopPolicy(); err != nil { + return err + } + } + return c.writeSnapshot() +} + +func (c *TieredStorage) Configure(_ bool) error { + log.Trace("TieredStorage::Configure : %s", c.Name()) + + conf := TieredStorageOptions{} + err := config.UnmarshalKey(c.Name(), &conf) + if err != nil { + log.Err("TieredStorage::Configure : config error [invalid config attributes]") + return fmt.Errorf("TieredStorage: config error [invalid config attributes]") + } + + c.tmpPath = filepath.Clean(common.ExpandPath(conf.TmpPath)) + if c.tmpPath == "" || c.tmpPath == "." { + return fmt.Errorf("TieredStorage: path not set in config") + } + err = os.MkdirAll(c.tmpPath, cacheDirPerm) + + if err != nil { + log.Err("TieredStorage::Configure : failed to create tmp path %s [%v]", c.tmpPath, err) + return fmt.Errorf("TieredStorage: failed to create tmp path: %w", err) + } + if conf.MaxSizeMB <= 0 { + return fmt.Errorf("TieredStorage: max-size-mb must be greater than 0") + } + + if conf.HighThreshold == 0 { + conf.HighThreshold = defaultHighThreshold + } + if conf.LowThreshold == 0 { + conf.LowThreshold = defaultLowThreshold + } + if conf.LowThreshold >= conf.HighThreshold || conf.HighThreshold > 100 { + return fmt.Errorf( + "TieredStorage: thresholds must satisfy 0 < low-threshold < high-threshold <= 100", + ) + } + if conf.MaxEviction == 0 { + conf.MaxEviction = defaultMaxEviction + } + if conf.Parallelism == 0 { + conf.Parallelism = defaultParallelism + } + pollInterval := time.Duration(conf.PollIntervalSec) * time.Second + if pollInterval == 0 { + pollInterval = capacityPollInterval + } + + c.maxCacheSize = conf.MaxSizeMB * common.MbToBytes + c.cacheSize = newCacheSizeTracker(c.tmpPath, reconcileCapacityInterval) + + c.policy = newLRUQueue(lruQueueConfig{ + cachePath: c.tmpPath, + maxCacheSize: c.maxCacheSize, + fileLocks: c.fileLocks, + size: c.cacheSize, + threshold: float64(conf.HighThreshold) / 100, + targetRatio: float64(conf.LowThreshold) / 100, + numWorkers: int(conf.Parallelism), + maxEviction: conf.MaxEviction, + pollInterval: pollInterval, + uploadandCleanFn: c.uploadandCleanFile, + }) + + return nil +} + +func (c *TieredStorage) OnConfigChange() { +} + +// localPath resolves an object name beneath the tiered storage root. +func (c *TieredStorage) localPath(name string) (string, error) { + name = filepath.FromSlash(common.NormalizeObjectName(name)) + if filepath.IsAbs(name) || filepath.VolumeName(name) != "" { + return "", syscall.EINVAL + } + + path := filepath.Join(c.tmpPath, name) + rel, err := filepath.Rel(c.tmpPath, path) + if err != nil || rel == ".." || strings.HasPrefix(rel, ".."+string(os.PathSeparator)) { + return "", syscall.EINVAL + } + return path, nil +} + +// Directory operations +func (c *TieredStorage) CreateDir(options internal.CreateDirOptions) error { + if _, err := c.GetAttr(internal.GetAttrOptions{Name: options.Name}); err == nil { + return syscall.EEXIST + } else if !errors.Is(err, os.ErrNotExist) { + return err + } + + localPath, err := c.localPath(options.Name) + if err != nil { + return err + } + mode := options.Mode + if mode == 0 || runtime.GOOS == "windows" { + mode = common.DefaultDirectoryPermissionBits + } + if err := os.Mkdir(localPath, mode); err != nil { + return err + } + if err := c.NextComponent().CreateDir(options); err != nil && !errors.Is(err, os.ErrExist) { + return err + } + return nil +} + +func (c *TieredStorage) DeleteDir(options internal.DeleteDirOptions) error { + localPath, err := c.localPath(options.Name) + if err != nil { + return err + } + cloudErr := c.NextComponent().DeleteDir(options) + if cloudErr != nil && !errors.Is(cloudErr, os.ErrNotExist) { + return cloudErr + } + localErr := os.Remove(localPath) + if localErr != nil && !errors.Is(localErr, os.ErrNotExist) { + return localErr + } + if errors.Is(localErr, os.ErrNotExist) && errors.Is(cloudErr, os.ErrNotExist) { + return syscall.ENOENT + } + return nil +} + +func (c *TieredStorage) IsDirEmpty(options internal.IsDirEmptyOptions) bool { + localPath, err := c.localPath(options.Name) + if err != nil { + return false + } + entries, localErr := os.ReadDir(localPath) + if localErr == nil && len(entries) > 0 { + return false + } + if localErr != nil && !errors.Is(localErr, os.ErrNotExist) { + return false + } + return c.NextComponent().IsDirEmpty(options) +} + +func (c *TieredStorage) OpenDir(options internal.OpenDirOptions) error { + localPath, err := c.localPath(options.Name) + if err != nil { + return err + } + if info, err := os.Stat(localPath); err == nil && info.IsDir() { + return nil + } + return c.NextComponent().OpenDir(options) +} + +func (c *TieredStorage) StreamDir( + options internal.StreamDirOptions, +) ([]*internal.ObjAttr, string, error) { + localPath, pathErr := c.localPath(options.Name) + if pathErr != nil { + return nil, "", pathErr + } + + attrs, token, cloudErr := c.NextComponent().StreamDir(options) + entries, localErr := os.ReadDir(localPath) + localExists := localErr == nil + if localErr != nil && !errors.Is(localErr, os.ErrNotExist) { + return nil, "", localErr + } + if cloudErr != nil && (!localExists || !errors.Is(cloudErr, os.ErrNotExist)) { + return attrs, token, cloudErr + } + if cloudErr != nil { + attrs = nil + token = "" + } + + localAttrs := make(map[string]*internal.ObjAttr, len(entries)) + for _, entry := range entries { + name := entry.Name() + if strings.HasSuffix(name, partialDownloadSuffix) || options.Name == "" && + (name == tieredStorageSnapshotPath || name == tieredStorageSnapshotPath+".tmp") { + continue + } + entryPath := common.JoinUnixFilepath(options.Name, name) + info, err := entry.Info() + if err != nil { + return nil, "", err + } + localAttr := newTieredStorageObjAttr(entryPath, info) + if value, tracked := c.fileMap.Load(entryPath); tracked { + localAttr.Size = value.(*FileNode).size.Load() + } + localAttrs[name] = localAttr + } + + listed := make(map[string]struct{}, len(attrs)) + for index, attr := range attrs { + listed[attr.Name] = struct{}{} + local, found := localAttrs[attr.Name] + if !found || attr.IsDir() { + continue + } + merged := *attr + merged.Size = local.Size + merged.Mtime = local.Mtime + attrs[index] = &merged + } + + if token == "" { + for name, attr := range localAttrs { + if _, found := listed[name]; !found { + attrs = append(attrs, attr) + } + } + } + slices.SortFunc(attrs, func(a, b *internal.ObjAttr) int { + return strings.Compare(a.Path, b.Path) + }) + return attrs, token, nil +} + +func (c *TieredStorage) CloseDir(options internal.CloseDirOptions) error { + localPath, err := c.localPath(options.Name) + if err != nil { + return err + } + if info, err := os.Stat(localPath); err == nil && info.IsDir() { + return nil + } + return c.NextComponent().CloseDir(options) +} + +func (c *TieredStorage) RenameDir(options internal.RenameDirOptions) error { + srcPath, err := c.localPath(options.Src) + if err != nil { + return err + } + dstPath, err := c.localPath(options.Dst) + if err != nil { + return err + } + _, localErr := os.Stat(srcPath) + localExists := !errors.Is(localErr, os.ErrNotExist) + if localErr != nil && localExists { + return localErr + } + + prefix := common.JoinUnixFilepath(options.Src) + "/" + var srcNames []string + c.fileMap.Range(func(key, _ any) bool { + name := key.(string) + if strings.HasPrefix(name, prefix) { + srcNames = append(srcNames, name) + } + return true + }) + + type renameEntry struct { + src string + dst string + } + entries := make([]renameEntry, 0, len(srcNames)) + lockNames := make([]string, 0, len(srcNames)*2) + for _, src := range srcNames { + dst := common.JoinUnixFilepath(options.Dst, strings.TrimPrefix(src, prefix)) + entries = append(entries, renameEntry{src: src, dst: dst}) + lockNames = append(lockNames, src, dst) + } + slices.Sort(lockNames) + lockNames = slices.Compact(lockNames) + locks := make([]*common.LockMapItem, 0, len(lockNames)) + for _, name := range lockNames { + flock := c.fileLocks.Get(name) + flock.Lock() + locks = append(locks, flock) + } + defer func() { + for _, flock := range slices.Backward(locks) { + flock.Unlock() + } + }() + + if !localExists { + return c.NextComponent().RenameDir(options) + } + if err := os.MkdirAll(filepath.Dir(dstPath), cacheDirPerm); err != nil { + return err + } + if err := os.Rename(srcPath, dstPath); err != nil { + return err + } + + moved := make([]*FileNode, 0, len(entries)) + for _, entry := range entries { + value, found := c.fileMap.LoadAndDelete(entry.src) + if !found { + continue + } + node := value.(*FileNode) + node.name = entry.dst + c.fileMap.Store(entry.dst, node) + moved = append(moved, node) + + c.policy.Rename(entry.src, entry.dst) + c.renameOpenHandles( + entry.src, + entry.dst, + c.fileLocks.Get(entry.src), + c.fileLocks.Get(entry.dst), + ) + } + + cloudErr := c.NextComponent().RenameDir(options) + if cloudErr == nil { + return nil + } + // The cloud copies may still be under src, so re-upload under dst. + for _, node := range moved { + if node.cloudBacked.Load() { + node.isDirty.Store(true) + } + } + if errors.Is(cloudErr, os.ErrNotExist) { + return nil + } + log.Err( + "TieredStorage::RenameDir : %s -> %s renamed locally but not in cloud [%v]", + options.Src, + options.Dst, + cloudErr, + ) + return cloudErr +} + +// File operations +func (c *TieredStorage) createFileUnlocked( + options internal.CreateFileOptions, +) (*handlemap.Handle, error) { + localPath, err := c.localPath(options.Name) + if err != nil { + return nil, err + } + err = os.MkdirAll(filepath.Dir(localPath), cacheDirPerm) + + if err != nil { + return nil, err + } + + localFile, err := common.OpenFile( + localPath, + os.O_CREATE|os.O_EXCL|os.O_RDWR, + c.cacheFileMode(options.Mode), + ) + if err != nil { + return nil, err + } + + node := &FileNode{name: options.Name} + node.isDirty.Store(true) + c.fileMap.Store(options.Name, node) + + handle := handlemap.NewHandle(options.Name) + handle.SetFileObject(localFile) + + c.setHandleDirty(handle) + + return handle, nil +} + +func (c *TieredStorage) CreateFile( + options internal.CreateFileOptions, +) (*handlemap.Handle, error) { + flock := c.fileLocks.Get(options.Name) + flock.Lock() + defer flock.Unlock() + + _, err := c.GetAttr(internal.GetAttrOptions{Name: options.Name}) + if err == nil { + return nil, syscall.EEXIST + } else if !errors.Is(err, os.ErrNotExist) { + return nil, err + } + + handle, err := c.createFileUnlocked(options) + if err != nil { + return nil, err + } + flock.Inc() + return handle, nil +} + +func (c *TieredStorage) DeleteFile(options internal.DeleteFileOptions) error { + log.Trace("TieredStorage::DeleteFile : name=%s", options.Name) + + flock := c.fileLocks.Get(options.Name) + flock.Lock() + defer flock.Unlock() + + val, exists := c.fileMap.Load(options.Name) + if !exists { + err := c.NextComponent().DeleteFile(options) + if errors.Is(err, os.ErrNotExist) { + return syscall.ENOENT + } + return err + } + + node := val.(*FileNode) + if node.cloudBacked.Load() { + // A failed rename can leave no cloud object under this name. + err := c.NextComponent().DeleteFile(options) + if err != nil && !errors.Is(err, os.ErrNotExist) { + return err + } + } + + // Cancel eviction before deleting state to prevent a stale upload. + c.policy.Dequeue(options.Name) + c.fileMap.Delete(options.Name) + + return c.purgeLocal(options.Name) +} + +func (c *TieredStorage) OpenFile(options internal.OpenFileOptions) (*handlemap.Handle, error) { + flock := c.fileLocks.Get(options.Name) + flock.Lock() + defer flock.Unlock() + + localPath, err := c.localPath(options.Name) + if err != nil { + return nil, err + } + attrs, attrErr := c.GetAttr(internal.GetAttrOptions{Name: options.Name}) + if attrErr != nil && !errors.Is(attrErr, os.ErrNotExist) { + return nil, attrErr + } + exists := attrErr == nil + if options.Flags&os.O_CREATE != 0 && options.Flags&os.O_EXCL != 0 && exists { + return nil, syscall.EEXIST + } + if !exists { + if options.Flags&os.O_CREATE == 0 { + return nil, syscall.ENOENT + } + handle, err := c.createFileUnlocked( + internal.CreateFileOptions{Name: options.Name, Mode: options.Mode}, + ) + if err != nil { + return nil, err + } + flock.Inc() + return handle, nil + } + + _, tracked := c.fileMap.Load(options.Name) + if !tracked { + info, statErr := os.Stat(localPath) + if statErr == nil { + log.Warn( + "TieredStorage::OpenFile : Warning file exists locally on disk but not in tiered storage cache: %s", + options.Name, + ) + node := &FileNode{name: options.Name} + node.size.Store(info.Size()) + node.isDirty.Store(true) + c.fileMap.Store(options.Name, node) + } else { + localCopyNode := &FileNode{name: options.Name} + localCopyNode.size.Store(attrs.Size) + localCopyNode.cloudBacked.Store(true) + + if options.Flags&os.O_TRUNC != 0 { + if err := os.MkdirAll(filepath.Dir(localPath), cacheDirPerm); err != nil { + return nil, err + } + file, err := common.OpenFile( + localPath, + os.O_CREATE|os.O_EXCL|os.O_RDWR, + c.cacheFileMode(options.Mode), + ) + if err != nil { + return nil, err + } + if err := file.Close(); err != nil { + return nil, err + } + localCopyNode.size.Store(0) + localCopyNode.isDirty.Store(true) + } else { + if err := c.reserveSpace(attrs.Size); err != nil { + return nil, err + } + if err := c.downloadCopyFromCloud(options); err != nil { + return nil, err + } + } + c.fileMap.Store(options.Name, localCopyNode) + } + } + + openFlags := options.Flags &^ (os.O_CREATE | os.O_EXCL) + localFile, err := common.OpenFile( + localPath, + openFlags, + c.cacheFileMode(options.Mode), + ) + if err != nil { + return nil, err + } + + handle := handlemap.NewHandle(options.Name) + handle.SetFileObject(localFile) + if options.Flags&os.O_APPEND != 0 { + handle.Flags.Set(handlemap.HandleOpenedAppend) + } + if options.Flags&os.O_TRUNC != 0 { + if value, found := c.fileMap.Load(options.Name); found { + node := value.(*FileNode) + c.cacheSize.Add(-node.size.Swap(0)) + node.isDirty.Store(true) + } + c.setHandleDirty(handle) + } + + flock.Inc() + + return handle, nil +} + +func (c *TieredStorage) cacheFileMode(mode os.FileMode) os.FileMode { + if mode == 0 || runtime.GOOS == "windows" { + return common.DefaultFilePermissionBits + } + return mode +} + +// downloadCopyFromCloud publishes the local path only after a complete download. +func (c *TieredStorage) downloadCopyFromCloud(options internal.OpenFileOptions) error { + localPath, err := c.localPath(options.Name) + if err != nil { + return err + } + err = os.MkdirAll(filepath.Dir(localPath), cacheDirPerm) + if err != nil { + return err + } + + partPath := fmt.Sprintf("%s.%d%s", localPath, os.Getpid(), partialDownloadSuffix) + partFile, err := common.OpenFile( + partPath, + os.O_CREATE|os.O_TRUNC|os.O_RDWR, + c.cacheFileMode(options.Mode), + ) + if err != nil { + return err + } + + err = c.NextComponent().CopyToFile(internal.CopyToFileOptions{ + Name: options.Name, + Offset: 0, + Count: 0, + File: partFile, + }) + if err == nil { + err = partFile.Sync() + } + if closeErr := partFile.Close(); err == nil { + err = closeErr + } + if err != nil { + log.Err("TieredStorage::downloadCopyFromCloud : %s download failed [%v]", options.Name, err) + _ = os.Remove(partPath) + return err + } + + if err := replaceFile(partPath, localPath); err != nil { + log.Err("TieredStorage::downloadCopyFromCloud : %s rename failed [%v]", options.Name, err) + _ = os.Remove(partPath) + return err + } + + if info, statErr := os.Stat(localPath); statErr == nil { + c.cacheSize.Add(info.Size()) + } + + return nil +} + +// replaceFile handles Windows rename-over-existing behavior. +func replaceFile(src, dst string) error { + err := os.Rename(src, dst) + if err == nil { + return nil + } + if rmErr := os.Remove(dst); rmErr != nil && !errors.Is(rmErr, os.ErrNotExist) { + return err + } + return os.Rename(src, dst) +} + +// purgeLocal removes an object's local copy and updates the usage counter. +// The object's file lock must be held. +func (c *TieredStorage) purgeLocal(name string) error { + localPath, err := c.localPath(name) + if err != nil { + return err + } + + info, statErr := os.Stat(localPath) + err = os.Remove(localPath) + if err == nil && statErr == nil { + c.cacheSize.Add(-info.Size()) + } + return err +} + +// reserveSpace evicts files as needed. Workers never block on file locks. +func (c *TieredStorage) reserveSpace(numBytes int64) error { + if c.maxCacheSize <= 0 || numBytes <= 0 { + return nil + } + if float64(c.cacheSize.Used()+numBytes) <= c.maxCacheSize { + return nil + } + if c.policy == nil || c.policy.EvictNow(numBytes) { + return nil + } + + log.Err( + "TieredStorage::reserveSpace : cache is full and nothing can be evicted (need %d bytes, using %d of %d)", + numBytes, + c.cacheSize.Used(), + int64(c.maxCacheSize), + ) + return syscall.ENOSPC +} + +func (c *TieredStorage) ReadInBuffer(options *internal.ReadInBufferOptions) (int, error) { + f := options.Handle.GetFileObject() + if f == nil { + log.Err( + "TieredStorage::ReadInBuffer : error [couldn't find fd in handle] %s", + options.Handle.Path, + ) + return 0, syscall.EBADF + } + + n, err := f.ReadAt(options.Data, options.Offset) + // ReadAt gives an error if it reads fewer bytes than the byte array. We discard that error. + if n < len(options.Data) && err == io.EOF { + return n, nil + } + return n, err +} + +func (c *TieredStorage) WriteFile(options *internal.WriteFileOptions) (int, error) { + f := options.Handle.GetFileObject() + if f == nil { + return 0, syscall.EBADF + } + + var node *FileNode + if val, ok := c.fileMap.Load(options.Handle.Path); ok { + node = val.(*FileNode) + } + + appending := options.Handle.Flags.IsSet(handlemap.HandleOpenedAppend) + growth := int64(len(options.Data)) + if !appending && node != nil { + growth = max(options.Offset+int64(len(options.Data))-node.size.Load(), 0) + } + if err := c.reserveSpace(growth); err != nil { + return 0, err + } + + var bytesWritten int + var err error + if appending { + bytesWritten, err = f.Write(options.Data) + } else { + bytesWritten, err = f.WriteAt(options.Data, options.Offset) + } + + if err == nil { + c.setHandleDirty(options.Handle) + if node != nil { + node.isDirty.Store(true) + if info, statErr := f.Stat(); statErr == nil { + c.cacheSize.Add(info.Size() - node.size.Swap(info.Size())) + } + } + } else { + log.Err( + "TieredStorage::WriteFile : failed to write %s [%s]", + options.Handle.Path, + err.Error(), + ) + } + + return bytesWritten, err +} + +func (c *TieredStorage) SyncFile(options internal.SyncFileOptions) error { + log.Trace( + "TieredStorage::SyncFile : handle=%d, path=%s", + options.Handle.ID, + options.Handle.Path, + ) + return c.FlushFile(internal.FlushFileOptions{Handle: options.Handle}) +} + +func (c *TieredStorage) FlushFile(options internal.FlushFileOptions) error { + log.Trace( + "TieredStorage::FlushFile : handle=%d, path=%s", + options.Handle.ID, + options.Handle.Path, + ) + + if !options.Handle.Dirty() { + return nil + } + f := options.Handle.GetFileObject() + if f == nil { + log.Err("TieredStorage::FlushFile : %s no file object in handle", options.Handle.Path) + return syscall.EBADF + } + err := f.Sync() + if err != nil { + log.Err("TieredStorage::FlushFile : %s sync failed [%v]", options.Handle.Path, err) + return syscall.EIO + } + + return nil +} + +func (c *TieredStorage) ReleaseFile(options internal.ReleaseFileOptions) error { + flock := c.fileLocks.Get(options.Handle.Path) + flock.Lock() + defer flock.Unlock() + + flock.Dec() + + var closeErr error + if f := options.Handle.GetFileObject(); f != nil { + closeErr = f.Close() + } + + c.clearHandleDirty(options.Handle) + options.Handle.Cleanup() + + handlemap.Delete(options.Handle.ID) + + if flock.Count() > 0 { + return closeErr + } + + val, ok := c.fileMap.Load(options.Handle.Path) + if !ok { + log.Debug( + "TieredStorage::ReleaseFile : %s has no local data left", + options.Handle.Path, + ) + return closeErr + } + node := val.(*FileNode) + + if !node.cloudBacked.Load() { + c.policy.Enqueue(options.Handle.Path) + return closeErr + } + + if node.isDirty.Load() { + if err := c.uploadCachedFile(options.Handle.Path); err != nil { + // Keep failed uploads local and eligible for eviction. + log.Err( + "TieredStorage::ReleaseFile : upload failed for %s [%v]", + options.Handle.Path, + err, + ) + c.policy.Enqueue(options.Handle.Path) + return err + } + node.isDirty.Store(false) + } + + if err := c.purgeLocal(options.Handle.Path); err != nil { + c.policy.Enqueue(options.Handle.Path) + return err + } + c.policy.Dequeue(options.Handle.Path) + c.fileMap.Delete(options.Handle.Path) + return closeErr +} + +func (c *TieredStorage) uploadCachedFile(name string) error { + localPath, err := c.localPath(name) + if err != nil { + return err + } + + f, err := common.Open(localPath) + if err != nil { + log.Err("TieredStorage::uploadFile : %s open failed [%v]", name, err) + return err + } + defer f.Close() + + err = c.NextComponent().CopyFromFile(internal.CopyFromFileOptions{Name: name, File: f}) + if err != nil { + log.Err("TieredStorage::uploadFile : %s upload failed [%v]", name, err) + return err + } + + value, found := c.fileMap.Load(name) + if !found { + return nil + } + node := value.(*FileNode) + node.cloudBacked.Store(true) + if node.modeDirty.Load() { + err = c.NextComponent().Chmod(internal.ChmodOptions{ + Name: name, + Mode: os.FileMode(node.mode.Load()), + }) + if err != nil && !errors.Is(err, syscall.ENOTSUP) { + return err + } + node.modeDirty.Store(false) + } + if node.ownerDirty.Load() { + err = c.NextComponent().Chown(internal.ChownOptions{ + Name: name, + Owner: int(node.owner.Load()), + Group: int(node.group.Load()), + }) + if err != nil && !errors.Is(err, syscall.ENOTSUP) { + return err + } + node.ownerDirty.Store(false) + } + return nil +} + +// uploadandCleanFile moves an object to cloud storage and removes the local +// copy. It is the LRU's eviction callback, so it runs with the object's file +// lock held. +func (c *TieredStorage) uploadandCleanFile(name string) error { + err := c.uploadCachedFile(name) + if err != nil { + return err + } + if value, found := c.fileMap.Load(name); found { + value.(*FileNode).isDirty.Store(false) + } + err = c.purgeLocal(name) + if err != nil { + log.Err("TieredStorage::uploadandCleanFile : %s remove failed [%v]", name, err) + return err + } + c.fileMap.Delete(name) + return nil +} + +func (c *TieredStorage) RenameFile(options internal.RenameFileOptions) error { + log.Trace("TieredStorage::RenameFile : src=%s, dst=%s", options.Src, options.Dst) + + sflock := c.fileLocks.Get(options.Src) + dflock := c.fileLocks.Get(options.Dst) + + if options.Src < options.Dst { + sflock.Lock() + dflock.Lock() + } else { + dflock.Lock() + sflock.Lock() + } + defer sflock.Unlock() + defer dflock.Unlock() + + val, exists := c.fileMap.Load(options.Src) + if !exists { + return c.NextComponent().RenameFile(options) + } + + node := val.(*FileNode) + srcPath, err := c.localPath(options.Src) + if err != nil { + return err + } + dstPath, err := c.localPath(options.Dst) + if err != nil { + return err + } + if err := os.Rename(srcPath, dstPath); err != nil { + return err + } + c.fileMap.Delete(options.Src) + node.name = options.Dst + c.fileMap.Store(options.Dst, node) + c.policy.Rename(options.Src, options.Dst) + c.renameOpenHandles(options.Src, options.Dst, sflock, dflock) + + if !node.cloudBacked.Load() { + return nil + } + err = c.NextComponent().RenameFile(options) + if err == nil { + return nil + } + // The cloud copy may still be under src, so re-upload under dst. + node.isDirty.Store(true) + if errors.Is(err, os.ErrNotExist) { + return nil + } + log.Err( + "TieredStorage::RenameFile : %s -> %s renamed locally but not in cloud [%v]", + options.Src, + options.Dst, + err, + ) + return err +} + +// Both file locks must be held. +func (c *TieredStorage) renameOpenHandles( + srcName, dstName string, + sflock, dflock *common.LockMapItem, +) { + if sflock.Count() > 0 { + handlemap.GetHandles().Range(func(key, value any) bool { + handle := value.(*handlemap.Handle) + // TODO: do we have to lock before checking the name? + handle.Lock() + if handle.Path == srcName { + handle.Path = dstName + } + handle.Unlock() + return true + }) + for sflock.Count() > 0 { + sflock.Dec() + dflock.Inc() + } + for sflock.DirtyCount() > 0 { + sflock.DecDirty() + dflock.IncDirty() + } + } +} + +func (c *TieredStorage) SyncDir(options internal.SyncDirOptions) error { + return c.NextComponent().SyncDir(options) +} + +// Symlink operations +func (c *TieredStorage) CreateLink(options internal.CreateLinkOptions) error { + return c.NextComponent().CreateLink(options) +} + +func (c *TieredStorage) ReadLink(options internal.ReadLinkOptions) (string, error) { + return c.NextComponent().ReadLink(options) +} + +func (c *TieredStorage) setHandleDirty(handle *handlemap.Handle) { + handle.Lock() + alreadyDirty := handle.Dirty() + if !alreadyDirty { + handle.Flags.Set(handlemap.HandleFlagDirty) + } + handle.Unlock() + if !alreadyDirty { + c.fileLocks.Get(handle.Path).IncDirty() + } +} + +func (c *TieredStorage) clearHandleDirty(handle *handlemap.Handle) { + handle.Lock() + wasDirty := handle.Dirty() + if wasDirty { + handle.Flags.Clear(handlemap.HandleFlagDirty) + } + handle.Unlock() + if wasDirty { + c.fileLocks.Get(handle.Path).DecDirty() + } +} + +func (c *TieredStorage) GetAttr(options internal.GetAttrOptions) (*internal.ObjAttr, error) { + // Lock-free: local copies are published by rename and removed only after + // upload, so each read sees a complete copy; libfuse orders this against + // rename/unlink of the same path. + localPath, err := c.localPath(options.Name) + if err != nil { + return nil, err + } + info, localErr := os.Stat(localPath) + if localErr != nil && !errors.Is(localErr, os.ErrNotExist) { + return nil, localErr + } + + attrs, cloudErr := c.NextComponent().GetAttr(options) + if localErr != nil { + return attrs, cloudErr + } + + localAttrs := newTieredStorageObjAttr(options.Name, info) + if cloudErr != nil || attrs == nil { + return localAttrs, nil + } + if info.IsDir() { + return attrs, nil + } + + merged := *attrs + merged.Size = localAttrs.Size + merged.Mtime = localAttrs.Mtime + return &merged, nil +} + +func (c *TieredStorage) Chmod(options internal.ChmodOptions) error { + flock := c.fileLocks.Get(options.Name) + flock.Lock() + defer flock.Unlock() + + localPath, err := c.localPath(options.Name) + if err != nil { + return err + } + info, localErr := os.Stat(localPath) + value, tracked := c.fileMap.Load(options.Name) + cloudBacked := !tracked || value.(*FileNode).cloudBacked.Load() || + localErr == nil && info.IsDir() + if cloudBacked { + err = c.NextComponent().Chmod(options) + if err != nil && (localErr != nil || !errors.Is(err, os.ErrNotExist)) { + return err + } + } + if localErr != nil { + if cloudBacked { + return nil + } + return localErr + } + if err := os.Chmod(localPath, options.Mode); err != nil { + return err + } + if tracked { + // Keep mode current so a pending retry cannot apply a stale value. + node := value.(*FileNode) + node.mode.Store(uint32(options.Mode)) + if !node.cloudBacked.Load() { + node.modeDirty.Store(true) + } + } + return nil +} + +func (c *TieredStorage) Chown(options internal.ChownOptions) error { + flock := c.fileLocks.Get(options.Name) + flock.Lock() + defer flock.Unlock() + + localPath, err := c.localPath(options.Name) + if err != nil { + return err + } + info, localErr := os.Stat(localPath) + value, tracked := c.fileMap.Load(options.Name) + cloudBacked := !tracked || value.(*FileNode).cloudBacked.Load() || + localErr == nil && info.IsDir() + if cloudBacked { + err = c.NextComponent().Chown(options) + if err != nil && (localErr != nil || !errors.Is(err, os.ErrNotExist)) { + return err + } + } + if localErr != nil { + if cloudBacked { + return nil + } + return localErr + } + if runtime.GOOS != "windows" { + if err := os.Chown(localPath, options.Owner, options.Group); err != nil { + return err + } + } + if tracked { + // Keep owner current so a pending retry cannot apply a stale value. + node := value.(*FileNode) + node.owner.Store(int64(options.Owner)) + node.group.Store(int64(options.Group)) + if !node.cloudBacked.Load() { + node.ownerDirty.Store(true) + } + } + return nil +} + +func (c *TieredStorage) TruncateFile(options internal.TruncateFileOptions) error { + if options.NewSize < 0 { + return syscall.EINVAL + } + if options.Handle == nil { + handle, err := c.OpenFile(internal.OpenFileOptions{ + Name: options.Name, + Flags: os.O_RDWR, + Mode: common.DefaultFilePermissionBits, + }) + if err != nil { + return err + } + options.Handle = handle + if err := c.TruncateFile(options); err != nil { + _ = c.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + return err + } + return c.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + } + + name := options.Handle.Path + flock := c.fileLocks.Get(name) + flock.Lock() + defer flock.Unlock() + + file := options.Handle.GetFileObject() + if file == nil { + return syscall.EBADF + } + info, err := file.Stat() + if err != nil { + return err + } + if err := c.reserveSpace(max(options.NewSize-info.Size(), 0)); err != nil { + return err + } + if err := file.Truncate(options.NewSize); err != nil { + return err + } + + c.setHandleDirty(options.Handle) + if value, found := c.fileMap.Load(name); found { + node := value.(*FileNode) + node.isDirty.Store(true) + c.cacheSize.Add(options.NewSize - node.size.Swap(options.NewSize)) + } + return nil +} + +func (c *TieredStorage) FileUsed(name string) error { + if _, found := c.fileMap.Load(name); found { + c.policy.Enqueue(name) + } + return nil +} + +func (c *TieredStorage) StatFs() (*common.Statfs_t, bool, error) { + const blockSize = 4096 + + physicalFree, err := c.getAvailableSize() + if err != nil { + return nil, false, err + } + available := max(int64(c.maxCacheSize)-c.cacheSize.Used(), 0) + stat := &common.Statfs_t{ + Blocks: uint64(c.maxCacheSize) / blockSize, + Bavail: uint64(available) / blockSize, + Bfree: physicalFree / blockSize, + Bsize: blockSize, + Frsize: blockSize, + Files: 1e9, + Ffree: 1e9, + Namemax: 255, + } + return stat, true, nil +} + +func (c *TieredStorage) GetFileBlockOffsets( + options internal.GetFileBlockOffsetsOptions, +) (*common.BlockOffsetList, error) { + return nil, syscall.ENOTSUP +} + +func (c *TieredStorage) GetCommittedBlockList( + name string, +) (*internal.CommittedBlockList, error) { + return nil, syscall.ENOTSUP +} + +func (c *TieredStorage) StageData(options internal.StageDataOptions) error { + return syscall.ENOTSUP +} + +func (c *TieredStorage) CommitData(options internal.CommitDataOptions) error { + return syscall.ENOTSUP +} + +// ------------------------- Factory ------------------------------------------- + +// Pipeline will call this method to create your object, initialize your variables here +// << DO NOT DELETE ANY AUTO GENERATED CODE HERE >> +func NewTieredStorageComponent() internal.Component { + comp := &TieredStorage{ + fileLocks: common.NewLockMap(), + } + comp.SetName(compName) + return comp +} + +// On init register this component to pipeline and supply your constructor +func init() { + internal.AddComponent(compName, NewTieredStorageComponent) +} diff --git a/component/tiered_storage/tiered_storage_linux.go b/component/tiered_storage/tiered_storage_linux.go new file mode 100644 index 000000000..2a6707851 --- /dev/null +++ b/component/tiered_storage/tiered_storage_linux.go @@ -0,0 +1,65 @@ +//go:build linux + +/* + Licensed under the MIT License . + + Copyright © 2026 Seagate Technology LLC and/or its Affiliates + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "io/fs" + "os" + "syscall" + "time" + + "github.com/Seagate/cloudfuse/internal" + "golang.org/x/sys/unix" +) + +func newTieredStorageObjAttr(path string, info fs.FileInfo) *internal.ObjAttr { + stat := info.Sys().(*syscall.Stat_t) + attrs := &internal.ObjAttr{ + Path: path, + Name: info.Name(), + Size: info.Size(), + Mode: info.Mode(), + Mtime: time.Unix(stat.Mtim.Sec, stat.Mtim.Nsec), + Atime: time.Unix(stat.Atim.Sec, stat.Atim.Nsec), + Ctime: time.Unix(stat.Ctim.Sec, stat.Ctim.Nsec), + } + + if info.Mode()&os.ModeSymlink != 0 { + attrs.Flags.Set(internal.PropFlagSymlink) + } else if info.IsDir() { + attrs.Flags.Set(internal.PropFlagIsDir) + } + return attrs +} + +func (c *TieredStorage) getAvailableSize() (uint64, error) { + stat := &unix.Statfs_t{} + if err := unix.Statfs(c.tmpPath, stat); err != nil { + return 0, err + } + return stat.Bavail * uint64(stat.Bsize), nil +} diff --git a/component/tiered_storage/tiered_storage_linux_test.go b/component/tiered_storage/tiered_storage_linux_test.go new file mode 100644 index 000000000..70dcfe5bc --- /dev/null +++ b/component/tiered_storage/tiered_storage_linux_test.go @@ -0,0 +1,66 @@ +//go:build linux + +/* + Licensed under the MIT License . + + Copyright © 2026 Seagate Technology LLC and/or its Affiliates + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "os" + "path/filepath" + "testing" + + "github.com/Seagate/cloudfuse/common" + "github.com/Seagate/cloudfuse/internal" + "github.com/stretchr/testify/require" + "go.uber.org/mock/gomock" +) + +func TestChownLocalFilePersistsOnUpload(t *testing.T) { + ctrl := gomock.NewController(t) + next := internal.NewMockComponent(ctrl) + path := "local-chown" + cachePath := t.TempDir() + require.NoError(t, os.WriteFile(filepath.Join(cachePath, path), []byte("data"), 0644)) + + storage := &TieredStorage{ + tmpPath: cachePath, + fileLocks: common.NewLockMap(), + cacheSize: newCacheSizeTracker(cachePath, 0), + } + storage.SetNextComponent(next) + node := &FileNode{name: path} + node.size.Store(4) + storage.fileMap.Store(path, node) + + owner, group := os.Getuid(), os.Getgid() + require.NoError(t, storage.Chown(internal.ChownOptions{ + Name: path, Owner: owner, Group: group, + })) + next.EXPECT().CopyFromFile(gomock.Any()).Return(nil) + next.EXPECT().Chown(internal.ChownOptions{ + Name: path, Owner: owner, Group: group, + }).Return(nil) + require.NoError(t, storage.uploadCachedFile(path)) +} diff --git a/component/tiered_storage/tiered_storage_test.go b/component/tiered_storage/tiered_storage_test.go new file mode 100644 index 000000000..20ba1ade1 --- /dev/null +++ b/component/tiered_storage/tiered_storage_test.go @@ -0,0 +1,1796 @@ +/* + Licensed under the MIT License . + + Copyright © 2026 Seagate Technology LLC and/or its Affiliates + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "context" + "crypto/rand" + "errors" + "fmt" + "os" + "path/filepath" + "runtime" + "strings" + "syscall" + "testing" + "time" + + "github.com/Seagate/cloudfuse/common" + "github.com/Seagate/cloudfuse/common/config" + "github.com/Seagate/cloudfuse/common/log" + "github.com/Seagate/cloudfuse/component/loopback" + "github.com/Seagate/cloudfuse/internal" + "github.com/Seagate/cloudfuse/internal/handlemap" + "go.uber.org/mock/gomock" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + "github.com/stretchr/testify/suite" +) + +var home_dir, _ = os.UserHomeDir() + +type tieredStorageTestSuite struct { + suite.Suite + assert *assert.Assertions + tieredStorage *TieredStorage + loopback internal.Component + cache_path string // uses os.Separator (filepath.Join) + fake_storage_path string // uses os.Separator (filepath.Join) + useMock bool + mockCtrl *gomock.Controller + mock *internal.MockComponent +} + +func newLoopbackFS(cachePath string) internal.Component { + loopback := loopback.NewLoopbackFSComponent() + _ = loopback.Configure(true) + return loopback +} + +func newTestTieredStorage(next internal.Component) *TieredStorage { + + tieredStorage := NewTieredStorageComponent() + tieredStorage.SetNextComponent(next) + err := tieredStorage.Configure(true) + if err != nil { + panic(fmt.Sprintf("Unable to configure tiered storage: %v", err)) + } + return tieredStorage.(*TieredStorage) +} + +func randomString(length int) string { + b := make([]byte, length) + _, err := rand.Read(b) + if err != nil { + panic(err) + } + return fmt.Sprintf("%x", b)[:length] +} + +func (suite *tieredStorageTestSuite) SetupTest() { + err := log.SetDefaultLogger("silent", common.LogConfig{Level: common.ELogLevel.LOG_DEBUG()}) + if err != nil { + panic(fmt.Sprintf("Unable to set silent logger as default: %v", err)) + } + rand := randomString(8) + suite.cache_path = filepath.Join(home_dir, "file_cache"+rand) + suite.fake_storage_path = filepath.Join(home_dir, "fake_storage"+rand) + defaultConfig := fmt.Sprintf( + "tiered_storage:\n path: %s\n max-size-mb: 1.0\n offload-io: true\n\nloopbackfs:\n path: %s", + suite.cache_path, + suite.fake_storage_path, + ) + suite.useMock = false + log.Debug("%s", defaultConfig) + + // Delete the temp directories created + err = os.RemoveAll(suite.cache_path) + if err != nil { + fmt.Printf( + "tieredStorageTestSuite::SetupTest : os.RemoveAll(%s) failed [%v]\n", + suite.cache_path, + err, + ) + } + err = os.RemoveAll(suite.fake_storage_path) + if err != nil { + fmt.Printf( + "tieredStorageTestSuite::SetupTest : os.RemoveAll(%s) failed [%v]\n", + suite.fake_storage_path, + err, + ) + } + suite.setupTestHelper(defaultConfig) +} + +func (suite *tieredStorageTestSuite) setupTestHelper(configuration string) { + suite.assert = assert.New(suite.T()) + + err := config.ReadConfigFromReader(strings.NewReader(configuration)) + suite.assert.NoError(err) + if suite.useMock { + suite.mockCtrl = gomock.NewController(suite.T()) + suite.mock = internal.NewMockComponent(suite.mockCtrl) + suite.tieredStorage = newTestTieredStorage(suite.mock) + // always simulate being offline + suite.mock.EXPECT().CloudConnected().AnyTimes().Return(false) + } else { + suite.loopback = newLoopbackFS(suite.fake_storage_path) + suite.tieredStorage = newTestTieredStorage(suite.loopback) + err = suite.loopback.Start(context.Background()) + suite.assert.NoError(err) + } + err = suite.tieredStorage.Start(context.Background()) + if err != nil { + panic(fmt.Sprintf("Unable to start tiered storage [%s]", err.Error())) + } + +} + +func (suite *tieredStorageTestSuite) cleanupTest() { + err := suite.tieredStorage.Stop() + if err != nil { + panic(fmt.Sprintf("Unable to stop tiered storage [%s]", err.Error())) + } + if suite.useMock { + suite.mockCtrl.Finish() + } else { + err = suite.loopback.Stop() + suite.assert.NoError(err) + } + + // Delete the temp directories created + err = os.RemoveAll(suite.cache_path) + suite.assert.NoError(err) + err = os.RemoveAll(suite.fake_storage_path) + suite.assert.NoError(err) +} + +func TestLocalPath(t *testing.T) { + storage := &TieredStorage{tmpPath: filepath.Join(string(os.PathSeparator), "cache")} + absoluteName, err := filepath.Abs("file") + require.NoError(t, err) + + tests := []struct { + name string + expected string + valid bool + }{ + {name: "", expected: storage.tmpPath, valid: true}, + {name: "dir/file", expected: filepath.Join(storage.tmpPath, "dir", "file"), valid: true}, + {name: `dir\file`, expected: filepath.Join(storage.tmpPath, "dir", "file"), valid: true}, + {name: "../file", valid: false}, + {name: "dir/../../file", valid: false}, + {name: absoluteName, valid: false}, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + path, err := storage.localPath(test.name) + if test.valid { + assert.NoError(t, err) + assert.Equal(t, test.expected, path) + return + } + assert.ErrorIs(t, err, syscall.EINVAL) + }) + } +} + +func TestTieredStoragePolicyConfig(t *testing.T) { + tests := []struct { + name string + config string + wantErr bool + high float64 + low float64 + maxEviction uint32 + parallelism int + pollInterval time.Duration + }{ + { + name: "defaults", + config: "max-size-mb: 1", + high: 0.9, low: 0.8, maxEviction: 5000, parallelism: 8, + pollInterval: time.Second, + }, + { + name: "custom", + config: "max-size-mb: 2\n high-threshold: 90\n low-threshold: 70\n" + + " max-eviction: 25\n parallelism: 3\n poll-interval-sec: 4", + high: 0.9, low: 0.7, maxEviction: 25, parallelism: 3, + pollInterval: 4 * time.Second, + }, + {name: "missing capacity", config: "high-threshold: 80", wantErr: true}, + { + name: "reversed thresholds", + config: "max-size-mb: 1\n high-threshold: 60\n low-threshold: 80", + wantErr: true, + }, + { + name: "high threshold over 100", + config: "max-size-mb: 1\n high-threshold: 101", + wantErr: true, + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + configuration := fmt.Sprintf( + "tiered_storage:\n path: %s\n %s", + t.TempDir(), + test.config, + ) + require.NoError(t, config.ReadConfigFromReader(strings.NewReader(configuration))) + + storage := NewTieredStorageComponent().(*TieredStorage) + err := storage.Configure(true) + if test.wantErr { + require.Error(t, err) + return + } + require.NoError(t, err) + assert.InEpsilon(t, test.high, storage.policy.threshold, 0.0001) + assert.InEpsilon(t, test.low, storage.policy.targetRatio, 0.0001) + assert.Equal(t, test.maxEviction, storage.policy.maxEviction) + assert.Equal(t, test.parallelism, storage.policy.numWorkers) + assert.Equal(t, test.pollInterval, storage.policy.pollInterval) + }) + } +} + +func (suite *tieredStorageTestSuite) TestGetAttrLocalOnly() { + defer suite.cleanupTest() + + const path = "local-only" + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: path, Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte("local data")}, + ) + suite.Require().NoError(err) + + attrs, err := suite.tieredStorage.GetAttr(internal.GetAttrOptions{Name: path}) + suite.Require().NoError(err) + suite.assert.Equal(path, attrs.Path) + suite.assert.EqualValues(len("local data"), attrs.Size) +} + +func (suite *tieredStorageTestSuite) TestGetAttrCloudOnly() { + defer suite.cleanupTest() + + const path = "cloud-only" + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + _, err = suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte("cloud data")}, + ) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + + attrs, err := suite.tieredStorage.GetAttr(internal.GetAttrOptions{Name: path}) + suite.Require().NoError(err) + suite.assert.EqualValues(len("cloud data"), attrs.Size) + suite.assert.NoFileExists(filepath.Join(suite.cache_path, path)) +} + +func (suite *tieredStorageTestSuite) TestGetAttrPrefersLocalData() { + defer suite.cleanupTest() + + const path = "local-and-cloud" + cloudHandle, err := suite.loopback.CreateFile( + internal.CreateFileOptions{Name: path, Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: cloudHandle, Data: []byte("cloud")}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: cloudHandle}), + ) + + handle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path, Flags: os.O_RDWR, Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: []byte("local data")}, + ) + suite.Require().NoError(err) + + attrs, err := suite.tieredStorage.GetAttr(internal.GetAttrOptions{Name: path}) + suite.Require().NoError(err) + suite.assert.EqualValues(len("local data"), attrs.Size) + + cloudAttrs, err := suite.loopback.GetAttr(internal.GetAttrOptions{Name: path}) + suite.Require().NoError(err) + suite.assert.EqualValues(len("cloud"), cloudAttrs.Size) +} + +func (suite *tieredStorageTestSuite) TestRestartRestoresLocalStateAndLRUOrder() { + defer suite.cleanupTest() + + for _, path := range []string{"older", "newer"} { + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: path, Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte(path)}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + } + + suite.Require().NoError(suite.tieredStorage.Stop()) + restarted := newTestTieredStorage(suite.loopback) + suite.Require().NoError(restarted.Start(context.Background())) + suite.tieredStorage = restarted + + for _, path := range []string{"older", "newer"} { + value, found := restarted.fileMap.Load(path) + suite.Require().True(found) + node := value.(*FileNode) + suite.assert.False(node.cloudBacked.Load()) + suite.assert.True(node.isDirty.Load()) + _, queued := restarted.policy.nodeMap.Load(path) + suite.assert.True(queued) + } + suite.Require().NotNil(restarted.policy.head) + suite.Require().NotNil(restarted.policy.tail) + suite.assert.Equal("newer", restarted.policy.head.name) + suite.assert.Equal("older", restarted.policy.tail.name) + suite.assert.NoFileExists(filepath.Join(suite.cache_path, tieredStorageSnapshotPath)) +} + +func (suite *tieredStorageTestSuite) TestRestartWithoutSnapshotRecoversConservatively() { + defer suite.cleanupTest() + + suite.Require().NoError(suite.tieredStorage.Stop()) + suite.Require().NoError(os.Remove(filepath.Join(suite.cache_path, tieredStorageSnapshotPath))) + suite.Require().NoError( + os.WriteFile(filepath.Join(suite.cache_path, "recovered"), []byte("data"), 0644), + ) + + restarted := newTestTieredStorage(suite.loopback) + suite.Require().NoError(restarted.Start(context.Background())) + suite.tieredStorage = restarted + + value, found := restarted.fileMap.Load("recovered") + suite.Require().True(found) + node := value.(*FileNode) + suite.assert.False(node.cloudBacked.Load()) + suite.assert.True(node.isDirty.Load()) + _, queued := restarted.policy.nodeMap.Load("recovered") + suite.assert.True(queued) +} + +func (suite *tieredStorageTestSuite) TestRestartWithCorruptSnapshotRecoversFile() { + defer suite.cleanupTest() + + suite.Require().NoError(suite.tieredStorage.Stop()) + suite.Require().NoError(os.WriteFile( + filepath.Join(suite.cache_path, tieredStorageSnapshotPath), + []byte("invalid snapshot"), + 0600, + )) + suite.Require().NoError( + os.WriteFile(filepath.Join(suite.cache_path, "recovered"), []byte("data"), 0644), + ) + + restarted := newTestTieredStorage(suite.loopback) + suite.Require().NoError(restarted.Start(context.Background())) + suite.tieredStorage = restarted + value, found := restarted.fileMap.Load("recovered") + suite.Require().True(found) + suite.assert.True(value.(*FileNode).isDirty.Load()) +} + +func (suite *tieredStorageTestSuite) TestRestartWithStaleSnapshotKeepsChangedFile() { + defer suite.cleanupTest() + + const path = "changed" + localPath := filepath.Join(suite.cache_path, path) + suite.Require().NoError(os.WriteFile(localPath, []byte("old"), 0644)) + node := &FileNode{name: path} + node.size.Store(3) + node.cloudBacked.Store(true) + suite.tieredStorage.fileMap.Store(path, node) + suite.Require().NoError(suite.tieredStorage.Stop()) + suite.Require().NoError(os.WriteFile(localPath, []byte("changed"), 0644)) + + restarted := newTestTieredStorage(suite.loopback) + suite.Require().NoError(restarted.Start(context.Background())) + suite.tieredStorage = restarted + value, found := restarted.fileMap.Load(path) + suite.Require().True(found) + recovered := value.(*FileNode) + suite.assert.False(recovered.cloudBacked.Load()) + suite.assert.True(recovered.isDirty.Load()) + suite.assert.FileExists(localPath) +} + +func (suite *tieredStorageTestSuite) TestRestartRemovesPartialDownload() { + defer suite.cleanupTest() + + suite.Require().NoError(suite.tieredStorage.Stop()) + partialPath := filepath.Join( + suite.cache_path, + "file.123"+partialDownloadSuffix, + ) + suite.Require().NoError(os.WriteFile(partialPath, []byte("partial"), 0644)) + + restarted := newTestTieredStorage(suite.loopback) + suite.Require().NoError(restarted.Start(context.Background())) + suite.tieredStorage = restarted + suite.assert.NoFileExists(partialPath) + _, found := restarted.fileMap.Load("file.123" + partialDownloadSuffix) + suite.assert.False(found) +} + +func TestUploadFailureKeepsLocalFile(t *testing.T) { + ctrl := gomock.NewController(t) + next := internal.NewMockComponent(ctrl) + cachePath := t.TempDir() + localPath := filepath.Join(cachePath, "file") + require.NoError(t, os.WriteFile(localPath, []byte("data"), 0644)) + + storage := &TieredStorage{ + tmpPath: cachePath, + fileLocks: common.NewLockMap(), + cacheSize: newCacheSizeTracker(cachePath, 0), + } + storage.SetNextComponent(next) + storage.fileMap.Store("file", &FileNode{name: "file"}) + next.EXPECT().CopyFromFile(gomock.Any()).Return(errors.New("upload failed")) + + err := storage.uploadandCleanFile("file") + require.Error(t, err) + assert.FileExists(t, localPath) + _, found := storage.fileMap.Load("file") + assert.True(t, found) +} + +func (suite *tieredStorageTestSuite) TestCreateFileRejectsExistingCloudObject() { + defer suite.cleanupTest() + + const path = "existing-cloud" + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + + _, err = suite.tieredStorage.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.assert.ErrorIs(err, syscall.EEXIST) +} + +func (suite *tieredStorageTestSuite) TestCreateFileRejectsExistingLocalObject() { + defer suite.cleanupTest() + + const path = "existing-local" + _, err := suite.tieredStorage.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + + _, err = suite.tieredStorage.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.assert.ErrorIs(err, syscall.EEXIST) +} + +func (suite *tieredStorageTestSuite) TestOpenFileExclusiveCreateRejectsExistingCloudObject() { + defer suite.cleanupTest() + + const path = "exclusive-cloud" + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + + _, err = suite.tieredStorage.OpenFile(internal.OpenFileOptions{ + Name: path, Flags: os.O_CREATE | os.O_EXCL | os.O_RDWR, Mode: 0644, + }) + suite.assert.ErrorIs(err, syscall.EEXIST) +} + +func (suite *tieredStorageTestSuite) TestOpenFileTruncatesCloudObjectWithoutDownloading() { + defer suite.cleanupTest() + + const path = "truncate-cloud" + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + _, err = suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte("old cloud data")}, + ) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + + handle, err = suite.tieredStorage.OpenFile(internal.OpenFileOptions{ + Name: path, Flags: os.O_WRONLY | os.O_TRUNC, Mode: 0644, + }) + suite.Require().NoError(err) + suite.assert.True(handle.Dirty()) + info, err := os.Stat(filepath.Join(suite.cache_path, path)) + suite.Require().NoError(err) + suite.assert.Zero(info.Size()) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + + attrs, err := suite.loopback.GetAttr(internal.GetAttrOptions{Name: path}) + suite.Require().NoError(err) + suite.assert.Zero(attrs.Size) +} + +func (suite *tieredStorageTestSuite) TestOpenFileHonorsReadOnlyFlag() { + defer suite.cleanupTest() + + const path = "read-only" + data := []byte("cloud data") + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + _, err = suite.loopback.WriteFile(&internal.WriteFileOptions{Handle: handle, Data: data}) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + + handle, err = suite.tieredStorage.OpenFile(internal.OpenFileOptions{ + Name: path, Flags: os.O_RDONLY, + }) + suite.Require().NoError(err) + buffer := make([]byte, len(data)) + _, err = suite.tieredStorage.ReadInBuffer( + &internal.ReadInBufferOptions{Handle: handle, Data: buffer}, + ) + suite.Require().NoError(err) + suite.assert.Equal(data, buffer) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) +} + +func (suite *tieredStorageTestSuite) TestTruncateOpenLocalFile() { + defer suite.cleanupTest() + + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "local-truncate", Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte("original")}, + ) + suite.Require().NoError(err) + + err = suite.tieredStorage.TruncateFile(internal.TruncateFileOptions{ + Name: "local-truncate", Handle: handle, NewSize: 3, + }) + suite.Require().NoError(err) + info, err := handle.GetFileObject().Stat() + suite.Require().NoError(err) + suite.assert.EqualValues(3, info.Size()) + suite.assert.True(handle.Dirty()) + value, found := suite.tieredStorage.fileMap.Load("local-truncate") + suite.Require().True(found) + suite.assert.EqualValues(3, value.(*FileNode).size.Load()) +} + +func (suite *tieredStorageTestSuite) TestTruncateCloudFileByPath() { + defer suite.cleanupTest() + + const path = "cloud-truncate" + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + _, err = suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte("cloud data")}, + ) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + + suite.Require().NoError(suite.tieredStorage.TruncateFile( + internal.TruncateFileOptions{Name: path, NewSize: 4}, + )) + attrs, err := suite.loopback.GetAttr(internal.GetAttrOptions{Name: path}) + suite.Require().NoError(err) + suite.assert.EqualValues(4, attrs.Size) + suite.assert.NoFileExists(filepath.Join(suite.cache_path, path)) +} + +func (suite *tieredStorageTestSuite) TestTruncateInvalidSize() { + defer suite.cleanupTest() + + err := suite.tieredStorage.TruncateFile( + internal.TruncateFileOptions{Name: "invalid", NewSize: -1}, + ) + suite.assert.ErrorIs(err, syscall.EINVAL) +} + +func (suite *tieredStorageTestSuite) TestStatFs() { + defer suite.cleanupTest() + + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "usage", Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.tieredStorage.WriteFile(&internal.WriteFileOptions{ + Handle: handle, + Data: make([]byte, 4096), + }) + suite.Require().NoError(err) + + stat, populated, err := suite.tieredStorage.StatFs() + suite.Require().NoError(err) + suite.assert.True(populated) + suite.assert.EqualValues(4096, stat.Bsize) + suite.assert.Equal(uint64(suite.tieredStorage.maxCacheSize)/4096, stat.Blocks) + expectedAvailable := max( + int64(suite.tieredStorage.maxCacheSize)-suite.tieredStorage.cacheSize.Used(), + 0, + ) + suite.assert.Equal(uint64(expectedAvailable)/4096, stat.Bavail) + suite.assert.Positive(stat.Bfree) +} + +func (suite *tieredStorageTestSuite) TestSymlink() { + defer suite.cleanupTest() + + suite.Require().NoError(suite.tieredStorage.CreateLink( + internal.CreateLinkOptions{Name: "link", Target: "target"}, + )) + target, err := suite.tieredStorage.ReadLink( + internal.ReadLinkOptions{Name: "link"}, + ) + suite.Require().NoError(err) + suite.assert.Equal("target", target) + suite.assert.NoFileExists(filepath.Join(suite.cache_path, "link")) +} + +func (suite *tieredStorageTestSuite) TestChmodLocalFilePersistsOnUpload() { + defer suite.cleanupTest() + if runtime.GOOS == "windows" { + suite.T().Skip("Windows only supports the read-only mode bit") + } + + const path = "local-chmod" + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: path, Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError(suite.tieredStorage.Chmod( + internal.ChmodOptions{Name: path, Mode: 0600}, + )) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + + flock := suite.tieredStorage.fileLocks.Get(path) + flock.Lock() + suite.Require().NoError(suite.tieredStorage.uploadandCleanFile(path)) + flock.Unlock() + info, err := os.Stat(filepath.Join(suite.fake_storage_path, path)) + suite.Require().NoError(err) + suite.assert.Equal(os.FileMode(0600), info.Mode().Perm()) +} + +func (suite *tieredStorageTestSuite) TestChmodCloudBackedFileUpdatesBothTiers() { + defer suite.cleanupTest() + if runtime.GOOS == "windows" { + suite.T().Skip("Windows only supports the read-only mode bit") + } + + const path = "cloud-chmod" + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + handle, err = suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path, Flags: os.O_RDWR, Mode: 0644}, + ) + suite.Require().NoError(err) + + suite.Require().NoError(suite.tieredStorage.Chmod( + internal.ChmodOptions{Name: path, Mode: 0600}, + )) + localInfo, err := os.Stat(filepath.Join(suite.cache_path, path)) + suite.Require().NoError(err) + cloudInfo, err := os.Stat(filepath.Join(suite.fake_storage_path, path)) + suite.Require().NoError(err) + suite.assert.Equal(os.FileMode(0600), localInfo.Mode().Perm()) + suite.assert.Equal(os.FileMode(0600), cloudInfo.Mode().Perm()) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) +} + +func (suite *tieredStorageTestSuite) TestChmodCloudOnlyFile() { + defer suite.cleanupTest() + if runtime.GOOS == "windows" { + suite.T().Skip("Windows only supports the read-only mode bit") + } + + const path = "cloud-only-chmod" + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0644}) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) + + suite.Require().NoError(suite.tieredStorage.Chmod( + internal.ChmodOptions{Name: path, Mode: 0600}, + )) + info, err := os.Stat(filepath.Join(suite.fake_storage_path, path)) + suite.Require().NoError(err) + suite.assert.Equal(os.FileMode(0600), info.Mode().Perm()) +} + +func (suite *tieredStorageTestSuite) TestStreamDirMergesLocalAndCloudEntries() { + defer suite.cleanupTest() + + localHandle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "local", Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: localHandle, Data: []byte("local data")}, + ) + suite.Require().NoError(err) + + for _, path := range []string{"cloud", "shared"} { + handle, err := suite.loopback.CreateFile( + internal.CreateFileOptions{Name: path, Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte("cloud")}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.loopback.ReleaseFile( + internal.ReleaseFileOptions{Handle: handle}, + ), + ) + } + sharedHandle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: "shared", Flags: os.O_RDWR, Mode: 0644}, + ) + suite.Require().NoError(err) + _, err = suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: sharedHandle, Data: []byte("local data")}, + ) + suite.Require().NoError(err) + + attrs, token, err := suite.tieredStorage.StreamDir(internal.StreamDirOptions{}) + suite.Require().NoError(err) + suite.assert.Empty(token) + suite.Require().Len(attrs, 3) + suite.assert.Equal([]string{"cloud", "local", "shared"}, []string{ + attrs[0].Name, attrs[1].Name, attrs[2].Name, + }) + suite.assert.EqualValues(len("local data"), attrs[1].Size) + suite.assert.EqualValues(len("local data"), attrs[2].Size) +} + +func (suite *tieredStorageTestSuite) TestCreateAndDeleteLocalDirectory() { + defer suite.cleanupTest() + + const path = "directory" + suite.Require().NoError( + suite.tieredStorage.CreateDir(internal.CreateDirOptions{Name: path, Mode: 0755}), + ) + attrs, err := suite.tieredStorage.GetAttr(internal.GetAttrOptions{Name: path}) + suite.Require().NoError(err) + suite.assert.True(attrs.IsDir()) + suite.assert.True(suite.tieredStorage.IsDirEmpty(internal.IsDirEmptyOptions{Name: path})) + suite.Require().NoError(suite.tieredStorage.DeleteDir(internal.DeleteDirOptions{Name: path})) + _, err = suite.tieredStorage.GetAttr(internal.GetAttrOptions{Name: path}) + suite.assert.ErrorIs(err, syscall.ENOENT) +} + +func (suite *tieredStorageTestSuite) TestStreamDirListsImplicitLocalDirectory() { + defer suite.cleanupTest() + + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "directory/file", Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + suite.Require().NoError( + suite.tieredStorage.OpenDir(internal.OpenDirOptions{Name: "directory"}), + ) + + attrs, token, err := suite.tieredStorage.StreamDir( + internal.StreamDirOptions{Name: "directory"}, + ) + suite.Require().NoError(err) + suite.assert.Empty(token) + suite.Require().Len(attrs, 1) + suite.assert.Equal("directory/file", attrs[0].Path) + suite.Require().NoError( + suite.tieredStorage.CloseDir(internal.CloseDirOptions{Name: "directory"}), + ) +} + +func (suite *tieredStorageTestSuite) TestRenameLocalOnlyDirectory() { + defer suite.cleanupTest() + + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "src/file", Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + + suite.Require().NoError(suite.tieredStorage.RenameDir( + internal.RenameDirOptions{Src: "src", Dst: "dst"}, + )) + suite.assert.NoFileExists(filepath.Join(suite.cache_path, "src", "file")) + suite.assert.FileExists(filepath.Join(suite.cache_path, "dst", "file")) + _, oldFound := suite.tieredStorage.fileMap.Load("src/file") + _, newFound := suite.tieredStorage.fileMap.Load("dst/file") + suite.assert.False(oldFound) + suite.assert.True(newFound) + _, oldQueued := suite.tieredStorage.policy.nodeMap.Load("src/file") + _, newQueued := suite.tieredStorage.policy.nodeMap.Load("dst/file") + suite.assert.False(oldQueued) + suite.assert.True(newQueued) +} + +func (suite *tieredStorageTestSuite) TestRenameMixedDirectory() { + defer suite.cleanupTest() + + suite.Require().NoError(suite.loopback.CreateDir( + internal.CreateDirOptions{Name: "src", Mode: 0755}, + )) + cloudHandle, err := suite.loopback.CreateFile( + internal.CreateFileOptions{Name: "src/cloud", Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: cloudHandle}), + ) + localHandle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "src/local", Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: localHandle}), + ) + + suite.Require().NoError(suite.tieredStorage.RenameDir( + internal.RenameDirOptions{Src: "src", Dst: "dst"}, + )) + suite.assert.FileExists(filepath.Join(suite.cache_path, "dst", "local")) + suite.assert.FileExists(filepath.Join(suite.fake_storage_path, "dst", "cloud")) + suite.assert.NoDirExists(filepath.Join(suite.cache_path, "src")) + suite.assert.NoDirExists(filepath.Join(suite.fake_storage_path, "src")) +} + +func (suite *tieredStorageTestSuite) TestRenameDirectoryUpdatesOpenHandle() { + defer suite.cleanupTest() + + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "src/open", Mode: 0644}, + ) + suite.Require().NoError(err) + handlemap.Add(handle) + err = suite.tieredStorage.RenameDir(internal.RenameDirOptions{Src: "src", Dst: "dst"}) + if runtime.GOOS == "windows" { + // Windows refuses to rename a directory containing open files. + suite.Require().Error(err) + suite.assert.Equal("src/open", handle.Path) + suite.assert.FileExists(filepath.Join(suite.cache_path, "src", "open")) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + return + } + suite.Require().NoError(err) + suite.assert.Equal("dst/open", handle.Path) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + _, queued := suite.tieredStorage.policy.nodeMap.Load("dst/open") + suite.assert.True(queued) +} + +func (suite *tieredStorageTestSuite) writeCloudFile(name, data string) { + handle, err := suite.loopback.CreateFile(internal.CreateFileOptions{Name: name, Mode: 0644}) + suite.Require().NoError(err) + _, err = suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Data: []byte(data)}, + ) + suite.Require().NoError(err) + suite.Require().NoError(suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle})) +} + +func (suite *tieredStorageTestSuite) TestRenameFileCloudFailureKeepsLocalRename() { + defer suite.cleanupTest() + + suite.writeCloudFile("src", "data") + handle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: "src", Flags: os.O_RDONLY, Mode: 0644}, + ) + suite.Require().NoError(err) + handlemap.Add(handle) + // An existing cloud directory makes the cloud rename fail. + blocker := filepath.Join(suite.fake_storage_path, "dst") + suite.Require().NoError(os.Mkdir(blocker, 0755)) + + err = suite.tieredStorage.RenameFile(internal.RenameFileOptions{Src: "src", Dst: "dst"}) + suite.Require().Error(err) + suite.assert.FileExists(filepath.Join(suite.cache_path, "dst")) + suite.assert.Equal("dst", handle.Path) + value, found := suite.tieredStorage.fileMap.Load("dst") + suite.Require().True(found) + suite.assert.True(value.(*FileNode).isDirty.Load()) + + suite.Require().NoError(os.Remove(blocker)) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + data, err := os.ReadFile(filepath.Join(suite.fake_storage_path, "dst")) + suite.Require().NoError(err) + suite.assert.Equal("data", string(data)) + suite.assert.NoFileExists(filepath.Join(suite.cache_path, "dst")) +} + +func (suite *tieredStorageTestSuite) TestRenameFileMissingCloudObjectSucceeds() { + defer suite.cleanupTest() + + suite.writeCloudFile("src", "data") + handle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: "src", Flags: os.O_RDONLY, Mode: 0644}, + ) + suite.Require().NoError(err) + handlemap.Add(handle) + suite.Require().NoError(os.Remove(filepath.Join(suite.fake_storage_path, "src"))) + + suite.Require().NoError( + suite.tieredStorage.RenameFile(internal.RenameFileOptions{Src: "src", Dst: "dst"}), + ) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + suite.assert.FileExists(filepath.Join(suite.fake_storage_path, "dst")) +} + +func (suite *tieredStorageTestSuite) TestRenameDirCloudFailureKeepsLocalRename() { + defer suite.cleanupTest() + if runtime.GOOS == "windows" { + suite.T().Skip("Windows refuses to rename a directory containing open files") + } + + suite.Require().NoError(suite.loopback.CreateDir( + internal.CreateDirOptions{Name: "src", Mode: 0755}, + )) + suite.writeCloudFile("src/cloud", "data") + cloudHandle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: "src/cloud", Flags: os.O_RDONLY, Mode: 0644}, + ) + suite.Require().NoError(err) + handlemap.Add(cloudHandle) + localHandle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "src/local", Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: localHandle}), + ) + // An existing cloud directory makes the cloud rename fail. + suite.Require().NoError(os.Mkdir(filepath.Join(suite.fake_storage_path, "dst"), 0755)) + + err = suite.tieredStorage.RenameDir(internal.RenameDirOptions{Src: "src", Dst: "dst"}) + suite.Require().Error(err) + suite.assert.NoDirExists(filepath.Join(suite.cache_path, "src")) + suite.assert.FileExists(filepath.Join(suite.cache_path, "dst", "local")) + suite.assert.Equal("dst/cloud", cloudHandle.Path) + cloudNode, found := suite.tieredStorage.fileMap.Load("dst/cloud") + suite.Require().True(found) + suite.assert.True(cloudNode.(*FileNode).isDirty.Load()) + localNode, found := suite.tieredStorage.fileMap.Load("dst/local") + suite.Require().True(found) + suite.assert.False(localNode.(*FileNode).cloudBacked.Load()) + + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: cloudHandle}), + ) + data, err := os.ReadFile(filepath.Join(suite.fake_storage_path, "dst", "cloud")) + suite.Require().NoError(err) + suite.assert.Equal("data", string(data)) +} + +//Testing OpenFile + +func (suite *tieredStorageTestSuite) TestOpenFileNotInCache() { + defer suite.cleanupTest() + path := "file7" + + //put file in cloud + handle, _ := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + testData := "test data" + data := []byte(testData) + _, err := suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + err = suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + //open file through tiered storage, should succeed and return a handle with correct path + handle, err = suite.tieredStorage.OpenFile( + internal.OpenFileOptions{ + Name: path, + Flags: os.O_RDWR, + Mode: 0666, //random mode, since we didn't do the other stuff yet + }, + ) + suite.assert.NoError(err) + suite.assert.Equal(path, handle.Path) + + // Verify it was now downloaded to the local tiered storage cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) +} + +func (suite *tieredStorageTestSuite) TestOpenFileInCache() { + defer suite.cleanupTest() + path := "file8" + handle, _ := suite.tieredStorage.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + testData := "test data" + data := []byte(testData) + _, err := suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + err = suite.tieredStorage.FlushFile(internal.FlushFileOptions{Handle: handle}) + suite.assert.NoError(err) + + // Download is required + handle, err = suite.tieredStorage.OpenFile(internal.OpenFileOptions{Name: path, Mode: 0777}) + suite.assert.NoError(err) + suite.assert.Equal(path, handle.Path) + suite.assert.False(handle.Dirty()) + + // File should exist in cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) +} + +func (suite *tieredStorageTestSuite) TestOpenFileOCreate() { + defer suite.cleanupTest() + path := "file9" + handle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path, Flags: os.O_CREATE, Mode: 0777}, + ) + suite.assert.NoError(err) + suite.assert.Equal(path, handle.Path) + suite.assert.True(handle.Dirty()) + // File should exist in cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) + +} + +func (suite *tieredStorageTestSuite) TestOpenFileOCreateExistsLocal() { + defer suite.cleanupTest() + path := "file10" + handle, _ := suite.tieredStorage.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + testData := "test data" + data := []byte(testData) + _, err := suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + err = suite.tieredStorage.FlushFile(internal.FlushFileOptions{Handle: handle}) + suite.assert.NoError(err) + + // Download is required + handle, err = suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path, Flags: os.O_CREATE, Mode: 0777}, + ) + suite.assert.NoError(err) + suite.assert.Equal(path, handle.Path) + suite.assert.False(handle.Dirty()) + + // File should exist in cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) + + //Make sure data didn't get modified + d, err := os.ReadFile(filepath.Join(suite.cache_path, path)) + suite.assert.NoError(err) + suite.assert.Equal(data, d) + +} + +func (suite *tieredStorageTestSuite) TestOpenFileOCreateExistsCloud() { + defer suite.cleanupTest() + path := "file11" + + //put file in cloud + handle, _ := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + testData := "test data" + data := []byte(testData) + _, err := suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + err = suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + //open file through tiered storage, should succeed and return a handle with correct path + handle, err = suite.tieredStorage.OpenFile( + internal.OpenFileOptions{ + Name: path, + Flags: os.O_CREATE, + Mode: 0777, + }, + ) + suite.assert.NoError(err) + suite.assert.Equal(path, handle.Path) + + // Verify it was now downloaded to the local tiered storage cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) +} + +//Testing WriteFile + +func (suite *tieredStorageTestSuite) TestWriteFile() { + defer suite.cleanupTest() + path := "file11" + handle, _ := suite.tieredStorage.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + handle.Flags.Clear( + handlemap.HandleFlagDirty, + ) // Technically create file will mark it as dirty, we just want to check write file updates the dirty flag, so temporarily set this to false + testData := "test data" + data := []byte(testData) + length, err := suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + + suite.assert.NoError(err) + suite.assert.Equal(len(data), length) + // Check that the local cache updated with data + d, _ := os.ReadFile(filepath.Join(suite.cache_path, path)) + suite.assert.Equal(data, d) + suite.assert.True(handle.Dirty()) +} + +func (suite *tieredStorageTestSuite) TestWriteFileErrorBadFd() { + defer suite.cleanupTest() + // Setup + file := "file20" + //bad handle + handle := handlemap.NewHandle(file) + bytesWrittength, err := suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle}, + ) + suite.assert.Error(err) + suite.assert.EqualValues(syscall.EBADF, err) + suite.assert.Equal(0, bytesWrittength) +} + +// Testing Create File +func (suite *tieredStorageTestSuite) TestCreateFile() { + defer suite.cleanupTest() + // Default is to not create empty files on create file to support immutable storage. + path := "file12" + options := internal.CreateFileOptions{Name: path} + f, err := suite.tieredStorage.CreateFile(options) + + suite.assert.NoError(err) + suite.assert.True(f.Dirty()) // Handle should be dirty since it was not created in cloud storage + + // Path should be added to the file cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) + // Path should not be in fake storage + suite.assert.NoFileExists(filepath.Join(suite.fake_storage_path, path)) +} + +// Testing Release File +func (suite *tieredStorageTestSuite) TestReleaseCloudNoDirtyFile() { + defer suite.cleanupTest() + path := "file13" + + //put file in cloud + handle, _ := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + err := suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + //open file through tiered storage, should succeed and return a handle with correct path + handle, openErr := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{ + Name: path, + Flags: os.O_RDWR, + Mode: 0666, //random mode, since we didn't do the other stuff yet + }, + ) + suite.assert.NoError(openErr) + + // Verify it was now downloaded to the local tiered storage cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) + + //As of now, the file would be cloudbacked and exist in map + // suite.tieredStorage.mu.Lock() + // node, exists := suite.tieredStorage.fileMap[path] + // suite.tieredStorage.mu.Unlock() + + val, ok := suite.tieredStorage.fileMap.Load(path) + node := val.(*FileNode) + + suite.assert.True(node.cloudBacked.Load(), "File should be marked as cloud-backed") + suite.assert.True(ok, "File should be tracked in the fileMap") + + //File should be "cloudBacked" and not dirty so on release the file should be deleted from local and the handle clean + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + _, err = os.Stat(filepath.Join(suite.cache_path, path)) + suite.assert.True(os.IsNotExist(err), "File should be deleted from cache after release") + +} + +func (suite *tieredStorageTestSuite) TestReleaseCloudDirtyFile() { + defer suite.cleanupTest() + path := "file13" + + //put file in cloud + handle, _ := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + testData := "test data" + data := []byte(testData) + _, err := suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + err = suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + //open file through tiered storage, should succeed and return a handle with correct path + handle, openErr := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{ + Name: path, + Flags: os.O_RDWR, + Mode: 0666, //random mode, since we didn't do the other stuff yet + }, + ) + suite.assert.NoError(openErr) + + // Verify it was now downloaded to the local tiered storage cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) + + _, err = suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + + // Handle should be dirty since it was not created in cloud storage + suite.assert.True(handle.Dirty()) + + //As of now, the file would be cloudbacked and exist in map + // suite.tieredStorage.mu.Lock() + // node, exists := suite.tieredStorage.fileMap[path] + // suite.tieredStorage.mu.Unlock() + + val, exists := suite.tieredStorage.fileMap.Load(path) + node := val.(*FileNode) + + suite.assert.True(node.cloudBacked.Load(), "File should be marked as cloud-backed") + suite.assert.True(exists, "File should be tracked in the fileMap") + + //File should be "cloudBacked" and dirty so on release the file should be deleted from local and the handle clean + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + _, err = os.Stat(filepath.Join(suite.cache_path, path)) + suite.assert.True(os.IsNotExist(err), "File should be deleted from cache after release") + + //Must check that file by its data is actually in the cloud + _, err = suite.tieredStorage.NextComponent().GetAttr( + internal.GetAttrOptions{Name: path, RetrieveMetadata: true}) + suite.assert.NoError(err) + + //tmpFile to hold cloud data || WARNING AI SLOP BELOW, I did not write below this + //It just checks if the data is preserved + tmpFile, err := os.CreateTemp("", "cloud_verify") + suite.assert.NoError(err) + defer os.Remove(tmpFile.Name()) + defer tmpFile.Close() + + // 2. Copy from the cloud (loopback) to the temporary file + err = suite.loopback.CopyToFile(internal.CopyToFileOptions{ + Name: path, + Offset: 0, + Count: 0, // 0 usually means the whole file + File: tmpFile, + }) + suite.assert.NoError(err) + + // 3. Read the data back from the temp file and verify + dataFromCloud, err := os.ReadFile(tmpFile.Name()) + suite.assert.NoError(err) + suite.assert.Equal( + data, + dataFromCloud, + "The cloud version should match the modified local version", + ) + +} + +func (suite *tieredStorageTestSuite) TestReadInBuffer() { + defer suite.cleanupTest() + // Setup + file := "file14" + + // put file in cloud and write to it + handle, _ := suite.loopback.CreateFile(internal.CreateFileOptions{Name: file, Mode: 0777}) + testData := "test data" + data := []byte(testData) + _, err := suite.loopback.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + err = suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + // Must check that file by its data is actually in the cloud + _, err = suite.tieredStorage.NextComponent().GetAttr( + internal.GetAttrOptions{Name: file, RetrieveMetadata: true}) + suite.assert.NoError(err) + + handle, _ = suite.tieredStorage.OpenFile(internal.OpenFileOptions{Name: file, Mode: 0777}) + + output := make([]byte, 9) + length, err := suite.tieredStorage.ReadInBuffer( + &internal.ReadInBufferOptions{Handle: handle, Offset: 0, Data: output}, + ) + suite.assert.NoError(err) + suite.assert.Equal(data, output) + suite.assert.Equal(len(data), length) +} + +func (suite *tieredStorageTestSuite) TestReadInBufferErrorBadFd() { + defer suite.cleanupTest() + // Setup + file := "file15" + handle := handlemap.NewHandle(file) + length, err := suite.tieredStorage.ReadInBuffer(&internal.ReadInBufferOptions{Handle: handle}) + suite.assert.Error(err) + suite.assert.EqualValues(syscall.EBADF, err) + suite.assert.Equal(0, length) +} + +func (suite *tieredStorageTestSuite) TestWriteReadDirtyState() { + defer suite.cleanupTest() + path := "file16" + + //put file in cloud + handle, _ := suite.loopback.CreateFile(internal.CreateFileOptions{Name: path, Mode: 0777}) + err := suite.loopback.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + //open file through tiered storage, should succeed and return a handle with correct path + handle, openErr := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{ + Name: path, + Flags: os.O_RDWR, + Mode: 0666, //random mode, since we didn't do the other stuff yet + }, + ) + suite.assert.NoError(openErr) + + // Verify it was now downloaded to the local tiered storage cache + in map + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) + // suite.tieredStorage.mu.Lock() + // node, exists := suite.tieredStorage.fileMap[path] + // suite.tieredStorage.mu.Unlock() + + val, exists := suite.tieredStorage.fileMap.Load(path) + node := val.(*FileNode) + + suite.assert.True(node.cloudBacked.Load(), "File should be marked as cloud-backed") + suite.assert.True(exists, "File should be tracked in the fileMap") + + //1. Write to handle + testData := "test data" + data := []byte(testData) + length, err := suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + suite.assert.Equal(len(data), length) + + //check the handle is dirty + suite.assert.True(handle.Dirty()) + + //2. New Read Handle to same file + handle2, openErr := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{ + Name: path, + Flags: os.O_RDWR, + Mode: 0666, //random mode, since we didn't do the other stuff yet + }, + ) + suite.assert.NoError(openErr) + output := make([]byte, 9) + length, err = suite.tieredStorage.ReadInBuffer( + &internal.ReadInBufferOptions{Handle: handle2, Offset: 0, Data: output}, + ) + suite.assert.NoError(err) + suite.assert.Equal(data, output) + suite.assert.Equal(len(data), length) + + //3. Release The write handle, should still be in local with a dirty handle + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + _, err = os.Stat(filepath.Join(suite.cache_path, path)) + suite.assert.False(os.IsNotExist(err), "File should not be uploaded") + + //check still dirty + suite.assert.False(handle2.Dirty()) + suite.assert.False(handle.Dirty()) + + // suite.tieredStorage.mu.Lock() + // node, _ = suite.tieredStorage.fileMap[path] + // suite.tieredStorage.mu.Unlock() + + val, _ = suite.tieredStorage.fileMap.Load(path) + node = val.(*FileNode) + + suite.assert.True(node.isDirty.Load(), "File should be marked as dirty") + + //4. Release the read should upload to cloud + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle2}) + suite.assert.NoError(err) + _, err = os.Stat(filepath.Join(suite.cache_path, path)) + suite.assert.True(os.IsNotExist(err), "File should be deleted from cache after release") + + //5. Check data + //It just checks if the data is preserved + tmpFile, err := os.CreateTemp("", "cloud_verify") + suite.assert.NoError(err) + defer os.Remove(tmpFile.Name()) + defer tmpFile.Close() + + // 2. Copy from the cloud (loopback) to the temporary file + err = suite.loopback.CopyToFile(internal.CopyToFileOptions{ + Name: path, + Offset: 0, + Count: 0, // 0 usually means the whole file + File: tmpFile, + }) + suite.assert.NoError(err) + + // 3. Read the data back from the temp file and verify + dataFromCloud, err := os.ReadFile(tmpFile.Name()) + suite.assert.NoError(err) + suite.assert.Equal( + data, + dataFromCloud, + "The cloud version should match the modified local version", + ) +} + +func (suite *tieredStorageTestSuite) TestReleaseLocalToLRUQueue() { + //Ok this next test is to essentially go through an iteration of LRU, + + //1. Initialize a local only file + defer suite.cleanupTest() + path := "file17" + handle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path, Flags: os.O_CREATE, Mode: 0777}, + ) + suite.assert.NoError(err) + suite.assert.Equal(path, handle.Path) + suite.assert.True(handle.Dirty()) + // File should exist in cache + suite.assert.FileExists(filepath.Join(suite.cache_path, path)) + + val, exists := suite.tieredStorage.fileMap.Load(path) + node := val.(*FileNode) + + suite.assert.False(node.cloudBacked.Load(), "File should not be marked as cloud-backed") + suite.assert.True(exists, "File should be tracked in the fileMap") + + // 2. Release this local file + //File is local only so it shouldn't be deleted from local knowledge + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + // 3. Check if its in the LRU Queue + suite.assert.Equal(path, suite.tieredStorage.policy.head.name) + suite.assert.Equal(path, suite.tieredStorage.policy.tail.name) + +} + +func (suite *tieredStorageTestSuite) TestFileUsedPromotesLocalFile() { + defer suite.cleanupTest() + + for _, path := range []string{"first", "second"} { + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: path, Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError( + suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}), + ) + } + suite.assert.Equal("first", suite.tieredStorage.policy.tail.name) + suite.Require().NoError(suite.tieredStorage.FileUsed("first")) + suite.assert.Equal("first", suite.tieredStorage.policy.head.name) +} + +func (suite *tieredStorageTestSuite) TestReleaseReturnsCloseError() { + defer suite.cleanupTest() + + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: "close-error", Mode: 0644}, + ) + suite.Require().NoError(err) + suite.Require().NoError(handle.GetFileObject().Close()) + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.Error(err) + _, queued := suite.tieredStorage.policy.nodeMap.Load("close-error") + suite.assert.True(queued) +} + +func (suite *tieredStorageTestSuite) TestBlockAPIsUnsupported() { + defer suite.cleanupTest() + + _, err := suite.tieredStorage.GetFileBlockOffsets(internal.GetFileBlockOffsetsOptions{}) + suite.assert.ErrorIs(err, syscall.ENOTSUP) + _, err = suite.tieredStorage.GetCommittedBlockList("file") + suite.assert.ErrorIs(err, syscall.ENOTSUP) + suite.assert.ErrorIs( + suite.tieredStorage.StageData(internal.StageDataOptions{}), + syscall.ENOTSUP, + ) + suite.assert.ErrorIs( + suite.tieredStorage.CommitData(internal.CommitDataOptions{}), + syscall.ENOTSUP, + ) +} + +func (suite *tieredStorageTestSuite) TestReleaseToTriggerEviction() { + defer suite.cleanupTest() + + // Ok this next test is to essentially go through an iteration of LRU, + // 1. Initialize many local only file + //2. Create files that exceed the 80% threshold, max set at 1MB + data := make([]byte, 250*1024) + path1 := "file18" + handle, err := suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path1, Flags: os.O_CREATE, Mode: 0777}, + ) + suite.assert.NoError(err) + suite.assert.Equal(path1, handle.Path) + + _, err = suite.tieredStorage.WriteFile(&internal.WriteFileOptions{Handle: handle, Data: data}) + suite.assert.NoError(err) + suite.assert.True(handle.Dirty()) + + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + path2 := "file19" + handle, err = suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path2, Flags: os.O_CREATE, Mode: 0777}, + ) + suite.assert.NoError(err) + suite.assert.Equal(path2, handle.Path) + + _, err = suite.tieredStorage.WriteFile(&internal.WriteFileOptions{Handle: handle, Data: data}) + suite.assert.NoError(err) + suite.assert.True(handle.Dirty()) + + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + path3 := "file20" + handle, err = suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path3, Flags: os.O_CREATE, Mode: 0777}, + ) + suite.assert.NoError(err) + suite.assert.Equal(path3, handle.Path) + + _, err = suite.tieredStorage.WriteFile(&internal.WriteFileOptions{Handle: handle, Data: data}) + suite.assert.NoError(err) + suite.assert.True(handle.Dirty()) + + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + path4 := "file21" + handle, err = suite.tieredStorage.OpenFile( + internal.OpenFileOptions{Name: path4, Flags: os.O_CREATE, Mode: 0777}, + ) + suite.assert.NoError(err) + suite.assert.Equal(path4, handle.Path) + + _, err = suite.tieredStorage.WriteFile(&internal.WriteFileOptions{Handle: handle, Data: data}) + suite.assert.NoError(err) + suite.assert.True(handle.Dirty()) + + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + // 3. Check if all in the LRU Queue initially + suite.assert.Equal(path4, suite.tieredStorage.policy.head.name) + suite.assert.Equal(path3, suite.tieredStorage.policy.head.next.name) + suite.assert.Equal(path2, suite.tieredStorage.policy.head.next.next.name) + suite.assert.Equal(path1, suite.tieredStorage.policy.tail.name) + + _, exists1 := suite.tieredStorage.policy.nodeMap.Load(path1) + _, exists2 := suite.tieredStorage.policy.nodeMap.Load(path2) + _, exists3 := suite.tieredStorage.policy.nodeMap.Load(path3) + _, exists4 := suite.tieredStorage.policy.nodeMap.Load(path4) + + suite.assert.True(exists1) + suite.assert.True(exists2) + suite.assert.True(exists3) + suite.assert.True(exists4) + + // 4. Wait for one eviction pass to drop the oldest file. + suite.assert.Eventually(func() bool { + _, exists1 := suite.tieredStorage.policy.nodeMap.Load(path1) + return !exists1 + }, 2*capacityPollInterval, 10*time.Millisecond) + + // 4. Some should then be released to the cloud essentially, the ones we wrote data to + //And the local files should be gone (uploaded and cleaned up), not in either map + + // 4a. Check state of NodeMap + _, exists1 = suite.tieredStorage.policy.nodeMap.Load(path1) + _, exists2 = suite.tieredStorage.policy.nodeMap.Load(path2) + _, exists3 = suite.tieredStorage.policy.nodeMap.Load(path3) + _, exists4 = suite.tieredStorage.policy.nodeMap.Load(path4) + + suite.assert.False(exists1) + suite.assert.True(exists2) + suite.assert.True(exists3) + suite.assert.True(exists4) + + //4b. Check state of fileMap + _, exists1 = suite.tieredStorage.fileMap.Load(path1) + _, exists2 = suite.tieredStorage.fileMap.Load(path2) + _, exists3 = suite.tieredStorage.fileMap.Load(path3) + _, exists4 = suite.tieredStorage.fileMap.Load(path4) + + suite.assert.False(exists1) + suite.assert.True(exists2) + suite.assert.True(exists3) + suite.assert.True(exists4) + + // 4c. Only the oldest file should be uploaded and removed locally. + suite.assert.NoFileExists(filepath.Join(suite.cache_path, path1)) + suite.assert.FileExists(filepath.Join(suite.cache_path, path2)) + + // 5. The evicted file must now exist in the cloud. + _, err = suite.tieredStorage.NextComponent().GetAttr( + internal.GetAttrOptions{Name: path1, RetrieveMetadata: true}) + suite.assert.NoError(err) + + //Validate the data matches what we have + //It just checks if the data is preserved + tmpFile, err := os.CreateTemp("", "cloud_verify") + suite.assert.NoError(err) + defer os.Remove(tmpFile.Name()) + defer tmpFile.Close() + + // 2. Copy from the cloud (loopback) to the temporary file + err = suite.loopback.CopyToFile(internal.CopyToFileOptions{ + Name: path1, + Offset: 0, + Count: 0, // 0 usually means the whole file + File: tmpFile, + }) + suite.assert.NoError(err) + + // 3. Read the data back from the temp file and verify + dataFromCloud, err := os.ReadFile(tmpFile.Name()) + suite.assert.NoError(err) + suite.assert.Equal( + data, + dataFromCloud, + "The cloud version should match the modified local version", + ) + +} + +// ok we gonna do file in local, cloud, file doesn't exist +func (suite *tieredStorageTestSuite) TestDeleteFileCloud() { + defer suite.cleanupTest() + // Setup + file := "file22" + + // put file in cloud and write to it + handle, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: file, Mode: 0777}, + ) + suite.assert.NoError(err) + err = suite.tieredStorage.ReleaseFile(internal.ReleaseFileOptions{Handle: handle}) + suite.assert.NoError(err) + + err = suite.tieredStorage.DeleteFile(internal.DeleteFileOptions{Name: file}) + suite.assert.NoError(err) + + // Path should not be in file cache + suite.assert.NoFileExists(filepath.Join(suite.cache_path, file)) + + //file should not exist in cloud + _, err = suite.tieredStorage.NextComponent().GetAttr( + internal.GetAttrOptions{Name: file, RetrieveMetadata: true}) + suite.assert.Error(err) + +} + +func (suite *tieredStorageTestSuite) TestDeleteFileLocal() { + defer suite.cleanupTest() + // Setup + file := "file23" + + //create local file + _, err := suite.tieredStorage.CreateFile( + internal.CreateFileOptions{Name: file, Mode: 0777}, + ) + suite.assert.NoError(err) + + err = suite.tieredStorage.DeleteFile(internal.DeleteFileOptions{Name: file}) + suite.assert.NoError(err) + + // Path should not be in file cache + suite.assert.NoFileExists(filepath.Join(suite.cache_path, file)) + +} + +func (suite *tieredStorageTestSuite) TestDeleteFileNotExists() { + defer suite.cleanupTest() + // Setup + file := "file24" + + err := suite.tieredStorage.DeleteFile(internal.DeleteFileOptions{Name: file}) + suite.assert.Error(err) + suite.assert.EqualValues(syscall.ENOENT, err) +} + +func (suite *tieredStorageTestSuite) TestFlushFile() { + defer suite.cleanupTest() + file := "file25" + handle, _ := suite.tieredStorage.CreateFile(internal.CreateFileOptions{Name: file, Mode: 0777}) + + testData := "test data" + data := []byte(testData) + _, err := suite.tieredStorage.WriteFile( + &internal.WriteFileOptions{Handle: handle, Offset: 0, Data: data}, + ) + suite.assert.NoError(err) + suite.assert.True(handle.Dirty()) + + err = suite.tieredStorage.FlushFile(internal.FlushFileOptions{Handle: handle}) + suite.assert.NoError(err) + + //Verify Data is still on the disk + d, _ := os.ReadFile(filepath.Join(suite.cache_path, file)) + suite.assert.Equal(data, d) + //Check that handle is still dirty + suite.assert.True(handle.Dirty()) + +} + +func TestTieredStorageTestSuite(t *testing.T) { + suite.Run(t, new(tieredStorageTestSuite)) +} diff --git a/component/tiered_storage/tiered_storage_windows.go b/component/tiered_storage/tiered_storage_windows.go new file mode 100644 index 000000000..36cd84de3 --- /dev/null +++ b/component/tiered_storage/tiered_storage_windows.go @@ -0,0 +1,71 @@ +//go:build windows + +/* + Licensed under the MIT License . + + Copyright © 2026 Seagate Technology LLC and/or its Affiliates + + Permission is hereby granted, free of charge, to any person obtaining a copy + of this software and associated documentation files (the "Software"), to deal + in the Software without restriction, including without limitation the rights + to use, copy, modify, merge, publish, distribute, sublicense, and/or sell + copies of the Software, and to permit persons to whom the Software is + furnished to do so, subject to the following conditions: + + The above copyright notice and this permission notice shall be included in all + copies or substantial portions of the Software. + + THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, + FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE + AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER + LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, + OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE + SOFTWARE +*/ + +package tiered_storage + +import ( + "io/fs" + "os" + "syscall" + "time" + + "github.com/Seagate/cloudfuse/common" + "github.com/Seagate/cloudfuse/internal" + "golang.org/x/sys/windows" +) + +func newTieredStorageObjAttr(path string, info fs.FileInfo) *internal.ObjAttr { + stat := info.Sys().(*syscall.Win32FileAttributeData) + attrs := &internal.ObjAttr{ + Path: common.NormalizeObjectName(path), + Name: common.NormalizeObjectName(info.Name()), + Size: info.Size(), + Mode: info.Mode() &^ os.ModePerm, + Mtime: time.Unix(0, stat.LastWriteTime.Nanoseconds()), + Atime: time.Unix(0, stat.LastAccessTime.Nanoseconds()), + Ctime: time.Unix(0, stat.CreationTime.Nanoseconds()), + } + + attrs.Flags.Set(internal.PropFlagModeDefault) + if info.Mode()&os.ModeSymlink != 0 { + attrs.Flags.Set(internal.PropFlagSymlink) + } else if info.IsDir() { + attrs.Flags.Set(internal.PropFlagIsDir) + } + return attrs +} + +func (c *TieredStorage) getAvailableSize() (uint64, error) { + path, err := windows.UTF16PtrFromString(c.tmpPath) + if err != nil { + return 0, err + } + var free, total, available uint64 + if err := windows.GetDiskFreeSpaceEx(path, &free, &total, &available); err != nil { + return 0, err + } + return available, nil +} diff --git a/sample_configs/sampleTieredStorageConfigS3.yaml b/sample_configs/sampleTieredStorageConfigS3.yaml new file mode 100644 index 000000000..1c419f197 --- /dev/null +++ b/sample_configs/sampleTieredStorageConfigS3.yaml @@ -0,0 +1,28 @@ +# Refer to setup/baseConfig.yaml for all configuration parameters. + +config-version: 1.0.0 + +logging: + type: syslog + level: log_warning + +components: + - libfuse + - tiered_storage + - attr_cache + - s3storage + +tiered_storage: + path: /// + max-size-mb: 102400 + +attr_cache: + timeout-sec: 120 + +s3storage: + bucket-name: + key-id: + secret-key: + endpoint: https://s3.us-east-1.lyvecloud.seagate.com + region: us-east-1 + enable-dir-marker: true diff --git a/setup/advancedConfig.yaml b/setup/advancedConfig.yaml index 66aae9390..a5b5f9bee 100644 --- a/setup/advancedConfig.yaml +++ b/setup/advancedConfig.yaml @@ -3,7 +3,7 @@ # 1. All boolean configs (true|false config) (except ignore-open-flags, virtual-directory) are set to 'false' by default. # No need to mention them in your config file unless you are setting them to true. # 2. 'loopbackfs' is purely for testing and shall not be used in production configuration. -# 3. 'stream', 'block-cache', and 'file_cache' can not co-exist and config file shall have only one of them based on your use case. +# 3. 'stream', 'xload', 'block_cache', 'file_cache', and 'tiered_storage' can not co-exist. Choose one based on your use case. # 4. By default log level is set to 'log_warning' level and are redirected to syslog. # Either use 'base' logging or syslog filters to redirect logs to separate file. # To install syslog filter follow below steps: @@ -23,7 +23,8 @@ # 8. If data in your storage account (non-HNS) is created using cloudfuse or AzCopy then there are marker files present # in your container to mark a directory. In such cases you can optimize your listing by setting 'virtual-directory' # flag to false in mount command. -# 9. If you are using 'file_cache' component then make sure you have enough disk space available for cache. +# 9. If using 'file_cache' or 'tiered_storage', make sure the configured path has enough disk space. +# Tiered storage may keep the only copy of a file locally, so never clean its path as a cache. # 10. 'sdk-trace' has been removed and setting log level to log_debug will auto enable these logs. # ----------------------------------------------------------------------------------------------------------------------- @@ -62,6 +63,7 @@ components: - xload - block_cache - file_cache + - tiered_storage - attr_cache - s3storage - azstorage @@ -129,6 +131,16 @@ file_cache: refresh-sec: hard-limit: true|false +# Tiered storage configuration +tiered_storage: + path: + max-size-mb: + high-threshold: <% local storage consumed which triggers eviction. Default - 80> + low-threshold: <% local storage consumed which eviction targets. Must be less than high-threshold. Default - 60> + max-eviction: + parallelism: + poll-interval-sec: + # Attribute cache related configuration attr_cache: timeout-sec: