* check for nil needle map before compaction sync When CommitCompact runs concurrently, it sets v.nm = nil under dataFileAccessLock. CompactByIndex does not hold that lock, so v.nm.Sync() can hit a nil pointer. Add an early nil check to return an error instead of crashing. Fixes #8591 * guard copyDataBasedOnIndexFile size check against nil needle map The post-compaction size validation at line 538 accesses v.nm.ContentSize() and v.nm.DeletedSize(). If CommitCompact has concurrently set v.nm to nil, this causes a SIGSEGV. Skip the validation when v.nm is nil since the actual data copy uses local needle maps (oldNm/newNm) and is unaffected. Fixes #8591 * use atomic.Bool for compaction flags to prevent concurrent vacuum races The isCompacting and isCommitCompacting flags were plain bools read and written from multiple goroutines without synchronization. This allowed concurrent vacuums on the same volume to pass the guard checks and run simultaneously, leading to the nil pointer crash. Using atomic.Bool with CompareAndSwap ensures only one compaction or commit can run per volume at a time. Fixes #8591 * use go-version-file in CI workflows instead of hardcoded versions Use go-version-file: 'go.mod' so CI automatically picks up the Go version from go.mod, avoiding future version drift. Reordered checkout before setup-go in go.yml and e2e.yml so go.mod is available. Removed the now-unused GO_VERSION env vars. * capture v.nm locally in CompactByIndex to close TOCTOU race A bare nil check on v.nm followed by v.nm.Sync() has a race window where CommitCompact can set v.nm = nil between the two. Snapshot the pointer into a local variable so the nil check and Sync operate on the same reference. * add dynamic timeouts to plugin worker vacuum gRPC calls All vacuum gRPC calls used context.Background() with no deadline, so the plugin scheduler's execution timeout could kill a job while a large volume compact was still in progress. Use volume-size-scaled timeouts matching the topology vacuum approach: 3 min/GB for compact, 1 min/GB for check, commit, and cleanup. Fixes #8591 * Revert "add dynamic timeouts to plugin worker vacuum gRPC calls" This reverts commit 80951934c37416bc4f6c1472a5d3f8d204a637d9. * unify compaction lifecycle into single atomic flag Replace separate isCompacting and isCommitCompacting flags with a single isCompactionInProgress atomic.Bool. This ensures CompactBy*, CommitCompact, Close, and Destroy are mutually exclusive — only one can run at a time per volume. Key changes: - All entry points use CompareAndSwap(false, true) to claim exclusive access. CompactByVolumeData and CompactByIndex now also guard v.nm and v.DataBackend with local captures. - Close() waits for the flag outside dataFileAccessLock to avoid deadlocking with CommitCompact (which holds the flag while waiting for the lock). It claims the flag before acquiring the lock so no new compaction can start. - Destroy() uses CAS instead of a racy Load check, preventing concurrent compaction from racing with volume teardown. - unmountVolumeByCollection no longer deletes from the map; DeleteCollectionFromDiskLocation removes entries only after successful Destroy, preventing orphaned volumes on failure. Fixes #8591
145 lines
5.7 KiB
YAML
145 lines
5.7 KiB
YAML
name: "End to End"
|
|
|
|
on:
|
|
push:
|
|
branches: [ master ]
|
|
pull_request:
|
|
branches: [ master ]
|
|
|
|
concurrency:
|
|
group: ${{ github.head_ref }}/e2e
|
|
cancel-in-progress: true
|
|
|
|
permissions:
|
|
contents: read
|
|
|
|
defaults:
|
|
run:
|
|
working-directory: docker
|
|
|
|
jobs:
|
|
e2e:
|
|
name: FUSE Mount
|
|
runs-on: ubuntu-22.04
|
|
timeout-minutes: 30
|
|
steps:
|
|
- name: Check out code into the Go module directory
|
|
uses: actions/checkout@v6
|
|
|
|
- name: Set up Go
|
|
uses: actions/setup-go@v6
|
|
with:
|
|
go-version-file: 'go.mod'
|
|
|
|
- name: Set up Docker Buildx
|
|
uses: docker/setup-buildx-action@v4
|
|
|
|
- name: Cache Docker layers
|
|
uses: actions/cache@v5
|
|
with:
|
|
path: /tmp/.buildx-cache
|
|
key: ${{ runner.os }}-buildx-e2e-${{ github.sha }}
|
|
restore-keys: |
|
|
${{ runner.os }}-buildx-e2e-
|
|
|
|
- name: Install dependencies
|
|
run: |
|
|
# Use faster mirrors and install with timeout
|
|
sudo rm -f /etc/apt/sources.list.d/azure-cli.list /etc/apt/sources.list.d/microsoft-prod.list
|
|
echo "deb http://azure.archive.ubuntu.com/ubuntu/ $(lsb_release -cs) main restricted universe multiverse" | sudo tee /etc/apt/sources.list
|
|
echo "deb http://azure.archive.ubuntu.com/ubuntu/ $(lsb_release -cs)-updates main restricted universe multiverse" | sudo tee -a /etc/apt/sources.list
|
|
|
|
sudo apt-get update --fix-missing
|
|
sudo DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends fuse
|
|
|
|
# Verify FUSE installation
|
|
echo "FUSE version: $(fusermount --version 2>&1 || echo 'fusermount not found')"
|
|
echo "FUSE device: $(ls -la /dev/fuse 2>&1 || echo '/dev/fuse not found')"
|
|
|
|
- name: Start SeaweedFS
|
|
timeout-minutes: 10
|
|
run: |
|
|
# Enable Docker buildkit for better caching
|
|
export DOCKER_BUILDKIT=1
|
|
export COMPOSE_DOCKER_CLI_BUILD=1
|
|
|
|
# Build with retry logic
|
|
for i in {1..3}; do
|
|
echo "Build attempt $i/3"
|
|
if make build_e2e; then
|
|
echo "Build successful on attempt $i"
|
|
break
|
|
elif [ $i -eq 3 ]; then
|
|
echo "Build failed after 3 attempts"
|
|
exit 1
|
|
else
|
|
echo "Build attempt $i failed, retrying in 30 seconds..."
|
|
sleep 30
|
|
fi
|
|
done
|
|
|
|
# Start services with wait
|
|
docker compose -f ./compose/e2e-mount.yml up --wait
|
|
|
|
- name: Run FIO 4k
|
|
timeout-minutes: 15
|
|
run: |
|
|
echo "Starting FIO at: $(date)"
|
|
# Concurrent r/w
|
|
echo 'Run randrw with size=16M bs=4k'
|
|
docker compose -f ./compose/e2e-mount.yml exec mount timeout -k5 60 fio --name=fiotest --filename=/mnt/seaweedfs/fiotest --size=16M --rw=randrw --bs=4k --direct=1 --numjobs=8 --ioengine=libaio --group_reporting --runtime=30 --time_based=1
|
|
|
|
echo "Verify FIO at: $(date)"
|
|
# Verified write
|
|
echo 'Run randwrite with size=16M bs=4k'
|
|
docker compose -f ./compose/e2e-mount.yml exec mount timeout -k5 60 fio --name=fiotest --filename=/mnt/seaweedfs/fiotest --size=16M --rw=randwrite --bs=4k --direct=1 --numjobs=8 --ioengine=libaio --iodepth=32 --group_reporting --runtime=30 --time_based=1 --do_verify=0 --verify=crc32c --verify_backlog=1
|
|
|
|
- name: Run FIO 128k
|
|
timeout-minutes: 15
|
|
run: |
|
|
echo "Starting FIO at: $(date)"
|
|
# Concurrent r/w
|
|
echo 'Run randrw with size=16M bs=128k'
|
|
docker compose -f ./compose/e2e-mount.yml exec mount timeout -k5 60 fio --name=fiotest --filename=/mnt/seaweedfs/fiotest --size=16M --rw=randrw --bs=128k --direct=1 --numjobs=8 --ioengine=libaio --iodepth=32 --group_reporting --runtime=30 --time_based=1
|
|
|
|
echo "Verify FIO at: $(date)"
|
|
# Verified write
|
|
echo 'Run randwrite with size=16M bs=128k'
|
|
docker compose -f ./compose/e2e-mount.yml exec mount timeout -k5 60 fio --name=fiotest --filename=/mnt/seaweedfs/fiotest --size=16M --rw=randwrite --bs=128k --direct=1 --numjobs=8 --ioengine=libaio --iodepth=32 --group_reporting --runtime=30 --time_based=1 --do_verify=0 --verify=crc32c --verify_backlog=1
|
|
|
|
- name: Run FIO 1MB
|
|
timeout-minutes: 15
|
|
run: |
|
|
echo "Starting FIO at: $(date)"
|
|
# Concurrent r/w
|
|
echo 'Run randrw with size=16M bs=1m'
|
|
docker compose -f ./compose/e2e-mount.yml exec mount timeout -k5 60 fio --name=fiotest --filename=/mnt/seaweedfs/fiotest --size=16M --rw=randrw --bs=1m --direct=1 --numjobs=8 --ioengine=libaio --iodepth=32 --group_reporting --runtime=30 --time_based=1
|
|
|
|
echo "Verify FIO at: $(date)"
|
|
# Verified write
|
|
echo 'Run randwrite with size=16M bs=1m'
|
|
docker compose -f ./compose/e2e-mount.yml exec mount timeout -k5 60 fio --name=fiotest --filename=/mnt/seaweedfs/fiotest --size=16M --rw=randwrite --bs=1m --direct=1 --numjobs=8 --ioengine=libaio --iodepth=32 --group_reporting --runtime=30 --time_based=1 --do_verify=0 --verify=crc32c --verify_backlog=1
|
|
|
|
- name: Save logs
|
|
if: always()
|
|
run: |
|
|
docker compose -f ./compose/e2e-mount.yml logs > output.log
|
|
echo 'Showing last 500 log lines of mount service:'
|
|
docker compose -f ./compose/e2e-mount.yml logs --tail 500 mount
|
|
|
|
- name: Check for data races
|
|
if: always()
|
|
continue-on-error: true # TODO: remove this comment to enable build failure on data races (after all are fixed)
|
|
run: grep -A50 'DATA RACE' output.log && exit 1 || exit 0
|
|
|
|
- name: Archive logs
|
|
if: always()
|
|
uses: actions/upload-artifact@v7
|
|
with:
|
|
name: output-logs
|
|
path: docker/output.log
|
|
|
|
- name: Cleanup
|
|
if: always()
|
|
run: docker compose -f ./compose/e2e-mount.yml down --volumes --remove-orphans --rmi all
|