From 133ef49de26ae251c2c86417f9c673ebc7166f76 Mon Sep 17 00:00:00 2001 From: Paul Buetow Date: Thu, 11 Jun 2026 08:40:10 +0300 Subject: Fix unsafe time.Timer.Reset on active timers in runlock and filelock Both filelock.AcquireExclusive and askcli.waitOrAcquireAskLockFD created a single time.Timer and called Reset() on it each retry iteration before it had necessarily fired. Per the Go timer API, Reset()-ing a timer that may still be pending is unsafe: a stale value can already be queued on the channel, causing a spurious early wake-up. Replace the reused-timer + Reset() pattern with a fresh time.After(...) per loop iteration, which guarantees a clean full retry interval (or context cancellation) every time and removes the reuse-while-active hazard. Dropped the now-unused *time.Timer parameter from waitOrAcquireAskLockFD and introduced named retry-interval constants. Add a context-cancel-while-blocked test for the runlock retry loop. Co-Authored-By: Claude Opus 4.8 --- internal/filelock/filelock.go | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) (limited to 'internal/filelock/filelock.go') diff --git a/internal/filelock/filelock.go b/internal/filelock/filelock.go index 192c3ca..bc3ae8a 100644 --- a/internal/filelock/filelock.go +++ b/internal/filelock/filelock.go @@ -21,22 +21,27 @@ func UnlockExclusive(f *os.File) error { return unlockExclusive(f.Fd()) } +// retryInterval is the backoff between successive non-blocking lock attempts. +const retryInterval = 5 * time.Millisecond + // AcquireExclusive spins with TryExclusive until the lock is acquired, ctx is done, or a non-would-block error occurs. func AcquireExclusive(ctx context.Context, f *os.File) (func() error, error) { fd := f.Fd() - retryTimer := time.NewTimer(5 * time.Millisecond) - defer retryTimer.Stop() for { err := tryLockExclusive(fd) if err == nil { return func() error { return unlockExclusive(fd) }, nil } if errors.Is(err, ErrWouldBlock) { - retryTimer.Reset(5 * time.Millisecond) + // Use a fresh timer per iteration via time.After instead of reusing and + // Reset()-ing a single timer. Resetting a timer that may still be pending is + // the documented hazard in the Go timer API: a stale value can already be + // queued on the channel, causing a spurious early wake-up. A new timer each + // loop guarantees a clean, full retryInterval delay (or ctx cancellation). select { case <-ctx.Done(): return nil, ctx.Err() - case <-retryTimer.C: + case <-time.After(retryInterval): } continue } -- cgit v1.2.3