From 7ff313f463bcc73225e8d62764c06607eda8f36d Mon Sep 17 00:00:00 2001 From: Ophestra Date: Wed, 7 Oct 2026 03:11:54 +0900 Subject: container: recover from intermittent EACCES This change works around a race in the vfs, where ACLs are ineffective for a very short window of time while they are being written. This remained undiscovered until now due to the previous integration test suite's complete inability to handle concurrency (the nixos vm test machinery could not even support python multithreading). At the time, it was deemed unnecessary to run this kind of test in the integration vm, as the entire container package up to the syscall wrappers had full test coverage and plenty of tests for race conditions, and a race in vfs was simply unexpected. This workaround adds around 60 milliseconds of latency for a truly inaccessible path. While this is not ideal, it is somewhat cheaper and a lot easier to implement than synchronising container startup all the way up from shim to the priv-side process. Regardless, EACCES should not occur at all during regular use, and a 60-millisecond delay during application development is hardly a problem. Signed-off-by: Ophestra --- container/dispatcher.go | 16 ++++++++++++---- container/errors.go | 27 +++++++++++++++++++++++++++ 2 files changed, 39 insertions(+), 4 deletions(-) (limited to 'container') diff --git a/container/dispatcher.go b/container/dispatcher.go index 4636a469..1bad7527 100644 --- a/container/dispatcher.go +++ b/container/dispatcher.go @@ -228,11 +228,19 @@ func (direct) seccompLoad(rules []std.NativeRule, flags seccomp.ExportFlag) erro func (direct) notify(c chan<- os.Signal, sig ...os.Signal) { signal.Notify(c, sig...) } func (direct) start(c *exec.Cmd) error { return c.Start() } func (direct) signal(c *exec.Cmd, sig os.Signal) error { return c.Process.Signal(sig) } -func (direct) evalSymlinks(path string) (string, error) { return filepath.EvalSymlinks(path) } +func (direct) evalSymlinks(path string) (string, error) { + return retryExp(func() (string, error) { + return filepath.EvalSymlinks(path) + }, syscall.EACCES) +} -func (direct) exit(code int) { os.Exit(code) } -func (direct) getpid() int { return os.Getpid() } -func (direct) stat(name string) (os.FileInfo, error) { return os.Stat(name) } +func (direct) exit(code int) { os.Exit(code) } +func (direct) getpid() int { return os.Getpid() } +func (direct) stat(name string) (os.FileInfo, error) { + return retryExp(func() (os.FileInfo, error) { + return os.Stat(name) + }, syscall.EACCES) +} func (direct) mkdir(name string, perm os.FileMode) error { return os.Mkdir(name, perm) } func (direct) mkdirTemp(dir, pattern string) (string, error) { return os.MkdirTemp(dir, pattern) } func (direct) mkdirAll(path string, perm os.FileMode) error { return os.MkdirAll(path, perm) } diff --git a/container/errors.go b/container/errors.go index 053bdeb7..8975dec4 100644 --- a/container/errors.go +++ b/container/errors.go @@ -4,12 +4,39 @@ import ( "errors" "os" "syscall" + "time" "hakurei.app/check" "hakurei.app/message" "hakurei.app/vfs" ) +// retryExp retries f if the resulting error is equivalent to e according to +// [errors.Is], up to 8 times. +func retryExp[T any](f func() (T, error), e error) (v T, err error) { + const statRetry = 8 // 64 ms + var n int + d := 256 * time.Microsecond + +retry: + v, err = f() + if err != nil { + if !errors.Is(err, e) { + return + } + + n++ + if n == statRetry { + return + } + + time.Sleep(d) + d *= 2 + goto retry + } + return +} + // messageFromError returns a printable error message for a supported concrete type. func messageFromError(err error) (m string, ok bool) { if m, ok = messagePrefixP[MountError]("cannot ", err); ok { -- cgit v1.3.1