From c918ea44ffad38b5ccafc867cf3a00c8e36ef1c2 Mon Sep 17 00:00:00 2001 From: mayankpande88 Date: Thu, 8 Oct 2026 12:24:25 +0530 Subject: [PATCH 1/2] fix: detect systemd units whose start event sees /init.scope systemd forks a unit's process inside its own /init.scope and moves it to the unit's cgroup before exec. The registry meant not to cache such a pid as ignored (cg.Id == "/init.scope"), but cgroup parsing skips /init.scope and the root cgroup, so their Id is "" and the check never matched. When the start event was handled before the move, the pid was cached as ignored for 15s, its exec was dropped, and a unit that did nothing else was never detected. Handling proc events on wakeup (#366) made that ordering common: 2 of 6 transient units on a warm agent. Match the empty Id instead, and pin the parsing it relies on in a test. --- cgroup/cgroup_test.go | 15 +++++++++++++++ containers/registry.go | 7 ++++++- 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/cgroup/cgroup_test.go b/cgroup/cgroup_test.go index 4ac82cdf..f2fcc27d 100644 --- a/cgroup/cgroup_test.go +++ b/cgroup/cgroup_test.go @@ -1,6 +1,7 @@ package cgroup import ( + "os" "path" "testing" @@ -236,3 +237,17 @@ func TestContainerByCgroup(t *testing.T) { as.Equal("ba7b10d15d16e10e3de7a2dcd408a3d971169ae303f46cfad4c5453c6326fee2", id) as.Nil(err) } + +// The registry relies on a process in systemd's /init.scope, or in the root +// cgroup, having an empty Id: it is not cached as ignored, so its exec after +// systemd moves it to a unit's cgroup is seen. +func TestInitScopeAndRootHaveEmptyId(t *testing.T) { + for _, content := range []string{"0::/init.scope\n", "0::/\n"} { + f := path.Join(t.TempDir(), "cgroup") + assert.Nil(t, os.WriteFile(f, []byte(content), 0644)) + cg, err := NewFromProcessCgroupFile(f) + assert.Nil(t, err) + assert.Equal(t, "", cg.Id, content) + assert.Equal(t, ContainerTypeStandaloneProcess, cg.ContainerType, content) + } +} diff --git a/containers/registry.go b/containers/registry.go index a873797a..96e85255 100644 --- a/containers/registry.go +++ b/containers/registry.go @@ -640,7 +640,12 @@ func (r *Registry) getOrCreateContainer(pid uint32) *Container { } id := calcId(cg, md) if id == "" { - if cg.Id == "/init.scope" && pid != 1 { + // systemd forks a unit's process inside its own /init.scope and + // moves it to the unit's cgroup before exec. Caching the pid as + // ignored here would drop that exec, and a unit that does nothing + // else would never be detected. /init.scope and the root cgroup + // both parse to an empty Id (cgroup.go skips them). + if cg.Id == "" && pid != 1 { klog.V(5).InfoS("ignoring without persisting", "cg", cg.Id, "pid", pid) } else { klog.V(5).InfoS("ignoring", "cg", cg.Id, "pid", pid) From 279373d06bce9825077d70d4e51b0112ae7723e2 Mon Sep 17 00:00:00 2001 From: mayankpande88 Date: Thu, 8 Oct 2026 13:05:12 +0530 Subject: [PATCH 2/2] test: stop on a failed parse instead of panicking on a nil cgroup --- cgroup/cgroup_test.go | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/cgroup/cgroup_test.go b/cgroup/cgroup_test.go index f2fcc27d..d9145a07 100644 --- a/cgroup/cgroup_test.go +++ b/cgroup/cgroup_test.go @@ -244,9 +244,9 @@ func TestContainerByCgroup(t *testing.T) { func TestInitScopeAndRootHaveEmptyId(t *testing.T) { for _, content := range []string{"0::/init.scope\n", "0::/\n"} { f := path.Join(t.TempDir(), "cgroup") - assert.Nil(t, os.WriteFile(f, []byte(content), 0644)) + require.NoError(t, os.WriteFile(f, []byte(content), 0644)) cg, err := NewFromProcessCgroupFile(f) - assert.Nil(t, err) + require.NoError(t, err) assert.Equal(t, "", cg.Id, content) assert.Equal(t, ContainerTypeStandaloneProcess, cg.ContainerType, content) }