Skip to content

Commit f2e4da4

Browse files
fix(profiler): reset ThreadContinuousScheduler lock after fork
If os.fork() runs while ThreadContinuousScheduler.lock is held, the child inherits a locked lock with no owner and ensure_running can deadlock. Mirror the Monitor #6159 pattern: register_at_fork after_in_child reset. Fixes #6165
1 parent 8fae8a6 commit f2e4da4

2 files changed

Lines changed: 94 additions & 1 deletion

File tree

‎sentry_sdk/profiler/continuous_profiler.py‎

Lines changed: 30 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -6,6 +6,7 @@
66
import time
77
import uuid
88
import warnings
9+
import weakref
910
from collections import deque
1011
from datetime import datetime, timezone
1112
from typing import TYPE_CHECKING
@@ -443,6 +444,8 @@ class ThreadContinuousScheduler(ContinuousScheduler):
443444
mode: "ContinuousProfilerMode" = "thread"
444445
name = "sentry.profiler.ThreadContinuousScheduler"
445446

447+
thread: "Optional[threading.Thread]"
448+
446449
def __init__(
447450
self,
448451
frequency: int,
@@ -452,8 +455,34 @@ def __init__(
452455
) -> None:
453456
super().__init__(frequency, options, sdk_info, capture_func)
454457

455-
self.thread: "Optional[threading.Thread]" = None
458+
self._reset_thread_state()
459+
460+
# See https://github.com/getsentry/sentry-python/issues/6165.
461+
# If os.fork() runs while another thread holds self.lock, the
462+
# child inherits the lock locked but the holding thread does
463+
# not exist in the child, so the lock can never be released and
464+
# ensure_running deadlocks forever. Reinitialise the lock,
465+
# cached thread/pid, running flag, and buffer in the child so
466+
# it starts clean regardless of inherited state. We bind via a
467+
# WeakMethod so the permanently-registered fork handler does
468+
# not pin this scheduler: register_at_fork has no unregister
469+
# API. POSIX-only; Windows uses spawn.
470+
if hasattr(os, "register_at_fork"):
471+
weak_reset = weakref.WeakMethod(self._reset_thread_state)
472+
473+
def _reset_in_child() -> None:
474+
method = weak_reset()
475+
if method is not None:
476+
method()
477+
478+
os.register_at_fork(after_in_child=_reset_in_child)
479+
480+
def _reset_thread_state(self) -> None:
481+
self.thread = None
456482
self.lock = threading.Lock()
483+
self.pid = None
484+
self.running = False
485+
self.buffer = None
457486

458487
def ensure_running(self) -> None:
459488
self.soft_shutdown = False

‎tests/profiler/test_continuous_profiler.py‎

Lines changed: 64 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -1,3 +1,5 @@
1+
import os
2+
import sys
13
import threading
24
import time
35
from collections import defaultdict
@@ -8,6 +10,7 @@
810
import sentry_sdk
911
from sentry_sdk.consts import VERSION
1012
from sentry_sdk.profiler.continuous_profiler import (
13+
ThreadContinuousScheduler,
1114
get_profiler_id,
1215
is_profile_session_sampled,
1316
setup_continuous_profiler,
@@ -1257,3 +1260,64 @@ def test_continuous_profiler_run_does_not_null_buffer_span_streaming(
12571260
"run() must not set self.buffer = None; "
12581261
"this would destroy buffers created by concurrent ensure_running() calls"
12591262
)
1263+
1264+
1265+
@pytest.mark.skipif(
1266+
sys.platform == "win32"
1267+
or not hasattr(os, "fork")
1268+
or not hasattr(os, "register_at_fork"),
1269+
reason="requires POSIX fork and os.register_at_fork (Python 3.7+)",
1270+
)
1271+
def test_thread_continuous_scheduler_lock_reset_in_child_after_fork():
1272+
"""Regression test for #6165.
1273+
1274+
If os.fork() runs while ThreadContinuousScheduler.lock is held, the
1275+
child inherits the lock locked. The holding thread does not exist in
1276+
the child, so the lock can never be released and ensure_running
1277+
deadlocks forever. The after-fork hook must replace the lock with a
1278+
fresh one in the child, and reset the inherited thread, pid, and
1279+
buffer so the child starts clean.
1280+
"""
1281+
scheduler = ThreadContinuousScheduler(
1282+
frequency=101,
1283+
options={"profile_lifecycle": "manual", "profile_session_sample_rate": 1.0},
1284+
sdk_info=mock_sdk_info,
1285+
capture_func=lambda envelope: None,
1286+
)
1287+
scheduler.reset_buffer()
1288+
scheduler.thread = threading.current_thread()
1289+
scheduler.pid = os.getpid()
1290+
scheduler.running = True
1291+
1292+
original_lock = scheduler.lock
1293+
original_buffer = scheduler.buffer
1294+
assert original_buffer is not None
1295+
1296+
original_lock.acquire()
1297+
pid = os.fork()
1298+
if pid == 0:
1299+
# Child: was the lock object replaced and is the new one not
1300+
# held? Without the fix, lock is `original_lock` inherited
1301+
# locked, so `replaced` is False. blocking=False guarantees
1302+
# the child can't hang on a regression.
1303+
replaced = scheduler.lock is not original_lock
1304+
unheld = scheduler.lock.acquire(blocking=False)
1305+
thread_reset = scheduler.thread is None
1306+
pid_reset = scheduler.pid is None
1307+
buffer_reset = scheduler.buffer is None
1308+
running_reset = scheduler.running is False
1309+
os._exit(
1310+
0
1311+
if replaced
1312+
and unheld
1313+
and thread_reset
1314+
and pid_reset
1315+
and buffer_reset
1316+
and running_reset
1317+
else 1
1318+
)
1319+
1320+
original_lock.release()
1321+
_, status = os.waitpid(pid, 0)
1322+
assert os.WIFEXITED(status) and os.WEXITSTATUS(status) == 0
1323+

0 commit comments

Comments
 (0)