Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
15 commits
Select commit Hold shift + click to select a range
2e93255
[3.15] gh-153364: Make frame, coroutine, and task-waiter chain walks …
miss-islington Oct 5, 2026
fd0dd34
[3.15] gh-155811: Add a seqcount to `gc_stats` to prevent torn reads …
maurycy Oct 5, 2026
65cdbbf
[3.15] gh-151292: `_remote_debugging`: Do not corrupt the binary file…
miss-islington Oct 5, 2026
a194a88
[3.15] gh-158583: Fix uninitialized memory read in bytes.fromhex() (G…
miss-islington Oct 5, 2026
8f7c713
[3.15] gh-154194: Degrade frames in Tachyon instead of failing the sa…
pablogsal Oct 5, 2026
cfe0cab
[3.15] Add MSan to CI (GH-158625) (#158832)
pablogsal Oct 5, 2026
7e8cfd5
[3.15] gh-156810: Write the profiler's collapsed-stack export as UTF-…
miss-islington Oct 5, 2026
d70f842
[3.15] gh-158552: Wait for Windows threads to suspend before blocking…
miss-islington Oct 5, 2026
43e640c
[3.15] gh-152721: Fix quadratic RLE replay time in the profiling bina…
pablogsal Oct 5, 2026
e27894e
[3.15] gh-156545: Fix flamegraph export RecursionError on deeply recu…
pablogsal Oct 5, 2026
c2cb278
[3.15] gh-158540: Add the profiled script's directory to sys.path (GH…
miss-islington Oct 5, 2026
78519d4
[3.15] gh-153838: Skip non-regular source files in the heatmap export…
pablogsal Oct 5, 2026
0e8dc09
[3.15] gh-158539: Fix exception mode missing handlers in generators/c…
pablogsal Oct 5, 2026
9fef40a
[3.15] gh-158522: Fix truncated stack for a task whose coroutine recu…
pablogsal Oct 5, 2026
82b7d1f
gh-156545, gh-158539: Fix deep flamegraph export on small C stacks an…
pablogsal Oct 5, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 3 additions & 0 deletions .github/workflows/build.yml
Original file line number Diff line number Diff line change
Expand Up @@ -605,6 +605,9 @@ jobs:
- check-name: Undefined behavior
sanitizer: UBSan
free-threading: false
- check-name: Memory
sanitizer: MSan
free-threading: false
uses: ./.github/workflows/reusable-san.yml
with:
sanitizer: ${{ matrix.sanitizer }}
Expand Down
22 changes: 20 additions & 2 deletions .github/workflows/reusable-san.yml
Original file line number Diff line number Diff line change
Expand Up @@ -60,7 +60,7 @@ jobs:
|| ''
}}
- name: UBSan option setup
if: inputs.sanitizer != 'TSan'
if: inputs.sanitizer == 'UBSan'
run: >-
echo
"UBSAN_OPTIONS=${SAN_LOG_OPTION}
Expand All @@ -69,6 +69,20 @@ jobs:
>> "$GITHUB_ENV"
env:
SAN_LOG_OPTION: log_path=${{ github.workspace }}/san_log
- name: MSan option setup
if: inputs.sanitizer == 'MSan'
run: |
echo "MSAN_OPTIONS=${SAN_LOG_OPTION} allocator_may_return_null=1 handle_segv=0" >> "$GITHUB_ENV"
# MSan reports false positives for memory initialized by libraries
# that are not built with MSan, so disable modules that use them.
# _remote_debugging links to libzstd directly, but we unpoision the memory.
{
echo '*disabled*'
echo '_bz2 _ctypes _curses _curses_panel _dbm _decimal _gdbm _hashlib'
echo '_lzma _sqlite3 _ssl _tkinter _uuid _zstd readline zlib'
} > Modules/Setup.local
env:
SAN_LOG_OPTION: log_path=${{ github.workspace }}/san_log
- name: Add ccache to PATH
run: |
echo "PATH=/usr/lib/ccache:$PATH" >> "$GITHUB_ENV"
Expand All @@ -93,6 +107,8 @@ jobs:
# gh-157958: -O2 instead of the pydebug default -Og to avoid a clang 21
# compile-time blowup on some interpreter files.
# (https://github.com/llvm/llvm-project/issues/179695)
# MSan uses --with-assertions instead of --with-pydebug because its
# hooks on the Python memory allocators hide uninitialized reads.
- name: Configure CPython
run: >-
./configure
Expand All @@ -101,9 +117,11 @@ jobs:
${{
inputs.sanitizer == 'TSan'
&& '--with-thread-sanitizer'
|| inputs.sanitizer == 'MSan'
&& '--with-memory-sanitizer'
|| '--with-undefined-behavior-sanitizer --with-strict-overflow'
}}
--with-pydebug
${{ inputs.sanitizer == 'MSan' && '--with-assertions' || '--with-pydebug' }}
${{ inputs.sanitizer == 'TSan' && '--with-openssl="$OPENSSL_DIR" --with-openssl-rpath=auto' || '' }}
${{ inputs.free-threading && '--disable-gil' || '' }}
- name: Build CPython
Expand Down
4 changes: 4 additions & 0 deletions Doc/using/configure.rst
Original file line number Diff line number Diff line change
Expand Up @@ -1015,6 +1015,10 @@ Debug options

Enable MemorySanitizer allocation error detector, ``msan`` (default is no).

MSan reports false positives for memory initialized by libraries that are
not built with MSan, so either build all dependencies with MSan or disable
the extension modules that use them in :file:`Modules/Setup.local`.

.. versionadded:: 3.6

.. option:: --with-undefined-behavior-sanitizer
Expand Down
1 change: 1 addition & 0 deletions Include/internal/pycore_interp_structs.h
Original file line number Diff line number Diff line change
Expand Up @@ -219,6 +219,7 @@ struct gc_old_stats_buffer {
struct gc_stats {
struct gc_young_stats_buffer young;
struct gc_old_stats_buffer old[2];
uint32_t update_seq;
};

struct _gc_runtime_state {
Expand Down
4 changes: 4 additions & 0 deletions Include/pyport.h
Original file line number Diff line number Diff line change
Expand Up @@ -554,6 +554,7 @@ extern "C" {
# define _Py_MEMORY_SANITIZER
# define _Py_NO_SANITIZE_MEMORY __attribute__((no_sanitize_memory))
# define _Py_MSAN_UNPOISON(PTR, SIZE) (__msan_unpoison(PTR, SIZE))
# define _Py_MSAN_UNPOISON_STRING(STR) (__msan_unpoison_string(STR))
# endif
# endif
# if __has_feature(address_sanitizer)
Expand Down Expand Up @@ -595,6 +596,9 @@ extern "C" {
#ifndef _Py_MSAN_UNPOISON
# define _Py_MSAN_UNPOISON(PTR, SIZE)
#endif
#ifndef _Py_MSAN_UNPOISON_STRING
# define _Py_MSAN_UNPOISON_STRING(STR)
#endif

/* AIX has __bool__ redefined in it's system header file. */
#if defined(_AIX) && defined(__bool__)
Expand Down
10 changes: 6 additions & 4 deletions Lib/asyncio/tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,10 @@ def __init__(
# ─── indexing helpers ───────────────────────────────────────────
def _format_stack_entry(elem: str|FrameInfo) -> str:
if not isinstance(elem, str):
if elem.location is None:
if elem.filename in ("", "~"):
return f"{elem.funcname}"
return f"{elem.funcname} {elem.filename}"
if elem.location.lineno == 0 and elem.filename == "":
return f"{elem.funcname}"
else:
Expand Down Expand Up @@ -190,8 +194,7 @@ def build_task_table(result):
# Build coroutine stack string
frames = [frame for coro in task_info.coroutine_stack
for frame in coro.call_stack]
coro_stack = " -> ".join(_format_stack_entry(x).split(" ")[0]
for x in frames)
coro_stack = " -> ".join(x.funcname for x in frames)

# Handle tasks with no awaiters
if not task_info.awaited_by:
Expand All @@ -202,8 +205,7 @@ def build_task_table(result):
# Handle tasks with awaiters
for coro_info in task_info.awaited_by:
parent_id = coro_info.task_name
awaiter_frames = [_format_stack_entry(x).split(" ")[0]
for x in coro_info.call_stack]
awaiter_frames = [x.funcname for x in coro_info.call_stack]
awaiter_chain = " -> ".join(awaiter_frames)
awaiter_name = id2name.get(parent_id, "Unknown")
parent_id_str = (hex(parent_id) if isinstance(parent_id, int)
Expand Down
5 changes: 5 additions & 0 deletions Lib/profiling/sampling/_sync_coordinator.py
Original file line number Diff line number Diff line change
Expand Up @@ -168,6 +168,11 @@ def _execute_script(script_path: str, script_args: List[str], cwd: str) -> None:
if not os.path.isfile(script_path):
raise TargetError(f"Script not found: {script_path}")

script_dir = os.path.dirname(os.path.realpath(script_path))
if script_dir in sys.path:
sys.path.remove(script_dir)
sys.path.insert(0, script_dir)

# Replace sys.argv to match original script call
sys.argv = [script_path] + script_args

Expand Down
22 changes: 15 additions & 7 deletions Lib/profiling/sampling/binary_collector.py
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
"""Thin Python wrapper around C binary writer for profiling data."""

import sys
import time

import _remote_debugging
Expand Down Expand Up @@ -81,6 +82,7 @@ def __init__(self, filename, sample_interval_usec, *, skip_idle=False,
self.filename = filename
self.sample_interval_usec = sample_interval_usec
self.skip_idle = skip_idle
self.running = True

compression_type = _resolve_compression(compression)
start_time_us = int(time.monotonic() * 1_000_000)
Expand All @@ -102,9 +104,19 @@ def collect(self, stack_frames, timestamp_us=None):
timestamp_us: Optional timestamp in microseconds. If not provided,
uses time.monotonic() to generate one.
"""
if not self.running:
return
if timestamp_us is None:
timestamp_us = int(time.monotonic() * 1_000_000)
self._writer.write_sample(stack_frames, timestamp_us)
try:
self._writer.write_sample(stack_frames, timestamp_us)
except OverflowError as e:
if not self._writer.limit_reached:
raise
self.running = False
print(f"Warning: {e}; stopping early and keeping the data "
"collected so far.",
file=sys.stderr)

def collect_failed_sample(self):
"""Record a failed sample attempt (no-op for binary format)."""
Expand Down Expand Up @@ -143,9 +155,5 @@ def __enter__(self):
return self

def __exit__(self, exc_type, exc_val, exc_tb):
"""Context manager exit - finalize unless there was an error."""
if exc_type is None:
self._writer.finalize()
else:
self._writer.close()
return False
"""Finalize if the writer can still produce a valid file."""
return self._writer.__exit__(exc_type, exc_val, exc_tb)
14 changes: 7 additions & 7 deletions Lib/profiling/sampling/heatmap_collector.py
Original file line number Diff line number Diff line change
Expand Up @@ -785,14 +785,14 @@ def _generate_file_html(self, output_path: Path, filename: str,
line_counts: Dict[int, int], self_counts: Dict[int, int],
file_stat: FileStats):
"""Generate HTML for a single source file with heatmap coloring."""
# Read source file
source_lines = [f"# Source file not available: {filename}"]
try:
source_lines = Path(filename).read_text(encoding='utf-8', errors='replace').splitlines()
except (IOError, OSError) as e:
if not (filename.startswith('<') or filename.startswith('[') or
filename in ('~', '...', '.') or len(filename) < 2):
print(f"Warning: Could not read source file {filename}: {e}")
source_lines = [f"# Source file not available: {filename}"]
path = Path(filename)
if path.is_file():
source_lines = path.read_text(
encoding='utf-8', errors='replace').splitlines()
except (IOError, OSError):
pass

# Generate HTML for each line
max_samples = max(line_counts.values()) if line_counts else 1
Expand Down
67 changes: 42 additions & 25 deletions Lib/profiling/sampling/stack_collector.py
Original file line number Diff line number Diff line change
Expand Up @@ -60,13 +60,18 @@ def export(self, filename):

lines.sort(key=lambda x: (-x[1], x[0]))

with open(filename, "w") as f:
with open(filename, "w",
encoding="utf-8", errors="surrogatepass") as f:
for stack, count in lines:
f.write(f"{stack} {count}\n")
print(f"Collapsed stack output written to {filename}")
return True


# Allow for tree conversion and the dict/list frames in the Python JSON encoder.
_FLAMEGRAPH_RECURSION_MARGIN = 6000


class FlamegraphCollector(StackTraceCollector):
def __init__(self, *args, **kwargs):
super().__init__(*args, **kwargs)
Expand Down Expand Up @@ -165,34 +170,41 @@ def set_mode(self, mode):
self.stats["mode"] = mode

def export(self, filename):
flamegraph_data = self._convert_to_flamegraph_format()

# Debug output with string table statistics
num_functions = len(flamegraph_data.get("children", []))
total_time = flamegraph_data.get("value", 0)
string_count = len(self._string_table)
s1 = "" if num_functions == 1 else "s"
s2 = "" if total_time == 1 else "s"
s3 = "" if string_count == 1 else "s"
print(
f"Flamegraph data: {num_functions} root function{s1}, "
f"{total_time} total sample{s2}, "
f"{string_count} unique string{s3}"
)

if num_functions == 0:
# Converting the call tree recurses to the sampled stack depth.
old_limit = sys.getrecursionlimit()
sys.setrecursionlimit(old_limit + _FLAMEGRAPH_RECURSION_MARGIN)
try:
flamegraph_data = self._convert_to_flamegraph_format()

# Debug output with string table statistics
num_functions = len(flamegraph_data.get("children", []))
total_time = flamegraph_data.get("value", 0)
string_count = len(self._string_table)
s1 = "" if num_functions == 1 else "s"
s2 = "" if total_time == 1 else "s"
s3 = "" if string_count == 1 else "s"
print(
"Warning: No functions found in profiling data. Check if sampling captured any data."
f"Flamegraph data: {num_functions} root function{s1}, "
f"{total_time} total sample{s2}, "
f"{string_count} unique string{s3}"
)
return False

html_content = self._create_flamegraph_html(flamegraph_data)
if num_functions == 0:
print(
"Warning: No functions found in profiling data. "
"Check if sampling captured any data."
)
return False

with open(filename, "w", encoding="utf-8") as f:
f.write(html_content)
html_content = self._create_flamegraph_html(flamegraph_data)

print(f"Flamegraph saved to: {filename}")
return True
with open(filename, "w", encoding="utf-8") as f:
f.write(html_content)

print(f"Flamegraph saved to: {filename}")
return True
finally:
sys.setrecursionlimit(old_limit)

@staticmethod
@functools.lru_cache(maxsize=None)
Expand Down Expand Up @@ -486,7 +498,12 @@ def _get_source_lines(self, func):
return None

def _create_flamegraph_html(self, data):
data_json = json.dumps(data)
try:
data_json = json.dumps(data)
except RecursionError:
# The C encoder can exhaust the C stack independently of the
# Python recursion limit. iterencode() uses the Python encoder.
data_json = "".join(json.JSONEncoder().iterencode(data))

template_dir = importlib.resources.files(__package__)
vendor_dir = template_dir / "_vendor"
Expand Down
Loading
Loading