From 51bad1f3fc6da3f3efd3b6d51e84c94534bb29fa Mon Sep 17 00:00:00 2001 From: Lindsey Gray Date: Fri, 25 Sep 2026 16:50:50 -0500 Subject: [PATCH 1/5] fix: ak.from_iter of np.clongdouble, 0-d arrays, and np.void scalars builder_fromiter had no branch for complex numpy scalars other than complex128, so np.clongdouble reached the tolist() branch; its tolist() returns itself and the builder recursed until the stack overflowed. - np.complexfloating scalars are built as complex128. - 0-d ndarrays are built from obj[()], keeping datetime64/timedelta64 units and structured field names that tolist() dropped. - structured np.void scalars are built as records; unstructured ones as bytes. - an object whose tolist() returns its own type raises TypeError instead of recursing. --- awkward-cpp/src/python/content.cpp | 34 ++++++++- tests/test_4392_from_iter_numpy_scalars.py | 81 ++++++++++++++++++++++ 2 files changed, 114 insertions(+), 1 deletion(-) create mode 100644 tests/test_4392_from_iter_numpy_scalars.py diff --git a/awkward-cpp/src/python/content.cpp b/awkward-cpp/src/python/content.cpp index 102231d97a..6e81eef102 100644 --- a/awkward-cpp/src/python/content.cpp +++ b/awkward-cpp/src/python/content.cpp @@ -122,6 +122,21 @@ builder_fromiter(ak::ArrayBuilder& self, const py::handle& obj) { } self.endtuple(); } + else if (py::isinstance(obj, py::module::import("numpy").attr("void"))) { + py::object names = obj.attr("dtype").attr("names"); + if (names.is_none()) { + self.bytestring(obj.attr("tolist")().cast()); + } + else { + self.beginrecord(); + for (auto name : names) { + std::string key = name.cast(); + self.field_check(key.c_str()); + builder_fromiter(self, obj[name]); + } + self.endrecord(); + } + } else if (py::isinstance(obj)) { py::dict dict = obj.cast(); self.beginrecord(); @@ -172,11 +187,28 @@ builder_fromiter(ak::ArrayBuilder& self, const py::handle& obj) { else if (py::isinstance(obj, py::module::import("numpy").attr("floating"))) { self.real(obj.cast()); } + else if (py::isinstance(obj, py::module::import("numpy").attr("complexfloating"))) { + self.complex(obj.cast>()); + } + // tolist() of a 0-d array drops the datetime/timedelta unit and structured field names. + else if (py::type::of(obj).is(py::module::import("numpy").attr("ndarray")) + && obj.attr("ndim").cast() == 0) { + builder_fromiter(self, obj[py::tuple()]); + } else if (py::hasattr(obj, "to_list")) { builder_fromiter(self, obj.attr("to_list")()); } else if (py::hasattr(obj, "tolist")) { - builder_fromiter(self, obj.attr("tolist")()); + py::object list = obj.attr("tolist")(); + if (py::type::of(list).is(py::type::of(obj))) { + throw py::type_error( + std::string("cannot convert ") + + obj.attr("__repr__")().cast() + std::string(" (type ") + + obj.attr("__class__").attr("__name__").cast() + + std::string(") to an array element: its tolist() returns the same type") + + FILENAME(__LINE__)); + } + builder_fromiter(self, list); } else { diff --git a/tests/test_4392_from_iter_numpy_scalars.py b/tests/test_4392_from_iter_numpy_scalars.py new file mode 100644 index 0000000000..026a0076cb --- /dev/null +++ b/tests/test_4392_from_iter_numpy_scalars.py @@ -0,0 +1,81 @@ +# BSD 3-Clause License; see https://github.com/scikit-hep/awkward/blob/main/LICENSE + +from __future__ import annotations + +import subprocess +import sys + +import numpy as np +import pytest + +import awkward as ak + +STRUCTURED = np.array([(1, 2.5)], dtype=[("a", "i4"), ("b", "f8")]) + + +@pytest.mark.parametrize("zero_d", [False, True]) +@pytest.mark.parametrize( + ("value", "expected_type"), + [ + (np.longdouble(1.5), "float64"), + (np.complex64(1.5 + 2.5j), "complex128"), + (np.datetime64("2020-01-02T03:04:05.000000006", "ns"), "datetime64[ns]"), + (np.datetime64("NaT", "s"), "datetime64[s]"), + (np.timedelta64(5, "s"), "timedelta64[s]"), + (np.timedelta64(5, "ns"), "timedelta64[ns]"), + (STRUCTURED[0], "{a: int64, b: float64}"), + ], +) +def test_matches_from_numpy(value, expected_type, zero_d): + result = ak.from_iter([np.array(value) if zero_d else value]) + assert str(result.type) == f"1 * {expected_type}" + assert result.to_list() == ak.from_numpy(np.asarray(value).reshape(1)).to_list() + + +@pytest.mark.parametrize("zero_d", [False, True]) +def test_unstructured_void(zero_d): + value = np.void(b"ab") + result = ak.from_iter([np.array(value) if zero_d else value]) + assert str(result.type) == "1 * bytes" + assert result.to_list() == [b"ab"] + + +def run(code): + # A stack overflow kills the interpreter, so it must fail this test rather than pytest. + return subprocess.run( + [sys.executable, "-c", f"import numpy as np, awkward as ak\n{code}"], + capture_output=True, + text=True, + check=False, + ) + + +@pytest.mark.parametrize( + ("expr", "expected"), + [ + ("[np.clongdouble(1.5 + 2.5j)]", "1 * complex128 [(1.5+2.5j)]"), + ("[np.array(np.clongdouble(1.5 + 2.5j))]", "1 * complex128 [(1.5+2.5j)]"), + ( + "np.full((2, 1), 1.5 + 2.5j, dtype=np.clongdouble)", + "2 * var * complex128 [[(1.5+2.5j)], [(1.5+2.5j)]]", + ), + ], +) +def test_clongdouble(expr, expected): + out = run(f"a = ak.from_iter({expr})\nprint(a.type, a.to_list())") + assert out.returncode == 0, out.stderr + assert out.stdout.strip() == expected + + +def test_tolist_returning_same_type(): + out = run( + "class S:\n" + " def tolist(self):\n" + " return S()\n" + "try:\n" + " ak.from_iter([S()])\n" + "except TypeError as err:\n" + " print('tolist() returns the same type' in str(err))" + ) + assert out.returncode == 0, out.stderr + assert out.stdout.strip() == "True" From 65b1cfae4de6fffbd23d0bab441adff76ffc2766 Mon Sep 17 00:00:00 2001 From: Lindsey Gray Date: Sat, 26 Sep 2026 01:55:18 -0500 Subject: [PATCH 2/5] fix: bound ak.from_iter recursion by the Python recursion limit builder_fromiter recursed on nested containers and on tolist()/to_list() results with nothing to stop it, so a tolist() 2-cycle, a to_list() that returns its own type, a 0-d object array containing itself, or a list, tuple or dict nested 200000 deep overflowed the stack. - A thread-local depth counter raises RecursionError once the nesting reaches sys.getrecursionlimit(). Py_EnterRecursiveCall is not enough: on Python 3.13 linux/amd64 its C-level limit lets tuples and dicts nested 200000 deep still overflow the stack. - The tolist() same-type TypeError is removed; the bound covers it. - The np.void branch moves after the dict branch and skips lists, so from_iter of records is not slowed by the numpy lookup. - Subprocess tests are skipped on emscripten, which has no subprocess. --- awkward-cpp/src/python/content.cpp | 64 ++++++++++++-------- tests/test_4392_from_iter_numpy_scalars.py | 69 ++++++++++++++++++---- 2 files changed, 98 insertions(+), 35 deletions(-) diff --git a/awkward-cpp/src/python/content.cpp b/awkward-cpp/src/python/content.cpp index 6e81eef102..3e53a9953f 100644 --- a/awkward-cpp/src/python/content.cpp +++ b/awkward-cpp/src/python/content.cpp @@ -90,8 +90,29 @@ builder_timedelta(ak::ArrayBuilder& self, const py::handle& obj) { } } +// Nesting, and tolist()/to_list() that never bottom out, recurse until the stack overflows. +// Bounded by the Python recursion limit, not Py_EnterRecursiveCall, whose C-level limit +// on Python 3.13 admits more of these frames than the stack holds. +static thread_local int builder_fromiter_depth = 0; + +struct builder_fromiter_depth_guard { + builder_fromiter_depth_guard() { + if (builder_fromiter_depth >= Py_GetRecursionLimit()) { + PyErr_SetString(PyExc_RecursionError, ( + std::string("maximum recursion depth exceeded in ak.from_iter") + + FILENAME(__LINE__)).c_str()); + throw py::error_already_set(); + } + builder_fromiter_depth++; + } + ~builder_fromiter_depth_guard() { + builder_fromiter_depth--; + } +}; + void builder_fromiter(ak::ArrayBuilder& self, const py::handle& obj) { + builder_fromiter_depth_guard guard; if (obj.is(py::none())) { self.null(); } @@ -122,21 +143,6 @@ builder_fromiter(ak::ArrayBuilder& self, const py::handle& obj) { } self.endtuple(); } - else if (py::isinstance(obj, py::module::import("numpy").attr("void"))) { - py::object names = obj.attr("dtype").attr("names"); - if (names.is_none()) { - self.bytestring(obj.attr("tolist")().cast()); - } - else { - self.beginrecord(); - for (auto name : names) { - std::string key = name.cast(); - self.field_check(key.c_str()); - builder_fromiter(self, obj[name]); - } - self.endrecord(); - } - } else if (py::isinstance(obj)) { py::dict dict = obj.cast(); self.beginrecord(); @@ -152,6 +158,23 @@ builder_fromiter(ak::ArrayBuilder& self, const py::handle& obj) { } self.endrecord(); } + // np.void is iterable, so it goes before the iterable branch; lists skip the numpy lookup. + else if (!PyList_Check(obj.ptr()) + && py::isinstance(obj, py::module::import("numpy").attr("void"))) { + py::object names = obj.attr("dtype").attr("names"); + if (names.is_none()) { + self.bytestring(obj.attr("tolist")().cast()); + } + else { + self.beginrecord(); + for (auto name : names) { + std::string key = name.cast(); + self.field_check(key.c_str()); + builder_fromiter(self, obj[name]); + } + self.endrecord(); + } + } else if (py::isinstance(obj)) { py::iterable seq = obj.cast(); self.beginlist(); @@ -199,16 +222,7 @@ builder_fromiter(ak::ArrayBuilder& self, const py::handle& obj) { builder_fromiter(self, obj.attr("to_list")()); } else if (py::hasattr(obj, "tolist")) { - py::object list = obj.attr("tolist")(); - if (py::type::of(list).is(py::type::of(obj))) { - throw py::type_error( - std::string("cannot convert ") - + obj.attr("__repr__")().cast() + std::string(" (type ") - + obj.attr("__class__").attr("__name__").cast() - + std::string(") to an array element: its tolist() returns the same type") - + FILENAME(__LINE__)); - } - builder_fromiter(self, list); + builder_fromiter(self, obj.attr("tolist")()); } else { diff --git a/tests/test_4392_from_iter_numpy_scalars.py b/tests/test_4392_from_iter_numpy_scalars.py index 026a0076cb..37c30d05a0 100644 --- a/tests/test_4392_from_iter_numpy_scalars.py +++ b/tests/test_4392_from_iter_numpy_scalars.py @@ -40,8 +40,13 @@ def test_unstructured_void(zero_d): assert result.to_list() == [b"ab"] +# A stack overflow kills the interpreter, so it must fail the test rather than pytest. +needs_subprocess = pytest.mark.skipif( + sys.platform.startswith("emscripten"), reason="no subprocess on emscripten" +) + + def run(code): - # A stack overflow kills the interpreter, so it must fail this test rather than pytest. return subprocess.run( [sys.executable, "-c", f"import numpy as np, awkward as ak\n{code}"], capture_output=True, @@ -50,6 +55,7 @@ def run(code): ) +@needs_subprocess @pytest.mark.parametrize( ("expr", "expected"), [ @@ -67,15 +73,58 @@ def test_clongdouble(expr, expected): assert out.stdout.strip() == expected -def test_tolist_returning_same_type(): +class Countdown: + def __init__(self, n): + self.n = n + + def tolist(self): + return Countdown(self.n - 1) if self.n else self.n + + +CYCLES = """ +class S: + def tolist(self): + return S() +class A: + def tolist(self): + return B() +class B: + def tolist(self): + return A() +class T: + def to_list(self): + return T() +""" + + +@needs_subprocess +@pytest.mark.parametrize( + "make_x", + [ + "x = S()", + "x = A()", + "x = T()", + "x = np.empty((), dtype=object)\nx[()] = x", + "x = 1\nfor _ in range(200_000):\n x = [x]", + "x = 1\nfor _ in range(200_000):\n x = (x,)", + "x = 1\nfor _ in range(200_000):\n x = {'a': x}", + ], +) +def test_unbounded_recursion(make_x): out = run( - "class S:\n" - " def tolist(self):\n" - " return S()\n" + f"{CYCLES}{make_x}\n" "try:\n" - " ak.from_iter([S()])\n" - "except TypeError as err:\n" - " print('tolist() returns the same type' in str(err))" + " ak.from_iter([x])\n" + "except RecursionError:\n" + " print('RecursionError')" ) - assert out.returncode == 0, out.stderr - assert out.stdout.strip() == "True" + assert out.returncode == 0, out.stderr[-2000:] + assert out.stdout.strip() == "RecursionError" + + +def test_bounded_recursion(): + assert ak.from_iter([Countdown(50)]).to_list() == [0] + x = 1 + for _ in range(200): + x = [x] + assert str(ak.from_iter(x).type) == "1 * " + "var * " * 199 + "int64" From 7b22b2d104314b45f81fd076fae7314b7f80e9c3 Mon Sep 17 00:00:00 2001 From: Lindsey Gray Date: Sat, 26 Sep 2026 02:23:51 -0500 Subject: [PATCH 3/5] fix: cap ak.from_iter depth below the stack's capacity The depth bound was sys.getrecursionlimit(), which users raise. Past about 7600 levels (tuples, linux/amd64) or 4800 (nested 0-d object arrays) the builder overflows the stack first, so setrecursionlimit(10_000) or higher brought the segfault back. The bound is now min(sys.getrecursionlimit(), 1000). 1000 is CPython's default limit, so default behavior is unchanged and no platform recurses deeper than it already did at the default; Windows gets roughly a third of the linux C-stack budget (Py_C_RECURSION_LIMIT 3000 vs 10000). A list nested deeper than 1000 now raises RecursionError even with a raised limit. --- awkward-cpp/src/python/content.cpp | 8 +++++--- tests/test_4392_from_iter_numpy_scalars.py | 2 ++ 2 files changed, 7 insertions(+), 3 deletions(-) diff --git a/awkward-cpp/src/python/content.cpp b/awkward-cpp/src/python/content.cpp index 3e53a9953f..483afafa7a 100644 --- a/awkward-cpp/src/python/content.cpp +++ b/awkward-cpp/src/python/content.cpp @@ -91,13 +91,15 @@ builder_timedelta(ak::ArrayBuilder& self, const py::handle& obj) { } // Nesting, and tolist()/to_list() that never bottom out, recurse until the stack overflows. -// Bounded by the Python recursion limit, not Py_EnterRecursiveCall, whose C-level limit -// on Python 3.13 admits more of these frames than the stack holds. +// Neither Py_EnterRecursiveCall's C-level limit nor a raised sys.getrecursionlimit() stays +// below the stack's capacity for these frames, so the depth is also capped at CPython's +// default recursion limit. static thread_local int builder_fromiter_depth = 0; +static const int builder_fromiter_max_depth = 1000; struct builder_fromiter_depth_guard { builder_fromiter_depth_guard() { - if (builder_fromiter_depth >= Py_GetRecursionLimit()) { + if (builder_fromiter_depth >= std::min(Py_GetRecursionLimit(), builder_fromiter_max_depth)) { PyErr_SetString(PyExc_RecursionError, ( std::string("maximum recursion depth exceeded in ak.from_iter") + FILENAME(__LINE__)).c_str()); diff --git a/tests/test_4392_from_iter_numpy_scalars.py b/tests/test_4392_from_iter_numpy_scalars.py index 37c30d05a0..e17f4cce2e 100644 --- a/tests/test_4392_from_iter_numpy_scalars.py +++ b/tests/test_4392_from_iter_numpy_scalars.py @@ -108,6 +108,8 @@ def to_list(self): "x = 1\nfor _ in range(200_000):\n x = [x]", "x = 1\nfor _ in range(200_000):\n x = (x,)", "x = 1\nfor _ in range(200_000):\n x = {'a': x}", + "import sys\nsys.setrecursionlimit(100_000)\nx = S()", + "import sys\nsys.setrecursionlimit(100_000)\nx = 1\nfor _ in range(200_000):\n x = (x,)", ], ) def test_unbounded_recursion(make_x): From ac7fac7d2ed89786949d224f5d88e1a4f526e15b Mon Sep 17 00:00:00 2001 From: Lindsey Gray Date: Sat, 26 Sep 2026 08:43:19 -0500 Subject: [PATCH 4/5] fix: bound the depth of argument reprs in error messages --- src/awkward/_errors.py | 35 +++++++++++++++++++++- tests/test_4392_from_iter_numpy_scalars.py | 18 +++++++++++ 2 files changed, 52 insertions(+), 1 deletion(-) diff --git a/src/awkward/_errors.py b/src/awkward/_errors.py index 9e840e6bd7..13de095418 100644 --- a/src/awkward/_errors.py +++ b/src/awkward/_errors.py @@ -1,10 +1,13 @@ # BSD 3-Clause License; see https://github.com/scikit-hep/awkward/blob/main/LICENSE +import reprlib +import sys import threading import warnings from collections.abc import Callable, Collection, Iterable, Mapping from functools import wraps +from itertools import islice from weakref import ref as weak_ref import numpy @@ -155,7 +158,10 @@ def format_argument(self, width, value): valuestr = valuestr[: width - 3] + "..." elif isinstance(value, (Collection, Mapping)) and len(value) < 10000: - valuestr = repr(value) + try: + valuestr = _ArgumentRepr(width).repr(value) + except Exception as err: + valuestr = f"repr-raised-{type(err).__name__}" if len(valuestr) > width: valuestr = valuestr[: width - 3] + "..." @@ -172,6 +178,33 @@ def note(self) -> str: raise NotImplementedError +class _ArgumentRepr(reprlib.Repr): + """ + repr() bounded in depth and length, so formatting an argument cannot overflow + the C stack; it gives the same text as repr() for anything that fits in `width`. + """ + + def __init__(self, width): + super().__init__() + self.maxlevel = self.maxlist = self.maxtuple = self.maxdict = width + self.maxset = self.maxfrozenset = self.maxdeque = self.maxarray = width + self.maxstring = self.maxother = self.maxlong = sys.maxsize + + def repr_dict(self, x, level): + # reprlib sorts keys; repr() keeps insertion order. + if not x: + return "{}" + if level <= 0: + return "{" + self.fillvalue + "}" + pieces = [ + f"{self.repr1(k, level - 1)}: {self.repr1(v, level - 1)}" + for k, v in islice(x.items(), self.maxdict) + ] + if len(x) > self.maxdict: + pieces.append(self.fillvalue) + return "{" + ", ".join(pieces) + "}" + + class OperationErrorContext(ErrorContext): _width = 80 - 8 diff --git a/tests/test_4392_from_iter_numpy_scalars.py b/tests/test_4392_from_iter_numpy_scalars.py index e17f4cce2e..ea3ba399d0 100644 --- a/tests/test_4392_from_iter_numpy_scalars.py +++ b/tests/test_4392_from_iter_numpy_scalars.py @@ -110,6 +110,7 @@ def to_list(self): "x = 1\nfor _ in range(200_000):\n x = {'a': x}", "import sys\nsys.setrecursionlimit(100_000)\nx = S()", "import sys\nsys.setrecursionlimit(100_000)\nx = 1\nfor _ in range(200_000):\n x = (x,)", + "import sys\nsys.setrecursionlimit(100_000)\nx = 1\nfor _ in range(200_000):\n x = {'a': x}", ], ) def test_unbounded_recursion(make_x): @@ -130,3 +131,20 @@ def test_bounded_recursion(): for _ in range(200): x = [x] assert str(ak.from_iter(x).type) == "1 * " + "var * " * 199 + "int64" + + +@pytest.mark.parametrize( + "value", + [ + [1, 2, 3], + (1,), + {"b": 1, "a": [2, (3,)]}, + [[1, [2, [3]]]], + ["x" * 100], + list(range(100)), + ], +) +def test_argument_text_matches_repr(value): + text = repr(value) + expected = text if len(text) <= 72 else text[:69] + "..." + assert ak._errors.ErrorContext().format_argument(72, value) == expected From 55f22712a71f63a2a81ced71488723d14a4b9cf9 Mon Sep 17 00:00:00 2001 From: Lindsey Gray Date: Sat, 26 Sep 2026 09:16:53 -0500 Subject: [PATCH 5/5] fix: cap argument repr depth without reprlib's fillvalue, absent on 3.10 --- src/awkward/_errors.py | 4 ++-- tests/test_4392_from_iter_numpy_scalars.py | 15 +++++++++++++++ 2 files changed, 17 insertions(+), 2 deletions(-) diff --git a/src/awkward/_errors.py b/src/awkward/_errors.py index 13de095418..89925485d8 100644 --- a/src/awkward/_errors.py +++ b/src/awkward/_errors.py @@ -195,13 +195,13 @@ def repr_dict(self, x, level): if not x: return "{}" if level <= 0: - return "{" + self.fillvalue + "}" + return "{...}" pieces = [ f"{self.repr1(k, level - 1)}: {self.repr1(v, level - 1)}" for k, v in islice(x.items(), self.maxdict) ] if len(x) > self.maxdict: - pieces.append(self.fillvalue) + pieces.append("...") return "{" + ", ".join(pieces) + "}" diff --git a/tests/test_4392_from_iter_numpy_scalars.py b/tests/test_4392_from_iter_numpy_scalars.py index ea3ba399d0..847bb27bd1 100644 --- a/tests/test_4392_from_iter_numpy_scalars.py +++ b/tests/test_4392_from_iter_numpy_scalars.py @@ -2,6 +2,7 @@ from __future__ import annotations +import functools import subprocess import sys @@ -142,9 +143,23 @@ def test_bounded_recursion(): [[1, [2, [3]]]], ["x" * 100], list(range(100)), + functools.reduce(lambda x, _: {"a": x}, range(100), 1), ], ) def test_argument_text_matches_repr(value): text = repr(value) expected = text if len(text) <= 72 else text[:69] + "..." assert ak._errors.ErrorContext().format_argument(72, value) == expected + + +def test_argument_whose_repr_raises(): + class Items(dict): + def items(self): + raise ValueError + + # reprlib picks its formatter by the type's name (and, in newer Pythons, its module) + Items.__name__, Items.__module__ = "dict", "builtins" + assert ( + ak._errors.ErrorContext().format_argument(72, Items(a=1)) + == "repr-raised-ValueError" + )