From 7a69d72cf63817c0a57c668268d51ae617fba37c Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 20:14:19 +0000 Subject: [PATCH 01/59] Name the field a record array's child failed on without a declared type A child that failed to convert was wrapped with its field name only on the declared path. With output_schema left as None the same failure named the output column alone, as in "output column 'out': Unsupported numpy type 15", and the field it choked on had to be guessed from the dtype. Both paths now convert a field through one helper that names it, the docstring says so, and the catalogue pins the inferred path. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 20 +++++++++++++------- test/test_mapinarrow_factory.py | 11 +++++++++++ 3 files changed, 30 insertions(+), 7 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 37ba8ba..2e02530 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -405,6 +405,12 @@ ' raise renamed(exc, f"field {field.name!r}") from exc', ' raise', ), + ( + "a record array converted as inferred stops naming a failing field", + "numbarrow/core/mapinarrow_factory.py", + " children = [_record_field(value, name, None) for name in names]", + " children = [_convert(value[name], None) for name in names]", + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index ee10287..d31266a 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -159,20 +159,29 @@ def _struct_column(children, rows, **layout): return pa.StructArray.from_arrays(children, **layout) +def _record_field(value, name, arrow_type): + """One field of a record array as an Arrow array; a failure names the field.""" + try: + return _convert(value[name], arrow_type) + except (pa.ArrowException, TypeError, ValueError, OverflowError) as exc: + raise renamed(exc, f"field {name!r}") from exc + + def _record_to_struct(value, arrow_type): """A numpy record array as a struct column, one child per field. A record array is what an ``@njit`` function returns for a numba record type, and the one ndarray shape that means struct, but ``pa.array`` refuses it with "Unsupported numpy type". Each field goes through the - same conversion as a column of its own, so a unicode field keeps its NULs - and a declared child type is honoured. A record array with no fields + same conversion as a column of its own, so a unicode field keeps its NULs, + a declared child type is honoured, and a field that fails to convert is + named whether or not a type was declared. A record array with no fields becomes that many empty structs: a struct array with no children has no length of its own. """ names = list(value.dtype.names) if arrow_type is None: - children = [_convert(value[name], None) for name in names] + children = [_record_field(value, name, None) for name in names] return _struct_column(children, len(value), names=names) if not pa.types.is_struct(arrow_type): raise TypeError(f"a record array with fields {names} cannot become {type_repr(arrow_type)}") @@ -188,10 +197,7 @@ def _record_to_struct(value, arrow_type): if field.name not in names: children.append(pa.nulls(len(value), type=field.type)) continue - try: - children.append(_convert(value[field.name], field.type)) - except (pa.ArrowException, TypeError, ValueError, OverflowError) as exc: - raise renamed(exc, f"field {field.name!r}") from exc + children.append(_record_field(value, field.name, field.type)) return _struct_column(children, len(value), fields=fields) diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index f979d0d..7acae60 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -558,6 +558,17 @@ def test_a_record_array_becomes_a_struct_column(): run_outputs({"r": wide}, pa.schema([("r", pa.struct([("i", pa.int32())]))])) +def test_a_record_array_field_that_fails_is_named_without_a_declared_type(): + # Only the declared path wrapped a child's failure with its field name; a + # record array converted as inferred failed naming the output column alone. + records = np.zeros(2, dtype=[("ok", "i8"), ("bad", "c16")]) + with pytest.raises(pa.ArrowException, match=r"'r'.*field 'bad'"): + run_outputs({"r": records}) + declared = pa.schema([("r", pa.struct([("ok", pa.int64()), ("bad", pa.float64())]))]) + with pytest.raises(pa.ArrowException, match=r"'r'.*field 'bad'"): + run_outputs({"r": records}, declared) + + def test_a_record_array_with_no_fields_keeps_its_rows(): # pa.StructArray.from_arrays([], names=[]) has no child to take a length # from, so the column came back with no rows and, as the only output From c3388fa53c4ee3b08c94edc63ae8ca53c659a486 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 20:14:21 +0000 Subject: [PATCH 02/59] Read input_columns once, when the function is made The names were read from the argument inside the batch loop, so a generator, map() or filter() handed in was used up by the first batch, every later batch was adapted with no columns at all, and the UDF died on a bare KeyError that said nothing about input_columns. The names are read once now, deduplicated in first-seen order as before, the docstring says a one-shot iterable serves as well as a list, and the catalogue pins the read-once line. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 14 +++++++++----- test/test_mapinarrow_factory.py | 17 +++++++++++++++++ 3 files changed, 32 insertions(+), 5 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 2e02530..dd099cf 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -411,6 +411,12 @@ " children = [_record_field(value, name, None) for name in names]", " children = [_convert(value[name], None) for name in names]", ), + ( + "input_columns is read again on every batch", + "numbarrow/core/mapinarrow_factory.py", + " input_columns_ = named if named is not None else list(dict.fromkeys(names))", + " input_columns_ = list(dict.fromkeys(input_columns if input_columns is not None else names))", + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index d31266a..dc773e4 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -527,7 +527,9 @@ def make_mapinarrow_func( does not have raises :class:`KeyError` listing the batch's columns, since Spark's case-insensitive projection may have spelled it differently, and a name the batch carries more than once, as an - unaliased join produces, raises :class:`ValueError`. + unaliased join produces, raises :class:`ValueError`. The names are + read once, when the function is made, so a one-shot iterable such as + a generator serves as well as a list. :param broadcasts: optional dictionary of broadcast values :param output_schema: optional :class:`pyarrow.Schema` for the batch that is yielded. When given, the dict returned by ``main_func`` is bound to it @@ -572,6 +574,11 @@ def make_mapinarrow_func( raise TypeError( f"input_columns must be a list of column names, not the string {input_columns!r}" ) + # dict.fromkeys keeps first-seen order. Naming a column twice produces the + # same arrays twice, so it stays harmless. Read once, here: read inside the + # batch loop, a generator, map() or filter() handed in was used up by the + # first batch, and every later batch then saw no columns at all. + named = None if input_columns is None else list(dict.fromkeys(input_columns)) if output_schema is not None and not isinstance(output_schema, pa.Schema): # A PySpark StructType is the schema mapInArrow itself takes, and it # carries .names too, so one handed here got as far as the first batch @@ -584,11 +591,8 @@ def _(iterator): for batch in iterator: data_dict: dict[str, np.ndarray | dict[str, np.ndarray]] = {} bitmap_dict: dict[str, np.ndarray | None | dict[str, np.ndarray | None]] = {} - requested = input_columns if input_columns is not None else batch.schema.names - # dict.fromkeys keeps first-seen order. Naming a column twice - # produces the same arrays twice, so it stays harmless. - input_columns_ = list(dict.fromkeys(requested)) names = batch.schema.names + input_columns_ = named if named is not None else list(dict.fromkeys(names)) for col in input_columns_: if col not in names: # Spark's projection is case-insensitive and may have diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 7acae60..183191b 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -92,6 +92,23 @@ def test_input_columns_selects_only_the_named_columns(): assert seen["data"]["c"].tolist() == [5, 6] +def test_input_columns_is_read_once_so_a_generator_serves_every_batch(): + # The names were read from the argument inside the batch loop, so a + # generator, map() or filter() was used up by the first batch and every + # later batch was adapted with no columns: the UDF died on a bare KeyError. + batches = [pa.RecordBatch.from_pydict({"x": [1, 2], "y": [0, 0]}), + pa.RecordBatch.from_pydict({"x": [3], "y": [0]})] + seen = [] + + def main(data_dict, bitmap_dict, broadcasts): + seen.append(list(data_dict)) + return {"out": data_dict["x"] * 2} + + got = list(make_mapinarrow_func(main, input_columns=(name for name in ["x"]))(iter(batches))) + assert seen == [["x"], ["x"]] + assert [batch.column("out").to_pylist() for batch in got] == [[2, 4], [6]] + + def test_a_struct_field_sharing_a_column_name_reaches_the_udf(): # Four ordinary Spark StructTypes convert to this shape: a top-level column # and a struct field sharing a name. Nested under its column, the field From f69aa70e034e9dc275b82e20b7348d1923041fb7 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 20:29:02 +0000 Subject: [PATCH 03/59] Point the catalogue's record-array entry at the helper that names the field The earlier commit moved the raise that entry mutated into _record_field, so the entry's old text was gone and the catalogue reported it stale instead of running it. The entry now mutates the helper's raise, which drops the field name on the declared and the inferred path alike, and the tests for both kill it. --- .github/scripts/mutation_guard_check.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index dd099cf..9267889 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -402,8 +402,8 @@ ( 'a record array field failure stops naming the field', 'numbarrow/core/mapinarrow_factory.py', - ' raise renamed(exc, f"field {field.name!r}") from exc', - ' raise', + ' raise renamed(exc, f"field {name!r}") from exc', + ' raise', ), ( "a record array converted as inferred stops naming a failing field", From 25c954ba32371d152e86ca5df9e70734d1736bc1 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:50:52 +0000 Subject: [PATCH 04/59] Fold a time unit's multiplier in, and take a coarse unit to seconds or days, before Arrow reads a datetime64 or timedelta64 output pa.array reads numpy's base unit and ignores a multiplier, so a datetime64[5s] column of five-second bins came back at one-second steps, 2020 read as 1980, and datetime64[2D] slipped past the day-unit inference into the misread it guards against. An hour or minute unit, and a week, month or year unit, raised ArrowNotImplementedError against the docstring's promise of a timestamp of the unit. A multiplier now folds into its base unit, hours and minutes become seconds, weeks, months and years become days and so date32, a unit finer than a nanosecond or a month or year timedelta is refused by name, and the docstring and README say so. --- .github/scripts/mutation_guard_check.py | 12 ++++++ README.md | 9 +++-- numbarrow/core/mapinarrow_factory.py | 44 +++++++++++++++++++++- test/test_docs_match_code.py | 10 +++-- test/test_mapinarrow_factory.py | 50 ++++++++++++++++++++++--- 5 files changed, 110 insertions(+), 15 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 9267889..ba1177a 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -417,6 +417,18 @@ " input_columns_ = named if named is not None else list(dict.fromkeys(names))", " input_columns_ = list(dict.fromkeys(input_columns if input_columns is not None else names))", ), + ( + 'a time unit multiplier stops being folded in', + 'numbarrow/core/mapinarrow_factory.py', + ' if target == unit and count == 1:\n return value', + ' if target == unit:\n return value', + ), + ( + 'a coarse time unit stops being taken to seconds', + 'numbarrow/core/mapinarrow_factory.py', + ' if unit in ("h", "m") or (family == "timedelta64" and unit in ("W", "D")):\n target = "s"', + ' if False:\n target = "s"', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/README.md b/README.md index d1debc4..c9223f0 100644 --- a/README.md +++ b/README.md @@ -72,10 +72,11 @@ and a naive `timestamp[us]` holding the same int64 adapt to the same `datetime64[us]`, exactly as pyarrow's `to_numpy` does, so a UDF's calendar arithmetic runs on UTC instants and can disagree with Spark's own `to_date` by the session offset. On the way back out of `make_mapinarrow_func` a -`datetime64` output becomes a naive timestamp of its unit, except -`datetime64[D]`, which becomes `date32`, and a `date64` input passed through -comes back `timestamp[ms]`; pass `output_schema` to restore a zone or a date -type. +`datetime64` output becomes a naive timestamp of its unit, with a multiplier +such as `datetime64[5s]` folded in and an hour or minute unit taken to +seconds; a day, week, month or year unit becomes `date32`, a unit finer than a +nanosecond is refused, and a `date64` input passed through comes back +`timestamp[ms]`; pass `output_schema` to restore a zone or a date type. A `ListArray` of structs flattens its elements and returns no offsets, so a null outer row can be neither reported nor accounted for in the element-to-row diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index dc773e4..143a8eb 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -252,6 +252,42 @@ def _convert(value, arrow_type): return pa.array(value, type=arrow_type) +def _at_arrow_unit(value): + """A datetime64 or timedelta64 array at a unit pyarrow models, with any multiplier folded in. + + ``pa.array`` reads numpy's base unit and ignores a multiplier, so a + ``datetime64[5s]`` column of five-second bins came back at one-second + steps, 2020 read as 1980, and ``datetime64[2D]`` slipped past the day-unit + inference in ``_ndarray_to_arrow`` into the misread it exists to prevent. + A multiplier folds into its base unit exactly; an hour or minute unit + becomes seconds and a week, month or year unit becomes days, exactly too, + where ``pa.array`` refused them outright. A unit finer than a nanosecond, + and a month or year timedelta, which has no fixed length, would not + convert exactly and are refused instead. + """ + family = "datetime64" if value.dtype.kind == "M" else "timedelta64" + unit, count = np.datetime_data(value.dtype) + if unit in ("ps", "fs", "as"): + raise TypeError( + f"a {value.dtype} array has no Arrow type: pyarrow models seconds down to nanoseconds; " + f"convert it to {family}[ns] first, which drops the finer digits" + ) + if family == "timedelta64" and unit in ("M", "Y"): + raise TypeError( + f"a {value.dtype} array has no fixed length in seconds; convert it to {family}[D] or " + f"{family}[s] first" + ) + if unit in ("h", "m") or (family == "timedelta64" and unit in ("W", "D")): + target = "s" + elif unit in ("W", "M", "Y"): + target = "D" + else: + target = unit + if target == unit and count == 1: + return value + return value.astype(f"{family}[{target}]") + + def _ndarray_to_arrow(value, arrow_type): """An ndarray of a non-object dtype as an Arrow array; see ``_convert``.""" if value.dtype.names is not None: @@ -261,6 +297,8 @@ def _ndarray_to_arrow(value, arrow_type): return pa.array(value.tolist(), type=arrow_type or pa.string()) if kind == "S": return pa.array(value.tolist(), type=arrow_type or pa.binary()) + if kind in ("M", "m"): + value = _at_arrow_unit(value) if value.dtype == np.dtype("datetime64[D]") and arrow_type is not None: # Under a declared timestamp or int32 ``pa.array`` reads a day-unit # array's 8-byte values as the 4-byte days of a date32, so every other @@ -565,8 +603,10 @@ def make_mapinarrow_func( order decides, and every type is inferred from the value, so a unicode or bytes array comes back ``string`` or ``binary`` whatever type went in, a ``datetime64`` array comes back a naive ``timestamp`` of its - unit, except a day-unit one, which comes back ``date32``, and an - object array holding only ``None`` comes back ``null``. + unit, with a multiplier such as ``datetime64[5s]`` folded in and an + hour or minute unit taken to seconds; a day, week, month or year unit + comes back ``date32``, a unit finer than a nanosecond is refused, and + an object array holding only ``None`` comes back ``null``. """ broadcasts = broadcasts if broadcasts is not None else {} if isinstance(input_columns, str): diff --git a/test/test_docs_match_code.py b/test/test_docs_match_code.py index 13cc1b4..769b690 100644 --- a/test/test_docs_match_code.py +++ b/test/test_docs_match_code.py @@ -161,10 +161,12 @@ def test_an_inferred_datetime64_column_comes_back_as_the_docs_say(): # Both sentences promised a timestamp of the array's unit for every unit, # and pa.array infers date32 for the day one, which the round-trip test's # own drift table admits by leaving date32 out of it. - assert ("a ``datetime64`` array comes back a naive ``timestamp`` of its unit, except a " - "day-unit one, which comes back ``date32``") in FACTORY_DOC - assert ("a `datetime64` output becomes a naive timestamp of its unit, except " - "`datetime64[D]`, which becomes `date32`") in README_TEXT + assert ("a ``datetime64`` array comes back a naive ``timestamp`` of its unit, with a multiplier such as " + "``datetime64[5s]`` folded in and an hour or minute unit taken to seconds; a day, week, month or year " + "unit comes back ``date32``, a unit finer than a nanosecond is refused") in FACTORY_DOC + assert ("a `datetime64` output becomes a naive timestamp of its unit, with a multiplier such as " + "`datetime64[5s]` folded in and an hour or minute unit taken to seconds; a day, week, month or year " + "unit becomes `date32`, a unit finer than a nanosecond is refused") in README_TEXT days = np.array(["2020-01-01", "2020-01-02"], dtype="datetime64[D]") column = _inferred_output_column(days) assert column.type == pa.date32() diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 183191b..edc30ce 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -361,7 +361,8 @@ def test_a_day_unit_datetime64_under_another_declared_type_is_its_date32_cast(): def test_a_declared_type_keeps_the_other_datetime64_conversions(): # What widening the day unit must leave alone: a unit change that drops # digits still raises, a date type still floors to the day without a word, - # and a timedelta is still a dtype pa.array has no converter for. + # a day-unit timedelta comes back in seconds, which pa.array models, and a + # month one has no fixed length and is refused. days = np.array(["2020-01-01", "2020-01-02"], dtype="datetime64[D]") sub_second = datetime.datetime(2020, 1, 1, 12, 34, 56, 789012) seconds = np.array([sub_second], dtype="datetime64[s]") @@ -376,10 +377,11 @@ def test_a_declared_type_keeps_the_other_datetime64_conversions(): dated = run_outputs({"t": days}, pa.schema([("t", pa.date64())])).column("t") assert dated.to_pylist() == [datetime.date(2020, 1, 1), datetime.date(2020, 1, 2)] spans = np.array([1, 2], dtype="timedelta64[D]") - with pytest.raises(pa.ArrowNotImplementedError, match=r"'t'.*timedelta64"): - run_outputs({"t": spans}, pa.schema([("t", pa.duration("s"))])) - with pytest.raises(pa.ArrowNotImplementedError, match=r"'t'.*timedelta64"): - run_outputs({"t": spans}) + for schema in (None, pa.schema([("t", pa.duration("s"))])): + got = run_outputs({"t": spans}, schema).column("t") + assert got.type == pa.duration("s") and got.to_pylist() == [datetime.timedelta(days=d) for d in (1, 2)] + with pytest.raises(TypeError, match=r"'t'.*no fixed length"): + run_outputs({"t": np.array([1, 2], dtype="timedelta64[M]")}) def test_output_schema_refuses_a_struct_key_no_field_has(): @@ -841,3 +843,41 @@ def test_a_bare_tuple_is_a_sequence_not_a_pair(): assert listed.type == pa.list_(pa.int64()) and listed.to_pylist() == [[1, 2], None] arrays = run_outputs({"a": (np.array([1, 2]), np.array([3, 4]))}).column("a") assert arrays.type == pa.list_(pa.int64()) and arrays.to_pylist() == [[1, 2], [3, 4]] + + +def test_a_time_unit_multiplier_is_folded_in_before_arrow_reads_it(): + # pa.array reads numpy's base unit and ignored the multiplier, so five-second + # bins came back at one-second steps, and datetime64[2D] slipped past the + # day-unit inference into the misread it guards against. + base = np.datetime64("2020-03-01T10:00:00") + stamps = (base + np.arange(3) * np.timedelta64(5, "s")).astype("datetime64[5s]") + expected = [datetime.datetime(2020, 3, 1, 10, 0, s) for s in (0, 5, 10)] + got = run_outputs({"t": stamps}).column("t") + assert got.type == pa.timestamp("s") and got.to_pylist() == expected + declared = pa.schema([("t", pa.timestamp("s"))]) + assert run_outputs({"t": stamps}, declared).column("t").to_pylist() == expected + days = np.array(["2020-03-01", "2020-03-03", "2020-03-05"], dtype="datetime64[2D]") + assert run_outputs({"d": days}).column("d").to_pylist() == [datetime.date(2020, 3, d) for d in (1, 3, 5)] + stamped = run_outputs({"d": days}, pa.schema([("d", pa.timestamp("s"))])).column("d") + assert stamped.to_pylist() == [datetime.datetime(2020, 3, d) for d in (1, 3, 5)] + deltas = np.array([15, 60, 45], dtype="timedelta64[s]").astype("timedelta64[15s]") + assert run_outputs({"e": deltas}).column("e").to_pylist() == [datetime.timedelta(seconds=s) for s in (15, 60, 45)] + + +def test_a_coarse_time_unit_becomes_seconds_or_days_and_a_finer_one_is_refused(): + # pa.array models seconds down to nanoseconds; an hour, minute, week, month + # or year unit raised ArrowNotImplementedError against the docstring's + # promise of a timestamp of the unit. + hours = np.array(["2020-03-01T10", "2020-03-01T11"], dtype="datetime64[h]") + got = run_outputs({"t": hours}).column("t") + assert got.type == pa.timestamp("s") + assert got.to_pylist() == [datetime.datetime(2020, 3, 1, h) for h in (10, 11)] + months = np.array(["2020-03", "2020-04"], dtype="datetime64[M]") + got = run_outputs({"d": months}).column("d") + assert got.type == pa.date32() and got.to_pylist() == [datetime.date(2020, 3, 1), datetime.date(2020, 4, 1)] + weeks = np.array([1, 2], dtype="timedelta64[W]") + assert run_outputs({"e": weeks}).column("e").to_pylist() == [datetime.timedelta(weeks=w) for w in (1, 2)] + with pytest.raises(TypeError, match=r"'t'.*nanoseconds"): + run_outputs({"t": np.array([1, 2], dtype="datetime64[ps]")}) + with pytest.raises(TypeError, match=r"'e'.*no fixed length"): + run_outputs({"e": np.array([1, 2], dtype="timedelta64[M]")}) From c1950ed79925d71ab068b1673ae82cb8c2a90708 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:50:54 +0000 Subject: [PATCH 05/59] Refuse a str, bytes or 0-d array output rather than spread it one character per row A str or bytes returned as a column went to pa.array, which iterates it, so {'country': 'US'} over a two-row batch was the rows U and S, and a 0-d unicode or bytes array's tolist() is that scalar, which defeated the one-dimensional refusal pa.array gives the array itself. Both are refused naming the column, and the message says how to build a constant column. --- .github/scripts/mutation_guard_check.py | 12 ++++++++++++ numbarrow/core/mapinarrow_factory.py | 17 +++++++++++++++-- test/test_messages.py | 11 +++++++++++ 3 files changed, 38 insertions(+), 2 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index ba1177a..4d2c647 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -429,6 +429,18 @@ ' if unit in ("h", "m") or (family == "timedelta64" and unit in ("W", "D")):\n target = "s"', ' if False:\n target = "s"', ), + ( + 'a str output stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' if isinstance(value, (str, bytes)):\n raise TypeError(', + ' if False:\n raise TypeError(', + ), + ( + 'a 0-d or 2-d array output stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' if value.ndim != 1:', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 143a8eb..b46d7ff 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -290,6 +290,11 @@ def _at_arrow_unit(value): def _ndarray_to_arrow(value, arrow_type): """An ndarray of a non-object dtype as an Arrow array; see ``_convert``.""" + if value.ndim != 1: + # A 0-d unicode or bytes array's tolist() is a bare scalar, which + # pa.array spreads one character per row, defeating the refusal it + # gives the array itself; every other dtype it refuses on its own. + raise TypeError(f"a {value.ndim}-dimensional {value.dtype} array; an output column is one-dimensional") if value.dtype.names is not None: return _record_to_struct(value, arrow_type) kind = value.dtype.kind @@ -377,8 +382,11 @@ def _to_arrow(value, name, arrow_type=None, handed=MappingProxyType({})): that case is refused here by identity rather than left to the byte check. ``pa.array`` iterates a Mapping, so a dict of arrays returned under one key silently became a string column of the dict's keys, with a different - row count and nothing raised; it is refused outright. Every other failure - on the output side named no column at all. + row count and nothing raised; it is refused outright. A str or bytes + returned as a column is refused the same way: ``pa.array`` spreads it one + character per row, so ``{"country": "US"}`` over a two-row batch was the + rows ``U`` and ``S``. Every other failure on the output side named no + column at all. """ value, bitmap = _split_pair(value) if isinstance(value, Mapping): @@ -387,6 +395,11 @@ def _to_arrow(value, name, arrow_type=None, handed=MappingProxyType({})): f"its keys; return an ndarray, a list or a pyarrow Array per column, and for a " f"struct column a list of dicts or a record array" ) + if isinstance(value, (str, bytes)): + raise TypeError( + f"output column {name!r} is a {type(value).__name__}, which pa.array would spread one " + f"character per row; a constant column is np.full(rows, value)" + ) try: array = _convert(value, arrow_type) if bitmap is None: diff --git a/test/test_messages.py b/test/test_messages.py index 3b7ee20..5092421 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -178,3 +178,14 @@ def test_an_exception_that_cannot_be_rebuilt_from_a_message_is_renamed_as_a_valu wrapped = renamed(UnicodeDecodeError("utf-8", b"\xff", 0, 1, "bad"), "column 'x'") assert type(wrapped) is ValueError assert str(wrapped).startswith("column 'x': ") + + +def test_a_scalar_string_output_is_refused_rather_than_spread(): + # {"country": "US"} over a two-row batch came back as the rows "U" and "S", + # and a 0-d unicode array's tolist() is that scalar, which defeated + # pa.array's own refusal of a 0-d array. + batch = _batch(v=[1, 2]) + for value in ("US", b"US", np.str_("US"), np.array("US"), np.array([["a", "b"]])): + fn = make_mapinarrow_func(lambda d, b, br, value=value: {"country": value}) + with pytest.raises(TypeError, match="'country'"): + list(fn(iter([batch]))) From 3e4e5d3ab232d4df40eee652bd6202619fea9606 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:51:02 +0000 Subject: [PATCH 06/59] Refuse a Nullable inside a Nullable by name A helper's Nullable wrapped once more with the input's bitmap went to pa.array as the 2-tuple it is and came back as a two-row list column, its data as one row and its bitmap's bytes as the other, silently on every batch whose column had no validity buffer and with a message blaming a resize otherwise. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 7 +++++++ test/test_mapinarrow_factory.py | 8 ++++++++ 3 files changed, 21 insertions(+) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 4d2c647..33450c0 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -441,6 +441,12 @@ ' if value.ndim != 1:', ' if False:', ), + ( + 'a Nullable inside a Nullable stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + " if isinstance(value, Nullable):\n # A helper's Nullable", + " if False:\n # A helper's Nullable", + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index b46d7ff..15661b0 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -389,6 +389,13 @@ def _to_arrow(value, name, arrow_type=None, handed=MappingProxyType({})): column at all. """ value, bitmap = _split_pair(value) + if isinstance(value, Nullable): + # A helper's Nullable wrapped once more went to pa.array as the 2-tuple + # it is: two rows, the data as one and the bitmap's bytes as the other. + raise TypeError( + f"output column {name!r} is a Nullable inside a Nullable, which pa.array would read as " + f"a two-row column of its data and its bitmap; wrap the data once" + ) if isinstance(value, Mapping): raise TypeError( f"output column {name!r} is a {type(value).__name__}, which pa.array would read as " diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index edc30ce..8b7e6eb 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -881,3 +881,11 @@ def test_a_coarse_time_unit_becomes_seconds_or_days_and_a_finer_one_is_refused() run_outputs({"t": np.array([1, 2], dtype="datetime64[ps]")}) with pytest.raises(TypeError, match=r"'e'.*no fixed length"): run_outputs({"e": np.array([1, 2], dtype="timedelta64[M]")}) + + +def test_a_nullable_inside_a_nullable_is_refused(): + # A helper that returned a Nullable, wrapped once more with the input's + # bitmap, went to pa.array as the 2-tuple it is and came back as two rows. + inner = Nullable(np.arange(2, dtype=np.int64), None) + with pytest.raises(TypeError, match=r"'out'.*Nullable inside a Nullable"): + run_outputs({"out": Nullable(inner, None)}) From 8dc54a1e6190429bb6f566ef8f74198ac78c595a Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:55:27 +0000 Subject: [PATCH 07/59] Look inside every row shape the key check let past: tuples by position, sequences by __getitem__, key/value entry dicts, and refuse a namedtuple in another field order The key check kept only Mapping rows and 2-tuple map entries, and pa.array accepts more: a struct row given as a tuple, a namedtuple or a pyspark Row binds by position, a sequence by __getitem__ alone iterates without an __iter__ attribute, and a map entry may be a {key, value} dict, the shape Spark's map_entries produces. Behind any of those a mistyped nested key was silently nulled, and a namedtuple or Row naming the declared fields in another order swapped every same-typed field. Tuple rows are now checked by position, iterability is tested with iter, entry dicts are read as pairs, a permuted namedtuple is refused by name, and a pyarrow scalar row is left to pa.array, which from pyarrow 21 made a MapScalar a Mapping whose values is an array and crashed the check. --- .github/scripts/mutation_guard_check.py | 30 +++++++++++++ numbarrow/core/mapinarrow_factory.py | 52 +++++++++++++++++----- test/test_mapinarrow_factory.py | 58 +++++++++++++++++++++++++ 3 files changed, 129 insertions(+), 11 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 33450c0..86e5dd9 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -447,6 +447,36 @@ " if isinstance(value, Nullable):\n # A helper's Nullable", " if False:\n # A helper's Nullable", ), + ( + 'tuple rows stop being checked by position', + 'numbarrow/core/mapinarrow_factory.py', + ' children.extend(row[index] for row in tuples if index < len(row))', + ' pass', + ), + ( + 'a namedtuple naming the fields in another order stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' if given is not None and set(given) == set(names) and list(given) != names:', + ' if False:', + ), + ( + 'iterability stops being tested with iter', + 'numbarrow/core/mapinarrow_factory.py', + ' try:\n iter(row)\n except TypeError:\n continue\n kept.append(row)', + ' if not hasattr(row, "__iter__"):\n continue\n kept.append(row)', + ), + ( + 'a key/value entry dict stops being read as a pair', + 'numbarrow/core/mapinarrow_factory.py', + ' if isinstance(pair, Mapping) and set(pair) == {"key", "value"}:', + ' if False:', + ), + ( + 'pyarrow scalar rows stop being passed over by the key check', + 'numbarrow/core/mapinarrow_factory.py', + ' if isinstance(row, (str, bytes, pa.Scalar)) or', + ' if isinstance(row, (str, bytes)) or', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 15661b0..e566461 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -88,13 +88,24 @@ def _iterable_rows(rows): they are passed over too: ``pa.array`` refuses such a row at its first element, where spreading it into a list here first took seconds and hundreds of megabytes for a long one. + + Iterability is tested with ``iter`` rather than by the ``__iter__`` + attribute: a sequence by ``__getitem__`` alone has no such attribute and + iterates all the same, and ``pa.array`` reads it item by item. A pyarrow + scalar row is passed over as well: ``pa.array`` checks one against the + declared type itself, and from pyarrow 21 a MapScalar is a Mapping whose + ``values`` is an array rather than a method. """ - return [ - row for row in rows - if hasattr(row, "__iter__") - and not isinstance(row, (str, bytes)) - and not (isinstance(row, np.ndarray) and row.dtype.kind != "O") - ] + kept = [] + for row in rows: + if isinstance(row, (str, bytes, pa.Scalar)) or (isinstance(row, np.ndarray) and row.dtype.kind != "O"): + continue + try: + iter(row) + except TypeError: + continue + kept.append(row) + return kept def _check_keys(rows, arrow_type): @@ -105,11 +116,16 @@ def _check_keys(rows, arrow_type): ``amount`` builds a whole column of nulls under an identical schema, without a word. The same typo on a top-level key raises; this makes the nested one raise too, however deep the struct sits inside a list, a map or - another struct. + another struct. A row given as a tuple, a namedtuple or a pyspark Row + binds by position, so its elements are checked against the fields in + declared order, and one that names the declared fields in another order, + which would swap every same-typed field without a word, is refused. """ if pa.types.is_struct(arrow_type): fields = {field.name: field.type for field in _struct_fields(arrow_type)} - dicts = [row for row in rows if isinstance(row, Mapping)] + names = list(fields) + dicts = [row for row in rows if isinstance(row, Mapping) and not isinstance(row, pa.Scalar)] + tuples = [row for row in rows if isinstance(row, tuple) and not isinstance(row, pa.Scalar)] seen = set() for row in dicts: seen.update(row) @@ -120,9 +136,18 @@ def _check_keys(rows, arrow_type): f"that no declared field has; Arrow matches struct fields by exact name and " f"fills a missing one with null" ) - for name, child_type in fields.items(): + for row in tuples: + given = getattr(row, "_fields", None) or getattr(row, "__fields__", None) + if given is not None and set(given) == set(names) and list(given) != names: + raise ValueError( + f"declared {type_repr(arrow_type)} but a row names its fields {list(given)}; a tuple's " + f"fields bind by position, so build it in the declared order or return dicts" + ) + for index, (name, child_type) in enumerate(fields.items()): if _carries_keys(child_type): - _check_keys([row[name] for row in dicts if name in row], child_type) + children = [row[name] for row in dicts if name in row] + children.extend(row[index] for row in tuples if index < len(row)) + _check_keys(children, child_type) elif _is_list_like(arrow_type): _check_keys([item for row in _iterable_rows(rows) for item in row], arrow_type.value_type) elif pa.types.is_map(arrow_type): @@ -146,7 +171,12 @@ def _map_entries(rows): items.extend(row.values()) else: for pair in row: - if isinstance(pair, (tuple, list)) and len(pair) == 2: + if isinstance(pair, Mapping) and set(pair) == {"key", "value"}: + # The entry shape Spark's map_entries produces, which + # pa.array reads as a pair. + keys.append(pair["key"]) + items.append(pair["value"]) + elif isinstance(pair, (tuple, list)) and len(pair) == 2: keys.append(pair[0]) items.append(pair[1]) return keys, items diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 8b7e6eb..44267a9 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -1,3 +1,4 @@ +import collections import datetime import weakref @@ -889,3 +890,60 @@ def test_a_nullable_inside_a_nullable_is_refused(): inner = Nullable(np.arange(2, dtype=np.int64), None) with pytest.raises(TypeError, match=r"'out'.*Nullable inside a Nullable"): run_outputs({"out": Nullable(inner, None)}) + + +def test_the_key_check_reaches_rows_that_are_not_dicts(): + # A tuple, a namedtuple, a pyspark Row and a sequence by __getitem__ alone + # bind by position in pa.array, and none of them was looked inside, so a + # mistyped nested key was silently nulled behind any of them. + Outer = collections.namedtuple("Outer", ["id", "inner"]) + + class Row(tuple): + __fields__ = ["id", "inner"] + + class Seq: + def __init__(self, items): + self._items = items + + def __len__(self): + return len(self._items) + + def __getitem__(self, index): + return self._items[index] + + nested = pa.schema([("s", pa.struct([("id", pa.int64()), ("inner", pa.struct([("amount", pa.int64())]))]))]) + for rows in ([(1, {"Amount": 5}), (2, {"Amount": 6})], + [Outer(1, {"Amount": 5}), Outer(2, {"Amount": 6})], + [Row((1, {"Amount": 5})), Row((2, {"Amount": 6}))]): + with pytest.raises(ValueError, match=r"'s'.*'Amount'"): + run_outputs({"s": rows}, nested) + listed = pa.schema([("s", pa.list_(pa.struct([("amount", pa.int64())])))]) + with pytest.raises(ValueError, match=r"'s'.*'Amount'"): + run_outputs({"s": [Seq([{"amount": 1}]), Seq([{"Amount": 2}])]}, listed) + mapped = pa.schema([("s", pa.map_(pa.string(), pa.struct([("amount", pa.int64())])))]) + entries = [[{"key": "k", "value": {"Amount": 1}}], [{"key": "j", "value": {"amount": 2}}]] + with pytest.raises(ValueError, match=r"'s'.*'Amount'"): + run_outputs({"s": entries}, mapped) + good = run_outputs({"s": [Outer(1, {"amount": 5}), (2, {"amount": 6})]}, nested).column("s") + assert good.to_pylist() == [{"id": 1, "inner": {"amount": 5}}, {"id": 2, "inner": {"amount": 6}}] + + +def test_a_namedtuple_row_naming_the_fields_in_another_order_is_refused(): + # pa.array binds a namedtuple by position, so YX(y=100, x=0) under + # struct put 100 in x and 0 in y without a word. + YX = collections.namedtuple("YX", ["y", "x"]) + schema = pa.schema([("s", pa.struct([("x", pa.int64()), ("y", pa.int64())]))]) + with pytest.raises(ValueError, match=r"'s'.*\['y', 'x'\].*position"): + run_outputs({"s": [YX(100, 0), YX(200, 1)]}, schema) + XY = collections.namedtuple("XY", ["x", "y"]) + got = run_outputs({"s": [XY(0, 100), XY(1, 200)]}, schema).column("s") + assert got.to_pylist() == [{"x": 0, "y": 100}, {"x": 1, "y": 200}] + + +def test_pyarrow_scalar_rows_are_left_to_pa_array(): + # From pyarrow 21 a MapScalar is a Mapping whose values is an array, so the + # key check died calling it; pa.array checks a scalar row itself. + mapped = pa.map_(pa.string(), pa.struct([("amount", pa.int64())])) + rows = list(pa.array([[("k", {"amount": 1})], [("j", {"amount": 2}), ("i", {"amount": 3})]], type=mapped)) + got = run_outputs({"s": rows}, pa.schema([("s", mapped)])).column("s") + assert got.to_pylist() == [[("k", {"amount": 1})], [("j", {"amount": 2}), ("i", {"amount": 3})]] From 58d3be8bf31a53654665610b45732d0ec3af0ebd Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:55:29 +0000 Subject: [PATCH 08/59] Refuse a pandas Series or DataFrame row by name, name the column on a KeyError, and combine a chunked array pa.array hands back pa.array reads a pandas Series row by its index labels, so a sorted or filtered one came back reordered without a word and one from a groupby died on a bare KeyError(0); a DataFrame under one output key died the same way or came back transposed; a KeyError raised inside pa.array escaped _to_arrow and _record_field unnamed although renamed has a branch for it; and a Series over a multi-chunk pyarrow array came back from pa.array as a ChunkedArray that RecordBatch.from_arrays refused naming no column. Series and DataFrame rows and a DataFrame column are refused naming the column and the remedy, KeyError joins both except tuples, and a ChunkedArray is combined. --- .github/scripts/mutation_guard_check.py | 28 +++++++++++++++++ numbarrow/core/mapinarrow_factory.py | 40 +++++++++++++++++++++---- test/test_mapinarrow_factory.py | 32 ++++++++++++++++++++ 3 files changed, 94 insertions(+), 6 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 86e5dd9..0aae4cd 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -477,6 +477,34 @@ ' if isinstance(row, (str, bytes, pa.Scalar)) or', ' if isinstance(row, (str, bytes)) or', ), + ( + 'a pandas Series row stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' if _is_pandas(row, "Series", "DataFrame"):', + ' if False:', + ), + ( + 'a KeyError from pa.array stops naming the column', + 'numbarrow/core/mapinarrow_factory.py', + ' except (pa.ArrowException, TypeError, ValueError, OverflowError, KeyError) as exc:\n' + ' raise renamed(exc, f"output column {name!r}") from exc', + ' except (pa.ArrowException, TypeError, ValueError, OverflowError) as exc:\n' + ' raise renamed(exc, f"output column {name!r}") from exc', + ), + ( + 'a KeyError from a record field stops naming the field', + 'numbarrow/core/mapinarrow_factory.py', + ' except (pa.ArrowException, TypeError, ValueError, OverflowError, KeyError) as exc:\n' + ' raise renamed(exc, f"field {name!r}") from exc', + ' except (pa.ArrowException, TypeError, ValueError, OverflowError) as exc:\n' + ' raise renamed(exc, f"field {name!r}") from exc', + ), + ( + 'a chunked array from pa.array stops being combined', + 'numbarrow/core/mapinarrow_factory.py', + ' if isinstance(array, pa.ChunkedArray):\n # A pandas Series over a multi-chunk', + ' if False:\n # A pandas Series over a multi-chunk', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index e566461..e39a6ac 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -193,7 +193,7 @@ def _record_field(value, name, arrow_type): """One field of a record array as an Arrow array; a failure names the field.""" try: return _convert(value[name], arrow_type) - except (pa.ArrowException, TypeError, ValueError, OverflowError) as exc: + except (pa.ArrowException, TypeError, ValueError, OverflowError, KeyError) as exc: raise renamed(exc, f"field {name!r}") from exc @@ -274,12 +274,27 @@ def _convert(value, arrow_type): return _ndarray_to_arrow(value, arrow_type) # An object array, a list, a tuple, a pandas Series or any other iterable # of Python objects. + if not hasattr(value, "__len__"): + # A generator would be consumed by the checks, so it is read once. + value = list(value) + if isinstance(value, (list, tuple)): + for row in value: + if _is_pandas(row, "Series", "DataFrame"): + # pa.array reads a Series row by its index labels, so a sorted + # or filtered one came back reordered, and one whose labels + # were not 0..n-1 died on a bare KeyError. + raise TypeError( + f"a row is a pandas {type(row).__name__}, which pa.array reads by its labels " + f"rather than in order; hand it over as row.to_numpy() or list(row)" + ) if arrow_type is not None and _carries_keys(arrow_type): - if not hasattr(value, "__len__"): - # A generator would be consumed by the check, so it is read once. - value = list(value) _check_keys(value, arrow_type) - return pa.array(value, type=arrow_type) + array = pa.array(value, type=arrow_type) + if isinstance(array, pa.ChunkedArray): + # A pandas Series over a multi-chunk pyarrow array comes back as one, + # which RecordBatch.from_arrays refused naming no column. + array = array.combine_chunks() + return array def _at_arrow_unit(value): @@ -344,6 +359,12 @@ def _ndarray_to_arrow(value, arrow_type): return pa.array(value, type=arrow_type) +def _is_pandas(value, *names): + """Whether *value* is a pandas object of one of the given class names, without importing pandas.""" + cls = type(value) + return cls.__name__ in names and cls.__module__.split(".")[0] == "pandas" + + def _split_pair(value): """The data and the bitmap of a :class:`Nullable`; any other shape carries no bitmap. @@ -437,6 +458,13 @@ def _to_arrow(value, name, arrow_type=None, handed=MappingProxyType({})): f"output column {name!r} is a {type(value).__name__}, which pa.array would spread one " f"character per row; a constant column is np.full(rows, value)" ) + if _is_pandas(value, "DataFrame"): + # Read by its column labels: a one-column frame died on a bare + # KeyError(0), and one with integer labels came back transposed. + raise TypeError( + f"output column {name!r} is a DataFrame, which pa.array reads by its column labels; " + f"return one Series or ndarray per column" + ) try: array = _convert(value, arrow_type) if bitmap is None: @@ -448,7 +476,7 @@ def _to_arrow(value, name, arrow_type=None, handed=MappingProxyType({})): f"{len(array)} rows; a resized column needs a bitmap of its own" ) return _with_validity(array, bitmap) - except (pa.ArrowException, TypeError, ValueError, OverflowError) as exc: + except (pa.ArrowException, TypeError, ValueError, OverflowError, KeyError) as exc: raise renamed(exc, f"output column {name!r}") from exc diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 44267a9..ff1a54e 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -947,3 +947,35 @@ def test_pyarrow_scalar_rows_are_left_to_pa_array(): rows = list(pa.array([[("k", {"amount": 1})], [("j", {"amount": 2}), ("i", {"amount": 3})]], type=mapped)) got = run_outputs({"s": rows}, pa.schema([("s", mapped)])).column("s") assert got.to_pylist() == [[("k", {"amount": 1})], [("j", {"amount": 2}), ("i", {"amount": 3})]] + + +def test_a_pandas_series_row_is_refused_rather_than_read_by_label(): + # pa.array reads a Series row by its index labels: a sorted one came back + # in label order, one from a groupby died on a bare KeyError, a frame under + # one key came back transposed, and a multi-chunk pyarrow-backed Series + # reached RecordBatch.from_arrays as a ChunkedArray. + pd = pytest.importorskip("pandas") + rows = [pd.Series([3, 1, 2]).sort_values(), pd.Series([6, 5, 4]).sort_values()] + schema = pa.schema([("s", pa.list_(pa.int64()))]) + with pytest.raises(TypeError, match=r"'s'.*Series.*list\(row\)"): + run_outputs({"s": rows}, schema) + got = run_outputs({"s": [list(row) for row in rows]}, schema).column("s") + assert got.to_pylist() == [[1, 2, 3], [4, 5, 6]] + frame = pd.DataFrame({"x": [1, 2]}) + with pytest.raises(TypeError, match=r"'out'.*DataFrame"): + run_outputs({"out": frame[["x"]]}) + chunked = pd.concat([pd.Series(["a"], dtype="string[pyarrow]"), pd.Series(["b"], dtype="string[pyarrow]")]) + assert run_outputs({"w": chunked}).column("w").to_pylist() == ["a", "b"] + + +def test_a_key_error_from_pa_array_names_the_column_and_the_field(): + # pa.array reads a UserDict row by index, and the KeyError it raised was + # outside the classes the output side renamed, so it escaped as "0". + rows = [collections.UserDict({"amount": 1}), collections.UserDict({"amount": 2})] + schema = pa.schema([("s", pa.struct([("amount", pa.int64())]))]) + with pytest.raises(KeyError, match=r"output column 's': 0"): + run_outputs({"s": rows}, schema) + records = np.array([(1, rows[0])], dtype=[("i", "i8"), ("o", "O")]) + declared = pa.schema([("r", pa.struct([("i", pa.int64()), ("o", pa.struct([("amount", pa.int64())]))]))]) + with pytest.raises(KeyError, match=r"'r'.*field 'o': 0"): + run_outputs({"r": records}, declared) From eb2b64214b7e14fe2da111fa0cf310fd458be07e Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:56:29 +0000 Subject: [PATCH 09/59] Pair the field guard's kinds through their layouts: extension storage, a map against a declared list of entries, and the list view layouts The ready-built-array field guard compared like kinds only, so an extension array with struct storage cast to a declared struct, a MapArray cast to a declared list of key/value structs, and a list_view or large_list_view container declared for dict rows all came back all-null with no refusal. An extension type is compared and key-checked through its storage, a map is paired with a list of entries through its entries struct, and the view layouts count as list-like where the installed pyarrow has them. --- .github/scripts/mutation_guard_check.py | 19 ++++++++++++++ numbarrow/core/mapinarrow_factory.py | 35 +++++++++++++++++++++++-- test/test_mapinarrow_factory.py | 21 +++++++++++++++ 3 files changed, 73 insertions(+), 2 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 0aae4cd..665a855 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -505,6 +505,25 @@ ' if isinstance(array, pa.ChunkedArray):\n # A pandas Series over a multi-chunk', ' if False:\n # A pandas Series over a multi-chunk', ), + ( + 'the field guard stops seeing through an extension type', + 'numbarrow/core/mapinarrow_factory.py', + ' source_type = _storage(source_type)\n' + ' declared_type = _storage(declared_type)', + ' declared_type = _storage(declared_type)', + ), + ( + 'a map source stops being paired with a declared list of entries', + 'numbarrow/core/mapinarrow_factory.py', + ' if pa.types.is_map(source_type) and _is_list_like(declared_type):', + ' if False:', + ), + ( + 'view layouts stop counting as list-like', + 'numbarrow/core/mapinarrow_factory.py', + ' or pa.types.is_fixed_size_list(arrow_type) or _is_list_view(arrow_type))', + ' or pa.types.is_fixed_size_list(arrow_type))', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index e39a6ac..8e3da5c 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -38,12 +38,28 @@ def _struct_fields(struct_type): return [struct_type[i] for i in range(struct_type.num_fields)] +def _storage(arrow_type): + """The type under any extension wrapping: a cast and a key check work on the storage.""" + while isinstance(arrow_type, pa.BaseExtensionType): + arrow_type = arrow_type.storage_type + return arrow_type + + +def _is_list_view(arrow_type): + # The view layouts arrived in pyarrow 16; on an older one nothing is a view. + is_view = getattr(pa.types, "is_list_view", None) + is_large_view = getattr(pa.types, "is_large_list_view", None) + return bool(is_view and is_view(arrow_type)) or bool(is_large_view and is_large_view(arrow_type)) + + def _is_list_like(arrow_type): - return pa.types.is_list(arrow_type) or pa.types.is_large_list(arrow_type) or pa.types.is_fixed_size_list(arrow_type) + return (pa.types.is_list(arrow_type) or pa.types.is_large_list(arrow_type) + or pa.types.is_fixed_size_list(arrow_type) or _is_list_view(arrow_type)) def _carries_keys(arrow_type): """Whether a value of this type is built from dicts somewhere inside it.""" + arrow_type = _storage(arrow_type) if pa.types.is_struct(arrow_type): return True if _is_list_like(arrow_type): @@ -54,7 +70,21 @@ def _carries_keys(arrow_type): def _unexpected_fields(source_type, declared_type): - """Field names the source type carries, at any depth, that the declared type does not.""" + """Field names the source type carries, at any depth, that the declared type does not. + + The kinds are paired through their layouts: an extension type through its + storage, and a map with a declared list of key/value structs through its + entries struct, since a cast matches those by name too and filled the + value struct with nulls behind either wrapping. + """ + source_type = _storage(source_type) + declared_type = _storage(declared_type) + if pa.types.is_map(source_type) and _is_list_like(declared_type): + entries = pa.struct([source_type.key_field, source_type.item_field]) + return _unexpected_fields(entries, declared_type.value_type) + if _is_list_like(source_type) and pa.types.is_map(declared_type): + entries = pa.struct([declared_type.key_field, declared_type.item_field]) + return _unexpected_fields(source_type.value_type, entries) if pa.types.is_dictionary(source_type) and pa.types.is_dictionary(declared_type): # A dictionary is a layout: the cast decodes it and matches the value # structs by name, and filled a whole column with nulls the same way. @@ -121,6 +151,7 @@ def _check_keys(rows, arrow_type): declared order, and one that names the declared fields in another order, which would swap every same-typed field without a word, is refused. """ + arrow_type = _storage(arrow_type) if pa.types.is_struct(arrow_type): fields = {field.name: field.type for field in _struct_fields(arrow_type)} names = list(fields) diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index ff1a54e..377dd3e 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -979,3 +979,24 @@ def test_a_key_error_from_pa_array_names_the_column_and_the_field(): declared = pa.schema([("r", pa.struct([("i", pa.int64()), ("o", pa.struct([("amount", pa.int64())]))]))]) with pytest.raises(KeyError, match=r"'r'.*field 'o': 0"): run_outputs({"r": records}, declared) + + +def test_the_field_guard_sees_through_extension_map_and_view_layouts(): + # An extension array with struct storage, a map declared as a list of + # key/value structs, and a list view were compared as unlike kinds, so the + # guard returned nothing and the cast filled the column with nulls. + point = pa.struct([("x", pa.float64()), ("y", pa.float64())]) + if hasattr(pa, "opaque"): + ext_type = pa.opaque(point, "point", "vendor") + ext = pa.ExtensionArray.from_storage(ext_type, pa.array([{"x": 1.0, "y": 2.0}], type=point)) + declared = pa.schema([("p", pa.struct([("lon", pa.float64()), ("lat", pa.float64())]))]) + with pytest.raises(ValueError, match=r"'p'.*\['x', 'y'\]"): + run_outputs({"p": ext}, declared) + mapped = pa.array([[("k", {"Amount": 1})]], type=pa.map_(pa.string(), pa.struct([("Amount", pa.int64())]))) + entries = pa.struct([("key", pa.string()), ("value", pa.struct([("amount", pa.int64())]))]) + with pytest.raises(ValueError, match=r"'m'.*\['Amount'\]"): + run_outputs({"m": mapped}, pa.schema([("m", pa.list_(entries))])) + if hasattr(pa, "list_view"): + viewed = pa.schema([("v", pa.list_view(pa.struct([("amount", pa.int64())])))]) + with pytest.raises(ValueError, match=r"'v'.*'Amount'"): + run_outputs({"v": [[{"Amount": 5}]]}, viewed) From 0677dc49a3002b849f1249fc8d349f4d2d584195 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:56:30 +0000 Subject: [PATCH 10/59] Refuse a batch whose inferred schema differs from the partition's first, naming the column Without output_schema every list, tuple or object column's type was inferred from its own batch, so an empty or all-None batch beside a full one, or ints beside floats, carried a second schema into Spark's writer, which refused it with Tried to write record batch with different schema and named nothing. The first batch's schema is now held for the partition and a later batch that differs is refused naming the column, both types and the remedy; the string and bytes pins stand as they were. --- .github/scripts/mutation_guard_check.py | 6 +++++ numbarrow/core/mapinarrow_factory.py | 33 +++++++++++++++++++++++-- test/test_mapinarrow_factory.py | 22 +++++++++++++++++ 3 files changed, 59 insertions(+), 2 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 665a855..133f2e5 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -524,6 +524,12 @@ ' or pa.types.is_fixed_size_list(arrow_type) or _is_list_view(arrow_type))', ' or pa.types.is_fixed_size_list(arrow_type))', ), + ( + "a later batch's inferred schema stops being compared with the first's", + 'numbarrow/core/mapinarrow_factory.py', + ' elif built.schema != inferred:', + ' elif False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 8e3da5c..ae11402 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -578,6 +578,20 @@ def _build_batch(outputs, output_schema, handed=MappingProxyType({})): return pa.RecordBatch.from_arrays(arrays, schema=output_schema) +def _schema_drift(first, later): + """Why a batch built by inference differs from the partition's first, naming the column.""" + remedy = ("Spark's writer refuses a batch whose schema differs from the first it wrote, so return the " + "same columns every batch and declare output_schema where a batch may be empty or all null") + if first.names != later.names: + return f"this batch built columns {later.names} where the first built {first.names}; {remedy}" + for name in first.names: + before, now = first.field(name).type, later.field(name).type + if before != now: + return (f"output column {name!r} was inferred as {type_repr(before)} from the first batch and " + f"{type_repr(now)} from this one; {remedy}") + return f"this batch's schema differs from the first batch's; {remedy}" + + def _fold_struct_validity(struct_bitmap, field_bitmap): """Combine a struct's own validity bits into one field's bits. @@ -715,7 +729,12 @@ def make_mapinarrow_func( unit, with a multiplier such as ``datetime64[5s]`` folded in and an hour or minute unit taken to seconds; a day, week, month or year unit comes back ``date32``, a unit finer than a nanosecond is refused, and - an object array holding only ``None`` comes back ``null``. + an object array holding only ``None`` comes back ``null``. The + first batch's inferred schema is held for the partition, and a later + batch whose inferred types differ, an all-``None`` list beside one + holding values, or ints beside floats, is refused naming the column, + since Spark's writer would refuse it naming nothing; declare + ``output_schema`` where a batch may be empty or all null. """ broadcasts = broadcasts if broadcasts is not None else {} if isinstance(input_columns, str): @@ -737,6 +756,7 @@ def make_mapinarrow_func( ) def _(iterator): + inferred = None for batch in iterator: data_dict: dict[str, np.ndarray | dict[str, np.ndarray]] = {} bitmap_dict: dict[str, np.ndarray | None | dict[str, np.ndarray | None]] = {} @@ -776,5 +796,14 @@ def _(iterator): else: bitmap_dict[col], data_dict[col] = adapted handed = _handed_bitmaps(data_dict, bitmap_dict) - yield _build_batch(main_func(data_dict, bitmap_dict, broadcasts), output_schema, handed) + built = _build_batch(main_func(data_dict, bitmap_dict, broadcasts), output_schema, handed) + if output_schema is None: + # An all-None list beside one holding values, or ints beside + # floats, inferred a second schema, and Spark's writer refused + # it naming nothing. + if inferred is None: + inferred = built.schema + elif built.schema != inferred: + raise ValueError(_schema_drift(inferred, built.schema)) + yield built return _ diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 377dd3e..e56bc46 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -1000,3 +1000,25 @@ def test_the_field_guard_sees_through_extension_map_and_view_layouts(): viewed = pa.schema([("v", pa.list_view(pa.struct([("amount", pa.int64())])))]) with pytest.raises(ValueError, match=r"'v'.*'Amount'"): run_outputs({"v": [[{"Amount": 5}]]}, viewed) + + +def test_a_batch_whose_inferred_type_differs_from_the_first_is_refused_by_name(): + # Spark's writer refused the second schema it saw, naming nothing: an + # all-None list beside one holding strings, or ints beside floats. + batches = [pa.RecordBatch.from_pydict({"x": [1, 2]}), pa.RecordBatch.from_pydict({"x": [3]})] + values = iter([[None, None], ["a"]]) + fn = make_mapinarrow_func(lambda d, b, br: {"s": next(values)}) + with pytest.raises(ValueError, match=r"'s'.*null.*string.*output_schema"): + list(fn(iter(batches))) + values = iter([[1, 2], [1.5]]) + fn = make_mapinarrow_func(lambda d, b, br: {"n": next(values)}) + with pytest.raises(ValueError, match=r"'n'.*int64.*double"): + list(fn(iter(batches))) + values = iter([{"a": [1, 2]}, {"b": [3]}]) + fn = make_mapinarrow_func(lambda d, b, br: next(values)) + with pytest.raises(ValueError, match=r"\['b'\].*\['a'\]"): + list(fn(iter(batches))) + values = iter([[None, None], ["a"]]) + declared = pa.schema([("s", pa.string())]) + fn = make_mapinarrow_func(lambda d, b, br: {"s": next(values)}, output_schema=declared) + assert [batch.column("s").to_pylist() for batch in fn(iter(batches))] == [[None, None], ["a"]] From 313338dd9cffb6eb929dd901bf6fe4838f6bbbaf Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:56:38 +0000 Subject: [PATCH 11/59] Mask a Nullable extension column through its storage _with_validity chose its path by the extension type, which reports no fields and no dictionary whatever its storage, so a Nullable over dictionary storage took the from_buffers path the dictionary term exists to avoid and aborted the interpreter, and one carrying a null went to if_else, which has no extension kernel. The storage is masked and the result rewrapped. --- .github/scripts/mutation_guard_check.py | 8 ++++++++ numbarrow/core/mapinarrow_factory.py | 10 +++++++++- test/test_mapinarrow_factory.py | 15 +++++++++++++++ 3 files changed, 32 insertions(+), 1 deletion(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 133f2e5..28c34d9 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -530,6 +530,14 @@ ' elif built.schema != inferred:', ' elif False:', ), + ( + 'an extension column stops being masked through its storage', + 'numbarrow/core/mapinarrow_factory.py', + ' if isinstance(array, pa.ExtensionArray):\n' + ' # The flat test below', + ' if False:\n' + ' # The flat test below', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index ae11402..267b994 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -419,7 +419,8 @@ def _with_validity(array, bitmap): contiguous, which is the case for every bitmap ``bitmap_dict`` hands out. Any other array, one that already carries nulls, a sliced Arrow array or a nested type, is masked through ``if_else``, which keeps the nulls it - had. The bitmap carries no row count of its own: the length check here is + had; an extension array is masked through its storage and rewrapped. The + bitmap carries no row count of its own: the length check here is per byte, eight rows to a byte, and the caller checks a bitmap the batch handed out against the count it was handed out for. A bitmap that is not an ndarray at all is refused before any attribute of it is read, since the @@ -444,6 +445,13 @@ def _with_validity(array, bitmap): ) if rows == 0: return array + if isinstance(array, pa.ExtensionArray): + # The flat test below reads the extension type, which reports no + # fields and no dictionary whatever its storage, so a dictionary + # storage took the from_buffers path and aborted the interpreter, and + # a null or a slice went to if_else, which has no extension kernel. + # The storage carries the layout; the result is rewrapped. + return pa.ExtensionArray.from_storage(array.type, _with_validity(array.storage, bitmap)) flat = (array.null_count == 0 and array.offset == 0 and array.type.num_fields == 0 and not pa.types.is_dictionary(array.type) and not pa.types.is_null(array.type)) if flat: diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index e56bc46..8239565 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -1022,3 +1022,18 @@ def test_a_batch_whose_inferred_type_differs_from_the_first_is_refused_by_name() declared = pa.schema([("s", pa.string())]) fn = make_mapinarrow_func(lambda d, b, br: {"s": next(values)}, output_schema=declared) assert [batch.column("s").to_pylist() for batch in fn(iter(batches))] == [[None, None], ["a"]] + + +def test_a_nullable_extension_column_is_masked_through_its_storage(): + # The flat test read the extension type, which reports no dictionary + # whatever its storage, so a dictionary storage took the from_buffers path + # and aborted the interpreter, and one carrying a null went to if_else, + # which has no extension kernel. + if not hasattr(pa, "opaque"): + pytest.skip("pa.opaque arrived in pyarrow 17") + labels = pa.opaque(pa.dictionary(pa.int32(), pa.string()), "label", "vendor") + bitmap = np.array([0b101], dtype=np.uint8) + for storage in (pa.array(["a", "b", "c"]).dictionary_encode(), pa.array(["a", None, "c"]).dictionary_encode()): + column = pa.ExtensionArray.from_storage(labels, storage) + got = run_outputs({"out": Nullable(column, bitmap)}).column("out") + assert got.type == labels and got.storage.to_pylist() == ["a", None, "c"] From a50ca85424ff8d7f0da741c4c3926e2081992533 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Fri, 25 Sep 2026 23:59:40 +0000 Subject: [PATCH 12/59] Refuse a union child before a struct is flattened structured_array_adapter called flatten() on every child of a struct with a null row before any child was dispatched, and flatten() hands the struct's validity to each child; a union carries no validity buffer of its own, so Arrow's C++ layer failed its check and aborted the process where the documented NotImplementedError naming the field was due. A union child, extension storage included, is refused by name before anything is flattened. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/utils/arrow_array_utils.py | 18 ++++++++++++++++++ test/test_messages.py | 12 ++++++++++++ 3 files changed, 36 insertions(+) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 28c34d9..24ef509 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -538,6 +538,12 @@ ' if False:\n' ' # The flat test below', ), + ( + 'a union child stops being refused before flatten', + 'numbarrow/utils/arrow_array_utils.py', + ' if _is_union_layout(raw_child.type):', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/utils/arrow_array_utils.py b/numbarrow/utils/arrow_array_utils.py index 14d0eea..728224e 100644 --- a/numbarrow/utils/arrow_array_utils.py +++ b/numbarrow/utils/arrow_array_utils.py @@ -240,6 +240,12 @@ def create_str_array(pa_str_array: pa.StringArray | pa.LargeStringArray) -> tupl # design. See: https://awkward-array.org/doc/main/reference/generated/ak.contents.BitMaskedArray.html +def _is_union_layout(arrow_type): + while isinstance(arrow_type, pa.BaseExtensionType): + arrow_type = arrow_type.storage_type + return pa.types.is_union(arrow_type) + + def structured_array_adapter(struct_array: pa.StructArray) -> tuple[ np.ndarray | None, dict[str, np.ndarray | None], dict[str, np.ndarray] ]: @@ -291,6 +297,18 @@ def structured_array_adapter(struct_array: pa.StructArray) -> tuple[ # `is_null_struct` takes both, so folding the struct layer into the field # bitmap here would collapse a distinction the caller needs. raw_children = [struct_array.field(i) for i in range(len(data_type))] + for field_ind, raw_child in enumerate(raw_children): + if _is_union_layout(raw_child.type): + # flatten() hands the struct's validity to each child, and a union + # carries no validity buffer of its own, so Arrow's C++ layer + # aborted the process on one under a struct with a null row, where + # the dispatcher's typed refusal was due. Refused before anything + # is flattened, null row or not. + raise NotImplementedError( + f"struct field {data_type[field_ind].name!r}: Not implemented for an array of " + f"{len(raw_child)} elements of type {type_repr(raw_child.type)}, a union layout, which " + f"cannot take the struct's validity" + ) masked = list(struct_array.flatten()) if struct_array.null_count else raw_children for field_ind in range(len(data_type)): field: pa.Field = data_type[field_ind] diff --git a/test/test_messages.py b/test/test_messages.py index 5092421..a2d4311 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -189,3 +189,15 @@ def test_a_scalar_string_output_is_refused_rather_than_spread(): fn = make_mapinarrow_func(lambda d, b, br, value=value: {"country": value}) with pytest.raises(TypeError, match="'country'"): list(fn(iter([batch]))) + + +def test_a_union_field_under_a_struct_is_refused_before_flatten(): + # flatten() hands the struct's validity to each child, and a union carries + # none, so Arrow's C++ layer aborted the process under a struct with a null + # row where the typed refusal was due. + types = pa.array([0, 1, 0], type=pa.int8()) + union = pa.UnionArray.from_sparse(types, [pa.array([1, 2, 3]), pa.array(["a", "b", "c"])]) + for mask in (None, pa.array([False, True, False])): + struct = pa.StructArray.from_arrays([pa.array([1, 2, 3]), union], names=["ok", "u"], mask=mask) + with pytest.raises(NotImplementedError, match=r"struct field 'u'.*union"): + arrow_array_adapter(struct) From f816f67b1739a58bde27b78907a73c6c73f8e6e2 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:03:26 +0000 Subject: [PATCH 13/59] Give a zero-copy view a base that cannot be released from under it np.frombuffer handed a memoryview keeps only a wrapper of it as the result's base, and that wrapper's release() drops the memoryview's hold on the source, so a caller who released it, dropped the source array and read the view read freed memory, a segfault at 64 Ki elements, reachable through the public adapters and a UDF's data_dict. The read-only memoryview is now wrapped in a foreign pyarrow buffer, which is what the view keeps as its base: it has no release(), holds the source through the memoryview, and exports read-only, so the flip stays refused. The docs test's view detection accepts a pyarrow buffer at the end of the chain. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/utils/arrow_array_utils.py | 9 ++++++++- test/test_arrow_array_utils.py | 22 ++++++++++++++++++++++ test/test_docs_match_code.py | 2 +- 4 files changed, 37 insertions(+), 2 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 24ef509..e2aace3 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -544,6 +544,12 @@ ' if _is_union_layout(raw_child.type):', ' if False:', ), + ( + "a view's base stops being a buffer that cannot be released", + 'numbarrow/utils/arrow_array_utils.py', + ' pa.py_buffer(memoryview(data_buf).toreadonly()),', + ' memoryview(data_buf).toreadonly(),', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/utils/arrow_array_utils.py b/numbarrow/utils/arrow_array_utils.py index 728224e..5343c0d 100644 --- a/numbarrow/utils/arrow_array_utils.py +++ b/numbarrow/utils/arrow_array_utils.py @@ -496,8 +496,15 @@ def uniform_arrow_array_adapter(pa_array: pa.Array) -> tuple[np.ndarray | None, # changed the source Arrow array. numpy refuses to set WRITEABLE on an # array whose base is read-only, which is also what makes pyarrow's own # to_numpy(zero_copy_only=True) refuse the flip. + # The read-only memoryview is wrapped in a foreign pyarrow buffer, and that + # is what the result keeps as its base. Handed the memoryview itself, + # np.frombuffer kept only a wrapper of it as .base, and that wrapper's + # release() dropped the memoryview's hold on the source, so a caller who + # released it, dropped the array and read the view read freed memory. A + # pa.Buffer has no release(), holds the memoryview and through it the + # source buffer, and exports read-only, so the flip stays refused. data = np.frombuffer( - memoryview(data_buf).toreadonly(), + pa.py_buffer(memoryview(data_buf).toreadonly()), dtype=data_np_ty, count=data_len, offset=pa_array.offset * data_item_byte_size diff --git a/test/test_arrow_array_utils.py b/test/test_arrow_array_utils.py index 413fb0c..49a962d 100644 --- a/test/test_arrow_array_utils.py +++ b/test/test_arrow_array_utils.py @@ -751,3 +751,25 @@ def test_a_ragged_map_raises_as_the_readme_says(): type=pa.map_(pa.string(), pa.int64())) _, _, datas = arrow_array_adapter(uniform) assert datas["value"].tolist() == [1, 2, 3, 4] + + +def test_the_base_of_a_view_cannot_be_released_from_under_it(): + # np.frombuffer kept only a wrapper of the memoryview it was handed as + # .base, and a caller who released it, dropped the source and read the + # view read freed memory. + n = 1 << 16 + + def view(): + source = pa.array(np.arange(n) * 7 + 3) + return arrow_array_adapter(source)[1] + + data = view() + before = data.copy() + assert not hasattr(data.base, "release") + with pytest.raises(ValueError, match="WRITEABLE"): + data.flags.writeable = True + gc.collect() + junk = [pa.allocate_buffer(n * 8) for _ in range(6)] + for buffer in junk: + np.frombuffer(buffer, dtype=np.uint8)[:] = 0xA5 + assert np.array_equal(data, before) diff --git a/test/test_docs_match_code.py b/test/test_docs_match_code.py index 769b690..ea819e7 100644 --- a/test/test_docs_match_code.py +++ b/test/test_docs_match_code.py @@ -88,7 +88,7 @@ def _is_view(array): """True when the result views an Arrow buffer rather than owning fresh memory.""" node = array for _ in range(8): - if isinstance(node, memoryview): + if isinstance(node, (memoryview, pa.Buffer)): return True node = getattr(node, "base", None) if node is None: From c0f741128df7cdb815523b8723b7c6b80b2c2464 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:03:35 +0000 Subject: [PATCH 14/59] Leave a multi-dimensional array of another dtype to pa.array's own refusal The one-dimensional guard was meant for the unicode and bytes route alone, where tolist() turns a 0-d array into a scalar and a 2-d one into nested lists and so defeats the refusal pa.array gives the array itself; every other dtype gets that refusal, which names the column, as before. --- .github/scripts/mutation_guard_check.py | 2 +- numbarrow/core/mapinarrow_factory.py | 11 ++++++----- 2 files changed, 7 insertions(+), 6 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index e2aace3..99a8497 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -438,7 +438,7 @@ ( 'a 0-d or 2-d array output stops being refused', 'numbarrow/core/mapinarrow_factory.py', - ' if value.ndim != 1:', + ' if kind in ("U", "S") and value.ndim != 1:', ' if False:', ), ( diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 267b994..968f69d 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -366,14 +366,15 @@ def _at_arrow_unit(value): def _ndarray_to_arrow(value, arrow_type): """An ndarray of a non-object dtype as an Arrow array; see ``_convert``.""" - if value.ndim != 1: - # A 0-d unicode or bytes array's tolist() is a bare scalar, which - # pa.array spreads one character per row, defeating the refusal it - # gives the array itself; every other dtype it refuses on its own. - raise TypeError(f"a {value.ndim}-dimensional {value.dtype} array; an output column is one-dimensional") if value.dtype.names is not None: return _record_to_struct(value, arrow_type) kind = value.dtype.kind + if kind in ("U", "S") and value.ndim != 1: + # A 0-d unicode or bytes array's tolist() is a bare scalar, which + # pa.array spreads one character per row, and a 2-d one's is nested + # lists; both defeat the one-dimensional refusal pa.array gives the + # array itself, which every other dtype still gets, naming the column. + raise TypeError(f"a {value.ndim}-dimensional {value.dtype} array; an output column is one-dimensional") if kind == "U": return pa.array(value.tolist(), type=arrow_type or pa.string()) if kind == "S": From 42a909ae6dc0a32c3dc7925898437627aede2c8e Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:03:44 +0000 Subject: [PATCH 15/59] Hold nothing of a batch across the yield The generator kept the adapted arrays, the views, the struct triples and the handed-out bitmaps bound in its frame while the consumer wrote the batch out and the next one was adapted, so a string column's |U copy was live twice at the peak, the README's 1.6 GB batch peaking at 3.2 GB, and a handed-out bitmap outlived the batch its docstring says it lives for. Those names are cleared before the yield, and the yielded batch is dropped on resume. --- .github/scripts/mutation_guard_check.py | 6 +++++ numbarrow/core/mapinarrow_factory.py | 10 +++++++- test/test_mapinarrow_factory.py | 33 +++++++++++++++++++++++++ 3 files changed, 48 insertions(+), 1 deletion(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 99a8497..4dd0114 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -550,6 +550,12 @@ ' pa.py_buffer(memoryview(data_buf).toreadonly()),', ' memoryview(data_buf).toreadonly(),', ), + ( + "a batch's arrays stay bound across the yield", + 'numbarrow/core/mapinarrow_factory.py', + ' data_dict = bitmap_dict = handed = col_pa = adapted = None', + ' pass', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 968f69d..45d5cf7 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -527,7 +527,7 @@ def _handed_bitmaps(data_dict, bitmap_dict): field's own elements, and for a list of structs those are the flattened elements rather than the outer rows. The count therefore comes from the data handed out beside the bitmap, never from the batch. The bitmap rides - along to stay alive for the batch: an id is reusable once its object is + along to stay alive for the batch, and no longer: an id is reusable once its object is freed, and a UDF that drops a bitmap from ``bitmap_dict`` frees it, after which a bitmap of its own could land on that id and be refused as the handed-out one. @@ -814,5 +814,13 @@ def _(iterator): inferred = built.schema elif built.schema != inferred: raise ValueError(_schema_drift(inferred, built.schema)) + # Nothing of this batch is held across the yield: the adapted + # arrays, the views and the handed-out bitmaps stayed bound in the + # frame while the consumer wrote the batch out and the next one + # was adapted, so a string column's |U copy was live twice at the + # peak and a handed-out bitmap outlived its batch. + data_dict = bitmap_dict = handed = col_pa = adapted = None + struct_bitmap = field_bitmaps = field_datas = None yield built + built = None return _ diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 8239565..7a4a7f1 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -1,5 +1,6 @@ import collections import datetime +import gc import weakref import numpy as np @@ -1037,3 +1038,35 @@ def test_a_nullable_extension_column_is_masked_through_its_storage(): column = pa.ExtensionArray.from_storage(labels, storage) got = run_outputs({"out": Nullable(column, bitmap)}).column("out") assert got.type == labels and got.storage.to_pylist() == ["a", None, "c"] + + +def test_nothing_of_a_batch_is_held_while_the_next_one_is_read(): + # The adapted arrays and the handed-out bitmaps stayed bound in the + # generator's frame across the yield, so a string column's |U copy was + # live twice while the next batch was adapted. + seen = [] + + def main(data_dict, bitmap_dict, broadcasts): + seen.append(weakref.ref(data_dict["s"])) + return {"n": np.zeros(len(data_dict["s"]), dtype=np.int64)} + + alive_when_the_next_is_read = [] + + class Batches: + def __init__(self): + self.batches = [pa.RecordBatch.from_pydict({"s": ["a", "b"]}), pa.RecordBatch.from_pydict({"s": ["c"]})] + + def __iter__(self): + return self + + def __next__(self): + if seen: + gc.collect() + alive_when_the_next_is_read.append(seen[-1]() is not None) + if not self.batches: + raise StopIteration + return self.batches.pop(0) + + for _ in make_mapinarrow_func(main)(Batches()): + pass + assert alive_when_the_next_is_read == [False, False] From 7eab1b322eadfe31157f9b4ee661613f0f62aba7 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:03:53 +0000 Subject: [PATCH 16/59] Give structured dtypes of one itemsize viewers of their own numpy names every structured dtype of one itemsize void, so two of them compiled under one qualname, landed in one cache index with identical argtypes, and a process that loaded both from the cache ran the first one's code for the second, boxing an int32 field's bytes as float32. A digest of the dtype's description joins the qualname of a structured dtype. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/utils/utils.py | 11 ++++++++++- test/test_utils.py | 15 ++++++++++++++- 3 files changed, 30 insertions(+), 2 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 4dd0114..d90c23a 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -556,6 +556,12 @@ ' data_dict = bitmap_dict = handed = col_pa = adapted = None', ' pass', ), + ( + 'structured dtypes stop getting viewers of their own', + 'numbarrow/utils/utils.py', + ' if dtype_.fields is not None:', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/utils/utils.py b/numbarrow/utils/utils.py index 66ac256..2aa5e00 100644 --- a/numbarrow/utils/utils.py +++ b/numbarrow/utils/utils.py @@ -6,6 +6,8 @@ ``@njit`` code to read Arrow buffer data directly without copying. """ +import hashlib + import numpy as np from numba import carray, from_dtype, int64, intp, njit from numba.core.types import Array, voidptr @@ -57,7 +59,14 @@ def viewer(ptr_as_int: int, sz: int): # NRT_adapt_ndarray_to_python. A qualname per dtype gives each viewer its # own index and data files, and one entry per index leaves nothing for two # writers to disagree about. - name = f"view_{np.dtype(dtype_).name}" + dtype_ = np.dtype(dtype_) + name = f"view_{dtype_.name}" + if dtype_.fields is not None: + # numpy names every structured dtype of one itemsize void, so + # two of them shared one index, and a process loading both from the + # cache ran the first one's code for the second; the description + # tells them apart. + name += "_" + hashlib.sha1(repr(dtype_.descr).encode()).hexdigest()[:12] viewer.__name__ = name viewer.__qualname__ = f"{numpy_array_from_ptr_factory.__qualname__}..{name}" return njit(Array(from_dtype(dtype_), 1, "C")(intp, int64), **jit_options)(viewer) diff --git a/test/test_utils.py b/test/test_utils.py index 58c8eba..32ce8e7 100644 --- a/test/test_utils.py +++ b/test/test_utils.py @@ -1,6 +1,6 @@ import numpy as np from numpy.testing import assert_equal -from numbarrow.utils.utils import arrays_viewers +from numbarrow.utils.utils import arrays_viewers, numpy_array_from_ptr_factory def test_int32_array_from_ptr_as_int(): @@ -21,3 +21,16 @@ def test_a_viewer_is_built_when_first_asked_for_and_kept(): if __name__ == "__main__": test_int32_array_from_ptr_as_int() + + +def test_structured_dtypes_of_one_itemsize_get_viewers_of_their_own(): + # numpy names both void32, so the two viewers shared one cache index and a + # process loading both from the cache ran the first one's code for the + # second. + first = numpy_array_from_ptr_factory(np.dtype([("a", " Date: Sat, 26 Sep 2026 00:04:02 +0000 Subject: [PATCH 17/59] Keep a zero-length temporal column's bitmap presence The date32, date64 and timestamp handlers took the bitmap from the result of an integer cast, and pyarrow drops the validity buffer when it casts a zero-length array, so a zero-row column that still carried one adapted to a bitmap of None, against the documented rule that None means no buffer, where every other type gives an empty uint8 array. The bitmap comes from the source array when the column is empty. --- .github/scripts/mutation_guard_check.py | 8 ++++++++ numbarrow/core/adapters.py | 8 ++++++++ test/test_adapters.py | 12 ++++++++++++ 3 files changed, 28 insertions(+) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index d90c23a..f38e534 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -562,6 +562,14 @@ ' if dtype_.fields is not None:', ' if False:', ), + ( + 'a zero-length temporal column stops keeping its bitmap presence', + 'numbarrow/core/adapters.py', + ' if not len(pa_array):\n' + ' # The cast of a zero-length array drops its validity buffer, and a', + ' if False:\n' + ' # The cast of a zero-length array drops its validity buffer, and a', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/adapters.py b/numbarrow/core/adapters.py index fafa2c2..1f9c830 100644 --- a/numbarrow/core/adapters.py +++ b/numbarrow/core/adapters.py @@ -38,6 +38,10 @@ def cast_64bit_date_arrow_to_numpy_array(pa_array: pa.Array, np_dtype: np.dtype) if len(pa_array): assert int64_array.buffers()[1].address == pa_array.buffers()[1].address, "got copied" bitmap, int64_data = uniform_arrow_array_adapter(int64_array) + if not len(pa_array): + # The cast of a zero-length array drops its validity buffer, and a + # bitmap is None only when the source carries none. + bitmap = create_bitmap(pa_array.buffers()[0], pa_array.offset, 0) data = int64_data.view(np_dtype) assert data.ctypes.data == int64_data.ctypes.data, "got copied" return bitmap, data @@ -123,6 +127,10 @@ def _(pa_array: pa.Date32Array): if len(pa_array): assert int32_array.buffers()[1].address == pa_array.buffers()[1].address, "got copied" bitmap, int32_data = uniform_arrow_array_adapter(int32_array) + if not len(pa_array): + # As in cast_64bit_date_arrow_to_numpy_array: the zero-length cast + # dropped the validity buffer the source still carries. + bitmap = create_bitmap(pa_array.buffers()[0], pa_array.offset, 0) data = int32_data.astype(np.dtype("datetime64[D]")) if len(pa_array): assert int32_data.ctypes.data != data.ctypes.data diff --git a/test/test_adapters.py b/test/test_adapters.py index 7bdd0ae..2b3fd7e 100644 --- a/test/test_adapters.py +++ b/test/test_adapters.py @@ -157,3 +157,15 @@ def test_zero_length_date_and_timestamp(): test_arrow_array_adapter_3() test_arrow_array_adapter_4() test_empty_str_array() + + +def test_a_zero_length_temporal_column_keeps_its_bitmap_presence(): + # The temporal handlers took the bitmap from the cast, and a zero-length + # cast drops the validity buffer, so the documented None-only-without-a- + # buffer rule broke for date32, date64 and timestamp columns. + for arrow_type, source_type in ((pa.date32(), pa.int32()), (pa.date64(), pa.int64()), + (pa.timestamp("us", "UTC"), pa.int64())): + source = pa.array([1, None], type=source_type).cast(arrow_type).slice(1, 0) + assert source.buffers()[0] is not None + bitmap, data = arrow_array_adapter(source) + assert bitmap is not None and bitmap.dtype == np.uint8 and len(bitmap) == 0 and len(data) == 0 From 1cdb5fccc63f70ae366422c3e773698956ccb64d Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:08:23 +0000 Subject: [PATCH 18/59] Cut the unexpected-keys listing with a count The struct key check listed every distinct key no declared field has, and that list is sized by the data rather than the schema: a UDF keying a dict by a row value put every key of a 100,000-row batch, 1.5 MB, into the exception and twice into the executor logs. Ten are shown with a count of the rest, as a wide type is cut. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 11 ++++++++++- test/test_messages.py | 12 ++++++++++++ 3 files changed, 28 insertions(+), 1 deletion(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index f38e534..8d34f08 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -570,6 +570,12 @@ ' if False:\n' ' # The cast of a zero-length array drops its validity buffer, and a', ), + ( + 'the unexpected-keys listing stops being cut', + 'numbarrow/core/mapinarrow_factory.py', + ' shown = unexpected_keys[:KEYS_SHOWN]', + ' shown = unexpected_keys', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 45d5cf7..2dd6ff9 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -34,6 +34,13 @@ class Nullable(NamedTuple): bitmap: np.ndarray | None +# How many of the keys no declared field has a refusal lists. The listing is +# sized by the data, not the schema: a UDF keying a dict by a row value put +# every key of a 100,000-row batch, 1.5 MB, into the exception and twice into +# the executor logs. +KEYS_SHOWN = 10 + + def _struct_fields(struct_type): return [struct_type[i] for i in range(struct_type.num_fields)] @@ -162,8 +169,10 @@ def _check_keys(rows, arrow_type): seen.update(row) unexpected_keys = sorted(str(key) for key in seen - set(fields)) if unexpected_keys: + shown = unexpected_keys[:KEYS_SHOWN] + more = f" and {len(unexpected_keys) - KEYS_SHOWN} more" if len(unexpected_keys) > KEYS_SHOWN else "" raise ValueError( - f"declared {type_repr(arrow_type)} but the dicts carry keys {unexpected_keys} " + f"declared {type_repr(arrow_type)} but the dicts carry keys {shown}{more} " f"that no declared field has; Arrow matches struct fields by exact name and " f"fills a missing one with null" ) diff --git a/test/test_messages.py b/test/test_messages.py index a2d4311..143c356 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -201,3 +201,15 @@ def test_a_union_field_under_a_struct_is_refused_before_flatten(): struct = pa.StructArray.from_arrays([pa.array([1, 2, 3]), union], names=["ok", "u"], mask=mask) with pytest.raises(NotImplementedError, match=r"struct field 'u'.*union"): arrow_array_adapter(struct) + + +def test_the_unexpected_keys_listing_is_cut_with_a_count(): + # A UDF keying a dict by a row value put every key of the batch into the + # exception, 1.5 MB for 100,000 rows, and twice into the executor logs. + rows = [{f"user_{i:06d}": 1} for i in range(1000)] + fn = make_mapinarrow_func(lambda d, b, br: {"counts": rows}, + output_schema=pa.schema([("counts", pa.struct([("total", pa.int64())]))])) + with pytest.raises(ValueError) as excinfo: + list(fn(iter([_batch(v=list(range(1000)))]))) + message = str(excinfo.value) + assert "'user_000000'" in message and "and 990 more" in message and len(message) < 600 From 6985ccebd85d72af2f99c4037b05e4c6b7537009 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:08:32 +0000 Subject: [PATCH 19/59] Describe a scalar as one at the dispatcher, and leak no column called type The dispatcher's fallback read len() and .type of whatever it was handed: a null list scalar died on len(), a struct or map scalar of a supported column was described as an unsupported array of that type, and a pandas frame or record array with a column called type put that column's values into the message or died in type_repr. A scalar is named as one, and only a DataType reaches the message. --- .github/scripts/mutation_guard_check.py | 12 ++++++++++++ numbarrow/core/adapters.py | 12 +++++++++++- test/test_messages.py | 14 ++++++++++++++ 3 files changed, 37 insertions(+), 1 deletion(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 8d34f08..b47cb4b 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -576,6 +576,18 @@ ' shown = unexpected_keys[:KEYS_SHOWN]', ' shown = unexpected_keys', ), + ( + 'a scalar stops being described as one at the dispatcher', + 'numbarrow/core/adapters.py', + ' if isinstance(pa_array, pa.Scalar):', + ' if False:', + ), + ( + 'a type attribute that is not a DataType stops being screened', + 'numbarrow/core/adapters.py', + ' if not isinstance(arrow_type, pa.DataType):', + ' if arrow_type is None:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/adapters.py b/numbarrow/core/adapters.py index 1f9c830..8b25677 100644 --- a/numbarrow/core/adapters.py +++ b/numbarrow/core/adapters.py @@ -69,8 +69,18 @@ def arrow_array_adapter(pa_array: pa.Array): f"Not implemented for a ChunkedArray of {pa_array.num_chunks} chunks of type " f"{type_repr(pa_array.type)}: pass one chunk, or combine_chunks() first" ) + if isinstance(pa_array, pa.Scalar): + # One row of a column, not the column: a null list scalar has no + # length to read, and a struct or map scalar of a supported column + # was described as an unsupported array of that type. + raise NotImplementedError( + f"Not implemented for a {type(pa_array).__name__} of type {type_repr(pa_array.type)}: " + f"pass the Array, not one of its rows" + ) arrow_type = getattr(pa_array, "type", None) - if arrow_type is None: + if not isinstance(arrow_type, pa.DataType): + # A pandas frame or a record array with a column called type answers + # the attribute with that column, which then went into the message. described = f"{type(pa_array).__name__}, which is not a pyarrow Array" elif hasattr(pa_array, "__len__"): described = f"an array of {len(pa_array)} elements of type {type_repr(arrow_type)}" diff --git a/test/test_messages.py b/test/test_messages.py index 143c356..584ff03 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -213,3 +213,17 @@ def test_the_unexpected_keys_listing_is_cut_with_a_count(): list(fn(iter([_batch(v=list(range(1000)))]))) message = str(excinfo.value) assert "'user_000000'" in message and "and 990 more" in message and len(message) < 600 + + +def test_the_dispatcher_describes_a_scalar_and_leaks_no_type_column(): + # A null list scalar died on len(), a struct scalar of a supported column + # was described as an unsupported array of that type, and a frame with a + # column called type put that column's values into the message. + for scalar in (pa.array([None], type=pa.list_(pa.int64()))[0], pa.array([{"a": 1}])[0]): + with pytest.raises(NotImplementedError, match=r"Scalar of type .*: pass the Array"): + arrow_array_adapter(scalar) + pd = pytest.importorskip("pandas") + frame = pd.DataFrame({"type": [f"secret-{i}" for i in range(50)], "v": range(50)}) + with pytest.raises(NotImplementedError, match="DataFrame, which is not a pyarrow Array") as excinfo: + arrow_array_adapter(frame) + assert "secret" not in str(excinfo.value) From 33210509fa1300ca2233f096f05b9fb69893836c Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:08:41 +0000 Subject: [PATCH 20/59] Refuse an output_schema that names a column or field twice A dict holds one value per name, so a name declared twice at any depth was filled twice from the same entry, and the batch died in the JVM with not all nodes and buffers were consumed, naming neither the column nor the repeat; the input side refuses the same shape by name. The schema is checked once, when the function is made. --- .github/scripts/mutation_guard_check.py | 8 ++++++++ numbarrow/core/mapinarrow_factory.py | 24 ++++++++++++++++++++++++ test/test_messages.py | 11 +++++++++++ 3 files changed, 43 insertions(+) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index b47cb4b..8301c78 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -588,6 +588,14 @@ ' if not isinstance(arrow_type, pa.DataType):', ' if arrow_type is None:', ), + ( + 'a repeated name in output_schema stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' repeated = _repeated_names(list(output_schema))\n' + ' if repeated:', + ' repeated = _repeated_names(list(output_schema))\n' + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 2dd6ff9..2c150b7 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -596,6 +596,21 @@ def _build_batch(outputs, output_schema, handed=MappingProxyType({})): return pa.RecordBatch.from_arrays(arrays, schema=output_schema) +def _repeated_names(fields): + """Names declared more than once among *fields* or inside any of their types, at any depth.""" + names = [field.name for field in fields] + repeated = sorted({name for name in names if names.count(name) > 1}) + for field in fields: + arrow_type = _storage(field.type) + if pa.types.is_struct(arrow_type): + repeated.extend(_repeated_names(_struct_fields(arrow_type))) + elif _is_list_like(arrow_type): + repeated.extend(_repeated_names([arrow_type.value_field])) + elif pa.types.is_map(arrow_type): + repeated.extend(_repeated_names([arrow_type.key_field, arrow_type.item_field])) + return repeated + + def _schema_drift(first, later): """Why a batch built by inference differs from the partition's first, naming the column.""" remedy = ("Spark's writer refuses a batch whose schema differs from the first it wrote, so return the " @@ -772,6 +787,15 @@ def make_mapinarrow_func( raise TypeError( f"output_schema must be a pyarrow.Schema, not a {type(output_schema).__name__}" ) + if output_schema is not None: + repeated = _repeated_names(list(output_schema)) + if repeated: + # A dict holds one value per name, so every copy was filled from it + # and Spark died in the JVM naming neither the column nor the copy. + raise ValueError( + f"output_schema names {repeated} more than once; the dict main_func returns holds one value " + f"per name, so alias one of them" + ) def _(iterator): inferred = None diff --git a/test/test_messages.py b/test/test_messages.py index 584ff03..3b69862 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -227,3 +227,14 @@ def test_the_dispatcher_describes_a_scalar_and_leaks_no_type_column(): with pytest.raises(NotImplementedError, match="DataFrame, which is not a pyarrow Array") as excinfo: arrow_array_adapter(frame) assert "secret" not in str(excinfo.value) + + +def test_an_output_schema_naming_a_field_twice_is_refused_at_factory_time(): + # One dict entry filled every copy, and Spark died in the JVM with "not + # all nodes and buffers were consumed", naming neither the column nor the + # repeated name. + with pytest.raises(ValueError, match=r"\['price'\] more than once"): + make_mapinarrow_func(lambda d, b, br: {}, output_schema=pa.schema([("price", pa.float64()), ("price", pa.int32())])) + nested = pa.schema([("s", pa.list_(pa.struct([("x", pa.int64()), ("x", pa.float64())])))]) + with pytest.raises(ValueError, match=r"\['x'\] more than once"): + make_mapinarrow_func(lambda d, b, br: {}, output_schema=nested) From 180be7efb6d764c0dd6c6bb14ad0a5ef4c895a01 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:08:50 +0000 Subject: [PATCH 21/59] Name the shape the function takes when it is handed a batch or a table The generator read .schema of whatever its iterator yielded, so udf(batch) and udf(table) walked the columns and died on an attribute of the first one, and mapInPandas's pandas frames died the same way. An item that is not a RecordBatch is refused naming the shape mapInArrow passes. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 8 ++++++++ test/test_messages.py | 11 +++++++++++ 3 files changed, 25 insertions(+) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 8301c78..1de3ad3 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -596,6 +596,12 @@ ' repeated = _repeated_names(list(output_schema))\n' ' if False:', ), + ( + 'an item that is not a RecordBatch stops being refused by name', + 'numbarrow/core/mapinarrow_factory.py', + ' if not isinstance(batch, pa.RecordBatch):', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 2c150b7..f3f6ef7 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -800,6 +800,14 @@ def make_mapinarrow_func( def _(iterator): inferred = None for batch in iterator: + if not isinstance(batch, pa.RecordBatch): + # Handed a RecordBatch or a Table instead of an iterator of + # them, the loop walked the columns and died on an attribute + # of the first one; mapInPandas hands over pandas frames. + raise TypeError( + f"pass an iterator of pyarrow.RecordBatch, as mapInArrow does, such as [batch] or " + f"table.to_batches(), not one yielding a {type(batch).__name__}" + ) data_dict: dict[str, np.ndarray | dict[str, np.ndarray]] = {} bitmap_dict: dict[str, np.ndarray | None | dict[str, np.ndarray | None]] = {} names = batch.schema.names diff --git a/test/test_messages.py b/test/test_messages.py index 3b69862..aba873f 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -238,3 +238,14 @@ def test_an_output_schema_naming_a_field_twice_is_refused_at_factory_time(): nested = pa.schema([("s", pa.list_(pa.struct([("x", pa.int64()), ("x", pa.float64())])))]) with pytest.raises(ValueError, match=r"\['x'\] more than once"): make_mapinarrow_func(lambda d, b, br: {}, output_schema=nested) + + +def test_the_function_names_the_shape_it_takes_when_handed_a_batch_or_a_table(): + # udf(batch) and udf(table) walked the columns and died on an attribute of + # the first one, and mapInPandas's frames died the same way. + fn = make_mapinarrow_func(lambda d, b, br: {"out": d["a"]}) + batch = pa.record_batch({"a": pa.array([1, 2, 3])}) + for handed in (batch, pa.Table.from_batches([batch]), [batch.to_pandas()]): + with pytest.raises(TypeError, match="iterator of pyarrow.RecordBatch"): + list(fn(handed)) + assert list(fn([batch]))[0].column("out").to_pylist() == [1, 2, 3] From 06ab92e2289fd22d4391df135dbf55e2c8dcf4b1 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:08:59 +0000 Subject: [PATCH 22/59] Say which rule NUMBARROW_JIT_OPTIONS broke, and show the value One message served both failures, so a value that was valid JSON but a list, a number, null, true or a string was told it must be valid JSON, and neither failure showed the value; the sibling refusals in the same function name the option and repr the value. Both messages now name the object requirement and the value, and the JSON error names its position. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/configurations.py | 14 ++++++++++---- test/test_configurations.py | 11 +++++++++++ 3 files changed, 27 insertions(+), 4 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 1de3ad3..496a149 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -602,6 +602,12 @@ ' if not isinstance(batch, pa.RecordBatch):', ' if False:', ), + ( + 'the options refusal stops showing the value', + 'numbarrow/core/configurations.py', + ' f"{invalid_jit_options_err}; {as_str!r} is valid JSON but a {type(as_json).__name__}, not an object"', + ' invalid_jit_options_err', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/configurations.py b/numbarrow/core/configurations.py index 65decdc..9ccd5b9 100644 --- a/numbarrow/core/configurations.py +++ b/numbarrow/core/configurations.py @@ -6,7 +6,9 @@ import json -invalid_jit_options_err = """Must be valid JSON, e.g., export NUMBARROW_JIT_OPTIONS='{"cache": false}'""" +invalid_jit_options_err = ( + """NUMBARROW_JIT_OPTIONS must be a JSON object, e.g., export NUMBARROW_JIT_OPTIONS='{"cache": false}'""" +) def get_jit_options(): @@ -32,10 +34,14 @@ def get_jit_options(): return {"cache": True} try: as_json = json.loads(as_str) - except json.JSONDecodeError: - raise ValueError(invalid_jit_options_err) + except json.JSONDecodeError as error: + raise ValueError(f"{invalid_jit_options_err}; {as_str!r} is not valid JSON: {error}") from None if not isinstance(as_json, dict): - raise ValueError(invalid_jit_options_err) + # One message for both failures told a value that was valid JSON that + # it must be valid JSON, and showed neither the value nor the rule. + raise ValueError( + f"{invalid_jit_options_err}; {as_str!r} is valid JSON but a {type(as_json).__name__}, not an object" + ) if "cache" in as_json and not isinstance(as_json["cache"], bool): raise ValueError( f'NUMBARROW_JIT_OPTIONS "cache" must be true or false, not {as_json["cache"]!r}: numba reads any ' diff --git a/test/test_configurations.py b/test/test_configurations.py index daa2dc1..38c94d2 100644 --- a/test/test_configurations.py +++ b/test/test_configurations.py @@ -70,3 +70,14 @@ def test_importing_with_an_empty_value_uses_the_default(): out = subprocess.run([sys.executable, "-c", src], capture_output=True, text=True, env=env) assert out.returncode == 0, out.stderr assert json.loads(out.stdout) == {"cache": True} + + +def test_the_refusal_names_the_requirement_and_shows_the_value(monkeypatch): + # One message for both failures told a value that was valid JSON that it + # must be valid JSON, and showed neither the value nor the rule. + monkeypatch.setenv("NUMBARROW_JIT_OPTIONS", "[1, 2]") + with pytest.raises(ValueError, match=r"JSON object.*'\[1, 2\]' is valid JSON but a list"): + get_jit_options() + monkeypatch.setenv("NUMBARROW_JIT_OPTIONS", "{cache: false}") + with pytest.raises(ValueError, match=r"'\{cache: false\}' is not valid JSON"): + get_jit_options() From 0f41f1388b36c2fb8dd09259c13fd3dc8c690dd9 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:09:08 +0000 Subject: [PATCH 23/59] Refuse a unit that is not the array's own in the 64-bit date view, and say what the function does cast_64bit_date_arrow_to_numpy_array's docstring offered a cast to various units and its body reinterpreted the int64 payload, so timestamp[ms] asked for as datetime64[s] came back in the year 51971 for a caller who followed the docs page. The unit must be the array's own now, a mismatch is refused naming both, and the docstring says it is a view. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/adapters.py | 24 ++++++++++++++++-------- test/test_adapters.py | 13 +++++++++++++ 3 files changed, 35 insertions(+), 8 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 496a149..620a61e 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -608,6 +608,12 @@ ' f"{invalid_jit_options_err}; {as_str!r} is valid JSON but a {type(as_json).__name__}, not an object"', ' invalid_jit_options_err', ), + ( + 'the 64-bit date view stops refusing another unit', + 'numbarrow/core/adapters.py', + ' if np_dtype != np.dtype(f"datetime64[{unit}]"):', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/adapters.py b/numbarrow/core/adapters.py index 8b25677..d61d437 100644 --- a/numbarrow/core/adapters.py +++ b/numbarrow/core/adapters.py @@ -23,15 +23,23 @@ def cast_64bit_date_arrow_to_numpy_array(pa_array: pa.Array, np_dtype: np.dtype): - """ Can be used to cast PyArrow arrays of date types that are represented by - 64-bit integers to numpy arrays of various date types (np.datetime64[...], - which are always represented by 64-bit integers whose meaning is determined - by the precision, such as, 's', 'ms', 'us'). - - Since underlying data layout of both arrays in int64, a copy is avoided, - - The associated bitmap (if any) is also returned. + """View a date64 or timestamp array as the ``datetime64`` of the same unit, without a copy. + + Both hold one int64 per value, so the Arrow buffer is viewed rather than + converted, and *np_dtype* must be ``datetime64`` at the array's own unit: + ``ms`` for a date64, and a timestamp's own unit. Any other unit is + refused, since a view cannot rescale and read every value at the wrong + instant, ``timestamp[ms]`` viewed as ``datetime64[s]`` landing in the + year 51971. The associated bitmap (if any) is also returned. """ + np_dtype = np.dtype(np_dtype) + unit = "ms" if pa.types.is_date64(pa_array.type) else getattr(pa_array.type, "unit", None) + if unit is None: + raise ValueError(f"{type_repr(pa_array.type)} is not a date64 or timestamp type") + if np_dtype != np.dtype(f"datetime64[{unit}]"): + raise ValueError( + f"{type_repr(pa_array.type)} holds {unit} instants, which view as datetime64[{unit}], not as {np_dtype}" + ) int64_array = pa_array.cast(pa.int64()) # A zero-length array's buffers are not required to survive a cast, and # nothing has been copied when there is nothing to copy. diff --git a/test/test_adapters.py b/test/test_adapters.py index 2b3fd7e..f9e3b07 100644 --- a/test/test_adapters.py +++ b/test/test_adapters.py @@ -169,3 +169,16 @@ def test_a_zero_length_temporal_column_keeps_its_bitmap_presence(): assert source.buffers()[0] is not None bitmap, data = arrow_array_adapter(source) assert bitmap is not None and bitmap.dtype == np.uint8 and len(bitmap) == 0 and len(data) == 0 + + +def test_the_64bit_date_view_refuses_a_unit_that_is_not_the_arrays_own(): + # The docstring promised a cast to any unit and the body reinterpreted the + # int64 payload, so timestamp[ms] viewed as datetime64[s] landed in 51971. + from numbarrow.core.adapters import cast_64bit_date_arrow_to_numpy_array + stamps = pa.array([datetime(2020, 1, 1, 12)], type=pa.timestamp("ms")) + _, data = cast_64bit_date_arrow_to_numpy_array(stamps, np.dtype("datetime64[ms]")) + assert data.tolist() == [datetime(2020, 1, 1, 12)] + with pytest.raises(ValueError, match=r"ms instants.*not as datetime64\[s\]"): + cast_64bit_date_arrow_to_numpy_array(stamps, np.dtype("datetime64[s]")) + with pytest.raises(ValueError, match="not a date64 or timestamp"): + cast_64bit_date_arrow_to_numpy_array(pa.array([1], type=pa.int64()), np.dtype("datetime64[s]")) From 104e34a346e42cfb10b29da48f0ae045f327b00a Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:09:18 +0000 Subject: [PATCH 24/59] Hand a fixed-width bytes column to pa.array directly under a fixed-size binary type tolist() drops a trailing NUL, so one digest in 256 came back a byte short and the whole batch was refused under fixed_size_binary(16), and the docstring blamed numpy for a byte the buffer still held. Under a fixed-size binary type pa.array keeps every byte of the array itself, and only there; variable-width binary still cuts at the first NUL and keeps the tolist() route. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 13 +++++++++++-- test/test_mapinarrow_factory.py | 11 +++++++++++ 3 files changed, 28 insertions(+), 2 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 620a61e..015c84a 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -614,6 +614,12 @@ ' if np_dtype != np.dtype(f"datetime64[{unit}]"):', ' if False:', ), + ( + 'a fixed-width binary column stops going to pa.array directly', + 'numbarrow/core/mapinarrow_factory.py', + ' if arrow_type is not None and pa.types.is_fixed_size_binary(arrow_type):', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index f3f6ef7..7b3e8a3 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -288,8 +288,10 @@ def _convert(value, arrow_type): ``"a\x00b"`` arrives as ``"a"`` and a leading NUL empties the value outright. Going via ``tolist()`` hands Arrow real Python strings and bytes, which carry NULs, at about 18% more time on a 200k-row column. A - trailing NUL is already gone before this point, dropped by numpy when the - array was built, which matches the adapter refusing one on the way in. + trailing NUL is dropped by ``tolist()`` itself, as numpy's own element + access drops it, which matches the adapter refusing one on the way in; + under a declared fixed-size binary type the array goes to ``pa.array`` + directly, which keeps every byte there. Without a declared type a unicode or bytes column is still named ``string`` or ``binary`` rather than inferred, because ``pa.array([])`` @@ -387,6 +389,13 @@ def _ndarray_to_arrow(value, arrow_type): if kind == "U": return pa.array(value.tolist(), type=arrow_type or pa.string()) if kind == "S": + if arrow_type is not None and pa.types.is_fixed_size_binary(arrow_type): + # tolist() drops a trailing NUL, so a digest ending in 0x00 came + # back a byte short and was refused under its fixed width. + # pa.array keeps every byte of a fixed-width array under a + # fixed-width type, and only there; variable-width binary still + # cuts at the first NUL, so it keeps the tolist() route. + return pa.array(value, type=arrow_type) return pa.array(value.tolist(), type=arrow_type or pa.binary()) if kind in ("M", "m"): value = _at_arrow_unit(value) diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 7a4a7f1..75418e9 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -1070,3 +1070,14 @@ def __next__(self): for _ in make_mapinarrow_func(main)(Batches()): pass assert alive_when_the_next_is_read == [False, False] + + +def test_a_fixed_width_binary_column_keeps_a_trailing_nul_under_its_declared_type(): + # tolist() drops a trailing NUL, so one digest in 256 came back a byte + # short and the batch was refused under fixed_size_binary(16); the + # docstring blamed numpy for a byte the buffer still held. + digests = np.array([b"0123456789abcde\x00", b"\x00fedcba987654321", b"0123456789abcdef"], dtype="|S16") + declared = pa.schema([("d", pa.binary(16))]) + got = run_outputs({"d": digests}, declared).column("d") + assert got.type == pa.binary(16) + assert got.to_pylist() == [b"0123456789abcde\x00", b"\x00fedcba987654321", b"0123456789abcdef"] From 0e6b38f5501550d0b0da3ff0f297643b48194bbf Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:10:22 +0000 Subject: [PATCH 25/59] Name the path to the field a nested key refusal is about Two same-typed sibling fields, a map's key and its value, and different nesting depths all raised a byte-identical message naming the column and the declared struct alone. The check carries the path through its recursion, so the refusal reads output column 's': field 'm': map value: declared ... --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 18 +++++++++--------- test/test_messages.py | 15 +++++++++++++++ 3 files changed, 30 insertions(+), 9 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 015c84a..a106df1 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -620,6 +620,12 @@ ' if arrow_type is not None and pa.types.is_fixed_size_binary(arrow_type):', ' if False:', ), + ( + 'a nested key refusal stops naming the field path', + 'numbarrow/core/mapinarrow_factory.py', + ' _check_keys(children, child_type, f"{where}field {name!r}: ")', + ' _check_keys(children, child_type, where)', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 7b3e8a3..463dcc3 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -145,8 +145,8 @@ def _iterable_rows(rows): return kept -def _check_keys(rows, arrow_type): - """Refuse, at any depth, a dict key that no declared struct field has. +def _check_keys(rows, arrow_type, where=""): + """Refuse, at any depth, a dict key that no declared struct field has, naming the path to it. Arrow matches struct fields by exact name and fills a missing one with null, so a list of dicts keyed ``Amount`` against a field called @@ -172,7 +172,7 @@ def _check_keys(rows, arrow_type): shown = unexpected_keys[:KEYS_SHOWN] more = f" and {len(unexpected_keys) - KEYS_SHOWN} more" if len(unexpected_keys) > KEYS_SHOWN else "" raise ValueError( - f"declared {type_repr(arrow_type)} but the dicts carry keys {shown}{more} " + f"{where}declared {type_repr(arrow_type)} but the dicts carry keys {shown}{more} " f"that no declared field has; Arrow matches struct fields by exact name and " f"fills a missing one with null" ) @@ -180,22 +180,22 @@ def _check_keys(rows, arrow_type): given = getattr(row, "_fields", None) or getattr(row, "__fields__", None) if given is not None and set(given) == set(names) and list(given) != names: raise ValueError( - f"declared {type_repr(arrow_type)} but a row names its fields {list(given)}; a tuple's " - f"fields bind by position, so build it in the declared order or return dicts" + f"{where}declared {type_repr(arrow_type)} but a row names its fields {list(given)}; a " + f"tuple's fields bind by position, so build it in the declared order or return dicts" ) for index, (name, child_type) in enumerate(fields.items()): if _carries_keys(child_type): children = [row[name] for row in dicts if name in row] children.extend(row[index] for row in tuples if index < len(row)) - _check_keys(children, child_type) + _check_keys(children, child_type, f"{where}field {name!r}: ") elif _is_list_like(arrow_type): - _check_keys([item for row in _iterable_rows(rows) for item in row], arrow_type.value_type) + _check_keys([item for row in _iterable_rows(rows) for item in row], arrow_type.value_type, f"{where}list item: ") elif pa.types.is_map(arrow_type): keys, items = _map_entries(rows) if _carries_keys(arrow_type.key_type): - _check_keys(keys, arrow_type.key_type) + _check_keys(keys, arrow_type.key_type, f"{where}map key: ") if _carries_keys(arrow_type.item_type): - _check_keys(items, arrow_type.item_type) + _check_keys(items, arrow_type.item_type, f"{where}map value: ") def _map_entries(rows): diff --git a/test/test_messages.py b/test/test_messages.py index aba873f..207173a 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -249,3 +249,18 @@ def test_the_function_names_the_shape_it_takes_when_handed_a_batch_or_a_table(): with pytest.raises(TypeError, match="iterator of pyarrow.RecordBatch"): list(fn(handed)) assert list(fn([batch]))[0].column("out").to_pylist() == [1, 2, 3] + + +def test_a_nested_key_refusal_names_the_path_to_the_field(): + # Two same-typed sibling fields, a map's key and its value, and different + # depths all raised a byte-identical message naming the column alone. + inner = pa.struct([("amount", pa.int64())]) + schema = pa.schema([("s", pa.struct([("a", inner), ("b", inner), ("m", pa.map_(pa.string(), inner))]))]) + rows = [{"a": {"amount": 1}, "b": {"Amount": 2}, "m": [("k", {"amount": 3})]}] + fn = make_mapinarrow_func(lambda d, b, br: {"s": rows}, output_schema=schema) + with pytest.raises(ValueError, match=r"output column 's': field 'b': declared"): + list(fn(iter([_batch(v=[1])]))) + rows = [{"a": {"amount": 1}, "b": {"amount": 2}, "m": [("k", {"Amount": 3})]}] + fn = make_mapinarrow_func(lambda d, b, br: {"s": rows}, output_schema=schema) + with pytest.raises(ValueError, match=r"output column 's': field 'm': map value: declared"): + list(fn(iter([_batch(v=[1])]))) From fc6bfa24b3666516c6c63998582a87381af6b476 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:10:32 +0000 Subject: [PATCH 26/59] Compile without a cache, with a warning naming the remedy, where numba can write none numba sets a cached function up when it is decorated and raises RuntimeError there when no cache location can be written: a read-only install, an unwritable site-packages and user cache directory, or an import from an .egg, .whl or .pyz archive, which Spark's --py-files ships. The import then failed with no locator available, naming neither NUMBA_CACHE_DIR nor NUMBARROW_JIT_OPTIONS. The three compiled functions and the viewers now decorate through one helper that compiles such a function without a cache and warns with both remedies, and the README says how the cache works. --- .github/scripts/mutation_guard_check.py | 13 +++++++++--- README.md | 13 ++++++++++++ numbarrow/core/configurations.py | 28 +++++++++++++++++++++++++ numbarrow/core/is_null.py | 11 +++++----- numbarrow/utils/utils.py | 6 +++--- test/test_cache.py | 22 +++++++++++++++++++ 6 files changed, 81 insertions(+), 12 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index a106df1..abd9167 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -215,9 +215,9 @@ ( "is_null_struct stops being compiled with one signature", "numbarrow/core/is_null.py", - '@njit(boolean(int64, Optional(Array(uint8, 1, "C", readonly=True)),\n' - ' Optional(Array(uint8, 1, "C", readonly=True))), **jit_options)', - "@njit(**jit_options)", + '@jit_with_options(boolean(int64, Optional(Array(uint8, 1, "C", readonly=True)),\n' + ' Optional(Array(uint8, 1, "C", readonly=True))))', + "@jit_with_options()", ), ( "viewers stop getting a cache name of their own", @@ -626,6 +626,13 @@ ' _check_keys(children, child_type, f"{where}field {name!r}: ")', ' _check_keys(children, child_type, where)', ), + ( + 'a function that numba cannot cache stops compiling uncached', + 'numbarrow/core/configurations.py', + ' if "no locator available" not in str(error) or not jit_options.get("cache"):\n' + ' raise', + ' raise', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/README.md b/README.md index c9223f0..cbe35ed 100644 --- a/README.md +++ b/README.md @@ -126,6 +126,19 @@ struct-level bitmap is the only record of a row that is null as a whole, since the fields of such a row carry no validity bits of their own; pass both layers to `is_null_struct`. +## Compilation and the cache + +numbarrow compiles its adapters with numba on first import, under the options +`NUMBARROW_JIT_OPTIONS` gives as a JSON object; unset, that is `{"cache": +true}`, so the compiled code is written to numba's on-disk cache, next to the +package or under `NUMBA_CACHE_DIR`. Where no cache location can be written, a +read-only install or an import from an `.egg`, `.whl` or `.pyz` archive such +as `spark-submit --py-files` ships, the functions compile without a cache and +a warning names the two remedies: point `NUMBA_CACHE_DIR` at a writable +directory, or set `NUMBARROW_JIT_OPTIONS='{"cache": false}'`. numba's cache +index does not record the options a function was compiled with, so point +`NUMBA_CACHE_DIR` at a fresh directory when an option changes. + ## PySpark Integration Use `make_mapinarrow_func` to create functions compatible with PySpark's `mapInArrow`: diff --git a/numbarrow/core/configurations.py b/numbarrow/core/configurations.py index 9ccd5b9..947131f 100644 --- a/numbarrow/core/configurations.py +++ b/numbarrow/core/configurations.py @@ -4,6 +4,9 @@ import os import json +import warnings + +from numba import njit invalid_jit_options_err = ( @@ -57,3 +60,28 @@ def get_jit_options(): jit_options = get_jit_options() + + +def jit_with_options(*signature): + """``njit`` under the options ``NUMBARROW_JIT_OPTIONS`` gives, compiling uncached where no cache can be written. + + numba sets a cached function up when it is decorated, and raises ``RuntimeError`` there when no cache + location can be written: a read-only install, an unwritable ``site-packages`` and user cache directory, or + an import from an ``.egg``, ``.whl`` or ``.pyz`` archive, which Spark's ``--py-files`` ships. Nothing then + named the way out. Such a function compiles without a cache, with a warning naming ``NUMBA_CACHE_DIR`` and + ``NUMBARROW_JIT_OPTIONS='{"cache": false}'``. A write that fails later, on a full disk, is numba's own error. + """ + def decorate(func): + try: + return njit(*signature, **jit_options)(func) + except RuntimeError as error: + if "no locator available" not in str(error) or not jit_options.get("cache"): + raise + warnings.warn( + f"numba cannot cache {func.__qualname__} here ({error}); it compiles without a cache. Set " + f"NUMBA_CACHE_DIR to a writable directory, or NUMBARROW_JIT_OPTIONS='{{\"cache\": false}}' to " + f"turn caching off and silence this warning", + RuntimeWarning, stacklevel=2, + ) + return njit(*signature, **{**jit_options, "cache": False})(func) + return decorate diff --git a/numbarrow/core/is_null.py b/numbarrow/core/is_null.py index ada674f..9f7350d 100644 --- a/numbarrow/core/is_null.py +++ b/numbarrow/core/is_null.py @@ -8,13 +8,12 @@ """ import numpy as np -from numba import njit from numba.core.types import boolean, int64, Array, Optional, uint8, bool_ -from numbarrow.core.configurations import jit_options +from numbarrow.core.configurations import jit_with_options -@njit(boolean(int64, Array(uint8, 1, "C", readonly=True)), **jit_options) +@jit_with_options(boolean(int64, Array(uint8, 1, "C", readonly=True))) def is_null(index_: int, bitmap: np.ndarray) -> bool: """Check whether element *index_* is null according to *bitmap*. @@ -39,7 +38,7 @@ def is_null(index_: int, bitmap: np.ndarray) -> bool: return not (byte_for_index >> bit_position_in_byte) % 2 -@njit(Array(bool_, 1, "C")(int64, int64, Array(uint8, 1, "C", readonly=True)), **jit_options) +@jit_with_options(Array(bool_, 1, "C")(int64, int64, Array(uint8, 1, "C", readonly=True))) def unpack_booleans(offset: int, length: int, packed_data: np.ndarray) -> np.ndarray: """Unpack bit-packed boolean data into a boolean array. @@ -64,8 +63,8 @@ def unpack_booleans(offset: int, length: int, packed_data: np.ndarray) -> np.nda # design. See: https://awkward-array.org/doc/main/reference/generated/ak.contents.BitMaskedArray.html -@njit(boolean(int64, Optional(Array(uint8, 1, "C", readonly=True)), - Optional(Array(uint8, 1, "C", readonly=True))), **jit_options) +@jit_with_options(boolean(int64, Optional(Array(uint8, 1, "C", readonly=True)), + Optional(Array(uint8, 1, "C", readonly=True)))) def is_null_struct(index_, struct_bitmap, field_bitmap): """Check whether a struct field value is null at either the struct or field layer. diff --git a/numbarrow/utils/utils.py b/numbarrow/utils/utils.py index 2aa5e00..083bb77 100644 --- a/numbarrow/utils/utils.py +++ b/numbarrow/utils/utils.py @@ -9,11 +9,11 @@ import hashlib import numpy as np -from numba import carray, from_dtype, int64, intp, njit +from numba import carray, from_dtype, int64, intp from numba.core.types import Array, voidptr from numba.extending import intrinsic -from numbarrow.core.configurations import jit_options +from numbarrow.core.configurations import jit_with_options @intrinsic @@ -69,7 +69,7 @@ def viewer(ptr_as_int: int, sz: int): name += "_" + hashlib.sha1(repr(dtype_.descr).encode()).hexdigest()[:12] viewer.__name__ = name viewer.__qualname__ = f"{numpy_array_from_ptr_factory.__qualname__}..{name}" - return njit(Array(from_dtype(dtype_), 1, "C")(intp, int64), **jit_options)(viewer) + return jit_with_options(Array(from_dtype(dtype_), 1, "C")(intp, int64))(viewer) class _Viewers(dict): diff --git a/test/test_cache.py b/test/test_cache.py index c13ebae..69d5989 100644 --- a/test/test_cache.py +++ b/test/test_cache.py @@ -21,6 +21,7 @@ import os import subprocess import sys +import zipfile from pathlib import Path REPO = Path(__file__).resolve().parent.parent @@ -171,3 +172,24 @@ def test_a_cold_cache_survives_a_concurrent_first_import_of_is_null_struct(tmp_p assert not failed, failed[0] out = _run(CHECK_EVERY_STRUCT_SHAPE, env, tmp_path) assert out.returncode == 0, out.stderr + + +def test_an_import_from_an_archive_compiles_uncached_with_a_warning_naming_the_remedy(tmp_path): + # numba's cache locators need the source file on disk, so an import from + # an .egg, .whl or .pyz archive, which Spark's --py-files ships, raised + # RuntimeError at decoration, naming neither NUMBA_CACHE_DIR nor the + # option that turns caching off. + archive = tmp_path / "numbarrow-0.0.0-py3.12.egg" + with zipfile.ZipFile(archive, "w") as zipped: + for path in sorted((REPO / "numbarrow").rglob("*.py")): + zipped.write(path, str(path.relative_to(REPO))) + env = dict(os.environ, PYTHONPATH=str(archive), NUMBA_CACHE_DIR=str(tmp_path / "cache")) + env.pop("NUMBARROW_JIT_OPTIONS", None) + probe = "import numbarrow.core.adapters as a; print(a.__file__)" + run = subprocess.run([sys.executable, "-W", "always", "-c", probe], + capture_output=True, text=True, env=env, cwd=str(tmp_path)) + assert run.returncode == 0 and str(archive) in run.stdout, run.stderr + assert "NUMBA_CACHE_DIR" in run.stderr and "compiles without a cache" in run.stderr + quiet = subprocess.run([sys.executable, "-W", "error", "-c", probe], capture_output=True, text=True, + env=dict(env, NUMBARROW_JIT_OPTIONS='{"cache": false}'), cwd=str(tmp_path)) + assert quiet.returncode == 0, quiet.stderr From 8ce83b5d91944a7f319dc23a0ee6bc00851fc308 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:13:45 +0000 Subject: [PATCH 27/59] Say that Spark reads a struct column's fields by position too, and how to derive output_schema from the Spark schema The positional-binding paragraph covered top-level columns; a record array's dtype order and a dict's key order decided where each struct field landed, batch by batch when some rows left a null field out, and the docstring's remedy did not mention that output_schema binds struct fields by name or that its column order has to match the schema mapInArrow is given, which this function never sees. The docstring and the README say both, and name to_arrow_schema as the way to keep the two orders equal. --- README.md | 8 ++++++++ numbarrow/core/mapinarrow_factory.py | 14 +++++++++++--- 2 files changed, 19 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index cbe35ed..552e1ce 100644 --- a/README.md +++ b/README.md @@ -182,6 +182,14 @@ a numpy masked array carry nulls out as well. See [test/test_mapinarrow_spark.py](test/test_mapinarrow_spark.py) for a complete runnable example. +Spark binds the batch's columns, and a struct column's fields, to the schema +given to `mapInArrow` by position, never by name: two columns or two fields +whose types share an accessor family swap silently when the dict or the record +dtype is built in the other order. Build them in the declared order, or pass +`output_schema` derived from the Spark schema, +`pyspark.sql.pandas.types.to_arrow_schema(spark_schema)`, and let Arrow bind +columns and struct fields by name. + ## Compatibility | Dependency | Versions | diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 463dcc3..0a49f96 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -695,9 +695,17 @@ def make_mapinarrow_func( ``java.lang.UnsupportedOperationException`` whatever its width, as float64 under ``LongType`` and int64 under ``DoubleType`` do, both 64 bits wide, and so does a width mismatch inside one family, such as - int32 under ``LongType``. Build the returned dict in the order the - output schema declares, or pass ``output_schema`` and let Arrow bind - it by name instead. + int32 under ``LongType``. A struct column's fields are read by + position too: without ``output_schema`` a record array's dtype order + and a dict's key order decide where each field lands, batch by batch + when some rows leave a null field out, so ``{"lat": .., "lon": ..}`` + under ``struct`` swaps the two without a word. Build the + returned dict, and any record dtype, in the order the output schema + declares, or pass ``output_schema`` and let Arrow bind columns and + struct fields by name; derive it from the Spark schema, + ``pyspark.sql.pandas.types.to_arrow_schema(spark_schema)``, so the + two orders cannot disagree, since this function never sees the schema + ``mapInArrow`` is given. ``data_dict`` maps each selected column's name to its data. A column of a uniform type maps to one array. A struct or list-of-struct From 6fe9ba4f9e1f05ed222c473189bea9cd5852f1fa Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:13:55 +0000 Subject: [PATCH 28/59] Complete the list of silent lossy conversions, and name OverflowError where an unsigned type is declared The closed list omitted a timestamp into a time type, which drops the date, a float into a decimal, which rounds, a number into bool, and a float into a narrower float than float32; and a negative value under an unsigned declared type raises OverflowError, which the documented except pa.ArrowInvalid does not catch, as does a value beyond the type's own ceiling rather than beyond int64. --- numbarrow/core/mapinarrow_factory.py | 20 ++++++++++++-------- 1 file changed, 12 insertions(+), 8 deletions(-) diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 0a49f96..e93522b 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -763,14 +763,18 @@ def make_mapinarrow_func( :class:`pyarrow.ArrowInvalid`. A Python list, and any other sequence of Python objects, an object-dtype ndarray included, goes through ``pa.array``'s sequence converter instead: an integer out of the - declared type's range still raises :class:`pyarrow.ArrowInvalid` and - one beyond int64 altogether raises :class:`OverflowError`, but a - float's fraction and a timestamp's extra digits are dropped silently. - So the lossy conversions that pass without a word are a timestamp - into ``date32`` or ``date64``, which floors to the day, ``float64`` - into ``float32``, which overflows to ``inf``, and, from a list alone, - a fraction into an integer type and a timestamp unit change that - drops digits. + declared type's range still raises :class:`pyarrow.ArrowInvalid`, a + negative value under an unsigned type and a value beyond the type's + own ceiling raise :class:`OverflowError`, which ``except + pa.ArrowInvalid`` does not catch, but a float's fraction and a + timestamp's extra digits are dropped silently. So the lossy + conversions that pass without a word are a timestamp into ``date32`` + or ``date64``, which floors to the day, a timestamp into ``time32`` + or ``time64``, which drops the date, a float into ``decimal``, which + rounds to the declared scale, a number into ``bool``, which is true + for anything but zero, a float into a narrower float, which overflows + to ``inf``, and, from a list alone, a fraction into an integer type + and a timestamp unit change that drops digits. Left as ``None`` the batch is built from the dict alone: insertion order decides, and every type is inferred from the value, so a unicode From 21f91becdc525aa4bdd4753ae3474621ebf23072 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:14:44 +0000 Subject: [PATCH 29/59] Bound the negative-index sentence, escape the underscored parameter names, and say which route flips a view writable is_null's docstring said every negative index stays in bounds and bounds checking cannot catch it, which holds only down to minus eight times the bitmap's length. Sphinx read the trailing underscore of :param index_: and :param dtype_: as a reference marker and rendered index and dtype, so a keyword call copied from the docs raised TypeError. A slice or reshape of a view that an njit function returns is boxed writable by numba and its flag can be flipped, a route the read-only paragraph did not mention; pyarrow's own view has it too. --- README.md | 6 +++++- numbarrow/core/is_null.py | 11 ++++++----- numbarrow/utils/arrow_array_utils.py | 8 +++++--- numbarrow/utils/utils.py | 2 +- 4 files changed, 17 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index 552e1ce..c68b8df 100644 --- a/README.md +++ b/README.md @@ -92,7 +92,11 @@ caller does not own and cannot be made writable, which is also why pyarrow's own `to_numpy(zero_copy_only=True)` refuses to hand out a writable one; the copies, booleans, `date32` and strings, start read-only as well, so the contract does not depend on the type, though a caller who flips the flag on a -copy writes into memory that is their own. Declare numba signatures that receive them with +copy writes into memory that is their own. A slice or reshape of a view that +an `@njit` function returns is a new array whose flag can be flipped, since +numba exports every buffer it boxes as writable, and a store through it +reaches the source; pyarrow's own view has the same route, and the whole +argument returned unchanged does not. Declare numba signatures that receive them with `readonly=True`, which accepts writable arrays as well, or leave the function lazily typed and numba will infer it. Returned bitmaps own their memory and are writable. diff --git a/numbarrow/core/is_null.py b/numbarrow/core/is_null.py index 9f7350d..b47177e 100644 --- a/numbarrow/core/is_null.py +++ b/numbarrow/core/is_null.py @@ -23,11 +23,12 @@ def is_null(index_: int, bitmap: np.ndarray) -> bool: ``index_`` must satisfy ``0 <= index_ < 8 * len(bitmap)``. Compiled without bounds checking, which is numba's default, an index past the bitmap reads memory that is not the bitmap's; ``NUMBARROW_JIT_OPTIONS='{"boundscheck": - true}'`` turns that into ``IndexError``. A negative index reads from the - bitmap's end like any numpy index, which is the wrong bit and in bounds, - so bounds checking does not catch it. + true}'`` turns that into ``IndexError``. A negative index down to + ``-8 * len(bitmap)`` reads from the bitmap's end like any numpy index, + which is the wrong bit and in bounds, so bounds checking does not catch + it; below that the read is out of bounds, and bounds checking does. - :param index_: zero-based element index + :param index\\_: zero-based element index :param bitmap: uint8 array containing the packed validity bitmap :returns: True if the element is null (bit is 0), False if valid (bit is 1) """ @@ -81,7 +82,7 @@ def is_null_struct(index_, struct_bitmap, field_bitmap): counting the entries in the index it just read, so a second entry is something two processes warming a cold cache can disagree about. - :param index_: zero-based element index, converted to ``int64`` + :param index\\_: zero-based element index, converted to ``int64`` :param struct_bitmap: uint8 packed bitmap for struct-level validity, or None :param field_bitmap: uint8 packed bitmap for field-level validity, or None :returns: True if null at either layer diff --git a/numbarrow/utils/arrow_array_utils.py b/numbarrow/utils/arrow_array_utils.py index 5343c0d..7a2d824 100644 --- a/numbarrow/utils/arrow_array_utils.py +++ b/numbarrow/utils/arrow_array_utils.py @@ -444,9 +444,11 @@ def uniform_arrow_array_adapter(pa_array: pa.Array) -> tuple[np.ndarray | None, Returns the validity bitmap, which owns its memory, and a zero-copy numpy view over the array's data buffer. The view is read-only and cannot be made writable: Arrow buffers are immutable by contract, and this is what - pyarrow's own ``Array.to_numpy(zero_copy_only=True)`` returns. Declare numba - signatures that receive it with ``readonly=True``, which accepts writable - arrays too. + pyarrow's own ``Array.to_numpy(zero_copy_only=True)`` returns. A slice or + reshape of it that an ``@njit`` function returns is a new array numba + boxes as writable, and its flag can be flipped, as pyarrow's own view's + can by the same route. Declare numba signatures that receive it with + ``readonly=True``, which accepts writable arrays too. """ data_arrow_ty = pa_array.type data_np_ty = arrow_to_numpy_dtypes.get(data_arrow_ty, None) diff --git a/numbarrow/utils/utils.py b/numbarrow/utils/utils.py index 083bb77..8d27a47 100644 --- a/numbarrow/utils/utils.py +++ b/numbarrow/utils/utils.py @@ -44,7 +44,7 @@ def numpy_array_from_ptr_factory(dtype_): read-only arrays tied to the Arrow array they view; this is the primitive it is built on. - :param dtype_: NumPy dtype for the resulting array (e.g. ``np.int32``) + :param dtype\\_: NumPy dtype for the resulting array (e.g. ``np.int32``) :returns: JIT-compiled function ``(int, int) -> np.ndarray`` """ def viewer(ptr_as_int: int, sz: int): From 159ffcf0c1ececf4e33ea0fdc55537499c9fa8b0 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:14:55 +0000 Subject: [PATCH 30/59] Say what the Spark struct-null test checks, since Spark nulls the child itself The test's comment credited the folded bitmap with seeing the null struct row, but Spark's own Arrow writer nulls the child of such a row before the batch arrives, so the test passes with the fold reduced to a no-op; the fold is pinned by the non-Spark tests that build a child carrying no null of its own, and this test is the end-to-end check that the row reaches the UDF at all. --- test/test_mapinarrow_spark.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/test/test_mapinarrow_spark.py b/test/test_mapinarrow_spark.py index 993e9d1..89806a6 100644 --- a/test/test_mapinarrow_spark.py +++ b/test/test_mapinarrow_spark.py @@ -179,9 +179,12 @@ def test_null_magnitude_is_excluded_from_the_product(spark): def test_struct_null_row_survives_the_arrow_transport(spark): - # A struct-null row reaches the worker with no child validity buffer, so - # the field's own bitmap cannot see it. Spark's transport is an Arrow IPC - # round trip, which preserves that shape, and the folded bitmap catches it. + # Spark's own Arrow writer nulls the child of a null struct row, so the + # field's bitmap alone sees this row and the fold changes nothing here: + # this is the end-to-end check that a struct null reaches the UDF through + # the transport at all. The fold is pinned where the child carries no + # null of its own, by test_both_null_layers_are_folded_together and its + # siblings in test_mapinarrow_factory.py. schema = StructType([ StructField("id", StringType()), StructField("point", StructType([StructField("v", LongType())])), From 58dc99b72f6f8d1a4ad6f721ebfcbfba2ce2cd94 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:15:05 +0000 Subject: [PATCH 31/59] Raise the pyspark floor to 3.5 on Python 3.13, and say what the extras and the floors really supply pyspark 3.4 does not import on Python 3.13, so the test extra's floor carries a marker; a struct output column needs pyspark 3.5, which the README's table says; the mapinarrow extra supplies setuptools for pyspark's distutils import on 3.12 as well as pandas, which the README said nothing about, so a reader who had pandas from pyspark[sql] saw no reason to install it and the example died on ModuleNotFoundError; and the 3.3 rationale named the worker-side symptom a trivial UDF hits where the factory's closure fails first on the driver, inside cloudpickle. --- README.md | 15 ++++++++++----- pyproject.toml | 7 +++++-- 2 files changed, 15 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index c68b8df..496d620 100644 --- a/README.md +++ b/README.md @@ -14,7 +14,7 @@ Optional dependencies for PySpark and pandas support: ```bash pip install numbarrow[test] # adds pyspark and everything the tests need -pip install numbarrow[mapinarrow] # adds pandas, which pyspark's mapInArrow requires +pip install numbarrow[mapinarrow] # adds pandas and setuptools, which pyspark's mapInArrow requires ``` The adapters themselves need only numba, numpy and pyarrow. @@ -201,16 +201,21 @@ columns and struct fields by name. | Python | 3.12+ | | numba | 0.60.0 – 0.67.0 | | pyarrow | 14.0 – 25.0 | -| pyspark | 3.4 – 3.x (optional) | -| pandas | 2.2.2+ (optional, required by pyspark's `mapInArrow`) | +| pyspark | 3.4 – 3.x (optional; 3.5+ on Python 3.13, and for a struct output column) | +| pandas | 2.2.2+ (optional, required by pyspark's `mapInArrow`, with setuptools for its `distutils` import on 3.12+) | `pyproject.toml` is authoritative. CI runs the newest numba the cap admits, with pandas 2.3.2 and pyspark 3.5.7, on Linux, Linux ARM and Windows, and both ends of the pyarrow row in a job of their own; the pandas and pyspark rows are not swept. The pyspark floor is 3.4.0 because pyspark 3.3 bundles cloudpickle 2.0.0, which predates the `co_qualname` argument Python -3.11 added to `code()`, so on the declared Python every UDF dies in the worker -with `TypeError: code() argument 13 must be str, not int`. The pandas floor is +3.11 added to `code()` and indexes `co_names` with the raw `LOAD_GLOBAL` +argument, so on the declared Python the function `make_mapinarrow_func` +returns fails on the driver while cloudpickle serialises it, with +`PicklingError: Could not serialize object: IndexError: tuple index out of +range`, and a trivial UDF dies in the worker with `TypeError: code() argument +13 must be str, not int`. pyspark 3.4 refuses a struct output column and does +not import on Python 3.13, so those need 3.5. The pandas floor is 2.2.2 in both extras: no pandas below 2.1.1 publishes a Python 3.12 wheel, and 2.1.1 installs next to numpy 2 but fails to import with `numpy.dtype size changed`; 2.2.2 is the first release built against numpy 2. The package also diff --git a/pyproject.toml b/pyproject.toml index 53e48c6..7bb32cd 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,8 +28,11 @@ test = [ "pandas>=2.2.2", # pyspark 3.3 bundles cloudpickle 2.0.0, which predates the co_qualname # argument Python 3.11 added to code(), so on the declared Python every - # UDF dies in the worker. 3.4.0 is the first release that runs one. - "pyspark>=3.4.0,<4.0.0", + # UDF dies: the factory's closure on the driver while cloudpickle walks its + # globals, a trivial one in the worker. 3.4.0 is the first release that + # runs one, and the first that imports on 3.13 is 3.5.0. + "pyspark>=3.4.0,<4.0.0; python_version < '3.13'", + "pyspark>=3.5.0,<4.0.0; python_version >= '3.13'", "pytest", "python-dateutil>=2.8.0", "setuptools>=75.0.0" From 52e536967c393e7907d543252c736017195f2c1b Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:15:15 +0000 Subject: [PATCH 32/59] Run the Windows install and lint steps under bash, parse the requires-python floor in the extras gate, and give the docs example its placeholders Without a shell the two steps ran under pwsh on the windows cells, where only the last command's exit decides a multi-line step, so a failing numba pin or a failing syntax gate went green on nine of thirty cells; the extras gate derived its interpreter by stripping leading characters from requires-python, so a capped specifier became the executable name python3.12,<3.14 and crashed the gate before any extra was installed; and the docs page's example used df_in and output_schema undefined, which the rst gate never saw because it matches only a code-block directive. --- .github/scripts/extras_sufficiency_check.py | 15 +++++++++++++-- .github/workflows/numbarrow_ci.yml | 5 +++++ docs/numbarrow.core.mapinarrow_factory.rst | 2 ++ 3 files changed, 20 insertions(+), 2 deletions(-) diff --git a/.github/scripts/extras_sufficiency_check.py b/.github/scripts/extras_sufficiency_check.py index 6be2062..e526c9e 100755 --- a/.github/scripts/extras_sufficiency_check.py +++ b/.github/scripts/extras_sufficiency_check.py @@ -16,6 +16,7 @@ """ import argparse import os +import re import shutil import subprocess import sys @@ -82,6 +83,14 @@ def check(repo: Path, extra: str, probe: str, python: str) -> str | None: return None +def floor_interpreter(requires_python): + """The interpreter name for the floor of a requires-python specifier set, such as python3.12.""" + floor = re.search(r">=\s*(\d+\.\d+)", requires_python) + if floor is None: + raise SystemExit(f"requires-python {requires_python!r} names no >= floor to test the extras on") + return "python" + floor.group(1) + + def main(argv=None): ap = argparse.ArgumentParser() ap.add_argument("--repo", default=str(Path(__file__).resolve().parents[2])) @@ -97,8 +106,10 @@ def main(argv=None): declared = set(meta.get("optional-dependencies", {})) # The floor is what matters: distutils is present below 3.12 and absent at # and above it, so testing a newer interpreter would hide the failure and - # testing the floor exposes it. - python = args.python or "python" + meta["requires-python"].lstrip(">=~^ ") + # testing the floor exposes it. requires-python is a specifier set, so the + # floor is the >= clause, wherever it sits: stripping leading characters + # turned ">=3.12,<3.14" into the executable name "python3.12,<3.14". + python = args.python or floor_interpreter(meta["requires-python"]) unprobed = declared - set(PROBES) - SKIP if unprobed: diff --git a/.github/workflows/numbarrow_ci.yml b/.github/workflows/numbarrow_ci.yml index 7ac6e03..8eec6b4 100644 --- a/.github/workflows/numbarrow_ci.yml +++ b/.github/workflows/numbarrow_ci.yml @@ -37,6 +37,10 @@ jobs: shell: bash run: echo "__version__ = '${{ env.VERSION }}'" > numbarrow/__init__.py - name: Install dependencies + # bash, as the steps around it: under pwsh, the windows default, only the + # last command's exit decides a multi-line step, so a failing pin here + # and a failing lint below went green on nine of the thirty cells. + shell: bash run: | python -m pip install --upgrade pip pip install flake8 pytest setuptools @@ -44,6 +48,7 @@ jobs: pip install pyspark==3.5.7 pip install -e . - name: Lint with flake8 + shell: bash run: | # stop the build if there are Python syntax errors or undefined names flake8 . --count --select=E9,F63,F7,F82 --show-source --statistics diff --git a/docs/numbarrow.core.mapinarrow_factory.rst b/docs/numbarrow.core.mapinarrow_factory.rst index 806ef3b..e91e6cd 100644 --- a/docs/numbarrow.core.mapinarrow_factory.rst +++ b/docs/numbarrow.core.mapinarrow_factory.rst @@ -28,6 +28,8 @@ Usage:: return {"output_col": Nullable(result, bitmap_dict["input_col"])} udf = make_mapinarrow_func(my_func, broadcasts={"scale": 1.5}) + df_in = ... # caller-provided PySpark DataFrame + output_schema = ... # caller-provided PySpark StructType df_out = df_in.mapInArrow(udf, output_schema) Every name in ``data_dict`` is also a key of ``bitmap_dict``, so a batch that From aeae42e3062edeb4629ee6d09f80f90d9e187785 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:15:27 +0000 Subject: [PATCH 33/59] Pin the terms the suite could not see: timestamp units, the key check's list layouts and null rows, the options reaching is_null, an empty bytes column, every refusal class of a record field, and the dispatcher's cut Adapting every timestamp as microseconds, dropping large_list or fixed_size_list from the key check, hard-coding the options on is_null.py's decorators, dropping the empty bytes column's binary default, narrowing the record field's except tuple to pyarrow errors, and spelling a chunked array's type out in full each left all the tests green; each now has a test, and three of them a catalogue entry. --- .github/scripts/mutation_guard_check.py | 20 +++++++++++++ test/test_adapters.py | 11 +++++++ test/test_cache.py | 25 ++++++++++++++++ test/test_mapinarrow_factory.py | 38 +++++++++++++++++++++++++ test/test_messages.py | 10 +++++++ 5 files changed, 104 insertions(+) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index abd9167..faf540d 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -633,6 +633,26 @@ ' raise', ' raise', ), + ( + 'a bytes output stops naming its type', + 'numbarrow/core/mapinarrow_factory.py', + ' return pa.array(value.tolist(), type=arrow_type or pa.binary())', + ' return pa.array(value.tolist(), type=arrow_type)', + ), + ( + "the dispatcher stops cutting a chunked array's type", + 'numbarrow/core/adapters.py', + ' f"Not implemented for a ChunkedArray of {pa_array.num_chunks} chunks of type "\n' + ' f"{type_repr(pa_array.type)}: pass one chunk, or combine_chunks() first"', + ' f"Not implemented for a ChunkedArray of {pa_array.num_chunks} chunks of type "\n' + ' f"{pa_array.type}: pass one chunk, or combine_chunks() first"', + ), + ( + 'a timestamp stops being read at its own unit', + 'numbarrow/core/adapters.py', + ' return cast_64bit_date_arrow_to_numpy_array(pa_array, np.dtype(f"datetime64[{timestamp_unit}]"))', + ' return cast_64bit_date_arrow_to_numpy_array(pa_array, np.dtype("datetime64[us]"))', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/test/test_adapters.py b/test/test_adapters.py index f9e3b07..76dea81 100644 --- a/test/test_adapters.py +++ b/test/test_adapters.py @@ -182,3 +182,14 @@ def test_the_64bit_date_view_refuses_a_unit_that_is_not_the_arrays_own(): cast_64bit_date_arrow_to_numpy_array(stamps, np.dtype("datetime64[s]")) with pytest.raises(ValueError, match="not a date64 or timestamp"): cast_64bit_date_arrow_to_numpy_array(pa.array([1], type=pa.int64()), np.dtype("datetime64[s]")) + + +def test_a_timestamp_is_read_at_its_own_unit(): + # Every test that read a timestamp value used microseconds, so adapting + # every unit as datetime64[us] passed the suite while a millisecond column + # read as 1970. + instant = datetime(2020, 9, 13, 12, 26, 40) + for unit in ("s", "ms", "us", "ns"): + _, data = arrow_array_adapter(pa.array([instant], type=pa.timestamp(unit))) + assert data.dtype == np.dtype(f"datetime64[{unit}]"), unit + assert data.astype("datetime64[us]").tolist() == [instant], unit diff --git a/test/test_cache.py b/test/test_cache.py index 69d5989..01cb67f 100644 --- a/test/test_cache.py +++ b/test/test_cache.py @@ -193,3 +193,28 @@ def test_an_import_from_an_archive_compiles_uncached_with_a_warning_naming_the_r quiet = subprocess.run([sys.executable, "-W", "error", "-c", probe], capture_output=True, text=True, env=dict(env, NUMBARROW_JIT_OPTIONS='{"cache": false}'), cwd=str(tmp_path)) assert quiet.returncode == 0, quiet.stderr + + +CHECK_BOUNDS = ( + "import numpy as np\n" + "from numbarrow.core.is_null import is_null, unpack_booleans\n" + "bitmap = np.zeros(1, dtype=np.uint8)\n" + "outcomes = []\n" + "for call in (lambda: is_null(100000, bitmap), lambda: unpack_booleans(0, 100000, bitmap)):\n" + " try:\n" + " call()\n" + " outcomes.append('returned')\n" + " except IndexError:\n" + " outcomes.append('IndexError')\n" + "print(' '.join(outcomes))\n" +) + + +def test_jit_options_reach_the_is_null_decorators(tmp_path): + # The options test imported only the viewers, so hard-coding the options + # on is_null.py's three decorators kept the suite green, and the documented + # boundscheck contract had no test at all. + checked = _run(CHECK_BOUNDS, _env(tmp_path / "checked", {"cache": False, "boundscheck": True}), tmp_path) + assert checked.returncode == 0 and checked.stdout.split() == ["IndexError", "IndexError"], checked.stderr + unchecked = _run(CHECK_BOUNDS, _env(tmp_path / "unchecked", {"cache": False}), tmp_path) + assert unchecked.returncode == 0 and unchecked.stdout.split() == ["returned", "returned"], unchecked.stderr diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 75418e9..6a0195b 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -1081,3 +1081,41 @@ def test_a_fixed_width_binary_column_keeps_a_trailing_nul_under_its_declared_typ got = run_outputs({"d": digests}, declared).column("d") assert got.type == pa.binary(16) assert got.to_pylist() == [b"0123456789abcde\x00", b"\x00fedcba987654321", b"0123456789abcdef"] + + +def test_the_key_check_covers_the_list_layouts_a_missing_field_and_a_none_row(): + # large_list and fixed_size_list, a row lacking a nested field and a None + # row in a struct column were terms of the check no test reached: dropping + # any of them left the suite green. + inner = pa.struct([("amount", pa.int64())]) + for layout in (pa.large_list(inner), pa.list_(inner, 1)): + with pytest.raises(ValueError, match=r"'s'.*'Amount'"): + run_outputs({"s": [[{"Amount": 1}], [{"amount": 2}]]}, pa.schema([("s", layout)])) + nested = pa.schema([("s", pa.struct([("id", pa.int64()), ("inner", inner)]))]) + got = run_outputs({"s": [{"id": 1, "inner": {"amount": 5}}, {"id": 2}]}, nested).column("s") + assert got.to_pylist() == [{"id": 1, "inner": {"amount": 5}}, {"id": 2, "inner": None}] + got = run_outputs({"s": [{"amount": 1}, None]}, pa.schema([("s", inner)])).column("s") + assert got.to_pylist() == [{"amount": 1}, None] + + +def test_an_empty_bytes_column_keeps_its_type(): + # The unicode half of the empty-batch pin had a test and the bytes half + # did not: an empty |S column inferred null with the default dropped. + got = run_outputs({"b": np.empty(0, dtype="S5"), "n": np.empty(0, dtype=np.int64)}) + assert [field.type for field in got.schema] == [pa.binary(), pa.int64()] + + +def test_every_refusal_class_of_a_record_field_names_the_field(): + # Only the pyarrow member of the field's except tuple was pinned; narrowed + # to it, a dict key no field has, a nested record declared as a list and + # an int beyond int64 all stopped naming the field. + inner = pa.struct([("amount", pa.int64())]) + with_dict = np.array([({"Amount": 1},)], dtype=[("meta", "O")]) + with pytest.raises(ValueError, match=r"'r': field 'meta': declared"): + run_outputs({"r": with_dict}, pa.schema([("r", pa.struct([("meta", inner)]))])) + nested = np.array([((1, 2),)], dtype=[("pair", [("c", "i8"), ("d", "i8")])]) + with pytest.raises(TypeError, match=r"'r': field 'pair': a record array"): + run_outputs({"r": nested}, pa.schema([("r", pa.struct([("pair", pa.list_(pa.int64()))]))])) + big = np.array([(2 ** 70,)], dtype=[("big", "O")]) + with pytest.raises(OverflowError, match=r"'r': field 'big'"): + run_outputs({"r": big}, pa.schema([("r", pa.struct([("big", pa.int64())]))])) diff --git a/test/test_messages.py b/test/test_messages.py index 207173a..bea774b 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -264,3 +264,13 @@ def test_a_nested_key_refusal_names_the_path_to_the_field(): fn = make_mapinarrow_func(lambda d, b, br: {"s": rows}, output_schema=schema) with pytest.raises(ValueError, match=r"output column 's': field 'm': map value: declared"): list(fn(iter([_batch(v=[1])]))) + + +def test_the_dispatcher_cuts_a_wide_type_for_a_chunked_array(): + # The wide-type test reached two of the twelve cut sites and none of the + # dispatcher's own: a Table column of a thousand-field struct put every + # field into the message under a mutant that spelled the type out. + column = pa.chunked_array([pa.array([{f"field_{i}": 1 for i in range(1000)}], type=_wide_struct(1000))]) + with pytest.raises(NotImplementedError) as excinfo: + arrow_array_adapter(column) + assert "ChunkedArray" in str(excinfo.value) and len(str(excinfo.value)) < 400 From 42aa41a2c9515ec4d4570448e8fb9be293e53601 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:18:34 +0000 Subject: [PATCH 34/59] Keep the new guards in helpers of their own The guards this branch added to make_mapinarrow_func, _convert and _check_keys took the factory's advisory complexity from 11 to 17 and the two helpers to 12. The permuted-fields refusal, the pandas-row scan, the repeated-names refusal and the held inferred schema each move into a helper that says what it refuses and why, and the functions that call them read as they did. --- .github/scripts/mutation_guard_check.py | 20 ++--- numbarrow/core/mapinarrow_factory.py | 101 ++++++++++++++++-------- 2 files changed, 77 insertions(+), 44 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index faf540d..60974e2 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -456,8 +456,8 @@ ( 'a namedtuple naming the fields in another order stops being refused', 'numbarrow/core/mapinarrow_factory.py', - ' if given is not None and set(given) == set(names) and list(given) != names:', - ' if False:', + ' if given is not None and set(given) == set(names) and list(given) != names:', + ' if False:', ), ( 'iterability stops being tested with iter', @@ -480,8 +480,8 @@ ( 'a pandas Series row stops being refused', 'numbarrow/core/mapinarrow_factory.py', - ' if _is_pandas(row, "Series", "DataFrame"):', - ' if False:', + ' if _is_pandas(row, "Series", "DataFrame"):', + ' if False:', ), ( 'a KeyError from pa.array stops naming the column', @@ -527,8 +527,8 @@ ( "a later batch's inferred schema stops being compared with the first's", 'numbarrow/core/mapinarrow_factory.py', - ' elif built.schema != inferred:', - ' elif False:', + ' if built.schema != inferred:', + ' if False:', ), ( 'an extension column stops being masked through its storage', @@ -591,10 +591,10 @@ ( 'a repeated name in output_schema stops being refused', 'numbarrow/core/mapinarrow_factory.py', - ' repeated = _repeated_names(list(output_schema))\n' - ' if repeated:', - ' repeated = _repeated_names(list(output_schema))\n' - ' if False:', + ' repeated = _repeated_names(list(output_schema))\n' + ' if repeated:', + ' repeated = _repeated_names(list(output_schema))\n' + ' if False:', ), ( 'an item that is not a RecordBatch stops being refused by name', diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index e93522b..2296e30 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -145,6 +145,21 @@ def _iterable_rows(rows): return kept +def _refuse_permuted_fields(tuples, names, arrow_type, where): + """Refuse a namedtuple or pyspark Row naming the declared fields in another order. + + ``pa.array`` binds a tuple's fields by position, so such a row swaps every + same-typed field without a word. + """ + for row in tuples: + given = getattr(row, "_fields", None) or getattr(row, "__fields__", None) + if given is not None and set(given) == set(names) and list(given) != names: + raise ValueError( + f"{where}declared {type_repr(arrow_type)} but a row names its fields {list(given)}; a " + f"tuple's fields bind by position, so build it in the declared order or return dicts" + ) + + def _check_keys(rows, arrow_type, where=""): """Refuse, at any depth, a dict key that no declared struct field has, naming the path to it. @@ -176,13 +191,7 @@ def _check_keys(rows, arrow_type, where=""): f"that no declared field has; Arrow matches struct fields by exact name and " f"fills a missing one with null" ) - for row in tuples: - given = getattr(row, "_fields", None) or getattr(row, "__fields__", None) - if given is not None and set(given) == set(names) and list(given) != names: - raise ValueError( - f"{where}declared {type_repr(arrow_type)} but a row names its fields {list(given)}; a " - f"tuple's fields bind by position, so build it in the declared order or return dicts" - ) + _refuse_permuted_fields(tuples, names, arrow_type, where) for index, (name, child_type) in enumerate(fields.items()): if _carries_keys(child_type): children = [row[name] for row in dicts if name in row] @@ -319,16 +328,7 @@ def _convert(value, arrow_type): if not hasattr(value, "__len__"): # A generator would be consumed by the checks, so it is read once. value = list(value) - if isinstance(value, (list, tuple)): - for row in value: - if _is_pandas(row, "Series", "DataFrame"): - # pa.array reads a Series row by its index labels, so a sorted - # or filtered one came back reordered, and one whose labels - # were not 0..n-1 died on a bare KeyError. - raise TypeError( - f"a row is a pandas {type(row).__name__}, which pa.array reads by its labels " - f"rather than in order; hand it over as row.to_numpy() or list(row)" - ) + _refuse_pandas_rows(value) if arrow_type is not None and _carries_keys(arrow_type): _check_keys(value, arrow_type) array = pa.array(value, type=arrow_type) @@ -409,6 +409,23 @@ def _ndarray_to_arrow(value, arrow_type): return pa.array(value, type=arrow_type) +def _refuse_pandas_rows(value): + """Refuse a pandas Series or DataFrame among the rows of a list or tuple column. + + ``pa.array`` reads a Series row by its index labels, so a sorted or + filtered one came back reordered, and one whose labels were not 0..n-1 + died on a bare KeyError. + """ + if not isinstance(value, (list, tuple)): + return + for row in value: + if _is_pandas(row, "Series", "DataFrame"): + raise TypeError( + f"a row is a pandas {type(row).__name__}, which pa.array reads by its labels " + f"rather than in order; hand it over as row.to_numpy() or list(row)" + ) + + def _is_pandas(value, *names): """Whether *value* is a pandas object of one of the given class names, without importing pandas.""" cls = type(value) @@ -620,6 +637,37 @@ def _repeated_names(fields): return repeated +def _refuse_repeated_names(output_schema): + """Refuse an output_schema that names a column or a field twice, at any depth. + + A dict holds one value per name, so every copy was filled from it and + Spark died in the JVM naming neither the column nor the copy. + """ + if output_schema is None: + return + repeated = _repeated_names(list(output_schema)) + if repeated: + raise ValueError( + f"output_schema names {repeated} more than once; the dict main_func returns holds one value " + f"per name, so alias one of them" + ) + + +def _held_schema(output_schema, inferred, built): + """The schema held for the partition when types are inferred: the first batch's, once *built* agrees. + + An all-None list beside one holding values, or ints beside floats, + inferred a second schema, and Spark's writer refused it naming nothing. + """ + if output_schema is not None: + return None + if inferred is None: + return built.schema + if built.schema != inferred: + raise ValueError(_schema_drift(inferred, built.schema)) + return inferred + + def _schema_drift(first, later): """Why a batch built by inference differs from the partition's first, naming the column.""" remedy = ("Spark's writer refuses a batch whose schema differs from the first it wrote, so return the " @@ -808,15 +856,7 @@ def make_mapinarrow_func( raise TypeError( f"output_schema must be a pyarrow.Schema, not a {type(output_schema).__name__}" ) - if output_schema is not None: - repeated = _repeated_names(list(output_schema)) - if repeated: - # A dict holds one value per name, so every copy was filled from it - # and Spark died in the JVM naming neither the column nor the copy. - raise ValueError( - f"output_schema names {repeated} more than once; the dict main_func returns holds one value " - f"per name, so alias one of them" - ) + _refuse_repeated_names(output_schema) def _(iterator): inferred = None @@ -868,14 +908,7 @@ def _(iterator): bitmap_dict[col], data_dict[col] = adapted handed = _handed_bitmaps(data_dict, bitmap_dict) built = _build_batch(main_func(data_dict, bitmap_dict, broadcasts), output_schema, handed) - if output_schema is None: - # An all-None list beside one holding values, or ints beside - # floats, inferred a second schema, and Spark's writer refused - # it naming nothing. - if inferred is None: - inferred = built.schema - elif built.schema != inferred: - raise ValueError(_schema_drift(inferred, built.schema)) + inferred = _held_schema(output_schema, inferred, built) # Nothing of this batch is held across the yield: the adapted # arrays, the views and the handed-out bitmaps stayed bound in the # frame while the consumer wrote the batch out and the next one From 0b25dff8b7e6cc0cdca96e97f6f5b1f20feda60d Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:18:36 +0000 Subject: [PATCH 35/59] Build the batch-shape test's batch the way pyarrow 14 accepts pa.record_batch takes a dict from pyarrow 15 on; the CI floor cell runs 14.0.0, where the test died building its own batch. --- test/test_messages.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/test/test_messages.py b/test/test_messages.py index bea774b..968c5d7 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -244,7 +244,7 @@ def test_the_function_names_the_shape_it_takes_when_handed_a_batch_or_a_table(): # udf(batch) and udf(table) walked the columns and died on an attribute of # the first one, and mapInPandas's frames died the same way. fn = make_mapinarrow_func(lambda d, b, br: {"out": d["a"]}) - batch = pa.record_batch({"a": pa.array([1, 2, 3])}) + batch = pa.RecordBatch.from_pydict({"a": [1, 2, 3]}) for handed in (batch, pa.Table.from_batches([batch]), [batch.to_pandas()]): with pytest.raises(TypeError, match="iterator of pyarrow.RecordBatch"): list(fn(handed)) From cffb02d437112e9b6c6ea55c825428d74e3f48e3 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 00:37:35 +0000 Subject: [PATCH 36/59] Point the catalogue's eight entries at the lines this branch moved The key check's row filter, its refusal, the map entry check, the split of a Nullable, the generator read and the uniform view's read-only wrapping were all rewritten on this branch, so the eight entries that mutated their old text reported STALE instead of running; each now mutates the same term where it lives now. --- .github/scripts/mutation_guard_check.py | 34 ++++++++++++------------- 1 file changed, 16 insertions(+), 18 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 60974e2..380f8c6 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -87,10 +87,8 @@ ( "a struct dict key no field has stops being refused", "numbarrow/core/mapinarrow_factory.py", - " if unexpected_keys:\n raise ValueError(\n" - " f\"declared {type_repr(arrow_type)} but the dicts", - " if False:\n raise ValueError(\n" - " f\"declared {type_repr(arrow_type)} but the dicts", + " if unexpected_keys:\n shown = unexpected_keys[:KEYS_SHOWN]", + " if False:\n shown = unexpected_keys[:KEYS_SHOWN]", ), ( "a struct key inside a struct field stops being checked", @@ -107,20 +105,20 @@ ( "the key check stops passing over a row it cannot look inside", "numbarrow/core/mapinarrow_factory.py", - ' if hasattr(row, "__iter__")\n', - " if True\n", + " try:\n iter(row)\n except TypeError:\n continue\n", + " try:\n iter(row)\n except TypeError:\n pass\n", ), ( "the key check spreads a str or bytes row again", "numbarrow/core/mapinarrow_factory.py", - " and not isinstance(row, (str, bytes))\n", - "", + " if isinstance(row, (str, bytes, pa.Scalar)) or", + " if isinstance(row, (pa.Scalar,)) or", ), ( "the key check spreads a numeric ndarray row again", "numbarrow/core/mapinarrow_factory.py", - ' and not (isinstance(row, np.ndarray) and row.dtype.kind != "O")\n', - "", + ' or (isinstance(row, np.ndarray) and row.dtype.kind != "O"):', + ' or False:', ), ( "a struct key inside a map's keys stops being checked", @@ -228,8 +226,8 @@ ( "uniform view stops being read-only at the buffer", "numbarrow/utils/arrow_array_utils.py", - " memoryview(data_buf).toreadonly(),", - " memoryview(data_buf),", + " pa.py_buffer(memoryview(data_buf).toreadonly()),", + " pa.py_buffer(memoryview(data_buf)),", ), ( "empty string result stops being read-only", @@ -255,8 +253,8 @@ ( "a Nullable stops being split into data and bitmap", "numbarrow/core/mapinarrow_factory.py", - " if isinstance(value, Nullable):", - " if False:", + " if isinstance(value, Nullable):\n return value.data, value.bitmap", + " if False:\n return value.data, value.bitmap", ), ( "a Nullable's bitmap stops being folded in", @@ -315,8 +313,8 @@ ( "a generator output stops being read into a list before the key check", "numbarrow/core/mapinarrow_factory.py", - ' if not hasattr(value, "__len__"):', - " if False:", + ' if not hasattr(value, "__len__"):', + " if False:", ), ( "a record array with no fields stops keeping its rows", @@ -396,8 +394,8 @@ ( "map entries stop checking a pair's shape", 'numbarrow/core/mapinarrow_factory.py', - ' if isinstance(pair, (tuple, list)) and len(pair) == 2:', - ' if True:', + ' elif isinstance(pair, (tuple, list)) and len(pair) == 2:', + ' elif True:', ), ( 'a record array field failure stops naming the field', From e1a2bc1c711e6d104bd7b6e6e19ef20f39603161 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 06:07:33 +0000 Subject: [PATCH 37/59] Name the cells the pwsh steps went green on without counting the fork's matrix The comment counted nine of thirty cells, which is this fork's expanded matrix; upstream's job has one windows cell, and the sentence holds for both without the count. --- .github/workflows/numbarrow_ci.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/numbarrow_ci.yml b/.github/workflows/numbarrow_ci.yml index 8eec6b4..8c72196 100644 --- a/.github/workflows/numbarrow_ci.yml +++ b/.github/workflows/numbarrow_ci.yml @@ -39,7 +39,7 @@ jobs: - name: Install dependencies # bash, as the steps around it: under pwsh, the windows default, only the # last command's exit decides a multi-line step, so a failing pin here - # and a failing lint below went green on nine of the thirty cells. + # and a failing lint below went green on every windows cell. shell: bash run: | python -m pip install --upgrade pip From f06df83f0e40e29c827f0ac6cbcb1b2bdcd6b4b7 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 06:41:14 +0000 Subject: [PATCH 38/59] Hand the batch-shape test a pandas frame only where pandas is installed The pyarrow-range cells install no pandas, and the test built its frame with to_pandas() unconditionally, so both cells went red on an import rather than on what the test checks. --- test/test_messages.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/test/test_messages.py b/test/test_messages.py index 968c5d7..ae477ed 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -245,9 +245,16 @@ def test_the_function_names_the_shape_it_takes_when_handed_a_batch_or_a_table(): # the first one, and mapInPandas's frames died the same way. fn = make_mapinarrow_func(lambda d, b, br: {"out": d["a"]}) batch = pa.RecordBatch.from_pydict({"a": [1, 2, 3]}) - for handed in (batch, pa.Table.from_batches([batch]), [batch.to_pandas()]): + handed = [batch, pa.Table.from_batches([batch])] + try: + import pandas # noqa: F401 + handed.append([batch.to_pandas()]) + except ImportError: + # The pyarrow-range cells install no pandas. + pass + for shape in handed: with pytest.raises(TypeError, match="iterator of pyarrow.RecordBatch"): - list(fn(handed)) + list(fn(shape)) assert list(fn([batch]))[0].column("out").to_pylist() == [1, 2, 3] From 70e0c9e04d9260324de5ffcc359820e9b9d017ec Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 06:42:03 +0000 Subject: [PATCH 39/59] Read a compatible-release clause as the floor in the extras gate too A requires-python written as ~=3.12 names the same floor as >=3.12, and the gate refused it; both clauses are read now, and a set with neither still stops the gate with a message naming the field. --- .github/scripts/extras_sufficiency_check.py | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/.github/scripts/extras_sufficiency_check.py b/.github/scripts/extras_sufficiency_check.py index e526c9e..170cc4a 100755 --- a/.github/scripts/extras_sufficiency_check.py +++ b/.github/scripts/extras_sufficiency_check.py @@ -84,10 +84,16 @@ def check(repo: Path, extra: str, probe: str, python: str) -> str | None: def floor_interpreter(requires_python): - """The interpreter name for the floor of a requires-python specifier set, such as python3.12.""" - floor = re.search(r">=\s*(\d+\.\d+)", requires_python) + """The interpreter name for the floor of a requires-python specifier set, such as python3.12. + + The floor is the ``>=`` clause, or the ``~=`` compatible-release clause, + which names its floor the same way; a set with neither names no + interpreter to test on, and the gate says so and stops, as it does for an + extra with no probe. + """ + floor = re.search(r"(?:>=|~=)\s*(\d+\.\d+)", requires_python) if floor is None: - raise SystemExit(f"requires-python {requires_python!r} names no >= floor to test the extras on") + raise SystemExit(f"requires-python {requires_python!r} names no >= or ~= floor to test the extras on") return "python" + floor.group(1) From c48b84308cc2965221b77e8cdc8a704570088083 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 15:53:08 +0000 Subject: [PATCH 40/59] Say that both cache variables are read when numbarrow is first imported The README's cache section named NUMBA_CACHE_DIR and NUMBARROW_JIT_OPTIONS as the remedies without saying when they are read. Set after the import, from inside Python, NUMBA_CACHE_DIR left numba's own setting empty and the index files landed in the package's __pycache__; set before it, they landed under the directory named. The section says so now. --- README.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 496d620..cb993a5 100644 --- a/README.md +++ b/README.md @@ -141,7 +141,9 @@ as `spark-submit --py-files` ships, the functions compile without a cache and a warning names the two remedies: point `NUMBA_CACHE_DIR` at a writable directory, or set `NUMBARROW_JIT_OPTIONS='{"cache": false}'`. numba's cache index does not record the options a function was compiled with, so point -`NUMBA_CACHE_DIR` at a fresh directory when an option changes. +`NUMBA_CACHE_DIR` at a fresh directory when an option changes. Both variables +are read when numbarrow is first imported, so set them before it: one set +afterwards from inside Python changes nothing. ## PySpark Integration From 8aa1aa8d9ecea386d2688dac4152148fcfd79869 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 15:53:08 +0000 Subject: [PATCH 41/59] Refuse an empty input_columns, which handed main_func no columns A generator already used up, or a filter that matched nothing, named no column, and the function was made without a word: a main_func reading data_dict["x"] died on a bare KeyError, and one doubling whatever arrived returned a batch of no rows and no columns. An empty selection is refused when the function is made, naming both causes and None as the way to every column; the docstring says so, the test covers a list, a used-up generator and a filter, and the catalogue pins the guard. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 19 ++++++++++++++++++- test/test_messages.py | 12 ++++++++++++ 3 files changed, 36 insertions(+), 1 deletion(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 380f8c6..d1608f0 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -651,6 +651,12 @@ ' return cast_64bit_date_arrow_to_numpy_array(pa_array, np.dtype(f"datetime64[{timestamp_unit}]"))', ' return cast_64bit_date_arrow_to_numpy_array(pa_array, np.dtype("datetime64[us]"))', ), + ( + 'an empty input_columns stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' if named == []:', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 2296e30..e8d355f 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -653,6 +653,20 @@ def _refuse_repeated_names(output_schema): ) +def _refuse_empty_selection(named): + """Refuse an input_columns that names no column, such as a generator already used up. + + An empty selection handed main_func no columns: a filter that matched + nothing, or a generator already used up, gave a batch of no rows and no + columns without a word, or a bare KeyError inside main_func. + """ + if named == []: + raise ValueError( + "input_columns names no column, so main_func would be handed none; pass None for every column, " + "and check for a generator already used up or a filter that matched nothing" + ) + + def _held_schema(output_schema, inferred, built): """The schema held for the partition when types are inferred: the first batch's, once *built* agrees. @@ -788,7 +802,9 @@ def make_mapinarrow_func( differently, and a name the batch carries more than once, as an unaliased join produces, raises :class:`ValueError`. The names are read once, when the function is made, so a one-shot iterable such as - a generator serves as well as a list. + a generator serves as well as a list, and an empty one, such as a + generator already used up, is refused rather than handing ``main_func`` + no columns. :param broadcasts: optional dictionary of broadcast values :param output_schema: optional :class:`pyarrow.Schema` for the batch that is yielded. When given, the dict returned by ``main_func`` is bound to it @@ -849,6 +865,7 @@ def make_mapinarrow_func( # batch loop, a generator, map() or filter() handed in was used up by the # first batch, and every later batch then saw no columns at all. named = None if input_columns is None else list(dict.fromkeys(input_columns)) + _refuse_empty_selection(named) if output_schema is not None and not isinstance(output_schema, pa.Schema): # A PySpark StructType is the schema mapInArrow itself takes, and it # carries .names too, so one handed here got as far as the first batch diff --git a/test/test_messages.py b/test/test_messages.py index ae477ed..4b01221 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -108,6 +108,18 @@ def test_input_columns_as_a_string_is_refused(): make_mapinarrow_func(lambda d, b, br: {}, input_columns="value") +def test_an_empty_input_columns_is_refused(): + # A generator already used up, or a filter that matched nothing, named no + # column: a main_func reading data_dict["x"] died on a bare KeyError, and + # one doubling whatever arrived returned a batch of no rows and no columns + # without a word. + used_up = (name for name in ["x"]) + list(used_up) + for empty in ([], used_up, filter(lambda name: name.startswith("feat_"), ["x", "y"])): + with pytest.raises(ValueError, match="input_columns names no column"): + make_mapinarrow_func(lambda d, b, br: {}, input_columns=empty) + + def test_an_output_schema_that_is_not_a_pyarrow_schema_is_refused(): # A PySpark StructType is what the README's own example calls # output_schema, and it carries .names too, so it got as far as the first From bced5c7b74161673c86759a708edce73bec39479 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:14:05 +0000 Subject: [PATCH 42/59] Refuse a namedtuple or Row naming a field the declared struct does not have The order check compared the two sets of names, so a row with one field misnamed, Point(lon=10, latitude=50) under struct, passed it and pa.array bound the row by position: the lon value landed in lat and the latitude value in lon, with nothing raised. The guard now refuses any field names that are not the declared ones in the declared order, saying whether the names are permuted or not the declared fields at all, with a test for the misnamed row and one for a row naming an extra field, and a catalogue entry for the new term beside the existing one, which follows the rewritten line. --- .github/scripts/mutation_guard_check.py | 10 ++++++++-- numbarrow/core/mapinarrow_factory.py | 23 ++++++++++++++--------- test/test_mapinarrow_factory.py | 13 +++++++++++++ 3 files changed, 35 insertions(+), 11 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index d1608f0..5dcb2b6 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -454,8 +454,14 @@ ( 'a namedtuple naming the fields in another order stops being refused', 'numbarrow/core/mapinarrow_factory.py', - ' if given is not None and set(given) == set(names) and list(given) != names:', - ' if False:', + ' if given is None or list(given) == names:\n continue', + ' if True:\n continue', + ), + ( + 'a namedtuple naming fields the declared type does not have stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' if given is None or list(given) == names:', + ' if given is None or set(given) != set(names):', ), ( 'iterability stops being tested with iter', diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index e8d355f..6316ed1 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -146,18 +146,22 @@ def _iterable_rows(rows): def _refuse_permuted_fields(tuples, names, arrow_type, where): - """Refuse a namedtuple or pyspark Row naming the declared fields in another order. + """Refuse a namedtuple or pyspark Row whose field names are not the declared ones in the declared order. - ``pa.array`` binds a tuple's fields by position, so such a row swaps every - same-typed field without a word. + ``pa.array`` binds a tuple's fields by position, so a row naming the + declared fields in another order swapped every same-typed field without a + word, and a row naming a field the declared type does not have, a typo or + an older name, put that value into whichever field sat at its position. """ for row in tuples: given = getattr(row, "_fields", None) or getattr(row, "__fields__", None) - if given is not None and set(given) == set(names) and list(given) != names: - raise ValueError( - f"{where}declared {type_repr(arrow_type)} but a row names its fields {list(given)}; a " - f"tuple's fields bind by position, so build it in the declared order or return dicts" - ) + if given is None or list(given) == names: + continue + how = "in another order" if set(given) == set(names) else "which are not the declared fields" + raise ValueError( + f"{where}declared {type_repr(arrow_type)} but a row names its fields {list(given)}, {how}; a " + f"tuple's fields bind by position, so name them as declared, in the declared order, or return dicts" + ) def _check_keys(rows, arrow_type, where=""): @@ -171,7 +175,8 @@ def _check_keys(rows, arrow_type, where=""): another struct. A row given as a tuple, a namedtuple or a pyspark Row binds by position, so its elements are checked against the fields in declared order, and one that names the declared fields in another order, - which would swap every same-typed field without a word, is refused. + which would swap every same-typed field without a word, or names a field + the declared type does not have, is refused. """ arrow_type = _storage(arrow_type) if pa.types.is_struct(arrow_type): diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 6a0195b..5a6ed04 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -941,6 +941,19 @@ def test_a_namedtuple_row_naming_the_fields_in_another_order_is_refused(): assert got.to_pylist() == [{"x": 0, "y": 100}, {"x": 1, "y": 200}] +def test_a_namedtuple_row_naming_other_fields_is_refused_rather_than_bound_by_position(): + # The order check compared the two sets of names, so Point(lon=10, + # latitude=50) under struct, one field misnamed, passed it and + # pa.array put the lon value in lat and the latitude value in lon. + Point = collections.namedtuple("Point", ["lon", "latitude"]) + schema = pa.schema([("p", pa.struct([("lat", pa.float64()), ("lon", pa.float64())]))]) + with pytest.raises(ValueError, match=r"'p'.*\['lon', 'latitude'\].*not the declared fields.*position"): + run_outputs({"p": [Point(lon=10.0, latitude=50.0)]}, schema) + Extra = collections.namedtuple("Extra", ["lat", "lon", "alt"]) + with pytest.raises(ValueError, match=r"'p'.*\['lat', 'lon', 'alt'\].*not the declared fields"): + run_outputs({"p": [Extra(50.0, 10.0, 0.0)]}, schema) + + def test_pyarrow_scalar_rows_are_left_to_pa_array(): # From pyarrow 21 a MapScalar is a Mapping whose values is an array, so the # key check died calling it; pa.array checks a scalar row itself. From 86df08d67525cfd7a8afbe28963553d3caa88c10 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:14:39 +0000 Subject: [PATCH 43/59] Refuse a pandas Series row inside an object array as well as inside a list The row check looked only inside a list or a tuple, so the same sorted Series handed over as the one element of an object array reached pa.array and came back in label order, [3, 1, 2] for values stored as [1, 2, 3], with nothing raised. Every shape of column that reaches the check is looked through now, with a test over an object array under a declared list type and under inference, and a catalogue entry that puts the list-or-tuple gate back. --- .github/scripts/mutation_guard_check.py | 7 +++++++ numbarrow/core/mapinarrow_factory.py | 8 ++++---- test/test_mapinarrow_factory.py | 13 +++++++++++++ 3 files changed, 24 insertions(+), 4 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 5dcb2b6..91e01c0 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -663,6 +663,13 @@ ' if named == []:', ' if False:', ), + ( + 'a pandas Series row inside an object array stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' for row in value:\n if _is_pandas(row, "Series", "DataFrame"):', + ' for row in (value if isinstance(value, (list, tuple)) else ()):\n' + ' if _is_pandas(row, "Series", "DataFrame"):', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 6316ed1..0d3b399 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -415,14 +415,14 @@ def _ndarray_to_arrow(value, arrow_type): def _refuse_pandas_rows(value): - """Refuse a pandas Series or DataFrame among the rows of a list or tuple column. + """Refuse a pandas Series or DataFrame among the rows of a list, tuple or object array column. ``pa.array`` reads a Series row by its index labels, so a sorted or filtered one came back reordered, and one whose labels were not 0..n-1 - died on a bare KeyError. + died on a bare KeyError. A row inside an object array is read the same + way, and the check looked only inside a list or a tuple, so every shape + of column that reaches here is looked through. """ - if not isinstance(value, (list, tuple)): - return for row in value: if _is_pandas(row, "Series", "DataFrame"): raise TypeError( diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 5a6ed04..2190bbc 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -982,6 +982,19 @@ def test_a_pandas_series_row_is_refused_rather_than_read_by_label(): assert run_outputs({"w": chunked}).column("w").to_pylist() == ["a", "b"] +def test_a_pandas_series_row_inside_an_object_array_is_refused_too(): + # The row check looked only inside a list or a tuple, so the same sorted + # Series as the one element of an object array reached pa.array and came + # back in label order, [3, 1, 2] for values stored as [1, 2, 3]. + pd = pytest.importorskip("pandas") + rows = np.empty(1, dtype=object) + rows[0] = pd.Series([3, 1, 2]).sort_values() + with pytest.raises(TypeError, match=r"'s'.*Series.*list\(row\)"): + run_outputs({"s": rows}, pa.schema([("s", pa.list_(pa.int64()))])) + with pytest.raises(TypeError, match=r"'s'.*Series"): + run_outputs({"s": rows}) + + def test_a_key_error_from_pa_array_names_the_column_and_the_field(): # pa.array reads a UserDict row by index, and the KeyError it raised was # outside the classes the output side renamed, so it escaped as "0". From d0fcb220cf57497142550769398ce383f839786b Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:15:27 +0000 Subject: [PATCH 44/59] Name row.to_dict() as the way over for a pandas Series row under a struct The refusal of a Series row named row.to_numpy() and list(row) as the remedies, and under a declared struct column pa.array refuses both in turn, wanting a dict or a tuple; the only shape that works there, row.to_dict(), went unnamed. The remedy now follows the declared type, with a test that follows it under a struct and a catalogue entry for the term. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 12 ++++++++---- test/test_mapinarrow_factory.py | 14 ++++++++++++++ 3 files changed, 28 insertions(+), 4 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 91e01c0..bc65595 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -670,6 +670,12 @@ ' for row in (value if isinstance(value, (list, tuple)) else ()):\n' ' if _is_pandas(row, "Series", "DataFrame"):', ), + ( + 'the remedy for a pandas Series row under a struct stops naming to_dict', + 'numbarrow/core/mapinarrow_factory.py', + ' remedy = "row.to_dict()" if under_struct else "row.to_numpy() or list(row)"', + ' remedy = "row.to_numpy() or list(row)"', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 0d3b399..e341366 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -333,7 +333,7 @@ def _convert(value, arrow_type): if not hasattr(value, "__len__"): # A generator would be consumed by the checks, so it is read once. value = list(value) - _refuse_pandas_rows(value) + _refuse_pandas_rows(value, arrow_type) if arrow_type is not None and _carries_keys(arrow_type): _check_keys(value, arrow_type) array = pa.array(value, type=arrow_type) @@ -414,20 +414,24 @@ def _ndarray_to_arrow(value, arrow_type): return pa.array(value, type=arrow_type) -def _refuse_pandas_rows(value): +def _refuse_pandas_rows(value, arrow_type): """Refuse a pandas Series or DataFrame among the rows of a list, tuple or object array column. ``pa.array`` reads a Series row by its index labels, so a sorted or filtered one came back reordered, and one whose labels were not 0..n-1 died on a bare KeyError. A row inside an object array is read the same way, and the check looked only inside a list or a tuple, so every shape - of column that reaches here is looked through. + of column that reaches here is looked through. The remedy depends on the + declared type: under a struct ``pa.array`` refuses an ndarray and a list + in turn, and a Series keyed by field name goes over as ``row.to_dict()``. """ for row in value: if _is_pandas(row, "Series", "DataFrame"): + under_struct = arrow_type is not None and pa.types.is_struct(_storage(arrow_type)) + remedy = "row.to_dict()" if under_struct else "row.to_numpy() or list(row)" raise TypeError( f"a row is a pandas {type(row).__name__}, which pa.array reads by its labels " - f"rather than in order; hand it over as row.to_numpy() or list(row)" + f"rather than in order; hand it over as {remedy}" ) diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 2190bbc..1d1279f 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -995,6 +995,20 @@ def test_a_pandas_series_row_inside_an_object_array_is_refused_too(): run_outputs({"s": rows}) +def test_a_pandas_series_row_under_a_struct_names_to_dict_as_the_remedy(): + # The refusal named row.to_numpy() and list(row) as the way over, and + # under a declared struct pa.array refuses both of those in turn; a Series + # keyed by field name goes over as row.to_dict(). + pd = pytest.importorskip("pandas") + schema = pa.schema([("p", pa.struct([("lat", pa.float64()), ("lon", pa.float64())]))]) + rows = [pd.Series({"lat": 50.0, "lon": 10.0}), pd.Series({"lat": 1.0, "lon": 2.0})] + with pytest.raises(TypeError, match=r"'p'.*Series.*row\.to_dict\(\)") as excinfo: + run_outputs({"p": rows}, schema) + assert "to_numpy" not in str(excinfo.value) + got = run_outputs({"p": [row.to_dict() for row in rows]}, schema).column("p") + assert got.to_pylist() == [{"lat": 50.0, "lon": 10.0}, {"lat": 1.0, "lon": 2.0}] + + def test_a_key_error_from_pa_array_names_the_column_and_the_field(): # pa.array reads a UserDict row by index, and the KeyError it raised was # outside the classes the output side renamed, so it escaped as "0". From 21944a98a604072b162176d7302b78013e5fc607 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:16:42 +0000 Subject: [PATCH 45/59] Refuse a coarse time unit under a declared integer or string type rather than rescaling its counts Taking an hour, minute, week or day unit to seconds or days happened before the declared type was applied, so counts of [1, 2] days declared int64 came back as [86400, 172800] with nothing raised, where pa.array had refused the unit outright. The rescale now waits for a temporal declared type, or none, and under any other type the array is refused naming the unit and the two ways out; a multiplier still folds under every type, since pa.array would otherwise read the count as one of the base unit. The factory docstring says so, a test covers the three units under int64 and string, the fold under int64 and the coarse unit under date32, and a catalogue entry pins the term. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/mapinarrow_factory.py | 22 ++++++++++++++++++---- test/test_mapinarrow_factory.py | 18 ++++++++++++++++++ 3 files changed, 42 insertions(+), 4 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index bc65595..d606552 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -676,6 +676,12 @@ ' remedy = "row.to_dict()" if under_struct else "row.to_numpy() or list(row)"', ' remedy = "row.to_numpy() or list(row)"', ), + ( + 'a coarse time unit under a declared non-temporal type stops being refused', + 'numbarrow/core/mapinarrow_factory.py', + ' if target != unit and arrow_type is not None and not pa.types.is_temporal(_storage(arrow_type)):', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index e341366..1d7c08e 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -344,7 +344,7 @@ def _convert(value, arrow_type): return array -def _at_arrow_unit(value): +def _at_arrow_unit(value, arrow_type): """A datetime64 or timedelta64 array at a unit pyarrow models, with any multiplier folded in. ``pa.array`` reads numpy's base unit and ignores a multiplier, so a @@ -355,7 +355,11 @@ def _at_arrow_unit(value): becomes seconds and a week, month or year unit becomes days, exactly too, where ``pa.array`` refused them outright. A unit finer than a nanosecond, and a month or year timedelta, which has no fixed length, would not - convert exactly and are refused instead. + convert exactly and are refused instead. Under a declared type that is + not temporal, an integer or a string, the rescaled counts would go over + as the values, two days as 172800, so a coarse unit is refused there as + ``pa.array`` refused it; a multiplier still folds, since the count would + otherwise be read as one of the base unit. """ family = "datetime64" if value.dtype.kind == "M" else "timedelta64" unit, count = np.datetime_data(value.dtype) @@ -375,6 +379,12 @@ def _at_arrow_unit(value): target = "D" else: target = unit + if target != unit and arrow_type is not None and not pa.types.is_temporal(_storage(arrow_type)): + raise TypeError( + f"a {value.dtype} array under {type_repr(arrow_type)}: pyarrow has no such unit, and taking the " + f"counts to {family}[{target}] would change them; convert the array to the values you mean, or " + f"declare a temporal type" + ) if target == unit and count == 1: return value return value.astype(f"{family}[{target}]") @@ -403,7 +413,7 @@ def _ndarray_to_arrow(value, arrow_type): return pa.array(value, type=arrow_type) return pa.array(value.tolist(), type=arrow_type or pa.binary()) if kind in ("M", "m"): - value = _at_arrow_unit(value) + value = _at_arrow_unit(value, arrow_type) if value.dtype == np.dtype("datetime64[D]") and arrow_type is not None: # Under a declared timestamp or int32 ``pa.array`` reads a day-unit # array's 8-byte values as the 4-byte days of a date32, so every other @@ -833,7 +843,11 @@ def make_mapinarrow_func( numeric or datetime dtype, or a :class:`pyarrow.Array`, an integer out of the declared type's range, a float with a fraction into an integer type and a timestamp unit change that drops digits all raise - :class:`pyarrow.ArrowInvalid`. A Python list, and any other sequence + :class:`pyarrow.ArrowInvalid`. A ``datetime64`` or ``timedelta64`` + array at a unit pyarrow does not model, an hour, minute, week, month + or year, or a day-unit ``timedelta64``, is taken to seconds or days + under a temporal type and refused under any other, since its counts + would go over rescaled. A Python list, and any other sequence of Python objects, an object-dtype ndarray included, goes through ``pa.array``'s sequence converter instead: an integer out of the declared type's range still raises :class:`pyarrow.ArrowInvalid`, a diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 1d1279f..a2c7f8f 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -386,6 +386,24 @@ def test_a_declared_type_keeps_the_other_datetime64_conversions(): run_outputs({"t": np.array([1, 2], dtype="timedelta64[M]")}) +def test_a_coarse_time_unit_under_a_declared_non_temporal_type_is_refused_rather_than_rescaled(): + # Taking a day or hour unit to seconds happened before the declared type + # was applied, so counts of [1, 2] days declared int64 came back as + # [86400, 172800], where pa.array had refused the unit. + for dtype in ("timedelta64[D]", "timedelta64[h]", "datetime64[h]"): + counts = np.array([1, 2], dtype=dtype) + for declared in (pa.int64(), pa.string()): + with pytest.raises(TypeError, match=r"'n'.*would change them"): + run_outputs({"n": counts}, pa.schema([("n", declared)])) + # A multiplier still folds, since pa.array reads the count as one of the + # base unit, and a temporal declared type still takes the coarse unit. + bins = np.array([1, 2], dtype="timedelta64[5s]") + assert run_outputs({"n": bins}, pa.schema([("n", pa.int64())])).column("n").to_pylist() == [5, 10] + hours = np.array([1, 2], dtype="datetime64[h]") + dated = run_outputs({"n": hours}, pa.schema([("n", pa.date32())])).column("n") + assert dated.to_pylist() == [datetime.date(1970, 1, 1), datetime.date(1970, 1, 1)] + + def test_output_schema_refuses_a_struct_key_no_field_has(): # Arrow matches struct fields by exact name and nulls a missing one, so a # list of dicts keyed Amount/Label against amount/label used to come back From 27d3593395598850281c1147d57938a75b4dc4cb Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:17:50 +0000 Subject: [PATCH 46/59] Refuse a time64 or duration array in the 64-bit date view by type family, not by a unit's presence The view refused a type with no unit as not a date64 or timestamp, and a time64 or a duration carries a unit too, so an hour of the day and a ninety-second span were viewed as instants on 1970-01-01. The type family is tested now, with the two shapes in the view's test and a catalogue entry that puts the unit test back. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/core/adapters.py | 12 +++++++++--- test/test_adapters.py | 6 ++++++ 3 files changed, 21 insertions(+), 3 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index d606552..8cda91e 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -682,6 +682,12 @@ ' if target != unit and arrow_type is not None and not pa.types.is_temporal(_storage(arrow_type)):', ' if False:', ), + ( + 'the 64-bit date view stops refusing a time64 or duration array', + 'numbarrow/core/adapters.py', + ' elif pa.types.is_timestamp(pa_array.type):\n unit = pa_array.type.unit', + ' elif hasattr(pa_array.type, "unit"):\n unit = pa_array.type.unit', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/adapters.py b/numbarrow/core/adapters.py index d61d437..cbb89b9 100644 --- a/numbarrow/core/adapters.py +++ b/numbarrow/core/adapters.py @@ -30,11 +30,17 @@ def cast_64bit_date_arrow_to_numpy_array(pa_array: pa.Array, np_dtype: np.dtype) ``ms`` for a date64, and a timestamp's own unit. Any other unit is refused, since a view cannot rescale and read every value at the wrong instant, ``timestamp[ms]`` viewed as ``datetime64[s]`` landing in the - year 51971. The associated bitmap (if any) is also returned. + year 51971. The associated bitmap (if any) is also returned. A time64 or + a duration array carries a unit too, and read as instants a time of day or + an elapsed span landed on 1970-01-01, so the type family is checked, not + the unit's presence. """ np_dtype = np.dtype(np_dtype) - unit = "ms" if pa.types.is_date64(pa_array.type) else getattr(pa_array.type, "unit", None) - if unit is None: + if pa.types.is_date64(pa_array.type): + unit = "ms" + elif pa.types.is_timestamp(pa_array.type): + unit = pa_array.type.unit + else: raise ValueError(f"{type_repr(pa_array.type)} is not a date64 or timestamp type") if np_dtype != np.dtype(f"datetime64[{unit}]"): raise ValueError( diff --git a/test/test_adapters.py b/test/test_adapters.py index 76dea81..a6a7b88 100644 --- a/test/test_adapters.py +++ b/test/test_adapters.py @@ -182,6 +182,12 @@ def test_the_64bit_date_view_refuses_a_unit_that_is_not_the_arrays_own(): cast_64bit_date_arrow_to_numpy_array(stamps, np.dtype("datetime64[s]")) with pytest.raises(ValueError, match="not a date64 or timestamp"): cast_64bit_date_arrow_to_numpy_array(pa.array([1], type=pa.int64()), np.dtype("datetime64[s]")) + # A time64 and a duration carry a unit as well, and the refusal tested + # only for one, so an hour of the day and a ninety-second span were + # viewed as instants on 1970-01-01. + for clocked in (pa.array([3_600_000_000], type=pa.time64("us")), pa.array([90_000], type=pa.duration("ms"))): + with pytest.raises(ValueError, match="not a date64 or timestamp"): + cast_64bit_date_arrow_to_numpy_array(clocked, np.dtype(f"datetime64[{clocked.type.unit}]")) def test_a_timestamp_is_read_at_its_own_unit(): From c3c1c0b24b7840d986c4d34c87e32048964eb61c Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:17:50 +0000 Subject: [PATCH 47/59] Key a structured viewer's name by the dtype's repr, which numpy defines for every dtype The cache-name suffix hashed dtype.descr, which numpy refuses to build for a dtype with out-of-order or overlapping fields, the very dtype its own multi-field indexing, rec[["b", "a"]], hands back, so the factory raised ValueError where it had built a viewer before. The repr tells the dtypes apart as well and exists for all of them; a test builds the viewer over the reordered dtype and reads through it, and a catalogue entry puts descr back. --- .github/scripts/mutation_guard_check.py | 6 ++++++ numbarrow/utils/utils.py | 8 +++++--- test/test_utils.py | 13 +++++++++++++ 3 files changed, 24 insertions(+), 3 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 8cda91e..44b8f52 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -688,6 +688,12 @@ ' elif pa.types.is_timestamp(pa_array.type):\n unit = pa_array.type.unit', ' elif hasattr(pa_array.type, "unit"):\n unit = pa_array.type.unit', ), + ( + 'a structured dtype with out-of-order fields stops getting a viewer', + 'numbarrow/utils/utils.py', + ' name += "_" + hashlib.sha1(repr(dtype_).encode()).hexdigest()[:12]', + ' name += "_" + hashlib.sha1(repr(dtype_.descr).encode()).hexdigest()[:12]', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/utils/utils.py b/numbarrow/utils/utils.py index 8d27a47..11a1a1a 100644 --- a/numbarrow/utils/utils.py +++ b/numbarrow/utils/utils.py @@ -64,9 +64,11 @@ def viewer(ptr_as_int: int, sz: int): if dtype_.fields is not None: # numpy names every structured dtype of one itemsize void, so # two of them shared one index, and a process loading both from the - # cache ran the first one's code for the second; the description - # tells them apart. - name += "_" + hashlib.sha1(repr(dtype_.descr).encode()).hexdigest()[:12] + # cache ran the first one's code for the second; the repr tells them + # apart. The repr rather than descr, which numpy refuses to build for + # a dtype with out-of-order or overlapping fields, the very dtype its + # own multi-field indexing, rec[["b", "a"]], hands back. + name += "_" + hashlib.sha1(repr(dtype_).encode()).hexdigest()[:12] viewer.__name__ = name viewer.__qualname__ = f"{numpy_array_from_ptr_factory.__qualname__}..{name}" return jit_with_options(Array(from_dtype(dtype_), 1, "C")(intp, int64))(viewer) diff --git a/test/test_utils.py b/test/test_utils.py index 32ce8e7..f80a95b 100644 --- a/test/test_utils.py +++ b/test/test_utils.py @@ -34,3 +34,16 @@ def test_structured_dtypes_of_one_itemsize_get_viewers_of_their_own(): floats = np.array([(1.5,), (-2.5,)], dtype=[("b", " Date: Sat, 26 Sep 2026 23:21:32 +0000 Subject: [PATCH 48/59] Pin the cache fallback's narrowing to numba's no-locator error with a test The decorator re-raises any RuntimeError at decoration but numba's "no locator available", and nothing tested that term: with it dropped, every such failure compiled uncached behind a warning about the cache, and the suite stayed green. A test hands the decorator another RuntimeError and expects it back, then the no-locator one and expects the uncached recompile behind the warning; a catalogue entry drops the term. --- .github/scripts/mutation_guard_check.py | 38 +++++++++++++++++++++++++ test/test_configurations.py | 29 +++++++++++++++++++ 2 files changed, 67 insertions(+) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 44b8f52..a138cd6 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -694,6 +694,44 @@ ' name += "_" + hashlib.sha1(repr(dtype_).encode()).hexdigest()[:12]', ' name += "_" + hashlib.sha1(repr(dtype_.descr).encode()).hexdigest()[:12]', ), + ( + 'the cache fallback stops being narrowed to the no-locator error', + 'numbarrow/core/configurations.py', + ' if "no locator available" not in str(error) or not jit_options.get("cache"):', + ' if not jit_options.get("cache"):', + ), + ( + 'the repeated-name check stops looking inside a map', + 'numbarrow/core/mapinarrow_factory.py', + ' repeated.extend(_repeated_names([arrow_type.key_field, arrow_type.item_field]))', + ' pass', + ), + ( + 'the repeated-name check stops seeing through an extension type', + 'numbarrow/core/mapinarrow_factory.py', + ' arrow_type = _storage(field.type)', + ' arrow_type = field.type', + ), + ( + 'the union check stops seeing through an extension type', + 'numbarrow/utils/arrow_array_utils.py', + ' while isinstance(arrow_type, pa.BaseExtensionType):\n' + ' arrow_type = arrow_type.storage_type\n' + ' return pa.types.is_union(arrow_type)', + ' return pa.types.is_union(arrow_type)', + ), + ( + 'the key check stops seeing through an extension type when asking whether a type carries keys', + 'numbarrow/core/mapinarrow_factory.py', + ' arrow_type = _storage(arrow_type)\n if pa.types.is_struct(arrow_type):\n return True', + ' if pa.types.is_struct(arrow_type):\n return True', + ), + ( + 'the key check stops seeing through an extension type over a struct', + 'numbarrow/core/mapinarrow_factory.py', + ' arrow_type = _storage(arrow_type)\n if pa.types.is_struct(arrow_type):\n fields = ', + ' if pa.types.is_struct(arrow_type):\n fields = ', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/test/test_configurations.py b/test/test_configurations.py index 38c94d2..92b7f38 100644 --- a/test/test_configurations.py +++ b/test/test_configurations.py @@ -81,3 +81,32 @@ def test_the_refusal_names_the_requirement_and_shows_the_value(monkeypatch): monkeypatch.setenv("NUMBARROW_JIT_OPTIONS", "{cache: false}") with pytest.raises(ValueError, match=r"'\{cache: false\}' is not valid JSON"): get_jit_options() + + +def test_a_runtime_error_other_than_numbas_no_locator_one_propagates(monkeypatch): + # The fallback is narrowed to numba's "no locator available", and nothing + # tested the narrowing: with the term dropped, every RuntimeError at + # decoration compiled uncached behind a warning about the cache. + from numbarrow.core import configurations + options_seen = [] + + def njit_raising_once(message): + def njit(*signature, **options): + def decorate(func): + options_seen.append(options) + if len(options_seen) == 1: + raise RuntimeError(message) + return func + return decorate + return njit + + monkeypatch.setattr(configurations, "jit_options", {"cache": True}) + monkeypatch.setattr(configurations, "njit", njit_raising_once("some other failure at decoration")) + with pytest.raises(RuntimeError, match="some other failure at decoration"): + configurations.jit_with_options()(lambda: None) + assert options_seen == [{"cache": True}] + options_seen.clear() + monkeypatch.setattr(configurations, "njit", njit_raising_once("cannot cache function: no locator available")) + with pytest.warns(RuntimeWarning, match="compiles without a cache"): + configurations.jit_with_options()(lambda: None) + assert options_seen == [{"cache": True}, {"cache": False}] From 5e9c4fdec33782c2192a486afbb77f3e754643fe Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:21:32 +0000 Subject: [PATCH 49/59] Pin the repeated-name check's map and extension paths with tests The check looks inside a map's entries and through an extension type's storage, and only its struct and list paths were exercised: deleting either term left the suite green. The factory-time test now declares the repeated name inside a map and under an extension type, over an extension type the tests' common module defines so that every pyarrow the matrix runs exercises it, and two catalogue entries delete the terms. --- test/conftest.py | 20 ++++++++++++++++++++ test/test_messages.py | 7 +++++++ 2 files changed, 27 insertions(+) diff --git a/test/conftest.py b/test/conftest.py index 9a2086d..21edf6b 100644 --- a/test/conftest.py +++ b/test/conftest.py @@ -2,9 +2,29 @@ import shutil import sys +import pyarrow as pa import pytest +class Wrapped(pa.ExtensionType): + """An extension type over any storage, for the tests that look through one. + + ``pa.opaque`` would do from pyarrow 17; this exists on every pyarrow the + matrix runs, so the guards that unwrap an extension type are exercised on + each of them. + """ + + def __init__(self, storage_type): + super().__init__(storage_type, "test.wrapped") + + def __arrow_ext_serialize__(self): + return b"" + + @classmethod + def __arrow_ext_deserialize__(cls, storage_type, serialized): + return cls(storage_type) + + def spark_leg_required(): """True when a Spark leg that cannot run must fail the run rather than skip it. diff --git a/test/test_messages.py b/test/test_messages.py index 4b01221..1cb7583 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -21,6 +21,7 @@ from numbarrow.core.adapters import arrow_array_adapter from numbarrow.core.mapinarrow_factory import make_mapinarrow_func from numbarrow.utils.arrow_array_utils import TYPE_REPR_WIDTH, renamed, type_repr +from test.conftest import Wrapped REPO = Path(__file__).resolve().parent.parent @@ -250,6 +251,12 @@ def test_an_output_schema_naming_a_field_twice_is_refused_at_factory_time(): nested = pa.schema([("s", pa.list_(pa.struct([("x", pa.int64()), ("x", pa.float64())])))]) with pytest.raises(ValueError, match=r"\['x'\] more than once"): make_mapinarrow_func(lambda d, b, br: {}, output_schema=nested) + # The check looks inside a map's entries and through an extension type's + # storage as well, and only the struct and list paths were exercised. + twice = pa.struct([("a", pa.int64()), ("a", pa.int64())]) + for shape in (pa.map_(pa.string(), twice), Wrapped(twice)): + with pytest.raises(ValueError, match=r"\['a'\] more than once"): + make_mapinarrow_func(lambda d, b, br: {}, output_schema=pa.schema([("s", shape)])) def test_the_function_names_the_shape_it_takes_when_handed_a_batch_or_a_table(): From 028ba586e88eef57b886248e2e850933efd935ec Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:21:32 +0000 Subject: [PATCH 50/59] Pin the union check's extension unwrap with a test The refusal of a union under a struct looks through an extension type over the union, and nothing exercised that: with the unwrap deleted the suite stayed green while flatten() would again hand the struct's validity to a child that cannot take it. The union test wraps its union in an extension type as well, and a catalogue entry deletes the unwrap. --- test/test_messages.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/test/test_messages.py b/test/test_messages.py index 1cb7583..df9cf4b 100644 --- a/test/test_messages.py +++ b/test/test_messages.py @@ -214,6 +214,12 @@ def test_a_union_field_under_a_struct_is_refused_before_flatten(): struct = pa.StructArray.from_arrays([pa.array([1, 2, 3]), union], names=["ok", "u"], mask=mask) with pytest.raises(NotImplementedError, match=r"struct field 'u'.*union"): arrow_array_adapter(struct) + # An extension type over the union shows no union of its own, so the + # check looks through it; nothing exercised that. + wrapped = pa.ExtensionArray.from_storage(Wrapped(union.type), union) + struct = pa.StructArray.from_arrays([pa.array([1, 2, 3]), wrapped], names=["ok", "u"]) + with pytest.raises(NotImplementedError, match=r"struct field 'u'.*union"): + arrow_array_adapter(struct) def test_the_unexpected_keys_listing_is_cut_with_a_count(): From d5c2768b8b695f511ec2fd48061191ec76ba40f1 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:21:32 +0000 Subject: [PATCH 51/59] Pin the key check's extension unwrap with a test The key check looks through an extension type over a struct twice, when it asks whether a type carries keys and when it reads the fields, and nothing exercised either: with one unwrap deleted the suite stayed green while pa.array built a column of nulls for dicts keyed Amount under an extension type over struct. The any-depth test gains that shape, and two catalogue entries delete the unwraps one at a time. --- test/test_mapinarrow_factory.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index a2c7f8f..13c9d6c 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -10,6 +10,7 @@ from numbarrow.core.is_null import is_null from numbarrow.core.mapinarrow_factory import Nullable, make_mapinarrow_func +from test.conftest import Wrapped def run_batch(batch, input_columns=None): @@ -462,6 +463,9 @@ def test_a_struct_key_no_field_has_is_refused_at_any_depth(): "map of structs": (pa.map_(pa.string(), inner), [{"k": {"Amount": 1}}]), "map of structs from pairs": (pa.map_(pa.string(), inner), [[("k", {"Amount": 1})]]), "struct-keyed map": (pa.map_(inner, pa.int64()), [[({"Amount": 1}, 5)]]), + # An extension type carries its storage's fields, and the check looks + # through it on both of its questions; nothing exercised either. + "extension over struct": (Wrapped(inner), [{"Amount": 1}]), } for label, (declared_type, value) in cases.items(): with pytest.raises(ValueError, match="Amount"): From afbd0e930c4d0b4d13cd106ac9515f3e3e24a8e4 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:25:08 +0000 Subject: [PATCH 52/59] Say that NUMBA_CACHE_DIR is read when numba is imported, not when numbarrow is The README said both cache variables are read when numbarrow is first imported. NUMBARROW_JIT_OPTIONS is; NUMBA_CACHE_DIR is numba's, read when numba is imported, which another library may have done earlier, and again only when numba compiles something, so a directory set between the two imports missed the first function numbarrow compiles, and all of them when the old cache was warm. The paragraph says so now, and a test imports numba under one directory, switches to another and imports numbarrow, expecting the first function's index under the first. --- README.md | 9 ++++++--- test/test_docs_match_code.py | 25 +++++++++++++++++++++++++ 2 files changed, 31 insertions(+), 3 deletions(-) diff --git a/README.md b/README.md index cb993a5..8538f40 100644 --- a/README.md +++ b/README.md @@ -141,9 +141,12 @@ as `spark-submit --py-files` ships, the functions compile without a cache and a warning names the two remedies: point `NUMBA_CACHE_DIR` at a writable directory, or set `NUMBARROW_JIT_OPTIONS='{"cache": false}'`. numba's cache index does not record the options a function was compiled with, so point -`NUMBA_CACHE_DIR` at a fresh directory when an option changes. Both variables -are read when numbarrow is first imported, so set them before it: one set -afterwards from inside Python changes nothing. +`NUMBA_CACHE_DIR` at a fresh directory when an option changes. Both are read +at import: `NUMBARROW_JIT_OPTIONS` when numbarrow is first imported and +`NUMBA_CACHE_DIR` when numba is, which another library may have done earlier, +so set both before either. numba re-reads its environment only when it +compiles something, so a directory set after its import misses at least the +first function numbarrow compiles, and all of them when the old cache is warm. ## PySpark Integration diff --git a/test/test_docs_match_code.py b/test/test_docs_match_code.py index ea819e7..eb27b7c 100644 --- a/test/test_docs_match_code.py +++ b/test/test_docs_match_code.py @@ -8,7 +8,10 @@ that stops it recurring. """ import datetime +import os import re +import subprocess +import sys from pathlib import Path import numpy as np @@ -187,6 +190,28 @@ def test_the_readme_names_the_pandas_floor_the_extras_declare(): assert "1.5.0" not in README_TEXT +def test_the_cache_dir_is_read_at_numbas_import_as_the_readme_says(tmp_path): + # The README said both variables are read when numbarrow is first + # imported. NUMBA_CACHE_DIR is numba's, read when numba is imported and + # again only when it compiles something, so a directory set between the + # two imports missed the first function numbarrow compiles, and all of + # them when the old cache was warm. + assert "`NUMBA_CACHE_DIR` when numba is" in README_TEXT + assert "misses at least the first function numbarrow compiles" in README_TEXT + first, late = tmp_path / "first", tmp_path / "late" + src = (f"import os; os.environ['NUMBA_CACHE_DIR'] = {str(first)!r}\n" + "import numba\n" + f"os.environ['NUMBA_CACHE_DIR'] = {str(late)!r}\n" + "import numbarrow.core.is_null\n") + env = {key: value for key, value in os.environ.items() if key not in ("NUMBA_CACHE_DIR", "NUMBARROW_JIT_OPTIONS")} + env["PYTHONPATH"] = str(README.parent) + run = subprocess.run([sys.executable, "-c", src], capture_output=True, text=True, env=env, cwd=str(tmp_path)) + assert run.returncode == 0, run.stderr + indexed = {where: sorted(path.name.split(".")[1].split("-")[0] for path in (tmp_path / where).rglob("*.nbi")) + for where in ("first", "late")} + assert "is_null" in indexed["first"] and "is_null" not in indexed["late"], indexed + + def _output_column(value): batch = pa.RecordBatch.from_arrays([pa.array([0, 0], type=pa.int64())], names=["c"]) fn = make_mapinarrow_func(lambda data, bitmap, broadcasts: {"o": value}, input_columns=["c"]) From b8752a3de9ea32b0443ef4e4792bc9c8a801f254 Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sat, 26 Sep 2026 23:25:08 +0000 Subject: [PATCH 53/59] Say that a float under a declared decimal type is refused, not rounded The factory docstring listed a float into decimal among the lossy conversions that pass without a word, rounding to the declared scale. Both routes refuse it: the sequence converter takes an int or a Decimal, and the typed one refuses a float dtype. The docstring says so, and a test holds it to that on both routes and passes a Decimal through. --- numbarrow/core/mapinarrow_factory.py | 12 +++++++----- test/test_docs_match_code.py | 15 +++++++++++++++ 2 files changed, 22 insertions(+), 5 deletions(-) diff --git a/numbarrow/core/mapinarrow_factory.py b/numbarrow/core/mapinarrow_factory.py index 1d7c08e..7e55367 100644 --- a/numbarrow/core/mapinarrow_factory.py +++ b/numbarrow/core/mapinarrow_factory.py @@ -857,11 +857,13 @@ def make_mapinarrow_func( timestamp's extra digits are dropped silently. So the lossy conversions that pass without a word are a timestamp into ``date32`` or ``date64``, which floors to the day, a timestamp into ``time32`` - or ``time64``, which drops the date, a float into ``decimal``, which - rounds to the declared scale, a number into ``bool``, which is true - for anything but zero, a float into a narrower float, which overflows - to ``inf``, and, from a list alone, a fraction into an integer type - and a timestamp unit change that drops digits. + or ``time64``, which drops the date, a number into ``bool``, which is + true for anything but zero, a float into a narrower float, which + overflows to ``inf``, and, from a list alone, a fraction into an + integer type and a timestamp unit change that drops digits. A float + under a ``decimal`` type is refused on both routes: the sequence + converter takes an int or a :class:`decimal.Decimal`, and the typed + one refuses a float dtype. Left as ``None`` the batch is built from the dict alone: insertion order decides, and every type is inferred from the value, so a unicode diff --git a/test/test_docs_match_code.py b/test/test_docs_match_code.py index eb27b7c..9a581cf 100644 --- a/test/test_docs_match_code.py +++ b/test/test_docs_match_code.py @@ -8,6 +8,7 @@ that stops it recurring. """ import datetime +import decimal import os import re import subprocess @@ -153,6 +154,20 @@ def test_the_truncations_the_docstring_admits_for_a_list_are_the_ones_it_makes() ] +def test_a_float_under_a_declared_decimal_is_refused_as_the_docstring_says(): + # The docstring listed a float into decimal among the conversions that + # pass without a word, rounding to the declared scale; both routes refuse + # it, the sequence converter wanting an int or a Decimal. + assert "A float under a ``decimal`` type is refused on both routes" in FACTORY_DOC + assert "a float into ``decimal``" not in FACTORY_DOC + with pytest.raises(pa.ArrowException): + _one_output_column([1.236], pa.decimal128(6, 2)) + with pytest.raises(pa.ArrowException): + _one_output_column(np.array([1.236]), pa.decimal128(6, 2)) + exact = _one_output_column([decimal.Decimal("1.23")], pa.decimal128(6, 2)) + assert exact.to_pylist() == [decimal.Decimal("1.23")] + + def _inferred_output_column(value): """The column a UDF returning *value* with no output_schema yields.""" batch = pa.RecordBatch.from_pydict({"v": [1.0]}) From d3638ee8a1bf9e480652b3866ddda43cd8e1926d Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sun, 27 Sep 2026 00:23:30 +0000 Subject: [PATCH 54/59] Pin the extras gate's floor parsing with a test The gate reads its interpreter from the floor of requires-python, and the invariants workflow runs it against this tree's own ">=3.12" alone, so the ~= clause it also accepts and the refusal of a set naming neither had no run behind them. The test loads the script by its path and skips where the tree carries no .github, as the catalogue's copies do not. --- test/test_extras_gate.py | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) create mode 100644 test/test_extras_gate.py diff --git a/test/test_extras_gate.py b/test/test_extras_gate.py new file mode 100644 index 0000000..5e3881e --- /dev/null +++ b/test/test_extras_gate.py @@ -0,0 +1,29 @@ +"""The extras gate's own parsing, which the workflow runs only against this tree's requires-python.""" +import importlib.util +from pathlib import Path + +import pytest + +SCRIPT = Path(__file__).resolve().parent.parent / ".github" / "scripts" / "extras_sufficiency_check.py" + + +def _gate(): + if not SCRIPT.exists(): + pytest.skip("the extras gate is not in this tree") + spec = importlib.util.spec_from_file_location("extras_sufficiency_check", SCRIPT) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_the_floor_interpreter_comes_from_a_ge_or_a_compatible_release_clause(): + # The gate reads its interpreter from the floor of requires-python, and + # the workflow runs it against this tree's own ">=3.12" alone, so the ~= + # clause it also accepts and the refusal of a set naming neither had no + # run behind them. + gate = _gate() + assert gate.floor_interpreter(">=3.12") == "python3.12" + assert gate.floor_interpreter(">= 3.12, <3.14") == "python3.12" + assert gate.floor_interpreter("~=3.12") == "python3.12" + with pytest.raises(SystemExit, match="names no >= or ~= floor"): + gate.floor_interpreter("==3.12.*") From 313ebe8f051a2eb09076ae8bc44f93d4d9bb508c Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Sun, 27 Sep 2026 00:56:23 +0000 Subject: [PATCH 55/59] Skip the extras gate's test where the script cannot load The gate imports tomllib, which arrived in Python 3.11, and the matrix runs 3.10 cells below the declared floor on purpose, so the test errored there with ModuleNotFoundError and fail-fast cancelled the rest of the matrix. The test now skips where the script cannot load, as it does where the tree carries no .github. --- test/test_extras_gate.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/test/test_extras_gate.py b/test/test_extras_gate.py index 5e3881e..d36d1cf 100644 --- a/test/test_extras_gate.py +++ b/test/test_extras_gate.py @@ -12,7 +12,12 @@ def _gate(): pytest.skip("the extras gate is not in this tree") spec = importlib.util.spec_from_file_location("extras_sufficiency_check", SCRIPT) module = importlib.util.module_from_spec(spec) - spec.loader.exec_module(module) + try: + spec.loader.exec_module(module) + except ImportError as exc: + # The gate runs on the floor interpreter, and CI also tests below the + # floor, where a module the script needs, tomllib on 3.10, is missing. + pytest.skip(f"the extras gate cannot load here: {exc}") return module From 83713bb18ce1b4d4e781a4550d0a37b49820f41a Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Mon, 28 Sep 2026 19:23:21 +0000 Subject: [PATCH 56/59] Give jit_with_options the one signature every caller passes; a second one reached numba as locals --- .github/scripts/mutation_guard_check.py | 2 +- numbarrow/core/configurations.py | 6 +++--- test/test_configurations.py | 5 +++-- 3 files changed, 7 insertions(+), 6 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index a138cd6..a864462 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -215,7 +215,7 @@ "numbarrow/core/is_null.py", '@jit_with_options(boolean(int64, Optional(Array(uint8, 1, "C", readonly=True)),\n' ' Optional(Array(uint8, 1, "C", readonly=True))))', - "@jit_with_options()", + "@jit_with_options(None)", ), ( "viewers stop getting a cache name of their own", diff --git a/numbarrow/core/configurations.py b/numbarrow/core/configurations.py index 947131f..eb7f8f5 100644 --- a/numbarrow/core/configurations.py +++ b/numbarrow/core/configurations.py @@ -62,7 +62,7 @@ def get_jit_options(): jit_options = get_jit_options() -def jit_with_options(*signature): +def jit_with_options(signature): """``njit`` under the options ``NUMBARROW_JIT_OPTIONS`` gives, compiling uncached where no cache can be written. numba sets a cached function up when it is decorated, and raises ``RuntimeError`` there when no cache @@ -73,7 +73,7 @@ def jit_with_options(*signature): """ def decorate(func): try: - return njit(*signature, **jit_options)(func) + return njit(signature, **jit_options)(func) except RuntimeError as error: if "no locator available" not in str(error) or not jit_options.get("cache"): raise @@ -83,5 +83,5 @@ def decorate(func): f"turn caching off and silence this warning", RuntimeWarning, stacklevel=2, ) - return njit(*signature, **{**jit_options, "cache": False})(func) + return njit(signature, **{**jit_options, "cache": False})(func) return decorate diff --git a/test/test_configurations.py b/test/test_configurations.py index 92b7f38..30b96e1 100644 --- a/test/test_configurations.py +++ b/test/test_configurations.py @@ -6,6 +6,7 @@ from pathlib import Path import pytest +from numba import void from numbarrow.core.configurations import get_jit_options, invalid_jit_options_err @@ -103,10 +104,10 @@ def decorate(func): monkeypatch.setattr(configurations, "jit_options", {"cache": True}) monkeypatch.setattr(configurations, "njit", njit_raising_once("some other failure at decoration")) with pytest.raises(RuntimeError, match="some other failure at decoration"): - configurations.jit_with_options()(lambda: None) + configurations.jit_with_options(void())(lambda: None) assert options_seen == [{"cache": True}] options_seen.clear() monkeypatch.setattr(configurations, "njit", njit_raising_once("cannot cache function: no locator available")) with pytest.warns(RuntimeWarning, match="compiles without a cache"): - configurations.jit_with_options()(lambda: None) + configurations.jit_with_options(void())(lambda: None) assert options_seen == [{"cache": True}, {"cache": False}] From 93eb2c9432d612b6f50c07fc0276e9c446809dad Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Mon, 28 Sep 2026 19:23:21 +0000 Subject: [PATCH 57/59] Lead the one-shot input_columns test's comment with what it checks --- test/test_mapinarrow_factory.py | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/test/test_mapinarrow_factory.py b/test/test_mapinarrow_factory.py index 13c9d6c..00a138d 100644 --- a/test/test_mapinarrow_factory.py +++ b/test/test_mapinarrow_factory.py @@ -96,8 +96,9 @@ def test_input_columns_selects_only_the_named_columns(): def test_input_columns_is_read_once_so_a_generator_serves_every_batch(): - # The names were read from the argument inside the batch loop, so a - # generator, map() or filter() was used up by the first batch and every + # A generator, map() or filter() given as input_columns must reach the UDF + # in every batch, not only the first. The names used to be read from the + # argument inside the batch loop, so the first batch used them up and every # later batch was adapted with no columns: the UDF died on a bare KeyError. batches = [pa.RecordBatch.from_pydict({"x": [1, 2], "y": [0, 0]}), pa.RecordBatch.from_pydict({"x": [3], "y": [0]})] From 105b9ce91110d21430d7dd1766cf4a6ff19385bd Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Tue, 29 Sep 2026 13:33:32 +0000 Subject: [PATCH 58/59] Name a remedy that works in the cache warning for an archive; NUMBA_CACHE_DIR has no effect there --- .github/scripts/mutation_guard_check.py | 12 ++++ README.md | 18 +++-- numbarrow/core/configurations.py | 24 +++++-- test/test_cache.py | 94 ++++++++++++++++++++++--- 4 files changed, 125 insertions(+), 23 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index a864462..259c02f 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -732,6 +732,18 @@ ' arrow_type = _storage(arrow_type)\n if pa.types.is_struct(arrow_type):\n fields = ', ' if pa.types.is_struct(arrow_type):\n fields = ', ), + ( + 'the cache warning offers NUMBA_CACHE_DIR to an import from an archive', + 'numbarrow/core/configurations.py', + ' if os.path.exists(inspect.getfile(func)):', + ' if True:', + ), + ( + 'the cache warning stops naming NUMBA_CACHE_DIR for a source file on disk', + 'numbarrow/core/configurations.py', + ' if os.path.exists(inspect.getfile(func)):', + ' if False:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/README.md b/README.md index 8538f40..6456a4f 100644 --- a/README.md +++ b/README.md @@ -135,13 +135,17 @@ to `is_null_struct`. numbarrow compiles its adapters with numba on first import, under the options `NUMBARROW_JIT_OPTIONS` gives as a JSON object; unset, that is `{"cache": true}`, so the compiled code is written to numba's on-disk cache, next to the -package or under `NUMBA_CACHE_DIR`. Where no cache location can be written, a -read-only install or an import from an `.egg`, `.whl` or `.pyz` archive such -as `spark-submit --py-files` ships, the functions compile without a cache and -a warning names the two remedies: point `NUMBA_CACHE_DIR` at a writable -directory, or set `NUMBARROW_JIT_OPTIONS='{"cache": false}'`. numba's cache -index does not record the options a function was compiled with, so point -`NUMBA_CACHE_DIR` at a fresh directory when an option changes. Both are read +package or under `NUMBA_CACHE_DIR`. Where no cache location can be written, +the functions compile without a cache and a warning names the remedy. For a +read-only install that is `NUMBA_CACHE_DIR`, pointed at a writable directory. +For an import from an `.egg`, `.whl` or `.pyz` archive such as `spark-submit +--py-files` ships, `NUMBA_CACHE_DIR` has no effect, since numba reads it only +for a source file on disk: install numbarrow unpacked, or ship it as a `.zip`, +which numba 0.61 and later cache in the user's cache directory. Either way +`NUMBARROW_JIT_OPTIONS='{"cache": false}'` turns caching off and silences the +warning. numba's cache index does not record the options a function was +compiled with, so point `NUMBA_CACHE_DIR` at a fresh directory when an option +changes. Both are read at import: `NUMBARROW_JIT_OPTIONS` when numbarrow is first imported and `NUMBA_CACHE_DIR` when numba is, which another library may have done earlier, so set both before either. numba re-reads its environment only when it diff --git a/numbarrow/core/configurations.py b/numbarrow/core/configurations.py index eb7f8f5..6409d17 100644 --- a/numbarrow/core/configurations.py +++ b/numbarrow/core/configurations.py @@ -2,6 +2,7 @@ Default configuration options for Numba JIT compilation used throughout numbarrow. """ +import inspect import os import json import warnings @@ -68,8 +69,11 @@ def jit_with_options(signature): numba sets a cached function up when it is decorated, and raises ``RuntimeError`` there when no cache location can be written: a read-only install, an unwritable ``site-packages`` and user cache directory, or an import from an ``.egg``, ``.whl`` or ``.pyz`` archive, which Spark's ``--py-files`` ships. Nothing then - named the way out. Such a function compiles without a cache, with a warning naming ``NUMBA_CACHE_DIR`` and - ``NUMBARROW_JIT_OPTIONS='{"cache": false}'``. A write that fails later, on a full disk, is numba's own error. + named the way out. Such a function compiles without a cache, with a warning naming the remedy: + ``NUMBA_CACHE_DIR`` for a source file on disk, and for an archive, where numba never reads it, an unpacked + install or a ``.zip``, which numba 0.61 and later cache in the user's cache directory. Either way + ``NUMBARROW_JIT_OPTIONS='{"cache": false}'`` turns caching off and silences the warning. A write that fails + later, on a full disk, is numba's own error. """ def decorate(func): try: @@ -77,10 +81,20 @@ def decorate(func): except RuntimeError as error: if "no locator available" not in str(error) or not jit_options.get("cache"): raise + silence = "NUMBARROW_JIT_OPTIONS='{\"cache\": false}' to turn caching off and silence this warning" + if os.path.exists(inspect.getfile(func)): + remedy = f"Set NUMBA_CACHE_DIR to a writable directory, or {silence}" + else: + # Every location numba reads NUMBA_CACHE_DIR for needs the source + # file on disk, so for an archive the warning named a remedy + # that changed nothing. + remedy = ( + "NUMBA_CACHE_DIR has no effect here, because the source is not a file on disk: to cache, " + "install numbarrow unpacked or import it from a .zip, which numba 0.61 and later cache in the " + f"user's cache directory. Set {silence}" + ) warnings.warn( - f"numba cannot cache {func.__qualname__} here ({error}); it compiles without a cache. Set " - f"NUMBA_CACHE_DIR to a writable directory, or NUMBARROW_JIT_OPTIONS='{{\"cache\": false}}' to " - f"turn caching off and silence this warning", + f"numba cannot cache {func.__qualname__} here ({error}); it compiles without a cache. {remedy}", RuntimeWarning, stacklevel=2, ) return njit(signature, **{**jit_options, "cache": False})(func) diff --git a/test/test_cache.py b/test/test_cache.py index 01cb67f..2b26c45 100644 --- a/test/test_cache.py +++ b/test/test_cache.py @@ -19,13 +19,22 @@ """ import json import os +import shutil import subprocess import sys import zipfile from pathlib import Path +import pytest + REPO = Path(__file__).resolve().parent.parent +# chmod takes write access from a directory neither on Windows nor from root. +needs_a_directory_it_cannot_write = pytest.mark.skipif( + os.name == "nt" or os.geteuid() == 0, reason="needs a directory this user cannot write to") + +IMPORT_AND_SHOW_FILE = "import numbarrow.core.adapters as a; print(a.__file__)" + IMPORT_AND_VIEW = ( "import numpy as np\n" "from numbarrow.utils.utils import arrays_viewers\n" @@ -174,27 +183,90 @@ def test_a_cold_cache_survives_a_concurrent_first_import_of_is_null_struct(tmp_p assert out.returncode == 0, out.stderr -def test_an_import_from_an_archive_compiles_uncached_with_a_warning_naming_the_remedy(tmp_path): +def _archive(path): + """numbarrow's modules zipped into ``path``, which goes on PYTHONPATH as it is.""" + with zipfile.ZipFile(path, "w") as zipped: + for source in sorted((REPO / "numbarrow").rglob("*.py")): + zipped.write(source, str(source.relative_to(REPO))) + return path + + +@pytest.mark.parametrize("name", ["numbarrow-0.0.0-py3.12.egg", "numbarrow-0.0.0-py3-none-any.whl"]) +def test_an_import_from_an_archive_compiles_uncached_with_a_warning_naming_the_remedy(tmp_path, name): # numba's cache locators need the source file on disk, so an import from # an .egg, .whl or .pyz archive, which Spark's --py-files ships, raised - # RuntimeError at decoration, naming neither NUMBA_CACHE_DIR nor the - # option that turns caching off. - archive = tmp_path / "numbarrow-0.0.0-py3.12.egg" - with zipfile.ZipFile(archive, "w") as zipped: - for path in sorted((REPO / "numbarrow").rglob("*.py")): - zipped.write(path, str(path.relative_to(REPO))) + # RuntimeError at decoration, naming neither the way to a cache nor the + # option that turns caching off. NUMBA_CACHE_DIR is no way to one: it is + # set and writable here, the warning fires all the same and nothing is + # written there, so the warning says so instead of offering it. + archive = _archive(tmp_path / name) env = dict(os.environ, PYTHONPATH=str(archive), NUMBA_CACHE_DIR=str(tmp_path / "cache")) env.pop("NUMBARROW_JIT_OPTIONS", None) - probe = "import numbarrow.core.adapters as a; print(a.__file__)" - run = subprocess.run([sys.executable, "-W", "always", "-c", probe], + run = subprocess.run([sys.executable, "-W", "always", "-c", IMPORT_AND_SHOW_FILE], capture_output=True, text=True, env=env, cwd=str(tmp_path)) assert run.returncode == 0 and str(archive) in run.stdout, run.stderr - assert "NUMBA_CACHE_DIR" in run.stderr and "compiles without a cache" in run.stderr - quiet = subprocess.run([sys.executable, "-W", "error", "-c", probe], capture_output=True, text=True, + assert "compiles without a cache" in run.stderr + assert "NUMBA_CACHE_DIR has no effect here" in run.stderr and "Set NUMBA_CACHE_DIR" not in run.stderr + assert _index_files(tmp_path / "cache") == [] + quiet = subprocess.run([sys.executable, "-W", "error", "-c", IMPORT_AND_SHOW_FILE], + capture_output=True, text=True, env=dict(env, NUMBARROW_JIT_OPTIONS='{"cache": false}'), cwd=str(tmp_path)) assert quiet.returncode == 0, quiet.stderr +def test_a_zip_import_is_cached_by_numba_from_0_61(tmp_path): + # The warning and the README send an archive's user to a .zip, which + # numba caches from 0.61 on, in the user's cache directory whatever + # NUMBA_CACHE_DIR says. Before that a .zip is one more archive. + import numba + archive = _archive(tmp_path / "numbarrow.zip") + home = tmp_path / "home" + env = dict(os.environ, PYTHONPATH=str(archive), HOME=str(home), XDG_CACHE_HOME=str(home / "cache"), + NUMBA_CACHE_DIR=str(tmp_path / "cache")) + env.pop("NUMBARROW_JIT_OPTIONS", None) + run = subprocess.run([sys.executable, "-W", "always", "-c", IMPORT_AND_SHOW_FILE], + capture_output=True, text=True, env=env, cwd=str(tmp_path)) + assert run.returncode == 0 and str(archive) in run.stdout, run.stderr + cached = tuple(int(part) for part in numba.__version__.split(".")[:2]) >= (0, 61) + assert ("compiles without a cache" not in run.stderr) == cached, run.stderr + assert _index_files(tmp_path / "cache") == [] + if os.name != "nt": + # On Windows numba asks the system for the user's cache directory, and + # no variable set here moves it. + assert bool(_index_files(home)) == cached + + +@needs_a_directory_it_cannot_write +def test_a_read_only_install_warns_naming_numba_cache_dir_and_setting_it_caches(tmp_path): + # The other way to have no cache location: the source is on disk, and + # neither its directory nor the user's cache directory can be written. + # NUMBA_CACHE_DIR is the remedy there, and nothing showed that the warning + # names it or that setting it works. + site = tmp_path / "site" + shutil.copytree(REPO / "numbarrow", site / "numbarrow", ignore=shutil.ignore_patterns("__pycache__")) + home = tmp_path / "home" + home.mkdir() + read_only = [home, *(path for path in site.rglob("*") if path.is_dir())] + for path in read_only: + path.chmod(0o555) + try: + env = dict(os.environ, PYTHONPATH=str(site), HOME=str(home), XDG_CACHE_HOME=str(home / "cache")) + env.pop("NUMBARROW_JIT_OPTIONS", None) + env.pop("NUMBA_CACHE_DIR", None) + run = subprocess.run([sys.executable, "-W", "always", "-c", IMPORT_AND_SHOW_FILE], + capture_output=True, text=True, env=env, cwd=str(tmp_path)) + assert run.returncode == 0 and str(site) in run.stdout, run.stderr + assert "compiles without a cache" in run.stderr and "Set NUMBA_CACHE_DIR" in run.stderr + cured = subprocess.run([sys.executable, "-W", "error", "-c", IMPORT_AND_SHOW_FILE], + capture_output=True, text=True, + env=dict(env, NUMBA_CACHE_DIR=str(tmp_path / "cache")), cwd=str(tmp_path)) + assert cured.returncode == 0, cured.stderr + assert _index_files(tmp_path / "cache") + finally: + for path in read_only: + path.chmod(0o755) + + CHECK_BOUNDS = ( "import numpy as np\n" "from numbarrow.core.is_null import is_null, unpack_booleans\n" From f45b510cb5dc8a3d3535558561585a0fc99f58cc Mon Sep 17 00:00:00 2001 From: "Nelson, Erik" Date: Tue, 29 Sep 2026 13:35:19 +0000 Subject: [PATCH 59/59] Compile uncached when the cache numba picked for a .zip cannot be written; the import died on its OSError --- .github/scripts/mutation_guard_check.py | 24 ++++++++++++++++++--- numbarrow/core/configurations.py | 18 +++++++++------- test/test_cache.py | 21 +++++++++++++++++++ test/test_configurations.py | 28 +++++++++++++++++++++++++ 4 files changed, 80 insertions(+), 11 deletions(-) diff --git a/.github/scripts/mutation_guard_check.py b/.github/scripts/mutation_guard_check.py index 259c02f..0380e81 100755 --- a/.github/scripts/mutation_guard_check.py +++ b/.github/scripts/mutation_guard_check.py @@ -633,7 +633,7 @@ ( 'a function that numba cannot cache stops compiling uncached', 'numbarrow/core/configurations.py', - ' if "no locator available" not in str(error) or not jit_options.get("cache"):\n' + ' if not cache_failed or not jit_options.get("cache"):\n' ' raise', ' raise', ), @@ -697,8 +697,8 @@ ( 'the cache fallback stops being narrowed to the no-locator error', 'numbarrow/core/configurations.py', - ' if "no locator available" not in str(error) or not jit_options.get("cache"):', - ' if not jit_options.get("cache"):', + ' cache_failed = isinstance(error, OSError) or "no locator available" in str(error)', + ' cache_failed = True', ), ( 'the repeated-name check stops looking inside a map', @@ -744,6 +744,24 @@ ' if os.path.exists(inspect.getfile(func)):', ' if False:', ), + ( + 'a cache write that fails at decoration stops being caught', + 'numbarrow/core/configurations.py', + ' except (RuntimeError, OSError) as error:', + ' except RuntimeError as error:', + ), + ( + 'a cache write that fails at decoration stops compiling uncached', + 'numbarrow/core/configurations.py', + ' cache_failed = isinstance(error, OSError) or "no locator available" in str(error)', + ' cache_failed = "no locator available" in str(error)', + ), + ( + 'an error at decoration with caching off stops reaching the caller', + 'numbarrow/core/configurations.py', + ' if not cache_failed or not jit_options.get("cache"):', + ' if not cache_failed:', + ), ] COPY = ["numbarrow", "test", "README.md", "pyproject.toml", "docs"] diff --git a/numbarrow/core/configurations.py b/numbarrow/core/configurations.py index 6409d17..928d4ca 100644 --- a/numbarrow/core/configurations.py +++ b/numbarrow/core/configurations.py @@ -68,18 +68,20 @@ def jit_with_options(signature): numba sets a cached function up when it is decorated, and raises ``RuntimeError`` there when no cache location can be written: a read-only install, an unwritable ``site-packages`` and user cache directory, or - an import from an ``.egg``, ``.whl`` or ``.pyz`` archive, which Spark's ``--py-files`` ships. Nothing then - named the way out. Such a function compiles without a cache, with a warning naming the remedy: - ``NUMBA_CACHE_DIR`` for a source file on disk, and for an archive, where numba never reads it, an unpacked - install or a ``.zip``, which numba 0.61 and later cache in the user's cache directory. Either way - ``NUMBARROW_JIT_OPTIONS='{"cache": false}'`` turns caching off and silences the warning. A write that fails - later, on a full disk, is numba's own error. + an import from an ``.egg``, ``.whl`` or ``.pyz`` archive, which Spark's ``--py-files`` ships. For a ``.zip`` + it takes the user's cache directory without checking that it can be written, and the first write raises + ``OSError`` instead. Nothing then named the way out. Such a function compiles without a cache, with a + warning naming the remedy: ``NUMBA_CACHE_DIR`` for a source file on disk, and for an archive, where numba + never reads it, an unpacked install or a ``.zip``, which numba 0.61 and later cache in the user's cache + directory. Either way ``NUMBARROW_JIT_OPTIONS='{"cache": false}'`` turns caching off and silences the + warning. An error that is not the cache's comes back from the uncached compile. """ def decorate(func): try: return njit(signature, **jit_options)(func) - except RuntimeError as error: - if "no locator available" not in str(error) or not jit_options.get("cache"): + except (RuntimeError, OSError) as error: + cache_failed = isinstance(error, OSError) or "no locator available" in str(error) + if not cache_failed or not jit_options.get("cache"): raise silence = "NUMBARROW_JIT_OPTIONS='{\"cache\": false}' to turn caching off and silence this warning" if os.path.exists(inspect.getfile(func)): diff --git a/test/test_cache.py b/test/test_cache.py index 2b26c45..d6f22b8 100644 --- a/test/test_cache.py +++ b/test/test_cache.py @@ -236,6 +236,27 @@ def test_a_zip_import_is_cached_by_numba_from_0_61(tmp_path): assert bool(_index_files(home)) == cached +@needs_a_directory_it_cannot_write +def test_a_zip_import_with_no_writable_user_cache_directory_compiles_uncached_with_a_warning(tmp_path): + # numba takes the user's cache directory for a .zip without checking that + # it can be written, so where it cannot, an executor with a read-only + # home, the first save raised PermissionError and the import died on it. + archive = _archive(tmp_path / "numbarrow.zip") + home = tmp_path / "home" + home.mkdir() + home.chmod(0o555) + try: + env = dict(os.environ, PYTHONPATH=str(archive), HOME=str(home), XDG_CACHE_HOME=str(home / "cache"), + NUMBA_CACHE_DIR=str(tmp_path / "cache")) + env.pop("NUMBARROW_JIT_OPTIONS", None) + run = subprocess.run([sys.executable, "-W", "always", "-c", IMPORT_AND_SHOW_FILE], + capture_output=True, text=True, env=env, cwd=str(tmp_path)) + assert run.returncode == 0 and str(archive) in run.stdout, run.stderr + assert "compiles without a cache" in run.stderr and "NUMBA_CACHE_DIR has no effect here" in run.stderr + finally: + home.chmod(0o755) + + @needs_a_directory_it_cannot_write def test_a_read_only_install_warns_naming_numba_cache_dir_and_setting_it_caches(tmp_path): # The other way to have no cache location: the source is on disk, and diff --git a/test/test_configurations.py b/test/test_configurations.py index 30b96e1..a0d1345 100644 --- a/test/test_configurations.py +++ b/test/test_configurations.py @@ -111,3 +111,31 @@ def decorate(func): with pytest.warns(RuntimeWarning, match="compiles without a cache"): configurations.jit_with_options(void())(lambda: None) assert options_seen == [{"cache": True}, {"cache": False}] + + +def test_a_cache_write_that_fails_at_decoration_compiles_uncached_only_when_caching_is_on(monkeypatch): + # numba takes a .zip's cache location unchecked, so there the failure is + # the first save's OSError and not the no-locator RuntimeError, and the + # import died on it. With caching off no cache is involved, and the error + # is the caller's to see. + from numbarrow.core import configurations + options_seen = [] + + def njit(*signature, **options): + def decorate(func): + options_seen.append(options) + if options.get("cache") or len(options_seen) == 1: + raise PermissionError(13, "Permission denied", "/nowhere/numba") + return func + return decorate + + monkeypatch.setattr(configurations, "njit", njit) + monkeypatch.setattr(configurations, "jit_options", {"cache": True}) + with pytest.warns(RuntimeWarning, match=r"Permission denied: '/nowhere/numba'.*compiles without a cache"): + configurations.jit_with_options(void())(lambda: None) + assert options_seen == [{"cache": True}, {"cache": False}] + options_seen.clear() + monkeypatch.setattr(configurations, "jit_options", {"cache": False}) + with pytest.raises(PermissionError): + configurations.jit_with_options(void())(lambda: None) + assert options_seen == [{"cache": False}]