timsaucer commented on code in PR #1763:
URL:
https://github.com/apache/datafusion-python/pull/1763#discussion_r4148555902
##########
python/datafusion/functions/__init__.py:
##########
@@ -5065,8 +5493,9 @@ def approx_percentile_cont_with_weight(
This aggregate function is similar to :py:func:`approx_percentile_cont`
except that
it uses the associated associated weights.
- If using the builder functions described in ref:`_aggregation` this
function ignores
- the options ``order_by``, ``null_treatment``, and ``distinct``.
+ If using the builder functions described in :ref:`aggregation` this
function ignores
+ the options ``null_treatment`` and ``distinct``, and ``order_by`` replaces
the
Review Comment:
Fixed in a4fe3933: `docs: correct what distinct does in count_star and the
weighted percentile`
##########
python/datafusion/functions/__init__.py:
##########
@@ -7091,14 +7618,29 @@ def string_agg(
... ).alias("s")])
>>> result.collect_column("s")[0].as_py()
'y,z'
+
+ >>> df = ctx.from_pydict({"a": ["y", "x", "y"]})
+ >>> result = df.aggregate(
+ ... [], [dfn.functions.string_agg(
+ ... dfn.col("a"), ",", distinct=True, order_by="a",
+ ... ).alias("s")])
+ >>> result.collect_column("s")[0].as_py()
+ 'x,y'
"""
+ if not isinstance(distinct, bool):
Review Comment:
Fixed in 47359dfe: `fix: accept numpy.bool_ as string_agg's distinct`
##########
python/datafusion/user_defined.py:
##########
@@ -388,7 +397,9 @@ def wrapper(*args: Any, **kwargs: Any) -> Callable:
return decorator
- if hasattr(args[0], "__datafusion_scalar_udf__"):
+ if args and (
+ hasattr(args[0], "__datafusion_scalar_udf__") or
_is_pycapsule(args[0])
+ ):
return ScalarUDF.from_pycapsule(args[0])
if args and callable(args[0]):
Review Comment:
Fixed in 14751239: `fix: accept the callable by keyword in udf, udaf, udwf,
and udtf`
##########
python/datafusion/dataframe.py:
##########
@@ -1868,13 +1925,51 @@ def fill_null(self, value: Any, subset: list[str] |
None = None) -> DataFrame:
>>> filled.sort(col("a")).collect()[0].column("a").to_pylist()
[0, 1, 3]
+ >>> df.fill_null(0, subset=[]).to_pydict()
+ {'a': [1, None, 3], 'b': [None, 5, 6]}
+
Notes:
- Only fills nulls in columns where the value can be cast to the
column type
- For columns where casting fails, the original column is kept
unchanged
- For columns not in subset, the original column is kept unchanged
"""
+ if subset is not None and not subset:
+ return self
return DataFrame(self.df.fill_null(value, subset))
+ def fill_nan(self, value: float, subset: list[str] | None = None) ->
DataFrame:
+ """Fill NaN values in floating-point columns with a value.
+
+ Only floating-point columns are changed; others are kept unchanged, as
is
+ any column ``value`` cannot be cast to. NaN is distinct from null,
which
+ :py:meth:`fill_null` handles.
+
+ Args:
+ value: Value to replace NaN with. Will be cast to match column
type.
+ subset: Optional list of column names to fill. If None, fills all
+ floating-point columns; an empty list fills none.
+
+ Returns:
+ DataFrame with NaN values replaced.
+
+ Examples:
+ >>> from datafusion import SessionContext
+ >>> ctx = SessionContext()
+ >>> nan = float("nan")
+ >>> df = ctx.from_pydict({"a": [1.0, nan, None], "b": [nan, 2.0,
3.0]})
+ >>> df.fill_nan(0.0).to_pydict()
+ {'a': [1.0, 0.0, None], 'b': [0.0, 2.0, 3.0]}
+
+ >>> df.fill_nan(0.0, subset=["a"]).collect_column("b")[0].as_py()
+ nan
+
+ >>> df.fill_nan(0.0, subset=[]).collect_column("b")[0].as_py()
+ nan
+ """
+ if subset is not None and not subset:
+ return self
+ return DataFrame(self.df.fill_nan(value, subset))
Review Comment:
Fixed in 354db9a6: `docs: note that fill_null and fill_nan fail on uppercase
or dotted names`
##########
docs/source/user-guide/upgrade-guides.md:
##########
@@ -198,6 +198,113 @@ ctx.execute(plan, partitions=0) # before
ctx.execute(plan, partition=0) # after
```
+### More aggregate functions accept `distinct`
+
+{py:func}`~datafusion.functions.bit_and`,
+{py:func}`~datafusion.functions.bit_or`,
+{py:func}`~datafusion.functions.mean`,
+{py:func}`~datafusion.functions.percentile_cont`,
+{py:func}`~datafusion.functions.quantile_cont`, and
+{py:func}`~datafusion.functions.string_agg` now accept a `distinct` argument.
+As with `sum` and `avg` in 54.0.0, `distinct` is inserted *before* `filter`, so
+code that passed `filter` (or, for `string_agg`, `order_by`) positionally must
+pass it by keyword.
+
+```python
+f.bit_and(column("a"), my_filter) # before
+f.bit_and(column("a"), filter=my_filter) # after
+```
+
+Passing `filter` to `mean` previously raised a `TypeError`, whether passed
Review Comment:
Fixed in 9f2b326a: `docs: say mean's filter now works only when passed by
keyword`
##########
python/datafusion/user_defined.py:
##########
@@ -31,7 +31,12 @@
from datafusion.expr import Expr
if TYPE_CHECKING:
- from _typeshed import CapsuleType as _PyCapsule
+ import sys
+
+ if sys.version_info >= (3, 13):
+ from types import CapsuleType as _PyCapsule
Review Comment:
Fixed in 6808f95f: `fix: resolve CapsuleType in context and extensions, and
at runtime` and ad4a4472: `fix: import CapsuleType from types on Python 3.13
and newer`
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]
---------------------------------------------------------------------
To unsubscribe, e-mail: [email protected]
For additional commands, e-mail: [email protected]