Skip to content

Commit d775ed8

Browse files
timsaucerclaude
andcommitted
Merge branch 'main' into feat/bundle-functions
Resolves conflicts with #1759, which hardened the extension API for 55.0.0: - context.py: drop the local PhysicalOptimizerRuleExportable (moved to datafusion.extensions) and route the bundle protocol isinstance checks in _collect_contributions and the planner-hook filter through the private _extensions alias, so the protocols stay type-only in datafusion.context. - extensions.py: keep _COMPONENT_NOUNS and apply kw_only=True to SessionExtensionComponents. - ffi-internals.md: keep this branch's four steps, where functions are resolved in step 3 and the planner is imported as each hook returns, so the commit itself cannot fail. Take main's carve-out for hook-side writes and the empty-call rebind skip. Drop main's note that the commit re-imports the capsule, which no longer holds here. - bundles.md: keep the collisions section and drop the shorter duplicate of the "Declaring a component" paragraph that auto-merge introduced. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
2 parents 6da87f6 + 6c5d9ff commit d775ed8

49 files changed

Lines changed: 3460 additions & 389 deletions

Some content is hidden

Large Commits have some content hidden by default. Use the searchbox below for content that may be hidden.

‎.ai/skills/check-upstream/SKILL.md‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -146,6 +146,7 @@ The user may specify an area via `$ARGUMENTS`. If no area is specified or "all"
146146
- `show_limit` — already covered by `DataFrame.show()`, which provides the same functionality with a simpler API
147147
- `with_param_values` — already covered by the `param_values` argument on `SessionContext.sql()`, which accomplishes the same thing more robustly
148148
- `union_by_name_distinct` — already covered by `DataFrame.union_by_name(distinct=True)`, which provides a more Pythonic API
149+
- `to_string` — `str(df)` is the Pythonic way to get a string and already goes through `__repr__` and the configurable formatter. A separate `to_string()` would either duplicate `str(df)` or render every row through a different path (session `datafusion.format.*` options, no formatter), giving a third text rendering alongside `repr` and `show()`
149150

150151
**How to check:**
151152
1. Fetch the upstream DataFrame documentation page listing all methods

‎.github/workflows/ci.yml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -21,7 +21,7 @@ name: CI
2121

2222
on:
2323
pull_request:
24-
branches: ["main"]
24+
branches: ["main", "branch-*"]
2525

2626
concurrency:
2727
group: ${{ github.repository }}-${{ github.head_ref || github.sha }}-${{ github.workflow }}

‎.github/workflows/release.yml‎

Lines changed: 1 addition & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -27,6 +27,7 @@ on:
2727
push:
2828
branches:
2929
- "main"
30+
- "branch-*"
3031
tags:
3132
- "*-rc*" # Release candidates (e.g., 45.0.0-rc1)
3233
- "[0-9]+.*" # Release tags (e.g., 45.0.0)

‎Cargo.lock‎

Lines changed: 10 additions & 10 deletions
Some generated files are not rendered by default. Learn more about customizing how changed files appear on GitHub.

‎Cargo.toml‎

Lines changed: 2 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -16,7 +16,7 @@
1616
# under the License.
1717

1818
[workspace.package]
19-
version = "54.0.0"
19+
version = "55.0.0"
2020
homepage = "https://datafusion.apache.org/python"
2121
repository = "https://github.com/apache/datafusion-python"
2222
authors = ["Apache DataFusion <dev@datafusion.apache.org>"]
@@ -69,7 +69,7 @@ log = "0.4.29"
6969
parking_lot = "0.12"
7070
prost-types = "0.14.3" # keep in line with `datafusion-substrait`
7171
pyo3-build-config = "0.29"
72-
datafusion-python-util = { path = "crates/util", version = "54.0.0" }
72+
datafusion-python-util = { path = "crates/util", version = "55.0.0" }
7373

7474
[profile.release]
7575
lto = "thin"

‎crates/core/src/dataframe.rs‎

Lines changed: 47 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -844,13 +844,24 @@ impl PyDataFrame {
844844
}
845845

846846
/// Print the query plan
847-
#[pyo3(signature = (verbose=false, analyze=false, format=None))]
847+
#[pyo3(signature = (
848+
verbose=false,
849+
analyze=false,
850+
format=None,
851+
show_statistics=None,
852+
analyze_level=None,
853+
analyze_categories=None
854+
))]
855+
#[allow(clippy::too_many_arguments)]
848856
fn explain(
849857
&self,
850858
py: Python,
851859
verbose: bool,
852860
analyze: bool,
853861
format: Option<&str>,
862+
show_statistics: Option<bool>,
863+
analyze_level: Option<&str>,
864+
analyze_categories: Option<Vec<String>>,
854865
) -> PyDataFusionResult<()> {
855866
let explain_format = match format {
856867
Some(f) => f
@@ -860,10 +871,24 @@ impl PyDataFrame {
860871
})?,
861872
None => datafusion::common::format::ExplainFormat::Indent,
862873
};
874+
let analyze_level = analyze_level
875+
.map(|l| l.parse::<datafusion::common::format::MetricType>())
876+
.transpose()?;
877+
let analyze_categories = analyze_categories
878+
.map(|cats| {
879+
cats.iter()
880+
.map(|c| c.parse::<datafusion::common::format::MetricCategory>())
881+
.collect::<datafusion::common::Result<Vec<_>>>()
882+
.map(datafusion::common::format::ExplainAnalyzeCategories::Only)
883+
})
884+
.transpose()?;
863885
let opts = datafusion::logical_expr::ExplainOption::default()
864886
.with_verbose(verbose)
865887
.with_analyze(analyze)
866-
.with_format(explain_format);
888+
.with_format(explain_format)
889+
.with_show_statistics(show_statistics)
890+
.with_analyze_level(analyze_level)
891+
.with_analyze_categories(analyze_categories);
867892
let df = self.df.as_ref().clone().explain_with_options(opts)?;
868893
print_dataframe(py, df)
869894
}
@@ -1320,6 +1345,26 @@ impl PyDataFrame {
13201345
let df = self.df.as_ref().fill_null(&scalar_value.0, &cols)?;
13211346
Ok(Self::new(df))
13221347
}
1348+
1349+
/// Fill NaN values with a specified value for specific floating-point columns
1350+
#[pyo3(signature = (value, columns=None))]
1351+
fn fill_nan(
1352+
&self,
1353+
value: Py<PyAny>,
1354+
columns: Option<Vec<PyBackedStr>>,
1355+
py: Python,
1356+
) -> PyDataFusionResult<Self> {
1357+
let scalar_value: PyScalarValue = value.extract(py)?;
1358+
1359+
let cols = match columns {
1360+
Some(col_names) => col_names.iter().map(|c| c.to_string()).collect(),
1361+
None => Vec::new(), // Empty vector means fill NaN for all columns
1362+
};
1363+
1364+
let cols = cols.iter().map(String::as_str).collect::<Vec<_>>();
1365+
let df = self.df.as_ref().fill_nan(&scalar_value.0, &cols)?;
1366+
Ok(Self::new(df))
1367+
}
13231368
}
13241369

13251370
#[derive(Debug, Clone, PartialEq, Eq, Hash, PartialOrd, Ord)]

0 commit comments

Comments
 (0)