Skip to content

Commit ee64f9b

Browse files
timsaucerclaude
andcommitted
feat: add show_statistics, analyze_level, and analyze_categories to explain
Expose the remaining upstream ExplainOption fields as keywords on DataFrame.explain. Each defaults to None, which falls back to the matching datafusion.explain.* session setting, so existing calls are unaffected. New ExplainAnalyzeLevel and ExplainMetricCategory enums sit beside ExplainFormat. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
1 parent 24764bd commit ee64f9b

4 files changed

Lines changed: 134 additions & 3 deletions

File tree

‎crates/core/src/dataframe.rs‎

Lines changed: 27 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -844,13 +844,24 @@ impl PyDataFrame {
844844
}
845845

846846
/// Print the query plan
847-
#[pyo3(signature = (verbose=false, analyze=false, format=None))]
847+
#[pyo3(signature = (
848+
verbose=false,
849+
analyze=false,
850+
format=None,
851+
show_statistics=None,
852+
analyze_level=None,
853+
analyze_categories=None
854+
))]
855+
#[allow(clippy::too_many_arguments)]
848856
fn explain(
849857
&self,
850858
py: Python,
851859
verbose: bool,
852860
analyze: bool,
853861
format: Option<&str>,
862+
show_statistics: Option<bool>,
863+
analyze_level: Option<&str>,
864+
analyze_categories: Option<Vec<String>>,
854865
) -> PyDataFusionResult<()> {
855866
let explain_format = match format {
856867
Some(f) => f
@@ -860,10 +871,24 @@ impl PyDataFrame {
860871
})?,
861872
None => datafusion::common::format::ExplainFormat::Indent,
862873
};
874+
let analyze_level = analyze_level
875+
.map(|l| l.parse::<datafusion::common::format::MetricType>())
876+
.transpose()?;
877+
let analyze_categories = analyze_categories
878+
.map(|cats| {
879+
cats.iter()
880+
.map(|c| c.parse::<datafusion::common::format::MetricCategory>())
881+
.collect::<datafusion::common::Result<Vec<_>>>()
882+
.map(datafusion::common::format::ExplainAnalyzeCategories::Only)
883+
})
884+
.transpose()?;
863885
let opts = datafusion::logical_expr::ExplainOption::default()
864886
.with_verbose(verbose)
865887
.with_analyze(analyze)
866-
.with_format(explain_format);
888+
.with_format(explain_format)
889+
.with_show_statistics(show_statistics)
890+
.with_analyze_level(analyze_level)
891+
.with_analyze_categories(analyze_categories);
867892
let df = self.df.as_ref().clone().explain_with_options(opts)?;
868893
print_dataframe(py, df)
869894
}

‎python/datafusion/__init__.py‎

Lines changed: 4 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -85,7 +85,9 @@
8585
from .dataframe import (
8686
DataFrame,
8787
DataFrameWriteOptions,
88+
ExplainAnalyzeLevel,
8889
ExplainFormat,
90+
ExplainMetricCategory,
8991
InsertOp,
9092
ParquetColumnOptions,
9193
ParquetWriterOptions,
@@ -126,7 +128,9 @@
126128
"DataFrame",
127129
"DataFrameWriteOptions",
128130
"ExecutionPlan",
131+
"ExplainAnalyzeLevel",
129132
"ExplainFormat",
133+
"ExplainMetricCategory",
130134
"Expr",
131135
"InsertOp",
132136
"LogicalPlan",

‎python/datafusion/dataframe.py‎

Lines changed: 50 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -108,6 +108,32 @@ class ExplainFormat(Enum):
108108
"""Graphviz DOT format for graph rendering."""
109109

110110

111+
class ExplainAnalyzeLevel(Enum):
112+
"""Which metrics :py:meth:`DataFrame.explain` reports when ``analyze=True``."""
113+
114+
SUMMARY = "summary"
115+
"""Common metrics for finding which operator is slow."""
116+
117+
DEV = "dev"
118+
"""All metrics, including those for deep operator-level introspection."""
119+
120+
121+
class ExplainMetricCategory(Enum):
122+
"""Category of metric reported by :py:meth:`DataFrame.explain` with ``analyze``."""
123+
124+
ROWS = "rows"
125+
"""Row counts, such as ``output_rows``."""
126+
127+
BYTES = "bytes"
128+
"""Byte sizes, such as ``output_bytes``."""
129+
130+
TIMING = "timing"
131+
"""Elapsed times, such as ``elapsed_compute``."""
132+
133+
UNCATEGORIZED = "uncategorized"
134+
"""Metrics that declare no category."""
135+
136+
111137
# excerpt from deltalake
112138
# https://github.com/apache/datafusion-python/pull/981#discussion_r1905619163
113139
class Compression(Enum):
@@ -1207,6 +1233,9 @@ def explain(
12071233
verbose: bool = False,
12081234
analyze: bool = False,
12091235
format: ExplainFormat | None = None,
1236+
show_statistics: bool | None = None,
1237+
analyze_level: ExplainAnalyzeLevel | None = None,
1238+
analyze_categories: Iterable[ExplainMetricCategory] | None = None,
12101239
) -> None:
12111240
"""Print an explanation of the DataFrame's plan so far.
12121241
@@ -1217,6 +1246,13 @@ def explain(
12171246
analyze: If ``True``, the plan will run and metrics reported.
12181247
format: Output format for the plan. Defaults to
12191248
:py:attr:`ExplainFormat.INDENT`.
1249+
show_statistics: If ``True``, include each operator's statistics.
1250+
``None`` uses the ``datafusion.explain.show_statistics`` setting.
1251+
analyze_level: Which metrics to report with ``analyze``. ``None``
1252+
uses the ``datafusion.explain.analyze_level`` setting.
1253+
analyze_categories: Report only metrics in these categories with
1254+
``analyze``; an empty iterable reports none. ``None`` uses the
1255+
``datafusion.explain.analyze_categories`` setting.
12201256
12211257
Examples:
12221258
Show the plan in tree format:
@@ -1229,9 +1265,22 @@ def explain(
12291265
Show plan with runtime metrics:
12301266
12311267
>>> df.explain(analyze=True) # doctest: +SKIP
1268+
1269+
Show only row-count metrics:
1270+
1271+
>>> from datafusion import ExplainMetricCategory
1272+
>>> df.explain(
1273+
... analyze=True, analyze_categories=[ExplainMetricCategory.ROWS]
1274+
... ) # doctest: +SKIP
12321275
"""
12331276
fmt = format.value if format is not None else None
1234-
self.df.explain(verbose, analyze, fmt)
1277+
level = analyze_level.value if analyze_level is not None else None
1278+
categories = (
1279+
[c.value for c in analyze_categories]
1280+
if analyze_categories is not None
1281+
else None
1282+
)
1283+
self.df.explain(verbose, analyze, fmt, show_statistics, level, categories)
12351284

12361285
def logical_plan(self) -> LogicalPlan:
12371286
"""Return the unoptimized ``LogicalPlan``.

‎python/tests/test_dataframe.py‎

Lines changed: 53 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -29,7 +29,9 @@
2929
import pytest
3030
from datafusion import (
3131
DataFrame,
32+
ExplainAnalyzeLevel,
3233
ExplainFormat,
34+
ExplainMetricCategory,
3335
InsertOp,
3436
ParquetColumnOptions,
3537
ParquetWriterOptions,
@@ -3871,6 +3873,57 @@ def test_explain_with_format(capsys, fmt, verbose, analyze, expected_substring):
38713873
assert expected_substring in captured.out
38723874

38733875

3876+
def _explain_output(capsys, **kwargs):
3877+
ctx = SessionContext()
3878+
df = ctx.from_pydict({"a": [1, 2]}).filter(column("a") > literal(1))
3879+
df.explain(**kwargs)
3880+
return capsys.readouterr().out
3881+
3882+
3883+
@pytest.mark.parametrize(
3884+
("kwargs", "present", "absent"),
3885+
[
3886+
pytest.param({}, [], ["statistics="], id="default_no_statistics"),
3887+
pytest.param(
3888+
{"show_statistics": True}, ["statistics=[Rows="], [], id="show_statistics"
3889+
),
3890+
pytest.param(
3891+
{"analyze": True, "analyze_level": ExplainAnalyzeLevel.DEV},
3892+
["output_rows=", "output_batches="],
3893+
[],
3894+
id="analyze_level_dev",
3895+
),
3896+
pytest.param(
3897+
{"analyze": True, "analyze_level": ExplainAnalyzeLevel.SUMMARY},
3898+
["output_rows="],
3899+
["output_batches="],
3900+
id="analyze_level_summary",
3901+
),
3902+
pytest.param(
3903+
{
3904+
"analyze": True,
3905+
"analyze_categories": [ExplainMetricCategory.ROWS],
3906+
},
3907+
["output_rows="],
3908+
["elapsed_compute=", "output_bytes="],
3909+
id="analyze_categories_rows",
3910+
),
3911+
pytest.param(
3912+
{"analyze": True, "analyze_categories": []},
3913+
["FilterExec: a@0 > 1, metrics=[]"],
3914+
["output_rows="],
3915+
id="analyze_categories_empty_suppresses_metrics",
3916+
),
3917+
],
3918+
)
3919+
def test_explain_options(capsys, kwargs, present, absent):
3920+
out = _explain_output(capsys, **kwargs)
3921+
for text in present:
3922+
assert text in out
3923+
for text in absent:
3924+
assert text not in out
3925+
3926+
38743927
@pytest.mark.parametrize(
38753928
("window_exprs", "expected_columns"),
38763929
[

0 commit comments

Comments
 (0)