Skip to content

Commit 67db06d

Browse files
committed
Run examples/*.py in CI, fix the broken example, and give silent ones output
Nothing in CI ran the top-level examples/*.py scripts, so they drifted out of sync with the API. examples/csv-read-options.py crashed because it reads data.csv and data.csv.gz that do not exist in the repository, and nine other examples end in asserts without printing anything, so a reader cannot tell a working example from a no-op. - Make csv-read-options.py self-contained: it now writes its own small CSV and gzipped CSV into a temporary directory. - Add a terminal call to the examples that printed nothing so each one shows its result. - Add a CI step, gated to the 3.12 abi3 entry, that runs every examples/*.py script against the built wheel with an explicit skip list for examples that need network/credentials, hand-downloaded data, generated TPC-H data, or the optional Ray dependency. Closes #1728
1 parent 59c1fd1 commit 67db06d

11 files changed

Lines changed: 86 additions & 6 deletions

‎.github/workflows/test.yml‎

Lines changed: 41 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -119,6 +119,47 @@ jobs:
119119
# free-threaded build and re-pick the system 3.12 (see install step).
120120
uv run --python "$PWD/.venv/bin/python" --no-project pytest -v --import-mode=importlib
121121
122+
# Run the top-level examples once. They exercise the public Python API
123+
# that the test suite does not: several are plain scripts with no pytest
124+
# coverage, so nothing caught a broken example before this job existed.
125+
- name: Run example scripts
126+
if: matrix.python-version == '3.12'
127+
run: |
128+
set -euo pipefail
129+
130+
# Extra runtime dependencies used by some examples but not part of
131+
# the dev group: pandas and polars for import.py/export.py, and
132+
# matplotlib for sql-to-pandas.py.
133+
uv pip install --python "$PWD/.venv/bin/python" pandas polars matplotlib
134+
135+
# Examples that cannot run here:
136+
# - sql-parquet-s3.py: needs network access and AWS credentials
137+
# - sql-parquet.py, dataframe-parquet.py, sql-to-pandas.py: need the
138+
# NYC taxi parquet file documented in examples/README.md
139+
# - python-udf-comparisons.py: needs the TPC-H dataset generated
140+
# later by the tpchgen-cli step
141+
# - ray_pickle_expr.py: needs the optional (heavy) `ray` dependency
142+
skip="sql-parquet-s3.py sql-parquet.py dataframe-parquet.py sql-to-pandas.py python-udf-comparisons.py ray_pickle_expr.py"
143+
144+
failed=""
145+
for script in examples/*.py; do
146+
name="$(basename "$script")"
147+
if [[ " $skip " == *" $name "* ]]; then
148+
echo "Skipping $script"
149+
continue
150+
fi
151+
echo "::group::Running $script"
152+
if ! uv run --python "$PWD/.venv/bin/python" --no-project python "$script"; then
153+
failed="$failed $script"
154+
fi
155+
echo "::endgroup::"
156+
done
157+
158+
if [[ -n "$failed" ]]; then
159+
echo "Example scripts failed:$failed"
160+
exit 1
161+
fi
162+
122163
# FFI + TPC-H examples only need to run once; gate to abi3 entries.
123164
- name: FFI unit tests
124165
if: matrix.wheel-tag == 'abi3'

‎examples/csv-read-options.py‎

Lines changed: 27 additions & 6 deletions
Original file line numberDiff line numberDiff line change
@@ -17,15 +17,31 @@
1717

1818
"""Example demonstrating CsvReadOptions usage."""
1919

20+
import gzip
21+
import tempfile
22+
from pathlib import Path
23+
2024
from datafusion import CsvReadOptions, SessionContext
2125

26+
# Write the CSV files used below into a temporary directory so the example is
27+
# self-contained and runnable without any external data.
28+
_tmpdir = tempfile.TemporaryDirectory()
29+
_data_dir = Path(_tmpdir.name)
30+
_csv_path = _data_dir / "data.csv"
31+
_gzip_path = _data_dir / "data.csv.gz"
32+
33+
_csv_path.write_text("a,b,c\n1,4,foo\n2,5,bar\n3,,baz\n")
34+
with gzip.open(_gzip_path, "wt") as _f:
35+
_f.write("a,b,c\n1,4,foo\n2,5,N/A\n3,6,baz\n")
36+
2237
# Create a SessionContext
2338
ctx = SessionContext()
2439

2540
# Example 1: Using CsvReadOptions with default values
2641
print("Example 1: Default CsvReadOptions")
2742
options = CsvReadOptions()
28-
df = ctx.read_csv("data.csv", options=options)
43+
df = ctx.read_csv(str(_csv_path), options=options)
44+
df.show()
2945

3046
# Example 2: Using CsvReadOptions with custom parameters
3147
print("\nExample 2: Custom CsvReadOptions")
@@ -36,7 +52,8 @@
3652
schema_infer_max_records=1000,
3753
file_extension=".csv",
3854
)
39-
df = ctx.read_csv("data.csv", options=options)
55+
df = ctx.read_csv(str(_csv_path), options=options)
56+
df.show()
4057

4158
# Example 3: Using the builder pattern (recommended for readability)
4259
print("\nExample 3: Builder pattern")
@@ -49,7 +66,8 @@
4966
.with_truncated_rows(False) # noqa: FBT003
5067
.with_newlines_in_values(True) # noqa: FBT003
5168
)
52-
df = ctx.read_csv("data.csv", options=options)
69+
df = ctx.read_csv(str(_csv_path), options=options)
70+
df.show()
5371

5472
# Example 4: Advanced options
5573
print("\nExample 4: Advanced options")
@@ -64,18 +82,21 @@
6482
.with_file_compression_type("gzip") # Read gzipped CSV
6583
.with_file_extension(".gz")
6684
)
67-
df = ctx.read_csv("data.csv.gz", options=options)
85+
df = ctx.read_csv(str(_gzip_path), options=options)
86+
df.show()
6887

6988
# Example 5: Register CSV table with options
7089
print("\nExample 5: Register CSV table")
7190
options = CsvReadOptions().with_has_header(True).with_delimiter(",") # noqa: FBT003
72-
ctx.register_csv("my_table", "data.csv", options=options)
91+
ctx.register_csv("my_table", str(_csv_path), options=options)
7392
df = ctx.sql("SELECT * FROM my_table")
93+
df.show()
7494

7595
# Example 6: Backward compatibility (without options)
7696
print("\nExample 6: Backward compatibility")
7797
# Still works the old way!
78-
df = ctx.read_csv("data.csv", has_header=True, delimiter=",")
98+
df = ctx.read_csv(str(_csv_path), has_header=True, delimiter=",")
99+
df.show()
79100

80101
print("\nAll examples completed!")
81102
print("\nFor all available options, see the CsvReadOptions documentation:")

‎examples/export.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -50,3 +50,5 @@
5050
# export to Python dictionary of columns
5151
pydict = df.to_pydict()
5252
assert pydict == {"a": [1, 2, 3], "b": [4, 5, 6]}
53+
54+
df.show()

‎examples/import.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -55,3 +55,5 @@
5555
arrow_table = pa.Table.from_pydict({"a": [1, 2, 3], "b": [4, 5, 6]})
5656
df = ctx.from_arrow(arrow_table)
5757
assert type(df) is datafusion.DataFrame
58+
59+
df.show()

‎examples/python-udaf.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -67,3 +67,5 @@ def evaluate(self) -> pa.Scalar:
6767
result = df.collect()[0]
6868

6969
assert result.column(0) == pa.array([6.0])
70+
71+
df.show()

‎examples/python-udf.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -41,3 +41,5 @@ def is_null(array: pa.Array) -> pa.Array:
4141
result = df.collect()[0]
4242

4343
assert result.column(0) == pa.array([False] * 3)
44+
45+
df.show()

‎examples/query-pyarrow-data.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -40,3 +40,5 @@
4040

4141
assert result.column(0) == pa.array([5, 7, 9])
4242
assert result.column(1) == pa.array([-3, -3, -3])
43+
44+
df.show()

‎examples/sql-to-pandas.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -40,3 +40,5 @@
4040
kind="bar", title="Trip Count by Number of Passengers"
4141
).get_figure()
4242
fig.savefig("chart.png")
43+
44+
print(pandas_df)

‎examples/sql-using-python-udaf.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -84,3 +84,5 @@ def evaluate(self) -> pa.Scalar:
8484
# +---+--------------+
8585
assert result_df.to_pydict()["a"] == [1, 3]
8686
assert result_df.to_pydict()["b_aggregated"] == [9, 6]
87+
88+
result_df.show()

‎examples/sql-using-python-udf.py‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -62,3 +62,5 @@ def is_null(array: pa.Array) -> pa.Array:
6262
# | 3 | false |
6363
# +---+-----------+
6464
assert result_df.to_pydict()["b_is_null"] == [False, True, False]
65+
66+
result_df.show()

0 commit comments

Comments
 (0)