|
17 | 17 |
|
18 | 18 | """Example demonstrating CsvReadOptions usage.""" |
19 | 19 |
|
| 20 | +import gzip |
| 21 | +import tempfile |
| 22 | +from pathlib import Path |
| 23 | + |
20 | 24 | from datafusion import CsvReadOptions, SessionContext |
21 | 25 |
|
| 26 | +# Write the CSV files used below into a temporary directory so the example is |
| 27 | +# self-contained and runnable without any external data. |
| 28 | +_tmpdir = tempfile.TemporaryDirectory() |
| 29 | +_data_dir = Path(_tmpdir.name) |
| 30 | +_csv_path = _data_dir / "data.csv" |
| 31 | +_gzip_path = _data_dir / "data.csv.gz" |
| 32 | + |
| 33 | +_csv_path.write_text("a,b,c\n1,4,foo\n2,5,bar\n3,,baz\n") |
| 34 | +with gzip.open(_gzip_path, "wt") as _f: |
| 35 | + _f.write("a,b,c\n1,4,foo\n2,5,N/A\n3,6,baz\n") |
| 36 | + |
22 | 37 | # Create a SessionContext |
23 | 38 | ctx = SessionContext() |
24 | 39 |
|
25 | 40 | # Example 1: Using CsvReadOptions with default values |
26 | 41 | print("Example 1: Default CsvReadOptions") |
27 | 42 | options = CsvReadOptions() |
28 | | -df = ctx.read_csv("data.csv", options=options) |
| 43 | +df = ctx.read_csv(str(_csv_path), options=options) |
| 44 | +df.show() |
29 | 45 |
|
30 | 46 | # Example 2: Using CsvReadOptions with custom parameters |
31 | 47 | print("\nExample 2: Custom CsvReadOptions") |
|
36 | 52 | schema_infer_max_records=1000, |
37 | 53 | file_extension=".csv", |
38 | 54 | ) |
39 | | -df = ctx.read_csv("data.csv", options=options) |
| 55 | +df = ctx.read_csv(str(_csv_path), options=options) |
| 56 | +df.show() |
40 | 57 |
|
41 | 58 | # Example 3: Using the builder pattern (recommended for readability) |
42 | 59 | print("\nExample 3: Builder pattern") |
|
49 | 66 | .with_truncated_rows(False) # noqa: FBT003 |
50 | 67 | .with_newlines_in_values(True) # noqa: FBT003 |
51 | 68 | ) |
52 | | -df = ctx.read_csv("data.csv", options=options) |
| 69 | +df = ctx.read_csv(str(_csv_path), options=options) |
| 70 | +df.show() |
53 | 71 |
|
54 | 72 | # Example 4: Advanced options |
55 | 73 | print("\nExample 4: Advanced options") |
|
64 | 82 | .with_file_compression_type("gzip") # Read gzipped CSV |
65 | 83 | .with_file_extension(".gz") |
66 | 84 | ) |
67 | | -df = ctx.read_csv("data.csv.gz", options=options) |
| 85 | +df = ctx.read_csv(str(_gzip_path), options=options) |
| 86 | +df.show() |
68 | 87 |
|
69 | 88 | # Example 5: Register CSV table with options |
70 | 89 | print("\nExample 5: Register CSV table") |
71 | 90 | options = CsvReadOptions().with_has_header(True).with_delimiter(",") # noqa: FBT003 |
72 | | -ctx.register_csv("my_table", "data.csv", options=options) |
| 91 | +ctx.register_csv("my_table", str(_csv_path), options=options) |
73 | 92 | df = ctx.sql("SELECT * FROM my_table") |
| 93 | +df.show() |
74 | 94 |
|
75 | 95 | # Example 6: Backward compatibility (without options) |
76 | 96 | print("\nExample 6: Backward compatibility") |
77 | 97 | # Still works the old way! |
78 | | -df = ctx.read_csv("data.csv", has_header=True, delimiter=",") |
| 98 | +df = ctx.read_csv(str(_csv_path), has_header=True, delimiter=",") |
| 99 | +df.show() |
79 | 100 |
|
80 | 101 | print("\nAll examples completed!") |
81 | 102 | print("\nFor all available options, see the CsvReadOptions documentation:") |
|
0 commit comments