diff --git a/.nextchanges/bundles/features.md b/.nextchanges/bundles/features.md new file mode 100644 index 00000000000..89cd02603cd --- /dev/null +++ b/.nextchanges/bundles/features.md @@ -0,0 +1 @@ +* Add support for the `features` resource type. ([#6952](https://github.com/databricks/cli/pull/6952)) diff --git a/acceptance/bundle/deployment/bind/feature/databricks.yml b/acceptance/bundle/deployment/bind/feature/databricks.yml new file mode 100644 index 00000000000..10c39741536 --- /dev/null +++ b/acceptance/bundle/deployment/bind/feature/databricks.yml @@ -0,0 +1,18 @@ +bundle: + name: test-bundle + +resources: + features: + feature1: + full_name: main.myschema.boundfeature + description: bound feature + source: + delta_table_source: + full_name: main.myschema.mysource + function: + column_selection: + column: amount + entities: + - name: account_id + timeseries_column: + name: event_ts diff --git a/acceptance/bundle/deployment/bind/feature/out.test.toml b/acceptance/bundle/deployment/bind/feature/out.test.toml new file mode 100644 index 00000000000..19b5b15d11f --- /dev/null +++ b/acceptance/bundle/deployment/bind/feature/out.test.toml @@ -0,0 +1,2 @@ +Cloud = false +EnvMatrix.DMS = [""] diff --git a/acceptance/bundle/deployment/bind/feature/output.txt b/acceptance/bundle/deployment/bind/feature/output.txt new file mode 100644 index 00000000000..d4d47f6a660 --- /dev/null +++ b/acceptance/bundle/deployment/bind/feature/output.txt @@ -0,0 +1,30 @@ + +>>> [CLI] bundle deployment bind feature1 main.myschema.boundfeature --auto-approve +Successfully bound feature with an id 'main.myschema.boundfeature' +Run 'bundle deploy' to deploy changes to your workspace + +>>> [CLI] bundle summary +Name: test-bundle +Target: default +Workspace: + User: [USERNAME] + Path: /Workspace/Users/[USERNAME]/.bundle/test-bundle/default +Resources: + Features: + feature1: + Name: main.myschema.boundfeature + URL: [DATABRICKS_URL]/explore/data/features/main/myschema/boundfeature?w=[WORKSPACE_ID] + +>>> [CLI] bundle deployment unbind feature1 + +>>> [CLI] bundle summary +Name: test-bundle +Target: default +Workspace: + User: [USERNAME] + Path: /Workspace/Users/[USERNAME]/.bundle/test-bundle/default +Resources: + Features: + feature1: + Name: main.myschema.boundfeature + URL: (not deployed) diff --git a/acceptance/bundle/deployment/bind/feature/script b/acceptance/bundle/deployment/bind/feature/script new file mode 100644 index 00000000000..8e27f69056c --- /dev/null +++ b/acceptance/bundle/deployment/bind/feature/script @@ -0,0 +1,5 @@ +trace $CLI bundle deployment bind feature1 main.myschema.boundfeature --auto-approve +trace $CLI bundle summary + +trace $CLI bundle deployment unbind feature1 +trace $CLI bundle summary diff --git a/acceptance/bundle/deployment/bind/feature/test.toml b/acceptance/bundle/deployment/bind/feature/test.toml new file mode 100644 index 00000000000..84e8bf9b8f4 --- /dev/null +++ b/acceptance/bundle/deployment/bind/feature/test.toml @@ -0,0 +1,22 @@ +Cloud = false + +Ignore = [ + ".databricks", +] + +# The bind flow issues a GET to confirm the remote feature exists before binding. +[[Server]] +Pattern = "GET /api/2.0/feature-engineering/features/{full_name}" +Response.Body = ''' +{ + "full_name": "main.myschema.boundfeature", + "catalog_name": "main", + "schema_name": "myschema", + "name": "boundfeature", + "description": "bound feature", + "source": {"delta_table_source": {"full_name": "main.myschema.mysource"}}, + "function": {"column_selection": {"column": "amount"}}, + "entities": [{"name": "account_id"}], + "timeseries_column": {"name": "event_ts"} +} +''' diff --git a/acceptance/bundle/invariant/configs/feature.yml.tmpl b/acceptance/bundle/invariant/configs/feature.yml.tmpl new file mode 100644 index 00000000000..f70627e0982 --- /dev/null +++ b/acceptance/bundle/invariant/configs/feature.yml.tmpl @@ -0,0 +1,23 @@ +bundle: + name: test-bundle-$UNIQUE_NAME + +resources: + features: + foo: + full_name: main.default.test_feature_$UNIQUE_NAME + description: This is a test feature # CAN_REMOVE + source: + delta_table_source: + full_name: main.default.test_source_$UNIQUE_NAME + function: + aggregation_function: + count_function: + input: value + time_window: + sliding: + window_duration: 604800s + slide_duration: 86400s + entities: + - name: id + timeseries_column: + name: timestamp diff --git a/acceptance/bundle/python/features-support/databricks.yml b/acceptance/bundle/python/features-support/databricks.yml new file mode 100644 index 00000000000..2a2c08e8cbc --- /dev/null +++ b/acceptance/bundle/python/features-support/databricks.yml @@ -0,0 +1,27 @@ +bundle: + name: my_project + +sync: {paths: []} # don't need to copy files + +python: + resources: + - "resources:load_resources" + mutators: + - "mutators:update_feature" + +resources: + features: + my_feature_1: + full_name: main.default.my_feature_1 + description: "My Feature" + source: + delta_table_source: + full_name: main.default.my_source_1 + function: + aggregation_function: + count_function: + input: value + time_window: + sliding: + window_duration: 604800s + slide_duration: 86400s diff --git a/acceptance/bundle/python/features-support/mutators.py b/acceptance/bundle/python/features-support/mutators.py new file mode 100644 index 00000000000..c9c8924b49e --- /dev/null +++ b/acceptance/bundle/python/features-support/mutators.py @@ -0,0 +1,11 @@ +from dataclasses import replace + +from databricks.bundles.core import feature_mutator +from databricks.bundles.features import Feature + + +@feature_mutator +def update_feature(feature: Feature) -> Feature: + assert isinstance(feature.description, str) + + return replace(feature, description=f"{feature.description} (updated)") diff --git a/acceptance/bundle/python/features-support/out.test.toml b/acceptance/bundle/python/features-support/out.test.toml new file mode 100644 index 00000000000..aeddf3e619e --- /dev/null +++ b/acceptance/bundle/python/features-support/out.test.toml @@ -0,0 +1,3 @@ +Cloud = false +EnvMatrix.DMS = ["", "true"] +EnvMatrix.PYDAB_VERSION = ["current"] diff --git a/acceptance/bundle/python/features-support/output.txt b/acceptance/bundle/python/features-support/output.txt new file mode 100644 index 00000000000..3dbfffc35ec --- /dev/null +++ b/acceptance/bundle/python/features-support/output.txt @@ -0,0 +1,62 @@ + +>>> uv run [UV_ARGS] -q [CLI] bundle validate --output json +{ + "experimental": { + "python": { + "mutators": [ + "mutators:update_feature" + ], + "resources": [ + "resources:load_resources" + ] + } + }, + "resources": { + "features": { + "my_feature_1": { + "description": "My Feature (updated)", + "full_name": "main.default.my_feature_1", + "function": { + "aggregation_function": { + "count_function": { + "input": "value" + }, + "time_window": { + "sliding": { + "slide_duration": "86400s", + "window_duration": "604800s" + } + } + } + }, + "source": { + "delta_table_source": { + "full_name": "main.default.my_source_1" + } + } + }, + "my_feature_2": { + "description": "My Feature (2) (updated)", + "full_name": "main.default.my_feature_2", + "function": { + "aggregation_function": { + "count_function": { + "input": "value" + }, + "time_window": { + "sliding": { + "slide_duration": "86400s", + "window_duration": "604800s" + } + } + } + }, + "source": { + "delta_table_source": { + "full_name": "main.default.my_source_2" + } + } + } + } + } +} diff --git a/acceptance/bundle/python/features-support/resources.py b/acceptance/bundle/python/features-support/resources.py new file mode 100644 index 00000000000..57e58acbd42 --- /dev/null +++ b/acceptance/bundle/python/features-support/resources.py @@ -0,0 +1,29 @@ +from databricks.bundles.core import Resources + + +def load_resources() -> Resources: + resources = Resources() + + resources.add_feature( + "my_feature_2", + { + "full_name": "main.default.my_feature_2", + "description": "My Feature (2)", + "source": { + "delta_table_source": {"full_name": "main.default.my_source_2"}, + }, + "function": { + "aggregation_function": { + "count_function": {"input": "value"}, + "time_window": { + "sliding": { + "window_duration": "604800s", + "slide_duration": "86400s", + }, + }, + }, + }, + }, + ) + + return resources diff --git a/acceptance/bundle/python/features-support/script b/acceptance/bundle/python/features-support/script new file mode 100644 index 00000000000..e273fb45a53 --- /dev/null +++ b/acceptance/bundle/python/features-support/script @@ -0,0 +1,5 @@ + +trace uv run $UV_ARGS -q $CLI bundle validate --output json | \ + jq "pick(.experimental.python, .resources)" + +rm -fr .databricks __pycache__ diff --git a/acceptance/bundle/python/features-support/test.toml b/acceptance/bundle/python/features-support/test.toml new file mode 100644 index 00000000000..2f4cd0ae557 --- /dev/null +++ b/acceptance/bundle/python/features-support/test.toml @@ -0,0 +1,4 @@ +Cloud = false # tests don't interact with APIs + +# features are only supported in the current version of the wheel +EnvMatrix.PYDAB_VERSION = ["current"] diff --git a/acceptance/bundle/refschema/out.fields.txt b/acceptance/bundle/refschema/out.fields.txt index 4e96efb1685..af1ce1bfcb9 100644 --- a/acceptance/bundle/refschema/out.fields.txt +++ b/acceptance/bundle/refschema/out.fields.txt @@ -832,6 +832,162 @@ resources.external_locations.*.grants[*] catalog.PrivilegeAssignment ALL resources.external_locations.*.grants[*].principal string ALL resources.external_locations.*.grants[*].privileges []catalog.Privilege ALL resources.external_locations.*.grants[*].privileges[*] catalog.Privilege ALL +resources.features.*.catalog_name string ALL +resources.features.*.created_at *time.Time ALL +resources.features.*.created_by string ALL +resources.features.*.description string ALL +resources.features.*.entities []ml.EntityColumn ALL +resources.features.*.entities[*] ml.EntityColumn ALL +resources.features.*.entities[*].name string ALL +resources.features.*.filter_condition string ALL +resources.features.*.full_name string ALL +resources.features.*.function ml.Function ALL +resources.features.*.function.aggregation_function *ml.AggregationFunction ALL +resources.features.*.function.aggregation_function.approx_count_distinct *ml.ApproxCountDistinctFunction ALL +resources.features.*.function.aggregation_function.approx_count_distinct.input string ALL +resources.features.*.function.aggregation_function.approx_count_distinct.relative_sd float64 ALL +resources.features.*.function.aggregation_function.approx_percentile *ml.ApproxPercentileFunction ALL +resources.features.*.function.aggregation_function.approx_percentile.accuracy int64 ALL +resources.features.*.function.aggregation_function.approx_percentile.input string ALL +resources.features.*.function.aggregation_function.approx_percentile.percentile float64 ALL +resources.features.*.function.aggregation_function.avg *ml.AvgFunction ALL +resources.features.*.function.aggregation_function.avg.input string ALL +resources.features.*.function.aggregation_function.count_function *ml.CountFunction ALL +resources.features.*.function.aggregation_function.count_function.input string ALL +resources.features.*.function.aggregation_function.first *ml.FirstFunction ALL +resources.features.*.function.aggregation_function.first.input string ALL +resources.features.*.function.aggregation_function.first_distinct *ml.FirstDistinctFunction ALL +resources.features.*.function.aggregation_function.first_distinct.input string ALL +resources.features.*.function.aggregation_function.first_distinct.n int64 ALL +resources.features.*.function.aggregation_function.first_n *ml.FirstNFunction ALL +resources.features.*.function.aggregation_function.first_n.input string ALL +resources.features.*.function.aggregation_function.first_n.n int64 ALL +resources.features.*.function.aggregation_function.last *ml.LastFunction ALL +resources.features.*.function.aggregation_function.last.input string ALL +resources.features.*.function.aggregation_function.last_distinct *ml.LastDistinctFunction ALL +resources.features.*.function.aggregation_function.last_distinct.input string ALL +resources.features.*.function.aggregation_function.last_distinct.n int64 ALL +resources.features.*.function.aggregation_function.last_n *ml.LastNFunction ALL +resources.features.*.function.aggregation_function.last_n.input string ALL +resources.features.*.function.aggregation_function.last_n.n int64 ALL +resources.features.*.function.aggregation_function.max *ml.MaxFunction ALL +resources.features.*.function.aggregation_function.max.input string ALL +resources.features.*.function.aggregation_function.min *ml.MinFunction ALL +resources.features.*.function.aggregation_function.min.input string ALL +resources.features.*.function.aggregation_function.stddev_pop *ml.StddevPopFunction ALL +resources.features.*.function.aggregation_function.stddev_pop.input string ALL +resources.features.*.function.aggregation_function.stddev_samp *ml.StddevSampFunction ALL +resources.features.*.function.aggregation_function.stddev_samp.input string ALL +resources.features.*.function.aggregation_function.sum *ml.SumFunction ALL +resources.features.*.function.aggregation_function.sum.input string ALL +resources.features.*.function.aggregation_function.time_window *ml.TimeWindow ALL +resources.features.*.function.aggregation_function.time_window.continuous *ml.ContinuousWindow ALL +resources.features.*.function.aggregation_function.time_window.continuous.offset string ALL +resources.features.*.function.aggregation_function.time_window.continuous.window_duration string ALL +resources.features.*.function.aggregation_function.time_window.rolling *ml.RollingWindow ALL +resources.features.*.function.aggregation_function.time_window.rolling.delay *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.rolling.window_duration *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.sawtooth *ml.SawtoothWindow ALL +resources.features.*.function.aggregation_function.time_window.sawtooth.delay *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.sawtooth.window_duration *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.sliding *ml.SlidingWindow ALL +resources.features.*.function.aggregation_function.time_window.sliding.delay *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.sliding.offset *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.sliding.slide_duration string ALL +resources.features.*.function.aggregation_function.time_window.sliding.window_duration string ALL +resources.features.*.function.aggregation_function.time_window.start_time *time.Time ALL +resources.features.*.function.aggregation_function.time_window.tumbling *ml.TumblingWindow ALL +resources.features.*.function.aggregation_function.time_window.tumbling.delay *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.tumbling.offset *duration.Duration ALL +resources.features.*.function.aggregation_function.time_window.tumbling.window_duration string ALL +resources.features.*.function.aggregation_function.var_pop *ml.VarPopFunction ALL +resources.features.*.function.aggregation_function.var_pop.input string ALL +resources.features.*.function.aggregation_function.var_samp *ml.VarSampFunction ALL +resources.features.*.function.aggregation_function.var_samp.input string ALL +resources.features.*.function.column_selection *ml.ColumnSelection ALL +resources.features.*.function.column_selection.column string ALL +resources.features.*.function.custom_udf *ml.CustomUdf ALL +resources.features.*.function.custom_udf.function_path string ALL +resources.features.*.function.custom_udf.input_bindings []ml.InputBinding ALL +resources.features.*.function.custom_udf.input_bindings[*] ml.InputBinding ALL +resources.features.*.function.custom_udf.input_bindings[*].column string ALL +resources.features.*.function.custom_udf.input_bindings[*].parameter string ALL +resources.features.*.function.extra_parameters []ml.FunctionExtraParameter ALL +resources.features.*.function.extra_parameters[*] ml.FunctionExtraParameter ALL +resources.features.*.function.extra_parameters[*].key string ALL +resources.features.*.function.extra_parameters[*].value string ALL +resources.features.*.function.function_type ml.FunctionFunctionType ALL +resources.features.*.id string INPUT +resources.features.*.inputs []string ALL +resources.features.*.inputs[*] string ALL +resources.features.*.lifecycle resources.Lifecycle INPUT +resources.features.*.lifecycle.prevent_destroy bool INPUT +resources.features.*.lineage_context *ml.LineageContext ALL +resources.features.*.lineage_context.job_context *ml.JobContext ALL +resources.features.*.lineage_context.job_context.job_id int64 ALL +resources.features.*.lineage_context.job_context.job_run_id int64 ALL +resources.features.*.lineage_context.notebook_id int64 ALL +resources.features.*.modified_status string INPUT +resources.features.*.name string ALL +resources.features.*.schema_name string ALL +resources.features.*.source ml.DataSource ALL +resources.features.*.source.delta_table_source *ml.DeltaTableSource ALL +resources.features.*.source.delta_table_source.dataframe_schema string ALL +resources.features.*.source.delta_table_source.entity_columns []string ALL +resources.features.*.source.delta_table_source.entity_columns[*] string ALL +resources.features.*.source.delta_table_source.filter_condition string ALL +resources.features.*.source.delta_table_source.full_name string ALL +resources.features.*.source.delta_table_source.timeseries_column string ALL +resources.features.*.source.delta_table_source.transformation_sql string ALL +resources.features.*.source.feature_view_source *ml.FeatureViewSource ALL +resources.features.*.source.feature_view_source.feature_references []ml.FeatureReference ALL +resources.features.*.source.feature_view_source.feature_references[*] ml.FeatureReference ALL +resources.features.*.source.feature_view_source.feature_references[*].feature string ALL +resources.features.*.source.kafka_source *ml.KafkaSource ALL +resources.features.*.source.kafka_source.entity_column_identifiers []ml.ColumnIdentifier ALL +resources.features.*.source.kafka_source.entity_column_identifiers[*] ml.ColumnIdentifier ALL +resources.features.*.source.kafka_source.entity_column_identifiers[*].variant_expr_path string ALL +resources.features.*.source.kafka_source.filter_condition string ALL +resources.features.*.source.kafka_source.name string ALL +resources.features.*.source.kafka_source.timeseries_column_identifier *ml.ColumnIdentifier ALL +resources.features.*.source.kafka_source.timeseries_column_identifier.variant_expr_path string ALL +resources.features.*.source.lateness *ml.SourceLateness ALL +resources.features.*.source.lateness.settling_delay *duration.Duration ALL +resources.features.*.source.request_source *ml.RequestSource ALL +resources.features.*.source.request_source.dataframe_schema string ALL +resources.features.*.source.request_source.flat_schema *ml.FlatSchema ALL +resources.features.*.source.request_source.flat_schema.fields []ml.FieldDefinition ALL +resources.features.*.source.request_source.flat_schema.fields[*] ml.FieldDefinition ALL +resources.features.*.source.request_source.flat_schema.fields[*].data_type ml.ScalarDataType ALL +resources.features.*.source.request_source.flat_schema.fields[*].name string ALL +resources.features.*.source.stream_source *ml.StreamSource ALL +resources.features.*.source.stream_source.dataframe_schema string ALL +resources.features.*.source.stream_source.filter_condition string ALL +resources.features.*.source.stream_source.full_name string ALL +resources.features.*.source.stream_source.transformation_sql string ALL +resources.features.*.time_window *ml.TimeWindow ALL +resources.features.*.time_window.continuous *ml.ContinuousWindow ALL +resources.features.*.time_window.continuous.offset string ALL +resources.features.*.time_window.continuous.window_duration string ALL +resources.features.*.time_window.rolling *ml.RollingWindow ALL +resources.features.*.time_window.rolling.delay *duration.Duration ALL +resources.features.*.time_window.rolling.window_duration *duration.Duration ALL +resources.features.*.time_window.sawtooth *ml.SawtoothWindow ALL +resources.features.*.time_window.sawtooth.delay *duration.Duration ALL +resources.features.*.time_window.sawtooth.window_duration *duration.Duration ALL +resources.features.*.time_window.sliding *ml.SlidingWindow ALL +resources.features.*.time_window.sliding.delay *duration.Duration ALL +resources.features.*.time_window.sliding.offset *duration.Duration ALL +resources.features.*.time_window.sliding.slide_duration string ALL +resources.features.*.time_window.sliding.window_duration string ALL +resources.features.*.time_window.start_time *time.Time ALL +resources.features.*.time_window.tumbling *ml.TumblingWindow ALL +resources.features.*.time_window.tumbling.delay *duration.Duration ALL +resources.features.*.time_window.tumbling.offset *duration.Duration ALL +resources.features.*.time_window.tumbling.window_duration string ALL +resources.features.*.timeseries_column *ml.TimeseriesColumn ALL +resources.features.*.timeseries_column.name string ALL +resources.features.*.url string INPUT resources.genie_spaces.*.description string ALL resources.genie_spaces.*.etag string ALL resources.genie_spaces.*.file_path string INPUT diff --git a/acceptance/bundle/resources/features/basic/databricks.yml b/acceptance/bundle/resources/features/basic/databricks.yml new file mode 100644 index 00000000000..cab7152110b --- /dev/null +++ b/acceptance/bundle/resources/features/basic/databricks.yml @@ -0,0 +1,23 @@ +bundle: + name: test-bundle + +resources: + features: + feature1: + full_name: main.myschema.myfeature + description: DESCRIPTION1 + source: + delta_table_source: + full_name: main.myschema.mysource + function: + aggregation_function: + count_function: + input: amount + time_window: + sliding: + window_duration: 604800s + slide_duration: 86400s + entities: + - name: account_id + timeseries_column: + name: event_ts diff --git a/acceptance/bundle/resources/features/basic/out.test.toml b/acceptance/bundle/resources/features/basic/out.test.toml new file mode 100644 index 00000000000..a927a5fbc06 --- /dev/null +++ b/acceptance/bundle/resources/features/basic/out.test.toml @@ -0,0 +1,2 @@ +Cloud = false +EnvMatrix.DMS = ["", "true"] diff --git a/acceptance/bundle/resources/features/basic/output.txt b/acceptance/bundle/resources/features/basic/output.txt new file mode 100644 index 00000000000..3f3b32dc99e --- /dev/null +++ b/acceptance/bundle/resources/features/basic/output.txt @@ -0,0 +1,170 @@ + +=== Initial summary before deploy +>>> [CLI] bundle summary -o json +{ + "description": "DESCRIPTION1", + "entities": [ + { + "name": "account_id" + } + ], + "full_name": "main.myschema.myfeature", + "function": { + "aggregation_function": { + "count_function": { + "input": "amount" + }, + "time_window": { + "sliding": { + "slide_duration": "86400s", + "window_duration": "604800s" + } + } + } + }, + "modified_status": "created", + "source": { + "delta_table_source": { + "full_name": "main.myschema.mysource" + } + }, + "timeseries_column": { + "name": "event_ts" + } +} + +=== Verify it does not exist yet +>>> musterr [CLI] feature-engineering get-feature main.myschema.myfeature +Error: Resource ml.Feature not found: main.myschema.myfeature + +>>> [CLI] bundle deploy +Uploading bundle files to /Workspace/Users/[USERNAME]/.bundle/test-bundle/default/files... +Created features.feature1 +Files: 1 uploaded, 0 deleted +Resources: 1 created, 0 changed, 0 deleted, 0 unchanged + +>>> print_requests.py //feature-engineering +{ + "method": "POST", + "path": "/api/2.0/feature-engineering/features", + "body": { + "description": "DESCRIPTION1", + "entities": [ + { + "name": "account_id" + } + ], + "full_name": "main.myschema.myfeature", + "function": { + "aggregation_function": { + "count_function": { + "input": "amount" + }, + "time_window": { + "sliding": { + "slide_duration": "86400s", + "window_duration": "604800s" + } + } + } + }, + "source": { + "delta_table_source": { + "full_name": "main.myschema.mysource" + } + }, + "timeseries_column": { + "name": "event_ts" + } + } +} + +=== Summary should show the id and the Catalog Explorer url +>>> [CLI] bundle summary -o json +{ + "id": "main.myschema.myfeature", + "url": "[DATABRICKS_URL]/explore/data/features/main/myschema/myfeature?w=[WORKSPACE_ID]" +} + +=== Verify deployment +>>> [CLI] feature-engineering get-feature main.myschema.myfeature +{ + "full_name": "main.myschema.myfeature", + "description": "DESCRIPTION1", + "catalog_name": "main", + "schema_name": "myschema", + "name": "myfeature" +} + +=== Update description (should update in place, not recreate) +>>> update_file.py databricks.yml DESCRIPTION1 DESCRIPTION2 + +>>> [CLI] bundle deploy +Uploading bundle files to /Workspace/Users/[USERNAME]/.bundle/test-bundle/default/files... +Updated features.feature1 +Files: 1 uploaded, 0 deleted +Resources: 0 created, 1 changed, 0 deleted, 0 unchanged + +>>> print_requests.py //feature-engineering +{ + "method": "PATCH", + "path": "/api/2.0/feature-engineering/features/main.myschema.myfeature", + "q": { + "update_mask": "description" + }, + "body": { + "description": "DESCRIPTION2", + "entities": [ + { + "name": "account_id" + } + ], + "full_name": "main.myschema.myfeature", + "function": { + "aggregation_function": { + "count_function": { + "input": "amount" + }, + "time_window": { + "sliding": { + "slide_duration": "86400s", + "window_duration": "604800s" + } + } + } + }, + "source": { + "delta_table_source": { + "full_name": "main.myschema.mysource" + } + }, + "timeseries_column": { + "name": "event_ts" + } + } +} + +>>> [CLI] feature-engineering get-feature main.myschema.myfeature +"DESCRIPTION2" + +=== Change an immutable field (should plan a recreate) +>>> update_file.py databricks.yml amount other_column + +>>> [CLI] bundle plan +recreate features.feature1 + +Plan: 1 to add, 0 to change, 1 to delete, 0 unchanged + +>>> [CLI] bundle destroy --auto-approve +The following resources will be deleted: + delete resources.features.feature1 + +All files and directories at the following location will be deleted: /Workspace/Users/[USERNAME]/.bundle/test-bundle/default + +Destroy: 1 deleted + +>>> print_requests.py //feature-engineering +{ + "method": "DELETE", + "path": "/api/2.0/feature-engineering/features/main.myschema.myfeature" +} diff --git a/acceptance/bundle/resources/features/basic/script b/acceptance/bundle/resources/features/basic/script new file mode 100755 index 00000000000..7d66a4bf0aa --- /dev/null +++ b/acceptance/bundle/resources/features/basic/script @@ -0,0 +1,29 @@ +title "Initial summary before deploy" +trace $CLI bundle summary -o json | jq .resources.features.feature1 + +title "Verify it does not exist yet" +trace musterr $CLI feature-engineering get-feature main.myschema.myfeature + +trace $CLI bundle deploy +trace print_requests.py //feature-engineering + +title "Summary should show the id and the Catalog Explorer url" +trace $CLI bundle summary -o json | jq ".resources.features.feature1 | {id, url}" + +title "Verify deployment" +trace $CLI feature-engineering get-feature main.myschema.myfeature | jq '{full_name, description, catalog_name, schema_name, name}' + +title "Update description (should update in place, not recreate)" +trace update_file.py databricks.yml DESCRIPTION1 DESCRIPTION2 +trace $CLI bundle deploy +trace print_requests.py //feature-engineering +trace $CLI feature-engineering get-feature main.myschema.myfeature | jq .description + +title "Change an immutable field (should plan a recreate)" +trace update_file.py databricks.yml amount other_column +trace $CLI bundle plan + +trace $CLI bundle destroy --auto-approve +trace print_requests.py //feature-engineering + +rm -f out.requests.txt diff --git a/acceptance/bundle/resources/features/basic/test.toml b/acceptance/bundle/resources/features/basic/test.toml new file mode 100644 index 00000000000..bd4b66fe75e --- /dev/null +++ b/acceptance/bundle/resources/features/basic/test.toml @@ -0,0 +1,3 @@ +Ignore = [ + ".databricks", +] diff --git a/acceptance/bundle/resources/features/lifecycle/databricks.yml.tmpl b/acceptance/bundle/resources/features/lifecycle/databricks.yml.tmpl new file mode 100644 index 00000000000..bde6fa2301b --- /dev/null +++ b/acceptance/bundle/resources/features/lifecycle/databricks.yml.tmpl @@ -0,0 +1,23 @@ +bundle: + name: test-features-lifecycle-$UNIQUE_NAME + +resources: + features: + feature1: + full_name: $SCHEMA_FULL_NAME.feature_$UNIQUE_NAME + description: DESCRIPTION1 + source: + delta_table_source: + full_name: $TABLE_NAME + function: + aggregation_function: + count_function: + input: value + time_window: + sliding: + window_duration: 604800s + slide_duration: 86400s + entities: + - name: id + timeseries_column: + name: timestamp diff --git a/acceptance/bundle/resources/features/lifecycle/out.test.toml b/acceptance/bundle/resources/features/lifecycle/out.test.toml new file mode 100644 index 00000000000..a927a5fbc06 --- /dev/null +++ b/acceptance/bundle/resources/features/lifecycle/out.test.toml @@ -0,0 +1,2 @@ +Cloud = false +EnvMatrix.DMS = ["", "true"] diff --git a/acceptance/bundle/resources/features/lifecycle/output.txt b/acceptance/bundle/resources/features/lifecycle/output.txt new file mode 100644 index 00000000000..144ebd4ef68 --- /dev/null +++ b/acceptance/bundle/resources/features/lifecycle/output.txt @@ -0,0 +1,40 @@ + +>>> [CLI] schemas create feat_test_[UNIQUE_NAME] main -o json +{ + "full_name": "main.feat_test_[UNIQUE_NAME]" +} + +>>> create_table.py main.feat_test_[UNIQUE_NAME].source_table +Created table main.feat_test_[UNIQUE_NAME].source_table +Inserted sample data into main.feat_test_[UNIQUE_NAME].source_table +Table main.feat_test_[UNIQUE_NAME].source_table is now visible (catalog_name=main) + +>>> [CLI] bundle deploy +Uploading bundle files to /Workspace/Users/[USERNAME]/.bundle/test-features-lifecycle-[UNIQUE_NAME]/default/files... +Created features.feature1 +Files: 2 uploaded, 0 deleted +Resources: 1 created, 0 changed, 0 deleted, 0 unchanged + +=== A second plan on an unchanged bundle must report no changes +>>> [CLI] bundle plan +Plan: 0 to add, 0 to change, 0 to delete, 1 unchanged + +=== Update the description in place, then confirm the plan settles again +>>> update_file.py databricks.yml DESCRIPTION1 DESCRIPTION2 + +>>> [CLI] bundle deploy +Uploading bundle files to /Workspace/Users/[USERNAME]/.bundle/test-features-lifecycle-[UNIQUE_NAME]/default/files... +Updated features.feature1 +Files: 1 uploaded, 0 deleted +Resources: 0 created, 1 changed, 0 deleted, 0 unchanged + +>>> [CLI] bundle plan +Plan: 0 to add, 0 to change, 0 to delete, 1 unchanged + +>>> [CLI] bundle destroy --auto-approve +The following resources will be deleted: + delete resources.features.feature1 + +All files and directories at the following location will be deleted: /Workspace/Users/[USERNAME]/.bundle/test-features-lifecycle-[UNIQUE_NAME]/default + +Destroy: 1 deleted diff --git a/acceptance/bundle/resources/features/lifecycle/script b/acceptance/bundle/resources/features/lifecycle/script new file mode 100644 index 00000000000..3e539c4e8c1 --- /dev/null +++ b/acceptance/bundle/resources/features/lifecycle/script @@ -0,0 +1,24 @@ +SCHEMA_NAME="feat_test_${UNIQUE_NAME}" +export SCHEMA_FULL_NAME="main.${SCHEMA_NAME}" +export TABLE_NAME="${SCHEMA_FULL_NAME}.source_table" + +trace $CLI schemas create "$SCHEMA_NAME" main -o json | jq '{full_name}' +trace create_table.py "$TABLE_NAME" + +cleanup() { + trace $CLI bundle destroy --auto-approve + $CLI schemas delete "$SCHEMA_FULL_NAME" --force 2>/dev/null || true +} +trap cleanup EXIT + +envsubst < databricks.yml.tmpl > databricks.yml + +trace $CLI bundle deploy + +title "A second plan on an unchanged bundle must report no changes" +trace $CLI bundle plan + +title "Update the description in place, then confirm the plan settles again" +trace update_file.py databricks.yml DESCRIPTION1 DESCRIPTION2 +trace $CLI bundle deploy +trace $CLI bundle plan diff --git a/acceptance/bundle/resources/features/lifecycle/test.toml b/acceptance/bundle/resources/features/lifecycle/test.toml new file mode 100644 index 00000000000..9286c0aeca9 --- /dev/null +++ b/acceptance/bundle/resources/features/lifecycle/test.toml @@ -0,0 +1,15 @@ +# Cloud = false until the feature-engineering service reaches the production stages: the cloud +# test environments are production workspaces and still serve a build that rejects the create +# payload this resource sends. Verified passing against a staging workspace running a build that +# carries the fix. Re-enable once production has it. +Cloud = false + +Ignore = [ + ".databricks", + "databricks.yml", +] + +# Fake SQL endpoint so create_table.py works against the local test server too. +[[Server]] +Pattern = "POST /api/2.0/sql/statements/" +Response.Body = '{"status": {"state": "SUCCEEDED"}, "manifest": {"schema": {"columns": []}}}' diff --git a/acceptance/bundle/resources/features/remote-delete/databricks.yml b/acceptance/bundle/resources/features/remote-delete/databricks.yml new file mode 100644 index 00000000000..59ad81a9d47 --- /dev/null +++ b/acceptance/bundle/resources/features/remote-delete/databricks.yml @@ -0,0 +1,18 @@ +bundle: + name: test-bundle + +resources: + features: + feature1: + full_name: main.myschema.myfeature + description: DESCRIPTION1 + source: + delta_table_source: + full_name: main.myschema.mysource + function: + column_selection: + column: amount + entities: + - name: account_id + timeseries_column: + name: event_ts diff --git a/acceptance/bundle/resources/features/remote-delete/out.test.toml b/acceptance/bundle/resources/features/remote-delete/out.test.toml new file mode 100644 index 00000000000..a927a5fbc06 --- /dev/null +++ b/acceptance/bundle/resources/features/remote-delete/out.test.toml @@ -0,0 +1,2 @@ +Cloud = false +EnvMatrix.DMS = ["", "true"] diff --git a/acceptance/bundle/resources/features/remote-delete/output.txt b/acceptance/bundle/resources/features/remote-delete/output.txt new file mode 100644 index 00000000000..3047716268a --- /dev/null +++ b/acceptance/bundle/resources/features/remote-delete/output.txt @@ -0,0 +1,20 @@ + +>>> [CLI] bundle deploy +Uploading bundle files to /Workspace/Users/[USERNAME]/.bundle/test-bundle/default/files... +Created features.feature1 +Files: 1 uploaded, 0 deleted +Resources: 1 created, 0 changed, 0 deleted, 0 unchanged + +=== Delete the feature out of band +>>> [CLI] feature-engineering delete-feature main.myschema.myfeature + +=== Plan should detect the feature is gone and re-create it +>>> [CLI] bundle plan +create features.feature1 + +Plan: 1 to add, 0 to change, 0 to delete, 0 unchanged + +>>> [CLI] bundle destroy --auto-approve +All files and directories at the following location will be deleted: /Workspace/Users/[USERNAME]/.bundle/test-bundle/default + +Destroy: 0 deleted diff --git a/acceptance/bundle/resources/features/remote-delete/script b/acceptance/bundle/resources/features/remote-delete/script new file mode 100644 index 00000000000..6c0b2522ad1 --- /dev/null +++ b/acceptance/bundle/resources/features/remote-delete/script @@ -0,0 +1,9 @@ +trace $CLI bundle deploy + +title "Delete the feature out of band" +trace $CLI feature-engineering delete-feature main.myschema.myfeature + +title "Plan should detect the feature is gone and re-create it" +trace $CLI bundle plan + +trace $CLI bundle destroy --auto-approve diff --git a/acceptance/bundle/resources/features/remote-delete/test.toml b/acceptance/bundle/resources/features/remote-delete/test.toml new file mode 100644 index 00000000000..b65b456fa41 --- /dev/null +++ b/acceptance/bundle/resources/features/remote-delete/test.toml @@ -0,0 +1,9 @@ +# Local only: simulates an out-of-band delete with a fixed feature name, so it can't run against a +# real workspace. +Cloud = false + +RecordRequests = false + +Ignore = [ + ".databricks", +] diff --git a/acceptance/experimental/open/output.txt b/acceptance/experimental/open/output.txt index 3be71a1c819..f178342ef54 100644 --- a/acceptance/experimental/open/output.txt +++ b/acceptance/experimental/open/output.txt @@ -9,7 +9,7 @@ === unknown resource type >>> [CLI] experimental open --url unknown 123 -Error: unknown resource type "unknown", must be one of: alerts, apps, catalogs, cluster_policies, clusters, dashboards, database_catalogs, database_instances, experiments, genie_spaces, instance_pools, jobs, mcp_services, model_provider_services, model_services, model_serving_endpoints, models, notebooks, pipelines, postgres_catalogs, postgres_synced_tables, quality_monitors, queries, registered_models, schemas, secrets, synced_database_tables, vector_search_endpoints, vector_search_indexes, volumes, warehouses +Error: unknown resource type "unknown", must be one of: alerts, apps, catalogs, cluster_policies, clusters, dashboards, database_catalogs, database_instances, experiments, features, genie_spaces, instance_pools, jobs, mcp_services, model_provider_services, model_services, model_serving_endpoints, models, notebooks, pipelines, postgres_catalogs, postgres_synced_tables, quality_monitors, queries, registered_models, schemas, secrets, synced_database_tables, vector_search_endpoints, vector_search_indexes, volumes, warehouses === test auto-completion handler >>> [CLI] __complete experimental open , @@ -22,6 +22,7 @@ dashboards database_catalogs database_instances experiments +features genie_spaces instance_pools jobs diff --git a/bundle/config/mutator/resourcemutator/apply_bundle_permissions_test.go b/bundle/config/mutator/resourcemutator/apply_bundle_permissions_test.go index c71bb2c5834..4ead9785e06 100644 --- a/bundle/config/mutator/resourcemutator/apply_bundle_permissions_test.go +++ b/bundle/config/mutator/resourcemutator/apply_bundle_permissions_test.go @@ -23,6 +23,7 @@ var unsupportedResources = []string{ "external_locations", "volumes", "schemas", + "features", "quality_monitors", "registered_models", "model_services", diff --git a/bundle/config/mutator/resourcemutator/apply_target_mode_test.go b/bundle/config/mutator/resourcemutator/apply_target_mode_test.go index 92867f9c61b..c040d91c0dc 100644 --- a/bundle/config/mutator/resourcemutator/apply_target_mode_test.go +++ b/bundle/config/mutator/resourcemutator/apply_target_mode_test.go @@ -156,6 +156,9 @@ func mockBundle(mode config.Mode) *bundle.Bundle { Volumes: map[string]*resources.Volume{ "volume1": {CreateVolumeRequestContent: catalog.CreateVolumeRequestContent{Name: "volume1"}}, }, + Features: map[string]*resources.Feature{ + "feature1": {Feature: ml.Feature{FullName: "catalog1.schema1.feature1"}}, + }, Clusters: map[string]*resources.Cluster{ "cluster1": {ClusterSpec: compute.ClusterSpec{ClusterName: "cluster1", SparkVersion: "13.2.x", NumWorkers: 1}}, }, @@ -491,6 +494,7 @@ func TestProcessTargetModeDefault(t *testing.T) { assert.Equal(t, "catalog1", b.Config.Resources.Catalogs["catalog1"].Name) assert.Equal(t, "schema1", b.Config.Resources.Schemas["schema1"].Name) assert.Equal(t, "volume1", b.Config.Resources.Volumes["volume1"].Name) + assert.Equal(t, "catalog1.schema1.feature1", b.Config.Resources.Features["feature1"].FullName) assert.Equal(t, "cluster1", b.Config.Resources.Clusters["cluster1"].ClusterName) assert.Equal(t, "instance_pool1", b.Config.Resources.InstancePools["instance_pool1"].InstancePoolName) assert.Equal(t, "sql_warehouse1", b.Config.Resources.SqlWarehouses["sql_warehouse1"].Name) @@ -535,6 +539,7 @@ func TestAppropriateResourcesAreRenamed(t *testing.T) { // Name field on these via embedded SDK types, hence the explicit skip. notUserNamed := []string{ "Apps", + "Features", "SecretScopes", "Secrets", "DatabaseInstances", diff --git a/bundle/config/mutator/resourcemutator/capture_uc_dependencies.go b/bundle/config/mutator/resourcemutator/capture_uc_dependencies.go index 488e215a2e2..a0e8d05fbde 100644 --- a/bundle/config/mutator/resourcemutator/capture_uc_dependencies.go +++ b/bundle/config/mutator/resourcemutator/capture_uc_dependencies.go @@ -190,6 +190,20 @@ func (m *captureUCDependencies) Apply(ctx context.Context, b *bundle.Bundle) dia catalogName, schemaName := parts[0], parts[1] idx.Name = resolveCatalog(b, catalogName) + "." + resolveSchema(b, catalogName, schemaName) + "." + parts[2] } + for _, f := range b.Config.Resources.Features { + if f == nil { + continue + } + if dynvar.ContainsVariableReference(f.FullName) { + continue + } + parts, ok := splitUCName(f.FullName, 3) + if !ok { + continue + } + catalogName, schemaName := parts[0], parts[1] + f.FullName = resolveCatalog(b, catalogName) + "." + resolveSchema(b, catalogName, schemaName) + "." + parts[2] + } for _, mse := range b.Config.Resources.ModelServingEndpoints { if mse == nil { continue diff --git a/bundle/config/mutator/resourcemutator/capture_uc_dependencies_test.go b/bundle/config/mutator/resourcemutator/capture_uc_dependencies_test.go index e5281a42a77..347f0b72caa 100644 --- a/bundle/config/mutator/resourcemutator/capture_uc_dependencies_test.go +++ b/bundle/config/mutator/resourcemutator/capture_uc_dependencies_test.go @@ -7,6 +7,7 @@ import ( "github.com/databricks/cli/bundle/config" "github.com/databricks/cli/bundle/config/resources" "github.com/databricks/databricks-sdk-go/service/catalog" + "github.com/databricks/databricks-sdk-go/service/ml" "github.com/databricks/databricks-sdk-go/service/pipelines" "github.com/databricks/databricks-sdk-go/service/serving" "github.com/databricks/databricks-sdk-go/service/vectorsearch" @@ -142,6 +143,11 @@ func TestCaptureUCDependencies(t *testing.T) { Name: "mycatalog.myschema.myindex", }}, }, + Features: map[string]*resources.Feature{ + "my_feature": {Feature: ml.Feature{ + FullName: "mycatalog.myschema.myfeature", + }}, + }, McpServices: map[string]*resources.McpService{ "my_mcp_service": {McpServiceConfig: resources.McpServiceConfig{ Parent: "schemas/mycatalog.myschema", McpServiceId: "mymcp", @@ -191,6 +197,9 @@ func TestCaptureUCDependencies(t *testing.T) { // Vector search index (three-part "catalog.schema.index" name). assert.Equal(t, catalogRef+"."+schemaRef+".myindex", b.Config.Resources.VectorSearchIndexes["my_index"].Name) + // Feature (three-part "catalog.schema.feature" full name). + assert.Equal(t, catalogRef+"."+schemaRef+".myfeature", b.Config.Resources.Features["my_feature"].FullName) + // MCP service (same compound parent field). assert.Equal(t, "schemas/"+catalogRef+"."+schemaRef, b.Config.Resources.McpServices["my_mcp_service"].Parent) diff --git a/bundle/config/mutator/resourcemutator/run_as_test.go b/bundle/config/mutator/resourcemutator/run_as_test.go index 5d1aa602d02..200939f2dd0 100644 --- a/bundle/config/mutator/resourcemutator/run_as_test.go +++ b/bundle/config/mutator/resourcemutator/run_as_test.go @@ -48,6 +48,7 @@ func allResourceTypes(t *testing.T) []string { "database_instances", "experiments", "external_locations", + "features", "genie_spaces", "instance_pools", "internal_immutable_snapshots", @@ -202,6 +203,7 @@ var allowList = []string{ "postgres_synced_tables", "registered_models", "experiments", + "features", "genie_spaces", "instance_pools", "job_runs", diff --git a/bundle/config/resources.go b/bundle/config/resources.go index f13a44617c4..6187ad8a144 100644 --- a/bundle/config/resources.go +++ b/bundle/config/resources.go @@ -15,6 +15,7 @@ type Resources struct { JobRuns map[string]*resources.JobRun `json:"job_runs,omitempty"` Pipelines map[string]*resources.Pipeline `json:"pipelines,omitempty"` + Features map[string]*resources.Feature `json:"features,omitempty"` Models map[string]*resources.MlflowModel `json:"models,omitempty"` Experiments map[string]*resources.MlflowExperiment `json:"experiments,omitempty"` ModelServingEndpoints map[string]*resources.ModelServingEndpoint `json:"model_serving_endpoints,omitempty"` @@ -113,6 +114,7 @@ func (r *Resources) AllResources() []ResourceGroup { collectResourceMap(descriptions["pipelines"], r.Pipelines), collectResourceMap(descriptions["models"], r.Models), collectResourceMap(descriptions["experiments"], r.Experiments), + collectResourceMap(descriptions["features"], r.Features), collectResourceMap(descriptions["model_serving_endpoints"], r.ModelServingEndpoints), collectResourceMap(descriptions["model_services"], r.ModelServices), collectResourceMap(descriptions["mcp_services"], r.McpServices), @@ -187,6 +189,7 @@ func SupportedResources() map[string]resources.ResourceDescription { "pipelines": (&resources.Pipeline{}).ResourceDescription(), "models": (&resources.MlflowModel{}).ResourceDescription(), "experiments": (&resources.MlflowExperiment{}).ResourceDescription(), + "features": (&resources.Feature{}).ResourceDescription(), "instance_pools": (&resources.InstancePool{}).ResourceDescription(), "model_serving_endpoints": (&resources.ModelServingEndpoint{}).ResourceDescription(), "model_services": (&resources.ModelService{}).ResourceDescription(), diff --git a/bundle/config/resources/feature.go b/bundle/config/resources/feature.go new file mode 100644 index 00000000000..05d27624a2e --- /dev/null +++ b/bundle/config/resources/feature.go @@ -0,0 +1,68 @@ +package resources + +import ( + "context" + "net/url" + + "github.com/databricks/databricks-sdk-go/apierr" + + "github.com/databricks/cli/libs/log" + "github.com/databricks/cli/libs/workspaceurls" + "github.com/databricks/databricks-sdk-go" + "github.com/databricks/databricks-sdk-go/marshal" + "github.com/databricks/databricks-sdk-go/service/ml" +) + +type Feature struct { + BaseResource + ID string `json:"id,omitempty" bundle:"readonly"` + ml.Feature +} + +func (f *Feature) UnmarshalJSON(b []byte) error { + return marshal.Unmarshal(b, f) +} + +func (f Feature) MarshalJSON() ([]byte, error) { + return marshal.Marshal(f) +} + +func (f *Feature) Exists(ctx context.Context, w *databricks.WorkspaceClient, fullName string) (bool, error) { + _, err := w.FeatureEngineering.GetFeature(ctx, ml.GetFeatureRequest{ + FullName: fullName, + }) + if err != nil { + log.Debugf(ctx, "feature with full name %s does not exist: %v", fullName, err) + + if apierr.IsMissing(err) { + return false, nil + } + + return false, err + } + + return true, nil +} + +func (*Feature) ResourceDescription() ResourceDescription { + return ResourceDescription{ + SingularName: "feature", + PluralName: "features", + SingularTitle: "Feature", + PluralTitle: "Features", + } +} + +func (f *Feature) InitializeURL(baseURL url.URL) { + if f.ID == "" { + return + } + f.URL = workspaceurls.ResourceURL(baseURL, "features", f.ID) +} + +// GetName returns the fully qualified name. Callers read it off the config, which only ever carries +// what the user wrote plus id and modified_status (see statemgmt.StateToBundle), so the OUTPUT_ONLY +// name field is always empty here. +func (f *Feature) GetName() string { + return f.FullName +} diff --git a/bundle/config/resources_test.go b/bundle/config/resources_test.go index c666c8964c0..9528e8cbd20 100644 --- a/bundle/config/resources_test.go +++ b/bundle/config/resources_test.go @@ -361,6 +361,13 @@ func TestResourcesBindSupport(t *testing.T) { }, }, }, + Features: map[string]*resources.Feature{ + "my_feature": { + Feature: ml.Feature{ + FullName: "main.myschema.my_feature", + }, + }, + }, VectorSearchEndpoints: map[string]*resources.VectorSearchEndpoint{ "my_vector_search_endpoint": { CreateEndpoint: vectorsearch.CreateEndpoint{ @@ -391,6 +398,7 @@ func TestResourcesBindSupport(t *testing.T) { m.GetMockJobsAPI().EXPECT().GetRun(mock.Anything, mock.Anything).Return(nil, nil) m.GetMockPipelinesAPI().EXPECT().Get(mock.Anything, mock.Anything).Return(nil, nil) m.GetMockExperimentsAPI().EXPECT().GetExperiment(mock.Anything, mock.Anything).Return(nil, nil) + m.GetMockFeatureEngineeringAPI().EXPECT().GetFeature(mock.Anything, mock.Anything).Return(nil, nil) m.GetMockRegisteredModelsAPI().EXPECT().Get(mock.Anything, mock.Anything).Return(nil, nil) m.GetMockCatalogsAPI().EXPECT().GetByName(mock.Anything, mock.Anything).Return(nil, nil) m.GetMockExternalLocationsAPI().EXPECT().GetByName(mock.Anything, mock.Anything).Return(nil, nil) diff --git a/bundle/direct/dresources/all.go b/bundle/direct/dresources/all.go index 27e8496b83e..88f08048235 100644 --- a/bundle/direct/dresources/all.go +++ b/bundle/direct/dresources/all.go @@ -11,6 +11,7 @@ var SupportedResources = map[string]any{ "job_runs": (*ResourceJobRun)(nil), "pipelines": (*ResourcePipeline)(nil), "experiments": (*ResourceExperiment)(nil), + "features": (*ResourceFeature)(nil), "catalogs": (*ResourceCatalog)(nil), "schemas": (*ResourceSchema)(nil), "external_locations": (*ResourceExternalLocation)(nil), diff --git a/bundle/direct/dresources/all_test.go b/bundle/direct/dresources/all_test.go index a0ef21b6f6d..e7cd954e003 100644 --- a/bundle/direct/dresources/all_test.go +++ b/bundle/direct/dresources/all_test.go @@ -133,6 +133,23 @@ var testConfig map[string]any = map[string]any{ }, }, + "features": &resources.Feature{ + Feature: ml.Feature{ + FullName: "main.default.my_feature", + Description: "Test feature", + Source: ml.DataSource{ + DeltaTableSource: &ml.DeltaTableSource{ + FullName: "main.default.my_source", + }, + }, + Function: ml.Function{ + ColumnSelection: &ml.ColumnSelection{Column: "amount"}, + }, + Entities: []ml.EntityColumn{{Name: "id"}}, + TimeseriesColumn: &ml.TimeseriesColumn{Name: "ts"}, + }, + }, + "experiments": &resources.MlflowExperiment{ CreateExperiment: ml.CreateExperiment{ Name: "my-experiment", diff --git a/bundle/direct/dresources/apitypes.generated.yml b/bundle/direct/dresources/apitypes.generated.yml index af96326fa4c..317a04442c4 100644 --- a/bundle/direct/dresources/apitypes.generated.yml +++ b/bundle/direct/dresources/apitypes.generated.yml @@ -20,6 +20,8 @@ experiments: ml.CreateExperiment external_locations: catalog.CreateExternalLocation +features: ml.Feature + genie_spaces: dashboards.GenieUpdateSpaceRequest instance_pools: compute.CreateInstancePool diff --git a/bundle/direct/dresources/configs/features.generated.yml b/bundle/direct/dresources/configs/features.generated.yml new file mode 100644 index 00000000000..1ca74675a5b --- /dev/null +++ b/bundle/direct/dresources/configs/features.generated.yml @@ -0,0 +1,21 @@ +# Generated, do not edit. + +recreate_on_changes: + - field: full_name + reason: spec:immutable + - field: function + reason: spec:immutable + - field: source + reason: spec:immutable + +ignore_remote_changes: + - field: catalog_name + reason: spec:output_only + - field: created_at + reason: spec:output_only + - field: created_by + reason: spec:output_only + - field: name + reason: spec:output_only + - field: schema_name + reason: spec:output_only diff --git a/bundle/direct/dresources/configs/features.yml b/bundle/direct/dresources/configs/features.yml new file mode 100644 index 00000000000..809f978b8dd --- /dev/null +++ b/bundle/direct/dresources/configs/features.yml @@ -0,0 +1,22 @@ +recreate_on_changes: + # UpdateFeature accepts only `description`, so a change to any of these cannot be applied in + # place. The spec deliberately leaves them OPTIONAL rather than IMMUTABLE so feature versioning + # stays open, which means features.generated.yml will never cover them and these entries are + # permanent. + - field: entities + reason: immutable + - field: timeseries_column + reason: immutable + - field: lineage_context + reason: immutable + - field: inputs + reason: immutable + - field: time_window + reason: immutable + - field: filter_condition + reason: immutable + +stable_output_fields: + # name is the last component of full_name, which is immutable, so it cannot change on an + # in-place update; see the stable_output_fields doc in config.go and ES-2202624. + - field: name diff --git a/bundle/direct/dresources/feature.go b/bundle/direct/dresources/feature.go new file mode 100644 index 00000000000..50cdbf3284c --- /dev/null +++ b/bundle/direct/dresources/feature.go @@ -0,0 +1,60 @@ +package dresources + +import ( + "context" + "slices" + "strings" + + "github.com/databricks/cli/bundle/config/resources" + "github.com/databricks/databricks-sdk-go" + "github.com/databricks/databricks-sdk-go/service/ml" +) + +// https://docs.databricks.com/api/workspace/featureengineering +// Terraform: https://github.com/databricks/terraform-provider-databricks/blob/main/internal/providers/pluginfw/products/featureengineering/ +type ResourceFeature struct { + client *databricks.WorkspaceClient +} + +func (*ResourceFeature) New(client *databricks.WorkspaceClient) *ResourceFeature { + return &ResourceFeature{client: client} +} + +func (*ResourceFeature) PrepareState(input *resources.Feature) *ml.Feature { + return &input.Feature +} + +func (r *ResourceFeature) DoRead(ctx context.Context, id string) (*ml.Feature, error) { + return r.client.FeatureEngineering.GetFeature(ctx, ml.GetFeatureRequest{FullName: id}) +} + +func (r *ResourceFeature) DoCreate(ctx context.Context, config *ml.Feature) (string, *ml.Feature, error) { + response, err := r.client.FeatureEngineering.CreateFeature(ctx, ml.CreateFeatureRequest{Feature: *config}) + if err != nil { + return "", nil, err + } + return response.FullName, response, nil +} + +// The update endpoint rejects a mask containing anything else, so "*" cannot be used. Every other +// field is excluded at plan level in configs/features.yml and features.generated.yml. +var featureUpdatableFields = []string{"description"} + +func (r *ResourceFeature) DoUpdate(ctx context.Context, id string, config *ml.Feature, _ *PlanEntry) (*ml.Feature, error) { + // The whole feature goes in the body, with the mask limiting what is applied: full_name, + // function and source have no omitempty, so a body built from just the updatable fields would + // send them blanked out. Description is forced so clearing it sends "" rather than omitting it. + update := *config + update.FullName = id + update.ForceSendFields = append(slices.Clone(config.ForceSendFields), "Description") + + return r.client.FeatureEngineering.UpdateFeature(ctx, ml.UpdateFeatureRequest{ + FullName: id, + Feature: update, + UpdateMask: strings.Join(featureUpdatableFields, ","), + }) +} + +func (r *ResourceFeature) DoDelete(ctx context.Context, id string, _ *ml.Feature) error { + return r.client.FeatureEngineering.DeleteFeature(ctx, ml.DeleteFeatureRequest{FullName: id}) +} diff --git a/bundle/internal/schema/annotations.yml b/bundle/internal/schema/annotations.yml index 710e4b8373f..1a5c4ccbfdd 100644 --- a/bundle/internal/schema/annotations.yml +++ b/bundle/internal/schema/annotations.yml @@ -917,6 +917,63 @@ resources: "lifecycle": "description": |- Settings that control the deployment lifecycle of the resource, such as preventing it from being destroyed. + "features": + "description": |- + The feature definitions for the bundle, where each key is the name of the feature. + "markdown_description": |- + The feature definitions for the bundle, where each key is the name of the feature. + "$fields": + "filter_condition": + "description": |- + PLACEHOLDER + "function": + "$fields": + "aggregation_function": + "$fields": + "time_window": + "$fields": + "continuous": + "description": |- + PLACEHOLDER + "sliding": + "description": |- + PLACEHOLDER + "tumbling": + "description": |- + PLACEHOLDER + "extra_parameters": + "description": |- + PLACEHOLDER + "function_type": + "description": |- + PLACEHOLDER + "inputs": + "description": |- + PLACEHOLDER + "lifecycle": + "description": |- + PLACEHOLDER + "source": + "$fields": + "delta_table_source": + "$fields": + "entity_columns": + "description": |- + PLACEHOLDER + "timeseries_column": + "description": |- + PLACEHOLDER + "kafka_source": + "$fields": + "entity_column_identifiers": + "description": |- + PLACEHOLDER + "timeseries_column_identifier": + "description": |- + PLACEHOLDER + "time_window": + "description": |- + PLACEHOLDER "genie_spaces": "description": |- PLACEHOLDER diff --git a/bundle/internal/validation/generated/enum_fields.go b/bundle/internal/validation/generated/enum_fields.go index cde86a67a1a..6eefec0fd8d 100644 --- a/bundle/internal/validation/generated/enum_fields.go +++ b/bundle/internal/validation/generated/enum_fields.go @@ -65,6 +65,9 @@ var EnumFields = map[string][]string{ "resources.external_locations.*.encryption_details.sse_encryption_details.algorithm": {"AWS_SSE_KMS", "AWS_SSE_S3"}, "resources.external_locations.*.grants[*].privileges[*]": {"ACCESS", "ALL_PRIVILEGES", "APPLY_TAG", "BROWSE", "CREATE", "CREATE_CATALOG", "CREATE_CLEAN_ROOM", "CREATE_CONNECTION", "CREATE_EXTERNAL_LOCATION", "CREATE_EXTERNAL_TABLE", "CREATE_EXTERNAL_VOLUME", "CREATE_FOREIGN_CATALOG", "CREATE_FOREIGN_SECURABLE", "CREATE_FUNCTION", "CREATE_MANAGED_STORAGE", "CREATE_MATERIALIZED_VIEW", "CREATE_MODEL", "CREATE_PROVIDER", "CREATE_RECIPIENT", "CREATE_SCHEMA", "CREATE_SERVICE_CREDENTIAL", "CREATE_SHARE", "CREATE_STORAGE_CREDENTIAL", "CREATE_TABLE", "CREATE_VIEW", "CREATE_VOLUME", "EXECUTE", "EXECUTE_CLEAN_ROOM_TASK", "EXTERNAL_USE_LOCATION", "EXTERNAL_USE_SCHEMA", "MANAGE", "MANAGE_ALLOWLIST", "MODIFY", "MODIFY_CLEAN_ROOM", "READ_FILES", "READ_METADATA", "READ_PRIVATE_FILES", "READ_VOLUME", "REFRESH", "SELECT", "SET_SHARE_PERMISSION", "USAGE", "USE_CATALOG", "USE_CONNECTION", "USE_MARKETPLACE_ASSETS", "USE_PROVIDER", "USE_RECIPIENT", "USE_SCHEMA", "USE_SHARE", "WRITE_FILES", "WRITE_PRIVATE_FILES", "WRITE_VOLUME"}, + "resources.features.*.function.function_type": {"APPROX_COUNT_DISTINCT", "APPROX_PERCENTILE", "AVG", "COUNT", "FIRST", "FUNCTION_TYPE_UNSPECIFIED", "LAST", "MAX", "MIN", "STDDEV_POP", "STDDEV_SAMP", "SUM", "VAR_POP", "VAR_SAMP"}, + "resources.features.*.source.request_source.flat_schema.fields[*].data_type": {"BINARY", "BOOLEAN", "DATE", "DECIMAL", "DOUBLE", "FLOAT", "INTEGER", "LONG", "SHORT", "STRING", "TIMESTAMP"}, + "resources.genie_spaces.*.permissions[*].level": {"CAN_ATTACH_TO", "CAN_BIND", "CAN_CREATE", "CAN_CREATE_APP", "CAN_EDIT", "CAN_EDIT_METADATA", "CAN_MANAGE", "CAN_MANAGE_PRODUCTION_VERSIONS", "CAN_MANAGE_RUN", "CAN_MANAGE_STAGING_VERSIONS", "CAN_MONITOR", "CAN_MONITOR_ONLY", "CAN_QUERY", "CAN_READ", "CAN_RESTART", "CAN_RUN", "CAN_USE", "CAN_VIEW", "CAN_VIEW_METADATA", "IS_OWNER"}, "resources.instance_pools.*.aws_attributes.availability": {"ON_DEMAND", "SPOT"}, diff --git a/bundle/internal/validation/generated/required_fields.go b/bundle/internal/validation/generated/required_fields.go index 6527fed534f..7049ae71b1b 100644 --- a/bundle/internal/validation/generated/required_fields.go +++ b/bundle/internal/validation/generated/required_fields.go @@ -75,6 +75,45 @@ var RequiredFields = map[string][]string{ "resources.external_locations.*": {"credential_name", "name", "url"}, + "resources.features.*": {"full_name", "function", "source"}, + "resources.features.*.entities[*]": {"name"}, + "resources.features.*.function.aggregation_function.approx_count_distinct": {"input"}, + "resources.features.*.function.aggregation_function.approx_percentile": {"input", "percentile"}, + "resources.features.*.function.aggregation_function.avg": {"input"}, + "resources.features.*.function.aggregation_function.count_function": {"input"}, + "resources.features.*.function.aggregation_function.first": {"input"}, + "resources.features.*.function.aggregation_function.first_distinct": {"input", "n"}, + "resources.features.*.function.aggregation_function.first_n": {"input", "n"}, + "resources.features.*.function.aggregation_function.last": {"input"}, + "resources.features.*.function.aggregation_function.last_distinct": {"input", "n"}, + "resources.features.*.function.aggregation_function.last_n": {"input", "n"}, + "resources.features.*.function.aggregation_function.max": {"input"}, + "resources.features.*.function.aggregation_function.min": {"input"}, + "resources.features.*.function.aggregation_function.stddev_pop": {"input"}, + "resources.features.*.function.aggregation_function.stddev_samp": {"input"}, + "resources.features.*.function.aggregation_function.sum": {"input"}, + "resources.features.*.function.aggregation_function.time_window.continuous": {"window_duration"}, + "resources.features.*.function.aggregation_function.time_window.sliding": {"slide_duration"}, + "resources.features.*.function.aggregation_function.time_window.tumbling": {"window_duration"}, + "resources.features.*.function.aggregation_function.var_pop": {"input"}, + "resources.features.*.function.aggregation_function.var_samp": {"input"}, + "resources.features.*.function.column_selection": {"column"}, + "resources.features.*.function.custom_udf": {"function_path"}, + "resources.features.*.function.custom_udf.input_bindings[*]": {"column", "parameter"}, + "resources.features.*.function.extra_parameters[*]": {"key", "value"}, + "resources.features.*.source.delta_table_source": {"full_name"}, + "resources.features.*.source.feature_view_source.feature_references[*]": {"feature"}, + "resources.features.*.source.kafka_source": {"name"}, + "resources.features.*.source.kafka_source.entity_column_identifiers[*]": {"variant_expr_path"}, + "resources.features.*.source.kafka_source.timeseries_column_identifier": {"variant_expr_path"}, + "resources.features.*.source.request_source.flat_schema": {"fields"}, + "resources.features.*.source.request_source.flat_schema.fields[*]": {"data_type", "name"}, + "resources.features.*.source.stream_source": {"full_name"}, + "resources.features.*.time_window.continuous": {"window_duration"}, + "resources.features.*.time_window.sliding": {"slide_duration"}, + "resources.features.*.time_window.tumbling": {"window_duration"}, + "resources.features.*.timeseries_column": {"name"}, + "resources.genie_spaces.*.permissions[*]": {"level"}, "resources.instance_pools.*": {"instance_pool_name", "node_type_id"}, diff --git a/bundle/schema/jsonschema.json b/bundle/schema/jsonschema.json index f52e8059ee4..c0c65425dc7 100644 --- a/bundle/schema/jsonschema.json +++ b/bundle/schema/jsonschema.json @@ -1026,6 +1026,94 @@ } ] }, + "resources.Feature": { + "oneOf": [ + { + "type": "object", + "properties": { + "description": { + "description": "[Private Preview] The description of the feature.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "entities": { + "description": "[Private Preview] The entity columns for the feature, used as aggregation keys and for query-time lookup.", + "$ref": "#/$defs/slice/github.com/databricks/databricks-sdk-go/service/ml.EntityColumn", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "filter_condition": { + "description": "[Private Preview]", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "full_name": { + "description": "[Private Preview] The full three-part name (catalog, schema, name) of the feature. This is the\nfeature's resource identifier; the catalog_name, schema_name, and name fields\nbelow are OUTPUT_ONLY decomposed views of this value.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "function": { + "description": "[Private Preview] The function by which the feature is computed.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.Function", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "inputs": { + "description": "[Private Preview]", + "$ref": "#/$defs/slice/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "lifecycle": { + "$ref": "#/$defs/github.com/databricks/cli/bundle/config/resources.Lifecycle" + }, + "lineage_context": { + "description": "[Private Preview] Lineage context information for this feature.\nWARNING: This field is primarily intended for internal use by Databricks systems and\nis automatically populated when features are created through Databricks notebooks or jobs.\nUsers should not manually set this field as incorrect values may lead to inaccurate lineage tracking or unexpected behavior.\nThis field will be set by feature-engineering client and should be left unset by SDK and terraform users.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.LineageContext", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "source": { + "description": "[Private Preview] The data source of the feature.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.DataSource", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "time_window": { + "description": "[Private Preview]", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.TimeWindow", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "timeseries_column": { + "description": "[Private Preview] Column recording time, used for point-in-time joins, backfills, and aggregations.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.TimeseriesColumn", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "full_name", + "function", + "source" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, "resources.GenieSpace": { "oneOf": [ { @@ -4023,6 +4111,11 @@ "external_locations": { "$ref": "#/$defs/map/github.com/databricks/cli/bundle/config/resources.ExternalLocation" }, + "features": { + "description": "The feature definitions for the bundle, where each key is the name of the feature.", + "$ref": "#/$defs/map/github.com/databricks/cli/bundle/config/resources.Feature", + "markdownDescription": "The feature definitions for the bundle, where each key is the name of the feature." + }, "genie_spaces": { "$ref": "#/$defs/map/github.com/databricks/cli/bundle/config/resources.GenieSpace" }, @@ -12113,16 +12206,122 @@ } ] }, - "ml.ExperimentPermissionLevel": { + "ml.AggregationFunction": { "oneOf": [ { - "type": "string", - "description": "Permission level", - "enum": [ - "CAN_MANAGE", - "CAN_EDIT", - "CAN_READ" - ] + "type": "object", + "description": "An aggregation function applied over a time window.", + "properties": { + "approx_count_distinct": { + "description": "[Private Preview] Computes the approximate count of distinct values.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.ApproxCountDistinctFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "approx_percentile": { + "description": "[Private Preview] Computes the approximate percentile of values.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.ApproxPercentileFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "avg": { + "description": "[Private Preview] Computes the average of values.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.AvgFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "count_function": { + "description": "[Private Preview] Computes the count of values.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.CountFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "first": { + "description": "[Private Preview] Returns the first value.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FirstFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "first_distinct": { + "description": "[Private Preview] Returns the first N distinct values, ordered by the feature's timeseries column.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FirstDistinctFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "first_n": { + "description": "[Private Preview] Returns the first N values, ordered by the feature's timeseries column.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FirstNFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "last": { + "description": "[Private Preview] Returns the last value.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.LastFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "last_distinct": { + "description": "[Private Preview] Returns the last N distinct values, ordered by the feature's timeseries column.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.LastDistinctFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "last_n": { + "description": "[Private Preview] Returns the last N values, ordered by the feature's timeseries column.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.LastNFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "max": { + "description": "[Private Preview] Computes the maximum value.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.MaxFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "min": { + "description": "[Private Preview] Computes the minimum value.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.MinFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "stddev_pop": { + "description": "[Private Preview] Computes the population standard deviation.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.StddevPopFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "stddev_samp": { + "description": "[Private Preview] Computes the sample standard deviation.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.StddevSampFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "sum": { + "description": "[Private Preview] Computes the sum of values.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.SumFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "time_window": { + "description": "[Private Preview] The time window over which the aggregation is computed.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.TimeWindow", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "var_pop": { + "description": "[Private Preview] Computes the population variance.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.VarPopFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "var_samp": { + "description": "[Private Preview] Computes the sample variance.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.VarSampFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false }, { "type": "string", @@ -12130,24 +12329,29 @@ } ] }, - "ml.ExperimentTag": { + "ml.ApproxCountDistinctFunction": { "oneOf": [ { "type": "object", - "description": "A tag for an experiment.", + "description": "Computes the approximate count of distinct values.", "properties": { - "key": { - "description": "The tag key.", + "input": { + "description": "[Private Preview] The input column from which the approximate count of distinct values is computed.", "$ref": "#/$defs/string", - "x-databricks-launch-stage": "GA" + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true }, - "value": { - "description": "The tag value.", - "$ref": "#/$defs/string", - "x-databricks-launch-stage": "GA" + "relative_sd": { + "description": "[Private Preview] The maximum relative standard deviation allowed (default defined by Spark).", + "$ref": "#/$defs/float64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true } }, - "additionalProperties": false + "additionalProperties": false, + "required": [ + "input" + ] }, { "type": "string", @@ -12155,20 +12359,36 @@ } ] }, - "ml.ExperimentTraceLocation": { + "ml.ApproxPercentileFunction": { "oneOf": [ { "type": "object", - "description": "The storage location for an experiment's traces.", + "description": "Computes the approximate percentile of values.", "properties": { - "uc_trace_location": { - "description": "[Private Preview] A Unity Catalog schema where the experiment's traces are stored as\nDelta tables.", - "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.UcTraceLocation", + "accuracy": { + "description": "[Private Preview] The accuracy parameter (higher is more accurate but slower).", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "input": { + "description": "[Private Preview] The input column from which the approximate percentile is computed.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "percentile": { + "description": "[Private Preview] The percentile value to compute (between 0 and 1).", + "$ref": "#/$defs/float64", "x-databricks-launch-stage": "PRIVATE_PREVIEW", "doNotSuggest": true } }, - "additionalProperties": false + "additionalProperties": false, + "required": [ + "input", + "percentile" + ] }, { "type": "string", @@ -12176,24 +12396,23 @@ } ] }, - "ml.ModelTag": { + "ml.AvgFunction": { "oneOf": [ { "type": "object", - "description": "Tag for a registered model", + "description": "Computes the average of values.", "properties": { - "key": { - "description": "The tag key.", - "$ref": "#/$defs/string", - "x-databricks-launch-stage": "GA" - }, - "value": { - "description": "The tag value.", + "input": { + "description": "[Private Preview] The input column from which the average is computed. For Kafka sources, use dot-prefixed path\nnotation (e.g., \"value.amount\"). For nested fields, the leaf node name is used.\nColon-prefixed notation (e.g., \"value:amount\") is supported for backwards\ncompatibility but is deprecated; migrate to dot notation.", "$ref": "#/$defs/string", - "x-databricks-launch-stage": "GA" + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true } }, - "additionalProperties": false + "additionalProperties": false, + "required": [ + "input" + ] }, { "type": "string", @@ -12201,17 +12420,21 @@ } ] }, - "ml.RegisteredModelPermissionLevel": { + "ml.ColumnIdentifier": { "oneOf": [ { - "type": "string", - "description": "Permission level", - "enum": [ - "CAN_MANAGE", - "CAN_MANAGE_PRODUCTION_VERSIONS", - "CAN_MANAGE_STAGING_VERSIONS", - "CAN_EDIT", - "CAN_READ" + "type": "object", + "properties": { + "variant_expr_path": { + "description": "[Private Preview] String representation of the column name using dot-prefixed path notation.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "variant_expr_path" ] }, { @@ -12220,26 +12443,14 @@ } ] }, - "ml.UcTraceLocation": { + "ml.ColumnSelection": { "oneOf": [ { "type": "object", - "description": "A Unity Catalog trace storage location. Traces are stored as Delta tables\nin the specified catalog and schema.", + "description": "A ColumnSelection function, equivalent to the LAST() record of an entity over a lifetime window", "properties": { - "catalog": { - "description": "[Private Preview] The name of the Unity Catalog catalog.", - "$ref": "#/$defs/string", - "x-databricks-launch-stage": "PRIVATE_PREVIEW", - "doNotSuggest": true - }, - "schema": { - "description": "[Private Preview] The name of the Unity Catalog schema within `catalog`.", - "$ref": "#/$defs/string", - "x-databricks-launch-stage": "PRIVATE_PREVIEW", - "doNotSuggest": true - }, - "table_prefix": { - "description": "[Private Preview] The prefix for the trace tables, which are named\n`{catalog}.{schema}.{table_prefix}_otel_*`. May only contain letters,\ndigits, and underscores, and may be at most 238 characters. When unset, a\nserver-generated prefix derived from the experiment ID is used and this\nfield stays empty on read; the resolved value is always available in\n`effective_table_prefix`.", + "column": { + "description": "[Private Preview] Column name from source to select as the feature value.", "$ref": "#/$defs/string", "x-databricks-launch-stage": "PRIVATE_PREVIEW", "doNotSuggest": true @@ -12247,8 +12458,7 @@ }, "additionalProperties": false, "required": [ - "catalog", - "schema" + "column" ] }, { @@ -12257,20 +12467,28 @@ } ] }, - "pipelines.ApiSourceConnectorConfig": { + "ml.ContinuousWindow": { "oneOf": [ { "type": "object", - "description": "Top-level configuration for API Source connectors with arbitrary configuration.", "properties": { - "configs": { - "description": "[Private Preview] Arbitrary key-value configuration values for the API Source connector.", - "$ref": "#/$defs/map/string", + "offset": { + "description": "[Private Preview] The offset of the continuous window (must be non-positive).", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "window_duration": { + "description": "[Private Preview] The duration of the continuous window (must be positive).", + "$ref": "#/$defs/string", "x-databricks-launch-stage": "PRIVATE_PREVIEW", "doNotSuggest": true } }, - "additionalProperties": false + "additionalProperties": false, + "required": [ + "window_duration" + ] }, { "type": "string", @@ -12278,20 +12496,23 @@ } ] }, - "pipelines.ApiSourceConnectorOptions": { + "ml.CountFunction": { "oneOf": [ { "type": "object", - "description": "Options for API Source connectors with arbitrary configuration.", + "description": "Computes the count of values.", "properties": { - "options": { - "description": "[Private Preview] Arbitrary key-value configuration options for the API Source connector.", - "$ref": "#/$defs/map/string", + "input": { + "description": "[Private Preview] The input column from which the count is computed. For Kafka sources, use dot-prefixed path\nnotation (e.g., \"value.amount\"). For nested fields, the leaf node name is used.\nColon-prefixed notation (e.g., \"value:amount\") is supported for backwards\ncompatibility but is deprecated; migrate to dot notation.", + "$ref": "#/$defs/string", "x-databricks-launch-stage": "PRIVATE_PREVIEW", "doNotSuggest": true } }, - "additionalProperties": false + "additionalProperties": false, + "required": [ + "input" + ] }, { "type": "string", @@ -12299,17 +12520,1387 @@ } ] }, - "pipelines.AutoFullRefreshPolicy": { + "ml.CustomUdf": { "oneOf": [ { "type": "object", - "description": "Policy for auto full refresh.", + "description": "A CustomUdf function applies a registered Unity Catalog function row-wise to\nsource columns, producing a single output column per row.", "properties": { - "enabled": { - "description": "[Public Preview] (Required, Mutable) Whether to enable auto full refresh or not.", - "$ref": "#/$defs/bool", - "x-databricks-launch-stage": "PUBLIC_PREVIEW" - }, + "function_path": { + "description": "[Private Preview] Fully qualified 3-part Unity Catalog path of the function to apply.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "input_bindings": { + "description": "[Private Preview] Binds each UC function parameter to a source column.\nMay be empty for zero-argument functions (e.g. a timestamp generator).", + "$ref": "#/$defs/slice/github.com/databricks/databricks-sdk-go/service/ml.InputBinding", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "function_path" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.DataSource": { + "oneOf": [ + { + "type": "object", + "description": "Specifies the data source backing a feature. Exactly one source type must be set.", + "properties": { + "delta_table_source": { + "description": "[Private Preview] A Delta table data source.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.DeltaTableSource", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "feature_view_source": { + "description": "[Private Preview] A data source composed from registered upstream Features.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FeatureViewSource", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "kafka_source": { + "description": "[Private Preview] A Kafka stream data source.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.KafkaSource", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "lateness": { + "description": "[Private Preview] Completeness timing for this Feature's use of the source. This configuration is part of the\nFeature definition; it does not modify the underlying table or stream.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.SourceLateness", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "request_source": { + "description": "[Private Preview] A request-time data source.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.RequestSource", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "stream_source": { + "description": "[Private Preview] A Stream data source.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.StreamSource", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.DeltaTableSource": { + "oneOf": [ + { + "type": "object", + "properties": { + "dataframe_schema": { + "description": "[Private Preview] Schema of the resulting dataframe after transformations, in Spark StructType JSON format (from df.schema.json()).\nRequired if transformation_sql is specified.\nExample: {\"type\":\"struct\",\"fields\":[{\"name\":\"col_a\",\"type\":\"integer\",\"nullable\":true,\"metadata\":{}},{\"name\":\"col_c\",\"type\":\"integer\",\"nullable\":true,\"metadata\":{}}]}", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "entity_columns": { + "description": "[Private Preview]", + "$ref": "#/$defs/slice/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "filter_condition": { + "description": "[Private Preview] Single WHERE clause to filter delta table before applying transformations. Will be row-wise evaluated, so should only include conditionals and projections.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "full_name": { + "description": "[Private Preview] The full three-part (catalog, schema, table) name of the Delta table.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "timeseries_column": { + "description": "[Private Preview]", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "transformation_sql": { + "description": "[Private Preview] A single SQL SELECT expression applied after filter_condition.\nShould contains all the columns needed (eg. \"SELECT *, col_a + col_b AS col_c FROM x.y.z WHERE col_a \u003e 0\" would have `transformation_sql` \"*, col_a + col_b AS col_c\")\nIf transformation_sql is not provided, all columns of the delta table are present in the DataSource dataframe.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "full_name" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.EntityColumn": { + "oneOf": [ + { + "type": "object", + "properties": { + "name": { + "description": "[Private Preview] The name of the entity column. For Kafka sources, use dot-prefixed path notation to reference\nfields within the key or value schema (e.g., \"value.user_id\", \"key.partition_key\"). For nested\nfields, the leaf node name (e.g., \"user_id\" from \"value.trip_details.user_id\") is what will\nbe present in materialized tables and expected to match at query time.\nColon-prefixed notation (e.g., \"value:user_id\") is supported for backwards\ncompatibility but is deprecated; migrate to dot notation.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "name" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.ExperimentPermissionLevel": { + "oneOf": [ + { + "type": "string", + "description": "Permission level", + "enum": [ + "CAN_MANAGE", + "CAN_EDIT", + "CAN_READ" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.ExperimentTag": { + "oneOf": [ + { + "type": "object", + "description": "A tag for an experiment.", + "properties": { + "key": { + "description": "The tag key.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "GA" + }, + "value": { + "description": "The tag value.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "GA" + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.ExperimentTraceLocation": { + "oneOf": [ + { + "type": "object", + "description": "The storage location for an experiment's traces.", + "properties": { + "uc_trace_location": { + "description": "[Private Preview] A Unity Catalog schema where the experiment's traces are stored as\nDelta tables.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.UcTraceLocation", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FeatureReference": { + "oneOf": [ + { + "type": "object", + "description": "A reference to one registered upstream Feature. A message rather than a bare name so an\nupstream can later be pinned more precisely (e.g. by version) without a breaking type change.", + "properties": { + "feature": { + "description": "[Private Preview] The three-part full name of the upstream Feature.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "feature" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FeatureViewSource": { + "oneOf": [ + { + "type": "object", + "description": "A data source composed from registered upstream Features.", + "properties": { + "feature_references": { + "description": "[Private Preview] The upstream Features this source reads. Must include at least one feature.", + "$ref": "#/$defs/slice/github.com/databricks/databricks-sdk-go/service/ml.FeatureReference", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FieldDefinition": { + "oneOf": [ + { + "type": "object", + "description": "A single field definition within a FlatSchema, specifying the field name and its scalar data type.\nDoes not support nested or complex types (arrays, maps, structs).", + "properties": { + "data_type": { + "description": "[Private Preview] The scalar data type of the field.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.ScalarDataType", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "name": { + "description": "[Private Preview] The name of the field.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "data_type", + "name" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FirstDistinctFunction": { + "oneOf": [ + { + "type": "object", + "description": "Returns the first N distinct values, ordered by the feature's timeseries column.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the first N distinct values are returned.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "n": { + "description": "[Private Preview] The number of distinct values to return.", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input", + "n" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FirstFunction": { + "oneOf": [ + { + "type": "object", + "description": "Returns the first value.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the first value is returned.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FirstNFunction": { + "oneOf": [ + { + "type": "object", + "description": "Returns the first N values, ordered by the feature's timeseries column.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the first N values are returned.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "n": { + "description": "[Private Preview] The number of values to return.", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input", + "n" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FlatSchema": { + "oneOf": [ + { + "type": "object", + "description": "A flat (non-nested) schema for request-time fields, defined as an ordered list of field definitions.\nThis schema only supports scalar types.", + "properties": { + "fields": { + "description": "[Private Preview] The list of fields in this schema.", + "$ref": "#/$defs/slice/github.com/databricks/databricks-sdk-go/service/ml.FieldDefinition", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "fields" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.Function": { + "oneOf": [ + { + "type": "object", + "properties": { + "aggregation_function": { + "description": "[Private Preview] An aggregation function applied over a time window.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.AggregationFunction", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "column_selection": { + "description": "[Private Preview] Selects the latest value of a single column in a data source", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.ColumnSelection", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "custom_udf": { + "description": "[Private Preview] Applies a registered Unity Catalog function row-wise to source columns.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.CustomUdf", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "extra_parameters": { + "description": "[Private Preview]", + "$ref": "#/$defs/slice/github.com/databricks/databricks-sdk-go/service/ml.FunctionExtraParameter", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "function_type": { + "description": "[Private Preview]", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FunctionFunctionType", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FunctionExtraParameter": { + "oneOf": [ + { + "type": "object", + "properties": { + "key": { + "description": "[Private Preview] The name of the parameter.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "value": { + "description": "[Private Preview] The value of the parameter.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "key", + "value" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FunctionFunctionType": { + "oneOf": [ + { + "type": "string", + "enum": [ + "FUNCTION_TYPE_UNSPECIFIED", + "AVG", + "COUNT", + "SUM", + "MIN", + "MAX", + "FIRST", + "LAST", + "APPROX_COUNT_DISTINCT", + "APPROX_PERCENTILE", + "STDDEV_POP", + "STDDEV_SAMP", + "VAR_POP", + "VAR_SAMP" + ], + "enumDescriptions": [ + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.InputBinding": { + "oneOf": [ + { + "type": "object", + "description": "Binds a single UC function parameter to a source column.", + "properties": { + "column": { + "description": "[Private Preview] Source column whose value is passed for this parameter at execution time.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "parameter": { + "description": "[Private Preview] Name of the UC function parameter.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "column", + "parameter" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.JobContext": { + "oneOf": [ + { + "type": "object", + "properties": { + "job_id": { + "description": "[Private Preview] The job ID where this API invoked.", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "job_run_id": { + "description": "[Private Preview] The job run ID where this API was invoked.", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.KafkaSource": { + "oneOf": [ + { + "type": "object", + "properties": { + "entity_column_identifiers": { + "description": "[Private Preview]", + "$ref": "#/$defs/slice/github.com/databricks/databricks-sdk-go/service/ml.ColumnIdentifier", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "filter_condition": { + "description": "[Private Preview] The filter condition applied to the source data before aggregation.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "name": { + "description": "[Private Preview] Name of the Kafka source, used to identify it. This is used to look up the corresponding KafkaConfig object. Can be distinct from topic name.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "timeseries_column_identifier": { + "description": "[Private Preview]", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.ColumnIdentifier", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + } + }, + "additionalProperties": false, + "required": [ + "name" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.LastDistinctFunction": { + "oneOf": [ + { + "type": "object", + "description": "Returns the last N distinct values, ordered by the feature's timeseries column.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the last N distinct values are returned.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "n": { + "description": "[Private Preview] The number of distinct values to return.", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input", + "n" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.LastFunction": { + "oneOf": [ + { + "type": "object", + "description": "Returns the last value.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the last value is returned.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.LastNFunction": { + "oneOf": [ + { + "type": "object", + "description": "Returns the last N values, ordered by the feature's timeseries column.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the last N values are returned.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "n": { + "description": "[Private Preview] The number of values to return.", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input", + "n" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.LineageContext": { + "oneOf": [ + { + "type": "object", + "description": "Lineage context information for tracking where an API was invoked. This will allow us to track lineage, which currently uses caller entity information for use across the Lineage Client and Observability in Lumberjack.", + "properties": { + "job_context": { + "description": "[Private Preview] Job context information including job ID and run ID.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.JobContext", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "notebook_id": { + "description": "[Private Preview] The notebook ID where this API was invoked.", + "$ref": "#/$defs/int64", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.MaxFunction": { + "oneOf": [ + { + "type": "object", + "description": "Computes the maximum value.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the maximum is computed.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.MinFunction": { + "oneOf": [ + { + "type": "object", + "description": "Computes the minimum value.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the minimum is computed.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.ModelTag": { + "oneOf": [ + { + "type": "object", + "description": "Tag for a registered model", + "properties": { + "key": { + "description": "The tag key.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "GA" + }, + "value": { + "description": "The tag value.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "GA" + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.RegisteredModelPermissionLevel": { + "oneOf": [ + { + "type": "string", + "description": "Permission level", + "enum": [ + "CAN_MANAGE", + "CAN_MANAGE_PRODUCTION_VERSIONS", + "CAN_MANAGE_STAGING_VERSIONS", + "CAN_EDIT", + "CAN_READ" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.RequestSource": { + "oneOf": [ + { + "type": "object", + "description": "A request-time data source whose value is provided at inference time: offline batch scoring or online serving endpoint", + "properties": { + "dataframe_schema": { + "description": "[Private Preview] A schema containing scalar or nested fields, in Spark StructType JSON format\n(from df.schema.json()). This preserves field, array-element, and map-value nullability.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "flat_schema": { + "description": "[Private Preview] A flat schema with scalar-typed fields only.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FlatSchema", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.RollingWindow": { + "oneOf": [ + { + "type": "object", + "description": "A rolling time window with an optional non-negative delay.", + "properties": { + "delay": { + "description": "[Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing\nvariations unrelated to source lateness, such as a 30-day count as of one week ago. If unset,\nthe analytic lag is zero. It composes with source.lateness when both are set.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "window_duration": { + "description": "[Private Preview] The duration of the rolling window. Must be positive when set; absent means lifetime\n(aggregate over the entity's entire history).", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.SawtoothWindow": { + "oneOf": [ + { + "type": "object", + "description": "A sawtooth window served via the hybrid batch + streaming path. The batch pipeline maintains\ndaily partial aggregates for the bulk of the window while the streaming pipeline maintains the\nmost recent day(s), and serving merges them on read. Same field shape as RollingWindow, but a\ndistinct type so the control plane can explicitly identify hybrid (sawtooth) features rather\nthan inferring hybrid behavior from window_duration.", + "properties": { + "delay": { + "description": "[Private Preview] Delay is not currently supported for Sawtooth windows.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "window_duration": { + "description": "[Private Preview] The duration of the window. Must be positive and span more than two days when set, so that both\nthe batch (N-1 day) and stale-path (N-2 day) partial aggregates are well defined. The duration\nneed not be a whole number of days (e.g. 3 days 15 minutes is allowed). Absent means lifetime\n(aggregate over the entity's entire history).", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.ScalarDataType": { + "oneOf": [ + { + "type": "string", + "description": "Scalar data types for request-time field definitions.\nOnly flat (non-nested) types are supported.", + "enum": [ + "INTEGER", + "FLOAT", + "BOOLEAN", + "STRING", + "DOUBLE", + "LONG", + "TIMESTAMP", + "DATE", + "SHORT", + "BINARY", + "DECIMAL" + ], + "enumDescriptions": [ + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]", + "[Private Preview]" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.SlidingWindow": { + "oneOf": [ + { + "type": "object", + "properties": { + "delay": { + "description": "[Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing\nvariations unrelated to source lateness, such as a 30-day count as of one week ago. If unset,\nthe analytic lag is zero. It composes with source.lateness when both are set.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "offset": { + "description": "[Private Preview] Non-negative phase shift from the default midnight UTC alignment. For example, offset=22h on\na 24h slide produces boundaries at 22:00 UTC (17:00 New York in standard time) instead of\nmidnight UTC. If unset, the offset is zero. Must be shorter than slide_duration (and therefore\nwindow_duration).", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "slide_duration": { + "description": "[Private Preview] The slide duration (interval by which windows advance, must be positive and less than duration).", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "window_duration": { + "description": "[Private Preview] The duration of the sliding window. Must be positive when set; absent means lifetime\n(aggregate over the entity's entire history).", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "slide_duration" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.SourceLateness": { + "oneOf": [ + { + "type": "object", + "description": "Configures when event-time data from this source is considered complete for a Feature.", + "properties": { + "settling_delay": { + "description": "[Private Preview] Non-negative time to wait after a window ends before treating its source data as complete.\nTraining shifts the eligible evaluation time backwards by this duration so it does not join\ndata that would still have been settling online. Materialization waits for the duration to\nelapse before publishing the window. If unset, source data is considered settled immediately.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.StddevPopFunction": { + "oneOf": [ + { + "type": "object", + "description": "Computes the population standard deviation.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the population standard deviation is computed. For Kafka sources,\nuse dot-prefixed path notation (e.g., \"value.amount\"). For nested fields, the leaf node name is used.\nColon-prefixed notation (e.g., \"value:amount\") is supported for backwards\ncompatibility but is deprecated; migrate to dot notation.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.StddevSampFunction": { + "oneOf": [ + { + "type": "object", + "description": "Computes the sample standard deviation.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the sample standard deviation is computed.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.StreamSource": { + "oneOf": [ + { + "type": "object", + "description": "A Stream entity used as a data source for a feature.", + "properties": { + "dataframe_schema": { + "description": "[Private Preview] Schema of the resulting dataframe after transformations, in Spark StructType\nJSON format (from df.schema.json()).\nAny subsequent functions operate against this dataframe.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "filter_condition": { + "description": "[Private Preview] The filter condition applied to the source data before aggregation.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "full_name": { + "description": "[Private Preview] Three-part full name of the Stream (catalog.schema.stream).", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "transformation_sql": { + "description": "[Private Preview] The pipeline runs these SQL statements immediately after conversion into\nthe schema specified on the Stream object.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "full_name" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.SumFunction": { + "oneOf": [ + { + "type": "object", + "description": "Computes the sum of values.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the sum is computed. For Kafka sources, use dot-prefixed path\nnotation (e.g., \"value.amount\"). For nested fields, the leaf node name is used.\nColon-prefixed notation (e.g., \"value:amount\") is supported for backwards\ncompatibility but is deprecated; migrate to dot notation.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.TimeWindow": { + "oneOf": [ + { + "type": "object", + "properties": { + "continuous": { + "description": "[Private Preview]", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.ContinuousWindow", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "deprecationMessage": "This field is deprecated", + "doNotSuggest": true, + "deprecated": true + }, + "rolling": { + "description": "[Private Preview] A rolling time window with an optional non-negative delay.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.RollingWindow", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "sawtooth": { + "description": "[Private Preview] A sawtooth window served via the hybrid batch + streaming path.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.SawtoothWindow", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "sliding": { + "description": "[Private Preview]", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.SlidingWindow", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "start_time": { + "description": "[Private Preview] Earliest event-time boundary at which the Feature may emit an output. This gates outputs, not\nthe historical inputs read by a window. For example, a 365-day window with\nstart_time=2026-01-01 begins emitting partial-window values on that date instead of waiting\nfor 365 days of data; a lifetime window produces no output before start_time. If unset,\ntumbling and fixed-duration sliding windows first emit at an offset-aligned boundary after a\nfull window can be formed. If unset, lifetime sliding windows and rolling windows emit as soon as\neligible source data exists.\nNot currently supported for sawtooth windows or for Features with a stream source.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/time.Time", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "tumbling": { + "description": "[Private Preview]", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.TumblingWindow", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.TimeseriesColumn": { + "oneOf": [ + { + "type": "object", + "properties": { + "name": { + "description": "[Private Preview] The name of the timeseries column. For Kafka sources, use dot-prefixed path notation to\nreference fields within the key or value schema (e.g., \"value.event_timestamp\"). For nested\nfields, the leaf node name (e.g., \"event_timestamp\" from \"value.event_details.event_timestamp\")\nis what will be present in materialized tables and expected to match at query time.\nColon-prefixed notation (e.g., \"value:event_timestamp\") is supported for\nbackwards compatibility but is deprecated; migrate to dot notation.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "name" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.TumblingWindow": { + "oneOf": [ + { + "type": "object", + "properties": { + "delay": { + "description": "[Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing\nvariations unrelated to source lateness, such as a 30-day count as of one week ago. If unset,\nthe analytic lag is zero. It composes with source.lateness when both are set.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "offset": { + "description": "[Private Preview] Non-negative phase shift from the default midnight UTC alignment. For example, offset=22h on\na 24h window produces boundaries at 22:00 UTC (17:00 New York in standard time) instead of\nmidnight UTC. If unset, the offset is zero. Must be shorter than window_duration.", + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/common/types/duration.Duration", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "window_duration": { + "description": "[Private Preview] The duration of each tumbling window (non-overlapping, fixed-duration windows).", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "window_duration" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.UcTraceLocation": { + "oneOf": [ + { + "type": "object", + "description": "A Unity Catalog trace storage location. Traces are stored as Delta tables\nin the specified catalog and schema.", + "properties": { + "catalog": { + "description": "[Private Preview] The name of the Unity Catalog catalog.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "schema": { + "description": "[Private Preview] The name of the Unity Catalog schema within `catalog`.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + }, + "table_prefix": { + "description": "[Private Preview] The prefix for the trace tables, which are named\n`{catalog}.{schema}.{table_prefix}_otel_*`. May only contain letters,\ndigits, and underscores, and may be at most 238 characters. When unset, a\nserver-generated prefix derived from the experiment ID is used and this\nfield stays empty on read; the resolved value is always available in\n`effective_table_prefix`.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "catalog", + "schema" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.VarPopFunction": { + "oneOf": [ + { + "type": "object", + "description": "Computes the population variance.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the population variance is computed.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.VarSampFunction": { + "oneOf": [ + { + "type": "object", + "description": "Computes the sample variance.", + "properties": { + "input": { + "description": "[Private Preview] The input column from which the sample variance is computed.", + "$ref": "#/$defs/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false, + "required": [ + "input" + ] + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "pipelines.ApiSourceConnectorConfig": { + "oneOf": [ + { + "type": "object", + "description": "Top-level configuration for API Source connectors with arbitrary configuration.", + "properties": { + "configs": { + "description": "[Private Preview] Arbitrary key-value configuration values for the API Source connector.", + "$ref": "#/$defs/map/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "pipelines.ApiSourceConnectorOptions": { + "oneOf": [ + { + "type": "object", + "description": "Options for API Source connectors with arbitrary configuration.", + "properties": { + "options": { + "description": "[Private Preview] Arbitrary key-value configuration options for the API Source connector.", + "$ref": "#/$defs/map/string", + "x-databricks-launch-stage": "PRIVATE_PREVIEW", + "doNotSuggest": true + } + }, + "additionalProperties": false + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "pipelines.AutoFullRefreshPolicy": { + "oneOf": [ + { + "type": "object", + "description": "Policy for auto full refresh.", + "properties": { + "enabled": { + "description": "[Public Preview] (Required, Mutable) Whether to enable auto full refresh or not.", + "$ref": "#/$defs/bool", + "x-databricks-launch-stage": "PUBLIC_PREVIEW" + }, "min_interval_hours": { "description": "[Public Preview] (Optional, Mutable) Specify the minimum interval in hours between the timestamp\nat which a table was last full refreshed and the current timestamp for triggering auto full\nIf unspecified and autoFullRefresh is enabled then by default min_interval_hours is 24 hours.", "$ref": "#/$defs/int", @@ -18151,6 +19742,20 @@ } ] }, + "resources.Feature": { + "oneOf": [ + { + "type": "object", + "additionalProperties": { + "$ref": "#/$defs/github.com/databricks/cli/bundle/config/resources.Feature" + } + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, "resources.GenieSpace": { "oneOf": [ { @@ -19330,6 +20935,34 @@ } ] }, + "ml.ColumnIdentifier": { + "oneOf": [ + { + "type": "array", + "items": { + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.ColumnIdentifier" + } + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.EntityColumn": { + "oneOf": [ + { + "type": "array", + "items": { + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.EntityColumn" + } + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, "ml.ExperimentTag": { "oneOf": [ { @@ -19344,6 +20977,62 @@ } ] }, + "ml.FeatureReference": { + "oneOf": [ + { + "type": "array", + "items": { + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FeatureReference" + } + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FieldDefinition": { + "oneOf": [ + { + "type": "array", + "items": { + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FieldDefinition" + } + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.FunctionExtraParameter": { + "oneOf": [ + { + "type": "array", + "items": { + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.FunctionExtraParameter" + } + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, + "ml.InputBinding": { + "oneOf": [ + { + "type": "array", + "items": { + "$ref": "#/$defs/github.com/databricks/databricks-sdk-go/service/ml.InputBinding" + } + }, + { + "type": "string", + "pattern": "\\$\\{(var(\\._*\\p{L}+([-_]*[\\p{L}\\p{N}]+)*(\\[[0-9]+\\])*)+)\\}" + } + ] + }, "ml.ModelTag": { "oneOf": [ { diff --git a/bundle/statemgmt/state_load_test.go b/bundle/statemgmt/state_load_test.go index b03ff469284..4ec77578d2b 100644 --- a/bundle/statemgmt/state_load_test.go +++ b/bundle/statemgmt/state_load_test.go @@ -31,6 +31,7 @@ func TestStateToBundleEmptyLocalResources(t *testing.T) { "resources.pipelines.test_pipeline": {ID: "1"}, "resources.models.test_mlflow_model": {ID: "1"}, "resources.experiments.test_mlflow_experiment": {ID: "1"}, + "resources.features.test_feature": {ID: "main.default.test_feature"}, "resources.model_serving_endpoints.test_model_serving": {ID: "1"}, "resources.model_services.test_model_service": {ID: "main.default.test_model_service"}, "resources.mcp_services.test_mcp_service": {ID: "main.default.test_mcp_service"}, @@ -217,6 +218,13 @@ func TestStateToBundleEmptyRemoteResources(t *testing.T) { }, }, }, + Features: map[string]*resources.Feature{ + "test_feature": { + Feature: ml.Feature{ + FullName: "main.default.test_feature", + }, + }, + }, ModelServingEndpoints: map[string]*resources.ModelServingEndpoint{ "test_model_serving": { CreateServingEndpoint: serving.CreateServingEndpoint{ @@ -644,6 +652,18 @@ func TestStateToBundleModifiedResources(t *testing.T) { }, }, }, + Features: map[string]*resources.Feature{ + "test_feature": { + Feature: ml.Feature{ + FullName: "main.default.test_feature", + }, + }, + "test_feature_new": { + Feature: ml.Feature{ + FullName: "main.default.test_feature_new", + }, + }, + }, ModelServingEndpoints: map[string]*resources.ModelServingEndpoint{ "test_model_serving": { CreateServingEndpoint: serving.CreateServingEndpoint{ @@ -1059,6 +1079,8 @@ func TestStateToBundleModifiedResources(t *testing.T) { "resources.models.test_mlflow_model_old": {ID: "2"}, "resources.experiments.test_mlflow_experiment": {ID: "1"}, "resources.experiments.test_mlflow_experiment_old": {ID: "2"}, + "resources.features.test_feature": {ID: "main.default.test_feature"}, + "resources.features.test_feature_old": {ID: "main.default.test_feature_old"}, "resources.model_serving_endpoints.test_model_serving": {ID: "1"}, "resources.model_serving_endpoints.test_model_serving_old": {ID: "2"}, "resources.registered_models.test_registered_model": {ID: "1"}, @@ -1161,6 +1183,13 @@ func TestStateToBundleModifiedResources(t *testing.T) { assert.Empty(t, config.Resources.Experiments["test_mlflow_experiment_new"].ID) assert.Equal(t, resources.ModifiedStatusCreated, config.Resources.Experiments["test_mlflow_experiment_new"].ModifiedStatus) + assert.Equal(t, "main.default.test_feature", config.Resources.Features["test_feature"].ID) + assert.Empty(t, config.Resources.Features["test_feature"].ModifiedStatus) + assert.Equal(t, "main.default.test_feature_old", config.Resources.Features["test_feature_old"].ID) + assert.Equal(t, resources.ModifiedStatusDeleted, config.Resources.Features["test_feature_old"].ModifiedStatus) + assert.Empty(t, config.Resources.Features["test_feature_new"].ID) + assert.Equal(t, resources.ModifiedStatusCreated, config.Resources.Features["test_feature_new"].ModifiedStatus) + assert.Equal(t, "1", config.Resources.ModelServingEndpoints["test_model_serving"].ID) assert.Empty(t, config.Resources.ModelServingEndpoints["test_model_serving"].ModifiedStatus) assert.Equal(t, "2", config.Resources.ModelServingEndpoints["test_model_serving_old"].ID) diff --git a/cmd/experimental/workspace_open_test.go b/cmd/experimental/workspace_open_test.go index 24f9d49c79b..91aa6d07f63 100644 --- a/cmd/experimental/workspace_open_test.go +++ b/cmd/experimental/workspace_open_test.go @@ -34,6 +34,7 @@ func TestBuildWorkspaceURLPathBasedResources(t *testing.T) { {"apps", "my-app", "https://myworkspace.databricks.com/apps/my-app"}, {"clusters", "0123-456789-abc", "https://myworkspace.databricks.com/compute/clusters/0123-456789-abc"}, {"registered_models", "catalog.schema.model", "https://myworkspace.databricks.com/explore/data/models/catalog/schema/model"}, + {"features", "catalog.schema.feature", "https://myworkspace.databricks.com/explore/data/features/catalog/schema/feature"}, } for _, tt := range tests { @@ -67,7 +68,7 @@ func TestBuildWorkspaceURLFragmentBasedResources(t *testing.T) { func TestBuildWorkspaceURLUnknownResourceType(t *testing.T) { _, err := workspaceurls.BuildResourceURL("https://myworkspace.databricks.com", "unknown", "123", "") assert.ErrorContains(t, err, "unknown resource type \"unknown\"") - assert.ErrorContains(t, err, "alerts, apps, catalogs, cluster_policies, clusters, dashboards, database_catalogs, database_instances, experiments, genie_spaces, instance_pools, jobs, mcp_services, model_provider_services, model_services, model_serving_endpoints, models, notebooks, pipelines, postgres_catalogs, postgres_synced_tables, quality_monitors, queries, registered_models, schemas, secrets, synced_database_tables, vector_search_endpoints, vector_search_indexes, volumes, warehouses") + assert.ErrorContains(t, err, "alerts, apps, catalogs, cluster_policies, clusters, dashboards, database_catalogs, database_instances, experiments, features, genie_spaces, instance_pools, jobs, mcp_services, model_provider_services, model_services, model_serving_endpoints, models, notebooks, pipelines, postgres_catalogs, postgres_synced_tables, quality_monitors, queries, registered_models, schemas, secrets, synced_database_tables, vector_search_endpoints, vector_search_indexes, volumes, warehouses") } func TestBuildWorkspaceURLHostWithTrailingSlash(t *testing.T) { @@ -116,6 +117,7 @@ func TestWorkspaceOpenCommandCompletion(t *testing.T) { "database_catalogs", "database_instances", "experiments", + "features", "genie_spaces", "instance_pools", "jobs", @@ -152,7 +154,7 @@ func TestWorkspaceOpenCommandCompletionSecondArg(t *testing.T) { func TestWorkspaceOpenCommandHelpText(t *testing.T) { cmd := newWorkspaceOpenCommand() - assert.Contains(t, cmd.Long, "Supported resource types: alerts, apps, catalogs, cluster_policies, clusters, dashboards, database_catalogs, database_instances, experiments, genie_spaces, instance_pools, jobs, mcp_services, model_provider_services, model_services, model_serving_endpoints, models, notebooks, pipelines, postgres_catalogs, postgres_synced_tables, quality_monitors, queries, registered_models, schemas, secrets, synced_database_tables, vector_search_endpoints, vector_search_indexes, volumes, warehouses.") + assert.Contains(t, cmd.Long, "Supported resource types: alerts, apps, catalogs, cluster_policies, clusters, dashboards, database_catalogs, database_instances, experiments, features, genie_spaces, instance_pools, jobs, mcp_services, model_provider_services, model_services, model_serving_endpoints, models, notebooks, pipelines, postgres_catalogs, postgres_synced_tables, quality_monitors, queries, registered_models, schemas, secrets, synced_database_tables, vector_search_endpoints, vector_search_indexes, volumes, warehouses.") assert.Contains(t, cmd.Long, "databricks experimental open jobs 123456789") assert.Contains(t, cmd.Long, "databricks experimental open notebooks /Users/user@example.com/my-notebook") assert.Contains(t, cmd.Long, "databricks experimental open registered_models catalog.schema.my_model") diff --git a/libs/testserver/fake_workspace.go b/libs/testserver/fake_workspace.go index b73803751b6..4edabd52660 100644 --- a/libs/testserver/fake_workspace.go +++ b/libs/testserver/fake_workspace.go @@ -217,6 +217,8 @@ type FakeWorkspace struct { ModelServices map[string]catalog.ModelService McpServices map[string]catalog.McpService ModelProviderServices map[string]catalog.ModelProviderService + + Features map[string]ml.Feature ServingEndpoints map[string]serving.ServingEndpointDetailed VectorSearchEndpoints map[string]vectorsearch.EndpointInfo VectorSearchIndexes map[string]fakeVectorSearchIndex @@ -497,6 +499,7 @@ func NewFakeWorkspace(url, token string) *FakeWorkspace { ModelServices: map[string]catalog.ModelService{}, McpServices: map[string]catalog.McpService{}, ModelProviderServices: map[string]catalog.ModelProviderService{}, + Features: map[string]ml.Feature{}, Volumes: map[string]catalog.VolumeInfo{}, Dashboards: NewEventualMap[string, *fakeDashboard](eventualConsistency), PublishedDashboards: map[string]dashboards.PublishedDashboard{}, diff --git a/libs/testserver/feature.go b/libs/testserver/feature.go new file mode 100644 index 00000000000..fe3289e38d6 --- /dev/null +++ b/libs/testserver/feature.go @@ -0,0 +1,69 @@ +package testserver + +import ( + "encoding/json" + "fmt" + "net/http" + "strings" + + "github.com/databricks/databricks-sdk-go/service/ml" +) + +// FeaturesCreate fakes POST /api/2.0/feature-engineering/features. The Feature body is sent +// directly and the map is keyed by full_name, which is also the resource id. catalog_name, +// schema_name and name are derived here because the backend derives them from full_name. +func (s *FakeWorkspace) FeaturesCreate(req Request) Response { + defer s.LockUnlock()() + + var feature ml.Feature + if err := json.Unmarshal(req.Body, &feature); err != nil { + return Response{ + Body: fmt.Sprintf("internal error: %s", err), + StatusCode: http.StatusInternalServerError, + } + } + + parts := strings.Split(feature.FullName, ".") + if len(parts) != 3 { + return Response{ + StatusCode: http.StatusBadRequest, + Body: fmt.Sprintf("full_name must be a three-part name, got %q", feature.FullName), + } + } + feature.CatalogName, feature.SchemaName, feature.Name = parts[0], parts[1], parts[2] + feature.CreatedBy = s.CurrentUser().UserName + + s.Features[feature.FullName] = feature + return Response{ + Body: feature, + } +} + +// FeaturesUpdate fakes PATCH /api/2.0/feature-engineering/features/{full_name}. Only description +// is applied: it is the sole field the real update endpoint accepts. +func (s *FakeWorkspace) FeaturesUpdate(req Request, fullName string) Response { + defer s.LockUnlock()() + + existing, ok := s.Features[fullName] + if !ok { + return Response{ + StatusCode: http.StatusNotFound, + Body: fmt.Sprintf("feature %s not found", fullName), + } + } + + var incoming ml.Feature + if err := json.Unmarshal(req.Body, &incoming); err != nil { + return Response{ + Body: fmt.Sprintf("internal error: %s", err), + StatusCode: http.StatusInternalServerError, + } + } + + existing.Description = incoming.Description + + s.Features[fullName] = existing + return Response{ + Body: existing, + } +} diff --git a/libs/testserver/handlers.go b/libs/testserver/handlers.go index a8217e3bb20..64ca0b102ee 100644 --- a/libs/testserver/handlers.go +++ b/libs/testserver/handlers.go @@ -675,6 +675,20 @@ func AddDefaultHandlers(server *Server) { return MapDelete(req.Workspace, req.Workspace.McpServices, req.Vars["name"]) }) + // Feature Engineering features: + server.Handle("POST", "/api/2.0/feature-engineering/features", func(req Request) any { + return req.Workspace.FeaturesCreate(req) + }) + server.Handle("GET", "/api/2.0/feature-engineering/features/{full_name}", func(req Request) any { + return MapGet(req.Workspace, req.Workspace.Features, req.Vars["full_name"]) + }) + server.Handle("PATCH", "/api/2.0/feature-engineering/features/{full_name}", func(req Request) any { + return req.Workspace.FeaturesUpdate(req, req.Vars["full_name"]) + }) + server.Handle("DELETE", "/api/2.0/feature-engineering/features/{full_name}", func(req Request) any { + return MapDelete(req.Workspace, req.Workspace.Features, req.Vars["full_name"]) + }) + // Model Provider Services (AI Gateway): server.Handle("POST", "/api/2.1/unity-catalog/model-provider-services", func(req Request) any { diff --git a/libs/workspaceurls/urls.go b/libs/workspaceurls/urls.go index 534bdc0fc83..0c31a7a0071 100644 --- a/libs/workspaceurls/urls.go +++ b/libs/workspaceurls/urls.go @@ -18,6 +18,7 @@ var resourceURLPatterns = map[string]string{ "database_catalogs": "explore/data/%s", "database_instances": "compute/database-instances/%s", "experiments": "ml/experiments/%s", + "features": "explore/data/features/%s", "genie_spaces": "genie/rooms/%s", "jobs": "jobs/%s", "mcp_services": "explore/data/mcp-services/%s", @@ -56,6 +57,7 @@ var resourceAliases = map[string]string{ // requires slash-separated segments. var dotSeparatedResources = map[string]bool{ "catalogs": true, + "features": true, "mcp_services": true, "model_services": true, "model_provider_services": true, diff --git a/python/databricks/bundles/core/__init__.py b/python/databricks/bundles/core/__init__.py index 6fb4f7acdd2..b40cb9f9684 100644 --- a/python/databricks/bundles/core/__init__.py +++ b/python/databricks/bundles/core/__init__.py @@ -23,6 +23,7 @@ "database_catalog_mutator", "database_instance_mutator", "external_location_mutator", + "feature_mutator", "genie_space_mutator", "instance_pool_mutator", "job_mutator", @@ -67,6 +68,7 @@ database_catalog_mutator, database_instance_mutator, external_location_mutator, + feature_mutator, genie_space_mutator, instance_pool_mutator, job_mutator, diff --git a/python/databricks/bundles/core/_generated/__init__.py b/python/databricks/bundles/core/_generated/__init__.py index eb7f3ce6494..6a56d2eb856 100644 --- a/python/databricks/bundles/core/_generated/__init__.py +++ b/python/databricks/bundles/core/_generated/__init__.py @@ -36,6 +36,10 @@ _ExternalLocationResources, external_location_mutator, ) +from databricks.bundles.core._generated.features import ( + _FeatureResources, + feature_mutator, +) from databricks.bundles.core._generated.genie_spaces import ( _GenieSpaceResources, genie_space_mutator, @@ -120,6 +124,7 @@ "database_catalog_mutator", "database_instance_mutator", "external_location_mutator", + "feature_mutator", "genie_space_mutator", "instance_pool_mutator", "job_mutator", @@ -155,6 +160,7 @@ class _GeneratedResources( _DatabaseInstanceResources, _MlflowExperimentResources, _ExternalLocationResources, + _FeatureResources, _GenieSpaceResources, _InstancePoolResources, _JobRunResources, @@ -191,6 +197,7 @@ def _all_resource_types() -> "tuple[_ResourceType, ...]": database_instances, experiments, external_locations, + features, genie_spaces, instance_pools, job_runs, @@ -224,6 +231,7 @@ def _all_resource_types() -> "tuple[_ResourceType, ...]": database_instances._resource_type(), experiments._resource_type(), external_locations._resource_type(), + features._resource_type(), genie_spaces._resource_type(), instance_pools._resource_type(), job_runs._resource_type(), diff --git a/python/databricks/bundles/core/_generated/features.py b/python/databricks/bundles/core/_generated/features.py new file mode 100644 index 00000000000..e9c4d96d179 --- /dev/null +++ b/python/databricks/bundles/core/_generated/features.py @@ -0,0 +1,115 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from collections.abc import Callable +from typing import TYPE_CHECKING, Optional, overload + +from databricks.bundles.core._bundle import Bundle +from databricks.bundles.core._location import Location +from databricks.bundles.core._resource_mutator import ResourceMutator +from databricks.bundles.core._transform import _transform + +if TYPE_CHECKING: + from databricks.bundles.core._resource_type import _ResourceType + from databricks.bundles.features._models.feature import Feature, FeatureParam + + +def _resource_type() -> "_ResourceType": + from databricks.bundles.core._resource_type import _ResourceType + from databricks.bundles.features._models.feature import Feature + + return _ResourceType( + resource_type=Feature, + singular_name="feature", + plural_name="features", + ) + + +class _FeatureResources: + """ + Generated feature accessors, mixed into Resources. + """ + + # Provided by the Resources subclass; declared here so the generated methods + # below type-check. + _resources: dict[str, dict] + + if TYPE_CHECKING: + + def add_location(self, path: tuple[str, ...], location: Location) -> None: ... + + def add_diagnostic_error( + self, + msg: str, + *, + detail: Optional[str] = None, + path: Optional[tuple[str, ...]] = None, + location: Optional[Location] = None, + ) -> None: ... + + @property + def features(self) -> dict[str, "Feature"]: + return self._resources["features"] + + def add_feature( + self, + resource_name: str, + feature: "FeatureParam", + *, + location: Optional[Location] = None, + ) -> None: + """ + Adds the resource feature to the collection of resources. Resource name must be unique across all features. + + :param resource_name: unique identifier for the feature + :param feature: the feature to add, can be Feature or dict + :param location: optional location of the feature in the source code + """ + from databricks.bundles.features._models.feature import Feature + + feature = _transform(Feature, feature) + path = ("resources", "features", resource_name) + location = location or Location.from_stack_frame(depth=1) + + if self._resources["features"].get(resource_name): + self.add_diagnostic_error( + msg=f"Duplicate resource name '{resource_name}' for resource 'feature'. Resource names must be unique.", + location=location, + path=path, + ) + else: + if location: + self.add_location(path, location) + + self._resources["features"][resource_name] = feature + + +@overload +def feature_mutator( + function: Callable[[Bundle, "Feature"], "Feature"], +) -> ResourceMutator["Feature"]: ... + + +@overload +def feature_mutator( + function: Callable[["Feature"], "Feature"], +) -> ResourceMutator["Feature"]: ... + + +def feature_mutator(function: Callable) -> ResourceMutator["Feature"]: + """ + Decorator for defining mutator for features. Function should return a new instance of the feature + with the desired changes, instead of mutating the input feature. + + Example: + + .. code-block:: python + + @feature_mutator + def my_feature_mutator(bundle: Bundle, feature: Feature) -> Feature: + return replace(feature, ...) + + :param function: Function that mutates features. + """ + from databricks.bundles.features._models.feature import Feature + + return ResourceMutator(resource_type=Feature, function=function) diff --git a/python/databricks/bundles/features/__init__.py b/python/databricks/bundles/features/__init__.py new file mode 100644 index 00000000000..24e3f143d65 --- /dev/null +++ b/python/databricks/bundles/features/__init__.py @@ -0,0 +1,356 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +__all__ = [ + "AggregationFunction", + "AggregationFunctionDict", + "AggregationFunctionParam", + "ApproxCountDistinctFunction", + "ApproxCountDistinctFunctionDict", + "ApproxCountDistinctFunctionParam", + "ApproxPercentileFunction", + "ApproxPercentileFunctionDict", + "ApproxPercentileFunctionParam", + "AvgFunction", + "AvgFunctionDict", + "AvgFunctionParam", + "ColumnSelection", + "ColumnSelectionDict", + "ColumnSelectionParam", + "CountFunction", + "CountFunctionDict", + "CountFunctionParam", + "CustomUdf", + "CustomUdfDict", + "CustomUdfParam", + "DataSource", + "DataSourceDict", + "DataSourceParam", + "DeltaTableSource", + "DeltaTableSourceDict", + "DeltaTableSourceParam", + "EntityColumn", + "EntityColumnDict", + "EntityColumnParam", + "Feature", + "FeatureDict", + "FeatureParam", + "FeatureReference", + "FeatureReferenceDict", + "FeatureReferenceParam", + "FeatureViewSource", + "FeatureViewSourceDict", + "FeatureViewSourceParam", + "FieldDefinition", + "FieldDefinitionDict", + "FieldDefinitionParam", + "FirstDistinctFunction", + "FirstDistinctFunctionDict", + "FirstDistinctFunctionParam", + "FirstFunction", + "FirstFunctionDict", + "FirstFunctionParam", + "FirstNFunction", + "FirstNFunctionDict", + "FirstNFunctionParam", + "FlatSchema", + "FlatSchemaDict", + "FlatSchemaParam", + "Function", + "FunctionDict", + "FunctionParam", + "InputBinding", + "InputBindingDict", + "InputBindingParam", + "JobContext", + "JobContextDict", + "JobContextParam", + "KafkaSource", + "KafkaSourceDict", + "KafkaSourceParam", + "LastDistinctFunction", + "LastDistinctFunctionDict", + "LastDistinctFunctionParam", + "LastFunction", + "LastFunctionDict", + "LastFunctionParam", + "LastNFunction", + "LastNFunctionDict", + "LastNFunctionParam", + "Lifecycle", + "LifecycleDict", + "LifecycleParam", + "LineageContext", + "LineageContextDict", + "LineageContextParam", + "MaxFunction", + "MaxFunctionDict", + "MaxFunctionParam", + "MinFunction", + "MinFunctionDict", + "MinFunctionParam", + "RequestSource", + "RequestSourceDict", + "RequestSourceParam", + "RollingWindow", + "RollingWindowDict", + "RollingWindowParam", + "SawtoothWindow", + "SawtoothWindowDict", + "SawtoothWindowParam", + "ScalarDataType", + "ScalarDataTypeParam", + "SlidingWindow", + "SlidingWindowDict", + "SlidingWindowParam", + "SourceLateness", + "SourceLatenessDict", + "SourceLatenessParam", + "StddevPopFunction", + "StddevPopFunctionDict", + "StddevPopFunctionParam", + "StddevSampFunction", + "StddevSampFunctionDict", + "StddevSampFunctionParam", + "StreamSource", + "StreamSourceDict", + "StreamSourceParam", + "SumFunction", + "SumFunctionDict", + "SumFunctionParam", + "TimeWindow", + "TimeWindowDict", + "TimeWindowParam", + "TimeseriesColumn", + "TimeseriesColumnDict", + "TimeseriesColumnParam", + "TumblingWindow", + "TumblingWindowDict", + "TumblingWindowParam", + "VarPopFunction", + "VarPopFunctionDict", + "VarPopFunctionParam", + "VarSampFunction", + "VarSampFunctionDict", + "VarSampFunctionParam", +] + + +from databricks.bundles.features._models.aggregation_function import ( + AggregationFunction, + AggregationFunctionDict, + AggregationFunctionParam, +) +from databricks.bundles.features._models.approx_count_distinct_function import ( + ApproxCountDistinctFunction, + ApproxCountDistinctFunctionDict, + ApproxCountDistinctFunctionParam, +) +from databricks.bundles.features._models.approx_percentile_function import ( + ApproxPercentileFunction, + ApproxPercentileFunctionDict, + ApproxPercentileFunctionParam, +) +from databricks.bundles.features._models.avg_function import ( + AvgFunction, + AvgFunctionDict, + AvgFunctionParam, +) +from databricks.bundles.features._models.column_selection import ( + ColumnSelection, + ColumnSelectionDict, + ColumnSelectionParam, +) +from databricks.bundles.features._models.count_function import ( + CountFunction, + CountFunctionDict, + CountFunctionParam, +) +from databricks.bundles.features._models.custom_udf import ( + CustomUdf, + CustomUdfDict, + CustomUdfParam, +) +from databricks.bundles.features._models.data_source import ( + DataSource, + DataSourceDict, + DataSourceParam, +) +from databricks.bundles.features._models.delta_table_source import ( + DeltaTableSource, + DeltaTableSourceDict, + DeltaTableSourceParam, +) +from databricks.bundles.features._models.entity_column import ( + EntityColumn, + EntityColumnDict, + EntityColumnParam, +) +from databricks.bundles.features._models.feature import ( + Feature, + FeatureDict, + FeatureParam, +) +from databricks.bundles.features._models.feature_reference import ( + FeatureReference, + FeatureReferenceDict, + FeatureReferenceParam, +) +from databricks.bundles.features._models.feature_view_source import ( + FeatureViewSource, + FeatureViewSourceDict, + FeatureViewSourceParam, +) +from databricks.bundles.features._models.field_definition import ( + FieldDefinition, + FieldDefinitionDict, + FieldDefinitionParam, +) +from databricks.bundles.features._models.first_distinct_function import ( + FirstDistinctFunction, + FirstDistinctFunctionDict, + FirstDistinctFunctionParam, +) +from databricks.bundles.features._models.first_function import ( + FirstFunction, + FirstFunctionDict, + FirstFunctionParam, +) +from databricks.bundles.features._models.first_n_function import ( + FirstNFunction, + FirstNFunctionDict, + FirstNFunctionParam, +) +from databricks.bundles.features._models.flat_schema import ( + FlatSchema, + FlatSchemaDict, + FlatSchemaParam, +) +from databricks.bundles.features._models.function import ( + Function, + FunctionDict, + FunctionParam, +) +from databricks.bundles.features._models.input_binding import ( + InputBinding, + InputBindingDict, + InputBindingParam, +) +from databricks.bundles.features._models.job_context import ( + JobContext, + JobContextDict, + JobContextParam, +) +from databricks.bundles.features._models.kafka_source import ( + KafkaSource, + KafkaSourceDict, + KafkaSourceParam, +) +from databricks.bundles.features._models.last_distinct_function import ( + LastDistinctFunction, + LastDistinctFunctionDict, + LastDistinctFunctionParam, +) +from databricks.bundles.features._models.last_function import ( + LastFunction, + LastFunctionDict, + LastFunctionParam, +) +from databricks.bundles.features._models.last_n_function import ( + LastNFunction, + LastNFunctionDict, + LastNFunctionParam, +) +from databricks.bundles.features._models.lifecycle import ( + Lifecycle, + LifecycleDict, + LifecycleParam, +) +from databricks.bundles.features._models.lineage_context import ( + LineageContext, + LineageContextDict, + LineageContextParam, +) +from databricks.bundles.features._models.max_function import ( + MaxFunction, + MaxFunctionDict, + MaxFunctionParam, +) +from databricks.bundles.features._models.min_function import ( + MinFunction, + MinFunctionDict, + MinFunctionParam, +) +from databricks.bundles.features._models.request_source import ( + RequestSource, + RequestSourceDict, + RequestSourceParam, +) +from databricks.bundles.features._models.rolling_window import ( + RollingWindow, + RollingWindowDict, + RollingWindowParam, +) +from databricks.bundles.features._models.sawtooth_window import ( + SawtoothWindow, + SawtoothWindowDict, + SawtoothWindowParam, +) +from databricks.bundles.features._models.scalar_data_type import ( + ScalarDataType, + ScalarDataTypeParam, +) +from databricks.bundles.features._models.sliding_window import ( + SlidingWindow, + SlidingWindowDict, + SlidingWindowParam, +) +from databricks.bundles.features._models.source_lateness import ( + SourceLateness, + SourceLatenessDict, + SourceLatenessParam, +) +from databricks.bundles.features._models.stddev_pop_function import ( + StddevPopFunction, + StddevPopFunctionDict, + StddevPopFunctionParam, +) +from databricks.bundles.features._models.stddev_samp_function import ( + StddevSampFunction, + StddevSampFunctionDict, + StddevSampFunctionParam, +) +from databricks.bundles.features._models.stream_source import ( + StreamSource, + StreamSourceDict, + StreamSourceParam, +) +from databricks.bundles.features._models.sum_function import ( + SumFunction, + SumFunctionDict, + SumFunctionParam, +) +from databricks.bundles.features._models.time_window import ( + TimeWindow, + TimeWindowDict, + TimeWindowParam, +) +from databricks.bundles.features._models.timeseries_column import ( + TimeseriesColumn, + TimeseriesColumnDict, + TimeseriesColumnParam, +) +from databricks.bundles.features._models.tumbling_window import ( + TumblingWindow, + TumblingWindowDict, + TumblingWindowParam, +) +from databricks.bundles.features._models.var_pop_function import ( + VarPopFunction, + VarPopFunctionDict, + VarPopFunctionParam, +) +from databricks.bundles.features._models.var_samp_function import ( + VarSampFunction, + VarSampFunctionDict, + VarSampFunctionParam, +) diff --git a/python/databricks/bundles/features/_models/aggregation_function.py b/python/databricks/bundles/features/_models/aggregation_function.py new file mode 100644 index 00000000000..df057143473 --- /dev/null +++ b/python/databricks/bundles/features/_models/aggregation_function.py @@ -0,0 +1,355 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional +from databricks.bundles.features._models.approx_count_distinct_function import ( + ApproxCountDistinctFunction, + ApproxCountDistinctFunctionParam, +) +from databricks.bundles.features._models.approx_percentile_function import ( + ApproxPercentileFunction, + ApproxPercentileFunctionParam, +) +from databricks.bundles.features._models.avg_function import ( + AvgFunction, + AvgFunctionParam, +) +from databricks.bundles.features._models.count_function import ( + CountFunction, + CountFunctionParam, +) +from databricks.bundles.features._models.first_distinct_function import ( + FirstDistinctFunction, + FirstDistinctFunctionParam, +) +from databricks.bundles.features._models.first_function import ( + FirstFunction, + FirstFunctionParam, +) +from databricks.bundles.features._models.first_n_function import ( + FirstNFunction, + FirstNFunctionParam, +) +from databricks.bundles.features._models.last_distinct_function import ( + LastDistinctFunction, + LastDistinctFunctionParam, +) +from databricks.bundles.features._models.last_function import ( + LastFunction, + LastFunctionParam, +) +from databricks.bundles.features._models.last_n_function import ( + LastNFunction, + LastNFunctionParam, +) +from databricks.bundles.features._models.max_function import ( + MaxFunction, + MaxFunctionParam, +) +from databricks.bundles.features._models.min_function import ( + MinFunction, + MinFunctionParam, +) +from databricks.bundles.features._models.stddev_pop_function import ( + StddevPopFunction, + StddevPopFunctionParam, +) +from databricks.bundles.features._models.stddev_samp_function import ( + StddevSampFunction, + StddevSampFunctionParam, +) +from databricks.bundles.features._models.sum_function import ( + SumFunction, + SumFunctionParam, +) +from databricks.bundles.features._models.time_window import TimeWindow, TimeWindowParam +from databricks.bundles.features._models.var_pop_function import ( + VarPopFunction, + VarPopFunctionParam, +) +from databricks.bundles.features._models.var_samp_function import ( + VarSampFunction, + VarSampFunctionParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class AggregationFunction: + """ + :meta private: [EXPERIMENTAL] + + An aggregation function applied over a time window. + """ + + approx_count_distinct: VariableOrOptional[ApproxCountDistinctFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the approximate count of distinct values. + """ + + approx_percentile: VariableOrOptional[ApproxPercentileFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the approximate percentile of values. + """ + + avg: VariableOrOptional[AvgFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the average of values. + """ + + count_function: VariableOrOptional[CountFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the count of values. + """ + + first: VariableOrOptional[FirstFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the first value. + """ + + first_distinct: VariableOrOptional[FirstDistinctFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the first N distinct values, ordered by the feature's timeseries column. + """ + + first_n: VariableOrOptional[FirstNFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the first N values, ordered by the feature's timeseries column. + """ + + last: VariableOrOptional[LastFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the last value. + """ + + last_distinct: VariableOrOptional[LastDistinctFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the last N distinct values, ordered by the feature's timeseries column. + """ + + last_n: VariableOrOptional[LastNFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the last N values, ordered by the feature's timeseries column. + """ + + max: VariableOrOptional[MaxFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the maximum value. + """ + + min: VariableOrOptional[MinFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the minimum value. + """ + + stddev_pop: VariableOrOptional[StddevPopFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the population standard deviation. + """ + + stddev_samp: VariableOrOptional[StddevSampFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the sample standard deviation. + """ + + sum: VariableOrOptional[SumFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the sum of values. + """ + + time_window: VariableOrOptional[TimeWindow] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The time window over which the aggregation is computed. + """ + + var_pop: VariableOrOptional[VarPopFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the population variance. + """ + + var_samp: VariableOrOptional[VarSampFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the sample variance. + """ + + @classmethod + def from_dict(cls, value: "AggregationFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "AggregationFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class AggregationFunctionDict(TypedDict, total=False): + """""" + + approx_count_distinct: VariableOrOptional[ApproxCountDistinctFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the approximate count of distinct values. + """ + + approx_percentile: VariableOrOptional[ApproxPercentileFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the approximate percentile of values. + """ + + avg: VariableOrOptional[AvgFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the average of values. + """ + + count_function: VariableOrOptional[CountFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the count of values. + """ + + first: VariableOrOptional[FirstFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the first value. + """ + + first_distinct: VariableOrOptional[FirstDistinctFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the first N distinct values, ordered by the feature's timeseries column. + """ + + first_n: VariableOrOptional[FirstNFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the first N values, ordered by the feature's timeseries column. + """ + + last: VariableOrOptional[LastFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the last value. + """ + + last_distinct: VariableOrOptional[LastDistinctFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the last N distinct values, ordered by the feature's timeseries column. + """ + + last_n: VariableOrOptional[LastNFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Returns the last N values, ordered by the feature's timeseries column. + """ + + max: VariableOrOptional[MaxFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the maximum value. + """ + + min: VariableOrOptional[MinFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the minimum value. + """ + + stddev_pop: VariableOrOptional[StddevPopFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the population standard deviation. + """ + + stddev_samp: VariableOrOptional[StddevSampFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the sample standard deviation. + """ + + sum: VariableOrOptional[SumFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the sum of values. + """ + + time_window: VariableOrOptional[TimeWindowParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The time window over which the aggregation is computed. + """ + + var_pop: VariableOrOptional[VarPopFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the population variance. + """ + + var_samp: VariableOrOptional[VarSampFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Computes the sample variance. + """ + + +AggregationFunctionParam = AggregationFunctionDict | AggregationFunction diff --git a/python/databricks/bundles/features/_models/approx_count_distinct_function.py b/python/databricks/bundles/features/_models/approx_count_distinct_function.py new file mode 100644 index 00000000000..f8ec66cc3d9 --- /dev/null +++ b/python/databricks/bundles/features/_models/approx_count_distinct_function.py @@ -0,0 +1,64 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class ApproxCountDistinctFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the approximate count of distinct values. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the approximate count of distinct values is computed. + """ + + relative_sd: VariableOrOptional[float] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The maximum relative standard deviation allowed (default defined by Spark). + """ + + @classmethod + def from_dict(cls, value: "ApproxCountDistinctFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "ApproxCountDistinctFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class ApproxCountDistinctFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the approximate count of distinct values is computed. + """ + + relative_sd: VariableOrOptional[float] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The maximum relative standard deviation allowed (default defined by Spark). + """ + + +ApproxCountDistinctFunctionParam = ( + ApproxCountDistinctFunctionDict | ApproxCountDistinctFunction +) diff --git a/python/databricks/bundles/features/_models/approx_percentile_function.py b/python/databricks/bundles/features/_models/approx_percentile_function.py new file mode 100644 index 00000000000..c39a000835e --- /dev/null +++ b/python/databricks/bundles/features/_models/approx_percentile_function.py @@ -0,0 +1,76 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class ApproxPercentileFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the approximate percentile of values. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the approximate percentile is computed. + """ + + percentile: VariableOr[float] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The percentile value to compute (between 0 and 1). + """ + + accuracy: VariableOrOptional[int] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The accuracy parameter (higher is more accurate but slower). + """ + + @classmethod + def from_dict(cls, value: "ApproxPercentileFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "ApproxPercentileFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class ApproxPercentileFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the approximate percentile is computed. + """ + + percentile: VariableOr[float] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The percentile value to compute (between 0 and 1). + """ + + accuracy: VariableOrOptional[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The accuracy parameter (higher is more accurate but slower). + """ + + +ApproxPercentileFunctionParam = ApproxPercentileFunctionDict | ApproxPercentileFunction diff --git a/python/databricks/bundles/features/_models/avg_function.py b/python/databricks/bundles/features/_models/avg_function.py new file mode 100644 index 00000000000..cb92afdcce3 --- /dev/null +++ b/python/databricks/bundles/features/_models/avg_function.py @@ -0,0 +1,54 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class AvgFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the average of values. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the average is computed. For Kafka sources, use dot-prefixed path + notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + @classmethod + def from_dict(cls, value: "AvgFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "AvgFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class AvgFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the average is computed. For Kafka sources, use dot-prefixed path + notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + +AvgFunctionParam = AvgFunctionDict | AvgFunction diff --git a/python/databricks/bundles/features/_models/column_selection.py b/python/databricks/bundles/features/_models/column_selection.py new file mode 100644 index 00000000000..5e482dc2986 --- /dev/null +++ b/python/databricks/bundles/features/_models/column_selection.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class ColumnSelection: + """ + :meta private: [EXPERIMENTAL] + + A ColumnSelection function, equivalent to the LAST() record of an entity over a lifetime window + """ + + column: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Column name from source to select as the feature value. + """ + + @classmethod + def from_dict(cls, value: "ColumnSelectionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "ColumnSelectionDict": + return _transform_to_json_value(self) # type:ignore + + +class ColumnSelectionDict(TypedDict, total=False): + """""" + + column: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Column name from source to select as the feature value. + """ + + +ColumnSelectionParam = ColumnSelectionDict | ColumnSelection diff --git a/python/databricks/bundles/features/_models/count_function.py b/python/databricks/bundles/features/_models/count_function.py new file mode 100644 index 00000000000..33f45978dd4 --- /dev/null +++ b/python/databricks/bundles/features/_models/count_function.py @@ -0,0 +1,54 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class CountFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the count of values. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the count is computed. For Kafka sources, use dot-prefixed path + notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + @classmethod + def from_dict(cls, value: "CountFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "CountFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class CountFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the count is computed. For Kafka sources, use dot-prefixed path + notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + +CountFunctionParam = CountFunctionDict | CountFunction diff --git a/python/databricks/bundles/features/_models/custom_udf.py b/python/databricks/bundles/features/_models/custom_udf.py new file mode 100644 index 00000000000..7130a8e59cb --- /dev/null +++ b/python/databricks/bundles/features/_models/custom_udf.py @@ -0,0 +1,69 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass, field +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrList +from databricks.bundles.features._models.input_binding import ( + InputBinding, + InputBindingParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class CustomUdf: + """ + :meta private: [EXPERIMENTAL] + + A CustomUdf function applies a registered Unity Catalog function row-wise to + source columns, producing a single output column per row. + """ + + function_path: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Fully qualified 3-part Unity Catalog path of the function to apply. + """ + + input_bindings: VariableOrList[InputBinding] = field(default_factory=list) + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Binds each UC function parameter to a source column. + May be empty for zero-argument functions (e.g. a timestamp generator). + """ + + @classmethod + def from_dict(cls, value: "CustomUdfDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "CustomUdfDict": + return _transform_to_json_value(self) # type:ignore + + +class CustomUdfDict(TypedDict, total=False): + """""" + + function_path: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Fully qualified 3-part Unity Catalog path of the function to apply. + """ + + input_bindings: VariableOrList[InputBindingParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Binds each UC function parameter to a source column. + May be empty for zero-argument functions (e.g. a timestamp generator). + """ + + +CustomUdfParam = CustomUdfDict | CustomUdf diff --git a/python/databricks/bundles/features/_models/data_source.py b/python/databricks/bundles/features/_models/data_source.py new file mode 100644 index 00000000000..fbd00cebb56 --- /dev/null +++ b/python/databricks/bundles/features/_models/data_source.py @@ -0,0 +1,144 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional +from databricks.bundles.features._models.delta_table_source import ( + DeltaTableSource, + DeltaTableSourceParam, +) +from databricks.bundles.features._models.feature_view_source import ( + FeatureViewSource, + FeatureViewSourceParam, +) +from databricks.bundles.features._models.kafka_source import ( + KafkaSource, + KafkaSourceParam, +) +from databricks.bundles.features._models.request_source import ( + RequestSource, + RequestSourceParam, +) +from databricks.bundles.features._models.source_lateness import ( + SourceLateness, + SourceLatenessParam, +) +from databricks.bundles.features._models.stream_source import ( + StreamSource, + StreamSourceParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class DataSource: + """ + :meta private: [EXPERIMENTAL] + + Specifies the data source backing a feature. Exactly one source type must be set. + """ + + delta_table_source: VariableOrOptional[DeltaTableSource] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A Delta table data source. + """ + + feature_view_source: VariableOrOptional[FeatureViewSource] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A data source composed from registered upstream Features. + """ + + kafka_source: VariableOrOptional[KafkaSource] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A Kafka stream data source. + """ + + lateness: VariableOrOptional[SourceLateness] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Completeness timing for this Feature's use of the source. This configuration is part of the + Feature definition; it does not modify the underlying table or stream. + """ + + request_source: VariableOrOptional[RequestSource] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A request-time data source. + """ + + stream_source: VariableOrOptional[StreamSource] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A Stream data source. + """ + + @classmethod + def from_dict(cls, value: "DataSourceDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "DataSourceDict": + return _transform_to_json_value(self) # type:ignore + + +class DataSourceDict(TypedDict, total=False): + """""" + + delta_table_source: VariableOrOptional[DeltaTableSourceParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A Delta table data source. + """ + + feature_view_source: VariableOrOptional[FeatureViewSourceParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A data source composed from registered upstream Features. + """ + + kafka_source: VariableOrOptional[KafkaSourceParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A Kafka stream data source. + """ + + lateness: VariableOrOptional[SourceLatenessParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Completeness timing for this Feature's use of the source. This configuration is part of the + Feature definition; it does not modify the underlying table or stream. + """ + + request_source: VariableOrOptional[RequestSourceParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A request-time data source. + """ + + stream_source: VariableOrOptional[StreamSourceParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A Stream data source. + """ + + +DataSourceParam = DataSourceDict | DataSource diff --git a/python/databricks/bundles/features/_models/delta_table_source.py b/python/databricks/bundles/features/_models/delta_table_source.py new file mode 100644 index 00000000000..ada2359ee72 --- /dev/null +++ b/python/databricks/bundles/features/_models/delta_table_source.py @@ -0,0 +1,96 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class DeltaTableSource: + """ + :meta private: [EXPERIMENTAL] + """ + + full_name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The full three-part (catalog, schema, table) name of the Delta table. + """ + + dataframe_schema: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Schema of the resulting dataframe after transformations, in Spark StructType JSON format (from df.schema.json()). + Required if transformation_sql is specified. + Example: {"type":"struct","fields":[{"name":"col_a","type":"integer","nullable":true,"metadata":{}},{"name":"col_c","type":"integer","nullable":true,"metadata":{}}]} + """ + + filter_condition: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Single WHERE clause to filter delta table before applying transformations. Will be row-wise evaluated, so should only include conditionals and projections. + """ + + transformation_sql: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A single SQL SELECT expression applied after filter_condition. + Should contains all the columns needed (eg. "SELECT *, col_a + col_b AS col_c FROM x.y.z WHERE col_a > 0" would have `transformation_sql` "*, col_a + col_b AS col_c") + If transformation_sql is not provided, all columns of the delta table are present in the DataSource dataframe. + """ + + @classmethod + def from_dict(cls, value: "DeltaTableSourceDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "DeltaTableSourceDict": + return _transform_to_json_value(self) # type:ignore + + +class DeltaTableSourceDict(TypedDict, total=False): + """""" + + full_name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The full three-part (catalog, schema, table) name of the Delta table. + """ + + dataframe_schema: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Schema of the resulting dataframe after transformations, in Spark StructType JSON format (from df.schema.json()). + Required if transformation_sql is specified. + Example: {"type":"struct","fields":[{"name":"col_a","type":"integer","nullable":true,"metadata":{}},{"name":"col_c","type":"integer","nullable":true,"metadata":{}}]} + """ + + filter_condition: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Single WHERE clause to filter delta table before applying transformations. Will be row-wise evaluated, so should only include conditionals and projections. + """ + + transformation_sql: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A single SQL SELECT expression applied after filter_condition. + Should contains all the columns needed (eg. "SELECT *, col_a + col_b AS col_c FROM x.y.z WHERE col_a > 0" would have `transformation_sql` "*, col_a + col_b AS col_c") + If transformation_sql is not provided, all columns of the delta table are present in the DataSource dataframe. + """ + + +DeltaTableSourceParam = DeltaTableSourceDict | DeltaTableSource diff --git a/python/databricks/bundles/features/_models/entity_column.py b/python/databricks/bundles/features/_models/entity_column.py new file mode 100644 index 00000000000..21158662c96 --- /dev/null +++ b/python/databricks/bundles/features/_models/entity_column.py @@ -0,0 +1,56 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class EntityColumn: + """ + :meta private: [EXPERIMENTAL] + """ + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The name of the entity column. For Kafka sources, use dot-prefixed path notation to reference + fields within the key or value schema (e.g., "value.user_id", "key.partition_key"). For nested + fields, the leaf node name (e.g., "user_id" from "value.trip_details.user_id") is what will + be present in materialized tables and expected to match at query time. + Colon-prefixed notation (e.g., "value:user_id") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + @classmethod + def from_dict(cls, value: "EntityColumnDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "EntityColumnDict": + return _transform_to_json_value(self) # type:ignore + + +class EntityColumnDict(TypedDict, total=False): + """""" + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The name of the entity column. For Kafka sources, use dot-prefixed path notation to reference + fields within the key or value schema (e.g., "value.user_id", "key.partition_key"). For nested + fields, the leaf node name (e.g., "user_id" from "value.trip_details.user_id") is what will + be present in materialized tables and expected to match at query time. + Colon-prefixed notation (e.g., "value:user_id") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + +EntityColumnParam = EntityColumnDict | EntityColumn diff --git a/python/databricks/bundles/features/_models/feature.py b/python/databricks/bundles/features/_models/feature.py new file mode 100644 index 00000000000..000ea04316a --- /dev/null +++ b/python/databricks/bundles/features/_models/feature.py @@ -0,0 +1,173 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass, field +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._resource import Resource +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import ( + VariableOr, + VariableOrList, + VariableOrOptional, +) +from databricks.bundles.features._models.data_source import ( + DataSource, + DataSourceParam, +) +from databricks.bundles.features._models.entity_column import ( + EntityColumn, + EntityColumnParam, +) +from databricks.bundles.features._models.function import ( + Function, + FunctionParam, +) +from databricks.bundles.features._models.lifecycle import ( + Lifecycle, + LifecycleParam, +) +from databricks.bundles.features._models.lineage_context import ( + LineageContext, + LineageContextParam, +) +from databricks.bundles.features._models.timeseries_column import ( + TimeseriesColumn, + TimeseriesColumnParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class Feature(Resource): + """""" + + full_name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The full three-part name (catalog, schema, name) of the feature. This is the + feature's resource identifier; the catalog_name, schema_name, and name fields + below are OUTPUT_ONLY decomposed views of this value. + """ + + function: VariableOr[Function] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The function by which the feature is computed. + """ + + source: VariableOr[DataSource] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The data source of the feature. + """ + + description: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The description of the feature. + """ + + entities: VariableOrList[EntityColumn] = field(default_factory=list) + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The entity columns for the feature, used as aggregation keys and for query-time lookup. + """ + + lifecycle: VariableOrOptional[Lifecycle] = None + + lineage_context: VariableOrOptional[LineageContext] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Lineage context information for this feature. + WARNING: This field is primarily intended for internal use by Databricks systems and + is automatically populated when features are created through Databricks notebooks or jobs. + Users should not manually set this field as incorrect values may lead to inaccurate lineage tracking or unexpected behavior. + This field will be set by feature-engineering client and should be left unset by SDK and terraform users. + """ + + timeseries_column: VariableOrOptional[TimeseriesColumn] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Column recording time, used for point-in-time joins, backfills, and aggregations. + """ + + @classmethod + def from_dict(cls, value: "FeatureDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FeatureDict": + return _transform_to_json_value(self) # type:ignore + + +class FeatureDict(TypedDict, total=False): + """""" + + full_name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The full three-part name (catalog, schema, name) of the feature. This is the + feature's resource identifier; the catalog_name, schema_name, and name fields + below are OUTPUT_ONLY decomposed views of this value. + """ + + function: VariableOr[FunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The function by which the feature is computed. + """ + + source: VariableOr[DataSourceParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The data source of the feature. + """ + + description: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The description of the feature. + """ + + entities: VariableOrList[EntityColumnParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The entity columns for the feature, used as aggregation keys and for query-time lookup. + """ + + lifecycle: VariableOrOptional[LifecycleParam] + + lineage_context: VariableOrOptional[LineageContextParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Lineage context information for this feature. + WARNING: This field is primarily intended for internal use by Databricks systems and + is automatically populated when features are created through Databricks notebooks or jobs. + Users should not manually set this field as incorrect values may lead to inaccurate lineage tracking or unexpected behavior. + This field will be set by feature-engineering client and should be left unset by SDK and terraform users. + """ + + timeseries_column: VariableOrOptional[TimeseriesColumnParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Column recording time, used for point-in-time joins, backfills, and aggregations. + """ + + +FeatureParam = FeatureDict | Feature diff --git a/python/databricks/bundles/features/_models/feature_reference.py b/python/databricks/bundles/features/_models/feature_reference.py new file mode 100644 index 00000000000..21decdf1547 --- /dev/null +++ b/python/databricks/bundles/features/_models/feature_reference.py @@ -0,0 +1,49 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class FeatureReference: + """ + :meta private: [EXPERIMENTAL] + + A reference to one registered upstream Feature. A message rather than a bare name so an + upstream can later be pinned more precisely (e.g. by version) without a breaking type change. + """ + + feature: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The three-part full name of the upstream Feature. + """ + + @classmethod + def from_dict(cls, value: "FeatureReferenceDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FeatureReferenceDict": + return _transform_to_json_value(self) # type:ignore + + +class FeatureReferenceDict(TypedDict, total=False): + """""" + + feature: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The three-part full name of the upstream Feature. + """ + + +FeatureReferenceParam = FeatureReferenceDict | FeatureReference diff --git a/python/databricks/bundles/features/_models/feature_view_source.py b/python/databricks/bundles/features/_models/feature_view_source.py new file mode 100644 index 00000000000..bca40ffcafa --- /dev/null +++ b/python/databricks/bundles/features/_models/feature_view_source.py @@ -0,0 +1,52 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass, field +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrList +from databricks.bundles.features._models.feature_reference import ( + FeatureReference, + FeatureReferenceParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class FeatureViewSource: + """ + :meta private: [EXPERIMENTAL] + + A data source composed from registered upstream Features. + """ + + feature_references: VariableOrList[FeatureReference] = field(default_factory=list) + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The upstream Features this source reads. Must include at least one feature. + """ + + @classmethod + def from_dict(cls, value: "FeatureViewSourceDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FeatureViewSourceDict": + return _transform_to_json_value(self) # type:ignore + + +class FeatureViewSourceDict(TypedDict, total=False): + """""" + + feature_references: VariableOrList[FeatureReferenceParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The upstream Features this source reads. Must include at least one feature. + """ + + +FeatureViewSourceParam = FeatureViewSourceDict | FeatureViewSource diff --git a/python/databricks/bundles/features/_models/field_definition.py b/python/databricks/bundles/features/_models/field_definition.py new file mode 100644 index 00000000000..4578a625e52 --- /dev/null +++ b/python/databricks/bundles/features/_models/field_definition.py @@ -0,0 +1,67 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr +from databricks.bundles.features._models.scalar_data_type import ( + ScalarDataType, + ScalarDataTypeParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class FieldDefinition: + """ + :meta private: [EXPERIMENTAL] + + A single field definition within a FlatSchema, specifying the field name and its scalar data type. + Does not support nested or complex types (arrays, maps, structs). + """ + + data_type: VariableOr[ScalarDataType] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The scalar data type of the field. + """ + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The name of the field. + """ + + @classmethod + def from_dict(cls, value: "FieldDefinitionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FieldDefinitionDict": + return _transform_to_json_value(self) # type:ignore + + +class FieldDefinitionDict(TypedDict, total=False): + """""" + + data_type: VariableOr[ScalarDataTypeParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The scalar data type of the field. + """ + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The name of the field. + """ + + +FieldDefinitionParam = FieldDefinitionDict | FieldDefinition diff --git a/python/databricks/bundles/features/_models/first_distinct_function.py b/python/databricks/bundles/features/_models/first_distinct_function.py new file mode 100644 index 00000000000..87d460fa228 --- /dev/null +++ b/python/databricks/bundles/features/_models/first_distinct_function.py @@ -0,0 +1,62 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class FirstDistinctFunction: + """ + :meta private: [EXPERIMENTAL] + + Returns the first N distinct values, ordered by the feature's timeseries column. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the first N distinct values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of distinct values to return. + """ + + @classmethod + def from_dict(cls, value: "FirstDistinctFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FirstDistinctFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class FirstDistinctFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the first N distinct values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of distinct values to return. + """ + + +FirstDistinctFunctionParam = FirstDistinctFunctionDict | FirstDistinctFunction diff --git a/python/databricks/bundles/features/_models/first_function.py b/python/databricks/bundles/features/_models/first_function.py new file mode 100644 index 00000000000..2b0eb8e0b3c --- /dev/null +++ b/python/databricks/bundles/features/_models/first_function.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class FirstFunction: + """ + :meta private: [EXPERIMENTAL] + + Returns the first value. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the first value is returned. + """ + + @classmethod + def from_dict(cls, value: "FirstFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FirstFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class FirstFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the first value is returned. + """ + + +FirstFunctionParam = FirstFunctionDict | FirstFunction diff --git a/python/databricks/bundles/features/_models/first_n_function.py b/python/databricks/bundles/features/_models/first_n_function.py new file mode 100644 index 00000000000..3368c234f68 --- /dev/null +++ b/python/databricks/bundles/features/_models/first_n_function.py @@ -0,0 +1,62 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class FirstNFunction: + """ + :meta private: [EXPERIMENTAL] + + Returns the first N values, ordered by the feature's timeseries column. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the first N values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of values to return. + """ + + @classmethod + def from_dict(cls, value: "FirstNFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FirstNFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class FirstNFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the first N values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of values to return. + """ + + +FirstNFunctionParam = FirstNFunctionDict | FirstNFunction diff --git a/python/databricks/bundles/features/_models/flat_schema.py b/python/databricks/bundles/features/_models/flat_schema.py new file mode 100644 index 00000000000..e592e7da7cc --- /dev/null +++ b/python/databricks/bundles/features/_models/flat_schema.py @@ -0,0 +1,53 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass, field +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrList +from databricks.bundles.features._models.field_definition import ( + FieldDefinition, + FieldDefinitionParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class FlatSchema: + """ + :meta private: [EXPERIMENTAL] + + A flat (non-nested) schema for request-time fields, defined as an ordered list of field definitions. + This schema only supports scalar types. + """ + + fields: VariableOrList[FieldDefinition] = field(default_factory=list) + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The list of fields in this schema. + """ + + @classmethod + def from_dict(cls, value: "FlatSchemaDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FlatSchemaDict": + return _transform_to_json_value(self) # type:ignore + + +class FlatSchemaDict(TypedDict, total=False): + """""" + + fields: VariableOrList[FieldDefinitionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The list of fields in this schema. + """ + + +FlatSchemaParam = FlatSchemaDict | FlatSchema diff --git a/python/databricks/bundles/features/_models/function.py b/python/databricks/bundles/features/_models/function.py new file mode 100644 index 00000000000..cfd131a30b1 --- /dev/null +++ b/python/databricks/bundles/features/_models/function.py @@ -0,0 +1,86 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional +from databricks.bundles.features._models.aggregation_function import ( + AggregationFunction, + AggregationFunctionParam, +) +from databricks.bundles.features._models.column_selection import ( + ColumnSelection, + ColumnSelectionParam, +) +from databricks.bundles.features._models.custom_udf import ( + CustomUdf, + CustomUdfParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class Function: + """ + :meta private: [EXPERIMENTAL] + """ + + aggregation_function: VariableOrOptional[AggregationFunction] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] An aggregation function applied over a time window. + """ + + column_selection: VariableOrOptional[ColumnSelection] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Selects the latest value of a single column in a data source + """ + + custom_udf: VariableOrOptional[CustomUdf] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Applies a registered Unity Catalog function row-wise to source columns. + """ + + @classmethod + def from_dict(cls, value: "FunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "FunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class FunctionDict(TypedDict, total=False): + """""" + + aggregation_function: VariableOrOptional[AggregationFunctionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] An aggregation function applied over a time window. + """ + + column_selection: VariableOrOptional[ColumnSelectionParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Selects the latest value of a single column in a data source + """ + + custom_udf: VariableOrOptional[CustomUdfParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Applies a registered Unity Catalog function row-wise to source columns. + """ + + +FunctionParam = FunctionDict | Function diff --git a/python/databricks/bundles/features/_models/input_binding.py b/python/databricks/bundles/features/_models/input_binding.py new file mode 100644 index 00000000000..7ade8c9405e --- /dev/null +++ b/python/databricks/bundles/features/_models/input_binding.py @@ -0,0 +1,62 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class InputBinding: + """ + :meta private: [EXPERIMENTAL] + + Binds a single UC function parameter to a source column. + """ + + column: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Source column whose value is passed for this parameter at execution time. + """ + + parameter: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Name of the UC function parameter. + """ + + @classmethod + def from_dict(cls, value: "InputBindingDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "InputBindingDict": + return _transform_to_json_value(self) # type:ignore + + +class InputBindingDict(TypedDict, total=False): + """""" + + column: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Source column whose value is passed for this parameter at execution time. + """ + + parameter: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Name of the UC function parameter. + """ + + +InputBindingParam = InputBindingDict | InputBinding diff --git a/python/databricks/bundles/features/_models/job_context.py b/python/databricks/bundles/features/_models/job_context.py new file mode 100644 index 00000000000..e2d70d0a0bf --- /dev/null +++ b/python/databricks/bundles/features/_models/job_context.py @@ -0,0 +1,60 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class JobContext: + """ + :meta private: [EXPERIMENTAL] + """ + + job_id: VariableOrOptional[int] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The job ID where this API invoked. + """ + + job_run_id: VariableOrOptional[int] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The job run ID where this API was invoked. + """ + + @classmethod + def from_dict(cls, value: "JobContextDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "JobContextDict": + return _transform_to_json_value(self) # type:ignore + + +class JobContextDict(TypedDict, total=False): + """""" + + job_id: VariableOrOptional[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The job ID where this API invoked. + """ + + job_run_id: VariableOrOptional[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The job run ID where this API was invoked. + """ + + +JobContextParam = JobContextDict | JobContext diff --git a/python/databricks/bundles/features/_models/kafka_source.py b/python/databricks/bundles/features/_models/kafka_source.py new file mode 100644 index 00000000000..f828284a95c --- /dev/null +++ b/python/databricks/bundles/features/_models/kafka_source.py @@ -0,0 +1,60 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class KafkaSource: + """ + :meta private: [EXPERIMENTAL] + """ + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Name of the Kafka source, used to identify it. This is used to look up the corresponding KafkaConfig object. Can be distinct from topic name. + """ + + filter_condition: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The filter condition applied to the source data before aggregation. + """ + + @classmethod + def from_dict(cls, value: "KafkaSourceDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "KafkaSourceDict": + return _transform_to_json_value(self) # type:ignore + + +class KafkaSourceDict(TypedDict, total=False): + """""" + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Name of the Kafka source, used to identify it. This is used to look up the corresponding KafkaConfig object. Can be distinct from topic name. + """ + + filter_condition: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The filter condition applied to the source data before aggregation. + """ + + +KafkaSourceParam = KafkaSourceDict | KafkaSource diff --git a/python/databricks/bundles/features/_models/last_distinct_function.py b/python/databricks/bundles/features/_models/last_distinct_function.py new file mode 100644 index 00000000000..f0ed97454a1 --- /dev/null +++ b/python/databricks/bundles/features/_models/last_distinct_function.py @@ -0,0 +1,62 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class LastDistinctFunction: + """ + :meta private: [EXPERIMENTAL] + + Returns the last N distinct values, ordered by the feature's timeseries column. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the last N distinct values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of distinct values to return. + """ + + @classmethod + def from_dict(cls, value: "LastDistinctFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "LastDistinctFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class LastDistinctFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the last N distinct values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of distinct values to return. + """ + + +LastDistinctFunctionParam = LastDistinctFunctionDict | LastDistinctFunction diff --git a/python/databricks/bundles/features/_models/last_function.py b/python/databricks/bundles/features/_models/last_function.py new file mode 100644 index 00000000000..f08b9e4d098 --- /dev/null +++ b/python/databricks/bundles/features/_models/last_function.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class LastFunction: + """ + :meta private: [EXPERIMENTAL] + + Returns the last value. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the last value is returned. + """ + + @classmethod + def from_dict(cls, value: "LastFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "LastFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class LastFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the last value is returned. + """ + + +LastFunctionParam = LastFunctionDict | LastFunction diff --git a/python/databricks/bundles/features/_models/last_n_function.py b/python/databricks/bundles/features/_models/last_n_function.py new file mode 100644 index 00000000000..03131e3182f --- /dev/null +++ b/python/databricks/bundles/features/_models/last_n_function.py @@ -0,0 +1,62 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class LastNFunction: + """ + :meta private: [EXPERIMENTAL] + + Returns the last N values, ordered by the feature's timeseries column. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the last N values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of values to return. + """ + + @classmethod + def from_dict(cls, value: "LastNFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "LastNFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class LastNFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the last N values are returned. + """ + + n: VariableOr[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The number of values to return. + """ + + +LastNFunctionParam = LastNFunctionDict | LastNFunction diff --git a/python/databricks/bundles/features/_models/lifecycle.py b/python/databricks/bundles/features/_models/lifecycle.py new file mode 100644 index 00000000000..697776a198e --- /dev/null +++ b/python/databricks/bundles/features/_models/lifecycle.py @@ -0,0 +1,40 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class Lifecycle: + """""" + + prevent_destroy: VariableOrOptional[bool] = None + """ + Lifecycle setting to prevent the resource from being destroyed. + """ + + @classmethod + def from_dict(cls, value: "LifecycleDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "LifecycleDict": + return _transform_to_json_value(self) # type:ignore + + +class LifecycleDict(TypedDict, total=False): + """""" + + prevent_destroy: VariableOrOptional[bool] + """ + Lifecycle setting to prevent the resource from being destroyed. + """ + + +LifecycleParam = LifecycleDict | Lifecycle diff --git a/python/databricks/bundles/features/_models/lineage_context.py b/python/databricks/bundles/features/_models/lineage_context.py new file mode 100644 index 00000000000..d11cb8082d6 --- /dev/null +++ b/python/databricks/bundles/features/_models/lineage_context.py @@ -0,0 +1,66 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional +from databricks.bundles.features._models.job_context import ( + JobContext, + JobContextParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class LineageContext: + """ + :meta private: [EXPERIMENTAL] + + Lineage context information for tracking where an API was invoked. This will allow us to track lineage, which currently uses caller entity information for use across the Lineage Client and Observability in Lumberjack. + """ + + job_context: VariableOrOptional[JobContext] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Job context information including job ID and run ID. + """ + + notebook_id: VariableOrOptional[int] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The notebook ID where this API was invoked. + """ + + @classmethod + def from_dict(cls, value: "LineageContextDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "LineageContextDict": + return _transform_to_json_value(self) # type:ignore + + +class LineageContextDict(TypedDict, total=False): + """""" + + job_context: VariableOrOptional[JobContextParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Job context information including job ID and run ID. + """ + + notebook_id: VariableOrOptional[int] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The notebook ID where this API was invoked. + """ + + +LineageContextParam = LineageContextDict | LineageContext diff --git a/python/databricks/bundles/features/_models/max_function.py b/python/databricks/bundles/features/_models/max_function.py new file mode 100644 index 00000000000..87efbb2d1aa --- /dev/null +++ b/python/databricks/bundles/features/_models/max_function.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class MaxFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the maximum value. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the maximum is computed. + """ + + @classmethod + def from_dict(cls, value: "MaxFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "MaxFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class MaxFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the maximum is computed. + """ + + +MaxFunctionParam = MaxFunctionDict | MaxFunction diff --git a/python/databricks/bundles/features/_models/min_function.py b/python/databricks/bundles/features/_models/min_function.py new file mode 100644 index 00000000000..8f89b2e925f --- /dev/null +++ b/python/databricks/bundles/features/_models/min_function.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class MinFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the minimum value. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the minimum is computed. + """ + + @classmethod + def from_dict(cls, value: "MinFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "MinFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class MinFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the minimum is computed. + """ + + +MinFunctionParam = MinFunctionDict | MinFunction diff --git a/python/databricks/bundles/features/_models/request_source.py b/python/databricks/bundles/features/_models/request_source.py new file mode 100644 index 00000000000..2badaf67dd4 --- /dev/null +++ b/python/databricks/bundles/features/_models/request_source.py @@ -0,0 +1,65 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional +from databricks.bundles.features._models.flat_schema import FlatSchema, FlatSchemaParam + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class RequestSource: + """ + :meta private: [EXPERIMENTAL] + + A request-time data source whose value is provided at inference time: offline batch scoring or online serving endpoint + """ + + dataframe_schema: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A schema containing scalar or nested fields, in Spark StructType JSON format + (from df.schema.json()). This preserves field, array-element, and map-value nullability. + """ + + flat_schema: VariableOrOptional[FlatSchema] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A flat schema with scalar-typed fields only. + """ + + @classmethod + def from_dict(cls, value: "RequestSourceDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "RequestSourceDict": + return _transform_to_json_value(self) # type:ignore + + +class RequestSourceDict(TypedDict, total=False): + """""" + + dataframe_schema: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A schema containing scalar or nested fields, in Spark StructType JSON format + (from df.schema.json()). This preserves field, array-element, and map-value nullability. + """ + + flat_schema: VariableOrOptional[FlatSchemaParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A flat schema with scalar-typed fields only. + """ + + +RequestSourceParam = RequestSourceDict | RequestSource diff --git a/python/databricks/bundles/features/_models/rolling_window.py b/python/databricks/bundles/features/_models/rolling_window.py new file mode 100644 index 00000000000..d677f8af673 --- /dev/null +++ b/python/databricks/bundles/features/_models/rolling_window.py @@ -0,0 +1,68 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class RollingWindow: + """ + :meta private: [EXPERIMENTAL] + + A rolling time window with an optional non-negative delay. + """ + + delay: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing + variations unrelated to source lateness, such as a 30-day count as of one week ago. If unset, + the analytic lag is zero. It composes with source.lateness when both are set. + """ + + window_duration: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of the rolling window. Must be positive when set; absent means lifetime + (aggregate over the entity's entire history). + """ + + @classmethod + def from_dict(cls, value: "RollingWindowDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "RollingWindowDict": + return _transform_to_json_value(self) # type:ignore + + +class RollingWindowDict(TypedDict, total=False): + """""" + + delay: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing + variations unrelated to source lateness, such as a 30-day count as of one week ago. If unset, + the analytic lag is zero. It composes with source.lateness when both are set. + """ + + window_duration: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of the rolling window. Must be positive when set; absent means lifetime + (aggregate over the entity's entire history). + """ + + +RollingWindowParam = RollingWindowDict | RollingWindow diff --git a/python/databricks/bundles/features/_models/sawtooth_window.py b/python/databricks/bundles/features/_models/sawtooth_window.py new file mode 100644 index 00000000000..303ab073d46 --- /dev/null +++ b/python/databricks/bundles/features/_models/sawtooth_window.py @@ -0,0 +1,72 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class SawtoothWindow: + """ + :meta private: [EXPERIMENTAL] + + A sawtooth window served via the hybrid batch + streaming path. The batch pipeline maintains + daily partial aggregates for the bulk of the window while the streaming pipeline maintains the + most recent day(s), and serving merges them on read. Same field shape as RollingWindow, but a + distinct type so the control plane can explicitly identify hybrid (sawtooth) features rather + than inferring hybrid behavior from window_duration. + """ + + delay: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Delay is not currently supported for Sawtooth windows. + """ + + window_duration: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of the window. Must be positive and span more than two days when set, so that both + the batch (N-1 day) and stale-path (N-2 day) partial aggregates are well defined. The duration + need not be a whole number of days (e.g. 3 days 15 minutes is allowed). Absent means lifetime + (aggregate over the entity's entire history). + """ + + @classmethod + def from_dict(cls, value: "SawtoothWindowDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "SawtoothWindowDict": + return _transform_to_json_value(self) # type:ignore + + +class SawtoothWindowDict(TypedDict, total=False): + """""" + + delay: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Delay is not currently supported for Sawtooth windows. + """ + + window_duration: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of the window. Must be positive and span more than two days when set, so that both + the batch (N-1 day) and stale-path (N-2 day) partial aggregates are well defined. The duration + need not be a whole number of days (e.g. 3 days 15 minutes is allowed). Absent means lifetime + (aggregate over the entity's entire history). + """ + + +SawtoothWindowParam = SawtoothWindowDict | SawtoothWindow diff --git a/python/databricks/bundles/features/_models/scalar_data_type.py b/python/databricks/bundles/features/_models/scalar_data_type.py new file mode 100644 index 00000000000..e15dde8b7fb --- /dev/null +++ b/python/databricks/bundles/features/_models/scalar_data_type.py @@ -0,0 +1,43 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from enum import Enum +from typing import Literal + + +class ScalarDataType(Enum): + """ + :meta private: [EXPERIMENTAL] + + Scalar data types for request-time field definitions. + Only flat (non-nested) types are supported. + """ + + INTEGER = "INTEGER" + FLOAT = "FLOAT" + BOOLEAN = "BOOLEAN" + STRING = "STRING" + DOUBLE = "DOUBLE" + LONG = "LONG" + TIMESTAMP = "TIMESTAMP" + DATE = "DATE" + SHORT = "SHORT" + BINARY = "BINARY" + DECIMAL = "DECIMAL" + + +ScalarDataTypeParam = ( + Literal[ + "INTEGER", + "FLOAT", + "BOOLEAN", + "STRING", + "DOUBLE", + "LONG", + "TIMESTAMP", + "DATE", + "SHORT", + "BINARY", + "DECIMAL", + ] + | ScalarDataType +) diff --git a/python/databricks/bundles/features/_models/sliding_window.py b/python/databricks/bundles/features/_models/sliding_window.py new file mode 100644 index 00000000000..994731f380a --- /dev/null +++ b/python/databricks/bundles/features/_models/sliding_window.py @@ -0,0 +1,100 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class SlidingWindow: + """ + :meta private: [EXPERIMENTAL] + """ + + slide_duration: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The slide duration (interval by which windows advance, must be positive and less than duration). + """ + + delay: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing + variations unrelated to source lateness, such as a 30-day count as of one week ago. If unset, + the analytic lag is zero. It composes with source.lateness when both are set. + """ + + offset: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative phase shift from the default midnight UTC alignment. For example, offset=22h on + a 24h slide produces boundaries at 22:00 UTC (17:00 New York in standard time) instead of + midnight UTC. If unset, the offset is zero. Must be shorter than slide_duration (and therefore + window_duration). + """ + + window_duration: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of the sliding window. Must be positive when set; absent means lifetime + (aggregate over the entity's entire history). + """ + + @classmethod + def from_dict(cls, value: "SlidingWindowDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "SlidingWindowDict": + return _transform_to_json_value(self) # type:ignore + + +class SlidingWindowDict(TypedDict, total=False): + """""" + + slide_duration: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The slide duration (interval by which windows advance, must be positive and less than duration). + """ + + delay: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing + variations unrelated to source lateness, such as a 30-day count as of one week ago. If unset, + the analytic lag is zero. It composes with source.lateness when both are set. + """ + + offset: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative phase shift from the default midnight UTC alignment. For example, offset=22h on + a 24h slide produces boundaries at 22:00 UTC (17:00 New York in standard time) instead of + midnight UTC. If unset, the offset is zero. Must be shorter than slide_duration (and therefore + window_duration). + """ + + window_duration: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of the sliding window. Must be positive when set; absent means lifetime + (aggregate over the entity's entire history). + """ + + +SlidingWindowParam = SlidingWindowDict | SlidingWindow diff --git a/python/databricks/bundles/features/_models/source_lateness.py b/python/databricks/bundles/features/_models/source_lateness.py new file mode 100644 index 00000000000..757eac320e8 --- /dev/null +++ b/python/databricks/bundles/features/_models/source_lateness.py @@ -0,0 +1,54 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class SourceLateness: + """ + :meta private: [EXPERIMENTAL] + + Configures when event-time data from this source is considered complete for a Feature. + """ + + settling_delay: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative time to wait after a window ends before treating its source data as complete. + Training shifts the eligible evaluation time backwards by this duration so it does not join + data that would still have been settling online. Materialization waits for the duration to + elapse before publishing the window. If unset, source data is considered settled immediately. + """ + + @classmethod + def from_dict(cls, value: "SourceLatenessDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "SourceLatenessDict": + return _transform_to_json_value(self) # type:ignore + + +class SourceLatenessDict(TypedDict, total=False): + """""" + + settling_delay: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative time to wait after a window ends before treating its source data as complete. + Training shifts the eligible evaluation time backwards by this duration so it does not join + data that would still have been settling online. Materialization waits for the duration to + elapse before publishing the window. If unset, source data is considered settled immediately. + """ + + +SourceLatenessParam = SourceLatenessDict | SourceLateness diff --git a/python/databricks/bundles/features/_models/stddev_pop_function.py b/python/databricks/bundles/features/_models/stddev_pop_function.py new file mode 100644 index 00000000000..fea19dcdcb8 --- /dev/null +++ b/python/databricks/bundles/features/_models/stddev_pop_function.py @@ -0,0 +1,54 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class StddevPopFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the population standard deviation. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the population standard deviation is computed. For Kafka sources, + use dot-prefixed path notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + @classmethod + def from_dict(cls, value: "StddevPopFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "StddevPopFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class StddevPopFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the population standard deviation is computed. For Kafka sources, + use dot-prefixed path notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + +StddevPopFunctionParam = StddevPopFunctionDict | StddevPopFunction diff --git a/python/databricks/bundles/features/_models/stddev_samp_function.py b/python/databricks/bundles/features/_models/stddev_samp_function.py new file mode 100644 index 00000000000..89f2e27dc1c --- /dev/null +++ b/python/databricks/bundles/features/_models/stddev_samp_function.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class StddevSampFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the sample standard deviation. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the sample standard deviation is computed. + """ + + @classmethod + def from_dict(cls, value: "StddevSampFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "StddevSampFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class StddevSampFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the sample standard deviation is computed. + """ + + +StddevSampFunctionParam = StddevSampFunctionDict | StddevSampFunction diff --git a/python/databricks/bundles/features/_models/stream_source.py b/python/databricks/bundles/features/_models/stream_source.py new file mode 100644 index 00000000000..792d8d3832f --- /dev/null +++ b/python/databricks/bundles/features/_models/stream_source.py @@ -0,0 +1,96 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class StreamSource: + """ + :meta private: [EXPERIMENTAL] + + A Stream entity used as a data source for a feature. + """ + + full_name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Three-part full name of the Stream (catalog.schema.stream). + """ + + dataframe_schema: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Schema of the resulting dataframe after transformations, in Spark StructType + JSON format (from df.schema.json()). + Any subsequent functions operate against this dataframe. + """ + + filter_condition: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The filter condition applied to the source data before aggregation. + """ + + transformation_sql: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The pipeline runs these SQL statements immediately after conversion into + the schema specified on the Stream object. + """ + + @classmethod + def from_dict(cls, value: "StreamSourceDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "StreamSourceDict": + return _transform_to_json_value(self) # type:ignore + + +class StreamSourceDict(TypedDict, total=False): + """""" + + full_name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Three-part full name of the Stream (catalog.schema.stream). + """ + + dataframe_schema: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Schema of the resulting dataframe after transformations, in Spark StructType + JSON format (from df.schema.json()). + Any subsequent functions operate against this dataframe. + """ + + filter_condition: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The filter condition applied to the source data before aggregation. + """ + + transformation_sql: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The pipeline runs these SQL statements immediately after conversion into + the schema specified on the Stream object. + """ + + +StreamSourceParam = StreamSourceDict | StreamSource diff --git a/python/databricks/bundles/features/_models/sum_function.py b/python/databricks/bundles/features/_models/sum_function.py new file mode 100644 index 00000000000..4c35653b35c --- /dev/null +++ b/python/databricks/bundles/features/_models/sum_function.py @@ -0,0 +1,54 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class SumFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the sum of values. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the sum is computed. For Kafka sources, use dot-prefixed path + notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + @classmethod + def from_dict(cls, value: "SumFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "SumFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class SumFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the sum is computed. For Kafka sources, use dot-prefixed path + notation (e.g., "value.amount"). For nested fields, the leaf node name is used. + Colon-prefixed notation (e.g., "value:amount") is supported for backwards + compatibility but is deprecated; migrate to dot notation. + """ + + +SumFunctionParam = SumFunctionDict | SumFunction diff --git a/python/databricks/bundles/features/_models/time_window.py b/python/databricks/bundles/features/_models/time_window.py new file mode 100644 index 00000000000..87b55f325cc --- /dev/null +++ b/python/databricks/bundles/features/_models/time_window.py @@ -0,0 +1,132 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOrOptional +from databricks.bundles.features._models.rolling_window import ( + RollingWindow, + RollingWindowParam, +) +from databricks.bundles.features._models.sawtooth_window import ( + SawtoothWindow, + SawtoothWindowParam, +) +from databricks.bundles.features._models.sliding_window import ( + SlidingWindow, + SlidingWindowParam, +) +from databricks.bundles.features._models.tumbling_window import ( + TumblingWindow, + TumblingWindowParam, +) + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class TimeWindow: + """ + :meta private: [EXPERIMENTAL] + """ + + rolling: VariableOrOptional[RollingWindow] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A rolling time window with an optional non-negative delay. + """ + + sawtooth: VariableOrOptional[SawtoothWindow] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A sawtooth window served via the hybrid batch + streaming path. + """ + + sliding: VariableOrOptional[SlidingWindow] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] + """ + + start_time: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Earliest event-time boundary at which the Feature may emit an output. This gates outputs, not + the historical inputs read by a window. For example, a 365-day window with + start_time=2026-01-01 begins emitting partial-window values on that date instead of waiting + for 365 days of data; a lifetime window produces no output before start_time. If unset, + tumbling and fixed-duration sliding windows first emit at an offset-aligned boundary after a + full window can be formed. If unset, lifetime sliding windows and rolling windows emit as soon as + eligible source data exists. + Not currently supported for sawtooth windows or for Features with a stream source. + """ + + tumbling: VariableOrOptional[TumblingWindow] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] + """ + + @classmethod + def from_dict(cls, value: "TimeWindowDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "TimeWindowDict": + return _transform_to_json_value(self) # type:ignore + + +class TimeWindowDict(TypedDict, total=False): + """""" + + rolling: VariableOrOptional[RollingWindowParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A rolling time window with an optional non-negative delay. + """ + + sawtooth: VariableOrOptional[SawtoothWindowParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] A sawtooth window served via the hybrid batch + streaming path. + """ + + sliding: VariableOrOptional[SlidingWindowParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] + """ + + start_time: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Earliest event-time boundary at which the Feature may emit an output. This gates outputs, not + the historical inputs read by a window. For example, a 365-day window with + start_time=2026-01-01 begins emitting partial-window values on that date instead of waiting + for 365 days of data; a lifetime window produces no output before start_time. If unset, + tumbling and fixed-duration sliding windows first emit at an offset-aligned boundary after a + full window can be formed. If unset, lifetime sliding windows and rolling windows emit as soon as + eligible source data exists. + Not currently supported for sawtooth windows or for Features with a stream source. + """ + + tumbling: VariableOrOptional[TumblingWindowParam] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] + """ + + +TimeWindowParam = TimeWindowDict | TimeWindow diff --git a/python/databricks/bundles/features/_models/timeseries_column.py b/python/databricks/bundles/features/_models/timeseries_column.py new file mode 100644 index 00000000000..534095cddc4 --- /dev/null +++ b/python/databricks/bundles/features/_models/timeseries_column.py @@ -0,0 +1,56 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class TimeseriesColumn: + """ + :meta private: [EXPERIMENTAL] + """ + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The name of the timeseries column. For Kafka sources, use dot-prefixed path notation to + reference fields within the key or value schema (e.g., "value.event_timestamp"). For nested + fields, the leaf node name (e.g., "event_timestamp" from "value.event_details.event_timestamp") + is what will be present in materialized tables and expected to match at query time. + Colon-prefixed notation (e.g., "value:event_timestamp") is supported for + backwards compatibility but is deprecated; migrate to dot notation. + """ + + @classmethod + def from_dict(cls, value: "TimeseriesColumnDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "TimeseriesColumnDict": + return _transform_to_json_value(self) # type:ignore + + +class TimeseriesColumnDict(TypedDict, total=False): + """""" + + name: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The name of the timeseries column. For Kafka sources, use dot-prefixed path notation to + reference fields within the key or value schema (e.g., "value.event_timestamp"). For nested + fields, the leaf node name (e.g., "event_timestamp" from "value.event_details.event_timestamp") + is what will be present in materialized tables and expected to match at query time. + Colon-prefixed notation (e.g., "value:event_timestamp") is supported for + backwards compatibility but is deprecated; migrate to dot notation. + """ + + +TimeseriesColumnParam = TimeseriesColumnDict | TimeseriesColumn diff --git a/python/databricks/bundles/features/_models/tumbling_window.py b/python/databricks/bundles/features/_models/tumbling_window.py new file mode 100644 index 00000000000..25bab3f5a2d --- /dev/null +++ b/python/databricks/bundles/features/_models/tumbling_window.py @@ -0,0 +1,82 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr, VariableOrOptional + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class TumblingWindow: + """ + :meta private: [EXPERIMENTAL] + """ + + window_duration: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of each tumbling window (non-overlapping, fixed-duration windows). + """ + + delay: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing + variations unrelated to source lateness, such as a 30-day count as of one week ago. If unset, + the analytic lag is zero. It composes with source.lateness when both are set. + """ + + offset: VariableOrOptional[str] = None + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative phase shift from the default midnight UTC alignment. For example, offset=22h on + a 24h window produces boundaries at 22:00 UTC (17:00 New York in standard time) instead of + midnight UTC. If unset, the offset is zero. Must be shorter than window_duration. + """ + + @classmethod + def from_dict(cls, value: "TumblingWindowDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "TumblingWindowDict": + return _transform_to_json_value(self) # type:ignore + + +class TumblingWindowDict(TypedDict, total=False): + """""" + + window_duration: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The duration of each tumbling window (non-overlapping, fixed-duration windows). + """ + + delay: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative analytic lag that evaluates the window this far in the past. Use this for timing + variations unrelated to source lateness, such as a 30-day count as of one week ago. If unset, + the analytic lag is zero. It composes with source.lateness when both are set. + """ + + offset: VariableOrOptional[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] Non-negative phase shift from the default midnight UTC alignment. For example, offset=22h on + a 24h window produces boundaries at 22:00 UTC (17:00 New York in standard time) instead of + midnight UTC. If unset, the offset is zero. Must be shorter than window_duration. + """ + + +TumblingWindowParam = TumblingWindowDict | TumblingWindow diff --git a/python/databricks/bundles/features/_models/var_pop_function.py b/python/databricks/bundles/features/_models/var_pop_function.py new file mode 100644 index 00000000000..398778c25e3 --- /dev/null +++ b/python/databricks/bundles/features/_models/var_pop_function.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class VarPopFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the population variance. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the population variance is computed. + """ + + @classmethod + def from_dict(cls, value: "VarPopFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "VarPopFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class VarPopFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the population variance is computed. + """ + + +VarPopFunctionParam = VarPopFunctionDict | VarPopFunction diff --git a/python/databricks/bundles/features/_models/var_samp_function.py b/python/databricks/bundles/features/_models/var_samp_function.py new file mode 100644 index 00000000000..066c5b668a6 --- /dev/null +++ b/python/databricks/bundles/features/_models/var_samp_function.py @@ -0,0 +1,48 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from dataclasses import dataclass +from typing import TYPE_CHECKING, TypedDict + +from databricks.bundles.core._transform import _transform +from databricks.bundles.core._transform_to_json import _transform_to_json_value +from databricks.bundles.core._variable import VariableOr + +if TYPE_CHECKING: + from typing_extensions import Self + + +@dataclass(kw_only=True) +class VarSampFunction: + """ + :meta private: [EXPERIMENTAL] + + Computes the sample variance. + """ + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the sample variance is computed. + """ + + @classmethod + def from_dict(cls, value: "VarSampFunctionDict") -> "Self": + return _transform(cls, value) + + def as_dict(self) -> "VarSampFunctionDict": + return _transform_to_json_value(self) # type:ignore + + +class VarSampFunctionDict(TypedDict, total=False): + """""" + + input: VariableOr[str] + """ + :meta private: [EXPERIMENTAL] + + [Private Preview] The input column from which the sample variance is computed. + """ + + +VarSampFunctionParam = VarSampFunctionDict | VarSampFunction diff --git a/python/databricks_tests/core/_generated/__init__.py b/python/databricks_tests/core/_generated/__init__.py index 762fff248ef..e1e07830307 100644 --- a/python/databricks_tests/core/_generated/__init__.py +++ b/python/databricks_tests/core/_generated/__init__.py @@ -11,6 +11,7 @@ database_instances, experiments, external_locations, + features, genie_spaces, instance_pools, job_runs, @@ -46,6 +47,7 @@ database_instances._test_case(), experiments._test_case(), external_locations._test_case(), + features._test_case(), genie_spaces._test_case(), instance_pools._test_case(), job_runs._test_case(), diff --git a/python/databricks_tests/core/_generated/features.py b/python/databricks_tests/core/_generated/features.py new file mode 100644 index 00000000000..a60a7008f46 --- /dev/null +++ b/python/databricks_tests/core/_generated/features.py @@ -0,0 +1,31 @@ +# Code generated by pydabs-codegen. DO NOT EDIT. + +from databricks.bundles.core import Resources, feature_mutator +from databricks.bundles.core._generated.features import _resource_type +from databricks.bundles.features._models.data_source import DataSource +from databricks.bundles.features._models.feature import Feature +from databricks.bundles.features._models.function import Function +from databricks.bundles.features._models.lifecycle import Lifecycle +from databricks_tests.core._resource_test_case import ResourceTestCase + + +def _test_case(): + return ( + ResourceTestCase( + add_resource=Resources.add_feature, + dict_example={ + "full_name": "full_name", + "function": {}, + "lifecycle": {}, + "source": {}, + }, + dataclass_example=Feature( + full_name="full_name", + function=Function(), + lifecycle=Lifecycle(), + source=DataSource(), + ), + mutator=feature_mutator, + ), + _resource_type(), + ) diff --git a/python/databricks_tests/core/public_api.txt b/python/databricks_tests/core/public_api.txt index 63e6ad0cf1f..3c34aac5aa2 100644 --- a/python/databricks_tests/core/public_api.txt +++ b/python/databricks_tests/core/public_api.txt @@ -22,6 +22,7 @@ __all__ = [ database_catalog_mutator, database_instance_mutator, external_location_mutator, + feature_mutator, genie_space_mutator, instance_pool_mutator, job_mutator, @@ -101,6 +102,7 @@ class Resources: def add_diagnostic_warning(self, msg: str, *, detail: Union[str, None] = None, path: Union[tuple[str, ...], None] = None, location: Union[Location, None] = None) -> None def add_diagnostics(self, other: Diagnostics) -> None def add_external_location(self, resource_name: str, external_location: ExternalLocationParam, *, location: Union[Location, None] = None) -> None + def add_feature(self, resource_name: str, feature: FeatureParam, *, location: Union[Location, None] = None) -> None def add_genie_space(self, resource_name: str, genie_space: GenieSpaceParam, *, location: Union[Location, None] = None) -> None def add_instance_pool(self, resource_name: str, instance_pool: InstancePoolParam, *, location: Union[Location, None] = None) -> None def add_job(self, resource_name: str, job: JobParam, *, location: Union[Location, None] = None) -> None @@ -136,6 +138,7 @@ class Resources: @property diagnostics -> Diagnostics @property experiments -> dict[str, MlflowExperiment] @property external_locations -> dict[str, ExternalLocation] + @property features -> dict[str, Feature] @property genie_spaces -> dict[str, GenieSpace] @property instance_pools -> dict[str, InstancePool] @property job_runs -> dict[str, JobRun] @@ -210,6 +213,10 @@ def database_instance_mutator(function: Callable) -> ResourceMutator[DatabaseIns @overload def external_location_mutator(function: Callable[[ExternalLocation], ExternalLocation]) -> ResourceMutator[ExternalLocation] def external_location_mutator(function: Callable) -> ResourceMutator[ExternalLocation] +@overload def feature_mutator(function: Callable[[Bundle, Feature], Feature]) -> ResourceMutator[Feature] +@overload def feature_mutator(function: Callable[[Feature], Feature]) -> ResourceMutator[Feature] +def feature_mutator(function: Callable) -> ResourceMutator[Feature] + @overload def genie_space_mutator(function: Callable[[Bundle, GenieSpace], GenieSpace]) -> ResourceMutator[GenieSpace] @overload def genie_space_mutator(function: Callable[[GenieSpace], GenieSpace]) -> ResourceMutator[GenieSpace] def genie_space_mutator(function: Callable) -> ResourceMutator[GenieSpace] @@ -314,6 +321,7 @@ singular_name=dashboard plural_name=dashboards resource_type=Dashboard singular_name=database_catalog plural_name=database_catalogs resource_type=DatabaseCatalog singular_name=database_instance plural_name=database_instances resource_type=DatabaseInstance singular_name=external_location plural_name=external_locations resource_type=ExternalLocation +singular_name=feature plural_name=features resource_type=Feature singular_name=genie_space plural_name=genie_spaces resource_type=GenieSpace singular_name=instance_pool plural_name=instance_pools resource_type=InstancePool singular_name=job plural_name=jobs resource_type=Job diff --git a/python/docs/databricks.bundles.features.rst b/python/docs/databricks.bundles.features.rst new file mode 100644 index 00000000000..277c3b7339b --- /dev/null +++ b/python/docs/databricks.bundles.features.rst @@ -0,0 +1,11 @@ +Features +=============================== + +.. currentmodule:: databricks.bundles.features + +**Package:** ``databricks.bundles.features`` + +Classes +--------------- + +.. automodule:: databricks.bundles.features diff --git a/python/docs/index.rst b/python/docs/index.rst index b758c0febd4..5ca4bda8ede 100644 --- a/python/docs/index.rst +++ b/python/docs/index.rst @@ -20,6 +20,7 @@ See `Bundle configuration in Python