@@ -892,8 +892,12 @@ def upsert(
892892 except ModuleNotFoundError as e :
893893 raise ModuleNotFoundError ("For writes PyArrow needs to be installed" ) from e
894894
895- from pyiceberg .io .pyarrow import expression_to_pyarrow
896- from pyiceberg .table import upsert_util
895+ from pyiceberg .io .pyarrow import (
896+ expression_to_pyarrow ,
897+ upsert_create_match_filter ,
898+ upsert_get_rows_to_update ,
899+ upsert_has_duplicate_rows ,
900+ )
897901
898902 if join_cols is None :
899903 join_cols = []
@@ -910,7 +914,7 @@ def upsert(
910914 if not when_matched_update_all and not when_not_matched_insert_all :
911915 raise ValueError ("no upsert options selected...exiting" )
912916
913- if upsert_util . has_duplicate_rows (df , join_cols ):
917+ if upsert_has_duplicate_rows (df , join_cols ):
914918 raise ValueError ("Duplicate rows found in source dataset based on the key columns. No upsert executed" )
915919
916920 from pyiceberg .io .pyarrow import _check_pyarrow_schema_compatible
@@ -924,7 +928,7 @@ def upsert(
924928 )
925929
926930 # get list of rows that exist so we don't have to load the entire target table
927- matched_predicate = upsert_util . create_match_filter (df , join_cols )
931+ matched_predicate = upsert_create_match_filter (df , join_cols )
928932
929933 # We must use Transaction.table_metadata for the scan. This includes all uncommitted - but relevant - changes.
930934
@@ -952,17 +956,17 @@ def upsert(
952956 # values have actually changed. We don't want to do just a blanket overwrite for matched
953957 # rows if the actual non-key column data hasn't changed.
954958 # this extra step avoids unnecessary IO and writes
955- rows_to_update = upsert_util . get_rows_to_update (df , rows , join_cols )
959+ rows_to_update = upsert_get_rows_to_update (df , rows , join_cols )
956960
957961 if len (rows_to_update ) > 0 :
958962 # build the match predicate filter
959- overwrite_mask_predicate = upsert_util . create_match_filter (rows_to_update , join_cols )
963+ overwrite_mask_predicate = upsert_create_match_filter (rows_to_update , join_cols )
960964
961965 batches_to_overwrite .append (rows_to_update )
962966 overwrite_predicates .append (overwrite_mask_predicate )
963967
964968 if when_not_matched_insert_all :
965- expr_match = upsert_util . create_match_filter (rows , join_cols )
969+ expr_match = upsert_create_match_filter (rows , join_cols )
966970 expr_match_bound = bind (self .table_metadata .schema (), expr_match , case_sensitive = case_sensitive )
967971 expr_match_arrow = expression_to_pyarrow (expr_match_bound )
968972
@@ -2663,8 +2667,9 @@ def plan_files(self) -> Iterable[FileScanTask]:
26632667 options = self .options ,
26642668 ).plan_files (
26652669 manifests = manifests ,
2666- manifest_entry_filter = lambda manifest_entry : manifest_entry .snapshot_id in append_snapshot_ids
2667- and manifest_entry .status == ManifestEntryStatus .ADDED ,
2670+ manifest_entry_filter = lambda manifest_entry : (
2671+ manifest_entry .snapshot_id in append_snapshot_ids and manifest_entry .status == ManifestEntryStatus .ADDED
2672+ ),
26682673 )
26692674
26702675 def to_arrow (self ) -> pa .Table :
0 commit comments