使用map_partitions
您可以修改该特定分区。
然后通过切换到延迟对象来替换数据帧中修改的分区来创建一个新帧,将延迟对象替换到列表中,然后切换回 dask 数据帧。
def append_row_dict(df, row_dict):
small_df = pd.DataFrame(row_dict)
return df.append(small_df)
p_df = pd.DataFrame({'a':np.arange(0,10)})
dask_df = dd.from_pandas(p_df,npartitions=4)
part_to_change = 1
new_partion = dask_df.get_partition(part_to_change).map_partitions(append_row_dict,{'a':[-1]})
list_of_delayed = dask_df.to_delayed()
## we only have 1 delayed object for 1 partition
assert new_partion.npartitions==1
list_of_delayed[part_to_change]=new_partion.to_delayed()[0]
new_dask_df = dd.from_delayed(list_of_delayed, meta=dask_df._meta)
new_dask_df.get_partition(part_to_change).compute()
a
3 3
4 4
5 5
0 -1