Skip to content

Commit

Permalink
Fix serialization bug in BroadcastJoinLayer (#9871)
Browse files Browse the repository at this point in the history
  • Loading branch information
rjzamora committed Jan 26, 2023
1 parent c6b7052 commit 3f8d092
Show file tree
Hide file tree
Showing 2 changed files with 27 additions and 6 deletions.
12 changes: 6 additions & 6 deletions dask/layers.py
Expand Up @@ -872,6 +872,8 @@ def __init__(
rhs_npartitions,
parts_out=None,
annotations=None,
left_on=None,
right_on=None,
**merge_kwargs,
):
super().__init__(annotations=annotations)
Expand All @@ -882,14 +884,12 @@ def __init__(
self.rhs_name = rhs_name
self.rhs_npartitions = rhs_npartitions
self.parts_out = parts_out or set(range(self.npartitions))
self.left_on = tuple(left_on) if isinstance(left_on, list) else left_on
self.right_on = tuple(right_on) if isinstance(right_on, list) else right_on
self.merge_kwargs = merge_kwargs
self.how = self.merge_kwargs.get("how")
self.left_on = self.merge_kwargs.get("left_on")
self.right_on = self.merge_kwargs.get("right_on")
if isinstance(self.left_on, list):
self.left_on = (list, tuple(self.left_on))
if isinstance(self.right_on, list):
self.right_on = (list, tuple(self.right_on))
self.merge_kwargs["left_on"] = self.left_on
self.merge_kwargs["right_on"] = self.right_on

def get_output_keys(self):
return {(self.name, part) for part in self.parts_out}
Expand Down
21 changes: 21 additions & 0 deletions dask/tests/test_distributed.py
Expand Up @@ -148,6 +148,27 @@ def test_fused_blockwise_dataframe_merge(c, fuse):
)


@pytest.mark.parametrize("on", ["a", ["a"]])
@pytest.mark.parametrize("broadcast", [True, False])
def test_dataframe_broadcast_merge(c, on, broadcast):
# See: https://github.com/dask/dask/issues/9870
pd = pytest.importorskip("pandas")
dd = pytest.importorskip("dask.dataframe")

pdfl = pd.DataFrame({"a": [1, 2] * 2, "b_left": range(4)})
pdfr = pd.DataFrame({"a": [2, 1], "b_right": range(2)})
dfl = dd.from_pandas(pdfl, npartitions=4)
dfr = dd.from_pandas(pdfr, npartitions=2)

ddfm = dd.merge(dfl, dfr, on=on, broadcast=broadcast, shuffle="tasks")
dfm = ddfm.compute()
dd.utils.assert_eq(
dfm.sort_values("a"),
pd.merge(pdfl, pdfr, on=on).sort_values("a"),
check_index=False,
)


@pytest.mark.parametrize(
"computation",
[
Expand Down

0 comments on commit 3f8d092

Please sign in to comment.