[SPARK-35349][SQL] Add code-gen for left/right outer sort merge join #32476

c21 · 2021-05-08T07:24:31Z

What changes were proposed in this pull request?

This PR is to add code-gen support for LEFT OUTER / RIGHT OUTER sort merge join. Currently sort merge join only supports inner join type (https://github.com/apache/spark/blob/master/sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala#L374 ). There's no fundamental reason why we cannot support code-gen for other join types. Here we add code-gen for LEFT OUTER / RIGHT OUTER join. Will submit followup PRs to add LEFT SEMI, LEFT ANTI and FULL OUTER code-gen separately.

The change is to extend current sort merge join logic to work with LEFT OUTER and RIGHT OUTER (should work with LEFT SEMI/ANTI as well, but FULL OUTER join needs some other more code change). Replace left/right with streamed/buffered to make code extendable to other join types besides inner join.

Example query:

val df1 = spark.range(10).select($"id".as("k1"), $"id".as("k3"))
val df2 = spark.range(4).select($"id".as("k2"), $"id".as("k4"))
df1.join(df2.hint("SHUFFLE_MERGE"), $"k1" === $"k2" && $"k3" + 1 < $"k4", "left_outer").explain("codegen")

Example generated code:

== Subtree 5 / 5 (maxMethodCodeSize:396; maxConstantPoolSize:159(0.24% used); numInnerClasses:0) ==
*(5) SortMergeJoin [k1#2L], [k2#8L], LeftOuter, ((k3#3L + 1) < k4#9L)
:- *(2) Sort [k1#2L ASC NULLS FIRST], false, 0
:  +- Exchange hashpartitioning(k1#2L, 5), ENSURE_REQUIREMENTS, [id=#26]
:     +- *(1) Project [id#0L AS k1#2L, id#0L AS k3#3L]
:        +- *(1) Range (0, 10, step=1, splits=2)
+- *(4) Sort [k2#8L ASC NULLS FIRST], false, 0
   +- Exchange hashpartitioning(k2#8L, 5), ENSURE_REQUIREMENTS, [id=#32]
      +- *(3) Project [id#6L AS k2#8L, id#6L AS k4#9L]
         +- *(3) Range (0, 4, step=1, splits=2)

Generated code:
/* 001 */ public Object generate(Object[] references) {
/* 002 */   return new GeneratedIteratorForCodegenStage5(references);
/* 003 */ }
/* 004 */
/* 005 */ // codegenStageId=5
/* 006 */ final class GeneratedIteratorForCodegenStage5 extends org.apache.spark.sql.execution.BufferedRowIterator {
/* 007 */   private Object[] references;
/* 008 */   private scala.collection.Iterator[] inputs;
/* 009 */   private scala.collection.Iterator smj_streamedInput_0;
/* 010 */   private scala.collection.Iterator smj_bufferedInput_0;
/* 011 */   private InternalRow smj_streamedRow_0;
/* 012 */   private InternalRow smj_bufferedRow_0;
/* 013 */   private long smj_value_2;
/* 014 */   private org.apache.spark.sql.execution.ExternalAppendOnlyUnsafeRowArray smj_matches_0;
/* 015 */   private long smj_value_3;
/* 016 */   private org.apache.spark.sql.catalyst.expressions.codegen.UnsafeRowWriter[] smj_mutableStateArray_0 = new org.apache.spark.sql.catalyst.expressions.codegen.UnsafeRowWriter[1];
/* 017 */
/* 018 */   public GeneratedIteratorForCodegenStage5(Object[] references) {
/* 019 */     this.references = references;
/* 020 */   }
/* 021 */
/* 022 */   public void init(int index, scala.collection.Iterator[] inputs) {
/* 023 */     partitionIndex = index;
/* 024 */     this.inputs = inputs;
/* 025 */     smj_streamedInput_0 = inputs[0];
/* 026 */     smj_bufferedInput_0 = inputs[1];
/* 027 */
/* 028 */     smj_matches_0 = new org.apache.spark.sql.execution.ExternalAppendOnlyUnsafeRowArray(2147483632, 2147483647);
/* 029 */     smj_mutableStateArray_0[0] = new org.apache.spark.sql.catalyst.expressions.codegen.UnsafeRowWriter(4, 0);
/* 030 */
/* 031 */   }
/* 032 */
/* 033 */   private boolean findNextJoinRows(
/* 034 */     scala.collection.Iterator streamedIter,
/* 035 */     scala.collection.Iterator bufferedIter) {
/* 036 */     smj_streamedRow_0 = null;
/* 037 */     int comp = 0;
/* 038 */     while (smj_streamedRow_0 == null) {
/* 039 */       if (!streamedIter.hasNext()) return false;
/* 040 */       smj_streamedRow_0 = (InternalRow) streamedIter.next();
/* 041 */       long smj_value_0 = smj_streamedRow_0.getLong(0);
/* 042 */       if (false) {
/* 043 */         if (!smj_matches_0.isEmpty()) {
/* 044 */           smj_matches_0.clear();
/* 045 */         }
/* 046 */         return false;
/* 047 */
/* 048 */       }
/* 049 */       if (!smj_matches_0.isEmpty()) {
/* 050 */         comp = 0;
/* 051 */         if (comp == 0) {
/* 052 */           comp = (smj_value_0 > smj_value_3 ? 1 : smj_value_0 < smj_value_3 ? -1 : 0);
/* 053 */         }
/* 054 */
/* 055 */         if (comp == 0) {
/* 056 */           return true;
/* 057 */         }
/* 058 */         smj_matches_0.clear();
/* 059 */       }
/* 060 */
/* 061 */       do {
/* 062 */         if (smj_bufferedRow_0 == null) {
/* 063 */           if (!bufferedIter.hasNext()) {
/* 064 */             smj_value_3 = smj_value_0;
/* 065 */             return !smj_matches_0.isEmpty();
/* 066 */           }
/* 067 */           smj_bufferedRow_0 = (InternalRow) bufferedIter.next();
/* 068 */           long smj_value_1 = smj_bufferedRow_0.getLong(0);
/* 069 */           if (false) {
/* 070 */             smj_bufferedRow_0 = null;
/* 071 */             continue;
/* 072 */           }
/* 073 */           smj_value_2 = smj_value_1;
/* 074 */         }
/* 075 */
/* 076 */         comp = 0;
/* 077 */         if (comp == 0) {
/* 078 */           comp = (smj_value_0 > smj_value_2 ? 1 : smj_value_0 < smj_value_2 ? -1 : 0);
/* 079 */         }
/* 080 */
/* 081 */         if (comp > 0) {
/* 082 */           smj_bufferedRow_0 = null;
/* 083 */         } else if (comp < 0) {
/* 084 */           if (!smj_matches_0.isEmpty()) {
/* 085 */             smj_value_3 = smj_value_0;
/* 086 */             return true;
/* 087 */           } else {
/* 088 */             return false;
/* 089 */           }
/* 090 */         } else {
/* 091 */           smj_matches_0.add((UnsafeRow) smj_bufferedRow_0);
/* 092 */           smj_bufferedRow_0 = null;
/* 093 */         }
/* 094 */       } while (smj_streamedRow_0 != null);
/* 095 */     }
/* 096 */     return false; // unreachable
/* 097 */   }
/* 098 */
/* 099 */   protected void processNext() throws java.io.IOException {
/* 100 */     while (smj_streamedInput_0.hasNext()) {
/* 101 */       findNextJoinRows(smj_streamedInput_0, smj_bufferedInput_0);
/* 102 */       long smj_value_4 = -1L;
/* 103 */       long smj_value_5 = -1L;
/* 104 */       boolean smj_loaded_0 = false;
/* 105 */       smj_value_5 = smj_streamedRow_0.getLong(1);
/* 106 */       scala.collection.Iterator<UnsafeRow> smj_iterator_0 = smj_matches_0.generateIterator();
/* 107 */       boolean smj_foundMatch_0 = false;
/* 108 */
/* 109 */       // the last iteration of this loop is to emit an empty row if there is no matched rows.
/* 110 */       while (smj_iterator_0.hasNext() || !smj_foundMatch_0) {
/* 111 */         InternalRow smj_bufferedRow_1 = smj_iterator_0.hasNext() ?
/* 112 */         (InternalRow) smj_iterator_0.next() : null;
/* 113 */         boolean smj_isNull_5 = true;
/* 114 */         long smj_value_9 = -1L;
/* 115 */         if (smj_bufferedRow_1 != null) {
/* 116 */           long smj_value_8 = smj_bufferedRow_1.getLong(1);
/* 117 */           smj_isNull_5 = false;
/* 118 */           smj_value_9 = smj_value_8;
/* 119 */         }
/* 120 */         if (smj_bufferedRow_1 != null) {
/* 121 */           boolean smj_isNull_6 = true;
/* 122 */           boolean smj_value_10 = false;
/* 123 */           long smj_value_11 = -1L;
/* 124 */
/* 125 */           smj_value_11 = smj_value_5 + 1L;
/* 126 */
/* 127 */           if (!smj_isNull_5) {
/* 128 */             smj_isNull_6 = false; // resultCode could change nullability.
/* 129 */             smj_value_10 = smj_value_11 < smj_value_9;
/* 130 */
/* 131 */           }
/* 132 */           if (smj_isNull_6 || !smj_value_10) {
/* 133 */             continue;
/* 134 */           }
/* 135 */         }
/* 136 */         if (!smj_loaded_0) {
/* 137 */           smj_loaded_0 = true;
/* 138 */           smj_value_4 = smj_streamedRow_0.getLong(0);
/* 139 */         }
/* 140 */         boolean smj_isNull_3 = true;
/* 141 */         long smj_value_7 = -1L;
/* 142 */         if (smj_bufferedRow_1 != null) {
/* 143 */           long smj_value_6 = smj_bufferedRow_1.getLong(0);
/* 144 */           smj_isNull_3 = false;
/* 145 */           smj_value_7 = smj_value_6;
/* 146 */         }
/* 147 */         smj_foundMatch_0 = true;
/* 148 */         ((org.apache.spark.sql.execution.metric.SQLMetric) references[0] /* numOutputRows */).add(1);
/* 149 */
/* 150 */         smj_mutableStateArray_0[0].reset();
/* 151 */
/* 152 */         smj_mutableStateArray_0[0].zeroOutNullBytes();
/* 153 */
/* 154 */         smj_mutableStateArray_0[0].write(0, smj_value_4);
/* 155 */
/* 156 */         smj_mutableStateArray_0[0].write(1, smj_value_5);
/* 157 */
/* 158 */         if (smj_isNull_3) {
/* 159 */           smj_mutableStateArray_0[0].setNullAt(2);
/* 160 */         } else {
/* 161 */           smj_mutableStateArray_0[0].write(2, smj_value_7);
/* 162 */         }
/* 163 */
/* 164 */         if (smj_isNull_5) {
/* 165 */           smj_mutableStateArray_0[0].setNullAt(3);
/* 166 */         } else {
/* 167 */           smj_mutableStateArray_0[0].write(3, smj_value_9);
/* 168 */         }
/* 169 */         append((smj_mutableStateArray_0[0].getRow()).copy());
/* 170 */
/* 171 */       }
/* 172 */       if (shouldStop()) return;
/* 173 */     }
/* 174 */     ((org.apache.spark.sql.execution.joins.SortMergeJoinExec) references[1] /* plan */).cleanupResources();
/* 175 */   }
/* 176 */
/* 177 */ }

Why are the changes needed?

Improve query CPU performance. Example micro benchmark below showed 10% run-time improvement.

def sortMergeJoinWithDuplicates(): Unit = {
    val N = 2 << 20
    codegenBenchmark("sort merge join with duplicates", N) {
      val df1 = spark.range(N)
        .selectExpr(s"(id * 15485863) % ${N*10} as k1", "id as k3")
      val df2 = spark.range(N)
        .selectExpr(s"(id * 15485867) % ${N*10} as k2", "id as k4")
      val df = df1.join(df2, col("k1") === col("k2") && col("k3") * 3 < col("k4"), "left_outer")
      assert(df.queryExecution.sparkPlan.find(_.isInstanceOf[SortMergeJoinExec]).isDefined)
      df.noop()
    }
 }

Running benchmark: sort merge join with duplicates
  Running case: sort merge join with duplicates outer-smj-codegen off
  Stopped after 2 iterations, 2696 ms
  Running case: sort merge join with duplicates outer-smj-codegen on
  Stopped after 5 iterations, 6058 ms

Java HotSpot(TM) 64-Bit Server VM 1.8.0_181-b13 on Mac OS X 10.16
Intel(R) Core(TM) i9-9980HK CPU @ 2.40GHz
sort merge join with duplicates:                       Best Time(ms)   Avg Time(ms)   Stdev(ms)    Rate(M/s)   Per Row(ns)   Relative
-------------------------------------------------------------------------------------------------------------------------------------
sort merge join with duplicates outer-smj-codegen off           1333           1348          21          1.6         635.7       1.0X
sort merge join with duplicates outer-smj-codegen on            1169           1212          47          1.8         557.4       1.1X

Does this PR introduce any user-facing change?

No.

How was this patch tested?

Added unit test in WholeStageCodegenSuite.scala and WholeStageCodegenSuite.scala.

c21 · 2021-05-08T07:29:39Z

cc @cloud-fan and @maropu could you help take a look when you have time? Thanks.

SparkQA · 2021-05-08T08:50:12Z

Kubernetes integration test starting
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42805/

SparkQA · 2021-05-08T08:54:29Z

Kubernetes integration test status failure
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42805/

SparkQA · 2021-05-08T10:46:41Z

Test build #138282 has finished for PR 32476 at commit 95e56b8.

This patch fails Spark unit tests.
This patch merges cleanly.
This patch adds the following public classes (experimental):
trait ShuffledJoin extends JoinCodegenSupport

SparkQA · 2021-05-09T00:13:54Z

Kubernetes integration test starting
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42820/

SparkQA · 2021-05-09T00:13:56Z

Kubernetes integration test status failure
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42820/

SparkQA · 2021-05-09T02:37:31Z

Test build #138298 has finished for PR 32476 at commit 2166c24.

This patch fails Spark unit tests.
This patch merges cleanly.
This patch adds no public classes.

SparkQA · 2021-05-09T04:55:42Z

Kubernetes integration test starting
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42824/

SparkQA · 2021-05-09T04:55:43Z

Kubernetes integration test status failure
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42824/

SparkQA · 2021-05-09T07:41:42Z

Test build #138301 has finished for PR 32476 at commit 13ffa8a.

This patch passes all tests.
This patch merges cleanly.
This patch adds no public classes.

maropu · 2021-05-09T13:13:18Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

+      throw new IllegalArgumentException(
+        s"SortMergeJoin.streamedPlan/bufferedPlan should not take $x as the JoinType")


How about this?

private lazy val ((streamedPlan, streamedKyes), (bufferedPlan, bufferedKeys)) = joinType match { case _: InnerLike | LeftOuter => ((left, leftKeys), (right, rightKeys)) case RightOuter => ((right, rightKeys), (left, leftKeys)) case x => throw new IllegalArgumentException( s"SortMergeJoin.streamedPlan/bufferedPlan should not take $x as the JoinType") } private lazy val streamOutput = streamedPlan.output private lazy val bufferedOutput = bufferedPlan.output

I think we don't need to repeat the joinType check.

Makes sense to me, addressed in #32495 .

maropu · 2021-05-09T13:14:41Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

@@ -353,12 +353,37 @@ case class SortMergeJoinExec(
    }
  }

-  override def supportCodegen: Boolean = {
-    joinType.isInstanceOf[InnerLike]
+  private lazy val (streamedPlan, bufferedPlan) = joinType match {


We need lazy here?

@maropu - yes, this is used for code-gen only. Note here we only pattern match inner/left outer/right outer join, so it will throw exception with val for other join types.

maropu · 2021-05-09T14:01:08Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

+           |if (!$matches.isEmpty()) {
+           |  $matches.clear();
+           |}
+           |return false;


// Eagerly return streamed row. s""" |$matches.clear(); |return false; """.stripMargin

?

Wanted to avoid clear() if isEmpty() is true. ExternalAppendOnlyUnsafeRowArray.isEmpty() is very cheap but clear() sets multiple variables.

I see. Could you leave some comments about it there?

@maropu - added comment.

maropu · 2021-05-09T23:36:31Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

+
+    lazy val outerJoin = {
+      val foundMatch = ctx.freshName("foundMatch")
+      val foundJoinRows = ctx.freshName("foundJoinRows")


foundJoinRows not used?

My bad, forget to remove it during code iterations. Will remove.

@maropu - removed.

maropu · 2021-05-09T23:47:28Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

-         |    scala.collection.Iterator leftIter,
-         |    scala.collection.Iterator rightIter) {
-         |  $leftRow = null;
+         |private boolean findNextJoinRows(


In the outer case, a return value is not used?

It looks reusing the inner-case code makes the outer-case code inefficient. For example, if there are too many matched duplicate rows in the buffered side, it seems we don't need to put all the rows in matches, right?

In the outer case, a return value is not used?

Yes. Otherwise it's very hard to re-use code in findNextJoinRows. I can further make more change to not return anything for findNextJoinRows in case it's an outer join. Do we want to do that?

For example, if there are too many matched duplicate rows in the buffered side, it seems we don't need to put all the rows in matches, right?

Why we don't need to put all the rows? We anyway need to evaluate all the rows on buffered side for join, right?

Why we don't need to put all the rows? We anyway need to evaluate all the rows on buffered side for join, right?

Oh, my bad. ya, you're right. I misunderstood it.

In the outer case, a return value is not used?
Yes. Otherwise it's very hard to re-use code in findNextJoinRows. I can further make more change to not return anything for findNextJoinRows in case it's an outer join. Do we want to do that?

okay, the current one looks fine. Let's just wait for a @cloud-fan comment here.

btw, in the current generated code, it seems conditionCheck is evaluated outside findNextJoinRows. We cannot evaluate it inside findNextJoinRows to avoid putting unmached rows in matches?

@maropu - No I think we need buffer anyway. The buffered rows has same join keys with current streamed row. But there can be multiple followed streamed rows having same join keys, as the buffered rows. Even though buffered rows cannot match condition with current streamed row, they may match condition with followed streamed rows. I think this is how current sort merge join (code-gen & iterator) is designed.

maropu · 2021-05-10T00:00:46Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

         |      }
-         |    } while ($leftRow != null);
+         |    } while ($streamedRow != null);
         |  }
         |  return false; // unreachable


(This is not related to this PR though) In this case, could we throw an illegal state exception?

@maropu - sounds good to me. Will update.

@maropu - Interesting, when I tried to add a throw new IllegalStateException before return false, janino compiler is clever enough to figure out the statement is unreachable and throws exception when trying to compile - https://gist.github.com/c21/196166411d5d0406d9a76b37be889194 . So I think we'd better keep this as it is for now?

Oh, I missed this comment. it looks interesting. sgtm.

maropu · 2021-05-10T00:20:43Z

Could you update the JoinBenchmark results, too?

c21 · 2021-05-10T00:29:38Z

@maropu - JoinBenchmark has only inner sort merge join, but not left/right outer join. So this PR does not affect the result of benchmark as it is. Shall we have a followup PR to update the join benchmark? I wanted to add other more test cases in JoinBenchmark as well.

maropu · 2021-05-10T00:30:59Z

@maropu - JoinBenchmark has only inner sort merge join, but not left/right outer join. So this PR does not affect the result of benchmark as it is. Shall we have a followup PR to update the join benchmark? I wanted to add other more test cases in JoinBenchmark as well.

Ah, okay. sgtm.

cloud-fan · 2021-05-10T09:36:00Z

can we open a PR to do the renaming first? left, right to streamed, buffered, to make this PR easier to review.

c21 · 2021-05-10T22:43:09Z

can we open a PR to do the renaming first? left, right to streamed, buffered, to make this PR easier to review.

@cloud-fan - sounds good, #32495 is for the renaming part only. Thanks.

…oin type ### What changes were proposed in this pull request? This is a pre-requisite of #32476, in discussion of #32476 (comment) . This is to refactor sort merge join code-gen to depend on streamed/buffered terminology, which makes the code-gen agnostic to different join types and can be extended to support other join types than inner join. ### Why are the changes needed? Pre-requisite of #32476. ### Does this PR introduce _any_ user-facing change? No. ### How was this patch tested? Existing unit test in `InnerJoinSuite.scala` for inner join code-gen. Closes #32495 from c21/smj-refactor. Authored-by: Cheng Su <chengsu@fb.com> Signed-off-by: Takeshi Yamamuro <yamamuro@apache.org>

Co-authored-by: Chen Li <meloli87@gmail.com>

c21 · 2021-05-11T06:12:10Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

@@ -501,7 +538,7 @@ case class SortMergeJoinExec(
      ctx: CodegenContext,
      streamedRow: String): (Seq[ExprCode], Seq[String]) = {
    ctx.INPUT_ROW = streamedRow
-    left.output.zipWithIndex.map { case (a, i) =>
+    streamedPlan.output.zipWithIndex.map { case (a, i) =>


sorry forgot to change this in #32495, fix it now here. cc @maropu.

c21 · 2021-05-11T06:13:38Z

@maropu, @cloud-fan - This PR is rebased on top of #32495, and ready for review now, thanks.

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

SparkQA · 2021-05-11T07:11:39Z

Kubernetes integration test starting
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42883/

SparkQA · 2021-05-11T07:11:41Z

Kubernetes integration test status failure
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42883/

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

SparkQA · 2021-05-11T10:09:56Z

Kubernetes integration test starting
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42897/

SparkQA · 2021-05-11T10:09:57Z

Kubernetes integration test status failure
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42897/

SparkQA · 2021-05-11T10:38:25Z

Test build #138360 has finished for PR 32476 at commit 44b210f.

This patch passes all tests.
This patch merges cleanly.
This patch adds no public classes.

cloud-fan · 2021-05-11T12:41:53Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

+    //            all matched rows into `matches`. Return true when getting all matched rows.
+    //            For `streamedRow` without `matches` (`handleStreamedWithoutMatch`):
+    //            1. Inner join: skip the row.
+    //            2. Left/Right Outer join: keep the row and return false.


keep the row and return false (with matches being empty)

@cloud-fan - updated.

cloud-fan · 2021-05-11T12:43:01Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

+    //            2. Left/Right Outer join: clear the previous `matches` if needed, keep the row,
+    //                                      and return false.
+    //  - Step 2: Find the `matches` from buffered side having same join keys with `streamedRow`.
+    //            If previous `matches` is not empty, check the join keys and clear the `matches`


nit: we can simply say Clear matches if we hit a new streamedRow, as we need to find new matches.

@cloud-fan - updated.

SparkQA · 2021-05-11T13:44:13Z

Test build #138374 has finished for PR 32476 at commit 765b247.

This patch passes all tests.
This patch merges cleanly.
This patch adds no public classes.

SparkQA · 2021-05-11T23:08:05Z

Kubernetes integration test starting
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42929/

SparkQA · 2021-05-11T23:13:27Z

Kubernetes integration test status failure
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42929/

SparkQA · 2021-05-12T03:08:22Z

Test build #138407 has finished for PR 32476 at commit 617f89c.

This patch passes all tests.
This patch merges cleanly.
This patch adds no public classes.

cloud-fan · 2021-05-12T06:34:52Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

+    // The function has the following step:
+    //  - Step 1: Find the next `streamedRow` with non-null join keys.
+    //            For `streamedRow` with null join keys (`handleStreamedAnyNull`):
+    //            1. Inner join: skip the row.


null join keys is also kind of a new streamRow, so ideally we should clear matches for inner join as well. It's ok because the matches will be cleared when hitting the next streamedRow without null join keys.

How about 1. Inner join: skip the row. matches will be cleared later when hitting the next streamedRow with non-null join keys.

@cloud-fan - sure, updated the comment.

cloud-fan · 2021-05-12T06:42:35Z

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala

+         |  ${streamedVarDecl.mkString("\n")}
+         |  ${beforeLoop.trim}
+         |  scala.collection.Iterator<UnsafeRow> $iterator = $matches.generateIterator();
+         |  boolean $foundMatch = false;


This name is a bit confusing, as we will set it to true even if there is no match. How about boolean firstIteration = true;?

Discussed offline for the naming. The variable is to indicate whether this streamed row has output row or not. So renamed to hasOutputRow.

SparkQA · 2021-05-12T08:22:23Z

Kubernetes integration test starting
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42957/

SparkQA · 2021-05-12T08:22:24Z

Kubernetes integration test status failure
URL: https://amplab.cs.berkeley.edu/jenkins/job/SparkPullRequestBuilder-K8s/42957/

SparkQA · 2021-05-12T12:25:24Z

Test build #138436 has finished for PR 32476 at commit 429edcc.

This patch passes all tests.
This patch merges cleanly.
This patch adds no public classes.

cloud-fan · 2021-05-12T14:10:14Z

thanks, merging to master!

c21 · 2021-05-12T18:31:01Z

Thank you @cloud-fan and @maropu for review!

maropu · 2021-05-19T12:01:07Z

NOTE: This fix improved TPCDS(sf=20) q78: 160883ms => 143171ms. Nice.

c21 · 2021-05-19T15:48:19Z

@maropu - thanks for heads up, this is great to know!

github-actions bot added the SQL label May 8, 2021

c21 force-pushed the smj-outer-codegen branch from 626d01f to 95e56b8 Compare May 8, 2021 07:37

maropu reviewed May 9, 2021

View reviewed changes

maropu reviewed May 10, 2021

View reviewed changes

c21 mentioned this pull request May 10, 2021

[SPARK-35363][SQL] Refactor sort merge join code-gen be agnostic to join type #32495

Closed

c21 and others added 3 commits May 10, 2021 22:59

Add codegen for left/right outer sort merge join

982af12

Co-authored-by: Chen Li <meloli87@gmail.com>

Fix unit test failures

4a57664

Fix more unit test failure

44b210f

c21 force-pushed the smj-outer-codegen branch from 13ffa8a to 44b210f Compare May 11, 2021 06:10

c21 commented May 11, 2021

View reviewed changes

cloud-fan reviewed May 11, 2021

View reviewed changes

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala Show resolved Hide resolved

Add comment for findNextJoinRows

765b247

cloud-fan reviewed May 11, 2021

View reviewed changes

sql/core/src/main/scala/org/apache/spark/sql/execution/joins/SortMergeJoinExec.scala Show resolved Hide resolved

cloud-fan reviewed May 11, 2021

View reviewed changes

Address all comments

617f89c

cloud-fan reviewed May 12, 2021

View reviewed changes

Address all comments

429edcc

cloud-fan approved these changes May 12, 2021

View reviewed changes

cloud-fan closed this in 7bcaded May 12, 2021

c21 deleted the smj-outer-codegen branch May 12, 2021 18:31

		throw new IllegalArgumentException(
		s"SortMergeJoin.streamedPlan/bufferedPlan should not take $x as the JoinType")

[SPARK-35349][SQL] Add code-gen for left/right outer sort merge join #32476

[SPARK-35349][SQL] Add code-gen for left/right outer sort merge join #32476

Conversation

c21 commented May 8, 2021

What changes were proposed in this pull request?

Why are the changes needed?

Does this PR introduce any user-facing change?

How was this patch tested?

c21 commented May 8, 2021

SparkQA commented May 8, 2021

SparkQA commented May 8, 2021

SparkQA commented May 8, 2021

SparkQA commented May 9, 2021

SparkQA commented May 9, 2021

SparkQA commented May 9, 2021

SparkQA commented May 9, 2021

SparkQA commented May 9, 2021

SparkQA commented May 9, 2021

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

maropu May 10, 2021 • edited

Choose a reason for hiding this comment

c21 May 10, 2021 • edited

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

maropu commented May 10, 2021

c21 commented May 10, 2021

maropu commented May 10, 2021

cloud-fan commented May 10, 2021

c21 commented May 10, 2021

c21 May 11, 2021 • edited

Choose a reason for hiding this comment

c21 commented May 11, 2021

SparkQA commented May 11, 2021

SparkQA commented May 11, 2021

SparkQA commented May 11, 2021

SparkQA commented May 11, 2021

SparkQA commented May 11, 2021

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

SparkQA commented May 11, 2021

SparkQA commented May 11, 2021

SparkQA commented May 11, 2021

SparkQA commented May 12, 2021

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

Choose a reason for hiding this comment

SparkQA commented May 12, 2021

SparkQA commented May 12, 2021

SparkQA commented May 12, 2021

cloud-fan commented May 12, 2021

c21 commented May 12, 2021

maropu commented May 19, 2021

c21 commented May 19, 2021

maropu May 10, 2021 •

edited

c21 May 10, 2021 •

edited

c21 May 11, 2021 •

edited