[SPARK-21167][SS] Decode the path generated by File sink to handle special characters

zsxwing · zsxwing · commit 6ef7a5bd32a4 · 2017-06-21T23:43:30.000-07:00
## What changes were proposed in this pull request? Decode the path generated by File sink to handle special characters. ## How was this patch tested? The added unit test. Author: Shixiong Zhu <shixiong@databricks.com> Closes apache#18381 from zsxwing/SPARK-21167. (cherry picked from commit d66b143) Signed-off-by: Shixiong Zhu <shixiong@databricks.com>
diff --git a/sql/core/src/main/scala/org/apache/spark/sql/execution/streaming/FileStreamSinkLog.scala b/sql/core/src/main/scala/org/apache/spark/sql/execution/streaming/FileStreamSinkLog.scala
@@ -17,6 +17,8 @@
 
 package org.apache.spark.sql.execution.streaming
 
+import java.net.URI
+
 import org.apache.hadoop.fs.{FileStatus, Path}
 import org.json4s.NoTypeHints
 import org.json4s.jackson.Serialization
@@ -47,7 +49,8 @@ case class SinkFileStatus(
     action: String) {
 
   def toFileStatus: FileStatus = {
-    new FileStatus(size, isDir, blockReplication, blockSize, modificationTime, new Path(path))
+    new FileStatus(
+      size, isDir, blockReplication, blockSize, modificationTime, new Path(new URI(path)))
   }
 }
 
diff --git a/sql/core/src/test/scala/org/apache/spark/sql/streaming/FileStreamSinkSuite.scala b/sql/core/src/test/scala/org/apache/spark/sql/streaming/FileStreamSinkSuite.scala
@@ -64,6 +64,35 @@ class FileStreamSinkSuite extends StreamTest {
     }
   }
 
+  test("SPARK-21167: encode and decode path correctly") {
+    val inputData = MemoryStream[String]
+    val ds = inputData.toDS()
+
+    val outputDir = Utils.createTempDir(namePrefix = "stream.output").getCanonicalPath
+    val checkpointDir = Utils.createTempDir(namePrefix = "stream.checkpoint").getCanonicalPath
+
+    val query = ds.map(s => (s, s.length))
+      .toDF("value", "len")
+      .writeStream
+      .partitionBy("value")
+      .option("checkpointLocation", checkpointDir)
+      .format("parquet")
+      .start(outputDir)
+
+    try {
+      // The output is partitoned by "value", so the value will appear in the file path.
+      // This is to test if we handle spaces in the path correctly.
+      inputData.addData("hello world")
+      failAfter(streamingTimeout) {
+        query.processAllAvailable()
+      }
+      val outputDf = spark.read.parquet(outputDir)
+      checkDatasetUnorderly(outputDf.as[(Int, String)], ("hello world".length, "hello world"))
+    } finally {
+      query.stop()
+    }
+  }
+
   test("partitioned writing and batch reading") {
     val inputData = MemoryStream[Int]
     val ds = inputData.toDS()

Original file line number	Diff line number	Diff line change
`@@ -17,6 +17,8 @@`
`17`	`17`
`18`	`18`	`package org.apache.spark.sql.execution.streaming`
`19`	`19`
	`20`	`+import java.net.URI`
	`21`	`+`
`20`	`22`	`import org.apache.hadoop.fs.{FileStatus, Path}`
`21`	`23`	`import org.json4s.NoTypeHints`
`22`	`24`	`import org.json4s.jackson.Serialization`
`@@ -47,7 +49,8 @@ case class SinkFileStatus(`
`47`	`49`	`action: String) {`
`48`	`50`
`49`	`51`	`def toFileStatus: FileStatus = {`
`50`		`- new FileStatus(size, isDir, blockReplication, blockSize, modificationTime, new Path(path))`
	`52`	`+ new FileStatus(`
	`53`	`+ size, isDir, blockReplication, blockSize, modificationTime, new Path(new URI(path)))`
`51`	`54`	`}`
`52`	`55`	`}`
`53`	`56`