[{"data":1,"prerenderedAt":60},["ShallowReactive",2],{"/2026/01/24-rename-pyspark-result-file":3},{"id":4,"title":5,"body":6,"date":51,"description":48,"extension":52,"meta":53,"navigation":54,"path":55,"robots":56,"seo":57,"stem":58,"__hash__":59},"posts/2026/01/24 rename-PySpark-result-file.md","24 Rename PySpark Result File",{"type":7,"value":8,"toc":47},"minimark",[9,18,25,28,31,33],[10,11,13],"post-title",{":date":12},"date",[14,15,17],"h1",{"id":16},"rename-pyspark-result-file","Rename PySpark Result File",[19,20,21],"notes",{},[22,23,24],"p",{},"This is a repost from my old blog. First posted in 9/5/2024.",[26,27],"br",{},[22,29,30],{},"Due to the distributed nature of Apache Spark, when writing result, we can't specify name for the result file. This makes the result file hard to predict which I need for my process orchestration. In my case, I need to write the result to S3 and I finally found a way to do this within a reasonable amount of time by utilizing aws wrangler, Panda, and optionally Arrow. I basically feed Spark dataframe to aws wrangler and have it write to S3 using a specific name.",[26,32],{},[22,34,35,36],{},"Here's link to my sample: ",[37,38,41],"span",{"className":39},[40],"text-blue-600",[42,43,44],"a",{"href":44,"rel":45},"https://github.com/nik-yo/PySparkFilename",[46],"nofollow",{"title":48,"searchDepth":49,"depth":49,"links":50},"",2,[],"2026-01-24T00:00:00.000Z","md",{},true,"/2026/01/24-rename-pyspark-result-file",null,{"title":5,"description":48},"2026/01/24 rename-PySpark-result-file","jwfSY8FuI_UZuKMjYO6I3j9WjLBQj2yTfcQE7fgw8ys",1785167451690]