<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:taxo="http://purl.org/rss/1.0/modules/taxonomy/" version="2.0">
  <channel>
    <title>topic Re: Iteration - Pyspark vs Pandas in Data Engineering</title>
    <link>https://community.databricks.com/t5/data-engineering/iteration-pyspark-vs-pandas/m-p/9005#M4506</link>
    <description>&lt;P&gt;Hi @ELENI GEORGOUSI​&amp;nbsp;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Hope everything is going great.&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Just wanted to check in if you were able to resolve your issue. If yes, would you be happy to mark an answer as best so that other members can find the solution more quickly? If not, please tell us so we can help you.&amp;nbsp;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Cheers!&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;</description>
    <pubDate>Sat, 22 Apr 2023 07:11:56 GMT</pubDate>
    <dc:creator>Anonymous</dc:creator>
    <dc:date>2023-04-22T07:11:56Z</dc:date>
    <item>
      <title>Iteration - Pyspark vs Pandas</title>
      <link>https://community.databricks.com/t5/data-engineering/iteration-pyspark-vs-pandas/m-p/9003#M4504</link>
      <description>&lt;P&gt;Hello. Could someone please explain why iteration over a Pyspark dataframe is way slower than over a Pandas dataframe?&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;&lt;B&gt;Pyspark&lt;/B&gt;&lt;/P&gt;&lt;P&gt;df_list = df.collect()&lt;/P&gt;&lt;P&gt;for index in range(0, len(df_list )):&lt;/P&gt;&lt;P&gt;.....&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;&lt;B&gt;Pandas&lt;/B&gt;&lt;/P&gt;&lt;P&gt;df_pnd = df.toPandas()&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;&lt;/P&gt;&lt;P&gt;for index, row in df_pnd.iterrows():&lt;/P&gt;&lt;P&gt;....&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Thank you in advance&lt;/P&gt;</description>
      <pubDate>Tue, 21 Feb 2023 11:21:41 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/iteration-pyspark-vs-pandas/m-p/9003#M4504</guid>
      <dc:creator>elgeo</dc:creator>
      <dc:date>2023-02-21T11:21:41Z</dc:date>
    </item>
    <item>
      <title>Re: Iteration - Pyspark vs Pandas</title>
      <link>https://community.databricks.com/t5/data-engineering/iteration-pyspark-vs-pandas/m-p/9005#M4506</link>
      <description>&lt;P&gt;Hi @ELENI GEORGOUSI​&amp;nbsp;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Hope everything is going great.&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Just wanted to check in if you were able to resolve your issue. If yes, would you be happy to mark an answer as best so that other members can find the solution more quickly? If not, please tell us so we can help you.&amp;nbsp;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Cheers!&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;</description>
      <pubDate>Sat, 22 Apr 2023 07:11:56 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/iteration-pyspark-vs-pandas/m-p/9005#M4506</guid>
      <dc:creator>Anonymous</dc:creator>
      <dc:date>2023-04-22T07:11:56Z</dc:date>
    </item>
  </channel>
</rss>

