<?xml version="1.0" encoding="UTF-8"?>
<rss xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:rdf="http://www.w3.org/1999/02/22-rdf-syntax-ns#" xmlns:taxo="http://purl.org/rss/1.0/modules/taxonomy/" version="2.0">
  <channel>
    <title>topic Re: Unable to read an XML file of 9 GB in Data Engineering</title>
    <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23079#M15893</link>
    <description>&lt;P&gt;Hi @Salah K.​&amp;nbsp; : What is the cluster size / configuration? pls share your code snippet. &lt;/P&gt;</description>
    <pubDate>Wed, 13 Apr 2022 04:12:15 GMT</pubDate>
    <dc:creator>RKNutalapati</dc:creator>
    <dc:date>2022-04-13T04:12:15Z</dc:date>
    <item>
      <title>Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23078#M15892</link>
      <description>&lt;P&gt;Hello,&lt;/P&gt;&lt;P&gt;We have a large XML file (9 GB) that we can't read.&lt;/P&gt;&lt;P&gt;We have this error : VM size limit&lt;/P&gt;&lt;P&gt;But how can we change the VM size limit ?&lt;/P&gt;&lt;P&gt;We have tested many clusters, but no one can read this file.&lt;/P&gt;&lt;P&gt;Thank you for your help.&lt;/P&gt;</description>
      <pubDate>Tue, 12 Apr 2022 12:12:10 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23078#M15892</guid>
      <dc:creator>wyzer</dc:creator>
      <dc:date>2022-04-12T12:12:10Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23079#M15893</link>
      <description>&lt;P&gt;Hi @Salah K.​&amp;nbsp; : What is the cluster size / configuration? pls share your code snippet. &lt;/P&gt;</description>
      <pubDate>Wed, 13 Apr 2022 04:12:15 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23079#M15893</guid>
      <dc:creator>RKNutalapati</dc:creator>
      <dc:date>2022-04-13T04:12:15Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23080#M15894</link>
      <description>&lt;P&gt;{&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"autoscale": {&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;"min_workers": 2,&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;"max_workers": 8&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;},&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"cluster_name": "GrosCluster",&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"spark_version": "10.4.x-scala2.12",&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"spark_conf": {&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;"spark.databricks.delta.preview.enabled": "true"&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;},&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"azure_attributes": {&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;"first_on_demand": 1,&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;"availability": "SPOT_WITH_FALLBACK_AZURE",&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;"spot_bid_max_price": -1&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;},&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"node_type_id": "Standard_L8s",&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"driver_node_type_id": "Standard_L8s",&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"ssh_public_keys": [],&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"custom_tags": {},&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"spark_env_vars": {&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;&amp;nbsp;&amp;nbsp;"PYSPARK_PYTHON": "/databricks/python3/bin/python3"&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;},&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"autotermination_minutes": 120,&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"enable_elastic_disk": true,&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"cluster_source": "UI",&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"init_scripts": [],&lt;/P&gt;&lt;P&gt;&amp;nbsp;&amp;nbsp;"cluster_id": "0408-123105-xj70dm6w"&lt;/P&gt;&lt;P&gt;}&lt;/P&gt;</description>
      <pubDate>Wed, 13 Apr 2022 07:23:37 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23080#M15894</guid>
      <dc:creator>wyzer</dc:creator>
      <dc:date>2022-04-13T07:23:37Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23081#M15895</link>
      <description>&lt;P&gt;@Salah K.​&amp;nbsp;, Did you tried with "Memory-optimized" cluster? My wild guess here is that it is doing a single thread operation and that thread does not have enough memory. Ensure each thread in the cluster has more than 9 GB of memory.&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Is your code has the &lt;B&gt;InferSchema&lt;/B&gt; option enabled?&lt;/P&gt;</description>
      <pubDate>Wed, 13 Apr 2022 08:15:56 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23081#M15895</guid>
      <dc:creator>RKNutalapati</dc:creator>
      <dc:date>2022-04-13T08:15:56Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23082#M15896</link>
      <description>&lt;P&gt;@Rama Krishna N​&amp;nbsp;, Yes we tried the "Memory-optimized" cluster.&lt;/P&gt;&lt;P&gt;And no, we didn't change the thread in the cluster.&lt;/P&gt;&lt;P&gt;How do you do that please?&lt;/P&gt;</description>
      <pubDate>Wed, 13 Apr 2022 08:52:38 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23082#M15896</guid>
      <dc:creator>wyzer</dc:creator>
      <dc:date>2022-04-13T08:52:38Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23084#M15898</link>
      <description>&lt;P&gt;Hi,&lt;/P&gt;&lt;P&gt;Yes I want to try it, but I don't know how to change the memory thread in the cluster.&lt;/P&gt;</description>
      <pubDate>Wed, 13 Apr 2022 09:03:07 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23084#M15898</guid>
      <dc:creator>wyzer</dc:creator>
      <dc:date>2022-04-13T09:03:07Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23085#M15899</link>
      <description>&lt;P&gt;Hi @Salah K.​&amp;nbsp; - I am sorry for the confusion.  I mean to say use bigger cluster and verify. Something like below. &lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Standard_M8ms&amp;nbsp;3 &lt;/P&gt;&lt;P&gt;&lt;B&gt;vCPU    = 8&lt;/B&gt;&lt;/P&gt;&lt;P&gt;&lt;B&gt;Memory: GiB = 218 &lt;/B&gt;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;&lt;A href="https://docs.microsoft.com/en-us/azure/virtual-machines/m-series" target="test_blank"&gt;https://docs.microsoft.com/en-us/azure/virtual-machines/m-series&lt;/A&gt;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;I am not that familiar with Azure. I am also doing xml parsing in AWS workspace. But the files I am loading are not this huge.&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;val df = spark.&lt;/P&gt;&lt;P&gt;read.format("com.databricks.spark.xml")&lt;/P&gt;&lt;P&gt;.option("rowTag", "&amp;lt;MyRowTag&amp;gt;")&amp;nbsp;&amp;nbsp;&lt;/P&gt;&lt;P&gt;.option("rootTag", "&amp;lt;MyRoootTag&amp;gt;")&lt;/P&gt;&lt;P&gt;.load("&amp;lt;XML File Path&amp;gt;")&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;&lt;B&gt;Turn off the below option, if it is true.&lt;/B&gt;&lt;/P&gt;&lt;P&gt;&lt;B&gt;.option("inferschema", "false")&lt;/B&gt;&lt;/P&gt;</description>
      <pubDate>Wed, 13 Apr 2022 09:54:33 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23085#M15899</guid>
      <dc:creator>RKNutalapati</dc:creator>
      <dc:date>2022-04-13T09:54:33Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23086#M15900</link>
      <description>&lt;P&gt;hello @Salah K.​&amp;nbsp;you can try configuring spark.executor.memory from cluster spark configuration.&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;total_executor_memory =  (total_ram_per_node -1) / executor_per_node&lt;/P&gt;&lt;P&gt;total_executor_memory = (64–1)/3 = 21(rounded down)&lt;/P&gt;&lt;P&gt;spark.executor.memory = total_executor_memory * 0.9&lt;/P&gt;&lt;P&gt;spark.executor.memory = 21*0.9 = 18 (rounded down)&lt;/P&gt;&lt;P&gt;memory_overhead = 21*0.1 = 3 (rounded up)&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;&lt;/P&gt;</description>
      <pubDate>Sat, 21 May 2022 13:48:26 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23086#M15900</guid>
      <dc:creator>Atanu</dc:creator>
      <dc:date>2022-05-21T13:48:26Z</dc:date>
    </item>
    <item>
      <title>Re: Unable to read an XML file of 9 GB</title>
      <link>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23087#M15901</link>
      <description>&lt;P&gt;Hi @Salah K.​,&lt;/P&gt;&lt;P&gt;&lt;/P&gt;&lt;P&gt;Just a friendly follow-up. Did any of the responses help you to resolve your question? if it did, please mark it as best. Otherwise, please let us know if you still need help.&lt;/P&gt;</description>
      <pubDate>Mon, 25 Jul 2022 21:14:39 GMT</pubDate>
      <guid>https://community.databricks.com/t5/data-engineering/unable-to-read-an-xml-file-of-9-gb/m-p/23087#M15901</guid>
      <dc:creator>jose_gonzalez</dc:creator>
      <dc:date>2022-07-25T21:14:39Z</dc:date>
    </item>
  </channel>
</rss>

