') + ')', 'gi'); if (regex.test(text)) { found = true; var frag = document.createDocumentFragment(); var parts = text.split(regex); parts.forEach(function(part, i) { if (i % 2 === 0) { frag.appendChild(document.createTextNode(part)); } else { var span = document.createElement('span'); span.className = 'userscript-highlight'; span.textContent = part; frag.appendChild(span); } }); node.parentNode.replaceChild(frag, node); } }); } else if (node.nodeType === 1 && node.childNodes) { // element var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT']; if (!skipTags.includes(node.tagName)) { Array.from(node.childNodes).forEach(highlight); } } } highlight(document.body); // Re-highlight on dynamic content var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1 || node.nodeType === 3) highlight(node); }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Strip utm_, fbclid, gclid, etc. from all links on page (function() { var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content', 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid', 'ref', 'ref_src', 'source', 'medium', 'campaign']; function cleanUrl(url) { try { var u = new URL(url, window.location.origin); var changed = false; trackingParams.forEach(function(p) { if (u.searchParams.has(p)) { u.searchParams.delete(p); changed = true; } }); return changed ? u.toString() : url; } catch (e) { return url; } } function cleanLinks() { document.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } cleanLinks(); var observer = new MutationObserver(function(mutations) { mutations.forEach(function(m) { m.addedNodes.forEach(function(node) { if (node.nodeType === 1) { if (node.tagName === 'A') cleanLinks(); node.querySelectorAll('a[href]').forEach(function(a) { var clean = cleanUrl(a.href); if (clean !== a.href) a.href = clean; }); } }); }); }); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Auto-enable theater mode on YouTube (function() { function tryTheater() { var btn = document.querySelector('button[aria-label="Theater mode"], ytd-player #player button[title="Theater mode"]'); if (btn && !btn.classList.contains('activated')) { btn.click(); } } // Try immediately tryTheater(); // Try after navigation (SPA) var lastUrl = location.href; setInterval(function() { if (location.href !== lastUrl) { lastUrl = location.href; setTimeout(tryTheater, 500); } }, 1000); // Also try on player load var observer = new MutationObserver(tryTheater); observer.observe(document.body, { childList: true, subtree: true }); })(); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + ', 'i'); if (__m === '*' || __re.test(location.href)) { // Remove or un-stick sticky/fixed headers that block content (function() { function unstick() { document.querySelectorAll('header, nav, [role="banner"], .header, .navbar, .sticky, .fixed-top, [style*="position: fixed"], [style*="position:sticky"]').forEach(function(el) { if (el.style.position === 'fixed' || el.style.position === 'sticky' || getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') { el.style.position = 'static'; el.style.top = 'auto'; el.style.zIndex = 'auto'; } }); } unstick(); var observer = new MutationObserver(unstick); observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] }); })(); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); })(); SpatialData repr breaks when Dask auto-partitions Parquet points · Issue #1084 · scverse/spatialdata · GitHub
Skip to content

SpatialData repr breaks when Dask auto-partitions Parquet points #1084

Description

@enric-bazz

Hi,
I'm encountering an issue when calling repr(sdata) which fails during the self-contained check for points elements backed by a Parquet file with the error:

AttributeError: 'list' object has no attribute 'values'

The failure originates in
_search_for_backing_files_recursively()
where the code assumes that each parquet-read task in the Dask graph has a dict in task.args[0] and therefore calls:

v.args[0].values()

However, when dask.dataframe.read_parquet() performs automatic partition aggregation, controlled by split_row_groups='infer' (default), task.args[0] can become a list of row-group dicts instead of a single dict.

Reproduce

To reproduce the error, first trigger Dask’s autopartitioning by forcing a DataFrame into a single partition, writing it to a Parquet file and reading it back with default read_parquet settings. Inspecting the graph reveals the list-of-dicts structure that breaks SpatialData. To trigger the original error, parse the DataFrame with PointsModel, build a SpatialData object, write it to a Zarr store (the error does not occur if unbacked), and finally call repr(sdata). The pseudocode is the following:

importdask.dataframeasddfromspatialdata._core.pointsimportPointsModelfromspatialdata._core.spatialdataimportSpatialData# 1. Create or load a Dask DataFramedf=dd.from_pandas(some_pandas_df, npartitions=4)
# 2. Force a single partitiondf_one_part=df.repartition(npartitions=1)
# 3. Write single-partition Parquetdf_one_part.to_parquet("example_points.parquet")
# 4. Read Parquet backdf_read=dd.read_parquet("example_points.parquet")
# 5. Inspect graph (optional)print(df_read.dask) # shows list-of-dicts if autopartitioning changed structure# 6. Parse with PointsModelpoints=PointsModel.parse(df_read)
# 7. Build SpatialData objectsdata=SpatialData(points=points)
# 8. Write to Zarr store (error triggers only when sdata.is_backed() == True)sdata.write("example.zarr")
# 9. Trigger error with reprprint(repr(sdata))

Metadata

Metadata

Assignees

No one assigned

    Labels

    No labels
    No labels

    Type

    No type

    Projects

    No projects

    Milestone

    No milestone

    Relationships

    None yet

    Development

    No branches or pull requests

    Issue actions