Skip to content
Merged

Fixes #201

Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
73 commits
Select commit Hold shift + click to select a range
f80b261
Register cellposev4 in benchmark run scripts
dariarom94 Jul 19, 2026
1a2fa09
fix anndata version mismatch with txsim
dariarom94 Jul 19, 2026
82add80
add segger to workflow (test)
dariarom94 Jul 19, 2026
53e1728
duplicates when FOV stiching cleaned up
dariarom94 Jul 19, 2026
1186b7a
chunks issue atera
dariarom94 Jul 20, 2026
18644d7
segger update image
dariarom94 Jul 20, 2026
ecb302d
claude fix for segger
dariarom94 Jul 20, 2026
7d66898
Merge branch 'main' into fixes
dariarom94 Jul 20, 2026
d400ebe
atera version fix
dariarom94 Jul 20, 2026
64d7b4e
wf for the custom rnaseq scripts
dariarom94 Jul 20, 2026
3edfbf1
adjust the loader image name
dariarom94 Jul 20, 2026
cbd2f12
adjust the memory
dariarom94 Jul 20, 2026
184260e
troubleshootig edges
dariarom94 Jul 20, 2026
9fa9a33
Merge branch 'main' into fixes
dariarom94 Jul 20, 2026
36631c4
segger update
dariarom94 Jul 21, 2026
0626127
cell type label correction
dariarom94 Jul 21, 2026
3186435
fix boundaries
dariarom94 Jul 21, 2026
d8a7d93
Merge branch 'main' into fixes
dariarom94 Jul 21, 2026
3505718
OOM fixes
dariarom94 Jul 21, 2026
d6e110a
fix code
dariarom94 Jul 21, 2026
4660f26
RCTD
dariarom94 Jul 21, 2026
5abd651
segger to RAPIDS
dariarom94 Jul 21, 2026
fe2e90a
Merge branch 'main' into fixes
dariarom94 Jul 21, 2026
0b23474
fix rctd
dariarom94 Jul 22, 2026
196ff1f
segger debug (torchvision)
dariarom94 Jul 22, 2026
4be7bd4
Merge branch 'main' into fixes
dariarom94 Jul 22, 2026
b8d3d7b
save the xenium version
dariarom94 Jul 22, 2026
202ac49
add atera to datasets
dariarom94 Jul 22, 2026
14be8d0
Add gene efficiency correction as a separate pipeline stage (#183)
dariarom94 Jul 22, 2026
0cf0243
moscot to pca and segger troubleshooting
dariarom94 Jul 22, 2026
d7afb84
added fastreseg
dariarom94 Jul 23, 2026
f87a1d9
segger bug new fix
dariarom94 Jul 23, 2026
123e112
fastreseg to workflow
dariarom94 Jul 23, 2026
7549589
add fastreseg test
dariarom94 Jul 23, 2026
9fa9604
Merge branch 'main' into fixes
dariarom94 Jul 23, 2026
ff04467
optimized fastreseg build
dariarom94 Jul 23, 2026
7ffc514
Merge branch 'main' into fixes
dariarom94 Jul 23, 2026
a7404d8
add s3 paths
dariarom94 Jul 23, 2026
19e5f83
troubleshoot comseg/segger
dariarom94 Jul 24, 2026
aaca151
segger update
dariarom94 Jul 25, 2026
c9bdb91
data loader bug
dariarom94 Jul 25, 2026
1df9834
Merge branch 'main' into fixes
dariarom94 Jul 25, 2026
4467d32
rctd adjustment (raw counts)
dariarom94 Jul 26, 2026
58912e4
fix segger and comseg
dariarom94 Jul 26, 2026
0a5aa99
optimize cosmx
dariarom94 Jul 26, 2026
298e666
Merge branch 'main' into fixes
dariarom94 Jul 26, 2026
c921937
parameter test for cellpose4
dariarom94 Jul 26, 2026
57c2d79
add atera
dariarom94 Jul 26, 2026
cf67e09
add a test in pciseq and dynamic memory for bruker
dariarom94 Jul 27, 2026
9f71692
add test to vizgen data
dariarom94 Jul 28, 2026
d795f33
Merge branch 'main' into fixes
dariarom94 Jul 28, 2026
fefaadc
param sweep
dariarom94 Jul 28, 2026
4dc08d3
add params to segmentation
dariarom94 Jul 28, 2026
e9505f1
adjust segger mem
dariarom94 Jul 29, 2026
3a155be
update fastreseg to tacco
dariarom94 Jul 29, 2026
acdd6c7
Add annotation + expression-correction parameter sweeps (rctd, ssam, …
dariarom94 Jul 30, 2026
fa462b8
Add moscot + split parameter sweeps (annotation, expression correction)
dariarom94 Jul 30, 2026
fd37aee
singler: read par['celltype_key'] instead of hardcoding "cell_type"
dariarom94 Jul 30, 2026
3449bd0
fastreseg
dariarom94 Jul 30, 2026
331d219
Merge branch 'main' into fixes
dariarom94 Jul 30, 2026
5961d0a
adjust labels
dariarom94 Jul 30, 2026
857e16a
bruker nsclc
dariarom94 Jul 31, 2026
54f023e
adjust bruker nsclc loader
dariarom94 Jul 31, 2026
6e8b9ea
setup
dariarom94 Jul 31, 2026
4eec493
Merge branch 'main' into fixes
dariarom94 Jul 31, 2026
44ad7af
method correction
dariarom94 Jul 31, 2026
b351148
Merge branch 'main' into fixes
dariarom94 Jul 31, 2026
2b8177c
pin anndata
dariarom94 Aug 1, 2026
6c16707
mirror nsclc
dariarom94 Aug 1, 2026
5618fb8
sync the vizgen files
dariarom94 Aug 2, 2026
b68c100
adjust mem for allen brain
dariarom94 Aug 2, 2026
98dbc69
claude notes
dariarom94 Aug 2, 2026
fa23294
claude notes
dariarom94 Aug 2, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
14 changes: 7 additions & 7 deletions scripts/create_resources/spatial/mirror_bruker_to_s3.sh
Original file line number Diff line number Diff line change
Expand Up @@ -19,7 +19,7 @@
# SCRATCH_DIR=/path/to/big/scratch ./mirror_bruker_to_s3.sh

set -euo pipefail

SCRATCH_DIR="/Volumes/SeagateHHD"
# --- Config -----------------------------------------------------------------

SRC_BASE="https://smi-public.objects.liquidweb.services"
Expand All @@ -33,11 +33,11 @@ SCRATCH_DIR="${SCRATCH_DIR:-$PWD/bruker_mirror_scratch}"
# - SOURCE is either a full https URL, or a name relative to SRC_BASE (the mouse/liver host).
# The URL-encoded names are what the liquidweb server serves.
# - LOCAL_NAME is the (decoded) name to store under on S3.
FILES=(
"HalfBrain.zip|HalfBrain.zip"
"Half%20%20Brain%20simple%20%20files%20.zip|Half Brain simple files.zip"
"NormalLiverFiles.zip|NormalLiverFiles.zip"
)
#FILES=(
# "HalfBrain.zip|HalfBrain.zip"
# "Half%20%20Brain%20simple%20%20files%20.zip|Half Brain simple files.zip"
# "NormalLiverFiles.zip|NormalLiverFiles.zip"
#)

# NSCLC lung-cancer samples: each ships a flat-files+cell-labels archive and a
# raw-morphology-images archive. The bruker_cosmx_nsclc loader streams both from S3.
Expand Down Expand Up @@ -132,7 +132,7 @@ for entry in "${FILES[@]}"; do

# Upload to S3
echo " Uploading to s3://$bucket/$key ..."
aws s3 cp "$local_path" "s3://$bucket/$key"
aws s3 cp "$local_path" "s3://$bucket/$key" --profile op

# Free scratch space before the next (much larger) file
echo " Removing local copy to free space"
Expand Down
185 changes: 185 additions & 0 deletions scripts/create_resources/spatial/upload_vizgen_merscope_2d.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,185 @@
#!/bin/bash

# Mirror a z3-only (2D) copy of the Vizgen MERSCOPE FFPE showcase raw data from Google
# Cloud Storage to S3, so the Nebius/Seqera benchmark can stage its inputs without any
# Google Cloud credentials.
#
# WHY THIS EXISTS
# process_vizgen_merscope_nebius.sh used to point `input:` directly at
# gs://vz-ffpe-showcase/<Patient>. That bucket is Vizgen's access-controlled
# data-release bucket (NOT public: an unauthenticated GET returns
# "Anonymous caller does not have storage.objects.get access ..."). The Nebius compute
# env has no GCP credentials, so Nextflow staged the input as an anonymous caller and
# the run died before the loader ran. Mirroring the data to s3://openproblems-data
# (which the Nebius env already reads for every other spatial dataset) removes the
# GCP-auth dependency from every future run.
#
# WHAT IS KEPT (lossless for this 2D pipeline)
# The vizgen_merscope loader calls spatialdata_io.merscope(..., z_layers=3), i.e. it
# only ever reads z-plane 3 of the mosaic images. So we mirror ONLY *_z3.tif and drop
# every *_z{0,1,2,4,5,6}.tif (7 focal planes -> 1). Everything else is copied verbatim:
# - cell_boundaries/*.hdf5 (OLD showcase format; the loader rebuilds
# cell_boundaries.parquet from these at run time)
# - images/mosaic_DAPI_z3.tif (DAPI only, by default — the loader keeps ONLY the DAPI
# channel: `sdata["morphology_mip"].sel(c=["DAPI"])`, so
# dropping PolyT/Cellbound stains is lossless for this
# pipeline and cuts the image bytes to ~1/5. If the
# keep-more-stains TODO in the loader is ever done, re-run
# with KEEP_ONLY_DAPI=0 to mirror all stains at z3.)
# - images/micron_to_mosaic_pixel_transform.csv, images/manifest.json
# - cell_by_gene.csv, cell_metadata.csv, detected_transcripts.csv
# KEEP_ONLY_DAPI defaults to 1 (DAPI-only). Set KEEP_ONLY_DAPI=0 to keep all stains at z3.
#
# PARALLEL
# Each patient ships ~1100-2500 tiny cell_boundaries/*.hdf5 files (~15k total across the 7
# patients). Streaming them one-at-a-time is dominated by per-file spawn/handshake overhead
# (~15 s/file => ~2-3 DAYS). This script runs the transfers through a pool of $JOBS parallel
# workers (default 16), which hides that overhead and cuts the whole mirror to a few hours.
# Big z3 tifs / detected_transcripts.csv are bandwidth-bound, so a handful running
# concurrently just keeps the uplink saturated. Tune with JOBS=<n> (e.g. 32 for the
# mostly-small-file tail).
#
# REQUIREMENTS (run this on a machine that is authenticated to BOTH clouds)
# - gcloud CLI, authenticated with an account granted access to gs://vz-ffpe-showcase
# (Vizgen data-release-program access — request it via https://info.vizgen.com/ffpe-showcase).
# - aws CLI, with a profile that can write s3://openproblems-data (default: profile "op").
# The transfer STREAMS each object gcloud->aws (no local disk staging).
#
# IDEMPOTENT + RESUMABLE
# Objects already in S3 with a matching byte size are skipped (one bulk `aws s3 ls` up
# front, so resuming does not cost a HEAD per object). A partially-copied file is aborted
# by aws and reads back as absent, so it is simply re-sent on the next run. Just re-run.
#
# Usage:
# AWS_PROFILE=op ./upload_vizgen_merscope_2d.sh
# # more workers for the small-file tail:
# JOBS=32 AWS_PROFILE=op ./upload_vizgen_merscope_2d.sh
# # or override anything:
# GCS_BASE=gs://vz-ffpe-showcase \
# S3_DEST=s3://openproblems-data/resources/raw_data/txSim_custom/vizgen_merscope \
# SAMPLES="HumanBreastCancerPatient1 HumanLungCancerPatient1" \
# JOBS=24 AWS_PROFILE=op ./upload_vizgen_merscope_2d.sh

set -euo pipefail

GCS_BASE="${GCS_BASE:-gs://vz-ffpe-showcase}"
S3_DEST="${S3_DEST:-s3://openproblems-data/resources/raw_data/txSim_custom/vizgen_merscope}"
AWS_PROFILE="${AWS_PROFILE:-op}"
Z="${Z:-3}" # which single z-plane to keep
KEEP_ONLY_DAPI="${KEEP_ONLY_DAPI:-1}" # 1 = DAPI-only (default, lossless for this loader); 0 = all stains at z3
JOBS="${JOBS:-12}" # number of concurrent gcloud->aws transfers (each = 2 TLS
# conns; >~16 tends to trigger transient TLS/read-timeout
# errors, which the worker retries with backoff anyway)

# The 7 patients currently active (uncommented) in process_vizgen_merscope_nebius.sh.
# Additional patients (melanoma/ovarian/prostate/uterine) are commented-out there and in
# the s3 sibling process_vizgen_merscope.sh — add their folder names here to mirror them.
SAMPLES=(${SAMPLES:-\
HumanBreastCancerPatient1 \
HumanLiverCancerPatient1 \
HumanLiverCancerPatient2 \
HumanLungCancerPatient1 \
HumanLungCancerPatient2 \
HumanColonCancerPatient1 \
HumanColonCancerPatient2})

export AWS_PROFILE
command -v gcloud >/dev/null || { echo "ERROR: gcloud CLI not found (needed to read gs://vz-ffpe-showcase)" >&2; exit 1; }
command -v aws >/dev/null || { echo "ERROR: aws CLI not found (needed to write $S3_DEST)" >&2; exit 1; }

gcs_bucket="$(echo "$GCS_BASE" | sed -E 's#^gs://([^/]+).*#\1#')"
s3_bucket="$(echo "$S3_DEST" | sed -E 's#^s3://([^/]+)/.*#\1#')"
s3_prefix="$(echo "$S3_DEST" | sed -E 's#^s3://[^/]+/(.*)#\1#')"
export gcs_bucket s3_bucket s3_prefix

# Should this object be dropped from the mirror?
drop_object() {
local rel="$1" base="${1##*/}"
# keep only z-plane $Z: drop any *_z<other>.tif (matches mosaic_*_z*.tif and boundaries_z*.tif)
if [[ "$base" =~ _z([0-9]+)\.tif$ && "${BASH_REMATCH[1]}" != "$Z" ]]; then return 0; fi
# optional: drop non-DAPI stains entirely
if [[ "$KEEP_ONLY_DAPI" == "1" && "$base" =~ ^mosaic_ && ! "$base" =~ ^mosaic_DAPI_ ]]; then return 0; fi
return 1
}

# Worker: stream one "<size>|<rel>" record gcloud->aws. Exported so `xargs -P` can run it in
# parallel `bash -c` subshells. Kept free of bash-4 features so it works on the stock macOS
# /bin/bash 3.2. The already-uploaded files are filtered out up front (see below), so the
# worker just transfers. A failed transfer is logged, not fatal: the batch keeps going and the
# file (absent, since aws aborts a partial multipart) is retried on the next resumable run.
transfer_one() {
local rec="$1"
local size="${rec%%|*}"
local rel="${rec#*|}"
local attempt=0 max=4
# Retry transient GCS/S3 errors (read timeouts, SSL/TLS handshake failures) that show up under
# high concurrency. pipefail (set by the caller) makes a gcloud-side failure fail the pipe too.
while :; do
attempt=$((attempt+1))
if gcloud storage cat "gs://$gcs_bucket/$rel" 2>/dev/null \
| aws s3 cp - "s3://$s3_bucket/$s3_prefix/$rel" --expected-size "$size" --only-show-errors 2>/dev/null; then
printf ' OK %s (%s B)\n' "$rel" "$size"
return 0
fi
if [ "$attempt" -ge "$max" ]; then
printf ' FAIL %s (after %s attempts; retry on re-run)\n' "$rel" "$attempt" >&2
return 0
fi
sleep $((attempt * 3)) # linear backoff: 3s, 6s, 9s
done
}
export -f transfer_one

present="$(mktemp -t viz_present)"
wanted_raw="$(mktemp -t viz_wanted_raw)"
wanted="$(mktemp -t viz_wanted)"
worklist="$(mktemp -t viz_worklist)"
trap 'rm -f "$present" "$wanted_raw" "$wanted" "$worklist"' EXIT

# 1) One bulk listing of what is already in S3 -> "<rel>\t<size>" (sorted). One API call for the
# whole resume state, instead of a HEAD per object.
echo "Listing existing objects under $S3_DEST ..."
aws s3 ls "s3://$s3_bucket/$s3_prefix/" --recursive 2>/dev/null \
| awk -v p="$s3_prefix/" 'BEGIN{OFS="\t"} {k=$4; sub("^"p,"",k); if(k!="") print k,$3}' \
| sort > "$present"
echo " $(wc -l < "$present" | tr -d ' ') objects already in S3."

# 2) Enumerate GCS, apply the keep-rules -> "<rel>\t<size>". Written to a file (not piped) so
# the counters stay in this shell and the progress echoes go to stderr, not into the data.
total=0; dropped=0
for sample in "${SAMPLES[@]}"; do
echo "Enumerating $GCS_BASE/$sample ..." >&2
while IFS= read -r line; do
[[ "$line" == *"gs://"* ]] || continue # skip the TOTAL: summary line
url="$(awk '{print $NF}' <<<"$line")"
size="$(awk '{print $1}' <<<"$line")"
[[ "$url" == */ ]] && continue # skip directory placeholders
[[ "$size" =~ ^[0-9]+$ ]] || continue # skip anything without a numeric size
rel="${url#gs://$gcs_bucket/}"
total=$((total+1))
if drop_object "$rel"; then dropped=$((dropped+1)); continue; fi
printf '%s\t%s\n' "$rel" "$size" >> "$wanted_raw"
done < <(gcloud storage ls -l "$GCS_BASE/$sample/**")
done
sort "$wanted_raw" > "$wanted"

# 3) Work list = wanted objects not already in S3 at the same size (comm on the sorted files).
comm -23 "$wanted" "$present" | awk -F'\t' '{print $2"|"$1}' > "$worklist"
queued="$(wc -l < "$worklist" | tr -d ' ')"
echo "================================================================"
echo "Enumerated $total objects: $dropped dropped, $(( $(wc -l < "$wanted" | tr -d ' ') - queued )) already in S3, $queued to transfer."
echo "Transferring with $JOBS parallel workers ..."

# 4) Run the transfers in parallel. -I REC feeds one record per worker; -P runs $JOBS at once.
# Use the SAME bash that is running this script ($BASH) so the exported transfer_one
# imports correctly regardless of which bash is first on PATH.
if [[ "$queued" -gt 0 ]]; then
xargs -P "$JOBS" -I REC "${BASH:-bash}" -c 'set -o pipefail; transfer_one "$@"' _ REC < "$worklist"
fi

echo "================================================================"
echo "Done. objects=$total dropped=$dropped queued=$queued (JOBS=$JOBS)"
echo "Loader --input paths (use these in process_vizgen_merscope_nebius.sh):"
for sample in "${SAMPLES[@]}"; do
echo " $S3_DEST/$sample"
done
1 change: 1 addition & 0 deletions src/base/labels_nebius.config
Original file line number Diff line number Diff line change
Expand Up @@ -193,6 +193,7 @@ process {

withLabel: hightime { time = 12.h }
withLabel: veryhightime { time = 24.h }
withLabel: veryveryhightime { time = 48.h } // 2 days (e.g. allen_brain_cell_atlas_merfish loader)

// make sure publishstates gets enough disk space and memory
withName:'.*publishStatesProc' {
Expand Down
2 changes: 1 addition & 1 deletion src/base/setup_spatialdata_partial.yaml
Original file line number Diff line number Diff line change
@@ -1,3 +1,3 @@
setup:
- type: python
pypi: ["spatialdata>=0.7.3", "anndata>=0.12.0", "zarr>=3.0.0"]
pypi: ["spatialdata>=0.7.3", "anndata>=0.12.0,<0.13", "zarr>=3.0.0"]
Original file line number Diff line number Diff line change
Expand Up @@ -91,4 +91,4 @@ runners:
- type: executable
- type: nextflow
directives:
label: [highmem, midcpu, hightime]
label: [veryhighmem, midcpu, veryveryhightime]
Loading
Loading