From 03eabc2e4a8fa5394c40ccaf2410b14300515182 Mon Sep 17 00:00:00 2001 From: "Guillaume V." Date: Wed, 16 Sep 2026 13:38:43 +0200 Subject: [PATCH 1/3] Update [Notebook] : Add bypass function to directly download outputs from execution name + add check on each step to verify needed files are here --- .../freesurfer_longitudinal.ipynb | 93 ++++++++++++++++++- 1 file changed, 92 insertions(+), 1 deletion(-) diff --git a/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb b/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb index d336c1c..fcfa0ff 100644 --- a/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb +++ b/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb @@ -88,6 +88,7 @@ "\n", "# VIP\n", "from vip_client import VipSession\n", + "from vip_client.utils import vip\n", "\n", "def flatten_folder(folder):\n", " parent = folder.parent\n", @@ -120,7 +121,31 @@ " parent_type = \"folder\"\n", " current = next_path\n", "\n", - " return resource" + " return resource\n", + "\n", + "def download_by_session_name(session_name, local_dir):\n", + " \"\"\"\n", + " Download outputs from a session by name only.\n", + " Only works with the 10 last executions.\n", + " \"\"\"\n", + " local_dir = Path(local_dir)\n", + " local_dir.mkdir(parents=True, exist_ok=True)\n", + " # Find workflows matching session name\n", + " workflows = [e[\"identifier\"] for e in vip.list_executions() if e.get(\"name\") == session_name]\n", + " print(f\"Found {len(workflows)} workflow(s)\")\n", + "\n", + " for wid in workflows:\n", + " files = [(f[\"path\"], str(local_dir / Path(f[\"path\"]).name))\n", + " for f in vip.get_exec_results(wid)\n", + " if not f.get(\"isDirectory\")]\n", + "\n", + " for (vip_path, local_path), success in vip.download_parallel(files):\n", + " print(f\"{'✓' if success else '✗'} {Path(vip_path).name}\")\n", + "\n", + " print(f\"Saved to: {local_dir}\")\n", + "\n", + "def is_empty(path: Path) -> bool:\n", + " return not any(p.name != \"session_data.json\" for p in path.iterdir())" ], "id": "18aa877afac9a865", "outputs": [], @@ -343,6 +368,9 @@ "metadata": {}, "cell_type": "code", "source": [ + "if not cfg.LOCAL_DATASET_PATH.exists() or is_empty(cfg.LOCAL_DATASET_PATH):\n", + " raise FileNotFoundError(f\"The local dataset path {cfg.LOCAL_DATASET_PATH} does not exist or is empty.\")\n", + "\n", "# Remove FreeSurfer CROSS output directory if existing\n", "if FS_CROSS_OUTPUT_DIR.exists():\n", " shutil.rmtree(FS_CROSS_OUTPUT_DIR)\n", @@ -412,6 +440,23 @@ "outputs": [], "execution_count": null }, + { + "metadata": {}, + "cell_type": "markdown", + "source": "⚠️ Use this last cell only if the previous one didn't work. It directly downloads the outputs from a session by a name you provide. You can find the session name on the vip portal as execution name, or in the CROSS/Outputs/session_data.json file.", + "id": "b2c407bdb11858d6" + }, + { + "metadata": {}, + "cell_type": "code", + "source": [ + "# Download outputs from a session by name only\n", + "download_by_session_name(\"SESSION_NAME\", str(FS_CROSS_OUTPUT_DIR))" + ], + "id": "d32ca116a43aeb15", + "outputs": [], + "execution_count": null + }, { "metadata": {}, "cell_type": "markdown", @@ -442,6 +487,9 @@ "metadata": {}, "cell_type": "code", "source": [ + "if not FS_CROSS_OUTPUT_DIR.exists() or is_empty(FS_CROSS_OUTPUT_DIR):\n", + " raise FileNotFoundError(f\"The CROSS output directory {FS_CROSS_OUTPUT_DIR} does not exist or is empty.\")\n", + "\n", "# Remove FreeSurfer BASE inputs/outputs directories if existing\n", "if FS_BASE_INPUT_DIR.exists():\n", " shutil.rmtree(FS_BASE_INPUT_DIR)\n", @@ -525,6 +573,9 @@ "metadata": {}, "cell_type": "code", "source": [ + "if not FS_BASE_INPUT_DIR.exists() or is_empty(FS_BASE_INPUT_DIR):\n", + " raise FileNotFoundError(f\"The BASE input directory {FS_BASE_INPUT_DIR} does not exist or is empty.\")\n", + "\n", "# Collect tarballs and BASE_IDs\n", "tp_tarballs = [f for f in FS_BASE_INPUT_DIR.iterdir() if f.suffix in (\".tgz\", \".tar.gz\") and \"_TPs\" in f.name]\n", "base_ids = [f.name.split(\"_\")[0] for f in tp_tarballs]\n", @@ -579,6 +630,23 @@ "outputs": [], "execution_count": null }, + { + "metadata": {}, + "cell_type": "markdown", + "source": "⚠️ Use this last cell only if the previous one didn't work. It directly downloads the outputs from a session by a name you provide. You can find the session name on the vip portal as execution name, or in the BASE/Outputs/session_data.json file.", + "id": "40677f8b4ddcc15c" + }, + { + "metadata": {}, + "cell_type": "code", + "source": [ + "# Download outputs from a session by name only\n", + "download_by_session_name(\"SESSION_NAME\", str(FS_BASE_OUTPUT_DIR))" + ], + "id": "1172112f1fafb734", + "outputs": [], + "execution_count": null + }, { "metadata": {}, "cell_type": "markdown", @@ -612,6 +680,9 @@ "metadata": {}, "cell_type": "code", "source": [ + "if not FS_BASE_OUTPUT_DIR.exists() or is_empty(FS_BASE_OUTPUT_DIR):\n", + " raise FileNotFoundError(f\"The BASE output directory {FS_BASE_OUTPUT_DIR} does not exist or is empty.\")\n", + "\n", "# Remove FreeSurfer LONG output directory if existing\n", "if FS_LONG_OUTPUT_DIR.exists():\n", " shutil.rmtree(FS_LONG_OUTPUT_DIR)\n", @@ -692,6 +763,23 @@ "outputs": [], "execution_count": null }, + { + "metadata": {}, + "cell_type": "markdown", + "source": "⚠️ Use this last cell only if the previous one didn't work. It directly downloads the outputs from a session by a name you provide. You can find the session name on the vip portal as execution name, or in the LONG/Outputs/session_data.json file.", + "id": "6841fad83b295709" + }, + { + "metadata": {}, + "cell_type": "code", + "source": [ + "# Download outputs from a session by name only\n", + "download_by_session_name(\"SESSION_NAME\", str(FS_LONG_OUTPUT_DIR))" + ], + "id": "3a8a884613a57a15", + "outputs": [], + "execution_count": null + }, { "metadata": {}, "cell_type": "markdown", @@ -718,6 +806,9 @@ "metadata": {}, "cell_type": "code", "source": [ + "if not FS_LONG_OUTPUT_DIR.exists() or is_empty(FS_LONG_OUTPUT_DIR):\n", + " raise FileNotFoundError(f\"The LONG output directory {FS_LONG_OUTPUT_DIR} does not exist or is empty.\")\n", + "\n", "# Restore all session_data.json and licenses files\n", "def restore_files(files):\n", " for original_path, temp_path in files:\n", From c63f37fe033cc3995b6efe3855e85179657dafe5 Mon Sep 17 00:00:00 2001 From: "Guillaume V." Date: Fri, 2 Oct 2026 10:50:48 +0200 Subject: [PATCH 2/3] Update [Freesurfer Notebook] : improve vip downloads safety + add checks, clean diffs and idempotence to freesurfer notebook --- .../freesurfer_longitudinal.ipynb | 453 +++++++++--------- src/vip_client/utils/vip.py | 74 ++- 2 files changed, 286 insertions(+), 241 deletions(-) diff --git a/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb b/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb index fcfa0ff..a032746 100644 --- a/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb +++ b/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb @@ -23,7 +23,7 @@ "\n", "⚠️ Please **DO NOT** run the full notebook at once, but proceed cell by cell to ensure the inputs and outputs are correct after each step, this will allow you to control your execution and revert changes if needed.\n", "\n", - "⚠️ If you are closing the notebook (pipelines running in background or pausing between steps), remember to run the Libraries and Parameters cells (before step 1) again before continuing your work." + "⚠️ If you are closing the notebook (pipelines running in background, pausing between steps, closing editor, etc...), remember to run the Setup cell (before step 1) again before continuing your work." ], "id": "eaa23e833be5ca5d" }, @@ -72,95 +72,9 @@ { "metadata": {}, "cell_type": "markdown", - "source": "## Libraries and functions", - "id": "7817023ff03e7dbf" - }, - { - "metadata": {}, - "cell_type": "code", "source": [ - "# Built-ins\n", - "import tarfile\n", - "import shutil\n", - "import ipywidgets as widgets\n", - "from IPython.display import display, clear_output\n", - "from pathlib import Path\n", - "\n", - "# VIP\n", - "from vip_client import VipSession\n", - "from vip_client.utils import vip\n", - "\n", - "def flatten_folder(folder):\n", - " parent = folder.parent\n", - " for f in folder.iterdir():\n", - " f.rename(parent / f.name)\n", - " shutil.rmtree(folder)\n", - "\n", - "def ensure_girder_path(client, path):\n", - " \"\"\"\n", - " Create missing Girder folders in a path like:\n", - " /collection/name/folder/subfolder\n", - " \"\"\"\n", - " parts = path.strip(\"/\").split(\"/\")\n", - " current = \"/\" + \"/\".join(parts[:2])\n", - " resource = client.resourceLookup(current)\n", - " parent_id = resource[\"_id\"]\n", - " parent_type = resource[\"_modelType\"]\n", - " for part in parts[2:-1]:\n", - " next_path = current + \"/\" + part\n", - " try:\n", - " resource = client.resourceLookup(next_path)\n", - " except Exception:\n", - " resource = client.createFolder(\n", - " parent_id,\n", - " part,\n", - " parentType=parent_type\n", - " )\n", - "\n", - " parent_id = resource[\"_id\"]\n", - " parent_type = \"folder\"\n", - " current = next_path\n", - "\n", - " return resource\n", - "\n", - "def download_by_session_name(session_name, local_dir):\n", - " \"\"\"\n", - " Download outputs from a session by name only.\n", - " Only works with the 10 last executions.\n", - " \"\"\"\n", - " local_dir = Path(local_dir)\n", - " local_dir.mkdir(parents=True, exist_ok=True)\n", - " # Find workflows matching session name\n", - " workflows = [e[\"identifier\"] for e in vip.list_executions() if e.get(\"name\") == session_name]\n", - " print(f\"Found {len(workflows)} workflow(s)\")\n", - "\n", - " for wid in workflows:\n", - " files = [(f[\"path\"], str(local_dir / Path(f[\"path\"]).name))\n", - " for f in vip.get_exec_results(wid)\n", - " if not f.get(\"isDirectory\")]\n", - "\n", - " for (vip_path, local_path), success in vip.download_parallel(files):\n", - " print(f\"{'✓' if success else '✗'} {Path(vip_path).name}\")\n", + "## User configuration\n", "\n", - " print(f\"Saved to: {local_dir}\")\n", - "\n", - "def is_empty(path: Path) -> bool:\n", - " return not any(p.name != \"session_data.json\" for p in path.iterdir())" - ], - "id": "18aa877afac9a865", - "outputs": [], - "execution_count": null - }, - { - "metadata": {}, - "cell_type": "markdown", - "source": "## Parameters", - "id": "3b01f462e227d9fa" - }, - { - "metadata": {}, - "cell_type": "markdown", - "source": [ "**User variables**: should be checked and supplied at each execution. This cell will create a user_config.py file. Please modify the variable values by editing the user_config.py file after it is created.\n", "\n", "⚠️ We strongly advise you not to modify the values directly in this cell in order to avoid conflicts when updating the notebook." @@ -171,6 +85,11 @@ "metadata": {}, "cell_type": "code", "source": [ + "import ipywidgets as widgets\n", + "\n", + "from pathlib import Path\n", + "from IPython.display import display, clear_output\n", + "\n", "CONFIG_PATH = Path('user_config.py')\n", "\n", "user_config = f\"\"\"\n", @@ -188,7 +107,7 @@ "GIRDER_OUTPUT_PATH = GIRDER_DATASET_PATH + '/derivatives/freesurfer' # Your Girder path to the final outputs folder, can be in collections or user folders. Defaults to `GIRDER_DATASET_PATH/derivatives/freesurfer` but can be changed to format '/user/USER_NAME/FOLDER_NAME' or '/collection/COLLECTION_NAME/FOLDER_NAME'. The girder folders will be created automatically if they do not exist, but the script cannot create collections.\n", "\n", "# Local\n", - "LOCAL_DATASET_PATH = Path('~/PATH').expanduser() # Your local dataset path, where the Girder dataset will be downloaded, must already exist and be empty. Do not add unrelated custom files in this directory and its subdirectories as it may break vip executions.\n", + "LOCAL_DATASET_PATH = Path('~/PATH').expanduser() # Your local dataset path, where the Girder dataset will be downloaded and all the necessary files/folders created, must already exist and be empty. Do not add unrelated custom files in this directory and its subdirectories as it may break vip executions.\n", "LICENSE_PATH = Path('~/PATH/LICENSE_NAME.txt').expanduser() # Your pipeline license path, will be copied into VIP input_dirs automatically\n", "\"\"\"\n", "\n", @@ -222,33 +141,168 @@ { "metadata": {}, "cell_type": "markdown", - "source": "Once you have filled the user_config.py file with your custom parameters, you can run the following cell to load the configuration.", + "source": [ + "## Setup\n", + "\n", + "Once you have filled the user_config.py file with your custom parameters, you can run the following cell to load the configuration, constant variables (should not be changed unless the dataset structure or the pipeline have been modified), libraries and functions that will be used all over the notebook.\n", + "\n", + "⚠️ This cell **MUST** be run once per use of the notebook, it has to be relaunched whenever the notebook has been closed or reset." + ], "id": "1c54141a752cab56" }, { "metadata": {}, "cell_type": "code", "source": [ + "# Built-ins\n", "import importlib\n", - "import user_config as cfg\n", + "import tarfile\n", + "import shutil\n", + "import json\n", + "from pathlib import Path\n", + "\n", + "# VIP\n", + "from vip_client import VipSession\n", + "from vip_client.utils import vip\n", "\n", "# Reload user config\n", - "importlib.reload(cfg)" - ], - "id": "5e9ad001a44701cd", - "outputs": [], - "execution_count": null - }, - { - "metadata": {}, - "cell_type": "markdown", - "source": "**Constant variables**: should not be changed unless the dataset structure or the pipeline have been modified\n", - "id": "8303e0953a323b31" - }, - { - "metadata": {}, - "cell_type": "code", - "source": [ + "import user_config as cfg\n", + "importlib.reload(cfg)\n", + "\n", + "### COMMON FUNCTIONS\n", + "\n", + "def ensure_girder_path(client, path):\n", + " \"\"\"\n", + " Create missing Girder folders in a path like:\n", + " /collection/name/folder/subfolder\n", + " \"\"\"\n", + " parts = path.strip(\"/\").split(\"/\")\n", + " current = \"/\" + \"/\".join(parts[:2])\n", + " resource = client.resourceLookup(current)\n", + " parent_id = resource[\"_id\"]\n", + " parent_type = resource[\"_modelType\"]\n", + " for part in parts[2:-1]:\n", + " next_path = current + \"/\" + part\n", + " try:\n", + " resource = client.resourceLookup(next_path)\n", + " except Exception:\n", + " resource = client.createFolder(\n", + " parent_id,\n", + " part,\n", + " parentType=parent_type\n", + " )\n", + "\n", + " parent_id = resource[\"_id\"]\n", + " parent_type = \"folder\"\n", + " current = next_path\n", + "\n", + " return resource\n", + "\n", + "def get_workflows(session_name):\n", + " return [e[\"identifier\"]\n", + " for e in vip.list_executions()\n", + " if e.get(\"name\") == session_name]\n", + "\n", + "def get_vip_files(workflow_id, local_dir):\n", + " return [(f[\"path\"], str(local_dir / Path(f[\"path\"]).name), f.get(\"size\"))\n", + " for f in vip.get_exec_results(workflow_id)\n", + " if not f.get(\"isDirectory\")]\n", + "\n", + "def download_by_session_name(session_name, local_dir):\n", + " \"\"\"\n", + " Download outputs from a session by name only.\n", + " Only works with the 10 last executions.\n", + " \"\"\"\n", + " local_dir = Path(local_dir)\n", + " local_dir.mkdir(parents=True, exist_ok=True)\n", + " # Find workflows matching session name\n", + " workflows = get_workflows(session_name)\n", + " print(f\"Found {len(workflows)} workflow(s)\")\n", + " successes = []\n", + " failures = []\n", + " for wid in workflows:\n", + " files = get_vip_files(wid, local_dir)\n", + " # Filter out files that already exist locally\n", + " files_to_download = [(vip_path, local_path)\n", + " for vip_path, local_path, size in files\n", + " if not Path(local_path).exists()\n", + " or (Path(local_path).exists()\n", + " and Path(local_path).stat().st_size != size)]\n", + "\n", + " if not files_to_download:\n", + " print(f\"All {len(files)} files already present, skipping download\")\n", + " continue\n", + "\n", + " print(f\"Downloading {len(files_to_download)} files (skipping {len(files) - len(files_to_download)} already present)\")\n", + " for (vip_path, local_path), success in vip.download_parallel(files_to_download):\n", + " if success:\n", + " successes.append({Path(vip_path).name})\n", + " else:\n", + " failures.append({Path(vip_path).name})\n", + "\n", + " if len(successes) > 0:\n", + " print(f\"{len(successes)} files downloaded successfully and saved to {local_dir}\")\n", + "\n", + " if len(failures) > 0:\n", + " print(f\"The following {len(failures)} files failed to download : \")\n", + " for f in failures:\n", + " print(f\"✗ {f}\")\n", + "\n", + "def is_empty(path: Path) -> bool:\n", + " return not any(p.name != \"session_data.json\" for p in path.iterdir())\n", + "\n", + "def check_downloaded_files(session_name, local_dir):\n", + " local_dir = Path(local_dir)\n", + " workflows = get_workflows(session_name)\n", + " expected = {\n", + " Path(vip_path).name\n", + " for wid in workflows\n", + " for vip_path, _, _ in get_vip_files(wid, local_dir)\n", + " }\n", + " downloaded = {\n", + " p.name\n", + " for p in local_dir.iterdir()\n", + " if p.is_file() and p.name != SESSION_DATA_NAME and not p.name.endswith(\".part\")\n", + " }\n", + "\n", + " missing = expected - downloaded\n", + " unexpected = downloaded - expected\n", + " print(\"=\" * 60)\n", + " print(\"DOWNLOAD CHECK\")\n", + " print(\"=\" * 60)\n", + " print(f\"Expected from VIP : {len(expected)}\")\n", + " print(f\"Found locally : {len(downloaded)}\")\n", + " print(f\"Missing : {len(missing)}\")\n", + " print(f\"Unexpected : {len(unexpected)}\")\n", + " if missing:\n", + " print()\n", + " print(\"Missing files:\")\n", + " for filename in sorted(missing):\n", + " print(f\" ✗ {filename}\")\n", + "\n", + " if unexpected:\n", + " print()\n", + " print(\"Unexpected local files:\")\n", + " for filename in sorted(unexpected):\n", + " print(f\" ! {filename}\")\n", + "\n", + " if not missing and not unexpected:\n", + " print()\n", + " print(\"✓ Local files match the files expected from VIP.\")\n", + "\n", + " print()\n", + "\n", + "def restore_session_from_output_dir(output_dir):\n", + " \"\"\"Restore session_name from session_data.json if it exists\"\"\"\n", + " session_file = Path(output_dir) / SESSION_DATA_NAME\n", + " if session_file.exists():\n", + " with open(session_file) as f:\n", + " data = json.load(f)\n", + " return data.get(\"session_name\")\n", + " return None\n", + "\n", + "### CONSTANTS\n", + "\n", "# Name of the pipelines\n", "PIPELINE_ALL_ID = 'FreeSurfer-Recon-all/7.3.1'\n", "PIPELINE_BASE_ID = 'FreeSurfer-Recon-all-BASE/7.3.1'\n", @@ -259,6 +313,7 @@ "\n", "# FreeSurfer CROSS\n", "FS_CROSS_ID = 'CROSS'\n", + "FS_CROSS_INPUT_DIR = Path(FREESURFER_DIR / FS_CROSS_ID / 'Inputs')\n", "FS_CROSS_OUTPUT_DIR = Path(FREESURFER_DIR / FS_CROSS_ID / 'Outputs')\n", "\n", "# FreeSurfer BASE\n", @@ -276,11 +331,11 @@ " FS_BASE_ID : FS_BASE_OUTPUT_DIR,\n", " FS_LONG_ID : FS_LONG_OUTPUT_DIR}\n", "\n", - "LICENSES = {FS_CROSS_ID : cfg.LOCAL_DATASET_PATH / f'license_{FS_CROSS_ID}.txt',\n", + "LICENSES = {FS_CROSS_ID : FS_CROSS_INPUT_DIR / f'license_{FS_CROSS_ID}.txt',\n", " FS_BASE_ID : FS_BASE_INPUT_DIR / f'license_{FS_BASE_ID}.txt',\n", " FS_LONG_ID : FS_BASE_OUTPUT_DIR / f'license_{FS_LONG_ID}.txt'}" ], - "id": "629bf8acf75b0901", + "id": "5e9ad001a44701cd", "outputs": [], "execution_count": null }, @@ -295,7 +350,7 @@ "### ⚙️ Inputs\n", "\n", "- `GIRDER_DATASET_PATH`: Path of the Girder folder containing MRI data\n", - "- `LOCAL_DATASET_PATH`: Local path where the data will be downloaded\n", + "- `LOCAL_DATASET_PATH`: Local path where all the INPUTS/OUTPUTS files/folders data will be downloaded\n", "\n", "### 📤 Outputs\n", "\n", @@ -313,14 +368,12 @@ "source": [ "# Authentication to Girder\n", "cfg.GIRDER_CLIENT.authenticate(apiKey=cfg.GIRDER_KEY)\n", - "\n", - "# Download dataset from Girder\n", - "cfg.GIRDER_CLIENT.downloadFolderRecursive(cfg.GIRDER_DATASET_PATH, str(cfg.LOCAL_DATASET_PATH))\n", - "\n", "# Create LOCAL_DATASET_PATH, derivatives and freesurfer folders if they don't exist\n", "FREESURFER_DIR.mkdir(parents=True, exist_ok=True)\n", + "# Download dataset from Girder\n", + "cfg.GIRDER_CLIENT.downloadFolderRecursive(cfg.GIRDER_DATASET_PATH, str(FS_CROSS_INPUT_DIR))\n", "\n", - "print(f\"✅ Done: 'derivatives/freesurfer' folder ensured at: {FREESURFER_DIR}, Girder dataset copied to : {cfg.LOCAL_DATASET_PATH}\")" + "print(f\"✅ Done: 'derivatives/freesurfer' folder ensured at: {FREESURFER_DIR}, Girder dataset copied to : {FS_CROSS_INPUT_DIR}\")" ], "outputs": [], "execution_count": null @@ -368,12 +421,8 @@ "metadata": {}, "cell_type": "code", "source": [ - "if not cfg.LOCAL_DATASET_PATH.exists() or is_empty(cfg.LOCAL_DATASET_PATH):\n", - " raise FileNotFoundError(f\"The local dataset path {cfg.LOCAL_DATASET_PATH} does not exist or is empty.\")\n", - "\n", - "# Remove FreeSurfer CROSS output directory if existing\n", - "if FS_CROSS_OUTPUT_DIR.exists():\n", - " shutil.rmtree(FS_CROSS_OUTPUT_DIR)\n", + "if not FS_CROSS_INPUT_DIR.exists() or is_empty(FS_CROSS_INPUT_DIR):\n", + " raise FileNotFoundError(f\"The local dataset path {FS_CROSS_INPUT_DIR} does not exist or is empty.\")\n", "\n", "# Create FreeSurfer CROSS output directory\n", "FS_CROSS_OUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n", @@ -381,7 +430,7 @@ "# Get T1w NIfTIs only sub-* folders, excluding run-02, run-03, ...\n", "nifti_files = [\n", " str(f)\n", - " for sub in cfg.LOCAL_DATASET_PATH.iterdir()\n", + " for sub in FS_CROSS_INPUT_DIR.iterdir()\n", " if sub.is_dir() and sub.name.startswith('sub-')\n", " for f in sub.rglob('*.nii.gz')\n", " if f.name.endswith('_T1w.nii.gz')\n", @@ -399,7 +448,7 @@ "\n", "session = VipSession.init(\n", " api_key=cfg.VIP_KEY,\n", - " input_dir=str(cfg.LOCAL_DATASET_PATH),\n", + " input_dir=str(FS_CROSS_INPUT_DIR),\n", " output_dir= str(FS_CROSS_OUTPUT_DIR),\n", " pipeline_id=PIPELINE_ALL_ID,\n", " input_settings=input_settings\n", @@ -417,7 +466,11 @@ { "metadata": {}, "cell_type": "markdown", - "source": "Run this cell after the execution ended to download outputs and setup output directory", + "source": [ + "Run this cell after the execution ended to download outputs and setup output directory.\n", + "\n", + "⚠️ You can also provide a specific session name if needed (replace the session_name value). You can find the session name on the vip portal as execution name, or in the CROSS/Outputs/session_data.json file." + ], "id": "ca06f1b3ede87d90" }, { @@ -426,37 +479,14 @@ "source": [ "# Connect back to session if needed\n", "VipSession.init(api_key=cfg.VIP_KEY)\n", - "session = VipSession(output_dir=str(FS_CROSS_OUTPUT_DIR))\n", - "\n", - "# Download outputs to the output_dir\n", - "session.download_outputs(False, get_status=['Finished', 'Killed'])\n", - "\n", - "# Since results may arrive in a VipSession folder (2026-X), when it's the case we flatten this folder to put files directly under FS_CROSS_OUTPUT_DIR\n", - "for f in FS_CROSS_OUTPUT_DIR.iterdir():\n", - " if f.is_dir():\n", - " flatten_folder(Path(f))" + "session_name = restore_session_from_output_dir(FS_CROSS_OUTPUT_DIR) # Replace with a specific session name if needed\n", + "download_by_session_name(session_name, str(FS_CROSS_OUTPUT_DIR))\n", + "check_downloaded_files(session_name, str(FS_CROSS_OUTPUT_DIR))" ], "id": "ae2c997bbc3d32", "outputs": [], "execution_count": null }, - { - "metadata": {}, - "cell_type": "markdown", - "source": "⚠️ Use this last cell only if the previous one didn't work. It directly downloads the outputs from a session by a name you provide. You can find the session name on the vip portal as execution name, or in the CROSS/Outputs/session_data.json file.", - "id": "b2c407bdb11858d6" - }, - { - "metadata": {}, - "cell_type": "code", - "source": [ - "# Download outputs from a session by name only\n", - "download_by_session_name(\"SESSION_NAME\", str(FS_CROSS_OUTPUT_DIR))" - ], - "id": "d32ca116a43aeb15", - "outputs": [], - "execution_count": null - }, { "metadata": {}, "cell_type": "markdown", @@ -490,52 +520,59 @@ "if not FS_CROSS_OUTPUT_DIR.exists() or is_empty(FS_CROSS_OUTPUT_DIR):\n", " raise FileNotFoundError(f\"The CROSS output directory {FS_CROSS_OUTPUT_DIR} does not exist or is empty.\")\n", "\n", - "# Remove FreeSurfer BASE inputs/outputs directories if existing\n", - "if FS_BASE_INPUT_DIR.exists():\n", - " shutil.rmtree(FS_BASE_INPUT_DIR)\n", - "if FS_BASE_OUTPUT_DIR.exists():\n", - " shutil.rmtree(FS_BASE_OUTPUT_DIR)\n", - "\n", - "# Create FreeSurfer BASE inputs/outputs directories\n", + "# Create FreeSurfer BASE inputs directory\n", "FS_BASE_INPUT_DIR.mkdir(parents=True, exist_ok=True)\n", - "FS_BASE_OUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n", "\n", "# Collect eligible tarballs (.tar.gz or .tgz)\n", "tarballs = [\n", " t for t in FS_CROSS_OUTPUT_DIR.iterdir()\n", " if t.is_file()\n", - " and t.suffixes in ([\".tar\", \".gz\"], [\".tgz\",])\n", - " and \"ses-\" in t.name\n", - " and \".long.\" not in t.name\n", + " and t.suffixes in ([\".tar\", \".gz\"], [\".tgz\",])\n", + " and \"ses-\" in t.name\n", + " and \".long.\" not in t.name\n", "]\n", "\n", "subjects = {}\n", "for tar_path in tarballs:\n", + " print(f\"\\n Processing: {tar_path.name}\")\n", " subj = tar_path.name.split(\"_\")[0] # sub-XXXX\n", "\n", " # Remove full archive suffixes safely\n", " base_name = tar_path.name[:-(sum(len(s) for s in tar_path.suffixes))]\n", - "\n", " extract_dir = FS_BASE_INPUT_DIR / \"extracted\" / base_name\n", - " extract_dir.mkdir(parents=True, exist_ok=True)\n", + " if extract_dir.exists() and any(extract_dir.iterdir()):\n", + " print(f\" → Already extracted, skipping.\")\n", + " subjects.setdefault(subj, []).append(extract_dir)\n", + " continue\n", "\n", + " extract_dir.mkdir(parents=True, exist_ok=True)\n", " # Untar (auto-detect compression)\n", - " with tarfile.open(tar_path, \"r:*\") as tar:\n", - " tar.extractall(path=extract_dir)\n", + " try:\n", + " with tarfile.open(tar_path, \"r:*\") as tar:\n", + " for member in tar:\n", + " try:\n", + " tar.extract(member, path=extract_dir)\n", + " except Exception as e:\n", + " print(f\" ❌ Error with internal file: {member.name}, {type(e).__name__}: {e}\")\n", + " raise\n", + " except Exception as e:\n", + " print(f\" ❌ ERROR: {tar_path.name} → {type(e).__name__}: {e}\")\n", + " continue\n", "\n", " subjects.setdefault(subj, []).append(extract_dir)\n", "\n", "# Group per subject\n", "for subj, tp_dirs in subjects.items():\n", " out_tar = FS_BASE_INPUT_DIR / f\"{subj}_TPs.tgz\"\n", + " if out_tar.exists() and out_tar.stat().st_size > 0:\n", + " print(f\" → {out_tar.name} already exists, skipping.\")\n", + " continue\n", + "\n", " with tarfile.open(out_tar, \"w:gz\") as tar:\n", " for tp_dir in tp_dirs:\n", " for item in tp_dir.iterdir():\n", " tar.add(item, arcname=item.name)\n", "\n", - "# Cleanup extracted files\n", - "shutil.rmtree(FS_BASE_INPUT_DIR / \"extracted\", ignore_errors=True)\n", - "\n", "print(f\"✅ Done: grouped longitudinal timepoints per subject at '{FS_BASE_INPUT_DIR}'.\")" ], "id": "e0896ac9ed719d4", @@ -576,6 +613,12 @@ "if not FS_BASE_INPUT_DIR.exists() or is_empty(FS_BASE_INPUT_DIR):\n", " raise FileNotFoundError(f\"The BASE input directory {FS_BASE_INPUT_DIR} does not exist or is empty.\")\n", "\n", + "# Cleanup previous extracted files\n", + "shutil.rmtree(FS_BASE_INPUT_DIR / \"extracted\", ignore_errors=True)\n", + "\n", + "# Create FreeSurfer BASE outputs directory\n", + "FS_BASE_OUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n", + "\n", "# Collect tarballs and BASE_IDs\n", "tp_tarballs = [f for f in FS_BASE_INPUT_DIR.iterdir() if f.suffix in (\".tgz\", \".tar.gz\") and \"_TPs\" in f.name]\n", "base_ids = [f.name.split(\"_\")[0] for f in tp_tarballs]\n", @@ -607,7 +650,11 @@ { "metadata": {}, "cell_type": "markdown", - "source": "Run this cell after the execution ended to download outputs and setup output directory", + "source": [ + "Run this cell after the execution ended to download outputs and setup output directory.\n", + "\n", + "⚠️ You can also provide a specific session name if needed (replace the session_name value). You can find the session name on the vip portal as execution name, or in the BASE/Outputs/session_data.json file." + ], "id": "26659b5150193e3f" }, { @@ -616,37 +663,14 @@ "source": [ "# Connect back to session if needed\n", "VipSession.init(api_key=cfg.VIP_KEY)\n", - "session = VipSession(output_dir=str(FS_BASE_OUTPUT_DIR))\n", - "\n", - "# Download outputs to the output_dir\n", - "session.download_outputs(False, get_status=['Finished', 'Killed'])\n", - "\n", - "# Since results may arrive in a VipSession folder (2026-X), when it's the case we flatten this folder to put files directly under FS_BASE_OUTPUT_DIR\n", - "for f in FS_BASE_OUTPUT_DIR.iterdir():\n", - " if f.is_dir():\n", - " flatten_folder(Path(f))" + "session_name = restore_session_from_output_dir(FS_BASE_OUTPUT_DIR) # Replace with a specific session name if needed\n", + "download_by_session_name(session_name, str(FS_BASE_OUTPUT_DIR))\n", + "check_downloaded_files(session_name, str(FS_BASE_OUTPUT_DIR))" ], "id": "2ca4dbc9998ae826", "outputs": [], "execution_count": null }, - { - "metadata": {}, - "cell_type": "markdown", - "source": "⚠️ Use this last cell only if the previous one didn't work. It directly downloads the outputs from a session by a name you provide. You can find the session name on the vip portal as execution name, or in the BASE/Outputs/session_data.json file.", - "id": "40677f8b4ddcc15c" - }, - { - "metadata": {}, - "cell_type": "code", - "source": [ - "# Download outputs from a session by name only\n", - "download_by_session_name(\"SESSION_NAME\", str(FS_BASE_OUTPUT_DIR))" - ], - "id": "1172112f1fafb734", - "outputs": [], - "execution_count": null - }, { "metadata": {}, "cell_type": "markdown", @@ -683,10 +707,6 @@ "if not FS_BASE_OUTPUT_DIR.exists() or is_empty(FS_BASE_OUTPUT_DIR):\n", " raise FileNotFoundError(f\"The BASE output directory {FS_BASE_OUTPUT_DIR} does not exist or is empty.\")\n", "\n", - "# Remove FreeSurfer LONG output directory if existing\n", - "if FS_LONG_OUTPUT_DIR.exists():\n", - " shutil.rmtree(FS_LONG_OUTPUT_DIR)\n", - "\n", "# Create FreeSurfer LONG output directory\n", "FS_LONG_OUTPUT_DIR.mkdir(parents=True, exist_ok=True)\n", "\n", @@ -740,7 +760,11 @@ { "metadata": {}, "cell_type": "markdown", - "source": "Run this cell after the execution ended to download outputs and setup output directory", + "source": [ + "Run this cell after the execution ended to download outputs and setup output directory.\n", + "\n", + "⚠️ You can also provide a specific session name if needed (replace the session_name value). You can find the session name on the vip portal as execution name, or in the LONG/Outputs/session_data.json file." + ], "id": "40aa3d4780a6b382" }, { @@ -749,37 +773,14 @@ "source": [ "# Connect back to session if needed\n", "VipSession.init(api_key=cfg.VIP_KEY)\n", - "session = VipSession(output_dir=str(FS_LONG_OUTPUT_DIR))\n", - "\n", - "# Download outputs to the output_dir\n", - "session.download_outputs(False, get_status=['Finished', 'Killed'])\n", - "\n", - "# Since results may arrive in a VipSession folder (2026-X), when it's the case we flatten this folder to put files directly under FS_LONG_OUTPUT_DIR\n", - "for f in FS_LONG_OUTPUT_DIR.iterdir():\n", - " if f.is_dir():\n", - " flatten_folder(Path(f))" + "session_name = restore_session_from_output_dir(FS_LONG_OUTPUT_DIR) # Replace with a specific session name if needed\n", + "download_by_session_name(session_name, str(FS_LONG_OUTPUT_DIR))\n", + "check_downloaded_files(session_name, str(FS_LONG_OUTPUT_DIR))" ], "id": "c2ba7293aebb2ea9", "outputs": [], "execution_count": null }, - { - "metadata": {}, - "cell_type": "markdown", - "source": "⚠️ Use this last cell only if the previous one didn't work. It directly downloads the outputs from a session by a name you provide. You can find the session name on the vip portal as execution name, or in the LONG/Outputs/session_data.json file.", - "id": "6841fad83b295709" - }, - { - "metadata": {}, - "cell_type": "code", - "source": [ - "# Download outputs from a session by name only\n", - "download_by_session_name(\"SESSION_NAME\", str(FS_LONG_OUTPUT_DIR))" - ], - "id": "3a8a884613a57a15", - "outputs": [], - "execution_count": null - }, { "metadata": {}, "cell_type": "markdown", diff --git a/src/vip_client/utils/vip.py b/src/vip_client/utils/vip.py index 4cb7aa5..1abdf48 100644 --- a/src/vip_client/utils/vip.py +++ b/src/vip_client/utils/vip.py @@ -11,7 +11,10 @@ from concurrent.futures import ThreadPoolExecutor from os.path import exists from pathlib import * +import os import threading +import tempfile +import shutil # Third-Party import requests @@ -258,13 +261,34 @@ def download(path, where_to_save) -> bool : """ # Parse arguments url = __PREFIX + 'path' + path + '?action=content' - rq = SESSION.get(url, headers=__headers, stream=True) - if rq.status_code != 200: + tmp_path = None + try: + with SESSION.get(url, headers=__headers, stream=True) as rq: + if rq.status_code != 200: + return False + + dest = Path(where_to_save) + dest.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile(dir=dest.parent, prefix=f".{dest.name}.", suffix=".part", delete=False) as tmp: + tmp_path = tmp.name + shutil.copyfileobj(rq.raw, tmp) + content_length = rq.headers.get("Content-Length") + if content_length is not None: + expected = int(content_length) + if tmp.tell() != expected: + return False + + os.replace(tmp.name, dest) + tmp_path = None + return True + except Exception: return False - else: - with open(where_to_save, 'wb') as out_file: - out_file.write(rq.content) - return True + finally: + if tmp_path is not None: + try: + os.remove(tmp_path) + except OSError: + pass # Methods for parallel downloads @@ -280,17 +304,37 @@ def download_thread(file: tuple) -> tuple : # Parameters path, where_to_save = map(str, file) # URL for request - url = __PREFIX + 'path' + str(path) + '?action=content' + url = __PREFIX + 'path' + path + '?action=content' # Parallel download - with (thread_local.session.get(url, headers=__headers, stream=True) as rq, - open(where_to_save, 'wb') as out_file): - # TODO: manage HTTP return code - if rq.status_code != 200: - return file, False - else: - with open(where_to_save, 'wb') as out_file: - out_file.write(rq.content) + tmp_path = None + try: + with thread_local.session.get(url, headers=__headers, stream=True) as rq: + # TODO: manage HTTP return code + if rq.status_code != 200: + return file, False + + dest = Path(where_to_save) + dest.parent.mkdir(parents=True, exist_ok=True) + with tempfile.NamedTemporaryFile(dir=dest.parent, prefix=f".{dest.name}.", suffix=".part", delete=False) as tmp: + tmp_path = tmp.name + shutil.copyfileobj(rq.raw, tmp) + content_length = rq.headers.get("Content-Length") + if content_length is not None: + expected = int(content_length) + if tmp.tell() != expected: + return file, False + + os.replace(tmp.name, dest) + tmp_path = None return file, True + except Exception: + return file, False + finally: + if tmp_path is not None: + try: + os.remove(tmp_path) + except OSError: + pass def download_parallel(files): """ From 8c99477eb4cb2f2c4d3751c6617cb5c066f94166 Mon Sep 17 00:00:00 2001 From: "Guillaume V." Date: Fri, 2 Oct 2026 13:23:43 +0200 Subject: [PATCH 3/3] Update [Freesurfer]: update doc --- .../freesurfer_longitudinal.ipynb | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb b/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb index a032746..950219d 100644 --- a/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb +++ b/examples/freesurfer/freesurfer_longitudinal_processing/freesurfer_longitudinal.ipynb @@ -840,7 +840,7 @@ " cfg.GIRDER_OUTPUT_PATH\n", " )\n", "\n", - " # Upload full freesurfer dir (outputs) to Girder selected path GIRDER_OUTPUT_PATH\n", + " # Upload full freesurfer dir (inputs/outputs) to Girder selected path GIRDER_OUTPUT_PATH\n", " cfg.GIRDER_CLIENT.upload(str(FREESURFER_DIR), output_folder[\"_id\"], leafFoldersAsItems=True, reuseExisting=True)\n", "finally:\n", " restore_files(moved_files)\n",