Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
58 changes: 58 additions & 0 deletions container_templates/base.def
Original file line number Diff line number Diff line change
@@ -0,0 +1,58 @@
# Written by scripts/build_containers.py from container_templates/base.def.
# The base image and install steps come from uv.def, or micromamba.def for
# projects with a conda lock file.
Bootstrap: docker
@@FROM@@

%files
@@FILES_BLOCK@@

%post
# Targets for the data directories bound in at run time
mkdir -p /cvmfs /hdfs /gpfs /ceph /hadoop

apt-get update
apt-get install -y --no-install-recommends build-essential git
rm -rf /var/lib/apt/lists/*

@@EXTRA_POST@@

# Installing with `--no-cache` unpacks packages into a temporary directory
# under TMPDIR. Apptainer bind-mounts the host's /tmp into %post, so the
# default would use the host's /tmp rather than the build filesystem, which
# can easily run out of space. Keep it inside the image instead, where it
# follows APPTAINER_TMPDIR, and remove it before the image is finalized.
export TMPDIR=/opt/build-tmp
mkdir -p $TMPDIR

@@INSTALL@@

rm -rf /opt/build-tmp

# Record the environment hash (scripts/env_hash.py) so the pipeline can
# tell whether this image matches the repo it's run from.
# Also record commit it was built from, with "-dirty" if its environment
# files had uncommitted changes.
echo @@ENV_HASH@@ > /opt/env_hash
echo @@BUILD_COMMIT@@ > /opt/build_commit

@@PURGE_BUILD_TOOLS@@

%environment
# Put the environment on PATH. The snakemake apptainer integration uses
# `apptainer exec`, which ignores %runscript, so this is what makes the
# project commands findable.
export PATH="/opt/env/bin:$PATH"

# Condor jobs may have no usable $HOME, so keep caches (astropy, gwpy,
# pycbc, matplotlib, torch inductor, triton) in the job's scratch directory.
# Outside condor, nothing changes.
if [ -n "${_CONDOR_SCRATCH_DIR:-}" ]; then
export HOME="$_CONDOR_SCRATCH_DIR"
export XDG_CACHE_HOME="$HOME/.cache"
export MPLCONFIGDIR="$XDG_CACHE_HOME/matplotlib"
export TORCHINDUCTOR_CACHE_DIR="$XDG_CACHE_HOME/torchinductor"
export TRITON_CACHE_DIR="$XDG_CACHE_HOME/triton"
fi

@@EXTRA_ENV@@
58 changes: 2 additions & 56 deletions container_templates/micromamba.def
Original file line number Diff line number Diff line change
@@ -1,28 +1,7 @@
Bootstrap: docker
From: mambaorg/micromamba:2
Stage: build

%files
@@FILES_BLOCK@@

%environment
# Put the conda env on PATH. The snakemake apptainer integration
# uses `apptainer exec`, which ignores %runscript; prepending
# to PATH here makes the project commands findable.
export PATH=/opt/env/bin:$PATH

@@EXTRA_ENV@@

%post
mkdir -p /cvmfs /hdfs /gpfs /ceph /hadoop

apt-get update
apt-get install -y --no-install-recommends git build-essential
rm -rf /var/lib/apt/lists/*

@@EXTRA_POST@@

# activate micromamba and create environment from lockfile
# create the environment from the project's lockfile
micromamba create -p /opt/env -f /opt/aframe/projects/@@PROJECT@@/@@PROJECT@@.conda-lock.yml

# install uv so we can install local deps of deps editably
Expand All @@ -32,15 +11,7 @@ micromamba run -p /opt/env python -m \
# Export rather than `uv sync`, so that uv installs into the existing conda
# env instead of managing its own. Export to pylock.toml rather than
# requirements.txt so that the index each package was locked from is
# recorded. This allows torch to be installed from the PyTorch CPU index.
# `--no-cache` makes uv unpack wheels into a temporary directory under
# TMPDIR. Apptainer bind-mounts the host's /tmp into %post, so the default
# would use the host's /tmp rather than the build filesystem, which
# can easily run out of space. Keep it inside the image instead, where it
# follows APPTAINER_TMPDIR, and remove it before the image is finalized.
export TMPDIR=/opt/build-tmp
mkdir -p $TMPDIR

# recorded. This allows torch to be installed from the PyTorch CPU index.
cd /opt/aframe
micromamba run -p /opt/env \
@@UV_CMD@@ --no-cache -o pylock.toml
Expand All @@ -49,29 +20,4 @@ micromamba run -p /opt/env \
uv pip install --no-cache -r pylock.toml

rm pylock.toml
rm -rf /opt/build-tmp
micromamba clean -ay

# Record the environment hash (scripts/env_hash.py) so the pipeline can
# tell whether this image matches the repo it's run from.
# Also record commit it was built from, with "-dirty" if its environment
# files had uncommitted changes.
echo @@ENV_HASH@@ > /opt/env_hash
echo @@BUILD_COMMIT@@ > /opt/build_commit

@@PURGE_BUILD_TOOLS@@

# initialize our shell so that we can execute
# commands in our environment at run time
# set path, and add it to /etc/profile
# so that it will be set if login shell
# is invoked
export PATH=/opt/env/bin:$PATH
echo export PATH=$PATH >> /etc/profile


%runscript
#!/bin/bash
eval "$(micromamba shell hook --shell bash)"
micromamba activate /opt/env
exec "$@"
37 changes: 0 additions & 37 deletions container_templates/uv.def
Original file line number Diff line number Diff line change
@@ -1,48 +1,11 @@
Bootstrap: docker
# Important: 0.9.30 is the last uv image built on Debian 12 (bookworm).
# Every tag afterwards is Debian 13 (trixie), and NVIDIA's Debian 13 repo
# doesn't have CUDA 12 support, which precludes running on V100s.
From: ghcr.io/astral-sh/uv:0.9.30-python3.12-bookworm-slim

%files
@@FILES_BLOCK@@

%post
apt-get update
apt-get install -y --no-install-recommends build-essential git
rm -rf /var/lib/apt/lists/*

@@EXTRA_POST@@

cd /opt/aframe/projects/@@PROJECT@@
# Set venv dir outside of project for
# when binding the repo into the container
export UV_PROJECT_ENVIRONMENT=/opt/env

# `--no-cache` makes uv unpack wheels into a temporary directory under
# TMPDIR. Apptainer bind-mounts the host's /tmp into %post, so the default
# would use the host's /tmp rather than the build filesystem, which
# can easily run out of space. Keep it inside the image instead, where it
# follows APPTAINER_TMPDIR, and remove it before the image is finalized.
export TMPDIR=/opt/build-tmp
mkdir -p $TMPDIR

@@UV_CMD@@ --no-cache

rm -rf /opt/build-tmp

# Record the environment hash (scripts/env_hash.py) so the pipeline can
# tell whether this image matches the repo it's run from.
# Also record commit it was built from, with "-dirty" if its environment
# files had uncommitted changes.
echo @@ENV_HASH@@ > /opt/env_hash
echo @@BUILD_COMMIT@@ > /opt/build_commit

@@PURGE_BUILD_TOOLS@@

%environment
# Append venv dir to PATH so the
# environment is active by default
export PATH="/opt/env/bin:$PATH"

@@EXTRA_ENV@@
14 changes: 14 additions & 0 deletions pipeline/config/config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,9 @@ run_dir: /path/to/run
# Can be shared across experiments with identical waveform parameters.
# waveforms_dir: /path/to/shared/waveforms

# Snakemake and job logs. Defaults to {run_dir}/logs if omitted.
# log_dir: /path/to/logs

# --- Interferometers ---------------------------------------------------------
ifos:
- H1
Expand Down Expand Up @@ -101,6 +104,16 @@ container_source: local
# own (/igwn/cit/staging/<USER>). Anyone's can be read.
osdf_staging_dir: null

# --- Rule placement ----------------------------------------------------------
# Whether rules run on the node snakemake runs on rather than as batch jobs.
# On LDG, both should be true, though the second is allowed to be false.
# On Delta, both should be false.
#
# Whether train and export, which need a GPU, run locally
gpu_rules_local: true
# Whether the aggregation rules run locally
aggregate_rules_local: true

# --- Resources ---------------------------------------------------------------
# Memory (MB) and walltime (minutes) for rules submitted as batch jobs,
# under slurm or condor. Rules and keys not listed here use the profile's
Expand Down Expand Up @@ -205,6 +218,7 @@ integration_window_length: 1.5 # seconds
cluster_window_length: 8.0 # seconds
Tb: 31536000.0 # target background livetime (seconds)
zero_lag: false # also analyze unshifted (zero-lag) data
return_timeseries: false # also write the network's output timeseries

# Triton server.
model_name: aframe-stream
Expand Down
5 changes: 3 additions & 2 deletions pipeline/profiles/ldg/config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -66,8 +66,9 @@ apptainer-args: >-
default-resources:
# Forward environment variables from the submit node.
# HOME is not accessible on the execute node, and the bearer
# token is delivered by condor.
getenv: AFRAME_*, PATH, KRB5*
# token is delivered by condor. Fetching proprietary data needs the
# datafind server.
getenv: AFRAME_*, PATH, KRB5*, GWDATAFIND_SERVER

# Pass through authentication and account info as raw job ad attributes
# using the classad_ prefix. The plugin supports passing attributes
Expand Down
14 changes: 7 additions & 7 deletions pipeline/resources.smk
Original file line number Diff line number Diff line change
Expand Up @@ -28,9 +28,9 @@ def container(project):
`osdf_staging_dir` (your own staging directory by default) through the
AP's `/osdf` mount.
"""
if config.get("container_source", "local") == "osdf" and project in OSDF_PROJECTS:
if config["container_source"] == "osdf" and project in OSDF_PROJECTS:
name = image_name(project, env_hash(project))
source = config.get("osdf_staging_dir") or staging_dir()
source = config["osdf_staging_dir"] or staging_dir()
return f"/osdf{source}/{name}"
return os.path.join(os.getenv("AFRAME_CONTAINER_ROOT", ""), f"{project}.sif")

Expand Down Expand Up @@ -78,9 +78,9 @@ def rule_resources(name):
default-resources. With `epnfs`, condor jobs only match execute points
that mount the AP's /home.
"""
res = config.get("resources", {}).get(name, {})
res = config["resources"].get(name, {})
out = {}
if config.get("epnfs"):
if config["epnfs"]:
out["requirements"] = "TARGET.EPNFS =?= True"
if "mem_mb" in res:
out["mem_mb"] = out["htcondor_request_mem_mb"] = res["mem_mb"]
Expand All @@ -99,14 +99,14 @@ def gpu_resources():
instead.
"""
res = {
"slurm_partition": config.get("inference_partition", "gpuA40x4"),
"slurm_partition": config["inference_partition"],
"gpu": 1,
"request_gpus": 1,
}
if config.get("inference_backend", "export") == "aoti":
if config["inference_backend"] == "aoti":
res["require_gpus"] = f"Capability == {config['aoti_gpu_capability']}"
else:
res["gpus_minimum_capability"] = config["gpu_min_capability"]
if config.get("gpu_min_memory_mb"):
if config["gpu_min_memory_mb"]:
res["gpus_minimum_memory"] = f"{config['gpu_min_memory_mb']}M"
return res
Loading
Loading