Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions pipeline/condor/job_wrapper.sh
Original file line number Diff line number Diff line change
Expand Up @@ -14,6 +14,12 @@
export HOME=$PWD
export TORCHINDUCTOR_CACHE_DIR=$PWD/.cache/torchinductor

# Condor writes the job's token here. Set the variable explicitly so
# that services can find it.
if [ -f "$_CONDOR_CREDS/scitokens.use" ]; then
export BEARER_TOKEN_FILE=$_CONDOR_CREDS/scitokens.use
fi

# The image installs the projects against /opt/aframe, which holds the code
# from when it was built. PYTHONPATH comes first, so the run's copy wins.
for package in "$PWD"/code/projects/*/ "$PWD"/code/libs/*/; do
Expand Down
12 changes: 6 additions & 6 deletions pipeline/profiles/ldg-transfer/config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -71,20 +71,20 @@ apptainer-args: >-
--bind ${AFRAME_REPO:?set at startup by set_container_binds in pipeline/resources.smk}:/opt/aframe
--home $HOME

# Passed to every job's environment with the submit node's values.
# Fetching proprietary data needs the datafind server.
envvars:
- GWDATAFIND_SERVER

default-resources:
# The executable that runs snakemake for the job in its image
job_wrapper: code/pipeline/condor/job_wrapper.sh
# Every job's environment, using the submit node's values. Fetching
# proprietary data needs the datafind server. Not set via `envvars`, which
# the plugin passes with literal quotes around each value.
environment: GWDATAFIND_SERVER=$ENV(GWDATAFIND_SERVER)
# Every job gets the run's copy of the code, and returns its logs
htcondor_transfer_input_files: code
htcondor_transfer_output_files: logs

# As in the ldg profile. Condor writes the token to
# $_CONDOR_CREDS/scitokens.use, which IGWN scitoken clients discover.
# $_CONDOR_CREDS/scitokens.use, and job_wrapper.sh sets BEARER_TOKEN_FILE
# to it.
classad_OAuthServicesNeeded: scitokens
classad_AcctGroup: $ENV(LIGO_GROUP)
classad_AcctGroupUser: $ENV(LIGO_USERNAME)
Expand Down
13 changes: 10 additions & 3 deletions pipeline/resources.smk
Original file line number Diff line number Diff line change
Expand Up @@ -208,13 +208,14 @@ def check_gpus():
)


def rule_resources(name, project):
def rule_resources(name, project, local_pool=False):
"""Memory and walltime for rule `name`, from the config's `resources`,
and for a condor job without a shared filesystem, `project`'s image.

Anything a rule doesn't set there fall back to the profile's
default-resources. With `epnfs`, condor jobs only match execute points
that mount the AP's /home.
that mount the AP's /home. With `local_pool`, they only match the AP's
local pool rather than glideins, for jobs that read the site's frames.
"""
res = config["resources"].get(name, {})
out = {}
Expand All @@ -223,8 +224,14 @@ def rule_resources(name, project):
if not SHARED_FS and not workflow.remote_exec:
out["universe"] = "container"
out["container_image"] = job_image(project)
requirements = []
if config["epnfs"]:
out["requirements"] = "TARGET.EPNFS =?= True"
requirements.append("TARGET.EPNFS =?= True")
if local_pool:
# Glideins advertise the site they run at
requirements.append("isUndefined(TARGET.GLIDEIN_Site)")
if requirements:
out["requirements"] = " && ".join(requirements)
if "mem_mb" in res:
out["mem_mb"] = out["htcondor_request_mem_mb"] = res["mem_mb"]
if "runtime" in res:
Expand Down
9 changes: 8 additions & 1 deletion projects/data/data.smk
Original file line number Diff line number Diff line change
Expand Up @@ -255,8 +255,15 @@ rule fetch_background:
DATA_CONTAINER
# `fetch` downloads with nproc=3
threads: 4
# Proprietary frames (full channel names) are only readable on the site's
# own nodes, not on glideins. Open data (bare IFO names) can come from
# anywhere.
resources:
**rule_resources("fetch_background", "data"),
**rule_resources(
"fetch_background",
"data",
local_pool=any(":" in channel for channel in config["channels"]),
),
params:
channels=_fmt_list(config["channels"]),
sample_rate=config["sample_rate"],
Expand Down
Loading