diff --git a/pipeline/condor/job_wrapper.sh b/pipeline/condor/job_wrapper.sh index 85039dc6d..e68e92676 100755 --- a/pipeline/condor/job_wrapper.sh +++ b/pipeline/condor/job_wrapper.sh @@ -14,6 +14,12 @@ export HOME=$PWD export TORCHINDUCTOR_CACHE_DIR=$PWD/.cache/torchinductor +# Condor writes the job's token here. Set the variable explicitly so +# that services can find it. +if [ -f "$_CONDOR_CREDS/scitokens.use" ]; then + export BEARER_TOKEN_FILE=$_CONDOR_CREDS/scitokens.use +fi + # The image installs the projects against /opt/aframe, which holds the code # from when it was built. PYTHONPATH comes first, so the run's copy wins. for package in "$PWD"/code/projects/*/ "$PWD"/code/libs/*/; do diff --git a/pipeline/profiles/ldg-transfer/config.yaml b/pipeline/profiles/ldg-transfer/config.yaml index 3ef5d6fd4..f228adc71 100644 --- a/pipeline/profiles/ldg-transfer/config.yaml +++ b/pipeline/profiles/ldg-transfer/config.yaml @@ -71,20 +71,20 @@ apptainer-args: >- --bind ${AFRAME_REPO:?set at startup by set_container_binds in pipeline/resources.smk}:/opt/aframe --home $HOME -# Passed to every job's environment with the submit node's values. -# Fetching proprietary data needs the datafind server. -envvars: - - GWDATAFIND_SERVER - default-resources: # The executable that runs snakemake for the job in its image job_wrapper: code/pipeline/condor/job_wrapper.sh + # Every job's environment, using the submit node's values. Fetching + # proprietary data needs the datafind server. Not set via `envvars`, which + # the plugin passes with literal quotes around each value. + environment: GWDATAFIND_SERVER=$ENV(GWDATAFIND_SERVER) # Every job gets the run's copy of the code, and returns its logs htcondor_transfer_input_files: code htcondor_transfer_output_files: logs # As in the ldg profile. Condor writes the token to - # $_CONDOR_CREDS/scitokens.use, which IGWN scitoken clients discover. + # $_CONDOR_CREDS/scitokens.use, and job_wrapper.sh sets BEARER_TOKEN_FILE + # to it. classad_OAuthServicesNeeded: scitokens classad_AcctGroup: $ENV(LIGO_GROUP) classad_AcctGroupUser: $ENV(LIGO_USERNAME) diff --git a/pipeline/resources.smk b/pipeline/resources.smk index b043984d2..626a3883f 100644 --- a/pipeline/resources.smk +++ b/pipeline/resources.smk @@ -208,13 +208,14 @@ def check_gpus(): ) -def rule_resources(name, project): +def rule_resources(name, project, local_pool=False): """Memory and walltime for rule `name`, from the config's `resources`, and for a condor job without a shared filesystem, `project`'s image. Anything a rule doesn't set there fall back to the profile's default-resources. With `epnfs`, condor jobs only match execute points - that mount the AP's /home. + that mount the AP's /home. With `local_pool`, they only match the AP's + local pool rather than glideins, for jobs that read the site's frames. """ res = config["resources"].get(name, {}) out = {} @@ -223,8 +224,14 @@ def rule_resources(name, project): if not SHARED_FS and not workflow.remote_exec: out["universe"] = "container" out["container_image"] = job_image(project) + requirements = [] if config["epnfs"]: - out["requirements"] = "TARGET.EPNFS =?= True" + requirements.append("TARGET.EPNFS =?= True") + if local_pool: + # Glideins advertise the site they run at + requirements.append("isUndefined(TARGET.GLIDEIN_Site)") + if requirements: + out["requirements"] = " && ".join(requirements) if "mem_mb" in res: out["mem_mb"] = out["htcondor_request_mem_mb"] = res["mem_mb"] if "runtime" in res: diff --git a/projects/data/data.smk b/projects/data/data.smk index f43c58b62..cf89c5287 100644 --- a/projects/data/data.smk +++ b/projects/data/data.smk @@ -255,8 +255,15 @@ rule fetch_background: DATA_CONTAINER # `fetch` downloads with nproc=3 threads: 4 + # Proprietary frames (full channel names) are only readable on the site's + # own nodes, not on glideins. Open data (bare IFO names) can come from + # anywhere. resources: - **rule_resources("fetch_background", "data"), + **rule_resources( + "fetch_background", + "data", + local_pool=any(":" in channel for channel in config["channels"]), + ), params: channels=_fmt_list(config["channels"]), sample_rate=config["sample_rate"],