Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
6 changes: 6 additions & 0 deletions GTEx/run/run_submission_models.sh
Original file line number Diff line number Diff line change
Expand Up @@ -5,4 +5,10 @@ mode="${1:---smoke}"
shift || true
root="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd -P)"

case "${mode}" in
--smoke|full) ;;
--full) mode="full" ;;
*) echo "usage: run_submission_models.sh [--smoke|--full]" >&2; exit 2 ;;
esac

exec "${PYTHON_BIN:-python3}" "${root}/src/dispatch_gtex_submission.py" "$@" -- "${mode}"
41 changes: 35 additions & 6 deletions GTEx/run/submit_submission_models_cluster_apptainer.sh
Original file line number Diff line number Diff line change
Expand Up @@ -25,7 +25,9 @@ enabled model and broad-tissue partition (currently 990 tasks). Without
--submit, it prints the delegated command and never calls qsub. --smoke runs
the small declared smoke reproduction as one Apptainer job only with --submit.
Use --model-id to select one or more enabled GTEx model IDs and --tissue-id to
select one configured broad-tissue partition.
select one configured broad-tissue partition. HZ2 can be combined with other
model IDs; it is submitted as its dedicated consensus task while the remaining
IDs are submitted through the standard model-by-tissue array.

Required environment: APPTAINER_IMAGE, DIG_REPO, SUBMISSION_WORK_DIR.
Full mode additionally requires GTEX_V10_COUNTS_GCT,
Expand Down Expand Up @@ -77,6 +79,34 @@ if [[ -n "${tissue_id}" ]]; then
|| { echo "Unknown GTEx broad tissue ID: ${tissue_id}" >&2; exit 2; }
fi

# HZ2 has a distinct input contract and no broad-tissue partition. Keep it as
# a dedicated job, but allow callers to select it alongside standard models in
# one command. The standard array receives only the remaining model IDs.
hz2_requested=0
standard_model_ids=()
if [[ -z "${model_id}" ]]; then
hz2_requested=1
else
IFS=',' read -r -a requested_model_ids <<< "${model_id}"
for requested_model in "${requested_model_ids[@]}"; do
requested_model="${requested_model//[[:space:]]/}"
[[ -n "${requested_model}" ]] || continue
if [[ "${requested_model}" == "HZ2" ]]; then
hz2_requested=1
else
standard_model_ids+=("${requested_model}")
fi
done
fi
if [[ ${hz2_requested} -eq 1 && -n "${tissue_id}" ]]; then
echo "HZ2 has no broad-tissue partition; omit --tissue-id when selecting HZ2" >&2
exit 2
fi
standard_model_id=""
if [[ ${#standard_model_ids[@]} -gt 0 ]]; then
standard_model_id="$(IFS=,; echo "${standard_model_ids[*]}")"
fi

submit_hz2() {
local runner="${root}/run/run_hz2_task_apptainer.sh"
[[ -n "${APPTAINER_IMAGE:-}" && -f "${APPTAINER_IMAGE}" ]] || { echo "--submit requires APPTAINER_IMAGE" >&2; return 1; }
Expand All @@ -86,8 +116,7 @@ submit_hz2() {
export_vars="GTEX_WRAPPER_ROOT=${root},SUBMISSION_WORK_DIR=${SUBMISSION_WORK_DIR},DIG_REPO=${DIG_REPO},APPTAINER_IMAGE=${APPTAINER_IMAGE},GTEX_V8_TPM_GCT=${GTEX_V8_TPM_GCT},GTEX_V8_SAMPLE_ATTRIBUTES_TSV=${GTEX_V8_SAMPLE_ATTRIBUTES_TSV},GTEX_V8_SUBJECT_PHENOTYPES_TSV=${GTEX_V8_SUBJECT_PHENOTYPES_TSV}"
"${QSUB_BIN:-qsub}" -N "${array_job_name}_hz2" -o "${SUBMISSION_WORK_DIR}/qsub_logs/gtex_hz2.out" -e "${SUBMISSION_WORK_DIR}/qsub_logs/gtex_hz2.err" -l "h_vmem=${array_memory},h_rt=${array_walltime}" -v "${export_vars}" "${runner}" full
}
if [[ "${model_id}" == "HZ2" ]]; then
[[ -z "${tissue_id}" ]] || { echo "HZ2 has no broad-tissue partition; omit --tissue-id" >&2; exit 2; }
if [[ ${hz2_requested} -eq 1 && -z "${standard_model_id}" ]]; then
if [[ ${submit} -eq 0 ]]; then echo "Would submit one HZ2 consensus task through run_hz2_task_apptainer.sh. Set --submit to call qsub."; exit 0; fi
submit_hz2; exit $?
fi
Expand All @@ -97,13 +126,13 @@ command=(env "WORK_ROOT=${SUBMISSION_WORK_DIR}" "GTEX_OUT_ROOT=${SUBMISSION_WORK
"GTEX_V10_COUNTS_GCT=${GTEX_V10_COUNTS_GCT:-}" "GTEX_V10_SAMPLE_ATTRIBUTES_TSV=${GTEX_V10_SAMPLE_ATTRIBUTES_TSV:-}" "GTEX_V10_SUBJECT_PHENOTYPES_TSV=${GTEX_V10_SUBJECT_PHENOTYPES_TSV:-}" \
"GTEX_V8_COUNTS_GCT=${GTEX_V8_COUNTS_GCT:-}" "GTEX_V8_SAMPLE_ATTRIBUTES_TSV=${GTEX_V8_SAMPLE_ATTRIBUTES_TSV:-}" "GTEX_V8_SUBJECT_PHENOTYPES_TSV=${GTEX_V8_SUBJECT_PHENOTYPES_TSV:-}" "GTEX_V8_HUMAN_GENE_INFO=${GTEX_V8_HUMAN_GENE_INFO:-}" "GTEX_GTF=${GTEX_GTF:-}" \
"${legacy_launcher}" --submit)
[[ -n "${model_id}" ]] && command+=(--model_id "${model_id}")
[[ -n "${standard_model_id}" ]] && command+=(--model_id "${standard_model_id}")
[[ -n "${tissue_id}" ]] && command+=(--tissue_id "${tissue_id}")
if [[ ${submit} -eq 0 ]]; then
if [[ ${hz2_requested} -eq 1 ]]; then echo "Would also submit one HZ2 consensus task through run_hz2_task_apptainer.sh."; fi
printf 'Would submit GTEx array: '
printf '%q ' "${command[@]}"
printf '\nSet --submit to call qsub.\n'
if [[ -z "${model_id}" ]]; then echo "Would also submit one HZ2 consensus task through run_hz2_task_apptainer.sh."; fi
exit 0
fi

Expand All @@ -112,6 +141,6 @@ fi
for variable in GTEX_V10_COUNTS_GCT GTEX_V10_SAMPLE_ATTRIBUTES_TSV GTEX_V10_SUBJECT_PHENOTYPES_TSV GTEX_V8_COUNTS_GCT GTEX_V8_SAMPLE_ATTRIBUTES_TSV GTEX_V8_SUBJECT_PHENOTYPES_TSV GTEX_V8_HUMAN_GENE_INFO GTEX_GTF; do
[[ -n "${!variable:-}" && -f "${!variable}" ]] || { echo "--submit requires existing ${variable}" >&2; exit 1; }
done
if [[ -z "${model_id}" ]]; then submit_hz2; fi
if [[ ${hz2_requested} -eq 1 ]]; then submit_hz2; fi
mkdir -p "${SUBMISSION_WORK_DIR}/qsub_logs"
exec "${command[@]}"
6 changes: 6 additions & 0 deletions GTEx/tests/test_hz2_consensus_smoke.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,6 +34,12 @@ def test_hz2_smoke_uses_dig_and_writes_deterministic_consensus() -> None:
"run_summary.txt",
):
assert (gmt.parent / name).is_file()
dapper_gmt = gmt.with_name("genesets.dapper-ids.gmt")
assert dapper_gmt.is_file()
assert all(
line.startswith("dapper:GeneSet.")
for line in dapper_gmt.read_text(encoding="utf-8").splitlines()
)
summary = json.loads((gmt.parent / "run_summary.json").read_text(encoding="utf-8"))
assert summary["support_fraction"] == 0.25
assert {"median", "min", "max"} == set(summary["sample_up_genes"])
Expand Down
18 changes: 18 additions & 0 deletions GTEx/tests/test_modern_submission.py
Original file line number Diff line number Diff line change
Expand Up @@ -79,6 +79,24 @@ def test_apptainer_scheduler_wrapper_forwards_full_task_filters(self) -> None:
self.assertIn("GTEX_ARRAY_WALLTIME=48:00:00", completed.stdout)
self.assertIn("GTEX_JOB_NAME=gtex_filtered_test", completed.stdout)

def test_apptainer_scheduler_wrapper_combines_hz2_with_standard_models(self) -> None:
root = Path(__file__).resolve().parents[1]
with tempfile.TemporaryDirectory() as temp:
completed = subprocess.run(
[
"bash", "run/submit_submission_models_cluster_apptainer.sh",
"--full", "--model-id", "HZ1,HZ2",
],
cwd=root,
env={**os.environ, "SUBMISSION_WORK_DIR": temp},
text=True,
capture_output=True,
)
self.assertEqual(completed.returncode, 0, completed.stdout + completed.stderr)
self.assertIn("Would also submit one HZ2 consensus task", completed.stdout)
self.assertIn("--model_id HZ1", completed.stdout)
self.assertNotIn("--model_id HZ1,HZ2", completed.stdout)

def test_full_contract_covers_all_enabled_model_tissue_pairs(self) -> None:
root = Path(__file__).resolve().parents[1]
with (root / "config/task_manifest.tsv").open(encoding="utf-8", newline="") as handle:
Expand Down
8 changes: 6 additions & 2 deletions HuBMAP/run/submit_submission_models_cluster_apptainer.sh
Original file line number Diff line number Diff line change
Expand Up @@ -9,9 +9,11 @@ model_id=""

usage() {
cat <<'EOF'
Usage: submit_submission_models_cluster_apptainer.sh [--model-id ID[,ID...]] [--submit]
Usage: submit_submission_models_cluster_apptainer.sh [--full] [--model-id ID[,ID...]] [--submit]

Writes the HZ1/HZ2 worklist unless --submit is supplied. Set
HuBMAP currently supports full execution only; --full is accepted explicitly
for consistency with the other library launchers. Writes the HZ1/HZ2 worklist
unless --submit is supplied. Set
SUBMISSION_WORK_DIR outside the HuBMAP checkout, APPTAINER_IMAGE, DIG_REPO,
HUBMAP_RAW_ASCTB_DIR, and HUBMAP_HUMAN_GENE_INFO. HZ2 retains its declared
GeneShot network dependency.
Expand All @@ -20,6 +22,8 @@ EOF

while [[ $# -gt 0 ]]; do
case "$1" in
--full) ;;
--smoke) echo "HuBMAP has no cluster smoke mode; use reproduction/reproduce.sh --smoke" >&2; exit 2 ;;
--submit) submit=1 ;;
--model-id) [[ $# -ge 2 ]] || { echo "Missing value for --model-id" >&2; exit 2; }; model_id="$2"; shift ;;
-h|--help) usage; exit 0 ;;
Expand Down
18 changes: 16 additions & 2 deletions HuBMAP/src/run_hubmap_hz_model.py
Original file line number Diff line number Diff line change
Expand Up @@ -154,7 +154,11 @@ def write_model_sidecar(
"extractor_name": "unsigned_term_gene",
"parameters": {
"term_prefix": "HuBMAP",
"signature_name": "HuBMAP_ASCTB" if model_id == "HZ1" else "HuBMAP_ASCTB_augmented",
"signature_name": (
"HuBMAP ASCT+B gene-set library (HZ1)"
if model_id == "HZ1"
else "HuBMAP ASCT+B augmented gene-set library (HZ2)"
),
"gmt_min_genes": 5,
"augmentation_threshold": manifest_value(settings, "workflow_augmentation_threshold", ""),
"cap_multiplier": manifest_value(settings, "workflow_cap_multiplier", ""),
Expand Down Expand Up @@ -253,7 +257,11 @@ def build_extractor_cmd(
provenance_mirror_local_prefix: str | None,
provenance_mirror_remote_prefix: str | None,
) -> list[str]:
signature_name = "HuBMAP_ASCTB" if model_id == "HZ1" else "HuBMAP_ASCTB_augmented"
signature_name = (
"HuBMAP ASCT+B gene-set library (HZ1)"
if model_id == "HZ1"
else "HuBMAP ASCT+B augmented gene-set library (HZ2)"
)
cmd = [
python_bin,
"-m",
Expand Down Expand Up @@ -286,6 +294,12 @@ def build_extractor_cmd(
"true",
"--emit_small_gene_sets",
"false",
"--dapper_gene_member_prefix",
"HGNC.SYMBOL",
"--dapper_gene_member_prefix_uri",
"https://identifiers.org/hgnc.symbol:",
"--dapper_row_display_separator",
"_",
]
if provenance_mirror_local_prefix:
cmd.extend(["--provenance_mirror_local_prefix", provenance_mirror_local_prefix])
Expand Down
34 changes: 31 additions & 3 deletions HuBMAP/tests/test_modern_submission.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,6 +6,9 @@
import sys
import unittest
from pathlib import Path
from tempfile import TemporaryDirectory

import yaml


ROOT = Path(__file__).resolve().parents[1]
Expand All @@ -14,6 +17,21 @@


class HuBMAPModernSubmissionTests(unittest.TestCase):
def test_cluster_launcher_accepts_explicit_full_and_hyphenated_model_id(self) -> None:
with TemporaryDirectory() as temp_dir:
result = subprocess.run(
[
"bash", str(ROOT / "run/submit_submission_models_cluster_apptainer.sh"),
"--full", "--model-id", "HZ1",
],
cwd=ROOT,
env={**os.environ, "SUBMISSION_WORK_DIR": temp_dir},
capture_output=True,
text=True,
)
self.assertEqual(result.returncode, 0, result.stderr + result.stdout)
self.assertIn("--model_id HZ1", result.stdout)

def test_declared_models_and_outputs_are_complete(self) -> None:
with (ROOT / "config/model_list.tsv").open(encoding="utf-8", newline="") as handle:
models = list(csv.DictReader(handle, delimiter="\t"))
Expand All @@ -27,18 +45,28 @@ def test_declared_models_and_outputs_are_complete(self) -> None:

def test_wrapper_declares_stable_hubmap_collection_names(self) -> None:
source = (ROOT / "src/run_hubmap_hz_model.py").read_text(encoding="utf-8")
self.assertIn('"HuBMAP_ASCTB" if model_id == "HZ1" else "HuBMAP_ASCTB_augmented"', source)
self.assertIn('"HuBMAP ASCT+B gene-set library (HZ1)"', source)
self.assertIn('"--signature_name"', source)

@unittest.skipUnless((DIG / "src/geneset_extractors").is_dir(), "requires sibling DIG checkout")
def test_smoke_reproduction_runs_hz1_without_network(self) -> None:
from tempfile import TemporaryDirectory
with TemporaryDirectory() as temp_dir:
result = subprocess.run(["bash", str(ROOT / "reproduction/reproduce.sh"), "--smoke"], cwd=ROOT, env={**os.environ, "SUBMISSION_WORK_DIR": temp_dir, "DIG_REPO": str(DIG), "PYTHON_BIN": sys.executable}, capture_output=True, text=True)
self.assertEqual(result.returncode, 0, result.stderr + result.stdout)
extractor = Path(temp_dir, "smoke/genesets/all_signatures/models/HZ1/extractor")
self.assertTrue((extractor / "genesets.gmt").is_file())
self.assertIn("name: HuBMAP_ASCTB", (extractor / "geneset.provenance.dapper.yaml").read_text(encoding="utf-8"))
dapper = yaml.safe_load((extractor / "geneset.provenance.dapper.yaml").read_text(encoding="utf-8"))
collection = dapper["gene_set_collections"][0]
self.assertEqual(collection["name"], "HuBMAP ASCT+B gene-set library (HZ1)")
self.assertTrue((extractor / "genesets.dapper-ids.gmt").is_file())
self.assertEqual(collection["members"], [row["id"] for row in dapper["gene_sets"]])
self.assertEqual(collection["n_genes"], len({gene for row in dapper["gene_sets"] for gene in row["members"]}))
companion = next(node for node in dapper["files"] if node["filename"] == "genesets.dapper-ids.gmt")
self.assertEqual(collection["has_gmt_file"], companion["id"])
for row, line in zip(dapper["gene_sets"], (extractor / "genesets.dapper-ids.gmt").read_text(encoding="utf-8").splitlines(), strict=True):
self.assertEqual(row["gmt_entry"], row["id"])
self.assertEqual(line.split("\t", 1)[0], row["id"])
self.assertEqual(row["in_gmt_file"], companion["id"])


if __name__ == "__main__":
Expand Down
4 changes: 3 additions & 1 deletion LINCS_L1000/tests/test_modern_submission.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,4 +53,6 @@ def test_smoke_reproduction_runs_hz1(self) -> None:
self.assertEqual(result.returncode, 0, result.stderr + result.stdout)
extractor = Path(temp_dir, "smoke/genesets/all_signatures/models/HZ1/extractor")
self.assertTrue((extractor / "genesets.gmt").is_file())
self.assertIn("name: LINCS_L1000_Chem_Pert", (extractor / "geneset.provenance.dapper.yaml").read_text(encoding="utf-8"))
# DAPPER keeps the legacy machine identifier as an alternate ID
# and emits the collection's readable name separately.
self.assertIn("name: LINCS L1000 Chem Pert", (extractor / "geneset.provenance.dapper.yaml").read_text(encoding="utf-8"))
7 changes: 6 additions & 1 deletion MoTrPAC/tests/test_modern_submission.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,8 +38,13 @@ def test_smoke_reproduction_runs_hz1(self) -> None:
self.assertTrue(output.with_name("geneset.provenance.legacy.json").is_file())
self.assertTrue(output.with_name("geneset.provenance.dapper.yaml").is_file())
self.assertFalse(output.with_name("geneset.provenance.json").exists())
dapper_gmt = output.with_name("genesets.dapper-ids.gmt")
self.assertTrue(dapper_gmt.is_file())
self.assertTrue(
all(line.startswith("dapper:GeneSet.") for line in dapper_gmt.read_text(encoding="utf-8").splitlines())
)
dapper = output.with_name("geneset.provenance.dapper.yaml").read_text(encoding="utf-8")
self.assertIn("name: MoTrPAC_Rat_Endurance_Training_HZ1", dapper)
self.assertIn("name: MoTrPAC Rat Endurance Training HZ1", dapper)


if __name__ == "__main__":
Expand Down
20 changes: 14 additions & 6 deletions run/submit_gtex_models_cluster.sh
Original file line number Diff line number Diff line change
Expand Up @@ -45,7 +45,7 @@ FILTER_MODEL_IDS=""
usage() {
cat <<'EOF'
Usage:
./geneset-extractor-dev/run/submit_gtex_models_cluster.sh --submit [--write_model_only|--refresh_metadata_and_provenance] [--model_group AB|AC|HZ] [--tissue_id TISSUE] [--model_id MODEL[,MODEL...]]
./geneset-extractor-dev/run/submit_gtex_models_cluster.sh --full --submit [--write-model-only|--refresh-metadata-and-provenance] [--model-group AB|AC|HZ] [--tissue-id TISSUE] [--model-id MODEL[,MODEL...]]
./geneset-extractor-dev/run/submit_gtex_models_cluster.sh --help

Required environment variables:
Expand All @@ -60,6 +60,7 @@ Optional environment variables:
LOCAL_INPUT_SOURCE_MAP_TSV

Notes:
- Hyphenated options are preferred; legacy underscore spellings remain accepted.
- Use --submit to submit the qsub array.
- Add --write_model_only to write only geneset.model.json sidecars.
- Add --refresh_metadata_and_provenance to patch metadata descriptions and
Expand Down Expand Up @@ -181,28 +182,35 @@ parse_cli() {
SUBMIT_MODE=1
shift
;;
--write_model_only)
--full)
shift
;;
--smoke)
echo "Smoke mode is available through GTEx/run/submit_submission_models_cluster_apptainer.sh" >&2
exit 2
;;
--write_model_only|--write-model-only)
WRITE_MODEL_ONLY=1
shift
;;
--refresh_metadata_and_provenance)
--refresh_metadata_and_provenance|--refresh-metadata-and-provenance)
REFRESH_METADATA_AND_PROVENANCE=1
shift
;;
--model_group)
--model_group|--model-group)
[[ $# -ge 2 ]] || { echo "Missing value for --model_group" >&2; exit 1; }
FILTER_MODEL_GROUP="$(canonicalize_model_group "$2")" || {
echo "Unsupported GTEx model group: $2" >&2
exit 1
}
shift 2
;;
--tissue_id)
--tissue_id|--tissue-id)
[[ $# -ge 2 ]] || { echo "Missing value for --tissue_id" >&2; exit 1; }
FILTER_TISSUE_ID="$2"
shift 2
;;
--model_id)
--model_id|--model-id)
[[ $# -ge 2 ]] || { echo "Missing value for --model_id" >&2; exit 1; }
FILTER_MODEL_IDS="$2"
shift 2
Expand Down
20 changes: 14 additions & 6 deletions run/submit_gtex_models_cluster_apptainer.sh
Original file line number Diff line number Diff line change
Expand Up @@ -59,7 +59,7 @@ FILTER_MODEL_IDS=""
usage() {
cat <<'EOF'
Usage:
./geneset-extractor-dev/run/submit_gtex_models_cluster_apptainer.sh --submit [--write_model_only|--refresh_metadata_and_provenance] [--model_group AB|AC|HZ] [--tissue_id TISSUE] [--model_id MODEL[,MODEL...]]
./geneset-extractor-dev/run/submit_gtex_models_cluster_apptainer.sh --full --submit [--write-model-only|--refresh-metadata-and-provenance] [--model-group AB|AC|HZ] [--tissue-id TISSUE] [--model-id MODEL[,MODEL...]]
./geneset-extractor-dev/run/submit_gtex_models_cluster_apptainer.sh --help

Required environment variables:
Expand All @@ -77,6 +77,7 @@ Optional environment variables:
LOCAL_INPUT_SOURCE_MAP_TSV

Notes:
- Hyphenated options are preferred; legacy underscore spellings remain accepted.
- Use --submit to submit the qsub array.
- Add --write_model_only to write only geneset.model.json sidecars.
- Add --refresh_metadata_and_provenance to patch metadata descriptions and
Expand Down Expand Up @@ -200,28 +201,35 @@ parse_cli() {
SUBMIT_MODE=1
shift
;;
--write_model_only)
--full)
shift
;;
--smoke)
echo "Smoke mode is available through GTEx/run/submit_submission_models_cluster_apptainer.sh" >&2
exit 2
;;
--write_model_only|--write-model-only)
WRITE_MODEL_ONLY=1
shift
;;
--refresh_metadata_and_provenance)
--refresh_metadata_and_provenance|--refresh-metadata-and-provenance)
REFRESH_METADATA_AND_PROVENANCE=1
shift
;;
--model_group)
--model_group|--model-group)
[[ $# -ge 2 ]] || { echo "Missing value for --model_group" >&2; exit 1; }
FILTER_MODEL_GROUP="$(canonicalize_model_group "$2")" || {
echo "Unsupported GTEx model group: $2" >&2
exit 1
}
shift 2
;;
--tissue_id)
--tissue_id|--tissue-id)
[[ $# -ge 2 ]] || { echo "Missing value for --tissue_id" >&2; exit 1; }
FILTER_TISSUE_ID="$2"
shift 2
;;
--model_id)
--model_id|--model-id)
[[ $# -ge 2 ]] || { echo "Missing value for --model_id" >&2; exit 1; }
FILTER_MODEL_IDS="$2"
shift 2
Expand Down
Loading
Loading