`make published-case` reads a GA4GH phenopacket from Monarch's Phenopacket Store and takes two things from it verbatim: the HPO terms the authors reported and the variant they called causal. The default is the TGFBR2 proband from Loeys et al., Nat Genet 2005 (doi:10.1038/ng1511), the paper that first defined Loeys-Dietz syndrome -- 30 reported terms and NM_003242.6:c.1069G>T p.(Gly357Trp). The rest of that patient's genome is not public, so background variants come from GIAB HG002 around the locus. They are drawn from coding exons where possible, via Ensembl's REST API: of ~4,000 HG002 variants in the window only 9 are coding, so a random sample is entirely intronic, the consequence filter discards all of it, and the causal variant is left as the only candidate -- a funnel that proves nothing. The real run ranks TGFBR2 first at 0.897 against an OSBPL10 missense at 0.547. Both are rare missense variants the model scores identically (0.887); only the phenotype separates them, which is the argument for phenotype-driven triage in one table. Documented with three caveats rather than left implicit: the phenotype match is partly circular because HPO's gene annotations are themselves curated from published cases; rarity contributes nothing without the VEP cache (--af_gnomade is rejected with --database, and plain --af returns nothing even for rs429358 at ~15% global frequency); and one healthy genome is not a diagnostic exome.
95 lines
4.2 KiB
Makefile
95 lines
4.2 KiB
Makefile
.PHONY: up down clean migrate test lint data hpo demo-case published-case training-set train loader pipeline annotate images kind serverless-deploy serverless-destroy gcp-configure gcp-secrets
|
|
|
|
VCF ?= data/example.vcf.gz
|
|
MLFLOW_URI ?= http://localhost:5001
|
|
TAG ?= latest
|
|
# The loader container reaches docker-compose's Postgres through the host.
|
|
HOST_DB_URL ?= postgresql://rarelens:[email protected]:5432/rarelens
|
|
|
|
up:
|
|
docker compose up -d --build
|
|
|
|
down:
|
|
docker compose down
|
|
|
|
clean: ## also deletes the Postgres and MLflow volumes
|
|
docker compose down -v
|
|
|
|
migrate:
|
|
docker compose exec api alembic upgrade head
|
|
|
|
test:
|
|
cd api && uv run --extra dev pytest -q
|
|
cd ml && uv run --extra dev pytest -q
|
|
cd pipeline && uv run --no-project --with-requirements requirements.txt --with pytest --with pgserver pytest -q tests
|
|
cd web && npm test
|
|
|
|
lint:
|
|
cd api && uv run --extra dev ruff check . && uv run --extra dev mypy app
|
|
cd web && npm run check
|
|
|
|
data: ## download the public demo slice: GIAB HG002 + ClinVar, chr22 (see docs/data.md)
|
|
scripts/fetch-demo-data.sh
|
|
|
|
hpo: ## load HPO gene-to-phenotype annotations, which the ranking matches against
|
|
scripts/load-hpo.py
|
|
|
|
demo-case: ## build the simulated proband: GIAB background + one ClinVar pathogenic variant
|
|
scripts/make-demo-case.sh
|
|
|
|
published-case: ## build a case from a published patient: a GA4GH phenopacket + GIAB background
|
|
scripts/make-published-case.py
|
|
|
|
training-set: ## build a ClinVar training table, shaped like VEP --tab output
|
|
scripts/make-training-set.sh
|
|
|
|
train: ## train the pathogenicity model and point the production alias at it (needs `make up`)
|
|
cd ml && MLFLOW_TRACKING_URI=$(MLFLOW_URI) uv run --extra dev \
|
|
python -m rarelens_ml.train --tsv ../data/clinvar-training.vep.tsv --register
|
|
|
|
loader:
|
|
docker build -t rarelens/loader:dev -f pipeline/loader.Dockerfile pipeline
|
|
|
|
pipeline: loader ## dry run: annotate $(VCF) without touching the database
|
|
cd pipeline && nextflow run main.nf -profile docker --vcf ../$(VCF)
|
|
|
|
annotate: loader ## make annotate JOB=<job id> [VCF=data/x.vcf.gz]
|
|
@test -n "$(JOB)" || (echo "usage: make annotate JOB=<job id> [VCF=...]"; exit 1)
|
|
cd pipeline && DATABASE_URL=$(HOST_DB_URL) nextflow run main.nf -profile docker \
|
|
--vcf ../$(VCF) --job_id $(JOB)
|
|
|
|
images:
|
|
docker build -t rarelens-api:dev api
|
|
docker build -t rarelens-web:dev web
|
|
|
|
kind: images
|
|
kind create cluster --name rarelens 2>/dev/null || true
|
|
kind load docker-image rarelens-api:dev rarelens-web:dev --name rarelens
|
|
kubectl apply -k infra/k8s/overlays/local
|
|
kubectl -n rarelens rollout status deploy/postgres deploy/api deploy/web
|
|
@echo "kubectl -n rarelens port-forward svc/web 8080:80 (UI)"
|
|
@echo "kubectl -n rarelens port-forward svc/api 8000:80 (API, used by the UI)"
|
|
|
|
serverless-deploy: ## deploy the Cloud Run track: make serverless-deploy PROJECT=<id> [TAG=<sha>]
|
|
@test -n "$(PROJECT)" || (echo "usage: make serverless-deploy PROJECT=<gcp project id> [TAG=<image tag>]"; exit 1)
|
|
@test -n "$$TF_VAR_database_url" || echo "note: TF_VAR_database_url is unset; add -var deploy_cloud_sql=true or export a Postgres URL"
|
|
cd infra/terraform && terraform apply -var project=$(PROJECT) -var image_tag=$(TAG)
|
|
|
|
serverless-destroy: ## tear it all down
|
|
cd infra/terraform && terraform destroy -var project=$(PROJECT) -var deletion_protection=false
|
|
|
|
gcp-configure: ## one-time: write your GCP project id into the gcp overlay and Argo manifests
|
|
@test -n "$(PROJECT)" || (echo "usage: make gcp-configure PROJECT=<gcp project id>"; exit 1)
|
|
grep -rl __GCP_PROJECT__ infra/k8s/overlays/gcp infra/argo-workflows \
|
|
| xargs sed -i.bak "s/__GCP_PROJECT__/$(PROJECT)/g"
|
|
find infra -name '*.bak' -delete
|
|
|
|
gcp-secrets: ## after terraform apply: copy DB URLs from Secret Manager into k8s secrets
|
|
@test -n "$(PROJECT)" || (echo "usage: make gcp-secrets PROJECT=<gcp project id>"; exit 1)
|
|
kubectl -n rarelens create secret generic api-secrets --dry-run=client -o yaml \
|
|
--from-literal=DATABASE_URL="$$(gcloud secrets versions access latest --project $(PROJECT) --secret rarelens-api-database-url)" \
|
|
| kubectl apply -f -
|
|
kubectl -n rarelens create secret generic pipeline-secrets --dry-run=client -o yaml \
|
|
--from-literal=DATABASE_URL="$$(gcloud secrets versions access latest --project $(PROJECT) --secret DATABASE_URL)" \
|
|
| kubectl apply -f -
|