Files
rarelens/Makefile
T
Kemal Yaylali 5588c9391d feat(data): build a case from a real published patient
`make published-case` reads a GA4GH phenopacket from Monarch's Phenopacket
Store and takes two things from it verbatim: the HPO terms the authors
reported and the variant they called causal. The default is the TGFBR2
proband from Loeys et al., Nat Genet 2005 (doi:10.1038/ng1511), the paper
that first defined Loeys-Dietz syndrome -- 30 reported terms and
NM_003242.6:c.1069G>T p.(Gly357Trp).

The rest of that patient's genome is not public, so background variants come
from GIAB HG002 around the locus. They are drawn from coding exons where
possible, via Ensembl's REST API: of ~4,000 HG002 variants in the window only
9 are coding, so a random sample is entirely intronic, the consequence filter
discards all of it, and the causal variant is left as the only candidate --
a funnel that proves nothing.

The real run ranks TGFBR2 first at 0.897 against an OSBPL10 missense at
0.547. Both are rare missense variants the model scores identically (0.887);
only the phenotype separates them, which is the argument for phenotype-driven
triage in one table.

Documented with three caveats rather than left implicit: the phenotype match
is partly circular because HPO's gene annotations are themselves curated from
published cases; rarity contributes nothing without the VEP cache
(--af_gnomade is rejected with --database, and plain --af returns nothing
even for rs429358 at ~15% global frequency); and one healthy genome is not a
diagnostic exome.
2026-09-12 10:27:08 +01:00

95 lines
4.2 KiB
Makefile

.PHONY: up down clean migrate test lint data hpo demo-case published-case training-set train loader pipeline annotate images kind serverless-deploy serverless-destroy gcp-configure gcp-secrets
VCF ?= data/example.vcf.gz
MLFLOW_URI ?= http://localhost:5001
TAG ?= latest
# The loader container reaches docker-compose's Postgres through the host.
HOST_DB_URL ?= postgresql://rarelens:[email protected]:5432/rarelens
up:
docker compose up -d --build
down:
docker compose down
clean: ## also deletes the Postgres and MLflow volumes
docker compose down -v
migrate:
docker compose exec api alembic upgrade head
test:
cd api && uv run --extra dev pytest -q
cd ml && uv run --extra dev pytest -q
cd pipeline && uv run --no-project --with-requirements requirements.txt --with pytest --with pgserver pytest -q tests
cd web && npm test
lint:
cd api && uv run --extra dev ruff check . && uv run --extra dev mypy app
cd web && npm run check
data: ## download the public demo slice: GIAB HG002 + ClinVar, chr22 (see docs/data.md)
scripts/fetch-demo-data.sh
hpo: ## load HPO gene-to-phenotype annotations, which the ranking matches against
scripts/load-hpo.py
demo-case: ## build the simulated proband: GIAB background + one ClinVar pathogenic variant
scripts/make-demo-case.sh
published-case: ## build a case from a published patient: a GA4GH phenopacket + GIAB background
scripts/make-published-case.py
training-set: ## build a ClinVar training table, shaped like VEP --tab output
scripts/make-training-set.sh
train: ## train the pathogenicity model and point the production alias at it (needs `make up`)
cd ml && MLFLOW_TRACKING_URI=$(MLFLOW_URI) uv run --extra dev \
python -m rarelens_ml.train --tsv ../data/clinvar-training.vep.tsv --register
loader:
docker build -t rarelens/loader:dev -f pipeline/loader.Dockerfile pipeline
pipeline: loader ## dry run: annotate $(VCF) without touching the database
cd pipeline && nextflow run main.nf -profile docker --vcf ../$(VCF)
annotate: loader ## make annotate JOB=<job id> [VCF=data/x.vcf.gz]
@test -n "$(JOB)" || (echo "usage: make annotate JOB=<job id> [VCF=...]"; exit 1)
cd pipeline && DATABASE_URL=$(HOST_DB_URL) nextflow run main.nf -profile docker \
--vcf ../$(VCF) --job_id $(JOB)
images:
docker build -t rarelens-api:dev api
docker build -t rarelens-web:dev web
kind: images
kind create cluster --name rarelens 2>/dev/null || true
kind load docker-image rarelens-api:dev rarelens-web:dev --name rarelens
kubectl apply -k infra/k8s/overlays/local
kubectl -n rarelens rollout status deploy/postgres deploy/api deploy/web
@echo "kubectl -n rarelens port-forward svc/web 8080:80 (UI)"
@echo "kubectl -n rarelens port-forward svc/api 8000:80 (API, used by the UI)"
serverless-deploy: ## deploy the Cloud Run track: make serverless-deploy PROJECT=<id> [TAG=<sha>]
@test -n "$(PROJECT)" || (echo "usage: make serverless-deploy PROJECT=<gcp project id> [TAG=<image tag>]"; exit 1)
@test -n "$$TF_VAR_database_url" || echo "note: TF_VAR_database_url is unset; add -var deploy_cloud_sql=true or export a Postgres URL"
cd infra/terraform && terraform apply -var project=$(PROJECT) -var image_tag=$(TAG)
serverless-destroy: ## tear it all down
cd infra/terraform && terraform destroy -var project=$(PROJECT) -var deletion_protection=false
gcp-configure: ## one-time: write your GCP project id into the gcp overlay and Argo manifests
@test -n "$(PROJECT)" || (echo "usage: make gcp-configure PROJECT=<gcp project id>"; exit 1)
grep -rl __GCP_PROJECT__ infra/k8s/overlays/gcp infra/argo-workflows \
| xargs sed -i.bak "s/__GCP_PROJECT__/$(PROJECT)/g"
find infra -name '*.bak' -delete
gcp-secrets: ## after terraform apply: copy DB URLs from Secret Manager into k8s secrets
@test -n "$(PROJECT)" || (echo "usage: make gcp-secrets PROJECT=<gcp project id>"; exit 1)
kubectl -n rarelens create secret generic api-secrets --dry-run=client -o yaml \
--from-literal=DATABASE_URL="$$(gcloud secrets versions access latest --project $(PROJECT) --secret rarelens-api-database-url)" \
| kubectl apply -f -
kubectl -n rarelens create secret generic pipeline-secrets --dry-run=client -o yaml \
--from-literal=DATABASE_URL="$$(gcloud secrets versions access latest --project $(PROJECT) --secret DATABASE_URL)" \
| kubectl apply -f -