fix: overhaul the platform skeleton, add a serverless deployment track
An end-to-end audit found the repo could not build, test or run as shipped. This fixes every finding, then adds a Cloud Run track so the demo costs about £1/month idle instead of ~£150. CI (red on its first run) - api: setuptools could not build the package (flat layout with app/ and alembic/) - web: missing @types/node; `vitest run` exited 1 with no test files - pipeline: the stub run needed a gitignored VCF, and no process had a stub block - ruff pinned, mypy configured, DB tests on real Postgres (pgserver locally, service in CI) ML serving (scores were meaningless) - the registered model now carries its own feature engineering and returns predict_proba, so serving sends raw columns and cannot drift from training - resolve by registry alias (stages are deprecated in MLflow 3) and record the real version; re-scoring upserts instead of failing on the unique constraint - ClinVar labels parsed from VEP's lowercase terms Pipeline - exact ref/alt recovered from a CHROM_POS_REF_ALT VCF ID; loading is idempotent - job status reaches running/failed/succeeded, so the UI stops polling dead jobs - DATABASE_URL travels in the environment or a Nextflow secret, never on a command line - VEP cache and plugins staged as inputs; the gcp profile runs tasks on Google Batch Deployment - the API serves /api (matching the ingress); the web app reads its API URL at runtime - migrations run in an init container under a Postgres advisory lock - terraform: custom VPC shared with Batch, private Cloud SQL, API enablement, Workload Identity bindings, Secret Manager, deletion protection - serverless track, now the default: Cloud Run services scaling to zero, a Cloud Run job for the Nextflow driver, and Neon or Cloud SQL behind one DATABASE_URL secret. GKE and Argo remain, behind -var deploy_kubernetes=true. See docs/cloud.md. Correctness and security - 409 on duplicate sample names, 422 on bad paging, natural chromosome ordering, wider VEP text columns, enum dropped on downgrade, the sample's assembly actually used - vcf_uri restricted to gs:// objects or files under the data root, blocking option injection - CORS restricted to configured origins; `make down` no longer deletes volumes Data - docs/data.md records the peer-reviewed, openly licensed sources (GIAB HG002, ClinVar, gnomAD) with citations and an honest evaluation plan; `make data` fetches a chr22 slice Verified: api 50 tests, ml 18, loader 16, web 12; ruff, mypy, svelte-check, terraform validate and both kustomize overlays clean.
This commit is contained in:
+128
-17
@@ -1,31 +1,142 @@
|
||||
"""Event publishing.
|
||||
"""Start the annotation pipeline for a job.
|
||||
|
||||
GCP: publish a JSON message to Pub/Sub; an Argo Events sensor turns it into an Argo Workflow.
|
||||
Local: run Nextflow directly in a background task so `docker compose up` gives a working demo.
|
||||
Serverless (CLOUDRUN_JOB set): execute a Cloud Run job that runs the Nextflow driver, which
|
||||
submits the pipeline tasks to Google Batch. Nothing runs, or is billed, between annotations.
|
||||
Kubernetes (PUBSUB_TOPIC set): publish to Pub/Sub; an Argo Events sensor submits the annotate-vcf
|
||||
WorkflowTemplate, whose exit handler marks the job failed if the workflow does not succeed.
|
||||
Local: run Nextflow as a child process and watch it. The docker-compose API image has neither
|
||||
Nextflow nor Docker, so there the job fails at once with instructions to run it from the host.
|
||||
"""
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shutil
|
||||
import uuid
|
||||
from datetime import UTC, datetime
|
||||
from functools import cache
|
||||
from typing import Any
|
||||
|
||||
from app.config import settings
|
||||
from app.db import SessionLocal
|
||||
from app.models import Job, JobStatus
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PIPELINE_ENTRYPOINT = "/pipeline/main.nf"
|
||||
LOG_TAIL_CHARS = 4000
|
||||
PUBLISH_TIMEOUT_S = 30
|
||||
_watchers: set[asyncio.Task[None]] = set()
|
||||
|
||||
|
||||
async def publish_vcf_uploaded(job_id: uuid.UUID, vcf_uri: str) -> None:
|
||||
message = {"job_id": str(job_id), "vcf_uri": vcf_uri}
|
||||
class LaunchError(Exception):
|
||||
"""The pipeline could not be started; the message is stored on the job for the user."""
|
||||
|
||||
|
||||
async def launch(job_id: uuid.UUID, vcf_uri: str, assembly: str) -> str:
|
||||
"""Start the pipeline and return a reference to the run (stored as jobs.workflow_ref)."""
|
||||
if settings.cloudrun_job:
|
||||
return await _run_cloud_run_job(job_id, vcf_uri, assembly)
|
||||
if settings.pubsub_topic:
|
||||
from google.cloud import pubsub_v1 # optional dependency, installed in the GCP image
|
||||
message = {"job_id": str(job_id), "vcf_uri": vcf_uri, "assembly": assembly}
|
||||
return await _publish(settings.pubsub_topic, message)
|
||||
return await _run_local(job_id, vcf_uri, assembly)
|
||||
|
||||
publisher = pubsub_v1.PublisherClient()
|
||||
publisher.publish(settings.pubsub_topic, json.dumps(message).encode())
|
||||
return
|
||||
|
||||
cmd = [
|
||||
"nextflow", "run", "/pipeline/main.nf", "-profile", "docker",
|
||||
"--vcf", vcf_uri, "--job_id", str(job_id), "--db_url", settings.database_url,
|
||||
async def drain() -> None:
|
||||
"""Wait for local pipeline watchers (tests, graceful shutdown)."""
|
||||
await asyncio.gather(*_watchers)
|
||||
|
||||
|
||||
@cache
|
||||
def _jobs_client() -> Any:
|
||||
from google.cloud import run_v2
|
||||
|
||||
return run_v2.JobsClient()
|
||||
|
||||
|
||||
async def _run_cloud_run_job(job_id: uuid.UUID, vcf_uri: str, assembly: str) -> str:
|
||||
if not settings.gcp_project:
|
||||
raise LaunchError("CLOUDRUN_JOB is set but GCP_PROJECT is not")
|
||||
name = (
|
||||
f"projects/{settings.gcp_project}/locations/{settings.gcp_region}"
|
||||
f"/jobs/{settings.cloudrun_job}"
|
||||
)
|
||||
args = [
|
||||
"run", PIPELINE_ENTRYPOINT, "-profile", "gcp",
|
||||
"--vcf", vcf_uri, "--job_id", str(job_id), "--assembly", assembly,
|
||||
]
|
||||
request = {"name": name, "overrides": {"container_overrides": [{"args": args}]}}
|
||||
try:
|
||||
await asyncio.create_subprocess_exec(*cmd)
|
||||
except FileNotFoundError:
|
||||
# Nextflow is not installed in this environment (no Java toolchain yet);
|
||||
# leave the job queued instead of crashing the request.
|
||||
print(f"[rarelens] nextflow not found; cannot start pipeline for job {job_id}", flush=True)
|
||||
operation = await asyncio.to_thread(_jobs_client().run_job, request=request)
|
||||
except Exception as e:
|
||||
logger.exception("starting Cloud Run job %s failed", name)
|
||||
raise LaunchError(f"could not start {settings.cloudrun_job}: {e}") from e
|
||||
name_or_job = getattr(getattr(operation, "metadata", None), "name", "")
|
||||
execution = name_or_job or settings.cloudrun_job or "started"
|
||||
return f"cloudrun:{execution.rsplit('/', 1)[-1]}"
|
||||
|
||||
|
||||
@cache
|
||||
def _publisher() -> Any:
|
||||
from google.cloud import pubsub_v1
|
||||
|
||||
return pubsub_v1.PublisherClient()
|
||||
|
||||
|
||||
async def _publish(topic_name: str, message: dict[str, str]) -> str:
|
||||
if not settings.gcp_project:
|
||||
raise LaunchError("PUBSUB_TOPIC is set but GCP_PROJECT is not")
|
||||
publisher = _publisher()
|
||||
topic = publisher.topic_path(settings.gcp_project, topic_name)
|
||||
try:
|
||||
future = publisher.publish(topic, json.dumps(message).encode())
|
||||
message_id = await asyncio.to_thread(future.result, timeout=PUBLISH_TIMEOUT_S)
|
||||
except Exception as e:
|
||||
logger.exception("publishing to %s failed", topic)
|
||||
raise LaunchError(f"could not publish to {topic}: {e}") from e
|
||||
return f"pubsub:{message_id}"
|
||||
|
||||
|
||||
async def _run_local(job_id: uuid.UUID, vcf_uri: str, assembly: str) -> str:
|
||||
nextflow = shutil.which("nextflow")
|
||||
if nextflow is None:
|
||||
raise LaunchError(
|
||||
"Nextflow is not installed where the API runs. Start the pipeline from the host with "
|
||||
f"`make annotate JOB={job_id} VCF=<path to the VCF>`."
|
||||
)
|
||||
cmd = [
|
||||
nextflow, "run", str(settings.pipeline_dir / "main.nf"),
|
||||
"-profile", settings.nextflow_profile,
|
||||
"--vcf", vcf_uri, "--job_id", str(job_id), "--assembly", assembly,
|
||||
]
|
||||
# The loader reads DATABASE_URL from its environment; keep it off the command line.
|
||||
env = {**os.environ, "DATABASE_URL": settings.database_url}
|
||||
try:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
*cmd,
|
||||
cwd=settings.pipeline_dir,
|
||||
env=env,
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.STDOUT,
|
||||
)
|
||||
except OSError as e:
|
||||
raise LaunchError(f"could not start Nextflow: {e}") from e
|
||||
task = asyncio.create_task(_watch(job_id, proc))
|
||||
_watchers.add(task)
|
||||
task.add_done_callback(_watchers.discard)
|
||||
return f"nextflow:pid-{proc.pid}"
|
||||
|
||||
|
||||
async def _watch(job_id: uuid.UUID, proc: Any) -> None:
|
||||
out, err = await proc.communicate()
|
||||
if proc.returncode == 0:
|
||||
return # the loader marks the job succeeded
|
||||
tail = ((out or b"") + (err or b"")).decode(errors="replace")[-LOG_TAIL_CHARS:]
|
||||
async with SessionLocal() as session:
|
||||
job = await session.get(Job, job_id)
|
||||
if job is not None and job.status != JobStatus.succeeded:
|
||||
job.status = JobStatus.failed
|
||||
job.log = f"nextflow exited with code {proc.returncode}\n{tail}"
|
||||
job.finished_at = datetime.now(UTC)
|
||||
await session.commit()
|
||||
|
||||
Reference in New Issue
Block a user