feat(pipeline): VEP database mode, and a pipeline-specific database URL

Makes a real annotation runnable locally without the 25 GB VEP cache, which is what
the demo needs and what a reviewer can reproduce in minutes.

- params.vep_database (VEP_DATABASE=true) queries Ensembl's public database instead of
  a local cache. Slower per variant and fewer fields, so --everything is swapped for the
  flags the loader actually stores. Its cache placeholder is NO_CACHE, not NO_FILE:
  Nextflow rejects two staged inputs sharing a filename.
- PIPELINE_DATABASE_URL is handed to the pipeline when set. The loader runs inside a
  container, where the API's own localhost URL would point at the container itself.
- README: how to run the UI's annotate button locally against host Nextflow + Docker.

Verified end to end on pipeline/tests/data/tiny.vcf: bcftools norm split the multiallelic
record, VEP 113 annotated 4 variants live, the loader wrote them and marked the job
succeeded, and the UI shows them. The deletion came back as 22:42126611 CT>C with exact
VCF alleles, which is the case the audit's ID-tagging fix exists for.

Tests: api 51, loader 16, stub run 3/3; ruff, mypy clean.
This commit is contained in:
Kemal Yaylali
2026-09-12 07:39:19 +01:00
parent 11fb6b3d73
commit ae58e33fe2
8 changed files with 64 additions and 4 deletions
+3
View File
@@ -23,6 +23,9 @@ class Settings(BaseSettings):
gcp_project: str | None = None # required with pubsub_topic or cloudrun_job
gcp_region: str = "europe-west2"
pipeline_dir: Path = REPO_ROOT / "pipeline"
# Handed to the pipeline when it differs from the API's own: the loader runs inside a
# container, where the API's localhost would be the container itself.
pipeline_database_url: str | None = None
nextflow_profile: str = "docker"
# Local (non-gs://) VCFs must live under this directory.
local_data_root: Path = Path("/data")
+1 -1
View File
@@ -111,7 +111,7 @@ async def _run_local(job_id: uuid.UUID, vcf_uri: str, assembly: str) -> str:
"--vcf", vcf_uri, "--job_id", str(job_id), "--assembly", assembly,
]
# The loader reads DATABASE_URL from its environment; keep it off the command line.
env = {**os.environ, "DATABASE_URL": settings.database_url}
env = {**os.environ, "DATABASE_URL": settings.pipeline_database_url or settings.database_url}
try:
proc = await asyncio.create_subprocess_exec(
*cmd,
+25
View File
@@ -124,3 +124,28 @@ async def test_pubsub_failure_marks_the_job_failed(
job = (await client.post(f"/api/samples/{sample_id}/annotate")).json()
assert job["status"] == "failed"
assert "403 denied" in job["log"]
@pytest.mark.usefixtures("db")
async def test_the_pipeline_gets_its_own_database_url(
client: AsyncClient, monkeypatch: pytest.MonkeyPatch
) -> None:
"""The loader runs in a container, where the API's own localhost URL would point at itself."""
monkeypatch.setattr(settings, "pubsub_topic", None)
monkeypatch.setattr(settings, "cloudrun_job", None)
monkeypatch.setattr(
settings, "pipeline_database_url", "postgresql+asyncpg://u:[email protected]:5432/db"
)
monkeypatch.setattr(events.shutil, "which", lambda _: "/usr/bin/nextflow")
launched: dict[str, Any] = {}
async def fake_exec(*cmd: str, **kw: Any) -> FakeProcess:
launched["env"] = kw["env"]
return FakeProcess(0, b"")
monkeypatch.setattr(events.asyncio, "create_subprocess_exec", fake_exec)
sample_id = await new_sample(client)
await client.post(f"/api/samples/{sample_id}/annotate")
await events.drain()
assert launched["env"]["DATABASE_URL"] == "postgresql+asyncpg://u:[email protected]:5432/db"