An end-to-end audit found the repo could not build, test or run as shipped. This fixes every finding, then adds a Cloud Run track so the demo costs about £1/month idle instead of ~£150. CI (red on its first run) - api: setuptools could not build the package (flat layout with app/ and alembic/) - web: missing @types/node; `vitest run` exited 1 with no test files - pipeline: the stub run needed a gitignored VCF, and no process had a stub block - ruff pinned, mypy configured, DB tests on real Postgres (pgserver locally, service in CI) ML serving (scores were meaningless) - the registered model now carries its own feature engineering and returns predict_proba, so serving sends raw columns and cannot drift from training - resolve by registry alias (stages are deprecated in MLflow 3) and record the real version; re-scoring upserts instead of failing on the unique constraint - ClinVar labels parsed from VEP's lowercase terms Pipeline - exact ref/alt recovered from a CHROM_POS_REF_ALT VCF ID; loading is idempotent - job status reaches running/failed/succeeded, so the UI stops polling dead jobs - DATABASE_URL travels in the environment or a Nextflow secret, never on a command line - VEP cache and plugins staged as inputs; the gcp profile runs tasks on Google Batch Deployment - the API serves /api (matching the ingress); the web app reads its API URL at runtime - migrations run in an init container under a Postgres advisory lock - terraform: custom VPC shared with Batch, private Cloud SQL, API enablement, Workload Identity bindings, Secret Manager, deletion protection - serverless track, now the default: Cloud Run services scaling to zero, a Cloud Run job for the Nextflow driver, and Neon or Cloud SQL behind one DATABASE_URL secret. GKE and Argo remain, behind -var deploy_kubernetes=true. See docs/cloud.md. Correctness and security - 409 on duplicate sample names, 422 on bad paging, natural chromosome ordering, wider VEP text columns, enum dropped on downgrade, the sample's assembly actually used - vcf_uri restricted to gs:// objects or files under the data root, blocking option injection - CORS restricted to configured origins; `make down` no longer deletes volumes Data - docs/data.md records the peer-reviewed, openly licensed sources (GIAB HG002, ClinVar, gnomAD) with citations and an honest evaluation plan; `make data` fetches a chr22 slice Verified: api 50 tests, ml 18, loader 16, web 12; ruff, mypy, svelte-check, terraform validate and both kustomize overlays clean.
78 lines
3.6 KiB
Python
78 lines
3.6 KiB
Python
"""SQLAlchemy 2.0 declarative models.
|
|
|
|
One sample -> many jobs; one job -> many variants; one variant -> one prediction (latest).
|
|
"""
|
|
import enum
|
|
import uuid
|
|
from datetime import datetime
|
|
|
|
from sqlalchemy import DateTime, Enum, Float, ForeignKey, Integer, String, Text, func
|
|
from sqlalchemy.dialects.postgresql import JSONB, UUID
|
|
from sqlalchemy.orm import DeclarativeBase, Mapped, mapped_column, relationship
|
|
|
|
|
|
class Base(DeclarativeBase):
|
|
pass
|
|
|
|
|
|
class JobStatus(str, enum.Enum):
|
|
queued = "queued"
|
|
running = "running"
|
|
succeeded = "succeeded"
|
|
failed = "failed"
|
|
|
|
|
|
class Sample(Base):
|
|
__tablename__ = "samples"
|
|
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
|
|
name: Mapped[str] = mapped_column(String(120), unique=True)
|
|
vcf_uri: Mapped[str] = mapped_column(Text)
|
|
assembly: Mapped[str] = mapped_column(String(10), default="GRCh38")
|
|
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())
|
|
jobs: Mapped[list["Job"]] = relationship(back_populates="sample")
|
|
|
|
|
|
class Job(Base):
|
|
__tablename__ = "jobs"
|
|
id: Mapped[uuid.UUID] = mapped_column(UUID(as_uuid=True), primary_key=True, default=uuid.uuid4)
|
|
sample_id: Mapped[uuid.UUID] = mapped_column(ForeignKey("samples.id", ondelete="CASCADE"))
|
|
status: Mapped[JobStatus] = mapped_column(Enum(JobStatus), default=JobStatus.queued)
|
|
workflow_ref: Mapped[str | None] = mapped_column(String(200)) # Argo workflow name / nf run id
|
|
vep_version: Mapped[str | None] = mapped_column(String(40))
|
|
log: Mapped[str | None] = mapped_column(Text)
|
|
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())
|
|
finished_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True))
|
|
sample: Mapped[Sample] = relationship(back_populates="jobs")
|
|
variants: Mapped[list["Variant"]] = relationship(back_populates="job")
|
|
|
|
|
|
class Variant(Base):
|
|
__tablename__ = "variants"
|
|
id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True)
|
|
job_id: Mapped[uuid.UUID] = mapped_column(ForeignKey("jobs.id", ondelete="CASCADE"), index=True)
|
|
chrom: Mapped[str] = mapped_column(String(10), index=True)
|
|
pos: Mapped[int] = mapped_column(Integer, index=True)
|
|
ref: Mapped[str] = mapped_column(Text)
|
|
alt: Mapped[str] = mapped_column(Text)
|
|
gene: Mapped[str | None] = mapped_column(String(60), index=True)
|
|
consequence: Mapped[str | None] = mapped_column(Text) # "&"-joined VEP terms
|
|
impact: Mapped[str | None] = mapped_column(String(20))
|
|
hgvsc: Mapped[str | None] = mapped_column(Text)
|
|
hgvsp: Mapped[str | None] = mapped_column(Text)
|
|
gnomad_af: Mapped[float | None] = mapped_column(Float)
|
|
clinvar_sig: Mapped[str | None] = mapped_column(Text) # ","-joined co-located ClinVar terms
|
|
annotations: Mapped[dict] = mapped_column(JSONB, default=dict) # full VEP CSQ record
|
|
job: Mapped[Job] = relationship(back_populates="variants")
|
|
prediction: Mapped["Prediction | None"] = relationship(back_populates="variant", uselist=False)
|
|
|
|
|
|
class Prediction(Base):
|
|
__tablename__ = "predictions"
|
|
id: Mapped[int] = mapped_column(Integer, primary_key=True, autoincrement=True)
|
|
variant_id: Mapped[int] = mapped_column(ForeignKey("variants.id", ondelete="CASCADE"), unique=True)
|
|
model_name: Mapped[str] = mapped_column(String(80))
|
|
model_version: Mapped[str] = mapped_column(String(40))
|
|
score: Mapped[float] = mapped_column(Float) # P(pathogenic)
|
|
created_at: Mapped[datetime] = mapped_column(DateTime(timezone=True), server_default=func.now())
|
|
variant: Mapped[Variant] = relationship(back_populates="prediction")
|