feat: redesign around phenotype-driven triage, not variant filtering
A table with filters made the user do the work. Rare disease triage is a different task:
which few variants could explain *this* patient's phenotype, and why. The app now answers
that, and lets a reviewer act on the answer.
Domain
- a case is a proband: a VCF plus the HPO terms observed in the patient (samples -> cases)
- HPO's gene-to-phenotype annotations are loaded as reference data (scripts/load-hpo.py)
- each candidate can be shortlisted or dismissed with a reason and a note
Ranking (app/services/triage.py, 21 tests)
- weighted sum of phenotype match, rarity, consequence severity and the model's score,
with every component shown next to the candidate
- rarity and consequence filter; phenotype only ranks, because a real diagnosis can sit in
a gene nobody has annotated yet and filtering on it would hide exactly that case
- ClinVar is deliberately not an input: it appears beside the result as independent
confirmation, so nothing ranks highly merely because ClinVar already said pathogenic
UI
- the funnel is the headline: variants called -> rare -> coding candidates -> phenotype-matched
- ranked candidates with evidence chips, not a grid of everything; filters are demoted
- a variant panel showing the score breakdown, the matched HPO terms, the raw VEP record and
links out to Ensembl/gnomAD/ClinVar, with the decision controls
- a printable case report: phenotype, funnel, shortlisted variants with reasons, provenance
API: /cases with phenotypes, /cases/{id}/candidates (funnel + ranked + weights),
/variants/{id}, /variants/{id}/decision, /cases/{id}/report, /phenotypes for the picker.
Scoring moved under the case and now answers 503 with the reason when no model registry is
reachable, instead of a 500.
Verified end to end on a simulated proband (scripts/make-demo-case.sh: real GIAB HG002
background + one real ClinVar 2-star pathogenic NF2 variant). 13 variants called -> 1 coding
candidate, and the planted variant ranks first at 0.80 on phenotype 1.00, rarity 1.00 and
consequence 1.00, with ClinVar agreeing afterwards.
Tests: api 75, ml 18, loader 16, web 27; ruff, mypy, svelte-check, terraform validate, both
kustomize overlays and the Nextflow stub run all clean.
This commit is contained in:
+99
-19
@@ -2,12 +2,15 @@ import re
|
||||
import uuid
|
||||
from datetime import datetime
|
||||
from pathlib import PurePosixPath
|
||||
from typing import Literal
|
||||
from typing import TYPE_CHECKING, Literal
|
||||
|
||||
from pydantic import BaseModel, ConfigDict, Field, field_validator
|
||||
|
||||
from app.config import settings
|
||||
from app.models import JobStatus
|
||||
from app.models import DecisionState, JobStatus
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from app.services.triage import Candidate
|
||||
|
||||
|
||||
class ORMModel(BaseModel):
|
||||
@@ -19,11 +22,18 @@ VCF_SUFFIXES = (".vcf", ".vcf.gz", ".vcf.bgz", ".bcf")
|
||||
GCS_URI = re.compile(r"gs://[a-z0-9][a-z0-9._-]{1,220}[a-z0-9]/\S+")
|
||||
|
||||
|
||||
class SampleCreate(BaseModel):
|
||||
class PhenotypeTerm(ORMModel):
|
||||
"""An HPO term: the id is what ranking matches on, the label is for people."""
|
||||
|
||||
hpo_id: str = Field(pattern=r"^HP:\d{7}$")
|
||||
label: str = Field(min_length=1, max_length=200)
|
||||
|
||||
|
||||
class CaseCreate(BaseModel):
|
||||
name: str = Field(min_length=1, max_length=120)
|
||||
vcf_uri: str
|
||||
# Passed to VEP --assembly; the VEP cache must contain it.
|
||||
assembly: Assembly = "GRCh38"
|
||||
phenotypes: list[PhenotypeTerm] = Field(default_factory=list, max_length=100)
|
||||
|
||||
@field_validator("vcf_uri")
|
||||
@classmethod
|
||||
@@ -44,17 +54,9 @@ class SampleCreate(BaseModel):
|
||||
return v
|
||||
|
||||
|
||||
class SampleOut(ORMModel):
|
||||
id: uuid.UUID
|
||||
name: str
|
||||
vcf_uri: str
|
||||
assembly: str
|
||||
created_at: datetime
|
||||
|
||||
|
||||
class JobOut(ORMModel):
|
||||
id: uuid.UUID
|
||||
sample_id: uuid.UUID
|
||||
case_id: uuid.UUID
|
||||
status: JobStatus
|
||||
workflow_ref: str | None
|
||||
vep_version: str | None
|
||||
@@ -63,6 +65,17 @@ class JobOut(ORMModel):
|
||||
finished_at: datetime | None
|
||||
|
||||
|
||||
class CaseOut(ORMModel):
|
||||
id: uuid.UUID
|
||||
name: str
|
||||
vcf_uri: str
|
||||
assembly: str
|
||||
created_at: datetime
|
||||
phenotypes: list[PhenotypeTerm] = Field(default_factory=list)
|
||||
latest_job: JobOut | None = None
|
||||
shortlisted: int = 0
|
||||
|
||||
|
||||
class PredictionOut(ORMModel):
|
||||
model_name: str
|
||||
model_version: str
|
||||
@@ -85,14 +98,81 @@ class VariantOut(ORMModel):
|
||||
prediction: PredictionOut | None = None
|
||||
|
||||
|
||||
class ScoreOut(BaseModel):
|
||||
job_id: uuid.UUID
|
||||
scored: int
|
||||
model_version: str
|
||||
class DecisionIn(BaseModel):
|
||||
state: DecisionState
|
||||
reason: str | None = Field(None, max_length=120)
|
||||
note: str | None = Field(None, max_length=2000)
|
||||
|
||||
|
||||
class VariantPage(BaseModel):
|
||||
items: list[VariantOut]
|
||||
class DecisionOut(ORMModel):
|
||||
state: DecisionState
|
||||
reason: str | None
|
||||
note: str | None
|
||||
decided_at: datetime
|
||||
|
||||
|
||||
class CandidateOut(BaseModel):
|
||||
variant: VariantOut
|
||||
score: float
|
||||
components: dict[str, float]
|
||||
matched_terms: list[PhenotypeTerm]
|
||||
scored: bool
|
||||
decision: DecisionOut | None = None
|
||||
|
||||
@classmethod
|
||||
def from_candidate(cls, candidate: "Candidate", labels: dict[str, str]) -> "CandidateOut":
|
||||
variant = candidate.variant
|
||||
return cls(
|
||||
variant=VariantOut.model_validate(variant),
|
||||
score=round(candidate.score, 4),
|
||||
components={name: round(v, 4) for name, v in candidate.components.items()},
|
||||
matched_terms=[
|
||||
PhenotypeTerm(hpo_id=term, label=labels.get(term, term))
|
||||
for term in candidate.matched_terms
|
||||
],
|
||||
scored=candidate.scored,
|
||||
decision=DecisionOut.model_validate(variant.decision) if variant.decision else None,
|
||||
)
|
||||
|
||||
|
||||
class VariantDetailOut(CandidateOut):
|
||||
annotations: dict
|
||||
|
||||
|
||||
class FunnelOut(BaseModel):
|
||||
total: int
|
||||
rare: int
|
||||
candidates: int
|
||||
phenotype_matched: int
|
||||
|
||||
|
||||
class CandidatePage(BaseModel):
|
||||
funnel: FunnelOut
|
||||
weights: dict[str, float]
|
||||
items: list[CandidateOut]
|
||||
total: int
|
||||
limit: int
|
||||
offset: int
|
||||
|
||||
|
||||
class ProvenanceOut(BaseModel):
|
||||
job_id: uuid.UUID | None = None
|
||||
vep_version: str | None = None
|
||||
finished_at: datetime | None = None
|
||||
model_name: str | None = None
|
||||
model_version: str | None = None
|
||||
|
||||
|
||||
class ReportOut(BaseModel):
|
||||
case: CaseOut
|
||||
funnel: FunnelOut
|
||||
generated_at: datetime
|
||||
provenance: ProvenanceOut
|
||||
shortlisted: list[CandidateOut]
|
||||
dismissed: list[CandidateOut]
|
||||
|
||||
|
||||
class ScoreOut(BaseModel):
|
||||
case_id: uuid.UUID
|
||||
scored: int
|
||||
model_version: str
|
||||
|
||||
Reference in New Issue
Block a user