diff --git a/.gitmodules b/.gitmodules index e28af6b..35f6c82 100644 --- a/.gitmodules +++ b/.gitmodules @@ -1,4 +1,4 @@ [submodule "submodules/va_spec"] path = submodules/va_spec url = https://github.com/ga4gh/va-spec - branch = 1.1.0-snapshot.2026-06 + branch = 1.1.0-ballot.2026-09 diff --git a/pyproject.toml b/pyproject.toml index 8eb1319..61ac8c8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -23,9 +23,9 @@ keywords = ["bioinformatics", "ga4gh", "genomics", "variation"] requires-python = ">=3.10" dynamic = ["version"] dependencies = [ - "ga4gh.vrs~=2.4.0-a3", - "ga4gh.cat_vrs~=0.8.0-a3", - "pydantic>=2.0,<3.0", + "ga4gh.vrs~=2.4.0-a5", + "ga4gh.cat_vrs~=0.8.0-a5", + "pydantic>=2.12,<3.0", "typing_extensions", ] diff --git a/src/ga4gh/va_spec/aac_2017/models.py b/src/ga4gh/va_spec/aac_2017/models.py index b840d64..3f83a1c 100644 --- a/src/ga4gh/va_spec/aac_2017/models.py +++ b/src/ga4gh/va_spec/aac_2017/models.py @@ -13,7 +13,11 @@ from typing_extensions import Self from ga4gh.core.metadata import Maturity -from ga4gh.core.models import BaseModelForbidExtra, MappableConcept, iriReference +from ga4gh.core.models import ( + BaseModelForbidExtra, + MappableConcept, + iriReference, +) from ga4gh.va_spec.aac_2017.metadata import AAC2017MetadataMixin from ga4gh.va_spec.base.core import ( Direction, @@ -62,7 +66,7 @@ class AmpAscoCapEvidenceLineStrength(str, Enum): class AmpAscoCapEvidenceLine(AAC2017MetadataMixin, EvidenceLine): - """Evidence line for AMP/ASCO/CAP""" + """General evidence line for AMP/ASCO/CAP""" _maturity: ClassVar[Maturity] = Maturity.DRAFT @@ -70,6 +74,7 @@ class AmpAscoCapEvidenceLine(AAC2017MetadataMixin, EvidenceLine): VariantPrognosticProposition | VariantDiagnosticProposition | VariantTherapeuticResponseProposition + | iriReference ) @field_validator("strengthOfEvidenceProvided", mode="after") @@ -90,7 +95,7 @@ def validate_strength_of_evidence_provided( class _PrognosticEvidenceLineObject(AmpAscoCapEvidenceLine): """Internal prognostic evidence line for AMP/ASCO/CAP""" - targetProposition: VariantPrognosticProposition + targetProposition: VariantPrognosticProposition | iriReference class PrognosticEvidenceLine( @@ -104,7 +109,7 @@ class PrognosticEvidenceLine( class _DiagnosticEvidenceLineObject(AmpAscoCapEvidenceLine): """Internal diagnostic evidence line for AMP/ASCO/CAP""" - targetProposition: VariantDiagnosticProposition + targetProposition: VariantDiagnosticProposition | iriReference class DiagnosticEvidenceLine( @@ -118,7 +123,7 @@ class DiagnosticEvidenceLine( class _TherapeuticEvidenceLineObject(AmpAscoCapEvidenceLine): """Internal therapeutic evidence line for AMP/ASCO/CAP""" - targetProposition: VariantTherapeuticResponseProposition + targetProposition: VariantTherapeuticResponseProposition | iriReference class TherapeuticEvidenceLine( @@ -204,59 +209,67 @@ class VariantClinicalSignificanceStatement( _maturity: ClassVar[Maturity] = Maturity.DRAFT - proposition: VariantClinicalSignificanceProposition - strength: MappableConcept | None = Field( + proposition: VariantClinicalSignificanceProposition | iriReference + strength: MappableConcept | iriReference | None = Field( default=None, description="The strength of support that the Statement is determined to provide for or against the Variant Clinical Significance Proposition for the assessed variant, based on the curation and reporting conventions of the AMP/ASCO/CAP 2017 Guidelines.", ) - classification: MappableConcept = Field( + classification: MappableConcept | iriReference = Field( ..., - description="A single term or phrase classifying the subject variant based on the outcome of direction and strength assessments of the Statement's Proposition, using terms from the AMP/ASCO/CAP 2017 Guidelines.", + description="A single term or phrase classifying the subject variant based on the result of direction and strength assessments of the Statement's Proposition, using terms from the AMP/ASCO/CAP 2017 Guidelines.", ) specifiedBy: Method | iriReference + @model_validator(mode="before") + @classmethod + def validate_tier_evidence_lines(cls, values: dict) -> dict: + """Validate tier I and II evidence-line types before base coercion.""" + if not isinstance(values, dict): + return values + + classification = values.get("classification") + if not isinstance(classification, dict): + return values + + primary_coding = classification.get("primaryCoding") + if not isinstance(primary_coding, dict) or primary_coding.get("code") not in { + AmpAscoCapClassificationCode.TIER_1, + AmpAscoCapClassificationCode.TIER_2, + }: + return values + + approved_el_classes = ( + DiagnosticEvidenceLine, + PrognosticEvidenceLine, + TherapeuticEvidenceLine, + iriReference, + ) + for evidence_line in values.get("hasEvidenceLines") or []: + for approved_el_cls in approved_el_classes: + try: + approved_el_cls.model_validate(evidence_line) + break + except Exception: # noqa: S112 + continue + else: + msg = "`hasEvidenceLines` must be one of: `DiagnosticEvidenceLine`, `PrognosticEvidenceLine`, `TherapeuticEvidenceLine`, or `iriReference`" + raise ValueError(msg) + + return values + @model_validator(mode="after") def validate_statement(self) -> Self: """Validate VariantClinicalSignificanceStatement""" - - def _validate_evidence_lines( - classification_code: AmpAscoCapClassificationCode, - has_evidence_lines: list, - ) -> None: - """Validate allowed evidence lines given classification code""" - approved_el_classes = [ - DiagnosticEvidenceLine, - PrognosticEvidenceLine, - TherapeuticEvidenceLine, - ] - if classification_code in { - AmpAscoCapClassificationCode.TIER_1, - AmpAscoCapClassificationCode.TIER_2, - }: - for evidence_line in has_evidence_lines: - if hasattr(evidence_line, "root"): - el_input = evidence_line.root - elif hasattr(evidence_line, "model_dump"): - el_input = evidence_line.model_dump() - else: - el_input = evidence_line - - for approved_el_cls in approved_el_classes: - try: - approved_el_cls.model_validate(el_input) - break - except Exception: # noqa: S112 - continue - else: - msg = "`hasEvidenceLines` must be one of: `DiagnosticEvidenceLine`, `PrognosticEvidenceLine`, or `TherapeuticEvidenceLine`" - raise ValueError(msg) + if isinstance(self.classification, iriReference) or isinstance( + self.strength, iriReference + ): + return self def _validate_amp_asco_cap_classification_constraints( classification_code: AmpAscoCapClassificationCode, classification_name: str | None, direction: str, strength_code: MappableConcept | None, - has_evidence_lines: list, ) -> None: """Validate that a classification code enforces required values for strength, name, direction, and when applicable allowed evidence line types. @@ -283,8 +296,6 @@ def _validate_amp_asco_cap_classification_constraints( msg = f"`direction` must be: {expected.direction.value}" raise ValueError(msg) - _validate_evidence_lines(classification_code, has_evidence_lines) - # Validate strength system. The actual value will be validated in # `_validate_amp_asco_cap_classification_constraints` validate_mappable_concept( @@ -307,7 +318,6 @@ def _validate_amp_asco_cap_classification_constraints( self.classification.name, self.direction, self.strength, - self.hasEvidenceLines or [], ) return self diff --git a/src/ga4gh/va_spec/acmg_2015/models.py b/src/ga4gh/va_spec/acmg_2015/models.py index 6a45379..89e34ae 100644 --- a/src/ga4gh/va_spec/acmg_2015/models.py +++ b/src/ga4gh/va_spec/acmg_2015/models.py @@ -14,7 +14,6 @@ from ga4gh.core.models import MappableConcept, iriReference from ga4gh.va_spec.acmg_2015.metadata import ACMG2015MetadataMixin from ga4gh.va_spec.base.core import ( - Direction, Document, EvidenceLine, Method, @@ -23,6 +22,7 @@ ) from ga4gh.va_spec.base.enums import ( CLIN_GEN_CLASSIFICATIONS, + NO_CRITERIA_MET, STRENGTH_CODES, STRENGTH_OF_EVIDENCE_PROVIDED_VALUES, System, @@ -75,15 +75,11 @@ class VariantPathogenicityEvidenceLine( _maturity: ClassVar[Maturity] = Maturity.DRAFT - targetProposition: VariantPathogenicityProposition | None = Field( + targetProposition: VariantPathogenicityProposition | iriReference | None = Field( default=None, description="A Variant Pathogenicity Proposition against which a specific type of evidence was assessed, to determine the strength and direction of support this evidence provides for or against the proposition's validity.", ) - directionOfEvidenceProvided: Direction = Field( - ..., - description="The direction of support that the Evidence Line is determined to provide toward its target Proposition (supports, disputes, neutral). For ACMG-based assessments, if a pathogenicity criterion is 'met' in the Evidence Line the direction is 'supports', if a benignity criterion is 'met' the direction is 'disputes', and if a criteria is 'not met' the direction is 'none'.", - ) - strengthOfEvidenceProvided: MappableConcept | None = Field( + strengthOfEvidenceProvided: MappableConcept | iriReference | None = Field( default=None, description="The strength of support that an Evidence Line is determined to provide for or against the proposed pathogenicity of the assessed variant. Strength is evaluated relative to the direction indicated by the 'directionOfEvidenceProvided' attribute, and captured using a MappableConcept, whose nested 'code' field is bound to an enumerated set of values. Conditional requirement: if `directionOfEvidenceProvided` is either 'supports' or 'disputes', then this attribute is required. If it is 'none', then this attribute is not allowed.", ) @@ -130,70 +126,70 @@ class MethodType(str, Enum): # Assessment of whether population control frequency refutes # pathogenicity or whether absence/extreme rarity in controls provides # supporting evidence for pathogenicity - POPULATION_DATA_ASSESSMENT = "Population Data Assessment" + POPULATION_DATA_ASSESSMENT = "population_data_assessment" # Prevalence in affected statistically increased over matched controls, # or enrichment in controls inconsistent with disease penetrance - CASE_CONTROL_ENRICHMENT_ASSESSMENT = "Case-Control Enrichment Assessment" + CASE_CONTROL_ENRICHMENT_ASSESSMENT = "case_control_enrichment_assessment" # Predicted null variant in a gene where LOF is a known mechanism of disease - NULL_VARIANT_ASSESSMENT = "Null variant assessment" + NULL_VARIANT_ASSESSMENT = "null_variant_assessment" # Same amino acid change as an established pathogenic variant - SAME_AMINO_ACID_CHANGE_ASSESSMENT = "Same amino acid change assessment" + SAME_AMINO_ACID_CHANGE_ASSESSMENT = "same_amino_acid_change_assessment" # Mutational hot spot or well-established functional domain without # benign variation MUTATIONAL_HOT_SPOT_AND_FUNCTIONAL_DOMAIN_ASSESSMENT = ( - "Mutational hot spot and functional domain assessment" + "mutational_hot_spot_and_functional_domain_assessment" ) # Protein length change, or in-frame indels changing counts of repeats # with no known function - PROTEIN_LENGTH_CHANGE_ASSESSMENT = "Protein length change assessment" + PROTEIN_LENGTH_CHANGE_ASSESSMENT = "protein_length_change_assessment" # Novel missense change at an amino acid where a different pathogenic # missense change has been seen before - NOVEL_MISSENSE_POSITION_ASSESSMENT = "Novel missense position assessment" + NOVEL_MISSENSE_POSITION_ASSESSMENT = "novel_missense_position_assessment" # Missense in a gene where only truncation causes disease, or with low # rate of benign missense variation - VARIANT_SPECTRUM_ASSESSMENT = "Variant spectrum assessment" + VARIANT_SPECTRUM_ASSESSMENT = "variant_spectrum_assessment" # Multiple lines of computational evidence support a deleterious effect # or no impact IN_SILICO_FUNCTIONAL_IMPACT_ASSESSMENT = ( - "In silico functional impact assessment" + "in_silico_functional_impact_assessment" ) # Silent variant with no predicted splicing impact - PREDICTED_SILENT_VARIANT_ASSESSMENT = "Predicted silent variant assessment" + PREDICTED_SILENT_VARIANT_ASSESSMENT = "predicted_silent_variant_assessment" # Well-established functional studies show or do not show deleterious # effect - FUNCTIONAL_DATA_ASSESSMENT = "Functional Data Assessment" + FUNCTIONAL_DATA_ASSESSMENT = "functional_data_assessment" # Consegregation with disease in multiple family members, or # nonsegregation - SEGREGATION_DATA_ASSESSMENT = "Segregation Data Assessment" + SEGREGATION_DATA_ASSESSMENT = "segregation_data_assessment" # De novo, with or without paternity and maternity confirmed - DE_NOVO_DATA_ASSESSMENT = "De Novo Data Assessment" + DE_NOVO_OCCURRENCE_ASSESSMENT = "de_novo_occurrence_assessment" # Observed in trans with a dominant variant, in cis with a pathogenic # variant, or in trans with a pathogenic variant in a recessive # disorder - CIS_TRANS_VARIANT_ASSESSMENT = "Cis/trans variant assessment" + CIS_TRANS_VARIANT_ASSESSMENT = "cis_trans_variant_assessment" # Benign or pathogenic according to a reputable source - REPUTABLE_SOURCE_ASSESSMENT = "Reputable Source Assessment" + REPUTABLE_SOURCE_ASSESSMENT = "reputable_source_assessment" # Patient's phenotype or family history highly specific for the gene # and disorder - PHENOTYPE_GENE_SPECIFICITY_ASSESSMENT = "Phenotype-gene specificity assessment" + PHENOTYPE_GENE_SPECIFICITY_ASSESSMENT = "phenotype_gene_specificity_assessment" # Found in a case with an alternate cause - ALTERNATIVE_CAUSE_ASSESSMENT = "Alternative cause assessment" + ALTERNATIVE_CAUSE_ASSESSMENT = "alternative_cause_assessment" ALLOWED_CRITERIA_BY_METHOD_TYPE: ClassVar[ MappingProxyType[ @@ -254,7 +250,7 @@ class MethodType(str, Enum): Criterion.BS4, } ), - MethodType.DE_NOVO_DATA_ASSESSMENT: frozenset( + MethodType.DE_NOVO_OCCURRENCE_ASSESSMENT: frozenset( { Criterion.PS2, Criterion.PM6, @@ -314,13 +310,18 @@ def validate_model(self) -> Self: ``strengthOfEvidenceProvided`` is provided when ``directionOfEvidenceProvided`` is neutral """ - self._validate_direction_of_evidence_provided() - acmg_code_pattern = r"^((?:PVS1)(?:_(?:not_met|(?:strong|moderate|supporting)))?|(?:PS[1-4]|BS[1-4])(?:_(?:not_met|(?:very_strong|moderate|supporting)))?|BA1(?:_not_met)?|(?:PM[1-6])(?:_(?:not_met|(?:very_strong|strong|supporting)))?|(PP[1-5]|BP[1-7])(?:_(?:not_met|very_strong|strong|moderate))?)$" + acmg_code_pattern = rf"^(?:{NO_CRITERIA_MET}|(?:PVS1)(?:_(?:not_met|(?:strong|moderate|supporting)))?|(?:PS[1-4]|BS[1-4])(?:_(?:not_met|(?:very_strong|moderate|supporting)))?|BA1(?:_not_met)?|(?:PM[1-6])(?:_(?:not_met|(?:very_strong|strong|supporting)))?|(PP[1-5]|BP[1-7])(?:_(?:not_met|very_strong|strong|moderate))?)$" self._validate_evidence_outcome(SYSTEM, acmg_code_pattern, is_required=True) + self._validate_noncontributing_evidence_outcome() + self._validate_direction_of_evidence_provided() self._validate_criterion_specified_by() - self._validate_method_type_evidence_outcome( - self.specifiedBy.methodType, self.evidenceOutcome.primaryCoding.code.root - ) + if isinstance(self.specifiedBy, Method) and isinstance( + self.evidenceOutcome, MappableConcept + ): + self._validate_method_type_evidence_outcome( + self.specifiedBy.methodType, + self.evidenceOutcome.primaryCoding.code.root, + ) return self @@ -329,15 +330,15 @@ class VariantPathogenicityStatement(ACMG2015MetadataMixin, Statement): _maturity: ClassVar[Maturity] = Maturity.DRAFT - proposition: VariantPathogenicityProposition = Field( + proposition: VariantPathogenicityProposition | iriReference = Field( ..., description="A proposition about the pathogenicity of a variant, the validity of which is assessed and reported by the Statement. A Statement can put forth the proposition as being true, false, or uncertain, and may provide an assessment of the level of confidence/evidence supporting this claim.", ) - strength: MappableConcept | None = Field( + strength: MappableConcept | iriReference | None = Field( default=None, description="The strength of support that an ACMG 2015 Variant Pathogenicity statement is determined to provide for or against the proposed pathogenicity of the assessed variant. Strength is evaluated relative to the direction indicated by the 'direction' attribute. The indicated enumeration constrains the nested MappableConcept.primaryCoding > Coding.code attribute when capturing evidence strength.", ) - classification: MappableConcept = Field( + classification: MappableConcept | iriReference = Field( ..., description="The classification of the variant's pathogenicity, based on the ACMG 2015 guidelines. These classifications should coincide with the direction and strength values as follows: 'pathogenic' with supports-strong, 'likely pathogenic' with supports-moderate, 'benign' with disputes-strong, 'likely benign' with disputes-moderate 'uncertain significance' can be one of three possibilities... supports-weak, disputes-weak or neutral for uncertain significance (favoring pathogenic), uncertain significance (favoring benign) or uncertain significance (favoring neither pathogenic nor benign). The 'low penetrance' and 'risk allele' versions of pathogenicity classifications would be applied based on whether the variant proposition was defined to have a 'penetrance' of 'low' or 'risk' respectively.", ) @@ -351,7 +352,9 @@ class VariantPathogenicityStatement(ACMG2015MetadataMixin, Statement): @field_validator("strength") @classmethod - def validate_strength(cls, v: MappableConcept | None) -> MappableConcept | None: + def validate_strength( + cls, v: MappableConcept | iriReference | None + ) -> MappableConcept | iriReference | None: """Validate strength :param v: strength @@ -364,13 +367,18 @@ def validate_strength(cls, v: MappableConcept | None) -> MappableConcept | None: @field_validator("classification") @classmethod - def validate_classification(cls, v: MappableConcept) -> MappableConcept: + def validate_classification( + cls, v: MappableConcept | iriReference + ) -> MappableConcept | iriReference: """Validate classification :param v: classification :raises ValueError: If invalid classification values are provided :return: Validated classification value """ + if isinstance(v, iriReference): + return v + if not v.primaryCoding: err_msg = "`primaryCoding` is required." raise ValueError(err_msg) diff --git a/src/ga4gh/va_spec/base/__init__.py b/src/ga4gh/va_spec/base/__init__.py index 9e5ca23..fa9a5c2 100644 --- a/src/ga4gh/va_spec/base/__init__.py +++ b/src/ga4gh/va_spec/base/__init__.py @@ -2,16 +2,19 @@ from .core import ( Agent, - ClinicalVariantProposition, CohortAlleleFrequencyStudyResult, + ComputationalVariantFunctionalImpactAnalysisResult, Contribution, CoreType, + DataItem, DataSet, Direction, Document, EvidenceLine, ExperimentalVariantFunctionalImpactProposition, ExperimentalVariantFunctionalImpactStudyResult, + GeneDiseaseValidityProposition, + GeneticContextVariantProposition, InformationEntity, Method, Proposition, @@ -22,15 +25,17 @@ TumorVariantFrequencyStudyResult, VariantClinicalSignificanceProposition, VariantDiagnosticProposition, + VariantMolecularConsequenceProposition, VariantOncogenicityProposition, VariantPathogenicityProposition, VariantPrognosticProposition, VariantTherapeuticResponseProposition, ) -from .domain_entities import Condition, ConditionSet, Therapeutic, TherapyGroup +from .domain_entities import Condition, ConditionSet, Therapy, TherapyGroup from .enums import ( CCV_CLASSIFICATIONS, CLIN_GEN_CLASSIFICATIONS, + NO_CRITERIA_MET, STRENGTH_CODES, STRENGTH_OF_EVIDENCE_PROVIDED_VALUES, CcvClassification, @@ -50,22 +55,26 @@ "CcvClassification", "ClinGenClassification", "ClinGenClassification", - "ClinicalVariantProposition", + "ComputationalVariantFunctionalImpactAnalysisResult", "CohortAlleleFrequencyStudyResult", "Condition", "ConditionSet", "Contribution", "CoreType", "DataSet", + "DataItem", "DiagnosticPredicate", "Direction", "Document", "EvidenceLine", "ExperimentalVariantFunctionalImpactProposition", "ExperimentalVariantFunctionalImpactStudyResult", + "GeneDiseaseValidityProposition", + "GeneticContextVariantProposition", "ExperimentalVariantFunctionalImpactStudyResult", "InformationEntity", "Method", + "NO_CRITERIA_MET", "PrognosticPredicate", "Proposition", "STRENGTH_CODES", @@ -77,10 +86,11 @@ "StudyResult", "SubjectVariantProposition", "System", - "Therapeutic", + "Therapy", "TherapeuticResponsePredicate", "TherapyGroup", "VariantClinicalSignificanceProposition", + "VariantMolecularConsequenceProposition", "VariantDiagnosticProposition", "VariantOncogenicityProposition", "VariantPathogenicityProposition", diff --git a/src/ga4gh/va_spec/base/core.py b/src/ga4gh/va_spec/base/core.py index 79e9f1a..a319e0a 100644 --- a/src/ga4gh/va_spec/base/core.py +++ b/src/ga4gh/va_spec/base/core.py @@ -2,28 +2,35 @@ from __future__ import annotations -from abc import ABC from datetime import date, datetime from enum import Enum -from typing import Annotated, ClassVar, Literal, TypeAlias +from typing import Annotated, ClassVar, Literal from pydantic import ( + BaseModel, ConfigDict, Field, - RootModel, StringConstraints, + field_validator, ) from ga4gh.cat_vrs.models import CategoricalVariant from ga4gh.core.metadata import Maturity from ga4gh.core.models import ( BaseModelForbidExtra, + ConceptSet, Entity, MappableConcept, iriReference, ) -from ga4gh.va_spec.base.domain_entities import Condition, Therapeutic +from ga4gh.va_spec.base.domain_entities import ( + Condition, + ConditionSet, + Therapy, + TherapyGroup, +) from ga4gh.va_spec.base.enums import ( + NO_CRITERIA_MET, DiagnosticPredicate, PrognosticPredicate, System, @@ -33,7 +40,7 @@ from ga4gh.va_spec.base.validators import ( validate_mappable_concept, ) -from ga4gh.vrs.models import Allele, MolecularVariation +from ga4gh.vrs.models import Adjacency, Allele, MolecularVariation class CoreType(str, Enum): @@ -49,6 +56,44 @@ class CoreType(str, Enum): STUDY_GROUP = "StudyGroup" +def _concrete_subclasses(model_class: type) -> list[type]: + """Return loaded descendants of a model class. + + :param model_class: Base class whose descendants to return. + :returns: Loaded descendant model classes. + """ + subclasses = [] + for subclass in model_class.__subclasses__(): + subclasses.append(subclass) + subclasses.extend(_concrete_subclasses(subclass)) + return subclasses + + +def _resolve_typed_model(value: object, base_class: type) -> object: + """Instantiate a loaded subclass selected by an object's ``type``. + + Resolves schema references without a separate type registry. + + :param value: Value to resolve when it is a typed object. + :param base_class: Base class of the candidate models. + :returns: A concrete model instance or the original value. + :raises ValidationError: If a matching concrete model rejects the value. + """ + if not isinstance(value, dict) or not isinstance(value.get("type"), str): + return value + + object_type = value["type"] + for model_class in _concrete_subclasses(base_class): + type_field = model_class.model_fields.get("type") + if ( + "type" in model_class.__annotations__ + and type_field is not None + and type_field.default == object_type + ): + return model_class.model_validate(value) + return value + + class Agent(BaseMetadataMixin, Entity, BaseModelForbidExtra): """An autonomous actor (person, organization, or software agent) that bears some form of responsibility for an activity taking place, for the existence of an entity, @@ -69,8 +114,8 @@ class Agent(BaseMetadataMixin, Entity, BaseModelForbidExtra): class Contribution(BaseMetadataMixin, Entity, BaseModelForbidExtra): """An action taken by an agent in contributing to the creation, modification, - assessment, or deprecation of a particular entity (e.g. a Statement, EvidenceLine, - DataSet, Publication, etc.) + assessment, or deprecation of a particular entity (e.g. a Statement, DataSet, + Publication, etc.) """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE @@ -79,7 +124,7 @@ class Contribution(BaseMetadataMixin, Entity, BaseModelForbidExtra): default=CoreType.CONTRIBUTION.value, description=f"MUST be '{CoreType.CONTRIBUTION.value}'.", ) - contributor: Agent | None = Field( + contributor: Agent | iriReference | None = Field( default=None, description="The agent that made the contribution." ) activityType: str | None = Field( @@ -153,6 +198,7 @@ class InformationEntity(BaseMetadataMixin, Entity): """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE + _abstract: ClassVar[bool] = True specifiedBy: Method | iriReference | None = Field( default=None, @@ -168,6 +214,23 @@ class InformationEntity(BaseMetadataMixin, Entity): ) +class DataItem(InformationEntity, BaseModelForbidExtra): + """An Information Entity representing an individual piece of data, generated or + acquired through methods which reliably produce truthful information about + something. + """ + + _maturity: ClassVar[Maturity] = Maturity.DRAFT + + type: Literal["DataItem"] = Field( + default="DataItem", description='Must be "DataItem"' + ) + value: dict | str | iriReference | None = Field( + default=None, + description="The value of the data item (could be a structured value, a quantitative or qualitative value, or an IRI reference to either).", + ) + + class DataSet(BaseMetadataMixin, Entity, BaseModelForbidExtra): """A collection of related data items or records that are organized together in a common format or structure, to enable their computational manipulation as a unit. @@ -194,7 +257,7 @@ class DataSet(BaseMetadataMixin, Entity, BaseModelForbidExtra): default=None, description="The version of the DataSet, as assigned by its creator.", ) - license: MappableConcept | None = Field( + license: MappableConcept | iriReference | None = Field( default=None, description="A specific license that dictates legal permissions for how a data set can be used (by whom, where, for what purposes, with what additional requirements, etc.)", ) @@ -217,21 +280,26 @@ class StudyGroup(BaseMetadataMixin, Entity, BaseModelForbidExtra): default=None, description="The total number of individual members in the StudyGroup.", ) - characteristics: list[MappableConcept] | None = Field( + characteristics: list[MappableConcept | iriReference] | None = Field( default=None, description="A feature or role shared by all members of the StudyGroup, representing a criterion for membership in the group.", ) -class _StudyResult(InformationEntity, ABC): - """A collection of data items from a single study that pertain to a particular subject - or experimental unit in the study, along with optional provenance information - describing how these data items were generated. +class StudyResult(InformationEntity): + """A collection of data items from a single study that pertain to a particular + subject or experimental unit in the study, along with optional provenance + information describing how these data items were generated. """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE + _abstract: ClassVar[bool] = True - sourceDataSet: DataSet | None = Field( + focus: Entity | iriReference = Field( + ..., + description="The specific participant, subject or experimental unit in a Study that data included in the StudyResult object is about - e.g. a particular variant in a population allele frequency dataset like ExAC or gnomAD.", + ) + sourceDataSet: DataSet | iriReference | None = Field( default=None, description="A larger DataSet from which the data included in the StudyResult was taken or derived.", ) @@ -245,7 +313,7 @@ class _StudyResult(InformationEntity, ABC): ) -class CohortAlleleFrequencyStudyResult(_StudyResult, BaseModelForbidExtra): +class CohortAlleleFrequencyStudyResult(StudyResult, BaseModelForbidExtra): """A StudyResult that reports measures related to the frequency of an Allele in a cohort""" _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE @@ -254,24 +322,24 @@ class CohortAlleleFrequencyStudyResult(_StudyResult, BaseModelForbidExtra): default="CohortAlleleFrequencyStudyResult", description="MUST be 'CohortAlleleFrequencyStudyResult'.", ) - sourceDataSet: DataSet | None = Field( + sourceDataSet: DataSet | iriReference | None = Field( default=None, description="The dataset from which the CohortAlleleFrequencyStudyResult was reported.", ) - focusAllele: Allele | iriReference = Field( + focus: Allele | iriReference = Field( ..., description="The Allele for which frequency results are reported." ) - focusAlleleCount: int = Field( - ..., description="The number of occurrences of the focusAllele in the cohort." + focusCount: int = Field( + ..., description="The number of occurrences of the focus Allele in the cohort." ) - locusAlleleCount: int = Field( + locusCount: int = Field( ..., description="The number of occurrences of all alleles at the locus in the cohort.", ) - focusAlleleFrequency: int | float = Field( - ..., description="The frequency of the focusAllele in the cohort." + alleleFrequency: int | float = Field( + ..., description="The frequency of the focus Allele in the cohort." ) - cohort: StudyGroup = Field( + cohort: StudyGroup | iriReference = Field( ..., description="The cohort from which the frequency was derived." ) subCohortFrequency: list[CohortAlleleFrequencyStudyResult] | None = Field( @@ -280,7 +348,7 @@ class CohortAlleleFrequencyStudyResult(_StudyResult, BaseModelForbidExtra): ) -class TumorVariantFrequencyStudyResult(_StudyResult, BaseModelForbidExtra): +class TumorVariantFrequencyStudyResult(StudyResult, BaseModelForbidExtra): """A Study Result that reports measures related to the frequency of an variant across different tumor types. """ @@ -291,11 +359,11 @@ class TumorVariantFrequencyStudyResult(_StudyResult, BaseModelForbidExtra): default="TumorVariantFrequencyStudyResult", description="MUST be 'TumorVariantFrequencyStudyResult'.", ) - sourceDataSet: DataSet | None = Field( + sourceDataSet: DataSet | iriReference | None = Field( default=None, description="The dataset from which data in the Tumor Variant Frequency Study Result was taken.", ) - focusVariant: Allele | CategoricalVariant | iriReference = Field( + focus: Allele | CategoricalVariant | iriReference = Field( ..., description="The variant for which frequency data is reported in the Study Result.", ) @@ -311,7 +379,7 @@ class TumorVariantFrequencyStudyResult(_StudyResult, BaseModelForbidExtra): ..., description="The frequency of tumor samples that include the focus variant in the sample group.", ) - sampleGroup: StudyGroup | None = Field( + sampleGroup: StudyGroup | iriReference | None = Field( default=None, description="The set of samples about which the frequency data was generated.", ) @@ -321,9 +389,7 @@ class TumorVariantFrequencyStudyResult(_StudyResult, BaseModelForbidExtra): ) -class ExperimentalVariantFunctionalImpactStudyResult( - _StudyResult, BaseModelForbidExtra -): +class ExperimentalVariantFunctionalImpactStudyResult(StudyResult, BaseModelForbidExtra): """A StudyResult that reports a functional impact score from a variant functional assay or study.""" _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE @@ -332,7 +398,7 @@ class ExperimentalVariantFunctionalImpactStudyResult( default="ExperimentalVariantFunctionalImpactStudyResult", description="MUST be 'ExperimentalVariantFunctionalImpactStudyResult'.", ) - focusVariant: MolecularVariation | iriReference = Field( + focus: MolecularVariation | iriReference = Field( ..., description="The genetic variant for which a functional impact score is generated.", ) @@ -344,32 +410,12 @@ class ExperimentalVariantFunctionalImpactStudyResult( default=None, description="The assay that was performed to generate the reported functional impact score.", ) - sourceDataSet: DataSet | None = Field( + sourceDataSet: DataSet | iriReference | None = Field( default=None, description="The full data set that provided the reported the functional impact score.", ) -class StudyResult(BaseMetadataMixin, RootModel): - """A collection of data items from a single study that pertain to a particular subject - or experimental unit in the study, along with optional provenance information - describing how these data items were generated. - """ - - _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - - root: ( - CohortAlleleFrequencyStudyResult - | ExperimentalVariantFunctionalImpactStudyResult - ) = Field( - ..., - json_schema_extra={ - "description": "A collection of data items from a single study that pertain to a particular subject or experimental unit in the study, along with optional provenance information describing how these data items were generated." - }, - discriminator="type", - ) - - class Proposition(BaseMetadataMixin, Entity): """An abstract entity representing a possible fact that may be true or false. As abstract entities, Propositions capture a 'sharable' piece of meaning whose identify @@ -378,36 +424,67 @@ class Proposition(BaseMetadataMixin, Entity): """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE + _abstract: ClassVar[bool] = True - subject: dict = Field( + subject: dict | iriReference = Field( ..., description="The Entity or concept about which the Proposition is made." ) predicate: str = Field( ..., description="The relationship declared to hold between the subject and the object of the Proposition.", ) - object: dict = Field( + object: dict | iriReference = Field( ..., description="An Entity or concept that is related to the subject of a Proposition via its predicate.", ) -class _SubjectVariantPropositionBase(BaseMetadataMixin, Entity, ABC): +class GeneDiseaseValidityProposition(Proposition, BaseModelForbidExtra): + """A proposition used to describe knowledge about a variant that includes or depends + on its genetic context - e.g. the allelic origin of the variant, or its relationship + to a specific gene. + """ + + _maturity: ClassVar[Maturity] = Maturity.DRAFT + + type: Literal["GeneDiseaseValidityProposition"] = Field( + default="GeneDiseaseValidityProposition", + description='MUST be "GeneDiseaseValidityProposition".', + ) + subject: MappableConcept | iriReference = Field( + ..., description="The Entity or concept about which the Proposition is made." + ) + predicate: Literal["variantsInGeneCausalFor"] = Field( + default="variantsInGeneCausalFor", + description='MUST be "variantsInGeneCausalFor".', + ) + object: MappableConcept | iriReference = Field( + ..., + description="An Entity or concept that is related to the subject of a Proposition via its predicate.", + ) + modeOfInheritanceQualifier: MappableConcept | iriReference | None = None + + +class SubjectVariantProposition(Proposition, BaseModel): + """A `Proposition` that has a variant as the subject.""" + _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE + _abstract: ClassVar[bool] = True - subjectVariant: MolecularVariation | CategoricalVariant | iriReference = Field( + subject: MolecularVariation | CategoricalVariant | iriReference = Field( ..., description="A variant that is the subject of the Proposition." ) -class ClinicalVariantProposition(_SubjectVariantPropositionBase): - """A proposition for use in describing the effect of variants in human subjects.""" +class GeneticContextVariantProposition(SubjectVariantProposition): + """A variant proposition that includes or depends on genetic context.""" _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE + _abstract: ClassVar[bool] = True condition_field_name: ClassVar[str] @property - def condition(self) -> Condition | iriReference: + def condition(self) -> Condition | ConditionSet | iriReference: """Return the condition associated with the proposition.""" return getattr(self, self.condition_field_name) @@ -421,8 +498,55 @@ def condition(self) -> Condition | iriReference: ) +class VariantMolecularConsequenceProposition( + SubjectVariantProposition, BaseModelForbidExtra +): + """A Proposition describing a predicted molecular consequence of a variant on + transcript and protein molecules - typically reporting the type of sequence feature + affected (e.g. 'intron variant', 'splice-site variant'), or an impact on the + processing of the molecule along the path from gene to transcript to polypeptide + (e.g. 'missense variant', 'frameshift variant'). Note that annotations about variant + impact on gene product function, which may occur downstream of a molecular + consequence, are not in scope here. These are covered by a Variant Functional Impact + Proposition classes. + """ + + _maturity: ClassVar[Maturity] = Maturity.DRAFT + + type: Literal["VariantMolecularConsequenceProposition"] = Field( + default="VariantMolecularConsequenceProposition", + description='MUST be "VariantMolecularConsequenceProposition".', + ) + subject: Allele | Adjacency | iriReference = Field( + ..., + description="MUST be a genomic variant specified against genomic reference sequence(s). This practice most directly reflects data conventions from sources like VEP, and aligns with the intended use of Molecular Consequence data. There are separate qualifier attributes in the model to capture transcript and/or protein level representations of the subject, that indicate the variant context in which the reported consequence(s) are actually manifest.", + ) + predicate: Literal["hasMolecularConsequence"] = Field( + default="hasMolecularConsequence", + description='The relationship the Proposition describes between the subject variant and object consequence terms for which the molecular consequence applies. MUST be "hasMolecularConsequence".', + ) + object: MappableConcept | ConceptSet | iriReference = Field( + ..., + description="The molecular consequence(s) of the subject variant, in the context of the qualifying transcript and/or protein variation context(s). These are typically terms from the 'structural_variant' branch of the Sequence Ontology, e.g. 'SO:0001627' (intron_variant), or 'SO:0001589' (frameshift_variant). If more than one consequence term apply, use a ConceptSet to capture them.", + ) + transcriptVariationContextQualifier: Allele | Adjacency | iriReference | None = ( + Field( + default=None, + description="The subject genomic variant as projected on a particular transcript or mRNA molecule. The reported relationship between the subject variant and object consequence terms holds specifically in the context of this transcript variation. A transcript variation context MUST be reported, unless the subject is an intergenic variant.", + ) + ) + proteinVariationContextQualifier: Allele | Adjacency | iriReference | None = Field( + default=None, + description="The subject genomic variant as projected on a particular protein molecule. The reported relationship between the subject variant and object consequence terms holds specifically in the context of this protein variation. A protein variation context MUST be accompanied by its corresponding transcript variation context.", + ) + geneContextQualifier: MappableConcept | iriReference | None = Field( + default=None, + description="The gene for which this statement reports a VariantMolecularConsequence.", + ) + + class ExperimentalVariantFunctionalImpactProposition( - _SubjectVariantPropositionBase, BaseModelForbidExtra + SubjectVariantProposition, BaseModelForbidExtra ): """A Proposition describing the impact of a variant on the function sequence feature (typically a gene or gene product). @@ -438,7 +562,7 @@ class ExperimentalVariantFunctionalImpactProposition( default="impactsFunctionOf", description="The relationship the Proposition describes between the subject variant and object sequence feature whose function it may alter. MUST be 'impactsFunctionOf'.", ) - objectSequenceFeature: iriReference | MappableConcept = Field( + object: MappableConcept | iriReference = Field( ..., description="The sequence feature (typically a gene or gene product) on whose function the impact of the subject variant is reported.", ) @@ -449,14 +573,14 @@ class ExperimentalVariantFunctionalImpactProposition( class VariantClinicalSignificanceProposition( - ClinicalVariantProposition, BaseModelForbidExtra + GeneticContextVariantProposition, BaseModelForbidExtra ): """A Proposition describing the clinical significance of a variant with respect to a condition. """ _maturity: ClassVar[Maturity] = Maturity.DRAFT - condition_field_name: ClassVar[str] = "objectCondition" + condition_field_name: ClassVar[str] = "object" model_config = ConfigDict(use_enum_values=True) @@ -468,12 +592,15 @@ class VariantClinicalSignificanceProposition( default="hasClinicalSignificanceFor", description="The predicate associating the subject variant to clinical significance for the object Condition. MUST be 'hasClinicalSignificanceFor'.", ) - objectCondition: Condition | iriReference = Field( - ..., description="The condition that is evaluated." + object: Condition | ConditionSet | iriReference = Field( + ..., + description="The condition that is evaluated.", ) -class VariantDiagnosticProposition(ClinicalVariantProposition, BaseModelForbidExtra): +class VariantDiagnosticProposition( + GeneticContextVariantProposition, BaseModelForbidExtra +): """A Proposition about whether a variant is associated with a disease (a diagnostic inclusion criterion), or absence of a disease (diagnostic exclusion criterion). """ @@ -481,7 +608,7 @@ class VariantDiagnosticProposition(ClinicalVariantProposition, BaseModelForbidEx model_config = ConfigDict(use_enum_values=True) _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - condition_field_name: ClassVar[str] = "objectCondition" + condition_field_name: ClassVar[str] = "object" type: Literal["VariantDiagnosticProposition"] = Field( default="VariantDiagnosticProposition", @@ -491,16 +618,19 @@ class VariantDiagnosticProposition(ClinicalVariantProposition, BaseModelForbidEx ..., description="The relationship the Proposition describes between the subject variant and object Condition. MUST be one of 'isDiagnosticInclusionCriterionFor' or 'isDiagnosticExclusionCriterionFor'.", ) - objectCondition: Condition | iriReference = Field( - ..., description="The disease that is evaluated for diagnosis." + object: Condition | ConditionSet | iriReference = Field( + ..., + description="The disease that is evaluated for diagnosis.", ) -class VariantOncogenicityProposition(ClinicalVariantProposition, BaseModelForbidExtra): +class VariantOncogenicityProposition( + GeneticContextVariantProposition, BaseModelForbidExtra +): """A proposition describing the role of a variant in causing a tumor type.""" _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - condition_field_name: ClassVar[str] = "objectTumorType" + condition_field_name: ClassVar[str] = "object" type: Literal["VariantOncogenicityProposition"] = Field( default="VariantOncogenicityProposition", @@ -510,16 +640,19 @@ class VariantOncogenicityProposition(ClinicalVariantProposition, BaseModelForbid default="isOncogenicFor", description="The relationship the Proposition describes between the subject variant and object tumor type. MUST be 'isOncogenicFor'.", ) - objectTumorType: Condition | iriReference = Field( - ..., description="The tumor type for which the variant impact is evaluated." + object: Condition | ConditionSet | iriReference = Field( + ..., + description="The tumor type for which the variant impact is evaluated.", ) -class VariantPathogenicityProposition(ClinicalVariantProposition, BaseModelForbidExtra): +class VariantPathogenicityProposition( + GeneticContextVariantProposition, BaseModelForbidExtra +): """A proposition describing the role of a variant in causing a heritable condition.""" _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - condition_field_name: ClassVar[str] = "objectCondition" + condition_field_name: ClassVar[str] = "object" type: Literal["VariantPathogenicityProposition"] = Field( default="VariantPathogenicityProposition", @@ -529,26 +662,29 @@ class VariantPathogenicityProposition(ClinicalVariantProposition, BaseModelForbi default="isCausalFor", description="The relationship the Proposition describes between the subject variant and object condition. MUST be 'isCausalFor'.", ) - objectCondition: Condition | iriReference = Field( - ..., description="The Condition for which the variant impact is stated." + object: Condition | ConditionSet | iriReference = Field( + ..., + description="The Condition or ConditionSet for which the variant impact is stated.", ) - penetranceQualifier: MappableConcept | None = Field( + penetranceQualifier: MappableConcept | iriReference | None = Field( default=None, description="Reports the penetrance of the pathogenic effect - i.e. the extent to which the variant impact is expressed by individuals carrying it as a measure of the proportion of carriers exhibiting the condition.", ) - modeOfInheritanceQualifier: MappableConcept | None = Field( + modeOfInheritanceQualifier: MappableConcept | iriReference | None = Field( default=None, description="Reports a pattern of inheritance expected for the pathogenic effect of the variant. Consider using terms or codes from community terminologies here - e.g. terms from the 'Mode of inheritance' branch of the Human Phenotype Ontology such as HP:0000006 (autosomal dominant inheritance).", ) -class VariantPrognosticProposition(ClinicalVariantProposition, BaseModelForbidExtra): +class VariantPrognosticProposition( + GeneticContextVariantProposition, BaseModelForbidExtra +): """A Proposition about whether a variant is associated with an improved or worse outcome for a disease.""" model_config = ConfigDict(use_enum_values=True) _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - condition_field_name: ClassVar[str] = "objectCondition" + condition_field_name: ClassVar[str] = "object" type: Literal["VariantPrognosticProposition"] = Field( default="VariantPrognosticProposition", @@ -558,13 +694,14 @@ class VariantPrognosticProposition(ClinicalVariantProposition, BaseModelForbidEx ..., description="The relationship the Proposition describes between the subject variant and object Condition. MUST be one of 'associatedWithBetterOutcomeFor' or 'associatedWithWorseOutcomeFor'.", ) - objectCondition: Condition | iriReference = Field( - ..., description="The disease that is evaluated for outcome." + object: Condition | ConditionSet | iriReference = Field( + ..., + description="The disease that is evaluated for outcome.", ) class VariantTherapeuticResponseProposition( - ClinicalVariantProposition, BaseModelForbidExtra + GeneticContextVariantProposition, BaseModelForbidExtra ): """A Proposition about the role of a variant in modulating the response of a neoplasm to drug administration or other therapeutic procedures. @@ -583,37 +720,16 @@ class VariantTherapeuticResponseProposition( ..., description='The relationship the Proposition describes between the subject variant and object therapeutic. MUST be one of "predictsSensitivityTo" or "predictsResistanceTo".', ) - objectTherapeutic: Therapeutic | iriReference = Field( + object: Therapy | TherapyGroup | iriReference = Field( ..., description="A drug administration or other therapeutic procedure that the neoplasm is intended to respond to.", ) - conditionQualifier: Condition | iriReference = Field( + conditionQualifier: Condition | ConditionSet | iriReference = Field( ..., description="Reports the disease context in which the variant's association with therapeutic sensitivity or resistance is evaluated. Note that this is a required qualifier in therapeutic response propositions.", ) -# Any new proposition type should be added to this union, and ONLY this union -# should be used when annotating a proposition property -_SubjectVariantPropositionType: TypeAlias = ( - ExperimentalVariantFunctionalImpactProposition - | VariantPathogenicityProposition - | VariantDiagnosticProposition - | VariantPrognosticProposition - | VariantOncogenicityProposition - | VariantTherapeuticResponseProposition - | VariantClinicalSignificanceProposition -) - - -class SubjectVariantProposition(BaseMetadataMixin, RootModel): - """A `Proposition` that has a variant as the subject.""" - - _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - - root: _SubjectVariantPropositionType = Field(discriminator="type") - - class Direction(str, Enum): """A term indicating whether the Statement supports, disputes, or remains neutral w.r.t. the validity of the Proposition it evaluates. @@ -624,6 +740,67 @@ class Direction(str, Enum): DISPUTES = "disputes" +class ComputationalVariantFunctionalImpactAnalysisResult( + StudyResult, BaseModelForbidExtra +): + """A computational assessment of a variant's functional impact.""" + + _maturity: ClassVar[Maturity] = Maturity.DRAFT + + focus: MolecularVariation | CategoricalVariant | iriReference = Field( + ..., + description="The genetic variant for which the in silico analysis was performed.", + ) + transcriptVariationContext: Allele | iriReference = Field( + ..., + description="The transcript in the context of which the in silico analysis was performed.", + ) + impactScore: float = Field( + ..., + description="The primary numeric score produced by the in silico tool for this variant.", + ) + specifiedBy: Method | iriReference | None = Field( + default=None, + description="The in silico method or algorithm that was applied to generate the reported score(s).", + ) + contributions: list[Contribution] | None = Field( + default=None, + description="Specific actions taken by an Agent toward the creation, modification, validation, or deprecation of this Information Entity.", + ) + type: Literal["ComputationalVariantFunctionalImpactAnalysisResult"] = Field( + default="ComputationalVariantFunctionalImpactAnalysisResult", + description='MUST be "ComputationalVariantFunctionalImpactAnalysisResult".', + ) + sourceDataSet: DataSet | iriReference | None = Field( + default=None, + description="The dataset from which the in silico scores were retrieved or derived (e.g., an Ensembl VEP annotation dataset).", + ) + ancillaryResults: dict | None = Field( + default=None, + description="An object in which implementers can define custom fields to capture additional scores or outputs produced by the in silico tool beyond the primary score. For example, CADD reports both a raw score and a Phred-scaled score; the primary score field would hold one, and ancillaryResults would hold the other.", + ) + qualityMeasures: dict | None = Field( + default=None, + description="An object in which implementers can define custom fields to capture metadata about the quality/provenance of the primary data items captured in standard attributes in the main body of the Study Result. e.g. a sequencing coverage metric in a Cohort Allele Frequency Study Result.", + ) + impactScoreType: MappableConcept | iriReference | None = Field( + default=None, + description="A descriptor indicating what the score represents (e.g., 'SIFT impact score', 'CADD Phred-scaled impact score').", + ) + categoricalImpact: MappableConcept | iriReference | None = Field( + default=None, + description="The categorical interpretation derived from the score by the in silico tool (e.g., 'tolerated', 'benign', 'pathogenic').", + ) + impactedFeatureType: MappableConcept | iriReference | None = Field( + default=None, + description="A descriptor indicating the type of feature for which the focus variant has a predicted impact", + ) + impactedFeature: MappableConcept | iriReference | None = Field( + default=None, + description="The specific feature for which the focus variant has a predicted impact", + ) + + class EvidenceLine(InformationEntity, BaseModelForbidExtra): """An independent, evidence-based argument that may support or refute the validity of a specific Proposition. The strength and direction of this argument is based on @@ -639,13 +816,11 @@ class EvidenceLine(InformationEntity, BaseModelForbidExtra): default=CoreType.EVIDENCE_LINE.value, description=f"MUST be '{CoreType.EVIDENCE_LINE.value}'.", ) - targetProposition: _SubjectVariantPropositionType | None = Field( + targetProposition: Proposition | iriReference | None = Field( default=None, description="The possible fact against which evidence items contained in an Evidence Line were collectively evaluated, in determining the overall strength and direction of support they provide. For example, in an ACMG Guideline-based assessment of variant pathogenicity, the support provided by distinct lines of evidence are assessed against a target proposition that the variant is pathogenic for a specific disease.", ) - hasEvidenceItems: ( - list[StudyResult | Statement | EvidenceLine | iriReference] | None - ) = Field( + hasEvidenceItems: list[InformationEntity | iriReference] | None = Field( default=None, description="An individual piece of information that was evaluated as evidence in building the argument represented by an Evidence Line.", ) @@ -653,7 +828,7 @@ class EvidenceLine(InformationEntity, BaseModelForbidExtra): ..., description="The direction of support that the Evidence Line is determined to provide toward its target Proposition (supports, disputes, neutral)", ) - strengthOfEvidenceProvided: MappableConcept | None = Field( + strengthOfEvidenceProvided: MappableConcept | iriReference | None = Field( default=None, description="The strength of support that an Evidence Line is determined to provide for or against its target Proposition, evaluated relative to the direction indicated by the directionOfEvidenceProvided value.", ) @@ -661,11 +836,39 @@ class EvidenceLine(InformationEntity, BaseModelForbidExtra): default=None, description="A quantitative score indicating the strength of support that an Evidence Line is determined to provide for or against its target Proposition, evaluated relative to the direction indicated by the directionOfEvidenceProvided value.", ) - evidenceOutcome: MappableConcept | None = Field( + evidenceOutcome: MappableConcept | iriReference | None = Field( default=None, description="A term summarizing the overall outcome of the evidence assessment represented by the Evidence Line, in terms of the direction and strength of support it provides for or against the target Proposition.", ) + @field_validator("targetProposition", mode="before") + @classmethod + def _resolve_target_proposition(cls, value: object) -> object: + """Resolve an inline target proposition to its concrete model. + + :param value: Inline proposition value. + :returns: Resolved proposition or the original value. + :raises ValidationError: If a matching proposition model rejects the value. + """ + return _resolve_typed_model(value, Proposition) + + @field_validator("hasEvidenceItems", mode="before") + @classmethod + def _resolve_evidence_items(cls, value: object) -> object: + """Resolve inline evidence items to their concrete models. + + :param value: Inline evidence item values. + :returns: Resolved evidence items or the original value. + :raises ValueError: If a matching evidence model rejects an item. + """ + if isinstance(value, list): + try: + return [_resolve_typed_model(item, InformationEntity) for item in value] + except ValueError as error: + msg = f"validation errors for {cls.__name__}" + raise ValueError(msg) from error + return value + def _validate_evidence_outcome( self, system: System, code_pattern: str, is_required: bool = False ) -> None: @@ -715,6 +918,36 @@ def _validate_direction_of_evidence_provided(self) -> None: err_msg = f"`strengthOfEvidenceProvided` is not allowed when `directionOfEvidenceProvided` is '{Direction.NEUTRAL.value}'." raise ValueError(err_msg) + def _validate_noncontributing_evidence_outcome(self) -> None: + """Validate direction and strength for noncontributing evidence outcomes. + + ``no_criteria_met`` and criterion outcomes ending in ``_not_met`` do not + contribute evidence for or against the target proposition. + + :raises ValueError: If a noncontributing outcome has a non-neutral + direction or a strength + """ + if not self.evidenceOutcome or isinstance(self.evidenceOutcome, iriReference): + return + + outcome_code = self.evidenceOutcome.primaryCoding.code.root + if outcome_code != NO_CRITERIA_MET and not outcome_code.endswith("_not_met"): + return + + if self.directionOfEvidenceProvided != Direction.NEUTRAL: + msg = ( + "`directionOfEvidenceProvided` must be 'neutral' when " + f"`evidenceOutcome` is '{NO_CRITERIA_MET}' or ends in '_not_met'." + ) + raise ValueError(msg) + + if self.strengthOfEvidenceProvided is not None: + msg = ( + "`strengthOfEvidenceProvided` must be null when " + f"`evidenceOutcome` is '{NO_CRITERIA_MET}' or ends in '_not_met'." + ) + raise ValueError(msg) + def _validate_criterion_specified_by(self) -> None: """Validate specifiedBy for criterion evidence lines @@ -752,16 +985,15 @@ class Statement(InformationEntity, BaseModelForbidExtra): default=CoreType.STATEMENT.value, description=f"MUST be '{CoreType.STATEMENT.value}'.", ) - proposition: _SubjectVariantPropositionType = Field( + proposition: Proposition | iriReference = Field( ..., description="A possible fact, the validity of which is assessed and reported by the Statement. A Statement can put forth the proposition as being true, false, or uncertain, and may provide an assessment of the level of confidence/evidence supporting this claim.", - discriminator="type", ) - direction: Direction = Field( - ..., + direction: Direction | None = Field( + default=None, description="A term indicating whether the Statement supports, disputes, or remains neutral w.r.t. the validity of the Proposition it evaluates.", ) - strength: MappableConcept | None = Field( + strength: MappableConcept | iriReference | None = Field( default=None, description="A term used to report the strength of a Proposition's assessment in the direction indicated (i.e. how strongly supported or disputed the Proposition is believed to be). Implementers may choose to frame a strength assessment in terms of how *confident* an agent is that the Proposition is true or false, or in terms of the *strength of all evidence* they believe supports or disputes it.", ) @@ -769,15 +1001,30 @@ class Statement(InformationEntity, BaseModelForbidExtra): default=None, description="A quantitative score that indicates the strength of a Proposition's assessment in the direction indicated (i.e. how strongly supported or disputed the Proposition is believed to be). Depending on its implementation, a score may reflect how *confident* that agent is that the Proposition is true or false, or the *strength of evidence* they believe supports or disputes it. Instructions for how to interpret the meaning of a given score may be gleaned from the method or document referenced in 'specifiedBy' attribute.", ) - classification: MappableConcept | None = Field( + classification: MappableConcept | iriReference | None = Field( default=None, - description="A single term or phrase summarizing the outcome of direction and strength assessments of a Statement's Proposition, in terms of a classification of its subject.", + description="A single term or phrase summarizing the result of direction and strength assessments of a Statement's Proposition, in terms of a classification of its subject.", + ) + quality: MappableConcept | iriReference | None = Field( + default=None, + description="A term used to report the quality of the assessment of a Proposition taking into consideration the reliability of the method, the contributor's self-reporting of the rigor of the evaluation, and the overall robustness of the supporting or disputing evidence. This is useful when there is a consistent policy and authority that manages and a governing framework for evaluating the quality of evidence. Also known as trust rating, review status or ranking.", + ) + hasEvidence: list[Statement | StudyResult | DataItem | iriReference] | None = Field( + default=None, + description="An individual piece of information that was evaluated as evidence in assessing the validity of the Proposition put forth by the Statement.", ) hasEvidenceLines: list[EvidenceLine | iriReference] | None = Field( default=None, description="An evidence-based argument that supports or disputes the validity of the proposition that a Statement assesses or puts forth as true. The strength and direction of this argument (whether it supports or disputes the proposition, and how strongly) is based on an interpretation of one or more pieces of information as evidence (i.e. 'Evidence Items).", ) + @field_validator("proposition", mode="before") + @classmethod + def _resolve_proposition(cls, value: object) -> object: + """Resolve an inline proposition to its concrete model. -Statement.model_rebuild() -EvidenceLine.model_rebuild() + :param value: Inline proposition value. + :returns: Resolved proposition or the original value. + :raises ValidationError: If a matching proposition model rejects the value. + """ + return _resolve_typed_model(value, Proposition) diff --git a/src/ga4gh/va_spec/base/domain_entities.py b/src/ga4gh/va_spec/base/domain_entities.py index aa19db2..518a16c 100644 --- a/src/ga4gh/va_spec/base/domain_entities.py +++ b/src/ga4gh/va_spec/base/domain_entities.py @@ -4,30 +4,27 @@ from typing import ClassVar -from pydantic import ConfigDict, Field, RootModel +from pydantic import Field from ga4gh.core.metadata import Maturity from ga4gh.core.models import ( - BaseModelForbidExtra, - Element, + ConceptSet, MappableConcept, MembershipOperator, + iriReference, ) from ga4gh.va_spec.base.metadata import BaseMetadataMixin -class ConditionSet(BaseMetadataMixin, Element, BaseModelForbidExtra): - """A set of conditions (diseases, phenotypes, traits) that occur together or are - related, depending on the membership operator, and may manifest together in the - same patient or individually in a different subset of participants in a research - study. +class ConditionSet(BaseMetadataMixin, ConceptSet): + """A specialization of ConceptSet representing a set of conditions (diseases, + phenotypes, traits) that occur together or are related, depending on the membership + operator. Concepts are restricted to Condition and ConditionSet members. """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - model_config = ConfigDict(use_enum_values=True) - - conditions: list[MappableConcept | ConditionSet] = Field( + concepts: list[Condition | ConditionSet | iriReference] = Field( ..., min_length=2, description="A list of conditions (diseases, phenotypes, traits) that are co-occurring or related, depending on the membership operator.", @@ -38,50 +35,54 @@ class ConditionSet(BaseMetadataMixin, Element, BaseModelForbidExtra): ) -class Condition(BaseMetadataMixin, RootModel): - """A single condition (disease, phenotype, or trait), or a set of conditions - (ConditionSet). +class Condition(BaseMetadataMixin, MappableConcept): + """A specialization of MappableConcept representing a single condition (disease, + phenotype, or trait). + + Allowed conceptType values include: Condition, Phenotype, + Disease, Trait, Absent. """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - root: ConditionSet | MappableConcept = Field( - ..., - json_schema_extra={ - "description": "A single condition (disease, phenotype, or trait), or a set of conditions (ConditionSet)." - }, + conceptType: str = Field( + default="Condition", + description="A term indicating the type of concept being represented by the MappableConcept.", ) -class TherapyGroup(BaseMetadataMixin, Element, BaseModelForbidExtra): - """A group of two or more therapies that are applied in combination to a single - patient/subject, or applied individually to a different subset of participants in a - research study. +class TherapyGroup(BaseMetadataMixin, ConceptSet): + """A specialization of ConceptSet representing a group of two or more therapies that + are applied in combination to a single patient/subject, or applied individually to a + different subset of participants in a research study. + + Concepts are restricted to Therapy and TherapyGroup members. """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - model_config = ConfigDict(use_enum_values=True) - - therapies: list[MappableConcept] = Field( + concepts: list[Therapy | TherapyGroup | iriReference] = Field( ..., min_length=2, description="A list of therapies that are applied to treat a condition.", ) membershipOperator: MembershipOperator = Field( ..., - description="The logical relationship between members of the group, that indicates how they were applied in treating participants in a study. The value 'AND' indicates that all therapies in the group were applied in combination to a given patient or subject. The value 'OR' indicates that each therapy was applied individually to a distinct subset of participants in the cohort that was interrogated in a given study.", + description="The logical relationship between members of the group, that indicates how they were applied in treating participants in a study. The value 'AND' indicates that all therapies in the group were applied in combination to a given patient or subject. The value 'OR' indicates that each therapy was applied individually to a distinct subset of participants in the cohort that was interrogated in a given study.", ) -class Therapeutic(BaseMetadataMixin, RootModel): - """An individual therapy (drug, procedure, behavioral intervention, etc.), or group of therapies (TherapyGroup).""" +class Therapy(BaseMetadataMixin, MappableConcept): + """A specialization of MappableConcept representing an individual therapy (drug, + procedure, behavioral intervention, etc.). + + Allowed conceptType values include: Therapy, Absent, Drug, Procedure, Behavioral + Intervention. + """ _maturity: ClassVar[Maturity] = Maturity.TRIAL_USE - root: TherapyGroup | MappableConcept = Field( - ..., - json_schema_extra={ - "description": "An individual therapy (drug, procedure, behavioral intervention, etc.), or group of therapies (TherapyGroup)." - }, + conceptType: str = Field( + default="Therapy", + description="A term indicating the type of concept being represented by the MappableConcept.", ) diff --git a/src/ga4gh/va_spec/base/enums.py b/src/ga4gh/va_spec/base/enums.py index 73b612e..1b64708 100644 --- a/src/ga4gh/va_spec/base/enums.py +++ b/src/ga4gh/va_spec/base/enums.py @@ -2,6 +2,9 @@ from enum import Enum +# Evidence outcome code used when an assessment meets no criteria. +NO_CRITERIA_MET = "no_criteria_met" + class DiagnosticPredicate(str, Enum): """Define constraints for diagnostic predicate""" diff --git a/src/ga4gh/va_spec/base/metadata.py b/src/ga4gh/va_spec/base/metadata.py index 7d4566e..0a214ac 100644 --- a/src/ga4gh/va_spec/base/metadata.py +++ b/src/ga4gh/va_spec/base/metadata.py @@ -6,4 +6,4 @@ class BaseMetadataMixin(VASpecMetadataMixin): """Provide metadata shared by models in the base namespace.""" - _schema_namespace = "base" + _schema_namespace = "" diff --git a/src/ga4gh/va_spec/base/validators.py b/src/ga4gh/va_spec/base/validators.py index 080af60..997e8bd 100644 --- a/src/ga4gh/va_spec/base/validators.py +++ b/src/ga4gh/va_spec/base/validators.py @@ -5,17 +5,17 @@ from types import MappingProxyType from typing import ClassVar, Generic, TypeVar -from ga4gh.core.models import MappableConcept -from ga4gh.va_spec.base.enums import System +from ga4gh.core.models import MappableConcept, iriReference +from ga4gh.va_spec.base.enums import NO_CRITERIA_MET, System def validate_mappable_concept( - mc: MappableConcept | None, + mc: MappableConcept | iriReference | None, valid_system: System, valid_codes: list[str] | None = None, code_pattern: str | None = None, mc_is_required: bool = False, -) -> MappableConcept | None: +) -> MappableConcept | iriReference | None: """Validate GKS Core Mappable Concept object :param mc: Mappable Concept object @@ -34,6 +34,9 @@ def validate_mappable_concept( raise ValueError(msg) return None + if isinstance(mc, iriReference): + return mc + if not mc.primaryCoding: msg = "`primaryCoding` is required." raise ValueError(msg) @@ -123,10 +126,14 @@ def _validate_method_type_evidence_outcome( :raises ValueError: If the evidence outcome criterion is invalid or is not valid for the specified method type, or if method type is invalid """ - parsed_method_type = cls.MethodType(method_type) + try: + parsed_method_type = cls.MethodType(method_type) + except ValueError as e: + msg = f"{method_type!r} is not a valid {cls.__qualname__}.MethodType" + raise ValueError(msg) from e allowed_criteria = cls.ALLOWED_CRITERIA_BY_METHOD_TYPE[parsed_method_type] - if not evidence_outcome_code: + if not evidence_outcome_code or evidence_outcome_code == NO_CRITERIA_MET: return criterion = cls._get_base_criterion_from_code(evidence_outcome_code) diff --git a/src/ga4gh/va_spec/ccv_2022/models.py b/src/ga4gh/va_spec/ccv_2022/models.py index fe0da59..df3db3d 100644 --- a/src/ga4gh/va_spec/ccv_2022/models.py +++ b/src/ga4gh/va_spec/ccv_2022/models.py @@ -13,7 +13,6 @@ from ga4gh.core.metadata import Maturity from ga4gh.core.models import MappableConcept, iriReference from ga4gh.va_spec.base.core import ( - Direction, Document, EvidenceLine, Method, @@ -22,6 +21,7 @@ ) from ga4gh.va_spec.base.enums import ( CCV_CLASSIFICATIONS, + NO_CRITERIA_MET, STRENGTH_CODES, STRENGTH_OF_EVIDENCE_PROVIDED_VALUES, System, @@ -34,7 +34,7 @@ SYSTEM = System.CCV CCV_CODE_PATTERN = ( - r"^(" + rf"^(?:{NO_CRITERIA_MET}|" r"(?:OVS1|SBVS1)(?:_(?:not_met|(?:strong|moderate|supporting)))?" r"|(?:OS[1-3]|SBS[1-2])(?:_(?:not_met|(?:very_strong|moderate|supporting)))?" r"|(?:OM[1-4])(?:_(?:not_met|(?:very_strong|strong|supporting)))?" @@ -73,15 +73,11 @@ class VariantOncogenicityEvidenceLine( _maturity: ClassVar[Maturity] = Maturity.DRAFT - targetProposition: VariantOncogenicityProposition | None = Field( + targetProposition: VariantOncogenicityProposition | iriReference | None = Field( default=None, description="A Variant Oncoogenicity Proposition against which a specific type of evidence was assessed, to determine the strength and direction of support this evidence provides for or against the proposition's validity.", ) - directionOfEvidenceProvided: Direction = Field( - ..., - description="The direction of support that the Evidence Line is determined to provide toward its target Proposition (supports, disputes, neutral). For CCV-based assessments, if a oncogenicity criterion is 'met' in the Evidence Line the direction is 'supports', if a benignity criterion is 'met' the direction is 'disputes', and if a criteria is 'not met' the direction is 'none'.", - ) - strengthOfEvidenceProvided: MappableConcept | None = Field( + strengthOfEvidenceProvided: MappableConcept | iriReference | None = Field( default=None, description="The strength of support that an Evidence Line is determined to provide for or against the proposed oncogenicity of the assessed variant. Strength is evaluated relative to the direction indicated by the 'directionOfEvidenceProvided' attribute, and captured using a MappableConcept, whose nested 'code' field is bound to an enumerated set of values. Conditional requirement: if `directionOfEvidenceProvided` is either 'supports' or 'disputes', then this attribute is required. If it is 'none', then this attribute is not allowed.", ) @@ -117,93 +113,95 @@ class MethodType(str, Enum): # Assessment of whether population control frequency refutes # oncogenicity or whether absence/extreme rarity in controls provides # supporting evidence for oncogenicity - POPULATION_FREQUENCY = "population_frequency" + POPULATION_DATA_ASSESSMENT = "population_data_assessment" # Assessment of well-established in vitro or in vivo functional studies # to determine whether experimental evidence supports or refutes an # oncogenic effect - FUNCTIONAL_ASSAY = "functional_assay" + FUNCTIONAL_DATA_ASSESSMENT = "functional_data_assessment" # Assessment of the primary sequence-level consequence of the variant, # including null/loss-of-function effects, protein-length changes, # stop-loss effects, or synonymous variants predicted to have no # splice or conservation impact - PRIMARY_SEQUENCE_CONSEQUENCE = "primary_sequence_consequence" + PRIMARY_SEQUENCE_CONSEQUENCE_ASSESSMENT = ( + "primary_sequence_consequence_assessment" + ) # Assessment of whether the variant occurs in a critical and # well-established functional domain or region, such as an enzyme # active site - FUNCTIONAL_DOMAIN_LOCATION = "functional_domain_location" + FUNCTIONAL_DOMAIN_ASSESSMENT = "functional_domain_assessment" # Assessment by analogy to previously established oncogenic variants, # including the same amino acid change or a different missense change # at the same residue - AMINO_ACID_OR_RESIDUE_ANALOGY = "amino_acid_or_residue_analogy" + AMINO_ACID_ANALOGY_ASSESSMENT = "amino_acid_analogy_assessment" # Assessment of somatic recurrence at cancer hotspots or recurrently # mutated residues, with evidence strength based on recurrence # thresholds - SOMATIC_HOTSPOT_RECURRENCE = "somatic_hotspot_recurrence" + SOMATIC_HOTSPOT_ASSESSMENT = "somatic_hotspot_assessment" # Aggregate assessment of computational predictions, including # conservation, missense-effect, and splice-effect tools, supporting # either oncogenic effect or no effect - COMPUTATIONAL_PREDICTION = "computational_prediction" + IN_SILICO_IMPACT_ASSESSMENT = "in_silico_impact_assessment" # Assessment of whether the variant occurs in a gene and malignancy # context where the disease has a single genetic etiology, making that # gene-level event supportive of oncogenicity - SINGLE_GENETIC_ETIOLOGY_CONTEXT = "single_genetic_etiology_context" + SINGLE_GENETIC_ETIOLOGY_ASSESSMENT = "single_genetic_etiology_assessment" ALLOWED_CRITERIA_BY_METHOD_TYPE: ClassVar[ MappingProxyType[MethodType, frozenset[Criterion]] ] = MappingProxyType( { - MethodType.POPULATION_FREQUENCY: frozenset( + MethodType.POPULATION_DATA_ASSESSMENT: frozenset( { Criterion.SBVS1, Criterion.SBS1, Criterion.OP4, } ), - MethodType.FUNCTIONAL_ASSAY: frozenset( + MethodType.FUNCTIONAL_DATA_ASSESSMENT: frozenset( { Criterion.OS2, Criterion.SBS2, } ), - MethodType.PRIMARY_SEQUENCE_CONSEQUENCE: frozenset( + MethodType.PRIMARY_SEQUENCE_CONSEQUENCE_ASSESSMENT: frozenset( { Criterion.OVS1, Criterion.OM2, Criterion.SBP2, } ), - MethodType.FUNCTIONAL_DOMAIN_LOCATION: frozenset( + MethodType.FUNCTIONAL_DOMAIN_ASSESSMENT: frozenset( { Criterion.OM1, } ), - MethodType.AMINO_ACID_OR_RESIDUE_ANALOGY: frozenset( + MethodType.AMINO_ACID_ANALOGY_ASSESSMENT: frozenset( { Criterion.OS1, Criterion.OM4, } ), - MethodType.SOMATIC_HOTSPOT_RECURRENCE: frozenset( + MethodType.SOMATIC_HOTSPOT_ASSESSMENT: frozenset( { Criterion.OS3, Criterion.OM3, Criterion.OP3, } ), - MethodType.COMPUTATIONAL_PREDICTION: frozenset( + MethodType.IN_SILICO_IMPACT_ASSESSMENT: frozenset( { Criterion.OP1, Criterion.SBP1, } ), - MethodType.SINGLE_GENETIC_ETIOLOGY_CONTEXT: frozenset( + MethodType.SINGLE_GENETIC_ETIOLOGY_ASSESSMENT: frozenset( { Criterion.OP2, } @@ -239,18 +237,20 @@ def validate_model(self) -> Self: ``strengthOfEvidenceProvided`` is provided when ``directionOfEvidenceProvided`` is neutral """ - self._validate_direction_of_evidence_provided() self._validate_evidence_outcome(SYSTEM, CCV_CODE_PATTERN, is_required=False) + self._validate_noncontributing_evidence_outcome() + self._validate_direction_of_evidence_provided() self._validate_criterion_specified_by() - evidence_outcome = ( - self.evidenceOutcome.primaryCoding.code.root - if self.evidenceOutcome - else None - ) - self._validate_method_type_evidence_outcome( - self.specifiedBy.methodType, evidence_outcome - ) + if isinstance(self.specifiedBy, Method): + evidence_outcome = ( + self.evidenceOutcome.primaryCoding.code.root + if isinstance(self.evidenceOutcome, MappableConcept) + else None + ) + self._validate_method_type_evidence_outcome( + self.specifiedBy.methodType, evidence_outcome + ) return self @@ -262,15 +262,18 @@ class VariantOncogenicityStatement(CCV2022MetadataMixin, Statement): _maturity: ClassVar[Maturity] = Maturity.DRAFT - proposition: VariantOncogenicityProposition = Field( + proposition: VariantOncogenicityProposition | iriReference = Field( ..., description="A proposition about the oncogenicity of a variant, for which the study provides evidence. The validity of this proposition, and the level of confidence/evidence supporting it, may be assessed and reported by the Statement.", ) - strength: MappableConcept | None = Field( + strength: MappableConcept | iriReference | None = Field( default=None, description="The strength of support that an CCV 2022 Oncogenicity statement is determined to provide for or against the proposed oncogenicity of the assessed variant. Strength is evaluated relative to the direction indicated by the 'direction' attribute. The indicated enumeration constrains the nested MappableConcept.primaryCoding > Coding.code attribute when capturing evidence strength. Conditional requirement: if directionOfEvidenceProvided is either 'supports' or 'disputes', then this attribute is required. If it is 'neutral', then this attribute is not allowed.", ) - classification: MappableConcept + classification: MappableConcept | iriReference = Field( + ..., + description="A single term or phrase classifying the subject variant based on the result of direction and strength assessments of the Statement's Proposition, using terms from the ClinGen/CGC/VICC 2022 Guidelines for Oncogenicity.", + ) specifiedBy: Method | iriReference = Field( ..., description="The method that specifies how the oncogenicity classification is ultimately assigned to the variant, based on assessment of evidence.", @@ -279,7 +282,9 @@ class VariantOncogenicityStatement(CCV2022MetadataMixin, Statement): @field_validator("strength") @classmethod - def validate_strength(cls, v: MappableConcept | None) -> MappableConcept | None: + def validate_strength( + cls, v: MappableConcept | iriReference | None + ) -> MappableConcept | iriReference | None: """Validate strength :param v: strength @@ -292,7 +297,9 @@ def validate_strength(cls, v: MappableConcept | None) -> MappableConcept | None: @field_validator("classification") @classmethod - def validate_classification(cls, v: MappableConcept) -> MappableConcept: + def validate_classification( + cls, v: MappableConcept | iriReference + ) -> MappableConcept | iriReference: """Validate classification :param v: classification diff --git a/src/ga4gh/va_spec/metadata.py b/src/ga4gh/va_spec/metadata.py index 4ba2396..864c0c0 100644 --- a/src/ga4gh/va_spec/metadata.py +++ b/src/ga4gh/va_spec/metadata.py @@ -2,11 +2,11 @@ from typing import ClassVar -from ga4gh.core.metadata import GKSMetadataMixin +from ga4gh.core.metadata import GKMMetadataMixin from ga4gh.va_spec.version import VASPEC_VERSION -class VASpecMetadataMixin(GKSMetadataMixin): +class VASpecMetadataMixin(GKMMetadataMixin): """Expose metadata for a concrete VA-Spec model.""" _schema_namespace: ClassVar[str] @@ -16,7 +16,8 @@ class VASpecMetadataMixin(GKSMetadataMixin): @classmethod def schema_id(cls) -> str: """Return the model's canonical VA-Spec JSON Schema identifier.""" + namespace = f"/{cls._schema_namespace}" if cls._schema_namespace else "" return ( f"{cls._schema_base_uri}/{cls._product_name}/{cls._product_version}/" - f"{cls._schema_namespace}/json/{cls.__name__}" + f"json{namespace}/{cls.__name__}" ) diff --git a/src/ga4gh/va_spec/version.py b/src/ga4gh/va_spec/version.py index 51d2060..c95c9d4 100644 --- a/src/ga4gh/va_spec/version.py +++ b/src/ga4gh/va_spec/version.py @@ -1,3 +1,3 @@ """Define the VA-Spec version.""" -VASPEC_VERSION = "1.1.0-snapshot.2026-06.1" +VASPEC_VERSION = "1.1.0-ballot.2026-09.1" diff --git a/submodules/va_spec b/submodules/va_spec index 56ada2e..2fa47d8 160000 --- a/submodules/va_spec +++ b/submodules/va_spec @@ -1 +1 @@ -Subproject commit 56ada2eee167abc8b03ad28fc3f7d94ecb6578cd +Subproject commit 2fa47d8faa3539ee62791e10402d22251a1ac0f3 diff --git a/tests/conftest.py b/tests/conftest.py index 369e19f..a747b8a 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -11,7 +11,7 @@ class VaSpecSchema(str, Enum): AAC_2017 = "aac-2017" ACMG_2015 = "acmg-2015" - BASE = "base" + BASE = "va-spec" CCV_2022 = "ccv-2022" @@ -25,7 +25,7 @@ def get_va_spec_schema(label: str) -> str | None: schema = VaSpecSchema.AAC_2017 elif label.endswith(VaSpecSchema.ACMG_2015): schema = VaSpecSchema.ACMG_2015 - elif label.endswith(VaSpecSchema.BASE): + elif label.endswith(VaSpecSchema.BASE) or label == "va-spec": schema = VaSpecSchema.BASE elif label.endswith(VaSpecSchema.CCV_2022): schema = VaSpecSchema.CCV_2022 diff --git a/tests/test_ccv_derived_evidence.py b/tests/test_ccv_derived_evidence.py index 13d3c8a..35dc387 100644 --- a/tests/test_ccv_derived_evidence.py +++ b/tests/test_ccv_derived_evidence.py @@ -15,103 +15,103 @@ VariantOncogenicityEvidenceLine.Criterion.OP1, "supporting", 1, - VariantOncogenicityEvidenceLine.MethodType.COMPUTATIONAL_PREDICTION, + VariantOncogenicityEvidenceLine.MethodType.IN_SILICO_IMPACT_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OP2, "supporting", 1, - VariantOncogenicityEvidenceLine.MethodType.SINGLE_GENETIC_ETIOLOGY_CONTEXT, + VariantOncogenicityEvidenceLine.MethodType.SINGLE_GENETIC_ETIOLOGY_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OP3, "supporting", 1, - VariantOncogenicityEvidenceLine.MethodType.SOMATIC_HOTSPOT_RECURRENCE, + VariantOncogenicityEvidenceLine.MethodType.SOMATIC_HOTSPOT_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OP4, "supporting", 1, - VariantOncogenicityEvidenceLine.MethodType.POPULATION_FREQUENCY, + VariantOncogenicityEvidenceLine.MethodType.POPULATION_DATA_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OM1, "moderate", 2, - VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_DOMAIN_LOCATION, + VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_DOMAIN_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OM2, "moderate", 2, - VariantOncogenicityEvidenceLine.MethodType.PRIMARY_SEQUENCE_CONSEQUENCE, + VariantOncogenicityEvidenceLine.MethodType.PRIMARY_SEQUENCE_CONSEQUENCE_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OM3, "moderate", 2, - VariantOncogenicityEvidenceLine.MethodType.SOMATIC_HOTSPOT_RECURRENCE, + VariantOncogenicityEvidenceLine.MethodType.SOMATIC_HOTSPOT_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OM4, "moderate", 2, - VariantOncogenicityEvidenceLine.MethodType.AMINO_ACID_OR_RESIDUE_ANALOGY, + VariantOncogenicityEvidenceLine.MethodType.AMINO_ACID_ANALOGY_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OS1, "strong", 4, - VariantOncogenicityEvidenceLine.MethodType.AMINO_ACID_OR_RESIDUE_ANALOGY, + VariantOncogenicityEvidenceLine.MethodType.AMINO_ACID_ANALOGY_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OS2, "strong", 4, - VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_ASSAY, + VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_DATA_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OS3, "strong", 4, - VariantOncogenicityEvidenceLine.MethodType.SOMATIC_HOTSPOT_RECURRENCE, + VariantOncogenicityEvidenceLine.MethodType.SOMATIC_HOTSPOT_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.OVS1, "very strong", 8, - VariantOncogenicityEvidenceLine.MethodType.PRIMARY_SEQUENCE_CONSEQUENCE, + VariantOncogenicityEvidenceLine.MethodType.PRIMARY_SEQUENCE_CONSEQUENCE_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.SBP1, "supporting", -1, - VariantOncogenicityEvidenceLine.MethodType.COMPUTATIONAL_PREDICTION, + VariantOncogenicityEvidenceLine.MethodType.IN_SILICO_IMPACT_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.SBP2, "supporting", -1, - VariantOncogenicityEvidenceLine.MethodType.PRIMARY_SEQUENCE_CONSEQUENCE, + VariantOncogenicityEvidenceLine.MethodType.PRIMARY_SEQUENCE_CONSEQUENCE_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.SBS1, "strong", -4, - VariantOncogenicityEvidenceLine.MethodType.POPULATION_FREQUENCY, + VariantOncogenicityEvidenceLine.MethodType.POPULATION_DATA_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.SBS2, "strong", -4, - VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_ASSAY, + VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_DATA_ASSESSMENT, ), ( VariantOncogenicityEvidenceLine.Criterion.SBVS1, "very strong", -8, - VariantOncogenicityEvidenceLine.MethodType.POPULATION_FREQUENCY, + VariantOncogenicityEvidenceLine.MethodType.POPULATION_DATA_ASSESSMENT, ), ], ) @@ -141,13 +141,13 @@ def test_derive_onco_evidence_attributes( "OS2_moderate", "supports", 2, - VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_ASSAY, + VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_DATA_ASSESSMENT, ), ( "SBS2_moderate", "disputes", -2, - VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_ASSAY, + VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_DATA_ASSESSMENT, ), ], ) @@ -205,6 +205,6 @@ def test_derive_onco_evidence_attributes_from_not_met_outcome(): assert onco_evidence_attrs.scoreOfEvidenceProvided is None assert ( onco_evidence_attrs.specifiedBy.methodType - == VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_ASSAY.value + == VariantOncogenicityEvidenceLine.MethodType.FUNCTIONAL_DATA_ASSESSMENT.value ) VariantOncogenicityEvidenceLine(**onco_evidence_attrs.model_dump()) diff --git a/tests/test_imports.py b/tests/test_imports.py new file mode 100644 index 0000000..719da60 --- /dev/null +++ b/tests/test_imports.py @@ -0,0 +1,34 @@ +"""Regression tests for public package imports.""" + +import subprocess +import sys + + +def test_public_modules_import_and_resolve_recursive_models(): + """Import public modules and construct recursive models in a clean process.""" + script = """ +import ga4gh.va_spec +import ga4gh.va_spec.aac_2017 +import ga4gh.va_spec.acmg_2015 +import ga4gh.va_spec.base +import ga4gh.va_spec.ccv_2022 +from ga4gh.va_spec.base import Statement + +Statement( + proposition={"type": "Proposition", "subject": {}, "predicate": "relatedTo", "object": {}}, + hasEvidence=[ + { + "type": "Statement", + "proposition": {"type": "Proposition", "subject": {}, "predicate": "relatedTo", "object": {}}, + } + ], +) +""" + result = subprocess.run( + [sys.executable, "-c", script], # noqa: S603 # Controlled test process. + capture_output=True, + text=True, + check=False, + ) + + assert result.returncode == 0, result.stderr diff --git a/tests/validation/test_model_metadata.py b/tests/validation/test_model_metadata.py index 1dd3241..a8a9a20 100644 --- a/tests/validation/test_model_metadata.py +++ b/tests/validation/test_model_metadata.py @@ -4,7 +4,6 @@ from pathlib import Path import pytest -import yaml from ga4gh.core.metadata import Maturity from ga4gh.va_spec import VASPEC_VERSION, base @@ -13,55 +12,54 @@ from ga4gh.va_spec.ccv_2022 import models as ccv_2022 SCHEMA_DIR = Path(__file__).parents[2] / "submodules" / "va_spec" / "schema" / "va-spec" -SCHEMAS = ( - (base, "base", SCHEMA_DIR / "base" / "va-core-source.yaml"), - (base, "base", SCHEMA_DIR / "base" / "domain-entities-source.yaml"), - (aac_2017, "aac-2017", SCHEMA_DIR / "aac-2017" / "profile-source.yaml"), - (acmg_2015, "acmg-2015", SCHEMA_DIR / "acmg-2015" / "profile-source.yaml"), - (ccv_2022, "ccv-2022", SCHEMA_DIR / "ccv-2022" / "profile-source.yaml"), -) +JSON_DIR = SCHEMA_DIR / "json" +SCHEMA_MODULES = { + "": base, + "aac-2017": aac_2017, + "acmg-2015": acmg_2015, + "ccv-2022": ccv_2022, +} def _model_params(): - """Return model metadata discovered from all VA-Spec source YAML files.""" + """Return model metadata discovered from VA-Spec JSON Schemas.""" params = [] - for model_module, namespace, source_path in SCHEMAS: - with source_path.open() as source_file: - definitions = yaml.safe_load(source_file)["$defs"] - for name, definition in definitions.items(): - schema_path = SCHEMA_DIR / namespace / "json" / name - if not schema_path.exists(): - continue - model = getattr(model_module, name) - with schema_path.open() as schema_file: - schema = json.load(schema_file) - params.append(pytest.param(model, namespace, definition, schema, id=name)) + for schema_path in JSON_DIR.rglob("*"): + if not schema_path.is_file(): + continue + with schema_path.open() as schema_file: + schema = json.load(schema_file) + namespace = str(schema_path.parent.relative_to(JSON_DIR)) + namespace = "" if namespace == "." else namespace + model_module = SCHEMA_MODULES.get(namespace) + if model_module is None: + continue + model = getattr(model_module, schema["title"]) + params.append(pytest.param(model, schema, id=schema["title"])) assert params, "No concrete VA-Spec models discovered" return params -def test_va_spec_version_matches_source_schema(): - """The package version matches the authoritative VA-Spec source schema.""" - with SCHEMAS[0][2].open() as source_file: - source_id = yaml.safe_load(source_file)["$id"] - source_version = source_id.split("/va-spec/", maxsplit=1)[1].split("/", maxsplit=1)[ - 0 - ] - assert source_version == VASPEC_VERSION +def test_va_spec_version_matches_json_schemas(): + """The package version matches every VA-Spec JSON Schema identifier.""" + for schema_path in JSON_DIR.rglob("*"): + if not schema_path.is_file(): + continue + with schema_path.open() as schema_file: + schema_id = json.load(schema_file)["$id"] + source_version = schema_id.split("/va-spec/", maxsplit=1)[1].split( + "/", maxsplit=1 + )[0] + assert source_version == VASPEC_VERSION -@pytest.mark.parametrize( - ("model", "namespace", "definition", "schema"), _model_params() -) -def test_model_metadata(model, namespace, definition, schema): - """Model metadata matches its source and generated JSON Schemas.""" - expected_schema_id = ( - f"https://w3id.org/ga4gh/schema/va-spec/{VASPEC_VERSION}/{namespace}/json/" - f"{model.__name__}" - ) +@pytest.mark.parametrize(("model", "schema"), _model_params()) +def test_model_metadata(model, schema): + """Model metadata matches its generated JSON Schema.""" + expected_schema_id = schema["$id"] assert model.schema_id() == expected_schema_id assert "_maturity" in model.__dict__ - assert model.maturity() == Maturity(definition["maturity"]) + assert model.maturity() == Maturity(schema["maturity"]) generated_schema = model.model_json_schema() assert generated_schema["$id"] == expected_schema_id diff --git a/tests/validation/test_va_spec_fixtures_validation.py b/tests/validation/test_va_spec_fixtures_validation.py index ab68c08..941ae68 100644 --- a/tests/validation/test_va_spec_fixtures_validation.py +++ b/tests/validation/test_va_spec_fixtures_validation.py @@ -22,9 +22,15 @@ for test_def in test_definitions: - if test_def["namespace"].startswith("va-spec."): - schema = get_va_spec_schema(test_def["namespace"].split("va-spec.")[-1]) - VA_SPEC_TEST_DEFINITIONS[schema].append(test_def) + namespace = test_def["namespace"] + # Base (core) fixtures use the namespace ``va-spec``; profiles use + # ``va-spec.``. Anything else (e.g. ``vrs``) is out of scope. + if namespace != "va-spec" and not namespace.startswith("va-spec."): + continue + schema = get_va_spec_schema(namespace.split("va-spec.")[-1]) + if schema is None: + continue + VA_SPEC_TEST_DEFINITIONS[schema].append(test_def) def test_va_spec_fixtures(): diff --git a/tests/validation/test_va_spec_models.py b/tests/validation/test_va_spec_models.py index b48f326..6b95d52 100644 --- a/tests/validation/test_va_spec_models.py +++ b/tests/validation/test_va_spec_models.py @@ -16,17 +16,20 @@ VariantPathogenicityStatement, ) from ga4gh.va_spec.base import ( + NO_CRITERIA_MET, Agent, CohortAlleleFrequencyStudyResult, ExperimentalVariantFunctionalImpactStudyResult, + TherapyGroup, + TumorVariantFrequencyStudyResult, ) from ga4gh.va_spec.base.core import ( Direction, EvidenceLine, Method, + Proposition, Statement, StudyGroup, - StudyResult, VariantClinicalSignificanceProposition, VariantDiagnosticProposition, VariantOncogenicityProposition, @@ -55,10 +58,10 @@ def test_definitions(): def caf(): """Create test fixture for CohortAlleleFrequencyStudyResult""" return CohortAlleleFrequencyStudyResult( - focusAllele="allele.json#/1", - focusAlleleCount=0, - focusAlleleFrequency=0, - locusAlleleCount=34086, + focus="allele.json#/1", + focusCount=0, + alleleFrequency=0, + locusCount=34086, cohort=StudyGroup(id="ALL", name="Overall"), ) @@ -77,7 +80,7 @@ def pathogenicity_evidence_line_params(): "pmid": "25741868", "name": "ACMG Guidelines, 2015", }, - "methodType": "Functional Data Assessment", + "methodType": "functional_data_assessment", }, "directionOfEvidenceProvided": "supports", "evidenceOutcome": { @@ -108,7 +111,7 @@ def oncogenicity_evidence_line_params(): "pmid": "35101336", "name": "ClinGen/CGC/VICC Guidelines for Oncogenicity, 2022", }, - "methodType": "functional_assay", + "methodType": "functional_data_assessment", }, "directionOfEvidenceProvided": "supports", "scoreOfEvidenceProvided": 1, @@ -132,19 +135,19 @@ def oncogenicity_evidence_line_params(): [ ( VariantClinicalSignificanceProposition, - "objectCondition", + "object", "hasClinicalSignificanceFor", ), ( VariantDiagnosticProposition, - "objectCondition", + "object", "isDiagnosticInclusionCriterionFor", ), - (VariantOncogenicityProposition, "objectTumorType", "isOncogenicFor"), - (VariantPathogenicityProposition, "objectCondition", "isCausalFor"), + (VariantOncogenicityProposition, "object", "isOncogenicFor"), + (VariantPathogenicityProposition, "object", "isCausalFor"), ( VariantPrognosticProposition, - "objectCondition", + "object", "associatedWithBetterOutcomeFor", ), ( @@ -160,12 +163,12 @@ def test_proposition_condition_helpers( """Test condition access without knowing the proposition field name.""" initial_condition = iriReference(root="conditions.json#/1") proposition_data = { - "subjectVariant": "alleles.json#/1", + "subject": "alleles.json#/1", "predicate": predicate, condition_field_name: initial_condition, } if proposition_class is VariantTherapeuticResponseProposition: - proposition_data["objectTherapeutic"] = "therapeutics.json#/1" + proposition_data["object"] = "therapeutics.json#/1" proposition = proposition_class(**proposition_data) @@ -179,7 +182,7 @@ def test_condition_set(): """Ensure ConditionSet model works as expected""" condition_set_dict = { "membershipOperator": "AND", - "conditions": [ + "concepts": [ { "conceptType": "Disease", "id": "civic.did:3387", @@ -195,7 +198,7 @@ def test_condition_set(): "name": "Diffuse Astrocytoma, MYB- Or MYBL1-altered", }, { - "conditions": [ + "concepts": [ { "conceptType": "Phenotype", "id": "civic.phenotype:8121", @@ -247,7 +250,7 @@ def test_condition_set(): assert ConditionSet(**condition_set_dict) invalid_params = deepcopy(condition_set_dict) - invalid_params["conditions"].pop() + invalid_params["concepts"].pop() with pytest.raises( ValidationError, match="List should have at least 2 items after validation" @@ -273,54 +276,32 @@ def test_agent(): def test_caf_study_result(caf): """Ensure CohortAlleleFrequencyStudyResult model works as expected""" - assert caf.focusAllele.root == "allele.json#/1" - assert caf.focusAlleleCount == 0 - assert caf.focusAlleleFrequency == 0 - assert caf.locusAlleleCount == 34086 + assert caf.focus.root == "allele.json#/1" + assert caf.focusCount == 0 + assert caf.alleleFrequency == 0 + assert caf.locusCount == 34086 assert caf.cohort.id == "ALL" assert caf.cohort.name == "Overall" assert caf.cohort.type == "StudyGroup" - assert "focus" not in caf.model_dump() - assert "focus" not in json.loads(caf.model_dump_json()) - - with pytest.raises( - AttributeError, - match="'CohortAlleleFrequencyStudyResult' object has no attribute 'focus'", - ): - caf.focus # noqa: B018 - - with pytest.raises( - ValueError, - match='"CohortAlleleFrequencyStudyResult" object has no field "focus"', - ): - caf.focus = "focus" + assert caf.model_dump()["focus"] == "allele.json#/1" + assert json.loads(caf.model_dump_json())["focus"] == "allele.json#/1" def test_experimental_func_impact_study_result(): """Ensure ExperimentalVariantFunctionalImpactStudyResult model works as expected""" experimental_func_impact_study_result = ( - ExperimentalVariantFunctionalImpactStudyResult(focusVariant="allele.json#/1") + ExperimentalVariantFunctionalImpactStudyResult(focus="allele.json#/1") ) - assert experimental_func_impact_study_result.focusVariant.root == "allele.json#/1" + assert experimental_func_impact_study_result.focus.root == "allele.json#/1" - assert "focus" not in experimental_func_impact_study_result.model_dump() - assert "focus" not in json.loads( + assert ( + experimental_func_impact_study_result.model_dump()["focus"] == "allele.json#/1" + ) + assert "focus" in json.loads( experimental_func_impact_study_result.model_dump_json() ) - with pytest.raises( - AttributeError, - match="'ExperimentalVariantFunctionalImpactStudyResult' object has no attribute 'focus'", - ): - experimental_func_impact_study_result.focus # noqa: B018 - - with pytest.raises( - ValueError, - match='"ExperimentalVariantFunctionalImpactStudyResult" object has no field "focus"', - ): - experimental_func_impact_study_result.focus = "focus" - def test_evidence_line(caf): """Ensure EvidenceLine model works as expected""" @@ -333,7 +314,7 @@ def test_evidence_line(caf): "type": "Statement", "proposition": { "type": "VariantTherapeuticResponseProposition", - "subjectVariant": { + "subject": { "id": "civic.mpid:33", "type": "CategoricalVariant", "name": "EGFR L858R", @@ -345,7 +326,7 @@ def test_evidence_line(caf): }, "alleleOriginQualifier": {"name": "somatic"}, "predicate": "predictsSensitivityTo", - "objectTherapeutic": { + "object": { "id": "civic.tid:146", "conceptType": "Therapy", "name": "Afatinib", @@ -388,6 +369,9 @@ def test_evidence_line(caf): el = EvidenceLine(**el_dict) assert isinstance(el.hasEvidenceItems[0], iriReference) assert isinstance(el.hasEvidenceItems[1], Statement) + assert isinstance( + el.hasEvidenceItems[1].proposition, VariantTherapeuticResponseProposition + ) el_dict = { "type": "EvidenceLine", @@ -395,8 +379,7 @@ def test_evidence_line(caf): "directionOfEvidenceProvided": "supports", } el = EvidenceLine(**el_dict) - assert isinstance(el.hasEvidenceItems[0], StudyResult) - assert isinstance(el.hasEvidenceItems[0].root, CohortAlleleFrequencyStudyResult) + assert isinstance(el.hasEvidenceItems[0], CohortAlleleFrequencyStudyResult) el_dict = { "type": "EvidenceLine", @@ -447,8 +430,8 @@ def test_variant_pathogenicity_stmt(pathogenicity_evidence_line_params): "proposition": { "type": "VariantPathogenicityProposition", "predicate": "isCausalFor", - "objectCondition": "conditions.json#/1", - "subjectVariant": "alleles.json#/1", + "object": "conditions.json#/1", + "subject": "alleles.json#/1", }, "classification": { "primaryCoding": {"code": "pathogenic", "system": "ACMG Guidelines, 2015"} @@ -499,9 +482,63 @@ def test_variant_pathogenicity_stmt(pathogenicity_evidence_line_params): VariantPathogenicityStatement(**invalid_params) +def test_statement_proposition_accepts_iri_reference(): + """Statements may reference a proposition instead of embedding one.""" + statement = Statement(proposition="propositions.json#/1") + + assert statement.proposition == iriReference(root="propositions.json#/1") + + +def test_base_statement_and_evidence_line_accept_generic_propositions(): + """Base models accept generic schema propositions.""" + proposition = Proposition( + type="Proposition", subject={}, predicate="relatedTo", object={} + ) + + statement = Statement(proposition=proposition) + evidence_line = EvidenceLine( + directionOfEvidenceProvided="neutral", targetProposition=proposition + ) + + assert statement.proposition == proposition + assert evidence_line.targetProposition == proposition + + +def test_concept_sets_accept_iri_references(): + """Concept sets accept schema-permitted IRIs.""" + condition_set = ConditionSet( + concepts=["conditions.json#/1", "conditions.json#/2"], + membershipOperator="OR", + ) + therapy_group = TherapyGroup( + concepts=["therapies.json#/1", "therapies.json#/2"], + membershipOperator="AND", + ) + + assert all(isinstance(concept, iriReference) for concept in condition_set.concepts) + assert all(isinstance(concept, iriReference) for concept in therapy_group.concepts) + + +def test_study_results_accept_iri_source_data_sets(caf): + """Study result models accept IRI source datasets.""" + assert CohortAlleleFrequencyStudyResult( + **(caf.model_dump() | {"sourceDataSet": "datasets.json#/1"}) + ).sourceDataSet == iriReference(root="datasets.json#/1") + assert TumorVariantFrequencyStudyResult( + focus="alleles.json#/1", + affectedSampleCount=1, + totalSampleCount=2, + affectedFrequency=0.5, + sourceDataSet="datasets.json#/1", + ).sourceDataSet == iriReference(root="datasets.json#/1") + assert ExperimentalVariantFunctionalImpactStudyResult( + focus="alleles.json#/1", sourceDataSet="datasets.json#/1" + ).sourceDataSet == iriReference(root="datasets.json#/1") + + def test_variant_pathogenicity_el(pathogenicity_evidence_line_params): """Ensure VariantPathogenicityEvidenceLine model works as expected""" - params = pathogenicity_evidence_line_params + params = deepcopy(pathogenicity_evidence_line_params) vp = VariantPathogenicityEvidenceLine(**params) assert isinstance(vp.specifiedBy, Method) @@ -584,6 +621,51 @@ def test_variant_pathogenicity_el(pathogenicity_evidence_line_params): VariantPathogenicityEvidenceLine(**invalid_params) +@pytest.mark.parametrize("outcome", [NO_CRITERIA_MET, "PS3_not_met"]) +def test_pathogenicity_noncontributing_outcomes_require_neutral_without_strength( + pathogenicity_evidence_line_params, outcome +): + """Noncontributing ACMG outcomes must have neutral direction and no strength.""" + params = deepcopy(pathogenicity_evidence_line_params) + strength = deepcopy(params["strengthOfEvidenceProvided"]) + params["evidenceOutcome"]["primaryCoding"]["code"] = outcome + params["directionOfEvidenceProvided"] = "neutral" + params["strengthOfEvidenceProvided"] = None + assert VariantPathogenicityEvidenceLine(**params) + + invalid_direction = deepcopy(params) + invalid_direction["directionOfEvidenceProvided"] = "supports" + invalid_direction["strengthOfEvidenceProvided"] = strength + with pytest.raises( + ValueError, match="`directionOfEvidenceProvided` must be 'neutral'" + ): + VariantPathogenicityEvidenceLine(**invalid_direction) + + invalid_strength = deepcopy(params) + invalid_strength["strengthOfEvidenceProvided"] = strength + with pytest.raises(ValueError, match="`strengthOfEvidenceProvided` must be null"): + VariantPathogenicityEvidenceLine(**invalid_strength) + + +def test_pathogenicity_profile_accepts_schema_permitted_references(): + """Pathogenicity models accept opaque IRI references.""" + evidence_line = VariantPathogenicityEvidenceLine( + specifiedBy="methods.json#/1", + directionOfEvidenceProvided="supports", + evidenceOutcome="outcomes.json#/1", + strengthOfEvidenceProvided="strengths.json#/1", + ) + statement = VariantPathogenicityStatement( + proposition="propositions.json#/1", + strength="strengths.json#/1", + classification="classifications.json#/1", + specifiedBy="methods.json#/1", + ) + + assert isinstance(evidence_line.evidenceOutcome, iriReference) + assert isinstance(statement.classification, iriReference) + + def test_variant_onco_stmt(oncogenicity_evidence_line_params): """Ensure VariantOncogenicityStatement model works as expected""" params = { @@ -591,8 +673,8 @@ def test_variant_onco_stmt(oncogenicity_evidence_line_params): "proposition": { "type": "VariantOncogenicityProposition", "predicate": "isOncogenicFor", - "objectTumorType": "conditions.json#/1", - "subjectVariant": "alleles.json#/1", + "object": "conditions.json#/1", + "subject": "alleles.json#/1", }, "classification": { "primaryCoding": { @@ -682,6 +764,51 @@ def test_variant_onco_el(oncogenicity_evidence_line_params): VariantOncogenicityEvidenceLine(**invalid_params) +@pytest.mark.parametrize("outcome", [NO_CRITERIA_MET, "OS2_not_met"]) +def test_oncogenicity_noncontributing_outcomes_require_neutral_without_strength( + oncogenicity_evidence_line_params, outcome +): + """Noncontributing CCV outcomes must have neutral direction and no strength.""" + params = deepcopy(oncogenicity_evidence_line_params) + strength = deepcopy(params["strengthOfEvidenceProvided"]) + params["evidenceOutcome"]["primaryCoding"]["code"] = outcome + params["directionOfEvidenceProvided"] = "neutral" + params["strengthOfEvidenceProvided"] = None + assert VariantOncogenicityEvidenceLine(**params) + + invalid_direction = deepcopy(params) + invalid_direction["directionOfEvidenceProvided"] = "supports" + invalid_direction["strengthOfEvidenceProvided"] = strength + with pytest.raises( + ValueError, match="`directionOfEvidenceProvided` must be 'neutral'" + ): + VariantOncogenicityEvidenceLine(**invalid_direction) + + invalid_strength = deepcopy(params) + invalid_strength["strengthOfEvidenceProvided"] = strength + with pytest.raises(ValueError, match="`strengthOfEvidenceProvided` must be null"): + VariantOncogenicityEvidenceLine(**invalid_strength) + + +def test_oncogenicity_profile_accepts_schema_permitted_references(): + """Oncogenicity models accept inherited IRI branches.""" + evidence_line = VariantOncogenicityEvidenceLine( + specifiedBy="methods.json#/1", + directionOfEvidenceProvided="supports", + evidenceOutcome="outcomes.json#/1", + strengthOfEvidenceProvided="strengths.json#/1", + ) + statement = VariantOncogenicityStatement( + proposition="propositions.json#/1", + strength="strengths.json#/1", + classification="classifications.json#/1", + specifiedBy="methods.json#/1", + ) + + assert isinstance(evidence_line.evidenceOutcome, iriReference) + assert isinstance(statement.classification, iriReference) + + def test_variant_onco_el_no_evidence_outcome(): """Test that VariantOncogenicityEvidenceLine validates without evidence outcome @@ -695,7 +822,7 @@ def test_variant_onco_el_no_evidence_outcome(): "pmid": "35101336", "name": "ClinGen/CGC/VICC Guidelines for Oncogenicity, 2022", }, - "methodType": "functional_assay", + "methodType": "functional_data_assessment", }, directionOfEvidenceProvided=Direction.NEUTRAL, scoreOfEvidenceProvided=0, @@ -714,21 +841,34 @@ def test_variant_onco_el_no_evidence_outcome(): VariantOncogenicityEvidenceLine.model_validate(invalid) +def test_aac_profile_accepts_schema_permitted_references(): + """AAC models accept opaque classification and strength IRIs.""" + statement = VariantClinicalSignificanceStatement( + proposition="propositions.json#/1", + strength="strengths.json#/1", + classification="classifications.json#/1", + specifiedBy="methods.json#/1", + ) + + assert isinstance(statement.strength, iriReference) + assert isinstance(statement.classification, iriReference) + + def test_aac_statement(): """Test that AMP/ASCO/CAP statement model validators work correctly""" prop = { "type": "VariantDiagnosticProposition", "predicate": "isDiagnosticExclusionCriterionFor", - "objectCondition": "conditions.json#/1", - "subjectVariant": "alleles.json#/1", + "object": "conditions.json#/1", + "subject": "alleles.json#/1", } params = { "direction": "supports", "proposition": { "type": "VariantClinicalSignificanceProposition", "predicate": "hasClinicalSignificanceFor", - "objectCondition": "conditions.json#/1", - "subjectVariant": "alleles.json#/1", + "object": "conditions.json#/1", + "subject": "alleles.json#/1", }, "strength": { "primaryCoding": { diff --git a/tests/validation/test_va_spec_schema.py b/tests/validation/test_va_spec_schema.py index 48b825a..3a2975f 100644 --- a/tests/validation/test_va_spec_schema.py +++ b/tests/validation/test_va_spec_schema.py @@ -8,7 +8,6 @@ from tests.conftest import ( SUBMODULES_DIR, VaSpecSchema, - get_va_spec_schema, ) from ga4gh.va_spec import aac_2017, acmg_2015, base, ccv_2022 @@ -39,7 +38,7 @@ def _update_va_spec_schema_mapping( spec_class = cls_def["title"] va_spec_schema_mapping.va_spec_schema[spec_class] = cls_def - if "properties" in cls_def: + if "properties" in cls_def and not cls_def.get("abstract", False): va_spec_schema_mapping.concrete_classes.add(spec_class) elif cls_def.get("type") in {"array", "integer", "string"}: va_spec_schema_mapping.primitives.add(spec_class) @@ -50,16 +49,14 @@ def _update_va_spec_schema_mapping( VA_SPEC_SCHEMA_MAPPING = {schema: VaSpecSchemaMapping() for schema in VaSpecSchema} -# Get core + profiles classes -for child in VA_SCHEMA_DIR.iterdir(): - child_str = str(child) - mapping_key = get_va_spec_schema(child_str) - if not mapping_key: - continue +# Core schemas are in ``va-spec/json``. Profile schemas use subdirectories. +for f in VA_SCHEMA_DIR.glob("json/*"): + if f.is_file(): + _update_va_spec_schema_mapping(f, VA_SPEC_SCHEMA_MAPPING[VaSpecSchema.BASE]) - mapping = VA_SPEC_SCHEMA_MAPPING[mapping_key] - for f in (child / "json").glob("*"): - _update_va_spec_schema_mapping(f, mapping) +for profile in (VaSpecSchema.AAC_2017, VaSpecSchema.ACMG_2015, VaSpecSchema.CCV_2022): + for f in (VA_SCHEMA_DIR / "json" / profile.value).glob("*"): + _update_va_spec_schema_mapping(f, VA_SPEC_SCHEMA_MAPPING[profile]) @pytest.mark.parametrize(