Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion src/ga4gh/vrs/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -5,7 +5,7 @@
from ga4gh.vrs import models
from ga4gh.vrs.enderef import vrs_deref, vrs_enref
from ga4gh.vrs.models import VrsType
from ga4gh.vrs.normalize import normalize
from ga4gh.vrs.normalize import RleSubunitMode, normalize

try:
__version__ = version(__name__)
Expand All @@ -19,6 +19,7 @@

__all__ = [
"VRS_VERSION",
"RleSubunitMode",
"VrsType",
"models",
"normalize",
Expand Down
36 changes: 34 additions & 2 deletions src/ga4gh/vrs/extras/translator.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,7 +17,11 @@
from ga4gh.vrs import models, normalize
from ga4gh.vrs.dataproxy import SequenceProxy, _DataProxy
from ga4gh.vrs.extras.decorators import lazy_property
from ga4gh.vrs.normalize import denormalize_reference_length_expression
from ga4gh.vrs.normalize import (
DEFAULT_RLE_SUBUNIT_MODE,
RleSubunitMode,
denormalize_reference_length_expression,
)
from ga4gh.vrs.utils.hgvs_tools import HgvsTools

_logger = logging.getLogger(__name__)
Expand Down Expand Up @@ -75,11 +79,13 @@ def __init__(
default_assembly_name: str = "GRCh38",
identify: bool = True,
rle_seq_limit: int | None = 50,
rle_subunit_mode: RleSubunitMode = DEFAULT_RLE_SUBUNIT_MODE,
) -> None:
self.default_assembly_name = default_assembly_name
self.data_proxy = data_proxy
self.identify = identify
self.rle_seq_limit = rle_seq_limit
self.rle_subunit_mode = RleSubunitMode(rle_subunit_mode)
self.from_translators: dict[str, VariationFromStrProtocol] = {}
self.to_translators: dict[str, VariationToStrProtocol] = {}

Expand Down Expand Up @@ -111,6 +117,10 @@ def translate_from(
To exclude `sequence` from the response, set to 0.
For no limit, set to `None`.
Defaults value set in instance variable, `rle_seq_limit`.
rle_subunit_mode (RleSubunitMode): Which valid factor of the seed
length to select as the `repeatSubunitLength` for a
reference-derived ambiguous insertion.
Defaults value set in instance variable, `rle_subunit_mode`.
do_normalize (bool): `True` if fully justified normalization should be
performed. `False` otherwise. Defaults to `True`
"""
Expand Down Expand Up @@ -181,9 +191,15 @@ def __init__(
data_proxy: _DataProxy,
default_assembly_name: str = "GRCh38",
identify: bool = True,
rle_subunit_mode: RleSubunitMode = DEFAULT_RLE_SUBUNIT_MODE,
) -> None:
"""Initialize AlleleTranslator class"""
super().__init__(data_proxy, default_assembly_name, identify)
super().__init__(
data_proxy,
default_assembly_name,
identify,
rle_subunit_mode=rle_subunit_mode,
)

self.from_translators = {
"beacon": self._from_beacon,
Expand Down Expand Up @@ -232,6 +248,10 @@ def _from_beacon(self, beacon_expr: str, **kwargs) -> models.Allele | None:
To exclude `sequence` from the response, set to 0.
For no limit, set to `None`.
Defaults value set in instance variable, `rle_seq_limit`.
rle_subunit_mode (RleSubunitMode): Which valid factor of the seed length
to select as the `repeatSubunitLength` for a reference-derived
ambiguous insertion.
Defaults value set in instance variable, `rle_subunit_mode`.
do_normalize (bool): `True` if fully justified normalization should be
performed. `False` otherwise. Defaults to `True`

Expand Down Expand Up @@ -297,6 +317,10 @@ def _from_gnomad(self, gnomad_expr: str, **kwargs) -> models.Allele | None:
To exclude `sequence` from the response, set to 0.
For no limit, set to `None`.
Defaults value set in instance variable, `rle_seq_limit`.
rle_subunit_mode (RleSubunitMode): Which valid factor of the seed length
to select as the `repeatSubunitLength` for a reference-derived
ambiguous insertion.
Defaults value set in instance variable, `rle_subunit_mode`.
do_normalize (bool): `True` if fully justified normalization should be
performed. `False` otherwise. Defaults to `True`

Expand Down Expand Up @@ -371,6 +395,10 @@ def _from_spdi(self, spdi_expr: str, **kwargs) -> models.Allele | None:
To exclude `sequence` from the response, set to 0.
For no limit, set to `None`.
Defaults value set in instance variable, `rle_seq_limit`.
rle_subunit_mode (RleSubunitMode): Which valid factor of the seed length
to select as the `repeatSubunitLength` for a reference-derived
ambiguous insertion.
Defaults value set in instance variable, `rle_subunit_mode`.
do_normalize (bool): `True` if fully justified normalization should be
performed. `False` otherwise. Defaults to `True`

Expand Down Expand Up @@ -509,6 +537,9 @@ def _post_process_imported_allele(
normalization, this sets the limit for the length of the `sequence`.
To exclude `sequence` from the response, set to 0.
For no limit, set to `None`.
rle_subunit_mode (RleSubunitMode): Which valid factor of the seed length to
select as the `repeatSubunitLength` for a reference-derived ambiguous
insertion. Defaults value set in instance variable, `rle_subunit_mode`.
do_normalize (bool): `True` if fully justified normalization should be
performed. `False` otherwise. Defaults to `True`
"""
Expand All @@ -517,6 +548,7 @@ def _post_process_imported_allele(
allele,
self.data_proxy,
rle_seq_limit=kwargs.get("rle_seq_limit", self.rle_seq_limit),
rle_subunit_mode=kwargs.get("rle_subunit_mode", self.rle_subunit_mode),
)

if self.identify:
Expand Down
67 changes: 57 additions & 10 deletions src/ga4gh/vrs/normalize.py
Original file line number Diff line number Diff line change
Expand Up @@ -6,7 +6,8 @@

import itertools
import logging
from enum import IntEnum
from collections.abc import Iterator
from enum import Enum, IntEnum
from typing import NamedTuple

from bioutils.normalize import NormalizationMode
Expand All @@ -20,6 +21,26 @@
_logger = logging.getLogger(__name__)


class RleSubunitMode(str, Enum):

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

In the 2.1 branch, we'll only allow smallest, right?

"""Define which repeat subunit length to select for a reference-derived ambiguous
insertion normalized to a ``ReferenceLengthExpression``. More than one factor of
the seed length may be circularly expandable to recreate the alternate sequence;
this selects which of them becomes the ``repeatSubunitLength``. The options are:
LARGEST - Select the greatest valid factor (VRS 2.0.x normalization)
SMALLEST - Select the smallest valid factor (VRS 2.1 normalization)

``repeatSubunitLength`` is part of the ``ReferenceLengthExpression`` digest, so the
two modes may produce different identifiers for the same input Allele.
"""

LARGEST = "largest"
SMALLEST = "smallest"


# VRS 2.0.x normalization selects the greatest valid factor
DEFAULT_RLE_SUBUNIT_MODE = RleSubunitMode.LARGEST


class PosType(IntEnum):
"""Define the kind of position on a location"""

Expand Down Expand Up @@ -85,7 +106,10 @@ def _get_new_allele_location_pos(


def _normalize_allele(
input_allele: models.Allele, data_proxy: _DataProxy, rle_seq_limit: int = 50
input_allele: models.Allele,
data_proxy: _DataProxy,
rle_seq_limit: int = 50,
rle_subunit_mode: RleSubunitMode = DEFAULT_RLE_SUBUNIT_MODE,
):
"""Normalize Allele using "fully-justified" normalization adapted from NCBI's
VOCA. Fully-justified normalization expands such ambiguous representation over the
Expand All @@ -111,7 +135,12 @@ def _normalize_allele(
of the `sequence`.
To exclude `sequence` from the response, set to 0.
For no limit, set to `None`.
:param rle_subunit_mode: Which valid factor of the seed length to select as the
`repeatSubunitLength` for a reference-derived ambiguous insertion. Defaults to
`DEFAULT_RLE_SUBUNIT_MODE`. See `RleSubunitMode`.
"""
rle_subunit_mode = RleSubunitMode(rle_subunit_mode)

# Algorithm applies to LiteralSequenceExpression alleles only; other states are returned unchanged
if not isinstance(input_allele.state, models.LiteralSequenceExpression):
_logger.warning(
Expand Down Expand Up @@ -204,7 +233,7 @@ def _normalize_allele(

# 5.c
if len_extended_alt > len_extended_ref:
factors = _factor_gen(seed_length)
factors = _factor_gen(seed_length, rle_subunit_mode)
for cycle_length in factors:
if cycle_length > len_extended_ref:
continue
Expand Down Expand Up @@ -308,17 +337,32 @@ def denormalize_reference_length_expression(
return alt


def _factor_gen(n: int):
"""Yield all factors of an integer `n`, in descending order"""
lower_factors = []
def _factor_gen(
n: int, mode: RleSubunitMode = DEFAULT_RLE_SUBUNIT_MODE
) -> Iterator[int]:
"""Yield all factors of an integer `n`

:param n: The integer to factor
:param mode: `RleSubunitMode.LARGEST` to yield factors in descending order,
`RleSubunitMode.SMALLEST` to yield them in ascending order. The first valid
factor found by the caller is the one selected, so the order determines whether
the largest or smallest valid factor wins.
"""
paired_factors = []
i = 1
while i * i <= n:
if n % i == 0:
yield n // i
if n // i != i:
lower_factors.append(i)
# i <= sqrt(n) <= n // i. Yield the factor that is already in the requested
# order, and save its pair to yield in reverse once the loop finishes.
if mode == RleSubunitMode.SMALLEST:
factor, pair = i, n // i
else:
factor, pair = n // i, i
yield factor
if pair != factor:
paired_factors.append(pair)
i += 1
yield from reversed(lower_factors)
yield from reversed(paired_factors)


def _define_rle_allele(
Expand Down Expand Up @@ -371,6 +415,9 @@ def normalize(vo, data_proxy: _DataProxy | None = None, **kwargs):
:param data_proxy: GA4GH sequence dataproxy instance, if needed
:keyword rle_seq_limit: If RLE is set as the new state, set the limit for the length
of the `sequence`. To exclude `state.sequence`, set to 0.
:keyword rle_subunit_mode: Which valid factor of the seed length to select as the
`repeatSubunitLength` for a reference-derived ambiguous insertion. Defaults to
`DEFAULT_RLE_SUBUNIT_MODE`. See `RleSubunitMode`.
:return: normalized object, or unmodified input object if the normalization algorithm
does not provide normalization steps for the given type.
:raise TypeError: if given object isn't a pydantic.BaseModel
Expand Down
Loading
Loading