Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
47 changes: 24 additions & 23 deletions alphabase/spectral_library/translate.py
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,7 @@
"""

import multiprocessing as mp
import warnings

import pandas as pd
import tqdm
Expand Down Expand Up @@ -44,8 +45,8 @@
"get_precursor_mz",
# this module's own
"WritingProcess",
"speclib_to_single_df",
"speclib_to_swath_df",
"speclib_to_single_df",
"translate_to_tsv",
]

Expand Down Expand Up @@ -73,7 +74,7 @@ def _precursors_to_swath_df( # noqa: PLR0913
) -> pd.DataFrame:
"""Convert precursor and fragment frames to a SWATH transition list.

The dataframe-in form of :func:`speclib_to_single_df`, so that one batch of
The dataframe-in form of :func:`speclib_to_swath_df`, so that one batch of

Copy link
Copy Markdown
Collaborator

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

(I know this is not part of this PR)

The naming SWATH here is a bit confusing. I would maybe call it just "spectral library" and do a full rename in a separate PR.

Copy link
Copy Markdown
Collaborator Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

I had a hard time naming this too. I don't think plain "spectral library" works though because the DIA-NN parquet export is also a spectral library, so the name wouldn't tell the two apart.

precursors can be converted without standing up a `SpecLibBase` around it.
"""
df = pd.DataFrame()
Expand Down Expand Up @@ -133,7 +134,7 @@ def _precursors_to_swath_df( # noqa: PLR0913
return df


def speclib_to_single_df(
def speclib_to_swath_df(
speclib: SpecLibBase,
*,
translate_mod_dict: dict = None,
Expand All @@ -145,10 +146,9 @@ def speclib_to_single_df(
modloss: str = "H3PO4",
verbose=True,
) -> pd.DataFrame:
"""Convert an alphabase library into a SWATH/Spectronaut transition list.
"""Convert an alphabase library to a SWATH/Spectronaut transition-list dataframe.

The in-memory counterpart of :func:`translate_to_tsv`, which writes the same
table to a file.
:func:`translate_to_tsv` writes the same table to a file.

Parameters
----------
Expand All @@ -167,7 +167,7 @@ def speclib_to_single_df(
Returns
-------
pd.DataFrame
a single dataframe in the SWATH-like format
One row per precursor/fragment pair, in SWATH column names.

"""
return _precursors_to_swath_df(
Expand All @@ -185,22 +185,15 @@ def speclib_to_single_df(
)


def speclib_to_swath_df(
speclib: SpecLibBase,
*,
keep_k_highest_fragments: int = 12,
min_frag_mz=200,
max_frag_mz=2000,
min_frag_intensity=0.01,
) -> pd.DataFrame:
speclib_to_single_df(
speclib,
translate_mod_dict=None,
keep_k_highest_fragments=keep_k_highest_fragments,
min_frag_mz=min_frag_mz,
max_frag_mz=max_frag_mz,
min_frag_intensity=min_frag_intensity,
def speclib_to_single_df(speclib: SpecLibBase, **kwargs) -> pd.DataFrame:
"""Deprecated alias of :func:`speclib_to_swath_df`."""
warnings.warn(
"speclib_to_single_df is deprecated; use speclib_to_swath_df, which names the "
"format it produces.",
FutureWarning,
stacklevel=2,
)
return speclib_to_swath_df(speclib, **kwargs)


class WritingProcess(mp.Process):
Expand Down Expand Up @@ -248,7 +241,7 @@ def translate_to_tsv(
Fragment m/z range; fragments outside it are dropped. Pass 0 for no lower bound
and `np.inf` for no upper bound.

See :func:`speclib_to_single_df`, whose parameters this shares, for the rest.
See :func:`speclib_to_swath_df`, whose parameters this shares, for the rest.
"""
if multiprocessing:
queue_size = 1000000 // batch_size
Expand Down Expand Up @@ -295,3 +288,11 @@ def translate_to_tsv(
"Translation finished, it will take several minutes to export the rest precursors to the tsv file..."
)
writing_process.join()
if writing_process.exitcode:
raise RuntimeError(
f"the process writing {tsv} exited with code "
f"{writing_process.exitcode}, so the file is incomplete; its traceback "
'is above. A script with no `if __name__ == "__main__":` guard is the '
"usual cause on macOS and Windows. Pass multiprocessing=False to write "
"from this process instead."
)
71 changes: 33 additions & 38 deletions alphabase/spectral_library/translate_core.py
Original file line number Diff line number Diff line change
Expand Up @@ -19,13 +19,16 @@
from alphabase.numba_wrapper import numba_njit
from alphabase.peptide.precursor import update_precursor_mz
from alphabase.psm_reader.keys import ConstantsClass, PsmDfCols
from alphabase.utils import explode_multiple_columns

# Candidate precursor columns in order of precedence, for libraries that carry more than
# one. The `*_pred` names are peptdeep's prediction outputs, which take priority over a
# measured value; `irt_pred` outranks `rt_pred` because an indexed RT is what a
# third-party library wants.
RT_COLUMNS = ["irt_pred", "rt_pred", PsmDfCols.RT, "irt", PsmDfCols.RT_NORM]
# Candidate precursor columns, in order of precedence.
RT_COLUMNS = [
"irt_pred",
"rt_pred",
"rt_norm_pred",
PsmDfCols.RT,
"irt",
PsmDfCols.RT_NORM,
]
MOBILITY_COLUMNS = ["mobility_pred", PsmDfCols.MOBILITY]
CCS_COLUMNS = ["ccs_pred", PsmDfCols.CCS]

Expand Down Expand Up @@ -56,17 +59,6 @@ class FragmentTableCols(metaclass=ConstantsClass):
LOSS_TYPE = "loss_type"


# the per-fragment columns, in the order the exports emit them
FRAGMENT_VALUE_COLUMNS = [
FragmentTableCols.FRAG_TYPE,
FragmentTableCols.MZ,
FragmentTableCols.INTENSITY,
FragmentTableCols.CHARGE,
FragmentTableCols.SERIES_NUMBER,
FragmentTableCols.LOSS_TYPE,
]


def get_precursor_mz(precursor_df: pd.DataFrame) -> pd.Series:
"""Return the precursors' m/z, leaving `precursor_df` alone.

Expand Down Expand Up @@ -186,7 +178,8 @@ def _get_frag_info_from_column_name(column: str) -> tuple:
"""Split a fragment column name into `(frag_type, loss_type, charge)`.

For example `y_modloss_z2` -> `('y', 'modloss', '2')` and `b_z1` -> `('b',
'noloss', '1')`. The charge is left as a string, as it is only written out.
'noloss', '1')`. The charge stays a string because numba cannot parse one;
:func:`fragment_table` casts the collected column.
"""
idx = column.rfind("_")
frag_type = column[:idx]
Expand Down Expand Up @@ -264,7 +257,9 @@ def get_fragment_table( # noqa: PLR0913
Returns
-------
pd.DataFrame
One row per kept fragment, in :class:`FragmentTableCols` columns.
One row per kept fragment, in :class:`FragmentTableCols` columns. `mz` and
`intensity` keep the dtype of the frame they came from; the charge and series
number are integers.

"""
frag_columns = fragment_mz_df.columns.to_numpy().astype("U")
Expand All @@ -280,12 +275,12 @@ def get_fragment_table( # noqa: PLR0913
)
max_frag_mz = np.inf

frag_types = []
frag_losses = []
frag_charges = []
frag_masses = []
frag_intensities = []
frag_numbers = []
frag_types: list = []
frag_losses: list = []
frag_charges: list = []
frag_numbers: list = []
frag_masses: list = []
frag_intensities: list = []
frag_idx_ranges = zip(frag_start_idx, frag_stop_idx)
if verbose:
frag_idx_ranges = tqdm.tqdm(frag_idx_ranges)
Expand Down Expand Up @@ -319,27 +314,27 @@ def get_fragment_table( # noqa: PLR0913

infos = [_get_frag_info_from_column_name(column) for column in columns]
types, losses, charges = zip(*infos) if infos else ((), (), ())
frag_types.append(types)
frag_losses.append(losses)
frag_charges.append(charges)
frag_types.extend(types)
Comment thread
mo-sameh marked this conversation as resolved.
frag_losses.extend(losses)
frag_charges.extend(charges)
frag_numbers.extend(_get_frag_num(columns, rows, frag_len))
frag_masses.append(masses[idx_in_df])
frag_intensities.append(intens[idx_in_df])
frag_numbers.append(_get_frag_num(columns, rows, frag_len))

fragments_df = pd.DataFrame(
return pd.DataFrame(
{
FragmentTableCols.PRECURSOR_ROW: np.arange(len(frag_start_idx)),
FragmentTableCols.PRECURSOR_ROW: np.repeat(
np.arange(len(frag_start_idx)),
[len(kept) for kept in frag_masses],
),
FragmentTableCols.FRAG_TYPE: frag_types,
FragmentTableCols.MZ: frag_masses,
FragmentTableCols.INTENSITY: frag_intensities,
FragmentTableCols.CHARGE: frag_charges,
FragmentTableCols.SERIES_NUMBER: frag_numbers,
FragmentTableCols.MZ: np.concatenate(frag_masses),
FragmentTableCols.INTENSITY: np.concatenate(frag_intensities),
FragmentTableCols.CHARGE: np.array(frag_charges, dtype=np.int8),
FragmentTableCols.SERIES_NUMBER: np.array(frag_numbers, dtype=np.int64),
FragmentTableCols.LOSS_TYPE: frag_losses,
}
)
fragments_df = explode_multiple_columns(fragments_df, FRAGMENT_VALUE_COLUMNS)
# a precursor that kept nothing explodes to one all-NaN row; drop those
return fragments_df.dropna(subset=[FragmentTableCols.MZ])


def join_fragments(
Expand Down
8 changes: 5 additions & 3 deletions alphabase/spectral_library/translate_diann.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,8 +62,10 @@ class DiannParquetCols(metaclass=ConstantsClass):
SOURCE_ID = "Source.Id"


# the DIA-NN names for the canonical fragment columns, in output order
# the DIA-NN names for the canonical fragment columns, in output order. `precursor_row`
# keeps its name for the base-peak flag; the schema selection at the end drops it.
DIANN_FRAGMENT_COLUMNS = {
FragmentTableCols.PRECURSOR_ROW: FragmentTableCols.PRECURSOR_ROW,
FragmentTableCols.FRAG_TYPE: DiannParquetCols.FRAGMENT_TYPE,
FragmentTableCols.MZ: DiannParquetCols.PRODUCT_MZ,
FragmentTableCols.INTENSITY: DiannParquetCols.RELATIVE_INTENSITY,
Expand Down Expand Up @@ -227,10 +229,10 @@ def _precursors_to_diann_df( # noqa: PLR0913
] = modloss
df = df.reset_index(drop=True)

# Flags: base bit on all fragments, base-peak bit on each precursor's most intense one
df[DiannParquetCols.FLAGS] = _DIANN_FLAG_BASE
if len(df):
base_peak_idx = df.groupby(DiannParquetCols.PRECURSOR_ID, sort=False)[
# by row: `Precursor.Id` repeats when a library holds the same precursor twice
base_peak_idx = df.groupby(FragmentTableCols.PRECURSOR_ROW, sort=False)[
DiannParquetCols.RELATIVE_INTENSITY
].idxmax()
df.loc[base_peak_idx, DiannParquetCols.FLAGS] |= _DIANN_FLAG_FIRST_FRAGMENT
Expand Down
8 changes: 4 additions & 4 deletions nbs_tests/spectral_library/translate.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,7 @@
"import pandas as pd\n",
"\n",
"from alphabase.spectral_library.base import SpecLibBase\n",
"from alphabase.spectral_library.translate import create_modified_sequence, speclib_to_single_df, translate_to_tsv"
"from alphabase.spectral_library.translate import create_modified_sequence, speclib_to_swath_df, translate_to_tsv"
]
},
{
Expand Down Expand Up @@ -538,10 +538,10 @@
"spec_lib._precursor_df = precursor_df\n",
"spec_lib._fragment_intensity_df = frag_mass_df.copy()\n",
"spec_lib._fragment_mz_df = frag_mass_df.copy()\n",
"df = speclib_to_single_df(spec_lib, min_frag_mz=300, max_frag_mz=1800)\n",
"df = speclib_to_swath_df(spec_lib, min_frag_mz=300, max_frag_mz=1800)\n",
"assert (df.FragmentMz>=300).all()\n",
"assert (df.FragmentMz<=1800).all()\n",
"df = speclib_to_single_df(spec_lib, min_frag_mz=200, min_frag_nAA=3)\n",
"df = speclib_to_swath_df(spec_lib, min_frag_mz=200, min_frag_nAA=3)\n",
"assert (df.FragmentNumber>=3).all()"
]
},
Expand Down Expand Up @@ -846,7 +846,7 @@
"spec_lib._precursor_df = precursor_df\n",
"spec_lib._fragment_intensity_df = frag_mass_df.copy()\n",
"spec_lib._fragment_mz_df = frag_mass_df.copy()\n",
"speclib_sdf = speclib_to_single_df(spec_lib)\n",
"speclib_sdf = speclib_to_swath_df(spec_lib)\n",
"with tempfile.TemporaryFile('w+') as f:\n",
" translate_to_tsv(spec_lib, f, batch_size=2, multiprocessing=False)\n",
" f.seek(0)\n",
Expand Down
Loading
Loading