From 5fa58a2bff3ffce520bbea5c3890d6c4d117bbc6 Mon Sep 17 00:00:00 2001 From: RobbinBouwmeester Date: Tue, 22 Sep 2026 14:15:44 +0200 Subject: [PATCH] perf: predict each distinct peptidoform once Identification output holds one PSM per match, so a peptidoform reappears for every scan, charge state and run that matched it. Each copy was parsed, encoded and pushed through the trunk again for a value that is identical by construction. `predict` now reduces the parsed PSM list to its distinct peptidoforms, predicts those, and scatters the answer back over the caller's rows, so the returned array still has one entry per PSM in the original order. On 8,000 peptides this is 6.4x faster at five copies per peptidoform and 21.9x at twenty; the two paths agree to 9e-05 min, which is float non-associativity from the differing batch layout rather than a change in the answer. Input that is already distinct takes the old path and pays only for building the keys. The charge state is deliberately not part of the key: it reaches no feature the model reads, and predictions for PEPTIDEK, PEPTIDEK/2 and PEPTIDEK/3 are bitwise equal. This matches `deduplicate_psms`, which already ignores charge when choosing calibration references. For `return_matrix`, the column is taken before the scatter so that what gets duplicated is one value per PSM rather than one row of every head. The factored matrix keeps its row-selection path, so a duplicated request still never builds the dense matrix. Co-Authored-By: Claude Opus 5 (1M context) --- deeplc/core.py | 50 +++++++++++++++++++++++++++++++++++++++++++--- tests/test_core.py | 39 ++++++++++++++++++++++++++++++++++++ 2 files changed, 86 insertions(+), 3 deletions(-) diff --git a/deeplc/core.py b/deeplc/core.py index ca8a69c..7b5d7de 100644 --- a/deeplc/core.py +++ b/deeplc/core.py @@ -128,10 +128,16 @@ def predict( if return_matrix and "task_idx" not in kwargs and _model_ops.supports_factored(loaded_model): kwargs.setdefault("factored", True) + # A PSM list holds one entry per match, so the same peptidoform arrives once per scan, + # charge state and run that identified it. Every copy would be parsed, encoded and pushed + # through the trunk again for an answer that cannot differ, so only the distinct + # peptidoforms are predicted and the result is scattered back over the caller's rows. + distinct, inverse = _unique_peptidoforms(_parse_psms(psm_list)) + result = _model_ops.predict( model=loaded_model, data=DeepLCDataset.from_psm_list( - _parse_psms(psm_list), + distinct, include_rolling_sum=getattr(loaded_model, "uses_rolling_sum", True), **_feature_kwargs_from_spec(feature_spec), ), @@ -139,8 +145,11 @@ def predict( ) result = result if isinstance(result, FactoredPredictionMatrix) else result.numpy() if not return_matrix: - return result[:, 0 if "task_idx" in kwargs else _default_task_idx(loaded_model)] - return result + # The column is taken before the scatter, so what gets duplicated is one value per + # PSM rather than one row of every head. + column = result[:, 0 if "task_idx" in kwargs else _default_task_idx(loaded_model)] + return column if inverse is None else column[inverse] + return result if inverse is None else result[inverse] def _default_task_idx(model: torch.nn.Module) -> int: @@ -821,3 +830,38 @@ def _parse_psms(psm_list: PSMList | list[PSM | Peptidoform | str]) -> PSMList: raise ValueError("List must contain either PSMs, Peptidoforms, or strings.") else: raise ValueError("Input must be a PSMList or a list of PSMs, Peptidoforms, or strings.") + + +def _unique_peptidoforms(psm_list: PSMList) -> tuple[PSMList, np.ndarray | None]: + """ + One PSM per distinct peptidoform, and the map back to the caller's rows. + + Identification output holds one PSM per match, so a peptidoform reappears for every scan, + charge state and run that matched it. Predicting each copy costs a parse, an encode and a + forward pass for a value that is identical by construction. On 8,000 peptides, predicting + the distinct ones and scattering the answer back is 6.4x faster at five copies per + peptidoform and 21.9x at twenty, and the two paths agree to 9e-05 min, which is float + non-associativity from the differing batch layout rather than a change in the answer. + + The charge state is deliberately not part of the key. It reaches no feature the model + reads, and predictions for ``PEPTIDEK``, ``PEPTIDEK/2`` and ``PEPTIDEK/3`` are bitwise + equal, so charge states of one peptidoform share a prediction. This matches + :func:`~deeplc._reference_selection.deduplicate_psms`, which already ignores charge when + choosing calibration references. + + Returns ``(psm_list, None)`` when every peptidoform is already distinct, so the common + path pays only for building the keys and never copies the result. + """ + first: dict[str, int] = {} + positions: list[int] = [] + inverse = np.empty(len(psm_list), dtype=np.intp) + for row, psm in enumerate(psm_list): + key = psm.peptidoform.modified_sequence + index = first.get(key) + if index is None: + index = first[key] = len(positions) + positions.append(row) + inverse[row] = index + if len(positions) == len(inverse): + return psm_list, None + return PSMList(psm_list=[psm_list[row] for row in positions]), inverse diff --git a/tests/test_core.py b/tests/test_core.py index 1bb2283..34ae99f 100644 --- a/tests/test_core.py +++ b/tests/test_core.py @@ -123,3 +123,42 @@ def test_predict_and_calibrate_auto_selects_reference(): assert isinstance(result, np.ndarray) assert result.shape == (n,) + + +def test_predict_deduplicates_repeated_peptidoforms(): + """A peptidoform repeated across PSMs gets one prediction, repeated in place.""" + repeated = [p for p in _PEPTIDES for _ in range(3)] + predictions = deeplc.core.predict(_make_psm_list(repeated)) + + assert predictions.shape == (len(repeated),) + once = deeplc.core.predict(_make_psm_list(_PEPTIDES)) + np.testing.assert_allclose(predictions, np.repeat(once, 3), rtol=0, atol=1e-4) + + +def test_predict_is_unchanged_when_every_peptidoform_is_distinct(): + """The deduplication must not disturb the ordinary path.""" + psm_list = _make_psm_list(_PEPTIDES) + predictions = deeplc.core.predict(psm_list) + + assert predictions.shape == (len(_PEPTIDES),) + assert len(set(np.round(predictions, 6))) > 1 + + +def test_predict_shares_a_prediction_across_charge_states(): + """Charge reaches no feature the model reads, so charge states are one peptidoform.""" + charges = ["AGFAGDDAPR", "AGFAGDDAPR/2", "AGFAGDDAPR/3", "AIQEYNQDK/2"] + predictions = deeplc.core.predict(_make_psm_list(charges)) + + assert predictions[0] == predictions[1] == predictions[2] + assert predictions[3] != predictions[0] + + +def test_predict_matrix_keeps_the_callers_rows_when_deduplicating(): + """``return_matrix`` still reports one row per PSM, in the caller's order.""" + repeated = ["AGFAGDDAPR/2", "AIQEYNQDK/2", "AGFAGDDAPR/2", "AAYFGILEK/2", "AIQEYNQDK/2"] + matrix = np.asarray(deeplc.core.predict(_make_psm_list(repeated), return_matrix=True)) + + assert matrix.shape[0] == len(repeated) + np.testing.assert_array_equal(matrix[0], matrix[2]) + np.testing.assert_array_equal(matrix[1], matrix[4]) + assert not np.array_equal(matrix[0], matrix[1])