diff --git a/.github/workflows/build-windows-executable-app.yaml b/.github/workflows/build-windows-executable-app.yaml index f33aac0..10cd25d 100644 --- a/.github/workflows/build-windows-executable-app.yaml +++ b/.github/workflows/build-windows-executable-app.yaml @@ -336,7 +336,7 @@ jobs: Reach out to us: - Join our Discord server for support and community discussions: https://discord.com/invite/4TAGhqJ7s5 - - Contribute or stay updated with the latest OpenMS web app developments on GitHub: https://github.com/OpenMS/streamlit-template + - Contribute or stay updated with the latest OpenDDA developments on GitHub: https://github.com/OpenMS/quantms-web - Visit our website for more information: https://openms.de/ Thank you for using ${{ env.APP_NAME }}! diff --git a/.github/workflows/test-win-exe-w-embed-py.yaml b/.github/workflows/test-win-exe-w-embed-py.yaml index deec56d..806a7b3 100644 --- a/.github/workflows/test-win-exe-w-embed-py.yaml +++ b/.github/workflows/test-win-exe-w-embed-py.yaml @@ -84,7 +84,7 @@ jobs: Reach out to us: - Join our Discord server for support and community discussions: https://discord.com/invite/4TAGhqJ7s5 - - Contribute or stay updated with the latest OpenMS web app developments on GitHub: https://github.com/OpenMS/streamlit-template + - Contribute or stay updated with the latest OpenDDA developments on GitHub: https://github.com/OpenMS/quantms-web - Visit our website for more information: https://openms.de/ Thank you for using ${{ env.APP_NAME }}! diff --git a/Dockerfile b/Dockerfile index 8f314b1..8f5a5cd 100644 --- a/Dockerfile +++ b/Dockerfile @@ -14,7 +14,7 @@ ARG PORT=8501 # Streamlit app GitHub user name (to download artifact from). ARG GITHUB_USER=OpenMS # Streamlit app GitHub repository name (to download artifact from). -ARG GITHUB_REPO=streamlit-template +ARG GITHUB_REPO=quantms-web USER root diff --git a/Dockerfile.arm b/Dockerfile.arm index 847ffcc..798cb20 100644 --- a/Dockerfile.arm +++ b/Dockerfile.arm @@ -14,7 +14,7 @@ ARG PORT=8501 # Streamlit app GitHub user name (to download artifact from). ARG GITHUB_USER=OpenMS # Streamlit app GitHub repository name (to download artifact from). -ARG GITHUB_REPO=streamlit-template +ARG GITHUB_REPO=quantms-web USER root diff --git a/Dockerfile_simple b/Dockerfile_simple index 163bcfe..c77892f 100644 --- a/Dockerfile_simple +++ b/Dockerfile_simple @@ -14,7 +14,7 @@ ARG PORT=8501 # Streamlit app GitHub user name (to download artifact from). ARG GITHUB_USER=OpenMS # Streamlit app GitHub repository name (to download artifact from). -ARG GITHUB_REPO=streamlit-template +ARG GITHUB_REPO=quantms-web # Step 1: set up a sane build system diff --git a/Dockerfile_simple.arm b/Dockerfile_simple.arm index 2d0181c..9c03063 100644 --- a/Dockerfile_simple.arm +++ b/Dockerfile_simple.arm @@ -14,7 +14,7 @@ ARG PORT=8501 # Streamlit app GitHub user name (to download artifact from). ARG GITHUB_USER=OpenMS # Streamlit app GitHub repository name (to download artifact from). -ARG GITHUB_REPO=streamlit-template +ARG GITHUB_REPO=quantms-web # Step 1: set up a sane build system diff --git a/content/quickstart.py b/content/quickstart.py index f79520f..060beb1 100644 --- a/content/quickstart.py +++ b/content/quickstart.py @@ -1,8 +1,8 @@ """ -DDA Label-Free Quantification Quickstart Page. +OpenDDA Quickstart Page. -This page provides an overview of the DDA-LFQ workflow and guidance -for getting started with the analysis pipeline. +This page provides an overview of the DDA quantification workflows (label-free +and TMT) and guidance for getting started with the analysis pipeline. """ from pathlib import Path @@ -27,7 +27,7 @@ def render_windows_download_box(app_bytes: bytes) -> None: st.markdown( """

- Want to run free and open DDA analysis offline? + Want to run free and open DDA analysis (LFQ and TMT) offline?

""", unsafe_allow_html=True, @@ -93,7 +93,7 @@ def render_windows_download_box(app_bytes: bytes) -> None: page_setup(page="main") -st.markdown("# DDA Label-Free Quantification") +st.markdown("# OpenDDA: DDA Quantitative Proteomics") windows_app_bytes = load_windows_app_bytes() if windows_app_bytes is not None: @@ -101,32 +101,56 @@ def render_windows_download_box(app_bytes: bytes) -> None: st.markdown( """ -This application provides a complete **Data-Dependent Acquisition (DDA) Label-Free Quantification** -workflow for proteomics data analysis. The pipeline enables identification and quantification of -proteins from mass spectrometry data. +This application provides complete **Data-Dependent Acquisition (DDA)** quantification +workflows for proteomics data analysis. The pipeline identifies and quantifies proteins +from mass spectrometry data and supports two quantification strategies: + +- **Label-free quantification (LFQ)**: compares precursor intensities across separately measured samples. +- **Tandem Mass Tag (TMT) quantification**: compares reporter-ion intensities of multiplexed samples measured in one run. + +Choose the **Analysis Mode** (LFQ or TMT) on the Configure page. """ ) st.info( - "This workflow mirrors the **dda-lfq branch of the quantms Nextflow workflow** " - "for DDA-based label-free quantification." + "These workflows mirror the **DDA-LFQ and DDA-ISO (TMT) branches of the quantms Nextflow workflow**." ) st.markdown("## Workflow Overview") -st.markdown( - """ -The analysis pipeline consists of five main stages: - -| Stage | Tool | Description | -|-------|------|-------------| -| **1. Identification** | Comet | Peptide-spectrum matching using a protein database | -| **2. Rescoring** | Percolator | Statistical validation using machine learning | -| **3. Filtering** | IDFilter | FDR-controlled filtering of identifications | -| **4. Quantification** | ProteomicsLFQ | Label-free quantification across samples | -| **5. Statistical Analysis** | Built-in | Volcano plots, PCA, and heatmaps | +lfq_col, tmt_col = st.columns(2) + +with lfq_col: + st.markdown( + """ +### Label-free (LFQ) + +| Stage | Tool | +|-------|------| +| **1. Identification** | Comet | +| **2. Rescoring** | Percolator | +| **3. Filtering** | IDFilter | +| **4. Quantification** | ProteomicsLFQ | +| **5. Statistical Analysis** | Built-in | """ -) + ) + +with tmt_col: + st.markdown( + """ +### TMT + +| Stage | Tool | +|-------|------| +| **1. Reporter extraction** | IsobaricAnalyzer | +| **2. Identification** | Comet | +| **3. Rescoring** | Percolator | +| **4. Filtering** | IDFilter | +| **5. Protein inference** | ProteinInference | +| **6. Quantification** | ProteinQuantifier | +| **7. Statistical Analysis** | Built-in | +""" + ) st.markdown("## Getting Started") @@ -143,7 +167,7 @@ def render_windows_download_box(app_bytes: bytes) -> None: st.markdown( """ ### 2. Configure Parameters -Set up search parameters, sample groups, and analysis settings. +Choose LFQ or TMT, then set up search parameters, sample groups (or TMT channels), and analysis settings. """ ) st.page_link("content/workflow_configure.py", label="Go to Configure", icon="⚙️") diff --git a/settings.json b/settings.json index c40a467..997ef00 100644 --- a/settings.json +++ b/settings.json @@ -1,8 +1,8 @@ { - "app-name": "quantms-web (DDA-LFQ)", + "app-name": "OpenDDA", "github-user": "OpenMS", "version": "1.1", - "repository-name": "streamlit-template", + "repository-name": "quantms-web", "legal_links": { "impressum": "https://openms.de/impressum", "privacy": "https://openms.de/privacy", diff --git a/src/WorkflowTest.py b/src/WorkflowTest.py index a2e1274..a0b518f 100644 --- a/src/WorkflowTest.py +++ b/src/WorkflowTest.py @@ -1,6 +1,8 @@ import streamlit as st from pathlib import Path import re +import json +import shutil import pandas as pd import plotly.express as px from streamlit_plotly_events import plotly_events @@ -16,6 +18,29 @@ from src.common.results_helpers import parse_idxml, build_spectra_cache, load_idxml from openms_insight import Table, Heatmap, LinePlot, SequenceView +# Accession prefixes that mark decoy entries in a FASTA database. +DECOY_PREFIXES = ("DECOY_", "decoy_", "rev_", "REV_", "XXX_", "reversed_") + + +@st.cache_data(show_spinner=False) +def detect_decoy_prefix(fasta_path: str, mtime: float) -> str | None: + """Return the decoy prefix used in a FASTA database, or None if it has no decoys. + + `mtime` is only part of the cache key, so a replaced file is scanned again. + """ + try: + with open(fasta_path, "r", errors="ignore") as f: + for line in f: + if line.startswith(">"): + header = line[1:] + for prefix in DECOY_PREFIXES: + if header.startswith(prefix): + return prefix + except OSError: + return None + return None + + # params = page_setup() class WorkflowTest(WorkflowManager): @@ -45,9 +70,11 @@ def upload(self) -> None: def configure(self) -> None: # reactive=True so Group Selection tab updates when selection changes self.ui.select_input_file("mzML-files", multiple=True, reactive=True) - self.ui.select_input_file("fasta-file", multiple=False) + # reactive=True so the decoy default below follows the selected database + self.ui.select_input_file("fasta-file", multiple=False, reactive=True) self.params = self.parameter_manager.get_parameters_from_json() + self.apply_decoy_default() saved_mode = self.params.get("analysis-mode", "LFQ") self.ui.input_widget( @@ -68,6 +95,80 @@ def configure(self) -> None: else: self.render_tmt_tabs() + def apply_decoy_default(self) -> None: + """Default decoy generation to on, unless the selected FASTA already has decoys. + + Runs once per database: a user's choice is kept until another FASTA is + selected. When decoys are found, their prefix becomes Comet's decoy_string + so PeptideIndexing, Percolator and IDFilter recognise them. + """ + prefix = self.parameter_manager.param_prefix + fasta_key = st.session_state.get(f"{prefix}fasta-file", self.params.get("fasta-file")) + if not fasta_key: + return + try: + fasta = Path(self.file_manager.get_files([fasta_key])[0]) + except (ValueError, IndexError): + return + if not fasta.is_file(): + return + + decoy_prefix = detect_decoy_prefix(str(fasta), fasta.stat().st_mtime) + marker = f"{fasta.name}:{decoy_prefix or ''}" + if self.params.get("decoy-check") == marker: + return + + generate = decoy_prefix is None + params = self.parameter_manager.get_parameters_from_json() + params["decoy-check"] = marker + params["generate-decoys"] = generate + if decoy_prefix: + for tool in ("CometAdapter", "CometAdapter-TMT"): + params.setdefault(tool, {})["PeptideIndexing:decoy_string"] = decoy_prefix + # Drop the widget's session value so it re-renders from params.json + st.session_state.pop( + f"{self.parameter_manager.topp_param_prefix}{tool}:1:PeptideIndexing:decoy_string", + None, + ) + with open(self.parameter_manager.params_file, "w", encoding="utf-8") as f: + json.dump(params, f, indent=4) + st.session_state[f"{prefix}decoy-check"] = marker + st.session_state[f"{prefix}generate-decoys"] = generate + self.params = params + self.ui.params = params + + def render_psm_fdr_widget(self) -> None: + self.ui.input_widget( + key="psm-fdr-percent", + default=1.0, + name="PSM FDR level (%)", + widget_type="number", + min_value=0.001, + max_value=100.0, + step_size=1.0, + help="Keep PSMs with a q-value up to this level. 100% disables FDR filtering " + "and passes every PSM on unfiltered.", + ) + + def psm_fdr(self) -> float: + """PSM FDR threshold as a fraction (1.0 means no filtering).""" + return min(float(self.params.get("psm-fdr-percent", 1.0)), 100.0) / 100.0 + + def filter_psms(self, in_files: list, out_files: list, tool_instance_name: str = "IDFilter") -> bool: + """Run IDFilter at the configured PSM FDR, or pass PSMs through at 100%.""" + fdr = self.psm_fdr() + if fdr >= 1.0: + self.logger.log("FDR level is 100%: skipping PSM FDR filtering") + for src, dst in zip(in_files, out_files): + shutil.copy(src, dst) + return True + return self.executor.run_topp( + "IDFilter", + {"in": in_files, "out": out_files}, + {"score:type_peptide": "q-value", "score:psm": fdr}, + tool_instance_name=tool_instance_name, + ) + def render_lfq_tabs(self): st.subheader("LFQ Analysis Mode") t = st.tabs(["**Identification**", "**Rescoring**", "**Filtering**", "**Library Generation**", "**Quantification**", "**Group Selection**"]) @@ -81,7 +182,7 @@ def render_lfq_tabs(self): default=True, name="Generate Decoy Database", widget_type="checkbox", - help="Generate reversed decoy sequences for FDR calculation. Disable if your FASTA already contains decoys.", + help="Generate decoy sequences for FDR calculation. Switched off automatically when the selected FASTA already contains decoys (e.g. DECOY_ or rev_ accessions).", reactive=True, ) @@ -186,18 +287,16 @@ def render_lfq_tabs(self): with t[2]: st.info(""" **Filtering (IDFilter):** - * **score:type_peptide**: Score used for filtering. If empty, the main score is used. - * **score:psm**: The score which should be reached by a peptide hit to be kept. (use 'NAN' to disable this filter) + PSMs are filtered on their Percolator q-value at the FDR level below. + Set it to 100% to keep all PSMs. """) + self.render_psm_fdr_widget() self.ui.input_TOPP( "IDFilter", custom_defaults={ "threads": 2, - "score:type_peptide": "q-value", - "score:psm": 0.10, }, - # include_parameters=["type_peptide", "score:psm"] - exclude_parameters=["type_protein"], + exclude_parameters=["type_protein", "type_peptide", "score:psm"], ) with t[3]: # Library Generation @@ -345,7 +444,7 @@ def render_tmt_tabs(self): default=True, name="Generate Decoy Database", widget_type="checkbox", - help="Generate reversed decoy sequences for FDR calculation. Disable if your FASTA already contains decoys.", + help="Generate decoy sequences for FDR calculation. Switched off automatically when the selected FASTA already contains decoys (e.g. DECOY_ or rev_ accessions).", reactive=True, ) @@ -465,12 +564,15 @@ def render_tmt_tabs(self): ) with t[3]: + st.info(""" + **Filtering (IDFilter):** + PSMs are filtered on their Percolator q-value at the FDR level below. + Set it to 100% to keep all PSMs. + """) + self.render_psm_fdr_widget() self.ui.input_TOPP( "IDFilter", - custom_defaults={ - "score:type_peptide": "q-value", - "score:psm": 0.10, - }, + exclude_parameters=["type_peptide", "score:psm"], tool_instance_name="IDFilter-strict", ) with t[4]: @@ -663,8 +765,9 @@ def execution(self) -> bool: st.success(f"Using decoy FASTA: {decoy_fasta.name}") database_fasta = decoy_fasta else: - # Get decoy_string from CometAdapter params - decoy_string = self.params.get("CometAdapter", {}).get("PeptideIndexing:decoy_string", "rev_") + # Get decoy_string from the active mode's CometAdapter params + comet_instance = "CometAdapter-TMT" if self.params.get("analysis-mode", "LFQ") == "TMT" else "CometAdapter" + decoy_string = self.params.get(comet_instance, {}).get("PeptideIndexing:decoy_string", "rev_") self.logger.log("📄 Using existing FASTA database") st.info(f"Using original FASTA: {fasta_path.name}") database_fasta = fasta_path @@ -930,13 +1033,7 @@ def execution(self) -> bool: # --- IDFilter --- self.logger.log("🔧 Filtering identifications...") with st.spinner(f"IDFilter ({stem})"): - if not self.executor.run_topp( - "IDFilter", - { - "in": percolator_results, - "out": filter_results, - }, - ): + if not self.filter_psms(percolator_results, filter_results): self.logger.log("Workflow stopped due to error") return False @@ -1339,6 +1436,7 @@ def execution(self) -> bool: "in": comet_results, "out": percolator_results, }, + {"decoy_pattern": decoy_string}, # Always propagated from upstream tool_instance_name="PercolatorAdapter-TMT", ): self.logger.log("Workflow stopped due to error") @@ -1422,14 +1520,7 @@ def execution(self) -> bool: # --- IDFilter --- self.logger.log("🔧 Filtering identifications...") with st.spinner(f"IDFilter"): - if not self.executor.run_topp( - "IDFilter", - { - "in": percolator_results, - "out": psm_filtered, - }, - tool_instance_name="IDFilter-strict" - ): + if not self.filter_psms(percolator_results, psm_filtered, "IDFilter-strict"): self.logger.log("Workflow stopped due to error") return False self.logger.log("✅ IDFilter-strict complete") @@ -1550,6 +1641,7 @@ def execution(self) -> bool: "in": [merged_id], "out": [protein_id], }, + {"picked_decoy_string": decoy_string}, # Always propagated from upstream tool_instance_name="ProteinInference-TMT", ): self.logger.log("Workflow stopped due to error") diff --git a/src/workflow/StreamlitUI.py b/src/workflow/StreamlitUI.py index 320e81d..b47fdc9 100644 --- a/src/workflow/StreamlitUI.py +++ b/src/workflow/StreamlitUI.py @@ -5,6 +5,7 @@ import subprocess from typing import Any, Union, List, Literal, Callable import json +import math import os import sys import importlib.util @@ -101,11 +102,14 @@ def upload_widget( name = key.replace("-", " ") c1, c2 = st.columns(2) - c1.markdown("**Upload file(s)**") mount_root = _mounted_data_root() if st.session_state.location == "online" else None if st.session_state.location == "local": + # A local install reads files straight from disk, so the browser + # upload widget (which streams every byte through the browser) is + # not offered. Files are picked from disk and copied or referenced. + c1.markdown("**Add file(s) from your computer**") c2_text, c2_checkbox = c2.columns([1.5, 1], gap="large") c2_text.markdown("**OR add files from local folder**") use_copy = c2_checkbox.checkbox( @@ -115,20 +119,21 @@ def upload_widget( help="Create a copy of files in workspace.", ) else: + c1.markdown("**Upload file(s)**") use_copy = True # Convert file_types to a list if it's a string if isinstance(file_types, str): file_types = [file_types] - if use_copy: + if st.session_state.location != "local": with c1.form(f"{key}-upload", clear_on_submit=True): # Streamlit file uploader accepts file types as a list or None file_type_for_uploader = file_types if file_types else None files = st.file_uploader( f"{name}", - accept_multiple_files=(st.session_state.location == "local"), + accept_multiple_files=False, type=file_type_for_uploader, label_visibility="collapsed", ) @@ -150,29 +155,20 @@ def upload_widget( else: st.error("Nothing to add, please upload file.") else: - # Create a temporary file to store the path to the local directories external_files = Path(files_dir, "external_files.txt") - # Check if the file exists, if not create it - if not external_files.exists(): - external_files.touch() c1.write("\n") with c1.container(border=True): dialog_button = st.button( rf"$\textsf{{\Large 📁 Add }} \textsf{{ \Large \textbf{{{name}}} }}$", type="primary", use_container_width=True, - key="local_browse_single", + key=f"local_browse_single_{key}", help="Browse for your local MS data files.", disabled=not TK_AVAILABLE, ) # Tk file dialog requires file types to be a list of tuples - if isinstance(file_types, str): - tk_file_types = [(f"{file_types}", f"*.{file_types}")] - elif isinstance(file_types, list): - tk_file_types = [(f"{ft}", f"*.{ft}") for ft in file_types] - else: - raise ValueError("'file_types' must be either of type str or list") + tk_file_types = [(f"{ft}", f"*.{ft}") for ft in file_types] if dialog_button: local_files = tk_file_dialog( @@ -183,8 +179,19 @@ def upload_widget( if local_files: my_bar = st.progress(0) for i, f in enumerate(local_files): - with open(external_files, "a") as f_handle: - f_handle.write(f"{f}\n") + my_bar.progress((i + 1) / len(local_files)) + if use_copy: + if os.path.isdir(f): + shutil.copytree( + f, + Path(files_dir, Path(f).name), + dirs_exist_ok=True, + ) + else: + shutil.copy(f, Path(files_dir, Path(f).name)) + else: + with open(external_files, "a") as f_handle: + f_handle.write(f"{f}\n") my_bar.empty() st.success("Successfully added files!") @@ -702,7 +709,9 @@ def seed(**kwargs: Any) -> dict: min_value=min_value, max_value=max_value, step=step_size, - format=None, + # "%g" keeps every decimal the user types; Streamlit's float + # default "%0.2f" rounds to two. + format="%g" if number_type is float else None, key=key, help=help, on_change=on_change, @@ -1158,10 +1167,21 @@ def display_TOPP_params(params: dict, num_cols): # floats elif isinstance(p["value"], float): + value = float(p["value"]) + # Streamlit's default "%0.2f" rounds what the user + # types to two decimals, which makes tolerances such as + # 0.015 Da impossible to enter. "%g" shows the value as + # typed, and small defaults step in their own decade. + step = ( + 10.0 ** math.floor(math.log10(abs(value))) + if 0 < abs(value) < 1 + else 1.0 + ) cols[i].number_input( name, - value=float(p["value"]), - step=1.0, + value=value, + step=step, + format="%g", help=p["description"], key=key, )