diff --git a/.github/workflows/build-windows-executable-app.yaml b/.github/workflows/build-windows-executable-app.yaml
index f33aac0..10cd25d 100644
--- a/.github/workflows/build-windows-executable-app.yaml
+++ b/.github/workflows/build-windows-executable-app.yaml
@@ -336,7 +336,7 @@ jobs:
Reach out to us:
- Join our Discord server for support and community discussions: https://discord.com/invite/4TAGhqJ7s5
- - Contribute or stay updated with the latest OpenMS web app developments on GitHub: https://github.com/OpenMS/streamlit-template
+ - Contribute or stay updated with the latest OpenDDA developments on GitHub: https://github.com/OpenMS/quantms-web
- Visit our website for more information: https://openms.de/
Thank you for using ${{ env.APP_NAME }}!
diff --git a/.github/workflows/test-win-exe-w-embed-py.yaml b/.github/workflows/test-win-exe-w-embed-py.yaml
index deec56d..806a7b3 100644
--- a/.github/workflows/test-win-exe-w-embed-py.yaml
+++ b/.github/workflows/test-win-exe-w-embed-py.yaml
@@ -84,7 +84,7 @@ jobs:
Reach out to us:
- Join our Discord server for support and community discussions: https://discord.com/invite/4TAGhqJ7s5
- - Contribute or stay updated with the latest OpenMS web app developments on GitHub: https://github.com/OpenMS/streamlit-template
+ - Contribute or stay updated with the latest OpenDDA developments on GitHub: https://github.com/OpenMS/quantms-web
- Visit our website for more information: https://openms.de/
Thank you for using ${{ env.APP_NAME }}!
diff --git a/Dockerfile b/Dockerfile
index 8f314b1..8f5a5cd 100644
--- a/Dockerfile
+++ b/Dockerfile
@@ -14,7 +14,7 @@ ARG PORT=8501
# Streamlit app GitHub user name (to download artifact from).
ARG GITHUB_USER=OpenMS
# Streamlit app GitHub repository name (to download artifact from).
-ARG GITHUB_REPO=streamlit-template
+ARG GITHUB_REPO=quantms-web
USER root
diff --git a/Dockerfile.arm b/Dockerfile.arm
index 847ffcc..798cb20 100644
--- a/Dockerfile.arm
+++ b/Dockerfile.arm
@@ -14,7 +14,7 @@ ARG PORT=8501
# Streamlit app GitHub user name (to download artifact from).
ARG GITHUB_USER=OpenMS
# Streamlit app GitHub repository name (to download artifact from).
-ARG GITHUB_REPO=streamlit-template
+ARG GITHUB_REPO=quantms-web
USER root
diff --git a/Dockerfile_simple b/Dockerfile_simple
index 163bcfe..c77892f 100644
--- a/Dockerfile_simple
+++ b/Dockerfile_simple
@@ -14,7 +14,7 @@ ARG PORT=8501
# Streamlit app GitHub user name (to download artifact from).
ARG GITHUB_USER=OpenMS
# Streamlit app GitHub repository name (to download artifact from).
-ARG GITHUB_REPO=streamlit-template
+ARG GITHUB_REPO=quantms-web
# Step 1: set up a sane build system
diff --git a/Dockerfile_simple.arm b/Dockerfile_simple.arm
index 2d0181c..9c03063 100644
--- a/Dockerfile_simple.arm
+++ b/Dockerfile_simple.arm
@@ -14,7 +14,7 @@ ARG PORT=8501
# Streamlit app GitHub user name (to download artifact from).
ARG GITHUB_USER=OpenMS
# Streamlit app GitHub repository name (to download artifact from).
-ARG GITHUB_REPO=streamlit-template
+ARG GITHUB_REPO=quantms-web
# Step 1: set up a sane build system
diff --git a/content/quickstart.py b/content/quickstart.py
index f79520f..060beb1 100644
--- a/content/quickstart.py
+++ b/content/quickstart.py
@@ -1,8 +1,8 @@
"""
-DDA Label-Free Quantification Quickstart Page.
+OpenDDA Quickstart Page.
-This page provides an overview of the DDA-LFQ workflow and guidance
-for getting started with the analysis pipeline.
+This page provides an overview of the DDA quantification workflows (label-free
+and TMT) and guidance for getting started with the analysis pipeline.
"""
from pathlib import Path
@@ -27,7 +27,7 @@ def render_windows_download_box(app_bytes: bytes) -> None:
st.markdown(
"""
- Want to run free and open DDA analysis offline?
+ Want to run free and open DDA analysis (LFQ and TMT) offline?
""",
unsafe_allow_html=True,
@@ -93,7 +93,7 @@ def render_windows_download_box(app_bytes: bytes) -> None:
page_setup(page="main")
-st.markdown("# DDA Label-Free Quantification")
+st.markdown("# OpenDDA: DDA Quantitative Proteomics")
windows_app_bytes = load_windows_app_bytes()
if windows_app_bytes is not None:
@@ -101,32 +101,56 @@ def render_windows_download_box(app_bytes: bytes) -> None:
st.markdown(
"""
-This application provides a complete **Data-Dependent Acquisition (DDA) Label-Free Quantification**
-workflow for proteomics data analysis. The pipeline enables identification and quantification of
-proteins from mass spectrometry data.
+This application provides complete **Data-Dependent Acquisition (DDA)** quantification
+workflows for proteomics data analysis. The pipeline identifies and quantifies proteins
+from mass spectrometry data and supports two quantification strategies:
+
+- **Label-free quantification (LFQ)**: compares precursor intensities across separately measured samples.
+- **Tandem Mass Tag (TMT) quantification**: compares reporter-ion intensities of multiplexed samples measured in one run.
+
+Choose the **Analysis Mode** (LFQ or TMT) on the Configure page.
"""
)
st.info(
- "This workflow mirrors the **dda-lfq branch of the quantms Nextflow workflow** "
- "for DDA-based label-free quantification."
+ "These workflows mirror the **DDA-LFQ and DDA-ISO (TMT) branches of the quantms Nextflow workflow**."
)
st.markdown("## Workflow Overview")
-st.markdown(
- """
-The analysis pipeline consists of five main stages:
-
-| Stage | Tool | Description |
-|-------|------|-------------|
-| **1. Identification** | Comet | Peptide-spectrum matching using a protein database |
-| **2. Rescoring** | Percolator | Statistical validation using machine learning |
-| **3. Filtering** | IDFilter | FDR-controlled filtering of identifications |
-| **4. Quantification** | ProteomicsLFQ | Label-free quantification across samples |
-| **5. Statistical Analysis** | Built-in | Volcano plots, PCA, and heatmaps |
+lfq_col, tmt_col = st.columns(2)
+
+with lfq_col:
+ st.markdown(
+ """
+### Label-free (LFQ)
+
+| Stage | Tool |
+|-------|------|
+| **1. Identification** | Comet |
+| **2. Rescoring** | Percolator |
+| **3. Filtering** | IDFilter |
+| **4. Quantification** | ProteomicsLFQ |
+| **5. Statistical Analysis** | Built-in |
"""
-)
+ )
+
+with tmt_col:
+ st.markdown(
+ """
+### TMT
+
+| Stage | Tool |
+|-------|------|
+| **1. Reporter extraction** | IsobaricAnalyzer |
+| **2. Identification** | Comet |
+| **3. Rescoring** | Percolator |
+| **4. Filtering** | IDFilter |
+| **5. Protein inference** | ProteinInference |
+| **6. Quantification** | ProteinQuantifier |
+| **7. Statistical Analysis** | Built-in |
+"""
+ )
st.markdown("## Getting Started")
@@ -143,7 +167,7 @@ def render_windows_download_box(app_bytes: bytes) -> None:
st.markdown(
"""
### 2. Configure Parameters
-Set up search parameters, sample groups, and analysis settings.
+Choose LFQ or TMT, then set up search parameters, sample groups (or TMT channels), and analysis settings.
"""
)
st.page_link("content/workflow_configure.py", label="Go to Configure", icon="⚙️")
diff --git a/settings.json b/settings.json
index c40a467..997ef00 100644
--- a/settings.json
+++ b/settings.json
@@ -1,8 +1,8 @@
{
- "app-name": "quantms-web (DDA-LFQ)",
+ "app-name": "OpenDDA",
"github-user": "OpenMS",
"version": "1.1",
- "repository-name": "streamlit-template",
+ "repository-name": "quantms-web",
"legal_links": {
"impressum": "https://openms.de/impressum",
"privacy": "https://openms.de/privacy",
diff --git a/src/WorkflowTest.py b/src/WorkflowTest.py
index a2e1274..a0b518f 100644
--- a/src/WorkflowTest.py
+++ b/src/WorkflowTest.py
@@ -1,6 +1,8 @@
import streamlit as st
from pathlib import Path
import re
+import json
+import shutil
import pandas as pd
import plotly.express as px
from streamlit_plotly_events import plotly_events
@@ -16,6 +18,29 @@
from src.common.results_helpers import parse_idxml, build_spectra_cache, load_idxml
from openms_insight import Table, Heatmap, LinePlot, SequenceView
+# Accession prefixes that mark decoy entries in a FASTA database.
+DECOY_PREFIXES = ("DECOY_", "decoy_", "rev_", "REV_", "XXX_", "reversed_")
+
+
+@st.cache_data(show_spinner=False)
+def detect_decoy_prefix(fasta_path: str, mtime: float) -> str | None:
+ """Return the decoy prefix used in a FASTA database, or None if it has no decoys.
+
+ `mtime` is only part of the cache key, so a replaced file is scanned again.
+ """
+ try:
+ with open(fasta_path, "r", errors="ignore") as f:
+ for line in f:
+ if line.startswith(">"):
+ header = line[1:]
+ for prefix in DECOY_PREFIXES:
+ if header.startswith(prefix):
+ return prefix
+ except OSError:
+ return None
+ return None
+
+
# params = page_setup()
class WorkflowTest(WorkflowManager):
@@ -45,9 +70,11 @@ def upload(self) -> None:
def configure(self) -> None:
# reactive=True so Group Selection tab updates when selection changes
self.ui.select_input_file("mzML-files", multiple=True, reactive=True)
- self.ui.select_input_file("fasta-file", multiple=False)
+ # reactive=True so the decoy default below follows the selected database
+ self.ui.select_input_file("fasta-file", multiple=False, reactive=True)
self.params = self.parameter_manager.get_parameters_from_json()
+ self.apply_decoy_default()
saved_mode = self.params.get("analysis-mode", "LFQ")
self.ui.input_widget(
@@ -68,6 +95,80 @@ def configure(self) -> None:
else:
self.render_tmt_tabs()
+ def apply_decoy_default(self) -> None:
+ """Default decoy generation to on, unless the selected FASTA already has decoys.
+
+ Runs once per database: a user's choice is kept until another FASTA is
+ selected. When decoys are found, their prefix becomes Comet's decoy_string
+ so PeptideIndexing, Percolator and IDFilter recognise them.
+ """
+ prefix = self.parameter_manager.param_prefix
+ fasta_key = st.session_state.get(f"{prefix}fasta-file", self.params.get("fasta-file"))
+ if not fasta_key:
+ return
+ try:
+ fasta = Path(self.file_manager.get_files([fasta_key])[0])
+ except (ValueError, IndexError):
+ return
+ if not fasta.is_file():
+ return
+
+ decoy_prefix = detect_decoy_prefix(str(fasta), fasta.stat().st_mtime)
+ marker = f"{fasta.name}:{decoy_prefix or ''}"
+ if self.params.get("decoy-check") == marker:
+ return
+
+ generate = decoy_prefix is None
+ params = self.parameter_manager.get_parameters_from_json()
+ params["decoy-check"] = marker
+ params["generate-decoys"] = generate
+ if decoy_prefix:
+ for tool in ("CometAdapter", "CometAdapter-TMT"):
+ params.setdefault(tool, {})["PeptideIndexing:decoy_string"] = decoy_prefix
+ # Drop the widget's session value so it re-renders from params.json
+ st.session_state.pop(
+ f"{self.parameter_manager.topp_param_prefix}{tool}:1:PeptideIndexing:decoy_string",
+ None,
+ )
+ with open(self.parameter_manager.params_file, "w", encoding="utf-8") as f:
+ json.dump(params, f, indent=4)
+ st.session_state[f"{prefix}decoy-check"] = marker
+ st.session_state[f"{prefix}generate-decoys"] = generate
+ self.params = params
+ self.ui.params = params
+
+ def render_psm_fdr_widget(self) -> None:
+ self.ui.input_widget(
+ key="psm-fdr-percent",
+ default=1.0,
+ name="PSM FDR level (%)",
+ widget_type="number",
+ min_value=0.001,
+ max_value=100.0,
+ step_size=1.0,
+ help="Keep PSMs with a q-value up to this level. 100% disables FDR filtering "
+ "and passes every PSM on unfiltered.",
+ )
+
+ def psm_fdr(self) -> float:
+ """PSM FDR threshold as a fraction (1.0 means no filtering)."""
+ return min(float(self.params.get("psm-fdr-percent", 1.0)), 100.0) / 100.0
+
+ def filter_psms(self, in_files: list, out_files: list, tool_instance_name: str = "IDFilter") -> bool:
+ """Run IDFilter at the configured PSM FDR, or pass PSMs through at 100%."""
+ fdr = self.psm_fdr()
+ if fdr >= 1.0:
+ self.logger.log("FDR level is 100%: skipping PSM FDR filtering")
+ for src, dst in zip(in_files, out_files):
+ shutil.copy(src, dst)
+ return True
+ return self.executor.run_topp(
+ "IDFilter",
+ {"in": in_files, "out": out_files},
+ {"score:type_peptide": "q-value", "score:psm": fdr},
+ tool_instance_name=tool_instance_name,
+ )
+
def render_lfq_tabs(self):
st.subheader("LFQ Analysis Mode")
t = st.tabs(["**Identification**", "**Rescoring**", "**Filtering**", "**Library Generation**", "**Quantification**", "**Group Selection**"])
@@ -81,7 +182,7 @@ def render_lfq_tabs(self):
default=True,
name="Generate Decoy Database",
widget_type="checkbox",
- help="Generate reversed decoy sequences for FDR calculation. Disable if your FASTA already contains decoys.",
+ help="Generate decoy sequences for FDR calculation. Switched off automatically when the selected FASTA already contains decoys (e.g. DECOY_ or rev_ accessions).",
reactive=True,
)
@@ -186,18 +287,16 @@ def render_lfq_tabs(self):
with t[2]:
st.info("""
**Filtering (IDFilter):**
- * **score:type_peptide**: Score used for filtering. If empty, the main score is used.
- * **score:psm**: The score which should be reached by a peptide hit to be kept. (use 'NAN' to disable this filter)
+ PSMs are filtered on their Percolator q-value at the FDR level below.
+ Set it to 100% to keep all PSMs.
""")
+ self.render_psm_fdr_widget()
self.ui.input_TOPP(
"IDFilter",
custom_defaults={
"threads": 2,
- "score:type_peptide": "q-value",
- "score:psm": 0.10,
},
- # include_parameters=["type_peptide", "score:psm"]
- exclude_parameters=["type_protein"],
+ exclude_parameters=["type_protein", "type_peptide", "score:psm"],
)
with t[3]: # Library Generation
@@ -345,7 +444,7 @@ def render_tmt_tabs(self):
default=True,
name="Generate Decoy Database",
widget_type="checkbox",
- help="Generate reversed decoy sequences for FDR calculation. Disable if your FASTA already contains decoys.",
+ help="Generate decoy sequences for FDR calculation. Switched off automatically when the selected FASTA already contains decoys (e.g. DECOY_ or rev_ accessions).",
reactive=True,
)
@@ -465,12 +564,15 @@ def render_tmt_tabs(self):
)
with t[3]:
+ st.info("""
+ **Filtering (IDFilter):**
+ PSMs are filtered on their Percolator q-value at the FDR level below.
+ Set it to 100% to keep all PSMs.
+ """)
+ self.render_psm_fdr_widget()
self.ui.input_TOPP(
"IDFilter",
- custom_defaults={
- "score:type_peptide": "q-value",
- "score:psm": 0.10,
- },
+ exclude_parameters=["type_peptide", "score:psm"],
tool_instance_name="IDFilter-strict",
)
with t[4]:
@@ -663,8 +765,9 @@ def execution(self) -> bool:
st.success(f"Using decoy FASTA: {decoy_fasta.name}")
database_fasta = decoy_fasta
else:
- # Get decoy_string from CometAdapter params
- decoy_string = self.params.get("CometAdapter", {}).get("PeptideIndexing:decoy_string", "rev_")
+ # Get decoy_string from the active mode's CometAdapter params
+ comet_instance = "CometAdapter-TMT" if self.params.get("analysis-mode", "LFQ") == "TMT" else "CometAdapter"
+ decoy_string = self.params.get(comet_instance, {}).get("PeptideIndexing:decoy_string", "rev_")
self.logger.log("📄 Using existing FASTA database")
st.info(f"Using original FASTA: {fasta_path.name}")
database_fasta = fasta_path
@@ -930,13 +1033,7 @@ def execution(self) -> bool:
# --- IDFilter ---
self.logger.log("🔧 Filtering identifications...")
with st.spinner(f"IDFilter ({stem})"):
- if not self.executor.run_topp(
- "IDFilter",
- {
- "in": percolator_results,
- "out": filter_results,
- },
- ):
+ if not self.filter_psms(percolator_results, filter_results):
self.logger.log("Workflow stopped due to error")
return False
@@ -1339,6 +1436,7 @@ def execution(self) -> bool:
"in": comet_results,
"out": percolator_results,
},
+ {"decoy_pattern": decoy_string}, # Always propagated from upstream
tool_instance_name="PercolatorAdapter-TMT",
):
self.logger.log("Workflow stopped due to error")
@@ -1422,14 +1520,7 @@ def execution(self) -> bool:
# --- IDFilter ---
self.logger.log("🔧 Filtering identifications...")
with st.spinner(f"IDFilter"):
- if not self.executor.run_topp(
- "IDFilter",
- {
- "in": percolator_results,
- "out": psm_filtered,
- },
- tool_instance_name="IDFilter-strict"
- ):
+ if not self.filter_psms(percolator_results, psm_filtered, "IDFilter-strict"):
self.logger.log("Workflow stopped due to error")
return False
self.logger.log("✅ IDFilter-strict complete")
@@ -1550,6 +1641,7 @@ def execution(self) -> bool:
"in": [merged_id],
"out": [protein_id],
},
+ {"picked_decoy_string": decoy_string}, # Always propagated from upstream
tool_instance_name="ProteinInference-TMT",
):
self.logger.log("Workflow stopped due to error")
diff --git a/src/workflow/StreamlitUI.py b/src/workflow/StreamlitUI.py
index 320e81d..b47fdc9 100644
--- a/src/workflow/StreamlitUI.py
+++ b/src/workflow/StreamlitUI.py
@@ -5,6 +5,7 @@
import subprocess
from typing import Any, Union, List, Literal, Callable
import json
+import math
import os
import sys
import importlib.util
@@ -101,11 +102,14 @@ def upload_widget(
name = key.replace("-", " ")
c1, c2 = st.columns(2)
- c1.markdown("**Upload file(s)**")
mount_root = _mounted_data_root() if st.session_state.location == "online" else None
if st.session_state.location == "local":
+ # A local install reads files straight from disk, so the browser
+ # upload widget (which streams every byte through the browser) is
+ # not offered. Files are picked from disk and copied or referenced.
+ c1.markdown("**Add file(s) from your computer**")
c2_text, c2_checkbox = c2.columns([1.5, 1], gap="large")
c2_text.markdown("**OR add files from local folder**")
use_copy = c2_checkbox.checkbox(
@@ -115,20 +119,21 @@ def upload_widget(
help="Create a copy of files in workspace.",
)
else:
+ c1.markdown("**Upload file(s)**")
use_copy = True
# Convert file_types to a list if it's a string
if isinstance(file_types, str):
file_types = [file_types]
- if use_copy:
+ if st.session_state.location != "local":
with c1.form(f"{key}-upload", clear_on_submit=True):
# Streamlit file uploader accepts file types as a list or None
file_type_for_uploader = file_types if file_types else None
files = st.file_uploader(
f"{name}",
- accept_multiple_files=(st.session_state.location == "local"),
+ accept_multiple_files=False,
type=file_type_for_uploader,
label_visibility="collapsed",
)
@@ -150,29 +155,20 @@ def upload_widget(
else:
st.error("Nothing to add, please upload file.")
else:
- # Create a temporary file to store the path to the local directories
external_files = Path(files_dir, "external_files.txt")
- # Check if the file exists, if not create it
- if not external_files.exists():
- external_files.touch()
c1.write("\n")
with c1.container(border=True):
dialog_button = st.button(
rf"$\textsf{{\Large 📁 Add }} \textsf{{ \Large \textbf{{{name}}} }}$",
type="primary",
use_container_width=True,
- key="local_browse_single",
+ key=f"local_browse_single_{key}",
help="Browse for your local MS data files.",
disabled=not TK_AVAILABLE,
)
# Tk file dialog requires file types to be a list of tuples
- if isinstance(file_types, str):
- tk_file_types = [(f"{file_types}", f"*.{file_types}")]
- elif isinstance(file_types, list):
- tk_file_types = [(f"{ft}", f"*.{ft}") for ft in file_types]
- else:
- raise ValueError("'file_types' must be either of type str or list")
+ tk_file_types = [(f"{ft}", f"*.{ft}") for ft in file_types]
if dialog_button:
local_files = tk_file_dialog(
@@ -183,8 +179,19 @@ def upload_widget(
if local_files:
my_bar = st.progress(0)
for i, f in enumerate(local_files):
- with open(external_files, "a") as f_handle:
- f_handle.write(f"{f}\n")
+ my_bar.progress((i + 1) / len(local_files))
+ if use_copy:
+ if os.path.isdir(f):
+ shutil.copytree(
+ f,
+ Path(files_dir, Path(f).name),
+ dirs_exist_ok=True,
+ )
+ else:
+ shutil.copy(f, Path(files_dir, Path(f).name))
+ else:
+ with open(external_files, "a") as f_handle:
+ f_handle.write(f"{f}\n")
my_bar.empty()
st.success("Successfully added files!")
@@ -702,7 +709,9 @@ def seed(**kwargs: Any) -> dict:
min_value=min_value,
max_value=max_value,
step=step_size,
- format=None,
+ # "%g" keeps every decimal the user types; Streamlit's float
+ # default "%0.2f" rounds to two.
+ format="%g" if number_type is float else None,
key=key,
help=help,
on_change=on_change,
@@ -1158,10 +1167,21 @@ def display_TOPP_params(params: dict, num_cols):
# floats
elif isinstance(p["value"], float):
+ value = float(p["value"])
+ # Streamlit's default "%0.2f" rounds what the user
+ # types to two decimals, which makes tolerances such as
+ # 0.015 Da impossible to enter. "%g" shows the value as
+ # typed, and small defaults step in their own decade.
+ step = (
+ 10.0 ** math.floor(math.log10(abs(value)))
+ if 0 < abs(value) < 1
+ else 1.0
+ )
cols[i].number_input(
name,
- value=float(p["value"]),
- step=1.0,
+ value=value,
+ step=step,
+ format="%g",
help=p["description"],
key=key,
)