From 3d9cce0f609a827ce684a10abc99cd1a7f4f827c Mon Sep 17 00:00:00 2001 From: Rakesh Ramakrishna Pai <41351936+developer-rpai@users.noreply.github.com> Date: Wed, 23 Sep 2026 23:26:51 -0700 Subject: [PATCH] Omit blank BioProject references from submission XML instead of writing "Not Provided" NCBI rejects submissions that carry a placeholder value such as "Not Provided" as a BioProject PrimaryId, which causes tostadas to hang waiting on the rejected submission. When ncbi-bioproject is blank, missing, or a placeholder, the BioProject element/AttributeRefId is now omitted entirely and a warning is logged, per the suggestion in #362. Fixes #362. --- bin/submission_helper.py | 56 ++++++++++++--- tests/test_submission_helper.py | 117 ++++++++++++++++++++++++++++++++ 2 files changed, 162 insertions(+), 11 deletions(-) create mode 100644 tests/test_submission_helper.py diff --git a/bin/submission_helper.py b/bin/submission_helper.py index 451b30bb..f633f318 100644 --- a/bin/submission_helper.py +++ b/bin/submission_helper.py @@ -694,6 +694,22 @@ def safe_text(self, value): if value is None or (isinstance(value, float) and math.isnan(value)): return "Not Provided" return str(value) + def bioproject_id(self): + """Return the BioProject accession from the metadata, or None if it is + blank, missing, or a placeholder. + + NCBI rejects submissions that carry a placeholder value such as + "Not Provided" as a BioProject PrimaryId (and tostadas then hangs + waiting on the rejected submission), so callers must omit the + BioProject element entirely instead of writing the placeholder. + """ + value = self.top_metadata.get('ncbi-bioproject') + if value is None or (isinstance(value, float) and math.isnan(value)): + return None + value = str(value).strip() + if not value or value == "Not Provided": + return None + return value def init_xml_root(self): self.submission_root = ET.Element('Submission') # Description @@ -796,9 +812,15 @@ def add_action_block(self, submission): organism_name = ET.SubElement(organism, 'OrganismName') organism_name.text = self.safe_text(self.biosample_metadata['organism']) # BioProject reference - bioproject = ET.SubElement(biosample, 'BioProject') - primary_id = ET.SubElement(bioproject, 'PrimaryId', {'db': 'BioProject'}) - primary_id.text = self.safe_text(self.top_metadata['ncbi-bioproject']) + # Omit the element entirely when no BioProject was provided: NCBI + # rejects submissions carrying a placeholder such as "Not Provided". + bioproject_id = self.bioproject_id() + if bioproject_id is not None: + bioproject = ET.SubElement(biosample, 'BioProject') + primary_id = ET.SubElement(bioproject, 'PrimaryId', {'db': 'BioProject'}) + primary_id.text = bioproject_id + else: + logging.warning("ncbi-bioproject is blank; omitting the BioProject reference from the BioSample submission XML") # Package bs_package = ET.SubElement(biosample, 'Package') bs_package.text = self.safe_text(self.submission_config['BioSample_package']) @@ -857,10 +879,16 @@ def add_attributes_block(self, add_files): attribute.text = self.safe_text(attr_value) spuid_namespace_value = self.safe_text(self.submission_config['NCBI_Namespace']) # BioProject reference - attribute_ref_id_bioproject = ET.SubElement(add_files, "AttributeRefId", name="BioProject") - refid_bioproject = ET.SubElement(attribute_ref_id_bioproject, "RefId") - primaryid_bioproject = ET.SubElement(refid_bioproject, "PrimaryId") - primaryid_bioproject.text = self.safe_text(self.top_metadata['ncbi-bioproject']) + # Omit the element entirely when no BioProject was provided: NCBI + # rejects submissions carrying a placeholder such as "Not Provided". + bioproject_id = self.bioproject_id() + if bioproject_id is not None: + attribute_ref_id_bioproject = ET.SubElement(add_files, "AttributeRefId", name="BioProject") + refid_bioproject = ET.SubElement(attribute_ref_id_bioproject, "RefId") + primaryid_bioproject = ET.SubElement(refid_bioproject, "PrimaryId") + primaryid_bioproject.text = bioproject_id + else: + logging.warning("ncbi-bioproject is blank; omitting the BioProject reference from the SRA submission XML") # BioSample reference attribute_ref_id_biosample = ET.SubElement(add_files, "AttributeRefId", name="BioSample") refid_biosample = ET.SubElement(attribute_ref_id_biosample, "RefId") @@ -953,10 +981,16 @@ def xml_create_wgs(self): ET.SubElement(description, "GenomeRepresentation").text = "Full" ET.SubElement(description, "ExpectedFinalVersion").text = "Yes" # AttributeRefId for BioProject - attribute_ref = ET.SubElement(add_files, "AttributeRefId") - ref_id = ET.SubElement(attribute_ref, "RefId") - primary_id = ET.SubElement(ref_id, "PrimaryId", db="BioProject") - primary_id.text = self.safe_text(self.top_metadata["ncbi-bioproject"]) + # Omit the element entirely when no BioProject was provided: NCBI + # rejects submissions carrying a placeholder such as "Not Provided". + bioproject_id = self.bioproject_id() + if bioproject_id is not None: + attribute_ref = ET.SubElement(add_files, "AttributeRefId") + ref_id = ET.SubElement(attribute_ref, "RefId") + primary_id = ET.SubElement(ref_id, "PrimaryId", db="BioProject") + primary_id.text = bioproject_id + else: + logging.warning("ncbi-bioproject is blank; omitting the BioProject reference from the GenBank submission XML") # AttributeRefId for BioSample attribute_ref = ET.SubElement(add_files, "AttributeRefId") ref_id = ET.SubElement(attribute_ref, "RefId") diff --git a/tests/test_submission_helper.py b/tests/test_submission_helper.py new file mode 100644 index 00000000..15c6697f --- /dev/null +++ b/tests/test_submission_helper.py @@ -0,0 +1,117 @@ +"""Regression tests for https://github.com/CDCgov/tostadas/issues/362. + +If `ncbi-bioproject` is blank in the metadata, the generated submission XML +must not contain a BioProject reference at all. Writing the placeholder +"Not Provided" (the old behavior) causes NCBI to reject the submission and +tostadas hangs waiting on it. +""" +import importlib.util +import logging +import xml.etree.ElementTree as ET +from pathlib import Path + +import pytest + +BIN_DIR = Path(__file__).resolve().parent.parent / "bin" + + +def load_submission_helper(): + spec = importlib.util.spec_from_file_location( + "submission_helper", BIN_DIR / "submission_helper.py" + ) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +sh = load_submission_helper() + + +def make_biosample_obj(bioproject_value): + obj = sh.BiosampleSubmission.__new__(sh.BiosampleSubmission) + obj.submission_config = { + "NCBI_Namespace": "TESTNS", + "BioSample_package": "Pathogen.cl.1.0", + } + obj.top_metadata = { + "ncbi-spuid": "SAMPLE1", + "ncbi-bioproject": bioproject_value, + "title": "test title", + } + obj.biosample_metadata = {"organism": "Test organism"} + obj.accession_id = None + obj.wastewater = False + return obj + + +def make_sra_obj(bioproject_value): + obj = sh.SRASubmission.__new__(sh.SRASubmission) + obj.submission_config = {"NCBI_Namespace": "TESTNS"} + obj.top_metadata = { + "ncbi-bioproject": bioproject_value, + "ncbi-spuid": "SAMPLE1", + "ncbi-spuid-sra": "SAMPLE1-SRA", + } + obj.sra_metadata = {} + return obj + + +def biosample_xml(bioproject_value): + obj = make_biosample_obj(bioproject_value) + root = ET.Element("Submission") + obj.add_action_block(root) + return ET.tostring(root, encoding="unicode") + + +def sra_xml(bioproject_value): + obj = make_sra_obj(bioproject_value) + add_files = ET.Element("AddFiles") + obj.add_attributes_block(add_files) + return ET.tostring(add_files, encoding="unicode") + + +@pytest.mark.parametrize("blank", [float("nan"), "", " ", "Not Provided", None]) +def test_blank_bioproject_omitted_from_biosample_xml(blank, caplog): + with caplog.at_level(logging.WARNING): + xml = biosample_xml(blank) + assert "BioProject" not in xml + assert "Not Provided" not in xml + assert "ncbi-bioproject is blank" in caplog.text + + +@pytest.mark.parametrize("blank", [float("nan"), "", " ", "Not Provided", None]) +def test_blank_bioproject_omitted_from_sra_xml(blank, caplog): + with caplog.at_level(logging.WARNING): + xml = sra_xml(blank) + assert "BioProject" not in xml + assert "Not Provided" not in xml + assert "ncbi-bioproject is blank" in caplog.text + + +def test_missing_bioproject_key_omitted(): + obj = make_biosample_obj("PRJNA123456") + del obj.top_metadata["ncbi-bioproject"] + root = ET.Element("Submission") + obj.add_action_block(root) + xml = ET.tostring(root, encoding="unicode") + assert "BioProject" not in xml + assert "Not Provided" not in xml + + +def test_real_bioproject_still_written(): + for xml in (biosample_xml("PRJNA123456"), sra_xml("PRJNA123456")): + assert "BioProject" in xml + assert "PRJNA123456" in xml + assert "Not Provided" not in xml + + +def test_bioproject_id_helper(): + obj = make_biosample_obj("PRJNA123456") + assert obj.bioproject_id() == "PRJNA123456" + obj.top_metadata["ncbi-bioproject"] = " PRJNA123456 " + assert obj.bioproject_id() == "PRJNA123456" + for blank in (float("nan"), "", " ", "Not Provided", None): + obj.top_metadata["ncbi-bioproject"] = blank + assert obj.bioproject_id() is None, blank + del obj.top_metadata["ncbi-bioproject"] + assert obj.bioproject_id() is None