From 625eb8e4c72bd12ca20474e3d57a935e764df66b Mon Sep 17 00:00:00 2001 From: nsheff Date: Wed, 25 Feb 2026 16:03:05 -0500 Subject: [PATCH 1/7] fix race condition --- pipestat/backends/file_backend/filebackend.py | 10 +++++++--- 1 file changed, 7 insertions(+), 3 deletions(-) diff --git a/pipestat/backends/file_backend/filebackend.py b/pipestat/backends/file_backend/filebackend.py index 10d0dce9..7beeb2f1 100644 --- a/pipestat/backends/file_backend/filebackend.py +++ b/pipestat/backends/file_backend/filebackend.py @@ -193,9 +193,13 @@ def get_status(self, record_identifier: str) -> Optional[str]: assert isinstance(flag_file, str), TypeError( "Flag file path is expected to be a str, were multiple flags found?" ) - with open(flag_file, "r") as f: - status = f.read() - return status + try: + with open(flag_file, "r") as f: + status = f.read() + return status + except FileNotFoundError: + _LOGGER.debug(f"Flag file disappeared: {flag_file}") + return None _LOGGER.debug( f"Could not determine status for '{r_id}' record. " f"No flags found in: {self.status_file_dir}" From 58c8051e0fa484b6f5208a94bcf69b0de501ab63 Mon Sep 17 00:00:00 2001 From: nsheff Date: Thu, 26 Feb 2026 18:11:17 -0500 Subject: [PATCH 2/7] improve rec identifier handling --- CLAUDE.md | 1 + pipestat/pipestat.py | 6 ++ pyproject.toml | 2 +- tests/test_record_identifier_setter.py | 107 +++++++++++++++++++++++++ 4 files changed, 115 insertions(+), 1 deletion(-) create mode 100644 tests/test_record_identifier_setter.py diff --git a/CLAUDE.md b/CLAUDE.md index b9247506..31c66d32 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -72,6 +72,7 @@ Classmethods for construction: - For project-level pipelines, `record_identifier` auto-defaults to `project_name` (which defaults to `"project"`) - `force_overwrite` defaults to `True` at manager level; it is a settable property - `result_formatter` is also a settable property (not an `__init__` param) +- `record_identifier` is a settable property for changing the default record after construction - File/image results require `{"path": "...", "title": "..."}` dict format (`thumbnail_path` is optional for images) ## Schema-Free Mode diff --git a/pipestat/pipestat.py b/pipestat/pipestat.py index d4894d07..e32ce029 100644 --- a/pipestat/pipestat.py +++ b/pipestat/pipestat.py @@ -1988,6 +1988,12 @@ def record_identifier(self) -> str | None: """ return self._resolve_record_identifier(None) + @record_identifier.setter + def record_identifier(self, value: str | None) -> None: + if value is not None and not value: + raise ValueError("record_identifier cannot be empty") + self.cfg[RECORD_IDENTIFIER] = value + @property def record_count(self) -> int: """Number of records reported. diff --git a/pyproject.toml b/pyproject.toml index a70943dd..ede0697f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "pipestat" -version = "0.13.0" +version = "0.13.1" description = "A pipeline results reporter" readme = "README.md" license = "BSD-2-Clause" diff --git a/tests/test_record_identifier_setter.py b/tests/test_record_identifier_setter.py new file mode 100644 index 00000000..79180b60 --- /dev/null +++ b/tests/test_record_identifier_setter.py @@ -0,0 +1,107 @@ +"""Tests for the record_identifier setter on PipestatManager.""" + +import pytest + +from pipestat import PipestatManager + + +class TestRecordIdentifierSetter: + """Tests for setting record_identifier after construction.""" + + def test_set_record_identifier_after_construction(self, tmp_path): + """Setting record_identifier updates the weak-bound default.""" + schema_content = """ +pipeline_name: test_pipeline +samples: + count: + type: integer + description: A count +""" + schema_file = tmp_path / "schema.yaml" + schema_file.write_text(schema_content) + + psm = PipestatManager( + schema_path=str(schema_file), + results_file_path=str(tmp_path / "results.yaml"), + pipeline_name="test", + ) + + # Initially None (no record_identifier set) + assert psm.record_identifier is None + + # Set it + psm.record_identifier = "sample1" + assert psm.record_identifier == "sample1" + + def test_set_record_identifier_enables_default_reporting(self, tmp_path): + """After setting record_identifier, report/retrieve use it as default.""" + schema_content = """ +pipeline_name: test_pipeline +samples: + count: + type: integer + description: A count +""" + schema_file = tmp_path / "schema.yaml" + schema_file.write_text(schema_content) + + psm = PipestatManager( + schema_path=str(schema_file), + results_file_path=str(tmp_path / "results.yaml"), + pipeline_name="test", + ) + + psm.record_identifier = "sample1" + + # Report without passing record_identifier -- uses the weak bound + psm.report(values={"count": 42}) + + # Retrieve without passing record_identifier + result = psm.retrieve_one() + assert result["count"] == 42 + + def test_set_record_identifier_to_none(self, tmp_path): + """Setting record_identifier to None clears it.""" + psm = PipestatManager( + results_file_path=str(tmp_path / "results.yaml"), + pipeline_name="test", + record_identifier="sample1", + validate_results=False, + ) + + assert psm.record_identifier == "sample1" + + psm.record_identifier = None + assert psm.record_identifier is None + + def test_set_record_identifier_empty_string_raises(self, tmp_path): + """Setting record_identifier to empty string raises ValueError.""" + psm = PipestatManager( + results_file_path=str(tmp_path / "results.yaml"), + pipeline_name="test", + validate_results=False, + ) + + with pytest.raises(ValueError, match="record_identifier cannot be empty"): + psm.record_identifier = "" + + def test_set_record_identifier_changes_default(self, tmp_path): + """Changing record_identifier switches which record is the default.""" + psm = PipestatManager( + results_file_path=str(tmp_path / "results.yaml"), + pipeline_name="test", + validate_results=False, + ) + + psm.record_identifier = "sample1" + psm.report(values={"metric": 10}) + + psm.record_identifier = "sample2" + psm.report(values={"metric": 20}) + + # Retrieve each + psm.record_identifier = "sample1" + assert psm.retrieve_one()["metric"] == 10 + + psm.record_identifier = "sample2" + assert psm.retrieve_one()["metric"] == 20 From c8c6edb09062e77f597746a1df90a68f383c68c1 Mon Sep 17 00:00:00 2001 From: nsheff Date: Thu, 26 Feb 2026 18:11:27 -0500 Subject: [PATCH 3/7] foramt --- pipestat/exceptions.py | 1 - tests/test_pephub.py | 9 --------- 2 files changed, 10 deletions(-) diff --git a/pipestat/exceptions.py b/pipestat/exceptions.py index 1c789011..5ebc5421 100644 --- a/pipestat/exceptions.py +++ b/pipestat/exceptions.py @@ -88,7 +88,6 @@ class SchemaValidationErrorDuringReport(SchemaError): """Adds clarity to JSON schema validation errors by providing additional information to error message.""" def __init__(self, msg, record_identifier, result_identifier, result): - txt = msg # original schema validation error txt += f"\nRecord identifier {record_identifier} \nResult_identifier {result_identifier} \nReported result: {result}" super(SchemaValidationErrorDuringReport, self).__init__(txt) diff --git a/tests/test_pephub.py b/tests/test_pephub.py index 355d847f..c3eee2e6 100644 --- a/tests/test_pephub.py +++ b/tests/test_pephub.py @@ -49,7 +49,6 @@ def test_pephub_backend_report( results_file_path, range_values, ): - psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) # Value already exists should give an error unless forcing overwrite @@ -74,7 +73,6 @@ def test_pephub_backend_retrieve_one( results_file_path, range_values, ): - psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) result = psm.retrieve_one(record_identifier=rec_id) @@ -94,7 +92,6 @@ def test_pephub_backend_config_file( config_file_path_pephub, schema_file_path, ): - # Can pipestat obtain pephub url from config file AND successfully retrieve values? psm = PipestatManager(config_file=config_file_path_pephub, schema_path=schema_file_path) @@ -109,7 +106,6 @@ def test_pephub_backend_retrieve_many( results_file_path, range_values, ): - rec_ids = ["test_pipestat_01", "test_pipestat_02"] psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) @@ -159,7 +155,6 @@ def test_pephub_backend_remove( results_file_path, range_values, ): - rec_ids = ["test_pipestat_01"] psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) @@ -175,7 +170,6 @@ def test_pephub_backend_remove_record( results_file_path, range_values, ): - rec_ids = ["test_pipestat_01"] psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) @@ -191,7 +185,6 @@ def test_pephub_unsupported_funcs( results_file_path, range_values, ): - rec_ids = ["test_pipestat_01"] psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) @@ -207,7 +200,6 @@ def test_pephub_backend_summarize( config_file_path, schema_file_path, ): - with TemporaryDirectory() as d: psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) report_path = psm.summarize(output_dir=d) @@ -219,7 +211,6 @@ def test_pephub_backend_link( config_file_path, schema_file_path, ): - with TemporaryDirectory() as d: psm = PipestatManager(pephub_path=PEPHUB_URL, schema_path=schema_file_path) report_path = psm.link(link_dir=d) From 09b7e7991bc65d65b55503e470d362f40af9d308 Mon Sep 17 00:00:00 2001 From: nsheff Date: Fri, 27 Feb 2026 13:21:16 -0500 Subject: [PATCH 4/7] remove pandas dep, gate imports for faster importing --- pipestat/backends/db_backend/__init__.py | 3 + pipestat/backends/pephub_backend/__init__.py | 1 + pipestat/pipestat.py | 48 +++++----- pipestat/reports.py | 98 ++++++-------------- pyproject.toml | 5 +- tests/test_profile_parsing.py | 71 ++++++++++++++ 6 files changed, 129 insertions(+), 97 deletions(-) create mode 100644 tests/test_profile_parsing.py diff --git a/pipestat/backends/db_backend/__init__.py b/pipestat/backends/db_backend/__init__.py index e69de29b..f119e474 100644 --- a/pipestat/backends/db_backend/__init__.py +++ b/pipestat/backends/db_backend/__init__.py @@ -0,0 +1,3 @@ +from .db_helpers import construct_db_url +from .db_parsed_schema import ParsedSchemaDB +from .dbbackend import DBBackend diff --git a/pipestat/backends/pephub_backend/__init__.py b/pipestat/backends/pephub_backend/__init__.py index e69de29b..18803b75 100644 --- a/pipestat/backends/pephub_backend/__init__.py +++ b/pipestat/backends/pephub_backend/__init__.py @@ -0,0 +1 @@ +from .pephubbackend import PEPHUBBACKEND diff --git a/pipestat/pipestat.py b/pipestat/pipestat.py index d4894d07..34929cba 100644 --- a/pipestat/pipestat.py +++ b/pipestat/pipestat.py @@ -51,26 +51,7 @@ SchemaNotFoundError, ) from .helpers import default_formatter, make_subdirectories, validate_type, zip_report -from .reports import HTMLReportBuilder, _create_stats_objs_summaries - -try: - from pipestat.backends.db_backend.db_parsed_schema import ParsedSchemaDB as ParsedSchema -except ImportError: - from .parsed_schema import ParsedSchema - -try: - from pipestat.backends.db_backend.db_helpers import construct_db_url - from pipestat.backends.db_backend.dbbackend import DBBackend -except ImportError: - # We let this pass, but if the user attempts to create DBBackend, check_dependencies raises exception. - pass - -try: - from pipestat.backends.pephub_backend.pephubbackend import PEPHUBBACKEND -except ImportError: - # Let this pass, if phc dependencies cannot be imported, raise exception - pass - +from .parsed_schema import ParsedSchema _LOGGER = getLogger(PKG_NAME) @@ -775,6 +756,14 @@ def initialize_pephubbackend( record_identifier (str, optional): The record identifier. pephub_path (str, optional): The path to the pephub registry. """ + try: + from pipestat.backends.pephub_backend import PEPHUBBACKEND + except ImportError: + raise PipestatDependencyError( + msg="Missing required dependencies for PEPhub backend. " + "Install them with: pip install pipestat[pephub]" + ) + self.backend = PEPHUBBACKEND( record_identifier, pephub_path, @@ -785,10 +774,6 @@ def initialize_pephubbackend( self.cfg[RESULT_FORMATTER], ) - @check_dependencies( - dependency_list=["DBBackend"], - msg="Missing required dependencies for this usage, e.g. try pip install pipestat['dbbackend']", - ) def initialize_dbbackend( self, record_identifier: str | None = None, show_db_logs: bool = False ) -> None: @@ -803,6 +788,14 @@ def initialize_dbbackend( NoBackendSpecifiedError: If database configuration is missing. PipestatDatabaseError: If database configuration is invalid. """ + try: + from pipestat.backends.db_backend import DBBackend, ParsedSchemaDB, construct_db_url + except ImportError: + raise PipestatDependencyError( + msg="Missing required dependencies for database backend. " + "Install them with: pip install pipestat[dbbackend]" + ) + _LOGGER.debug("Determined database as backend") if not self.cfg.get(PROJECT_NAME): raise ValueError( @@ -827,6 +820,10 @@ def initialize_dbbackend( raise PipestatDatabaseError(f"No database section ('{CFG_DATABASE_KEY}') in config") self._show_db_logs = show_db_logs + # Re-parse schema with DB-aware parser for model building + if self._schema_path is not None: + self.cfg[SCHEMA_KEY] = ParsedSchemaDB(self._schema_path) + self.backend = DBBackend( record_identifier, self.cfg[PIPELINE_NAME], @@ -1694,6 +1691,7 @@ def summarize( Raises: PipestatSummarizeError: If no results are found at the backend. """ + from .reports import HTMLReportBuilder if output_dir: self.cfg[OUTPUT_DIR] = output_dir @@ -1758,6 +1756,8 @@ def table( Returns: list[str]: File paths of the generated stats and objects files. """ + from .reports import _create_stats_objs_summaries + if output_dir: self.cfg[OUTPUT_DIR] = output_dir diff --git a/pipestat/reports.py b/pipestat/reports.py index 758c16a8..94e272ab 100644 --- a/pipestat/reports.py +++ b/pipestat/reports.py @@ -12,10 +12,9 @@ from logging import getLogger import jinja2 -import pandas as _pd import yaml from peppy.const import AMENDMENTS_KEY -from ubiquerg import mkabs +from ubiquerg import mkabs, parse_timedelta from .const import ( BUTTON_APPEARANCE_BY_FLAG, @@ -1273,43 +1272,6 @@ def _make_relpath(file_name, wd, context=None): return relpath if not context else os.path.join(os.path.join(*context), relpath) -def _read_csv_encodings(path, encodings=["utf-8", "ascii"], **kwargs): - """ - Try to read file with the provided encodings. - - Args: - path (str): Path to file. - encodings (list): List of encodings to try. - **kwargs: Additional keyword arguments. - """ - idx = 0 - while idx < len(encodings): - e = encodings[idx] - try: - t = _pd.read_csv(path, encoding=e, **kwargs) - return t - except UnicodeDecodeError: - pass - idx = idx + 1 - _LOGGER.warning(f"Could not read the log file '{path}' with encodings '{encodings}'") - - -def _read_tsv_to_json(path): - """ - Read a tsv file to a JSON formatted string. - - Args: - path (str): To file path. - - Returns: - str: JSON formatted string. - """ - assert os.path.exists(path), "The file '{}' does not exist".format(path) - _LOGGER.debug("Reading TSV from '{}'".format(path)) - df = _pd.read_csv(path, sep="\t", index_col=False, header=None) - return df.to_json() - - def fetch_pipeline_results( project, sample_name=None, @@ -1453,10 +1415,21 @@ def _warn(what, e, sn): 0 ] # Assumes the profile file will be in status dir assert os.path.exists(profile), FileNotFoundError(f"Not found: {profile}") - df = _pd.read_csv(profile, sep="\t", comment="#", names=PROFILE_COLNAMES) - df["runtime"] = _pd.to_timedelta(df["runtime"]) - times.append(_get_runtime(df)) - mems.append(_get_maxmem(df)) + rows = [] + with open(profile) as fh: + reader = csv.reader(fh, delimiter="\t") + for row in reader: + line = row[0].strip() if row else "" + if not line or line.startswith("#"): + continue + if len(row) < len(PROFILE_COLNAMES): + continue + rows.append(dict(zip(PROFILE_COLNAMES, row))) + for r in rows: + r["runtime"] = parse_timedelta(r["runtime"]) + r["mem"] = float(r["mem"]) + times.append(_get_runtime(rows)) + mems.append(_get_maxmem(rows)) except Exception as e: _warn("profile", e, sample) times.append(NO_DATA_PLACEHOLDER) @@ -1496,35 +1469,20 @@ def create_glossary_table(project): return render_jinja_template("glossary_table.html", get_jinja_env(), template_vars) -def _get_maxmem(profile: _pd.DataFrame) -> str: - """ - Get current peak memory. - - Args: - profile (pandas.DataFrame): A data frame representing the current profile.tsv - for a sample. - - Returns: - str: Max memory. - """ - return f"{str(max(profile['mem']) if not profile['mem'].empty else 0)} GB" - +def _get_maxmem(rows: list[dict]) -> str: + """Get peak memory across all profile rows.""" + if not rows: + return "0 GB" + return f"{max(r['mem'] for r in rows)} GB" -def _get_runtime(profile_df: _pd.DataFrame) -> str: - """ - Collect the unique and last duplicated runtimes, sum them and then return in str format. - Args: - profile_df (pandas.DataFrame): A data frame representing the current profile.tsv - for a sample. - - Returns: - str: Sum of runtimes. - """ - unique_df = profile_df[~profile_df.duplicated("cid", keep="last").values] - return str( - timedelta(seconds=sum(unique_df["runtime"].apply(lambda x: x.total_seconds()))) - ).split(".")[0] +def _get_runtime(rows: list[dict]) -> str: + """Sum unique command runtimes, deduplicating reruns by cid (keeping last).""" + seen = {} + for r in rows: + seen[r["cid"]] = r + total = sum(r["runtime"].total_seconds() for r in seen.values()) + return str(timedelta(seconds=total)).split(".")[0] def get_file_for_table(prj, pipeline_name: str, appendix=None, directory=None) -> str: diff --git a/pyproject.toml b/pyproject.toml index a70943dd..1b7fa1f5 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "pipestat" -version = "0.13.0" +version = "0.13.1" description = "A pipeline results reporter" readme = "README.md" license = "BSD-2-Clause" @@ -27,9 +27,8 @@ dependencies = [ "jsonschema", "logmuse>=0.2.5", "pyyaml", - "ubiquerg>=0.8.0", + "ubiquerg>=0.9.1", "yacman>=0.9.5", - "pandas", "eido", "jinja2", ] diff --git a/tests/test_profile_parsing.py b/tests/test_profile_parsing.py new file mode 100644 index 00000000..a1cecef7 --- /dev/null +++ b/tests/test_profile_parsing.py @@ -0,0 +1,71 @@ +"""Tests for profile-parsing functions in reports.py.""" + +from ubiquerg import parse_timedelta + +from pipestat.const import PROFILE_COLNAMES +from pipestat.reports import _get_maxmem, _get_runtime + + +def _make_profile_rows(raw_rows: list[list]) -> list[dict]: + """Build a list of dicts mimicking csv-parsed profile rows.""" + rows = [] + for raw in raw_rows: + r = dict(zip(PROFILE_COLNAMES, raw)) + r["runtime"] = parse_timedelta(r["runtime"]) + r["mem"] = float(r["mem"]) + rows.append(r) + return rows + + +class TestGetRuntime: + def test_basic_sum(self): + rows = _make_profile_rows( + [ + ["1", "abc", "1", "0:00:10", " 100.0", "cmd1", "lock.1"], + ["1", "abc", "2", "0:00:20", " 200.0", "cmd2", "lock.2"], + ] + ) + assert _get_runtime(rows) == "0:00:30" + + def test_dedup_keeps_last(self): + rows = _make_profile_rows( + [ + ["1", "abc", "1", "0:00:05", " 100.0", "cmd1", "lock.1"], + ["1", "abc", "2", "0:01:30", " 200.0", "cmd2", "lock.2"], + ["1", "abc", "2", "0:02:00", " 300.0", "cmd2", "lock.2"], + ] + ) + assert _get_runtime(rows) == "0:02:05" + + def test_single_row(self): + rows = _make_profile_rows( + [ + ["1", "abc", "1", "1:00:00", " 512.0", "cmd1", "lock.1"], + ] + ) + assert _get_runtime(rows) == "1:00:00" + + +class TestGetMaxmem: + def test_returns_max(self): + rows = _make_profile_rows( + [ + ["1", "abc", "1", "0:00:10", " 128.5", "cmd1", "lock.1"], + ["1", "abc", "2", "0:00:20", " 512.0", "cmd2", "lock.2"], + ["1", "abc", "3", "0:00:30", " 256.0", "cmd3", "lock.3"], + ] + ) + result = _get_maxmem(rows) + assert "512.0" in result + assert "GB" in result + + def test_single_row(self): + rows = _make_profile_rows( + [ + ["1", "abc", "1", "0:00:10", " 64.25", "cmd1", "lock.1"], + ] + ) + assert "64.25" in _get_maxmem(rows) + + def test_empty_rows(self): + assert _get_maxmem([]) == "0 GB" From eb62178b48c87bd9397a3adbb72c94ecdffa35de Mon Sep 17 00:00:00 2001 From: nsheff Date: Sat, 28 Feb 2026 08:04:12 -0500 Subject: [PATCH 5/7] lint --- pipestat/backends/db_backend/__init__.py | 6 +++--- pipestat/backends/pephub_backend/__init__.py | 2 +- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/pipestat/backends/db_backend/__init__.py b/pipestat/backends/db_backend/__init__.py index f119e474..24b60ed5 100644 --- a/pipestat/backends/db_backend/__init__.py +++ b/pipestat/backends/db_backend/__init__.py @@ -1,3 +1,3 @@ -from .db_helpers import construct_db_url -from .db_parsed_schema import ParsedSchemaDB -from .dbbackend import DBBackend +from .db_helpers import construct_db_url as construct_db_url +from .db_parsed_schema import ParsedSchemaDB as ParsedSchemaDB +from .dbbackend import DBBackend as DBBackend diff --git a/pipestat/backends/pephub_backend/__init__.py b/pipestat/backends/pephub_backend/__init__.py index 18803b75..cfc8cb27 100644 --- a/pipestat/backends/pephub_backend/__init__.py +++ b/pipestat/backends/pephub_backend/__init__.py @@ -1 +1 @@ -from .pephubbackend import PEPHUBBACKEND +from .pephubbackend import PEPHUBBACKEND as PEPHUBBACKEND From fd750343eb187ffad214ee2bb6e8b6ec817dd3c6 Mon Sep 17 00:00:00 2001 From: nsheff Date: Thu, 5 Mar 2026 14:21:54 -0500 Subject: [PATCH 6/7] fix template path bug --- pipestat/backends/file_backend/filebackend.py | 16 ++++++++++---- tests/test_multi_result_files.py | 22 +++++++++++++++++++ 2 files changed, 34 insertions(+), 4 deletions(-) diff --git a/pipestat/backends/file_backend/filebackend.py b/pipestat/backends/file_backend/filebackend.py index 7beeb2f1..65e73fe0 100644 --- a/pipestat/backends/file_backend/filebackend.py +++ b/pipestat/backends/file_backend/filebackend.py @@ -79,8 +79,8 @@ def determine_results_file(self) -> None: """ if "{record_identifier}" in self.results_file_path: - # In the special case where the user wants to use {record_identifier} in file path - pass + # Template path: _data will be initialized per-record when resolved + self._data = None else: if not os.path.exists(self.results_file_path): _LOGGER.debug( @@ -105,6 +105,8 @@ def check_record_exists( bool: Whether the record exists in the table. """ + if self._data is None: + return False return ( self.pipeline_name in self._data and record_identifier in self._data[self.pipeline_name][self.pipeline_type] @@ -148,8 +150,12 @@ def count_records(self) -> int: Returns: int: Number of records. """ - - return len(self._data[self.pipeline_name][self.pipeline_type]) + if self._data is None: + return 0 + try: + return len(self._data[self.pipeline_name][self.pipeline_type]) + except (KeyError, TypeError): + return 0 def get_flag_file( self, record_identifier: Optional[str] = None @@ -243,6 +249,8 @@ def list_results( """ record_identifier = record_identifier or self.record_identifier + if self._data is None: + return [] try: results = list( self._data[self.pipeline_name][self.pipeline_type][record_identifier].keys() diff --git a/tests/test_multi_result_files.py b/tests/test_multi_result_files.py index d7ad0f07..37601546 100644 --- a/tests/test_multi_result_files.py +++ b/tests/test_multi_result_files.py @@ -83,3 +83,25 @@ def test_multi_results_summarize( os.path.join(temp_dir, "aggregate_results.yaml") ) assert r_id in data[psm.pipeline_name][psm.pipeline_type].keys() + + @pytest.mark.parametrize("backend", ["file"]) + def test_template_path_without_record_identifier( + self, + config_file_path, + results_file_path, + recursive_schema_file_path, + backend, + range_values, + ): + """PSM created with template path and no record_identifier can check status and list results.""" + with TemporaryDirectory() as temp_dir: + results_file_path = os.path.join(temp_dir, "{record_identifier}/results.yaml") + psm = SamplePipestatManager( + results_file_path=results_file_path, + schema_path=recursive_schema_file_path, + ) + # These are what looper calls on a PSM without record_identifier + assert psm.count_records() == 0 + r_id = range_values[0][0] + assert psm.backend.check_record_exists(record_identifier=r_id) is False + assert psm.backend.list_results(record_identifier=r_id) == [] From b0166026e2763f9c2e19633645932ff89ff67714 Mon Sep 17 00:00:00 2001 From: nsheff Date: Thu, 5 Mar 2026 20:38:22 -0500 Subject: [PATCH 7/7] update yacman req --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 1b7fa1f5..d6faf606 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -28,7 +28,7 @@ dependencies = [ "logmuse>=0.2.5", "pyyaml", "ubiquerg>=0.9.1", - "yacman>=0.9.5", + "yacman>=1.0.0", "eido", "jinja2", ]