From 5ff1cfe42a8e97f6cfe4e5aae61fc7c616617eb1 Mon Sep 17 00:00:00 2001 From: Ladme Date: Thu, 3 Sep 2026 16:45:58 +0200 Subject: [PATCH 01/19] feat/fix: number of free GPUs no longer relevant for node state --- CHANGELOG.md | 6 ++++++ src/qq_lib/nodes/presenter.py | 2 +- tests/nodes/test_nodes_presenter.py | 1 + 3 files changed, 8 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index cb02251..04d3665 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,12 @@ - Fixed build for IT4Innovation's Karolina. +## Version 0.13.0 + +- Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. + +--- + ## Version 0.12.1 - Fixed a bug where `qq clear` would report incorrect number of files excluded from clearing. diff --git a/src/qq_lib/nodes/presenter.py b/src/qq_lib/nodes/presenter.py index 0c93a4b..e5d7b71 100644 --- a/src/qq_lib/nodes/presenter.py +++ b/src/qq_lib/nodes/presenter.py @@ -662,7 +662,7 @@ def _format_state_mark( style = CFG.nodes_presenter.unavailable_node_style elif free_cpus == total_cpus and free_gpus == total_gpus: style = CFG.nodes_presenter.free_node_style - elif free_cpus != 0 or free_gpus != 0: + elif free_cpus != 0: style = CFG.nodes_presenter.part_free_node_style else: style = CFG.nodes_presenter.busy_node_style diff --git a/tests/nodes/test_nodes_presenter.py b/tests/nodes/test_nodes_presenter.py index 2404978..942412a 100644 --- a/tests/nodes/test_nodes_presenter.py +++ b/tests/nodes/test_nodes_presenter.py @@ -725,6 +725,7 @@ def test_nodes_presenter_format_properties_section_returns_expected_text(): (8, 8, 1, 1, True, CFG.nodes_presenter.free_node_style), (4, 8, 0, 1, True, CFG.nodes_presenter.part_free_node_style), (0, 8, 0, 1, True, CFG.nodes_presenter.busy_node_style), + (0, 8, 2, 2, True, CFG.nodes_presenter.busy_node_style), ], ) def test_nodes_presenter_format_state_mark_returns_correct_style( From 011a28f0dad518becff697fe0a3269a26d5645c0 Mon Sep 17 00:00:00 2001 From: Ladme Date: Thu, 3 Sep 2026 16:54:20 +0200 Subject: [PATCH 02/19] docs/feat: clarify that excluded files are copied to input directory --- CHANGELOG.md | 1 + src/qq_lib/submit/cli.py | 1 + 2 files changed, 2 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 04d3665..f11fe3a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,7 @@ ## Version 0.13.0 - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. +- Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. --- diff --git a/src/qq_lib/submit/cli.py b/src/qq_lib/submit/cli.py index 955edf0..db4ebce 100644 --- a/src/qq_lib/submit/cli.py +++ b/src/qq_lib/submit/cli.py @@ -124,6 +124,7 @@ def complete_script( help=( f"Colon-, comma-, or space-separated list of files or directories that should {click.style('not', bold=True)} be copied to the working directory.\n" "Paths to files and directories to exclude must be relative to the input directory.\n" + f"Excluded files and directories that the job creates in the working directory {click.style('will', bold=True)} be copied back to the input directory.\n" ), ) @optgroup.option( From c1b75a516d8be6977e8d3a4a8c8fd755614beb65 Mon Sep 17 00:00:00 2001 From: Ladme Date: Thu, 3 Sep 2026 18:36:58 +0200 Subject: [PATCH 03/19] feat: new env vars for loop jobs --- CHANGELOG.md | 8 ++++++++ src/qq_lib/core/config.py | 6 ++++++ src/qq_lib/submit/submitter.py | 27 +++++++++++++++++++++++++++ tests/submit/test_submit_submitter.py | 20 +++++++++++++++++++- 4 files changed, 60 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f11fe3a..5560f80 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,14 @@ ## Version 0.13.0 +### New environment variables for loop jobs + +- Loop jobs now expose three additional environment variables: + - `QQ_LOOP_NEXT` which specifies the index of the next loop cycle. + - `QQ_ARCHIVE_CURRENT` and `QQ_ARCHIVE_NEXT` which specify strings expected in files archived by qq for the current and the next loop cycles, respectively. If the archive format is not a printf pattern, the values of these variables are empty strings. + +### Other changes + - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. diff --git a/src/qq_lib/core/config.py b/src/qq_lib/core/config.py index a2b28b1..640ac1b 100644 --- a/src/qq_lib/core/config.py +++ b/src/qq_lib/core/config.py @@ -58,10 +58,16 @@ class EnvironmentVariables: batch_system: str = "QQ_BATCH_SYSTEM" # Current loop-cycle index. loop_current: str = "QQ_LOOP_CURRENT" + # Loop-cycle index for the next cycle. + loop_next: str = "QQ_LOOP_NEXT" # Starting loop-cycle index. loop_start: str = "QQ_LOOP_START" # Final loop-cycle index. loop_end: str = "QQ_LOOP_END" + # Archive pattern used for the current cycle. + archive_current: str = "QQ_ARCHIVE_CURRENT" + # Archive pattern used for the next cycle. + archive_next: str = "QQ_ARCHIVE_NEXT" # Non-resubmit flag returned by a job script. no_resubmit: str = "QQ_NO_RESUBMIT" # Archive filename pattern. diff --git a/src/qq_lib/submit/submitter.py b/src/qq_lib/submit/submitter.py index 13d79dd..6315bf8 100644 --- a/src/qq_lib/submit/submitter.py +++ b/src/qq_lib/submit/submitter.py @@ -14,6 +14,7 @@ construct_loop_job_name, get_info_file, hhmmss_to_duration, + is_printf_pattern, ) from qq_lib.core.config import CFG from qq_lib.core.error import QQError @@ -381,9 +382,16 @@ def _create_env_vars_dict(self) -> dict[str, str]: # loop job-specific environment variables if self._loop_info: env_vars[CFG.env_vars.loop_current] = str(self._loop_info.current) + env_vars[CFG.env_vars.loop_next] = str(self._loop_info.current + 1) env_vars[CFG.env_vars.loop_start] = str(self._loop_info.start) env_vars[CFG.env_vars.loop_end] = str(self._loop_info.end) env_vars[CFG.env_vars.archive_format] = self._loop_info.archive_format + env_vars[CFG.env_vars.archive_current] = self._make_pattern( + self._loop_info.archive_format, self._loop_info.current + ) + env_vars[CFG.env_vars.archive_next] = self._make_pattern( + self._loop_info.archive_format, self._loop_info.current + 1 + ) # loop job- or continuous job-specific environment variables if self._job_type in [JobType.LOOP, JobType.CONTINUOUS]: @@ -391,6 +399,25 @@ def _create_env_vars_dict(self) -> dict[str, str]: return env_vars + @staticmethod + def _make_pattern(archive_format: str, cycle: int) -> str: + """ + Create a pattern for archived files in the specified cycle. + + If the archive_format is not a printf pattern, returns an empty string. + + Args: + archive_format (str): The provided archive format. + cycle (int): Cycle number to use. + + Returns: + str: The pattern or an empty string if the archive format is not a printf pattern. + """ + if is_printf_pattern(archive_format): + return archive_format % cycle + + return "" + def _has_valid_shebang(self, script: Path) -> bool: """ Verify that the script has a valid shebang for qq run. diff --git a/tests/submit/test_submit_submitter.py b/tests/submit/test_submit_submitter.py index 6e86529..e49a75d 100644 --- a/tests/submit/test_submit_submitter.py +++ b/tests/submit/test_submit_submitter.py @@ -317,7 +317,7 @@ class DummyLoop: current = 1 start = 0 end = 5 - archive_format = "zip" + archive_format = "job%02d" submitter = Submitter.__new__(Submitter) submitter._info_file = tmp_path / "job.qqinfo" @@ -340,10 +340,13 @@ class DummyLoop: assert env[CFG.env_vars.input_dir] == str(submitter._input_dir) assert env[CFG.env_vars.loop_current] == str(DummyLoop.current) + assert env[CFG.env_vars.loop_next] == str(DummyLoop.current + 1) assert env[CFG.env_vars.loop_start] == str(DummyLoop.start) assert env[CFG.env_vars.loop_end] == str(DummyLoop.end) assert env[CFG.env_vars.archive_format] == DummyLoop.archive_format assert env[CFG.env_vars.no_resubmit] == str(CFG.exit_codes.qq_run_no_resubmit) + assert env[CFG.env_vars.archive_current] == "job01" + assert env[CFG.env_vars.archive_next] == "job02" if debug_mode: assert env[CFG.env_vars.debug_mode] == "true" else: @@ -719,3 +722,18 @@ def test_submitter_submit(tmp_path): ) mock_info_instance.to_file.assert_called_once_with(submitter._info_file) assert result == "jobid123" + + +@pytest.mark.parametrize( + "input_pattern, cycle, expected", + [ + ("job%04d", 1, "job0001"), + ("md%03d", 643, "md643"), + ("job%2d", 5, "job 5"), + ("^abc\\d+$", 7, ""), + ("file\\d{3}", 123, ""), + ], +) +def test_submitter_make_pattern(input_pattern, cycle, expected): + result = Submitter._make_pattern(input_pattern, cycle) + assert result == expected From 0946637c112c580a87f575951c87855cf81699ad Mon Sep 17 00:00:00 2001 From: Ladme Date: Thu, 3 Sep 2026 18:37:49 +0200 Subject: [PATCH 04/19] docs: missing changelog entry for qq v0.12.2 --- CHANGELOG.md | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5560f80..5c19ed9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,6 +21,10 @@ --- +## Version 0.12.2 + +- Fixed build for IT4Innovation's Karolina. + ## Version 0.12.1 - Fixed a bug where `qq clear` would report incorrect number of files excluded from clearing. From 8210ed23d61ef05a1bf8b2eb5e5cd0b672195d6e Mon Sep 17 00:00:00 2001 From: Ladme Date: Thu, 3 Sep 2026 18:43:02 +0200 Subject: [PATCH 05/19] feat: more explicit info about when qq killall will be removed --- src/qq_lib/killall/cli.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/src/qq_lib/killall/cli.py b/src/qq_lib/killall/cli.py index 9e2e8e2..834c94e 100644 --- a/src/qq_lib/killall/cli.py +++ b/src/qq_lib/killall/cli.py @@ -51,7 +51,8 @@ def killall( ) -> NoReturn: try: logger.warning( - "This command is deprecated! It will be removed in a future version of qq. Use `qq kill --all` instead." + "This command is deprecated! It will be removed in qq v0.14. Use `qq kill --all` instead.", + extra={"markup": False, "highlighter": None}, ) BatchSystem = BatchInterface.from_env_var_or_guess() @@ -91,7 +92,8 @@ def killall( logger.info("Operation aborted.") logger.warning( - "This command is deprecated! It will be removed in a future version of qq. Use `qq kill --all` instead." + "This command is deprecated! It will be removed in qq v0.14. Use `qq kill --all` instead.", + extra={"markup": False, "highlighter": None}, ) sys.exit(0) except QQError as e: From 5c9aaa49e47f524f5077606f12aebb9dc9e42791 Mon Sep 17 00:00:00 2001 From: Ladme Date: Thu, 3 Sep 2026 18:51:06 +0200 Subject: [PATCH 06/19] docs: update obsolete comments --- src/qq_lib/sync/syncer.py | 2 +- src/qq_lib/wipe/wiper.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/src/qq_lib/sync/syncer.py b/src/qq_lib/sync/syncer.py index 33a5bb8..201393d 100644 --- a/src/qq_lib/sync/syncer.py +++ b/src/qq_lib/sync/syncer.py @@ -63,7 +63,7 @@ def sync(self, files: list[str] | None = None) -> None: ) # hint for type checker - # work_dir and main_node must be set - we check that in self.hasDestination + # work_dir and main_node must be set - we check that in self.has_destination assert self._work_dir and self._main_node if files: diff --git a/src/qq_lib/wipe/wiper.py b/src/qq_lib/wipe/wiper.py index 2ed0961..78a0f16 100644 --- a/src/qq_lib/wipe/wiper.py +++ b/src/qq_lib/wipe/wiper.py @@ -66,7 +66,7 @@ def wipe(self) -> str: ) # hint for type checker - # work_dir and main_node must be set - we check that in self.hasDestination + # work_dir and main_node must be set - we check that in self.has_destination assert self._work_dir and self._main_node # we cannot delete the input directory even if the `--force` flag is used From 76f8d5e2f56c6ee3e0bac37e97416b6412f31cd4 Mon Sep 17 00:00:00 2001 From: Ladme Date: Sat, 5 Sep 2026 10:24:14 +0200 Subject: [PATCH 07/19] glob patterns for include and exclude --- src/qq_lib/core/common.py | 74 +++++- src/qq_lib/respawn/respawner.py | 4 +- src/qq_lib/resubmit/resubmitter.py | 4 +- src/qq_lib/submit/cli.py | 8 +- src/qq_lib/submit/factory.py | 14 +- src/qq_lib/submit/parser.py | 15 +- src/qq_lib/submit/submitter.py | 18 +- tests/core/test_common.py | 256 ++++++++++++++++++-- tests/respawn/test_respawn_respawner.py | 4 +- tests/resubmit/test_resubmit_resubmitter.py | 4 +- tests/submit/test_submit_factory.py | 16 +- tests/submit/test_submit_parser.py | 18 +- tests/submit/test_submit_submitter.py | 61 ++++- 13 files changed, 409 insertions(+), 87 deletions(-) diff --git a/src/qq_lib/core/common.py b/src/qq_lib/core/common.py index 454e0a9..e431e37 100644 --- a/src/qq_lib/core/common.py +++ b/src/qq_lib/core/common.py @@ -9,9 +9,11 @@ """ import re +from collections.abc import Iterable from datetime import timedelta from functools import lru_cache from pathlib import Path +from typing import Final import readchar import yaml @@ -28,6 +30,9 @@ logger = get_logger(__name__) +_GLOB_MAGIC: Final[tuple[str, ...]] = ("*", "?", "[") + + @lru_cache(maxsize=1) def load_yaml_dumper() -> type[yaml.Dumper]: """Return the fastest available YAML dumper (CDumper if possible).""" @@ -605,25 +610,25 @@ def is_printf_pattern(pattern: str) -> bool: return bool(re.search(r"%0?\d*d", pattern)) -def split_files_list(string: str | None) -> list[Path]: +def split_string_list(string: str | None) -> list[str]: """ - Split a string containing multiple file paths into a list of relative Path objects. + Split a string containing multiple substrings into a list of strings. - The string can contain file paths separated by colons (:), commas (,), or + The string can contain substrings separated by colons (:), commas (,), or any whitespace characters (space, tab, newline). Args: - string (str | None): The string containing file paths. If None or empty, + string (str | None): The string containing substrings. If None or empty, an empty list is returned. Returns: - list[Path]: A list of Path objects corresponding to the individual - relative file paths in the input string. + list[str]: A list of strings corresponding to the individual + substrings in the input string. """ if not string: return [] - return [Path(f) for f in re.split(r"[:,\s]+", string)] + return list(re.split(r"[:,\s]+", string)) def to_snake_case(s: str) -> str: @@ -766,3 +771,58 @@ def default_resubmit_from_hosts() -> str: # if no batch system is available except QQError: return "??? (no batch system detected)" + + +def expand_paths(patterns: Iterable[str], directory: Path) -> list[Path]: + """ + Convert paths to absolute paths, expanding glob patterns along the way. + + Args: + patterns (Iterable[str]): Paths or glob patterns, relative or absolute. + directory (Path): Directory to resolve relative patterns against. + + Returns: + list[Path]: Absolute paths, with glob patterns expanded. + + Raises: + QQError: If a pattern cannot be expanded. + """ + expanded: list[Path] = [] + for pattern in patterns: + expanded.extend(expand_pattern(pattern, directory)) + + return list(dict.fromkeys(expanded)) + + +def expand_pattern(pattern: str, directory: Path) -> list[Path]: + """ + Expand a single path or glob pattern into absolute paths. + + Args: + pattern (str): Path or glob pattern, relative or absolute. + directory (Path): Directory to resolve a relative pattern against. + + Returns: + list[Path]: Absolute paths matching the pattern, sorted. A pattern + without glob metacharacters yields exactly one path. + + Raises: + QQError: If the pattern is not a valid glob pattern or the file system + could not be searched. + """ + path = Path(pattern) + + if not any(char in pattern for char in _GLOB_MAGIC): + return [path if path.is_absolute() else directory / path] + + if path.is_absolute(): + anchor = Path(path.anchor) + relative = path.relative_to(path.anchor) + else: + anchor = directory + relative = path + + try: + return sorted(anchor.glob(str(relative))) + except Exception as e: + raise QQError(f"Could not expand pattern '{pattern}': {e}.") from e diff --git a/src/qq_lib/respawn/respawner.py b/src/qq_lib/respawn/respawner.py index 39fa3be..6cc2dbb 100644 --- a/src/qq_lib/respawn/respawner.py +++ b/src/qq_lib/respawn/respawner.py @@ -94,8 +94,8 @@ def _build_submitter(self, informer: Informer) -> Submitter: job_type=informer.info.job_type, resources=informer.info.resources, loop_info=loop_info, - exclude=informer.info.excluded_files, - include=informer.info.included_files, + exclude=[str(x) for x in informer.info.excluded_files], + include=[str(x) for x in informer.info.included_files], # we need to remove dependencies that are no longer present in the batch system depend=filter_dependencies(informer.batch_system, informer.info.depend), transfer_mode=informer.info.transfer_mode, diff --git a/src/qq_lib/resubmit/resubmitter.py b/src/qq_lib/resubmit/resubmitter.py index 2f5033c..15ee578 100644 --- a/src/qq_lib/resubmit/resubmitter.py +++ b/src/qq_lib/resubmit/resubmitter.py @@ -85,8 +85,8 @@ def _build_submitter(informer: Informer, input_dir: Path) -> Submitter: job_type=informer.info.job_type, resources=informer.info.resources, loop_info=informer.info.loop_info, - exclude=informer.info.excluded_files, - include=informer.info.included_files, + exclude=[str(x) for x in informer.info.excluded_files], + include=[str(x) for x in informer.info.included_files], depend=[Depend(type=DependType.AFTER_SUCCESS, jobs=[informer.info.job_id])], transfer_mode=informer.info.transfer_mode, server=informer.info.server, diff --git a/src/qq_lib/submit/cli.py b/src/qq_lib/submit/cli.py index db4ebce..a3770c1 100644 --- a/src/qq_lib/submit/cli.py +++ b/src/qq_lib/submit/cli.py @@ -123,7 +123,8 @@ def complete_script( default=None, help=( f"Colon-, comma-, or space-separated list of files or directories that should {click.style('not', bold=True)} be copied to the working directory.\n" - "Paths to files and directories to exclude must be relative to the input directory.\n" + "Paths to files and directories to exclude must be absolute or relative to the input directory.\n" + "You can use glob patterns to match multiple files or directories.\n" f"Excluded files and directories that the job creates in the working directory {click.style('will', bold=True)} be copied back to the input directory.\n" ), ) @@ -135,8 +136,9 @@ def complete_script( f"Colon-, comma-, or space-separated list of files or directories to copy into the working directory " f"in addition to the input directory contents.\n" f"These files are {click.style('not', bold=True)} copied back after job completion. " - f"Paths must be absolute or relative to the input directory. " - f"Ignored if the input directory is used as the working directory.\n" + f"Paths must be absolute or relative to the input directory.\n" + "You can use glob patterns to match multiple files or directories.\n" + f"This option is ignored if the input directory is used as the working directory.\n" ), ) @optgroup.option( diff --git a/src/qq_lib/submit/factory.py b/src/qq_lib/submit/factory.py index 6902836..2ad969d 100644 --- a/src/qq_lib/submit/factory.py +++ b/src/qq_lib/submit/factory.py @@ -5,7 +5,7 @@ from pathlib import Path from qq_lib.batch.interface import AnyBatchClass, BatchInterface -from qq_lib.core.common import split_files_list, translate_server +from qq_lib.core.common import split_string_list, translate_server from qq_lib.core.config import CFG from qq_lib.core.error import QQError from qq_lib.core.logger import get_logger @@ -233,7 +233,7 @@ def _print_warning_if_resubmit_from_defined(self, job_type: JobType) -> None: f"Option 'resubmit_from' is specified but job type is '{str(job_type)}', not 'loop' or 'continuous' - 'resubmit_from' will be ignored." ) - def _get_exclude(self) -> list[Path]: + def _get_exclude(self) -> list[str]: """ Determine the files to exclude from being copied to the job's working directory. @@ -244,13 +244,13 @@ def _get_exclude(self) -> list[Path]: The lists are NOT merged. Returns: - list[Path]: List of relative file paths to exclude. + list[str]: List of files or glob patterns to exclude. """ return ( - split_files_list(self._kwargs.get("exclude")) or self._parser.get_exclude() + split_string_list(self._kwargs.get("exclude")) or self._parser.get_exclude() ) - def _get_include(self) -> list[Path]: + def _get_include(self) -> list[str]: """ Determine the files to explicitly copy to the job's working directory. @@ -261,10 +261,10 @@ def _get_include(self) -> list[Path]: The lists are NOT merged. Returns: - list[Path]: List of file paths to include. + list[str]: List of files or glob patterns to include. """ return ( - split_files_list(self._kwargs.get("include")) or self._parser.get_include() + split_string_list(self._kwargs.get("include")) or self._parser.get_include() ) def _get_depend(self) -> list[Depend]: diff --git a/src/qq_lib/submit/parser.py b/src/qq_lib/submit/parser.py index a678bcb..4c94c1c 100644 --- a/src/qq_lib/submit/parser.py +++ b/src/qq_lib/submit/parser.py @@ -9,7 +9,7 @@ from click_option_group import GroupedOption from qq_lib.batch.interface import BatchInterface -from qq_lib.core.common import split_files_list, to_snake_case +from qq_lib.core.common import split_string_list, to_snake_case from qq_lib.core.error import QQError from qq_lib.core.logger import get_logger from qq_lib.properties.depend import Depend @@ -158,27 +158,28 @@ def get_resources(self) -> Resources: # only select fields that are part of Resources return Resources(**{k: v for k, v in self._options.items() if k in field_names}) # ty: ignore[invalid-argument-type] - def get_exclude(self) -> list[Path]: + def get_exclude(self) -> list[str]: """ Determine the files to exclude from being copied to the job's working directory. Returns: - list[Path]: List of excluded file paths. Returns an empty list if none specified. + list[str]: List of excluded files or glob patterns. + Returns an empty list if none specified. """ if (exclude := self._options.get("exclude")) is not None: - return split_files_list(str(exclude)) + return split_string_list(str(exclude)) return [] - def get_include(self) -> list[Path]: + def get_include(self) -> list[str]: """ Determine the files to explicitly copy to the job's working directory. Returns: - list[Path]: List of included file paths. Returns an empty list if none specified. + list[Path]: List of included files or glob patterns. Returns an empty list if none specified. """ if (include := self._options.get("include")) is not None: - return split_files_list(str(include)) + return split_string_list(str(include)) return [] diff --git a/src/qq_lib/submit/submitter.py b/src/qq_lib/submit/submitter.py index 6315bf8..cd03525 100644 --- a/src/qq_lib/submit/submitter.py +++ b/src/qq_lib/submit/submitter.py @@ -12,6 +12,7 @@ from qq_lib.core.common import ( construct_info_file_path, construct_loop_job_name, + expand_paths, get_info_file, hhmmss_to_duration, is_printf_pattern, @@ -57,8 +58,8 @@ def __init__( job_type: JobType, resources: Resources, loop_info: LoopInfo | None = None, - exclude: list[Path] | None = None, - include: list[Path] | None = None, + exclude: list[str] | None = None, + include: list[str] | None = None, depend: list[Depend] | None = None, transfer_mode: list[TransferMode] | None = None, server: str | None = None, @@ -77,9 +78,9 @@ def __init__( job_type (JobType): Type of the job to submit (e.g. standard, loop). resources (Resources): Job resource requirements (e.g., CPUs, memory, walltime). loop_info (LoopInfo | None): Optional information for loop jobs. Pass None if not applicable. - exclude (list[Path] | None): Optional list of files which should not be copied to the working directory. - Paths are provided relative to the input directory. - include (list[Path] | None): Optional list of files which should be copied to the working directory + exclude (list[str] | None): Optional list of files or glob patterns which should not be copied to the working directory. + Paths are provided relative to the input directory or absolute. + include (list[str] | None): Optional list of files or glob patterns which should be copied to the working directory even though they are not part of the job's input directory. Paths are provided either absolute or relative to the input directory. depend (list[Depend] | None): Optional list of job dependencies. @@ -108,11 +109,8 @@ def __init__( self._job_name = self._construct_job_name() self._info_file = construct_info_file_path(self._input_dir, self._job_name) self._resources = resources - # convert relative paths to absolute paths by prepending the input dir path - self._exclude = [self._input_dir / e for e in (exclude or [])] - self._include = [ - i if i.is_absolute() else self._input_dir / i for i in (include or []) - ] + self._exclude = expand_paths(exclude or [], self._input_dir) + self._include = expand_paths(include or [], self._input_dir) self._depend = depend or [] self._transfer_mode = transfer_mode or TransferMode.multi_from_str( CFG.transfer_files_options.default_transfer_mode diff --git a/tests/core/test_common.py b/tests/core/test_common.py index 669f303..c0584b6 100644 --- a/tests/core/test_common.py +++ b/tests/core/test_common.py @@ -22,6 +22,8 @@ convert_absolute_to_relative, dhhmmss_to_duration, equals_normalized, + expand_paths, + expand_pattern, format_duration, format_duration_wdhhmmss, get_files_with_suffix, @@ -37,7 +39,7 @@ load_yaml_dumper, load_yaml_loader, printf_to_regex, - split_files_list, + split_string_list, to_snake_case, translate_server, wdhms_to_hhmmss, @@ -476,52 +478,63 @@ def test_is_printf_pattern(pattern, expected): assert is_printf_pattern(pattern) == expected -def test_split_files_list_none_or_empty(): +def test_split_string_list_none_or_empty(): # None input - assert split_files_list(None) == [] + assert split_string_list(None) == [] # empty string - assert split_files_list("") == [] + assert split_string_list("") == [] -def test_split_files_list_whitespace(tmp_path): +def test_split_string_list_whitespace(tmp_path): string = ( f"{tmp_path / 'file1.txt'} {tmp_path / 'file2.txt'}\t{tmp_path / 'file3.txt'}" ) expected = [ - Path(tmp_path / "file1.txt"), - Path(tmp_path / "file2.txt"), - Path(tmp_path / "file3.txt"), + str(tmp_path / "file1.txt"), + str(tmp_path / "file2.txt"), + str(tmp_path / "file3.txt"), ] - assert split_files_list(string) == expected + assert split_string_list(string) == expected -def test_split_files_list_commas_and_colons(tmp_path): +def test_split_string_list_commas_and_colons(tmp_path): string = ( f"{tmp_path / 'file1.txt'},{tmp_path / 'file2.txt'}:{tmp_path / 'file3.txt'}" ) expected = [ - Path(tmp_path / "file1.txt"), - Path(tmp_path / "file2.txt"), - Path(tmp_path / "file3.txt"), + str(tmp_path / "file1.txt"), + str(tmp_path / "file2.txt"), + str(tmp_path / "file3.txt"), ] - assert split_files_list(string) == expected + assert split_string_list(string) == expected -def test_split_files_list_mixed_separators(tmp_path): +def test_split_string_list_mixed_separators(tmp_path): string = f"{tmp_path / 'file1.txt'}, {tmp_path / 'file2.txt'}:{tmp_path / 'file3.txt'} {tmp_path / 'file4.txt'}" expected = [ - Path(tmp_path / "file1.txt"), - Path(tmp_path / "file2.txt"), - Path(tmp_path / "file3.txt"), - Path(tmp_path / "file4.txt"), + str(tmp_path / "file1.txt"), + str(tmp_path / "file2.txt"), + str(tmp_path / "file3.txt"), + str(tmp_path / "file4.txt"), ] - assert split_files_list(string) == expected + assert split_string_list(string) == expected -def test_split_files_list_single_file(tmp_path): +def test_split_string_list_mixed_separators_glob_patterns(tmp_path): + string = f"{tmp_path / 'file*.txt'}, {tmp_path / 'file2.txt'}:{tmp_path / 'file??.txt'} {tmp_path / 'file4.txt'}" + expected = [ + str(tmp_path / "file*.txt"), + str(tmp_path / "file2.txt"), + str(tmp_path / "file??.txt"), + str(tmp_path / "file4.txt"), + ] + assert split_string_list(string) == expected + + +def test_split_string_list_single_file(tmp_path): string = str(tmp_path / "single_file.txt") - expected = [Path(tmp_path / "single_file.txt")] - assert split_files_list(string) == expected + expected = [str(tmp_path / "single_file.txt")] + assert split_string_list(string) == expected @pytest.mark.parametrize( @@ -1112,3 +1125,200 @@ def test_translate_server_returns_unknown_value_unchanged(): assert ( translate_server("unknown-server.example.com") == "unknown-server.example.com" ) + + +def _touch(directory: Path, *relative: str) -> None: + """ + Create empty files under a directory, making parent directories as needed. + + Args: + directory (Path): Directory to create the files in. + *relative (str): Paths of the files, relative to `directory`. + """ + for rel in relative: + path = directory / rel + path.parent.mkdir(parents=True, exist_ok=True) + path.touch() + + +def test_expand_paths_empty_input(tmp_path: Path) -> None: + assert expand_paths([], tmp_path) == [] + + +def test_expand_paths_relative_literals_are_prefixed(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt", "sub/b.txt") + + assert expand_paths(["a.txt", "sub/b.txt"], tmp_path) == [ + tmp_path / "a.txt", + tmp_path / "sub" / "b.txt", + ] + + +def test_expand_paths_absolute_literals_are_unchanged(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt") + other = tmp_path / "elsewhere" + other.mkdir() + + assert expand_paths([str(other / "c.txt")], tmp_path) == [other / "c.txt"] + + +def test_expand_paths_keeps_literals_that_do_not_exist( + tmp_path: Path, +) -> None: + assert expand_paths(["missing.txt"], tmp_path) == [tmp_path / "missing.txt"] + + +def test_expand_paths_drops_patterns_that_match_nothing( + tmp_path: Path, +) -> None: + _touch(tmp_path, "a.txt") + + assert expand_paths(["*.log"], tmp_path) == [] + + +def test_expand_paths_mixes_literals_and_patterns(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt", "b.txt") + + assert expand_paths(["missing.dat", "*.txt"], tmp_path) == [ + tmp_path / "missing.dat", + tmp_path / "a.txt", + tmp_path / "b.txt", + ] + + +def test_expand_paths_deduplicates_preserving_order(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt", "b.txt") + + assert expand_paths(["b.txt", "*.txt"], tmp_path) == [ + tmp_path / "b.txt", + tmp_path / "a.txt", + ] + + +def test_expand_paths_deduplicates_relative_and_absolute_forms( + tmp_path: Path, +) -> None: + _touch(tmp_path, "a.txt") + + assert expand_paths(["a.txt", str(tmp_path / "a.txt")], tmp_path) == [ + tmp_path / "a.txt" + ] + + +def test_expand_paths_returns_absolute_paths(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt", "sub/b.txt") + + result = expand_paths(["a.txt", "**/*.txt", "missing.dat"], tmp_path) + + assert result + assert all(path.is_absolute() for path in result) + + +def test_expand_paths_propagates_expansion_errors( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + def _raise(_self: Path, _pattern: str) -> list[Path]: + raise OSError("boom") + + monkeypatch.setattr(Path, "glob", _raise) + + with pytest.raises(QQError): + expand_paths(["*.txt"], tmp_path) + + +def test_expand_pattern_relative_literal(tmp_path: Path) -> None: + assert expand_pattern("a.txt", tmp_path) == [tmp_path / "a.txt"] + + +def test_expand_pattern_absolute_literal(tmp_path: Path) -> None: + other = tmp_path / "elsewhere" / "a.txt" + + assert expand_pattern(str(other), tmp_path) == [other] + + +def test_expand_pattern_literal_does_not_require_existence(tmp_path: Path) -> None: + missing = tmp_path / "nowhere" / "deeply" / "nested.txt" + + assert expand_pattern(str(missing), tmp_path) == [missing] + + +@pytest.mark.parametrize( + ("pattern", "expected"), + [ + ("*.txt", ["a.txt", "ab.txt"]), + ("a?.txt", ["ab.txt"]), + ("[ab].txt", ["a.txt"]), + ], +) +def test_expand_pattern_metacharacters( + tmp_path: Path, pattern: str, expected: list[str] +) -> None: + _touch(tmp_path, "a.txt", "ab.txt") + + assert expand_pattern(pattern, tmp_path) == [tmp_path / name for name in expected] + + +def test_expand_pattern_returns_sorted_matches(tmp_path: Path) -> None: + _touch(tmp_path, "c.txt", "a.txt", "b.txt") + + assert expand_pattern("*.txt", tmp_path) == [ + tmp_path / "a.txt", + tmp_path / "b.txt", + tmp_path / "c.txt", + ] + + +def test_expand_pattern_recursive_pattern(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt", "sub/b.txt", "sub/deeper/c.txt") + + assert expand_pattern("**/*.txt", tmp_path) == [ + tmp_path / "a.txt", + tmp_path / "sub" / "b.txt", + tmp_path / "sub" / "deeper" / "c.txt", + ] + + +def test_expand_pattern_absolute_pattern(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt", "b.log") + elsewhere = tmp_path / "elsewhere" + _touch(elsewhere, "c.txt") + + assert expand_pattern(str(elsewhere / "*.txt"), tmp_path) == [elsewhere / "c.txt"] + + +def test_expand_pattern_ignores_the_directory_for_absolute_patterns( + tmp_path: Path, +) -> None: + _touch(tmp_path, "a.txt") + elsewhere = tmp_path / "elsewhere" + elsewhere.mkdir() + + assert expand_pattern(str(elsewhere / "*.txt"), tmp_path) == [] + + +def test_expand_pattern_matches_directories(tmp_path: Path) -> None: + _touch(tmp_path, "logs.txt") + (tmp_path / "logs").mkdir() + + assert expand_pattern("log*", tmp_path) == [ + tmp_path / "logs", + tmp_path / "logs.txt", + ] + + +def test_expand_pattern_no_matches(tmp_path: Path) -> None: + _touch(tmp_path, "a.txt") + + assert expand_pattern("*.log", tmp_path) == [] + + +def test_expand_pattern_raises_on_glob_failure( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + def _raise(_self: Path, _pattern: str) -> list[Path]: + raise OSError("boom") + + monkeypatch.setattr(Path, "glob", _raise) + + with pytest.raises(QQError, match=r"\*\.txt"): + expand_pattern("*.txt", tmp_path) diff --git a/tests/respawn/test_respawn_respawner.py b/tests/respawn/test_respawn_respawner.py index 576272c..bd49c53 100644 --- a/tests/respawn/test_respawn_respawner.py +++ b/tests/respawn/test_respawn_respawner.py @@ -131,8 +131,8 @@ def test_respawner_build_submitter_creates_submitter_with_correct_params( job_type=informer.info.job_type, resources=informer.info.resources, loop_info=None, - exclude=informer.info.excluded_files, - include=informer.info.included_files, + exclude=[str(x) for x in informer.info.excluded_files], + include=[str(x) for x in informer.info.included_files], depend=dependencies, transfer_mode=informer.info.transfer_mode, server=informer.info.server, diff --git a/tests/resubmit/test_resubmit_resubmitter.py b/tests/resubmit/test_resubmit_resubmitter.py index 27d83a8..00a8a71 100644 --- a/tests/resubmit/test_resubmit_resubmitter.py +++ b/tests/resubmit/test_resubmit_resubmitter.py @@ -52,8 +52,8 @@ def test_resubmitter_build_submitter_creates_submitter_with_correct_params(): job_type=informer.info.job_type, resources=informer.info.resources, loop_info=informer.info.loop_info, - exclude=informer.info.excluded_files, - include=informer.info.included_files, + exclude=[str(x) for x in informer.info.excluded_files], + include=[str(x) for x in informer.info.included_files], depend=[Depend(type=DependType.AFTER_SUCCESS, jobs=["12345"])], transfer_mode=informer.info.transfer_mode, server=informer.info.server, diff --git a/tests/submit/test_submit_factory.py b/tests/submit/test_submit_factory.py index 2031783..7424449 100644 --- a/tests/submit/test_submit_factory.py +++ b/tests/submit/test_submit_factory.py @@ -117,17 +117,17 @@ def test_submitter_factory_get_transfer_mode_from_parser(): def test_submitter_factory_get_exclude_from_command_line(): mock_parser = MagicMock() - parser_excludes = [Path("/tmp/file1"), Path("/tmp/file2")] + parser_excludes = ["/tmp/file1", "/tmp/file2"] mock_parser.get_exclude.return_value = parser_excludes factory = SubmitterFactory.__new__(SubmitterFactory) factory._parser = mock_parser factory._kwargs = {"exclude": "/tmp/file3,/tmp/file4"} - cli_excludes = [Path("/tmp/file3"), Path("/tmp/file4")] + cli_excludes = ["/tmp/file3", "/tmp/file4"] with patch( - "qq_lib.submit.factory.split_files_list", return_value=cli_excludes + "qq_lib.submit.factory.split_string_list", return_value=cli_excludes ) as mock_split: result = factory._get_exclude() @@ -137,7 +137,7 @@ def test_submitter_factory_get_exclude_from_command_line(): def test_submitter_factory_get_exclude_from_parser(): mock_parser = MagicMock() - parser_excludes = [Path("/tmp/file1"), Path("/tmp/file2")] + parser_excludes = ["/tmp/file1", "/tmp/file2"] mock_parser.get_exclude.return_value = parser_excludes factory = SubmitterFactory.__new__(SubmitterFactory) @@ -150,17 +150,17 @@ def test_submitter_factory_get_exclude_from_parser(): def test_submitter_factory_get_include_from_command_line(): mock_parser = MagicMock() - parser_includes = [Path("/tmp/file1"), Path("/tmp/file2")] + parser_includes = ["/tmp/file1", "/tmp/file2"] mock_parser.get_include.return_value = parser_includes factory = SubmitterFactory.__new__(SubmitterFactory) factory._parser = mock_parser factory._kwargs = {"include": "/tmp/file3,/tmp/file4"} - cli_includes = [Path("/tmp/file3"), Path("/tmp/file4")] + cli_includes = ["/tmp/file3", "/tmp/file4"] with patch( - "qq_lib.submit.factory.split_files_list", return_value=cli_includes + "qq_lib.submit.factory.split_string_list", return_value=cli_includes ) as mock_split: result = factory._get_include() @@ -170,7 +170,7 @@ def test_submitter_factory_get_include_from_command_line(): def test_submitter_factory_get_include_from_parser(): mock_parser = MagicMock() - parser_includes = [Path("/tmp/file1"), Path("/tmp/file2")] + parser_includes = ["/tmp/file1", "/tmp/file2"] mock_parser.get_include.return_value = parser_includes factory = SubmitterFactory.__new__(SubmitterFactory) diff --git a/tests/submit/test_submit_parser.py b/tests/submit/test_submit_parser.py index 2e7b5b2..282d06e 100644 --- a/tests/submit/test_submit_parser.py +++ b/tests/submit/test_submit_parser.py @@ -201,14 +201,14 @@ def test_parser_get_exclude_empty_list(): assert result == [] -def test_parser_get_exclude_calls_split_files_list(): +def test_parser_get_exclude_calls_split_string_list(): parser = Parser.__new__(Parser) parser._options = {"exclude": "file1,file2"} mock_split_result = [Path("file1"), Path("file2")] with patch( - "qq_lib.submit.parser.split_files_list", return_value=mock_split_result + "qq_lib.submit.parser.split_string_list", return_value=mock_split_result ) as mock_split: result = parser.get_exclude() @@ -224,14 +224,14 @@ def test_parser_get_include_empty_list(): assert result == [] -def test_parser_get_include_calls_split_files_list_single_numeric_value(): +def test_parser_get_include_calls_split_string_list_single_numeric_value(): parser = Parser.__new__(Parser) parser._options = {"include": 16} - mock_split_result = [Path("16")] + mock_split_result = ["16"] with patch( - "qq_lib.submit.parser.split_files_list", return_value=mock_split_result + "qq_lib.submit.parser.split_string_list", return_value=mock_split_result ) as mock_split: result = parser.get_include() @@ -239,14 +239,14 @@ def test_parser_get_include_calls_split_files_list_single_numeric_value(): assert result == mock_split_result -def test_parser_get_include_calls_split_files_list(): +def test_parser_get_include_calls_split_string_list(): parser = Parser.__new__(Parser) parser._options = {"include": "file1,file2"} - mock_split_result = [Path("file1"), Path("file2")] + mock_split_result = ["file1", "file2"] with patch( - "qq_lib.submit.parser.split_files_list", return_value=mock_split_result + "qq_lib.submit.parser.split_string_list", return_value=mock_split_result ) as mock_split: result = parser.get_include() @@ -753,7 +753,7 @@ def test_parser_integration(): assert resources.props == {"vnode": "node"} exclude = parser.get_exclude() - assert exclude == [Path("file1.txt"), Path("file2.txt")] + assert exclude == ["file1.txt", "file2.txt"] assert parser.get_loop_start() == 2 assert parser.get_loop_end() == 10 diff --git a/tests/submit/test_submit_submitter.py b/tests/submit/test_submit_submitter.py index e49a75d..6d7e5ab 100644 --- a/tests/submit/test_submit_submitter.py +++ b/tests/submit/test_submit_submitter.py @@ -10,7 +10,9 @@ import pytest +from qq_lib.batch.interface import AnyBatchClass from qq_lib.batch.pbs.pbs import PBS +from qq_lib.batch.slurm import Slurm from qq_lib.core.error import QQError from qq_lib.info.informer import Informer from qq_lib.properties.depend import Depend, DependType @@ -39,8 +41,8 @@ def test_submitter_init_sets_all_attributes_correctly(tmp_path): script=script, job_type=JobType.STANDARD, resources=Resources(), - exclude=[Path("exclude")], - include=[Path("include"), Path("/tmp/include")], + exclude=["exclude", "/tmp/exclude"], + include=["include", "/tmp/include"], transfer_mode=[Always()], server="pbs-m1.metacentrum.cz", interpreter=Interpreter(executable="bash"), @@ -57,7 +59,7 @@ def test_submitter_init_sets_all_attributes_correctly(tmp_path): assert submitter._job_name == "job1" assert submitter._info_file == tmp_path / f"job1{CFG.suffixes.qq_info}" assert submitter._resources == Resources() - assert submitter._exclude == [tmp_path / "exclude"] + assert submitter._exclude == [tmp_path / "exclude", Path("/tmp/exclude")] assert submitter._include == [tmp_path / "include", Path("/tmp/include")] assert submitter._depend == [] assert isinstance(submitter._transfer_mode[0], Always) @@ -104,7 +106,7 @@ def test_submitter_init_sets_all_optional_arguments_correctly(tmp_path): script.write_text("#!/usr/bin/env -S qq run\n") loop_info = LoopInfo(1, 5, Path("storage"), "job%04d") - exclude_files = [tmp_path / "file1.txt", tmp_path / "file2.txt"] + exclude_files = [str(tmp_path / "file1.txt"), str(tmp_path / "file2.txt")] depend_jobs = [ Depend(DependType.AFTER_SUCCESS, ["12345"]), Depend(DependType.AFTER_START, ["23456"]), @@ -139,7 +141,7 @@ def test_submitter_init_sets_all_optional_arguments_correctly(tmp_path): assert submitter._job_name == "job" assert submitter._info_file == tmp_path / f"job{CFG.suffixes.qq_info}" assert submitter._resources == Resources() - assert submitter._exclude == exclude_files + assert submitter._exclude == [Path(x) for x in exclude_files] assert submitter._depend == depend_jobs assert submitter._server == "fake.server.com" assert submitter._resubmit_from == [WorkHost(), ExplicitHost("node01")] @@ -737,3 +739,52 @@ def test_submitter_submit(tmp_path): def test_submitter_make_pattern(input_pattern, cycle, expected): result = Submitter._make_pattern(input_pattern, cycle) assert result == expected + + +@pytest.mark.parametrize("batch_system", [PBS, Slurm]) +def test_submitter_expands_glob_patterns_in_exclude_and_include( + tmp_path: Path, batch_system: AnyBatchClass +) -> None: + input_dir = tmp_path / "job" + input_dir.mkdir() + + script = input_dir / "run.sh" + script.write_text(f"#!/usr/bin/env -S {CFG.binary_name} run\n") + + (input_dir / "topology.pdb").touch() + (input_dir / "start.gro").touch() + (input_dir / "old.log").touch() + (input_dir / "run.log").touch() + (input_dir / "notes.md").touch() + (input_dir / "sub").mkdir() + (input_dir / "sub" / "nested.log").touch() + + shared = tmp_path / "shared" + shared.mkdir() + (shared / "params.itp").touch() + (shared / "forcefield.itp").touch() + (shared / "readme.md").touch() + + submitter = Submitter( + batch_system=batch_system, + queue="default", + account=None, + script=script, + job_type=JobType.STANDARD, + resources=Resources(ncpus=1, mem="1gb", walltime="1:00:00"), + exclude=["*.log", "notes.md", "missing.dat"], + include=[str(shared / "*.itp"), "sub/nested.log"], + ) + + assert submitter._exclude == [ + input_dir / "old.log", + input_dir / "run.log", + input_dir / "notes.md", + input_dir / "missing.dat", + ] + + assert submitter._include == [ + shared / "forcefield.itp", + shared / "params.itp", + input_dir / "sub" / "nested.log", + ] From 24afbf4ad0199c8fa572d8b73da8f21552595ec6 Mon Sep 17 00:00:00 2001 From: Ladme Date: Sat, 5 Sep 2026 12:24:54 +0200 Subject: [PATCH 08/19] fix/feat: make archiver respect include and exclude options --- CHANGELOG.md | 9 ++- src/qq_lib/archive/archiver.py | 41 +++++++++-- src/qq_lib/core/common.py | 14 ++++ src/qq_lib/run/runner.py | 31 +++++---- tests/archive/test_archive.py | 120 ++++++++++++++++++++++++++++++++- tests/core/test_common.py | 99 +++++++++++++++++++++++++++ tests/run/test_run_runner.py | 50 ++++++++++++-- 7 files changed, 338 insertions(+), 26 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5c19ed9..a35e07e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -14,8 +14,15 @@ - `QQ_LOOP_NEXT` which specifies the index of the next loop cycle. - `QQ_ARCHIVE_CURRENT` and `QQ_ARCHIVE_NEXT` which specify strings expected in files archived by qq for the current and the next loop cycles, respectively. If the archive format is not a printf pattern, the values of these variables are empty strings. -### Other changes +### `--include` and `--exclude` revisited + +- Glob patterns are now supported in `--include` and `--exclude` options. +- If you explicitly include a file into a working directory using the `--include` submission option, it will no longer be archived even if it matches the archive pattern. +- If you explicitly exclude a file from a working directory using the `--exclude` submission option, it will no longer be fetched from the archive, even if it matches the archive pattern for this loop job cycle. + +### Bug fixes and other changes +- Directories can be now properly archived. - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. diff --git a/src/qq_lib/archive/archiver.py b/src/qq_lib/archive/archiver.py index 105d0d7..43547f6 100644 --- a/src/qq_lib/archive/archiver.py +++ b/src/qq_lib/archive/archiver.py @@ -2,12 +2,13 @@ # Copyright (c) 2025-2026 Ladislav Bartos and Robert Vacha Lab import re +import shutil import socket from collections.abc import Iterable from pathlib import Path from qq_lib.batch.interface import AnyBatchClass -from qq_lib.core.common import is_printf_pattern, printf_to_regex +from qq_lib.core.common import is_printf_pattern, printf_to_regex, relocate_by_name from qq_lib.core.config import CFG from qq_lib.core.logger import get_logger from qq_lib.core.logical_paths import logical_resolve @@ -28,6 +29,8 @@ def __init__( input_machine: str, input_dir: Path, batch_system: AnyBatchClass, + included_files: list[Path], + excluded_files: list[Path], ): """ Initialize the Archiver. @@ -38,12 +41,18 @@ def __init__( input_machine (str): The hostname from which the job was submitted. input_dir (Path): The directory from which the job was submitted. batch_system (AnyBatchClass): The batch system which manages the job. + included_files (list[Path]): List that were explicitly included + in the working directory and should not be archived. + excluded_files (list[Path]): List of files that were explicitly excluded + from the working directory and should not be fetched from archive. """ self._batch_system = batch_system self._archive = archive self._archive_format = archive_format self._input_machine = input_machine self._input_dir = input_dir + self._included_files = included_files + self._excluded_files = excluded_files def make_archive_dir(self) -> None: """ @@ -64,6 +73,8 @@ def from_archive(self, dir: Path, cycle: int | None = None) -> None: fetched. If no cycle is provided, all files matching the pattern in the archive are fetched. + Files that were explicitly excluded via the `exclude` submission option are not fetched. + Args: dir (Path): The directory where files will be copied to. cycle (int | None): The cycle number to filter files for. @@ -81,6 +92,12 @@ def from_archive(self, dir: Path, cycle: int | None = None) -> None: logger.debug("Nothing to fetch from archive.") return + # files that were explicitly excluded via the `exclude` submission option are not fetched + logger.debug( + f"Files that were excluded from work dir using the `--exclude` option: {self._excluded_files}." + ) + files = [file for file in files if file not in self._excluded_files] + logger.debug(f"Files to fetch from archive: {files}.") Retryer( @@ -102,6 +119,8 @@ def to_archive(self, dir: Path) -> None: `dir` to the archive directory. After successfully transferring the files, they are removed from the working directory. + Files that were explicitly included via the `include` submission option are not archived. + Args: work_dir (Path): The directory containing files to archive. @@ -116,6 +135,15 @@ def to_archive(self, dir: Path) -> None: logger.debug("Nothing to archive.") return + # files that were explicitly included into the working directory + # are not transferred back to the input directory + # and also should not be archived + included_files = relocate_by_name(self._included_files, dir) + logger.debug( + f"Files that were copied to work dir using the `--include` option: {included_files}." + ) + files = [file for file in files if file not in included_files] + logger.debug(f"Files to archive: {files}.") Retryer( @@ -281,13 +309,16 @@ def _prepare_regex_pattern(pattern: str) -> re.Pattern[str]: @staticmethod def _remove_files(files: Iterable[Path]) -> None: """ - Remove a list of files from the filesystem. + Remove a list of files or directories from the filesystem. Args: - files (Iterable[Path]): Files to delete. + files (Iterable[Path]): Files or directories to delete. Raises: - OSError: If file removal fails for any file. + OSError: If file or directory removal fails for any file. """ for file in files: - file.unlink() + if file.is_symlink() or not file.is_dir(): + file.unlink() + else: + shutil.rmtree(file) diff --git a/src/qq_lib/core/common.py b/src/qq_lib/core/common.py index e431e37..970c5d9 100644 --- a/src/qq_lib/core/common.py +++ b/src/qq_lib/core/common.py @@ -826,3 +826,17 @@ def expand_pattern(pattern: str, directory: Path) -> list[Path]: return sorted(anchor.glob(str(relative))) except Exception as e: raise QQError(f"Could not expand pattern '{pattern}': {e}.") from e + + +def relocate_by_name(files: Iterable[Path], directory: Path) -> list[Path]: + """ + Map paths to their counterparts in another directory, matching by file name. + + Args: + files (Iterable[Path]): Original paths of the files. + directory (Path): Directory the files were placed in. + + Returns: + list[Path]: Logical absolute paths to the files inside `directory`. + """ + return [logical_resolve(directory / file.name) for file in files] diff --git a/src/qq_lib/run/runner.py b/src/qq_lib/run/runner.py index 73b15c0..ad84b28 100644 --- a/src/qq_lib/run/runner.py +++ b/src/qq_lib/run/runner.py @@ -16,7 +16,7 @@ import qq_lib from qq_lib.archive.archiver import Archiver from qq_lib.batch.interface import BatchInterface -from qq_lib.core.common import construct_loop_job_name +from qq_lib.core.common import construct_loop_job_name, relocate_by_name from qq_lib.core.config import CFG from qq_lib.core.error import ( QQError, @@ -122,11 +122,13 @@ def __init__(self, info_file: Path, host: str): # initialize archiver, if this is a loop job if loop_info := self._informer.info.loop_info: self._archiver = Archiver( - loop_info.archive, - loop_info.archive_format, - self._informer.info.input_machine, - self._informer.info.input_dir, - self._batch_system, + archive=loop_info.archive, + archive_format=loop_info.archive_format, + input_machine=self._informer.info.input_machine, + input_dir=self._informer.info.input_dir, + batch_system=self._batch_system, + included_files=self._informer.info.included_files, + excluded_files=self._informer.info.excluded_files, ) self._should_resubmit = True else: @@ -722,14 +724,20 @@ def _archive_files_from_work_dir(self) -> None: ) # get the files to archive corresponding to the next loop job cycle - if not self._archiver.get_files_matching_pattern( + files_matching_pattern = self._archiver.get_files_matching_pattern( self._work_dir, None, loop_info.archive_format, loop_info.current + 1, False, - ): - # if there are no files matching the next loop job cycle, create an empty .init file + ) + included_files = self._get_explicitly_included_files_in_work_dir() + files = [f for f in files_matching_pattern if f not in included_files] + + if not files: + # if there are no files matching the next loop job cycle + # which were not explicitly included, + # create an empty .init file # so that the loop job continues normally logger.debug( f"Creating .init file for loop job cycle {loop_info.current + 1}." @@ -744,10 +752,7 @@ def _get_explicitly_included_files_in_work_dir(self) -> list[Path]: Return absolute paths to files and directories in the working directory that were explicitly copied via the `--include` submission option. """ - files = [ - logical_resolve(self._work_dir / f.name) - for f in self._informer.info.included_files - ] + files = relocate_by_name(self._informer.info.included_files, self._work_dir) logger.debug( f"Files that were copied to work dir using the `--include` option: {files}." diff --git a/tests/archive/test_archive.py b/tests/archive/test_archive.py index 077795f..0475374 100644 --- a/tests/archive/test_archive.py +++ b/tests/archive/test_archive.py @@ -24,6 +24,19 @@ def test_remove_files(tmp_path): assert not f.exists() +def test_remove_files_directory(tmp_path): + files = [] + for i in range(3): + d = tmp_path / f"dir{i}" + d.mkdir() + files.append(d) + + Archiver._remove_files(files) + + for f in files: + assert not f.exists() + + def test_remove_files_raises(tmp_path): f = tmp_path / "file.txt" f.write_text("test") @@ -95,6 +108,8 @@ def test_make_archive_dir_creates_directory(monkeypatch, archive_dir, input_dir) input_machine="fake_host", input_dir=input_dir, batch_system=PBS, + included_files=[], + excluded_files=[], ) assert not archive_dir.exists() @@ -114,6 +129,8 @@ def test_make_archive_dir_already_exists(monkeypatch, archive_dir, input_dir): input_machine="fake_host", input_dir=input_dir, batch_system=PBS, + included_files=[], + excluded_files=[], ) archiver.make_archive_dir() @@ -129,6 +146,34 @@ def archiver(input_dir, archive_dir): input_machine="fake_host", input_dir=input_dir, batch_system=PBS, + included_files=[], + excluded_files=[], + ) + + +@pytest.fixture +def archiver_with_included_files(input_dir, archive_dir): + return Archiver( + archive=archive_dir, + archive_format="job%04d", + input_machine="fake_host", + input_dir=input_dir, + batch_system=PBS, + included_files=[Path("external/shared/job0002.dat")], + excluded_files=[], + ) + + +@pytest.fixture +def archiver_with_excluded_files(input_dir, archive_dir): + return Archiver( + archive=archive_dir, + archive_format="job%04d", + input_machine="fake_host", + input_dir=input_dir, + batch_system=PBS, + included_files=[], + excluded_files=[archive_dir / "job0001.dat"], ) @@ -295,7 +340,7 @@ def test_get_files_printf_pattern_without_cycle_partial_match( @pytest.mark.parametrize("cycle", [None, 1]) -def test_archive_from_copies_files(monkeypatch, archiver, archive_dir, work_dir, cycle): +def test_from_archive_copies_files(monkeypatch, archiver, archive_dir, work_dir, cycle): monkeypatch.setenv(CFG.env_vars.shared_submit, "true") archiver.make_archive_dir() @@ -322,6 +367,36 @@ def test_archive_from_copies_files(monkeypatch, archiver, archive_dir, work_dir, assert not (work_dir / "other.txt").exists() +@pytest.mark.parametrize("cycle", [None, 1]) +def test_from_archive_copies_files_except_excluded( + monkeypatch, archiver_with_excluded_files, archive_dir, work_dir, cycle +): + monkeypatch.setenv(CFG.env_vars.shared_submit, "true") + archiver_with_excluded_files.make_archive_dir() + + filenames = ["job0001.dat", "job0002.dat", "other.txt", "job0001.qqinfo"] + touch_files(archive_dir, filenames) + + archiver_with_excluded_files.from_archive(work_dir, cycle=cycle) + + expected_files = [] if cycle == 1 else [archive_dir / "job0002.dat"] + + for f in expected_files: + copied_file = work_dir / f.name + assert copied_file.exists() + assert copied_file.is_file() + + # files still exist in the archive + assert f.exists() + assert f.is_file() + + # files not matching pattern should not be copied + assert not (work_dir / "other.txt").exists() + + # files excluded by pattern should not be copied + assert not (work_dir / "job0001.dat").exists() + + def test_archive_from_nothing_to_fetch(monkeypatch, archiver, work_dir): monkeypatch.setenv(CFG.env_vars.shared_submit, "true") archiver.make_archive_dir() @@ -333,7 +408,7 @@ def test_archive_from_nothing_to_fetch(monkeypatch, archiver, work_dir): assert list(work_dir.iterdir()) == [] -def test_archive_to_copies_and_removes_files( +def test_archive_to_archive_copies_and_removes_files( monkeypatch, archiver, archive_dir, work_dir ): monkeypatch.setenv(CFG.env_vars.shared_submit, "true") @@ -366,6 +441,41 @@ def test_archive_to_copies_and_removes_files( assert (work_dir / "job0002.err").exists() +def test_archive_to_archive_copies_and_removes_files_except_included( + monkeypatch, archiver_with_included_files, archive_dir, work_dir +): + monkeypatch.setenv(CFG.env_vars.shared_submit, "true") + archiver_with_included_files.make_archive_dir() + + filenames = [ + "job0001.dat", + "job0002.dat", + "other.txt", + "job0001.out", + "job0002.err", + ] + touch_files(work_dir, filenames) + + archiver_with_included_files.to_archive(work_dir) + + expected_copied = [archive_dir / "job0001.dat"] + for f in expected_copied: + assert f.exists() and f.is_file() + + # matching files should be removed from work_dir + assert not (work_dir / "job0001.dat").exists() + + # non-matching files should remain + assert (work_dir / "other.txt").exists() + + # included files should also remain + assert (work_dir / "job0002.dat").exists() + + # qq runtime files should remain as well + assert (work_dir / "job0001.out").exists() + assert (work_dir / "job0002.err").exists() + + def test_archive_to_nothing_to_archive(monkeypatch, archiver, archive_dir, work_dir): monkeypatch.setenv(CFG.env_vars.shared_submit, "true") archiver.make_archive_dir() @@ -503,6 +613,8 @@ def test_create_init_file_creates_file_for_given_cycle(tmp_path: Path): input_machine="localhost", input_dir=tmp_path, batch_system=PBS, + included_files=[], + excluded_files=[], ) archiver.create_init_file(cycle=1) @@ -518,6 +630,8 @@ def test_create_init_file_creates_empty_file(tmp_path: Path): input_machine="localhost", input_dir=tmp_path, batch_system=PBS, + included_files=[], + excluded_files=[], ) archiver.create_init_file(cycle=1) @@ -533,6 +647,8 @@ def test_create_init_file_uses_correct_cycle_number(tmp_path: Path): input_machine="localhost", input_dir=tmp_path, batch_system=PBS, + included_files=[], + excluded_files=[], ) archiver.create_init_file(cycle=42) diff --git a/tests/core/test_common.py b/tests/core/test_common.py index c0584b6..4e543ec 100644 --- a/tests/core/test_common.py +++ b/tests/core/test_common.py @@ -39,6 +39,7 @@ load_yaml_dumper, load_yaml_loader, printf_to_regex, + relocate_by_name, split_string_list, to_snake_case, translate_server, @@ -1322,3 +1323,101 @@ def _raise(_self: Path, _pattern: str) -> list[Path]: with pytest.raises(QQError, match=r"\*\.txt"): expand_pattern("*.txt", tmp_path) + + expand_pattern("*.txt", tmp_path) + + +def test_relocate_by_name_empty_input(tmp_path: Path) -> None: + assert relocate_by_name([], tmp_path / "work") == [] + + +def test_relocate_by_name_single_file(tmp_path: Path) -> None: + assert relocate_by_name([tmp_path / "input" / "start.gro"], tmp_path / "work") == [ + tmp_path / "work" / "start.gro" + ] + + +def test_relocate_by_name_preserves_order(tmp_path: Path) -> None: + files = [ + tmp_path / "input" / "c.itp", + tmp_path / "input" / "a.itp", + tmp_path / "input" / "b.itp", + ] + + assert relocate_by_name(files, tmp_path / "work") == [ + tmp_path / "work" / "c.itp", + tmp_path / "work" / "a.itp", + tmp_path / "work" / "b.itp", + ] + + +def test_relocate_by_name_flattens_nested_paths(tmp_path: Path) -> None: + files = [ + tmp_path / "input" / "topology.pdb", + tmp_path / "input" / "sub" / "deeper" / "params.itp", + ] + + assert relocate_by_name(files, tmp_path / "work") == [ + tmp_path / "work" / "topology.pdb", + tmp_path / "work" / "params.itp", + ] + + +def test_relocate_by_name_flattens_files_from_unrelated_directories( + tmp_path: Path, +) -> None: + files = [ + tmp_path / "input" / "start.gro", + Path("/shared/forcefield/params.itp"), + ] + + assert relocate_by_name(files, tmp_path / "work") == [ + tmp_path / "work" / "start.gro", + tmp_path / "work" / "params.itp", + ] + + +def test_relocate_by_name_duplicate_basenames_collide(tmp_path: Path) -> None: + files = [ + tmp_path / "first" / "params.itp", + tmp_path / "second" / "params.itp", + ] + + assert relocate_by_name(files, tmp_path / "work") == [ + tmp_path / "work" / "params.itp", + tmp_path / "work" / "params.itp", + ] + + +def test_relocate_by_name_relocates_directories(tmp_path: Path) -> None: + assert relocate_by_name([tmp_path / "input" / "frames"], tmp_path / "work") == [ + tmp_path / "work" / "frames" + ] + + +def test_relocate_by_name_collapses_dot_segments_in_directory(tmp_path: Path) -> None: + directory = tmp_path / "work" / "sub" / ".." / "." / "final" + + assert relocate_by_name([tmp_path / "input" / "a.txt"], directory) == [ + tmp_path / "work" / "final" / "a.txt" + ] + + +def test_relocate_by_name_does_not_expand_symlinks(tmp_path: Path) -> None: + real = tmp_path / "real" + real.mkdir() + link = tmp_path / "link" + link.symlink_to(real) + + assert relocate_by_name([tmp_path / "input" / "a.txt"], link) == [link / "a.txt"] + + assert relocate_by_name([tmp_path / "input" / "a.txt"], link) == [link / "a.txt"] + + +def test_relocate_by_name_returns_absolute_paths(tmp_path: Path) -> None: + files = [tmp_path / "input" / "a.txt", Path("/shared/b.txt")] + + result = relocate_by_name(files, tmp_path / "work") + + assert result + assert all(path.is_absolute() for path in result) diff --git a/tests/run/test_run_runner.py b/tests/run/test_run_runner.py index 27e9de7..77e27f2 100644 --- a/tests/run/test_run_runner.py +++ b/tests/run/test_run_runner.py @@ -187,11 +187,13 @@ def test_runner_init_creates_archiver_when_loop_info_present(): runner = Runner(Path("job.qqinfo"), "host") mock_archiver.assert_called_once_with( - loop_info.archive, - loop_info.archive_format, - informer.info.input_machine, - informer.info.input_dir, - batch, + archive=loop_info.archive, + archive_format=loop_info.archive_format, + input_machine=informer.info.input_machine, + input_dir=informer.info.input_dir, + batch_system=batch, + included_files=informer.info.included_files, + excluded_files=informer.info.excluded_files, ) mock_batchmeta.assert_called_once() mock_retryer.assert_called_once() @@ -1688,6 +1690,24 @@ def _make_runner_with_archiver( informer = MagicMock() informer.info.loop_info = loop_info + informer.info.included_files = [] + runner._informer = informer + + return runner, archiver + + +def _make_runner_with_archiver_and_included_files( + tmp_path: Path, + loop_info: LoopInfo, +) -> tuple[Runner, MagicMock]: + runner = Runner.__new__(Runner) + archiver = MagicMock(spec=Archiver) + runner._archiver = archiver + runner._work_dir = tmp_path + + informer = MagicMock() + informer.info.loop_info = loop_info + informer.info.included_files = [Path("/path/to/job0003.dat")] runner._informer = informer return runner, archiver @@ -1727,6 +1747,26 @@ def test_archive_files_from_work_dir_creates_init_file_when_no_matching_files( archiver.create_init_file.assert_called_once_with(4) +def test_archive_files_from_work_dir_creates_init_file_even_when_matching_files_exist_if_they_were_included( + tmp_path: Path, +): + loop_info = LoopInfo( + start=1, + end=10, + archive=tmp_path / "archive", + archive_format="job%04d", + current=3, + ) + runner, archiver = _make_runner_with_archiver_and_included_files( + tmp_path, loop_info + ) + archiver.get_files_matching_pattern.return_value = [tmp_path / "job0003.dat"] + + runner._archive_files_from_work_dir() + + archiver.create_init_file.assert_called_once_with(4) + + def test_archive_files_from_work_dir_does_not_create_init_file_when_matching_files_exist( tmp_path: Path, ): From ff43c22f59189811036ab4da4bb44c3912eac249 Mon Sep 17 00:00:00 2001 From: Ladme Date: Sat, 5 Sep 2026 12:57:11 +0200 Subject: [PATCH 09/19] chore: fix changelog --- CHANGELOG.md | 12 +++--------- 1 file changed, 3 insertions(+), 9 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a35e07e..fae8215 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,11 +1,3 @@ -## Version 0.12.3 - -- More robust installation script for Metacentrum. - -## Version 0.12.2 - -- Fixed build for IT4Innovation's Karolina. - ## Version 0.13.0 ### New environment variables for loop jobs @@ -26,7 +18,9 @@ - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. ---- +## Version 0.12.3 + +- More robust installation script for Metacentrum. ## Version 0.12.2 From ad2941a30915c3ae9e8bcbdf9cd72f9048655fd2 Mon Sep 17 00:00:00 2001 From: Ladme Date: Sat, 5 Sep 2026 13:00:14 +0200 Subject: [PATCH 10/19] chore: separator to changelog --- CHANGELOG.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index fae8215..30247c1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -18,6 +18,8 @@ - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. +--- + ## Version 0.12.3 - More robust installation script for Metacentrum. From 62d48b4e184ac404fb3ae7a531245b945fd33e0a Mon Sep 17 00:00:00 2001 From: Ladme Date: Sat, 5 Sep 2026 20:41:48 +0200 Subject: [PATCH 11/19] feat: add 'ignore' submission option --- src/qq_lib/archive/archiver.py | 76 +++- src/qq_lib/properties/info.py | 3 + src/qq_lib/respawn/respawner.py | 1 + src/qq_lib/resubmit/resubmitter.py | 1 + src/qq_lib/run/runner.py | 84 +++-- src/qq_lib/submit/cli.py | 12 +- src/qq_lib/submit/factory.py | 18 + src/qq_lib/submit/parser.py | 12 + src/qq_lib/submit/submitter.py | 10 + tests/archive/test_archive.py | 371 ++++++++++++++++++-- tests/respawn/test_respawn_respawner.py | 1 + tests/resubmit/test_resubmit_resubmitter.py | 1 + tests/run/test_run_runner.py | 266 ++++++++++++-- tests/submit/test_submit_factory.py | 57 ++- tests/submit/test_submit_parser.py | 25 +- tests/submit/test_submit_submitter.py | 11 + 16 files changed, 847 insertions(+), 102 deletions(-) diff --git a/src/qq_lib/archive/archiver.py b/src/qq_lib/archive/archiver.py index 43547f6..7782e72 100644 --- a/src/qq_lib/archive/archiver.py +++ b/src/qq_lib/archive/archiver.py @@ -31,6 +31,7 @@ def __init__( batch_system: AnyBatchClass, included_files: list[Path], excluded_files: list[Path], + ignored_files: list[Path], ): """ Initialize the Archiver. @@ -45,6 +46,8 @@ def __init__( in the working directory and should not be archived. excluded_files (list[Path]): List of files that were explicitly excluded from the working directory and should not be fetched from archive. + ignored_files (list[Path]): List of files that are ignored and should be neither + archived nor fetched from archive. """ self._batch_system = batch_system self._archive = archive @@ -53,6 +56,14 @@ def __init__( self._input_dir = input_dir self._included_files = included_files self._excluded_files = excluded_files + self._ignored_files = ignored_files + + @property + def archive(self) -> Path: + """ + Returns the absolute path to the archive. + """ + return self._archive def make_archive_dir(self) -> None: """ @@ -93,11 +104,13 @@ def from_archive(self, dir: Path, cycle: int | None = None) -> None: return # files that were explicitly excluded via the `exclude` submission option are not fetched + # as are not fetched files that are explicitly ignored via the `ignore` submission option + exclude = self._get_excluded_from_copying_from_archive() logger.debug( - f"Files that were excluded from work dir using the `--exclude` option: {self._excluded_files}." + f"Files that are excluded or ignored from being copied from the archive: {exclude}." ) - files = [file for file in files if file not in self._excluded_files] + files = [file for file in files if file not in exclude] logger.debug(f"Files to fetch from archive: {files}.") Retryer( @@ -135,14 +148,13 @@ def to_archive(self, dir: Path) -> None: logger.debug("Nothing to archive.") return - # files that were explicitly included into the working directory - # are not transferred back to the input directory - # and also should not be archived - included_files = relocate_by_name(self._included_files, dir) + # files that were explicitly included via the `include` submission option are not archived + # as well as files that are explicitly ignored via the `ignore` submission option + exclude = self._get_excluded_from_copying_to_archive(dir) logger.debug( - f"Files that were copied to work dir using the `--include` option: {included_files}." + f"Files that are excluded or ignored from being copied to the archive: {exclude}." ) - files = [file for file in files if file not in included_files] + files = [file for file in files if file not in exclude] logger.debug(f"Files to archive: {files}.") @@ -280,6 +292,54 @@ def get_files_matching_pattern( if regex.search(f.stem) and f.suffix not in CFG.suffixes.all_suffixes ] + def _get_excluded_from_copying_to_archive(self, dir: Path) -> list[Path]: + """ + Return paths that must not be copied to the archive. + + Collects the files ignored and explicitly included by the user, + and the archive directory itself. Duplicates are removed, + preserving the order of first occurrence. + + Args: + dir (Path): The directory from which we are archiving the files. + + Returns: + list[Path]: Paths that should not be copied to the archive. + """ + + return list( + dict.fromkeys( + relocate_by_name( + [ + *self._ignored_files, + *self._included_files, + self._archive, + ], + dir, + ) + ) + ) + + def _get_excluded_from_copying_from_archive(self) -> list[Path]: + """ + Return paths that must not be copied from the archive. + + Collects files ignored and explicitly excluded by the user. + Duplicates are removed, preserving the order of first occurrence. + + Returns: + list[Path]: Paths that should not be copied from the archive. + """ + + return list( + dict.fromkeys( + [ + *self._ignored_files, + *self._excluded_files, + ] + ) + ) + def create_init_file(self, cycle: int) -> None: """ Create an empty init file for the given cycle. diff --git a/src/qq_lib/properties/info.py b/src/qq_lib/properties/info.py index 6789685..aafead8 100644 --- a/src/qq_lib/properties/info.py +++ b/src/qq_lib/properties/info.py @@ -104,6 +104,9 @@ class Info: # List of files and directories to explicitly copy to the working directory. included_files: list[Path] = field(default_factory=list) + # List of files and directories to ignore completely. + ignored_files: list[Path] = field(default_factory=list) + # Mode of transferring files from the working directory to the input directory after job completion. transfer_mode: list[TransferMode] = field(default_factory=lambda: [Success()]) diff --git a/src/qq_lib/respawn/respawner.py b/src/qq_lib/respawn/respawner.py index 6cc2dbb..d78e09d 100644 --- a/src/qq_lib/respawn/respawner.py +++ b/src/qq_lib/respawn/respawner.py @@ -96,6 +96,7 @@ def _build_submitter(self, informer: Informer) -> Submitter: loop_info=loop_info, exclude=[str(x) for x in informer.info.excluded_files], include=[str(x) for x in informer.info.included_files], + ignore=[str(x) for x in informer.info.ignored_files], # we need to remove dependencies that are no longer present in the batch system depend=filter_dependencies(informer.batch_system, informer.info.depend), transfer_mode=informer.info.transfer_mode, diff --git a/src/qq_lib/resubmit/resubmitter.py b/src/qq_lib/resubmit/resubmitter.py index 15ee578..f83a822 100644 --- a/src/qq_lib/resubmit/resubmitter.py +++ b/src/qq_lib/resubmit/resubmitter.py @@ -87,6 +87,7 @@ def _build_submitter(informer: Informer, input_dir: Path) -> Submitter: loop_info=informer.info.loop_info, exclude=[str(x) for x in informer.info.excluded_files], include=[str(x) for x in informer.info.included_files], + ignore=[str(x) for x in informer.info.ignored_files], depend=[Depend(type=DependType.AFTER_SUCCESS, jobs=[informer.info.job_id])], transfer_mode=informer.info.transfer_mode, server=informer.info.server, diff --git a/src/qq_lib/run/runner.py b/src/qq_lib/run/runner.py index ad84b28..bc28d11 100644 --- a/src/qq_lib/run/runner.py +++ b/src/qq_lib/run/runner.py @@ -129,6 +129,7 @@ def __init__(self, info_file: Path, host: str): batch_system=self._batch_system, included_files=self._informer.info.included_files, excluded_files=self._informer.info.excluded_files, + ignored_files=self._informer.info.ignored_files, ) self._should_resubmit = True else: @@ -288,9 +289,9 @@ def finalize(self) -> None: self._input_dir, socket.getfqdn(), self._informer.info.input_machine, - # exclude files that were copied to workdir from the outside of input dir (--include option) - # these files should not be copied to the input directory, since they were never inside it - self._get_explicitly_included_files_in_work_dir(), + # exclude files that were specifically included via the `--include` option + # and files that were specifically chosen to be ignored via `--ignore` option + self._get_excluded_from_input_dir(), max_tries=CFG.runner.retry_tries, wait_seconds=CFG.runner.retry_wait, ).run() @@ -378,17 +379,12 @@ def _set_up_scratch_dir(self) -> None: ).run() # files excluded from copying to the working directory - qq_out = ( - self._informer.info.input_dir / self._informer.info.job_name - ).with_suffix(CFG.suffixes.qq_out) - excluded = self._informer.info.excluded_files + [self._info_file, qq_out] - if self._archiver: - excluded.append(self._archiver._archive) - - # copy files from the input directory to the working directory + excluded = self._get_excluded_from_work_dir() logger.debug( f"Files excluded from being copied to the working directory: {excluded}." ) + + # copy files from the input directory to the working directory Retryer( self._batch_system.sync_with_exclusions, self._input_dir, @@ -731,12 +727,13 @@ def _archive_files_from_work_dir(self) -> None: loop_info.current + 1, False, ) - included_files = self._get_explicitly_included_files_in_work_dir() - files = [f for f in files_matching_pattern if f not in included_files] + exclude = self._get_excluded_from_input_dir() + logger.debug(f"Files excluded from archiving: {exclude}.") + files = [f for f in files_matching_pattern if f not in exclude] if not files: # if there are no files matching the next loop job cycle - # which were not explicitly included, + # (which are not excluded via `--include` or `--ignore` options) # create an empty .init file # so that the loop job continues normally logger.debug( @@ -747,19 +744,6 @@ def _archive_files_from_work_dir(self) -> None: # archive all files matching the archive format self._archiver.to_archive(self._work_dir) - def _get_explicitly_included_files_in_work_dir(self) -> list[Path]: - """ - Return absolute paths to files and directories in the working directory - that were explicitly copied via the `--include` submission option. - """ - files = relocate_by_name(self._informer.info.included_files, self._work_dir) - - logger.debug( - f"Files that were copied to work dir using the `--include` option: {files}." - ) - - return files - def _copy_files(self, files: list[Path]): """ Copy files and directories using the provided absolute paths to the working directory. @@ -775,6 +759,52 @@ def _copy_files(self, files: list[Path]): [file], ) + def _get_excluded_from_work_dir(self) -> list[Path]: + """ + Return paths that must not be copied to the working directory. + + Collects the files excluded and ignored by the user, the qq info file, + the qq output file, and the archive if the job is a loop job. + Duplicates are removed, preserving the order of first occurrence. + + Returns: + list[Path]: Paths that should not be copied to the working directory. + """ + info = self._informer.info + + qq_out = (info.input_dir / info.job_name).with_suffix(CFG.suffixes.qq_out) + + excluded = [ + *info.excluded_files, + *info.ignored_files, + self._info_file, + qq_out, + ] + + if self._archiver: + excluded.append(self._archiver.archive) + + return list(dict.fromkeys(excluded)) + + def _get_excluded_from_input_dir(self) -> list[Path]: + """ + Return paths that must not be copied to the input directory. + + Collects explicitly included files and ignored files, and the + archive if the job is a loop job. Duplicates are removed, + preserving the order of first occurrence. + + Returns: + list[Path]: Paths that should not be copied to the input directory. + """ + info = self._informer.info + + excluded = [*info.included_files, *info.ignored_files] + if self._archiver: + excluded.append(self._archiver.archive) + + return list(dict.fromkeys(relocate_by_name(excluded, self._work_dir))) + def _cleanup(self) -> None: """ Clean up after execution is interrupted or killed. diff --git a/src/qq_lib/submit/cli.py b/src/qq_lib/submit/cli.py index a3770c1..01c29be 100644 --- a/src/qq_lib/submit/cli.py +++ b/src/qq_lib/submit/cli.py @@ -138,7 +138,17 @@ def complete_script( f"These files are {click.style('not', bold=True)} copied back after job completion. " f"Paths must be absolute or relative to the input directory.\n" "You can use glob patterns to match multiple files or directories.\n" - f"This option is ignored if the input directory is used as the working directory.\n" + ), +) +@optgroup.option( + "--ignore", + type=str, + default=None, + help=( + "Colon-, comma-, or space-separated list of files or directories to ignore in all transfer operations.\n" + "These files are neither copied to the working directory nor copied back after job completion.\n" + "Paths to files and directories to ignore must be absolute or relative to the input directory.\n" + "You can use glob patterns to match multiple files or directories.\n" ), ) @optgroup.option( diff --git a/src/qq_lib/submit/factory.py b/src/qq_lib/submit/factory.py index 2ad969d..1d9dadb 100644 --- a/src/qq_lib/submit/factory.py +++ b/src/qq_lib/submit/factory.py @@ -84,6 +84,7 @@ def make_submitter(self) -> Submitter: loop_info=loop_info, exclude=self._get_exclude(), include=self._get_include(), + ignore=self._get_ignore(), depend=self._get_depend(), transfer_mode=self._get_transfer_mode(), server=server, @@ -267,6 +268,23 @@ def _get_include(self) -> list[str]: split_string_list(self._kwargs.get("include")) or self._parser.get_include() ) + def _get_ignore(self) -> list[str]: + """ + Determine the files that transfer operations should ignore completely. + + Priority: + 1. Ignored files specified on the command line. + 2. Ignored files specified inside the submitted script. + + The lists are NOT merged. + + Returns: + list[str]: List of files or glob patterns to ignore. + """ + return ( + split_string_list(self._kwargs.get("ignore")) or self._parser.get_ignore() + ) + def _get_depend(self) -> list[Depend]: """ Determine the list of dependencies. diff --git a/src/qq_lib/submit/parser.py b/src/qq_lib/submit/parser.py index 4c94c1c..ce979f9 100644 --- a/src/qq_lib/submit/parser.py +++ b/src/qq_lib/submit/parser.py @@ -183,6 +183,18 @@ def get_include(self) -> list[str]: return [] + def get_ignore(self) -> list[str]: + """ + Determine the files to completely ignore during transfer operations. + + Returns: + list[Path]: List of ignored files or glob patterns. Returns an empty list if none specified. + """ + if (ignore := self._options.get("ignore")) is not None: + return split_string_list(str(ignore)) + + return [] + def get_loop_start(self) -> int | None: """ Return the starting cycle number for loop jobs. diff --git a/src/qq_lib/submit/submitter.py b/src/qq_lib/submit/submitter.py index cd03525..f100743 100644 --- a/src/qq_lib/submit/submitter.py +++ b/src/qq_lib/submit/submitter.py @@ -60,6 +60,7 @@ def __init__( loop_info: LoopInfo | None = None, exclude: list[str] | None = None, include: list[str] | None = None, + ignore: list[str] | None = None, depend: list[Depend] | None = None, transfer_mode: list[TransferMode] | None = None, server: str | None = None, @@ -83,6 +84,9 @@ def __init__( include (list[str] | None): Optional list of files or glob patterns which should be copied to the working directory even though they are not part of the job's input directory. Paths are provided either absolute or relative to the input directory. + ignore (list[str] | None): Optional list of files or glob patterns which should be ignored completely. + These files will not be copied to the working directory and if they are created in the working directory, they are also not copied back. + Paths are provided either absolute or relative to the input directory. depend (list[Depend] | None): Optional list of job dependencies. transfer_mode (list[TransferMode] | None): Mode specifying when files whould be transferred from the working directory to the input directory. Defaults to [`Success()`]. @@ -111,6 +115,7 @@ def __init__( self._resources = resources self._exclude = expand_paths(exclude or [], self._input_dir) self._include = expand_paths(include or [], self._input_dir) + self._ignore = expand_paths(ignore or [], self._input_dir) self._depend = depend or [] self._transfer_mode = transfer_mode or TransferMode.multi_from_str( CFG.transfer_files_options.default_transfer_mode @@ -183,6 +188,7 @@ def submit(self, remote: str | None = None) -> str: loop_info=self._loop_info, excluded_files=self._exclude, included_files=self._include, + ignored_files=self._ignore, depend=self._depend, account=self._account, transfer_mode=self._transfer_mode, @@ -307,6 +313,10 @@ def get_include(self) -> list[Path]: """Get a list of included files.""" return self._include + def get_ignore(self) -> list[Path]: + """Get a list of ignored files.""" + return self._ignore + def get_depend(self) -> list[Depend]: """Get the list of dependencies.""" return self._depend diff --git a/tests/archive/test_archive.py b/tests/archive/test_archive.py index 0475374..f60b336 100644 --- a/tests/archive/test_archive.py +++ b/tests/archive/test_archive.py @@ -110,6 +110,7 @@ def test_make_archive_dir_creates_directory(monkeypatch, archive_dir, input_dir) batch_system=PBS, included_files=[], excluded_files=[], + ignored_files=[], ) assert not archive_dir.exists() @@ -131,6 +132,7 @@ def test_make_archive_dir_already_exists(monkeypatch, archive_dir, input_dir): batch_system=PBS, included_files=[], excluded_files=[], + ignored_files=[], ) archiver.make_archive_dir() @@ -148,32 +150,26 @@ def archiver(input_dir, archive_dir): batch_system=PBS, included_files=[], excluded_files=[], + ignored_files=[], ) -@pytest.fixture -def archiver_with_included_files(input_dir, archive_dir): - return Archiver( - archive=archive_dir, - archive_format="job%04d", - input_machine="fake_host", - input_dir=input_dir, - batch_system=PBS, - included_files=[Path("external/shared/job0002.dat")], - excluded_files=[], - ) - - -@pytest.fixture -def archiver_with_excluded_files(input_dir, archive_dir): +def _make_archiver( + archive: Path, + input_dir: Path, + included_files: list[Path] | None = None, + excluded_files: list[Path] | None = None, + ignored_files: list[Path] | None = None, +) -> Archiver: return Archiver( - archive=archive_dir, + archive=archive, archive_format="job%04d", - input_machine="fake_host", + input_machine="input", input_dir=input_dir, batch_system=PBS, - included_files=[], - excluded_files=[archive_dir / "job0001.dat"], + included_files=included_files or [], + excluded_files=excluded_files or [], + ignored_files=ignored_files or [], ) @@ -369,15 +365,18 @@ def test_from_archive_copies_files(monkeypatch, archiver, archive_dir, work_dir, @pytest.mark.parametrize("cycle", [None, 1]) def test_from_archive_copies_files_except_excluded( - monkeypatch, archiver_with_excluded_files, archive_dir, work_dir, cycle + monkeypatch, archive_dir, work_dir, cycle, input_dir ): monkeypatch.setenv(CFG.env_vars.shared_submit, "true") - archiver_with_excluded_files.make_archive_dir() + archiver = _make_archiver( + archive_dir, input_dir, excluded_files=[archive_dir / "job0001.dat"] + ) + archiver.make_archive_dir() filenames = ["job0001.dat", "job0002.dat", "other.txt", "job0001.qqinfo"] touch_files(archive_dir, filenames) - archiver_with_excluded_files.from_archive(work_dir, cycle=cycle) + archiver.from_archive(work_dir, cycle=cycle) expected_files = [] if cycle == 1 else [archive_dir / "job0002.dat"] @@ -397,6 +396,51 @@ def test_from_archive_copies_files_except_excluded( assert not (work_dir / "job0001.dat").exists() +@pytest.mark.parametrize("cycle", [None, 1]) +def test_from_archive_copies_files_except_excluded_and_ignored( + monkeypatch, archive_dir, work_dir, cycle, input_dir +): + monkeypatch.setenv(CFG.env_vars.shared_submit, "true") + archiver = _make_archiver( + archive_dir, + input_dir, + excluded_files=[archive_dir / "job0001.dat"], + ignored_files=[input_dir / "file_to_ignore.txt"], + ) + archiver.make_archive_dir() + + filenames = [ + "job0001.dat", + "job0002.dat", + "other.txt", + "file_to_ignore.txt", + "job0001.qqinfo", + ] + touch_files(archive_dir, filenames) + + archiver.from_archive(work_dir, cycle=cycle) + + expected_files = [] if cycle == 1 else [archive_dir / "job0002.dat"] + + for f in expected_files: + copied_file = work_dir / f.name + assert copied_file.exists() + assert copied_file.is_file() + + # files still exist in the archive + assert f.exists() + assert f.is_file() + + # files not matching pattern should not be copied + assert not (work_dir / "other.txt").exists() + + # excluded files should not be copied + assert not (work_dir / "job0001.dat").exists() + + # ignored files should not be copied + assert not (work_dir / "file_to_ignore.txt").exists() + + def test_archive_from_nothing_to_fetch(monkeypatch, archiver, work_dir): monkeypatch.setenv(CFG.env_vars.shared_submit, "true") archiver.make_archive_dir() @@ -442,10 +486,16 @@ def test_archive_to_archive_copies_and_removes_files( def test_archive_to_archive_copies_and_removes_files_except_included( - monkeypatch, archiver_with_included_files, archive_dir, work_dir + monkeypatch, + archive_dir, + work_dir, + input_dir, ): monkeypatch.setenv(CFG.env_vars.shared_submit, "true") - archiver_with_included_files.make_archive_dir() + archiver = _make_archiver( + archive_dir, input_dir, included_files=[Path("external/shared/job0002.dat")] + ) + archiver.make_archive_dir() filenames = [ "job0001.dat", @@ -456,7 +506,48 @@ def test_archive_to_archive_copies_and_removes_files_except_included( ] touch_files(work_dir, filenames) - archiver_with_included_files.to_archive(work_dir) + archiver.to_archive(work_dir) + + expected_copied = [archive_dir / "job0001.dat"] + for f in expected_copied: + assert f.exists() and f.is_file() + + # matching files should be removed from work_dir + assert not (work_dir / "job0001.dat").exists() + + # non-matching files should remain + assert (work_dir / "other.txt").exists() + + # included files should also remain + assert (work_dir / "job0002.dat").exists() + + # qq runtime files should remain as well + assert (work_dir / "job0001.out").exists() + assert (work_dir / "job0002.err").exists() + + +def test_archive_to_archive_copies_and_removes_files_except_ignored( + monkeypatch, + archive_dir, + work_dir, + input_dir, +): + monkeypatch.setenv(CFG.env_vars.shared_submit, "true") + archiver = _make_archiver( + archive_dir, input_dir, ignored_files=[Path("external/shared/job0002.dat")] + ) + archiver.make_archive_dir() + + filenames = [ + "job0001.dat", + "job0002.dat", + "other.txt", + "job0001.out", + "job0002.err", + ] + touch_files(work_dir, filenames) + + archiver.to_archive(work_dir) expected_copied = [archive_dir / "job0001.dat"] for f in expected_copied: @@ -615,6 +706,7 @@ def test_create_init_file_creates_file_for_given_cycle(tmp_path: Path): batch_system=PBS, included_files=[], excluded_files=[], + ignored_files=[], ) archiver.create_init_file(cycle=1) @@ -632,6 +724,7 @@ def test_create_init_file_creates_empty_file(tmp_path: Path): batch_system=PBS, included_files=[], excluded_files=[], + ignored_files=[], ) archiver.create_init_file(cycle=1) @@ -649,8 +742,236 @@ def test_create_init_file_uses_correct_cycle_number(tmp_path: Path): batch_system=PBS, included_files=[], excluded_files=[], + ignored_files=[], ) archiver.create_init_file(cycle=42) assert (tmp_path / "md042.init").exists() + + +def test_archiver_get_excluded_from_copying_to_archive_collects_all_sources( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + archive = input_dir / "archive" + work_dir = tmp_path / "scratch" / "main" + + archiver = _make_archiver( + archive=archive, + input_dir=input_dir, + ignored_files=[input_dir / "core.dump"], + included_files=[tmp_path / "shared" / "params.itp"], + ) + + assert archiver._get_excluded_from_copying_to_archive(work_dir) == [ + work_dir / "core.dump", + work_dir / "params.itp", + work_dir / "archive", + ] + + +def test_archiver_get_excluded_from_copying_to_archive_only_the_archive( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + work_dir = tmp_path / "scratch" / "main" + + archiver = _make_archiver(archive=input_dir / "archive", input_dir=input_dir) + + assert archiver._get_excluded_from_copying_to_archive(work_dir) == [ + work_dir / "archive" + ] + + +def test_archiver_get_excluded_from_copying_to_archive_ignores_excluded_files( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + work_dir = tmp_path / "scratch" / "main" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + excluded_files=[input_dir / "old.log"], + ) + + assert archiver._get_excluded_from_copying_to_archive(work_dir) == [ + work_dir / "archive" + ] + + +def test_archiver_get_excluded_from_copying_to_archive_relocates_to_the_given_dir( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + archive = input_dir / "archive" + other_dir = tmp_path / "scratch" / "other" + + archiver = _make_archiver( + archive=archive, + input_dir=input_dir, + ignored_files=[input_dir / "core.dump"], + ) + + result = archiver._get_excluded_from_copying_to_archive(other_dir) + + assert result == [other_dir / "core.dump", other_dir / "archive"] + assert all(path.parent == other_dir for path in result) + + +def test_archiver_get_excluded_from_copying_to_archive_flattens_nested_paths( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + work_dir = tmp_path / "scratch" / "main" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + included_files=[tmp_path / "shared" / "forcefield" / "deeper" / "params.itp"], + ) + + assert archiver._get_excluded_from_copying_to_archive(work_dir) == [ + work_dir / "params.itp", + work_dir / "archive", + ] + + +def test_archiver_get_excluded_from_copying_to_archive_deduplicates( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + shared = tmp_path / "shared" + work_dir = tmp_path / "scratch" / "main" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + ignored_files=[shared / "params.itp", input_dir / "core.dump"], + included_files=[shared / "params.itp"], + ) + + assert archiver._get_excluded_from_copying_to_archive(work_dir) == [ + work_dir / "params.itp", + work_dir / "core.dump", + work_dir / "archive", + ] + + +def test_archiver_get_excluded_from_copying_to_archive_deduplicates_after_relocating( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + work_dir = tmp_path / "scratch" / "main" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + included_files=[ + tmp_path / "first" / "params.itp", + tmp_path / "second" / "params.itp", + ], + ) + + assert archiver._get_excluded_from_copying_to_archive(work_dir) == [ + work_dir / "params.itp", + work_dir / "archive", + ] + + +def test_archiver_get_excluded_from_copying_to_archive_preserves_order( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + work_dir = tmp_path / "scratch" / "main" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + ignored_files=[input_dir / "b.dump", input_dir / "a.dump"], + included_files=[input_dir / "c.itp"], + ) + + assert archiver._get_excluded_from_copying_to_archive(work_dir) == [ + work_dir / "b.dump", + work_dir / "a.dump", + work_dir / "c.itp", + work_dir / "archive", + ] + + +def test_archiver_get_excluded_from_copying_from_archive_collects_all_sources( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + ignored_files=[input_dir / "core.dump"], + excluded_files=[input_dir / "old.log"], + ) + + assert archiver._get_excluded_from_copying_from_archive() == [ + input_dir / "core.dump", + input_dir / "old.log", + ] + + +def test_archiver_get_excluded_from_copying_from_archive_empty(tmp_path: Path) -> None: + input_dir = tmp_path / "input" + + archiver = _make_archiver(archive=input_dir / "archive", input_dir=input_dir) + + assert archiver._get_excluded_from_copying_from_archive() == [] + + +def test_archiver_get_excluded_from_copying_from_archive_ignores_included_files( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + included_files=[tmp_path / "shared" / "params.itp"], + ) + + assert archiver._get_excluded_from_copying_from_archive() == [] + + +def test_archiver_get_excluded_from_copying_from_archive_deduplicates( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + ignored_files=[input_dir / "core.dump", input_dir / "notes.md"], + excluded_files=[input_dir / "core.dump"], + ) + + assert archiver._get_excluded_from_copying_from_archive() == [ + input_dir / "core.dump", + input_dir / "notes.md", + ] + + +def test_archiver_get_excluded_from_copying_from_archive_keeps_duplicate_basenames( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + + archiver = _make_archiver( + archive=input_dir / "archive", + input_dir=input_dir, + ignored_files=[tmp_path / "first" / "params.itp"], + excluded_files=[tmp_path / "second" / "params.itp"], + ) + + assert archiver._get_excluded_from_copying_from_archive() == [ + tmp_path / "first" / "params.itp", + tmp_path / "second" / "params.itp", + ] diff --git a/tests/respawn/test_respawn_respawner.py b/tests/respawn/test_respawn_respawner.py index bd49c53..0f5c08d 100644 --- a/tests/respawn/test_respawn_respawner.py +++ b/tests/respawn/test_respawn_respawner.py @@ -133,6 +133,7 @@ def test_respawner_build_submitter_creates_submitter_with_correct_params( loop_info=None, exclude=[str(x) for x in informer.info.excluded_files], include=[str(x) for x in informer.info.included_files], + ignore=[str(x) for x in informer.info.ignored_files], depend=dependencies, transfer_mode=informer.info.transfer_mode, server=informer.info.server, diff --git a/tests/resubmit/test_resubmit_resubmitter.py b/tests/resubmit/test_resubmit_resubmitter.py index 00a8a71..2237cde 100644 --- a/tests/resubmit/test_resubmit_resubmitter.py +++ b/tests/resubmit/test_resubmit_resubmitter.py @@ -54,6 +54,7 @@ def test_resubmitter_build_submitter_creates_submitter_with_correct_params(): loop_info=informer.info.loop_info, exclude=[str(x) for x in informer.info.excluded_files], include=[str(x) for x in informer.info.included_files], + ignore=[str(x) for x in informer.info.ignored_files], depend=[Depend(type=DependType.AFTER_SUCCESS, jobs=["12345"])], transfer_mode=informer.info.transfer_mode, server=informer.info.server, diff --git a/tests/run/test_run_runner.py b/tests/run/test_run_runner.py index 77e27f2..cbc4e63 100644 --- a/tests/run/test_run_runner.py +++ b/tests/run/test_run_runner.py @@ -194,6 +194,7 @@ def test_runner_init_creates_archiver_when_loop_info_present(): batch_system=batch, included_files=informer.info.included_files, excluded_files=informer.info.excluded_files, + ignored_files=informer.info.ignored_files, ) mock_batchmeta.assert_called_once() mock_retryer.assert_called_once() @@ -723,8 +724,9 @@ def test_runner_set_up_scratch_dir_calls_retryers_with_correct_arguments(): runner._info_file = Path("job.qqinfo") runner._input_dir = Path("/input") runner._informer.info.job_id = "123" - runner._informer.info.excluded_files = ["ignore.txt"] + runner._informer.info.excluded_files = [Path("ignore.txt")] runner._informer.info.included_files = ["include1.txt", "include2.txt"] + runner._informer.info.ignored_files = [Path("something.dat")] runner._informer.info.input_machine = "random.host.org" runner._informer.info.input_dir = Path("/input") runner._informer.info.job_name = "job+0002" @@ -755,13 +757,18 @@ def test_runner_set_up_scratch_dir_calls_retryers_with_correct_arguments(): # third Retryer call: sync_with_exclusions sync_call = retryer_cls.call_args_list[2] - expected_excluded = ["ignore.txt", runner._info_file, Path("/input/job+0002.qqout")] + expected_excluded = [ + Path("ignore.txt"), + runner._info_file, + Path("/input/job+0002.qqout"), + ] + expected_ignored = [Path("something.dat")] assert sync_call.args[0] == runner._batch_system.sync_with_exclusions assert sync_call.args[1] == runner._input_dir assert sync_call.args[2] == work_dir assert sync_call.args[3] == "random.host.org" assert sync_call.args[4] == "localhost" - assert set(sync_call.args[5]) == set(expected_excluded) + assert set(sync_call.args[5]) == set(expected_excluded + expected_ignored) assert sync_call.kwargs["max_tries"] == CFG.runner.retry_tries assert sync_call.kwargs["wait_seconds"] == CFG.runner.retry_wait @@ -788,7 +795,7 @@ def test_runner_set_up_scratch_dir_with_archiver_adds_archive_to_excluded(): # set archiver with a dummy _archive attribute archiver_mock = MagicMock() - archiver_mock._archive = Path("storage") + archiver_mock.archive = Path("storage") runner._archiver = archiver_mock scratch_dir = Path("/scratch") @@ -921,8 +928,8 @@ def test_runner_finalize_with_scratch_and_archiver(mock_logger_info): patch("qq_lib.run.runner.Retryer") as retryer_mock, patch("socket.getfqdn", return_value="host"), patch.object( - Runner, "_get_explicitly_included_files_in_work_dir", return_value=[] - ) as included_mock, + Runner, "_get_excluded_from_input_dir", return_value=[] + ) as excluded_mock, ): runner.finalize() @@ -930,7 +937,7 @@ def test_runner_finalize_with_scratch_and_archiver(mock_logger_info): retryer_mock.assert_called_once() runner._delete_work_dir.assert_called_once() runner._update_info_finished.assert_called_once() - included_mock.assert_called_once() + excluded_mock.assert_called_once() mock_logger_info.assert_any_call("Finalizing the execution.") mock_logger_info.assert_any_call("Job completed with an exit code of 0.") @@ -1617,33 +1624,6 @@ def test_runner_copy_runtime_files_to_input_dir_retry_false(): ) -def test_runner_get_included_files_in_work_dir_resolves_paths(tmp_path): - runner = Runner.__new__(Runner) - - runner._work_dir = tmp_path / "workdir" - runner._work_dir.mkdir() - - abs_file = tmp_path / "abs.txt" - abs_file.write_text("abs") - - rel_file = Path("rel.txt") - (tmp_path / "rel.txt").write_text("rel") - - included = [abs_file, rel_file] - - runner._informer = MagicMock() - runner._informer.info.included_files = included - - expected = [ - (runner._work_dir / abs_file.name).resolve(), - (runner._work_dir / rel_file.name).resolve(), - ] - - result = runner._get_explicitly_included_files_in_work_dir() - - assert result == expected - - @patch("qq_lib.run.runner.socket.getfqdn", return_value="local") def test_runner_copy_files_calls_sync_selected(tmp_path): runner = Runner.__new__(Runner) @@ -1814,3 +1794,221 @@ def test_archive_files_from_work_dir_raises_when_loop_info_is_none(tmp_path: Pat with pytest.raises(QQError, match="Loop info is undefined"): runner._archive_files_from_work_dir() + + +def _make_info( + input_dir: Path, + job_name: str = "job", + excluded_files: list[Path] | None = None, + ignored_files: list[Path] | None = None, + included_files: list[Path] | None = None, +) -> MagicMock: + info = MagicMock() + info.input_dir = input_dir + info.job_name = job_name + info.excluded_files = excluded_files or [] + info.ignored_files = ignored_files or [] + info.included_files = included_files or [] + + return info + + +def _make_runner( + input_dir: Path, + work_dir: Path, + job_name: str = "job", + excluded_files: list[Path] | None = None, + ignored_files: list[Path] | None = None, + included_files: list[Path] | None = None, + archive: Path | None = None, +) -> Runner: + runner = Runner.__new__(Runner) + runner._info_file = input_dir / f"{job_name}.qqinfo" + runner._work_dir = work_dir + + if archive: + archiver = MagicMock() + archiver.archive = archive + runner._archiver = archiver + else: + runner._archiver = None + + informer = MagicMock() + informer.info = _make_info( + input_dir=input_dir, + job_name=job_name, + excluded_files=excluded_files, + ignored_files=ignored_files, + included_files=included_files, + ) + runner._informer = informer + + return runner + + +def test_runner_get_excluded_from_work_dir_collects_all_sources(tmp_path: Path) -> None: + input_dir = tmp_path / "input" + + runner = _make_runner( + input_dir=input_dir, + work_dir=tmp_path / "scratch" / "main", + excluded_files=[input_dir / "old.log"], + ignored_files=[input_dir / "core.dump"], + ) + + assert runner._get_excluded_from_work_dir() == [ + input_dir / "old.log", + input_dir / "core.dump", + input_dir / "job.qqinfo", + (input_dir / "job").with_suffix(CFG.suffixes.qq_out), + ] + + +def test_runner_get_excluded_from_work_dir_appends_archive(tmp_path: Path) -> None: + input_dir = tmp_path / "input" + archive = input_dir / "storage" + + runner = _make_runner( + input_dir=input_dir, + work_dir=tmp_path / "scratch" / "main", + archive=archive, + ) + + assert runner._get_excluded_from_work_dir() == [ + input_dir / "job.qqinfo", + (input_dir / "job").with_suffix(CFG.suffixes.qq_out), + archive, + ] + + +def test_runner_get_excluded_from_work_dir_deduplicates(tmp_path: Path) -> None: + input_dir = tmp_path / "input" + + runner = _make_runner( + input_dir=input_dir, + work_dir=tmp_path / "scratch" / "main", + excluded_files=[input_dir / "old.log", input_dir / "notes.md"], + ignored_files=[input_dir / "old.log", input_dir / "job.qqinfo"], + ) + + assert runner._get_excluded_from_work_dir() == [ + input_dir / "old.log", + input_dir / "notes.md", + input_dir / "job.qqinfo", + (input_dir / "job").with_suffix(CFG.suffixes.qq_out), + ] + + +def test_runner_get_excluded_from_work_dir_ignores_included_files( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + shared = tmp_path / "shared" + + runner = _make_runner( + input_dir=input_dir, + work_dir=tmp_path / "scratch" / "main", + included_files=[shared / "params.itp"], + ) + + assert shared / "params.itp" not in runner._get_excluded_from_work_dir() + + +def test_runner_get_excluded_from_input_dir_relocates_to_work_dir( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + shared = tmp_path / "shared" + work_dir = tmp_path / "scratch" / "main" + + runner = _make_runner( + input_dir=input_dir, + work_dir=work_dir, + ignored_files=[input_dir / "core.dump"], + included_files=[shared / "params.itp"], + ) + + assert runner._get_excluded_from_input_dir() == [ + work_dir / "params.itp", + work_dir / "core.dump", + ] + + +def test_runner_get_excluded_from_input_dir_flattens_nested_paths( + tmp_path: Path, +) -> None: + work_dir = tmp_path / "scratch" / "main" + + runner = _make_runner( + input_dir=tmp_path / "input", + work_dir=work_dir, + included_files=[tmp_path / "shared" / "forcefield" / "deeper" / "params.itp"], + ) + + assert runner._get_excluded_from_input_dir() == [work_dir / "params.itp"] + + +def test_runner_get_excluded_from_input_dir_appends_archive(tmp_path: Path) -> None: + input_dir = tmp_path / "input" + work_dir = tmp_path / "scratch" / "main" + + runner = _make_runner( + input_dir=input_dir, + work_dir=work_dir, + ignored_files=[input_dir / "core.dump"], + archive=input_dir / "storage", + ) + + assert runner._get_excluded_from_input_dir() == [ + work_dir / "core.dump", + work_dir / "storage", + ] + + +def test_runner_get_excluded_from_input_dir_ignores_excluded_files( + tmp_path: Path, +) -> None: + input_dir = tmp_path / "input" + + runner = _make_runner( + input_dir=input_dir, + work_dir=tmp_path / "scratch" / "main", + excluded_files=[input_dir / "old.log"], + ) + + assert runner._get_excluded_from_input_dir() == [] + + +def test_runner_get_excluded_from_input_dir_deduplicates(tmp_path: Path) -> None: + input_dir = tmp_path / "input" + shared = tmp_path / "shared" + work_dir = tmp_path / "scratch" / "main" + + runner = _make_runner( + input_dir=input_dir, + work_dir=work_dir, + ignored_files=[shared / "params.itp", input_dir / "core.dump"], + included_files=[shared / "params.itp"], + ) + + assert runner._get_excluded_from_input_dir() == [ + work_dir / "params.itp", + work_dir / "core.dump", + ] + + +def test_runner_get_excluded_from_input_dir_deduplicates_after_relocating( + tmp_path: Path, +) -> None: + work_dir = tmp_path / "scratch" / "main" + + runner = _make_runner( + input_dir=tmp_path / "input", + work_dir=work_dir, + included_files=[ + tmp_path / "first" / "params.itp", + tmp_path / "second" / "params.itp", + ], + ) + + assert runner._get_excluded_from_input_dir() == [work_dir / "params.itp"] diff --git a/tests/submit/test_submit_factory.py b/tests/submit/test_submit_factory.py index 7424449..9c8f3b6 100644 --- a/tests/submit/test_submit_factory.py +++ b/tests/submit/test_submit_factory.py @@ -181,6 +181,39 @@ def test_submitter_factory_get_include_from_parser(): assert result == parser_includes +def test_submitter_factory_get_ignore_from_command_line(): + mock_parser = MagicMock() + parser_ignores = ["/tmp/file1", "/tmp/file2"] + mock_parser.get_ignore.return_value = parser_ignores + + factory = SubmitterFactory.__new__(SubmitterFactory) + factory._parser = mock_parser + factory._kwargs = {"ignore": "/tmp/file3,/tmp/file4"} + + cli_ignores = ["/tmp/file3", "/tmp/file4"] + + with patch( + "qq_lib.submit.factory.split_string_list", return_value=cli_ignores + ) as mock_split: + result = factory._get_ignore() + + mock_split.assert_called_once_with("/tmp/file3,/tmp/file4") + assert result == cli_ignores + + +def test_submitter_factory_get_ignore_from_parser(): + mock_parser = MagicMock() + parser_ignores = ["/tmp/file1", "/tmp/file2"] + mock_parser.get_ignore.return_value = parser_ignores + + factory = SubmitterFactory.__new__(SubmitterFactory) + factory._parser = mock_parser + factory._kwargs = {} + + result = factory._get_ignore() + assert result == parser_ignores + + def test_submitter_factory_get_loop_info_uses_cli_over_parser(): mock_parser = MagicMock() mock_parser.get_loop_start.return_value = 2 @@ -719,8 +752,9 @@ def test_submitter_factory_make_submitter_standard_job(server): mock_parser.parse = MagicMock() mock_parser.get_job_type.return_value = JobType.STANDARD resources = Resources() - excludes = [Path("/tmp/file1")] - includes = [Path("included_file")] + excludes = ["/tmp/file1"] + includes = ["included_file"] + ignores = ["ignored_file"] depends = [] account = "fake-account" transfer = [Always()] @@ -744,6 +778,7 @@ def test_submitter_factory_make_submitter_standard_job(server): patch.object(factory, "_get_resources", return_value=resources) as mock_get_res, patch.object(factory, "_get_exclude", return_value=excludes) as mock_get_excl, patch.object(factory, "_get_include", return_value=includes) as mock_get_incl, + patch.object(factory, "_get_ignore", return_value=ignores) as mock_get_ignore, patch.object(factory, "_get_depend", return_value=depends) as mock_get_dep, patch.object(factory, "_get_account", return_value=account) as mock_get_acct, patch.object( @@ -777,6 +812,7 @@ def test_submitter_factory_make_submitter_standard_job(server): mock_get_res.assert_called_once_with(BatchSystem, queue, server) mock_get_excl.assert_called_once() mock_get_incl.assert_called_once() + mock_get_ignore.assert_called_once() mock_get_dep.assert_called_once() mock_get_acct.assert_called_once() mock_get_transfer.assert_called_once() @@ -795,6 +831,7 @@ def test_submitter_factory_make_submitter_standard_job(server): loop_info=None, # loop_info is None for STANDARD job exclude=excludes, include=includes, + ignore=ignores, depend=depends, transfer_mode=transfer, server=server, @@ -810,8 +847,9 @@ def test_submitter_factory_make_submitter_loop_job(server): mock_parser.parse = MagicMock() mock_parser.get_job_type.return_value = JobType.LOOP resources = Resources() - excludes = [Path("/tmp/file1")] - includes = [Path("included_file")] + excludes = ["/tmp/file1"] + includes = ["included_file"] + ignores = ["ignored_file"] depends = [] account = None transfer = [Always()] @@ -838,6 +876,7 @@ def test_submitter_factory_make_submitter_loop_job(server): patch.object(factory, "_get_resources", return_value=resources) as mock_get_res, patch.object(factory, "_get_exclude", return_value=excludes) as mock_get_excl, patch.object(factory, "_get_include", return_value=includes) as mock_get_incl, + patch.object(factory, "_get_ignore", return_value=ignores) as mock_get_ignore, patch.object(factory, "_get_depend", return_value=depends) as mock_get_dep, patch.object(factory, "_get_account", return_value=account) as mock_get_acct, patch.object( @@ -871,6 +910,7 @@ def test_submitter_factory_make_submitter_loop_job(server): mock_get_res.assert_called_once_with(BatchSystem, queue, server) mock_get_excl.assert_called_once() mock_get_incl.assert_called_once() + mock_get_ignore.assert_called_once() mock_get_dep.assert_called_once() mock_get_acct.assert_called_once() mock_get_transfer.assert_called_once() @@ -889,6 +929,7 @@ def test_submitter_factory_make_submitter_loop_job(server): loop_info=loop_info, exclude=excludes, include=includes, + ignore=ignores, depend=depends, transfer_mode=transfer, server=server, @@ -904,8 +945,9 @@ def test_submitter_factory_make_submitter_continuous_job(server): mock_parser.parse = MagicMock() mock_parser.get_job_type.return_value = JobType.CONTINUOUS resources = Resources() - excludes = [Path("/tmp/file1")] - includes = [Path("included_file")] + excludes = ["/tmp/file1"] + includes = ["included_file"] + ignores = ["ignored_file"] depends = [] account = None transfer = [Always()] @@ -932,6 +974,7 @@ def test_submitter_factory_make_submitter_continuous_job(server): patch.object(factory, "_get_resources", return_value=resources) as mock_get_res, patch.object(factory, "_get_exclude", return_value=excludes) as mock_get_excl, patch.object(factory, "_get_include", return_value=includes) as mock_get_incl, + patch.object(factory, "_get_ignore", return_value=ignores) as mock_get_ignore, patch.object(factory, "_get_depend", return_value=depends) as mock_get_dep, patch.object(factory, "_get_account", return_value=account) as mock_get_acct, patch.object( @@ -965,6 +1008,7 @@ def test_submitter_factory_make_submitter_continuous_job(server): mock_get_res.assert_called_once_with(BatchSystem, queue, server) mock_get_excl.assert_called_once() mock_get_incl.assert_called_once() + mock_get_ignore.assert_called_once() mock_get_dep.assert_called_once() mock_get_acct.assert_called_once() mock_get_transfer.assert_called_once() @@ -983,6 +1027,7 @@ def test_submitter_factory_make_submitter_continuous_job(server): loop_info=None, exclude=excludes, include=includes, + ignore=ignores, depend=depends, transfer_mode=transfer, server=server, diff --git a/tests/submit/test_submit_parser.py b/tests/submit/test_submit_parser.py index 282d06e..291148f 100644 --- a/tests/submit/test_submit_parser.py +++ b/tests/submit/test_submit_parser.py @@ -205,7 +205,7 @@ def test_parser_get_exclude_calls_split_string_list(): parser = Parser.__new__(Parser) parser._options = {"exclude": "file1,file2"} - mock_split_result = [Path("file1"), Path("file2")] + mock_split_result = ["file1", "file2"] with patch( "qq_lib.submit.parser.split_string_list", return_value=mock_split_result @@ -254,6 +254,29 @@ def test_parser_get_include_calls_split_string_list(): assert result == mock_split_result +def test_parser_get_ignore_empty_list(): + parser = Parser.__new__(Parser) + parser._options = {} + + result = parser.get_ignore() + assert result == [] + + +def test_parser_get_ignore_calls_split_string_list(): + parser = Parser.__new__(Parser) + parser._options = {"ignore": "file1,file2"} + + mock_split_result = ["file1", "file2"] + + with patch( + "qq_lib.submit.parser.split_string_list", return_value=mock_split_result + ) as mock_split: + result = parser.get_ignore() + + mock_split.assert_called_once_with("file1,file2") + assert result == mock_split_result + + def test_parser_get_resources_returns_empty_resources_if_no_matching_options(): parser = Parser.__new__(Parser) parser._options = {"foo": "bar"} # not a Resources field diff --git a/tests/submit/test_submit_submitter.py b/tests/submit/test_submit_submitter.py index 6d7e5ab..02913a5 100644 --- a/tests/submit/test_submit_submitter.py +++ b/tests/submit/test_submit_submitter.py @@ -43,6 +43,7 @@ def test_submitter_init_sets_all_attributes_correctly(tmp_path): resources=Resources(), exclude=["exclude", "/tmp/exclude"], include=["include", "/tmp/include"], + ignore=["ignore", "/tmp/ignore"], transfer_mode=[Always()], server="pbs-m1.metacentrum.cz", interpreter=Interpreter(executable="bash"), @@ -61,6 +62,7 @@ def test_submitter_init_sets_all_attributes_correctly(tmp_path): assert submitter._resources == Resources() assert submitter._exclude == [tmp_path / "exclude", Path("/tmp/exclude")] assert submitter._include == [tmp_path / "include", Path("/tmp/include")] + assert submitter._ignore == [tmp_path / "ignore", Path("/tmp/ignore")] assert submitter._depend == [] assert isinstance(submitter._transfer_mode[0], Always) assert submitter._server == "pbs-m1.metacentrum.cz" @@ -107,6 +109,8 @@ def test_submitter_init_sets_all_optional_arguments_correctly(tmp_path): loop_info = LoopInfo(1, 5, Path("storage"), "job%04d") exclude_files = [str(tmp_path / "file1.txt"), str(tmp_path / "file2.txt")] + include_files = [str(tmp_path / "file3.txt"), str(tmp_path / "file4.txt")] + ignore_files = [str(tmp_path / "file5.txt"), str(tmp_path / "file6.txt")] depend_jobs = [ Depend(DependType.AFTER_SUCCESS, ["12345"]), Depend(DependType.AFTER_START, ["23456"]), @@ -125,6 +129,8 @@ def test_submitter_init_sets_all_optional_arguments_correctly(tmp_path): resources=Resources(), loop_info=loop_info, exclude=exclude_files, + include=include_files, + ignore=ignore_files, depend=depend_jobs, server="fake.server.com", resubmit_from=[WorkHost(), ExplicitHost("node01")], @@ -139,6 +145,8 @@ def test_submitter_init_sets_all_optional_arguments_correctly(tmp_path): assert submitter._input_dir == tmp_path assert submitter._script_name == script.name assert submitter._job_name == "job" + assert submitter._include == [Path(x) for x in include_files] + assert submitter._ignore == [Path(x) for x in ignore_files] assert submitter._info_file == tmp_path / f"job{CFG.suffixes.qq_info}" assert submitter._resources == Resources() assert submitter._exclude == [Path(x) for x in exclude_files] @@ -603,6 +611,7 @@ def test_submitter_submit_calls_all_steps_and_returns_job_id(tmp_path): submitter._loop_info = None submitter._exclude = [] submitter._include = [] + submitter._ignore = [] submitter._depend = [] submitter._transfer_mode = [Success()] submitter._info_file = tmp_path / f"{submitter._job_name}.qqinfo" @@ -657,6 +666,7 @@ def test_submitter_submit(tmp_path): submitter._loop_info = None submitter._exclude = ["exclude1"] submitter._include = ["include1"] + submitter._ignore = ["ignore1"] submitter._depend = [] submitter._transfer_mode = [Always()] submitter._server = "fake.server.com" @@ -716,6 +726,7 @@ def test_submitter_submit(tmp_path): loop_info=submitter._loop_info, excluded_files=submitter._exclude, included_files=submitter._include, + ignored_files=submitter._ignore, depend=submitter._depend, transfer_mode=[Always()], server=submitter._server, From 2bca35c40f48fd32b5ce789109d1ae4c95acf679 Mon Sep 17 00:00:00 2001 From: Ladme Date: Sat, 5 Sep 2026 21:09:30 +0200 Subject: [PATCH 12/19] docs: add note about ignored automatically generated .init file --- src/qq_lib/run/runner.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/src/qq_lib/run/runner.py b/src/qq_lib/run/runner.py index bc28d11..df2ff7a 100644 --- a/src/qq_lib/run/runner.py +++ b/src/qq_lib/run/runner.py @@ -742,6 +742,9 @@ def _archive_files_from_work_dir(self) -> None: self._archiver.create_init_file(loop_info.current + 1) # archive all files matching the archive format + # note that if the .init file created in the previous block of code + # is in a list of ignored files, it will not be included in the archive + # thus, its creation is pointless self._archiver.to_archive(self._work_dir) def _copy_files(self, files: list[Path]): From 17543e19d45c94e27593843acb91da6b97093da8 Mon Sep 17 00:00:00 2001 From: Ladme Date: Sat, 5 Sep 2026 21:15:18 +0200 Subject: [PATCH 13/19] docs: ignore option to changelog --- CHANGELOG.md | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 30247c1..02f7587 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,9 +12,14 @@ - If you explicitly include a file into a working directory using the `--include` submission option, it will no longer be archived even if it matches the archive pattern. - If you explicitly exclude a file from a working directory using the `--exclude` submission option, it will no longer be fetched from the archive, even if it matches the archive pattern for this loop job cycle. +### `--ignore` option + +- As a complement to `--include` and `--exclude`, `--ignore` allows to specify files that should be completely ignored by qq's transfer operations. Ignored files will never be copied to working directory; if they are created in working directory, they will never be transferred to the input directory. If the job is a loop job, ignored files will also never be archived nor fetched from the archive. + ### Bug fixes and other changes - Directories can be now properly archived. +- Archive directory created in a working directory is no longer merged with the actual archive directory in the input directory. - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. From 93ef8c84ce082a38458ba85d24c8a2ec7d6bbfaa Mon Sep 17 00:00:00 2001 From: Ladme Date: Fri, 25 Sep 2026 12:13:27 +0200 Subject: [PATCH 14/19] chore/fix: removing terminating dot from error messages --- CHANGELOG.md | 1 + src/qq_lib/batch/interface/interface.py | 24 ++++----- src/qq_lib/batch/pbs/node.py | 2 +- src/qq_lib/batch/pbs/pbs.py | 52 +++++++++---------- src/qq_lib/batch/pbs/queue.py | 2 +- src/qq_lib/batch/slurm/job.py | 4 +- src/qq_lib/batch/slurm/node.py | 2 +- src/qq_lib/batch/slurm/queue.py | 4 +- src/qq_lib/batch/slurm/slurm.py | 28 +++++----- src/qq_lib/batch/slurmit4i/slurm.py | 20 +++---- src/qq_lib/batch/slurmlumi/slurm.py | 4 +- src/qq_lib/cd/cder.py | 4 +- src/qq_lib/cd/cli.py | 2 +- src/qq_lib/clear/cli.py | 2 +- src/qq_lib/core/command_runner.py | 8 +-- src/qq_lib/core/common.py | 24 ++++----- src/qq_lib/core/config.py | 2 +- src/qq_lib/core/error.py | 38 ++++++++++++-- src/qq_lib/core/error_handlers.py | 10 ++-- src/qq_lib/core/retryer.py | 2 +- src/qq_lib/go/goer.py | 6 +-- src/qq_lib/info/informer.py | 6 +-- src/qq_lib/jobs/cli.py | 2 +- src/qq_lib/kill/killer.py | 6 +-- src/qq_lib/killall/cli.py | 4 +- src/qq_lib/nodes/cli.py | 2 +- src/qq_lib/properties/depend.py | 4 +- src/qq_lib/properties/info.py | 14 ++--- src/qq_lib/properties/interpreter.py | 2 +- src/qq_lib/properties/job_type.py | 2 +- src/qq_lib/properties/loop.py | 22 ++++---- src/qq_lib/properties/resources.py | 2 +- src/qq_lib/properties/size.py | 10 ++-- src/qq_lib/properties/transfer_mode.py | 2 +- src/qq_lib/queues/cli.py | 2 +- src/qq_lib/respawn/respawner.py | 4 +- src/qq_lib/resubmit/resubmitter.py | 6 +-- src/qq_lib/run/cli.py | 15 ++++-- src/qq_lib/run/runner.py | 19 ++++--- src/qq_lib/shebang/cli.py | 4 +- src/qq_lib/stat/cli.py | 2 +- src/qq_lib/submit/cli.py | 6 +-- src/qq_lib/submit/factory.py | 2 +- src/qq_lib/submit/parser.py | 6 +-- src/qq_lib/submit/submitter.py | 6 +-- src/qq_lib/sync/syncer.py | 10 ++-- src/qq_lib/wipe/wiper.py | 16 +++--- tests/batch/interface/test_batch_interface.py | 6 +-- tests/batch/pbs/test_pbs.py | 2 +- tests/batch/pbs/test_pbs_node.py | 2 +- tests/batch/pbs/test_pbs_queue.py | 2 +- tests/batch/slurm/test_slurm.py | 4 +- tests/batch/slurm/test_slurm_node.py | 2 +- tests/batch/slurm/test_slurm_queue.py | 4 +- tests/batch/slurmit4i/test_slurmit4i.py | 4 +- tests/core/test_common.py | 2 +- tests/core/test_error_handlers.py | 16 +++--- tests/go/test_go_goer.py | 4 +- tests/info/test_info_informer.py | 6 +-- tests/kill/test_kill_killer.py | 12 ++--- tests/properties/test_resources.py | 2 +- tests/run/test_run_runner.py | 7 +-- tests/submit/test_submit_factory.py | 2 +- tests/sync/test_sync_syncer.py | 4 +- tests/wipe/test_wipe_wiper.py | 2 +- 65 files changed, 270 insertions(+), 230 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 02f7587..e592543 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -22,6 +22,7 @@ - Archive directory created in a working directory is no longer merged with the actual archive directory in the input directory. - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. +- Error messages reformatted to be easier to compose. --- diff --git a/src/qq_lib/batch/interface/interface.py b/src/qq_lib/batch/interface/interface.py index c813701..ce89832 100644 --- a/src/qq_lib/batch/interface/interface.py +++ b/src/qq_lib/batch/interface/interface.py @@ -485,11 +485,11 @@ def navigate_to_destination(cls, host: str, directory: Path) -> None: # we ignore user exit codes entirely and only treat _SSH_FAIL and _CD_FAIL as errors if result.returncode == cls._SSH_FAIL: raise QQError( - f"Could not reach '{host}:{str(directory)}': Could not connect to host." + f"Could not reach '{host}:{str(directory)}': could not connect to host" ) if result.returncode == cls._CD_FAIL: raise QQError( - f"Could not reach '{host}:{str(directory)}': Could not change directory." + f"Could not reach '{host}:{str(directory)}': could not change directory" ) @classmethod @@ -531,7 +531,7 @@ def read_remote_file(cls, host: str, file: Path) -> str: if result.returncode != 0: raise QQError( - f"Could not read remote file '{file}' on '{host}': {result.stderr.strip()}." + f"Could not read remote file '{file}' on '{host}': {result.stderr.strip()}" ) return result.stdout @@ -573,7 +573,7 @@ def write_remote_file(cls, host: str, file: Path, content: str) -> None: if result.returncode != 0: raise QQError( - f"Could not write to remote file '{file}' on '{host}': {result.stderr.strip()}." + f"Could not write to remote file '{file}' on '{host}': {result.stderr.strip()}" ) @classmethod @@ -612,7 +612,7 @@ def make_remote_dir(cls, host: str, directory: Path) -> None: if result.returncode != 0: raise QQError( - f"Could not make remote directory '{directory}' on '{host}': {result.stderr.strip()}." + f"Could not make remote directory '{directory}' on '{host}': {result.stderr.strip()}" ) @classmethod @@ -654,7 +654,7 @@ def list_remote_dir(cls, host: str, directory: Path) -> list[Path]: if result.returncode != 0: raise QQError( - f"Could not list remote directory '{directory}' on '{host}': {result.stderr.strip()}." + f"Could not list remote directory '{directory}' on '{host}': {result.stderr.strip()}" ) # split by newline and filter out empty lines @@ -699,7 +699,7 @@ def delete_remote_dir(cls, host: str, directory: Path) -> None: if result.returncode != 0: raise QQError( - f"Could not delete remote directory '{directory}' on '{host}': {result.stderr.strip()}." + f"Could not delete remote directory '{directory}' on '{host}': {result.stderr.strip()}" ) @classmethod @@ -744,7 +744,7 @@ def move_remote_files( if result.returncode != 0: raise QQError( - f"Could not move files on a remote host '{host}': {result.stderr.strip()}." + f"Could not move files on a remote host '{host}': {result.stderr.strip()}" ) @classmethod @@ -977,7 +977,7 @@ def _navigate_same_host(cls, directory: Path) -> None: logger.debug("Current host is the same as target host. Using 'cd'.") if not directory.is_dir(): raise QQError( - f"Could not reach '{socket.getfqdn()}:{str(directory)}': Could not change directory." + f"Could not reach '{socket.getfqdn()}:{str(directory)}': could not change directory" ) subprocess.run(["bash"], cwd=directory) @@ -1007,7 +1007,7 @@ def _translate_move_command(cls, files: list[Path], moved_files: list[Path]) -> """ if len(files) != len(moved_files): raise QQError( - "The provided 'files' and 'moved_files' must have the same length." + "The provided 'files' and 'moved_files' must have the same length" ) mv_commands: list[str] = [] @@ -1159,10 +1159,10 @@ def _run_rsync( ) except subprocess.TimeoutExpired as e: raise QQError( - f"Could not rsync files between '{src}' and '{dest}': Connection timed out after {CFG.timeouts.rsync} seconds." + f"Could not rsync files between '{src}' and '{dest}': connection timed out after {CFG.timeouts.rsync} seconds" ) from e if result.returncode != 0: raise QQError( - f"Could not rsync files between '{src}' and '{dest}': {result.stderr.strip()}." + f"Could not rsync files between '{src}' and '{dest}': {result.stderr.strip()}" ) diff --git a/src/qq_lib/batch/pbs/node.py b/src/qq_lib/batch/pbs/node.py index 9d22b6d..3115a2b 100644 --- a/src/qq_lib/batch/pbs/node.py +++ b/src/qq_lib/batch/pbs/node.py @@ -115,7 +115,7 @@ def update(self) -> None: ) if result.returncode != 0: - raise QQError(f"Node '{self._name}' does not exist.") + raise QQError(f"Node '{self._name}' does not exist") self._info = parse_pbs_dump_to_dictionary(result.stdout) diff --git a/src/qq_lib/batch/pbs/pbs.py b/src/qq_lib/batch/pbs/pbs.py index 5ecb48d..dfa1678 100644 --- a/src/qq_lib/batch/pbs/pbs.py +++ b/src/qq_lib/batch/pbs/pbs.py @@ -138,7 +138,7 @@ def job_submit( if result.returncode != 0: raise QQError( - f"Failed to submit script '{str(script)}': {result.stderr.strip()}." + f"Failed to submit script '{str(script)}': {result.stderr.strip()}" ) return result.stdout.strip() @@ -159,7 +159,7 @@ def job_kill(cls, job_id: str) -> None: ) if result.returncode != 0: - raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}.") + raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}") @classmethod def job_kill_force(cls, job_id: str) -> None: @@ -177,7 +177,7 @@ def job_kill_force(cls, job_id: str) -> None: ) if result.returncode != 0: - raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}.") + raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}") @classmethod def get_batch_job(cls, job_id: str) -> PBSJob: @@ -269,7 +269,7 @@ def get_queues(cls, server: str | None = None) -> list[PBSQueue]: if result.returncode != 0: raise QQError( - f"Could not retrieve information about queues: {result.stderr.strip()}." + f"Could not retrieve information about queues: {result.stderr.strip()}" ) queues = [] @@ -298,7 +298,7 @@ def get_nodes(cls, server: str | None = None) -> list[PBSNode]: if result.returncode != 0: raise QQError( - f"Could not retrieve information about nodes: {result.stderr.strip()}." + f"Could not retrieve information about nodes: {result.stderr.strip()}" ) queues = [] @@ -326,7 +326,7 @@ def read_remote_file(cls, host: str, file: Path) -> str: try: return file.read_text() except Exception as e: - raise QQError(f"Could not read file '{file}': {e}.") from e + raise QQError(f"Could not read file '{file}': {e}") from e else: # otherwise, we fall back to the default implementation logger.debug(f"Reading a remote file '{file}' on '{host}'.") @@ -341,7 +341,7 @@ def write_remote_file(cls, host: str, file: Path, content: str) -> None: try: file.write_text(content) except Exception as e: - raise QQError(f"Could not write file '{file}': {e}.") from e + raise QQError(f"Could not write file '{file}': {e}") from e else: # otherwise, we fall back to the default implementation logger.debug(f"Writing a remote file '{file}' on '{host}'.") @@ -355,9 +355,7 @@ def make_remote_dir(cls, host: str, directory: Path) -> None: try: directory.mkdir(exist_ok=True) except Exception as e: - raise QQError( - f"Could not create a directory '{directory}': {e}." - ) from e + raise QQError(f"Could not create a directory '{directory}': {e}") from e else: # otherwise we fall back to the default implementation logger.debug(f"Creating a directory '{directory}' on '{host}'.") @@ -371,7 +369,7 @@ def list_remote_dir(cls, host: str, directory: Path) -> list[Path]: try: return list(directory.iterdir()) except Exception as e: - raise QQError(f"Could not list a directory '{directory}': {e}.") from e + raise QQError(f"Could not list a directory '{directory}': {e}") from e else: # otherwise we fall back to the default implementation logger.debug(f"Listing a directory '{directory}' on '{host}'.") @@ -385,7 +383,7 @@ def delete_remote_dir(cls, host: str, directory: Path) -> None: try: shutil.rmtree(directory) except Exception as e: - raise QQError(f"Could not delete directory '{directory}': {e}.") from e + raise QQError(f"Could not delete directory '{directory}': {e}") from e else: # otherwise we fall back to the default implementation logger.debug(f"Deleting a directory '{directory}' on '{host}'.") @@ -463,7 +461,7 @@ def transform_resources( ) if not resources.work_dir: raise QQError( - "Work-dir is not set after filling in default attributes. This is a bug." + "Work-dir is not set after filling in default attributes. This is a bug, please report it" ) # sanity check input_dir @@ -512,7 +510,7 @@ def transform_resources( # unknown work-dir type raise QQError( - f"Unknown working directory type specified: work-dir='{resources.work_dir}'. Supported types for {cls.env_name()} are: '{' '.join(cls.get_supported_work_dir_types())}'." + f"Unknown working directory type specified: work-dir='{resources.work_dir}'. Supported types for {cls.env_name()} are: '{' '.join(cls.get_supported_work_dir_types())}'" ) @classmethod @@ -575,19 +573,19 @@ def _shared_guard( elif not res.uses_scratch(): # if job directory is used as working directory, it must always be shared raise QQError( - "Job was requested to run directly in the submission directory (work-dir='job_dir' or 'input_dir'), but submission is done from a local filesystem." + "Job was requested to run directly in the submission directory (work-dir='job_dir' or 'input_dir'), but submission is done from a local filesystem" ) elif server is not None: # if we are submitting to a different server raise QQError( - f"Job was requested to be submitted to server '{server}' which is potentially non-local, but the submission is done from a local filesystem." + f"Job was requested to be submitted to server '{server}' which is potentially non-local, but the submission is done from a local filesystem" ) elif ( remote_host is not None and socket.getfqdn(remote_host) != socket.getfqdn() ): # if we are submitting from a different host than the current one raise QQError( - f"Job was requested to be submitted from host '{remote_host}', but the submission is done from a local filesystem." + f"Job was requested to be submitted from host '{remote_host}', but the submission is done from a local filesystem" ) @classmethod @@ -758,18 +756,18 @@ def _translate_per_chunk_resources(cls, res: Resources) -> list[str]: # sanity checking per-chunk resources if not res.nnodes: raise QQError( - "Attribute 'nnodes' should not be undefined. This is a bug, please report it." + "Attribute 'nnodes' should not be undefined. This is a bug, please report it" ) if res.nnodes == 0: - raise QQError("Attribute 'nnodes' cannot be 0.") + raise QQError("Attribute 'nnodes' cannot be 0") if res.ncpus and res.ncpus != 0 and res.ncpus % res.nnodes != 0: raise QQError( - f"Attribute 'ncpus' ({res.ncpus}) must be divisible by 'nnodes' ({res.nnodes})." + f"Attribute 'ncpus' ({res.ncpus}) must be divisible by 'nnodes' ({res.nnodes})" ) if res.ngpus and res.ngpus != 0 and res.ngpus % res.nnodes != 0: raise QQError( - f"Attribute 'ngpus' ({res.ngpus}) must be divisible by 'nnodes' ({res.nnodes})." + f"Attribute 'ngpus' ({res.ngpus}) must be divisible by 'nnodes' ({res.nnodes})" ) # translate per-chunk resources @@ -798,12 +796,12 @@ def _translate_per_chunk_resources(cls, res: Resources) -> list[str]: ) else: raise QQError( - "Attribute 'mem-per-cpu' requires attributes 'ncpus' or 'ncpus-per-node' to be defined." + "Attribute 'mem-per-cpu' requires attributes 'ncpus' or 'ncpus-per-node' to be defined" ) else: # memory not set in any way raise QQError( - "None of the attributes 'mem', 'mem-per-node', or 'mem-per-cpu' is defined." + "None of the attributes 'mem', 'mem-per-node', or 'mem-per-cpu' is defined" ) if res.ngpus: @@ -847,11 +845,11 @@ def _translate_work_dir(cls, res: Resources) -> str | None: return f"{res.work_dir}={(res.work_size_per_cpu * res.ncpus_per_node).to_str_exact()}" raise QQError( - "Attribute 'work-size-per-cpu' requires attributes 'ncpus' or 'ncpus-per-node' to be defined." + "Attribute 'work-size-per-cpu' requires attributes 'ncpus' or 'ncpus-per-node' to be defined" ) raise QQError( - "None of the attributes 'work-size', 'work-size-per-node', or 'work-size-per-cpu' is defined." + "None of the attributes 'work-size', 'work-size-per-node', or 'work-size-per-cpu' is defined" ) @classmethod @@ -1029,7 +1027,7 @@ def _sync_directories( sync_function(src_dir, dest_dir, src, dest, files) else: raise QQError( - f"The source '{src_host}' and destination '{dest_host}' cannot be both remote." + f"The source '{src_host}' and destination '{dest_host}' cannot be both remote" ) @classmethod @@ -1076,7 +1074,7 @@ def _get_batch_jobs_using_command( if not ignore_exit_code and result.returncode != 0: raise QQError( # standard error is written to stdout - f"Could not retrieve information about jobs: {result.stdout.strip()}." + f"Could not retrieve information about jobs: {result.stdout.strip()}" ) jobs = [] diff --git a/src/qq_lib/batch/pbs/queue.py b/src/qq_lib/batch/pbs/queue.py index 8b5c120..f8d059b 100644 --- a/src/qq_lib/batch/pbs/queue.py +++ b/src/qq_lib/batch/pbs/queue.py @@ -109,7 +109,7 @@ def update(self) -> None: ) if result.returncode != 0: - raise QQError(f"Queue '{self._name}' does not exist.") + raise QQError(f"Queue '{self._name}' does not exist") self._info = parse_pbs_dump_to_dictionary(result.stdout) self._set_attributes() diff --git a/src/qq_lib/batch/slurm/job.py b/src/qq_lib/batch/slurm/job.py index 5b6b12d..c4fcc45 100644 --- a/src/qq_lib/batch/slurm/job.py +++ b/src/qq_lib/batch/slurm/job.py @@ -408,7 +408,7 @@ def from_sacct_string(cls, string: str) -> Self: split = string.split("|") if len(fields) != len(split): raise QQError( - f"Number of items in a sacct string '{string}' ('{len(split)}') does not match the expected number of items ('{len(fields)}'). This is a bug, please report it!" + f"Number of items in a sacct string '{string}' ('{len(split)}') does not match the expected number of items ('{len(fields)}'). This is a bug, please report it" ) info: dict[str, str] = dict(zip(fields, split)) @@ -443,7 +443,7 @@ def _step_from_sacct_string(cls, string: str) -> Self: split = string.split("|") if len(fields) != len(split): raise QQError( - f"Number of items in a sacct string for a slurm step '{string}' ('{len(split)}') does not match the expected number of items ('{len(fields)}'). This is a bug, please report it!" + f"Number of items in a sacct string for a slurm step '{string}' ('{len(split)}') does not match the expected number of items ('{len(fields)}'). This is a bug, please report it" ) info: dict[str, str] = dict(zip(fields, split)) diff --git a/src/qq_lib/batch/slurm/node.py b/src/qq_lib/batch/slurm/node.py index 5240c12..79db19d 100644 --- a/src/qq_lib/batch/slurm/node.py +++ b/src/qq_lib/batch/slurm/node.py @@ -43,7 +43,7 @@ def update(self) -> None: ) if result.returncode != 0: - raise QQError(f"Node '{self._name}' does not exist.") + raise QQError(f"Node '{self._name}' does not exist") self._info = parse_slurm_dump_to_dictionary(result.stdout) diff --git a/src/qq_lib/batch/slurm/queue.py b/src/qq_lib/batch/slurm/queue.py index 8a28e72..97eccac 100644 --- a/src/qq_lib/batch/slurm/queue.py +++ b/src/qq_lib/batch/slurm/queue.py @@ -117,7 +117,7 @@ def update(self) -> None: ) if result.returncode != 0: - raise QQError(f"Queue '{self._name}' does not exist.") + raise QQError(f"Queue '{self._name}' does not exist") self._info = parse_slurm_dump_to_dictionary(result.stdout) self._set_job_numbers() @@ -268,7 +268,7 @@ def _set_job_numbers(self) -> None: if result.returncode != 0: raise QQError( - f"Could not get job numbers for queue '{self._name}': {result.stderr.strip()}." + f"Could not get job numbers for queue '{self._name}': {result.stderr.strip()}" ) for line in result.stdout.splitlines(): diff --git a/src/qq_lib/batch/slurm/slurm.py b/src/qq_lib/batch/slurm/slurm.py index 164ea96..c964341 100644 --- a/src/qq_lib/batch/slurm/slurm.py +++ b/src/qq_lib/batch/slurm/slurm.py @@ -108,7 +108,7 @@ def job_submit( if result.returncode != 0: raise QQError( - f"Failed to submit script '{str(script)}': {result.stderr.strip()}." + f"Failed to submit script '{str(script)}': {result.stderr.strip()}" ) return result.stdout.split()[-1] @@ -129,7 +129,7 @@ def job_kill(cls, job_id: str) -> None: ) if result.returncode != 0: - raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}.") + raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}") @classmethod def job_kill_force(cls, job_id: str) -> None: @@ -147,7 +147,7 @@ def job_kill_force(cls, job_id: str) -> None: ) if result.returncode != 0: - raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}.") + raise QQError(f"Failed to kill job '{job_id}': {result.stderr.strip()}") @classmethod def get_batch_job(cls, job_id: str) -> SlurmJob: @@ -286,7 +286,7 @@ def get_queues(cls, server: str | None = None) -> list[SlurmQueue]: if result.returncode != 0: raise QQError( - f"Could not retrieve information about queues: {result.stderr.strip()}." + f"Could not retrieve information about queues: {result.stderr.strip()}" ) queues = [] @@ -315,7 +315,7 @@ def get_nodes(cls, server: str | None = None) -> list[SlurmNode]: if result.returncode != 0: raise QQError( - f"Could not retrieve information about nodes: {result.stderr.strip()}." + f"Could not retrieve information about nodes: {result.stderr.strip()}" ) nodes = [] @@ -470,7 +470,7 @@ def _translate_submit( for k, v in res.props.items(): if v != "true": raise QQError( - f"Slurm only supports properties with a value of 'true', not '{k}={v}'." + f"Slurm only supports properties with a value of 'true', not '{k}={v}'" ) constraints.append(k) @@ -535,18 +535,18 @@ def _translate_per_chunk_resources(cls, res: Resources) -> list[str]: # sanity checking per-chunk resources if res.nnodes is None: raise QQError( - "Attribute 'nnodes' should not be undefined. This is a bug, please report it." + "Attribute 'nnodes' should not be undefined. This is a bug, please report it" ) if res.nnodes == 0: - raise QQError("Attribute 'nnodes' cannot be 0.") + raise QQError("Attribute 'nnodes' cannot be 0") if res.ncpus and res.ncpus != 0 and res.ncpus % res.nnodes != 0: raise QQError( - f"Attribute 'ncpus' ({res.ncpus}) must be divisible by 'nnodes' ({res.nnodes})." + f"Attribute 'ncpus' ({res.ncpus}) must be divisible by 'nnodes' ({res.nnodes})" ) if res.ngpus and res.ngpus != 0 and res.ngpus % res.nnodes != 0: raise QQError( - f"Attribute 'ngpus' ({res.ngpus}) must be divisible by 'nnodes' ({res.nnodes})." + f"Attribute 'ngpus' ({res.ngpus}) must be divisible by 'nnodes' ({res.nnodes})" ) # translate per-chunk resources @@ -569,7 +569,7 @@ def _translate_per_chunk_resources(cls, res: Resources) -> list[str]: else: # memory not set in any way raise QQError( - "None of the attributes 'mem', 'mem-per-node', or 'mem-per-cpu' is defined." + "None of the attributes 'mem', 'mem-per-node', or 'mem-per-cpu' is defined" ) if res.ngpus: @@ -665,7 +665,7 @@ def _get_batch_jobs_using_sacct_command(cls, command: str) -> list[SlurmJob]: if result.returncode != 0: raise QQError( - f"Could not retrieve information about jobs: {result.stderr.strip()}." + f"Could not retrieve information about jobs: {result.stderr.strip()}" ) jobs = [] @@ -709,7 +709,7 @@ def _get_batch_jobs_using_squeue_command(cls, command: str) -> list[SlurmJob]: if result.returncode != 0: raise QQError( - f"Could not retrieve information about jobs: {result.stderr.strip()}." + f"Could not retrieve information about jobs: {result.stderr.strip()}" ) ids = [line.strip() for line in result.stdout.split("\n") if line.strip()] @@ -743,6 +743,6 @@ def get_job(job_id: str) -> SlurmJob: jobs.append(future.result()) except Exception as e: id = future_to_id[future] - raise QQError(f"Failed to load job {id}: {e}.") from e + raise QQError(f"Failed to load job {id}: {e}") from e return jobs diff --git a/src/qq_lib/batch/slurmit4i/slurm.py b/src/qq_lib/batch/slurmit4i/slurm.py index b91dcd2..4e37238 100644 --- a/src/qq_lib/batch/slurmit4i/slurm.py +++ b/src/qq_lib/batch/slurmit4i/slurm.py @@ -38,7 +38,7 @@ def is_available(cls) -> bool: @classmethod def create_work_dir_on_scratch(cls, job_id: str) -> Path: if not (account := os.environ.get(CFG.env_vars.slurm_job_account)): - raise QQError(f"No account is defined for job '{job_id}'.") + raise QQError(f"No account is defined for job '{job_id}'") user = getpass.getuser() @@ -87,7 +87,7 @@ def read_remote_file(cls, host: str, file: Path) -> str: try: return file.read_text() except Exception as e: - raise QQError(f"Could not read file '{file}': {e}.") from e + raise QQError(f"Could not read file '{file}': {e}") from e @classmethod def write_remote_file(cls, host: str, file: Path, content: str) -> None: @@ -96,7 +96,7 @@ def write_remote_file(cls, host: str, file: Path, content: str) -> None: try: file.write_text(content) except Exception as e: - raise QQError(f"Could not write file '{file}': {e}.") from e + raise QQError(f"Could not write file '{file}': {e}") from e @classmethod def make_remote_dir(cls, host: str, directory: Path) -> None: @@ -105,7 +105,7 @@ def make_remote_dir(cls, host: str, directory: Path) -> None: try: directory.mkdir(exist_ok=True) except Exception as e: - raise QQError(f"Could not create a directory '{directory}': {e}.") from e + raise QQError(f"Could not create a directory '{directory}': {e}") from e @classmethod def list_remote_dir(cls, host: str, directory: Path) -> list[Path]: @@ -114,7 +114,7 @@ def list_remote_dir(cls, host: str, directory: Path) -> list[Path]: try: return list(directory.iterdir()) except Exception as e: - raise QQError(f"Could not list a directory '{directory}': {e}.") from e + raise QQError(f"Could not list a directory '{directory}': {e}") from e @classmethod def delete_remote_dir(cls, host: str, directory: Path) -> None: @@ -123,7 +123,7 @@ def delete_remote_dir(cls, host: str, directory: Path) -> None: try: shutil.rmtree(directory) except Exception as e: - raise QQError(f"Could not delete directory '{directory}': {e}.") from e + raise QQError(f"Could not delete directory '{directory}': {e}") from e @classmethod def move_remote_files( @@ -131,7 +131,7 @@ def move_remote_files( ) -> None: if len(files) != len(moved_files): raise QQError( - "The provided 'files' and 'moved_files' must have the same length." + "The provided 'files' and 'moved_files' must have the same length" ) # always on shared storage @@ -187,12 +187,12 @@ def transform_resources( ) if not resources.work_dir: raise QQError( - "Work-dir is not set after filling in default attributes. This is a bug." + "Work-dir is not set after filling in default attributes. This is a bug, please report it" ) if provided_resources.work_size_per_cpu or provided_resources.work_size: logger.warning( - "Setting work-size is not supported in this environment. Working directory has a virtually unlimited capacity." + "Setting work-size is not supported in this environment. Working directory has a virtually unlimited capacity" ) if not any( @@ -200,7 +200,7 @@ def transform_resources( for dir in cls.get_supported_work_dir_types() ): raise QQError( - f"Unknown working directory type specified: work-dir='{resources.work_dir}'. Supported types for {cls.env_name()} are: {' '.join(cls.get_supported_work_dir_types())}." + f"Unknown working directory type specified: work-dir='{resources.work_dir}'. Supported types for {cls.env_name()} are: {' '.join(cls.get_supported_work_dir_types())}" ) return resources diff --git a/src/qq_lib/batch/slurmlumi/slurm.py b/src/qq_lib/batch/slurmlumi/slurm.py index dcf34d7..65b3fcf 100644 --- a/src/qq_lib/batch/slurmlumi/slurm.py +++ b/src/qq_lib/batch/slurmlumi/slurm.py @@ -61,12 +61,12 @@ def job_submit( @classmethod def create_work_dir_on_scratch(cls, job_id: str) -> Path: if not (account := os.environ.get(CFG.env_vars.slurm_job_account)): - raise QQError(f"No account is defined for job '{job_id}'.") + raise QQError(f"No account is defined for job '{job_id}'") # get the storage type (scratch or flash) if not (storage_type := os.environ.get(CFG.env_vars.lumi_scratch_type)): raise QQError( - f"Environment variable '{CFG.env_vars.lumi_scratch_type}' is not defined. This is a bug!" + f"Environment variable '{CFG.env_vars.lumi_scratch_type}' is not defined. This is a bug, please report it" ) user = getpass.getuser() diff --git a/src/qq_lib/cd/cder.py b/src/qq_lib/cd/cder.py index 5d136ea..d50de2f 100644 --- a/src/qq_lib/cd/cder.py +++ b/src/qq_lib/cd/cder.py @@ -76,9 +76,9 @@ def _get_input_dir_from_job_id[ job_info: BatchJobInterface = BatchSystem.get_batch_job(job_id) if job_info.is_empty(): - raise QQError(f"Job '{job_id}' does not exist.") + raise QQError(f"Job '{job_id}' does not exist") if not (input_dir := job_info.get_input_dir()): - raise QQError(f"Job '{job_id}' has an unknown input directory.") + raise QQError(f"Job '{job_id}' has an unknown input directory") return input_dir diff --git a/src/qq_lib/cd/cli.py b/src/qq_lib/cd/cli.py index 4ff76b4..f485a3f 100644 --- a/src/qq_lib/cd/cli.py +++ b/src/qq_lib/cd/cli.py @@ -44,7 +44,7 @@ def cd(job: str) -> NoReturn: print(cder.cd()) sys.exit(0) except QQError as e: - logger.error(e) + logger.error(e.terminated) sys.exit(CFG.exit_codes.default) except Exception as e: logger.critical(e, exc_info=True, stack_info=True) diff --git a/src/qq_lib/clear/cli.py b/src/qq_lib/clear/cli.py index b7e09de..2b1722c 100644 --- a/src/qq_lib/clear/cli.py +++ b/src/qq_lib/clear/cli.py @@ -54,7 +54,7 @@ def clear(dir: tuple[Path, ...], force: bool) -> NoReturn: clearer.clear(force) sys.exit(0) except QQError as e: - logger.error(e) + logger.error(e.terminated) sys.exit(CFG.exit_codes.default) except Exception as e: logger.critical(e, exc_info=True, stack_info=True) diff --git a/src/qq_lib/core/command_runner.py b/src/qq_lib/core/command_runner.py index 4549a1e..723a618 100644 --- a/src/qq_lib/core/command_runner.py +++ b/src/qq_lib/core/command_runner.py @@ -125,7 +125,7 @@ def run(self) -> NoReturn: self._run_pipeline(targets) sys.exit(0) except QQError as e: - self._logger.error(e) + self._logger.error(e.terminated) sys.exit(CFG.exit_codes.default) except Exception as e: self._logger.critical(e, exc_info=True, stack_info=True) @@ -173,8 +173,8 @@ def _build_targets(self) -> list[Callable[[], Informer]]: if not targets: if not self._job_ids and not self._all: - raise QQError("No qq job info file found.") - raise QQError("No jobs found.") + raise QQError("No qq job info file found") + raise QQError("No jobs found") return targets @@ -285,7 +285,7 @@ def prepare(index: int, target: Callable[[], Informer]) -> None: self._execute(result) else: raise ValueError( - f"Unexpected result type: {type(result)}. This is a bug, please report it." + f"Unexpected result type: {type(result)}. This is a bug, please report it" ) def _execute(self, informer: Informer) -> None: diff --git a/src/qq_lib/core/common.py b/src/qq_lib/core/common.py index 970c5d9..a7b0766 100644 --- a/src/qq_lib/core/common.py +++ b/src/qq_lib/core/common.py @@ -118,9 +118,9 @@ def get_info_file(directory: Path) -> Path: """ info_files = get_info_files(directory) if len(info_files) == 0: - raise QQError("No qq job info file found.") + raise QQError("No qq job info file found") if len(info_files) > 1: - raise QQError("Multiple qq job info files found.") + raise QQError("Multiple qq job info files found") return info_files[0] @@ -167,10 +167,10 @@ def get_info_file_from_job_id(job_id: str) -> Path: job_info: BatchJobInterface = BatchSystem.get_batch_job(job_id) if job_info.is_empty(): - raise QQError(f"Job '{job_id}' does not exist.") + raise QQError(f"Job '{job_id}' does not exist") if not (path := job_info.get_info_file()): - raise QQError(f"Job '{job_id}' is not a valid qq job.") + raise QQError(f"Job '{job_id}' is not a valid qq job") return path @@ -203,7 +203,7 @@ def get_info_files_from_job_id_or_dir(job_id: str | None) -> list[Path]: if missing: raise QQError( - f"Info file for job '{job_id}' does not exist or is not reachable." + f"Info file for job '{job_id}' does not exist or is not reachable" ) return [info_file] @@ -211,7 +211,7 @@ def get_info_files_from_job_id_or_dir(job_id: str | None) -> list[Path]: # get info files from the directory info_files = get_info_files(Path()) if not info_files: - raise QQError("No qq job info file found.") + raise QQError("No qq job info file found") return info_files @@ -353,7 +353,7 @@ def hhmmss_to_duration(timestr: str) -> timedelta: pattern = re.compile(r"^\s*(\d+):([0-5]?\d):([0-5]?\d)\s*$") match = pattern.fullmatch(timestr) if not match: - raise QQError(f"Invalid HH:MM:SS time string '{timestr}'.") + raise QQError(f"Invalid HH:MM:SS time string '{timestr}'") hours, minutes, seconds = map(int, match.groups()) @@ -382,7 +382,7 @@ def dhhmmss_to_duration(timestr: str) -> timedelta: pattern = re.compile(r"^\s*(?:(\d+)-)?(\d+):([0-5]?\d):([0-5]?\d)\s*$") match = pattern.fullmatch(timestr) if not match: - raise QQError(f"Invalid D-HH:MM:SS time string '{timestr}'.") + raise QQError(f"Invalid D-HH:MM:SS time string '{timestr}'") days_str, hours_str, minutes_str, seconds_str = match.groups() days = int(days_str) if days_str else 0 @@ -450,7 +450,7 @@ def convert_absolute_to_relative(files: list[Path], target: Path) -> list[Path]: # file must starts with the target path if file_parts[: len(target_parts)] != target_parts: - raise QQError(f"Item '{file}' is not in target directory '{target}'.") + raise QQError(f"Item '{file}' is not in target directory '{target}'") # create a relative path rel_path = Path(*file_parts[len(target_parts) :]) @@ -494,7 +494,7 @@ def wdhms_to_hhmmss(timestr: str) -> str: # validation full_pattern = re.compile(r"^\s*(?:\d+\s*[wdhms]\s*)+$", re.IGNORECASE) if not full_pattern.fullmatch(timestr): - raise QQError(f"Invalid time string '{timestr}'.") + raise QQError(f"Invalid time string '{timestr}'") # extract tokens token_pattern = re.compile(r"(\d+)\s*([wdhms])", re.IGNORECASE) @@ -554,7 +554,7 @@ def hhmmss_to_wdhms(timestr: str) -> str: pattern = re.compile(r"^\s*(\d+):([0-5]?\d):([0-5]?\d)\s*$") match = pattern.fullmatch(timestr) if not match: - raise QQError(f"Invalid HH:MM:SS time string '{timestr}'.") + raise QQError(f"Invalid HH:MM:SS time string '{timestr}'") hours, minutes, seconds = map(int, match.groups()) total_seconds = hours * 3600 + minutes * 60 + seconds @@ -825,7 +825,7 @@ def expand_pattern(pattern: str, directory: Path) -> list[Path]: try: return sorted(anchor.glob(str(relative))) except Exception as e: - raise QQError(f"Could not expand pattern '{pattern}': {e}.") from e + raise QQError(f"Could not expand pattern '{pattern}': {e}") from e def relocate_by_name(files: Iterable[Path], directory: Path) -> list[Path]: diff --git a/src/qq_lib/core/config.py b/src/qq_lib/core/config.py index 640ac1b..7c4c648 100644 --- a/src/qq_lib/core/config.py +++ b/src/qq_lib/core/config.py @@ -560,7 +560,7 @@ def load(cls, config_path: Path | None = None) -> Self: return _dict_to_dataclass(cls, config_data) except Exception as e: print( - f"[ FATAL CONFIGURATION ERROR ] Could not read qq config '{config_path}': {e}." + f"[ FATAL CONFIGURATION ERROR ] Could not read qq config '{config_path}': {e}" ) print( "[ FATAL CONFIGURATION ERROR ] Falling back to default configuration.\n\n" diff --git a/src/qq_lib/core/error.py b/src/qq_lib/core/error.py index 86d6b4a..d102ebc 100644 --- a/src/qq_lib/core/error.py +++ b/src/qq_lib/core/error.py @@ -16,8 +16,40 @@ logger = get_logger(__name__) +_TERMINAL_PUNCTUATION: str = ".!?:;" -class QQError(Exception): + +def terminate(message: str) -> str: + """ + Add a terminal dot to a message unless it already ends with punctuation. + + Args: + message (str): Message to terminate. + + Returns: + str: The message ending with a single punctuation mark. An empty or + blank message is returned as an empty string. + """ + stripped = message.rstrip() + if not stripped: + return "" + if stripped[-1] in _TERMINAL_PUNCTUATION: + return stripped + return f"{stripped}." + + +class QQTerminatedMixin: + """Provides a terminally punctuated form of an exception message.""" + + @property + def terminated(self) -> str: + """ + str: The exception message with a terminal dot added. + """ + return terminate(str(self)) + + +class QQError(QQTerminatedMixin, Exception): """Common exception type for all recoverable qq errors.""" exit_code = CFG.exit_codes.default @@ -35,7 +67,7 @@ class QQNotSuitableError(QQError): pass -class QQRunFatalError(Exception): +class QQRunFatalError(QQTerminatedMixin, Exception): """ Raised when qq runner is unable to load a qq info file. @@ -45,7 +77,7 @@ class QQRunFatalError(Exception): exit_code = CFG.exit_codes.qq_run_fatal -class QQRunCommunicationError(Exception): +class QQRunCommunicationError(QQTerminatedMixin, Exception): """ Raised when qq runner detects an inconsistency between the information it has and the information in the corresponding qq info file. diff --git a/src/qq_lib/core/error_handlers.py b/src/qq_lib/core/error_handlers.py index e902a5f..9233e90 100644 --- a/src/qq_lib/core/error_handlers.py +++ b/src/qq_lib/core/error_handlers.py @@ -16,7 +16,7 @@ from qq_lib.core.command_runner import CommandRunner from .config import CFG -from .error import QQNotSuitableError +from .error import QQNotSuitableError, terminate from .logger import get_logger logger = get_logger(__name__) @@ -28,11 +28,11 @@ def handle_not_suitable_error( ) -> None: """Handle cases where a job is unsuitable for a qq operation.""" if runner.n_jobs == 1: - logger.error(exception) + logger.error(terminate(str(exception))) sys.exit(CFG.exit_codes.default) if runner.n_jobs > 1: - logger.info(exception) + logger.info(terminate(str(exception))) if ( sum( @@ -50,7 +50,7 @@ def handle_job_mismatch_error( _runner: CommandRunner, ) -> NoReturn: """Handle cases where the provided job ID does not match the qq info file.""" - logger.error(exception) + logger.error(terminate(str(exception))) sys.exit(CFG.exit_codes.default) @@ -59,7 +59,7 @@ def handle_general_qq_error( runner: CommandRunner, ) -> None: """Handle general qq errors that occur during a qq operation.""" - logger.error(exception) + logger.error(terminate(str(exception))) if runner.n_jobs == len(runner.encountered_errors): sys.exit(CFG.exit_codes.default) diff --git a/src/qq_lib/core/retryer.py b/src/qq_lib/core/retryer.py index 2f0db40..2c8a2ff 100644 --- a/src/qq_lib/core/retryer.py +++ b/src/qq_lib/core/retryer.py @@ -73,5 +73,5 @@ def run(self) -> Any: # should never get here raise QQError( - "Execution got into an unexpected part of the Retryer.run method. This is a bug, please report it." + "Execution got into an unexpected part of the Retryer.run method. This is a bug, please report it" ) diff --git a/src/qq_lib/go/goer.py b/src/qq_lib/go/goer.py index 642e7a0..c69d39b 100644 --- a/src/qq_lib/go/goer.py +++ b/src/qq_lib/go/goer.py @@ -25,12 +25,12 @@ def ensure_suitable(self) -> None: """ if self._is_synchronized() and not self._work_dir_is_input_dir(): raise QQNotSuitableError( - "Job has been completed and was synchronized: working directory no longer exists." + "Job has been completed and was synchronized: working directory no longer exists" ) if self._is_killed() and not self.has_destination(): raise QQNotSuitableError( - "Job has been killed and no working directory has been created." + "Job has been killed and no working directory has been created" ) def go(self) -> None: @@ -76,7 +76,7 @@ def go(self) -> None: if not self.has_destination(): raise QQError( - "Host ('main_node') or working directory ('work_dir') are not defined." + "Host ('main_node') or working directory ('work_dir') are not defined" ) # hint for type checker diff --git a/src/qq_lib/info/informer.py b/src/qq_lib/info/informer.py index 94d07fd..13aea91 100644 --- a/src/qq_lib/info/informer.py +++ b/src/qq_lib/info/informer.py @@ -86,7 +86,7 @@ def from_job_id(cls, job_id: str) -> Self: batch_job: BatchJobInterface = BatchSystem.get_batch_job(job_id) if batch_job.is_empty(): - raise QQError(f"Job '{job_id}' does not exist.") + raise QQError(f"Job '{job_id}' does not exist") return cls.from_batch_job(batch_job) @@ -110,14 +110,14 @@ def from_batch_job(cls, batch_job: BatchJobInterface) -> Self: QQJobMismatchError: If the info file does not correspond to the job's ID. """ if not (path := batch_job.get_info_file()): - raise QQError(f"Job '{batch_job.get_id()}' is not a valid qq job.") + raise QQError(f"Job '{batch_job.get_id()}' is not a valid qq job") informer = cls.from_file(path) # check that the loaded info file actually corresponds to the batch job's ID if not informer.matches_job(batch_job.get_id()): raise QQJobMismatchError( - f"Info file for job '{batch_job.get_id()}' does not exist or is not reachable." + f"Info file for job '{batch_job.get_id()}' does not exist or is not reachable" ) informer._batch_info = batch_job diff --git a/src/qq_lib/jobs/cli.py b/src/qq_lib/jobs/cli.py index cd3af93..ebadf6a 100644 --- a/src/qq_lib/jobs/cli.py +++ b/src/qq_lib/jobs/cli.py @@ -82,7 +82,7 @@ def jobs(user: str, extra: bool, all: bool, server: str | None, yaml: bool) -> N sys.exit(0) except QQError as e: - logger.error(e) + logger.error(e.terminated) print() sys.exit(CFG.exit_codes.default) except Exception as e: diff --git a/src/qq_lib/kill/killer.py b/src/qq_lib/kill/killer.py index 7ba19b4..aa04a0b 100644 --- a/src/qq_lib/kill/killer.py +++ b/src/qq_lib/kill/killer.py @@ -28,17 +28,17 @@ def ensure_suitable(self) -> None: """ if self._is_completed(): raise QQNotSuitableError( - "Job cannot be terminated. Job is already completed." + "Job cannot be terminated. Job is already completed" ) if self._is_killed(): raise QQNotSuitableError( - "Job cannot be terminated. Job has already been killed." + "Job cannot be terminated. Job has already been killed" ) if self._is_exiting(): raise QQNotSuitableError( - "Job cannot be terminated. Job is in an exiting state." + "Job cannot be terminated. Job is in an exiting state" ) def kill(self, force: bool = False) -> str: diff --git a/src/qq_lib/killall/cli.py b/src/qq_lib/killall/cli.py index 834c94e..4794558 100644 --- a/src/qq_lib/killall/cli.py +++ b/src/qq_lib/killall/cli.py @@ -14,7 +14,7 @@ from qq_lib.core.click_format import GNUHelpColorsCommand from qq_lib.core.common import translate_server, yes_or_no_prompt from qq_lib.core.config import CFG -from qq_lib.core.error import QQError, QQJobMismatchError, QQNotSuitableError +from qq_lib.core.error import QQError, QQJobMismatchError, QQNotSuitableError, terminate from qq_lib.core.logger import get_logger from qq_lib.core.repeater import Repeater from qq_lib.info.informer import Informer @@ -125,4 +125,4 @@ def _log_error_and_continue( """ Log error as error and continue the execution. """ - logger.error(exception) + logger.error(terminate(str(exception))) diff --git a/src/qq_lib/nodes/cli.py b/src/qq_lib/nodes/cli.py index 1f1db33..267a70d 100644 --- a/src/qq_lib/nodes/cli.py +++ b/src/qq_lib/nodes/cli.py @@ -68,7 +68,7 @@ def nodes(all: bool, server: str | None, yaml: bool) -> NoReturn: console.print(panel) sys.exit(0) except QQError as e: - logger.error(e) + logger.error(e.terminated) print() sys.exit(CFG.exit_codes.default) except Exception as e: diff --git a/src/qq_lib/properties/depend.py b/src/qq_lib/properties/depend.py index d760080..4352962 100644 --- a/src/qq_lib/properties/depend.py +++ b/src/qq_lib/properties/depend.py @@ -95,7 +95,7 @@ def to_str(self) -> str: return "afterany" raise QQError( - f"Unknown dependency type '{self}'. This is a bug; please report it." + f"Unknown dependency type '{self}'. This is a bug, please report it" ) @@ -140,7 +140,7 @@ def from_str(cls, raw_depend: str): return cls(type, jobs) except Exception as e: raise QQError( - f"Could not parse dependency specification '{raw_depend}': {e}." + f"Could not parse dependency specification '{raw_depend}': {e}" ) from e @classmethod diff --git a/src/qq_lib/properties/info.py b/src/qq_lib/properties/info.py index aafead8..7a184e6 100644 --- a/src/qq_lib/properties/info.py +++ b/src/qq_lib/properties/info.py @@ -186,23 +186,23 @@ def from_file(cls, file: Path, host: str | None = None) -> Self: with file.open("r") as input: data: dict[str, object] = yaml.load(input, Loader=SafeLoader) except FileNotFoundError: - raise QQError(f"qq info file '{file}' does not exist.") + raise QQError(f"qq info file '{file}' does not exist") except PermissionError: raise QQError( - f"No permission to read file '{file}' or access its parent directory." + f"No permission to read file '{file}' or access its parent directory" ) except IsADirectoryError: - raise QQError(f"Expected a file but path is a directory: {file}.") + raise QQError(f"Expected a file but path is a directory: {file}") except UnicodeDecodeError as e: - raise QQError(f"File is not valid UTF-8 text: {file}.") from e + raise QQError(f"File is not valid UTF-8 text: {file}") from e except yaml.YAMLError as e: - raise QQError(f"Failed to parse YAML in {file}: {e}.") from e + raise QQError(f"Failed to parse YAML in {file}: {e}") from e return cls._from_dict(data) except yaml.YAMLError as e: - raise QQError(f"Could not parse the qq info file '{file}': {e}.") from e + raise QQError(f"Could not parse the qq info file '{file}': {e}") from e except TypeError as e: - raise QQError(f"Invalid qq info file '{file}': {e}.") from e + raise QQError(f"Invalid qq info file '{file}': {e}") from e def to_file(self, file: Path, host: str | None = None) -> None: """ diff --git a/src/qq_lib/properties/interpreter.py b/src/qq_lib/properties/interpreter.py index 0a949bb..5fc8000 100644 --- a/src/qq_lib/properties/interpreter.py +++ b/src/qq_lib/properties/interpreter.py @@ -93,7 +93,7 @@ def to_command_list(self) -> list[str]: if not (full := shutil.which(self.executable)): raise QQError( - f"Interpreter '{self.executable}' is not available on node '{socket.getfqdn()}'." + f"Interpreter '{self.executable}' is not available on node '{socket.getfqdn()}'" ) return [full] + self.arguments diff --git a/src/qq_lib/properties/job_type.py b/src/qq_lib/properties/job_type.py index 90514b5..4571715 100644 --- a/src/qq_lib/properties/job_type.py +++ b/src/qq_lib/properties/job_type.py @@ -43,4 +43,4 @@ def from_str(cls, s: str) -> Self: try: return cls[s.upper()] except KeyError: - raise QQError(f"Could not recognize a job type '{s}'.") + raise QQError(f"Could not recognize a job type '{s}'") diff --git a/src/qq_lib/properties/loop.py b/src/qq_lib/properties/loop.py index 925542c..5edb7b0 100644 --- a/src/qq_lib/properties/loop.py +++ b/src/qq_lib/properties/loop.py @@ -69,11 +69,11 @@ def __init__( or if the archive path is invalid. """ if not end: - raise QQError("Attribute 'loop-end' is undefined.") + raise QQError("Attribute 'loop-end' is undefined") self.archive = logical_resolve(archive) if input_dir and self.archive == logical_resolve(input_dir): - raise QQError("Input directory cannot be used as the loop job's archive.") + raise QQError("Input directory cannot be used as the loop job's archive") self.archive_format = archive_format self.archive_mode = archive_mode or TransferMode.multi_from_str( @@ -85,16 +85,16 @@ def __init__( self.current = current or self.determine_cycle_from_archive() if self.start < 0: - raise QQError(f"Attribute 'loop-start' ({self.start}) cannot be negative.") + raise QQError(f"Attribute 'loop-start' ({self.start}) cannot be negative") if self.start > self.end: raise QQError( - f"Attribute 'loop-start' ({self.start}) cannot be higher than 'loop-end' ({self.end})." + f"Attribute 'loop-start' ({self.start}) cannot be higher than 'loop-end' ({self.end})" ) if self.current > self.end: raise QQError( - f"Current cycle number ({self.current}) cannot be higher than 'loop-end' ({self.end})." + f"Current cycle number ({self.current}) cannot be higher than 'loop-end' ({self.end})" ) @classmethod @@ -116,25 +116,25 @@ def from_dict(cls, data: dict[str, object]) -> Self: archive_mode = data.get("archive_mode", ["success"]) if not isinstance(start, int): - raise QQError(f"Field 'start' must be an int, got {type(start).__name__}.") + raise QQError(f"Field 'start' must be an int, got {type(start).__name__}") if not isinstance(end, int): - raise QQError(f"Field 'end' must be an int, got {type(end).__name__}.") + raise QQError(f"Field 'end' must be an int, got {type(end).__name__}") if not isinstance(archive, str): raise QQError( - f"Field 'archive' must be a str, got {type(archive).__name__}." + f"Field 'archive' must be a str, got {type(archive).__name__}" ) if not isinstance(archive_format, str): raise QQError( - f"Field 'archive_format' must be a str, got {type(archive_format).__name__}." + f"Field 'archive_format' must be a str, got {type(archive_format).__name__}" ) if not isinstance(current, int): raise QQError( - f"Field 'current' must be an int, got {type(current).__name__}." + f"Field 'current' must be an int, got {type(current).__name__}" ) if not isinstance(archive_mode, list) or not all( isinstance(m, str) for m in archive_mode ): - raise QQError("Field 'archive_mode' must be a list of strings.") + raise QQError("Field 'archive_mode' must be a list of strings") return cls( start=start, diff --git a/src/qq_lib/properties/resources.py b/src/qq_lib/properties/resources.py index a96db83..36efb5c 100644 --- a/src/qq_lib/properties/resources.py +++ b/src/qq_lib/properties/resources.py @@ -274,7 +274,7 @@ def _parse_props(props: str) -> dict[str, str]: key, value = part, "true" if key in result: - raise QQError(f"Property '{key}' is defined multiple times.") + raise QQError(f"Property '{key}' is defined multiple times") result[key] = value return result diff --git a/src/qq_lib/properties/size.py b/src/qq_lib/properties/size.py index 776026b..20b7e6c 100644 --- a/src/qq_lib/properties/size.py +++ b/src/qq_lib/properties/size.py @@ -45,7 +45,7 @@ def __init__(self, value: int, unit: str = "kb"): self.value = 0 if value == 0 else 1 return - raise QQError(f"Unsupported unit for size '{unit}'.") + raise QQError(f"Unsupported unit for size '{unit}'") self.value = value * self._unit_map[unit] @@ -65,7 +65,7 @@ def from_string(cls, s: str) -> Self: """ match = re.match(r"^\s*(\d+)\s*([a-zA-Z]+)\s*$", s) if not match: - raise QQError(f"Invalid size string: '{s}'.") + raise QQError(f"Invalid size string: '{s}'") value, unit = match.groups() # normalize single-letter units to their full form by appending 'b' @@ -139,7 +139,7 @@ def __floordiv__(self, n: int) -> "Size": if not isinstance(n, int): return NotImplemented if n == 0: - raise ZeroDivisionError("Division by zero.") + raise ZeroDivisionError("Division by zero") return Size(math.ceil(self.value / n), "kb") @@ -167,7 +167,7 @@ def __truediv__(self, other: "Size") -> float: ) if other.value == 0: - raise ZeroDivisionError("Division by zero size.") + raise ZeroDivisionError("Division by zero size") return self.value / other.value @@ -192,6 +192,6 @@ def __sub__(self, other: "Size") -> "Size": result_kb = self.value - other.value if result_kb < 0: - raise ValueError("Resulting Size cannot be negative.") + raise ValueError("Resulting Size cannot be negative") return Size(result_kb, "kb") diff --git a/src/qq_lib/properties/transfer_mode.py b/src/qq_lib/properties/transfer_mode.py index e88b7fd..30328d7 100644 --- a/src/qq_lib/properties/transfer_mode.py +++ b/src/qq_lib/properties/transfer_mode.py @@ -43,7 +43,7 @@ def from_str(cls, s: str) -> "TransferMode": if bool(re.match(r"^-?\d+$", s.strip())): return ExitCode(int(s)) - raise QQError(f"Could not recognize a transfer mode variant '{s}'.") + raise QQError(f"Could not recognize a transfer mode variant '{s}'") @classmethod def multi_from_str(cls, raw: str) -> list["TransferMode"]: diff --git a/src/qq_lib/queues/cli.py b/src/qq_lib/queues/cli.py index 4b20f12..5290a89 100644 --- a/src/qq_lib/queues/cli.py +++ b/src/qq_lib/queues/cli.py @@ -66,7 +66,7 @@ def queues(all: bool, server: str | None, yaml: bool) -> NoReturn: console.print(panel) sys.exit(0) except QQError as e: - logger.error(e) + logger.error(e.terminated) print() sys.exit(CFG.exit_codes.default) except Exception as e: diff --git a/src/qq_lib/respawn/respawner.py b/src/qq_lib/respawn/respawner.py index d78e09d..38f03fc 100644 --- a/src/qq_lib/respawn/respawner.py +++ b/src/qq_lib/respawn/respawner.py @@ -31,7 +31,7 @@ def ensure_suitable(self) -> None: """ if self._state not in {RealState.FAILED, RealState.KILLED}: raise QQNotSuitableError( - f"Job cannot be respawned. Job is {str(self._state)}." + f"Job cannot be respawned. Job is {str(self._state)}" ) def respawn(self) -> str: @@ -122,5 +122,5 @@ def _ensure_archive_consistent(loop_info: LoopInfo) -> None: ) != loop_info.current: raise QQError( f"Respawning loop job in cycle '{loop_info.current}' but the loop job should continue from cycle '{archive_cycle}' " - "based on the contents of the archive directory. Canceling job respawn." + "based on the contents of the archive directory. Canceling job respawn" ) diff --git a/src/qq_lib/resubmit/resubmitter.py b/src/qq_lib/resubmit/resubmitter.py index f83a822..32c27c4 100644 --- a/src/qq_lib/resubmit/resubmitter.py +++ b/src/qq_lib/resubmit/resubmitter.py @@ -125,12 +125,12 @@ def _try_resubmit( main_node = informer.info.main_node if not main_node: raise QQError( - "Job cannot be resubmitted. The 'main_node' of the job is not defined." + "Job cannot be resubmitted. The 'main_node' of the job is not defined" ) if not hosts: raise QQError( - "Job cannot be resubmitted. No resubmission hosts defined. This is a bug." + "Job cannot be resubmitted. No resubmission hosts defined. This is a bug, please report it" ) for host in hosts: @@ -146,4 +146,4 @@ def _try_resubmit( except Exception as e: logger.warning(f"Failed resubmission from host '{hostname}': {e}") - raise QQError("Could not resubmit the job.") + raise QQError("Could not resubmit the job") diff --git a/src/qq_lib/run/cli.py b/src/qq_lib/run/cli.py index bcfed02..cc5b92b 100644 --- a/src/qq_lib/run/cli.py +++ b/src/qq_lib/run/cli.py @@ -10,7 +10,12 @@ import click from qq_lib.core.config import CFG -from qq_lib.core.error import QQError, QQRunCommunicationError, QQRunFatalError +from qq_lib.core.error import ( + QQError, + QQRunCommunicationError, + QQRunFatalError, + terminate, +) from qq_lib.core.logger import get_logger from .runner import Runner, log_fatal_error_and_exit @@ -61,19 +66,19 @@ def run(script_path: str) -> NoReturn: # make sure that qq run is being run as a batch job ensure_qq_env() except Exception as e: - logger.error(e) + logger.error(terminate(str(e))) sys.exit(CFG.exit_codes.not_qq_env) try: # get the destination of the info file from env vars if not (info_file := os.environ.get(CFG.env_vars.info_file)): raise QQRunFatalError( - f"'{CFG.env_vars.info_file}' environment variable is not set." + f"'{CFG.env_vars.info_file}' environment variable is not set" ) if not (input_machine := os.environ.get(CFG.env_vars.input_machine)): raise QQRunFatalError( - f"'{CFG.env_vars.input_machine}' environment variable is not set." + f"'{CFG.env_vars.input_machine}' environment variable is not set" ) # initialize the runner @@ -109,5 +114,5 @@ def ensure_qq_env() -> None: if not os.environ.get(CFG.env_vars.guard): raise QQError( "This script must be run as a qq job within the batch system. " - f"To submit it properly, use: '{CFG.binary_name} submit'." + f"To submit it properly, use: '{CFG.binary_name} submit'" ) diff --git a/src/qq_lib/run/runner.py b/src/qq_lib/run/runner.py index df2ff7a..86ffb79 100644 --- a/src/qq_lib/run/runner.py +++ b/src/qq_lib/run/runner.py @@ -23,6 +23,7 @@ QQJobMismatchError, QQRunCommunicationError, QQRunFatalError, + terminate, ) from qq_lib.core.logger import get_logger from qq_lib.core.logical_paths import logical_resolve @@ -329,7 +330,7 @@ def log_failure_and_exit(self, exception: BaseException) -> NoReturn: exit_code = getattr(exception, "exit_code", CFG.exit_codes.unexpected_error) try: self._update_info_failed(exit_code) - logger.error(exception) + logger.error(terminate(str(exception))) sys.exit(exit_code) except Exception as e: # unable to log the current state into the info file @@ -456,7 +457,7 @@ def _update_info_running(self) -> None: ).run() except Exception as e: raise QQError( - f"Could not update qqinfo file '{self._info_file}' at JOB START: {e}." + f"Could not update qqinfo file '{self._info_file}' at JOB START: {e}" ) from e def _get_nodes(self) -> list[str]: @@ -636,7 +637,7 @@ def _ensure_matches_job(self, job_id: str) -> None: """ if not self._informer.matches_job(job_id): raise QQJobMismatchError( - f"Info file '{self._info_file}' does not correspond to job '{job_id}'." + f"Info file '{self._info_file}' does not correspond to job '{job_id}'" ) def _ensure_not_killed(self) -> None: @@ -648,7 +649,7 @@ def _ensure_not_killed(self) -> None: """ if self._informer.info.job_state == NaiveState.KILLED: raise QQRunCommunicationError( - "Job has been killed without informing qq run. Aborting the job!" + "Job has been killed without informing qq run. Aborting the job" ) def _reload_info_and_ensure_valid(self, retry: bool = False) -> None: @@ -689,7 +690,7 @@ def _resubmit(self) -> None: if self._informer.info.job_type == JobType.LOOP: if not (loop_info := self._informer.info.loop_info): raise QQError( - "Loop info is undefined while resubmiting a loop job. This is a bug!" + "Loop info is undefined while resubmiting a loop job. This is a bug, please report it" ) return @@ -712,11 +713,13 @@ def _archive_files_from_work_dir(self) -> None: If no file exists for the next loop cycle, creates an empty init file to ensure the loop job continues normally. """ if not self._archiver: - raise QQError("Archiver is undefined while archiving files. This is a bug!") + raise QQError( + "Archiver is undefined while archiving files. This is a bug, please report it" + ) if not (loop_info := self._informer.info.loop_info): raise QQError( - "Loop info is undefined while archiving files. This is a bug!" + "Loop info is undefined while archiving files. This is a bug, please report it" ) # get the files to archive corresponding to the next loop job cycle @@ -861,7 +864,7 @@ def log_fatal_error_and_exit(exception: BaseException) -> NoReturn: Raises: SystemExit: Exits with an exit code associated with the exception. """ - logger.error(f"Fatal qq run error: {exception}") + logger.error(f"Fatal qq run error: {exception}.") logger.error("Failure state was NOT logged into the job info file.") if isinstance(exception, (QQRunFatalError, QQRunCommunicationError, QQError)): diff --git a/src/qq_lib/shebang/cli.py b/src/qq_lib/shebang/cli.py index 3553945..a2e6dec 100644 --- a/src/qq_lib/shebang/cli.py +++ b/src/qq_lib/shebang/cli.py @@ -45,7 +45,7 @@ def shebang(script: str | None) -> NoReturn: print(SHEBANG) sys.exit(0) except QQError as e: - logger.error(e) + logger.error(e.terminated) print() sys.exit(CFG.exit_codes.default) except Exception as e: @@ -66,7 +66,7 @@ def _replace_or_add_shebang(file: Path) -> None: """ if not file.is_file(): - raise QQError(f"File '{str(file)}' does not exist.") + raise QQError(f"File '{str(file)}' does not exist") content = file.read_text().splitlines() diff --git a/src/qq_lib/stat/cli.py b/src/qq_lib/stat/cli.py index c6e8609..18c94f1 100644 --- a/src/qq_lib/stat/cli.py +++ b/src/qq_lib/stat/cli.py @@ -71,7 +71,7 @@ def stat(extra: bool, all: bool, server: str | None, yaml: bool) -> NoReturn: sys.exit(0) except QQError as e: - logger.error(e) + logger.error(e.terminated) print() sys.exit(CFG.exit_codes.default) except Exception as e: diff --git a/src/qq_lib/submit/cli.py b/src/qq_lib/submit/cli.py index 01c29be..a7ca888 100644 --- a/src/qq_lib/submit/cli.py +++ b/src/qq_lib/submit/cli.py @@ -373,7 +373,7 @@ def _submit_job(script: str, kwargs: dict[str, Any]) -> str | None: """ try: if not (script_path := Path(script)).is_file(): - raise QQError(f"Script '{script}' does not exist or is not a file.") + raise QQError(f"Script '{script}' does not exist or is not a file") factory = SubmitterFactory(logical_resolve(script_path), **kwargs) submitter = factory.make_submitter() @@ -383,14 +383,14 @@ def _submit_job(script: str, kwargs: dict[str, Any]) -> str | None: and not submitter.continues_loop() ): raise QQError( - "Detected qq runtime files in the submission directory. Submission aborted." + "Detected qq runtime files in the submission directory. Submission aborted" ) job_id = submitter.submit() logger.info(f"Job '{job_id}' submitted successfully.") return job_id except QQError as e: - logger.error(e) + logger.error(e.terminated) return None except Exception as e: logger.critical(e, exc_info=True, stack_info=True) diff --git a/src/qq_lib/submit/factory.py b/src/qq_lib/submit/factory.py index 1d9dadb..ae1ea50 100644 --- a/src/qq_lib/submit/factory.py +++ b/src/qq_lib/submit/factory.py @@ -142,7 +142,7 @@ def _get_queue(self) -> str: QQError: If no queue is specified either in kwargs or in the script. """ if not (queue := self._kwargs.get("queue") or self._parser.get_queue()): - raise QQError("Submission queue not specified.") + raise QQError("Submission queue not specified") return queue def _get_resources( diff --git a/src/qq_lib/submit/parser.py b/src/qq_lib/submit/parser.py index ce979f9..a5577fb 100644 --- a/src/qq_lib/submit/parser.py +++ b/src/qq_lib/submit/parser.py @@ -64,7 +64,7 @@ def parse(self) -> None: contains an unknown option. """ if not self._script.is_file(): - raise QQError(f"Could not open '{self._script}' as a file.") + raise QQError(f"Could not open '{self._script}' as a file") with self._script.open() as f: # skip the first line (shebang) @@ -88,7 +88,7 @@ def parse(self) -> None: parts = Parser._strip_and_split(line) if len(parts) < 2: raise QQError( - f"Invalid qq submit option line in '{str(self._script)}': {line}." + f"Invalid qq submit option line in '{str(self._script)}': {line}" ) key, value = parts[-2], parts[-1] @@ -107,7 +107,7 @@ def parse(self) -> None: self._options[snake_case_key] = value else: raise QQError( - f"Unknown qq submit option '{key}' in '{str(self._script)}': {line.strip()}.\nKnown options are '{' '.join(self._known_options)}'." + f"Unknown qq submit option '{key}' in '{str(self._script)}': {line.strip()}.\nKnown options are '{' '.join(self._known_options)}'" ) logger.debug(f"Parsed options from '{self._script}': {self._options}.") diff --git a/src/qq_lib/submit/submitter.py b/src/qq_lib/submit/submitter.py index f100743..b0ce8f1 100644 --- a/src/qq_lib/submit/submitter.py +++ b/src/qq_lib/submit/submitter.py @@ -125,12 +125,12 @@ def __init__( # script must exist if not self._script.is_file(): - raise QQError(f"Script '{script}' does not exist or is not a file.") + raise QQError(f"Script '{script}' does not exist or is not a file") # script must have a valid qq shebang if not self._has_valid_shebang(self._script): raise QQError( - f"Script '{self._script}' has an invalid shebang. The first line of the script should be '#!/usr/bin/env -S {CFG.binary_name} run'." + f"Script '{self._script}' has an invalid shebang. The first line of the script should be '#!/usr/bin/env -S {CFG.binary_name} run'" ) def submit(self, remote: str | None = None) -> str: @@ -223,7 +223,7 @@ def continues_loop(self) -> bool: ) return False except QQError as e: - logger.debug(f"Could not read an info file: {e}.") + logger.debug(f"Could not read an info file: {e}") return False def _loop_job_continues_loop(self, previous: Informer) -> bool: diff --git a/src/qq_lib/sync/syncer.py b/src/qq_lib/sync/syncer.py index 201393d..bbfb394 100644 --- a/src/qq_lib/sync/syncer.py +++ b/src/qq_lib/sync/syncer.py @@ -24,23 +24,23 @@ def ensure_suitable(self): """ if self._work_dir_is_input_dir(): raise QQNotSuitableError( - "Working directory of the job is the input directory of the job: implicitly synchronized." + "Working directory of the job is the input directory of the job: implicitly synchronized" ) if self._is_synchronized(): raise QQNotSuitableError( - "Job has been completed and was synchronized: working directory no longer exists." + "Job has been completed and was synchronized: working directory no longer exists" ) # killed jobs may not have working directory if self._is_killed() and not self.has_destination(): raise QQNotSuitableError( - "Job has been killed and no working directory is available." + "Job has been killed and no working directory is available" ) # queued jobs do not have working directory if self._is_queued(): - raise QQNotSuitableError("Job is queued or booting: nothing to sync.") + raise QQNotSuitableError("Job is queued or booting: nothing to sync") def sync(self, files: list[str] | None = None) -> None: """ @@ -59,7 +59,7 @@ def sync(self, files: list[str] | None = None) -> None: """ if not self.has_destination(): raise QQError( - "Host ('main_node') or working directory ('work_dir') are not defined." + "Host ('main_node') or working directory ('work_dir') are not defined" ) # hint for type checker diff --git a/src/qq_lib/wipe/wiper.py b/src/qq_lib/wipe/wiper.py index 78a0f16..a6ba715 100644 --- a/src/qq_lib/wipe/wiper.py +++ b/src/qq_lib/wipe/wiper.py @@ -24,31 +24,31 @@ def ensure_suitable(self) -> None: """ if self._work_dir_is_input_dir(): raise QQNotSuitableError( - "Working directory of the job is the input directory of the job. Cannot delete the input directory." + "Working directory of the job is the input directory of the job. Cannot delete the input directory" ) if self._is_queued(): raise QQNotSuitableError( - f"Job is {str(self._informer.get_real_state()).lower()} and does not have a working directory yet." + f"Job is {str(self._informer.get_real_state()).lower()} and does not have a working directory yet" ) if self._is_running() or self._is_suspended(): raise QQNotSuitableError( - f"Job is {str(self._informer.get_real_state()).lower()}. It is not safe to delete the working directory." + f"Job is {str(self._informer.get_real_state()).lower()}. It is not safe to delete the working directory" ) if self._is_synchronized(): raise QQNotSuitableError( - "Job has been completed and was synchronized: working directory no longer exists." + "Job has been completed and was synchronized: working directory no longer exists" ) if self._is_finished(): raise QQNotSuitableError( - "It may not be safe to delete the working directory of a successfully finished job. Rerun as 'qq wipe --force' if sure." + "It may not be safe to delete the working directory of a successfully finished job. Rerun as 'qq wipe --force' if sure" ) if not self.has_destination(): - raise QQNotSuitableError("Job does not have a working directory.") + raise QQNotSuitableError("Job does not have a working directory") def wipe(self) -> str: """ @@ -62,7 +62,7 @@ def wipe(self) -> str: """ if not self.has_destination(): raise QQError( - "Host ('main_node') or working directory ('work_dir') are not defined." + "Host ('main_node') or working directory ('work_dir') are not defined" ) # hint for type checker @@ -72,7 +72,7 @@ def wipe(self) -> str: # we cannot delete the input directory even if the `--force` flag is used if self._work_dir_is_input_dir(): raise QQError( - "Working directory of the job is the input directory of the job. Cannot delete the input directory." + "Working directory of the job is the input directory of the job. Cannot delete the input directory" ) logger.info( diff --git a/tests/batch/interface/test_batch_interface.py b/tests/batch/interface/test_batch_interface.py index 8c4feae..9a80041 100644 --- a/tests/batch/interface/test_batch_interface.py +++ b/tests/batch/interface/test_batch_interface.py @@ -670,7 +670,7 @@ def test_batch_interface_sort_jobs_empty_list(): @patch("qq_lib.batch.interface.interface.subprocess.run") -def test_batchinterface_delete_remote_dir_success(mock_run): +def test_batch_interface_delete_remote_dir_success(mock_run): mock_run.return_value = MagicMock(returncode=0) BatchInterface.delete_remote_dir("remote_host", Path("/remote/dir")) @@ -691,12 +691,12 @@ def test_batchinterface_delete_remote_dir_success(mock_run): @patch("qq_lib.batch.interface.interface.subprocess.run") -def test_batchinterface_delete_remote_dir_raises_error(mock_run): +def test_batch_interface_delete_remote_dir_raises_error(mock_run): mock_run.return_value = MagicMock(returncode=1, stderr="permission denied") with pytest.raises( QQError, - match="Could not delete remote directory '/remote/dir' on 'remote_host': permission denied.", + match="Could not delete remote directory '/remote/dir' on 'remote_host': permission denied", ): BatchInterface.delete_remote_dir("remote_host", Path("/remote/dir")) diff --git a/tests/batch/pbs/test_pbs.py b/tests/batch/pbs/test_pbs.py index 87691e1..cf5a070 100644 --- a/tests/batch/pbs/test_pbs.py +++ b/tests/batch/pbs/test_pbs.py @@ -1927,7 +1927,7 @@ def mock_rmtree(_): host = socket.getfqdn() with pytest.raises( - QQError, match=f"Could not delete directory '{test_dir}': access denied." + QQError, match=f"Could not delete directory '{test_dir}': access denied" ): PBS.delete_remote_dir(host, test_dir) diff --git a/tests/batch/pbs/test_pbs_node.py b/tests/batch/pbs/test_pbs_node.py index c0db777..8e34097 100644 --- a/tests/batch/pbs/test_pbs_node.py +++ b/tests/batch/pbs/test_pbs_node.py @@ -171,7 +171,7 @@ def test_pbs_node_update_raises_on_nonzero_return(mock_run): node._name = "nodeX" node._server = None mock_run.return_value = MagicMock(returncode=1, stdout="", stderr="error") - with pytest.raises(QQError, match="Node 'nodeX' does not exist."): + with pytest.raises(QQError, match="Node 'nodeX' does not exist"): node.update() mock_run.assert_called_once() diff --git a/tests/batch/pbs/test_pbs_queue.py b/tests/batch/pbs/test_pbs_queue.py index aff9ce7..e5862d4 100644 --- a/tests/batch/pbs/test_pbs_queue.py +++ b/tests/batch/pbs/test_pbs_queue.py @@ -170,7 +170,7 @@ def test_pbsqueue_update_failure(): with ( patch("qq_lib.batch.pbs.queue.subprocess.run", return_value=mock_result), - pytest.raises(QQError, match="Queue 'nonexistent' does not exist."), + pytest.raises(QQError, match="Queue 'nonexistent' does not exist"), ): queue.update() diff --git a/tests/batch/slurm/test_slurm.py b/tests/batch/slurm/test_slurm.py index 6e99016..09d9708 100644 --- a/tests/batch/slurm/test_slurm.py +++ b/tests/batch/slurm/test_slurm.py @@ -307,7 +307,7 @@ def test_slurm_translate_per_chunk_resources_raises_when_mem_missing(): res.mem_per_cpu = None with pytest.raises( QQError, - match="None of the attributes 'mem', 'mem-per-node', or 'mem-per-cpu' is defined.", + match="None of the attributes 'mem', 'mem-per-node', or 'mem-per-cpu' is defined", ): Slurm._translate_per_chunk_resources(res) @@ -880,7 +880,7 @@ def test_slurm_get_nodes_failure_raises_qqerror(mock_run): mock_run.return_value = mock_result with pytest.raises( - QQError, match="Could not retrieve information about nodes: some error." + QQError, match="Could not retrieve information about nodes: some error" ): Slurm.get_nodes() diff --git a/tests/batch/slurm/test_slurm_node.py b/tests/batch/slurm/test_slurm_node.py index 75133da..12cdffd 100644 --- a/tests/batch/slurm/test_slurm_node.py +++ b/tests/batch/slurm/test_slurm_node.py @@ -57,7 +57,7 @@ def test_slurm_node_update_failure_raises_qqerror(mock_run): mock_result.returncode = 1 mock_run.return_value = mock_result - with pytest.raises(QQError, match="Node 'node2' does not exist."): + with pytest.raises(QQError, match="Node 'node2' does not exist"): SlurmNode("node2") diff --git a/tests/batch/slurm/test_slurm_queue.py b/tests/batch/slurm/test_slurm_queue.py index 8d5a06a..8a8a22e 100644 --- a/tests/batch/slurm/test_slurm_queue.py +++ b/tests/batch/slurm/test_slurm_queue.py @@ -113,7 +113,7 @@ def test_slurm_queue_update_raises_on_failure(mock_run): queue = SlurmQueue.__new__(SlurmQueue) queue._name = "cpu" mock_run.return_value = MagicMock(returncode=1, stdout="") - with pytest.raises(QQError, match="Queue 'cpu' does not exist."): + with pytest.raises(QQError, match="Queue 'cpu' does not exist"): queue.update() @@ -426,7 +426,7 @@ def test_slurm_queue_set_job_numbers_raises_on_failure(mock_run): queue._name = "cpu" mock_run.return_value = MagicMock(returncode=1, stderr="permission denied") with pytest.raises( - QQError, match="Could not get job numbers for queue 'cpu': permission denied." + QQError, match="Could not get job numbers for queue 'cpu': permission denied" ): queue._set_job_numbers() diff --git a/tests/batch/slurmit4i/test_slurmit4i.py b/tests/batch/slurmit4i/test_slurmit4i.py index 7923ec3..e8af90c 100644 --- a/tests/batch/slurmit4i/test_slurmit4i.py +++ b/tests/batch/slurmit4i/test_slurmit4i.py @@ -222,7 +222,7 @@ def test_slurmit4i_move_remote_files_raises_on_length_mismatch(): moved_files = [Path("/data/a_moved.txt"), Path("/data/b_moved.txt")] with pytest.raises( QQError, - match="The provided 'files' and 'moved_files' must have the same length.", + match="The provided 'files' and 'moved_files' must have the same length", ): SlurmIT4I.move_remote_files("host", files, moved_files) @@ -370,7 +370,7 @@ def mock_rmtree(_): monkeypatch.setattr(shutil, "rmtree", mock_rmtree) with pytest.raises( - QQError, match=f"Could not delete directory '{test_dir}': access denied." + QQError, match=f"Could not delete directory '{test_dir}': access denied" ): SlurmIT4I.delete_remote_dir("some_host", test_dir) diff --git a/tests/core/test_common.py b/tests/core/test_common.py index 4e543ec..96c962b 100644 --- a/tests/core/test_common.py +++ b/tests/core/test_common.py @@ -82,7 +82,7 @@ def test_get_info_file_no_info_file(): with tempfile.TemporaryDirectory() as tmpdir: tmp_path = Path(tmpdir) - with pytest.raises(QQError, match="No qq job info file found."): + with pytest.raises(QQError, match="No qq job info file found"): get_info_file(tmp_path) diff --git a/tests/core/test_error_handlers.py b/tests/core/test_error_handlers.py index 9b06109..d4a5d18 100644 --- a/tests/core/test_error_handlers.py +++ b/tests/core/test_error_handlers.py @@ -7,7 +7,7 @@ import pytest from qq_lib.core.command_runner import CommandRunner -from qq_lib.core.error import QQNotSuitableError +from qq_lib.core.error import QQNotSuitableError, terminate from qq_lib.core.error_handlers import ( CFG, handle_general_qq_error, @@ -28,7 +28,7 @@ def test_not_suitable_single_item_logs_error_and_exits(): ): handle_not_suitable_error(exc, metadata) - mock_logger.error.assert_called_once_with(exc) + mock_logger.error.assert_called_once_with(terminate(str(exc))) mock_exit.assert_called_once_with(CFG.exit_codes.default) @@ -44,7 +44,7 @@ def test_not_suitable_multiple_items_logs_info(): ): handle_not_suitable_error(exc, metadata) - mock_logger.info.assert_called_once_with(exc) + mock_logger.info.assert_called_once_with(terminate(str(exc))) mock_exit.assert_not_called() @@ -63,7 +63,7 @@ def test_not_suitable_multiple_items_multiple_errors_logs_and_exits(): ): handle_not_suitable_error(exc, metadata) - mock_logger.info.assert_called_once_with(exc) + mock_logger.info.assert_called_once_with(terminate(str(exc))) mock_logger.error.assert_called_once_with("No suitable qq job.\n") mock_exit.assert_called_once_with(CFG.exit_codes.default) @@ -81,7 +81,7 @@ def test_job_mismatch_logs_and_exits(n_jobs): ): handle_job_mismatch_error(exc, metadata) - mock_logger.error.assert_called_once_with(exc) + mock_logger.error.assert_called_once_with(terminate(str(exc))) mock_exit.assert_called_once_with(CFG.exit_codes.default) @@ -97,7 +97,7 @@ def test_job_general_qq_error_single_item_logs_and_exits(): ): handle_general_qq_error(exc, metadata) - mock_logger.error.assert_called_once_with(exc) + mock_logger.error.assert_called_once_with(terminate(str(exc))) mock_exit.assert_called_once_with(CFG.exit_codes.default) @@ -113,7 +113,7 @@ def test_job_general_qq_error_multiple_items_logs(): ): handle_general_qq_error(exc, metadata) - mock_logger.error.assert_called_once_with(exc) + mock_logger.error.assert_called_once_with(terminate(str(exc))) mock_exit.assert_not_called() @@ -132,5 +132,5 @@ def test_job_general_qq_error_multiple_items_multiple_errors_logs_and_exits(): ): handle_general_qq_error(exc, metadata) - mock_logger.error.assert_called_once_with(exc) + mock_logger.error.assert_called_once_with(terminate(str(exc))) mock_exit.assert_called_once_with(CFG.exit_codes.default) diff --git a/tests/go/test_go_goer.py b/tests/go/test_go_goer.py index 42cc1d1..663a693 100644 --- a/tests/go/test_go_goer.py +++ b/tests/go/test_go_goer.py @@ -147,7 +147,7 @@ def test_goer_ensure_suitable_raises_killed_without_destination(destination): with pytest.raises( QQNotSuitableError, - match="Job has been killed and no working directory has been created.", + match="Job has been killed and no working directory has been created", ): goer.ensure_suitable() @@ -553,7 +553,7 @@ def test_goer_go_no_destination_raises_error(): patch("qq_lib.go.goer.logger"), pytest.raises( QQError, - match="Host \\('main_node'\\) or working directory \\('work_dir'\\) are not defined.", + match="Host \\('main_node'\\) or working directory \\('work_dir'\\) are not defined", ), ): goer.go() diff --git a/tests/info/test_info_informer.py b/tests/info/test_info_informer.py index fe5c4ce..9f3b077 100644 --- a/tests/info/test_info_informer.py +++ b/tests/info/test_info_informer.py @@ -488,7 +488,7 @@ def test_informer_from_job_id_raises_when_empty(): "qq_lib.info.informer.BatchInterface.from_env_var_or_guess", return_value=batch_system, ), - pytest.raises(QQError, match="Job '123' does not exist."), + pytest.raises(QQError, match="Job '123' does not exist"), ): Informer.from_job_id("123") @@ -521,7 +521,7 @@ def test_informer_from_batch_job_raises_when_no_info_file(): batch_job.get_info_file.return_value = None batch_job.get_id.return_value = "123" - with pytest.raises(QQError, match="Job '123' is not a valid qq job."): + with pytest.raises(QQError, match="Job '123' is not a valid qq job"): Informer.from_batch_job(batch_job) @@ -537,7 +537,7 @@ def test_informer_from_batch_job_raises_on_mismatch(): patch("qq_lib.info.informer.Informer.from_file", return_value=informer_mock), pytest.raises( QQJobMismatchError, - match="Info file for job '123' does not exist or is not reachable.", + match="Info file for job '123' does not exist or is not reachable", ), ): Informer.from_batch_job(batch_job) diff --git a/tests/kill/test_kill_killer.py b/tests/kill/test_kill_killer.py index 8cda919..074c76b 100644 --- a/tests/kill/test_kill_killer.py +++ b/tests/kill/test_kill_killer.py @@ -247,20 +247,20 @@ def test_killer_kill_does_not_update_info_file(): @pytest.mark.parametrize( "state,exit,expected_message", [ - (RealState.FINISHED, 0, "Job cannot be terminated. Job is already completed."), - (RealState.FAILED, 1, "Job cannot be terminated. Job is already completed."), + (RealState.FINISHED, 0, "Job cannot be terminated. Job is already completed"), + (RealState.FAILED, 1, "Job cannot be terminated. Job is already completed"), ( RealState.KILLED, None, - "Job cannot be terminated. Job has already been killed.", + "Job cannot be terminated. Job has already been killed", ), ( RealState.EXITING, None, - "Job cannot be terminated. Job has already been killed.", + "Job cannot be terminated. Job has already been killed", ), - (RealState.EXITING, 0, "Job cannot be terminated. Job is in an exiting state."), - (RealState.EXITING, 1, "Job cannot be terminated. Job is in an exiting state."), + (RealState.EXITING, 0, "Job cannot be terminated. Job is in an exiting state"), + (RealState.EXITING, 1, "Job cannot be terminated. Job is in an exiting state"), ], ) def test_killer_ensure_suitable_raises(state, exit, expected_message): diff --git a/tests/properties/test_resources.py b/tests/properties/test_resources.py index 9faa86d..a3c08e7 100644 --- a/tests/properties/test_resources.py +++ b/tests/properties/test_resources.py @@ -445,7 +445,7 @@ def test_parse_props_strips_empty_parts(): ], ) def test_parse_props_raises_on_duplicate_keys(props): - with pytest.raises(QQError, match="Property 'foo' is defined multiple times."): + with pytest.raises(QQError, match="Property 'foo' is defined multiple times"): Resources._parse_props(props) diff --git a/tests/run/test_run_runner.py b/tests/run/test_run_runner.py index cbc4e63..9bfadfb 100644 --- a/tests/run/test_run_runner.py +++ b/tests/run/test_run_runner.py @@ -16,6 +16,7 @@ QQJobMismatchError, QQRunCommunicationError, QQRunFatalError, + terminate, ) from qq_lib.properties.interpreter import Interpreter from qq_lib.properties.job_type import JobType @@ -846,7 +847,7 @@ def test_runner_log_failure_and_exit_calls_update_and_exits(): runner.log_failure_and_exit(exc) runner._update_info_failed.assert_called_once_with(42) - mock_logger.error.assert_called_once_with(exc) + mock_logger.error.assert_called_once_with(terminate(str(exc))) mock_exit.assert_called_once_with(42) @@ -1370,7 +1371,7 @@ def test_log_fatal_error_and_exit_known_exception(): ): log_fatal_error_and_exit(exc) - mock_logger.error.assert_any_call("Fatal qq run error: fatal") + mock_logger.error.assert_any_call("Fatal qq run error: fatal.") mock_logger.error.assert_any_call( "Failure state was NOT logged into the job info file." ) @@ -1386,7 +1387,7 @@ def test_log_fatal_error_and_exit_unknown_exception(): ): log_fatal_error_and_exit(exc) - mock_logger.error.assert_any_call("Fatal qq run error: unknown") + mock_logger.error.assert_any_call("Fatal qq run error: unknown.") mock_logger.error.assert_any_call( "Failure state was NOT logged into the job info file." ) diff --git a/tests/submit/test_submit_factory.py b/tests/submit/test_submit_factory.py index 9c8f3b6..b317120 100644 --- a/tests/submit/test_submit_factory.py +++ b/tests/submit/test_submit_factory.py @@ -420,7 +420,7 @@ def test_submitter_factory_get_queue_raises_error_if_missing(): factory._parser = mock_parser factory._kwargs = {} - with pytest.raises(QQError, match="Submission queue not specified."): + with pytest.raises(QQError, match="Submission queue not specified"): factory._get_queue() diff --git a/tests/sync/test_sync_syncer.py b/tests/sync/test_sync_syncer.py index 5c9174e..132b470 100644 --- a/tests/sync/test_sync_syncer.py +++ b/tests/sync/test_sync_syncer.py @@ -163,7 +163,7 @@ def test_syncer_ensure_suitable_raises_killed_without_destination(destination): with pytest.raises( QQNotSuitableError, - match="Job has been killed and no working directory is available.", + match="Job has been killed and no working directory is available", ): syncer.ensure_suitable() @@ -283,7 +283,7 @@ def test_syncer_sync_raises_without_destination(destination): with pytest.raises( QQError, - match=r"Host \('main_node'\) or working directory \('work_dir'\) are not defined\.", + match=r"Host \('main_node'\) or working directory \('work_dir'\) are not defined", ): syncer.sync() diff --git a/tests/wipe/test_wipe_wiper.py b/tests/wipe/test_wipe_wiper.py index baa2eab..9d1464f 100644 --- a/tests/wipe/test_wipe_wiper.py +++ b/tests/wipe/test_wipe_wiper.py @@ -175,7 +175,7 @@ def test_wiper_ensure_suitable_raises_when_destination_missing(): wiper._batch_system = MagicMock() with pytest.raises( - QQNotSuitableError, match="Job does not have a working directory." + QQNotSuitableError, match="Job does not have a working directory" ): wiper.ensure_suitable() From b0e9a8042b45988ac8842bf5268cf7dccc6d0184 Mon Sep 17 00:00:00 2001 From: Ladme Date: Fri, 25 Sep 2026 12:36:47 +0200 Subject: [PATCH 15/19] fix: parse nonexistent jobs/queues/nodes from PBS dump --- CHANGELOG.md | 1 + src/qq_lib/batch/pbs/common.py | 37 ++++++ src/qq_lib/info/informer.py | 3 + tests/batch/pbs/test_pbs_common.py | 180 +++++++++++++++++++++++++++++ tests/info/test_info_informer.py | 12 ++ 5 files changed, 233 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index e592543..59a4f23 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,6 +19,7 @@ ### Bug fixes and other changes - Directories can be now properly archived. +- qq commands operating on multiple job IDs no longer fail on the first occurence of a job that does not exist. They report an error and continue processing the remaining jobs. - Archive directory created in a working directory is no longer merged with the actual archive directory in the input directory. - Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. diff --git a/src/qq_lib/batch/pbs/common.py b/src/qq_lib/batch/pbs/common.py index bea6036..0ffca0d 100644 --- a/src/qq_lib/batch/pbs/common.py +++ b/src/qq_lib/batch/pbs/common.py @@ -9,6 +9,32 @@ logger = get_logger(__name__) +_UNKNOWN_OBJECT_PATTERNS = ( + # qstat: Unknown queue gpu1234 + # qstat: Unknown Job Id 100.robox-pro.ceitec.muni.cz + re.compile(r"^\s*\w+:\s+Unknown\s+(?:Job Id|queue|node)\s+(\S+)\s*$"), + # Node: node.ceitec.muni.cz, Error: Unknown node + re.compile(r"^\s*Node:\s*([^,\s]+)\s*,\s*Error:\s*Unknown node\s*$"), +) + + +def _parse_unknown_pbs_object(line: str) -> str | None: + """ + Extract the identifier from a PBS line reporting an unknown job, queue or node. + + Args: + line (str): A single line of a PBS dump. + + Returns: + str | None: Identifier of the unknown PBS object, or `None` if the line does + not report an unknown object or carries no identifier. + """ + for pattern in _UNKNOWN_OBJECT_PATTERNS: + match = pattern.match(line) + if match: + return match.group(1) + return None + def parse_pbs_dump_to_dictionary(text: str) -> dict[str, str]: """ @@ -60,6 +86,17 @@ def parse_multi_pbs_dump_to_dictionaries( ) for line in text.splitlines(): + # lines reporting unknown objects are parsed into empty dictionaries + unknown = _parse_unknown_pbs_object(line) + if unknown is not None: + if block: + data.append( + (parse_pbs_dump_to_dictionary("\n".join(block)), identifier) + ) + block, identifier = [], None + data.append(({}, unknown)) + continue + # if the line is empty, start a new block if not line.strip(): if block: diff --git a/src/qq_lib/info/informer.py b/src/qq_lib/info/informer.py index 13aea91..f11f234 100644 --- a/src/qq_lib/info/informer.py +++ b/src/qq_lib/info/informer.py @@ -109,6 +109,9 @@ def from_batch_job(cls, batch_job: BatchJobInterface) -> Self: QQError: If the job is not a valid qq job (missing info file). QQJobMismatchError: If the info file does not correspond to the job's ID. """ + if batch_job.is_empty(): + raise QQError(f"Job '{batch_job.get_id()}' does not exist") + if not (path := batch_job.get_info_file()): raise QQError(f"Job '{batch_job.get_id()}' is not a valid qq job") diff --git a/tests/batch/pbs/test_pbs_common.py b/tests/batch/pbs/test_pbs_common.py index 28ab246..3b37206 100644 --- a/tests/batch/pbs/test_pbs_common.py +++ b/tests/batch/pbs/test_pbs_common.py @@ -5,6 +5,7 @@ import pytest from qq_lib.batch.pbs.common import ( + _parse_unknown_pbs_object, parse_multi_pbs_dump_to_dictionaries, parse_pbs_dump_to_dictionary, ) @@ -298,6 +299,71 @@ def test_parse_multi_pbs_dump_to_dictionaries_jobs(): assert isinstance(name, str) +def test_parse_multi_pbs_dump_to_dictionaries_jobs_with_nonexistent(): + pbs_dump = """Job Id: 101.fake-cluster.example.com + Job_Name = job_one + Job_Owner = user1@EXAMPLE + job_state = R + queue = gpu + ctime = Sun Sep 21 00:00:00 2025 + Resource_List.ncpus = 8 + Resource_List.ngpus = 1 + Resource_List.mem = 8gb + Resource_List.walltime = 24:00:00 + started = True + +qstat: Unknown Job Id 99.fake-cluster.example.com +qstat: Unknown Job Id 104.fake-cluster.example.com +Job Id: 102.fake-cluster.example.com + Job_Name = job_two + Job_Owner = user2@EXAMPLE + job_state = Q + queue = cpu + ctime = Sun Sep 21 01:00:00 2025 + Resource_List.ncpus = 16 + Resource_List.ngpus = 0 + Resource_List.mem = 16gb + Resource_List.walltime = 12:00:00 + started = False + +Job Id: 103.fake-cluster.example.com + Job_Name = job_three + Job_Owner = user3@EXAMPLE + job_state = H + queue = maintenance + ctime = Sun Sep 21 02:00:00 2025 + Resource_List.ncpus = 4 + Resource_List.ngpus = 0 + Resource_List.mem = 4gb + Resource_List.walltime = 06:00:00 + started = False + """ + + result = parse_multi_pbs_dump_to_dictionaries(pbs_dump, "Job Id") + + assert len(result) == 5 + + job_names = [name for _, name in result] + assert job_names == [ + "101.fake-cluster.example.com", + "99.fake-cluster.example.com", + "104.fake-cluster.example.com", + "102.fake-cluster.example.com", + "103.fake-cluster.example.com", + ] + + for data_dict, name in result: + assert isinstance(data_dict, dict) + if name in ["99.fake-cluster.example.com", "104.fake-cluster.example.com"]: + assert data_dict == {} + continue + + assert "Job_Name" in data_dict + assert "job_state" in data_dict + assert "queue" in data_dict + assert isinstance(name, str) + + def test_parse_pbs_dump_to_dictionary_node(): pbs_dump = """zero21 Mom = zero21.cluster.local @@ -422,3 +488,117 @@ def test_parse_multi_pbs_dump_to_dictionaries_invalid_format_raises_error(): invalid_dump = "Invalid text without queue name line" with pytest.raises(QQError, match="Invalid PBS dump format"): parse_multi_pbs_dump_to_dictionaries(invalid_dump, "Job Id") + + +@pytest.mark.parametrize( + ("line", "expected"), + [ + ("qstat: Unknown queue gpu1234", "gpu1234"), + ( + "qstat: Unknown Job Id 100.robox-pro.ceitec.muni.cz", + "100.robox-pro.ceitec.muni.cz", + ), + ( + "Node: node.ceitec.muni.cz, Error: Unknown node", + "node.ceitec.muni.cz", + ), + ], +) +def test_parse_unknown_pbs_object_extracts_identifier(line: str, expected: str) -> None: + assert _parse_unknown_pbs_object(line) == expected + + +@pytest.mark.parametrize( + ("line", "expected"), + [ + (" qstat: Unknown queue gpu1234", "gpu1234"), + ("qstat: Unknown queue gpu1234 ", "gpu1234"), + ("qstat: Unknown queue gpu1234\n", "gpu1234"), + ("\tqstat: Unknown Job Id 100.robox-pro\t", "100.robox-pro"), + ("Node: node.ceitec.muni.cz, Error: Unknown node", "node.ceitec.muni.cz"), + ("Node: node.ceitec.muni.cz,Error: Unknown node", "node.ceitec.muni.cz"), + (" Node: node.ceitec.muni.cz, Error: Unknown node ", "node.ceitec.muni.cz"), + ], +) +def test_parse_unknown_pbs_object_tolerates_surrounding_whitespace( + line: str, expected: str +) -> None: + assert _parse_unknown_pbs_object(line) == expected + + +@pytest.mark.parametrize( + "identifier", + [ + "gpu", + "gpu1234", + "q_2h", + "default@server.ceitec.muni.cz", + ], +) +def test_parse_unknown_pbs_object_extracts_queue_names(identifier: str) -> None: + assert _parse_unknown_pbs_object(f"qstat: Unknown queue {identifier}") == identifier + + +@pytest.mark.parametrize( + "identifier", + [ + "100", + "100.robox-pro.ceitec.muni.cz", + "100[].robox-pro.ceitec.muni.cz", + "100[7].robox-pro.ceitec.muni.cz", + "1234567.pbs-m1.metacentrum.cz", + ], +) +def test_parse_unknown_pbs_object_extracts_job_ids(identifier: str) -> None: + assert ( + _parse_unknown_pbs_object(f"qstat: Unknown Job Id {identifier}") == identifier + ) + + +@pytest.mark.parametrize( + "identifier", + [ + "konos1", + "node.ceitec.muni.cz", + "elmo3-1.elmo.metacentrum.cz", + ], +) +def test_parse_unknown_pbs_object_extracts_node_names(identifier: str) -> None: + assert ( + _parse_unknown_pbs_object(f"Node: {identifier}, Error: Unknown node") + == identifier + ) + + +@pytest.mark.parametrize( + "line", + [ + "", + " ", + "\n", + "Job Id: 100.robox-pro.ceitec.muni.cz", + "Queue: gpu1234", + " job_state = R", + " resources_used.walltime = 00:10:00", + " comment = Job run at Mon Sep 21", + "qstat: invalid option", + "pbsnodes: Server has no node list", + ], +) +def test_parse_unknown_pbs_object_returns_none_for_regular_lines(line: str) -> None: + assert _parse_unknown_pbs_object(line) is None + + +@pytest.mark.parametrize( + "line", + [ + "Job Id: unknown.robox-pro.ceitec.muni.cz", + " Variable_List = QQ_NODE=unknown node", + " comment = Unknown queue gpu1234 was requested", + "Node: node.ceitec.muni.cz, Error: Node is down", + ], +) +def test_parse_unknown_pbs_object_ignores_unknown_inside_regular_lines( + line: str, +) -> None: + assert _parse_unknown_pbs_object(line) is None diff --git a/tests/info/test_info_informer.py b/tests/info/test_info_informer.py index 9f3b077..26b0ffa 100644 --- a/tests/info/test_info_informer.py +++ b/tests/info/test_info_informer.py @@ -518,6 +518,7 @@ def test_informer_from_job_id_returns_informer_when_valid(): def test_informer_from_batch_job_raises_when_no_info_file(): batch_job = MagicMock() + batch_job.is_empty.return_value = False batch_job.get_info_file.return_value = None batch_job.get_id.return_value = "123" @@ -527,6 +528,7 @@ def test_informer_from_batch_job_raises_when_no_info_file(): def test_informer_from_batch_job_raises_on_mismatch(): batch_job = MagicMock() + batch_job.is_empty.return_value = False batch_job.get_info_file.return_value = "info_path" batch_job.get_id.return_value = "123" @@ -545,6 +547,7 @@ def test_informer_from_batch_job_raises_on_mismatch(): def test_informer_from_batch_job_returns_informer_on_success(): batch_job = MagicMock() + batch_job.is_empty.return_value = False batch_job.get_info_file.return_value = "info_path" batch_job.get_id.return_value = "123" @@ -558,6 +561,15 @@ def test_informer_from_batch_job_returns_informer_on_success(): assert result._batch_info is batch_job +def test_informer_from_batch_job_raises_when_empty(): + batch_job = MagicMock() + batch_job.is_empty.return_value = True + batch_job.get_id.return_value = "123" + + with pytest.raises(QQError, match="Job '123' does not exist"): + Informer.from_batch_job(batch_job) + + def test_informer_should_transfer_files_returns_true_on_success(): info_mock = MagicMock(spec=Info) info_mock.transfer_mode = [Success()] From a4ace47c9500de5d4fb13259a524e32305f2a834 Mon Sep 17 00:00:00 2001 From: Ladme Date: Fri, 25 Sep 2026 12:39:16 +0200 Subject: [PATCH 16/19] chore: test for terminate --- tests/core/test_error.py | 77 ++++++++++++++++++++++++++++++++++++++++ 1 file changed, 77 insertions(+) create mode 100644 tests/core/test_error.py diff --git a/tests/core/test_error.py b/tests/core/test_error.py new file mode 100644 index 0000000..90f4a5d --- /dev/null +++ b/tests/core/test_error.py @@ -0,0 +1,77 @@ +# Released under MIT License. +# Copyright (c) 2025-2026 Ladislav Bartos and Robert Vacha Lab + + +import pytest + +from qq_lib.core.error import ( + QQError, + QQJobMismatchError, + QQNotSuitableError, + QQRunCommunicationError, + QQRunFatalError, + terminate, +) + + +@pytest.mark.parametrize( + "message, expected", + [ + ("could not submit script", "could not submit script."), + ("could not submit script.", "could not submit script."), + ("what went wrong?", "what went wrong?"), + ("job failed!", "job failed!"), + ("reasons:", "reasons:"), + ("first; second;", "first; second;"), + ], +) +def test_terminate_adds_dot_only_when_needed(message: str, expected: str) -> None: + assert terminate(message) == expected + + +@pytest.mark.parametrize( + "message, expected", + [ + ("could not submit script ", "could not submit script."), + ("could not submit script\n", "could not submit script."), + ("could not submit script. \n", "could not submit script."), + (" leading space kept", " leading space kept."), + ], +) +def test_terminate_strips_trailing_whitespace(message: str, expected: str) -> None: + assert terminate(message) == expected + + +@pytest.mark.parametrize("message", ["", " ", "\n", " \t\n "]) +def test_terminate_returns_empty_string_for_blank_message(message: str) -> None: + assert terminate(message) == "" + + +def test_terminate_does_not_double_dot_ellipsis() -> None: + assert terminate("waiting...") == "waiting..." + + +def test_terminate_appends_after_quote() -> None: + assert terminate("could not find 'job.sh'") == "could not find 'job.sh'." + + +@pytest.mark.parametrize( + "exception_type", + [ + QQError, + QQJobMismatchError, + QQNotSuitableError, + QQRunFatalError, + QQRunCommunicationError, + ], +) +def test_qq_terminated_mixin_adds_dot(exception_type: type[Exception]) -> None: + exception = exception_type("something broke") + assert str(exception) == "something broke" + assert exception.terminated == "something broke." # ty:ignore[unresolved-attribute] + + +def test_qq_terminated_mixin_keeps_nested_message_undotted() -> None: + inner = QQError("scratch directory is missing") + outer = QQError(f"could not run job: {inner}") + assert outer.terminated == "could not run job: scratch directory is missing." From c3277ec4b15bf059d23190b508c0a454a6ecb0c6 Mon Sep 17 00:00:00 2001 From: Ladme Date: Fri, 25 Sep 2026 13:05:25 +0200 Subject: [PATCH 17/19] feat: updated gromacs run scripts for qq 0.13 --- scripts/run_scripts/qq_flex_md | 24 +++++++++++++++--------- scripts/run_scripts/qq_flex_re | 25 +++++++++++++++---------- scripts/run_scripts/qq_loop_md | 21 +++++++++++++-------- scripts/run_scripts/qq_loop_re | 23 ++++++++++++++--------- 4 files changed, 57 insertions(+), 36 deletions(-) diff --git a/scripts/run_scripts/qq_flex_md b/scripts/run_scripts/qq_flex_md index 2674dc5..3fbb59f 100755 --- a/scripts/run_scripts/qq_flex_md +++ b/scripts/run_scripts/qq_flex_md @@ -1,8 +1,9 @@ #!/usr/bin/env -S qq run ######################################## # Script for running Gromacs # -# flexible-length loop jobs using qq # -# script version: 0.11 # +# flexible-length loop jobs # +# script version: 0.12 # +# requires qq version: >=0.13 # # support: ladmeb@gmail.com # ######################################## @@ -122,11 +123,16 @@ if [[ -z "${MAX_TIME}" ]]; then fi echo "[QQ_FLEX_MD] INFO Setting the maximum Gromacs runtime to ${MAX_TIME} hours." -# create STAGE strings -# CURR - prefix for the data produced in this run -# NEXT - prefix for the checkpoint file for the next run -printf -v CURR "${QQ_ARCHIVE_FORMAT}" "${QQ_LOOP_CURRENT}" -printf -v NEXT "${QQ_ARCHIVE_FORMAT}" "$((QQ_LOOP_CURRENT + 1))" + +# set up stage strings +if [ -z "${QQ_ARCHIVE_CURRENT:-}" ] || [ -z "${QQ_ARCHIVE_NEXT:-}" ]; then + echo "[QQ_FLEX_MD] ERROR QQ_ARCHIVE_CURRENT and QQ_ARCHIVE_NEXT are not set." >&2 + echo "[QQ_FLEX_MD] ERROR You are either not using qq version 0.13 or higher, or your archive format is not a printf pattern." >&2 + exit 9 +fi + +CURR=${QQ_ARCHIVE_CURRENT} +NEXT=${QQ_ARCHIVE_NEXT} # determine gromacs version to set the appropriate appending method GMX_VER="$(gmx_mpi --version | awk 'BEGIN {i = 0} tolower($0) ~ /gromacs version/ {split($NF, N, "."); if (N[1] > 6) i = 1} END {print i}')" @@ -193,7 +199,7 @@ if [[ "${QQ_LOOP_CURRENT}" -eq "${QQ_LOOP_START}" ]]; then GEN_VEL="$(awk 'BEGIN {i = 0} /gen[-_]vel/ {if (toupper($3) == "YES") i = 1} END {print i}' "${MDP}")" if [[ "${GEN_VEL}" -eq 0 ]]; then MAXWARN=0; else MAXWARN=1; fi fi - + # compile the tpr file gmx_mpi grompp -f "${MDP}" -c "${GRO}" ${REF} ${CPT} -p "${TOP}" -n "${NDX}" -o "${CURR}.tpr" -quiet -maxwarn "${MAXWARN}" @@ -225,7 +231,7 @@ if [[ "${QQ_BATCH_SYSTEM}" == *"Slurm"* ]]; then --cpus-per-task=${NTOMP} \ gmx_mpi mdrun -v -deffnm ${CURR} ${CPI} \ -cpo ${NEXT} -cpt 1 -pf ${CURR}_pf.xvg -px ${CURR}_px.xvg \ - ${PLUMED} -ntomp ${NTOMP} ${APPEND} -nb ${NB} -pin on -maxh ${MAX_TIME} + ${PLUMED} -ntomp ${NTOMP} ${APPEND} -nb ${NB} -pin on -maxh ${MAX_TIME} else mpirun \ -display-allocation \ diff --git a/scripts/run_scripts/qq_flex_re b/scripts/run_scripts/qq_flex_re index 1d786d9..39b801a 100755 --- a/scripts/run_scripts/qq_flex_re +++ b/scripts/run_scripts/qq_flex_re @@ -1,9 +1,9 @@ #!/usr/bin/env -S qq run ######################################## # Script for running Gromacs # -# flexible-length replica exchange # -# using qq # -# script version: 0.7 # +# flexible-length replica exchange # +# script version: 0.8 # +# requires qq version: >=0.13 # # support: ladmeb@gmail.com # ######################################## @@ -152,11 +152,16 @@ if [[ -z "${MAX_TIME}" ]]; then fi echo "[QQ_FLEX_RE] INFO Setting the maximum Gromacs runtime to ${MAX_TIME} hours." -# create STAGE strings -# CURR - prefix for the data produced in this run -# NEXT - prefix for the checkpoint file for the next run -printf -v CURR "${QQ_ARCHIVE_FORMAT}" "${QQ_LOOP_CURRENT}" -printf -v NEXT "${QQ_ARCHIVE_FORMAT}" "$((QQ_LOOP_CURRENT + 1))" + +# set up stage strings +if [ -z "${QQ_ARCHIVE_CURRENT:-}" ] || [ -z "${QQ_ARCHIVE_NEXT:-}" ]; then + echo "[QQ_FLEX_RE] ERROR QQ_ARCHIVE_CURRENT and QQ_ARCHIVE_NEXT are not set." >&2 + echo "[QQ_FLEX_RE] ERROR You are either not using qq version 0.13 or higher, or your archive format is not a printf pattern." >&2 + exit 9 +fi + +CURR=${QQ_ARCHIVE_CURRENT} +NEXT=${QQ_ARCHIVE_NEXT} # determine gromacs version to set the appropriate appending method GMX_VER="$(gmx_mpi --version | awk 'BEGIN {i = 0} tolower($0) ~ /gromacs version/ {split($NF, N, "."); if (N[1] > 6) i = 1} END {print i}')" @@ -303,7 +308,7 @@ for ((I=0; I<${#CLIENTS[@]}; I++)); do SUFFIX=${CLI_SUFFIXES[I]} cd ${CLI} - + # rename output files from gromacs 2016 and later for PART in *part*; do [[ -e "${PART}" ]] || break @@ -349,4 +354,4 @@ echo "[QQ_FLEX_RE] INFO Last step: ${LAST_STEP} and target step: ${TARGET_ if [[ "${LAST_STEP}" -ge "${TARGET_STEP}" ]]; then # signal to qq that the job should not be resubmitted exit "${QQ_NO_RESUBMIT}" -fi \ No newline at end of file +fi diff --git a/scripts/run_scripts/qq_loop_md b/scripts/run_scripts/qq_loop_md index 389054b..8c17f5e 100755 --- a/scripts/run_scripts/qq_loop_md +++ b/scripts/run_scripts/qq_loop_md @@ -1,8 +1,9 @@ #!/usr/bin/env -S qq run ######################################## # Script for running Gromacs # -# loop jobs using qq # -# script version: 0.11 # +# fixed length loop jobs # +# script version: 0.12 # +# requires qq version: >=0.13 # # support: ladmeb@gmail.com # ######################################## @@ -113,11 +114,15 @@ else NB="gpu" fi -# create STAGE strings -# CURR - prefix for the data produced in this run -# NEXT - prefix for the checkpoint file for the next run -printf -v CURR "${QQ_ARCHIVE_FORMAT}" "${QQ_LOOP_CURRENT}" -printf -v NEXT "${QQ_ARCHIVE_FORMAT}" "$((QQ_LOOP_CURRENT + 1))" +# set up stage strings +if [ -z "${QQ_ARCHIVE_CURRENT:-}" ] || [ -z "${QQ_ARCHIVE_NEXT:-}" ]; then + echo "[QQ_LOOP_MD] ERROR QQ_ARCHIVE_CURRENT and QQ_ARCHIVE_NEXT are not set." >&2 + echo "[QQ_LOOP_MD] ERROR You are either not using qq version 0.13 or higher, or your archive format is not a printf pattern." >&2 + exit 9 +fi + +CURR=${QQ_ARCHIVE_CURRENT} +NEXT=${QQ_ARCHIVE_NEXT} # determine gromacs version to set the appropriate appending method GMX_VER="$(gmx_mpi --version | awk 'BEGIN {i = 0} tolower($0) ~ /gromacs version/ {split($NF, N, "."); if (N[1] > 6) i = 1} END {print i}')" @@ -184,7 +189,7 @@ if [[ "${QQ_LOOP_CURRENT}" -eq "${QQ_LOOP_START}" ]]; then GEN_VEL="$(awk 'BEGIN {i = 0} /gen[-_]vel/ {if (toupper($3) == "YES") i = 1} END {print i}' "${MDP}")" if [[ "${GEN_VEL}" -eq 0 ]]; then MAXWARN=0; else MAXWARN=1; fi fi - + # compile the tpr file gmx_mpi grompp -f "${MDP}" -c "${GRO}" ${REF} ${CPT} -p "${TOP}" -n "${NDX}" -o "${CURR}.tpr" -quiet -maxwarn "${MAXWARN}" diff --git a/scripts/run_scripts/qq_loop_re b/scripts/run_scripts/qq_loop_re index 1db666c..27ca61d 100755 --- a/scripts/run_scripts/qq_loop_re +++ b/scripts/run_scripts/qq_loop_re @@ -1,8 +1,9 @@ #!/usr/bin/env -S qq run ######################################## # Script for running Gromacs # -# multidir loop jobs using qq # -# script version: 0.7 # +# multidir fixed length loop jobs # +# script version: 0.8 # +# requires qq version: >=0.13 # # support: ladmeb@gmail.com # ######################################## @@ -142,11 +143,15 @@ if [[ -n "${RE_FREQ}" ]]; then fi fi -# create STAGE strings -# CURR - prefix for the data produced in this run -# NEXT - prefix for the checkpoint file for the next run -printf -v CURR "${QQ_ARCHIVE_FORMAT}" "${QQ_LOOP_CURRENT}" -printf -v NEXT "${QQ_ARCHIVE_FORMAT}" "$((QQ_LOOP_CURRENT + 1))" +# set up stage strings +if [ -z "${QQ_ARCHIVE_CURRENT:-}" ] || [ -z "${QQ_ARCHIVE_NEXT:-}" ]; then + echo "[QQ_LOOP_RE] ERROR QQ_ARCHIVE_CURRENT and QQ_ARCHIVE_NEXT are not set." >&2 + echo "[QQ_LOOP_RE] ERROR You are either not using qq version 0.13 or higher, or your archive format is not a printf pattern." >&2 + exit 9 +fi + +CURR=${QQ_ARCHIVE_CURRENT} +NEXT=${QQ_ARCHIVE_NEXT} # determine gromacs version to set the appropriate appending method GMX_VER="$(gmx_mpi --version | awk 'BEGIN {i = 0} tolower($0) ~ /gromacs version/ {split($NF, N, "."); if (N[1] > 6) i = 1} END {print i}')" @@ -293,7 +298,7 @@ for ((I=0; I<${#CLIENTS[@]}; I++)); do SUFFIX=${CLI_SUFFIXES[I]} cd ${CLI} - + # rename output files from gromacs 2016 and later for PART in *part*; do [[ -e "${PART}" ]] || break @@ -330,4 +335,4 @@ for ((I=0; I<${#CLIENTS[@]}; I++)); do done cd .. -done \ No newline at end of file +done From 72858f9952860f8d677cdab5fab77defabb13da4 Mon Sep 17 00:00:00 2001 From: Ladme Date: Fri, 25 Sep 2026 13:09:44 +0200 Subject: [PATCH 18/19] chore: update gmx-eta to qq v0.13 --- scripts/qq_scripts/gmx-eta | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/qq_scripts/gmx-eta b/scripts/qq_scripts/gmx-eta index b126172..5a4819f 100755 --- a/scripts/qq_scripts/gmx-eta +++ b/scripts/qq_scripts/gmx-eta @@ -5,7 +5,7 @@ """ Get the estimated time of a Gromacs simulation finishing. -Version qq 0.12.1. +Version qq 0.13.0. Requires `uv`: https://docs.astral.sh/uv """ @@ -16,7 +16,7 @@ Requires `uv`: https://docs.astral.sh/uv # ] # # [tool.uv.sources] -# qq = { git = "https://github.com/VachaLab/qq.git", tag = "v0.12.1" } +# qq = { git = "https://github.com/VachaLab/qq.git", tag = "v0.13.0" } # /// import argparse From 45f374b7cb5f107f4ca0a72dac8fb2e8b2d335c3 Mon Sep 17 00:00:00 2001 From: Ladme Date: Fri, 25 Sep 2026 14:33:25 +0200 Subject: [PATCH 19/19] docs: fix issues in changelog --- CHANGELOG.md | 18 +++++++++--------- 1 file changed, 9 insertions(+), 9 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 59a4f23..cd7b6d2 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -3,25 +3,25 @@ ### New environment variables for loop jobs - Loop jobs now expose three additional environment variables: - - `QQ_LOOP_NEXT` which specifies the index of the next loop cycle. - - `QQ_ARCHIVE_CURRENT` and `QQ_ARCHIVE_NEXT` which specify strings expected in files archived by qq for the current and the next loop cycles, respectively. If the archive format is not a printf pattern, the values of these variables are empty strings. + - `QQ_LOOP_NEXT`, which specifies the index of the next loop cycle. + - `QQ_ARCHIVE_CURRENT` and `QQ_ARCHIVE_NEXT`, which specify strings expected in files archived by qq for the current and the next loop cycles, respectively. If the archive format is not a printf pattern, the values of these variables are empty strings. ### `--include` and `--exclude` revisited - Glob patterns are now supported in `--include` and `--exclude` options. -- If you explicitly include a file into a working directory using the `--include` submission option, it will no longer be archived even if it matches the archive pattern. -- If you explicitly exclude a file from a working directory using the `--exclude` submission option, it will no longer be fetched from the archive, even if it matches the archive pattern for this loop job cycle. +- If you explicitly include a file in the working directory using the `--include` submission option, it will no longer be archived even if it matches the archive pattern. +- If you explicitly exclude a file from the working directory using the `--exclude` submission option, it will no longer be fetched from the archive, even if it matches the archive pattern for this loop job cycle. ### `--ignore` option -- As a complement to `--include` and `--exclude`, `--ignore` allows to specify files that should be completely ignored by qq's transfer operations. Ignored files will never be copied to working directory; if they are created in working directory, they will never be transferred to the input directory. If the job is a loop job, ignored files will also never be archived nor fetched from the archive. +- As a complement to `--include` and `--exclude`, `--ignore` allows you to specify files that should be completely ignored by qq's transfer operations. Ignored files will never be copied to the working directory; if they are created in the working directory, they will never be transferred to the input directory. If the job is a loop job, ignored files will also never be archived nor fetched from the archive. ### Bug fixes and other changes -- Directories can be now properly archived. -- qq commands operating on multiple job IDs no longer fail on the first occurence of a job that does not exist. They report an error and continue processing the remaining jobs. -- Archive directory created in a working directory is no longer merged with the actual archive directory in the input directory. -- Number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. +- Directories can now be properly archived. +- qq commands operating on multiple job IDs no longer fail on the first occurrence of a job that does not exist. They report an error and continue processing the remaining jobs. +- The archive directory created in the working directory is no longer merged with the actual archive directory in the input directory. +- The number of free GPUs is no longer relevant for determining node state in `qq nodes`. Nodes with exhausted CPUs will always be marked as busy even if they have free GPUs. - Clarified in `qq submit -h` that files and directories that the job creates in the working directory **are** copied back to the input directory **even if they are excluded** from being copied to the working directory. - Error messages reformatted to be easier to compose.