From d3d7f0db75bfd6d3441cdf9a998606151623aad1 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 3 Dec 2024 15:13:04 -0600 Subject: [PATCH 001/161] Exporting OTF data to CryoSparc --- emtools/scripts/emt-scipion-otf.py | 194 ++++++++++++++++++++++++++++- 1 file changed, 193 insertions(+), 1 deletion(-) diff --git a/emtools/scripts/emt-scipion-otf.py b/emtools/scripts/emt-scipion-otf.py index b947d39..34d3d4a 100755 --- a/emtools/scripts/emt-scipion-otf.py +++ b/emtools/scripts/emt-scipion-otf.py @@ -24,6 +24,7 @@ from collections import OrderedDict import datetime as dt import re +from pprint import pprint from emtools.utils import Process, Color, System from emtools.metadata import EPU, SqliteFile, StarFile, Table @@ -314,7 +315,7 @@ def _path(*p): sphericalAberration=acq['cs'], doseInitial=0.0, dosePerFrame=acq['dose'], - gainFile=gain, + gainFile=os.path.abspath(gain), dataStreaming=True ) @@ -346,6 +347,7 @@ def _path(*p): 'motioncorr.protocols.ProtMotionCorrTasks', objLabel='motioncor', patchX=patchX, patchY=patchY, + gainFlip=1, # Fli numberOfThreads=1, streamingBatchSize=16, gpuList=' '.join(str(g) for g in params['mcGpus']) @@ -701,6 +703,179 @@ def fix_run_links(workingDir, srcRuns): logger.system(f"cd Runs && ln -s runs/{fn}") +class CryoSparc: + STATUS_FAILED = "failed" + STATUS_ABORTED = "aborted" + STATUS_COMPLETED = "completed" + STATUS_KILLED = "killed" + STATUS_RUNNING = "running" + STATUS_QUEUED = "queued" + STATUS_LAUNCHED = "launched" + STATUS_STARTED = "started" + STATUS_BUILDING = "building" + + STOP_STATUSES = [STATUS_ABORTED, STATUS_COMPLETED, STATUS_FAILED, STATUS_KILLED] + ACTIVE_STATUSES = [STATUS_QUEUED, STATUS_RUNNING, STATUS_STARTED, + STATUS_LAUNCHED, STATUS_BUILDING] + + def __init__(self, projId): + self.projId = projId + from cryosparc.tools import CryoSPARC, CommandClient + cs_config = os.environ.get('CRYOSPARC_CONFIG', None) + if cs_config is None: + raise Exception('Please define CRYOSPARC_CONFIG="LICENSE|URL|PORT"') + + license, url, port = cs_config.split('|') + print("\n>>> Using license: ", Color.green(license)) + print(">>> URL/port: ", Color.bold(f"{url}:{port}")) + self._cli = CommandClient(host=url, port=port, headers={"License-ID": license}) + projInfo = self.cli('get_project', projId) + print("\n", "=" * 20, Color.green(f"PROJECT: {projId}"), "=" * 20) + pprint(projInfo) + print("=" * 50, "\n") + self.userId = projInfo['owner_user_id'] + lanes = self.cli('get_scheduler_lanes') + pprint(lanes) + + def __call__(self, cmd, **kwargs): + p = Process(self.csm, 'cli', cmd) + lines = list(p.lines()) + + try: + for i, line in enumerate(lines): + print(Color.cyan(i), Color.bold(line)) + return lines[0] + except Exception as e: + print(Color.red(f"Error: running command {cmd}")) + print(e) + + def _argstr(self, args): + return json.dumps(args).replace('true', 'True') + + def cli(self, function, *args, **kwargs): + def _val(v): + return Color.bold(json.dumps(v)) + + argsStr = ','.join(_val(a) for a in args) + sepStr = ', ' if argsStr else '' + kwargsStr = ','.join("%s=%s" % (Color.cyan(k), _val(v)) for k, v in kwargs.items()) + print(f"\n{Color.green(function)}({argsStr}{sepStr}{kwargsStr})") + func = getattr(self._cli, function) + return func(*args, **kwargs) + + def job_status(self, jobId): + """ Return the job status. """ + status = self.cli('get_job_status', project_uid=self.projId, job_uid=jobId) + print(status) + return status + + def job_wait(self, jobId): + """ Wait for a job to complete (in any stop status). """ + while self.job_status(jobId) not in self.STOP_STATUSES: + time.sleep(10) + + def job_run(self, wsId, jobType, args, inputs={}, wait=True): + #cmd = (f'make_job("{jobType}", "{self.projId}", "{wsId}", "{self.userId}", None, None, None, ' + # f'{self._argstr(args)}, {self._argstr(inputs)})') + #jobId = self(cmd) + # jobId = self._cli.make_job(job_type=jobType, project_uid=self.projId, workspace_uid=wsId, + # user_id=self.userId, params=args, input_group_connects=inputs) + jobId = self.cli('make_job', + job_type=jobType, project_uid=self.projId, workspace_uid=wsId, + user_id=self.userId, params=args, input_group_connects=inputs) + #cmd = f'enqueue_job("{self.projId}", "{jobId}", "default", "{self.userId}")' + #self(cmd) + #self._cli.enqueue_job(project_uid=self.projId, user_id=self.userId, job_uid=jobId, lane='default') + self.cli('enqueue_job', + project_uid=self.projId, user_id=self.userId, job_uid=jobId) + if wait: + self.job_wait(jobId) + + return jobId + + +def cryosparc_prepare(): + if os.path.exists('CS'): + raise Exception("CS folder already exists. Remove it before running this command.") + + logger = Process.Logger(format="%(message)s", only_log=False)#True) + + for folder in ['Micrographs', 'Movies', 'XML']: + logger.mkdir(f'CS/{folder}') + + fn = 'micrographs_ctf.star' + + with StarFile(fn) as sf: + with StarFile('CS/particles.star', 'w') as sfOut: + ctfCols = ['rlnDefocusU', 'rlnDefocusV', 'rlnDefocusAngle', 'rlnCtfFigureOfMerit', 'rlnCtfMaxResolution'] + ctfCols = [] # CS is giving an error when using CTF + t = Table(['rlnMicrographName', 'rlnCoordinateX', 'rlnCoordinateY'] + ctfCols) + print("cols", len(t.getColumnNames()), t.getColumnNames()) + + sfOut.writeHeader('particles', t) + + for row in sf.iterTable('micrographs'): + micFn = row.rlnMicrographName + movFn = row.rlnMicrographMovieName.replace('Images-Disc1_', '') + xmlFn = movFn.replace('_EER.eer', '.xml') + base = os.path.basename(micFn) + micName = base.replace('_DW.mrc', '') + movName = micName.replace('mic_', 'mov_') + logger.system(f'ln -s ../../{micFn} CS/Micrographs/{micName}.mrc') + logger.system(f'ln -s ../../{movFn} CS/Movies/{movName}.eer') + logger.system(f'ln -s ../../{xmlFn} CS/XML/{movName}.xml') + coordsFn = f'Coordinates/{micName}_DW_coordinates.star' + ctfValues = [getattr(row, k) for k in ctfCols] + print(len(ctfValues)) + with StarFile(coordsFn) as sfCoords: + for rowCoord in sfCoords.iterTable(''): + sfOut.writeRow(t.Row(f'{micName}.mrc', + rowCoord.rlnCoordinateX, + rowCoord.rlnCoordinateY, + *ctfValues)) + + +def cryosparc_import(projId, dataRoot): + + acq = { + "psize_A": 0.724, + "accel_kv": 300, + "cs_mm": 0.1, + } + + cs = CryoSparc(projId) + csRoot = os.path.join(dataRoot, 'CS') + + print(f">>> Importing data from: {Color.green(dataRoot)}") + + args = { + "blob_paths": f"{csRoot}/Micrographs/mic_*.mrc", + "total_dose_e_per_A2": 40, + "parse_xml_files": True, + "xml_paths": f"{csRoot}/XML/mov_*.xml", + "mov_cut_prefix_xml": 4, + "mov_cut_suffix_xml": 4, + "xml_cut_prefix_xml": 4, + "xml_cut_suffix_xml": 4 + } + args.update(acq) + micsImport = cs.job_run("W1", "import_micrographs", args) + time.sleep(5) # FIXME: wait for job completion + args = { + "ignore_blob": True, + "particle_meta_path": f"{csRoot}/particles.star", + "query_cut_suff": 4, + "remove_leading_uid": True, + "source_cut_suff": 4, + "enable_validation": True, + "location_exists": True, + "amp_contrast": 2.7, + } + args.update(acq) + ptsImport = cs.job_run("W1", "import_particles", args, + {'micrographs': f'{micsImport}.imported_micrographs'}) + + def main(): p = argparse.ArgumentParser(prog='scipion-otf') g = p.add_mutually_exclusive_group() @@ -728,6 +903,15 @@ def main(): "and the Cryolo picking for picking. One can pass a string" "with the protocol ids for ctfs and/or picking. For example:" "--write_starts 'ctfs=1524 picking=1711'") + g.add_argument('--cs_prepare', action='store_true', + help="Prepare a folder CS to be used to import movies, micrographs " + "and particles into CryoSparc. ") + g.add_argument('--cs_import', nargs='+', + metavar=('CRYOSPARC_PROJECT_ID', 'DATA_ROOT'), + help="Import data from CS into a running project. ") + #get_scheduler_lanes + g.add_argument('--cs_test', metavar='CRYOSPARC_PROJECT_ID', + help="Test connection to CryoSparc server. ") g.add_argument('--clone_project', nargs=2, metavar=('SRC', 'DST'), help="Clone an existing Scipion project") g.add_argument('--fix_run_links', metavar='RUNS_SRC', @@ -762,6 +946,14 @@ def main(): fix_run_links(cwd, args.fix_run_links) elif protId := args.print_protocol: print_protocol(cwd, protId) + elif cs := args.cs_prepare: + cryosparc_prepare() + elif projId := args.cs_test: + cs = CryoSparc(projId) + elif cs := args.cs_import: + projId = cs[0] + dataRoot = cs[1] + cryosparc_import(projId, dataRoot) else: # by default open the GUI from pyworkflow.gui.project import ProjectWindow ProjectWindow(cwd).show() From e8d76702ff169d7d93551309d918008e44fe9cce Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Thu, 19 Dec 2024 15:14:03 -0600 Subject: [PATCH 002/161] Some updates in ProcessingPipeline --- emtools/jobs/pipeline.py | 26 ++++++++++++++++++++++++-- 1 file changed, 24 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/pipeline.py b/emtools/jobs/pipeline.py index 5f40420..69784df 100644 --- a/emtools/jobs/pipeline.py +++ b/emtools/jobs/pipeline.py @@ -20,6 +20,9 @@ import threading import signal import traceback +from uuid import uuid4 + +from emtools.utils import Process class Pipeline: @@ -195,10 +198,15 @@ class ProcessingPipeline(Pipeline): It will also add some helper functions to manipulate file paths relative to the working dir. """ - def __init__(self, workingDir, outputDir, **kwargs): - Pipeline.__init__(self, **kwargs) + def __init__(self, **kwargs): + workingDir = kwargs.pop('working_dir', os.getcwd()) + outputDir = kwargs.pop('output_dir', None) + scratchDir = kwargs.pop('scratch', None) + Pipeline.__init__(self, debug=kwargs.get('debug', False)) self.workingDir = self.__validate(workingDir, 'working') self.outputDir = self.__validate(outputDir, 'output') + self.scratchDir = self.__validate(scratchDir, 'scratch') if scratchDir else None + self.tmpDir = self.join('tmp') def __validate(self, path, key): if not path: @@ -208,6 +216,18 @@ def __validate(self, path, key): return path + def __clean_tmp(self): + Process.system(f"rm -rf {self.tmpDir}") + + def __create_tmp(self): + self.__clean_tmp() + + if self.scratchDir: + scratchTmp = os.path.join(self.scratchDir, str(uuid4())) + Process.system(f"ln -s {scratchTmp} {self.tmpDir}") + else: + Process.system(f"mkdir {self.tmpDir}") + def get_arg(self, argDict, key, envKey, default=None): """ Get an argument from the argDict or from the environment. @@ -245,9 +265,11 @@ def run(self): try: signal.signal(signal.SIGINT, self.__abort) signal.signal(signal.SIGTERM, self.__abort) + self.__create_tmp() self.prerun() Pipeline.run(self) self.postrun() + self.__clean_tmp() self.__file('SUCCESS') except Exception as e: self.__file('FAILURE') From 21b699f2edac44d31286936b0c569751ab3aecd0 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sat, 4 Jan 2025 12:37:15 -0600 Subject: [PATCH 003/161] Added context manager for tmpDir in Path class --- emtools/utils/path.py | 30 ++++++++++++++++++++++++++++++ 1 file changed, 30 insertions(+) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index e256170..8a6da1a 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -15,12 +15,16 @@ # ************************************************************************** import os +import shutil import time +import tempfile from datetime import datetime as dt from collections import OrderedDict +from contextlib import contextmanager from .pretty import Pretty from .process import Process +from .color import Color class Path: @@ -163,6 +167,32 @@ def _mkdir(d): _copy(os.path.join(root, f), os.path.join(root2, f), **kwargs) + @staticmethod + @contextmanager + def tmpDir(**kwargs): + tmp = tempfile.mkdtemp(prefix=kwargs.get('prefix', '')) + + chdir = kwargs.get('chdir', False) + cwd = os.getcwd() + if chdir: + os.chdir(tmp) + + if kwargs.get('verbose', True): + print(f"Using temporary dir: {tmp}") + + yield tmp + + if chdir: + os.chdir(cwd) + + globalClean = int(os.environ.get('EMWRAP_CLEAN', 1)) + if kwargs.get('clean', globalClean): + shutil.rmtree(tmp) + else: + print(f"Temporary directy was not deleted, " + f"remove it with the following command: \n" + f"{Color.bold('rm -rf %s' % tmp)}") + @staticmethod def replaceExt(filename, newExt): """ Replace the current path extension(from last .) From 4cba58b8b2445a2ba623353933163d35b15d1cb8 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sat, 4 Jan 2025 17:02:27 -0600 Subject: [PATCH 004/161] Created Batch class as a utility subclass from dict --- emtools/jobs/batch_manager.py | 32 ++++++++++++++++++++++++++++++-- 1 file changed, 30 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index e4ebcd2..4b36dda 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -21,6 +21,34 @@ from emtools.utils import Process +class Batch(dict): + """ Subclass from dict with some utilities related to Batch logic. """ + @property + def id(self): + return self['id'] + + @property + def path(self): + return self['path'] + + @property + def index(self): + return self['index'] + + @property + def items(self): + return self['items'] + + def join(self, *p): + return os.path.join(self['path'], *p) + + def mkdir(self, *p): + return os.mkdir(self.join(*p)) + + def exists(self, *p): + return os.path.exists(self.join(*p)) + + class BatchManager: """ Class used to generate and handle the creation of batches from an input stream of items. @@ -70,12 +98,12 @@ def _createBatch(self, items, inputFolder=None): os.path.join(batch_path, baseName)) self._batchCount += 1 - return { + return Batch({ 'items': items, 'id': batch_id, 'path': batch_path, 'index': self._batchCount - } + }) def generate(self): """ Generate batches based on the input items. """ From c483317b35212e2288fac145db968d0f577d9290 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sun, 5 Jan 2025 12:46:44 -0600 Subject: [PATCH 005/161] Added helper Args class --- emtools/jobs/__init__.py | 4 ++-- emtools/jobs/batch_manager.py | 22 ++++++++++++++++++++++ emtools/metadata/starfile.py | 7 +++++++ 3 files changed, 31 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index 672d307..88f32fe 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -15,6 +15,6 @@ # ************************************************************************** from .pipeline import Pipeline, ProcessingPipeline -from .batch_manager import BatchManager +from .batch_manager import Args, Batch, BatchManager -__all__ = ["Pipeline", "BatchManager", "ProcessingPipeline"] \ No newline at end of file +__all__ = ["Pipeline", "BatchManager", "ProcessingPipeline", "Batch"] diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 4b36dda..a478a71 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -17,10 +17,22 @@ import os from uuid import uuid4 from datetime import datetime +import itertools +import json from emtools.utils import Process +class Args(dict): + """ Subclass from dict with some utilities related to arguments. """ + + def toList(self): + return list(itertools.chain.from_iterable([str(k), str(v)] for k, v in self.items())) + + def toLine(self): + return ' '.join("%s %s" % (k, v) for k, v in self.items()) + + class Batch(dict): """ Subclass from dict with some utilities related to Batch logic. """ @property @@ -39,6 +51,12 @@ def index(self): def items(self): return self['items'] + @property + def info(self): + if 'info' not in self: + self['info'] = {} + return self['info'] + def join(self, *p): return os.path.join(self['path'], *p) @@ -48,6 +66,10 @@ def mkdir(self, *p): def exists(self, *p): return os.path.exists(self.join(*p)) + def dump_info(self): + with open(self.join('info.json'), 'w') as batch_info: + json.dump(self.info, batch_info, indent=4) + class BatchManager: """ Class used to generate and handle the creation of batches diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index a2f2a96..141fd05 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -29,6 +29,8 @@ from collections import OrderedDict from datetime import datetime, timedelta +import emtools +from emtools.utils import Pretty from .table import ColumnList, Table @@ -319,6 +321,11 @@ def writeLine(self, line): """ Write a line to the opened file. """ self._file.write(f"{line}\n") + def writeTimeStamp(self): + """ Write a comment line with current datetime and library version. """ + self.writeLine(f"\n# StarFile written on {Pretty.now()} " + f"by emtools ({emtools.__version__})\n") + def _writeTableName(self, tableName): self._file.write("\ndata_%s\n\n" % (tableName or '')) From f2d54134b3c2a760d624764fc35531058ab8cc44 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sun, 5 Jan 2025 12:50:15 -0600 Subject: [PATCH 006/161] Bumped devel version --- emtools/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/emtools/__init__.py b/emtools/__init__.py index 26dc68e..4f29437 100644 --- a/emtools/__init__.py +++ b/emtools/__init__.py @@ -24,5 +24,5 @@ # * # ************************************************************************** -__version__ = '0.1.3' +__version__ = '0.1.4rc' From d31dfa8cb8547e61c97f0ff8bc6b260043b6396f Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sun, 5 Jan 2025 17:04:31 -0600 Subject: [PATCH 007/161] Allow to pass input to process as stdin --- emtools/jobs/__init__.py | 2 +- emtools/utils/process.py | 5 +++-- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index 88f32fe..fe62aac 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -17,4 +17,4 @@ from .pipeline import Pipeline, ProcessingPipeline from .batch_manager import Args, Batch, BatchManager -__all__ = ["Pipeline", "BatchManager", "ProcessingPipeline", "Batch"] +__all__ = ["Pipeline", "BatchManager", "ProcessingPipeline", "Batch", "Args"] diff --git a/emtools/utils/process.py b/emtools/utils/process.py index 7f47c6c..134c3b6 100644 --- a/emtools/utils/process.py +++ b/emtools/utils/process.py @@ -29,7 +29,8 @@ def __init__(self, *args, **kwargs): self.args = args error = '' try: - self._p = subprocess.run(args, capture_output=True, text=True) + self._p = subprocess.run(args, capture_output=True, text=True, + input=kwargs.get('input', None)) self.stdout = self._p.stdout self.stderr = self._p.stderr self.returncode = self._p.returncode @@ -46,7 +47,7 @@ def __init__(self, *args, **kwargs): def lines(self): """ Iterate over the lines of the process output. """ - for line in self.stdout.split('\n'): + for line in self.stdout.splitlines(): yield line def print(self, args=True, stdout=False): From 99a6ae1af738779a4ea8d667588fd6349daed348 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 7 Jan 2025 11:19:29 -0600 Subject: [PATCH 008/161] Added functions to Batch class --- emtools/jobs/batch_manager.py | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index a478a71..2e27c33 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -19,8 +19,9 @@ from datetime import datetime import itertools import json +import subprocess -from emtools.utils import Process +from emtools.utils import Process, Color class Args(dict): @@ -61,7 +62,9 @@ def join(self, *p): return os.path.join(self['path'], *p) def mkdir(self, *p): - return os.mkdir(self.join(*p)) + d = self.join(*p) + os.mkdir(d) + return d def exists(self, *p): return os.path.exists(self.join(*p)) @@ -70,6 +73,11 @@ def dump_info(self): with open(self.join('info.json'), 'w') as batch_info: json.dump(self.info, batch_info, indent=4) + def call(self, args, logfile): + with open(logfile, 'w') as f: + print(">>>", Color.green(args[0]), Color.bold(' '.join(args[1:]))) + subprocess.call(args, cwd=self.path, stderr=f, stdout=f) + class BatchManager: """ Class used to generate and handle the creation of batches From 17593cf3e1caeb2de27a5d32cfe586f011a41459 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 7 Jan 2025 16:57:36 -0600 Subject: [PATCH 009/161] Some fixs in Args --- emtools/jobs/batch_manager.py | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 2e27c33..30e2925 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -28,7 +28,12 @@ class Args(dict): """ Subclass from dict with some utilities related to arguments. """ def toList(self): - return list(itertools.chain.from_iterable([str(k), str(v)] for k, v in self.items())) + args = [] + for k, v in self.items(): + args.append(str(k)) + if v: + args.append((str(v))) + return args def toLine(self): return ' '.join("%s %s" % (k, v) for k, v in self.items()) @@ -61,6 +66,9 @@ def info(self): def join(self, *p): return os.path.join(self['path'], *p) + def relpath(self, p): + return os.path.relpath(p, self.path) + def mkdir(self, *p): d = self.join(*p) os.mkdir(d) @@ -73,7 +81,8 @@ def dump_info(self): with open(self.join('info.json'), 'w') as batch_info: json.dump(self.info, batch_info, indent=4) - def call(self, args, logfile): + def call(self, program, kwargs, logfile): + args = [program] + Args(kwargs).toList() with open(logfile, 'w') as f: print(">>>", Color.green(args[0]), Color.bold(' '.join(args[1:]))) subprocess.call(args, cwd=self.path, stderr=f, stdout=f) From b4543faa3f4bfaad4c9542c24b0671b2ac962c16 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Thu, 9 Jan 2025 12:58:54 -0600 Subject: [PATCH 010/161] Minor fixes and new Table.fromDict function --- emtools/jobs/batch_manager.py | 6 ++++-- emtools/metadata/table.py | 6 ++++++ 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 30e2925..85ae5d1 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -31,7 +31,7 @@ def toList(self): args = [] for k, v in self.items(): args.append(str(k)) - if v: + if v != '': args.append((str(v))) return args @@ -84,7 +84,9 @@ def dump_info(self): def call(self, program, kwargs, logfile): args = [program] + Args(kwargs).toList() with open(logfile, 'w') as f: - print(">>>", Color.green(args[0]), Color.bold(' '.join(args[1:]))) + cmd = f">>> {Color.green(args[0])} {Color.bold(' '.join(args[1:]))}" + print(cmd) + f.write(f"\n{cmd}\n") subprocess.call(args, cwd=self.path, stderr=f, stdout=f) diff --git a/emtools/metadata/table.py b/emtools/metadata/table.py index 26126f3..f1f73b8 100644 --- a/emtools/metadata/table.py +++ b/emtools/metadata/table.py @@ -143,6 +143,12 @@ def __init__(self, columns=None): self.Row = self.createRowClass() self._rows = [] + @staticmethod + def fromDict(valuesDict): + t = Table(list(valuesDict.keys())) + t.addRowValues(**valuesDict) + return t + def clear(self): self.Row = None self._columns.clear() From 2eae5bfb5b1ada73a7ae905ec63403af645fa412 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Thu, 9 Jan 2025 16:23:59 -0600 Subject: [PATCH 011/161] Moved ProcessingPipeline from emtools to emwrap --- emtools/jobs/__init__.py | 4 +- emtools/jobs/batch_manager.py | 1 + emtools/jobs/pipeline.py | 90 ----------------------------------- emtools/utils/path.py | 2 +- 4 files changed, 4 insertions(+), 93 deletions(-) diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index fe62aac..c760f41 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -14,7 +14,7 @@ # * # ************************************************************************** -from .pipeline import Pipeline, ProcessingPipeline +from .pipeline import Pipeline from .batch_manager import Args, Batch, BatchManager -__all__ = ["Pipeline", "BatchManager", "ProcessingPipeline", "Batch", "Args"] +__all__ = ["Pipeline", "BatchManager", "Batch", "Args"] diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 85ae5d1..adce05c 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -87,6 +87,7 @@ def call(self, program, kwargs, logfile): cmd = f">>> {Color.green(args[0])} {Color.bold(' '.join(args[1:]))}" print(cmd) f.write(f"\n{cmd}\n") + f.flush() subprocess.call(args, cwd=self.path, stderr=f, stdout=f) diff --git a/emtools/jobs/pipeline.py b/emtools/jobs/pipeline.py index 69784df..266b671 100644 --- a/emtools/jobs/pipeline.py +++ b/emtools/jobs/pipeline.py @@ -18,9 +18,6 @@ import sys from collections import OrderedDict import threading -import signal -import traceback -from uuid import uuid4 from emtools.utils import Process @@ -189,90 +186,3 @@ def _process(self): self._print("Got task: None") - -class ProcessingPipeline(Pipeline): - """ Subclass of Pipeline that is commonly used to run programs. - - This class will define a workingDir (usually os.getcwd) - and an output dir where all output should be generated. - It will also add some helper functions to manipulate file - paths relative to the working dir. - """ - def __init__(self, **kwargs): - workingDir = kwargs.pop('working_dir', os.getcwd()) - outputDir = kwargs.pop('output_dir', None) - scratchDir = kwargs.pop('scratch', None) - Pipeline.__init__(self, debug=kwargs.get('debug', False)) - self.workingDir = self.__validate(workingDir, 'working') - self.outputDir = self.__validate(outputDir, 'output') - self.scratchDir = self.__validate(scratchDir, 'scratch') if scratchDir else None - self.tmpDir = self.join('tmp') - - def __validate(self, path, key): - if not path: - raise Exception(f'Invalid {key} directory: {path}') - if not os.path.exists(path): - raise Exception(f'Non-existing {key} directory: {path}') - - return path - - def __clean_tmp(self): - Process.system(f"rm -rf {self.tmpDir}") - - def __create_tmp(self): - self.__clean_tmp() - - if self.scratchDir: - scratchTmp = os.path.join(self.scratchDir, str(uuid4())) - Process.system(f"ln -s {scratchTmp} {self.tmpDir}") - else: - Process.system(f"mkdir {self.tmpDir}") - - def get_arg(self, argDict, key, envKey, default=None): - """ Get an argument from the argDict or from the environment. - - Args: - argDict: arguments dict from where to get the 'key' value - key: string key of the argument name in argDict - envKey: string key of the environment variable - default: default value if not found in argDict or environ - """ - return argDict.get(key, os.environ.get(envKey, default)) - - def join(self, *p): - return os.path.join(self.outputDir, *p) - - def relpath(self, p): - return os.path.relpath(p, self.workingDir) - - def prerun(self): - """ This method will be called before the run. """ - pass - - def postrun(self): - """ This method will be called after the run. """ - pass - - def __file(self, suffix): - with open(self.join(f'RELION_JOB_EXIT_{suffix}'), 'w'): - pass - - def __abort(self, signum, frame): - self.__file('ABORTED') - sys.exit(0) - - def run(self): - try: - signal.signal(signal.SIGINT, self.__abort) - signal.signal(signal.SIGTERM, self.__abort) - self.__create_tmp() - self.prerun() - Pipeline.run(self) - self.postrun() - self.__clean_tmp() - self.__file('SUCCESS') - except Exception as e: - self.__file('FAILURE') - traceback.print_exc() - - diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 8a6da1a..2a49eb0 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -189,7 +189,7 @@ def tmpDir(**kwargs): if kwargs.get('clean', globalClean): shutil.rmtree(tmp) else: - print(f"Temporary directy was not deleted, " + print(f"Temporary directory was not deleted, " f"remove it with the following command: \n" f"{Color.bold('rm -rf %s' % tmp)}") From 9a6cb030d15794bf380a5a0f07d5a89bd477b632 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sat, 11 Jan 2025 15:01:55 -0600 Subject: [PATCH 012/161] Introduced FolderManager class --- emtools/jobs/batch_manager.py | 31 +++++++++---------------------- emtools/utils/__init__.py | 6 +++--- emtools/utils/path.py | 23 +++++++++++++++++++++++ 3 files changed, 35 insertions(+), 25 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index adce05c..6974162 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -21,7 +21,7 @@ import json import subprocess -from emtools.utils import Process, Color +from emtools.utils import Process, Color, FolderManager class Args(dict): @@ -39,16 +39,16 @@ def toLine(self): return ' '.join("%s %s" % (k, v) for k, v in self.items()) -class Batch(dict): +class Batch(dict, FolderManager): """ Subclass from dict with some utilities related to Batch logic. """ + def __init__(self, *args, **kwargs): + dict.__init__(self, *args, **kwargs) + FolderManager.__init__(self, self['path']) + @property def id(self): return self['id'] - @property - def path(self): - return self['path'] - @property def index(self): return self['index'] @@ -63,22 +63,9 @@ def info(self): self['info'] = {} return self['info'] - def join(self, *p): - return os.path.join(self['path'], *p) - - def relpath(self, p): - return os.path.relpath(p, self.path) - - def mkdir(self, *p): - d = self.join(*p) - os.mkdir(d) - return d - - def exists(self, *p): - return os.path.exists(self.join(*p)) - - def dump_info(self): - with open(self.join('info.json'), 'w') as batch_info: + def dump_info(self, infoFile=None): + infoFile = infoFile or self.join('info.json') + with open(infoFile, 'w') as batch_info: json.dump(self.info, batch_info, indent=4) def call(self, program, kwargs, logfile): diff --git a/emtools/utils/__init__.py b/emtools/utils/__init__.py index 01f256c..c5869a3 100644 --- a/emtools/utils/__init__.py +++ b/emtools/utils/__init__.py @@ -19,12 +19,12 @@ from .time import Timer from .process import Process -from .path import Path +from .path import Path, FolderManager from .system import System from .server import JsonTCPServer, JsonTCPClient -__all__ = ["Color", "Pretty", "Timer", "Process", "Path", "System", - "JsonTCPServer", "JsonTCPClient"] +__all__ = ["Color", "Pretty", "Timer", "Process", "Path", "FolderManager", + "System", "JsonTCPServer", "JsonTCPClient"] diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 2a49eb0..c780ce7 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -228,3 +228,26 @@ def exists(path): """ return path and os.path.exists(path) + +class FolderManager: + """ Helper class with some path utilities from a given path. """ + def __init__(self, path): + self.__path = path + + def join(self, *p): + return os.path.join(self.__path, *p) + + def relpath(self, p): + return os.path.relpath(p, self.path) + + def mkdir(self, *p): + d = self.join(*p) + os.mkdir(d) + return d + + def exists(self, *p): + return os.path.exists(self.join(*p)) + + @property + def path(self): + return self['path'] From 918d65f4373d9a846351e7c25b28b6018b025b11 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 14 Jan 2025 16:45:29 -0600 Subject: [PATCH 013/161] Check if folder exists --- emtools/utils/path.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index c780ce7..bc5ffde 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -242,7 +242,8 @@ def relpath(self, p): def mkdir(self, *p): d = self.join(*p) - os.mkdir(d) + if not os.path.exists(d): + os.mkdir(d) return d def exists(self, *p): From 68448647a659fc06a6d809685abc0493ea1b1f12 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 14 Jan 2025 16:47:26 -0600 Subject: [PATCH 014/161] Added helper function StarFile.getTableFromFile --- emtools/metadata/starfile.py | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 141fd05..a555dc9 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -30,7 +30,7 @@ from datetime import datetime, timedelta import emtools -from emtools.utils import Pretty +from emtools.utils import Pretty, Color from .table import ColumnList, Table @@ -131,6 +131,14 @@ def getTable(self, tableName, **kwargs): return self._table + @staticmethod + def getTableFromFile(starFileName, tableName, **kwargs): + """ Shortcut to read a table from file. + **kwargs are the same expected by getTable function. + """ + with StarFile(starFileName) as sf: + return sf.getTable(tableName, **kwargs) + def getTableSize(self, tableName): """ Return the number of elements in the given table without parsing @@ -423,11 +431,10 @@ def __init__(self, fileName, tableName, rowKeyFunc, **kwargs): self.fileName = fileName self._tableName = tableName self._rowKeyFunc = rowKeyFunc - self._wait = kwargs.get('wait', 10) + self._wait = kwargs.get('wait', 30) self._timeout = timedelta(seconds=kwargs.get('timeout', 300)) self.lastCheck = None # Last timestamp when input was checked self.lastUpdate = None # Last timestamp when new items were found - self.inputCount = 0 # Count all input elements # Black list some items to not be monitored again # We are not interested in the items but just skip them from @@ -447,7 +454,6 @@ def update(self): for row in sf.iterTable(self._tableName): rowKey = self._rowKeyFunc(row) if rowKey not in self._seenItems: - self.inputCount += 1 self._seenItems.add(rowKey) newRows.append(row) @@ -462,7 +468,7 @@ def timedOut(self): if self.lastCheck is None or self.lastUpdate is None: return False else: - return self.lastCheck - self.lastUpdate > self._timeout + return (self.lastCheck - self.lastUpdate) > self._timeout def newItems(self, sleep=10): """ Yield new items since last update until the stream is closed. """ From bb3fad0bfcdd420e8d4be3e1903b7a5ea2b75b72 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Wed, 15 Jan 2025 16:19:08 -0600 Subject: [PATCH 015/161] added emt-star utility program --- emtools/scripts/emt_star.py | 52 +++++++++++++++++++++++++++++++++++++ setup.py | 3 ++- 2 files changed, 54 insertions(+), 1 deletion(-) create mode 100755 emtools/scripts/emt_star.py diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py new file mode 100755 index 0000000..1936380 --- /dev/null +++ b/emtools/scripts/emt_star.py @@ -0,0 +1,52 @@ +#!/usr/bin/env python +# ************************************************************************** +# * +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) +# * +# * This program is free software; you can redistribute it and/or modify +# * it under the terms of the GNU General Public License as published by +# * the Free Software Foundation; either version 3 of the License, or +# * (at your option) any later version. +# * +# * This program is distributed in the hope that it will be useful, +# * but WITHOUT ANY WARRANTY; without even the implied warranty of +# * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# * GNU General Public License for more details. +# * +# ************************************************************************** + +import os +import time +import argparse +from glob import glob +from datetime import datetime, timedelta +from pprint import pprint +import numpy as np + +from emtools.utils import Process, Color, Path, Timer, Pretty +from emtools.metadata import StarFile + + +def printStarInfo(starFile): + with StarFile(starFile) as sf: + tables = sf.getTableNames() + for t in tables: + cols = sf.getTableInfo(t).getColumnNames() + tSize = sf.getTableSize(t) + print(f">>> {Color.bold('Table')}: {Color.green(t)}" + f"\n - Columns: {Color.cyan(len(cols))} [{' '.join(c for c in cols)}]" + f"\n - Rows: {Color.cyan(tSize)}") + + +def main(): + p = argparse.ArgumentParser(prog='emt-star') + p.add_argument('input', + help="Input STAR file. ") + + args = p.parse_args() + printStarInfo(args.input) + + +if __name__ == '__main__': + main() + diff --git a/setup.py b/setup.py index 486211c..6492ee2 100644 --- a/setup.py +++ b/setup.py @@ -77,7 +77,8 @@ 'emt-files = emtools.scripts.emt_files:main', 'emt-epu = emtools.scripts.emt_epu:main', 'emt-beamshifts = emtools.scripts.emt_beamshifts:main', - 'emt-angdist = emtools.scripts.emt_angdist:main' + 'emt-angdist = emtools.scripts.emt_angdist:main', + 'emt-star = emtools.scripts.emt_star:main' ], }, From 125c3e265705ed9713707d9b7a87fcf3119c777a Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 24 Jan 2025 09:14:29 -0600 Subject: [PATCH 016/161] Allow to dump json files in the folder --- emtools/jobs/batch_manager.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 6974162..ca1d1bf 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -63,10 +63,13 @@ def info(self): self['info'] = {} return self['info'] - def dump_info(self, infoFile=None): - infoFile = infoFile or self.join('info.json') - with open(infoFile, 'w') as batch_info: - json.dump(self.info, batch_info, indent=4) + def dump(self, obj, fn): + filePath = self.join(fn) + with open(filePath, 'w') as f: + json.dump(obj, f, indent=4) + + def dump_info(self): + self.dump(self.info, 'info.json') def call(self, program, kwargs, logfile): args = [program] + Args(kwargs).toList() From d9fba2087b9218e57ce5b68e4e242094bd111efc Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Mon, 27 Jan 2025 11:02:01 -0600 Subject: [PATCH 017/161] Add more methods to Batch --- emtools/jobs/batch_manager.py | 33 +++++++++++++++++++++++---------- 1 file changed, 23 insertions(+), 10 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index ca1d1bf..112610f 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -21,7 +21,7 @@ import json import subprocess -from emtools.utils import Process, Color, FolderManager +from emtools.utils import Process, Color, FolderManager, Pretty class Args(dict): @@ -53,10 +53,6 @@ def id(self): def index(self): return self['index'] - @property - def items(self): - return self['items'] - @property def info(self): if 'info' not in self: @@ -71,11 +67,28 @@ def dump(self, obj, fn): def dump_info(self): self.dump(self.info, 'info.json') - def call(self, program, kwargs, logfile): - args = [program] + Args(kwargs).toList() - with open(logfile, 'w') as f: - cmd = f">>> {Color.green(args[0])} {Color.bold(' '.join(args[1:]))}" - print(cmd) + def dump_all(self): + self.dump(self, 'batch.json') + + def load_all(self): + with open(self.join('batch.json')) as f: + self.update(json.load(f)) + + def call(self, program, kwargs, logfile=None, verbose=False): + if isinstance(kwargs, dict): + args = Args(kwargs).toList() + elif isinstance(kwargs, list): + args = list(kwargs) + else: + raise Exception("Expecting dict or list as arguments") + + args.insert(0, program) + logfile = logfile or self.join('batch.log') + + with open(logfile, 'a') as f: + cmd = f"{Pretty.now()}: >>> {Color.green(args[0])} {Color.bold(' '.join(args[1:]))}" + if verbose: + print(cmd) f.write(f"\n{cmd}\n") f.flush() subprocess.call(args, cwd=self.path, stderr=f, stdout=f) From a34fc3578ba2349ec4f1d4a6240dba6418c233c4 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 31 Jan 2025 15:57:47 -0600 Subject: [PATCH 018/161] Used another threading.Condition to allow maxsize in TaskQueue --- emtools/jobs/batch_manager.py | 10 ++++- emtools/jobs/pipeline.py | 82 ++++++++++++++++++---------------- emtools/tests/test_pipeline.py | 27 +++++++++++ 3 files changed, 78 insertions(+), 41 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 112610f..ba83825 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -74,7 +74,10 @@ def load_all(self): with open(self.join('batch.json')) as f: self.update(json.load(f)) - def call(self, program, kwargs, logfile=None, verbose=False): + def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): + """ + If cwd is True, call the program from the batch directory. + """ if isinstance(kwargs, dict): args = Args(kwargs).toList() elif isinstance(kwargs, list): @@ -91,7 +94,10 @@ def call(self, program, kwargs, logfile=None, verbose=False): print(cmd) f.write(f"\n{cmd}\n") f.flush() - subprocess.call(args, cwd=self.path, stderr=f, stdout=f) + kwargs = {'stderr': f, 'stdout': f} + if cwd: + kwargs['cwd'] = self.path + subprocess.call(args, **kwargs) class BatchManager: diff --git a/emtools/jobs/pipeline.py b/emtools/jobs/pipeline.py index 266b671..d86fe45 100644 --- a/emtools/jobs/pipeline.py +++ b/emtools/jobs/pipeline.py @@ -64,35 +64,36 @@ class TaskQueue: """ Queue of tasks where producers can deposit tasks and consumers can get it. """ - def __init__(self): + def __init__(self, maxsize=None): self._activeGenerators = 0 - self._condition = threading.Condition() self._tasks = [] + self._maxsize = maxsize + self._lock = threading.Lock() # Lock to access tasks + self._condEmpty = threading.Condition(self._lock) + self._condFull = threading.Condition(self._lock) def getTask(self, proc): """ This function should be called from a consumer of this output instance. """ - self._condition.acquire() - - proc._print("Inside condition lock, queue._activeGenerators: ", - self._activeGenerators) - doWait = True - task = None - - while doWait: - doWait = False - if self._tasks: - proc._print("There are tasks") - task = self._tasks.pop(0) - elif self._activeGenerators > 0: - proc._print("No tasks, but not Done, waiting...") - self._condition.wait() - doWait = True - else: - proc._print("No tasks and done, should return None task.") - - self._condition.release() + with self._lock: + proc._print("Inside condition lock, queue._activeGenerators: ", + self._activeGenerators) + doWait = True + task = None + + while doWait: + doWait = False + if self._tasks: + proc._print("There are tasks") + task = self._tasks.pop(0) + self._condFull.notify() + elif self._activeGenerators > 0: + proc._print("No tasks, but not Done, waiting...") + self._condEmpty.wait() + doWait = True + else: + proc._print("No tasks and done, should return None task.") # Return the task, either None if nothing else should be # done, or a task to be processed @@ -102,44 +103,47 @@ def putTask(self, task): """ This function should be used by subclasses of Output that produces items that will be used by consumers. """ - self._condition.acquire() - self._tasks.append(task) - self._condition.notify() - self._condition.release() + with self._lock: + if self._maxsize and len(self._tasks) == self._maxsize: + self._condFull.wait() + self._tasks.append(task) + self._condEmpty.notify() def notifyGeneratorStarts(self): """ When this queue is associated to a generator, this method should be used to notify that the generator has started to run. """ - self._condition.acquire() - self._activeGenerators += 1 - self._condition.release() + with self._lock: + self._activeGenerators += 1 def notifyGeneratorEnds(self): """ This function should be used by generators associated to this queue to notify that they are done and not more tasks will be produced. """ - self._condition.acquire() - self._activeGenerators -= 1 - if self._activeGenerators == 0: - self._condition.notifyAll() - self._condition.release() + with self._lock: + self._activeGenerators -= 1 + if self._activeGenerators == 0: + self._condEmpty.notifyAll() def isDone(self): - self._condition.acquire() - is_done = self._activeGenerators == 0 - self._condition.release() + with self._lock: + is_done = self._activeGenerators == 0 + return is_done class TaskGenerator(threading.Thread): def __init__(self, generator, outputQueue=None, - name='', debug=False): + name='', debug=False, queueMaxSize=None): """ Params: generator: function generating new tasks outputQueue: queue to put new tasks. If None, a new queue will be created + queueMaxSize: maximum number of task that can be in + output queue. After that, a call to putTask block + the generator. If outputQueue is not None, this + parameter is ignored. """ threading.Thread.__init__(self) self.id = None @@ -148,7 +152,7 @@ def __init__(self, generator, outputQueue=None, self._generator = generator if outputQueue is None: - self.outputQueue = TaskQueue() + self.outputQueue = TaskQueue(maxsize=queueMaxSize) else: self.outputQueue = outputQueue diff --git a/emtools/tests/test_pipeline.py b/emtools/tests/test_pipeline.py index 33687c5..d1bfdb1 100644 --- a/emtools/tests/test_pipeline.py +++ b/emtools/tests/test_pipeline.py @@ -18,9 +18,12 @@ import numpy as np import time + +from emtools.utils import Color from emtools.jobs import Pipeline + class TestThreading(unittest.TestCase): def test_threads_processors(self): @@ -65,4 +68,28 @@ def picking(mic): pipeline.run() + print("PROCESSING DONE!!!") + + def test_queueMaxSize(self): + def generate(): + n = 8 + for i in range(1, n+1): + batch = "batch_%03d" % i + print("Generated batch: %s" % Color.green(batch)) + yield batch + time.sleep(1) + + def process(batch): + print("Processing batch: %s" % Color.warn(batch)) + time.sleep(8) + return batch + + pipeline = Pipeline(debug=False) + + g = pipeline.addGenerator(generate, + name='GENERATOR', + queueMaxSize=2) + + pipeline.addProcessor(g.outputQueue, process, name='PROC') + pipeline.run() print("PROCESSING DONE!!!") \ No newline at end of file From 727766488e926637b586da560a91b37962e54977 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sun, 2 Feb 2025 22:23:36 -0600 Subject: [PATCH 019/161] Added utility to read image dimensions --- emtools/image/__init__.py | 16 ++-------- emtools/image/thumbnail.py | 38 ++++++++++++++---------- emtools/tests/test_image.py | 58 +++++++++++++++++++++++++++++++++++++ 3 files changed, 83 insertions(+), 29 deletions(-) create mode 100644 emtools/tests/test_image.py diff --git a/emtools/image/__init__.py b/emtools/image/__init__.py index 619aec6..4563128 100644 --- a/emtools/image/__init__.py +++ b/emtools/image/__init__.py @@ -1,8 +1,6 @@ # ************************************************************************** # * -# * Authors: J.M. De la Rosa Trevin (delarosatrevin@scilifelab.se) [1] -# * -# * [1] SciLifeLab, Stockholm University +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) # * # * This program is free software; you can redistribute it and/or modify # * it under the terms of the GNU General Public License as published by @@ -14,18 +12,10 @@ # * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # * GNU General Public License for more details. # * -# * You should have received a copy of the GNU General Public License -# * along with this program; if not, write to the Free Software -# * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA -# * 02111-1307 USA -# * -# * All comments concerning this program package may be sent to the -# * e-mail address 'delarosatrevin@scilifelab.se' -# * # ************************************************************************** -from .thumbnail import Thumbnail +from .thumbnail import Thumbnail, Image -__all__ = [Thumbnail] +__all__ = ["Thumbnail", "Image"] diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index 0da7d7d..d44c60a 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -1,8 +1,6 @@ # ************************************************************************** # * -# * Authors: J.M. De la Rosa Trevin (delarosatrevin@scilifelab.se) [1] -# * -# * [1] SciLifeLab, Stockholm University +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) # * # * This program is free software; you can redistribute it and/or modify # * it under the terms of the GNU General Public License as published by @@ -14,22 +12,15 @@ # * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the # * GNU General Public License for more details. # * -# * You should have received a copy of the GNU General Public License -# * along with this program; if not, write to the Free Software -# * Foundation, Inc., 59 Temple Place, Suite 330, Boston, MA -# * 02111-1307 USA -# * -# * All comments concerning this program package may be sent to the -# * e-mail address 'delarosatrevin@scilifelab.se' -# * # ************************************************************************** import io import numpy as np import base64 import mrcfile +import tifffile -from PIL import Image, ImageOps, ImageFilter +import PIL class Thumbnail: @@ -79,10 +70,10 @@ def from_pil(self, pil_img): self.scale = scale if self.contrast_factor is not None: - pil_img = ImageOps.autocontrast(pil_img, cutoff=self.contrast_factor) + pil_img = PIL.ImageOps.autocontrast(pil_img, cutoff=self.contrast_factor) if self.gaussian_radius is not None: - pil_img = pil_img.filter(ImageFilter.GaussianBlur(radius=self.gaussian_radius)) + pil_img = pil_img.filter(PIL.ImageFilter.GaussianBlur(radius=self.gaussian_radius)) return self.__format(pil_img) @@ -90,7 +81,7 @@ def from_path(self, path): """ Read the image path as a PIL image and encode it as base64. """ try: - img = Image.open(path) + img = PIL.Image.open(path) encoded = self.from_pil(img) img.close() except: @@ -121,7 +112,7 @@ def from_array(self, imageArray): im255 = ((array - iMin) / (iMax - iMin) * 255).astype(np.uint8) - pil_img = Image.fromarray(im255) + pil_img = PIL.Image.fromarray(im255) return self.from_pil(pil_img) @@ -168,3 +159,18 @@ def Psd(**kwargs): defaults.update(kwargs) return Thumbnail(**defaults) + +class Image: + @staticmethod + def get_dimensions(imagePath): + imageLower = imagePath.lower() + if imageLower.endswith('.mrc') or imageLower.endswith('.mrcs'): + with mrcfile.open(imagePath) as mrc: + return mrc.data.shape[::-1] # in reverse order + elif (imageLower.endswith('.tif') or + imageLower.endswith('.tiff') or + imageLower.endswith('.eer')): + with tifffile.TiffFile(imagePath) as tif: + n = len(tif.pages) + y, x = tif.pages[0].shape + return (x, y, n) if n > 1 else (x, y) diff --git a/emtools/tests/test_image.py b/emtools/tests/test_image.py new file mode 100644 index 0000000..d5a9681 --- /dev/null +++ b/emtools/tests/test_image.py @@ -0,0 +1,58 @@ +# ************************************************************************** +# * +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) +# * +# * This program is free software; you can redistribute it and/or modify +# * it under the terms of the GNU General Public License as published by +# * the Free Software Foundation; either version 3 of the License, or +# * (at your option) any later version. +# * +# * This program is distributed in the hope that it will be useful, +# * but WITHOUT ANY WARRANTY; without even the implied warranty of +# * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# * GNU General Public License for more details. +# * +# ************************************************************************** + +import os +import unittest +import tempfile +import random +import time +import threading +import tempfile +from pprint import pprint +from datetime import datetime + +from emtools.utils import Timer, Color, Pretty +from emtools.metadata import StarFile, SqliteFile, EPU, StarMonitor +from emtools.jobs import BatchManager +from emtools.tests import testpath +from emtools.image import Image + +from .star_pipeline_tester import StarPipelineTester + + +class TestImage(unittest.TestCase): + """ + Tests for Image class. + """ + + def test_dimensions(self): + """ + Read a star file with several blocks + """ + names = ['May08_03.05.02.bin.mrc', + 'gain.mrc', + '20170629_00021_frameImage.tiff'] + dims = [(1240, 1200, 50), + (3710, 3838), + (3710, 3838, 24)] + files = [testpath('movies', n) for n in names] + + if any(f is None for f in files): + return + + for f, d in zip(files, dims): + self.assertEqual(Image.get_dimensions(f), d) + From bd80dbd69aa5e689df38c90d5074bd224b60f22b Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 3 Feb 2025 09:06:23 -0600 Subject: [PATCH 020/161] Added 'group_by' option to emt-star --- emtools/scripts/emt_star.py | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py index 1936380..69170c6 100755 --- a/emtools/scripts/emt_star.py +++ b/emtools/scripts/emt_star.py @@ -22,6 +22,7 @@ from datetime import datetime, timedelta from pprint import pprint import numpy as np +from collections import defaultdict from emtools.utils import Process, Color, Path, Timer, Pretty from emtools.metadata import StarFile @@ -37,14 +38,32 @@ def printStarInfo(starFile): f"\n - Columns: {Color.cyan(len(cols))} [{' '.join(c for c in cols)}]" f"\n - Rows: {Color.cyan(tSize)}") +def groupBy(starFile, table, column): + group = defaultdict(lambda: 0) + + with StarFile(starFile) as sf: + for row in sf.iterTable(table): + group[row.get(column)] += 1 + + for k, v in group.items(): + print(k, v) + def main(): p = argparse.ArgumentParser(prog='emt-star') p.add_argument('input', help="Input STAR file. ") + p.add_argument('--group_by', '-g', nargs=2, + metavar=('TABLE', 'COLUMN'), + help="Count rows grouped by a given label") args = p.parse_args() - printStarInfo(args.input) + + if args.group_by: + table, column = args.group_by + groupBy(args.input, table, column) + else: + printStarInfo(args.input) if __name__ == '__main__': From 19a12b6ef44f75fc0371070b024f45dda0536def Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 5 Feb 2025 06:54:42 -0600 Subject: [PATCH 021/161] Added prototype to split star files --- emtools/scripts/emt_star.py | 50 +++++++++++++++++++++++++++++++++++-- 1 file changed, 48 insertions(+), 2 deletions(-) diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py index 69170c6..19420c0 100755 --- a/emtools/scripts/emt_star.py +++ b/emtools/scripts/emt_star.py @@ -38,6 +38,7 @@ def printStarInfo(starFile): f"\n - Columns: {Color.cyan(len(cols))} [{' '.join(c for c in cols)}]" f"\n - Rows: {Color.cyan(tSize)}") + def groupBy(starFile, table, column): group = defaultdict(lambda: 0) @@ -49,6 +50,45 @@ def groupBy(starFile, table, column): print(k, v) +def splitBy(starFile, column, minSize): + with StarFile(starFile) as sf: + tOptics = sf.getTable('optics') + tParticles = sf.getTableInfo('particles') + rows = [] + count = 0 + map = {} + + def _writeStar(minSize=0): + nonlocal count + nonlocal rows + + if len(rows) <= minSize: + return + + count += 1 + outStarFile = Path.replaceExt(starFile, f'_{count:03}.star') + with StarFile(outStarFile, 'w') as sfOut: + sfOut.writeTimeStamp() + sfOut.writeTable('optics', tOptics) + sfOut.writeHeader('particles', tParticles) + for row in rows: + sfOut.writeRow(row) + rows = [] + + lastValue = None + lastIndex = 0 + + for row in sf.iterTable('particles'): + value = getattr(row, column) + if lastValue is not None and lastValue != value: + _writeStar(int(minSize)) + rows.append(row) + lastValue = value + + if rows: + _writeStar(0) # Write all remaining + + def main(): p = argparse.ArgumentParser(prog='emt-star') p.add_argument('input', @@ -56,16 +96,22 @@ def main(): p.add_argument('--group_by', '-g', nargs=2, metavar=('TABLE', 'COLUMN'), help="Count rows grouped by a given label") + p.add_argument('--split_particles', '-s', nargs='+', metavar=('COLUMN', 'minsize'), + help="Split input particles by some column") args = p.parse_args() + inputStar = args.input if args.group_by: table, column = args.group_by - groupBy(args.input, table, column) + groupBy(inputStar, table, column) + elif split := args.split_particles: + column = split[0] + minSize = split[1] if len(split) > 1 else 0 + splitBy(inputStar, column, minSize) else: printStarInfo(args.input) if __name__ == '__main__': main() - From c9e710b99d9d9274710869a8859660804ece27d7 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 6 Feb 2025 13:49:14 -0600 Subject: [PATCH 022/161] Allow to pass print function --- emtools/utils/process.py | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/emtools/utils/process.py b/emtools/utils/process.py index 134c3b6..7db2b56 100644 --- a/emtools/utils/process.py +++ b/emtools/utils/process.py @@ -23,6 +23,10 @@ import logging +def _print(*msgs): + print(*msgs) + + class Process: def __init__(self, *args, **kwargs): """ Create a process using subprocess.""" @@ -57,7 +61,7 @@ def print(self, args=True, stdout=False): print(self.stdout) @staticmethod - def system(cmd, only_print=False, color=None, do_print=True): + def system(cmd, only_print=False, color=None, print=_print): """ Execute and print a command. Args: @@ -66,7 +70,7 @@ def system(cmd, only_print=False, color=None, do_print=True): not executed color: Optional color for the command """ - if do_print: + if print: printCmd = cmd if color is None else color(cmd) print(printCmd) if not only_print: From 6d23991e92833a76d15a72af28c3f6020162cbce Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 6 Feb 2025 13:50:06 -0600 Subject: [PATCH 023/161] Added more logging to Batch --- emtools/jobs/batch_manager.py | 37 +++++++++++++++++++++-------------- 1 file changed, 22 insertions(+), 15 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index ba83825..8893204 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -89,9 +89,7 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): logfile = logfile or self.join('batch.log') with open(logfile, 'a') as f: - cmd = f"{Pretty.now()}: >>> {Color.green(args[0])} {Color.bold(' '.join(args[1:]))}" - if verbose: - print(cmd) + cmd = self.log(f"{Color.green(args[0])} {Color.bold(' '.join(args[1:]))}") f.write(f"\n{cmd}\n") f.flush() kwargs = {'stderr': f, 'stdout': f} @@ -99,6 +97,17 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): kwargs['cwd'] = self.path subprocess.call(args, **kwargs) + def log(self, msg): + logMsg = f"{Pretty.now()}: {self.id}: {msg}" + print(logMsg) + return logMsg + + def create(self): + """ Create batch folder. """ + self.log(f"Creating folder: {self.path}") + Process.system(f"rm -rf '{self.path}'", print=False) + Process.system(f"mkdir '{self.path}'", print=False) + class BatchManager: """ Class used to generate and handle the creation of batches @@ -127,18 +136,22 @@ def __init__(self, batchSize, inputItemsIterator, workingPath, def _createBatchId(self): # We will use batchCount, before the batch is created nowPrefix = datetime.now().strftime('%y%m%d-%H%M%S') - countStr = '%02d' % (self._batchCount + 1) + countStr = '%02d' % self._batchCount uuidSuffix = str(uuid4()).split('-')[0] return f"{nowPrefix}_{countStr}_{uuidSuffix}" def _createBatch(self, items, inputFolder=None): + self._batchCount += 1 batch_id = self._createBatchId() batch_path = os.path.join(self._workingPath, batch_id) - print(f"Creating batch: {batch_path}") - Process.system(f"rm -rf '{batch_path}'") - Process.system(f"mkdir '{batch_path}'") + batch = Batch(id=batch_id, + index=self._batchCount, + path=batch_path, + items=items) + batch.create() + if inputFolder is not None: - Process.system(f"mkdir '{batch_path}/{inputFolder}'") + batch.mkdir(inputFolder) for item in items: fn = self._itemFileNameFunc(item) @@ -148,13 +161,7 @@ def _createBatch(self, items, inputFolder=None): os.symlink(os.path.abspath(fn), os.path.join(batch_path, baseName)) - self._batchCount += 1 - return Batch({ - 'items': items, - 'id': batch_id, - 'path': batch_path, - 'index': self._batchCount - }) + return batch def generate(self): """ Generate batches based on the input items. """ From 0f88eea2c5eaf653f7af5f4ab3f02025422e3e46 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 6 Feb 2025 13:50:37 -0600 Subject: [PATCH 024/161] Allow missing coordinate files --- emtools/scripts/emt-scipion-otf.py | 13 +++++++------ 1 file changed, 7 insertions(+), 6 deletions(-) diff --git a/emtools/scripts/emt-scipion-otf.py b/emtools/scripts/emt-scipion-otf.py index 34d3d4a..69d2185 100755 --- a/emtools/scripts/emt-scipion-otf.py +++ b/emtools/scripts/emt-scipion-otf.py @@ -827,12 +827,13 @@ def cryosparc_prepare(): coordsFn = f'Coordinates/{micName}_DW_coordinates.star' ctfValues = [getattr(row, k) for k in ctfCols] print(len(ctfValues)) - with StarFile(coordsFn) as sfCoords: - for rowCoord in sfCoords.iterTable(''): - sfOut.writeRow(t.Row(f'{micName}.mrc', - rowCoord.rlnCoordinateX, - rowCoord.rlnCoordinateY, - *ctfValues)) + if os.path.exists(coordsFn): + with StarFile(coordsFn) as sfCoords: + for rowCoord in sfCoords.iterTable(''): + sfOut.writeRow(t.Row(f'{micName}.mrc', + rowCoord.rlnCoordinateX, + rowCoord.rlnCoordinateY, + *ctfValues)) def cryosparc_import(projId, dataRoot): From 39f1f6f09fe79a9fd087fe0c8d529c9fd99f6345 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 6 Feb 2025 16:54:01 -0600 Subject: [PATCH 025/161] Added queueMaxSize for TaskProcessor --- emtools/jobs/pipeline.py | 13 +++++++++---- 1 file changed, 9 insertions(+), 4 deletions(-) diff --git a/emtools/jobs/pipeline.py b/emtools/jobs/pipeline.py index d86fe45..d404d4f 100644 --- a/emtools/jobs/pipeline.py +++ b/emtools/jobs/pipeline.py @@ -99,7 +99,7 @@ def getTask(self, proc): # done, or a task to be processed return task - def putTask(self, task): + def putTask(self, task, proc): """ This function should be used by subclasses of Output that produces items that will be used by consumers. """ @@ -160,8 +160,11 @@ def run(self): self.outputQueue.notifyGeneratorStarts() self.id = threading.get_ident() + self._print(">>>>>> Iterating generator tasks") for task in self._generator(): - self.outputQueue.putTask(task) + self._print(">>>>>>>> Got task: ", task['id'], "...putting it queue.") + self.outputQueue.putTask(task, self) + self._print(">>>>>>>> SENT task: ", task['id']) self.outputQueue.notifyGeneratorEnds() @@ -173,8 +176,10 @@ def _print(self, *args): class TaskProcessor(TaskGenerator): def __init__(self, inputQueue, processor, outputQueue=None, - name='', debug=False): - TaskGenerator.__init__(self, self._process, outputQueue, name, debug) + name='', debug=False, queueMaxSize=None): + TaskGenerator.__init__(self, self._process, + outputQueue=outputQueue, name=name, + debug=debug, queueMaxSize=queueMaxSize) self._processor = processor self._inputQueue = inputQueue From 8ec00bb94fadf5b3b2fca05d960449a771f7cbc0 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 7 Feb 2025 16:03:18 -0600 Subject: [PATCH 026/161] Moved create to FolderManager --- emtools/utils/path.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index bc5ffde..bfa41d8 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -251,4 +251,9 @@ def exists(self, *p): @property def path(self): - return self['path'] + return self.__path + + def create(self, **kwargs): + """ Create batch folder. """ + Process.system(f"rm -rf '{self.path}'", **kwargs) + Process.system(f"mkdir -p '{self.path}'", **kwargs) From 3193b16bd6156ea29135f8e38e11b8fd8f6c08b2 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 7 Feb 2025 16:03:48 -0600 Subject: [PATCH 027/161] Added methods to Mdoc --- emtools/metadata/misc.py | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 85286bd..0eccd7d 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -15,7 +15,7 @@ # ************************************************************************** import os - +import pathlib from datetime import datetime, timedelta from emtools.utils import Path, Pretty, Process @@ -274,9 +274,20 @@ def parse(mdocFn): return mdoc + @staticmethod + def getSubFrameBase(section): + """ Helper method to extract the subframe base filename. """ + subFramePath = section.get('SubFramePath', '') + return pathlib.PureWindowsPath(subFramePath).parts[-1] + @property def zvalues(self): - return [(k, v) for k, v in self.items() if k.startswith('ZValue')] + return [(k, v) for k, v in self.zsections()] + + def zsections(self): + for k, v in self.items(): + if k.startswith('ZValue'): + yield k, v def write(self, path): with open(path, 'w') as f: From 69ea1ca5775700a7dd5268d84ae440f3a0684d3f Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 7 Feb 2025 16:04:18 -0600 Subject: [PATCH 028/161] Moved create to FolderManager --- emtools/jobs/batch_manager.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 8893204..e2785f2 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -105,8 +105,7 @@ def log(self, msg): def create(self): """ Create batch folder. """ self.log(f"Creating folder: {self.path}") - Process.system(f"rm -rf '{self.path}'", print=False) - Process.system(f"mkdir '{self.path}'", print=False) + FolderManager.create(self, print=False) class BatchManager: From a333ffc54854d8cbbd9abab4220eb2cafcaeb20e Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 11 Feb 2025 13:44:05 -0600 Subject: [PATCH 029/161] Working on Workflow tools --- emtools/jobs/__init__.py | 3 +- emtools/jobs/workflow.py | 106 +++++++++++++++++++++++++++++++++ emtools/tests/test_workflow.py | 42 +++++++++++++ 3 files changed, 150 insertions(+), 1 deletion(-) create mode 100644 emtools/jobs/workflow.py create mode 100644 emtools/tests/test_workflow.py diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index c760f41..fa04441 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -16,5 +16,6 @@ from .pipeline import Pipeline from .batch_manager import Args, Batch, BatchManager +from .workflow import Workflow -__all__ = ["Pipeline", "BatchManager", "Batch", "Args"] +__all__ = ["Pipeline", "BatchManager", "Batch", "Args", "Workflow"] diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py new file mode 100644 index 0000000..590d732 --- /dev/null +++ b/emtools/jobs/workflow.py @@ -0,0 +1,106 @@ +# ************************************************************************** +# * +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) +# * +# * This program is free software; you can redistribute it and/or modify +# * it under the terms of the GNU General Public License as published by +# * the Free Software Foundation; either version 3 of the License, or +# * (at your option) any later version. +# * +# * This program is distributed in the hope that it will be useful, +# * but WITHOUT ANY WARRANTY; without even the implied warranty of +# * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# * GNU General Public License for more details. +# * +# ************************************************************************** + +import os +import sys +from collections import OrderedDict +import threading + +from emtools.utils import Process + + +class Workflow: + """ + Simple implementation of a Workflow class for management of Jobs and + produced Data. The workflow is represented as a directed acyclic graph. + """ + + def __init__(self, jobCounter=0): + self._jobs = {} + self.data = {} + self._jobCounter = jobCounter + + def jobs(self): + """ Iterate over the jobs sorted by index. """ + return sorted(self._jobs.values(), key=lambda j: j.index) + + def root(self): + """ Iterator over nodes that does not have any input. """ + for j in self.jobs(): + if not j.inputs: + yield j + + def getJob(self, jobId): + return self._jobs[jobId] + + def registerJob(self, jobId, inputs=[]): + job = Workflow.Job(self, jobId, self._jobCounter + 1, inputs=inputs) + self._jobCounter += 1 + self._jobs[jobId] = job + return job + + def print(self): + """ Print the workflow to the terminal. """ + dot = 'digraph G {\n compound=true;\n' + links = '' + + for j in self.jobs(): + dot += (f' subgraph cluster_{j.index} {{\n' + f' style=filled; color=lightgrey; \n' + f' node [style=filled,color=white];\n' + f' i{j.index} [color=lightgrey,fontcolor=lightgrey];\n' + f' label="{j.id}";\n') + for o in j.outputs: + dot += f' {o.id}\n' + for c in o.childs: + links += f'{o.id} -> i{c.index} [lhead=cluster_{c.index}];\n' + dot += ' }\n' + + dot += f'\n{links}\n}}\n' + print(dot) + + class Job(dict): + def __init__(self, wf, id, index, inputs=[], **kwargs): + dict.__init__(self, **kwargs) + self.wf = wf + self.id = id + self.index = index + self.inputs = [] + self.outputs = [] + self.addInputs(inputs) + + def registerOutput(self, dataId, **kwargs): + data = Workflow.Data(self, dataId, **kwargs) + self.wf.data[dataId] = data + self.outputs.append(data) + return data + + def addInputs(self, inputs): + if any(i in self.inputs for i in inputs): + raise Exception(f'Input {i} was already added.') + + # TODO validate cyclic dependencies + for i in inputs: + self.inputs.append(i) + i.childs.append(self) + + class Data(dict): + def __init__(self, parent, dataId, **kwargs): + dict.__init__(self, **kwargs) + self.id = dataId + self.parent = parent + self.childs = [] + diff --git a/emtools/tests/test_workflow.py b/emtools/tests/test_workflow.py new file mode 100644 index 0000000..5da9f9d --- /dev/null +++ b/emtools/tests/test_workflow.py @@ -0,0 +1,42 @@ +# ************************************************************************** +# * +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) +# * +# * This program is free software; you can redistribute it and/or modify +# * it under the terms of the GNU General Public License as published by +# * the Free Software Foundation; either version 3 of the License, or +# * (at your option) any later version. +# * +# * This program is distributed in the hope that it will be useful, +# * but WITHOUT ANY WARRANTY; without even the implied warranty of +# * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# * GNU General Public License for more details. +# * +# ************************************************************************** + +import unittest +import numpy as np +import time + + +from emtools.utils import Color +from emtools.jobs import Pipeline, Workflow + + +class TestWorkflow(unittest.TestCase): + def test_basic(self): + wf = Workflow() + + j1 = wf.registerJob('job01') + d1 = j1.registerOutput('d1') + j2 = wf.registerJob('job02', inputs=[d1]) + j3 = wf.registerJob('job03', inputs=[d1]) + d3a = j3.registerOutput('d3a') + d3b = j3.registerOutput('d3b') + j5 = wf.registerJob('job05') + d5 = j5.registerOutput('d5') + j6 = wf.registerJob('job06', inputs=[d3b, d5]) + + wf.print() + + From 122ddc42f059e4042249ae583f6f5d35049eac6f Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 12 Feb 2025 09:20:09 -0600 Subject: [PATCH 030/161] More on Workflow --- emtools/jobs/workflow.py | 56 ++++++++++++++++++++++++++++------ emtools/tests/test_workflow.py | 8 ++++- 2 files changed, 54 insertions(+), 10 deletions(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index 590d732..f200b28 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -20,6 +20,7 @@ import threading from emtools.utils import Process +from emtools.metadata import StarFile class Workflow: @@ -46,13 +47,17 @@ def root(self): def getJob(self, jobId): return self._jobs[jobId] - def registerJob(self, jobId, inputs=[]): - job = Workflow.Job(self, jobId, self._jobCounter + 1, inputs=inputs) + def getData(self, dataId): + return self.data[dataId] + + def registerJob(self, jobId, inputs=[], **kwargs): + job = Workflow.Job(self, jobId, self._jobCounter + 1, + inputs=inputs, **kwargs) self._jobCounter += 1 self._jobs[jobId] = job return job - def print(self): + def dot(self): """ Print the workflow to the terminal. """ dot = 'digraph G {\n compound=true;\n' links = '' @@ -64,13 +69,14 @@ def print(self): f' i{j.index} [color=lightgrey,fontcolor=lightgrey];\n' f' label="{j.id}";\n') for o in j.outputs: - dot += f' {o.id}\n' + oid = o.id.replace('/', '_').replace('.', '_') + dot += f' {oid}\n' for c in o.childs: - links += f'{o.id} -> i{c.index} [lhead=cluster_{c.index}];\n' + links += f'{oid} -> i{c.index} [lhead=cluster_{c.index}];\n' dot += ' }\n' dot += f'\n{links}\n}}\n' - print(dot) + return dot class Job(dict): def __init__(self, wf, id, index, inputs=[], **kwargs): @@ -88,11 +94,17 @@ def registerOutput(self, dataId, **kwargs): self.outputs.append(data) return data + def _validateInputs(self, inputs): + for i in inputs: + if not isinstance(i, Workflow.Data): + raise Exception(f"Input {i} is not of type Workflow.Data") + if i in self.inputs: + Exception(f'Input {i} was already added.') + # TODO validate cyclic dependencies + def addInputs(self, inputs): - if any(i in self.inputs for i in inputs): - raise Exception(f'Input {i} was already added.') + self._validateInputs(inputs) - # TODO validate cyclic dependencies for i in inputs: self.inputs.append(i) i.childs.append(self) @@ -104,3 +116,29 @@ def __init__(self, parent, dataId, **kwargs): self.parent = parent self.childs = [] + @staticmethod + def fromRelionPipeline(pipelineStar): + """ Load pipeline Graph from the default_pipeline.star file. """ + wf = Workflow() + + with StarFile(pipelineStar) as sf: + for row in sf.iterTable('pipeline_processes'): + wf.registerJob(row.rlnPipeLineProcessName, + alias=row.rlnPipeLineProcessAlias, + status=row.rlnPipeLineProcessStatusLabel, + type=row.rlnPipeLineProcessTypeLabel) + + nodes = {row.rlnPipeLineNodeName: {'type': row.rlnPipeLineNodeTypeLabel} + for row in sf.iterTable('pipeline_nodes')} + + for row in sf.iterTable('pipeline_output_edges'): + job = wf.getJob(row.rlnPipeLineEdgeProcess) + nodeName = row.rlnPipeLineEdgeToNode + job.registerOutput(nodeName) #, type=nodes[nodeName]) + + for row in sf.iterTable('pipeline_input_edges'): + job = wf.getJob(row.rlnPipeLineEdgeProcess) + job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) + + return wf + diff --git a/emtools/tests/test_workflow.py b/emtools/tests/test_workflow.py index 5da9f9d..2c1282c 100644 --- a/emtools/tests/test_workflow.py +++ b/emtools/tests/test_workflow.py @@ -37,6 +37,12 @@ def test_basic(self): d5 = j5.registerOutput('d5') j6 = wf.registerJob('job06', inputs=[d3b, d5]) - wf.print() + dot = wf.dot() + def test_relion_pipeline(self): + pipelineStar = '/Users/jdela80/work/data/emwrap/testing/Relion5-Tutorial-emwrap/default_pipeline.star' + wf = Workflow.fromRelionPipeline(pipelineStar) + + print("\n") + print(wf.dot()) From 1e113fbec9979500b2d5413dfe85c91546d1e36a Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Wed, 12 Feb 2025 17:57:18 -0600 Subject: [PATCH 031/161] Moved log function and more on Workflow --- emtools/jobs/batch_manager.py | 11 +-------- emtools/jobs/workflow.py | 42 +++++++++-------------------------- emtools/utils/path.py | 12 +++++++--- 3 files changed, 20 insertions(+), 45 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index e2785f2..da8c4d9 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -44,6 +44,7 @@ class Batch(dict, FolderManager): def __init__(self, *args, **kwargs): dict.__init__(self, *args, **kwargs) FolderManager.__init__(self, self['path']) + self._logId = f" {self.id}:" @property def id(self): @@ -97,16 +98,6 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): kwargs['cwd'] = self.path subprocess.call(args, **kwargs) - def log(self, msg): - logMsg = f"{Pretty.now()}: {self.id}: {msg}" - print(logMsg) - return logMsg - - def create(self): - """ Create batch folder. """ - self.log(f"Creating folder: {self.path}") - FolderManager.create(self, print=False) - class BatchManager: """ Class used to generate and handle the creation of batches diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index f200b28..83a37fa 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -29,10 +29,10 @@ class Workflow: produced Data. The workflow is represented as a directed acyclic graph. """ - def __init__(self, jobCounter=0): + def __init__(self, **kwargs): self._jobs = {} self.data = {} - self._jobCounter = jobCounter + self.jobNextIndex = 1 def jobs(self): """ Iterate over the jobs sorted by index. """ @@ -50,10 +50,10 @@ def getJob(self, jobId): def getData(self, dataId): return self.data[dataId] - def registerJob(self, jobId, inputs=[], **kwargs): - job = Workflow.Job(self, jobId, self._jobCounter + 1, + def registerJob(self, jobId, inputs=None, **kwargs): + job = Workflow.Job(self, jobId, self.jobNextIndex, inputs=inputs, **kwargs) - self._jobCounter += 1 + self.jobNextIndex += 1 self._jobs[jobId] = job return job @@ -79,10 +79,10 @@ def dot(self): return dot class Job(dict): - def __init__(self, wf, id, index, inputs=[], **kwargs): + def __init__(self, wf, jobId, index, inputs=None, **kwargs): dict.__init__(self, **kwargs) self.wf = wf - self.id = id + self.id = jobId self.index = index self.inputs = [] self.outputs = [] @@ -103,6 +103,9 @@ def _validateInputs(self, inputs): # TODO validate cyclic dependencies def addInputs(self, inputs): + if not inputs: + return + self._validateInputs(inputs) for i in inputs: @@ -116,29 +119,4 @@ def __init__(self, parent, dataId, **kwargs): self.parent = parent self.childs = [] - @staticmethod - def fromRelionPipeline(pipelineStar): - """ Load pipeline Graph from the default_pipeline.star file. """ - wf = Workflow() - - with StarFile(pipelineStar) as sf: - for row in sf.iterTable('pipeline_processes'): - wf.registerJob(row.rlnPipeLineProcessName, - alias=row.rlnPipeLineProcessAlias, - status=row.rlnPipeLineProcessStatusLabel, - type=row.rlnPipeLineProcessTypeLabel) - - nodes = {row.rlnPipeLineNodeName: {'type': row.rlnPipeLineNodeTypeLabel} - for row in sf.iterTable('pipeline_nodes')} - - for row in sf.iterTable('pipeline_output_edges'): - job = wf.getJob(row.rlnPipeLineEdgeProcess) - nodeName = row.rlnPipeLineEdgeToNode - job.registerOutput(nodeName) #, type=nodes[nodeName]) - - for row in sf.iterTable('pipeline_input_edges'): - job = wf.getJob(row.rlnPipeLineEdgeProcess) - job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) - - return wf diff --git a/emtools/utils/path.py b/emtools/utils/path.py index bfa41d8..c8a96c2 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -233,6 +233,7 @@ class FolderManager: """ Helper class with some path utilities from a given path. """ def __init__(self, path): self.__path = path + self._logId = "" def join(self, *p): return os.path.join(self.__path, *p) @@ -240,10 +241,9 @@ def join(self, *p): def relpath(self, p): return os.path.relpath(p, self.path) - def mkdir(self, *p): + def mkdir(self, *p, **kwargs): d = self.join(*p) - if not os.path.exists(d): - os.mkdir(d) + Process.system(f"mkdir -p '{d}'", **kwargs) return d def exists(self, *p): @@ -255,5 +255,11 @@ def path(self): def create(self, **kwargs): """ Create batch folder. """ + self.log(f"Creating folder: {self.path}") Process.system(f"rm -rf '{self.path}'", **kwargs) Process.system(f"mkdir -p '{self.path}'", **kwargs) + + def log(self, msg): + logMsg = f"{Pretty.now()}:{self._logId} {msg}" + print(logMsg) + return logMsg From b2b868a9a64d3d590587c746f406b6ac8dfa6b62 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Thu, 13 Feb 2025 14:01:05 -0600 Subject: [PATCH 032/161] Minor fixes in Workflow jobIndex --- emtools/jobs/workflow.py | 5 +++-- emtools/utils/path.py | 4 +--- 2 files changed, 4 insertions(+), 5 deletions(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index 83a37fa..188b439 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -51,9 +51,10 @@ def getData(self, dataId): return self.data[dataId] def registerJob(self, jobId, inputs=None, **kwargs): - job = Workflow.Job(self, jobId, self.jobNextIndex, + jobIndex = kwargs.get('jobindex', self.jobNextIndex) + job = Workflow.Job(self, jobId, jobIndex, inputs=inputs, **kwargs) - self.jobNextIndex += 1 + self.jobNextIndex = jobIndex + 1 self._jobs[jobId] = job return job diff --git a/emtools/utils/path.py b/emtools/utils/path.py index c8a96c2..5df140f 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -83,7 +83,7 @@ def splitall(path): @staticmethod def addslash(path): - """ Add an slash (/) to the end of the path if not present. """ + """ Add a slash (/) to the end of the path if not present. """ return path if path.endswith('/') else path + '/' @staticmethod @@ -136,7 +136,6 @@ def copyFile(file1, file2, sleep=0): f2.write(rbytes) if sleep: time.sleep(sleep) - #Process.system(f'cp {file1} {file2}') @staticmethod def copyDir(dir1, dir2, copyFileFunc=None, pl=None, **kwargs): @@ -166,7 +165,6 @@ def _mkdir(d): for f in files: _copy(os.path.join(root, f), os.path.join(root2, f), **kwargs) - @staticmethod @contextmanager def tmpDir(**kwargs): From c938f7d40635b6141fc86b8ac9b63d2dd6ef6d7b Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 14 Feb 2025 09:07:24 -0600 Subject: [PATCH 033/161] Moved Acqusition and RelionStar to metadata --- emtools/metadata/__init__.py | 9 +- emtools/metadata/misc.py | 35 +++++++ emtools/metadata/starfile.py | 197 ++++++++++++++++++++++++++++++++++- 3 files changed, 236 insertions(+), 5 deletions(-) diff --git a/emtools/metadata/__init__.py b/emtools/metadata/__init__.py index ece86c5..bda4fa2 100644 --- a/emtools/metadata/__init__.py +++ b/emtools/metadata/__init__.py @@ -15,12 +15,13 @@ # ************************************************************************** from .table import Column, ColumnList, Table -from .starfile import StarFile, StarMonitor +from .starfile import StarFile, StarMonitor, RelionStar from .epu import EPU -from .misc import Bins, TsBins, DataFiles, MovieFiles, Mdoc, TextFile +from .misc import Bins, TsBins, DataFiles, MovieFiles, Mdoc, TextFile, Acquisition from .sqlite import SqliteFile -__all__ = ["Column", "ColumnList", "Table", "StarFile", "StarMonitor", "EPU", +__all__ = ["Column", "ColumnList", "Table", + "StarFile", "StarMonitor", "RelionStar", "EPU", "Bins", "TsBins", "SqliteFile", "DataFiles", "MovieFiles", - "Mdoc", "TextFile"] + "Mdoc", "TextFile", "Acquisition"] diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 0eccd7d..8d48140 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -307,3 +307,38 @@ def stripLines(fn, **kwargs): if line and not line.startswith('#'): yield line + +class Acquisition(dict): + """ Subclass from dict with some utilities related to Acquisition. """ + + @property + def pixel_size(self): + return self['pixel_size'] + + @pixel_size.setter + def pixel_size(self, value): + self['pixel_size'] = value + + @property + def voltage(self): + return self['voltage'] + + @voltage.setter + def voltage(self, value): + self['voltage'] = value + + @property + def cs(self): + return self['cs'] + + @cs.setter + def cs(self, value): + self['cs'] = value + + @property + def amplitude_contrast(self): + return self.get('amplitude_contrast', 0.1) + + @amplitude_contrast.setter + def amplitude_contrast(self, value): + self['amplitude_contrast'] = value \ No newline at end of file diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index a555dc9..f0f7a43 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -26,12 +26,14 @@ import time import re from contextlib import AbstractContextManager -from collections import OrderedDict from datetime import datetime, timedelta import emtools from emtools.utils import Pretty, Color +from emtools.jobs import Workflow + from .table import ColumnList, Table +from .misc import Acquisition class StarFile(AbstractContextManager): @@ -491,3 +493,196 @@ def _escapeStrValue(v): """ Escape string values by adding quotes if the string is empty or contains spaces. """ return '"%s"' % v if isinstance(v, str) and (not v or ' ' in v) else v + + +class RelionStar: + + JOB_INDEX = re.compile('job(\d{3})') + + @staticmethod + def optics_table(acq, opticsGroup=1, opticsGroupName="opticsGroup1", + mtf=None, originalPixelSize=None): + origPs = originalPixelSize or acq['pixel_size'] + + values = { + 'rlnOpticsGroupName': opticsGroupName, + 'rlnOpticsGroup': opticsGroup, + 'rlnMicrographOriginalPixelSize': origPs, + 'rlnVoltage': acq['voltage'], + 'rlnSphericalAberration': acq['cs'], + 'rlnAmplitudeContrast': acq.get('amplitude_constrast', 0.1), + 'rlnMicrographPixelSize': acq['pixel_size'] + } + if mtf: + values['rlnMtfFileName'] = mtf + return Table.fromDict(values) + + @staticmethod + def movies_table(**kwargs): + extra_cols = kwargs.get('extra_cols', []) + return Table([ + 'rlnMicrographMovieName', + 'rlnOpticsGroup' + ] + extra_cols) + + @staticmethod + def micrograph_table(**kwargs): + extra_cols = kwargs.get('extra_cols', []) + return Table([ + 'rlnMicrographName', + 'rlnOpticsGroup', + 'rlnCtfImage', + 'rlnDefocusU', + 'rlnDefocusV', + 'rlnCtfAstigmatism', + 'rlnDefocusAngle', + 'rlnCtfFigureOfMerit', + 'rlnCtfMaxResolution' + ] + extra_cols) + + @staticmethod + def coordinates_table(**kwargs): + return Table(['rlnMicrographName', 'rlnMicrographCoordinates']) + + @staticmethod + def get_acquisition(inputTableOrFile): + """ Load acquisition parameters from an optics table + or a given input STAR file. + """ + if isinstance(inputTableOrFile, Table): + tOptics = inputTableOrFile + else: + with StarFile(inputTableOrFile) as sf: + tOptics = sf.getTable('optics') + + o = tOptics[0]._asdict() # get first row + + return Acquisition( + pixel_size=o.get('rlnMicrographPixelSize', + o['rlnMicrographOriginalPixelSize']), + voltage=o['rlnVoltage'], + cs=o['rlnSphericalAberration'], + amplitude_constrast=o.get('rlnAmplitudeContrast', 0.1) + ) + + @staticmethod + def pipeline_tables(): + return { + 'processes': Table(['rlnPipeLineProcessName', + 'rlnPipeLineProcessAlias', + 'rlnPipeLineProcessTypeLabel', + 'rlnPipeLineProcessStatusLabel']), + 'nodes': Table(['rlnPipeLineNodeName', + 'rlnPipeLineNodeTypeLabel', + 'rlnPipeLineNodeTypeLabelDepth']), + 'output_edges': Table(['rlnPipeLineEdgeProcess', + 'rlnPipeLineEdgeToNode']), + 'intput_edges': Table(['rlnPipeLineEdgeFromNode', + 'rlnPipeLineEdgeProcess']) + } + + @staticmethod + def write_pipeline(pipeline_star, jobCounter=1, tables=None): + with StarFile(pipeline_star, 'w') as sf: + sf.writeTimeStamp() + tGeneral = Table(['rlnPipeLineJobCounter']) + tGeneral.addRowValues(jobCounter) + sf.writeTable('pipeline_general', tGeneral, singleRow=True) + + if tables: + for name, t in tables.items(): + if len(t): + sf.writeTable(f"pipeline_{name}", t) + + @staticmethod + def job_index(jobId): + """ Return the integer job index from the name of the form Folder/jobXXX. """ + m = RelionStar.JOB_INDEX.search(jobId) + if m is None: + return None + else: + return int(m.groups()[0]) + + @staticmethod + def pipeline_to_workflow(pipelineStar): + """ Read the Relion pipeline star file and build the proper Workflow. """ + wf = Workflow() + with StarFile(pipelineStar) as sf: + tables = sf.getTableNames() + + def _table(name): + fullname = f"pipeline_{name}" + return sf.getTable(fullname) if fullname in tables else None + + if tGeneral := _table('general'): + wf.jobNextIndex = int(tGeneral[0].rlnPipeLineJobCounter) + else: + wf.jobNextIndex = 1 + + if tProc := _table('processes'): + for row in tProc: + jobId = row.rlnPipeLineProcessName + wf.registerJob(jobId, + alias=row.rlnPipeLineProcessAlias, + status=row.rlnPipeLineProcessStatusLabel, + jobtype=row.rlnPipeLineProcessTypeLabel, + jobindex=RelionStar.job_index(jobId)) + + if tNodes := _table('nodes'): + nodes = {row.rlnPipeLineNodeName: row.rlnPipeLineNodeTypeLabel + for row in tNodes} + else: + nodes = {} + + if tOutput := _table('output_edges'): + for row in tOutput: + job = wf.getJob(row.rlnPipeLineEdgeProcess) + nodeName = row.rlnPipeLineEdgeToNode + job.registerOutput(nodeName, datatype=nodes[nodeName]) + + if tInput := _table('input_edges'): + for row in tInput: + job = wf.getJob(row.rlnPipeLineEdgeProcess) + job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) + + return wf + + @staticmethod + def workflow_to_pipeline(wf, pipelineStar): + """ Write the input workflow as the expected Relion pipeline STAR file. """ + tables = RelionStar.pipeline_tables() + tProc = tables['processes'] + tNodes = tables['nodes'] + tOutput = tables['output_edges'] + + for job in wf.jobs(): + tProc.addRowValues( + rlnPipeLineProcessName=job.id, + rlnPipeLineProcessAlias=job['alias'], + rlnPipeLineProcessStatusLabel=job['status'], + rlnPipeLineProcessTypeLabel=job['jobtype'] + ) + + for o in job.outputs: + tNodes.addRowValues( + rlnPipeLineNodeName=o.id, + rlnPipeLineNodeTypeLabel=o['datatype'], + rlnPipeLineNodeTypeLabelDepth=1 + ) + tOutput.addRowValues( + rlnPipeLineEdgeProcess=job.id, + rlnPipeLineEdgeToNode=o.id + ) + + # if tOutput := _table('output_edges'): + # for row in tOutput: + # job = wf.getJob(row.rlnPipeLineEdgeProcess) + # nodeName = row.rlnPipeLineEdgeToNode + # job.registerOutput(nodeName, datatype=nodes[nodeName]) + # + # if tInput := _table('input_edges'): + # for row in tInput: + # job = wf.getJob(row.rlnPipeLineEdgeProcess) + # job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) + + RelionStar.write_pipeline(pipelineStar, wf.jobNextIndex, tables) From b244b8a87c466dbdc83a09e63f6c5e98feb088b8 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 14 Feb 2025 09:07:58 -0600 Subject: [PATCH 034/161] Added function Job.hasOutput --- emtools/jobs/batch_manager.py | 3 +-- emtools/jobs/pipeline.py | 4 ---- emtools/jobs/workflow.py | 11 +++-------- 3 files changed, 4 insertions(+), 14 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index da8c4d9..8d70975 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -17,11 +17,10 @@ import os from uuid import uuid4 from datetime import datetime -import itertools import json import subprocess -from emtools.utils import Process, Color, FolderManager, Pretty +from emtools.utils import Color, FolderManager class Args(dict): diff --git a/emtools/jobs/pipeline.py b/emtools/jobs/pipeline.py index d404d4f..48cda02 100644 --- a/emtools/jobs/pipeline.py +++ b/emtools/jobs/pipeline.py @@ -14,13 +14,9 @@ # * # ************************************************************************** -import os -import sys from collections import OrderedDict import threading -from emtools.utils import Process - class Pipeline: """ diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index 188b439..1762c02 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -14,14 +14,6 @@ # * # ************************************************************************** -import os -import sys -from collections import OrderedDict -import threading - -from emtools.utils import Process -from emtools.metadata import StarFile - class Workflow: """ @@ -95,6 +87,9 @@ def registerOutput(self, dataId, **kwargs): self.outputs.append(data) return data + def hasOutput(self, dataId): + return any(o.id == dataId for o in self.outputs) + def _validateInputs(self, inputs): for i in inputs: if not isinstance(i, Workflow.Data): From cdfa949e608a443c5d41834713d7d32cd388bdec Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Mon, 17 Feb 2025 07:38:32 -0600 Subject: [PATCH 035/161] Added Timer to Batch --- emtools/jobs/batch_manager.py | 23 ++++++++++++++++++++++- emtools/utils/time.py | 3 +++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 8d70975..4f4873a 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -20,7 +20,7 @@ import json import subprocess -from emtools.utils import Color, FolderManager +from emtools.utils import Color, FolderManager, Timer, Pretty class Args(dict): @@ -44,6 +44,8 @@ def __init__(self, *args, **kwargs): dict.__init__(self, *args, **kwargs) FolderManager.__init__(self, self['path']) self._logId = f" {self.id}:" + self._timer = Timer() # Create a timer to monitor batch execution + self._timerPrefix = '' @property def id(self): @@ -59,6 +61,14 @@ def info(self): self['info'] = {} return self['info'] + @property + def error(self): + return self.info.get('error', None) + + @error.setter + def error(self, value): + self.info['error'] = str(value) + def dump(self, obj, fn): filePath = self.join(fn) with open(filePath, 'w') as f: @@ -97,6 +107,17 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): kwargs['cwd'] = self.path subprocess.call(args, **kwargs) + def tic(self, prefix=''): + self._timer.tic() + self._timerPrefix = prefix + + def toc(self): + self.info.update({ + f'{self._timerPrefix}start': self._timer.getTic(), + f'{self._timerPrefix}end': Pretty.now(), + f'{self._timerPrefix}elapsed': str(self._timer.getElapsedTime()) + }) + class BatchManager: """ Class used to generate and handle the creation of batches diff --git a/emtools/utils/time.py b/emtools/utils/time.py index b7de988..e1a336e 100644 --- a/emtools/utils/time.py +++ b/emtools/utils/time.py @@ -38,6 +38,9 @@ def getElapsedTime(self): def toc(self, message=None, pretty=False): print(self.getToc(message=message, pretty=pretty)) + def getTic(self): + return Pretty.datetime(self._dt) + def getToc(self, message=None, pretty=False): if message: self.message = message From 41ab96d77c38f1ee9b38e6f08a5c6d99e9dfcb85 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 18 Feb 2025 12:02:22 -0600 Subject: [PATCH 036/161] Added Mdoc.glob and fixed circular import --- emtools/metadata/misc.py | 11 +++++++++++ emtools/metadata/starfile.py | 4 +++- 2 files changed, 14 insertions(+), 1 deletion(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 8d48140..a96dfc7 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -17,6 +17,7 @@ import os import pathlib from datetime import datetime, timedelta +from glob import glob from emtools.utils import Path, Pretty, Process @@ -274,6 +275,16 @@ def parse(mdocFn): return mdoc + @staticmethod + def glob(mdocPattern): + mdocs = [] + for mdocFn in glob(mdocPattern): + mdoc = Mdoc.parse(mdocFn) + mdoc['MdocFile'] = {'Path': mdocFn} + mdocs.append(mdoc) + + return mdocs + @staticmethod def getSubFrameBase(section): """ Helper method to extract the subframe base filename. """ diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index f0f7a43..1879100 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -30,7 +30,7 @@ import emtools from emtools.utils import Pretty, Color -from emtools.jobs import Workflow + from .table import ColumnList, Table from .misc import Acquisition @@ -606,6 +606,8 @@ def job_index(jobId): @staticmethod def pipeline_to_workflow(pipelineStar): """ Read the Relion pipeline star file and build the proper Workflow. """ + from emtools.jobs import Workflow # import here to avoid circular imports + wf = Workflow() with StarFile(pipelineStar) as sf: tables = sf.getTableNames() From 11643ed0f91cc1a0b9e55e880b7b5d91772045cc Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 18 Feb 2025 12:03:04 -0600 Subject: [PATCH 037/161] Added more funcs to Batch and new class MdocBatchManager --- emtools/jobs/__init__.py | 4 +- emtools/jobs/batch_manager.py | 75 ++++++++++++++++++++++++++++++++--- 2 files changed, 71 insertions(+), 8 deletions(-) diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index fa04441..a48607d 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -15,7 +15,7 @@ # ************************************************************************** from .pipeline import Pipeline -from .batch_manager import Args, Batch, BatchManager +from .batch_manager import Args, Batch, BatchManager, MdocBatchManager from .workflow import Workflow -__all__ = ["Pipeline", "BatchManager", "Batch", "Args", "Workflow"] +__all__ = ["Pipeline", "BatchManager", "Batch", "Args", "Workflow", "MdocBatchManager"] diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 4f4873a..6c9a4ea 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -19,8 +19,11 @@ from datetime import datetime import json import subprocess +import traceback +from contextlib import contextmanager -from emtools.utils import Color, FolderManager, Timer, Pretty +from emtools.utils import Color, FolderManager, Timer, Pretty, Path +from emtools.metadata import Mdoc class Args(dict): @@ -47,6 +50,9 @@ def __init__(self, *args, **kwargs): self._timer = Timer() # Create a timer to monitor batch execution self._timerPrefix = '' + def clone(self): + return Batch(self) + @property def id(self): return self['id'] @@ -118,6 +124,16 @@ def toc(self): f'{self._timerPrefix}elapsed': str(self._timer.getElapsedTime()) }) + @contextmanager + def execute(self): + try: + self.tic() + yield self + except Exception as e: + self.error = traceback.format_exc() + finally: + self.toc() + class BatchManager: """ Class used to generate and handle the creation of batches @@ -150,16 +166,20 @@ def _createBatchId(self): uuidSuffix = str(uuid4()).split('-')[0] return f"{nowPrefix}_{countStr}_{uuidSuffix}" - def _createBatch(self, items, inputFolder=None): + def _createBatch(self, items, inputFolder=None, **batchAttrs): self._batchCount += 1 batch_id = self._createBatchId() batch_path = os.path.join(self._workingPath, batch_id) batch = Batch(id=batch_id, index=self._batchCount, path=batch_path, - items=items) + items=items, + **batchAttrs) batch.create() + self._createBatchLinks(batch, items, inputFolder=inputFolder) + return batch + def _createBatchLinks(self, batch, items, inputFolder=None): if inputFolder is not None: batch.mkdir(inputFolder) @@ -168,10 +188,8 @@ def _createBatch(self, items, inputFolder=None): baseName = os.path.basename(fn) if inputFolder is not None: baseName = os.path.join(inputFolder, baseName) - os.symlink(os.path.abspath(fn), - os.path.join(batch_path, baseName)) + os.symlink(os.path.abspath(fn), batch.join(baseName)) - return batch def generate(self): """ Generate batches based on the input items. """ @@ -187,3 +205,48 @@ def generate(self): if items: yield self._createBatch(items) + +class MdocBatchManager(BatchManager): + """ Batch manager for Tilt-series. """ + + def __init__(self, tsIterator, workingPath, suffix=None, movies=None): + """ + Args: + tsIterator: input tilt-series iterator + workingPath: path where the batches folder will be created + suffix: suffix to be removed from mdoc filename to generate + the tilt-series name + """ + BatchManager.__init__(self, 0, tsIterator, workingPath, + itemFileNameFunc=lambda item: item[1]['SubFramePath']) + self._suffix = suffix + self._movies = movies + + def _subframePath(self, mdocFn, section): + movieFolder = self._movies or os.path.dirname(mdocFn) + return os.path.join(movieFolder, Mdoc.getSubFrameBase(section)) + + def _tsName(self, mdocFn): + name = Path.removeBaseExt(mdocFn) + if self._suffix: + name = name.replace(self._suffix, '') + return name + + def generate(self): + """ Generate batches based on the input items. """ + for mdoc in self._items: + mdocFn = mdoc['MdocFile']['Path'] + yield self._createBatch(mdoc.zvalues, mdoc=mdoc, tsName=self._tsName(mdocFn)) + + def _createBatchLinks(self, batch, items, inputFolder=None): + mdocFn = batch['mdoc']['MdocFile']['Path'] + + def _absfn(item): + return os.path.abspath(self._subframePath(mdocFn, item[1])) + + framesFolder = os.path.dirname(_absfn(items[0])) + os.symlink(framesFolder, batch.join('frames')) + + for item in items: + baseName = os.path.basename(_absfn(item)) + os.symlink(os.path.join('frames', baseName), batch.join(baseName)) \ No newline at end of file From 064744c65886a34a9eeb4979cca9f3685f73adad Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 18 Feb 2025 16:17:35 -0600 Subject: [PATCH 038/161] Added table for tiltseries --- emtools/metadata/starfile.py | 38 ++++++++++++++++++++++++++++++++++++ 1 file changed, 38 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 1879100..168db09 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -544,6 +544,44 @@ def micrograph_table(**kwargs): def coordinates_table(**kwargs): return Table(['rlnMicrographName', 'rlnMicrographCoordinates']) + @staticmethod + def tiltseries_table(mc=True, ctf=True, **kwargs): + cols = [ + 'rlnMicrographMovieName', + 'rlnTomoTiltMovieFrameCount', + 'rlnTomoNominalStageTiltAngle', + 'rlnTomoNominalTiltAxisAngle', + 'rlnMicrographPreExposure', + 'rlnTomoNominalDefocus', + 'rlnMicrographNameEven', + 'rlnMicrographNameOdd', + 'rlnMicrographName' + ] + + if mc: + cols.extend([ + 'rlnMicrographMetadata', + 'rlnAccumMotionTotal', + 'rlnAccumMotionEarly', + 'rlnAccumMotionLate' + ]) + + if ctf: + cols.extend([ + 'rlnCtfImage', + 'rlnDefocusU', + 'rlnDefocusV', + 'rlnCtfAstigmatism', + 'rlnDefocusAngle', + 'rlnCtfFigureOfMerit', + 'rlnCtfMaxResolution', + 'rlnCtfIceRingDensity' + ]) + + cols.extend(kwargs.get('extra_cols', [])) + + return Table(cols) + @staticmethod def get_acquisition(inputTableOrFile): """ Load acquisition parameters from an optics table From eefb14fc1ca66238160d56f57bc07e6fe848598d Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 18 Feb 2025 16:56:36 -0600 Subject: [PATCH 039/161] Added global table for tiltseries --- emtools/metadata/starfile.py | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 168db09..0518890 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -582,6 +582,23 @@ def tiltseries_table(mc=True, ctf=True, **kwargs): return Table(cols) + @staticmethod + def global_tiltseries_table(**kwargs): + cols = [ + 'rlnTomoName', + 'rlnTomoTiltSeriesStarFile', + 'rlnVoltage', + 'rlnSphericalAberration', + 'rlnAmplitudeContrast', + 'rlnMicrographOriginalPixelSize', + 'rlnTomoHand', + 'rlnOpticsGroupName', + 'rlnTomoTiltSeriesPixelSize' + ] + cols.extend(kwargs.get('extra_cols', [])) + + return Table(cols) + @staticmethod def get_acquisition(inputTableOrFile): """ Load acquisition parameters from an optics table From edad74c6748061db22a455907e1ae7c6a8c6074c Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Thu, 20 Feb 2025 09:02:25 -0600 Subject: [PATCH 040/161] Added Path.rsync method --- emtools/scripts/emt_files.py | 7 +++++++ emtools/utils/path.py | 23 ++++++++++++++++++++--- 2 files changed, 27 insertions(+), 3 deletions(-) diff --git a/emtools/scripts/emt_files.py b/emtools/scripts/emt_files.py index 2b8b99f..267ee52 100755 --- a/emtools/scripts/emt_files.py +++ b/emtools/scripts/emt_files.py @@ -158,6 +158,9 @@ def main(): help='Copy directory with some delay') g.add_argument('--check_dirs', nargs=2, metavar=('DIR1', 'DIR2'), help='Check if the two directories are synchronized. ') + g.add_argument('--rsync_dirs', nargs=2, metavar=('DIR1', 'DIR2'), + help='Rsync both directories and print the number of ' + 'transferred files. ') p.add_argument('--bin', '-b', type=int, default=6000, help="Create bins of the given time in minutes " @@ -214,6 +217,10 @@ def _mkdir(d): s = Color.green('in SYNC') if sync else Color.red('NOT in SYNC') print(f"Dirs are {s}") + elif dirs := args.rsync_dirs: + n = Path.rsync(dirs[0], dirs[1], verbose=True) + print(f"Transferred files: {n}") + elif pattern := args.timing: timeStats(pattern, args.bin, args.plot) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 5df140f..19f5ab7 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -97,19 +97,36 @@ def inSync(dir1, dir2, verbose=False): Use rsync as a subprocess to check if the two directories are synchronized. Both directories must exist. """ + return Path.rsync(dir1, dir2, dry=True, verbose=verbose) == 0 + + @staticmethod + def rsync(dir1, dir2, dry=False, verbose=False): + """ Run rsync to synchronize dir1 and dir2 are synchronized (i.e. same content) + Use rsync as a subprocess to synchronize dir1 and dir2 and return + the number of files transferred. + """ dir1 = Path.addslash(dir1) dir2 = Path.addslash(dir2) - p = Process('rsync', '--dry-run', '-a', '--stats', dir1, dir2) + args = ['rsync', '-a', '--stats', dir1, dir2] + if dry: + args.insert(1, '--dry-run') + + p = Process(*args) + if verbose: p.print(stdout=True) transf = 1 for line in p.lines(): if 'files transferred:' in line: - transf = int(line.split(':')[1]) + value = line.split(':')[1] + # Remove , that is used to separate thousands + transf = int(value.replace(',', '')) break - return transf == 0 + + return transf + @staticmethod def lastModified(folder): From 525e6dff4e99c7fe9f034bbed5c45ffec89f4870 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 21 Feb 2025 08:51:15 -0600 Subject: [PATCH 041/161] Fixed typo --- emtools/utils/pretty.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/emtools/utils/pretty.py b/emtools/utils/pretty.py index 626f4c3..0fb248b 100644 --- a/emtools/utils/pretty.py +++ b/emtools/utils/pretty.py @@ -80,7 +80,7 @@ def modified(fn, **kwargs): @staticmethod def elapsed(timestamp, now=None): """ - Get a datetime object or a int() Epoch timestamp and return a + Get a datetime object or an int() Epoch timestamp and return a pretty string like 'an hour ago', 'Yesterday', '3 months ago', 'just now', etc """ From c7b30bffa111cb324e21ee85b5636adfcbe48968 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Mon, 24 Feb 2025 14:01:03 -0600 Subject: [PATCH 042/161] Fixed typos --- emtools/metadata/starfile.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 0518890..a3609a0 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -510,7 +510,7 @@ def optics_table(acq, opticsGroup=1, opticsGroupName="opticsGroup1", 'rlnMicrographOriginalPixelSize': origPs, 'rlnVoltage': acq['voltage'], 'rlnSphericalAberration': acq['cs'], - 'rlnAmplitudeContrast': acq.get('amplitude_constrast', 0.1), + 'rlnAmplitudeContrast': acq.get('amplitude_contrast', 0.1), 'rlnMicrographPixelSize': acq['pixel_size'] } if mtf: @@ -617,7 +617,7 @@ def get_acquisition(inputTableOrFile): o['rlnMicrographOriginalPixelSize']), voltage=o['rlnVoltage'], cs=o['rlnSphericalAberration'], - amplitude_constrast=o.get('rlnAmplitudeContrast', 0.1) + amplitude_contrast=o.get('rlnAmplitudeContrast', 0.1) ) @staticmethod From 7f6a7514aa354fb859f33c843c697ba7267667a3 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Wed, 26 Feb 2025 16:12:22 -0600 Subject: [PATCH 043/161] Minor updates --- emtools/metadata/misc.py | 16 ++++++++-------- emtools/utils/process.py | 10 +++++++++- 2 files changed, 17 insertions(+), 9 deletions(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index a96dfc7..2a725ea 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -324,32 +324,32 @@ class Acquisition(dict): @property def pixel_size(self): - return self['pixel_size'] + return float(self['pixel_size']) @pixel_size.setter def pixel_size(self, value): - self['pixel_size'] = value + self['pixel_size'] = float(value) @property def voltage(self): - return self['voltage'] + return float(self['voltage']) @voltage.setter def voltage(self, value): - self['voltage'] = value + self['voltage'] = float(value) @property def cs(self): - return self['cs'] + return float(self['cs']) @cs.setter def cs(self, value): - self['cs'] = value + self['cs'] = float(value) @property def amplitude_contrast(self): - return self.get('amplitude_contrast', 0.1) + return float(self.get('amplitude_contrast', 0.1)) @amplitude_contrast.setter def amplitude_contrast(self, value): - self['amplitude_contrast'] = value \ No newline at end of file + self['amplitude_contrast'] = float(value) diff --git a/emtools/utils/process.py b/emtools/utils/process.py index 7db2b56..c287f7c 100644 --- a/emtools/utils/process.py +++ b/emtools/utils/process.py @@ -96,10 +96,18 @@ def _addProc(f, proc): pids.add(proc.pid) attrs = ['pid', 'ppid', 'name', 'cwd', 'username', 'memory_percent', 'cpu_percent'] + def _filter_name(proc): + if program and program not in proc.info['name']: + cmdline = proc.cmdline() + if len(cmdline) == 0 or program not in cmdline[0]: + return False + return True + for proc in psutil.process_iter(attrs): - if not program or program in proc.info['name']: + if _filter_name(proc): folder = proc.info['cwd'] if workingDir is None or folder == workingDir: + print(f"program: {program}, proc_info: {proc.info['name']}") _addProc(folder, proc) if children: for child in proc.children(recursive=True): From d0078f4040a19c4c4108808e2a2aeda1ddc18b9d Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Fri, 28 Feb 2025 10:32:10 -0600 Subject: [PATCH 044/161] Added flush to log function --- emtools/utils/path.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 19f5ab7..d2f5570 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -249,6 +249,7 @@ class FolderManager: def __init__(self, path): self.__path = path self._logId = "" + self.__extraLog = None def join(self, *p): return os.path.join(self.__path, *p) @@ -274,7 +275,12 @@ def create(self, **kwargs): Process.system(f"rm -rf '{self.path}'", **kwargs) Process.system(f"mkdir -p '{self.path}'", **kwargs) - def log(self, msg): + def log(self, msg, flush=False): logMsg = f"{Pretty.now()}:{self._logId} {msg}" - print(logMsg) + print(logMsg, flush=flush) + if self.__extraLog: + self.__extraLog(logMsg, flush=flush) return logMsg + + def setExtraLog(self, logFunc): + self.__extraLog = logFunc From 92fc202c084be6c4d29b27d65675f71d871567d2 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sat, 1 Mar 2025 15:51:57 -0600 Subject: [PATCH 045/161] Added Vars helper class --- emtools/jobs/__init__.py | 5 +++-- emtools/jobs/batch_manager.py | 23 +++++++++++++++++++++++ 2 files changed, 26 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index a48607d..8ae82d2 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -15,7 +15,8 @@ # ************************************************************************** from .pipeline import Pipeline -from .batch_manager import Args, Batch, BatchManager, MdocBatchManager +from .batch_manager import Args, Batch, BatchManager, MdocBatchManager, Vars from .workflow import Workflow -__all__ = ["Pipeline", "BatchManager", "Batch", "Args", "Workflow", "MdocBatchManager"] +__all__ = ["Pipeline", "BatchManager", "Batch", "Args", "Vars", + "Workflow", "MdocBatchManager"] diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 6c9a4ea..6db5698 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -41,6 +41,29 @@ def toLine(self): return ' '.join("%s %s" % (k, v) for k, v in self.items()) +class Vars: + """ Handle variable definitions, either from input dict + or from os.environ. + """ + def __init__(self, vars={}): + self._vars = vars + + def get(self, key, is_path=False): + """ Get the var for that Key, raising exception if the var does not exist. + If is_path = True, validates that the path exists. + """ + value = self._vars.get(key, os.environ.get(key, None)) + + if value is None: + raise Exception(f"ERROR: Missing expected variable {key}.") + + if is_path and not os.path.exists(value): + raise Exception(f"ERROR: Variable {key}={value} does not exist.") + + return value + + + class Batch(dict, FolderManager): """ Subclass from dict with some utilities related to Batch logic. """ def __init__(self, *args, **kwargs): From 49c2af239d25b52e172f094fbb4aa11709e56c16 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Mon, 3 Mar 2025 10:33:21 -0600 Subject: [PATCH 046/161] emt-image utility and added fractions.mrc as movie suffix --- emtools/image/__main__.py | 32 ++++++++++++++++++++++++++++++++ emtools/metadata/epu.py | 3 ++- emtools/metadata/misc.py | 2 +- setup.py | 4 +++- 4 files changed, 38 insertions(+), 3 deletions(-) create mode 100644 emtools/image/__main__.py diff --git a/emtools/image/__main__.py b/emtools/image/__main__.py new file mode 100644 index 0000000..5254c7f --- /dev/null +++ b/emtools/image/__main__.py @@ -0,0 +1,32 @@ +# ************************************************************************** +# * +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) +# * +# * This program is free software; you can redistribute it and/or modify +# * it under the terms of the GNU General Public License as published by +# * the Free Software Foundation; either version 3 of the License, or +# * (at your option) any later version. +# * +# * This program is distributed in the hope that it will be useful, +# * but WITHOUT ANY WARRANTY; without even the implied warranty of +# * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# * GNU General Public License for more details. +# * +# ************************************************************************** + +import argparse +from .thumbnail import Image + + +def main(): + p = argparse.ArgumentParser() + p.add_argument('path', metavar="IMAGE_PATH", + help="Image path") + + args = p.parse_args() + + print(Image.get_dimensions(args.path)) + + +if __name__ == '__main__': + main() diff --git a/emtools/metadata/epu.py b/emtools/metadata/epu.py index 2190866..0d99ab5 100644 --- a/emtools/metadata/epu.py +++ b/emtools/metadata/epu.py @@ -25,7 +25,8 @@ class EPU: - MOVIES_SUFFICES = ['_fractions.tiff', '_EER.eer'] + MOVIES_SUFFICES = ['_fractions.tiff', '_EER.eer', '_fractions.mrc'] + @staticmethod def get_acquisition(movieXmlFn): """ Parse acquisition parameters from EPU's xml movie file. """ diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 2a725ea..4038a35 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -208,7 +208,7 @@ class MovieFiles(DataFiles): def __init__(self, **kwargs): DataFiles.__init__(self, filters=[self.is_movie], **kwargs) self._moviesSuffix = kwargs.get('moviesSuffix', - ['fractions.tiff', '.eer']) + ['fractions.tiff', '.eer', 'fractions.mrc']) def is_movie(self, fn): return any(fn.endswith(s) for s in self._moviesSuffix) diff --git a/setup.py b/setup.py index 6492ee2..35b91cf 100644 --- a/setup.py +++ b/setup.py @@ -78,7 +78,9 @@ 'emt-epu = emtools.scripts.emt_epu:main', 'emt-beamshifts = emtools.scripts.emt_beamshifts:main', 'emt-angdist = emtools.scripts.emt_angdist:main', - 'emt-star = emtools.scripts.emt_star:main' + 'emt-star = emtools.scripts.emt_star:main', + 'emt-image = emtools.image.__main__:main' + ], }, From 8c121bb391f1608c48ca1742188ea2901b3c1fd8 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 4 Mar 2025 16:16:22 -0600 Subject: [PATCH 047/161] Added some utilities --- emtools/scripts/emt_ps.py | 37 +---------------------------------- emtools/utils/path.py | 7 +++++++ emtools/utils/process.py | 41 +++++++++++++++++++++++++++++++++++++++ 3 files changed, 49 insertions(+), 36 deletions(-) diff --git a/emtools/scripts/emt_ps.py b/emtools/scripts/emt_ps.py index bcc4bab..9ad65ef 100755 --- a/emtools/scripts/emt_ps.py +++ b/emtools/scripts/emt_ps.py @@ -55,44 +55,9 @@ def main(): print(System.hostname()) sys.exit(0) - v = args.verbose - - kill = args.kill folderPath = os.path.abspath(args.folder) if args.folder else args.folder print('path', folderPath) - processes = Process.ps(args.name, workingDir=folderPath, children=args.children) - - color = Color.red if kill else Color.bold - - for folder, procs in processes.items(): - print(Color.warn(f"{folder}")) - header = f" {'USER':<15} {'PPID/PID':<15} {color('PROGRAM'):<30}" - if v > 0: - header += f" {'CPU(%)':>10} {'MEMORY(%)':>10}" - if v > 1: - header += f" {'COMMAND LINE'}" - - print(Color.bold(header)) - - prefix = 'Killing' if kill else '' - for p in procs: - pidstr = f"{p.info['ppid']}/{p.pid}" - msg = f" {prefix} {p.info['username']:<15} {pidstr:<15} {color(p.info['name']):<30}" - if v > 0: - try: - cpu_percent = p.cpu_percent(interval=1) / cpus - except: - continue - - msg += f" {cpu_percent:>10,.2f} {p.info['memory_percent']:>10,.2f}" - if v > 1: - msg += f" {p.cmdline()}" - print(msg) - if kill: - try: - p.kill() - except: - pass + Process.checkChilds(args.name, folderPath, kill=args.kill, verbose=args.verbose) if __name__ == '__main__': diff --git a/emtools/utils/path.py b/emtools/utils/path.py index d2f5570..f0a844d 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -27,6 +27,9 @@ from .color import Color +GLOB_CHARS = ['*', '?', '[', ']'] + + class Path: """ Group some path utility functions. @@ -243,6 +246,10 @@ def exists(path): """ return path and os.path.exists(path) + @staticmethod + def isPattern(path): + return any(c in path for c in GLOB_CHARS) + class FolderManager: """ Helper class with some path utilities from a given path. """ diff --git a/emtools/utils/process.py b/emtools/utils/process.py index c287f7c..3f43cf8 100644 --- a/emtools/utils/process.py +++ b/emtools/utils/process.py @@ -22,6 +22,8 @@ import subprocess import logging +from .color import Color + def _print(*msgs): print(*msgs) @@ -116,6 +118,45 @@ def _filter_name(proc): return processes + @staticmethod + def checkChilds(programName, folderPath, kill=False, verbose=0): + from .system import System + specs = System.specs() + cpus = specs['CPUs'] + processes = Process.ps(programName, workingDir=folderPath, children=True) + + color = Color.red if kill else Color.bold + + for folder, procs in processes.items(): + print(Color.warn(f"{folder}")) + header = f" {'USER':<15} {'PPID/PID':<15} {color('PROGRAM'):<30}" + if verbose > 0: + header += f" {'CPU(%)':>10} {'MEMORY(%)':>10}" + if verbose > 1: + header += f" {'COMMAND LINE'}" + + print(Color.bold(header)) + + prefix = 'Killing' if kill else '' + for p in procs: + pidstr = f"{p.info['ppid']}/{p.pid}" + msg = f" {prefix} {p.info['username']:<15} {pidstr:<15} {color(p.info['name']):<30}" + if verbose > 0: + try: + cpu_percent = p.cpu_percent(interval=1) / cpus + except: + continue + + msg += f" {cpu_percent:>10,.2f} {p.info['memory_percent']:>10,.2f}" + if verbose > 1: + msg += f" {p.cmdline()}" + print(msg) + if kill: + try: + p.kill() + except: + pass + class Logger: """ Use a logger to log commands that are executed via os.system. """ def __init__(self, logger=None, only_log=False, From 99be009285da3ba7ccf4494bc0402be08977d4df Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sun, 9 Mar 2025 15:09:16 -0500 Subject: [PATCH 048/161] Allow to avoid creation of batch folder --- emtools/jobs/batch_manager.py | 21 ++++++++++++--------- emtools/utils/path.py | 5 +++++ 2 files changed, 17 insertions(+), 9 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 6db5698..a6df540 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -63,7 +63,6 @@ def get(self, key, is_path=False): return value - class Batch(dict, FolderManager): """ Subclass from dict with some utilities related to Batch logic. """ def __init__(self, *args, **kwargs): @@ -106,11 +105,13 @@ def dump(self, obj, fn): def dump_info(self): self.dump(self.info, 'info.json') - def dump_all(self): - self.dump(self, 'batch.json') + def dump_all(self, fn=None): + fileName = fn or 'batch.json' + self.dump(self, fileName) - def load_all(self): - with open(self.join('batch.json')) as f: + def load_all(self, fn=None): + filePath = fn or self.join('batch.json') + with open(filePath) as f: self.update(json.load(f)) def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): @@ -167,7 +168,8 @@ class BatchManager: folder. """ def __init__(self, batchSize, inputItemsIterator, workingPath, - itemFileNameFunc=lambda item: item.getFileName()): + itemFileNameFunc=lambda item: item.getFileName(), + createBatch=True): """ Args: batchSize: Number of items that will be grouped into one batch @@ -181,6 +183,7 @@ def __init__(self, batchSize, inputItemsIterator, workingPath, self._batchCount = 0 self._workingPath = workingPath self._itemFileNameFunc = itemFileNameFunc + self._create = createBatch def _createBatchId(self): # We will use batchCount, before the batch is created @@ -198,8 +201,9 @@ def _createBatch(self, items, inputFolder=None, **batchAttrs): path=batch_path, items=items, **batchAttrs) - batch.create() - self._createBatchLinks(batch, items, inputFolder=inputFolder) + if self._create: + batch.create() + self._createBatchLinks(batch, items, inputFolder=inputFolder) return batch def _createBatchLinks(self, batch, items, inputFolder=None): @@ -213,7 +217,6 @@ def _createBatchLinks(self, batch, items, inputFolder=None): baseName = os.path.join(inputFolder, baseName) os.symlink(os.path.abspath(fn), batch.join(baseName)) - def generate(self): """ Generate batches based on the input items. """ items = [] diff --git a/emtools/utils/path.py b/emtools/utils/path.py index f0a844d..61789dd 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -276,6 +276,10 @@ def exists(self, *p): def path(self): return self.__path + @path.setter + def path(self, value): + self.__path = value + def create(self, **kwargs): """ Create batch folder. """ self.log(f"Creating folder: {self.path}") @@ -291,3 +295,4 @@ def log(self, msg, flush=False): def setExtraLog(self, logFunc): self.__extraLog = logFunc + From ee190c3e4f6d2daf95e62272e71ae989118062d2 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 10 Mar 2025 15:31:36 -0500 Subject: [PATCH 049/161] Fixed bug in getTable with non-existing table --- emtools/metadata/starfile.py | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index a3609a0..20cbeac 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -124,7 +124,11 @@ def getTable(self, tableName, **kwargs): types=None, optional types dict with {columnName: columnType} pairs that allows to specify types for certain columns. """ - self.__createTable(tableName, **kwargs) + try: + self.__createTable(tableName, **kwargs) + except: + return None + if self._singleRow: self._table.addRow(self.__rowFromValues(self._values)) else: @@ -286,6 +290,8 @@ def _findDataLine(self, dataName): break line = f.readline() # Start from the beginning and scann until complete the full loop + if initial_offset == 0: + break f.seek(0) offset = 0 line = f.readline() From 1c428bb3b95f47b3467ec74fd602044a12504ed6 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 11 Mar 2025 20:35:26 -0500 Subject: [PATCH 050/161] Allow imageId in micrograph_table --- emtools/metadata/starfile.py | 11 ++++++++--- emtools/utils/path.py | 4 ++++ 2 files changed, 12 insertions(+), 3 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 20cbeac..a98240f 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -533,8 +533,10 @@ def movies_table(**kwargs): @staticmethod def micrograph_table(**kwargs): - extra_cols = kwargs.get('extra_cols', []) - return Table([ + cols = [] + if image_id := kwargs.get('image_id', None): + cols.append(image_id) + cols.extend([ 'rlnMicrographName', 'rlnOpticsGroup', 'rlnCtfImage', @@ -544,7 +546,10 @@ def micrograph_table(**kwargs): 'rlnDefocusAngle', 'rlnCtfFigureOfMerit', 'rlnCtfMaxResolution' - ] + extra_cols) + ]) + if extra_cols := kwargs.get('extra_cols', []): + cols.extend(extra_cols) + return Table(cols) @staticmethod def coordinates_table(**kwargs): diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 61789dd..240c75e 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -296,3 +296,7 @@ def log(self, msg, flush=False): def setExtraLog(self, logFunc): self.__extraLog = logFunc + def listdir(self): + """ Return files relative to the path. """ + return os.listdir(self.path) + From f97daaf005ca4e3b70c60ec507228bcbe48f40d6 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 12 Mar 2025 13:42:44 -0500 Subject: [PATCH 051/161] Added Timer.parse_timedelta function --- emtools/utils/time.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/emtools/utils/time.py b/emtools/utils/time.py index e1a336e..4aa4570 100644 --- a/emtools/utils/time.py +++ b/emtools/utils/time.py @@ -14,7 +14,7 @@ # * # ************************************************************************** -from datetime import datetime +from datetime import datetime, timedelta from functools import wraps from .pretty import Pretty @@ -65,3 +65,8 @@ def wrap(*args, **kw): t.toc(f"Function {func.__name__} took: ") return result return wrap + + @staticmethod + def parse_timedelta(tdStr): + hours, minutes, seconds = tuple(map(float, tdStr.split(':'))) + return timedelta(hours=hours, minutes=minutes, seconds=seconds) From c4d59b597e4d91a68402cbd84adb386ad7cd5202 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sun, 16 Mar 2025 08:42:14 -0500 Subject: [PATCH 052/161] Moved dump function to FolderManager --- emtools/jobs/batch_manager.py | 5 ----- emtools/utils/path.py | 6 ++++++ 2 files changed, 6 insertions(+), 5 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index a6df540..a931314 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -97,11 +97,6 @@ def error(self): def error(self, value): self.info['error'] = str(value) - def dump(self, obj, fn): - filePath = self.join(fn) - with open(filePath, 'w') as f: - json.dump(obj, f, indent=4) - def dump_info(self): self.dump(self.info, 'info.json') diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 240c75e..5c4f358 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -18,6 +18,7 @@ import shutil import time import tempfile +import json from datetime import datetime as dt from collections import OrderedDict from contextlib import contextmanager @@ -300,3 +301,8 @@ def listdir(self): """ Return files relative to the path. """ return os.listdir(self.path) + def dump(self, obj, fn): + filePath = self.join(fn) + with open(filePath, 'w') as f: + json.dump(obj, f, indent=4) + From d8b183c4c860ffea9fd10b75f3f9c3c8d394faa3 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 18 Mar 2025 10:18:29 -0500 Subject: [PATCH 053/161] Check if input star file exists --- emtools/metadata/starfile.py | 31 +++++++++++++++++-------------- 1 file changed, 17 insertions(+), 14 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index a98240f..6714b44 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -454,20 +454,23 @@ def __init__(self, fileName, tableName, rowKeyFunc, **kwargs): def update(self): newRows = [] - now = datetime.now() - mTime = datetime.fromtimestamp(os.path.getmtime(self.fileName)) - - if self.lastCheck is None or mTime > self.lastCheck: - with StarFile(self.fileName) as sf: - for row in sf.iterTable(self._tableName): - rowKey = self._rowKeyFunc(row) - if rowKey not in self._seenItems: - self._seenItems.add(rowKey) - newRows.append(row) - - self.lastCheck = now - if newRows: - self.lastUpdate = now + + if os.path.exists(self.fileName): + now = datetime.now() + mTime = datetime.fromtimestamp(os.path.getmtime(self.fileName)) + + if self.lastCheck is None or mTime > self.lastCheck: + with StarFile(self.fileName) as sf: + for row in sf.iterTable(self._tableName): + rowKey = self._rowKeyFunc(row) + if rowKey not in self._seenItems: + self._seenItems.add(rowKey) + newRows.append(row) + + self.lastCheck = now + if newRows: + self.lastUpdate = now + return newRows def timedOut(self): From 380371480bc76e2017f79fb146e59e96acdae0d7 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 25 Mar 2025 17:53:26 -0500 Subject: [PATCH 054/161] Updated some utils methods --- emtools/utils/path.py | 15 +++++++++++++++ emtools/utils/time.py | 7 ++++++- 2 files changed, 21 insertions(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index f0a844d..5c4f358 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -18,6 +18,7 @@ import shutil import time import tempfile +import json from datetime import datetime as dt from collections import OrderedDict from contextlib import contextmanager @@ -276,6 +277,10 @@ def exists(self, *p): def path(self): return self.__path + @path.setter + def path(self, value): + self.__path = value + def create(self, **kwargs): """ Create batch folder. """ self.log(f"Creating folder: {self.path}") @@ -291,3 +296,13 @@ def log(self, msg, flush=False): def setExtraLog(self, logFunc): self.__extraLog = logFunc + + def listdir(self): + """ Return files relative to the path. """ + return os.listdir(self.path) + + def dump(self, obj, fn): + filePath = self.join(fn) + with open(filePath, 'w') as f: + json.dump(obj, f, indent=4) + diff --git a/emtools/utils/time.py b/emtools/utils/time.py index e1a336e..4aa4570 100644 --- a/emtools/utils/time.py +++ b/emtools/utils/time.py @@ -14,7 +14,7 @@ # * # ************************************************************************** -from datetime import datetime +from datetime import datetime, timedelta from functools import wraps from .pretty import Pretty @@ -65,3 +65,8 @@ def wrap(*args, **kw): t.toc(f"Function {func.__name__} took: ") return result return wrap + + @staticmethod + def parse_timedelta(tdStr): + hours, minutes, seconds = tuple(map(float, tdStr.split(':'))) + return timedelta(hours=hours, minutes=minutes, seconds=seconds) From 986ca8717a918e82d2793c644f3472cf818c079b Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Tue, 25 Mar 2025 17:53:46 -0500 Subject: [PATCH 055/161] Move utils methods --- emtools/jobs/batch_manager.py | 26 +++++++++--------- emtools/metadata/starfile.py | 50 ++++++++++++++++++++++------------- 2 files changed, 44 insertions(+), 32 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 6db5698..a931314 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -63,7 +63,6 @@ def get(self, key, is_path=False): return value - class Batch(dict, FolderManager): """ Subclass from dict with some utilities related to Batch logic. """ def __init__(self, *args, **kwargs): @@ -98,19 +97,16 @@ def error(self): def error(self, value): self.info['error'] = str(value) - def dump(self, obj, fn): - filePath = self.join(fn) - with open(filePath, 'w') as f: - json.dump(obj, f, indent=4) - def dump_info(self): self.dump(self.info, 'info.json') - def dump_all(self): - self.dump(self, 'batch.json') + def dump_all(self, fn=None): + fileName = fn or 'batch.json' + self.dump(self, fileName) - def load_all(self): - with open(self.join('batch.json')) as f: + def load_all(self, fn=None): + filePath = fn or self.join('batch.json') + with open(filePath) as f: self.update(json.load(f)) def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): @@ -167,7 +163,8 @@ class BatchManager: folder. """ def __init__(self, batchSize, inputItemsIterator, workingPath, - itemFileNameFunc=lambda item: item.getFileName()): + itemFileNameFunc=lambda item: item.getFileName(), + createBatch=True): """ Args: batchSize: Number of items that will be grouped into one batch @@ -181,6 +178,7 @@ def __init__(self, batchSize, inputItemsIterator, workingPath, self._batchCount = 0 self._workingPath = workingPath self._itemFileNameFunc = itemFileNameFunc + self._create = createBatch def _createBatchId(self): # We will use batchCount, before the batch is created @@ -198,8 +196,9 @@ def _createBatch(self, items, inputFolder=None, **batchAttrs): path=batch_path, items=items, **batchAttrs) - batch.create() - self._createBatchLinks(batch, items, inputFolder=inputFolder) + if self._create: + batch.create() + self._createBatchLinks(batch, items, inputFolder=inputFolder) return batch def _createBatchLinks(self, batch, items, inputFolder=None): @@ -213,7 +212,6 @@ def _createBatchLinks(self, batch, items, inputFolder=None): baseName = os.path.join(inputFolder, baseName) os.symlink(os.path.abspath(fn), batch.join(baseName)) - def generate(self): """ Generate batches based on the input items. """ items = [] diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index a3609a0..6714b44 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -124,7 +124,11 @@ def getTable(self, tableName, **kwargs): types=None, optional types dict with {columnName: columnType} pairs that allows to specify types for certain columns. """ - self.__createTable(tableName, **kwargs) + try: + self.__createTable(tableName, **kwargs) + except: + return None + if self._singleRow: self._table.addRow(self.__rowFromValues(self._values)) else: @@ -286,6 +290,8 @@ def _findDataLine(self, dataName): break line = f.readline() # Start from the beginning and scann until complete the full loop + if initial_offset == 0: + break f.seek(0) offset = 0 line = f.readline() @@ -448,20 +454,23 @@ def __init__(self, fileName, tableName, rowKeyFunc, **kwargs): def update(self): newRows = [] - now = datetime.now() - mTime = datetime.fromtimestamp(os.path.getmtime(self.fileName)) - - if self.lastCheck is None or mTime > self.lastCheck: - with StarFile(self.fileName) as sf: - for row in sf.iterTable(self._tableName): - rowKey = self._rowKeyFunc(row) - if rowKey not in self._seenItems: - self._seenItems.add(rowKey) - newRows.append(row) - - self.lastCheck = now - if newRows: - self.lastUpdate = now + + if os.path.exists(self.fileName): + now = datetime.now() + mTime = datetime.fromtimestamp(os.path.getmtime(self.fileName)) + + if self.lastCheck is None or mTime > self.lastCheck: + with StarFile(self.fileName) as sf: + for row in sf.iterTable(self._tableName): + rowKey = self._rowKeyFunc(row) + if rowKey not in self._seenItems: + self._seenItems.add(rowKey) + newRows.append(row) + + self.lastCheck = now + if newRows: + self.lastUpdate = now + return newRows def timedOut(self): @@ -527,8 +536,10 @@ def movies_table(**kwargs): @staticmethod def micrograph_table(**kwargs): - extra_cols = kwargs.get('extra_cols', []) - return Table([ + cols = [] + if image_id := kwargs.get('image_id', None): + cols.append(image_id) + cols.extend([ 'rlnMicrographName', 'rlnOpticsGroup', 'rlnCtfImage', @@ -538,7 +549,10 @@ def micrograph_table(**kwargs): 'rlnDefocusAngle', 'rlnCtfFigureOfMerit', 'rlnCtfMaxResolution' - ] + extra_cols) + ]) + if extra_cols := kwargs.get('extra_cols', []): + cols.extend(extra_cols) + return Table(cols) @staticmethod def coordinates_table(**kwargs): From 7ae3bffebdbbcfff72d3f2f7cbc5c955351e88a7 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sun, 13 Apr 2025 16:46:02 -0500 Subject: [PATCH 056/161] Allow additional options to rsync --- emtools/utils/path.py | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 5c4f358..ac6f30d 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -101,22 +101,24 @@ def inSync(dir1, dir2, verbose=False): Use rsync as a subprocess to check if the two directories are synchronized. Both directories must exist. """ - return Path.rsync(dir1, dir2, dry=True, verbose=verbose) == 0 + return Path.rsync(dir1, dir2, '--dry-run', verbose=verbose) == 0 @staticmethod - def rsync(dir1, dir2, dry=False, verbose=False): + def rsync(dir1, dir2, *args, verbose=False): """ Run rsync to synchronize dir1 and dir2 are synchronized (i.e. same content) Use rsync as a subprocess to synchronize dir1 and dir2 and return the number of files transferred. + Args: + dir1: source directory + dir2: destination directory + *args: extra arguments to rsync + verbose: If True, print the command to stdout """ dir1 = Path.addslash(dir1) dir2 = Path.addslash(dir2) - args = ['rsync', '-a', '--stats', dir1, dir2] - if dry: - args.insert(1, '--dry-run') - - p = Process(*args) + cmd = ['rsync'] + list(args) + ['-a', '--stats', dir1, dir2] + p = Process(*cmd) if verbose: p.print(stdout=True) From 64d9633e3252180fef1334e5949e8558b692fd2d Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sun, 13 Apr 2025 16:46:50 -0500 Subject: [PATCH 057/161] Fixed error counting dirs --- emtools/metadata/misc.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 4038a35..7874e0b 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -19,7 +19,7 @@ from datetime import datetime, timedelta from glob import glob -from emtools.utils import Path, Pretty, Process +from emtools.utils import Path, Pretty, Process, Timer class Bins: @@ -135,7 +135,8 @@ def print(self, name): f"\n\ttime: {last_dt}") if self.first and self.last_ts: - print(f"Duration: {(last_dt - first_dt).seconds / 3600:0.2f} hours") + #print(f"Duration: {(last_dt - first_dt).seconds / 3600:0.2f} hours") + print(f"Duration: {Pretty.delta(last_dt - first_dt)}") print(f"Total {name}s: {self.total}, size: {Pretty.size(self.total_size)}") @@ -161,13 +162,18 @@ def __init__(self, filters=[], root=None): def scan(self, folder): """ Scan a folder and register all files recursively. """ + t = Timer() + self.root = Path.addslash(folder) + self._total_dirs = 0 for root, dirs, files in os.walk(folder): for fn in files: self.register(os.path.join(root, fn)) self._total_dirs += len(dirs) + #t.toc("Scanned") + def register(self, filename, stat=None): """ Register a file, if stat is None it will be calculated. """ if stat or os.path.exists(filename): From ec7774e20a7ef68350f53cdcc0e69c2df149386e Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sun, 13 Apr 2025 16:47:25 -0500 Subject: [PATCH 058/161] Fixed import --- emtools/image/thumbnail.py | 1 + 1 file changed, 1 insertion(+) diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index d44c60a..e3fd0ad 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -21,6 +21,7 @@ import tifffile import PIL +from PIL import Image class Thumbnail: From de0262528a04217422d26c20652a99dd198e84f7 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa" Date: Sun, 13 Apr 2025 16:47:37 -0500 Subject: [PATCH 059/161] Added option to count movies --- emtools/scripts/emt_files.py | 47 +++++++++++++++++++++++++++++------- 1 file changed, 38 insertions(+), 9 deletions(-) diff --git a/emtools/scripts/emt_files.py b/emtools/scripts/emt_files.py index 267ee52..f7e92f5 100755 --- a/emtools/scripts/emt_files.py +++ b/emtools/scripts/emt_files.py @@ -31,11 +31,24 @@ def statsDir(folder, sort): df = MovieFiles() df.scan(folder) df.print(sort=sort) - df.counters[1].print('movie') -def timeStats(pattern, bin, plot): - files = glob(pattern) +def countMovies(folder): + m = 0 + for root, dirs, files in os.walk(folder): + for fn in files: + if EPU.is_movie_fn(fn): + m += 1 + return m + + +def timeStats(pattern, bin, plot, data): + files = [] + if os.path.isdir(pattern): + for root, dirs, dfiles in os.walk(pattern): + files.extend(os.path.join(root, fn) for fn in dfiles) + else: + files = glob(pattern) total_size = 0 filesDict = {} @@ -51,6 +64,8 @@ def timeStats(pattern, bin, plot): first = fs[0] last = fs[-1] + to_GB = 1 / (1024 ** 3) + if bin: bindelta = timedelta(minutes=bin) start = datetime.fromtimestamp(first[1]['ts']) @@ -60,7 +75,8 @@ def timeStats(pattern, bin, plot): end = last_bin['end'] ts = datetime.fromtimestamp(v['ts']) if ts <= end: - last_bin['count'] += 1 + value = 1 if not data else v['size'] * to_GB + last_bin['count'] += value else: bins.append({'start': end, 'end': end + bindelta, @@ -117,7 +133,8 @@ def _addDt(b, onlyTime=False): w = width * 0.9 ax.bar(x + w / 2, values, w, label='Men') # Add some text for labels, title and custom x-axis tick labels, etc. - ax.set_ylabel('Files') + ylabel = 'Files' if not data else 'Data (Gb)' + ax.set_ylabel(ylabel) ax.set_title(f'Files generated every {bin} minutes') ax.set_xticks(x) ax.set_xticklabels(labels) @@ -151,9 +168,11 @@ def main(): g = p.add_mutually_exclusive_group() g.add_argument('--stats', '-s', metavar='FOLDER', help="Statistics of the files in a given folder.") - g.add_argument('--timing', metavar='PATTERN', + g.add_argument('--timing', metavar='FOLDER_OR_PATTERN', help="Compute histogram from the timestamps of files " - "matching the pattern.") + "in the folder or matching the pattern.") + g.add_argument('--count_movies', '-m', nargs='+', + help="Count number of movies for each input folder") g.add_argument('--copy_dir', nargs=2, metavar=('SRC_DIR', 'NEW_DIR'), help='Copy directory with some delay') g.add_argument('--check_dirs', nargs=2, metavar=('DIR1', 'DIR2'), @@ -167,7 +186,9 @@ def main(): "(with --timing)") p.add_argument('--plot', '-p', action='store_true', help="Plot the number of files per bin " - "(with --stats)") + "(with --timing)") + p.add_argument('--data', '-a', action='store_true', + help="Use file size for the timing plot") p.add_argument('--delay', '-d', type=float, default=0, help="Delay in seconds when copying files " "(with --copy_dir)") @@ -217,12 +238,20 @@ def _mkdir(d): s = Color.green('in SYNC') if sync else Color.red('NOT in SYNC') print(f"Dirs are {s}") + elif dirs := args.count_movies: + maxlen = max(len(d) for d in dirs) + def _pad(s): + return (maxlen - len(s)) * ' ' + s + + for d in dirs: + print(f"{_pad(d)}: {countMovies(d):>8}") + elif dirs := args.rsync_dirs: n = Path.rsync(dirs[0], dirs[1], verbose=True) print(f"Transferred files: {n}") elif pattern := args.timing: - timeStats(pattern, args.bin, args.plot) + timeStats(pattern, args.bin, args.plot, args.data) # TODO: check from here elif args.transfer: From ce4b2c7e48e36ed3dd93f0a7ca4cdf6a63c20830 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 18 Apr 2025 15:14:33 -0500 Subject: [PATCH 060/161] Added method EPU.count_movies --- emtools/metadata/epu.py | 15 +++++++++++++-- emtools/scripts/emt_files.py | 11 +---------- 2 files changed, 14 insertions(+), 12 deletions(-) diff --git a/emtools/metadata/epu.py b/emtools/metadata/epu.py index 0d99ab5..6a23f65 100644 --- a/emtools/metadata/epu.py +++ b/emtools/metadata/epu.py @@ -41,7 +41,7 @@ def get_acquisition(movieXmlFn): def _pixelSize(k): if not pixelSize: return '' - ps = float(pixelSize[k]['numericValue']) * (10**10) + ps = float(pixelSize[k]['numericValue']) * (10 ** 10) return f'{ps:0.5f}' data = { @@ -57,7 +57,7 @@ def _pixelSize(k): 'ExposureTime': camera['ExposureTime'], 'ReadoutArea': {'height': camera['ReadoutArea']['a:height'], 'width': camera['ReadoutArea']['a:width']} - } + } } return data @@ -130,10 +130,20 @@ def get_movie_xml(fn): return fn.replace(s, '.xml') return '' + @staticmethod + def count_movies(folder): + m = 0 + for root, dirs, files in os.walk(folder): + for fn in files: + if EPU.is_movie_fn(fn) and not os.path.islink(os.path.join(root, fn)): + m += 1 + return m + class Data: """ Class to keep track of EPU files and associated metadata. The information can be read/write from/to a STAR file. """ + def __init__(self, rootFolder, epuStar): self._acq = None self._rootFolder = rootFolder @@ -230,6 +240,7 @@ class Session: Monitor EPU session files and allow to make a copy of GridSquares images and xml files. """ + def __init__(self, inputDir, outputStar=None, backupFolder=None, pl=None): """ Create a new EPU.Session instance. diff --git a/emtools/scripts/emt_files.py b/emtools/scripts/emt_files.py index f7e92f5..83a18e7 100755 --- a/emtools/scripts/emt_files.py +++ b/emtools/scripts/emt_files.py @@ -33,15 +33,6 @@ def statsDir(folder, sort): df.print(sort=sort) -def countMovies(folder): - m = 0 - for root, dirs, files in os.walk(folder): - for fn in files: - if EPU.is_movie_fn(fn): - m += 1 - return m - - def timeStats(pattern, bin, plot, data): files = [] if os.path.isdir(pattern): @@ -244,7 +235,7 @@ def _pad(s): return (maxlen - len(s)) * ' ' + s for d in dirs: - print(f"{_pad(d)}: {countMovies(d):>8}") + print(f"{_pad(d)}: {EPU.count_movies(d):>8}") elif dirs := args.rsync_dirs: n = Path.rsync(dirs[0], dirs[1], verbose=True) From 2f872da943f48064797fc88225d2b415b118ae29 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 5 May 2025 12:11:30 -0500 Subject: [PATCH 061/161] Make Process fail for Rsync and allow to check transfer size --- emtools/utils/path.py | 27 ++++++++++++++++++--------- 1 file changed, 18 insertions(+), 9 deletions(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index ac6f30d..c6fd78b 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -104,7 +104,9 @@ def inSync(dir1, dir2, verbose=False): return Path.rsync(dir1, dir2, '--dry-run', verbose=verbose) == 0 @staticmethod - def rsync(dir1, dir2, *args, verbose=False): + def rsync(dir1, dir2, *args, + verbose=False, + size=False): """ Run rsync to synchronize dir1 and dir2 are synchronized (i.e. same content) Use rsync as a subprocess to synchronize dir1 and dir2 and return the number of files transferred. @@ -113,25 +115,32 @@ def rsync(dir1, dir2, *args, verbose=False): dir2: destination directory *args: extra arguments to rsync verbose: If True, print the command to stdout + size: If True, a tuple is returned with transferred files and transferred data size """ dir1 = Path.addslash(dir1) dir2 = Path.addslash(dir2) cmd = ['rsync'] + list(args) + ['-a', '--stats', dir1, dir2] - p = Process(*cmd) + p = Process(*cmd, doRaise=True) if verbose: p.print(stdout=True) - transf = 1 + def _value(line): + # Get the value after the colon (:) + # and remove , that is used to separate thousands + return int(line.split(':')[1].replace(',', '')) + + transf = 0 + transfSize = 0 + for line in p.lines(): - if 'files transferred:' in line: - value = line.split(':')[1] - # Remove , that is used to separate thousands - transf = int(value.replace(',', '')) - break + if 'Number of regular files transferred:' in line: + transf = _value(line) + elif 'Total transferred file size:' in line: + transfSize = _value(line.replace('bytes', '')) - return transf + return (transf, transfSize) if size else transf @staticmethod From 34ebd7ba7ec58f4fbb97df4cb2780340d2a25f39 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 15 May 2025 11:38:52 -0500 Subject: [PATCH 062/161] Added duplicates check to emt-star and rename function to FolderManager --- emtools/scripts/emt_star.py | 18 ++++++++++++++++++ emtools/utils/path.py | 3 +++ 2 files changed, 21 insertions(+) diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py index 19420c0..fc1e8a6 100755 --- a/emtools/scripts/emt_star.py +++ b/emtools/scripts/emt_star.py @@ -50,6 +50,18 @@ def groupBy(starFile, table, column): print(k, v) +def checkDuplicates(inputStar, table, column): + items = set() + total = 0 + + with StarFile(inputStar) as sf: + for row in sf.iterTable(table): + items.add(row.get(column)) + total += 1 + + print(f">>> Duplicates: {total - len(items)}") + + def splitBy(starFile, column, minSize): with StarFile(starFile) as sf: tOptics = sf.getTable('optics') @@ -98,6 +110,9 @@ def main(): help="Count rows grouped by a given label") p.add_argument('--split_particles', '-s', nargs='+', metavar=('COLUMN', 'minsize'), help="Split input particles by some column") + p.add_argument('--duplicates', '-d', nargs=2, + metavar=('TABLE', 'COLUMN'), + help="Check duplicates values for a given label") args = p.parse_args() inputStar = args.input @@ -109,6 +124,9 @@ def main(): column = split[0] minSize = split[1] if len(split) > 1 else 0 splitBy(inputStar, column, minSize) + elif args.duplicates: + table, column = args.duplicates + checkDuplicates(inputStar, table, column) else: printStarInfo(args.input) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index c6fd78b..432399d 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -317,3 +317,6 @@ def dump(self, obj, fn): with open(filePath, 'w') as f: json.dump(obj, f, indent=4) + def rename(self, oldFn, newFn): + os.rename(self.join(oldFn), self.join(newFn)) + From e6b2e897f611e39968352ebe3c948a6ea77932d2 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 15 May 2025 12:37:15 -0500 Subject: [PATCH 063/161] Print all duplicates --- emtools/scripts/emt_star.py | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py index fc1e8a6..c016f6f 100755 --- a/emtools/scripts/emt_star.py +++ b/emtools/scripts/emt_star.py @@ -52,14 +52,18 @@ def groupBy(starFile, table, column): def checkDuplicates(inputStar, table, column): items = set() - total = 0 + duplicates = [] with StarFile(inputStar) as sf: for row in sf.iterTable(table): - items.add(row.get(column)) - total += 1 - - print(f">>> Duplicates: {total - len(items)}") + value = row.get(column) + if value in items: + duplicates.append(value) + else: + items.add(value) + + print(f">>> Duplicates: {len(duplicates)}\n" + f" {duplicates}") def splitBy(starFile, column, minSize): From 5274a162d1e868e7df96c1784d7f9b07fccb5430 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 20 May 2025 15:11:11 -0500 Subject: [PATCH 064/161] Added helper methods to determine filetype based on extension --- emtools/utils/path.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 432399d..243981f 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -30,6 +30,9 @@ GLOB_CHARS = ['*', '?', '[', ']'] +IMAGE_EXT = ['tiff', 'tif', 'png', 'jpg', 'jpeg'] +TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv'] + class Path: """ @@ -262,6 +265,14 @@ def exists(path): def isPattern(path): return any(c in path for c in GLOB_CHARS) + @staticmethod + def isImage(path): + return Path.getExt(path).lower()[1:] in IMAGE_EXT + + @staticmethod + def isText(path): + return Path.getExt(path).lower()[1:] in TEXT_EXT + class FolderManager: """ Helper class with some path utilities from a given path. """ From b11656d66d0bfbcf2627cbae4ee26da4ef7c84ef Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 20 Jun 2025 15:38:43 -0500 Subject: [PATCH 065/161] Allow dict as input --- emtools/metadata/starfile.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 6714b44..c5d2ac8 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -371,6 +371,9 @@ def writeRowValues(self, values): """ Write to file a line for these row values. Order should be ensured that is the same of the expected columns. """ + if isinstance(values, dict): + values = values.values() + if not self._format: self._computeLineFormat([values]) From 33309cbf2edfb27c679d8e78d3ab7a8d141d75b9 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 20 Jun 2025 15:39:40 -0500 Subject: [PATCH 066/161] Implemented TS batch manager from star input --- emtools/jobs/__init__.py | 8 +++++--- emtools/jobs/batch_manager.py | 29 +++++++++++++++++++++++++++-- emtools/jobs/workflow.py | 3 +++ 3 files changed, 35 insertions(+), 5 deletions(-) diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index 8ae82d2..88b1afd 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -15,8 +15,10 @@ # ************************************************************************** from .pipeline import Pipeline -from .batch_manager import Args, Batch, BatchManager, MdocBatchManager, Vars +from .batch_manager import (Args, Batch, Vars, + BatchManager, MdocBatchManager, TsStarBatchManager) from .workflow import Workflow -__all__ = ["Pipeline", "BatchManager", "Batch", "Args", "Vars", - "Workflow", "MdocBatchManager"] +__all__ = ["Pipeline", "Workflow", + "Args", "Vars", + "Batch", "BatchManager", "MdocBatchManager", "TsStarBatchManager"] diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index a931314..833c56d 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -23,7 +23,7 @@ from contextlib import contextmanager from emtools.utils import Color, FolderManager, Timer, Pretty, Path -from emtools.metadata import Mdoc +from emtools.metadata import Mdoc, StarFile class Args(dict): @@ -270,4 +270,29 @@ def _absfn(item): for item in items: baseName = os.path.basename(_absfn(item)) - os.symlink(os.path.join('frames', baseName), batch.join(baseName)) \ No newline at end of file + os.symlink(os.path.join('frames', baseName), batch.join(baseName)) + + +class TsStarBatchManager(BatchManager): + """ + Batch manager from a Relion tilt_series.star file. + (e.g. after the TS import job) + """ + + def __init__(self, tsIterator, workingPath): + """ + Args: + tsIterator: input tilt-series iterator + workingPath: path where the batches folder will be created + """ + BatchManager.__init__(self, 0, tsIterator, workingPath, + itemFileNameFunc=lambda item: item.rlnMicrographMovieName) + self._create = False # Do not create batch folder until processing + + def generate(self): + """ Generate batches based on the input items. """ + for tsRow in self._items: + tsName = tsRow.rlnTomoName + with StarFile(tsRow.rlnTomoTiltSeriesStarFile) as sf: + items = [row._asdict() for row in sf.iterTable(tsName)] + yield self._createBatch(items, tsName=tsName) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index 1762c02..bb66938 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -36,6 +36,9 @@ def root(self): if not j.inputs: yield j + def hasJob(self, jobId): + return jobId in self._jobs + def getJob(self, jobId): return self._jobs[jobId] From 3e5f37a9290db3ffc16706fb24d0e2b4a32f22a0 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 30 Jun 2025 09:34:16 -0500 Subject: [PATCH 067/161] Added utility functions to Batch and FolderManager --- emtools/jobs/batch_manager.py | 14 ++++++++------ emtools/utils/path.py | 4 ++++ 2 files changed, 12 insertions(+), 6 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 833c56d..197cbb1 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -33,7 +33,9 @@ def toList(self): args = [] for k, v in self.items(): args.append(str(k)) - if v != '': + if isinstance(v, list): + args.extend(str(e) for e in v) + elif v != '': args.append((str(v))) return args @@ -138,15 +140,15 @@ def tic(self, prefix=''): def toc(self): self.info.update({ - f'{self._timerPrefix}start': self._timer.getTic(), - f'{self._timerPrefix}end': Pretty.now(), - f'{self._timerPrefix}elapsed': str(self._timer.getElapsedTime()) + f'{self._timerPrefix}_start': self._timer.getTic(), + f'{self._timerPrefix}_end': Pretty.now(), + f'{self._timerPrefix}_elapsed': str(self._timer.getElapsedTime()) }) @contextmanager - def execute(self): + def execute(self, prefix=''): try: - self.tic() + self.tic(prefix=prefix) yield self except Exception as e: self.error = traceback.format_exc() diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 243981f..633ccfd 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -19,6 +19,7 @@ import time import tempfile import json +from glob import glob from datetime import datetime as dt from collections import OrderedDict from contextlib import contextmanager @@ -323,6 +324,9 @@ def listdir(self): """ Return files relative to the path. """ return os.listdir(self.path) + def glob(self, pattern): + return glob(self.join(pattern)) + def dump(self, obj, fn): filePath = self.join(fn) with open(filePath, 'w') as f: From fd46e50f9be4576716cb92920fe1542b127082af Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 11 Jul 2025 21:04:36 -0500 Subject: [PATCH 068/161] working on a gpu monitor tool --- emtools/utils/__init__.py | 4 +-- emtools/utils/path.py | 6 ++++ emtools/utils/system.py | 63 +++++++++++++++++++++++++++++++++++++++ 3 files changed, 71 insertions(+), 2 deletions(-) diff --git a/emtools/utils/__init__.py b/emtools/utils/__init__.py index c5869a3..e3c1f3d 100644 --- a/emtools/utils/__init__.py +++ b/emtools/utils/__init__.py @@ -20,11 +20,11 @@ from .process import Process from .path import Path, FolderManager -from .system import System +from .system import System, GpuMonitor from .server import JsonTCPServer, JsonTCPClient __all__ = ["Color", "Pretty", "Timer", "Process", "Path", "FolderManager", - "System", "JsonTCPServer", "JsonTCPClient"] + "System", "JsonTCPServer", "JsonTCPClient", "GpuMonitor"] diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 633ccfd..09c5a68 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -335,3 +335,9 @@ def dump(self, obj, fn): def rename(self, oldFn, newFn): os.rename(self.join(oldFn), self.join(newFn)) + def link(self, fn, absolute=False): + """ Link a file inside the folder and return the basename. """ + base = os.path.basename(fn) + src = os.path.abspath(fn) if absolute else self.relpath(fn) + os.symlink(src, self.join(base)) + return base diff --git a/emtools/utils/system.py b/emtools/utils/system.py index 8f54c5d..f907e0c 100644 --- a/emtools/utils/system.py +++ b/emtools/utils/system.py @@ -23,6 +23,10 @@ import socket import platform import psutil +import time +import json +import threading +from datetime import datetime from .process import Process @@ -97,3 +101,62 @@ def specs(): def hostname(): """ Return the hostname. """ return socket.gethostname() + + +class GpuMonitor(threading.Thread): + """ Monitor GPU utilization. + Keeps an internal record of utilization data points, indexed by time. """ + + def __init__(self): + super().__init__() + self._stopEvent = threading.Event() + self._data = { + "sample": System.gpus(), + "columns": ["timestamp", ["temperature.gpu", + "utilization.gpu", + "utilization.memory"]], + "rows": [] + } + self.sleep = 1 + self.outputLog = 'gpu_monitor.json' + + def sample(self, verbose=False): + now = datetime.now() + gpus = System.gpus() + gpuLine = f'\r{now} ' + row = [str(now), []] + gpuEntries = {} + for gpuDict in sorted(gpus, key=lambda r: r['index']): + i = gpuDict['index'] + ugpu = gpuDict["utilization.gpu"].split()[0] # Remove % character + umem = gpuDict["utilization.memory"].split()[0] + gpuStr = f'{i}: gpu {ugpu}, mem {umem}' + gpuLine += f"{gpuStr:<30}" + gpuEntries[i] = [ugpu, umem] + if verbose: + print(gpuLine, end="") + self._data['rows'].append([str(now), gpuEntries]) + + def monitor(self, outputLog=None): + if outputLog: + self.outputLog = outputLog + c = 0 + while not self._stopEvent.is_set(): + self.sample() + c += 1 + if self.outputLog and c % 10 == 1: + with open(self.outputLog, 'w') as f: + json.dump(self._data, f) + c = 0 + + time.sleep(self.sleep) + + def run(self): + self.monitor() + + def stop(self): + """ Stop the current thread. """ + self._stopEvent.set() + self.join() + + From 6e0873196ee14e453b8cd2deb68eb612ce1d9f12 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 16 Jul 2025 15:44:48 -0500 Subject: [PATCH 069/161] Added WarpXml helper class --- emtools/metadata/__init__.py | 5 +++-- emtools/metadata/misc.py | 17 +++++++++++++++++ 2 files changed, 20 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/__init__.py b/emtools/metadata/__init__.py index bda4fa2..d5bf226 100644 --- a/emtools/metadata/__init__.py +++ b/emtools/metadata/__init__.py @@ -17,11 +17,12 @@ from .table import Column, ColumnList, Table from .starfile import StarFile, StarMonitor, RelionStar from .epu import EPU -from .misc import Bins, TsBins, DataFiles, MovieFiles, Mdoc, TextFile, Acquisition +from .misc import (Bins, TsBins, DataFiles, MovieFiles, + Mdoc, TextFile, Acquisition, WarpXml) from .sqlite import SqliteFile __all__ = ["Column", "ColumnList", "Table", "StarFile", "StarMonitor", "RelionStar", "EPU", "Bins", "TsBins", "SqliteFile", "DataFiles", "MovieFiles", - "Mdoc", "TextFile", "Acquisition"] + "Mdoc", "TextFile", "Acquisition", "WarpXml"] diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 7874e0b..85ac0ef 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -18,6 +18,7 @@ import pathlib from datetime import datetime, timedelta from glob import glob +import xmltodict from emtools.utils import Path, Pretty, Process, Timer @@ -325,6 +326,22 @@ def stripLines(fn, **kwargs): yield line +class WarpXml: + """ Helper class to read Warp's XML files. """ + def __init__(self, xmlPath): + with open(xmlPath) as f: + self._data = xmltodict.parse(f.read()) + + def getDict(self, *keys): + """ Navigate the provided keys and get a dict from Name=Value pairs. + """ + d = self._data + for k in keys: + d = d[k] + + return {e['@Name']: e['@Value'] for e in d} + + class Acquisition(dict): """ Subclass from dict with some utilities related to Acquisition. """ From b55f6f0096e6423768bd1d9611752b8c703eeeb7 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 22 Jul 2025 15:33:24 -0500 Subject: [PATCH 070/161] Added Batch.copy method --- emtools/utils/path.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 09c5a68..263f41b 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -341,3 +341,9 @@ def link(self, fn, absolute=False): src = os.path.abspath(fn) if absolute else self.relpath(fn) os.symlink(src, self.join(base)) return base + + def copy(self, *paths): + """ Copy one or many files into the path. """ + for p in paths: + shutil.copy(p, self.__path) + From b5f3db28847fdb952bf9101c178f53ceb306808b Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 24 Jul 2025 15:05:44 -0500 Subject: [PATCH 071/161] added dose properties to Acquisition class --- emtools/metadata/misc.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 85ac0ef..28c8c60 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -376,3 +376,19 @@ def amplitude_contrast(self): @amplitude_contrast.setter def amplitude_contrast(self, value): self['amplitude_contrast'] = float(value) + + @property + def dose(self): + return float(self.get('dose', 0.0)) + + @dose.setter + def dose(self, value): + self['dose'] = float(value) + + @property + def total_dose(self): + return float(self.get('total_dose', 0.0)) + + @total_dose.setter + def total_dose(self, value): + self['total_dose'] = float(value) From 17dfd1819dde7da5ec76090bd60d50188eb62098 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 7 Aug 2025 10:30:58 -0500 Subject: [PATCH 072/161] allow string arguments --- emtools/jobs/batch_manager.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 197cbb1..03dd0fb 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -15,11 +15,12 @@ # ************************************************************************** import os -from uuid import uuid4 -from datetime import datetime import json import subprocess import traceback +import shlex +from uuid import uuid4 +from datetime import datetime from contextlib import contextmanager from emtools.utils import Color, FolderManager, Timer, Pretty, Path @@ -119,6 +120,8 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): args = Args(kwargs).toList() elif isinstance(kwargs, list): args = list(kwargs) + elif isinstance(kwargs, str): + args = shlex.split(kwargs) else: raise Exception("Expecting dict or list as arguments") From 1e5613465160e258c458f0306d25ba81cc80b532 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 22 Aug 2025 08:44:33 -0500 Subject: [PATCH 073/161] emt-ps to catch program name in args --- emtools/utils/process.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/emtools/utils/process.py b/emtools/utils/process.py index 3f43cf8..78a7bdd 100644 --- a/emtools/utils/process.py +++ b/emtools/utils/process.py @@ -101,7 +101,7 @@ def _addProc(f, proc): def _filter_name(proc): if program and program not in proc.info['name']: cmdline = proc.cmdline() - if len(cmdline) == 0 or program not in cmdline[0]: + if len(cmdline) == 0 or all(program not in cmd for cmd in cmdline): return False return True From 285e839f91a93d7381a73445ce2a8a18467c0f99 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sat, 23 Aug 2025 10:59:14 -0500 Subject: [PATCH 074/161] Improved Mdocs batch manager --- emtools/jobs/batch_manager.py | 84 ++++++++++++++++++++++++++++------- 1 file changed, 69 insertions(+), 15 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 03dd0fb..7e91678 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -19,8 +19,10 @@ import subprocess import traceback import shlex +import time +from glob import glob from uuid import uuid4 -from datetime import datetime +from datetime import datetime, timedelta from contextlib import contextmanager from emtools.utils import Color, FolderManager, Timer, Pretty, Path @@ -178,7 +180,7 @@ def __init__(self, batchSize, inputItemsIterator, workingPath, itemFileNameFunc: function to extract a filename from each item (by default: lambda item: item.getFileName()) """ - self._items = inputItemsIterator + self._itemsIterator = inputItemsIterator self._batchSize = batchSize self._batchCount = 0 self._workingPath = workingPath @@ -221,7 +223,7 @@ def generate(self): """ Generate batches based on the input items. """ items = [] - for item in self._items: + for item in self._itemsIterator: items.append(item) if len(items) == self._batchSize: @@ -235,32 +237,84 @@ def generate(self): class MdocBatchManager(BatchManager): """ Batch manager for Tilt-series. """ - def __init__(self, tsIterator, workingPath, suffix=None, movies=None): + def __init__(self, mdocsPattern, workingPath, + moviesPath=None, **kwargs): """ Args: - tsIterator: input tilt-series iterator + mdocsPattern: input pattern of Mdocs files workingPath: path where the batches folder will be created - suffix: suffix to be removed from mdoc filename to generate - the tilt-series name + moviesPath: path where the frames pointed by Mdocs are + + Kwargs: + wait: waiting time in seconds to check for new files + timeout: time in seconds to quit after no new files found + blacklist: container of tsName that have been processed or want + to be avoided """ - BatchManager.__init__(self, 0, tsIterator, workingPath, + if not glob(mdocsPattern): + raise Exception(f"No mdoc files were found with pattern: {mdocsPattern}") + + BatchManager.__init__(self, 0, self._iterMdocs(mdocsPattern), workingPath, itemFileNameFunc=lambda item: item[1]['SubFramePath']) - self._suffix = suffix - self._movies = movies + self._moviesPath = moviesPath + self._wait = kwargs.get('wait', 60) + self._timeout = timedelta(seconds=kwargs.get('timeout', 3600)) + self._blacklist = set(kwargs.get('blacklist', [])) + + def _iterMdocs(self, mdocsPattern): + """ Iterate over a provided Mdocs pattern. """ + one_min = timedelta(minutes=1) + + def _newMdoc(now, fn): + """ Return True if the file meets the following two conditions: + - It has not been processed (in blacklist) + - Modification time is more than 1 minute. + """ + tsName = self._tsName(fn) + if tsName not in self._blacklist: + s = os.stat(fn) + dt = datetime.fromtimestamp(s.st_mtime) + # Ignore also sessions that have not been updated for + # more than X days or that have not been modified since last check + if now - dt > one_min: + self._blacklist.add(tsName) + return True + return False + + last_found = datetime.now() + now = datetime.now() + + def _print(msg): + print(f"INPUT MDOCS: {Pretty.now()}: {msg}", flush=True) + + while now - last_found < self._timeout: + _print("Checking for new mdocs") + if new_mdocs := [fn for fn in glob(mdocsPattern) if _newMdoc(now, fn)]: + _print(f"New mdocs found: {str(new_mdocs)}") + for mdocFn in new_mdocs: + mdoc = Mdoc.parse(mdocFn) + mdoc['MdocFile'] = {'Path': mdocFn} + yield mdoc + last_found = now + else: + _print("No new Mdocs found, sleeping.") + + time.sleep(self._wait) + now = datetime.now() def _subframePath(self, mdocFn, section): - movieFolder = self._movies or os.path.dirname(mdocFn) + movieFolder = self._moviesPath or os.path.dirname(mdocFn) return os.path.join(movieFolder, Mdoc.getSubFrameBase(section)) def _tsName(self, mdocFn): name = Path.removeBaseExt(mdocFn) - if self._suffix: - name = name.replace(self._suffix, '') + # if self._suffix: + # name = name.replace(self._suffix, '') return name def generate(self): """ Generate batches based on the input items. """ - for mdoc in self._items: + for mdoc in self._itemsIterator: mdocFn = mdoc['MdocFile']['Path'] yield self._createBatch(mdoc.zvalues, mdoc=mdoc, tsName=self._tsName(mdocFn)) @@ -296,7 +350,7 @@ def __init__(self, tsIterator, workingPath): def generate(self): """ Generate batches based on the input items. """ - for tsRow in self._items: + for tsRow in self._itemsIterator: tsName = tsRow.rlnTomoName with StarFile(tsRow.rlnTomoTiltSeriesStarFile) as sf: items = [row._asdict() for row in sf.iterTable(tsName)] From b4a940c22c95da677f73e41cec2059f12ddfc9e3 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 25 Aug 2025 10:52:07 -0500 Subject: [PATCH 075/161] Added debug shortcut --- emtools/utils/process.py | 1 + 1 file changed, 1 insertion(+) diff --git a/emtools/utils/process.py b/emtools/utils/process.py index 78a7bdd..2c58979 100644 --- a/emtools/utils/process.py +++ b/emtools/utils/process.py @@ -173,6 +173,7 @@ def __init__(self, logger=None, only_log=False, # Shortcuts self.logger = logger self.info = logger.info + self.debug = logger.debug self.error = logger.error self.warning = logger.warning From bcda1934c05cc44c5658361d17f81dc24bd83bab Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 2 Sep 2025 09:01:51 -0500 Subject: [PATCH 076/161] Allow to pass basename to link --- emtools/utils/path.py | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 263f41b..d135260 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -335,9 +335,11 @@ def dump(self, obj, fn): def rename(self, oldFn, newFn): os.rename(self.join(oldFn), self.join(newFn)) - def link(self, fn, absolute=False): - """ Link a file inside the folder and return the basename. """ - base = os.path.basename(fn) + def link(self, fn, absolute=False, name=None): + """ Link a file inside the folder and return the basename. + If name is None, the basename of the fn will be used. + """ + base = name or os.path.basename(fn) src = os.path.abspath(fn) if absolute else self.relpath(fn) os.symlink(src, self.join(base)) return base From 8a956ce8acd06c5f17c7276a85ef6f76d8a5a776 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 4 Sep 2025 09:41:58 -0500 Subject: [PATCH 077/161] Fix when there is no last file --- emtools/utils/path.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index d135260..e6d5b4f 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -159,7 +159,10 @@ def lastModified(folder): t = (f, s.st_mtime) last = t if not last or s.st_mtime > last[1] else last - return last[0], dt.fromtimestamp(last[1]) + if last: + return last[0], dt.fromtimestamp(last[1]) + else: + return None, None @staticmethod def copyFile(file1, file2, sleep=0): From ba7cee0a7939e2e4dc8238ffd8b8fa17b063e8b9 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 30 Sep 2025 10:55:37 -0500 Subject: [PATCH 078/161] Some utils --- emtools/jobs/batch_manager.py | 21 +++++++++++++++++++++ emtools/utils/pretty.py | 5 +++++ 2 files changed, 26 insertions(+) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 7e91678..3269c2a 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -45,6 +45,27 @@ def toList(self): def toLine(self): return ' '.join("%s %s" % (k, v) for k, v in self.items()) + @staticmethod + def fromList(iterable): + args = Args() + for p in iterable: + if p.startswith('--'): + last_key = p + args[p] = '' + else: + v = args[last_key] + + if v: + if isinstance(v, list): + v.append(p) + else: + v = [v, p] + else: + v = p + args[last_key] = v + + return args + class Vars: """ Handle variable definitions, either from input dict diff --git a/emtools/utils/pretty.py b/emtools/utils/pretty.py index 0fb248b..a32f3b6 100644 --- a/emtools/utils/pretty.py +++ b/emtools/utils/pretty.py @@ -129,4 +129,9 @@ def _plural(div, noun): return _plural(365, 'year') + @staticmethod + def dprint(msg): + """ DEBUG print with timestamp and flush. """ + print(f"{Pretty.now()}: >>> DEBUG: {msg}", flush=True) + From 957449efdfeddcf80f1829a1983be7c26729df1a Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 4 Nov 2025 22:57:52 -0600 Subject: [PATCH 079/161] Add rsync extra args at the end --- emtools/utils/path.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index e6d5b4f..526054e 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -124,7 +124,7 @@ def rsync(dir1, dir2, *args, dir1 = Path.addslash(dir1) dir2 = Path.addslash(dir2) - cmd = ['rsync'] + list(args) + ['-a', '--stats', dir1, dir2] + cmd = ['rsync', '-a', '--stats'] + list(args) + [dir1, dir2] p = Process(*cmd, doRaise=True) if verbose: From e5038df47147a6f11d757bf3c61601105d4d32b3 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sun, 9 Nov 2025 22:44:12 -0600 Subject: [PATCH 080/161] Fixed bug about new option computeFormat when writing StarFile tables --- emtools/metadata/starfile.py | 32 ++++++++++++++++++++++++++++---- 1 file changed, 28 insertions(+), 4 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index c5d2ac8..70a67c4 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -389,7 +389,7 @@ def writeRow(self, row): def _writeNewline(self): self._file.write('\n') - def _computeLineFormat(self, valuesList): + def _computeLineFormat(self, valuesList, all=False): """ Compute format base on row values width. """ # Take a hint for the columns width from the first row widths = [len(_formatValue(v)) for v in valuesList[0]] @@ -398,7 +398,8 @@ def _computeLineFormat(self, valuesList): if n > 1: # Check middle and last row, just in case ;) - for index in [n // 2, -1]: + indexes = list(range(len(valuesList))) if all else [n // 2, -1] + for index in indexes: for i, v in enumerate(valuesList[index]): w = len(_formatValue(v)) if w > widths[i]: @@ -407,19 +408,24 @@ def _computeLineFormat(self, valuesList): self._format = " ".join("{:>%d%s} " % (w + 1, f) for w, f in zip(widths, formats)) + '\n' - def writeTable(self, tableName, table, singleRow=False): + def writeTable(self, tableName, table, singleRow=False, computeFormat=False): """ Write a Table in Star format to the given file. Args: tableName: The name of the table to write. table: Table that is going to be written singleRow: If True, don't write *loop\_*, just label/value pairs. + computeFormat: compute format based on widest first column, + just for aesthetics and not recommended for large tables """ if table.size(): if singleRow: self.writeSingleRow(tableName, table[0]) else: self.writeHeader(tableName, table) + if computeFormat: + valuesList = [row._asdict().values() for row in table] + self._computeLineFormat(valuesList, all=True) for row in table: self.writeRow(row) @@ -511,6 +517,24 @@ class RelionStar: JOB_INDEX = re.compile('job(\d{3})') + @staticmethod + def to_bool(strValue): + """ Convert Relion Yes/No to True/False. """ + if strValue == 'Yes': + return True + elif strValue == 'False': + return False + else: + raise Exception(f"Invalid Relion bool value: {strValue}") + + @staticmethod + def from_bool(boolValue): + """ Return Yes or No string from True/False. """ + if not isinstance(boolValue): + raise Exception("Expecting bool value for Yes/No conversion") + + return 'Yes' if boolValue else 'No' + @staticmethod def optics_table(acq, opticsGroup=1, opticsGroupName="opticsGroup1", mtf=None, originalPixelSize=None): @@ -664,7 +688,7 @@ def write_pipeline(pipeline_star, jobCounter=1, tables=None): if tables: for name, t in tables.items(): if len(t): - sf.writeTable(f"pipeline_{name}", t) + sf.writeTable(f"pipeline_{name}", t, computeFormat=True) @staticmethod def job_index(jobId): From bb00d65eb9339c3f41d28c5fec4089a4be80b502 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 11 Nov 2025 09:58:19 -0600 Subject: [PATCH 081/161] Method to clear a folder --- emtools/utils/path.py | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 526054e..e9c5c60 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -307,6 +307,10 @@ def path(self): def path(self, value): self.__path = value + def clear(self): + """ Remove existing path. """ + Process.system(f"rm -rf '{self.path}'") + def create(self, **kwargs): """ Create batch folder. """ self.log(f"Creating folder: {self.path}") From 40e240d380060cc13fadc04e409bb295729bbfab Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 11 Nov 2025 09:58:55 -0600 Subject: [PATCH 082/161] Added methods to read/write job.star and fixed bug in pipeline.star --- emtools/metadata/starfile.py | 51 +++++++++++++++++++++++++++--------- 1 file changed, 38 insertions(+), 13 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 70a67c4..9d5ec18 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -535,6 +535,36 @@ def from_bool(boolValue): return 'Yes' if boolValue else 'No' + @staticmethod + def read_jobstar(jobStarFile): + tValues = StarFile.getTableFromFile(jobStarFile, + 'joboptions_values', + guessType=False) + + def _val(v): + if v == 'Yes': + return True + elif v == 'No': + return False + else: + return v + + return {row.rlnJobOptionVariable: _val(row.rlnJobOptionValue) for row in tValues} + + @staticmethod + def write_jobstar(jobType, values, jobStarFile, isTomo=0, isContinue=0): + """ Convert params dict to a Relion job.star file. """ + with StarFile(jobStarFile, 'w') as sfOut: + tJob = Table(['rlnJobTypeLabel', 'rlnJobIsContinue', 'rlnJobIsTomo']) + tJob.addRowValues(jobType, isContinue, isTomo) # FIXME check continue and isTomo + sfOut.writeTimeStamp() + sfOut.writeTable('job', tJob, singleRow=True) + tValues = Table(['rlnJobOptionVariable', 'rlnJobOptionValue']) + for k, v in values.items(): + val = ('Yes' if v else 'No') if isinstance(v, bool) else v + tValues.addRowValues(k, val) + sfOut.writeTable('joboptions_values', tValues, computeFormat=True) + @staticmethod def optics_table(acq, opticsGroup=1, opticsGroupName="opticsGroup1", mtf=None, originalPixelSize=None): @@ -673,8 +703,8 @@ def pipeline_tables(): 'rlnPipeLineNodeTypeLabelDepth']), 'output_edges': Table(['rlnPipeLineEdgeProcess', 'rlnPipeLineEdgeToNode']), - 'intput_edges': Table(['rlnPipeLineEdgeFromNode', - 'rlnPipeLineEdgeProcess']) + 'input_edges': Table(['rlnPipeLineEdgeFromNode', + 'rlnPipeLineEdgeProcess']) } @staticmethod @@ -752,6 +782,7 @@ def workflow_to_pipeline(wf, pipelineStar): tProc = tables['processes'] tNodes = tables['nodes'] tOutput = tables['output_edges'] + tInput = tables['input_edges'] for job in wf.jobs(): tProc.addRowValues( @@ -760,6 +791,11 @@ def workflow_to_pipeline(wf, pipelineStar): rlnPipeLineProcessStatusLabel=job['status'], rlnPipeLineProcessTypeLabel=job['jobtype'] ) + for i in job.inputs: + tInput.addRowValues( + rlnPipeLineEdgeProcess=job.id, + rlnPipeLineEdgeFromNode=i.id + ) for o in job.outputs: tNodes.addRowValues( @@ -772,15 +808,4 @@ def workflow_to_pipeline(wf, pipelineStar): rlnPipeLineEdgeToNode=o.id ) - # if tOutput := _table('output_edges'): - # for row in tOutput: - # job = wf.getJob(row.rlnPipeLineEdgeProcess) - # nodeName = row.rlnPipeLineEdgeToNode - # job.registerOutput(nodeName, datatype=nodes[nodeName]) - # - # if tInput := _table('input_edges'): - # for row in tInput: - # job = wf.getJob(row.rlnPipeLineEdgeProcess) - # job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) - RelionStar.write_pipeline(pipelineStar, wf.jobNextIndex, tables) From ee4271494ea93df7612c08c01859518c4875b350 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 11 Nov 2025 09:59:37 -0600 Subject: [PATCH 083/161] Allow default to Workflow.getJob and added deleteJob --- emtools/jobs/workflow.py | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index bb66938..bbf5a1f 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -39,8 +39,8 @@ def root(self): def hasJob(self, jobId): return jobId in self._jobs - def getJob(self, jobId): - return self._jobs[jobId] + def getJob(self, jobId, default=None): + return self._jobs.get(jobId, default) def getData(self, dataId): return self.data[dataId] @@ -53,6 +53,11 @@ def registerJob(self, jobId, inputs=None, **kwargs): self._jobs[jobId] = job return job + def deleteJob(self, job): + for o in job.outputs: + del self.data[o.id] + del self._jobs[job.id] + def dot(self): """ Print the workflow to the terminal. """ dot = 'digraph G {\n compound=true;\n' From 3084574cb2070f2d6113bdb22797cc67837f2b34 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 12 Nov 2025 16:58:26 -0600 Subject: [PATCH 084/161] Allow left alignment in format and fixed boolean values --- emtools/metadata/starfile.py | 24 +++++++++++++----------- 1 file changed, 13 insertions(+), 11 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 9d5ec18..c666225 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -138,7 +138,7 @@ def getTable(self, tableName, **kwargs): return self._table @staticmethod - def getTableFromFile(starFileName, tableName, **kwargs): + def getTableFromFile(tableName, starFileName, **kwargs): """ Shortcut to read a table from file. **kwargs are the same expected by getTable function. """ @@ -389,23 +389,26 @@ def writeRow(self, row): def _writeNewline(self): self._file.write('\n') - def _computeLineFormat(self, valuesList, all=False): + def _computeLineFormat(self, valuesList, computeFormat=False): """ Compute format base on row values width. """ # Take a hint for the columns width from the first row widths = [len(_formatValue(v)) for v in valuesList[0]] formats = [_getFormatStr(v) for v in valuesList[0]] n = len(valuesList) + a = '>' if n > 1: # Check middle and last row, just in case ;) - indexes = list(range(len(valuesList))) if all else [n // 2, -1] + indexes = list(range(len(valuesList))) if computeFormat else [n // 2, -1] + if computeFormat == 'left': + a = '<' for index in indexes: for i, v in enumerate(valuesList[index]): w = len(_formatValue(v)) if w > widths[i]: widths[i] = w - self._format = " ".join("{:>%d%s} " % (w + 1, f) + self._format = " ".join("{:%s%d%s} " % (a, w + 1, f) for w, f in zip(widths, formats)) + '\n' def writeTable(self, tableName, table, singleRow=False, computeFormat=False): @@ -425,7 +428,7 @@ def writeTable(self, tableName, table, singleRow=False, computeFormat=False): self.writeHeader(tableName, table) if computeFormat: valuesList = [row._asdict().values() for row in table] - self._computeLineFormat(valuesList, all=True) + self._computeLineFormat(valuesList, computeFormat=computeFormat) for row in table: self.writeRow(row) @@ -537,14 +540,13 @@ def from_bool(boolValue): @staticmethod def read_jobstar(jobStarFile): - tValues = StarFile.getTableFromFile(jobStarFile, - 'joboptions_values', + tValues = StarFile.getTableFromFile('joboptions_values', + jobStarFile, guessType=False) - def _val(v): - if v == 'Yes': + if v in ['Yes', 'True', 'true']: return True - elif v == 'No': + elif v in ['No', 'False', 'false']: return False else: return v @@ -563,7 +565,7 @@ def write_jobstar(jobType, values, jobStarFile, isTomo=0, isContinue=0): for k, v in values.items(): val = ('Yes' if v else 'No') if isinstance(v, bool) else v tValues.addRowValues(k, val) - sfOut.writeTable('joboptions_values', tValues, computeFormat=True) + sfOut.writeTable('joboptions_values', tValues, computeFormat='left') @staticmethod def optics_table(acq, opticsGroup=1, opticsGroupName="opticsGroup1", From f44430d6d830d245bc3fe348d634ca1da1e801fa Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 12 Nov 2025 16:59:00 -0600 Subject: [PATCH 085/161] Removed some debug print --- emtools/jobs/batch_manager.py | 8 +++----- emtools/jobs/pipeline.py | 6 +++--- 2 files changed, 6 insertions(+), 8 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 3269c2a..5a2aa38 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -276,7 +276,8 @@ def __init__(self, mdocsPattern, workingPath, raise Exception(f"No mdoc files were found with pattern: {mdocsPattern}") BatchManager.__init__(self, 0, self._iterMdocs(mdocsPattern), workingPath, - itemFileNameFunc=lambda item: item[1]['SubFramePath']) + itemFileNameFunc=lambda item: item[1]['SubFramePath'], + createBatch=kwargs.get('createBatch', True)) self._moviesPath = moviesPath self._wait = kwargs.get('wait', 60) self._timeout = timedelta(seconds=kwargs.get('timeout', 3600)) @@ -328,10 +329,7 @@ def _subframePath(self, mdocFn, section): return os.path.join(movieFolder, Mdoc.getSubFrameBase(section)) def _tsName(self, mdocFn): - name = Path.removeBaseExt(mdocFn) - # if self._suffix: - # name = name.replace(self._suffix, '') - return name + return Path.removeBaseExt(mdocFn) def generate(self): """ Generate batches based on the input items. """ diff --git a/emtools/jobs/pipeline.py b/emtools/jobs/pipeline.py index 48cda02..53f91a6 100644 --- a/emtools/jobs/pipeline.py +++ b/emtools/jobs/pipeline.py @@ -156,11 +156,11 @@ def run(self): self.outputQueue.notifyGeneratorStarts() self.id = threading.get_ident() - self._print(">>>>>> Iterating generator tasks") + # self._print(">>>>>> Iterating generator tasks") for task in self._generator(): - self._print(">>>>>>>> Got task: ", task['id'], "...putting it queue.") + # self._print(">>>>>>>> Got task: ", task['id'], "...putting it queue.") self.outputQueue.putTask(task, self) - self._print(">>>>>>>> SENT task: ", task['id']) + # self._print(">>>>>>>> SENT task: ", task['id']) self.outputQueue.notifyGeneratorEnds() From 25a8a26dbd6e4780b6628fa28c59081c7070d1d9 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 13 Nov 2025 08:39:51 -0600 Subject: [PATCH 086/161] Added method to worklflow Job.getOutput --- emtools/jobs/workflow.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index bbf5a1f..24485bc 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -98,6 +98,12 @@ def registerOutput(self, dataId, **kwargs): def hasOutput(self, dataId): return any(o.id == dataId for o in self.outputs) + def getOutput(self, dataId): + for o in self.outputs: + if o.id == dataId: + return o + return None + def _validateInputs(self, inputs): for i in inputs: if not isinstance(i, Workflow.Data): From 2d67b09f535b8ac1448770dea592a58f607edf46 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 13 Nov 2025 12:03:48 -0600 Subject: [PATCH 087/161] Added argument to write timeStamp --- emtools/metadata/starfile.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index c666225..d9f74ab 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -411,7 +411,10 @@ def _computeLineFormat(self, valuesList, computeFormat=False): self._format = " ".join("{:%s%d%s} " % (a, w + 1, f) for w, f in zip(widths, formats)) + '\n' - def writeTable(self, tableName, table, singleRow=False, computeFormat=False): + def writeTable(self, tableName, table, + singleRow=False, + computeFormat=False, + timeStamp=False): """ Write a Table in Star format to the given file. Args: @@ -419,8 +422,12 @@ def writeTable(self, tableName, table, singleRow=False, computeFormat=False): table: Table that is going to be written singleRow: If True, don't write *loop\_*, just label/value pairs. computeFormat: compute format based on widest first column, - just for aesthetics and not recommended for large tables + just for aesthetics and not recommended for large tables. + Values can be 'left' or 'rigth' for alignment. """ + if timeStamp: + self.writeTimeStamp() + if table.size(): if singleRow: self.writeSingleRow(tableName, table[0]) From 87fc681f7bfd96f46ffcdfa583aa8a7158984504 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sun, 16 Nov 2025 11:52:37 -0600 Subject: [PATCH 088/161] Added missing requirement --- requirements.txt | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/requirements.txt b/requirements.txt index 9e549ea..7a15b9e 100644 --- a/requirements.txt +++ b/requirements.txt @@ -2,4 +2,5 @@ mrcfile numpy Pillow>=9.0.1 xmltodict -psutil \ No newline at end of file +psutil +tifffile From 905d47dca490a51870ee826d374600af8ee39359 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 11 Dec 2025 15:03:40 -0600 Subject: [PATCH 089/161] Do not check stat if file does not exist --- emtools/utils/path.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index e9c5c60..a66b819 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -155,9 +155,10 @@ def lastModified(folder): for fn in files: f = os.path.join(folder, fn) - s = os.stat(f) - t = (f, s.st_mtime) - last = t if not last or s.st_mtime > last[1] else last + if os.path.exists(f): + s = os.stat(f) + t = (f, s.st_mtime) + last = t if not last or s.st_mtime > last[1] else last if last: return last[0], dt.fromtimestamp(last[1]) From c2c555366cb38cb7ec38f1b57ef6e8b076e5c0f4 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 16 Dec 2025 11:20:41 -0600 Subject: [PATCH 090/161] Added .eer as tiff image --- emtools/image/thumbnail.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index e3fd0ad..0edcf20 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -170,7 +170,8 @@ def get_dimensions(imagePath): return mrc.data.shape[::-1] # in reverse order elif (imageLower.endswith('.tif') or imageLower.endswith('.tiff') or - imageLower.endswith('.eer')): + imageLower.endswith('.eer') or + imageLower.endswith('.gain')): with tifffile.TiffFile(imagePath) as tif: n = len(tif.pages) y, x = tif.pages[0].shape From c102a8f191c6da404344b4e546dbb42602e417e4 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 22 Dec 2025 10:26:40 -0600 Subject: [PATCH 091/161] Added Table.cloneColumns method to create empty table using columns from another table --- emtools/metadata/table.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/emtools/metadata/table.py b/emtools/metadata/table.py index f1f73b8..30ba169 100644 --- a/emtools/metadata/table.py +++ b/emtools/metadata/table.py @@ -49,6 +49,9 @@ def getType(self): def setType(self, colType): self._type = colType + def clone(self): + return Column(self._name, type=self._type) + class ColumnList: def __init__(self, columns=None): @@ -111,6 +114,18 @@ def get(self, key, default=None): return Row + def cloneColumns(self, exclude=None): + """ Create a new Table that will have exactly the same columns + as this table. Optionally, some columns can be excluded. """ + excludeList = exclude or [] + newCols = [] + + for colName, col in self._columns.items(): + if colName not in excludeList: + newCols.append(col.clone()) + + return Table(newCols) + @staticmethod def createColumns(colNames, values, guessType=True, types=None): """ Return a list of Columns create from the names. From ae84c6467b959531c55c07570329ea3541c36b5f Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 23 Jan 2026 22:43:08 -0600 Subject: [PATCH 092/161] Fixed double extension in .mdoc files --- emtools/jobs/batch_manager.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 5a2aa38..cc0a871 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -329,7 +329,11 @@ def _subframePath(self, mdocFn, section): return os.path.join(movieFolder, Mdoc.getSubFrameBase(section)) def _tsName(self, mdocFn): - return Path.removeBaseExt(mdocFn) + # Remove all extensions, there are cases like .mrc.mdoc + name = mdocFn + while Path.getExt(name): + name = Path.removeBaseExt(name) + return name def generate(self): """ Generate batches based on the input items. """ From fe1c2437dd1700d7c94d3bee03f2faa1c8f9caf1 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 23 Jan 2026 22:43:23 -0600 Subject: [PATCH 093/161] Fixed handling of Relion's bool values --- emtools/metadata/starfile.py | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index d9f74ab..871ff2c 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -526,6 +526,8 @@ def _escapeStrValue(v): class RelionStar: JOB_INDEX = re.compile('job(\d{3})') + TRUE_VALUES = ['Yes', 'True', 'true'] + FALSE_VALUES = ['No', 'False', 'false'] @staticmethod def to_bool(strValue): @@ -545,15 +547,23 @@ def from_bool(boolValue): return 'Yes' if boolValue else 'No' + @staticmethod + def true_value(v): + return v in RelionStar.TRUE_VALUES + + @staticmethod + def false_value(v): + return v in RelionStar.FALSE_VALUES + @staticmethod def read_jobstar(jobStarFile): tValues = StarFile.getTableFromFile('joboptions_values', jobStarFile, guessType=False) def _val(v): - if v in ['Yes', 'True', 'true']: + if RelionStar.true_value(v): return True - elif v in ['No', 'False', 'false']: + elif RelionStar.false_value(v): return False else: return v From 02aaadc42944d996c689d95160ef3755169ae780 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 26 Jan 2026 20:09:55 -0600 Subject: [PATCH 094/161] Added method to validate missing data --- emtools/jobs/workflow.py | 3 +++ emtools/metadata/starfile.py | 5 ++++- 2 files changed, 7 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index 24485bc..bbc0faa 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -42,6 +42,9 @@ def hasJob(self, jobId): def getJob(self, jobId, default=None): return self._jobs.get(jobId, default) + def hasData(self, dataId): + return dataId in self.data + def getData(self, dataId): return self.data[dataId] diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 871ff2c..2c49dea 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -790,7 +790,10 @@ def _table(name): if tInput := _table('input_edges'): for row in tInput: job = wf.getJob(row.rlnPipeLineEdgeProcess) - job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) + if wf.hasData(row.rlnPipeLineEdgeFromNode): + job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) + else: + print(f"WARNING: Missing input edge: {row.rlnPipeLineEdgeFromNode}") return wf From 7941ba219ecb7b1e3b0e20b6cb6bd01501f0a2f3 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sun, 1 Feb 2026 07:46:50 -0600 Subject: [PATCH 095/161] Updated Workflow class to use dict instead of lists --- emtools/jobs/workflow.py | 30 +++++++++++++++++++----------- 1 file changed, 19 insertions(+), 11 deletions(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index bbc0faa..fba4341 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -88,33 +88,38 @@ def __init__(self, wf, jobId, index, inputs=None, **kwargs): self.wf = wf self.id = jobId self.index = index - self.inputs = [] - self.outputs = [] + self._inputs = {} + self._outputs = {} self.addInputs(inputs) + @property + def outputs(self): + return self._outputs.values() + def registerOutput(self, dataId, **kwargs): data = Workflow.Data(self, dataId, **kwargs) self.wf.data[dataId] = data - self.outputs.append(data) + self._outputs[dataId] = data return data def hasOutput(self, dataId): - return any(o.id == dataId for o in self.outputs) + return dataId in self._outputs - def getOutput(self, dataId): - for o in self.outputs: - if o.id == dataId: - return o - return None + def getOutput(self, dataId, default=None): + return self._outputs.get(dataId, default) def _validateInputs(self, inputs): for i in inputs: if not isinstance(i, Workflow.Data): raise Exception(f"Input {i} is not of type Workflow.Data") - if i in self.inputs: + if i.id in self._inputs: Exception(f'Input {i} was already added.') # TODO validate cyclic dependencies + @property + def inputs(self): + return self._inputs.values() + def addInputs(self, inputs): if not inputs: return @@ -122,9 +127,12 @@ def addInputs(self, inputs): self._validateInputs(inputs) for i in inputs: - self.inputs.append(i) + self._inputs[i.id] = i i.childs.append(self) + def hasInput(self, inputId): + return inputId in self._inputs + class Data(dict): def __init__(self, parent, dataId, **kwargs): dict.__init__(self, **kwargs) From 2ed7f223f2b937a2cb8b17e36b9c27e7e9e539d5 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Sun, 1 Feb 2026 22:26:00 -0600 Subject: [PATCH 096/161] Added method to clear inputs and more text extensions --- emtools/jobs/workflow.py | 3 +++ emtools/utils/path.py | 2 +- 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index fba4341..b93dd63 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -133,6 +133,9 @@ def addInputs(self, inputs): def hasInput(self, inputId): return inputId in self._inputs + def clearInputs(self): + self._inputs = {} + class Data(dict): def __init__(self, parent, dataId, **kwargs): dict.__init__(self, **kwargs) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index a66b819..43a895e 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -32,7 +32,7 @@ GLOB_CHARS = ['*', '?', '[', ']'] IMAGE_EXT = ['tiff', 'tif', 'png', 'jpg', 'jpeg'] -TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv'] +TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc'] class Path: From 44b6966c2af4672e5358d935c44db52660d5f169 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 3 Feb 2026 12:18:33 -0600 Subject: [PATCH 097/161] Added subset method to Args --- emtools/jobs/batch_manager.py | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index cc0a871..f3a2ee8 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -66,6 +66,33 @@ def fromList(iterable): return args + def subset(self, prefix, new_prefix='', filters=None): + """ Return a new Args object with a subset of the keys. + """ + filters = filters or [] + full_prefix = f'{prefix}.' + + def _filter(k, v): + return k.startswith(full_prefix) + + result = Args() + + for k, v in self.items(): + if _filter(k, v): + nk = k.replace(full_prefix, new_prefix) + + if isinstance(v, bool): + if 'remove_false' in filters: + if v: + result[nk] = '' + else: + result[nk] = v + else: + if v or 'remove_empty' not in filters: + result[nk] = v + + return result + class Vars: """ Handle variable definitions, either from input dict From 533a906dd3681a91808faf7e3759dc384a791ddd Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 6 Feb 2026 10:40:59 -0600 Subject: [PATCH 098/161] Added method to parse timedelta --- emtools/utils/pretty.py | 16 +++++++++++++++- 1 file changed, 15 insertions(+), 1 deletion(-) diff --git a/emtools/utils/pretty.py b/emtools/utils/pretty.py index a32f3b6..7ad7ec7 100644 --- a/emtools/utils/pretty.py +++ b/emtools/utils/pretty.py @@ -16,7 +16,7 @@ import math import os -from datetime import datetime +from datetime import datetime, timedelta class Pretty: @@ -69,6 +69,20 @@ def parse_datetime(dt_str, **kwargs): f = kwargs.get('format', Pretty.DATETIME_FORMAT) return datetime.strptime(dt_str, f) + @staticmethod + def parse_timedelta(td_str, **kwargs): + """Parse 'HH:MM:SS' or 'D days, HH:MM:SS' format""" + parts = td_str.split(', ') + days = 0 + if len(parts) == 2: + days = int(parts[0].split()[0]) + time_part = parts[1] + else: + time_part = parts[0] + + h, m, s = map(float, time_part.split(':')) + return timedelta(days=days, hours=h, minutes=m, seconds=s) + @staticmethod def modified(fn, **kwargs): if not os.path.exists(fn): From b2a93cccdf538dbf10c22f17d0656a5e671b3d93 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 11 Feb 2026 11:31:21 -0600 Subject: [PATCH 099/161] Add folder scanning and JSON snapshot functionality Implemented methods to scan a folder, save its snapshot to a JSON file, and compare the current scan with a previously saved snapshot. Added command-line arguments for scanning, outputting, and comparing folder states. --- emtools/scripts/emt_files.py | 114 ++++++++++++++++++++++++++++++++++- 1 file changed, 113 insertions(+), 1 deletion(-) diff --git a/emtools/scripts/emt_files.py b/emtools/scripts/emt_files.py index 83a18e7..b0a992d 100755 --- a/emtools/scripts/emt_files.py +++ b/emtools/scripts/emt_files.py @@ -18,6 +18,7 @@ import os import time import argparse +import json from glob import glob from datetime import datetime, timedelta from pprint import pprint @@ -27,6 +28,99 @@ from emtools.metadata import EPU, MovieFiles +def scan_folder(folder): + """Scan a folder; return (files_dict, dirs_set). + files_dict: relative_path -> {size, mtime} + dirs_set: set of relative directory paths (including '.' for the root). + """ + folder = os.path.abspath(os.path.expanduser(folder)) + if not os.path.isdir(folder): + raise SystemExit(f"ERROR: Not a directory: {folder}") + files_result = {} + dirs_set = set() + for root, _dirs, files in os.walk(folder): + rel_root = os.path.relpath(root, folder) + if rel_root == '.': + dirs_set.add('.') + else: + dirs_set.add(rel_root) + for fn in files: + path = os.path.join(root, fn) + try: + st = os.stat(path) + except OSError: + continue + rel = os.path.relpath(path, folder) + files_result[rel] = {'size': st.st_size, 'mtime': st.st_mtime} + return files_result, dirs_set + + +def scan_save(folder, output_path): + """Scan folder and write snapshot to a JSON file.""" + files_snapshot, dirs_set = scan_folder(folder) + folder_abs = os.path.abspath(os.path.expanduser(folder)) + data = { + 'folder': folder_abs, + 'scanned_at': datetime.now().isoformat(), + 'files': files_snapshot, + 'dirs': sorted(dirs_set), + } + output_path = os.path.abspath(os.path.expanduser(output_path)) + os.makedirs(os.path.dirname(output_path) or '.', exist_ok=True) + with open(output_path, 'w') as f: + json.dump(data, f, indent=2) + print(f"Scan saved: {len(files_snapshot)} files, {len(dirs_set)} dirs -> {output_path}") + + +def scan_compare(folder, compare_path): + """Scan folder and compare to a previously saved JSON snapshot.""" + folder_abs = os.path.abspath(os.path.expanduser(folder)) + compare_path = os.path.abspath(os.path.expanduser(compare_path)) + if not os.path.isfile(compare_path): + raise SystemExit(f"ERROR: Compare file not found: {compare_path}") + + with open(compare_path) as f: + data = json.load(f) + previous_files = data.get('files', data) if 'files' in data else data + if isinstance(previous_files, dict) and not previous_files and 'files' in data: + previous_files = data['files'] + previous_dirs = set(data.get('dirs', [])) + + current_files, current_dirs = scan_folder(folder) + prev_file_keys = set(previous_files) + curr_file_keys = set(current_files) + + new_files = sorted(curr_file_keys - prev_file_keys) + deleted_files = sorted(prev_file_keys - curr_file_keys) + modified = [] + for k in sorted(prev_file_keys & curr_file_keys): + p, c = previous_files[k], current_files[k] + if p.get('size') != c.get('size') or p.get('mtime') != c.get('mtime'): + modified.append(k) + + new_dirs = sorted(current_dirs - previous_dirs) + deleted_dirs = sorted(previous_dirs - current_dirs) + + def _report(label, items, color_fn=Color.red): + if not items: + return + print(color_fn(f"\n{label} ({len(items)}):")) + for rel in items: + print(f" {rel}") + + print(f"Comparison: current scan vs {compare_path}") + print(f" Files: previous {len(prev_file_keys)} | current {len(curr_file_keys)}") + print(f" Dirs: previous {len(previous_dirs)} | current {len(current_dirs)}") + _report("New folders", new_dirs, Color.green) + _report("Deleted folders", deleted_dirs, Color.red) + _report("New files", new_files, Color.green) + _report("Deleted files", deleted_files, Color.red) + _report("Modified files", modified, Color.red if modified else lambda x: x) + + if not new_files and not deleted_files and not modified and not new_dirs and not deleted_dirs: + print(Color.green("\nNo changes detected.")) + + def statsDir(folder, sort): df = MovieFiles() df.scan(folder) @@ -171,7 +265,14 @@ def main(): g.add_argument('--rsync_dirs', nargs=2, metavar=('DIR1', 'DIR2'), help='Rsync both directories and print the number of ' 'transferred files. ') - + g.add_argument('--scan', metavar='FOLDER', + help='Scan folder. Use with --output to save snapshot to JSON, ' + 'or with --compare to diff against a saved snapshot.') + + p.add_argument('--output', '-o', metavar='FILE', + help='Save scan snapshot to this JSON file (with --scan)') + p.add_argument('--compare', '-c', metavar='FILE', + help='Compare current scan to this JSON snapshot (with --scan)') p.add_argument('--bin', '-b', type=int, default=6000, help="Create bins of the given time in minutes " "(with --timing)") @@ -244,6 +345,17 @@ def _pad(s): elif pattern := args.timing: timeStats(pattern, args.bin, args.plot, args.data) + elif folder := args.scan: + if args.output and args.compare: + p.error("--scan: use either --output or --compare, not both") + elif args.output: + scan_save(folder, args.output) + elif args.compare: + scan_compare(folder, args.compare) + else: + p.error("--scan requires either --output FILE (save snapshot) " + "or --compare FILE (compare to snapshot)") + # TODO: check from here elif args.transfer: frames, raw, epu = args.transfer From 699f17372d1af816a152e5d96b37e7624657b49c Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 11 Feb 2026 11:32:36 -0600 Subject: [PATCH 100/161] added new textfiles extensions --- emtools/utils/path.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 43a895e..6e8f685 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -32,7 +32,9 @@ GLOB_CHARS = ['*', '?', '[', ']'] IMAGE_EXT = ['tiff', 'tif', 'png', 'jpg', 'jpeg'] -TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc'] +TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', + 'star', 'sh', 'out', 'err', 'bashrc', + 'script', 'settings', 'job'] class Path: From 0bf876255297aa570a50b6a99596d3b88e6ea5d6 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 11 Feb 2026 11:32:53 -0600 Subject: [PATCH 101/161] Added utility methods --- emtools/image/thumbnail.py | 9 ++++----- emtools/jobs/batch_manager.py | 4 ++++ 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index 0edcf20..2690dd5 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -41,7 +41,6 @@ def __init__(self, **kwargs): self.min_max = kwargs.get('min_max', None) self.std_threshold = kwargs.get('std_threshold', 0) - def __format(self, pil_img): format = self.output_format @@ -135,8 +134,8 @@ def from_mrc(self, mrc_path): @staticmethod def Micrograph(**kwargs): - """ Shortcut method with presets for Micrograph thumbail. - All settings can be overwriten with kwargs. + """ Shortcut method with presets for Micrograph thumbnail. + All settings can be overwritten with kwargs. """ defaults = { 'output_format': 'base64', @@ -149,8 +148,8 @@ def Micrograph(**kwargs): @staticmethod def Psd(**kwargs): - """ Shortcut method with presets for PSD thumbails. - All settings can be overwriten with kwargs. + """ Shortcut method with presets for PSD thumbnails. + All settings can be overwritten with kwargs. """ defaults = { 'output_format': 'base64', diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index f3a2ee8..c39ffbf 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -45,6 +45,10 @@ def toList(self): def toLine(self): return ' '.join("%s %s" % (k, v) for k, v in self.items()) + @staticmethod + def fromString(string): + return Args.fromList(shlex.split(string)) + @staticmethod def fromList(iterable): args = Args() From 61e36b1edfeff34b4ff3aad22ec8fcf79a7b3987 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 23 Feb 2026 16:08:10 -0600 Subject: [PATCH 102/161] Added Thumbnail.Preview class and more file extensions --- emtools/image/thumbnail.py | 64 ++++++++++++++++++++++++++++++++++++++ emtools/utils/path.py | 8 ++++- 2 files changed, 71 insertions(+), 1 deletion(-) diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index 2690dd5..14f5e86 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -14,6 +14,7 @@ # * # ************************************************************************** +from doctest import OutputChecker import io import numpy as np import base64 @@ -23,6 +24,8 @@ import PIL from PIL import Image +from emtools.utils import Path, Pretty + class Thumbnail: """ Create image thumbnails from different input types. @@ -159,6 +162,67 @@ def Psd(**kwargs): defaults.update(kwargs) return Thumbnail(**defaults) + @staticmethod + def Preview(imagePath, **kwargs): + imageLower = imagePath.lower() + thumb = Thumbnail.Micrograph(max_size=(256, 256)) + + if not (Path.isImage(imagePath) or Path.isEmImage(imagePath)): + raise Exception("Can not generate preview for: %s" % imagePath) + + if Path.isImage(imagePath): + return thumb.from_path(imagePath) + + if imageLower.endswith('.mrc'): + dims = Image.get_dimensions(imagePath) + mrc = mrcfile.open(imagePath, permissive=True) + thumb = Thumbnail.Micrograph() + if len(dims) == 2: + array = mrc.data + elif len(dims) == 3: + x, y, z = dims + if mrc.is_volume() or (x == y and y == z): + thumb = Thumbnail(max_size=(256, 256), output_format='base64') + iMax = mrc.data.max() # min(imean + 10 * isd, imageArray.max()) + iMin = mrc.data.min() # max(imean - 10 * isd, imageArray.min()) + im255 = ((mrc.data - iMin) / (iMax - iMin) * 255).astype(np.uint8) + + # 1. Setup + ximg = PIL.Image.fromarray(im255[:, :, x // 2]) + yimg = PIL.Image.fromarray(im255[:, y // 2, :]) + zimg = PIL.Image.fromarray(im255[z // 2, :, :]) + + xw, xh = ximg.size + yw, yh = yimg.size + zw, zh = zimg.size + + pad = 2 # The thickness of the dark gray lines/borders + + # 2. Calculate canvas size for a full grid with outer borders + # Total Width = (2 * image width) + (3 * padding for left, middle, right) + canvas_w = (xw + yw) + (3 * pad) + canvas_h = (xh + zh) + (3 * pad) + + # Create canvas with a white background (matching your image) + bg_color = (256, 256, 256) + montage = PIL.Image.new('RGB', (canvas_w, canvas_h), bg_color) + + # Top-Left: x slice + montage.paste(ximg, (pad, pad)) + # Bottom-Left: z slice + montage.paste(zimg, (pad, xh + 2 * pad)) + # Bottom-Right: y slice + montage.paste(yimg, (xw + 2 * pad, xh + 2 * pad)) + + return thumb.from_pil(montage) + Pretty.dprint("Loading MRC volume: %s" % str(array.shape)) + else: + array = mrc.data[z//2, :, :] # FIXME + Pretty.dprint("Loading MRC 2D: %s" % str(array.shape)) + return thumb.from_array(array) + else: + raise Exception("Invalid dimensions: %s" % dims) + class Image: @staticmethod diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 6e8f685..6218d10 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -32,9 +32,10 @@ GLOB_CHARS = ['*', '?', '[', ']'] IMAGE_EXT = ['tiff', 'tif', 'png', 'jpg', 'jpeg'] +EM_EXT = ['mrc', 'mrcs', 'eer', 'gain'] TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc', - 'script', 'settings', 'job'] + 'script', 'settings', 'job', 'tomostar'] class Path: @@ -280,6 +281,11 @@ def isImage(path): def isText(path): return Path.getExt(path).lower()[1:] in TEXT_EXT + @staticmethod + def isEmImage(path): + return Path.getExt(path).lower()[1:] in EM_EXT + + class FolderManager: """ Helper class with some path utilities from a given path. """ From cf3f1b1ff410d36e3b4a3e875256c7342c89dc7f Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 3 Mar 2026 09:56:42 -0600 Subject: [PATCH 103/161] Added helper class WarpPopulation --- emtools/metadata/__init__.py | 4 ++-- emtools/metadata/misc.py | 46 +++++++++++++++++++++++++++++++++++- emtools/utils/path.py | 3 ++- 3 files changed, 49 insertions(+), 4 deletions(-) diff --git a/emtools/metadata/__init__.py b/emtools/metadata/__init__.py index d5bf226..eab7984 100644 --- a/emtools/metadata/__init__.py +++ b/emtools/metadata/__init__.py @@ -18,11 +18,11 @@ from .starfile import StarFile, StarMonitor, RelionStar from .epu import EPU from .misc import (Bins, TsBins, DataFiles, MovieFiles, - Mdoc, TextFile, Acquisition, WarpXml) + Mdoc, TextFile, Acquisition, WarpXml, WarpPopulation) from .sqlite import SqliteFile __all__ = ["Column", "ColumnList", "Table", "StarFile", "StarMonitor", "RelionStar", "EPU", "Bins", "TsBins", "SqliteFile", "DataFiles", "MovieFiles", - "Mdoc", "TextFile", "Acquisition", "WarpXml"] + "Mdoc", "TextFile", "Acquisition", "WarpXml", "WarpPopulation"] diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 28c8c60..84f3342 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -18,9 +18,10 @@ import pathlib from datetime import datetime, timedelta from glob import glob +from readline import insert_text import xmltodict -from emtools.utils import Path, Pretty, Process, Timer +from emtools.utils import Path, Pretty, Color, Timer class Bins: @@ -342,6 +343,49 @@ def getDict(self, *keys): return {e['@Name']: e['@Value'] for e in d} +class WarpPopulation: + """ Helper class to read Warp's .population files. """ + def __init__(self, populationFile): + with open(populationFile) as f: + xmlDict = xmltodict.parse(f.read()) + self._data = xmlDict['Population'] + self.Name = self._data['Param']['@Value'] + self.LastRefinementOptions = {e['@Name']: e['@Value'] for e in self._data['LastRefinementOptions']['Param']} + print("\n>>>>>>> Sources\n") + self.Sources = self._parseList(self._data['Sources'], 'Source') + print("\n>>>>>>> SPECIES\n") + self.Species = self._parseList(self._data['Species'], 'Species') + + def __repr__(self): + r = f"Population: {self.Name}\n" + r += f" {Color.bold('Last Refinement Options:')}\n" + for k, v in self.LastRefinementOptions.items(): + r += f" {k:<30}: {v:<}\n" + r += f" {Color.green('Species:')}\n" + for s in self.Species: + r += f" {s['name']:<30}: {s['path']:<}\n" + r += f" {Color.cyan('Sources:')}\n" + for s in self.Sources: + r += f" {s['name']:<30}: {s['path']:<}\n" + return r + + def _parseList(self, data, key): + suffix = f".{key.lower()}" + + def _parseItem(item): + p = item['@Path'] + name = os.path.basename(p).replace(suffix, '') + return {'id': item['@GUID'], 'path': p, 'name': name} + + from pprint import pprint + d = data[key] + + if isinstance(d, list): + return [_parseItem(item) for item in d] + else: + return [_parseItem(d)] + + class Acquisition(dict): """ Subclass from dict with some utilities related to Acquisition. """ diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 6218d10..a34f0f9 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -35,7 +35,8 @@ EM_EXT = ['mrc', 'mrcs', 'eer', 'gain'] TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc', - 'script', 'settings', 'job', 'tomostar'] + 'script', 'settings', 'job', 'tomostar', + 'population'] class Path: From a515cd3495618bd7358b75e9dc8a69cdea8d2ce3 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 31 Mar 2026 09:40:45 -0500 Subject: [PATCH 104/161] Added a method to get starfile tables as dict --- emtools/metadata/starfile.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 2c49dea..584bc5c 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -145,6 +145,14 @@ def getTableFromFile(tableName, starFileName, **kwargs): with StarFile(starFileName) as sf: return sf.getTable(tableName, **kwargs) + @staticmethod + def getTablesDict(starFileName, **kwargs): + """ Shortcut to read all tables from file as a dictionary. + **kwargs are the same expected by getTable function. + """ + with StarFile(starFileName) as sf: + return {table: sf.getTable(table, **kwargs) for table in sf.getTableNames()} + def getTableSize(self, tableName): """ Return the number of elements in the given table without parsing From a1b00a37d28256b62408d349e07718419e6fce51 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 23 Apr 2026 06:16:06 -0500 Subject: [PATCH 105/161] Added extra validation for setting path in FolderManager --- emtools/utils/path.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index a34f0f9..91e7bfd 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -315,6 +315,8 @@ def path(self): @path.setter def path(self, value): + if not isinstance(value, str): + raise Exception(f"FolderManger: Path must be a string, got {type(value)}") self.__path = value def clear(self): From eaf99285e8b35bbf12532a2870bdcd56657ce827 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 27 Apr 2026 09:21:35 -0500 Subject: [PATCH 106/161] Added --relink option and fixed old options --- emtools/scripts/emt_files.py | 40 ++++++++++++++++++++++++++++-------- 1 file changed, 31 insertions(+), 9 deletions(-) diff --git a/emtools/scripts/emt_files.py b/emtools/scripts/emt_files.py index b0a992d..21bf616 100755 --- a/emtools/scripts/emt_files.py +++ b/emtools/scripts/emt_files.py @@ -268,6 +268,10 @@ def main(): g.add_argument('--scan', metavar='FOLDER', help='Scan folder. Use with --output to save snapshot to JSON, ' 'or with --compare to diff against a saved snapshot.') + g.add_argument('--relink', nargs=2, metavar=('OLD_PREFIX', 'NEW_PREFIX'), + help='Relink the symbolic links in the current directory, changing the prefix to the new one') + g.add_argument('--transfer', nargs=3, metavar=('FRAMES_DIR', 'RAW_DIR', 'EPU_DIR'), + help='REVIEW: Transfer files from FRAMES_DIR to RAW_DIR and EPU_DIR') p.add_argument('--output', '-o', metavar='FILE', help='Save scan snapshot to this JSON file (with --scan)') @@ -287,6 +291,8 @@ def main(): p.add_argument('--sort', choices=['count', 'size'], help="Sort results from --stats with a folder" "based on count or size (with --stats FOLDER)") + p.add_argument('--dry-run', action='store_true', + help="Dry run, without actually performing the operation") args = p.parse_args() @@ -397,15 +403,31 @@ def _moveFile(srcFile, dstFile): pprint(epuData.info()) - elif args.parse: - ed = Path.ExtDict() - for root, dirs, files in os.walk(args.parse): - for f in files: - srcFn = os.path.join(root, f) - if os.path.isfile(srcFn): - ed.register(os.path.join(root, f)) - ed.print() - + # elif args.parse: + # ed = Path.ExtDict() + # for root, dirs, files in os.walk(args.parse): + # for f in files: + # srcFn = os.path.join(root, f) + # if os.path.isfile(srcFn): + # ed.register(os.path.join(root, f)) + # ed.print() + + elif args.relink: + old_prefix, new_prefix = args.relink + cwd = os.getcwd() + print(f"Relinking files in {cwd} from {old_prefix} to {new_prefix}") + for fn in os.listdir(cwd): + filepath = os.path.join(cwd, fn) + if os.path.islink(filepath): + target = os.readlink(filepath) + if target.startswith(old_prefix): + new_target = target.replace(old_prefix, new_prefix) + print(f"LINK: {Color.bold(filepath)}\n" + f" OLD: {Color.red(target)}\n" + f" NEW: {Color.green(new_target)}") + if not args.dry_run: + os.unlink(filepath) + os.symlink(new_target, filepath) if __name__ == '__main__': main() From 0e6275c07daa7029186689ec90e8c769e7246d03 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 4 May 2026 16:02:56 -0500 Subject: [PATCH 107/161] Added support for inverted_booleans and possible filters --- emtools/jobs/batch_manager.py | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index c39ffbf..b922396 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -70,7 +70,7 @@ def fromList(iterable): return args - def subset(self, prefix, new_prefix='', filters=None): + def subset(self, prefix, new_prefix='', filters=None, inverted_booleans=[], possitive=[]): """ Return a new Args object with a subset of the keys. """ filters = filters or [] @@ -82,18 +82,26 @@ def _filter(k, v): result = Args() for k, v in self.items(): + k_suffix = k.replace(full_prefix, '') + if _filter(k, v): nk = k.replace(full_prefix, new_prefix) if isinstance(v, bool): if 'remove_false' in filters: - if v: + add_boolean = not v if k_suffix in inverted_booleans else v + if add_boolean: result[nk] = '' else: result[nk] = v else: if v or 'remove_empty' not in filters: - result[nk] = v + if k_suffix in possitive: + value = float(v) + if value > 0: + result[nk] = '' + else: + result[nk] = v return result From ddf48743d7a39bd9e4562322e4d6a83285bd6159 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 5 May 2026 12:55:11 -0500 Subject: [PATCH 108/161] Fixed bug for possible checking --- emtools/jobs/batch_manager.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index b922396..7384345 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -97,9 +97,8 @@ def _filter(k, v): else: if v or 'remove_empty' not in filters: if k_suffix in possitive: - value = float(v) - if value > 0: - result[nk] = '' + if float(v) > 0: + result[nk] = v else: result[nk] = v From 595f46860fe98a20952e204ca67a6fd28c6a2fc9 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 22 May 2026 15:08:11 -0500 Subject: [PATCH 109/161] Added compute hash and fixed rsync parsing for Mac --- emtools/utils/path.py | 46 +++++++++++++++++++++++++++++++++++++++++-- 1 file changed, 44 insertions(+), 2 deletions(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 91e7bfd..6194e4f 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -19,6 +19,7 @@ import time import tempfile import json +import hashlib from glob import glob from datetime import datetime as dt from collections import OrderedDict @@ -36,7 +37,7 @@ TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc', 'script', 'settings', 'job', 'tomostar', - 'population'] + 'population', 'species'] class Path: @@ -137,7 +138,10 @@ def rsync(dir1, dir2, *args, def _value(line): # Get the value after the colon (:) # and remove , that is used to separate thousands - return int(line.split(':')[1].replace(',', '')) + v = line.split(':')[1].replace(',', '') + if ' ' in v: # MacOS have a different rsync output format + v = v.strip().split()[0] + return int(v) transf = 0 transfSize = 0 @@ -286,6 +290,44 @@ def isText(path): def isEmImage(path): return Path.getExt(path).lower()[1:] in EM_EXT + @staticmethod + def computeHashDict(path, verbose=False): + """ Get the hash of a file. """ + import hashlib + + result = {} + + # Ensure the input path is absolute for consistent splitting + base_path = os.path.abspath(path) + + for root, dirs, files in os.walk(base_path): + # 1. Handle folder entries (directories) + for dir_name in dirs: + dir_full_path = os.path.join(root, dir_name) + # Calculate path relative to the input folder + rel_dir_path = os.path.relpath(dir_full_path, base_path) + result[rel_dir_path] = "" + + # 2. Handle file entries + for file_name in files: + file_full_path = os.path.join(root, file_name) + rel_file_path = os.path.relpath(file_full_path, base_path) + + + + # Calculate MD5 by reading the entire file into memory + try: + if verbose: + print(f"Computing hash for {rel_file_path}") + with open(file_full_path, "rb") as f: + file_bytes = f.read() # Loads the whole file into RAM + + # Hash the complete byte string at once + result[rel_file_path] = hashlib.md5(file_bytes).hexdigest() + except (PermissionError, FileNotFoundError): + result[rel_file_path] = "ERROR: Cannot read file" + + return result class FolderManager: From f1163a8639cc18733655de4ee3bafe4edf13950e Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 22 May 2026 15:08:37 -0500 Subject: [PATCH 110/161] Getting Species info from Warp output --- emtools/metadata/misc.py | 29 +++++++++++++++++++++++++++-- 1 file changed, 27 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 84f3342..b313d13 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -343,17 +343,25 @@ def getDict(self, *keys): return {e['@Name']: e['@Value'] for e in d} +class WarpSpecies(dict): + """ Helper class to read Warp's .species files. """ + def __init__(self, speciesFile): + with open(speciesFile) as f: + xmlDict = xmltodict.parse(f.read()) + self._data = xmlDict['Species'] + for item in self._data['Param']: + self[item['@Name']] = item['@Value'] + class WarpPopulation: """ Helper class to read Warp's .population files. """ def __init__(self, populationFile): + self._filepath = populationFile with open(populationFile) as f: xmlDict = xmltodict.parse(f.read()) self._data = xmlDict['Population'] self.Name = self._data['Param']['@Value'] self.LastRefinementOptions = {e['@Name']: e['@Value'] for e in self._data['LastRefinementOptions']['Param']} - print("\n>>>>>>> Sources\n") self.Sources = self._parseList(self._data['Sources'], 'Source') - print("\n>>>>>>> SPECIES\n") self.Species = self._parseList(self._data['Species'], 'Species') def __repr__(self): @@ -385,6 +393,23 @@ def _parseItem(item): else: return [_parseItem(d)] + def getSpecies(self, nameOrIndex): + entry = None + if isinstance(nameOrIndex, int): + entry = self.Species[nameOrIndex] + else: + for s in self.Species: + if s['name'] == nameOrIndex: + entry = s + break + + if not entry: + raise Exception(f"Species {nameOrIndex} not found") + + folder = os.path.dirname(self._filepath) + + return WarpSpecies(os.path.join(folder, entry['path'])) + class Acquisition(dict): """ Subclass from dict with some utilities related to Acquisition. """ From f246b88d8ed2b54138444809949e523c31e6eb09 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa Trevin" Date: Sat, 13 Jun 2026 13:36:22 +0100 Subject: [PATCH 111/161] Handling ending slash in jobids for relion pipeline --- emtools/metadata/starfile.py | 15 +++++++-------- 1 file changed, 7 insertions(+), 8 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 584bc5c..5e17413 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -29,8 +29,7 @@ from datetime import datetime, timedelta import emtools -from emtools.utils import Pretty, Color - +from emtools.utils import Pretty, Color, Path from .table import ColumnList, Table from .misc import Acquisition @@ -776,7 +775,7 @@ def _table(name): if tProc := _table('processes'): for row in tProc: - jobId = row.rlnPipeLineProcessName + jobId = Path.rmslash(row.rlnPipeLineProcessName) wf.registerJob(jobId, alias=row.rlnPipeLineProcessAlias, status=row.rlnPipeLineProcessStatusLabel, @@ -791,13 +790,13 @@ def _table(name): if tOutput := _table('output_edges'): for row in tOutput: - job = wf.getJob(row.rlnPipeLineEdgeProcess) + job = wf.getJob(Path.rmslash(row.rlnPipeLineEdgeProcess)) nodeName = row.rlnPipeLineEdgeToNode job.registerOutput(nodeName, datatype=nodes[nodeName]) if tInput := _table('input_edges'): for row in tInput: - job = wf.getJob(row.rlnPipeLineEdgeProcess) + job = wf.getJob(Path.rmslash(row.rlnPipeLineEdgeProcess)) if wf.hasData(row.rlnPipeLineEdgeFromNode): job.addInputs([wf.getData(row.rlnPipeLineEdgeFromNode)]) else: @@ -816,14 +815,14 @@ def workflow_to_pipeline(wf, pipelineStar): for job in wf.jobs(): tProc.addRowValues( - rlnPipeLineProcessName=job.id, + rlnPipeLineProcessName=Path.addslash(job.id), rlnPipeLineProcessAlias=job['alias'], rlnPipeLineProcessStatusLabel=job['status'], rlnPipeLineProcessTypeLabel=job['jobtype'] ) for i in job.inputs: tInput.addRowValues( - rlnPipeLineEdgeProcess=job.id, + rlnPipeLineEdgeProcess=Path.addslash(job.id), rlnPipeLineEdgeFromNode=i.id ) @@ -834,7 +833,7 @@ def workflow_to_pipeline(wf, pipelineStar): rlnPipeLineNodeTypeLabelDepth=1 ) tOutput.addRowValues( - rlnPipeLineEdgeProcess=job.id, + rlnPipeLineEdgeProcess=Path.addslash(job.id), rlnPipeLineEdgeToNode=o.id ) From 376e96960f0ce2660734c6b6e757091341719449 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa Trevin" Date: Tue, 23 Jun 2026 15:48:35 +0100 Subject: [PATCH 112/161] Added relion tomo alignment constants --- emtools/metadata/starfile.py | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 5e17413..973978e 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -536,6 +536,14 @@ class RelionStar: TRUE_VALUES = ['Yes', 'True', 'true'] FALSE_VALUES = ['No', 'False', 'false'] + TOMO_ALIGNMENT_COLUMNS = [ + "rlnTomoXTilt", + "rlnTomoYTilt", + "rlnTomoZRot", + "rlnTomoXShiftAngst", + "rlnTomoYShiftAngst" + ] + @staticmethod def to_bool(strValue): """ Convert Relion Yes/No to True/False. """ From 53721dbd47ca5165dbbf7d894045059007aaa199 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa Trevin" Date: Fri, 26 Jun 2026 13:52:32 +0100 Subject: [PATCH 113/161] Added more text file extensions --- emtools/utils/path.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 6194e4f..822faad 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -35,9 +35,9 @@ IMAGE_EXT = ['tiff', 'tif', 'png', 'jpg', 'jpeg'] EM_EXT = ['mrc', 'mrcs', 'eer', 'gain'] TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', - 'star', 'sh', 'out', 'err', 'bashrc', + 'star', 'sh', 'out', 'err', 'bashrc', 'xml', 'script', 'settings', 'job', 'tomostar', - 'population', 'species'] + 'population', 'species', 'aln', 'com', 'rawtlt'] class Path: @@ -313,8 +313,6 @@ def computeHashDict(path, verbose=False): file_full_path = os.path.join(root, file_name) rel_file_path = os.path.relpath(file_full_path, base_path) - - # Calculate MD5 by reading the entire file into memory try: if verbose: From e9d7452d78c2fa937f055d1933ccf2bd9b0191a2 Mon Sep 17 00:00:00 2001 From: "J.M. de la Rosa Trevin" Date: Fri, 26 Jun 2026 13:52:51 +0100 Subject: [PATCH 114/161] Added option to print some STAR file columns --- emtools/scripts/emt_star.py | 37 ++++++++++++++++++++++++++++++++++++- 1 file changed, 36 insertions(+), 1 deletion(-) diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py index c016f6f..46cae0e 100755 --- a/emtools/scripts/emt_star.py +++ b/emtools/scripts/emt_star.py @@ -25,7 +25,7 @@ from collections import defaultdict from emtools.utils import Process, Color, Path, Timer, Pretty -from emtools.metadata import StarFile +from emtools.metadata import StarFile, Table def printStarInfo(starFile): @@ -65,6 +65,28 @@ def checkDuplicates(inputStar, table, column): print(f">>> Duplicates: {len(duplicates)}\n" f" {duplicates}") +def printColumns(inputStar, tableName, columns): + + if not os.path.exists(inputStar): + raise Exception(f"Input star file does not exist: {inputStar}") + + with StarFile(inputStar) as sf: + existingTables = sf.getTableNames() + if tableName is None: + tableName = existingTables[0] + else: + if not tableName in existingTables: + raise Exception(f"Table name does not exist: {tableName}") + + table = StarFile.getTableFromFile(tableName, inputStar) + columnList = columns.split() + newTable = Table(columns=[col for col in table.getColumns() if col.getName() in columnList]) + for row in table: + values = {k: getattr(row, k) for k in columnList} + newTable.addRowValues(**values) + + StarFile.printTable(newTable, tableName) + def splitBy(starFile, column, minSize): with StarFile(starFile) as sf: @@ -117,6 +139,9 @@ def main(): p.add_argument('--duplicates', '-d', nargs=2, metavar=('TABLE', 'COLUMN'), help="Check duplicates values for a given label") + p.add_argument('--print', '-p', nargs='+', + metavar=('COLUMNS', 'TABLE'), + help="Print some columns from the given table.") args = p.parse_args() inputStar = args.input @@ -131,6 +156,16 @@ def main(): elif args.duplicates: table, column = args.duplicates checkDuplicates(inputStar, table, column) + elif args.print: + tableName = None + n = len(args.print) + cols = args.print[0] + if n > 2: + raise Exception(f"Only pass columns and optionally the tableName") + elif n > 1: # n == 2 + tableName = args.print[1] + + printColumns(args.input, tableName, cols) else: printStarInfo(args.input) From a02e2602f604281ad743846833bdfc327595d4b4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?March=C3=A1n?= Date: Mon, 6 Jul 2026 11:13:10 -0500 Subject: [PATCH 115/161] TsStarBatchManager now also reads tsMdoc and is passed to the batch --- emtools/jobs/batch_manager.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 7384345..b98574e 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -413,6 +413,7 @@ def generate(self): """ Generate batches based on the input items. """ for tsRow in self._itemsIterator: tsName = tsRow.rlnTomoName + tsMdoc = tsRow.rlnMdocFile with StarFile(tsRow.rlnTomoTiltSeriesStarFile) as sf: items = [row._asdict() for row in sf.iterTable(tsName)] - yield self._createBatch(items, tsName=tsName) + yield self._createBatch(items, tsName=tsName, tsMdoc=tsMdoc) From d3a9215b368da2c24d1e1213671779648e0aa939 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 6 Jul 2026 13:31:13 -0500 Subject: [PATCH 116/161] Added binary_boolean filter to get 0/1 for boolean params --- emtools/jobs/batch_manager.py | 13 ++++++++----- 1 file changed, 8 insertions(+), 5 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index b98574e..de0ed08 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -88,12 +88,15 @@ def _filter(k, v): nk = k.replace(full_prefix, new_prefix) if isinstance(v, bool): - if 'remove_false' in filters: - add_boolean = not v if k_suffix in inverted_booleans else v - if add_boolean: - result[nk] = '' + if 'binary_boolean' in filters: + result[nk] = '1' if v else '0' else: - result[nk] = v + if 'remove_false' in filters: + add_boolean = not v if k_suffix in inverted_booleans else v + if add_boolean: + result[nk] = '' + else: + result[nk] = v else: if v or 'remove_empty' not in filters: if k_suffix in possitive: From 9c494b3596ec288ea32d8d12806f2e686b70972d Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 6 Jul 2026 13:31:57 -0500 Subject: [PATCH 117/161] Condensed in elif --- emtools/jobs/batch_manager.py | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index de0ed08..78282db 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -90,13 +90,12 @@ def _filter(k, v): if isinstance(v, bool): if 'binary_boolean' in filters: result[nk] = '1' if v else '0' + elif 'remove_false' in filters: + add_boolean = not v if k_suffix in inverted_booleans else v + if add_boolean: + result[nk] = '' else: - if 'remove_false' in filters: - add_boolean = not v if k_suffix in inverted_booleans else v - if add_boolean: - result[nk] = '' - else: - result[nk] = v + result[nk] = v else: if v or 'remove_empty' not in filters: if k_suffix in possitive: From 94a3569a1f9516343aeaf35c58da77dc572d8b7b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?March=C3=A1n?= Date: Tue, 7 Jul 2026 14:53:32 -0500 Subject: [PATCH 118/161] Args subset function modified with multiple values --- emtools/jobs/batch_manager.py | 60 ++++++++++++++++++++--------------- 1 file changed, 34 insertions(+), 26 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 78282db..ce1143f 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -69,40 +69,48 @@ def fromList(iterable): args[last_key] = v return args - - def subset(self, prefix, new_prefix='', filters=None, inverted_booleans=[], possitive=[]): - """ Return a new Args object with a subset of the keys. - """ + + def subset(self, prefix, new_prefix='', filters=None, + inverted_booleans=None, possitive=None, multiple_values=None): + """Return a new Args object with a subset of the keys.""" filters = filters or [] - full_prefix = f'{prefix}.' - - def _filter(k, v): - return k.startswith(full_prefix) + inverted_booleans = inverted_booleans or [] + possitive = possitive or [] + multiple_values = multiple_values or [] + full_prefix = f'{prefix}.' result = Args() for k, v in self.items(): + if not k.startswith(full_prefix): + continue + k_suffix = k.replace(full_prefix, '') + nk = k.replace(full_prefix, new_prefix) + + if isinstance(v, bool): + if 'binary_boolean' in filters: + result[nk] = '1' if v else '0' + elif 'remove_false' in filters: + add_boolean = not v if k_suffix in inverted_booleans else v + if add_boolean: + result[nk] = '' + else: + result[nk] = v - if _filter(k, v): - nk = k.replace(full_prefix, new_prefix) + continue - if isinstance(v, bool): - if 'binary_boolean' in filters: - result[nk] = '1' if v else '0' - elif 'remove_false' in filters: - add_boolean = not v if k_suffix in inverted_booleans else v - if add_boolean: - result[nk] = '' - else: - result[nk] = v - else: - if v or 'remove_empty' not in filters: - if k_suffix in possitive: - if float(v) > 0: - result[nk] = v - else: - result[nk] = v + if not v and 'remove_empty' in filters: + continue + + if k_suffix in possitive and float(v) <= 0: + continue + + if 'multiple_values' in filters and k_suffix in multiple_values: + tokens = str(v).split() + result[nk] = tokens if len(tokens) > 1 else v + else: + result[nk] = v return result From aadfed9191640fba8f9e8592d0d2fe5b77a307af Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 7 Jul 2026 15:04:27 -0500 Subject: [PATCH 119/161] Added Job.getInput method and renamed label to rlnTomoMdocFile --- emtools/jobs/batch_manager.py | 2 +- emtools/jobs/workflow.py | 3 +++ 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 78282db..143367c 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -415,7 +415,7 @@ def generate(self): """ Generate batches based on the input items. """ for tsRow in self._itemsIterator: tsName = tsRow.rlnTomoName - tsMdoc = tsRow.rlnMdocFile + tsMdoc = tsRow.rlnTomoMdocFile with StarFile(tsRow.rlnTomoTiltSeriesStarFile) as sf: items = [row._asdict() for row in sf.iterTable(tsName)] yield self._createBatch(items, tsName=tsName, tsMdoc=tsMdoc) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index b93dd63..f4ffbcb 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -120,6 +120,9 @@ def _validateInputs(self, inputs): def inputs(self): return self._inputs.values() + def getInput(self, inputId, default=None): + return self._inputs.get(inputId, default) + def addInputs(self, inputs): if not inputs: return From a471a37ca94e5fa79318c486c88f37c20ea74da8 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 7 Jul 2026 15:27:03 -0500 Subject: [PATCH 120/161] Using regex to detect args with - or -- --- emtools/jobs/batch_manager.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index f898e2c..f4552f8 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -20,6 +20,7 @@ import traceback import shlex import time +import re from glob import glob from uuid import uuid4 from datetime import datetime, timedelta @@ -51,9 +52,13 @@ def fromString(string): @staticmethod def fromList(iterable): + r = re.compile(r"^-{1,2}[a-zA-Z][a-zA-Z0-9_-]+$") + def _is_arg(v): + return r.match(v) is not None + args = Args() for p in iterable: - if p.startswith('--'): + if _is_arg(p): last_key = p args[p] = '' else: From 3fd53ff2eb30c44699bcb46602d93197c908f96d Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 7 Jul 2026 15:58:58 -0500 Subject: [PATCH 121/161] Added mdoc as text file --- emtools/utils/path.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 822faad..0a4635d 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -36,7 +36,7 @@ EM_EXT = ['mrc', 'mrcs', 'eer', 'gain'] TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc', 'xml', - 'script', 'settings', 'job', 'tomostar', + 'script', 'settings', 'job', 'tomostar', 'mdoc', 'population', 'species', 'aln', 'com', 'rawtlt'] From e10db1ddd963f45b574a4dd9fc79842362e1cb1a Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 8 Jul 2026 11:52:02 -0500 Subject: [PATCH 122/161] Some Relion tomo colums refactoring --- emtools/metadata/starfile.py | 24 ++++++++++++++---------- 1 file changed, 14 insertions(+), 10 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 973978e..31b29ac 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -536,6 +536,15 @@ class RelionStar: TRUE_VALUES = ['Yes', 'True', 'true'] FALSE_VALUES = ['No', 'False', 'false'] + TOMO_FRAME_SERIES_COLUMNS = [ + 'rlnMicrographMovieName', + 'rlnTomoTiltMovieFrameCount', + 'rlnTomoNominalStageTiltAngle', + 'rlnTomoNominalTiltAxisAngle', + 'rlnMicrographPreExposure', + 'rlnTomoNominalDefocus' + ] + TOMO_ALIGNMENT_COLUMNS = [ "rlnTomoXTilt", "rlnTomoYTilt", @@ -651,17 +660,12 @@ def coordinates_table(**kwargs): @staticmethod def tiltseries_table(mc=True, ctf=True, **kwargs): - cols = [ - 'rlnMicrographMovieName', - 'rlnTomoTiltMovieFrameCount', - 'rlnTomoNominalStageTiltAngle', - 'rlnTomoNominalTiltAxisAngle', - 'rlnMicrographPreExposure', - 'rlnTomoNominalDefocus', + cols = list(RelionStar.TOMO_FRAME_SERIES_COLUMNS) + cols.extend([ + 'rlnMicrographName', 'rlnMicrographNameEven', - 'rlnMicrographNameOdd', - 'rlnMicrographName' - ] + 'rlnMicrographNameOdd' + ]) if mc: cols.extend([ From 2ffde735a5534b29062574153a7df3d0bd7259bf Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 15 Jul 2026 17:15:29 -0500 Subject: [PATCH 123/161] Pass the whole row as a dict --- emtools/jobs/batch_manager.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index f4552f8..59a9e3c 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -431,4 +431,4 @@ def generate(self): tsMdoc = tsRow.rlnTomoMdocFile with StarFile(tsRow.rlnTomoTiltSeriesStarFile) as sf: items = [row._asdict() for row in sf.iterTable(tsName)] - yield self._createBatch(items, tsName=tsName, tsMdoc=tsMdoc) + yield self._createBatch(items, tsName=tsName, tsMdoc=tsMdoc, rowDict=tsRow._asdict()) From 96781610595a997493919e26e6382e19aa6cbbcf Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 17 Jul 2026 13:29:16 -0500 Subject: [PATCH 124/161] Added some more text extension --- emtools/utils/path.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 0a4635d..806dc64 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -37,7 +37,8 @@ TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc', 'xml', 'script', 'settings', 'job', 'tomostar', 'mdoc', - 'population', 'species', 'aln', 'com', 'rawtlt'] + 'population', 'species', + 'aln', 'com', 'rawtlt', 'tlt', 'xf', 'xtilt'] class Path: From 1f6485c65637fd57883f38919f533fb211a0149c Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 23 Jul 2026 11:32:42 -0500 Subject: [PATCH 125/161] Mapped unknown status in Relion to 'Saved' --- emtools/metadata/starfile.py | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 31b29ac..fd84fe0 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -825,11 +825,18 @@ def workflow_to_pipeline(wf, pipelineStar): tOutput = tables['output_edges'] tInput = tables['input_edges'] + # There are some job'status that are not supported by Relion, so we need to map them to the expected values + status_map = { + 'Launched': 'Scheduled', + 'Saved': 'Scheduled' + } + for job in wf.jobs(): + status = status_map.get(job['status'], job['status']) tProc.addRowValues( rlnPipeLineProcessName=Path.addslash(job.id), rlnPipeLineProcessAlias=job['alias'], - rlnPipeLineProcessStatusLabel=job['status'], + rlnPipeLineProcessStatusLabel=status, rlnPipeLineProcessTypeLabel=job['jobtype'] ) for i in job.inputs: From 585b57e2beb4d70e73050ae656cef6ddab881815 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 23 Jul 2026 11:33:11 -0500 Subject: [PATCH 126/161] Added Imod util class and methods to read .tlt and .xf files --- emtools/metadata/__init__.py | 4 ++-- emtools/metadata/misc.py | 26 ++++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 2 deletions(-) diff --git a/emtools/metadata/__init__.py b/emtools/metadata/__init__.py index eab7984..0618916 100644 --- a/emtools/metadata/__init__.py +++ b/emtools/metadata/__init__.py @@ -18,11 +18,11 @@ from .starfile import StarFile, StarMonitor, RelionStar from .epu import EPU from .misc import (Bins, TsBins, DataFiles, MovieFiles, - Mdoc, TextFile, Acquisition, WarpXml, WarpPopulation) + Mdoc, TextFile, Acquisition, WarpXml, WarpPopulation, Imod) from .sqlite import SqliteFile __all__ = ["Column", "ColumnList", "Table", "StarFile", "StarMonitor", "RelionStar", "EPU", "Bins", "TsBins", "SqliteFile", "DataFiles", "MovieFiles", - "Mdoc", "TextFile", "Acquisition", "WarpXml", "WarpPopulation"] + "Mdoc", "TextFile", "Acquisition", "WarpXml", "WarpPopulation", "Imod"] diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index b313d13..722a1ff 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -461,3 +461,29 @@ def total_dose(self): @total_dose.setter def total_dose(self, value): self['total_dose'] = float(value) + + +class Imod: + @staticmethod + def get_angles_from_tlt(tltFile): + """ Read AreTomo3/IMOD file with tilt angles. + + Expected file: + TS_NAME_Imod/TS_NAME_st.tlt + Returns: + list[float]: list of tilt angles (as floats) in the same order as in the input file. + """ + return [float(line) for line in TextFile.stripLines(tltFile)] + + @staticmethod + def get_alignment_from_xf(xfFile): + """ Read IMOD XF transformation matrices from .xf file. + + Expected file: + TS_NAME_Imod/TS_NAME_st.xf + Each row contains: + A11 A12 A21 A22 DX DY + Returns: + list[list[float]] + """ + return [list(map(float, line.split())) for line in TextFile.stripLines(xfFile)] \ No newline at end of file From fb2fa49743a5cdbd1adaafcfb669f1f847765941 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 23 Jul 2026 16:54:57 -0500 Subject: [PATCH 127/161] Write the #order for starfiles and allow table from list of rows --- emtools/metadata/starfile.py | 11 +++++++---- emtools/metadata/table.py | 21 +++++++++++++++++++-- 2 files changed, 26 insertions(+), 6 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index fd84fe0..55e76da 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -49,9 +49,12 @@ class StarFile(AbstractContextManager): _splitRegex = re.compile('\"[^"]*\"|[^"\s]+') @staticmethod - def printTable(table, tableName=''): + def printTable(table, tableName='', + computeFormat=False, + timeStamp=False): w = StarFile(sys.stdout, closeFile=False) - w.writeTable(tableName, table, singleRow=len(table) <= 1) + w.writeTable(tableName, table, singleRow=len(table) <= 1, + computeFormat=computeFormat, timeStamp=timeStamp) def __init__(self, inputFile, mode='r', **kwargs): """ @@ -371,8 +374,8 @@ def writeHeader(self, tableName, table): self._file.write("loop_\n") self._columns = table.getColumns() # Write column names - for col in self._columns: - self._file.write("_%s \n" % col.getName()) + for i, col in enumerate(self._columns, start=1): + self._file.write(f"_{col.getName()} # {i}\n") def writeRowValues(self, values): """ Write to file a line for these row values. diff --git a/emtools/metadata/table.py b/emtools/metadata/table.py index 30ba169..9ebc63c 100644 --- a/emtools/metadata/table.py +++ b/emtools/metadata/table.py @@ -160,8 +160,25 @@ def __init__(self, columns=None): @staticmethod def fromDict(valuesDict): - t = Table(list(valuesDict.keys())) - t.addRowValues(**valuesDict) + """ Create a Table from a dictionary of values or a list of dictionaries. + If it is a list, all dictionaries must have the same keys. + + Args: + valuesDict: a dictionary of values or a list of dictionaries + Returns: + Table: a Table object + """ + if isinstance(valuesDict, dict): + rows = [valuesDict] + elif isinstance(valuesDict, list): + rows = valuesDict + else: + raise ValueError(f"Invalid type {type(valuesDict)} for valuesDict") + + t = Table(list(rows[0].keys())) + for row in rows: + t.addRowValues(**row) + return t def clear(self): From 7025d8a83d6b5a9c530b4ee0925ff8f0e2588e7b Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 24 Jul 2026 10:30:32 -0500 Subject: [PATCH 128/161] Allow to sort zsections by the DateTime value --- emtools/metadata/misc.py | 35 ++++++++++++++++++++++++++++++----- 1 file changed, 30 insertions(+), 5 deletions(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index 722a1ff..fcf449a 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -299,14 +299,39 @@ def getSubFrameBase(section): subFramePath = section.get('SubFramePath', '') return pathlib.PureWindowsPath(subFramePath).parts[-1] + MDOC_DATE_FMTS = ('%d-%b-%Y %H:%M:%S', '%d-%b-%y %H:%M:%S') + + @staticmethod + def parseDate(dateStr): + """ Parse an mdoc DateTime field (e.g. '31-Jul-19 17:20:05'). + + SerialEM used dd-Mon-yy before 4.1; yyyy since 4.1 (July 2022). + """ + for fmt in Mdoc.MDOC_DATE_FMTS: + try: + return datetime.strptime(dateStr, fmt) + except ValueError: + continue + raise ValueError(f"Could not parse mdoc DateTime: {dateStr!r}") + @property def zvalues(self): - return [(k, v) for k, v in self.zsections()] + """ Get the Z values from the mdoc file. + Returns: + list[tuple[str, dict]]: list of Z values with the section data + """ + return list(self.zsections()) - def zsections(self): - for k, v in self.items(): - if k.startswith('ZValue'): - yield k, v + def zsections(self, sort=None): + """ Iterate over ZValue sections in the mdoc file. + Args: + sort: Use 'date' to sort by acquisition date (newest first). + """ + sections = [(k, v) for k, v in self.items() if k.startswith('ZValue')] + if sort == 'date': + sections.sort(key=lambda x: Mdoc.parseDate(x[1]['DateTime'])) + for k, v in sections: + yield k, v def write(self, path): with open(path, 'w') as f: From 4340bedfc9f8d5e4a5c4e1d0f967f30bbc1da0db Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 24 Jul 2026 10:30:51 -0500 Subject: [PATCH 129/161] Added methods to clearOutputs --- emtools/jobs/workflow.py | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index f4ffbcb..1908273 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -57,8 +57,7 @@ def registerJob(self, jobId, inputs=None, **kwargs): return job def deleteJob(self, job): - for o in job.outputs: - del self.data[o.id] + job.clearOutputs() del self._jobs[job.id] def dot(self): @@ -139,6 +138,16 @@ def hasInput(self, inputId): def clearInputs(self): self._inputs = {} + def removeOutput(self, output_id): + if output_id in self._outputs: + if output_id in self.wf.data: + del self.wf.data[output_id] + del self._outputs[output_id] + + def clearOutputs(self): + for output_id in list(self._outputs.keys()): + self.removeOutput(output_id) + class Data(dict): def __init__(self, parent, dataId, **kwargs): dict.__init__(self, **kwargs) From 28caa8b41c326d6a970134fd481a4a0df8854397 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 27 Jul 2026 12:56:27 -0500 Subject: [PATCH 130/161] Added alignment conversion from Imod to Relion and fixed comment space in STAR column headers --- emtools/metadata/starfile.py | 58 +++++++++++++++++++++++++++++++++++- 1 file changed, 57 insertions(+), 1 deletion(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 55e76da..01943be 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -25,6 +25,7 @@ import sys import time import re +import math from contextlib import AbstractContextManager from datetime import datetime, timedelta @@ -375,7 +376,7 @@ def writeHeader(self, tableName, table): self._columns = table.getColumns() # Write column names for i, col in enumerate(self._columns, start=1): - self._file.write(f"_{col.getName()} # {i}\n") + self._file.write(f"_{col.getName()} #{i}\n") def writeRowValues(self, values): """ Write to file a line for these row values. @@ -732,6 +733,61 @@ def get_acquisition(inputTableOrFile): amplitude_contrast=o.get('rlnAmplitudeContrast', 0.1) ) + @staticmethod + def alignment_from_xf(xf_row, pixel_size): + """Convert one IMOD XF row into Relion alignment labels. + IMOD XF row: + A11 A12 A21 A22 DX DY + The translation should be taken from the inverse transform, then + converted from pixels to Angstroms. + """ + a11, a12, a21, a22, dx, dy = xf_row + + det = a11 * a22 - a12 * a21 + if abs(det) < 1e-12: + return { + 'rlnTomoZRot': '', + 'rlnTomoXShiftAngst': '', + 'rlnTomoYShiftAngst': '', + } + + z_rot = math.degrees(math.atan2(a12, a11)) + + # Inverse affine translation: + # inv(M) * -t + inv_dx = -((a22 * dx - a12 * dy) / det) + inv_dy = -((-a21 * dx + a11 * dy) / det) + + return { + 'rlnTomoZRot': z_rot, + 'rlnTomoXShiftAngst': inv_dx * pixel_size, + 'rlnTomoYShiftAngst': inv_dy * pixel_size, + } + + @staticmethod + def alignments_from_imod(tlt_angles, xf_alignments, pixel_size): + """ Read tilt angles (.tlt file) and IMOD transforms (.xf file) to compute Relion alignments. + Returns: + list[dict]: list of Relion alignments + """ + rln_alignments = [] + + for tilt, xf_row in zip(tlt_angles, xf_alignments): + xf_values = RelionStar.alignment_from_xf(xf_row, pixel_size) + ctf_scale = math.cos(math.radians(tilt)) + + rln_alignments.append({ + 'tilt': tilt, + 'rlnTomoXTilt': 0.0 if tilt != '' else '', + 'rlnTomoYTilt': tilt, + 'rlnTomoZRot': xf_values.get('rlnTomoZRot', ''), + 'rlnTomoXShiftAngst': xf_values.get('rlnTomoXShiftAngst', ''), + 'rlnTomoYShiftAngst': xf_values.get('rlnTomoYShiftAngst', ''), + 'rlnCtfScalefactor': ctf_scale, + }) + + return rln_alignments + @staticmethod def pipeline_tables(): return { From a481153ab345fec8a5312de160fcfa65c1318cf5 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 28 Jul 2026 09:40:13 -0500 Subject: [PATCH 131/161] Added some tomo functions to RelionStar --- emtools/metadata/starfile.py | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 01943be..063e2de 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -583,6 +583,25 @@ def true_value(v): def false_value(v): return v in RelionStar.FALSE_VALUES + @staticmethod + def getTomoBinning(row): + return float(getattr(row, 'rlnTomoTomogramBinning', 1)) + + @staticmethod + def getTomoPixelSize(row): + """Compute the tomogram pixel size from TS pixel size and binning.""" + return (float(getattr(row, 'rlnTomoTiltSeriesPixelSize', 0)) + * RelionStar.getTomoBinning(row)) + + @staticmethod + def getTomogram(row): + """Return tomogram path, trying from different columns.""" + cols = ['rlnTomoReconstructedTomogram', 'rlnTomoReconstructedTomogramDenoised'] + for col in cols: + if value := row.get(col): + return value + raise ValueError(f"No tomogram column ({', '.join(cols)}) found in row: {row}") + @staticmethod def read_jobstar(jobStarFile): tValues = StarFile.getTableFromFile('joboptions_values', From 68884032cff371f4fce92591a432fa9e63d29fff Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 29 Jul 2026 11:58:37 -0500 Subject: [PATCH 132/161] Added helper methods --- emtools/metadata/starfile.py | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 063e2de..0e93b75 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -593,6 +593,17 @@ def getTomoPixelSize(row): return (float(getattr(row, 'rlnTomoTiltSeriesPixelSize', 0)) * RelionStar.getTomoBinning(row)) + @staticmethod + def reconstructedTomoSize(row, axis): + """Return reconstructed tomogram size in pixels along X/Y/Z.""" + return float(getattr(row, axis)) / RelionStar.getTomoBinning(row) + + @staticmethod + def centeredAngstToPixel(centered_angst, row, axis): + """Convert Relion centered Angstrom coordinates to tomogram pixels.""" + return (float(centered_angst) / RelionStar.getTomoPixelSize(row) + + RelionStar.reconstructedTomoSize(row, axis) / 2) + @staticmethod def getTomogram(row): """Return tomogram path, trying from different columns.""" From ff12028a0faba826dd7e40303d359f6f98afe496 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 5 Aug 2026 08:46:31 -0500 Subject: [PATCH 133/161] Added method to getAcquisition from STAR file --- emtools/metadata/starfile.py | 84 +++++++++++++++++++++++++++++------- 1 file changed, 68 insertions(+), 16 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 0e93b75..28c18b9 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -743,25 +743,77 @@ def global_tiltseries_table(**kwargs): return Table(cols) @staticmethod - def get_acquisition(inputTableOrFile): - """ Load acquisition parameters from an optics table - or a given input STAR file. - """ - if isinstance(inputTableOrFile, Table): - tOptics = inputTableOrFile + def _acquisition_from_row(row): + """ Build Acquisition from an optics or tomography global row. """ + if getattr(row, 'rlnTomoTiltSeriesPixelSize', None): + pixel_size = RelionStar.getTomoPixelSize(row) else: - with StarFile(inputTableOrFile) as sf: - tOptics = sf.getTable('optics') + pixel_size = (getattr(row, 'rlnMicrographPixelSize', None) + or row.rlnMicrographOriginalPixelSize) + + acq = Acquisition( + pixel_size=pixel_size, + voltage=row.rlnVoltage, + cs=row.rlnSphericalAberration, + amplitude_contrast=getattr(row, 'rlnAmplitudeContrast', 0.1) + ) + if gain := getattr(row, 'rlnMicrographGainName', None): + acq['gain'] = gain + if dose := getattr(row, 'rlnMicrographDoseRate', None): + acq['total_dose'] = float(dose) - o = tOptics[0]._asdict() # get first row + return acq - return Acquisition( - pixel_size=o.get('rlnMicrographPixelSize', - o['rlnMicrographOriginalPixelSize']), - voltage=o['rlnVoltage'], - cs=o['rlnSphericalAberration'], - amplitude_contrast=o.get('rlnAmplitudeContrast', 0.1) - ) + @staticmethod + def _resolve_linked_star(baseStarFile, linkedPath): + if not linkedPath: + return None + if os.path.isabs(linkedPath): + return linkedPath + + candidates = [ + os.path.normpath(os.path.join(os.path.dirname(baseStarFile), + linkedPath)), + os.path.normpath(os.path.join(os.getcwd(), linkedPath)), + ] + for candidate in candidates: + if os.path.exists(candidate): + return candidate + return candidates[0] + + @staticmethod + def getAcquisition(inputTableOrFile): + """ Load acquisition parameters from an optics/global table row, + or a given input STAR file (movies, tilt series, tomograms, etc.). + """ + if hasattr(inputTableOrFile, 'rlnVoltage'): + return RelionStar._acquisition_from_row(inputTableOrFile) + + if isinstance(inputTableOrFile, Table): + return RelionStar._acquisition_from_row(inputTableOrFile[0]) + + starFile = inputTableOrFile + if starFile.endswith('optimisation_set.star'): + with StarFile(starFile) as sf: + tableNames = sf.getTableNames() + tableName = ('optimisation_set' if 'optimisation_set' in tableNames + else tableNames[0]) + t = sf.getTable(tableName) + row = t[0] + if tomogramsStar := getattr(row, 'rlnTomoTomogramsFile', None): + return RelionStar.getAcquisition( + RelionStar._resolve_linked_star(starFile, tomogramsStar)) + if particlesStar := getattr(row, 'rlnTomoParticlesFile', None): + return RelionStar.getAcquisition( + RelionStar._resolve_linked_star(starFile, particlesStar)) + + with StarFile(starFile) as sf: + if t := sf.getTable('optics'): + return RelionStar._acquisition_from_row(t[0]) + if t := sf.getTable('global'): + return RelionStar._acquisition_from_row(t[0]) + + raise Exception(f"Could not read acquisition parameters from {starFile}") @staticmethod def alignment_from_xf(xf_row, pixel_size): From 4f390d5db302a505b65f930413860e76314756df Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 5 Aug 2026 09:33:29 -0500 Subject: [PATCH 134/161] Bumped version to 0.2.0-rc260805 --- emtools/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/emtools/__init__.py b/emtools/__init__.py index 4f29437..50a8321 100644 --- a/emtools/__init__.py +++ b/emtools/__init__.py @@ -24,5 +24,5 @@ # * # ************************************************************************** -__version__ = '0.1.4rc' +__version__ = '0.2.0-rc260805' From caebbeeb16f055bcce4473e99f3d2c5cd0aae181 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?March=C3=A1n?= Date: Wed, 5 Aug 2026 09:38:24 -0500 Subject: [PATCH 135/161] add alignment to xf conversion function --- emtools/metadata/starfile.py | 66 ++++++++++++++++++++++++++++++++++++ 1 file changed, 66 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 0e93b75..70119f0 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -794,6 +794,72 @@ def alignment_from_xf(xf_row, pixel_size): 'rlnTomoYShiftAngst': inv_dy * pixel_size, } + @staticmethod + def alignment_to_xf(alignment, pixel_size): + """Convert one RELION alignment row into an AreTomo/IMOD XF row. + + The AreTomo XF convention used here stores: + + A11 A12 A21 A22 DX DY + + Matrix coefficients are rounded to three decimals before computing + DX and DY. Translations are then rounded to two decimals. This is + required to reverse the existing XF-to-RELION conversion exactly + for AreTomo's rounded rigid transforms. + """ + if pixel_size <= 0: + raise ValueError( + f"Alignment pixel size must be greater than zero: {pixel_size}" + ) + + def _get_value(name): + if isinstance(alignment, dict): + value = alignment.get(name) + else: + value = getattr(alignment, name, None) + + if value in (None, ''): + raise ValueError(f"Missing RELION alignment value: {name}") + + return float(value) + + z_rot = _get_value('rlnTomoZRot') + + shift_x_pixels = ( + _get_value('rlnTomoXShiftAngst') / pixel_size + ) + shift_y_pixels = ( + _get_value('rlnTomoYShiftAngst') / pixel_size + ) + + angle = math.radians(z_rot) + + # Important: AreTomo writes the matrix coefficients with three + # decimals. Round them before reconstructing the translation. + a11 = float(f'{math.cos(angle):.3f}') + a12 = float(f'{math.sin(angle):.3f}') + a21 = float(f'{-math.sin(angle):.3f}') + a22 = float(f'{math.cos(angle):.3f}') + + # Existing forward conversion: + # relion_shift = -inverse(M) @ imod_translation + # Therefore: + # imod_translation = -M @ relion_shift + dx = -(a11 * shift_x_pixels + a12 * shift_y_pixels) + dy = -(a21 * shift_x_pixels + a22 * shift_y_pixels) + + # AreTomo writes translations with two decimals. + dx = float(f'{dx:.2f}') + dy = float(f'{dy:.2f}') + + # Avoid writing "-0.00". + if dx == 0: + dx = 0.0 + if dy == 0: + dy = 0.0 + + return [a11, a12, a21, a22, dx, dy] + @staticmethod def alignments_from_imod(tlt_angles, xf_alignments, pixel_size): """ Read tilt angles (.tlt file) and IMOD transforms (.xf file) to compute Relion alignments. From f8558e0573282ddb681dfbe81f7a09e1554c9015 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 10 Aug 2026 15:42:48 -0500 Subject: [PATCH 136/161] allow to checkChild given job pid --- emtools/utils/process.py | 32 ++++++++++++++++++++++++++++++-- 1 file changed, 30 insertions(+), 2 deletions(-) diff --git a/emtools/utils/process.py b/emtools/utils/process.py index 2c58979..5307c0e 100644 --- a/emtools/utils/process.py +++ b/emtools/utils/process.py @@ -119,11 +119,37 @@ def _filter_name(proc): return processes @staticmethod - def checkChilds(programName, folderPath, kill=False, verbose=0): + def checkChilds(programName, folderPath, kill=False, verbose=0, pid=None): from .system import System specs = System.specs() cpus = specs['CPUs'] - processes = Process.ps(programName, workingDir=folderPath, children=True) + attrs = ['pid', 'ppid', 'name', 'cwd', 'username', + 'memory_percent', 'cpu_percent'] + + if pid is not None: + try: + root = psutil.Process(int(pid)) + except (psutil.NoSuchProcess, psutil.AccessDenied, ValueError): + return False + + procs = [] + seen = set() + + def _add(proc): + if proc.pid in seen: + return + proc.info = proc.as_dict(attrs) + procs.append(proc) + seen.add(proc.pid) + + _add(root) + for child in root.children(recursive=True): + _add(child) + folder = root.info.get('cwd') or folderPath or '' + processes = {folder: procs} + else: + processes = Process.ps(programName, workingDir=folderPath, + children=True) color = Color.red if kill else Color.bold @@ -157,6 +183,8 @@ def checkChilds(programName, folderPath, kill=False, verbose=0): except: pass + return True if pid is not None else None + class Logger: """ Use a logger to log commands that are executed via os.system. """ def __init__(self, logger=None, only_log=False, From 97ffc638379bdc21f950b11821ff1947ac5df8ef Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 10 Aug 2026 15:43:16 -0500 Subject: [PATCH 137/161] Added methods to validate optimisation_sets.tar and particles.star --- emtools/metadata/starfile.py | 108 ++++++++++++++++++++++++++++++--- emtools/tests/test_metadata.py | 70 ++++++++++++++++++++- 2 files changed, 170 insertions(+), 8 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 28c18b9..ca2da58 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -557,6 +557,105 @@ class RelionStar: "rlnTomoYShiftAngst" ] + TOMO_OPTIMISATION_SET_TABLE = 'optimisation_set' + TOMO_OPTIMISATION_SET_COLUMNS = [ + 'rlnTomoParticlesFile', + 'rlnTomoTomogramsFile' + ] + + TOMO_PARTICLES_TABLE = 'particles' + TOMO_PARTICLES_COLUMNS = [ + 'rlnTomoName', + ] + TOMO_PARTICLES_PIXEL_COORD_COLUMNS = [ + 'rlnCoordinateX', + 'rlnCoordinateY', + 'rlnCoordinateZ', + ] + TOMO_PARTICLES_CENTERED_COORD_COLUMNS = [ + 'rlnCenteredCoordinateXAngst', + 'rlnCenteredCoordinateYAngst', + 'rlnCenteredCoordinateZAngst', + ] + + @staticmethod + def hasTomoParticleCoordinates(table): + """Return True if the table has tomography particle coordinates.""" + return ( + table.hasAllColumns(RelionStar.TOMO_PARTICLES_PIXEL_COORD_COLUMNS) + or table.hasAllColumns(RelionStar.TOMO_PARTICLES_CENTERED_COORD_COLUMNS) + ) + + @staticmethod + def isTomoOptimisationSet(starFile): + """Return True if the STAR file has a compliant optimisation_set table.""" + if not starFile or not os.path.isfile(starFile): + return False + try: + with StarFile(starFile) as sf: + if RelionStar.TOMO_OPTIMISATION_SET_TABLE not in sf.getTableNames(): + return False + if sf.getTableSize(RelionStar.TOMO_OPTIMISATION_SET_TABLE) < 1: + return False + table = sf.getTableInfo(RelionStar.TOMO_OPTIMISATION_SET_TABLE) + return ( + table is not None + and table.hasAllColumns(RelionStar.TOMO_OPTIMISATION_SET_COLUMNS) + ) + except (OSError, IOError, Exception): + return False + + @staticmethod + def isTomoParticles(starFile): + """Return True if the STAR file has a compliant tomography particles table.""" + if not starFile or not os.path.isfile(starFile): + return False + try: + with StarFile(starFile) as sf: + if RelionStar.TOMO_PARTICLES_TABLE not in sf.getTableNames(): + return False + if sf.getTableSize(RelionStar.TOMO_PARTICLES_TABLE) < 1: + return False + table = sf.getTableInfo(RelionStar.TOMO_PARTICLES_TABLE) + return ( + table is not None + and table.hasAllColumns(RelionStar.TOMO_PARTICLES_COLUMNS) + and RelionStar.hasTomoParticleCoordinates(table) + ) + except (OSError, IOError, Exception): + return False + + @staticmethod + def readTomoOptimisationSet(starFile): + """Read the optimisation_set table or raise ValueError.""" + if not RelionStar.isTomoOptimisationSet(starFile): + raise ValueError( + f"{starFile} is not a compliant tomography optimisation_set STAR file." + ) + table = StarFile.getTableFromFile( + RelionStar.TOMO_OPTIMISATION_SET_TABLE, starFile) + if not table: + raise ValueError( + f"Could not read '{RelionStar.TOMO_OPTIMISATION_SET_TABLE}' " + f"table from {starFile}." + ) + return table + + @staticmethod + def readTomoParticles(starFile): + """Read the particles table or raise ValueError.""" + if not RelionStar.isTomoParticles(starFile): + raise ValueError( + f"{starFile} is not a compliant tomography particles STAR file." + ) + table = StarFile.getTableFromFile(RelionStar.TOMO_PARTICLES_TABLE, starFile) + if not table: + raise ValueError( + f"Could not read '{RelionStar.TOMO_PARTICLES_TABLE}' " + f"table from {starFile}." + ) + return table + @staticmethod def to_bool(strValue): """ Convert Relion Yes/No to True/False. """ @@ -793,13 +892,8 @@ def getAcquisition(inputTableOrFile): return RelionStar._acquisition_from_row(inputTableOrFile[0]) starFile = inputTableOrFile - if starFile.endswith('optimisation_set.star'): - with StarFile(starFile) as sf: - tableNames = sf.getTableNames() - tableName = ('optimisation_set' if 'optimisation_set' in tableNames - else tableNames[0]) - t = sf.getTable(tableName) - row = t[0] + if RelionStar.isTomoOptimisationSet(starFile): + row = RelionStar.readTomoOptimisationSet(starFile)[0] if tomogramsStar := getattr(row, 'rlnTomoTomogramsFile', None): return RelionStar.getAcquisition( RelionStar._resolve_linked_star(starFile, tomogramsStar)) diff --git a/emtools/tests/test_metadata.py b/emtools/tests/test_metadata.py index cdf9ab4..df8e0ae 100644 --- a/emtools/tests/test_metadata.py +++ b/emtools/tests/test_metadata.py @@ -24,7 +24,7 @@ from datetime import datetime from emtools.utils import Timer, Color, Pretty -from emtools.metadata import StarFile, SqliteFile, EPU, StarMonitor +from emtools.metadata import StarFile, SqliteFile, EPU, StarMonitor, RelionStar, Table from emtools.jobs import BatchManager from emtools.tests import testpath @@ -374,6 +374,74 @@ def _pipeline(monitor): self.__test_star_streaming(_pipeline, inputStreaming=False) +class TestRelionStarTomo(unittest.TestCase): + """Tests for Relion tomography STAR file validation helpers.""" + + def _write_star(self, tables): + ftmp = tempfile.NamedTemporaryFile(mode='w', delete=False, suffix='.star') + with StarFile(ftmp.name, 'w') as sf: + for table_name, table in tables.items(): + sf.writeTable(table_name, table, timeStamp=False, singleRow=len(table) <= 1) + ftmp.close() + return ftmp.name + + def test_isTomoOptimisationSet(self): + opt_star = self._write_star({ + 'optimisation_set': Table.fromDict({ + 'rlnTomoParticlesFile': 'particles.star', + 'rlnTomoTomogramsFile': 'tomograms.star', + }), + }) + wrong_table = self._write_star({ + 'global': Table.fromDict({'rlnTomoName': 'tomo1'}), + }) + missing_link = self._write_star({ + 'optimisation_set': Table.fromDict({'rlnTomoTomogramsFile': 'tomograms.star'}), + }) + + self.assertTrue(RelionStar.isTomoOptimisationSet(opt_star)) + self.assertFalse(RelionStar.isTomoOptimisationSet(wrong_table)) + self.assertFalse(RelionStar.isTomoOptimisationSet(missing_link)) + self.assertFalse(RelionStar.isTomoOptimisationSet(__file__)) + + row = RelionStar.readTomoOptimisationSet(opt_star)[0] + self.assertEqual(row.rlnTomoParticlesFile, 'particles.star') + + for fn in (opt_star, wrong_table, missing_link): + os.unlink(fn) + + def test_isTomoParticles(self): + particles_star = self._write_star({ + 'particles': Table.fromDict({ + 'rlnTomoName': 'tomo1', + 'rlnCoordinateX': 1.0, + 'rlnCoordinateY': 2.0, + 'rlnCoordinateZ': 3.0, + }), + }) + centered_star = self._write_star({ + 'particles': Table.fromDict({ + 'rlnTomoName': 'tomo1', + 'rlnCenteredCoordinateXAngst': 10.0, + 'rlnCenteredCoordinateYAngst': 20.0, + 'rlnCenteredCoordinateZAngst': 30.0, + }), + }) + missing_coords = self._write_star({ + 'particles': Table.fromDict({'rlnTomoName': 'tomo1'}), + }) + + self.assertTrue(RelionStar.isTomoParticles(particles_star)) + self.assertTrue(RelionStar.isTomoParticles(centered_star)) + self.assertFalse(RelionStar.isTomoParticles(missing_coords)) + + table = RelionStar.readTomoParticles(particles_star) + self.assertEqual(table[0].rlnTomoName, 'tomo1') + + for fn in (particles_star, centered_star, missing_coords): + os.unlink(fn) + + class TestEPU(unittest.TestCase): """ Tests for EPU class. """ From 0db72e38474fbe5785c232f56ad9181eda0df047 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 12 Aug 2026 16:45:33 -0500 Subject: [PATCH 138/161] fixed optimisation set validation --- emtools/metadata/starfile.py | 30 +++++++++--------------------- 1 file changed, 9 insertions(+), 21 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index ca2da58..a67a4b4 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -589,19 +589,9 @@ def hasTomoParticleCoordinates(table): @staticmethod def isTomoOptimisationSet(starFile): """Return True if the STAR file has a compliant optimisation_set table.""" - if not starFile or not os.path.isfile(starFile): - return False try: - with StarFile(starFile) as sf: - if RelionStar.TOMO_OPTIMISATION_SET_TABLE not in sf.getTableNames(): - return False - if sf.getTableSize(RelionStar.TOMO_OPTIMISATION_SET_TABLE) < 1: - return False - table = sf.getTableInfo(RelionStar.TOMO_OPTIMISATION_SET_TABLE) - return ( - table is not None - and table.hasAllColumns(RelionStar.TOMO_OPTIMISATION_SET_COLUMNS) - ) + RelionStar.readTomoOptimisationSet(starFile) + return True except (OSError, IOError, Exception): return False @@ -628,18 +618,16 @@ def isTomoParticles(starFile): @staticmethod def readTomoOptimisationSet(starFile): """Read the optimisation_set table or raise ValueError.""" - if not RelionStar.isTomoOptimisationSet(starFile): + if not starFile or not os.path.isfile(starFile): + raise ValueError(f"Invalid STAR file: {starFile}") + all_tables = StarFile.getTablesDict(starFile) + first_table = next(iter(all_tables.values())) + + if not first_table.hasAllColumns(RelionStar.TOMO_OPTIMISATION_SET_COLUMNS): raise ValueError( f"{starFile} is not a compliant tomography optimisation_set STAR file." ) - table = StarFile.getTableFromFile( - RelionStar.TOMO_OPTIMISATION_SET_TABLE, starFile) - if not table: - raise ValueError( - f"Could not read '{RelionStar.TOMO_OPTIMISATION_SET_TABLE}' " - f"table from {starFile}." - ) - return table + return first_table @staticmethod def readTomoParticles(starFile): From e33269d3075c3348d3d1824ac1834550ac784fcb Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Wed, 12 Aug 2026 16:45:43 -0500 Subject: [PATCH 139/161] Added image helper methods --- emtools/image/thumbnail.py | 206 ++++++++++++++++++++++++++++--------- 1 file changed, 158 insertions(+), 48 deletions(-) diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index 14f5e86..64cbf41 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -162,6 +162,91 @@ def Psd(**kwargs): defaults.update(kwargs) return Thumbnail(**defaults) + @staticmethod + def _preview_stack_indices(n): + """Return 4 frame indices: first, two equally spaced, last.""" + if n <= 1: + return [0, 0, 0, 0] + if n == 2: + return [0, 0, 1, 1] + if n == 3: + return [0, 1, 1, 2] + return [ + 0, + int(round((n - 1) / 3)), + int(round(2 * (n - 1) / 3)), + n - 1, + ] + + @staticmethod + def _array_to_uint8(array): + i_min = array.min() + i_max = array.max() + if i_max == i_min: + return np.zeros(array.shape, dtype=np.uint8) + return ((array - i_min) / (i_max - i_min) * 255).astype(np.uint8) + + @staticmethod + def _montage_grid(pil_images, cols, pad=2, bg_color=(255, 255, 255)): + w, h = pil_images[0].size + rows = (len(pil_images) + cols - 1) // cols + canvas_w = cols * w + (cols + 1) * pad + canvas_h = rows * h + (rows + 1) * pad + montage = PIL.Image.new('RGB', (canvas_w, canvas_h), bg_color) + + for i, img in enumerate(pil_images): + row, col = divmod(i, cols) + x = pad + col * (w + pad) + y = pad + row * (h + pad) + if img.mode != 'RGB': + img = img.convert('RGB') + montage.paste(img, (x, y)) + + return montage + + @staticmethod + def _volume_slice_montage(data): + im255 = Thumbnail._array_to_uint8(data) + _, y, x = im255.shape + z = im255.shape[0] + + ximg = PIL.Image.fromarray(im255[:, :, x // 2]) + yimg = PIL.Image.fromarray(im255[:, y // 2, :]) + zimg = PIL.Image.fromarray(im255[z // 2, :, :]) + + xw, xh = ximg.size + yw, yh = yimg.size + pad = 2 + canvas_w = (xw + yw) + (3 * pad) + canvas_h = (xh + yh) + (3 * pad) + montage = PIL.Image.new('RGB', (canvas_w, canvas_h), (255, 255, 255)) + + montage.paste(ximg.convert('RGB'), (pad, pad)) + montage.paste(zimg.convert('RGB'), (pad, xh + 2 * pad)) + montage.paste(yimg.convert('RGB'), (xw + 2 * pad, xh + 2 * pad)) + + return montage + + @staticmethod + def _stack_frame_montage(data): + n = data.shape[0] + im255 = Thumbnail._array_to_uint8(data) + indices = Thumbnail._preview_stack_indices(n) + images = [PIL.Image.fromarray(im255[i, :, :]) for i in indices] + return Thumbnail._montage_grid(images, cols=2) + + @staticmethod + def _mrc_is_stack(mrc, image_lower): + if image_lower.endswith('.mrcs'): + return True + if len(mrc.data.shape) < 3: + return False + if mrc.is_volume(): + return False + if mrc.is_image_stack(): + return True + return True + @staticmethod def Preview(imagePath, **kwargs): imageLower = imagePath.lower() @@ -173,58 +258,83 @@ def Preview(imagePath, **kwargs): if Path.isImage(imagePath): return thumb.from_path(imagePath) - if imageLower.endswith('.mrc'): - dims = Image.get_dimensions(imagePath) - mrc = mrcfile.open(imagePath, permissive=True) - thumb = Thumbnail.Micrograph() - if len(dims) == 2: - array = mrc.data - elif len(dims) == 3: - x, y, z = dims - if mrc.is_volume() or (x == y and y == z): - thumb = Thumbnail(max_size=(256, 256), output_format='base64') - iMax = mrc.data.max() # min(imean + 10 * isd, imageArray.max()) - iMin = mrc.data.min() # max(imean - 10 * isd, imageArray.min()) - im255 = ((mrc.data - iMin) / (iMax - iMin) * 255).astype(np.uint8) - - # 1. Setup - ximg = PIL.Image.fromarray(im255[:, :, x // 2]) - yimg = PIL.Image.fromarray(im255[:, y // 2, :]) - zimg = PIL.Image.fromarray(im255[z // 2, :, :]) - - xw, xh = ximg.size - yw, yh = yimg.size - zw, zh = zimg.size - - pad = 2 # The thickness of the dark gray lines/borders - - # 2. Calculate canvas size for a full grid with outer borders - # Total Width = (2 * image width) + (3 * padding for left, middle, right) - canvas_w = (xw + yw) + (3 * pad) - canvas_h = (xh + zh) + (3 * pad) - - # Create canvas with a white background (matching your image) - bg_color = (256, 256, 256) - montage = PIL.Image.new('RGB', (canvas_w, canvas_h), bg_color) - - # Top-Left: x slice - montage.paste(ximg, (pad, pad)) - # Bottom-Left: z slice - montage.paste(zimg, (pad, xh + 2 * pad)) - # Bottom-Right: y slice - montage.paste(yimg, (xw + 2 * pad, xh + 2 * pad)) - - return thumb.from_pil(montage) - Pretty.dprint("Loading MRC volume: %s" % str(array.shape)) + if imageLower.endswith('.mrc') or imageLower.endswith('.mrcs'): + with mrcfile.open(imagePath, permissive=True) as mrc: + data = mrc.data + if len(data.shape) == 2: + return thumb.from_array(data) + + if len(data.shape) != 3: + raise Exception("Invalid dimensions: %s" % (data.shape,)) + + preview_thumb = Thumbnail( + max_size=(256, 256), output_format='base64') + + if Thumbnail._mrc_is_stack(mrc, imageLower): + montage = Thumbnail._stack_frame_montage(data) else: - array = mrc.data[z//2, :, :] # FIXME - Pretty.dprint("Loading MRC 2D: %s" % str(array.shape)) - return thumb.from_array(array) - else: - raise Exception("Invalid dimensions: %s" % dims) + montage = Thumbnail._volume_slice_montage(data) + + return preview_thumb.from_pil(montage) + + raise Exception("Can not generate preview for: %s" % imagePath) class Image: + @staticmethod + def _mrc_data_type(mrc, image_lower): + if image_lower.endswith('.mrcs'): + return '2D stack' + if mrc.is_volume(): + return '3D volume' + if mrc.is_image_stack(): + return '2D stack' + return None + + @staticmethod + def get_metadata(imagePath): + """Return structured metadata for EM image files, or None.""" + imageLower = imagePath.lower() + if imageLower.endswith('.mrc') or imageLower.endswith('.mrcs'): + with mrcfile.open(imagePath) as mrc: + dims = mrc.data.shape[::-1] + if len(dims) == 2: + x, y = dims + return { + 'info': f'{x} x {y}', + } + if len(dims) == 3: + x, y, third = dims + data_type = Image._mrc_data_type(mrc, imageLower) + if data_type == '3D volume': + return { + 'dataType': data_type, + 'info': f'{x} x {y} x {third}', + } + return { + 'dataType': '2D stack', + 'info': f'{x} x {y} x {third}', + } + return None + + if (imageLower.endswith('.tif') or + imageLower.endswith('.tiff') or + imageLower.endswith('.eer') or + imageLower.endswith('.gain')): + dims = Image.get_dimensions(imagePath) + if len(dims) == 2: + x, y = dims + return { + 'info': f'{x} x {y}', + } + if len(dims) == 3: + x, y, n = dims + return { + 'dataType': '2D stack', + 'info': f'{x} x {y} x {n}', + } + return None + @staticmethod def get_dimensions(imagePath): imageLower = imagePath.lower() From 838d142f55888ba1fc5c632bbadc6d8834567442 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 14 Aug 2026 10:22:47 -0500 Subject: [PATCH 140/161] Minor fix to detect volumes --- emtools/image/thumbnail.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index 64cbf41..7ff26bf 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -305,7 +305,8 @@ def get_metadata(imagePath): } if len(dims) == 3: x, y, third = dims - data_type = Image._mrc_data_type(mrc, imageLower) + is_cube = x == y == third + data_type = '3D volume' if is_cube else Image._mrc_data_type(mrc, imageLower) if data_type == '3D volume': return { 'dataType': data_type, From 92987d5b162d89e1f62b50dc54fe0beda107b4bb Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 14 Aug 2026 10:23:53 -0500 Subject: [PATCH 141/161] Added Table.filter and Table.update, also accessible via emt-star --- emtools/metadata/table.py | 88 ++++++++++++ emtools/scripts/emt_star.py | 249 +++++++++++++++++++++++++++++---- emtools/tests/test_metadata.py | 95 +++++++++++++ 3 files changed, 406 insertions(+), 26 deletions(-) diff --git a/emtools/metadata/table.py b/emtools/metadata/table.py index 9ebc63c..e068168 100644 --- a/emtools/metadata/table.py +++ b/emtools/metadata/table.py @@ -23,6 +23,7 @@ from collections import OrderedDict, namedtuple +import re class Column: @@ -300,6 +301,51 @@ def keyFunc(r): return getattr(r, key) if isinstance(key, str) else key self._rows.sort(key=keyFunc, reverse=reverse) + def update(self, spec): + """Update column values from a comma-separated spec of col=expr assignments. + + Each assignment must contain exactly one '=' character. Expressions are + evaluated per row using column names as variables. + + Args: + spec: Comma-separated column=expression pairs, e.g. + 'rlnPixelSize=rlnOriginalPixelSize/2, rlnOpticsGroup=1' + + Returns: + self, to allow method chaining. + """ + updates = _parseUpdateSpec(spec) + colNames = self.getColumnNames() + for col, _ in updates: + if col not in colNames: + raise ValueError(f"Column '{col}' not found in table") + + newRows = [] + for row in self._rows: + values = row._asdict() + for col, expr in updates: + values[col] = _evalRowExpr(expr, row, values=values) + newRows.append(self.Row(**values)) + self._rows = newRows + return self + + def filter(self, spec): + """Keep rows where the expression evaluates to True. + + Args: + spec: Boolean expression evaluated per row using column names + as variables, e.g. 'rlnOpticsGroup == 1' + + Returns: + self, to allow method chaining. + """ + spec = spec.strip() + if not spec: + raise ValueError("FILTER requires a non-empty expression") + _validateFilterSpec(spec) + self._rows = [row for row in self._rows if _evalRowExpr(spec, row)] + return self + def print(self, formatStr=None): for row in self._rows: print(formatStr.format(**row._asdict())) @@ -318,6 +364,48 @@ def __setitem__(self, key, value): # --------- Helper functions ------------------------ +def _parseUpdateSpec(spec): + """Parse 'col1=expr1, col2=expr2' into a list of (column, expression) pairs.""" + spec = spec.strip() + if not spec: + raise ValueError("UPDATE requires a non-empty expression spec") + + updates = [] + for part in spec.split(','): + part = part.strip() + if not part: + continue + if part.count('=') != 1: + raise ValueError( + f"Invalid UPDATE spec (expected exactly one '=' per column): {part}") + col, expr = part.split('=', 1) + col = col.strip() + expr = expr.strip() + if not col: + raise ValueError(f"Invalid UPDATE spec (empty column name): {part}") + if not expr: + raise ValueError(f"Invalid UPDATE spec (empty expression): {part}") + updates.append((col, expr)) + + if not updates: + raise ValueError("UPDATE requires at least one column assignment") + return updates + + +def _validateFilterSpec(spec): + """Reject bare column names that often mean the shell ate a comparison.""" + if re.fullmatch(r'[A-Za-z_][A-Za-z0-9_]*', spec): + raise ValueError( + f"FILTER expression '{spec}' looks like a column name only. " + "Quote the expression in the shell, e.g. 'rlnDefocusAngle > 50'") + + +def _evalRowExpr(expr, row, values=None): + """Evaluate an expression using row column values as variables.""" + namespace = dict(values if values is not None else row._asdict()) + return eval(expr, {"__builtins__": {}}, namespace) + + def _str(s): """ Get the string value but stripping quotes if present. """ return s[1:-1] if s.startswith('"') and s.endswith('"') else s diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py index 46cae0e..31318ce 100755 --- a/emtools/scripts/emt_star.py +++ b/emtools/scripts/emt_star.py @@ -16,6 +16,7 @@ # ************************************************************************** import os +import sys import time import argparse from glob import glob @@ -28,6 +29,161 @@ from emtools.metadata import StarFile, Table +ALL_TABLES = 'all' +OP_DROP = 'DROP' +OP_UPDATE = 'UPDATE' +OP_FILTER = 'FILTER' +VALID_OPERATIONS = {OP_DROP, OP_UPDATE, OP_FILTER} + + +def _normalizeOperation(op): + op = op.strip().upper() + if op not in VALID_OPERATIONS: + raise ValueError( + f"Unknown operation '{op}'. Valid values: {', '.join(sorted(VALID_OPERATIONS))}") + return op + + +def _resolveTableName(table, tableNames): + """Resolve table name; only the special 'all' token is case-insensitive.""" + table = table.strip() + if table.lower() == ALL_TABLES: + return ALL_TABLES + if table not in tableNames: + raise ValueError(f"Table '{table}' not found in input STAR file") + return table + + +def _parseOperateArgs(operateArgs): + """Flatten --operate argument groups into (TABLE, OPERATION, EXPR) tuples.""" + operations = [] + for group in operateArgs: + if len(group) % 3 != 0: + raise ValueError( + "--operate expects groups of TABLE OPERATION EXPR " + f"(multiple of 3 arguments), got {len(group)} in one --operate") + for i in range(0, len(group), 3): + operations.append((group[i], group[i + 1], group[i + 2])) + return operations + + +def _parseOperateActions(operations, tableNames): + """Build per-table action lists and the set of tables to drop.""" + explicitTables = set() + dropTables = set() + tableActions = defaultdict(list) + allActions = [] + allDrop = False + + for table, operation, expr in operations: + tableName = _resolveTableName(table, tableNames) + operation = _normalizeOperation(operation) + + if tableName == ALL_TABLES: + if operation == OP_DROP: + allDrop = True + else: + allActions.append((operation, expr)) + continue + + explicitTables.add(tableName) + + if operation == OP_DROP: + dropTables.add(tableName) + else: + tableActions[tableName].append((operation, expr)) + + if allDrop: + for tableName in tableNames: + if tableName not in explicitTables: + dropTables.add(tableName) + + return explicitTables, dropTables, tableActions, allActions + + +def _iterTableRows(sf, tableName, subset=None): + kwargs = {'limit': subset} if subset is not None else {} + return list(sf.iterTable(tableName, **kwargs)) + + +def _writeProcessedTable(sfOut, tableName, tableInfo, rows, singleRow): + if not rows: + sfOut.writeTable(tableName, tableInfo) + return + + if singleRow: + sfOut.writeSingleRow(tableName, rows[0]) + else: + result = tableInfo.cloneColumns() + for row in rows: + result.addRow(row) + sfOut.writeTable(tableName, result) + + +def _writeUnchangedTable(sfIn, sfOut, tableName, subset=None): + sfIn.getTableInfo(tableName) + singleRow = sfIn._singleRow + rows = _iterTableRows(sfIn, tableName, subset=subset) + tableInfo = sfIn.getTableInfo(tableName) + _writeProcessedTable(sfOut, tableName, tableInfo, rows, singleRow) + + +def _processTable(sf, tableName, actions, subset=None): + tableInfo = sf.getTableInfo(tableName) + singleRow = sf._singleRow + table = tableInfo.cloneColumns() + kwargs = {'limit': subset} if subset is not None else {} + for row in sf.iterTable(tableName, **kwargs): + table.addRow(row) + + for operation, expr in actions: + if operation == OP_UPDATE: + table.update(expr) + elif operation == OP_FILTER: + table.filter(expr) + + return tableInfo, list(table), singleRow + + +def operateStarFile(inputStar, operations, subset=None, output=None): + if not operations: + raise ValueError("At least one --operate action is required") + + if not os.path.exists(inputStar): + raise FileNotFoundError(f"Input star file does not exist: {inputStar}") + + closeOutput = output is not None + out = open(output, 'w') if closeOutput else sys.stdout + + try: + with StarFile(inputStar) as sfIn: + tableNames = sfIn.getTableNames() + explicitTables, dropTables, tableActions, allActions = ( + _parseOperateActions(operations, tableNames)) + + with StarFile(out, closeFile=closeOutput) as sfOut: + sfOut.writeTimeStamp() + + for tableName in tableNames: + if tableName in dropTables: + continue + + if tableName in explicitTables: + actions = tableActions.get(tableName, []) + else: + actions = list(allActions) + + if not actions: + _writeUnchangedTable(sfIn, sfOut, tableName, subset=subset) + else: + tableInfo, rows, singleRow = _processTable( + sfIn, tableName, actions, subset=subset) + _writeProcessedTable(sfOut, tableName, tableInfo, rows, singleRow) + finally: + if closeOutput: + out.close() + + def printStarInfo(starFile): with StarFile(starFile) as sf: tables = sf.getTableNames() @@ -65,27 +221,45 @@ def checkDuplicates(inputStar, table, column): print(f">>> Duplicates: {len(duplicates)}\n" f" {duplicates}") -def printColumns(inputStar, tableName, columns): - +def printColumns(inputStar, tableName=None, columns=None, subset=None): if not os.path.exists(inputStar): - raise Exception(f"Input star file does not exist: {inputStar}") + raise FileNotFoundError(f"Input star file does not exist: {inputStar}") with StarFile(inputStar) as sf: existingTables = sf.getTableNames() if tableName is None: + if not existingTables: + return tableName = existingTables[0] - else: - if not tableName in existingTables: - raise Exception(f"Table name does not exist: {tableName}") + elif tableName not in existingTables: + raise ValueError(f"Table name does not exist: {tableName}") + + columnList = columns.split() if columns else sf.getTableInfo(tableName).getColumnNames() + table = _buildPrintTable(sf, tableName, columnList, subset=subset) + StarFile.printTable(table, tableName) + - table = StarFile.getTableFromFile(tableName, inputStar) - columnList = columns.split() - newTable = Table(columns=[col for col in table.getColumns() if col.getName() in columnList]) - for row in table: +def printAllTables(inputStar, subset=None): + if not os.path.exists(inputStar): + raise FileNotFoundError(f"Input star file does not exist: {inputStar}") + + with StarFile(inputStar) as sf: + for tableName in sf.getTableNames(): + tableInfo = sf.getTableInfo(tableName) + table = _buildPrintTable(sf, tableName, tableInfo.getColumnNames(), + subset=subset) + StarFile.printTable(table, tableName) + + +def _buildPrintTable(sf, tableName, columnList, subset=None): + tableInfo = sf.getTableInfo(tableName) + cols = [col for col in tableInfo.getColumns() if col.getName() in columnList] + newTable = Table(columns=cols) + kwargs = {'limit': subset} if subset is not None else {} + for row in sf.iterTable(tableName, **kwargs): values = {k: getattr(row, k) for k in columnList} newTable.addRowValues(**values) - - StarFile.printTable(newTable, tableName) + return newTable def splitBy(starFile, column, minSize): @@ -139,14 +313,35 @@ def main(): p.add_argument('--duplicates', '-d', nargs=2, metavar=('TABLE', 'COLUMN'), help="Check duplicates values for a given label") - p.add_argument('--print', '-p', nargs='+', - metavar=('COLUMNS', 'TABLE'), - help="Print some columns from the given table.") + outputMode = p.add_mutually_exclusive_group() + outputMode.add_argument('--print', '-p', nargs='*', + metavar=('COLUMNS', 'TABLE'), + help="Print columns from STAR file tables to stdout. " + "With no arguments, print all tables and all columns. " + "With COLUMNS only, print from the first table. " + "With COLUMNS and TABLE, print from the given table.") + outputMode.add_argument('--operate', '-e', action='append', nargs='+', + metavar='TRIPLET', + help="Apply one or more operations. Each operation is a triplet " + "TABLE OPERATION EXPR; multiple triplets can be passed in a " + "single --operate. TABLE is the exact table name from the " + "input STAR file, or 'all' (case insensitive) for all tables " + "not explicitly listed in other operations. OPERATION can be " + "UPDATE, FILTER, or DROP. For UPDATE, EXPR is comma-separated " + "column=expression assignments. For FILTER, EXPR is a boolean " + "expression per row. For DROP, EXPR is ignored.") + p.add_argument('--subset', '-n', type=int, default=None, metavar='N', + help="Process at most N rows per table (for debugging)") + p.add_argument('--output', '-o', default=None, metavar='FILE', + help="Write output STAR file to this path (default: stdout)") args = p.parse_args() inputStar = args.input - if args.group_by: + if args.operate: + operateStarFile(inputStar, _parseOperateArgs(args.operate), + subset=args.subset, output=args.output) + elif args.group_by: table, column = args.group_by groupBy(inputStar, table, column) elif split := args.split_particles: @@ -156,16 +351,18 @@ def main(): elif args.duplicates: table, column = args.duplicates checkDuplicates(inputStar, table, column) - elif args.print: - tableName = None - n = len(args.print) - cols = args.print[0] - if n > 2: - raise Exception(f"Only pass columns and optionally the tableName") - elif n > 1: # n == 2 - tableName = args.print[1] - - printColumns(args.input, tableName, cols) + elif args.print is not None: + if len(args.print) == 0: + printAllTables(inputStar, subset=args.subset) + else: + tableName = None + cols = args.print[0] + if len(args.print) > 2: + raise ValueError("Only pass columns and optionally the table name") + elif len(args.print) > 1: + tableName = args.print[1] + + printColumns(inputStar, tableName, cols, subset=args.subset) else: printStarInfo(args.input) diff --git a/emtools/tests/test_metadata.py b/emtools/tests/test_metadata.py index df8e0ae..129651c 100644 --- a/emtools/tests/test_metadata.py +++ b/emtools/tests/test_metadata.py @@ -374,6 +374,101 @@ def _pipeline(monitor): self.__test_star_streaming(_pipeline, inputStreaming=False) +class TestTable(unittest.TestCase): + """Tests for Table.update and Table.filter.""" + + @staticmethod + def _sampleTable(): + t = Table(['rlnA', 'rlnB', 'rlnStatus']) + t.addRowValues(1, 10, 'Scheduled') + t.addRowValues(2, 20, 'Scheduled') + t.addRowValues(3, 30, 'Finished') + return t + + def test_update_constant(self): + t = self._sampleTable() + t.update('rlnB=100') + self.assertEqual(t.size(), 3) + for row in t: + self.assertEqual(row.rlnB, 100) + + def test_update_from_column_expression(self): + t = self._sampleTable() + t.update('rlnB = rlnA * 10') + self.assertEqual(t[0].rlnB, 10) + self.assertEqual(t[1].rlnB, 20) + self.assertEqual(t[2].rlnB, 30) + + def test_update_multiple_columns(self): + t = self._sampleTable() + t.update('rlnB = rlnA * 10, rlnStatus = "Updated"') + self.assertEqual(t[0].rlnB, 10) + self.assertEqual(t[0].rlnStatus, 'Updated') + self.assertEqual(t[2].rlnB, 30) + + def test_update_later_assignments_see_earlier_updates(self): + t = Table(['rlnA', 'rlnB', 'rlnC']) + t.addRowValues(2, 0, 0) + t.update('rlnB = rlnA * 10, rlnC = rlnB + 1') + self.assertEqual(t[0].rlnB, 20) + self.assertEqual(t[0].rlnC, 21) + + def test_update_whitespace_in_spec(self): + t = self._sampleTable() + t.update(' rlnB = rlnA * 10 , rlnStatus = "Updated" ') + self.assertEqual(t[1].rlnB, 20) + self.assertEqual(t[1].rlnStatus, 'Updated') + + def test_update_returns_self(self): + t = self._sampleTable() + self.assertIs(t.update('rlnB=0'), t) + + def test_update_raises_on_invalid_spec(self): + t = self._sampleTable() + for spec in ['', 'rlnB', 'rlnB=1=2', 'rlnB==1']: + with self.subTest(spec=spec): + with self.assertRaises(ValueError): + t.update(spec) + + def test_update_raises_on_unknown_column(self): + t = self._sampleTable() + with self.assertRaisesRegex(ValueError, "not found"): + t.update('missingCol=1') + + def test_filter_by_numeric_condition(self): + t = self._sampleTable() + t.filter('rlnA > 1') + self.assertEqual(t.size(), 2) + self.assertEqual([row.rlnA for row in t], [2, 3]) + + def test_filter_by_string_condition(self): + t = self._sampleTable() + t.filter('rlnStatus == "Scheduled"') + self.assertEqual(t.size(), 2) + self.assertTrue(all(row.rlnStatus == 'Scheduled' for row in t)) + + def test_filter_returns_self(self): + t = self._sampleTable() + self.assertIs(t.filter('rlnA > 0'), t) + + def test_filter_raises_on_empty_spec(self): + t = self._sampleTable() + with self.assertRaisesRegex(ValueError, "non-empty"): + t.filter(' ') + + def test_filter_raises_on_bare_column_name(self): + t = self._sampleTable() + with self.assertRaisesRegex(ValueError, "column name only"): + t.filter('rlnA') + + def test_update_then_filter_chaining(self): + t = self._sampleTable() + t.update('rlnB = rlnA * 10').filter('rlnB >= 20') + self.assertEqual(t.size(), 2) + self.assertEqual([row.rlnA for row in t], [2, 3]) + self.assertEqual([row.rlnB for row in t], [20, 30]) + + class TestRelionStarTomo(unittest.TestCase): """Tests for Relion tomography STAR file validation helpers.""" From f3f5927d0d149a51a1bb2b322c3ec2caf23b7a71 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 21 Aug 2026 11:05:17 -0500 Subject: [PATCH 142/161] Added more image info and constant labels --- emtools/datatypes.py | 21 +++++++++++++++++++++ emtools/image/__init__.py | 3 ++- emtools/image/thumbnail.py | 16 +++++++++------- emtools/metadata/starfile.py | 30 +++++++++++++++++++++++++----- 4 files changed, 57 insertions(+), 13 deletions(-) create mode 100644 emtools/datatypes.py diff --git a/emtools/datatypes.py b/emtools/datatypes.py new file mode 100644 index 0000000..e153e85 --- /dev/null +++ b/emtools/datatypes.py @@ -0,0 +1,21 @@ +# ************************************************************************** +# * +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) +# * +# * This program is free software; you can redistribute it and/or modify +# * it under the terms of the GNU General Public License as published by +# * the Free Software Foundation; either version 3 of the License, or +# * (at your option) any later version. +# * +# * This program is distributed in the hope that it will be useful, +# * but WITHOUT ANY WARRANTY; without even the implied warranty of +# * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# * GNU General Public License for more details. +# * +# ************************************************************************** + + +"""Shared data-type labels for EM file and workflow object metadata.""" + +VOLUME = 'Volume' +STACK_2D = '2D stack' diff --git a/emtools/image/__init__.py b/emtools/image/__init__.py index 4563128..7341c21 100644 --- a/emtools/image/__init__.py +++ b/emtools/image/__init__.py @@ -16,6 +16,7 @@ from .thumbnail import Thumbnail, Image +from emtools.datatypes import STACK_2D, VOLUME -__all__ = ["Thumbnail", "Image"] +__all__ = ["Thumbnail", "Image", "STACK_2D", "VOLUME"] diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index 7ff26bf..47d7bc2 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -24,6 +24,8 @@ import PIL from PIL import Image +from emtools.datatypes import STACK_2D, VOLUME + from emtools.utils import Path, Pretty @@ -284,11 +286,11 @@ class Image: @staticmethod def _mrc_data_type(mrc, image_lower): if image_lower.endswith('.mrcs'): - return '2D stack' + return STACK_2D if mrc.is_volume(): - return '3D volume' + return VOLUME if mrc.is_image_stack(): - return '2D stack' + return STACK_2D return None @staticmethod @@ -306,14 +308,14 @@ def get_metadata(imagePath): if len(dims) == 3: x, y, third = dims is_cube = x == y == third - data_type = '3D volume' if is_cube else Image._mrc_data_type(mrc, imageLower) - if data_type == '3D volume': + data_type = VOLUME if is_cube else Image._mrc_data_type(mrc, imageLower) + if data_type == VOLUME: return { 'dataType': data_type, 'info': f'{x} x {y} x {third}', } return { - 'dataType': '2D stack', + 'dataType': STACK_2D, 'info': f'{x} x {y} x {third}', } return None @@ -331,7 +333,7 @@ def get_metadata(imagePath): if len(dims) == 3: x, y, n = dims return { - 'dataType': '2D stack', + 'dataType': STACK_2D, 'info': f'{x} x {y} x {n}', } return None diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index a67a4b4..e28ce76 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -353,6 +353,10 @@ def writeTimeStamp(self): self.writeLine(f"\n# StarFile written on {Pretty.now()} " f"by emtools ({emtools.__version__})\n") + def writeVersion(self, version): + """ Write a RELION metadata-table version tag (e.g. 50001). """ + self.writeLine(f"# version {version}") + def _writeTableName(self, tableName): self._file.write("\ndata_%s\n\n" % (tableName or '')) @@ -425,7 +429,8 @@ def _computeLineFormat(self, valuesList, computeFormat=False): def writeTable(self, tableName, table, singleRow=False, computeFormat=False, - timeStamp=False): + timeStamp=False, + version=None): """ Write a Table in Star format to the given file. Args: @@ -435,9 +440,12 @@ def writeTable(self, tableName, table, computeFormat: compute format based on widest first column, just for aesthetics and not recommended for large tables. Values can be 'left' or 'rigth' for alignment. + version: If set, write a RELION ``# version`` tag before the table. """ if timeStamp: self.writeTimeStamp() + if version is not None: + self.writeVersion(version) if table.size(): if singleRow: @@ -536,6 +544,9 @@ def _escapeStrValue(v): class RelionStar: + # RELION 5 pipeline STAR tables; prevents re-running pre-5.0 label migration. + PIPELINE_VERSION = 50001 + JOB_INDEX = re.compile('job(\d{3})') TRUE_VALUES = ['Yes', 'True', 'true'] FALSE_VALUES = ['No', 'False', 'false'] @@ -720,12 +731,18 @@ def write_jobstar(jobType, values, jobStarFile, isTomo=0, isContinue=0): """ Convert params dict to a Relion job.star file. """ with StarFile(jobStarFile, 'w') as sfOut: tJob = Table(['rlnJobTypeLabel', 'rlnJobIsContinue', 'rlnJobIsTomo']) - tJob.addRowValues(jobType, isContinue, isTomo) # FIXME check continue and isTomo + tJob.addRowValues(jobType, isContinue, isTomo) sfOut.writeTimeStamp() sfOut.writeTable('job', tJob, singleRow=True) tValues = Table(['rlnJobOptionVariable', 'rlnJobOptionValue']) for k, v in values.items(): - val = ('Yes' if v else 'No') if isinstance(v, bool) else v + if isinstance(v, bool): + val = 'Yes' if v else 'No' + elif v is None: + val = '' + else: + # Job option values are always strings in STAR files. + val = str(v) tValues.addRowValues(k, val) sfOut.writeTable('joboptions_values', tValues, computeFormat='left') @@ -970,16 +987,19 @@ def pipeline_tables(): @staticmethod def write_pipeline(pipeline_star, jobCounter=1, tables=None): + version = RelionStar.PIPELINE_VERSION with StarFile(pipeline_star, 'w') as sf: sf.writeTimeStamp() tGeneral = Table(['rlnPipeLineJobCounter']) tGeneral.addRowValues(jobCounter) - sf.writeTable('pipeline_general', tGeneral, singleRow=True) + sf.writeTable('pipeline_general', tGeneral, + singleRow=True, version=version) if tables: for name, t in tables.items(): if len(t): - sf.writeTable(f"pipeline_{name}", t, computeFormat=True) + sf.writeTable(f"pipeline_{name}", t, + computeFormat=True, version=version) @staticmethod def job_index(jobId): From fabe01fc6414887a2d95d530d270e34faa878ca2 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Mon, 24 Aug 2026 14:09:21 -0500 Subject: [PATCH 143/161] Make default output datatype more robust --- emtools/jobs/workflow.py | 1 + emtools/metadata/starfile.py | 4 ++-- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index 1908273..f3b87e1 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -96,6 +96,7 @@ def outputs(self): return self._outputs.values() def registerOutput(self, dataId, **kwargs): + kwargs.setdefault('datatype', 'File') data = Workflow.Data(self, dataId, **kwargs) self.wf.data[dataId] = data self._outputs[dataId] = data diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index de3e288..22779ce 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -1113,7 +1113,7 @@ def _table(name): for row in tOutput: job = wf.getJob(Path.rmslash(row.rlnPipeLineEdgeProcess)) nodeName = row.rlnPipeLineEdgeToNode - job.registerOutput(nodeName, datatype=nodes[nodeName]) + job.registerOutput(nodeName, datatype=nodes.get(nodeName, 'File')) if tInput := _table('input_edges'): for row in tInput: @@ -1157,7 +1157,7 @@ def workflow_to_pipeline(wf, pipelineStar): for o in job.outputs: tNodes.addRowValues( rlnPipeLineNodeName=o.id, - rlnPipeLineNodeTypeLabel=o['datatype'], + rlnPipeLineNodeTypeLabel=o.get('datatype', 'File'), rlnPipeLineNodeTypeLabelDepth=1 ) tOutput.addRowValues( From 6f0776f1bf3f930bb6095cad9e38d1d29e08b420 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 28 Aug 2026 11:33:22 -0500 Subject: [PATCH 144/161] Added some methods to WarpPopulation helper class --- emtools/metadata/misc.py | 42 ++++++++++++++++++++++++++++++++-------- 1 file changed, 34 insertions(+), 8 deletions(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index fcf449a..c0e162b 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -386,8 +386,8 @@ def __init__(self, populationFile): self._data = xmlDict['Population'] self.Name = self._data['Param']['@Value'] self.LastRefinementOptions = {e['@Name']: e['@Value'] for e in self._data['LastRefinementOptions']['Param']} - self.Sources = self._parseList(self._data['Sources'], 'Source') - self.Species = self._parseList(self._data['Species'], 'Species') + self.Sources = self._parseList(self._data.get('Sources'), 'Source') + self.Species = self._parseList(self._data.get('Species'), 'Species') def __repr__(self): r = f"Population: {self.Name}\n" @@ -403,20 +403,43 @@ def __repr__(self): return r def _parseList(self, data, key): + if not data: + return [] + suffix = f".{key.lower()}" def _parseItem(item): p = item['@Path'] name = os.path.basename(p).replace(suffix, '') return {'id': item['@GUID'], 'path': p, 'name': name} - - from pprint import pprint - d = data[key] - if isinstance(d, list): + d = data.get(key) if isinstance(data, dict) else None + if not d: + return [] + + if isinstance(d, list): return [_parseItem(item) for item in d] + return [_parseItem(d)] + + def getSource(self, nameOrIndex): + entry = None + if isinstance(nameOrIndex, int): + entry = self.Sources[nameOrIndex] else: - return [_parseItem(d)] + for s in self.Sources: + if s['name'] == nameOrIndex: + entry = s + break + + if not entry: + raise Exception(f"Source {nameOrIndex} not found") + + folder = os.path.dirname(self._filepath) + source_file = os.path.join(folder, entry['path']) + if not os.path.isfile(source_file): + raise Exception(f"Source file not found: {source_file}") + + return source_file def getSpecies(self, nameOrIndex): entry = None @@ -432,8 +455,11 @@ def getSpecies(self, nameOrIndex): raise Exception(f"Species {nameOrIndex} not found") folder = os.path.dirname(self._filepath) + species_file = os.path.join(folder, entry['path']) + if not os.path.isfile(species_file): + raise Exception(f"Species file not found: {species_file}") - return WarpSpecies(os.path.join(folder, entry['path'])) + return WarpSpecies(species_file) class Acquisition(dict): From 68deb8be050edc8ed538d6254332115220480d76 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 28 Aug 2026 11:33:36 -0500 Subject: [PATCH 145/161] Allow batch.call to pass cwd --- emtools/jobs/batch_manager.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 59a9e3c..46162ea 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -191,6 +191,8 @@ def load_all(self, fn=None): def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): """ If cwd is True, call the program from the batch directory. + If cwd is a string, call the program from that directory. + If cwd is False, use the current working directory. """ if isinstance(kwargs, dict): args = Args(kwargs).toList() @@ -209,8 +211,8 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): f.write(f"\n{cmd}\n") f.flush() kwargs = {'stderr': f, 'stdout': f} - if cwd: - kwargs['cwd'] = self.path + if cwd is not False: + kwargs['cwd'] = self.path if cwd is True else cwd subprocess.call(args, **kwargs) def tic(self, prefix=''): From a70e500cf40885d89d4307fc93a7d6bbbbd395ab Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 28 Aug 2026 11:33:56 -0500 Subject: [PATCH 146/161] Added method to convert Relion centered coordinates --- emtools/metadata/starfile.py | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 22779ce..8d65543 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -702,6 +702,24 @@ def centeredAngstToPixel(centered_angst, row, axis): return (float(centered_angst) / RelionStar.getTomoPixelSize(row) + RelionStar.reconstructedTomoSize(row, axis) / 2) + @staticmethod + def particleCoordsToPixel(particle_row, tomo_row): + """Return particle X/Y/Z coordinates in reconstructed tomogram pixels.""" + if hasattr(particle_row, 'rlnCoordinateX'): + return ( + float(particle_row.rlnCoordinateX), + float(particle_row.rlnCoordinateY), + float(particle_row.rlnCoordinateZ), + ) + return ( + RelionStar.centeredAngstToPixel( + particle_row.rlnCenteredCoordinateXAngst, tomo_row, 'rlnTomoSizeX'), + RelionStar.centeredAngstToPixel( + particle_row.rlnCenteredCoordinateYAngst, tomo_row, 'rlnTomoSizeY'), + RelionStar.centeredAngstToPixel( + particle_row.rlnCenteredCoordinateZAngst, tomo_row, 'rlnTomoSizeZ'), + ) + @staticmethod def getTomogram(row): """Return tomogram path, trying from different columns.""" From adecd67484717f3951d31b0fb700d62c8b0e454e Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 11 Sep 2026 11:22:33 -0500 Subject: [PATCH 147/161] Some fixes when clearing inputs --- emtools/jobs/workflow.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/workflow.py b/emtools/jobs/workflow.py index f3b87e1..5471e87 100644 --- a/emtools/jobs/workflow.py +++ b/emtools/jobs/workflow.py @@ -131,12 +131,16 @@ def addInputs(self, inputs): for i in inputs: self._inputs[i.id] = i - i.childs.append(self) + if self not in i.childs: + i.childs.append(self) def hasInput(self, inputId): return inputId in self._inputs def clearInputs(self): + for data in self._inputs.values(): + if self in data.childs: + data.childs.remove(self) self._inputs = {} def removeOutput(self, output_id): From 3d0a26f7fb5d9dea7fd1f619c66152cdb59ee690 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 11 Sep 2026 11:22:49 -0500 Subject: [PATCH 148/161] Added NumericList helper class --- emtools/jobs/__init__.py | 4 ++-- emtools/jobs/batch_manager.py | 37 +++++++++++++++++++++++++++++++++++ 2 files changed, 39 insertions(+), 2 deletions(-) diff --git a/emtools/jobs/__init__.py b/emtools/jobs/__init__.py index 88b1afd..c230e72 100644 --- a/emtools/jobs/__init__.py +++ b/emtools/jobs/__init__.py @@ -15,10 +15,10 @@ # ************************************************************************** from .pipeline import Pipeline -from .batch_manager import (Args, Batch, Vars, +from .batch_manager import (Args, NumericList, Batch, Vars, BatchManager, MdocBatchManager, TsStarBatchManager) from .workflow import Workflow __all__ = ["Pipeline", "Workflow", - "Args", "Vars", + "Args", "NumericList", "Vars", "Batch", "BatchManager", "MdocBatchManager", "TsStarBatchManager"] diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 46162ea..b16cc59 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -120,6 +120,43 @@ def subset(self, prefix, new_prefix='', filters=None, return result +class NumericList(list): + """ List of integers parsed from a string, subclassing 'list' so the + result can be used directly wherever a list of ints is expected. + + Accepted syntax (used e.g. for GPU ids, tilt/frame indices, etc.): + - Individual values, separated by spaces and/or commas: + '1,2,3' or '1 2 3' or '1, 2 3' + - Inclusive ranges, written as 'start-end' (no spaces around the + dash), which get expanded to every value in between: + '10-12' -> 10, 11, 12 + '1, 5-7, 9' -> 1, 5, 6, 7, 9 + """ + + _RANGE = re.compile(r'^(?P\d+)-(?P\d+)$') + + @classmethod + def fromString(cls, string): + values = cls() + for token in re.split(r'[,\s]+', str(string).strip()): + if not token: + continue + m = cls._RANGE.match(token) + if m: + start, end = int(m.group('start')), int(m.group('end')) + if end < start: + raise ValueError( + f"Invalid range '{token}': end must be >= start.") + values.extend(range(start, end + 1)) + else: + try: + values.append(int(token)) + except ValueError: + raise ValueError( + f"Invalid numeric value or range '{token}' in '{string}'.") + return values + + class Vars: """ Handle variable definitions, either from input dict or from os.environ. From 8335ca7aabe17bd7bcfc12b501e676f7e79fe4a3 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 11 Sep 2026 11:23:12 -0500 Subject: [PATCH 149/161] Working on fourier_crop and image display --- emtools/image/__main__.py | 71 ++++++++++++++++++++- emtools/image/thumbnail.py | 122 +++++++++++++++++++++++++++++++++++-- 2 files changed, 186 insertions(+), 7 deletions(-) diff --git a/emtools/image/__main__.py b/emtools/image/__main__.py index 5254c7f..5f6de66 100644 --- a/emtools/image/__main__.py +++ b/emtools/image/__main__.py @@ -15,17 +15,82 @@ # ************************************************************************** import argparse -from .thumbnail import Image +import mrcfile +import numpy as np + +from .thumbnail import Image, Thumbnail + + + +def show2D(inputFile, **kwargs): + import matplotlib.pyplot as plt + size = kwargs.get('size', 512) + print("size: ", size) + thumb = Thumbnail(max_size=(size, size), contrast_factor=0.15, std_threshold=1) + array = Image.get_array(inputFile) + plt.imshow(thumb.from_array(array), cmap='gray') + plt.axis('off') # Optional: Turn off axis labels and ticks + plt.show() + + +def fourier_crop(inputFile, scale): + array = Image.get_array(inputFile) + return Image.fourier_crop(inputFile, scale) + + +def save_array(array, outputFile): + if outputFile.endswith('.mrc'): + with mrcfile.new(outputFile, overwrite=True) as mrc: + mrc.set_data(array) + elif outputFile.endswith('.png'): + thumb = Thumbnail(max_size=None, contrast_factor=0.15)#, std_threshold=1) + pil_img = thumb.from_array(array) + pil_img.save(outputFile) + else: + raise ValueError(f"Unsupported file type: {outputFile}") + return outputFile + + +def compare_images(inputFile1, inputFile2): + array1 = Image.get_array(inputFile1) + array2 = Image.get_array(inputFile2) + return np.array_equal(array1, array2) + def main(): p = argparse.ArgumentParser() p.add_argument('path', metavar="IMAGE_PATH", help="Image path") - + p.add_argument('--show', '-s', action='store_true', + help="Show the image") + p.add_argument('--max-size', '-m', type=int, default=512, + help="Size of the image") + p.add_argument('--bin', '-b', type=float, default=None, + help="Bin factor for the image, using Fourier cropping. Bin 1 means no cropping, bin 2 means half the original size, etc.") + p.add_argument('--output', '-o', type=str, default=None, + help="Output file name") + p.add_argument('--compare', '-c', + help="Compare the image with another image.") args = p.parse_args() - print(Image.get_dimensions(args.path)) + if args.show: + show2D(args.path, size=args.max_size) + elif args.bin: + if args.bin > 1: + scale = 1 / args.bin + array = fourier_crop(args.path, scale) + if args.output: + save_array(array, args.output) + else: + print("Bin factor must be greater than 1") + elif args.compare: + if compare_images(args.path, args.compare): + print("Images are the same") + else: + print("Images are different") + else: + print(Image.get_dimensions(args.path)) if __name__ == '__main__': diff --git a/emtools/image/thumbnail.py b/emtools/image/thumbnail.py index 47d7bc2..f20b036 100644 --- a/emtools/image/thumbnail.py +++ b/emtools/image/thumbnail.py @@ -22,7 +22,7 @@ import tifffile import PIL -from PIL import Image +from PIL import ImageFilter, ImageOps from emtools.datatypes import STACK_2D, VOLUME @@ -75,10 +75,11 @@ def from_pil(self, pil_img): self.scale = scale if self.contrast_factor is not None: - pil_img = PIL.ImageOps.autocontrast(pil_img, cutoff=self.contrast_factor) + pil_img = ImageOps.autocontrast(pil_img, cutoff=self.contrast_factor) if self.gaussian_radius is not None: - pil_img = pil_img.filter(PIL.ImageFilter.GaussianBlur(radius=self.gaussian_radius)) + pil_img = pil_img.filter( + ImageFilter.GaussianBlur(radius=self.gaussian_radius)) return self.__format(pil_img) @@ -101,7 +102,7 @@ def from_array(self, imageArray): array = imageArray else: if self.std_threshold > 0: - array = np.array(imageArray) + array = np.asarray(imageArray, dtype=np.float64) imean = array.mean() isd = array.std() isdTh = self.std_threshold * isd @@ -352,3 +353,116 @@ def get_dimensions(imagePath): n = len(tif.pages) y, x = tif.pages[0].shape return (x, y, n) if n > 1 else (x, y) + + @staticmethod + def get_array(imagePath): + imageLower = imagePath.lower() + if imageLower.endswith('.mrc') or imageLower.endswith('.mrcs'): + with mrcfile.open(imagePath) as mrc: + return mrc.data + elif (imageLower.endswith('.tif') or + imageLower.endswith('.tiff') or + imageLower.endswith('.eer') or + imageLower.endswith('.gain')): + with tifffile.TiffFile(imagePath) as tif: + return tif.asarray() + return None + + @staticmethod + def _fourier_output_size(size, scale): + """Return rounded output size for a given scale factor.""" + return max(1, int(round(size * scale))) + + @staticmethod + def _fourier_axis_slices(in_size, out_size): + """Return source/destination slices for DC-centered crop or pad. + + The DC component lives at in_size // 2 in the shifted spectrum, so the + crop/pad is centered there rather than on the array geometric center. + This keeps even and odd input/output sizes aligned correctly. + """ + if out_size <= in_size: + src_start = in_size // 2 - out_size // 2 + dst_start = 0 + length = out_size + else: + src_start = 0 + dst_start = out_size // 2 - in_size // 2 + length = in_size + return src_start, dst_start, length + + @staticmethod + def _is_volume(imagePath, array): + """Return True when a 3D array should be treated as a volume.""" + if array.ndim != 3: + return False + + if Path.exists(imagePath): + metadata = Image.get_metadata(imagePath) + if metadata and metadata.get('dataType') == VOLUME: + return True + + z, y, x = array.shape + return z == y == x + + @staticmethod + def _fourier_rescale(image, scale): + """Resize an n-D image by Fourier cropping or zero-padding.""" + shape = image.shape + new_shape = tuple(Image._fourier_output_size(s, scale) for s in shape) + + if new_shape == shape: + return np.array(image, copy=True) + + spectrum = np.fft.fftshift(np.fft.fftn(image)) + resized = np.zeros(new_shape, dtype=spectrum.dtype) + + src_slices = [] + dst_slices = [] + for in_size, out_size in zip(shape, new_shape): + src_start, dst_start, length = Image._fourier_axis_slices( + in_size, out_size) + src_slices.append(slice(src_start, src_start + length)) + dst_slices.append(slice(dst_start, dst_start + length)) + + resized[tuple(dst_slices)] = spectrum[tuple(src_slices)] + + # Preserve average intensity when changing the number of pixels. + amp = np.prod(new_shape) / np.prod(shape) + result = np.real(np.fft.ifftn(np.fft.ifftshift(resized * amp))) + + if np.issubdtype(image.dtype, np.floating): + return result.astype(image.dtype, copy=False) + return result + + @staticmethod + def fourier_crop(imagePath, scale): + """Resize an image by Fourier cropping or padding. + + Args: + imagePath: Path to a supported image file. + scale: Output-size multiplier per axis (e.g. 0.5 halves the size). + + Returns: + Rescaled numpy array with the same dimensionality as the input. + """ + array = Image.get_array(imagePath) + if array is None: + raise ValueError("Unsupported image format: %s" % imagePath) + if scale <= 0: + raise ValueError("Scale must be positive, got: %s" % scale) + + if array.ndim == 2: + return Image._fourier_rescale(array, scale) + + if array.ndim == 3: + if Image._is_volume(imagePath, array): + return Image._fourier_rescale(array, scale) + + return np.stack( + [Image._fourier_rescale(array[i], scale) + for i in range(array.shape[0])], + axis=0, + ) + + raise ValueError("Expected 2D or 3D image, got shape: %s" % (array.shape,)) From 934ccd9c5d06faf1ea8e859aad48278febd7b51e Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 15 Sep 2026 09:17:29 -0500 Subject: [PATCH 150/161] Added function to merge star files --- emtools/scripts/emt_star.py | 63 ++++++++++++++++++++++++++++++++++++- 1 file changed, 62 insertions(+), 1 deletion(-) diff --git a/emtools/scripts/emt_star.py b/emtools/scripts/emt_star.py index 31318ce..0da65a2 100755 --- a/emtools/scripts/emt_star.py +++ b/emtools/scripts/emt_star.py @@ -145,6 +145,62 @@ def _processTable(sf, tableName, actions, subset=None): return tableInfo, list(table), singleRow +def mergeStarFiles(pattern, tableName, output=None): + files = sorted(glob(pattern)) + if not files: + raise FileNotFoundError(f"No STAR files match pattern: {pattern}") + + tableInfo = None + refColumnNames = None + singleRow = None + rows = [] + + for starPath in files: + with StarFile(starPath) as sf: + if tableName not in sf.getTableNames(): + raise ValueError( + f"Table '{tableName}' not found in {starPath}") + + info = sf.getTableInfo(tableName) + columnNames = info.getColumnNames() + fileSingleRow = sf._singleRow + + if tableInfo is None: + tableInfo = info + refColumnNames = columnNames + singleRow = fileSingleRow + else: + if len(columnNames) != len(refColumnNames): + raise ValueError( + f"Column count mismatch in {starPath}: " + f"expected {len(refColumnNames)}, got {len(columnNames)}") + if columnNames != refColumnNames: + raise ValueError( + f"Column names mismatch in {starPath}: " + f"expected {refColumnNames}, got {columnNames}") + if fileSingleRow != singleRow: + raise ValueError( + f"Table layout mismatch in {starPath}: " + f"expected {'single-row' if singleRow else 'loop'} " + f"format") + + for row in sf.iterTable(tableName): + rows.append(row) + + if len(rows) > 1: + singleRow = False + + closeOutput = output is not None + out = open(output, 'w') if closeOutput else sys.stdout + try: + with StarFile(out, closeFile=closeOutput) as sfOut: + sfOut.writeTimeStamp() + _writeProcessedTable(sfOut, tableName, tableInfo, rows, singleRow) + finally: + if closeOutput: + out.close() + + def operateStarFile(inputStar, operations, subset=None, output=None): if not operations: raise ValueError("At least one --operate action is required") @@ -304,7 +360,7 @@ def _writeStar(minSize=0): def main(): p = argparse.ArgumentParser(prog='emt-star') p.add_argument('input', - help="Input STAR file. ") + help="Input STAR file, or a glob pattern when using --merge.") p.add_argument('--group_by', '-g', nargs=2, metavar=('TABLE', 'COLUMN'), help="Count rows grouped by a given label") @@ -330,6 +386,9 @@ def main(): "UPDATE, FILTER, or DROP. For UPDATE, EXPR is comma-separated " "column=expression assignments. For FILTER, EXPR is a boolean " "expression per row. For DROP, EXPR is ignored.") + outputMode.add_argument('--merge', '-m', metavar='TABLE', + help="Merge TABLE from every STAR file matching the input " + "glob pattern into one table.") p.add_argument('--subset', '-n', type=int, default=None, metavar='N', help="Process at most N rows per table (for debugging)") p.add_argument('--output', '-o', default=None, metavar='FILE', @@ -341,6 +400,8 @@ def main(): if args.operate: operateStarFile(inputStar, _parseOperateArgs(args.operate), subset=args.subset, output=args.output) + elif args.merge: + mergeStarFiles(inputStar, args.merge, output=args.output) elif args.group_by: table, column = args.group_by groupBy(inputStar, table, column) From d0075c3a0a7285e85b270f657b78ee9b09b0d149 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?March=C3=A1n?= Date: Tue, 15 Sep 2026 10:32:02 -0500 Subject: [PATCH 151/161] iterTable define types for NominalStageTiltAngle and TiltAxisAngle --- emtools/jobs/batch_manager.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index b16cc59..5648f58 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -469,5 +469,10 @@ def generate(self): tsName = tsRow.rlnTomoName tsMdoc = tsRow.rlnTomoMdocFile with StarFile(tsRow.rlnTomoTiltSeriesStarFile) as sf: - items = [row._asdict() for row in sf.iterTable(tsName)] + # Force float typing for angle columns: guessing from the + # first row alone can lock the column to int and then break + # on later rows with decimal values (e.g. '3.00013'). + items = [row._asdict() for row in sf.iterTable( + tsName, types={'rlnTomoNominalStageTiltAngle': float, + 'rlnTomoNominalTiltAxisAngle': float})] yield self._createBatch(items, tsName=tsName, tsMdoc=tsMdoc, rowDict=tsRow._asdict()) From 0006987bac95e545e88d2ce7a7525ff8b4147117 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 18 Sep 2026 15:27:10 -0500 Subject: [PATCH 152/161] Improved system info functions --- emtools/utils/path.py | 2 +- emtools/utils/system.py | 46 +++++++++++++++++++++++++++++++++++++++++ setup.py | 1 + 3 files changed, 48 insertions(+), 1 deletion(-) diff --git a/emtools/utils/path.py b/emtools/utils/path.py index 806dc64..243f7b7 100644 --- a/emtools/utils/path.py +++ b/emtools/utils/path.py @@ -37,7 +37,7 @@ TEXT_EXT = ['txt', 'log', 'err', 'out', 'json', 'csv', 'star', 'sh', 'out', 'err', 'bashrc', 'xml', 'script', 'settings', 'job', 'tomostar', 'mdoc', - 'population', 'species', + 'population', 'species', 'yaml', 'id', 'aln', 'com', 'rawtlt', 'tlt', 'xf', 'xtilt'] diff --git a/emtools/utils/system.py b/emtools/utils/system.py index f907e0c..4c5a7dd 100644 --- a/emtools/utils/system.py +++ b/emtools/utils/system.py @@ -21,6 +21,7 @@ """ import socket +import shutil import platform import psutil import time @@ -102,6 +103,51 @@ def hostname(): """ Return the hostname. """ return socket.gethostname() + @staticmethod + def distro(): + """ Return a human-readable OS distribution name and version. + On Linux, it reads /etc/os-release (PRETTY_NAME); on macOS it uses + the product version; falls back to platform.platform() otherwise. + """ + system = platform.system() + + if system == 'Linux': + info = {} + try: + with open('/etc/os-release') as f: + for line in f: + line = line.strip() + if not line or line.startswith('#') or '=' not in line: + continue + key, _, value = line.partition('=') + info[key] = value.strip().strip('"') + except (FileNotFoundError, OSError): + pass + + name = info.get('PRETTY_NAME') or info.get('NAME') + return name or f"Linux {platform.release()}" + + elif system == 'Darwin': + version, _, _ = platform.mac_ver() + return f"macOS {version}" if version else "macOS" + + return platform.platform() + + @staticmethod + def kernel(): + """ Return the kernel release (e.g. what 'uname -r' would print). """ + return platform.release() + + @staticmethod + def disk(path='/'): + """ Return disk usage (in bytes) for the filesystem containing path. + Keys: 'total', 'used', 'free'. Returns None if it can't be read. """ + try: + usage = shutil.disk_usage(path) + except OSError: + return None + return {'total': usage.total, 'used': usage.used, 'free': usage.free} + class GpuMonitor(threading.Thread): """ Monitor GPU utilization. diff --git a/setup.py b/setup.py index 35b91cf..acec377 100644 --- a/setup.py +++ b/setup.py @@ -74,6 +74,7 @@ entry_points={ # Optional 'console_scripts': [ 'emt-ps = emtools.scripts.emt_ps:main', + 'emt-sysinfo = emtools.scripts.emt_sysinfo:main', 'emt-files = emtools.scripts.emt_files:main', 'emt-epu = emtools.scripts.emt_epu:main', 'emt-beamshifts = emtools.scripts.emt_beamshifts:main', From 72246308e5040102a214f8e32d9332fba165af9c Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 22 Sep 2026 08:50:38 -0500 Subject: [PATCH 153/161] Missed sysinfo script --- emtools/scripts/emt_sysinfo.py | 96 ++++++++++++++++++++++++++++++++++ 1 file changed, 96 insertions(+) create mode 100755 emtools/scripts/emt_sysinfo.py diff --git a/emtools/scripts/emt_sysinfo.py b/emtools/scripts/emt_sysinfo.py new file mode 100755 index 0000000..5a5768e --- /dev/null +++ b/emtools/scripts/emt_sysinfo.py @@ -0,0 +1,96 @@ +#!/usr/bin/env python +# ************************************************************************** +# * +# * Authors: J.M. de la Rosa Trevin (delarosatrevin@gmail.com) +# * +# * This program is free software; you can redistribute it and/or modify +# * it under the terms of the GNU General Public License as published by +# * the Free Software Foundation; either version 3 of the License, or +# * (at your option) any later version. +# * +# * This program is distributed in the hope that it will be useful, +# * but WITHOUT ANY WARRANTY; without even the implied warranty of +# * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the +# * GNU General Public License for more details. +# * +# ************************************************************************** + +""" +Quick summary of this workstation: OS/distro, kernel, CPUs, RAM, disk and +GPUs. Mostly a thin, friendly wrapper around emtools.utils.System. +""" + +import json +import argparse +import platform + +from emtools.utils import Color, Pretty, System + + +def get_info(disk_path='/'): + """ Collect a dict with a quick summary of this workstation. + Reuses emtools.utils.System for all the underlying lookups. """ + return { + 'hostname': System.hostname(), + 'os': System.distro(), + 'kernel': System.kernel(), + 'arch': platform.machine(), + 'cpus': System.cpus(), + 'memory_gb': System.memory(), + 'disk': System.disk(disk_path), + 'disk_path': disk_path, + 'gpus': System.gpus(), + } + + +def print_info(info): + """ Print a nicely formatted report from the dict returned by get_info(). """ + width = 70 + title = f" SYSTEM INFO: {info['hostname']} " + print(Color.bold(title.center(width, '='))) + print(f" {'OS':<10}: {info['os']}") + print(f" {'Kernel':<10}: {info['kernel']} ({info['arch']})") + print(f" {'CPUs':<10}: {info['cpus']}") + print(f" {'Memory':<10}: {info['memory_gb']} GB") + + disk = info['disk'] + if disk: + pct = 100 * disk['used'] / disk['total'] if disk['total'] else 0 + print(f" {'Disk (' + info['disk_path'] + ')':<10}: " + f"{Pretty.size(disk['total'])} total, " + f"{Pretty.size(disk['free'])} free ({pct:.0f}% used)") + + gpus = info['gpus'] + print(f" {'GPUs':<10}: {len(gpus)}") + for g in gpus: + line = (f" [{g.get('index', '?')}] {g.get('name', 'Unknown GPU'):<28} " + f"{g.get('memory.total', '?'):>10} total " + f"{g.get('memory.used', '?'):>10} used " + f"driver {g.get('driver_version', '?')}") + print(Color.cyan(line)) + + print(Color.bold('=' * width)) + + +def main(): + p = argparse.ArgumentParser( + prog='emt-sysinfo', + description="Print a quick summary of this workstation's hardware " + "and OS: Linux distro/version, kernel, CPUs, RAM, disk " + "usage and GPUs.") + p.add_argument('--json', '-j', action='store_true', + help='Print the info as JSON instead of the formatted report.') + p.add_argument('--disk-path', default='/', + help='Path used to report disk usage (default: /).') + + args = p.parse_args() + info = get_info(disk_path=args.disk_path) + + if args.json: + print(json.dumps(info, indent=2)) + else: + print_info(info) + + +if __name__ == '__main__': + main() From 5366bf0abaf607e6647a33d306fe0a6c05e37a9e Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Tue, 22 Sep 2026 11:51:49 -0500 Subject: [PATCH 154/161] Return subprocess exit status and created Population digest --- emtools/jobs/batch_manager.py | 4 +++- emtools/metadata/misc.py | 21 +++++++++++++++++++++ 2 files changed, 24 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 5648f58..da2c5b4 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -230,6 +230,8 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): If cwd is True, call the program from the batch directory. If cwd is a string, call the program from that directory. If cwd is False, use the current working directory. + + Returns the exit code of the program. """ if isinstance(kwargs, dict): args = Args(kwargs).toList() @@ -250,7 +252,7 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): kwargs = {'stderr': f, 'stdout': f} if cwd is not False: kwargs['cwd'] = self.path if cwd is True else cwd - subprocess.call(args, **kwargs) + return subprocess.call(args, **kwargs) def tic(self, prefix=''): self._timer.tic() diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index c0e162b..bcb2324 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -14,6 +14,7 @@ # * # ************************************************************************** +import hashlib import os import pathlib from datetime import datetime, timedelta @@ -421,6 +422,26 @@ def _parseItem(item): return [_parseItem(item) for item in d] return [_parseItem(d)] + def digest(self): + """ Content digest of this population and the species it points to. + + Refinements rewrite the species files while leaving the .population + file itself untouched, so both are needed to tell two states apart. + """ + md5 = hashlib.md5() + folder = os.path.dirname(self._filepath) + entries = [('', self._filepath)] + entries += [(s['path'], os.path.join(folder, s['path'])) + for s in self.Species] + + for relpath, path in entries: + md5.update(relpath.encode()) + if os.path.isfile(path): + with open(path, 'rb') as f: + md5.update(f.read()) + + return md5.hexdigest() + def getSource(self, nameOrIndex): entry = None if isinstance(nameOrIndex, int): From 0a57665c300cd0284553408a2eb75def78b69edd Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Fri, 25 Sep 2026 11:22:07 -0500 Subject: [PATCH 155/161] Fixed unused tableNames cache --- emtools/metadata/starfile.py | 6 ++--- emtools/tests/test_metadata.py | 47 ++++++++++++++++++++++++++++++++++ 2 files changed, 50 insertions(+), 3 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 8d65543..f3a87c1 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -97,9 +97,9 @@ def getTableNames(self): # While searching for a data line, we will store the offsets # for any data_ line that we find if line.startswith('data_'): - tn = line.strip().replace('data_', '') - self._offsets[tn] = offset - self._names.append(tn) + ds = line.strip() + self._offsets[ds] = offset + self._names.append(ds.replace('data_', '', 1)) offset = f.tell() line = f.readline() diff --git a/emtools/tests/test_metadata.py b/emtools/tests/test_metadata.py index 129651c..56797a4 100644 --- a/emtools/tests/test_metadata.py +++ b/emtools/tests/test_metadata.py @@ -373,6 +373,53 @@ def _pipeline(monitor): #self.__test_star_streaming(_pipeline, inputStreaming=True) self.__test_star_streaming(_pipeline, inputStreaming=False) + def test_table_offsets_cache(self): + ftmp = tempfile.NamedTemporaryFile(mode='w', delete=False, suffix='.star') + ftmp.write(""" +# version 50001 + +data_general + +_rlnTomoSubTomosAre2DStacks 1 + +# version 50001 + +data_optics + +loop_ +_rlnOpticsGroup #1 +_rlnOpticsGroupName #2 + 1 opticsGroup1 + 2 opticsGroup2 + +# version 50001 + +data_particles + +loop_ +_rlnTomoName #1 +_rlnTomoParticleId #2 +TS_01 1 +TS_01 2 +TS_02 3 +""") + ftmp.close() + + with StarFile(ftmp.name) as sf: + self.assertEqual(sf.getTableNames(), ['general', 'optics', 'particles']) + self.assertEqual(set(sf._offsets), + {'data_general', 'data_optics', 'data_particles'}) + + # Read out of order to exercise seeking from cached offsets + self.assertEqual(len(sf.getTable('particles')), 3) + self.assertEqual(sf.getTable('general')[0].rlnTomoSubTomosAre2DStacks, 1) + self.assertEqual(sf.getTableSize('optics'), 2) + self.assertEqual(sf.getTableInfo('particles').getColumnNames(), + ['rlnTomoName', 'rlnTomoParticleId']) + self.assertIsNone(sf.getTable('missing')) + + os.unlink(ftmp.name) + class TestTable(unittest.TestCase): """Tests for Table.update and Table.filter.""" From 081d48323acbbfd796bbb67dc4708b4d45bd2346 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 1 Oct 2026 12:58:57 -0500 Subject: [PATCH 156/161] Parse single letter args --- emtools/jobs/batch_manager.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index da2c5b4..803d15f 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -52,7 +52,8 @@ def fromString(string): @staticmethod def fromList(iterable): - r = re.compile(r"^-{1,2}[a-zA-Z][a-zA-Z0-9_-]+$") + # Also single letter options (e.g. Relion's --K or cryolo's -t) + r = re.compile(r"^-{1,2}[a-zA-Z][a-zA-Z0-9_-]*$") def _is_arg(v): return r.match(v) is not None From 9c4ca353f12a37d05ccb35099241f651dadb5649 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 1 Oct 2026 12:59:26 -0500 Subject: [PATCH 157/161] Some fixes when monitoring STAR files --- emtools/metadata/starfile.py | 70 +++++++++++++++++++++++++++++------- 1 file changed, 58 insertions(+), 12 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index f3a87c1..33f0ae3 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -468,17 +468,37 @@ class StarMonitor: """ Monitor a STAR file for changes and return new items in a given table. - This class will subclass OrderedDict to hold a clone of each new element. - It will also keep internally the last access timestamp to prevent loading - the STAR file if it has not been modified since the last check. + It will keep internally the last modification time of the file to + prevent loading the STAR file if it has not been modified since the + last check. The file might be written (appended or re-written) while + it is read, so reads that fail or where the file changed during the + read are discarded and retried in the next check. """ def __init__(self, fileName, tableName, rowKeyFunc, **kwargs): + """ + Args: + fileName: STAR file to monitor + tableName: table to read new rows from + rowKeyFunc: function returning a unique key for each row + Keyword Args: + wait: seconds to wait between checks (default 30) + timeout: stop after these seconds without new items (default 300) + blacklist: rows that will not be returned (e.g. already processed) + tableKwargs: dict with extra args to read the table + (e.g. {'guessType': False}, see StarFile.iterTable) + maxErrors: raise the reading error after these consecutive + failed reads (default 10) + """ self._seenItems = set() self.fileName = fileName self._tableName = tableName self._rowKeyFunc = rowKeyFunc self._wait = kwargs.get('wait', 30) self._timeout = timedelta(seconds=kwargs.get('timeout', 300)) + self._tableKwargs = kwargs.get('tableKwargs', {}) + self._maxErrors = kwargs.get('maxErrors', 10) + self._errors = 0 + self._lastMTime = None # File modification time of last read self.lastCheck = None # Last timestamp when input was checked self.lastUpdate = None # Last timestamp when new items were found @@ -490,20 +510,46 @@ def __init__(self, fileName, tableName, rowKeyFunc, **kwargs): for row in blacklist: self._seenItems.add(self._rowKeyFunc(row)) + def _readNewRows(self): + """ Read the table and return new rows (not marked as seen yet). """ + newRows = [] + newKeys = set() + with StarFile(self.fileName) as sf: + for row in sf.iterTable(self._tableName, **self._tableKwargs): + rowKey = self._rowKeyFunc(row) + if rowKey not in self._seenItems and rowKey not in newKeys: + newKeys.add(rowKey) + newRows.append(row) + return newRows, newKeys + def update(self): newRows = [] if os.path.exists(self.fileName): now = datetime.now() - mTime = datetime.fromtimestamp(os.path.getmtime(self.fileName)) - - if self.lastCheck is None or mTime > self.lastCheck: - with StarFile(self.fileName) as sf: - for row in sf.iterTable(self._tableName): - rowKey = self._rowKeyFunc(row) - if rowKey not in self._seenItems: - self._seenItems.add(rowKey) - newRows.append(row) + mTime = os.path.getmtime(self.fileName) + + if mTime != self._lastMTime: + try: + newRows, newKeys = self._readNewRows() + if os.path.getmtime(self.fileName) != mTime: + # Modified while reading, read again in next check + # (after the file is complete) + newRows = [] + else: + self._seenItems.update(newKeys) + # With coarse mtime resolution (e.g. 1s in NFS), a + # later write could keep the same mtime, so read + # again if the file was modified very recently + if time.time() - mTime > 2: + self._lastMTime = mTime + self._errors = 0 + except Exception: + # The file might be partially written, try in next check + newRows = [] + self._errors += 1 + if self._errors >= self._maxErrors: + raise self.lastCheck = now if newRows: From 55969aca09748403f101f59fcebd3036cda7a381 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 1 Oct 2026 14:43:30 -0500 Subject: [PATCH 158/161] Fixed partial writes on star file monitoring --- emtools/metadata/starfile.py | 54 ++++++++++++++++++++---------------- 1 file changed, 30 insertions(+), 24 deletions(-) diff --git a/emtools/metadata/starfile.py b/emtools/metadata/starfile.py index 33f0ae3..348814d 100644 --- a/emtools/metadata/starfile.py +++ b/emtools/metadata/starfile.py @@ -471,8 +471,9 @@ class StarMonitor: It will keep internally the last modification time of the file to prevent loading the STAR file if it has not been modified since the last check. The file might be written (appended or re-written) while - it is read, so reads that fail or where the file changed during the - read are discarded and retried in the next check. + it is read: reads that fail are retried in the next check and, if the + file changed during the read, the last row (that might be incomplete) + is left for the next check. """ def __init__(self, fileName, tableName, rowKeyFunc, **kwargs): """ @@ -510,17 +511,24 @@ def __init__(self, fileName, tableName, rowKeyFunc, **kwargs): for row in blacklist: self._seenItems.add(self._rowKeyFunc(row)) - def _readNewRows(self): - """ Read the table and return new rows (not marked as seen yet). """ - newRows = [] - newKeys = set() + def _readTable(self): + """ Read the table rows as (key, row) pairs. """ with StarFile(self.fileName) as sf: - for row in sf.iterTable(self._tableName, **self._tableKwargs): - rowKey = self._rowKeyFunc(row) - if rowKey not in self._seenItems and rowKey not in newKeys: - newKeys.add(rowKey) - newRows.append(row) - return newRows, newKeys + return [(self._rowKeyFunc(row), row) + for row in sf.iterTable(self._tableName, **self._tableKwargs)] + + def _newRows(self, rows, skipLast=False): + """ Return rows not seen before and mark them as seen. + If skipLast, the last row is ignored (and not marked). """ + if skipLast and rows: + rows = rows[:-1] + + newRows = [] + for rowKey, row in rows: + if rowKey not in self._seenItems: + self._seenItems.add(rowKey) + newRows.append(row) + return newRows def update(self): newRows = [] @@ -531,18 +539,16 @@ def update(self): if mTime != self._lastMTime: try: - newRows, newKeys = self._readNewRows() - if os.path.getmtime(self.fileName) != mTime: - # Modified while reading, read again in next check - # (after the file is complete) - newRows = [] - else: - self._seenItems.update(newKeys) - # With coarse mtime resolution (e.g. 1s in NFS), a - # later write could keep the same mtime, so read - # again if the file was modified very recently - if time.time() - mTime > 2: - self._lastMTime = mTime + rows = self._readTable() + # If modified while reading, the last row might be + # incomplete, so leave it for the next check + modified = os.path.getmtime(self.fileName) != mTime + newRows = self._newRows(rows, skipLast=modified) + # With coarse mtime resolution (e.g. 1s in NFS), a + # later write could keep the same mtime, so read + # again if the file was modified very recently + if not modified and time.time() - mTime > 2: + self._lastMTime = mTime self._errors = 0 except Exception: # The file might be partially written, try in next check From 8e0e8a4baf787a942b746db90bc8a0d19ac6db55 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?March=C3=A1n?= Date: Tue, 6 Oct 2026 15:26:40 -0500 Subject: [PATCH 159/161] Add imagecodecs requirement tifffile needs imagecodecs to read and write LZW compressed TIFF files, e.g. EPU .gain references upsampled for EER super-resolution in emwrap. Co-Authored-By: Claude Opus 5.5 --- requirements.txt | 1 + 1 file changed, 1 insertion(+) diff --git a/requirements.txt b/requirements.txt index 7a15b9e..351e2da 100644 --- a/requirements.txt +++ b/requirements.txt @@ -4,3 +4,4 @@ Pillow>=9.0.1 xmltodict psutil tifffile +imagecodecs From 57ac31668b0f36c21450f2c23c2d9f12bbe56db8 Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 8 Oct 2026 14:27:13 -0500 Subject: [PATCH 160/161] Added more logs --- emtools/jobs/batch_manager.py | 46 +++++++++++++++++++++++++++++------ 1 file changed, 39 insertions(+), 7 deletions(-) diff --git a/emtools/jobs/batch_manager.py b/emtools/jobs/batch_manager.py index 803d15f..da2df13 100644 --- a/emtools/jobs/batch_manager.py +++ b/emtools/jobs/batch_manager.py @@ -232,6 +232,10 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): If cwd is a string, call the program from that directory. If cwd is False, use the current working directory. + The command is written to the log file (by default batch.log), in a + way that it can be copied to re-run it manually (see logCommand), + followed by the program output and its exit code. + Returns the exit code of the program. """ if isinstance(kwargs, dict): @@ -243,17 +247,45 @@ def call(self, program, kwargs, logfile=None, verbose=False, cwd=True): else: raise Exception("Expecting dict or list as arguments") - args.insert(0, program) + args = [str(a) for a in [program] + args] logfile = logfile or self.join('batch.log') + workDir = os.getcwd() if cwd is False else (self.path if cwd is True else cwd) + self.log(f"{Color.green(args[0])} {Color.bold(shlex.join(args[1:]))}") with open(logfile, 'a') as f: - cmd = self.log(f"{Color.green(args[0])} {Color.bold(' '.join(args[1:]))}") - f.write(f"\n{cmd}\n") + f.write(self._commandText(args, workDir)) f.flush() - kwargs = {'stderr': f, 'stdout': f} - if cwd is not False: - kwargs['cwd'] = self.path if cwd is True else cwd - return subprocess.call(args, **kwargs) + returncode = subprocess.call(args, stderr=f, stdout=f, cwd=workDir) + f.write(f"# Exit code: {returncode}\n") + return returncode + + def _commandText(self, args, cwd, input=None): + """ Command text to re-run it: cd to the working dir and run the + command (quoted arguments), with the standard input if any. """ + cmd = f"cd {shlex.quote(os.path.abspath(cwd))} && {shlex.join(args)}" + if input is not None: + cmd += f" <<'EOF'\n{input.rstrip()}\nEOF" + return f"\n# {Pretty.now()}{self._logId} command:\n{cmd}\n" + + def logCommand(self, args, cwd=None, input=None, returncode=None, + output=None, logfile=None): + """ Write to the log file (by default batch.log) a command executed + outside this batch's call method, e.g. one that needs its output. + + Args: + args: command arguments, including the program + cwd: working directory where the command was run (default: current) + input: text passed as standard input + returncode: exit code of the command + output: output of the command (e.g. stdout and stderr) + """ + args = [str(a) for a in args] + with open(logfile or self.join('batch.log'), 'a') as f: + f.write(self._commandText(args, cwd or os.getcwd(), input=input)) + if output: + f.write(output if output.endswith('\n') else output + '\n') + if returncode is not None: + f.write(f"# Exit code: {returncode}\n") def tic(self, prefix=''): self._timer.tic() From 4e6a32ed197ceea888ec70c085aa3cc2939ca51c Mon Sep 17 00:00:00 2001 From: "J.M de la Rosa Trevin" Date: Thu, 8 Oct 2026 14:27:32 -0500 Subject: [PATCH 161/161] Added more functions to WarpXml helper class --- emtools/metadata/misc.py | 35 +++++++++++++++++++++++++++++++---- 1 file changed, 31 insertions(+), 4 deletions(-) diff --git a/emtools/metadata/misc.py b/emtools/metadata/misc.py index bcb2324..7b523aa 100644 --- a/emtools/metadata/misc.py +++ b/emtools/metadata/misc.py @@ -358,15 +358,42 @@ class WarpXml: def __init__(self, xmlPath): with open(xmlPath) as f: self._data = xmltodict.parse(f.read()) + # Root element name, e.g. 'Movie' or 'TiltSeries' + self.root = next(iter(self._data)) - def getDict(self, *keys): - """ Navigate the provided keys and get a dict from Name=Value pairs. - """ + def _get(self, *keys): d = self._data for k in keys: d = d[k] + return d - return {e['@Name']: e['@Value'] for e in d} + def getDict(self, *keys): + """ Navigate the provided keys and get a dict from Name=Value pairs. + """ + return {e['@Name']: e['@Value'] for e in self._get(*keys)} + + def getAttributes(self, *keys): + """ Return the attributes of the element at the provided keys + (the root element if no keys are given) as a dict. """ + d = self._get(*(keys or (self.root,))) + return {k[1:]: v for k, v in d.items() if k.startswith('@')} + + def getValues(self, *keys): + """ Return the whitespace-separated values in the text of the + element at the provided keys, e.g. ('TiltSeries', 'UseTilt'). """ + d = self._get(*keys) + text = d.get('#text', '') if isinstance(d, dict) else d + return (text or '').split() + + def isSelected(self): + """ Return False if the item (movie or tilt series) is deselected in + Warp, either manually or by Warp filters. UnselectManual, when set, + overrides UnselectFilter. """ + attrs = self.getAttributes() + manual = attrs.get('UnselectManual') or 'null' + if manual != 'null': + return manual != 'True' + return attrs.get('UnselectFilter', 'False') != 'True' class WarpSpecies(dict):