From b6f5553615397655639ea4ed60d429293b634825 Mon Sep 17 00:00:00 2001 From: Charles Coulombe Date: Tue, 19 Aug 2025 11:06:35 -0400 Subject: [PATCH 01/14] Format with 3 decimals and display seconds --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 94e1c5528a3..41a2665fb02 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -273,7 +273,7 @@ def administrative_delete_datasets( stop = time.time() print() print(f"Marked {deleted_instance_count} dataset instances as deleted") - print("Total elapsed time: ", stop - start) + print(f"Total elapsed time: {stop - start:.3f} seconds") print("##########################################") From 39047212083e4c6508bcbd842fd43828ff0183af Mon Sep 17 00:00:00 2001 From: Charles Coulombe Date: Tue, 19 Aug 2025 19:01:31 -0400 Subject: [PATCH 02/14] Fixed query results mapping for compatibility with sqlalchemy 2.x --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 41a2665fb02..1c5284e5560 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -235,14 +235,14 @@ def administrative_delete_datasets( .select_from( sa.join(model.User.__table__, model.History.__table__).join(model.HistoryDatasetAssociation.__table__) ) - .set_label_style() + .set_label_style(sa.LABEL_STYLE_DEFAULT) ) for result in app.sa_session.execute(user_query): - user_notifications[result[model.User.__table__.c.email]].append( + user_notifications[result._mapping[model.User.__table__.c.email]].append( ( - result[model.HistoryDatasetAssociation.__table__.c.name], - result[model.History.__table__.c.name], + result._mapping[model.HistoryDatasetAssociation.__table__.c.name], + result._mapping[model.History.__table__.c.name], ) ) deleted_instance_count += 1 From 860e2cda72aba8bb008a309625bbbed0414157cc Mon Sep 17 00:00:00 2001 From: Charles Coulombe Date: Tue, 19 Aug 2025 19:50:04 -0400 Subject: [PATCH 03/14] Refactored and updated administrative_delete_datasets function to be more readable but mainly compatible with sqlalchemy 2.x --- .../admin_cleanup_datasets.py | 63 ++++++++++--------- 1 file changed, 32 insertions(+), 31 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 1c5284e5560..05c6f90c207 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -55,7 +55,9 @@ from mako.template import Template from sqlalchemy import ( and_, false, + select, ) +from sqlalchemy.orm import aliased sys.path.insert(1, os.path.abspath(os.path.join(os.path.dirname(__file__), os.pardir, os.pardir, "lib"))) @@ -196,24 +198,26 @@ def administrative_delete_datasets( ): # Marks dataset history association deleted and email users start = time.time() + session = app.sa_session + + # Aliases for ORM‑mapped classes + HDA = aliased(app.model.HistoryDatasetAssociation) + Hist = aliased(app.model.History) + User = aliased(app.model.User) + Dataset = aliased(app.model.Dataset) + # Get HDAs older than cutoff time (ignore tool_id at this point) - # We really only need the id column here, but sqlalchemy barfs when - # trying to select only 1 column hda_ids_query = ( - sa.select(model.HistoryDatasetAssociation.__table__.c.id, model.HistoryDatasetAssociation.__table__.c.deleted) - .where( - and_( - model.Dataset.__table__.c.deleted == false(), - model.HistoryDatasetAssociation.__table__.c.update_time < cutoff_time, - model.HistoryDatasetAssociation.__table__.c.deleted == false(), - ) - ) - .select_from(sa.outerjoin(model.Dataset.__table__, model.HistoryDatasetAssociation.__table__)) + select(HDA.id) + .join(Dataset, Dataset.id == HDA.dataset_id, isouter=True) + .where(and_( + Dataset.deleted.is_(False), + HDA.update_time < cutoff_time, + HDA.deleted.is_(False))) ) # Add all datasets associated with Histories to our list - hda_ids = [] - hda_ids.extend([row.id for row in app.sa_session.execute(hda_ids_query)]) + hda_ids = session.execute(hda_ids_query).scalars().all() # Now find the tool_id that generated the dataset (even if it was copied) tool_matched_ids = [] @@ -229,32 +233,29 @@ def administrative_delete_datasets( # Process each of the Dataset objects for hda_id in hda_ids: - user_query = ( - sa.select(model.HistoryDatasetAssociation.__table__, model.History.__table__, model.User.__table__) - .where(and_(model.HistoryDatasetAssociation.__table__.c.id == hda_id)) - .select_from( - sa.join(model.User.__table__, model.History.__table__).join(model.HistoryDatasetAssociation.__table__) - ) - .set_label_style(sa.LABEL_STYLE_DEFAULT) - ) + # Bind hda_id for current iteration + rows = session.execute( + select(User.email, HDA.name, Hist.name) + .join(Hist, Hist.user_id == User.id) + .join(HDA, HDA.history_id == Hist.id) + .where(HDA.id == hda_id) + ).all() - for result in app.sa_session.execute(user_query): - user_notifications[result._mapping[model.User.__table__.c.email]].append( - ( - result._mapping[model.HistoryDatasetAssociation.__table__.c.name], - result._mapping[model.History.__table__.c.name], - ) - ) + for email, dataset_name, history_name in rows: + user_notifications[email].append((dataset_name, history_name)) deleted_instance_count += 1 + if not info_only and not email_only: # Get the HistoryDatasetAssociation objects - hda = app.sa_session.query(model.HistoryDatasetAssociation).get(hda_id) + hda = session.get(app.model.HistoryDatasetAssociation, hda_id) if not hda.deleted: # Mark the HistoryDatasetAssociation as deleted hda.deleted = True - app.sa_session.add(hda) + session.add(hda) print(f"Marked HistoryDatasetAssociation id {hda.id} as deleted") - app.sa_session().commit() + + if not info_only and not email_only: + session.commit() emailtemplate = Template(filename=template_file) for email, dataset_list in user_notifications.items(): From 651426ae39ba305e927cc8649eda7b2812a7f797 Mon Sep 17 00:00:00 2001 From: Charles Coulombe Date: Wed, 20 Aug 2025 08:57:04 -0400 Subject: [PATCH 04/14] Updated _get_tool_id_for_hda function to be more readable but mainly compatible with sqlalchemy 2.x --- .../admin_cleanup_datasets.py | 33 ++++++++++++------- 1 file changed, 22 insertions(+), 11 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 05c6f90c207..9399ca456ac 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -279,20 +279,31 @@ def administrative_delete_datasets( def _get_tool_id_for_hda(app, hda_id): - # TODO Some datasets don't seem to have an entry in jtod or a copied_from if hda_id is None: return None - job = ( - app.sa_session.query(model.Job) - .join(model.JobToOutputDatasetAssociation) - .filter(model.JobToOutputDatasetAssociation.__table__.c.dataset_id == hda_id) - .first() + + # Aliases for ORM‑mapped classes + Job = aliased(app.model.Job) + JTODA = aliased(app.model.JobToOutputDatasetAssociation) + HDA = aliased(app.model.HistoryDatasetAssociation) + + session = app.sa_session + + job_query = ( + select(Job.tool_id) + .join(JTODA, JTODA.job_id == Job.id) + .where(JTODA.dataset_id == hda_id) ) - if job is not None: - return job.tool_id - else: - hda = app.sa_session.query(model.HistoryDatasetAssociation).get(hda_id) - return _get_tool_id_for_hda(app, hda.copied_from_history_dataset_association_id) + + tool_id = session.execute(job_query).scalars().first() + if tool_id is not None: + return tool_id + + hda = session.get(app.model.HistoryDatasetAssociation, hda_id) + if hda is None: + return None + + return _get_tool_id_for_hda(app, hda.copied_from_history_dataset_association_id) if __name__ == "__main__": From f0e63ff652e7dd75386381b3dcbc577a19be3b3a Mon Sep 17 00:00:00 2001 From: Charles Coulombe Date: Mon, 25 Aug 2025 15:38:57 -0400 Subject: [PATCH 05/14] Added `state` to email subject. Differentiate between the information and the actual deletion email --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 9399ca456ac..ae23fb83d4a 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -260,7 +260,8 @@ def administrative_delete_datasets( emailtemplate = Template(filename=template_file) for email, dataset_list in user_notifications.items(): msgtext = emailtemplate.render(email=email, datasets=dataset_list, cutoff=cutoff_days) - subject = f"Galaxy Server Cleanup - {len(dataset_list)} datasets DELETED" + state = "" if info_only or email_only else " DELETED" + subject = f"Galaxy Server Cleanup - {len(dataset_list)} datasets{state}" fromaddr = config.email_from print() print(f"From: {fromaddr}") From 85d8e6a78f2e2ccff71db761bde4a18274fcce34 Mon Sep 17 00:00:00 2001 From: Charles Coulombe Date: Tue, 26 Aug 2025 10:29:34 -0400 Subject: [PATCH 06/14] Added --no-send option, to allow not sending an email upon deletion --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index ae23fb83d4a..a966542fbcd 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -22,6 +22,7 @@ Optional Arguments: -i --info_only - Print results, but don't email or delete anything -e --email_only - Email notifications, but don't delete anything Useful for notifying users of pending deletion + --no-send - Do no send email (Default: false) --smtp - Specify smtp server If not specified, use smtp settings specified in config file @@ -121,6 +122,7 @@ def main(): help="Send emails only, don't delete", default=False, ) + parser.add_argument("--no-send", action="store_true", help="Do not send email", default=False) parser.add_argument( "--smtp", default=None, help="SMTP Server to use to send email. Default: [read from galaxy config file]" ) @@ -188,13 +190,14 @@ def main(): config=config, email_only=args.email_only, info_only=args.info_only, + no_send=args.no_send, ) app.shutdown() sys.exit(0) def administrative_delete_datasets( - app, cutoff_time, cutoff_days, tool_id, template_file, config, email_only=False, info_only=False + app, cutoff_time, cutoff_days, tool_id, template_file, config, email_only=False, info_only=False, no_send=False ): # Marks dataset history association deleted and email users start = time.time() @@ -269,7 +272,7 @@ def administrative_delete_datasets( print(f"Subject: {subject}") print("----------") print(msgtext) - if not info_only: + if not no_send and not info_only: galaxy.util.send_mail(fromaddr, email, subject, msgtext, config) stop = time.time() From f14f34dca72df143a7fae6ea1078a4cfe77a7d66 Mon Sep 17 00:00:00 2001 From: Charles Coulombe Date: Tue, 26 Aug 2025 13:18:46 -0400 Subject: [PATCH 07/14] Removed unsused imports --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 4 ---- 1 file changed, 4 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index a966542fbcd..c593ffa000d 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -51,11 +51,9 @@ from datetime import ( ) from time import strftime -import sqlalchemy as sa from mako.template import Template from sqlalchemy import ( and_, - false, select, ) from sqlalchemy.orm import aliased @@ -66,7 +64,6 @@ from cleanup_datasets import CleanupDatasetsApplication import galaxy.config import galaxy.util -from galaxy import model from galaxy.util.script import ( app_properties_from_args, populate_config_args, @@ -289,7 +286,6 @@ def _get_tool_id_for_hda(app, hda_id): # Aliases for ORM‑mapped classes Job = aliased(app.model.Job) JTODA = aliased(app.model.JobToOutputDatasetAssociation) - HDA = aliased(app.model.HistoryDatasetAssociation) session = app.sa_session From 1bd316e9518267524949f5221c640e8c65fc4400 Mon Sep 17 00:00:00 2001 From: John Davis Date: Tue, 27 Jan 2026 11:20:42 -0500 Subject: [PATCH 08/14] Use == false() instead of is_(False) --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index c593ffa000d..2c11ef92f93 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -211,9 +211,9 @@ def administrative_delete_datasets( select(HDA.id) .join(Dataset, Dataset.id == HDA.dataset_id, isouter=True) .where(and_( - Dataset.deleted.is_(False), + Dataset.deleted == false(), HDA.update_time < cutoff_time, - HDA.deleted.is_(False))) + HDA.deleted == false())) ) # Add all datasets associated with Histories to our list From cce6f59c1c31d82461084bc64fbf54d54c2931ac Mon Sep 17 00:00:00 2001 From: John Davis Date: Tue, 27 Jan 2026 11:28:27 -0500 Subject: [PATCH 09/14] Do not use aliased on models for readability: just import them --- .../admin_cleanup_datasets.py | 32 +++++++++---------- 1 file changed, 16 insertions(+), 16 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 2c11ef92f93..6154821d72c 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -64,6 +64,12 @@ from cleanup_datasets import CleanupDatasetsApplication import galaxy.config import galaxy.util +from galaxy.model import ( + Dataset, + History, + HistoryDatasetAssociation, + User, +) from galaxy.util.script import ( app_properties_from_args, populate_config_args, @@ -200,20 +206,14 @@ def administrative_delete_datasets( start = time.time() session = app.sa_session - # Aliases for ORM‑mapped classes - HDA = aliased(app.model.HistoryDatasetAssociation) - Hist = aliased(app.model.History) - User = aliased(app.model.User) - Dataset = aliased(app.model.Dataset) - # Get HDAs older than cutoff time (ignore tool_id at this point) hda_ids_query = ( - select(HDA.id) - .join(Dataset, Dataset.id == HDA.dataset_id, isouter=True) + select(HistoryDatasetAssociation.id) + .join(Dataset, Dataset.id == HistoryDatasetAssociation.dataset_id, isouter=True) .where(and_( Dataset.deleted == false(), - HDA.update_time < cutoff_time, - HDA.deleted == false())) + HistoryDatasetAssociation.update_time < cutoff_time, + HistoryDatasetAssociation.deleted == false())) ) # Add all datasets associated with Histories to our list @@ -235,10 +235,10 @@ def administrative_delete_datasets( for hda_id in hda_ids: # Bind hda_id for current iteration rows = session.execute( - select(User.email, HDA.name, Hist.name) - .join(Hist, Hist.user_id == User.id) - .join(HDA, HDA.history_id == Hist.id) - .where(HDA.id == hda_id) + select(User.email, HistoryDatasetAssociation.name, History.name) + .join(History, History.user_id == User.id) + .join(HistoryDatasetAssociation, HistoryDatasetAssociation.history_id == History.id) + .where(HistoryDatasetAssociation.id == hda_id) ).all() for email, dataset_name, history_name in rows: @@ -247,7 +247,7 @@ def administrative_delete_datasets( if not info_only and not email_only: # Get the HistoryDatasetAssociation objects - hda = session.get(app.model.HistoryDatasetAssociation, hda_id) + hda = session.get(HistoryDatasetAssociation, hda_id) if not hda.deleted: # Mark the HistoryDatasetAssociation as deleted hda.deleted = True @@ -299,7 +299,7 @@ def _get_tool_id_for_hda(app, hda_id): if tool_id is not None: return tool_id - hda = session.get(app.model.HistoryDatasetAssociation, hda_id) + hda = session.get(HistoryDatasetAssociation, hda_id) if hda is None: return None From 3a59617837e8fffb4ee794903f72713f75fb453e Mon Sep 17 00:00:00 2001 From: John Davis Date: Tue, 27 Jan 2026 11:30:40 -0500 Subject: [PATCH 10/14] Restore old comment --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 6154821d72c..9a0df0692ea 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -280,6 +280,7 @@ def administrative_delete_datasets( def _get_tool_id_for_hda(app, hda_id): + # TODO Some datasets don't seem to have an entry in jtod or a copied_from if hda_id is None: return None From 5ffdf789c1d4d682147618f708d58d47fcc4db7e Mon Sep 17 00:00:00 2001 From: John Davis Date: Tue, 27 Jan 2026 11:49:22 -0500 Subject: [PATCH 11/14] Remove redundant join criteria --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 9a0df0692ea..89163a7b9f3 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -209,7 +209,7 @@ def administrative_delete_datasets( # Get HDAs older than cutoff time (ignore tool_id at this point) hda_ids_query = ( select(HistoryDatasetAssociation.id) - .join(Dataset, Dataset.id == HistoryDatasetAssociation.dataset_id, isouter=True) + .join(Dataset, isouter=True) .where(and_( Dataset.deleted == false(), HistoryDatasetAssociation.update_time < cutoff_time, @@ -236,8 +236,8 @@ def administrative_delete_datasets( # Bind hda_id for current iteration rows = session.execute( select(User.email, HistoryDatasetAssociation.name, History.name) - .join(History, History.user_id == User.id) - .join(HistoryDatasetAssociation, HistoryDatasetAssociation.history_id == History.id) + .join(History) + .join(HistoryDatasetAssociation) .where(HistoryDatasetAssociation.id == hda_id) ).all() @@ -292,7 +292,7 @@ def _get_tool_id_for_hda(app, hda_id): job_query = ( select(Job.tool_id) - .join(JTODA, JTODA.job_id == Job.id) + .join(JTODA) .where(JTODA.dataset_id == hda_id) ) From 01a3f2707be77c4ec7666fa5f0078cea70135051 Mon Sep 17 00:00:00 2001 From: John Davis Date: Tue, 27 Jan 2026 11:50:36 -0500 Subject: [PATCH 12/14] Do not change existing behavior; raising error is correct --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 89163a7b9f3..d995cc385ca 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -301,9 +301,6 @@ def _get_tool_id_for_hda(app, hda_id): return tool_id hda = session.get(HistoryDatasetAssociation, hda_id) - if hda is None: - return None - return _get_tool_id_for_hda(app, hda.copied_from_history_dataset_association_id) From 8a40ef19fba2526c9e3c87da6d50b47868b3d6fc Mon Sep 17 00:00:00 2001 From: John Davis Date: Tue, 27 Jan 2026 11:51:12 -0500 Subject: [PATCH 13/14] Format --- .../cleanup_datasets/admin_cleanup_datasets.py | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index d995cc385ca..097e0f3cedc 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -210,10 +210,13 @@ def administrative_delete_datasets( hda_ids_query = ( select(HistoryDatasetAssociation.id) .join(Dataset, isouter=True) - .where(and_( - Dataset.deleted == false(), - HistoryDatasetAssociation.update_time < cutoff_time, - HistoryDatasetAssociation.deleted == false())) + .where( + and_( + Dataset.deleted == false(), + HistoryDatasetAssociation.update_time < cutoff_time, + HistoryDatasetAssociation.deleted == false(), + ) + ) ) # Add all datasets associated with Histories to our list @@ -290,11 +293,7 @@ def _get_tool_id_for_hda(app, hda_id): session = app.sa_session - job_query = ( - select(Job.tool_id) - .join(JTODA) - .where(JTODA.dataset_id == hda_id) - ) + job_query = select(Job.tool_id).join(JTODA).where(JTODA.dataset_id == hda_id) tool_id = session.execute(job_query).scalars().first() if tool_id is not None: From 4f66c001245503a98c8944c823ddac1b55d52e75 Mon Sep 17 00:00:00 2001 From: John Davis Date: Tue, 27 Jan 2026 15:16:04 -0500 Subject: [PATCH 14/14] Update scripts/cleanup_datasets/admin_cleanup_datasets.py Co-authored-by: Nicola Soranzo --- scripts/cleanup_datasets/admin_cleanup_datasets.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scripts/cleanup_datasets/admin_cleanup_datasets.py b/scripts/cleanup_datasets/admin_cleanup_datasets.py index 097e0f3cedc..d2259823480 100755 --- a/scripts/cleanup_datasets/admin_cleanup_datasets.py +++ b/scripts/cleanup_datasets/admin_cleanup_datasets.py @@ -54,6 +54,7 @@ from time import strftime from mako.template import Template from sqlalchemy import ( and_, + false, select, ) from sqlalchemy.orm import aliased