From f6ff679d637076fe6f5f6e37e6bb105ed795719c Mon Sep 17 00:00:00 2001 From: M Bernt Date: Thu, 1 Aug 2019 09:17:35 +0200 Subject: [PATCH] cleanup: use logging instead of print --- scripts/cleanup_datasets/cleanup_datasets.py | 116 +++++++++---------- 1 file changed, 58 insertions(+), 58 deletions(-) diff --git a/scripts/cleanup_datasets/cleanup_datasets.py b/scripts/cleanup_datasets/cleanup_datasets.py index 28f4cb53458..b05cb0e9034 100755 --- a/scripts/cleanup_datasets/cleanup_datasets.py +++ b/scripts/cleanup_datasets/cleanup_datasets.py @@ -105,15 +105,15 @@ def main(): cutoff_time = datetime.utcnow() - timedelta(days=args.days) now = strftime("%Y-%m-%d %H:%M:%S") - print("##########################################") - print("\n# %s - Handling stuff older than %i days" % (now, args.days)) + log.info("##########################################") + log.info("\n# %s - Handling stuff older than %i days" % (now, args.days)) if args.info_only: - print("# Displaying info only ( --info_only )\n") + log.info("# Displaying info only ( --info_only )\n") elif args.remove_from_disk: - print("Datasets will be removed from disk.\n") + log.info("Datasets will be removed from disk.\n") else: - print("Datasets will NOT be removed from disk.\n") + log.info("Datasets will NOT be removed from disk.\n") if args.delete_userless_histories: delete_userless_histories(app, cutoff_time, info_only=args.info_only, force_retry=args.force_retry) @@ -149,15 +149,15 @@ def delete_userless_histories(app, cutoff_time, info_only=False, force_retry=Fal app.model.History.table.c.update_time < cutoff_time)) for history in histories: if not info_only: - print("Deleting history id ", history.id) + log.info("Deleting history id ", history.id) history.deleted = True app.sa_session.add(history) app.sa_session.flush() history_count += 1 stop = time.time() - print("Deleted %d histories" % history_count) - print("Elapsed time: ", stop - start) - print("##########################################") + log.info("Deleted %d histories" % history_count) + log.info("Elapsed time: ", stop - start) + log.info("##########################################") def purge_histories(app, cutoff_time, remove_from_disk, info_only=False, force_retry=False): @@ -180,7 +180,7 @@ def purge_histories(app, cutoff_time, remove_from_disk, info_only=False, force_r app.model.History.table.c.update_time < cutoff_time)) \ .options(eagerload('datasets')) for history in histories: - print("### Processing history id %d (%s)" % (history.id, unicodify(history.name))) + log.info("### Processing history id %d (%s)" % (history.id, unicodify(history.name))) for dataset_assoc in history.datasets: _purge_dataset_instance(dataset_assoc, app, remove_from_disk, info_only=info_only) # mark a DatasetInstance as deleted, clear associated files, and mark the Dataset as deleted if it is deletable if not info_only: @@ -189,17 +189,17 @@ def purge_histories(app, cutoff_time, remove_from_disk, info_only=False, force_r # if we should ever delete info like this from the db though, so commented out for now... # for dhp in history.default_permissions: # dhp.delete() - print("Purging history id ", history.id) + log.info("Purging history id ", history.id) history.purged = True app.sa_session.add(history) app.sa_session.flush() else: - print("History id %d will be purged (without 'info_only' mode)" % history.id) + log.info("History id %d will be purged (without 'info_only' mode)" % history.id) history_count += 1 stop = time.time() - print('Purged %d histories.' % history_count) - print("Elapsed time: ", stop - start) - print("##########################################") + log.info('Purged %d histories.' % history_count) + log.info("Elapsed time: ", stop - start) + log.info("##########################################") def purge_libraries(app, cutoff_time, remove_from_disk, info_only=False, force_retry=False): @@ -222,15 +222,15 @@ def purge_libraries(app, cutoff_time, remove_from_disk, info_only=False, force_r for library in libraries: _purge_folder(library.root_folder, app, remove_from_disk, info_only=info_only) if not info_only: - print("Purging library id ", library.id) + log.info("Purging library id ", library.id) library.purged = True app.sa_session.add(library) app.sa_session.flush() library_count += 1 stop = time.time() - print('# Purged %d libraries .' % library_count) - print("Elapsed time: ", stop - start) - print("##########################################") + log.info('# Purged %d libraries .' % library_count) + log.info("Elapsed time: ", stop - start) + log.info("##########################################") def purge_folders(app, cutoff_time, remove_from_disk, info_only=False, force_retry=False): @@ -254,9 +254,9 @@ def purge_folders(app, cutoff_time, remove_from_disk, info_only=False, force_ret _purge_folder(folder, app, remove_from_disk, info_only=info_only) folder_count += 1 stop = time.time() - print('# Purged %d folders.' % folder_count) - print("Elapsed time: ", stop - start) - print("##########################################") + log.info('# Purged %d folders.' % folder_count) + log.info("Elapsed time: ", stop - start) + log.info("##########################################") def delete_datasets(app, cutoff_time, remove_from_disk, info_only=False, force_retry=False): @@ -301,7 +301,7 @@ def delete_datasets(app, cutoff_time, remove_from_disk, info_only=False, force_r library_dataset_ids = [row.id for row in library_dataset_ids_query.execute()] dataset_ids = [] for library_dataset_id in library_dataset_ids: - print("######### Processing LibraryDataset id:", library_dataset_id) + log.info("######### Processing LibraryDataset id:", library_dataset_id) # Get the LibraryDataset and the current LibraryDatasetDatasetAssociation objects ld = app.sa_session.query(app.model.LibraryDataset).get(library_dataset_id) ldda = ld.library_dataset_dataset_association @@ -311,16 +311,16 @@ def delete_datasets(app, cutoff_time, remove_from_disk, info_only=False, force_r if not ldda.deleted: ldda.deleted = True app.sa_session.add(ldda) - print("Marked associated LibraryDatasetDatasetAssociation id %d as deleted" % ldda.id) + log.info("Marked associated LibraryDatasetDatasetAssociation id %d as deleted" % ldda.id) for expired_ldda in ld.expired_datasets: if not expired_ldda.deleted: expired_ldda.deleted = True app.sa_session.add(expired_ldda) - print("Marked associated expired LibraryDatasetDatasetAssociation id %d as deleted" % ldda.id) + log.info("Marked associated expired LibraryDatasetDatasetAssociation id %d as deleted" % ldda.id) # Mark the LibraryDataset as purged ld.purged = True app.sa_session.add(ld) - print("Marked LibraryDataset id %d as purged" % ld.id) + log.info("Marked LibraryDataset id %d as purged" % ld.id) app.sa_session.flush() # Add all datasets associated with Histories to our list dataset_ids.extend([row.id for row in history_dataset_ids_query.execute()]) @@ -330,9 +330,9 @@ def delete_datasets(app, cutoff_time, remove_from_disk, info_only=False, force_r if dataset.id in skip: continue skip.append(dataset.id) - print("######### Processing dataset id:", dataset_id) + log.info("######### Processing dataset id:", dataset_id) if not _dataset_is_deletable(dataset): - print("Dataset is not deletable (shared between multiple histories/libraries, at least one is not deleted)") + log.info("Dataset is not deletable (shared between multiple histories/libraries, at least one is not deleted)") continue deleted_dataset_count += 1 for dataset_instance in dataset.history_associations + dataset.library_associations: @@ -340,9 +340,9 @@ def delete_datasets(app, cutoff_time, remove_from_disk, info_only=False, force_r _purge_dataset_instance(dataset_instance, app, remove_from_disk, info_only=info_only, is_deletable=True) deleted_instance_count += 1 stop = time.time() - print("Examined %d datasets, marked %d datasets and %d dataset instances (HDA) as deleted" % (len(skip), deleted_dataset_count, deleted_instance_count)) - print("Total elapsed time: ", stop - start) - print("##########################################") + log.info("Examined %d datasets, marked %d datasets and %d dataset instances (HDA) as deleted" % (len(skip), deleted_dataset_count, deleted_instance_count)) + log.info("Total elapsed time: ", stop - start) + log.info("##########################################") def purge_datasets(app, cutoff_time, remove_from_disk, info_only=False, force_retry=False): @@ -371,18 +371,18 @@ def purge_datasets(app, cutoff_time, remove_from_disk, info_only=False, force_re except Exception: pass stop = time.time() - print('Purged %d datasets' % dataset_count) + log.info('Purged %d datasets' % dataset_count) if remove_from_disk: - print('Freed disk space: ', disk_space) - print("Elapsed time: ", stop - start) - print("##########################################") + log.info('Freed disk space: ', disk_space) + log.info("Elapsed time: ", stop - start) + log.info("##########################################") def _purge_dataset_instance(dataset_instance, app, remove_from_disk, info_only=False, is_deletable=False): # A dataset_instance is either a HDA or an LDDA. Purging a dataset instance marks the instance as deleted, # and marks the associated dataset as deleted if it is not associated with another active DatsetInstance. if not info_only: - print("Marking as deleted: %s id %d (for dataset id %d)" % + log.info("Marking as deleted: %s id %d (for dataset id %d)" % (dataset_instance.__class__.__name__, dataset_instance.id, dataset_instance.dataset.id)) dataset_instance.mark_deleted() dataset_instance.clear_associated_files() @@ -390,16 +390,16 @@ def _purge_dataset_instance(dataset_instance, app, remove_from_disk, info_only=F app.sa_session.flush() app.sa_session.refresh(dataset_instance.dataset) else: - print("%s id %d (for dataset id %d) will be marked as deleted (without 'info_only' mode)" % + log.info("%s id %d (for dataset id %d) will be marked as deleted (without 'info_only' mode)" % (dataset_instance.__class__.__name__, dataset_instance.id, dataset_instance.dataset.id)) if is_deletable or _dataset_is_deletable(dataset_instance.dataset): # Calling methods may have already checked _dataset_is_deletable, if so, is_deletable should be True _delete_dataset(dataset_instance.dataset, app, remove_from_disk, info_only=info_only, is_deletable=is_deletable) else: if info_only: - print("Not deleting dataset ", dataset_instance.dataset.id, " (will be possibly deleted without 'info_only' mode)") + log.info("Not deleting dataset ", dataset_instance.dataset.id, " (will be possibly deleted without 'info_only' mode)") else: - print("Not deleting dataset %d (shared between multiple histories/libraries, at least one not deleted)" % dataset_instance.dataset.id) + log.info("Not deleting dataset %d (shared between multiple histories/libraries, at least one not deleted)" % dataset_instance.dataset.id) def _dataset_is_deletable(dataset): @@ -411,7 +411,7 @@ def _delete_dataset(dataset, app, remove_from_disk, info_only=False, is_deletabl # Marks a base dataset as deleted, hdas/lddas associated with dataset can no longer be undeleted. # Metadata files attached to associated dataset Instances is removed now. if not is_deletable and not _dataset_is_deletable(dataset): - print("This Dataset (%i) is not deletable, associated Metadata Files will not be removed.\n" % (dataset.id)) + log.info("This Dataset (%i) is not deletable, associated Metadata Files will not be removed.\n" % (dataset.id)) else: # Mark all associated MetadataFiles as deleted and purged and remove them from disk metadata_files = [] @@ -429,29 +429,29 @@ def _delete_dataset(dataset, app, remove_from_disk, info_only=False, is_deletabl if remove_from_disk: op_description = op_description + " and purged from disk" if info_only: - print("The following metadata files attached to associations of Dataset '%s' will be %s (without 'info_only' mode):" % (dataset.id, op_description)) + log.info("The following metadata files attached to associations of Dataset '%s' will be %s (without 'info_only' mode):" % (dataset.id, op_description)) else: - print("The following metadata files attached to associations of Dataset '%s' have been %s:" % (dataset.id, op_description)) + log.info("The following metadata files attached to associations of Dataset '%s' have been %s:" % (dataset.id, op_description)) if remove_from_disk: try: - print("Removing disk file ", metadata_file.file_name) + log.info("Removing disk file ", metadata_file.file_name) os.unlink(metadata_file.file_name) except Exception as e: - print("Error, exception: %s caught attempting to purge metadata file %s\n" % (str(e), metadata_file.file_name)) + log.info("Error, exception: %s caught attempting to purge metadata file %s\n" % (str(e), metadata_file.file_name)) metadata_file.purged = True app.sa_session.add(metadata_file) app.sa_session.flush() metadata_file.deleted = True app.sa_session.add(metadata_file) app.sa_session.flush() - print("%s" % metadata_file.file_name) + log.info("%s" % metadata_file.file_name) if not info_only: - print("Deleting dataset id", dataset.id) + log.info("Deleting dataset id", dataset.id) dataset.deleted = True app.sa_session.add(dataset) app.sa_session.flush() else: - print("Dataset %i will be deleted (without 'info_only' mode)" % (dataset.id)) + log.info("Dataset %i will be deleted (without 'info_only' mode)" % (dataset.id)) def _purge_dataset(app, dataset, remove_from_disk, info_only=False): @@ -462,7 +462,7 @@ def _purge_dataset(app, dataset, remove_from_disk, info_only=False): # Remove files from disk and update the database if remove_from_disk: # TODO: should permissions on the dataset be deleted here? - print("Removing disk, file ", dataset.file_name) + log.info("Removing disk, file ", dataset.file_name) os.unlink(dataset.file_name) # Remove associated extra files from disk if they exist if dataset.extra_files_path and os.path.exists(dataset.extra_files_path): @@ -476,32 +476,32 @@ def _purge_dataset(app, dataset, remove_from_disk, info_only=False): for user in usage_users: user.adjust_total_disk_usage(-dataset.get_total_size()) app.sa_session.add(user) - print("Purging dataset id", dataset.id) + log.info("Purging dataset id", dataset.id) dataset.purged = True app.sa_session.add(dataset) app.sa_session.flush() else: - print("Dataset %i will be purged (without 'info_only' mode)" % (dataset.id)) + log.info("Dataset %i will be purged (without 'info_only' mode)" % (dataset.id)) else: - print("This dataset (%i) is not purgable, the file (%s) will not be removed.\n" % (dataset.id, dataset.file_name)) + log.info("This dataset (%i) is not purgable, the file (%s) will not be removed.\n" % (dataset.id, dataset.file_name)) except OSError as exc: - print("Error, dataset file has already been removed: %s" % str(exc)) - print("Purging dataset id", dataset.id) + log.error("Error, dataset file has already been removed: %s" % str(exc)) + log.error("Purging dataset id", dataset.id) dataset.purged = True app.sa_session.add(dataset) app.sa_session.flush() except ObjectNotFound: - print("Dataset %i cannot be found in the object store" % dataset.id) + log.error("Dataset %i cannot be found in the object store" % dataset.id) except Exception as exc: - print("Error attempting to purge data file: ", dataset.file_name, " error: ", str(exc)) + log.error("Error attempting to purge data file: ", dataset.file_name, " error: ", str(exc)) else: - print("Error: '%s' has not previously been deleted, so it cannot be purged\n" % dataset.file_name) + log.info("Error: '%s' has not previously been deleted, so it cannot be purged\n" % dataset.file_name) def _purge_folder(folder, app, remove_from_disk, info_only=False): """Purges a folder and its contents, recursively""" for ld in folder.datasets: - print("Deleting library dataset id ", ld.id) + log.info("Deleting library dataset id ", ld.id) ld.deleted = True for ldda in [ld.library_dataset_dataset_association] + ld.expired_datasets: _purge_dataset_instance(ldda, app, remove_from_disk, info_only=info_only) # mark a DatasetInstance as deleted, clear associated files, and mark the Dataset as deleted if it is deletable @@ -509,7 +509,7 @@ def _purge_folder(folder, app, remove_from_disk, info_only=False): _purge_folder(sub_folder, app, remove_from_disk, info_only=info_only) if not info_only: # TODO: should the folder permissions be deleted here? - print("Purging folder id ", folder.id) + log.info("Purging folder id ", folder.id) folder.purged = True app.sa_session.add(folder) app.sa_session.flush()