From 61f1cdf5d7e2078569cb4c1efb3c1df3bda6d0e2 Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 14:16:45 -0500 Subject: [PATCH 01/11] Added script to upload a directory to a data library --- scripts/api/library_upload_dir.py | 134 ++++++++++++++++++++++++++++++ 1 file changed, 134 insertions(+) create mode 100644 scripts/api/library_upload_dir.py diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py new file mode 100644 index 00000000000..d6b1912faaf --- /dev/null +++ b/scripts/api/library_upload_dir.py @@ -0,0 +1,134 @@ +#!/usr/bin/env python +import argparse +import os +from bioblend import galaxy + + +class Uploader: + + def __init__(self, url, api, local_dir, library_id, folder_id): + self.gi = galaxy.GalaxyInstance(url=url, key=api) + self.local_dir = local_dir + self.library_id = library_id + self.folder_id = folder_id + + self.memo_path = {} + + def memoized_path(self, path_parts, base_folder=None): + """Get the folder ID for a given folder path specified by path_parts. + + If the folder does not exist, it will be created ONCE (during the + instantiation of this Uploader object). After that it is stored and + recycled. If the Uploader object is re-created, it is not aware of + previously existing paths and will not respect those. TODO: handle + existing paths. + """ + if base_folder is None: + base_folder = self.folder_id + dropped_prefix = [] + + fk = '/'.join(path_parts) + if fk in self.memo_path: + # print "Cache hit %s" % fk + return self.memo_path[fk] + else: + # print "Cache miss %s" % fk + for i in reversed(range(len(path_parts))): + fk = '/'.join(path_parts[0:i + 1]) + if fk in self.memo_path: + # print "Parent folder hit %s" % fk + dropped_prefix = path_parts[0:i + 1] + path_parts = path_parts[i + 1:] + base_folder = self.memo_path[fk] + break + + nfk = [] + for i in range(len(path_parts)): + nfk.append('/'.join(list(dropped_prefix) + list(path_parts[0:i + 1]))) + + # Recursively create the path from our base_folder starting points, + # gettting the IDs of each folder per path component + ids = self.recursively_build_path(path_parts, base_folder) + + # These are then associated with the paths. + for (key, fid) in zip(nfk, ids): + self.memo_path[key] = fid + return ids[-1] + + def recursively_build_path(self, path_parts, parent_folder_id, ids=None): + """Given an iterable of path components and a parent folder id, recursively + create directories below parent_folder_id""" + if ids is None: + ids = [] + if len(path_parts) == 0: + return ids + else: + pf = self.gi.libraries.create_folder(self.library_id, path_parts[0], base_folder_id=parent_folder_id) + ids.append(pf[0]['id']) + # print "create_folder(%s, %s, %s) = %s" % (self.library_id, path_parts[0], parent_folder_id, pf[0]['id']) + return self.recursively_build_path(path_parts[1:], pf[0]['id'], ids=ids) + + # http://stackoverflow.com/questions/13505819/python-split-path-recursively/13505966#13505966 + def rec_split(self, s): + rest, tail = os.path.split(s) + if tail == '.': + return () + if rest == '': + return tail, + return self.rec_split(rest) + (tail,) + + def file_filter(dirName, fname): + bad = [ + '/x' in dirName, + '/x' in fname, + fname.startswith('xa'), + 'CONTIG' in fname, + 'NODE' in fname, + ] + if any(bad): + return False + + good = [ + fname.endswith('.fa'), + fname.endswith('.fna'), + fname.endswith('.fastq'), + fname.endswith('.sff'), + ] + + return any(good) + + def collect_files(self, rootDir): + for dirName, subdirList, fileList in os.walk(rootDir): + for fname in fileList: + if self.file_filter(dirName, fname): + yield (dirName, fname) + + def upload(self): + all_files = list(self.collect_files(self.local_dir)) + + for idx, (dirName, fname) in enumerate(all_files): + if idx < 35: + continue + if not os.path.exists(os.path.join(dirName, fname)): + continue + fid = self.memoized_path(self.rec_split(dirName), base_folder=self.folder_id) + print('[%s/%s] %s/%s' % (idx, len(all_files), fid, fname)) + print self.gi.libraries.upload_file_from_local_path( + self.library_id, + os.path.join(dirName, fname), + folder_id=fid, + ) + + +if __name__ == '__main__': + parser = argparse.ArgumentParser(description='Upload a directory into a data library') + parser.add_argument( "-u", "--url", dest="uri", required=True, help="Galaxy URL" ) + parser.add_argument( "-a", "--api", dest="api", required=True, help="API Key" ) + + parser.add_argument( "-d", "--dir", dest="local_dir", required=True, help="Local directory" ) + parser.add_argument( "-l", "--lib", dest="library_id", required=True, help="Library ID" ) + parser.add_argument( "-f", "--folder", dest="folder_id", help="Folder ID. If not specified, will go to root of library." ) + args = parser.parse_args() + + u = Uploader(**vars(args)) + u.upload() From a8b727791ffbdcbe70d2b638080101ec2a8a3a4d Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 14:17:32 -0500 Subject: [PATCH 02/11] Correct language in galaxy dev doc --- doc/source/dev/faq.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/doc/source/dev/faq.rst b/doc/source/dev/faq.rst index d8883d7e82c..e60a0a9eea8 100644 --- a/doc/source/dev/faq.rst +++ b/doc/source/dev/faq.rst @@ -2,7 +2,7 @@ How Do I... =========== This section contains a number of smaller topics with links and examples meant -to provide relatively concrete answers for specific tool development scenarios. +to provide relatively concrete answers for specific Galaxy development scenarios. ... interact with the Galaxy codebase interactively? ---------------------------------------------------- From decca0e96a43bf9cf23af56d09a966898018b354 Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 14:37:21 -0500 Subject: [PATCH 03/11] Refactor to accept output of find --- scripts/api/library_upload_dir.py | 35 ++++--------------------------- 1 file changed, 4 insertions(+), 31 deletions(-) diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py index d6b1912faaf..f91e10fe621 100644 --- a/scripts/api/library_upload_dir.py +++ b/scripts/api/library_upload_dir.py @@ -1,4 +1,5 @@ #!/usr/bin/env python +import sys import argparse import os from bioblend import galaxy @@ -6,9 +7,8 @@ from bioblend import galaxy class Uploader: - def __init__(self, url, api, local_dir, library_id, folder_id): + def __init__(self, url, api, library_id, folder_id): self.gi = galaxy.GalaxyInstance(url=url, key=api) - self.local_dir = local_dir self.library_id = library_id self.folder_id = folder_id @@ -77,34 +77,8 @@ class Uploader: return tail, return self.rec_split(rest) + (tail,) - def file_filter(dirName, fname): - bad = [ - '/x' in dirName, - '/x' in fname, - fname.startswith('xa'), - 'CONTIG' in fname, - 'NODE' in fname, - ] - if any(bad): - return False - - good = [ - fname.endswith('.fa'), - fname.endswith('.fna'), - fname.endswith('.fastq'), - fname.endswith('.sff'), - ] - - return any(good) - - def collect_files(self, rootDir): - for dirName, subdirList, fileList in os.walk(rootDir): - for fname in fileList: - if self.file_filter(dirName, fname): - yield (dirName, fname) - def upload(self): - all_files = list(self.collect_files(self.local_dir)) + all_files = list(sys.stdin.readlines()) for idx, (dirName, fname) in enumerate(all_files): if idx < 35: @@ -122,10 +96,9 @@ class Uploader: if __name__ == '__main__': parser = argparse.ArgumentParser(description='Upload a directory into a data library') - parser.add_argument( "-u", "--url", dest="uri", required=True, help="Galaxy URL" ) + parser.add_argument( "-u", "--url", dest="url", required=True, help="Galaxy URL" ) parser.add_argument( "-a", "--api", dest="api", required=True, help="API Key" ) - parser.add_argument( "-d", "--dir", dest="local_dir", required=True, help="Local directory" ) parser.add_argument( "-l", "--lib", dest="library_id", required=True, help="Library ID" ) parser.add_argument( "-f", "--folder", dest="folder_id", help="Folder ID. If not specified, will go to root of library." ) args = parser.parse_args() From aec9961407717f305b4ee3a5fdd6f655f4885944 Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 14:43:08 -0500 Subject: [PATCH 04/11] Strip newlines --- scripts/api/library_upload_dir.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py index f91e10fe621..e142c5dfdec 100644 --- a/scripts/api/library_upload_dir.py +++ b/scripts/api/library_upload_dir.py @@ -78,7 +78,7 @@ class Uploader: return self.rec_split(rest) + (tail,) def upload(self): - all_files = list(sys.stdin.readlines()) + all_files = [x.strip() for x in list(sys.stdin.readlines())] for idx, (dirName, fname) in enumerate(all_files): if idx < 35: From 1f9be2410bb9efe8cd647e240bd45a9e99a3bdce Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 14:48:30 -0500 Subject: [PATCH 05/11] Documented use of library upload tool --- doc/source/admin/index.rst | 2 ++ doc/source/admin/useful_scripts.rst | 17 +++++++++++++++++ 2 files changed, 19 insertions(+) create mode 100644 doc/source/admin/useful_scripts.rst diff --git a/doc/source/admin/index.rst b/doc/source/admin/index.rst index a3abf3d072e..af7df7a4783 100644 --- a/doc/source/admin/index.rst +++ b/doc/source/admin/index.rst @@ -9,3 +9,5 @@ documentation. These resources should be used together. :maxdepth: 3 interactive_environments.rst + + useful_scripts.rst diff --git a/doc/source/admin/useful_scripts.rst b/doc/source/admin/useful_scripts.rst new file mode 100644 index 00000000000..46746a03b99 --- /dev/null +++ b/doc/source/admin/useful_scripts.rst @@ -0,0 +1,17 @@ +Useful Scripts and Administration Tricks +======================================== + +This page aims to help ease the burden of administration with some easy to use scripts and documentation on what is available for admins to use. + +Uploading a directory into a Data Library +----------------------------------------- + +Data libraries can really ease the use of Galaxy for your administrators and end users. They provide a form of shared folders that users can copy datasets from into their history. + +This script was developed to be as general as possible, allowing you to pipe the output of a much more complex find command to this script, uploading all of the files into a data library: + +.. code-block:: console + + $ find /path/to/sequencing-data/ -name '*.fastq' -or -name '*.fa' | python $GALAXY_ROOT/scripts/api/library_upload_files.py + +Find has an extremely expressive command line for selecting specific files that are of interest to you. These will then be recursively uploaded into Galaxy, maintaining the folder hierarchy, a useful feature when moving legacy data into Galaxy. From 33925747b8b3b554b7b1cbb185ab4f15cfa3586b Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 14:50:35 -0500 Subject: [PATCH 06/11] Allow linking only --- scripts/api/library_upload_dir.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py index e142c5dfdec..4da6e5e8dc2 100644 --- a/scripts/api/library_upload_dir.py +++ b/scripts/api/library_upload_dir.py @@ -7,10 +7,11 @@ from bioblend import galaxy class Uploader: - def __init__(self, url, api, library_id, folder_id): + def __init__(self, url, api, library_id, folder_id, should_link): self.gi = galaxy.GalaxyInstance(url=url, key=api) self.library_id = library_id self.folder_id = folder_id + self.should_link self.memo_path = {} @@ -91,6 +92,7 @@ class Uploader: self.library_id, os.path.join(dirName, fname), folder_id=fid, + link_data_only=self.should_link, ) @@ -101,6 +103,8 @@ if __name__ == '__main__': parser.add_argument( "-l", "--lib", dest="library_id", required=True, help="Library ID" ) parser.add_argument( "-f", "--folder", dest="folder_id", help="Folder ID. If not specified, will go to root of library." ) + + parser.add_argument( "--link", dest="should_link", action="store_true", default=False, help="Link datasets only, do not upload to Galaxy.") args = parser.parse_args() u = Uploader(**vars(args)) From 705445a2f2605d45d31fb11ef773460f252b0afd Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 14:51:21 -0500 Subject: [PATCH 07/11] Correct doc script name --- doc/source/admin/useful_scripts.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/doc/source/admin/useful_scripts.rst b/doc/source/admin/useful_scripts.rst index 46746a03b99..908b0ffff5c 100644 --- a/doc/source/admin/useful_scripts.rst +++ b/doc/source/admin/useful_scripts.rst @@ -12,6 +12,6 @@ This script was developed to be as general as possible, allowing you to pipe the .. code-block:: console - $ find /path/to/sequencing-data/ -name '*.fastq' -or -name '*.fa' | python $GALAXY_ROOT/scripts/api/library_upload_files.py + $ find /path/to/sequencing-data/ -name '*.fastq' -or -name '*.fa' | python $GALAXY_ROOT/scripts/api/library_upload_dir.py -Find has an extremely expressive command line for selecting specific files that are of interest to you. These will then be recursively uploaded into Galaxy, maintaining the folder hierarchy, a useful feature when moving legacy data into Galaxy. +Find has an extremely expressive command line for selecting specific files that are of interest to you. These will then be recursively uploaded into Galaxy, maintaining the folder hierarchy, a useful feature when moving legacy data into Galaxy. For a complete description of the options of this script, you can run ``python $GALAXY_ROOT/scripts/api/library_upload_dir.py --help`` From 6be301ca8afcbc3cd1c3c7e3d74d3dee17a6a607 Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 15:02:27 -0500 Subject: [PATCH 08/11] Support local/non local data and paths including root --- scripts/api/library_upload_dir.py | 40 +++++++++++++++++++++---------- 1 file changed, 27 insertions(+), 13 deletions(-) diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py index 4da6e5e8dc2..02b31ba08ca 100644 --- a/scripts/api/library_upload_dir.py +++ b/scripts/api/library_upload_dir.py @@ -7,11 +7,12 @@ from bioblend import galaxy class Uploader: - def __init__(self, url, api, library_id, folder_id, should_link): + def __init__(self, url, api, library_id, folder_id, should_link, non_local): self.gi = galaxy.GalaxyInstance(url=url, key=api) self.library_id = library_id self.folder_id = folder_id - self.should_link + self.should_link = should_link + self.non_local = non_local self.memo_path = {} @@ -71,6 +72,9 @@ class Uploader: # http://stackoverflow.com/questions/13505819/python-split-path-recursively/13505966#13505966 def rec_split(self, s): + if s == '/': + return () + rest, tail = os.path.split(s) if tail == '.': return () @@ -81,19 +85,26 @@ class Uploader: def upload(self): all_files = [x.strip() for x in list(sys.stdin.readlines())] - for idx, (dirName, fname) in enumerate(all_files): - if idx < 35: - continue + for idx, path in enumerate(all_files): + (dirName, fname) = path.rsplit(os.path.sep, 1) if not os.path.exists(os.path.join(dirName, fname)): continue fid = self.memoized_path(self.rec_split(dirName), base_folder=self.folder_id) - print('[%s/%s] %s/%s' % (idx, len(all_files), fid, fname)) - print self.gi.libraries.upload_file_from_local_path( - self.library_id, - os.path.join(dirName, fname), - folder_id=fid, - link_data_only=self.should_link, - ) + print('[%s/%s] %s/%s' % (idx + 1, len(all_files), fid, fname)) + + if self.non_local: + self.gi.libraries.upload_file_from_local_path( + self.library_id, + os.path.join(dirName, fname), + folder_id=fid, + ) + else: + self.gi.libraries.upload_from_galaxy_filesystem( + self.library_id, + os.path.join(dirName, fname), + folder_id=fid, + link_data_only='link_to_files' if self.should_link else 'copy_files', + ) if __name__ == '__main__': @@ -104,7 +115,10 @@ if __name__ == '__main__': parser.add_argument( "-l", "--lib", dest="library_id", required=True, help="Library ID" ) parser.add_argument( "-f", "--folder", dest="folder_id", help="Folder ID. If not specified, will go to root of library." ) - parser.add_argument( "--link", dest="should_link", action="store_true", default=False, help="Link datasets only, do not upload to Galaxy.") + parser.add_argument( "--nonlocal", dest="non_local", action="store_true", default=False, + help="Set this flag if you are NOT running this script on your Galaxy head node with access to the full filesystem" ) + parser.add_argument( "--link", dest="should_link", action="store_true", default=False, + help="Link datasets only, do not upload to Galaxy. ONLY Avaialble if you run 'locally' relative to your Galaxy head node/filesystem ") args = parser.parse_args() u = Uploader(**vars(args)) From 2cca823a787dfcd48b5cfb7ef685d81e9341d189 Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Wed, 14 Oct 2015 16:05:44 -0500 Subject: [PATCH 09/11] Prepopulate memoization cache --- scripts/api/library_upload_dir.py | 36 ++++++++++++++++++++++++++++++- 1 file changed, 35 insertions(+), 1 deletion(-) diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py index 02b31ba08ca..8f3c1fd958f 100644 --- a/scripts/api/library_upload_dir.py +++ b/scripts/api/library_upload_dir.py @@ -7,7 +7,8 @@ from bioblend import galaxy class Uploader: - def __init__(self, url, api, library_id, folder_id, should_link, non_local): + def __init__(self, url, api, library_id, folder_id, should_link, + non_local): self.gi = galaxy.GalaxyInstance(url=url, key=api) self.library_id = library_id self.folder_id = folder_id @@ -15,6 +16,39 @@ class Uploader: self.non_local = non_local self.memo_path = {} + self.prepopulate_memo() + + def prepopulate_memo(self): + """ + Because the Galaxy Data Libraries API/system does not act like any + other file system in existence, and allows multiple files/folders with + identical names in the same parent directory, we have to prepopulate + the memoization cache with everything currently in the target + directory. + + Because the Galaxy Data Libraries API does not work from a perspective + of "show me what is in this directory", we are forced to get the entire + contents of the data library, and then filter out things that are + interesting to us based on a folder prefix. + """ + existing = self.gi.libraries.show_library(self.library_id, contents=True) + + uploading_to = [x for x in existing if x['id'] == self.folder_id] + if len(uploading_to) == 0: + raise Exception("Unknown folder [%s] in library [%s]" % + (self.folder_id, self.library_id)) + else: + uploading_to = uploading_to[0] + + for x in existing: + # We only care if it's a subdirectory of where we're uploading to + if not x['name'].startswith(uploading_to['name']): + continue + + name_part = x['name'].split(uploading_to['name'], 1)[-1] + if name_part.startswith('/'): + name_part = name_part[1:] + self.memo_path[name_part] = x['id'] def memoized_path(self, path_parts, base_folder=None): """Get the folder ID for a given folder path specified by path_parts. From 8a66701d4f622e3a57e39274fb3046dc67640be5 Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Thu, 15 Oct 2015 16:02:05 -0500 Subject: [PATCH 10/11] Don't re-upload if already there --- scripts/api/library_upload_dir.py | 40 +++++++++++++++++++------------ 1 file changed, 25 insertions(+), 15 deletions(-) diff --git a/scripts/api/library_upload_dir.py b/scripts/api/library_upload_dir.py index 8f3c1fd958f..9e9b34322b7 100644 --- a/scripts/api/library_upload_dir.py +++ b/scripts/api/library_upload_dir.py @@ -123,22 +123,32 @@ class Uploader: (dirName, fname) = path.rsplit(os.path.sep, 1) if not os.path.exists(os.path.join(dirName, fname)): continue - fid = self.memoized_path(self.rec_split(dirName), base_folder=self.folder_id) - print('[%s/%s] %s/%s' % (idx + 1, len(all_files), fid, fname)) - - if self.non_local: - self.gi.libraries.upload_file_from_local_path( - self.library_id, - os.path.join(dirName, fname), - folder_id=fid, - ) + # Figure out what the memo key will be early + basepath = self.rec_split(dirName) + if len(basepath) == 0: + memo_key = fname else: - self.gi.libraries.upload_from_galaxy_filesystem( - self.library_id, - os.path.join(dirName, fname), - folder_id=fid, - link_data_only='link_to_files' if self.should_link else 'copy_files', - ) + memo_key = os.path.join(os.path.join(*basepath), fname) + + # So that we can check if it really needs to be uploaded. + already_uploaded = memo_key in self.memo_path.keys() + fid = self.memoized_path(basepath, base_folder=self.folder_id) + print('[%s/%s] %s/%s uploaded=%' % (idx + 1, len(all_files), fid, fname, already_uploaded)) + + if not already_uploaded: + if self.non_local: + self.gi.libraries.upload_file_from_local_path( + self.library_id, + os.path.join(dirName, fname), + folder_id=fid, + ) + else: + self.gi.libraries.upload_from_galaxy_filesystem( + self.library_id, + os.path.join(dirName, fname), + folder_id=fid, + link_data_only='link_to_files' if self.should_link else 'copy_files', + ) if __name__ == '__main__': From 5dd7370b845202b45b30b88c67e8ca98362935a5 Mon Sep 17 00:00:00 2001 From: Eric Rasche Date: Thu, 15 Oct 2015 16:06:18 -0500 Subject: [PATCH 11/11] Add a line of documentation about syncing --- doc/source/admin/useful_scripts.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/doc/source/admin/useful_scripts.rst b/doc/source/admin/useful_scripts.rst index 908b0ffff5c..5c17ddbcda8 100644 --- a/doc/source/admin/useful_scripts.rst +++ b/doc/source/admin/useful_scripts.rst @@ -15,3 +15,5 @@ This script was developed to be as general as possible, allowing you to pipe the $ find /path/to/sequencing-data/ -name '*.fastq' -or -name '*.fa' | python $GALAXY_ROOT/scripts/api/library_upload_dir.py Find has an extremely expressive command line for selecting specific files that are of interest to you. These will then be recursively uploaded into Galaxy, maintaining the folder hierarchy, a useful feature when moving legacy data into Galaxy. For a complete description of the options of this script, you can run ``python $GALAXY_ROOT/scripts/api/library_upload_dir.py --help`` + +This tool will not overwrite or re-upload already uploaded datasets. As a result, one can imagine running this on a cron job to keep an "incoming sequencing data" directory synced with a data library.