From 7ce54fe460f3d5a574a276b50609f4eee42e9253 Mon Sep 17 00:00:00 2001 From: Peter Cock Date: Mon, 25 Oct 2010 10:28:39 +0100 Subject: [PATCH] Remove FASTA filter script from BLAST+ tools (I want to generalise it) --- tool_conf.xml.sample | 1 - tools/ncbi_blast_plus/blast_filter_fasta.py | 38 -------------------- tools/ncbi_blast_plus/blast_filter_fasta.xml | 37 ------------------- 3 files changed, 76 deletions(-) delete mode 100644 tools/ncbi_blast_plus/blast_filter_fasta.py delete mode 100644 tools/ncbi_blast_plus/blast_filter_fasta.xml diff --git a/tool_conf.xml.sample b/tool_conf.xml.sample index adee724a8b9..419d780ff81 100644 --- a/tool_conf.xml.sample +++ b/tool_conf.xml.sample @@ -269,7 +269,6 @@ -
diff --git a/tools/ncbi_blast_plus/blast_filter_fasta.py b/tools/ncbi_blast_plus/blast_filter_fasta.py deleted file mode 100644 index 98fe7647122..00000000000 --- a/tools/ncbi_blast_plus/blast_filter_fasta.py +++ /dev/null @@ -1,38 +0,0 @@ -#!/usr/bin/env python -"""Filter a FASTA file using tabular output, e.g. from BLAST. - -Takes five command line options, tabular BLAST filename, ID column number -(using one based counting), input FASTA filename, and two output FASTA -filenames (for records with and without any BLAST hits). - -In the default NCBI BLAST+ tabular output, the query sequence ID is in column -one, and the ID of the match from the database is in column two. -""" -import sys -from galaxy_utils.sequence.fasta import fastaReader, fastaWriter - -#Parse Command Line -blast_file, blast_col, in_file, out_positive_file, out_negative_file = sys.argv[1:] -blast_col = int(blast_col)-1 -assert blast_col >= 0 - -#Read tabular BLAST file and record all queries with hit(s) -ids = set() -blast_handle = open(blast_file, "rU") -for line in blast_handle: - ids.add(line.split("\t")[blast_col]) -blast_handle.close() - -#Write filtered FASTA file based on IDs from BLAST file -reader = fastaReader(open(in_file, "rU")) -positive_writer = fastaWriter(open(out_positive_file, "w")) -negative_writer = fastaWriter(open(out_negative_file, "w")) -for record in reader: - #The [1:] is because the fastaReader leaves the > on the identifer. - if record.identifier and record.identifier.split()[0][1:] in ids: - positive_writer.write(record) - else: - negative_writer.write(record) -positive_writer.close() -negative_writer.close() -reader.close() diff --git a/tools/ncbi_blast_plus/blast_filter_fasta.xml b/tools/ncbi_blast_plus/blast_filter_fasta.xml deleted file mode 100644 index a8528ea075e..00000000000 --- a/tools/ncbi_blast_plus/blast_filter_fasta.xml +++ /dev/null @@ -1,37 +0,0 @@ - - Divide a FASTA file based on BLAST hits - - blast_filter_fasta.py $blast_file $blast_col $in_file $out_positive_file $out_negative_file - - - - - - - - - - - - - - - - - - - -**What it does** - -Typical use would be to take a multi-sequence FASTA and the tabular output of -running BLAST on it, and divide the FASTA file in two: those sequence with a -BLAST hit, and those without. - -In the default NCBI BLAST+ tabular output, the query sequence ID is in column -one, and the ID of the match from the database is in column two. - -This allows you to filter the FASTA file for the subjects in the BLAST search, -rather than filtering the FASTA file for the queries in the BLAST search. - - -