mirror of
https://github.com/galaxyproject/galaxy.git
synced 2026-09-24 16:30:27 +08:00
Updates to disease ontology and lps tools
This commit is contained in:
@@ -1,10 +0,0 @@
|
||||
#This is a sample file distributed with Galaxy that is used by the disease
|
||||
#ontology tool. The disease_ontology.loc file has this format (white space
|
||||
#characters are TAB characters):
|
||||
#
|
||||
#<build> <description> <path to disease ontology files>
|
||||
#
|
||||
#Your disease_ontology.loc file should include an entry per line for each
|
||||
#disease ontology file you have stored.
|
||||
#hg18 disease associated genes /galaxy/data/hg18/misc/disease_ontology/genes-disease.bedlike
|
||||
#hg18 ontology /galaxy/data/hg18/misc/disease_ontology/human_disease.obo
|
||||
@@ -0,0 +1,11 @@
|
||||
#This is a sample file distributed with Galaxy that is used by the FunDO tool.
|
||||
#The funDo.loc file has this format (white space characters are TAB
|
||||
#characters):
|
||||
#
|
||||
#<build> <description> <path to disease associated genes file>
|
||||
#
|
||||
#Your funDo.loc file should include an entry per line for each disease
|
||||
#associated genes file you have stored.
|
||||
#
|
||||
hg18 disease associated genes /galaxy/data/hg18/misc/funDo/genes-disease.Sept2010.interval
|
||||
hg19 disease associated genes /galaxy/data/hg19/misc/funDo/genes-disease.Sept2010.interval
|
||||
@@ -333,7 +333,7 @@
|
||||
<tool file="human_genome_variation/linkToGProfile.xml" />
|
||||
<tool file="human_genome_variation/linkToDavid.xml"/>
|
||||
<tool file="human_genome_variation/ctd.xml" />
|
||||
<tool file="human_genome_variation/disease_ontology_options.xml" />
|
||||
<tool file="human_genome_variation/funDo.xml" />
|
||||
<tool file="human_genome_variation/snpFreq.xml" />
|
||||
<tool file="human_genome_variation/ldtools.xml" />
|
||||
<tool file="human_genome_variation/pass.xml" />
|
||||
|
||||
@@ -8,6 +8,7 @@ use warnings;
|
||||
# disease ontology.
|
||||
# ontology: http://do-wiki.nubic.northwestern.edu/index.php/Main_Page
|
||||
# gene associations by FunDO: http://projects.bioinformatics.northwestern.edu/do_rif/
|
||||
# Sept 2010, switch to doLite
|
||||
# input: build outfile sourceFileLoc.loc term or partial term
|
||||
##################################################################
|
||||
|
||||
@@ -23,7 +24,6 @@ my $term = shift @ARGV;
|
||||
$term =~ s/^'//; #remove quotes protecting from shell
|
||||
$term =~ s/'$//;
|
||||
my $data;
|
||||
my $obo;
|
||||
open(LOC, $in) or die "Couldn't open $in, $!\n";
|
||||
while (<LOC>) {
|
||||
chomp;
|
||||
@@ -32,8 +32,6 @@ while (<LOC>) {
|
||||
if ($f[0] eq $build) {
|
||||
if ($f[1] eq 'disease associated genes') {
|
||||
$data = $f[2];
|
||||
}elsif ($f[1] eq 'ontology') {
|
||||
$obo = $f[2];
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -53,8 +51,8 @@ open(FH, $data) or die "Couldn't open data file $data, $!\n";
|
||||
$term =~ s/\s+/|/g; #use OR between words
|
||||
while (<FH>) {
|
||||
chomp;
|
||||
my @f = split(/\t/); #chrom start end strand geneID geneName DOID disease
|
||||
if ($f[7] =~ /($term)/i) {
|
||||
my @f = split(/\t/); #chrom start end strand geneName geneID disease
|
||||
if ($f[6] =~ /($term)/i) {
|
||||
print OUT join("\t", @f), "\n";
|
||||
}elsif ($term eq 'disease') { #print all with disease
|
||||
print OUT join("\t", @f), "\n";
|
||||
|
||||
@@ -1,80 +0,0 @@
|
||||
<tool id="hgv_disease_ontology" name="FunDO" Version="1.0.0">
|
||||
<description>human genes associated with Disease Ontology terms</description>
|
||||
|
||||
<command interpreter="perl">
|
||||
disease_ontology_gene_fuzzy_selector.pl $build $out_file1 ${GALAXY_DATA_INDEX_DIR}/disease_ontology.loc '$term'
|
||||
</command>
|
||||
|
||||
<inputs>
|
||||
<param name="build" type="select" label="Database build">
|
||||
<options from_file="disease_ontology.loc">
|
||||
<column name="name" index="0"/>
|
||||
<column name="value" index="0"/>
|
||||
<filter type="unique_value" column="0"/>
|
||||
</options>
|
||||
</param>
|
||||
<param name="term" size="40" type="text" label="Enter Term " help="Entering disease will result in using all genes associated with any disease, or enter a term or space delimited list of terms."/>
|
||||
</inputs>
|
||||
|
||||
<outputs>
|
||||
<data format="tabular" name="out_file1"/>
|
||||
</outputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<param name="term" value="lung"/>
|
||||
<param name="build" value="hg18"/>
|
||||
<output name="out_file1" file="disease_ontology_1.out" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<help>
|
||||
**Data format**
|
||||
|
||||
There is no input dataset. The output is tabular_ format.
|
||||
|
||||
.. _tabular: ./static/formatHelp.html#tab
|
||||
|
||||
-----
|
||||
|
||||
**What it does**
|
||||
|
||||
This returns a set of genes that are associated with the disease term(s). This uses the entered text to search for terms matching a pattern. The search is case insensitive and the text is split on spaces and matches to any words are returned. The output includes gene coordinates, name, disease ontology ID and term. Entering disease would return all genes associated with disease.
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
Typing::
|
||||
|
||||
carcinoma
|
||||
|
||||
results in::
|
||||
|
||||
1. 2. 3. 4. 5. 6. 7. 8.
|
||||
chr8 18293034 18303003 + 10 NAT2 DOID:1994 carcinoma of the large Intestine
|
||||
chr8 18293034 18303003 + 10 NAT2 DOID:2876 laryngeal squamous cell carcinoma
|
||||
chr8 18293034 18303003 + 10 NAT2 DOID:3459 breast carcinoma
|
||||
chr8 18293034 18303003 + 10 NAT2 DOID:4947 cholangiocarcinoma
|
||||
chr1 241733106 242073176 - 10000 AKT3 DOID:3744 cervical squamous cell carcinoma
|
||||
chr1 241718157 242073176 - 10000 AKT3 DOID:3744 cervical squamous cell carcinoma
|
||||
etc.
|
||||
|
||||
Where::
|
||||
|
||||
1. is the chromosome name.
|
||||
2. is the start position of the gene.
|
||||
3. is the end position of the gene.
|
||||
4. is the strand.
|
||||
5. is the Entrez Gene ID.
|
||||
6. is the gene name.
|
||||
7. is the disease ontology ID.
|
||||
8. is the disease ontology term.
|
||||
|
||||
-----
|
||||
|
||||
**Reference**
|
||||
|
||||
Osborne JD, Flatow J, Holko M, Lin SM, Kibbe WA, Zhu LJ, Danila MI, Feng G, Chisholm RL. Annotating the human genome with Disease Ontology. BMC Genomics (2009) 10 S1:S6.
|
||||
</help>
|
||||
</tool>
|
||||
@@ -0,0 +1,96 @@
|
||||
<tool id="hgv_funDo" name="FunDO" Version="1.0.0">
|
||||
<description>human genes associated with disease terms</description>
|
||||
|
||||
<command interpreter="perl">
|
||||
disease_ontology_gene_fuzzy_selector.pl $build $out_file1 ${GALAXY_DATA_INDEX_DIR}/funDo.loc '$term'
|
||||
</command>
|
||||
|
||||
<inputs>
|
||||
<param name="build" type="select" label="Database build">
|
||||
<options from_file="funDo.loc">
|
||||
<column name="name" index="0"/>
|
||||
<column name="value" index="0"/>
|
||||
<filter type="unique_value" column="0"/>
|
||||
</options>
|
||||
</param>
|
||||
<param name="term" size="40" type="text" label="Disease term(s)" />
|
||||
</inputs>
|
||||
|
||||
<outputs>
|
||||
<data format="interval" name="out_file1">
|
||||
</data>
|
||||
</outputs>
|
||||
|
||||
<tests>
|
||||
<test>
|
||||
<param name="term" value="lung"/>
|
||||
<param name="build" value="hg18"/>
|
||||
<output name="out_file1" file="funDo_output1.interval" />
|
||||
</test>
|
||||
</tests>
|
||||
|
||||
<help>
|
||||
**Data formats**
|
||||
|
||||
There is no input dataset. The output is in interval_ format.
|
||||
|
||||
.. _interval: ./static/formatHelp.html#interval
|
||||
|
||||
-----
|
||||
|
||||
**What it does**
|
||||
|
||||
This tool searches the disease-term field of the DOLite mappings
|
||||
used by the FunDO project and returns a set of genes that
|
||||
are associated with terms matching the specified pattern. (This is the
|
||||
reverse of what FunDO's own server does.)
|
||||
|
||||
The search is case insensitive, and selects terms that contain any of
|
||||
the given words, either exactly or within a longer word (e.g. "nemia"
|
||||
selects not only "anemia", but also "hyperglycinemia", "tyrosinemias",
|
||||
and many other things). Multiple words should be separated by spaces,
|
||||
not commas. As a special case, entering the word "disease" returns all
|
||||
genes associated with any disease, even if that word does not actually
|
||||
appear in the term field.
|
||||
|
||||
home page: http://django.nubic.northwestern.edu/fundo/
|
||||
|
||||
-----
|
||||
|
||||
**Example**
|
||||
|
||||
Typing::
|
||||
|
||||
carcinoma
|
||||
|
||||
results in::
|
||||
|
||||
1. 2. 3. 4. 5. 6. 7.
|
||||
chr11 89507465 89565427 + NAALAD2 10003 Adenocarcinoma
|
||||
chr15 50189113 50192264 - BCL2L10 10017 Carcinoma
|
||||
chr7 150535855 150555250 - ABCF2 10061 Clear cell carcinoma
|
||||
chr7 150540508 150555250 - ABCF2 10061 Clear cell carcinoma
|
||||
chr10 134925911 134940397 - ADAM8 101 Adenocarcinoma
|
||||
chr10 134925911 134940397 - ADAM8 101 Adenocarcinoma
|
||||
etc.
|
||||
|
||||
where the column contents are as follows::
|
||||
|
||||
1. chromosome name.
|
||||
2. start position of the gene.
|
||||
3. end position of the gene.
|
||||
4. strand.
|
||||
4. gene name.
|
||||
6. Entrez Gene ID.
|
||||
7. disease term.
|
||||
|
||||
-----
|
||||
|
||||
**References**
|
||||
|
||||
Pan Du, Gang Feng,Jared Flatow, Jie Song, Michelle Holko, Warren A. Kibbe1 and
|
||||
Simon M. Lin. From disease ontology to disease-ontology lite: statistical methods to adapt a general-purpose ontology for the test of gene-ontology associations. Bioinformatics (2009) 25 (12):i63-i68.
|
||||
|
||||
Osborne JD, Flatow J, Holko M, Lin SM, Kibbe WA, Zhu LJ, Danila MI, Feng G, Chisholm RL. Annotating the human genome with Disease Ontology. BMC Genomics (2009) 10 S1:S6.
|
||||
</help>
|
||||
</tool>
|
||||
@@ -5,7 +5,8 @@
|
||||
# the specified command.
|
||||
#
|
||||
|
||||
MCRROOT=/opt/MATLAB/MATLAB_Compiler_Runtime/v711
|
||||
|
||||
MCRROOT=${MCRROOT:-/galaxy/software/linux2.6-x86_64/bin/MCR-7.11}
|
||||
MWE_ARCH=glnxa64
|
||||
|
||||
exe_name=$( basename $0 )
|
||||
|
||||
Reference in New Issue
Block a user