annotate variant_effect_predictor/Bio/DB/RefSeq.pm @ 3:d30fa12e4cc5 default tip

Merge heads 2:a5976b2dce6f and 1:09613ce8151e which were created as a result of a recently fixed bug.
author devteam <devteam@galaxyproject.org>
date Mon, 13 Jan 2014 10:38:30 -0500
parents 1f6dce3d34e0
children
Ignore whitespace changes - Everywhere: Within whitespace: At end of lines:
rev   line source
0
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
1 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
2 # $Id: RefSeq.pm,v 1.5 2002/10/22 07:38:29 lapp Exp $
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
3 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
4 # BioPerl module for Bio::DB::EMBL
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
5 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
6 # Cared for by Heikki Lehvaslaiho <Heikki@ebi.ac.uk>
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
7 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
8 # Copyright Jason Stajich
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
9 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
10 # You may distribute this module under the same terms as perl itself
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
11
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
12 # POD documentation - main docs before the code
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
13
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
14 =head1 NAME
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
15
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
16 Bio::DB::RefSeq - Database object interface for RefSeq retrieval
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
17
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
18 =head1 SYNOPSIS
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
19 use Bio::DB::RefSeq;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
20
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
21 $db = new Bio::DB::RefSeq;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
22
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
23 # most of the time RefSeq_ID eq RefSeq acc
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
24 $seq = $db->get_Seq_by_id('NM_006732'); # RefSeq ID
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
25 print "accession is ", $seq->accession_number, "\n";
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
26
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
27 # or changeing to accession number and Fasta format ...
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
28 $db->request_format('fasta');
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
29 $seq = $db->get_Seq_by_acc('NM_006732'); # RefSeq ACC
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
30 print "seq is ", $seq->seq, "\n";
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
31
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
32 # especially when using versions, you better be prepared
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
33 # in not getting what what want
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
34 eval {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
35 $seq = $db->get_Seq_by_version('NM_006732.1'); # RefSeq VERSION
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
36 };
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
37 print "accesion is ", $seq->accession_number, "\n" unless $@;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
38
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
39 # or ... best when downloading very large files, prevents
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
40 # keeping all of the file in memory
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
41
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
42 # also don't want features, just sequence so let's save bandwith
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
43 # and request Fasta sequence
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
44 $db = new Bio::DB::RefSeq(-retrievaltype => 'tempfile' ,
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
45 -format => 'fasta');
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
46 my $seqio = $db->get_Stream_by_batch(['NM_006732', 'NM_005252'] );
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
47 while( my $seq = $seqio->next_seq ) {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
48 print "seqid is ", $seq->id, "\n";
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
49 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
50
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
51 =head1 DESCRIPTION
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
52
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
53 Allows the dynamic retrieval of sequence objects L<Bio::Seq> from the
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
54 RefSeq database using the dbfetch script at EBI:
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
55 L<http:E<sol>E<sol>www.ebi.ac.ukE<sol>cgi-binE<sol>dbfetch>.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
56
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
57 In order to make changes transparent we have host type (currently only
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
58 ebi) and location (defaults to ebi) separated out. This allows later
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
59 additions of more servers in different geographical locations.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
60
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
61 The functionality of this module is inherited from L<Bio::DB::DBFetch>
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
62 which implements L<Bio::DB::WebDBSeqI>.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
63
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
64 This module retrieves entries from EBI although it
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
65 retrives database entries produced at NCBI. When read into bioperl
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
66 objects, the parser for GenBank format it used. RefSeq is a
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
67 NONSTANDARD GenBank file so be ready for surprises.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
68
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
69 =head1 FEEDBACK
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
70
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
71 =head2 Mailing Lists
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
72
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
73 User feedback is an integral part of the evolution of this and other
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
74 Bioperl modules. Send your comments and suggestions preferably to one
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
75 of the Bioperl mailing lists. Your participation is much appreciated.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
76
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
77 bioperl-l@bioperl.org - General discussion
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
78 http://bio.perl.org/MailList.html - About the mailing lists
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
79
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
80 =head2 Reporting Bugs
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
81
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
82 Report bugs to the Bioperl bug tracking system to help us keep track
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
83 the bugs and their resolution.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
84 Bug reports can be submitted via email or the web:
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
85
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
86 bioperl-bugs@bio.perl.org
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
87 http://bugzilla.bioperl.org/
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
88
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
89 =head1 AUTHOR - Heikki Lehvaslaiho
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
90
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
91 Email Heikki Lehvaslaiho E<lt>Heikki@ebi.ac.ukE<gt>
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
92
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
93 =head1 APPENDIX
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
94
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
95 The rest of the documentation details each of the object
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
96 methods. Internal methods are usually preceded with a _
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
97
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
98 =cut
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
99
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
100 # Let the code begin...
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
101
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
102 package Bio::DB::RefSeq;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
103 use strict;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
104 use vars qw(@ISA $MODVERSION %HOSTS %FORMATMAP $DEFAULTFORMAT);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
105
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
106 $MODVERSION = '0.1';
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
107 use Bio::DB::DBFetch;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
108
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
109 @ISA = qw(Bio::DB::DBFetch);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
110
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
111 BEGIN {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
112 # you can add your own here theoretically.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
113 %HOSTS = (
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
114 'dbfetch' => {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
115 baseurl => 'http://%s/cgi-bin/dbfetch?db=refseq&style=raw',
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
116 hosts => {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
117 'ebi' => 'www.ebi.ac.uk'
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
118 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
119 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
120 );
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
121 %FORMATMAP = ( 'embl' => 'embl',
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
122 'genbank' => 'genbank',
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
123 'fasta' => 'fasta'
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
124 );
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
125 $DEFAULTFORMAT = 'genbank';
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
126 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
127
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
128 sub new {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
129 my ($class, @args ) = @_;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
130 my $self = $class->SUPER::new(@args);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
131
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
132 $self->{ '_hosts' } = {};
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
133 $self->{ '_formatmap' } = {};
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
134
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
135 $self->hosts(\%HOSTS);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
136 $self->formatmap(\%FORMATMAP);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
137 $self->{'_default_format'} = $DEFAULTFORMAT;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
138
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
139 return $self;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
140 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
141
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
142 1;