annotate variant_effect_predictor/Bio/DB/EMBL.pm @ 0:1f6dce3d34e0

Uploaded
author mahtabm
date Thu, 11 Apr 2013 02:01:53 -0400
parents
children
Ignore whitespace changes - Everywhere: Within whitespace: At end of lines:
rev   line source
0
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
1 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
2 # $Id: EMBL.pm,v 1.12.2.1 2003/06/25 13:44:18 heikki Exp $
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
3 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
4 # BioPerl module for Bio::DB::EMBL
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
5 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
6 # Cared for by Heikki Lehvaslaiho <Heikki@ebi.ac.uk>
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
7 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
8 # Copyright Jason Stajich
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
9 #
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
10 # You may distribute this module under the same terms as perl itself
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
11
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
12 # POD documentation - main docs before the code
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
13
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
14 =head1 NAME
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
15
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
16 Bio::DB::EMBL - Database object interface for EMBL entry retrieval
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
17
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
18 =head1 SYNOPSIS
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
19
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
20 use Bio::DB::EMBL;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
21
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
22 $embl = new Bio::DB::EMBL;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
23
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
24 # remember that EMBL_ID does not equal GenBank_ID!
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
25 $seq = $embl->get_Seq_by_id('BUM'); # EMBL ID
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
26 print "cloneid is ", $seq->id, "\n";
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
27
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
28 # or changeing to accession number and Fasta format ...
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
29 $embl->request_format('fasta');
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
30 $seq = $embl->get_Seq_by_acc('J02231'); # EMBL ACC
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
31 print "cloneid is ", $seq->id, "\n";
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
32
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
33 # especially when using versions, you better be prepared
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
34 # in not getting what what want
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
35 eval {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
36 $seq = $embl->get_Seq_by_version('J02231.1'); # EMBL VERSION
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
37 };
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
38 print "cloneid is ", $seq->id, "\n" unless $@;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
39
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
40 # or ... best when downloading very large files, prevents
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
41 # keeping all of the file in memory
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
42
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
43 # also don't want features, just sequence so let's save bandwith
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
44 # and request Fasta sequence
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
45 $embl = new Bio::DB::EMBL(-retrievaltype => 'tempfile' ,
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
46 -format => 'fasta');
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
47 my $seqio = $embl->get_Stream_by_batch(['AC013798', 'AC021953'] );
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
48 while( my $clone = $seqio->next_seq ) {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
49 print "cloneid is ", $clone->id, "\n";
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
50 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
51
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
52 =head1 DESCRIPTION
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
53
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
54 Allows the dynamic retrieval of sequence objects L<Bio::Seq> from the
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
55 EMBL database using the dbfetch script at EBI:
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
56 L<http://www.ebi.ac.uk/cgi-bin/dbfetch>.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
57
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
58 In order to make changes transparent we have host type (currently only
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
59 ebi) and location (defaults to ebi) separated out. This allows later
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
60 additions of more servers in different geographical locations.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
61
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
62 The functionality of this module is inherited from L<Bio::DB::DBFetch>
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
63 which implements L<Bio::DB::WebDBSeqI>.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
64
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
65 =head1 FEEDBACK
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
66
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
67 =head2 Mailing Lists
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
68
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
69 User feedback is an integral part of the evolution of this and other
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
70 Bioperl modules. Send your comments and suggestions preferably to one
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
71 of the Bioperl mailing lists. Your participation is much appreciated.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
72
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
73 bioperl-l@bioperl.org - General discussion
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
74 http://bio.perl.org/MailList.html - About the mailing lists
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
75
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
76 =head2 Reporting Bugs
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
77
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
78 Report bugs to the Bioperl bug tracking system to help us keep track
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
79 the bugs and their resolution.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
80 Bug reports can be submitted via email or the web:
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
81
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
82 bioperl-bugs@bio.perl.org
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
83 http://bugzilla.bioperl.org/
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
84
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
85 =head1 AUTHOR - Heikki Lehvaslaiho
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
86
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
87 Email Heikki Lehvaslaiho E<lt>Heikki@ebi.ac.ukE<gt>
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
88
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
89 =head1 APPENDIX
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
90
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
91 The rest of the documentation details each of the object
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
92 methods. Internal methods are usually preceded with a _
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
93
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
94 =cut
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
95
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
96 # Let the code begin...
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
97
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
98 package Bio::DB::EMBL;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
99 use strict;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
100 use vars qw(@ISA $MODVERSION %HOSTS %FORMATMAP $DEFAULTFORMAT);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
101
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
102 $MODVERSION = '0.2';
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
103 use Bio::DB::DBFetch;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
104 use Bio::DB::RefSeq;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
105
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
106 @ISA = qw(Bio::DB::DBFetch);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
107
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
108 BEGIN {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
109 # you can add your own here theoretically.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
110 %HOSTS = (
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
111 'dbfetch' => {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
112 baseurl => 'http://%s/cgi-bin/dbfetch?db=embl&style=raw',
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
113 hosts => {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
114 'ebi' => 'www.ebi.ac.uk'
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
115 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
116 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
117 );
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
118 %FORMATMAP = ( 'embl' => 'embl',
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
119 'fasta' => 'fasta'
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
120 );
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
121 $DEFAULTFORMAT = 'embl';
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
122 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
123
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
124 =head2 new
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
125
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
126 Title : new
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
127 Usage : $gb = Bio::DB::GenBank->new(@options)
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
128 Function: Creates a new genbank handle
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
129 Returns : New genbank handle
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
130 Args : -delay number of seconds to delay between fetches (3s)
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
131
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
132 NOTE: There are other options that are used internally.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
133
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
134 =cut
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
135
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
136 sub new {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
137 my ($class, @args ) = @_;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
138 my $self = $class->SUPER::new(@args);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
139
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
140 $self->{ '_hosts' } = {};
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
141 $self->{ '_formatmap' } = {};
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
142
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
143 $self->hosts(\%HOSTS);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
144 $self->formatmap(\%FORMATMAP);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
145 $self->{'_default_format'} = $DEFAULTFORMAT;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
146
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
147 return $self;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
148 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
149
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
150
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
151 =head2 Bio::DB::WebDBSeqI methods
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
152
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
153 Overriding WebDBSeqI method to help newbies to retrieve sequences.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
154 EMBL database is all too often passed RefSeq accessions. This
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
155 redirects those calls. See L<Bio::DB::RefSeq>.
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
156
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
157
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
158 =head2 get_Stream_by_acc
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
159
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
160 Title : get_Stream_by_acc
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
161 Usage : $seq = $db->get_Seq_by_acc([$acc1, $acc2]);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
162 Function: Gets a series of Seq objects by accession numbers
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
163 Returns : a Bio::SeqIO stream object
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
164 Args : $ref : a reference to an array of accession numbers for
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
165 the desired sequence entries
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
166 Note : For GenBank, this just calls the same code for get_Stream_by_id()
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
167
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
168 =cut
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
169
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
170 sub get_Stream_by_acc {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
171 my ($self, $ids ) = @_;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
172 my $newdb = $self->_check_id($ids);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
173 if ($newdb && $newdb->isa('Bio::DB::RefSeq')) {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
174 return $newdb->get_seq_stream('-uids' => $ids, '-mode' => 'single');
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
175 } else {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
176 return $self->get_seq_stream('-uids' => $ids, '-mode' => 'single');
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
177 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
178 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
179
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
180
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
181 =head2 _check_id
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
182
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
183 Title : _check_id
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
184 Usage :
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
185 Function:
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
186 Returns : A Bio::DB::RefSeq reference or throws
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
187 Args : $id(s), $string
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
188 =cut
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
189
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
190 sub _check_id {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
191 my ($self, $ids) = @_;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
192
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
193 # NT contigs can not be retrieved
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
194 $self->throw("NT_ contigs are whole chromosome files which are not part of regular".
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
195 "database distributions. Go to ftp://ftp.ncbi.nih.gov/genomes/.")
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
196 if $ids =~ /NT_/;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
197
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
198 # Asking for a RefSeq from EMBL/GenBank
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
199
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
200 if ($ids =~ /N._/) {
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
201 $self->warn("[$ids] is not a normal sequence database but a RefSeq entry.".
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
202 " Redirecting the request.\n")
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
203 if $self->verbose >= 0;
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
204 return new Bio::DB::RefSeq(-verbose => $self->verbose);
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
205 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
206 }
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
207
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
208
1f6dce3d34e0 Uploaded
mahtabm
parents:
diff changeset
209 1;