comparison tophat2_wrapper.xml @ 0:ffa30bedbee3

Imported from capsule None
author devteam
date Mon, 27 Jan 2014 09:26:57 -0500
parents
children ae06af1118dc
comparison
equal deleted inserted replaced
-1:000000000000 0:ffa30bedbee3
1 <tool id="tophat2" name="Tophat2" version="0.6">
2 <!-- Wrapper compatible with Tophat version 2.0.0+ -->
3 <description>Gapped-read mapper for RNA-seq data</description>
4 <version_command>tophat2 --version</version_command>
5 <requirements>
6 <requirement type="package" version="0.1.18">samtools</requirement>
7 <requirement type="package" version="2.1.0">bowtie2</requirement>
8 <requirement type="package" version="2.0.9">tophat2</requirement>
9 </requirements>
10
11 <command>
12 ##
13 ## Set path to index, building the reference if necessary.
14 ##
15
16 #set index_path = ''
17 #if $refGenomeSource.genomeSource == "history":
18 bowtie2-build "$refGenomeSource.ownFile" genome ; ln -s "$refGenomeSource.ownFile" genome.fa ;
19 #set index_path = 'genome'
20 #else:
21 #set index_path = $refGenomeSource.index.fields.path
22 #end if
23
24 ##
25 ## Run tophat.
26 ##
27
28 tophat2
29
30 ## Change this to accommodate the number of threads you have available.
31 --num-threads \${GALAXY_SLOTS:-4}
32
33 ## Set params.
34 #if $params.settingsType == "full":
35 --read-mismatches $params.read_mismatches
36 #if str($params.bowtie_n) == "Yes":
37 --bowtie-n
38 #end if
39
40 --read-edit-dist $params.read_edit_dist
41 --read-realign-edit-dist $params.read_realign_edit_dist
42 -a $params.anchor_length
43 -m $params.splice_mismatches
44 -i $params.min_intron_length
45 -I $params.max_intron_length
46 -g $params.max_multihits
47 --min-segment-intron $params.min_segment_intron
48 --max-segment-intron $params.max_segment_intron
49 --segment-mismatches $params.seg_mismatches
50 --segment-length $params.seg_length
51 --library-type $params.library_type
52
53 ## Indel search.
54 #if $params.indel_search.allow_indel_search == "Yes":
55 ## --allow-indels
56 --max-insertion-length $params.indel_search.max_insertion_length
57 --max-deletion-length $params.indel_search.max_deletion_length
58 #else:
59 --no-novel-indels
60 #end if
61
62 ## Supplying junctions parameters.
63 #if $params.own_junctions.use_junctions == "Yes":
64 #if $params.own_junctions.gene_model_ann.use_annotations == "Yes":
65 -G $params.own_junctions.gene_model_ann.gene_annotation_model
66 #end if
67 #if $params.own_junctions.raw_juncs.use_juncs == "Yes":
68 -j $params.own_junctions.raw_juncs.raw_juncs
69 #end if
70 #if str($params.own_junctions.no_novel_juncs) == "Yes":
71 --no-novel-juncs
72 #end if
73 #end if
74
75 #if $params.coverage_search.use_search == "Yes":
76 --coverage-search
77 --min-coverage-intron $params.coverage_search.min_coverage_intron
78 --max-coverage-intron $params.coverage_search.max_coverage_intron
79 #else:
80 --no-coverage-search
81 #end if
82
83 #if str($params.microexon_search) == "Yes":
84 --microexon-search
85 #end if
86
87 #if $params.fusion_search.do_search == "Yes":
88 --fusion-search
89 --fusion-anchor-length $params.fusion_search.anchor_len
90 --fusion-min-dist $params.fusion_search.min_dist
91 --fusion-read-mismatches $params.fusion_search.read_mismatches
92 --fusion-multireads $params.fusion_search.multireads
93 --fusion-multipairs $params.fusion_search.multipairs
94 --fusion-ignore-chromosomes "$params.fusion_search.ignore_chromosomes"
95 #end if
96
97 #if $params.bowtie2_settings.b2_settings == "Yes":
98 #if $params.bowtie2_settings.preset.b2_preset == "Yes":
99 --b2-$params.bowtie2_settings.preset.b2_preset_select
100 #end if
101 #end if
102
103 #end if
104
105 ## Read group information.
106 #if $readGroup.specReadGroup == "yes"
107 --rg-id "$readGroup.rgid"
108 --rg-library "$readGroup.rglb"
109 --rg-platform "$readGroup.rgpl"
110 --rg-sample "$readGroup.rgsm"
111 #end if
112
113 ## Set index path, inputs and parameters specific to paired data.
114 #if $singlePaired.sPaired == "paired"
115 -r $singlePaired.mate_inner_distance
116 --mate-std-dev=$singlePaired.mate_std_dev
117
118 #if str($singlePaired.report_discordant_pairs) == "No":
119 --no-discordant
120 #end if
121
122 ${index_path} $singlePaired.input1 $singlePaired.input2
123 #else
124 ${index_path} $singlePaired.input1
125 #end if
126 </command>
127
128 <inputs>
129 <conditional name="singlePaired">
130 <param name="sPaired" type="select" label="Is this library mate-paired?">
131 <option value="single">Single-end</option>
132 <option value="paired">Paired-end</option>
133 </param>
134 <when value="single">
135 <param format="fastqsanger" name="input1" type="data" label="RNA-Seq FASTQ file" help="Nucleotide-space: Must have Sanger-scaled quality values with ASCII offset 33"/>
136 </when>
137 <when value="paired">
138 <param format="fastqsanger" name="input1" type="data" label="RNA-Seq FASTQ file, forward reads" help="Nucleotide-space: Must have Sanger-scaled quality values with ASCII offset 33" />
139 <param format="fastqsanger" name="input2" type="data" label="RNA-Seq FASTQ file, reverse reads" help="Nucleotide-space: Must have Sanger-scaled quality values with ASCII offset 33" />
140 <param name="mate_inner_distance" type="integer" value="300" label="Mean Inner Distance between Mate Pairs" />
141 <param name="mate_std_dev" type="integer" value="20" label="Std. Dev for Distance between Mate Pairs" help="The standard deviation for the distribution on inner distances between mate pairs."/>
142 <!-- Discordant pairs. -->
143 <param name="report_discordant_pairs" type="select" label="Report discordant pair alignments?">
144 <option value="No">No</option>
145 <option selected="True" value="Yes">Yes</option>
146 </param>
147 </when>
148 </conditional>
149 <expand macro="refGenomeSourceConditional">
150 <options from_data_table="tophat2_indexes">
151 <filter type="sort_by" column="2"/>
152 <validator type="no_options" message="No genomes are available for the selected input dataset"/>
153 </options>
154 </expand>
155 <conditional name="params">
156 <param name="settingsType" type="select" label="TopHat settings to use" help="You can use the default settings or set custom values for any of Tophat's parameters.">
157 <option value="preSet">Use Defaults</option>
158 <option value="full">Full parameter list</option>
159 </param>
160 <when value="preSet" />
161 <!-- Full/advanced params. -->
162 <when value="full">
163 <param name="read_realign_edit_dist" type="integer" value="1000" label="Max realign edit distance" help="Some of the reads spanning multiple exons may be mapped incorrectly as a contiguous alignment to the genome even though the correct alignment should be a spliced one - this can happen in the presence of processed pseudogenes that are rarely (if at all) transcribed or expressed. This option can direct TopHat to re-align reads for which the edit distance of an alignment obtained in a previous mapping step is above or equal to this option value. If you set this option to 0, TopHat will map every read in all the mapping steps (transcriptome if you provided gene annotations, genome, and finally splice variants detected by TopHat), reporting the best possible alignment found in any of these mapping steps. This may greatly increase the mapping accuracy at the expense of an increase in running time. The default value for this option is set such that TopHat will not try to realign reads already mapped in earlier steps." />
164
165 <param name="read_edit_dist" type="integer" value="2" label="Max edit distance" help="Final read alignments having more than these many edit distance are discarded." />
166
167 <param name="library_type" type="select" label="Library Type" help="TopHat will treat the reads as strand specific. Every read alignment will have an XS attribute tag. Consider supplying library type options below to select the correct RNA-seq protocol.">
168 <option value="fr-unstranded">FR Unstranded</option>
169 <option value="fr-firststrand">FR First Strand</option>
170 <option value="fr-secondstrand">FR Second Strand</option>
171 </param>
172 <param name="read_mismatches" type="integer" value="2" label="Final read mismatches" help="Final read alignments having more than these many mismatches are discarded." />
173 <param name="bowtie_n" type="select" label="Use bowtie -n mode">
174 <option selected="true" value="No">No</option>
175 <option value="Yes">Yes</option>
176 </param>
177 <param name="anchor_length" type="integer" value="8" label="Anchor length (at least 3)" help="Report junctions spanned by reads with at least this many bases on each side of the junction." />
178 <param name="splice_mismatches" type="integer" value="0" label="Maximum number of mismatches that can appear in the anchor region of spliced alignment" />
179 <param name="min_intron_length" type="integer" value="70" label="The minimum intron length" help="TopHat will ignore donor/acceptor pairs closer than this many bases apart." />
180 <param name="max_intron_length" type="integer" value="500000" label="The maximum intron length" help="When searching for junctions ab initio, TopHat will ignore donor/acceptor pairs farther than this many bases apart, except when such a pair is supported by a split segment alignment of a long read." />
181 <expand macro="indel_searchConditional" />
182 alignments (number of reads divided by average depth of coverage)" help="0.0 to 1.0 (0 to turn off)" />
183 <param name="max_multihits" type="integer" value="20" label="Maximum number of alignments to be allowed" />
184 <param name="min_segment_intron" type="integer" value="50" label="Minimum intron length that may be found during split-segment (default) search" />
185 <param name="max_segment_intron" type="integer" value="500000" label="Maximum intron length that may be found during split-segment (default) search" />
186 <param name="seg_mismatches" type="integer" min="0" max="3" value="2" label="Number of mismatches allowed in each segment alignment for reads mapped independently" />
187 <param name="seg_length" type="integer" value="25" label="Minimum length of read segments" />
188
189 <!-- Options for supplying own junctions. -->
190 <expand macro="own_junctionsConditional" />
191 <!-- Coverage search. -->
192 <conditional name="coverage_search">
193 <param name="use_search" type="select" label="Use Coverage Search" help="Enables the coverage based search for junctions. Use when coverage search is disabled by default (such as for reads 75bp or longer), for maximum sensitivity.">
194 <option selected="true" value="No">No</option>
195 <option value="Yes">Yes</option>
196 </param>
197 <when value="Yes">
198 <param name="min_coverage_intron" type="integer" value="50" label="Minimum intron length that may be found during coverage search" />
199 <param name="max_coverage_intron" type="integer" value="20000" label="Maximum intron length that may be found during coverage search" />
200 </when>
201 <when value="No" />
202 </conditional>
203
204 <!-- Microexon search params -->
205 <param name="microexon_search" type="select" label="Use Microexon Search" help="With this option, the pipeline will attempt to find alignments incident to microexons. Works only for reads 50bp or longer.">
206 <option value="No">No</option>
207 <option value="Yes">Yes</option>
208 </param>
209
210 <!-- Fusion mapping. -->
211 <conditional name="fusion_search">
212 <param name="do_search" type="select" label="Do Fusion Search">
213 <option selected="true" value="No">No</option>
214 <option value="Yes">Yes</option>
215 </param>
216 <when value="No" />
217 <when value="Yes">
218 <param name="anchor_len" type="integer" value="20" label="Anchor Length" help="A 'supporting' read must map to both sides of a fusion by at least this many bases."/>
219 <param name="min_dist" type="integer" value="10000000" label="Minimum Distance" help="For intra-chromosomal fusions, TopHat-Fusion tries to find fusions separated by at least this distance."/>
220 <param name="read_mismatches" type="integer" value="2" label="Read Mismatches" help="Reads support fusions if they map across fusion with at most this many mismatches."/>
221 <param name="multireads" type="integer" value="2" label="Multireads" help="Reads that map to more than this many places will be ignored. It may be possible that a fusion is supported by reads (or pairs) that map to multiple places."/>
222 <param name="multipairs" type="integer" value="2" label="Multipairs" help="Pairs that map to more than this many places will be ignored."/>
223 <param name="ignore_chromosomes" type="text" value='' label="Ignore some chromosomes such as chrM when detecting fusion break points"/>
224 </when>
225 </conditional>
226
227 <!-- Bowtie2 settings. -->
228 <conditional name="bowtie2_settings">
229 <param name="b2_settings" type="select" label="Set Bowtie2 settings">
230 <option selected="true" value="No">No</option>
231 <option value="Yes">Yes</option>
232 </param>
233 <when value="No" />
234 <when value="Yes">
235 <conditional name="preset">
236 <param name="b2_preset" type="select" label="Use Preset options">
237 <option selected="true" value="Yes">Yes</option>
238 <option value="No">No</option>
239 </param>
240 <when value="Yes">
241 <param name="b2_preset_select" type="select" label="Preset option">
242 <option value="very-fast">Very fast</option>
243 <option value="fast">Fast</option>
244 <option selected="true" value="sensitive">Sensitive</option>
245 <option value="very-sensitive">Very sensitive</option>
246 </param>
247 </when>
248 <!-- TODO: -->
249 <when value="No" />
250 </conditional>
251 </when>
252 </conditional>
253 </when> <!-- full -->
254 </conditional> <!-- params -->
255 <conditional name="readGroup">
256 <param name="specReadGroup" type="select" label="Specify read group?">
257 <option value="yes">Yes</option>
258 <option value="no" selected="True">No</option>
259 </param>
260 <when value="yes">
261 <param name="rgid" type="text" size="25" label="Read group identifier (ID). Each @RG line must have a unique ID. The value of ID is used in the RG tags of alignment records. Must be unique among all read groups in header section." help="Required if RG specified. Read group IDs may be modified when merging SAM files in order to handle collisions." />
262 <param name="rglb" type="text" size="25" label="Library name (LB)" help="Required if RG specified" />
263 <param name="rgpl" type="text" size="25" label="Platform/technology used to produce the reads (PL)" help="Required if RG specified. Valid values : CAPILLARY, LS454, ILLUMINA, SOLID, HELICOS, IONTORRENT and PACBIO" />
264 <param name="rgsm" type="text" size="25" label="Sample (SM)" help="Required if RG specified. Use pool name where a pool is being sequenced" />
265 </when>
266 <when value="no" />
267 </conditional> <!-- readGroup -->
268 </inputs>
269
270 <stdio>
271 <regex match="Exception|Error" source="both" level="fatal" description="Tool execution failed"/>
272 <regex match=".*" source="both" level="log" description="tool progress"/>
273 </stdio>
274
275 <outputs>
276 <data format="txt" name="align_summary" label="${tool.name} on ${on_string}: align_summary" from_work_dir="tophat_out/align_summary.txt"/>
277 <data format="tabular" name="fusions" label="${tool.name} on ${on_string}: fusions" from_work_dir="tophat_out/fusions.out">
278 <filter>(params['settingsType'] == 'full' and params['fusion_search']['do_search'] == 'Yes')</filter>
279 </data>
280 <data format="bed" name="insertions" label="${tool.name} on ${on_string}: insertions" from_work_dir="tophat_out/insertions.bed">
281 <expand macro="dbKeyActions" />
282 </data>
283 <data format="bed" name="deletions" label="${tool.name} on ${on_string}: deletions" from_work_dir="tophat_out/deletions.bed">
284 <expand macro="dbKeyActions" />
285 </data>
286 <data format="bed" name="junctions" label="${tool.name} on ${on_string}: splice junctions" from_work_dir="tophat_out/junctions.bed">
287 <expand macro="dbKeyActions" />
288 </data>
289 <data format="bam" name="accepted_hits" label="${tool.name} on ${on_string}: accepted_hits" from_work_dir="tophat_out/accepted_hits.bam">
290 <expand macro="dbKeyActions" />
291 </data>
292 </outputs>
293
294 <macros>
295 <import>tophat_macros.xml</import>
296 <macro name="dbKeyActions">
297 <actions>
298 <conditional name="refGenomeSource.genomeSource">
299 <when value="indexed">
300 <action type="metadata" name="dbkey">
301 <option type="from_data_table" name="tophat2_indexes" column="1" offset="0">
302 <filter type="param_value" column="0" value="#" compare="startswith" keep="False"/>
303 <filter type="param_value" ref="refGenomeSource.index" column="0"/>
304 </option>
305 </action>
306 </when>
307 <when value="history">
308 <action type="metadata" name="dbkey">
309 <option type="from_param" name="refGenomeSource.ownFile" param_attribute="dbkey" />
310 </action>
311 </when>
312 </conditional>
313 </actions>
314 </macro>
315 </macros>
316
317 <tests>
318 <!-- Test base-space single-end reads with pre-built index and preset parameters -->
319 <test>
320 <!-- TopHat commands:
321 tophat2 -o tmp_dir -p 1 tophat_in1 test-data/tophat_in2.fastqsanger
322 Rename the files in tmp_dir appropriately
323 -->
324 <param name="sPaired" value="single" />
325 <param name="input1" ftype="fastqsanger" value="tophat_in2.fastqsanger" />
326 <param name="genomeSource" value="indexed" />
327 <param name="index" value="tophat_test" />
328 <param name="settingsType" value="preSet" />
329 <param name="specReadGroup" value="No" />
330 <output name="junctions" file="tophat2_out1j.bed" />
331 <output name="accepted_hits" file="tophat_out1h.bam" compare="sim_size" />
332 </test>
333 <!-- Test using base-space test data: paired-end reads, index from history. -->
334 <test>
335 <!-- TopHat commands:
336 bowtie2-build -f test-data/tophat_in1.fasta tophat_in1
337 tophat2 -o tmp_dir -p 1 -r 20 tophat_in1 test-data/tophat_in2.fastqsanger test-data/tophat_in3.fastqsanger
338 Rename the files in tmp_dir appropriately
339 -->
340 <param name="sPaired" value="paired" />
341 <param name="input1" ftype="fastqsanger" value="tophat_in2.fastqsanger" />
342 <param name="input2" ftype="fastqsanger" value="tophat_in3.fastqsanger" />
343 <param name="genomeSource" value="history" />
344 <param name="ownFile" ftype="fasta" value="tophat_in1.fasta" />
345 <param name="mate_inner_distance" value="20" />
346 <param name="settingsType" value="preSet" />
347 <param name="specReadGroup" value="No" />
348 <output name="junctions" file="tophat2_out2j.bed" />
349 <output name="accepted_hits" file="tophat_out2h.bam" compare="sim_size" />
350 </test>
351 <!-- Test base-space single-end reads with user-supplied reference fasta and full parameters -->
352 <test>
353 <!-- Tophat commands:
354 bowtie2-build -f test-data/tophat_in1.fasta tophat_in1
355 tophat2 -o tmp_dir -p 1 -a 8 -m 0 -i 70 -I 500000 -g 40 +coverage-search +min-coverage-intron 50 +max-coverage-intro 20000 +segment-mismatches 2 +segment-length 25 +microexon-search tophat_in1 test-data/tophat_in2.fastqsanger
356 Replace the + with double-dash
357 Rename the files in tmp_dir appropriately
358 -->
359 <param name="sPaired" value="single"/>
360 <param name="input1" ftype="fastqsanger" value="tophat_in2.fastqsanger"/>
361 <param name="genomeSource" value="history"/>
362 <param name="ownFile" value="tophat_in1.fasta"/>
363 <param name="settingsType" value="full"/>
364 <param name="library_type" value="FR Unstranded"/>
365 <param name="read_mismatches" value="2"/>
366 <param name="bowtie_n" value="No"/>
367 <param name="anchor_length" value="8"/>
368 <param name="splice_mismatches" value="0"/>
369 <param name="min_intron_length" value="70"/>
370 <param name="max_intron_length" value="500000"/>
371 <param name="max_multihits" value="40"/>
372 <param name="min_segment_intron" value="50" />
373 <param name="max_segment_intron" value="500000" />
374 <param name="seg_mismatches" value="2"/>
375 <param name="seg_length" value="25"/>
376 <param name="allow_indel_search" value="Yes"/>
377 <param name="max_insertion_length" value="3"/>
378 <param name="max_deletion_length" value="3"/>
379 <param name="use_junctions" value="Yes" />
380 <param name="use_annotations" value="No" />
381 <param name="use_juncs" value="No" />
382 <param name="no_novel_juncs" value="No" />
383 <param name="use_search" value="Yes" />
384 <param name="min_coverage_intron" value="50" />
385 <param name="max_coverage_intron" value="20000" />
386 <param name="microexon_search" value="Yes" />
387 <param name="b2_settings" value="No" />
388 <!-- Fusion search params -->
389 <param name="do_search" value="Yes" />
390 <param name="anchor_len" value="21" />
391 <param name="min_dist" value="10000021" />
392 <param name="read_mismatches" value="3" />
393 <param name="multireads" value="4" />
394 <param name="multipairs" value="5" />
395 <param name="ignore_chromosomes" value="chrM"/>
396 <param name="specReadGroup" value="No" />
397 <output name="insertions" file="tophat_out3i.bed" />
398 <output name="deletions" file="tophat_out3d.bed" />
399 <output name="junctions" file="tophat2_out3j.bed" />
400 <output name="accepted_hits" file="tophat_out3h.bam" compare="sim_size" />
401 </test>
402 <!-- Test base-space paired-end reads with user-supplied reference fasta and full parameters -->
403 <test>
404 <!-- TopHat commands:
405 tophat2 -o tmp_dir -r 20 -p 1 -a 8 -m 0 -i 70 -I 500000 -g 40 +coverage-search +min-coverage-intron 50 +max-coverage-intro 20000 +segment-mismatches 2 +segment-length 25 +microexon-search +report_discordant_pairs tophat_in1 test-data/tophat_in2.fastqsanger test-data/tophat_in3.fastqsanger
406 Replace the + with double-dash
407 Rename the files in tmp_dir appropriately
408 -->
409 <param name="sPaired" value="paired"/>
410 <param name="input1" ftype="fastqsanger" value="tophat_in2.fastqsanger"/>
411 <param name="input2" ftype="fastqsanger" value="tophat_in3.fastqsanger"/>
412 <param name="genomeSource" value="indexed"/>
413 <param name="index" value="tophat_test"/>
414 <param name="mate_inner_distance" value="20"/>
415 <param name="settingsType" value="full"/>
416 <param name="library_type" value="FR Unstranded"/>
417 <param name="read_mismatches" value="5"/>
418 <param name="bowtie_n" value="Yes"/>
419 <param name="mate_std_dev" value="20"/>
420 <param name="anchor_length" value="8"/>
421 <param name="splice_mismatches" value="0"/>
422 <param name="min_intron_length" value="70"/>
423 <param name="max_intron_length" value="500000"/>
424 <param name="max_multihits" value="40"/>
425 <param name="min_segment_intron" value="50" />
426 <param name="max_segment_intron" value="500000" />
427 <param name="seg_mismatches" value="2"/>
428 <param name="seg_length" value="25"/>
429 <param name="allow_indel_search" value="No"/>
430 <param name="use_junctions" value="Yes" />
431 <param name="use_annotations" value="No" />
432 <param name="use_juncs" value="No" />
433 <param name="no_novel_juncs" value="No" />
434 <param name="report_discordant_pairs" value="Yes" />
435 <param name="use_search" value="No" />
436 <param name="microexon_search" value="Yes" />
437 <param name="b2_settings" value="No" />
438 <!-- Fusion search params -->
439 <param name="do_search" value="Yes" />
440 <param name="anchor_len" value="21" />
441 <param name="min_dist" value="10000021" />
442 <param name="read_mismatches" value="3" />
443 <param name="multireads" value="4" />
444 <param name="multipairs" value="5" />
445 <param name="ignore_chromosomes" value="chrM"/>
446 <param name="specReadGroup" value="No" />
447 <output name="junctions" file="tophat2_out4j.bed" />
448 <output name="accepted_hits" file="tophat_out4h.bam" compare="sim_size" />
449 </test>
450 </tests>
451
452 <help>
453 **Tophat Overview**
454
455 TopHat_ is a fast splice junction mapper for RNA-Seq reads. It aligns RNA-Seq reads to mammalian-sized genomes using the ultra high-throughput short read aligner Bowtie(2), and then analyzes the mapping results to identify splice junctions between exons. Please cite: Kim D, Pertea G, Trapnell C, Pimentel H, Kelley R, and Salzberg SL. TopHat2: accurate alignment
456 of transcriptomes in the presence of insertions, deletions and gene fusions. Genome Biol 14:R36, 2013.
457
458 .. _Tophat: http://tophat.cbcb.umd.edu/
459
460 ------
461
462 **Know what you are doing**
463
464 .. class:: warningmark
465
466 There is no such thing (yet) as an automated gearshift in splice junction identification. It is all like stick-shift driving in San Francisco. In other words, running this tool with default parameters will probably not give you meaningful results. A way to deal with this is to **understand** the parameters by carefully reading the `documentation`__ and experimenting. Fortunately, Galaxy makes experimenting easy.
467
468 .. __: http://tophat.cbcb.umd.edu/manual.html
469
470 ------
471
472 **Input formats**
473
474 Tophat accepts files in Sanger FASTQ format. Use the FASTQ Groomer to prepare your files.
475
476 ------
477
478 **Outputs**
479
480 Tophat produces two output files:
481
482 - junctions -- A UCSC BED_ track of junctions reported by TopHat. Each junction consists of two connected BED blocks, where each block is as long as the maximal overhang of any read spanning the junction. The score is the number of alignments spanning the junction.
483 - accepted_hits -- A list of read alignments in BAM_ format.
484
485 .. _BED: http://genome.ucsc.edu/FAQ/FAQformat.html#format1
486 .. _BAM: http://samtools.sourceforge.net/
487
488 Two other possible outputs, depending on the options you choose, are insertions and deletions, both of which are in BED format.
489
490 -------
491
492 **Tophat settings**
493
494 All of the options have a default value. You can change any of them. Some of the options in Tophat have been implemented here.
495
496 ------
497
498 **Tophat parameter list**
499
500 This is a list of implemented Tophat options::
501
502 -r This is the expected (mean) inner distance between mate pairs. For, example, for paired end runs with fragments
503 selected at 300bp, where each end is 50bp, you should set -r to be 200. There is no default, and this parameter
504 is required for paired end runs.
505 --mate-std-dev INT The standard deviation for the distribution on inner distances between mate pairs. The default is 20bp.
506 -a/--min-anchor-length INT The "anchor length". TopHat will report junctions spanned by reads with at least this many bases on each side of the junction. Note that individual spliced
507 alignments may span a junction with fewer than this many bases on one side. However, every junction involved in spliced alignments is supported by at least one
508 read with this many bases on each side. This must be at least 3 and the default is 8.
509 -m/--splice-mismatches INT The maximum number of mismatches that may appear in the "anchor" region of a spliced alignment. The default is 0.
510 -i/--min-intron-length INT The minimum intron length. TopHat will ignore donor/acceptor pairs closer than this many bases apart. The default is 70.
511 -I/--max-intron-length INT The maximum intron length. When searching for junctions ab initio, TopHat will ignore donor/acceptor pairs farther than this many bases apart, except when such a pair is supported by a split segment alignment of a long read. The default is 500000.
512 -g/--max-multihits INT Instructs TopHat to allow up to this many alignments to the reference for a given read, and suppresses all alignments for reads with more than this many
513 alignments. The default is 40.
514 -G/--GTF [GTF 2.2 file] Supply TopHat with a list of gene model annotations. TopHat will use the exon records in this file to build a set of known splice junctions for each gene, and will attempt to align reads to these junctions even if they would not normally be covered by the initial mapping.
515 -j/--raw-juncs [juncs file] Supply TopHat with a list of raw junctions. Junctions are specified one per line, in a tab-delimited format. Records look like: [chrom] [left] [right] [+/-], left and right are zero-based coordinates, and specify the last character of the left sequenced to be spliced to the first character of the right sequence, inclusive.
516 -no-novel-juncs Only look for junctions indicated in the supplied GFF file. (ignored without -G)
517 --no-coverage-search Disables the coverage based search for junctions.
518 --coverage-search Enables the coverage based search for junctions. Use when coverage search is disabled by default (such as for reads 75bp or longer), for maximum sensitivity.
519 --microexon-search With this option, the pipeline will attempt to find alignments incident to microexons. Works only for reads 50bp or longer.
520 --segment-mismatches Read segments are mapped independently, allowing up to this many mismatches in each segment alignment. The default is 2.
521 --segment-length Each read is cut up into segments, each at least this long. These segments are mapped independently. The default is 25.
522 --min-coverage-intron The minimum intron length that may be found during coverage search. The default is 50.
523 --max-coverage-intron The maximum intron length that may be found during coverage search. The default is 20000.
524 --min-segment-intron The minimum intron length that may be found during split-segment search. The default is 50.
525 --max-segment-intron The maximum intron length that may be found during split-segment search. The default is 500000.
526 </help>
527 </tool>