Mercurial > repos > jankanis > blast2html
annotate blast_html.py @ 18:4434ffab721a
add a parameter for the template
author | Jan Kanis <jan.code@jankanis.nl> |
---|---|
date | Tue, 13 May 2014 15:26:20 +0200 |
parents | db7e4ee3be03 |
children | 67ddcb807b7d |
rev | line source |
---|---|
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
1 #!/usr/bin/env python3 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
2 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
3 # Copyright The Hyve B.V. 2014 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
4 # License: GPL version 3 or higher |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
5 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
6 import sys |
7
9e7927673089
intermediate commit before converting some tables to divs
Jan Kanis <jan.code@jankanis.nl>
parents:
5
diff
changeset
|
7 import math |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
8 import warnings |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
9 from os import path |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
10 from itertools import repeat |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
11 import argparse |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
12 from lxml import objectify |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
13 import jinja2 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
14 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
15 |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
16 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
17 _filters = {} |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
18 def filter(func_or_name): |
11
7660519f2dc9
proper layout for alignments, added some links
Jan Kanis <jan.code@jankanis.nl>
parents:
10
diff
changeset
|
19 "Decorator to register a function as filter in the current jinja environment" |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
20 if isinstance(func_or_name, str): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
21 def inner(func): |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
22 _filters[func_or_name] = func |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
23 return func |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
24 return inner |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
25 else: |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
26 _filters[func_or_name.__name__] = func_or_name |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
27 return func_or_name |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
28 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
29 |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
30 def color_idx(length): |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
31 if length < 40: |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
32 return 0 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
33 elif length < 50: |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
34 return 1 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
35 elif length < 80: |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
36 return 2 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
37 elif length < 200: |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
38 return 3 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
39 return 4 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
40 |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
41 @filter |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
42 def fmt(val, fmt): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
43 return format(float(val), fmt) |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
44 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
45 @filter |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
46 def firsttitle(hit): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
47 return hit.Hit_def.text.split('>')[0] |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
48 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
49 @filter |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
50 def othertitles(hit): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
51 """Split a hit.Hit_def that contains multiple titles up, splitting out the hit ids from the titles.""" |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
52 id_titles = hit.Hit_def.text.split('>') |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
53 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
54 titles = [] |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
55 for t in id_titles[1:]: |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
56 fullid, title = t.split(' ', 1) |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
57 hitid, id = fullid.split('|', 2)[1:3] |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
58 titles.append(dict(id = id, |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
59 hitid = hitid, |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
60 fullid = fullid, |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
61 title = title)) |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
62 return titles |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
63 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
64 @filter |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
65 def hitid(hit): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
66 return hit.Hit_id.text.split('|', 2)[1] |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
67 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
68 @filter |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
69 def seqid(hit): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
70 return hit.Hit_id.text.split('|', 2)[2] |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
71 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
72 @filter |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
73 def alignment_pre(hsp): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
74 return ( |
15 | 75 "Query {:>7s} {} {}\n".format(hsp['Hsp_query-from'].text, hsp.Hsp_qseq, hsp['Hsp_query-to']) + |
11
7660519f2dc9
proper layout for alignments, added some links
Jan Kanis <jan.code@jankanis.nl>
parents:
10
diff
changeset
|
76 " {:7s} {}\n".format('', hsp.Hsp_midline) + |
15 | 77 "Subject{:>7s} {} {}".format(hsp['Hsp_hit-from'].text, hsp.Hsp_hseq, hsp['Hsp_hit-to']) |
11
7660519f2dc9
proper layout for alignments, added some links
Jan Kanis <jan.code@jankanis.nl>
parents:
10
diff
changeset
|
78 ) |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
79 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
80 @filter('len') |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
81 def hsplen(node): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
82 return int(node['Hsp_align-len']) |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
83 |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
84 @filter |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
85 def asframe(frame): |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
86 if frame == 1: |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
87 return 'Plus' |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
88 elif frame == -1: |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
89 return 'Minus' |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
90 raise Exception("frame should be either +1 or -1") |
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
91 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
92 def genelink(hit, type='genbank', hsp=None): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
93 if not isinstance(hit, str): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
94 hit = hitid(hit) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
95 link = "http://www.ncbi.nlm.nih.gov/nucleotide/{}?report={}&log$=nuclalign".format(hit, type) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
96 if hsp != None: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
97 link += "&from={}&to={}".format(hsp['Hsp_hit-from'], hsp['Hsp_hit-to']) |
16 | 98 return link |
99 | |
100 | |
101 # javascript escape filter based on Django's, from https://github.com/dsissitka/khan-website/blob/master/templatefilters.py#L112-139 | |
102 # I've removed the html escapes, since html escaping is already being performed by the template engine. | |
7
9e7927673089
intermediate commit before converting some tables to divs
Jan Kanis <jan.code@jankanis.nl>
parents:
5
diff
changeset
|
103 |
16 | 104 _base_js_escapes = ( |
105 ('\\', r'\u005C'), | |
106 ('\'', r'\u0027'), | |
107 ('"', r'\u0022'), | |
108 # ('>', r'\u003E'), | |
109 # ('<', r'\u003C'), | |
110 # ('&', r'\u0026'), | |
111 # ('=', r'\u003D'), | |
112 # ('-', r'\u002D'), | |
113 # (';', r'\u003B'), | |
114 # (u'\u2028', r'\u2028'), | |
115 # (u'\u2029', r'\u2029') | |
116 ) | |
117 | |
118 # Escape every ASCII character with a value less than 32. This is | |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
119 # needed a.o. to prevent html parsers from jumping out of javascript |
16 | 120 # parsing mode. |
121 _js_escapes = (_base_js_escapes + | |
122 tuple(('%c' % z, '\\u%04X' % z) for z in range(32))) | |
123 | |
124 @filter | |
125 def js_string_escape(value): | |
126 """Escape javascript string literal escapes. Note that this only works | |
127 within javascript string literals, not in general javascript | |
128 snippets.""" | |
129 | |
130 value = str(value) | |
131 | |
132 for bad, good in _js_escapes: | |
133 value = value.replace(bad, good) | |
134 | |
135 return value | |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
136 |
7
9e7927673089
intermediate commit before converting some tables to divs
Jan Kanis <jan.code@jankanis.nl>
parents:
5
diff
changeset
|
137 |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
138 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
139 class BlastVisualize: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
140 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
141 colors = ('black', 'blue', 'green', 'magenta', 'red') |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
142 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
143 max_scale_labels = 10 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
144 |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
145 def __init__(self, input, templatedir, templatename): |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
146 self.input = input |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
147 self.templatename = templatename |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
148 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
149 self.blast = objectify.parse(self.input).getroot() |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
150 self.loader = jinja2.FileSystemLoader(searchpath=templatedir) |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
151 self.environment = jinja2.Environment(loader=self.loader, |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
152 lstrip_blocks=True, trim_blocks=True, autoescape=True) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
153 |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
154 self.environment.filters.update(_filters) |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
155 self.environment.filters['color'] = lambda length: match_colors[color_idx(length)] |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
156 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
157 self.query_length = int(self.blast["BlastOutput_query-len"]) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
158 self.hits = self.blast.BlastOutput_iterations.Iteration.Iteration_hits.Hit |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
159 # sort hits by longest hotspot first |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
160 self.ordered_hits = sorted(self.hits, |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
161 key=lambda h: max(hsplen(hsp) for hsp in h.Hit_hsps.Hsp), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
162 reverse=True) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
163 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
164 def render(self, output): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
165 template = self.environment.get_template(self.templatename) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
166 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
167 params = (('Query ID', self.blast["BlastOutput_query-ID"]), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
168 ('Query definition', self.blast["BlastOutput_query-def"]), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
169 ('Query length', self.blast["BlastOutput_query-len"]), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
170 ('Program', self.blast.BlastOutput_version), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
171 ('Database', self.blast.BlastOutput_db), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
172 ) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
173 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
174 if len(self.blast.BlastOutput_iterations.Iteration) > 1: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
175 warnings.warn("Multiple 'Iteration' elements found, showing only the first") |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
176 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
177 output.write(template.render(blast=self.blast, |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
178 length=self.query_length, |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
179 hits=self.blast.BlastOutput_iterations.Iteration.Iteration_hits.Hit, |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
180 colors=self.colors, |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
181 match_colors=self.match_colors(), |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
182 queryscale=self.queryscale(), |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
183 hit_info=self.hit_info(), |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
184 genelink=genelink, |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
185 params=params)) |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
186 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
187 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
188 def match_colors(self): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
189 """ |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
190 An iterator that yields lists of length-color pairs. |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
191 """ |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
192 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
193 percent_multiplier = 100 / self.query_length |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
194 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
195 for hit in self.hits: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
196 # sort hotspots from short to long, so we can overwrite index colors of |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
197 # short matches with those of long ones. |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
198 hotspots = sorted(hit.Hit_hsps.Hsp, key=lambda hsp: hsplen(hsp)) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
199 table = bytearray([255]) * self.query_length |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
200 for hsp in hotspots: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
201 frm = hsp['Hsp_query-from'] - 1 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
202 to = int(hsp['Hsp_query-to']) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
203 table[frm:to] = repeat(color_idx(hsplen(hsp)), to - frm) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
204 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
205 matches = [] |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
206 last = table[0] |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
207 count = 0 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
208 for i in range(self.query_length): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
209 if table[i] == last: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
210 count += 1 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
211 continue |
16 | 212 matches.append((count * percent_multiplier, self.colors[last] if last != 255 else 'transparent')) |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
213 last = table[i] |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
214 count = 1 |
16 | 215 matches.append((count * percent_multiplier, self.colors[last] if last != 255 else 'transparent')) |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
216 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
217 yield dict(colors=matches, link="#hit"+hit.Hit_num.text, defline=firsttitle(hit)) |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
218 |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
219 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
220 def queryscale(self): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
221 skip = math.ceil(self.query_length / self.max_scale_labels) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
222 percent_multiplier = 100 / self.query_length |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
223 for i in range(1, self.query_length+1): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
224 if i % skip == 0: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
225 yield dict(label = i, width = skip * percent_multiplier) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
226 if self.query_length % skip != 0: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
227 yield dict(label = self.query_length, width = (self.query_length % skip) * percent_multiplier) |
7
9e7927673089
intermediate commit before converting some tables to divs
Jan Kanis <jan.code@jankanis.nl>
parents:
5
diff
changeset
|
228 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
229 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
230 def hit_info(self): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
231 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
232 for hit in self.ordered_hits: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
233 hsps = hit.Hit_hsps.Hsp |
7
9e7927673089
intermediate commit before converting some tables to divs
Jan Kanis <jan.code@jankanis.nl>
parents:
5
diff
changeset
|
234 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
235 cover = [False] * self.query_length |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
236 for hsp in hsps: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
237 cover[hsp['Hsp_query-from']-1 : int(hsp['Hsp_query-to'])] = repeat(True, hsplen(hsp)) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
238 cover_count = cover.count(True) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
239 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
240 def hsp_val(path): |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
241 return (float(hsp[path]) for hsp in hsps) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
242 |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
243 yield dict(hit = hit, |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
244 title = firsttitle(hit), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
245 link_id = hit.Hit_num, |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
246 maxscore = "{:.1f}".format(max(hsp_val('Hsp_bit-score'))), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
247 totalscore = "{:.1f}".format(sum(hsp_val('Hsp_bit-score'))), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
248 cover = "{:.0%}".format(cover_count / self.query_length), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
249 e_value = "{:.4g}".format(min(hsp_val('Hsp_evalue'))), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
250 # FIXME: is this the correct formula vv? |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
251 ident = "{:.0%}".format(float(min(hsp.Hsp_identity / hsplen(hsp) for hsp in hsps))), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
252 accession = hit.Hit_accession) |
10
2fbdf2eb27b4
All data is displayed now, still some formatting to do
Jan Kanis <jan.code@jankanis.nl>
parents:
7
diff
changeset
|
253 |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
254 def main(): |
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
255 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
256 parser = argparse.ArgumentParser(description="Convert a BLAST XML result into a nicely readable html page", |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
257 usage="{} [-i] INPUT [-o OUTPUT]".format(sys.argv[0])) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
258 input_group = parser.add_mutually_exclusive_group(required=True) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
259 input_group.add_argument('positional_arg', metavar='INPUT', nargs='?', type=argparse.FileType(mode='r'), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
260 help='The input Blast XML file, same as -i/--input') |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
261 input_group.add_argument('-i', '--input', type=argparse.FileType(mode='r'), |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
262 help='The input Blast XML file') |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
263 parser.add_argument('-o', '--output', type=argparse.FileType(mode='w'), default=sys.stdout, |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
264 help='The output html file') |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
265 # We just want the file name here, so jinja can open the file |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
266 # itself. But it is easier to just use a FileType so argparse can |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
267 # handle the errors. This introduces a small race condition when |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
268 # jinja later tries to re-open the template file, but we don't |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
269 # care too much. |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
270 parser.add_argument('--template', type=argparse.FileType(mode='r'), default='blast_html.html.jinja', |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
271 help='The template file to use. Defaults to blast_html.html.jinja') |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
272 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
273 args = parser.parse_args() |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
274 if args.input == None: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
275 args.input = args.positional_arg |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
276 if args.input == None: |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
277 parser.error('no input specified') |
5
1df2bfce5c24
first features are working, partial match table
Jan Kanis <jan.code@jankanis.nl>
parents:
diff
changeset
|
278 |
18
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
279 templatedir, templatename = path.split(args.template.name) |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
280 args.template.close() |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
281 if not templatedir: |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
282 templatedir = '.' |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
283 |
4434ffab721a
add a parameter for the template
Jan Kanis <jan.code@jankanis.nl>
parents:
16
diff
changeset
|
284 b = BlastVisualize(args.input, templatedir, templatename) |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
285 b.render(args.output) |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
286 |
11
7660519f2dc9
proper layout for alignments, added some links
Jan Kanis <jan.code@jankanis.nl>
parents:
10
diff
changeset
|
287 |
12
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
288 if __name__ == '__main__': |
a459c754cdb5
add links, refactor, proper commandline arguments
Jan Kanis <jan.code@jankanis.nl>
parents:
11
diff
changeset
|
289 main() |
11
7660519f2dc9
proper layout for alignments, added some links
Jan Kanis <jan.code@jankanis.nl>
parents:
10
diff
changeset
|
290 |