annotate naive_output.r @ 67:ba33b94637ca draft

Uploaded
author davidvanzessen
date Tue, 29 Jan 2019 03:54:09 -0500
parents c33d93683a09
children
Ignore whitespace changes - Everywhere: Within whitespace: At end of lines:
rev   line source
0
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
1 args <- commandArgs(trailingOnly = TRUE)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
2
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
3 naive.file = args[1]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
4 shm.file = args[2]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
5 output.file.ca = args[3]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
6 output.file.cg = args[4]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
7 output.file.cm = args[5]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
8
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
9 naive = read.table(naive.file, sep="\t", header=T, quote="", fill=T)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
10 shm.merge = read.table(shm.file, sep="\t", header=T, quote="", fill=T)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
11
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
12
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
13 final = merge(naive, shm.merge[,c("Sequence.ID", "best_match")], by.x="ID", by.y="Sequence.ID")
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
14 print(paste("nrow final:", nrow(final)))
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
15 names(final)[names(final) == "best_match"] = "Sample"
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
16 final.numeric = final[,sapply(final, is.numeric)]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
17 final.numeric[is.na(final.numeric)] = 0
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
18 final[,sapply(final, is.numeric)] = final.numeric
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
19
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
20 final.ca = final[grepl("^ca", final$Sample),]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
21 final.cg = final[grepl("^cg", final$Sample),]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
22 final.cm = final[grepl("^cm", final$Sample),]
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
23
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
24 if(nrow(final.ca) > 0){
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
25 final.ca$Replicate = 1
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
26 }
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
27
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
28 if(nrow(final.cg) > 0){
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
29 final.cg$Replicate = 1
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
30 }
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
31
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
32 if(nrow(final.cm) > 0){
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
33 final.cm$Replicate = 1
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
34 }
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
35
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
36 #print(paste("nrow final:", nrow(final)))
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
37 #final2 = final
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
38 #final2$Sample = gsub("[0-9]", "", final2$Sample)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
39 #final = rbind(final, final2)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
40 #final$Replicate = 1
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
41
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
42 write.table(final.ca, output.file.ca, quote=F, sep="\t", row.names=F, col.names=T)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
43 write.table(final.cg, output.file.cg, quote=F, sep="\t", row.names=F, col.names=T)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
44 write.table(final.cm, output.file.cm, quote=F, sep="\t", row.names=F, col.names=T)
c33d93683a09 Uploaded
davidvanzessen
parents:
diff changeset
45