mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
Add files via upload
This commit is contained in:
parent
5e9a793275
commit
75a9dd10ab
4 changed files with 328 additions and 42 deletions
62
pfam_folder_pred.py
Normal file
62
pfam_folder_pred.py
Normal file
|
|
@ -0,0 +1,62 @@
|
||||||
|
import json
|
||||||
|
import gc
|
||||||
|
import pandas as pd
|
||||||
|
import numpy as np
|
||||||
|
import pickle
|
||||||
|
import sys
|
||||||
|
import progressbar
|
||||||
|
import os
|
||||||
|
from neighbor_genes import read_genome_maps
|
||||||
|
from process_data import create_data_homology_ls
|
||||||
|
from read_get_gene_seq import read_gene_sequences
|
||||||
|
from access_data_rest import update_rest,update_rest_protein
|
||||||
|
from prepare_synteny_matrix import write_fasta
|
||||||
|
from process_data import create_map_list
|
||||||
|
|
||||||
|
def read_database(fname,dirname):
|
||||||
|
df=pd.read_csv(dirname+"/"+fname,sep="\t",header=None)
|
||||||
|
label_dict=dict(ortholog_one2one=1,
|
||||||
|
other_paralog=0,
|
||||||
|
non_homolog=2,
|
||||||
|
ortholog_one2many=1,
|
||||||
|
ortholog_many2many=1,
|
||||||
|
within_species_paralog=0,
|
||||||
|
gene_split=4)
|
||||||
|
label=[]
|
||||||
|
for _,row in df.iterrows():
|
||||||
|
label.append(label_dict[row[7]])
|
||||||
|
df=df.assign(label=label)
|
||||||
|
df=df.drop(7,axis=1)
|
||||||
|
df=df.drop(0,axis=1)
|
||||||
|
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","goc","wga","label"]
|
||||||
|
return df
|
||||||
|
|
||||||
|
def read_prediction_file_folder(dir_name):
|
||||||
|
lf=os.listdir(dir_name)
|
||||||
|
a_h=[]
|
||||||
|
d_h=[]
|
||||||
|
for x in progressbar.progressbar(lf):
|
||||||
|
df=read_database(x,dir_name)
|
||||||
|
a_h.append(df)
|
||||||
|
d_h.append(x.split(".")[0])
|
||||||
|
return a_h,d_h
|
||||||
|
|
||||||
|
def create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,name):
|
||||||
|
lsy=create_data_homology_ls(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,0)
|
||||||
|
protein_sequences=read_gene_sequences(a_h,lsy,"pro_seq","prediction_"+name)
|
||||||
|
protein_sequences=update_rest_protein(protein_sequences,"prediction_"+name)
|
||||||
|
write_fasta(protein_sequences,"prediction_"+name)
|
||||||
|
print("Protein Sequences Loaded")
|
||||||
|
|
||||||
|
def main():
|
||||||
|
arg=sys.argv
|
||||||
|
dirname=arg[-1]
|
||||||
|
a_h,d_h=read_prediction_file_folder(dirname)
|
||||||
|
n=3
|
||||||
|
|
||||||
|
a,d,ld,ldg,cmap,cimap=read_genome_maps()#read the genome maps
|
||||||
|
print("Genome Maps Loaded.")
|
||||||
|
|
||||||
|
create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,dirname)
|
||||||
|
if __name__=="__main__":
|
||||||
|
main()
|
||||||
213
prediction_pfam.py
Normal file
213
prediction_pfam.py
Normal file
|
|
@ -0,0 +1,213 @@
|
||||||
|
import json
|
||||||
|
import gc
|
||||||
|
import pandas as pd
|
||||||
|
import numpy as np
|
||||||
|
import pickle
|
||||||
|
import tensorflow as tf
|
||||||
|
import sys
|
||||||
|
import os
|
||||||
|
import progressbar
|
||||||
|
from neighbor_genes import read_genome_maps
|
||||||
|
from process_data import create_data_homology_ls
|
||||||
|
from read_get_gene_seq import read_gene_sequences
|
||||||
|
from access_data_rest import update_rest,update_rest_protein
|
||||||
|
from threads import Procerssrunner
|
||||||
|
from prepare_synteny_matrix import read_data_synteny
|
||||||
|
from tree_data import create_tree_data
|
||||||
|
from process_data import create_map_list
|
||||||
|
from pfam_parser import pfam_parse
|
||||||
|
from pfam_matrix import create_pfam_map,create_pfam_matrix
|
||||||
|
|
||||||
|
def read_database(fname):
|
||||||
|
df=pd.read_csv(fname,sep="\t",header=None)
|
||||||
|
label_dict=dict(ortholog_one2one=1,
|
||||||
|
other_paralog=0,
|
||||||
|
non_homolog=2,
|
||||||
|
ortholog_one2many=1,
|
||||||
|
ortholog_many2many=1,
|
||||||
|
within_species_paralog=0,
|
||||||
|
gene_split=4)
|
||||||
|
label=[]
|
||||||
|
for _,row in df.iterrows():
|
||||||
|
label.append(label_dict[row[7]])
|
||||||
|
df=df.assign(label=label)
|
||||||
|
df=df.drop(7,axis=1)
|
||||||
|
df=df.drop(0,axis=1)
|
||||||
|
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","goc","wga","label"]
|
||||||
|
return df
|
||||||
|
|
||||||
|
def select_data_by_length(df,st,end):
|
||||||
|
try:
|
||||||
|
if end<len(df):
|
||||||
|
if st<end:
|
||||||
|
df=df.loc[df.index.values[st:end]]
|
||||||
|
else:
|
||||||
|
raise ValueError()
|
||||||
|
except:
|
||||||
|
print("Making Predictions for the complete dataframe:)")
|
||||||
|
print(len(df))
|
||||||
|
return df
|
||||||
|
|
||||||
|
def create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,name):
|
||||||
|
lsy=create_data_homology_ls(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,0)
|
||||||
|
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","prediction_"+name)
|
||||||
|
gene_sequences=update_rest(gene_sequences,"prediction_"+name)
|
||||||
|
print("Gene Sequences Loaded.")
|
||||||
|
return lsy,gene_sequences
|
||||||
|
|
||||||
|
def threadmaker(nop,df,lsy,gene_sequences,n,name):
|
||||||
|
part=len(df)//nop
|
||||||
|
pr=Procerssrunner()
|
||||||
|
pr.start_processes(nop,df,gene_sequences,lsy,part,n,name)
|
||||||
|
smg,sml,indexes=read_data_synteny(nop,name)
|
||||||
|
sml=np.array(sml)
|
||||||
|
smg=np.array(smg)
|
||||||
|
indexes=np.array(indexes)
|
||||||
|
return sml,smg,indexes
|
||||||
|
|
||||||
|
def get_prediction(smg,sml,pfam_matrices,indexes,bls,blhs,dis,dps,dphs,model_name,no_of_model,w):
|
||||||
|
preds=np.zeros((len(smg),no_of_model))
|
||||||
|
pfam_matrices=pfam_matrices.reshape((len(smg),7,7,1))
|
||||||
|
for i in range(1,no_of_model+1):
|
||||||
|
try:
|
||||||
|
model=tf.train.import_meta_graph(model_name+'_v'+str(i)+'/model.ckpt.meta')
|
||||||
|
except:
|
||||||
|
print("Something wrong with the model.")
|
||||||
|
continue
|
||||||
|
with tf.Session() as sess:
|
||||||
|
try:
|
||||||
|
model.restore(sess,model_name+'_v'+str(i)+"/model.ckpt")
|
||||||
|
graph = tf.get_default_graph()
|
||||||
|
synmgt,synmlt,pfamt,blst,blhst,dpst,dphst,dist,lrt,yt=graph.get_collection("input_nodes")
|
||||||
|
predictions=graph.get_tensor_by_name("Predictions/BiasAdd:0")
|
||||||
|
print("Model Loaded Successfully :)")
|
||||||
|
except:
|
||||||
|
print(":(")
|
||||||
|
sys.exit()
|
||||||
|
|
||||||
|
fd={synmgt:smg,
|
||||||
|
synmlt:sml,
|
||||||
|
pfamt:pfam_matrices,
|
||||||
|
blst:bls,
|
||||||
|
blhst:blhs,
|
||||||
|
dpst:dps.reshape((len(blhs),1)),
|
||||||
|
dist:dis.reshape((len(blhs),1)),
|
||||||
|
dphst:dphs.reshape((len(blhs),1))}
|
||||||
|
preds_t_1=sess.run([predictions],feed_dict=fd)
|
||||||
|
preds_t_1=np.array(preds_t_1)[0]
|
||||||
|
fd={synmgt:smg.transpose((0,2,1,3)),
|
||||||
|
synmlt:sml.transpose((0,2,1,3)),
|
||||||
|
pfamt:pfam_matrices.transpose((0,2,1,3)),
|
||||||
|
blst:blhs,
|
||||||
|
blhst:bls,
|
||||||
|
dpst:dphs.reshape((len(blhs),1)),
|
||||||
|
dist:dis.reshape((len(blhs),1)),
|
||||||
|
dphst:dps.reshape((len(blhs),1))}
|
||||||
|
preds_t_2=sess.run([predictions],feed_dict=fd)
|
||||||
|
preds_t_2=np.array(preds_t_2)[0]
|
||||||
|
preds=preds+w[i-1]*(preds_t_1+preds_t_2)/2
|
||||||
|
tf.reset_default_graph()
|
||||||
|
preds=np.argmax(preds,axis=1)
|
||||||
|
print(preds.shape)
|
||||||
|
return preds
|
||||||
|
|
||||||
|
def write_preds(fname,model_name,name,preds,index_dict,df):
|
||||||
|
print("Writing predcitions to:","prediction_"+fname+"_"+model_name+"_"+name+"_multiple_pfam.txt")
|
||||||
|
with open("prediction_"+fname+"_"+model_name+"_"+name+"_multiple_pfam.txt","w") as file:
|
||||||
|
for index,row in progressbar.progressbar(df.iterrows()):
|
||||||
|
file.write(str(row[0]))
|
||||||
|
file.write("\t")
|
||||||
|
file.write(row[1])
|
||||||
|
file.write("\t")
|
||||||
|
file.write(row[3])
|
||||||
|
file.write("\t")
|
||||||
|
file.write(str(row["label"]))
|
||||||
|
file.write("\t")
|
||||||
|
if index in index_dict:
|
||||||
|
file.write(str(preds[index_dict[index]]))
|
||||||
|
file.write("\t")
|
||||||
|
if preds[index_dict[index]]==row["label"]:
|
||||||
|
file.write(str(1))
|
||||||
|
else:
|
||||||
|
file.write(str(0))
|
||||||
|
else:
|
||||||
|
file.write("Error")
|
||||||
|
file.write("\t")
|
||||||
|
file.write("NaN")
|
||||||
|
file.write("\n")
|
||||||
|
|
||||||
|
def main():
|
||||||
|
arg=sys.argv
|
||||||
|
arg=arg[1:]
|
||||||
|
fname=arg[0]
|
||||||
|
model_name=arg[1]
|
||||||
|
no_of_model=int(arg[2])
|
||||||
|
nop=int(arg[3])
|
||||||
|
st=int(arg[4])
|
||||||
|
end=int(arg[5])
|
||||||
|
name=arg[6]
|
||||||
|
pfam_fname=arg[7]
|
||||||
|
weight=arg[8]
|
||||||
|
if weight=="e":
|
||||||
|
w=[1]*no_of_model
|
||||||
|
else:
|
||||||
|
w=[]
|
||||||
|
for i in range(no_of_model):
|
||||||
|
w.append(float(arg[9+i]))
|
||||||
|
df=read_database(fname)
|
||||||
|
df=select_data_by_length(df,st,end)
|
||||||
|
n=3
|
||||||
|
|
||||||
|
if os.path.exists("prediction_data/data_"+fname):
|
||||||
|
|
||||||
|
with open("prediction_data/data_"+fname,"rb") as file:
|
||||||
|
save_dict=pickle.load(file)
|
||||||
|
smg=save_dict["smg"]
|
||||||
|
sml=save_dict["sml"]
|
||||||
|
pfam_matrices=save_dict["pfam"]
|
||||||
|
indexes=save_dict["indexes"]
|
||||||
|
bls=save_dict["bls"]
|
||||||
|
blhs=save_dict["blhs"]
|
||||||
|
dis=save_dict["dis"]
|
||||||
|
dps=save_dict["dps"]
|
||||||
|
dphs=save_dict["dphs"]
|
||||||
|
|
||||||
|
else:
|
||||||
|
a,d,ld,ldg,cmap,cimap=read_genome_maps()#read the genome maps
|
||||||
|
print("Genome Maps Loaded.")
|
||||||
|
a_h=[df]
|
||||||
|
d_h=["prediction"]
|
||||||
|
|
||||||
|
lsy,gene_sequences=create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,name)
|
||||||
|
sml,smg,indexes=threadmaker(nop,df,lsy,gene_sequences,n,name)
|
||||||
|
a=""
|
||||||
|
d=""
|
||||||
|
ld=""
|
||||||
|
ldg=""
|
||||||
|
cmap=""
|
||||||
|
cimap=""
|
||||||
|
gene_sequences=""
|
||||||
|
gc.collect()
|
||||||
|
df_temp=df.loc[indexes]
|
||||||
|
pfam_db=pd.read_hdf(pfam_fname.split(".")[0]+"_pfam_db.h5")
|
||||||
|
with open(pfam_fname.split(".")[0]+"_pfam_map","rb") as file:
|
||||||
|
pfam_map=pickle.load(file)
|
||||||
|
pfam_matrices,indexes_pfam=create_pfam_matrix(df_temp,lsy,pfam_db,pfam_map)
|
||||||
|
pfam_db=""
|
||||||
|
gc.collect()
|
||||||
|
bls,blhs,dis,dps,dphs=create_tree_data("species_tree.tree",df_temp)
|
||||||
|
save_dict=dict(smg=smg,sml=sml,pfam=pfam_matrices,indexes=indexes,bls=bls,blhs=blhs,dis=dis,dps=dps,dphs=dphs)
|
||||||
|
|
||||||
|
if not os.path.isdir("prediction_data"):
|
||||||
|
os.mkdir("prediction_data")
|
||||||
|
|
||||||
|
with open("prediction_data/data_"+fname,"wb") as file:
|
||||||
|
pickle.dump(save_dict,file)
|
||||||
|
|
||||||
|
index_dict=create_map_list(indexes)
|
||||||
|
preds=get_prediction(smg,sml,pfam_matrices,indexes,bls,blhs,dis,dps,dphs,model_name,no_of_model,w)
|
||||||
|
|
||||||
|
write_preds(fname,model_name,name,preds,index_dict,df)
|
||||||
|
|
||||||
|
if __name__=="__main__":
|
||||||
|
main()
|
||||||
|
|
@ -8,7 +8,18 @@ from select_data import read_db_homology
|
||||||
from threads import Procerssrunner
|
from threads import Procerssrunner
|
||||||
from read_get_gene_seq import read_gene_sequences
|
from read_get_gene_seq import read_gene_sequences
|
||||||
from access_data_rest import update_rest,update_rest_protein
|
from access_data_rest import update_rest,update_rest_protein
|
||||||
from process_negative import write_fasta
|
from Bio import SeqIO
|
||||||
|
from Bio.Seq import Seq
|
||||||
|
from Bio.SeqRecord import SeqRecord
|
||||||
|
from Bio.Alphabet import IUPAC
|
||||||
|
|
||||||
|
def write_fasta(sequences,name):
|
||||||
|
with open(name+".fa","w") as file:
|
||||||
|
for seq in sequences:
|
||||||
|
if sequences[seq]=="":
|
||||||
|
continue
|
||||||
|
record=SeqRecord(Seq(sequences[seq],IUPAC.protein),id=seq)
|
||||||
|
SeqIO.write(record,file,"fasta")
|
||||||
|
|
||||||
def read_data_synteny(nop,name):
|
def read_data_synteny(nop,name):
|
||||||
smg=[]
|
smg=[]
|
||||||
|
|
|
||||||
|
|
@ -1,63 +1,63 @@
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import progressbar
|
||||||
|
import json
|
||||||
import sys
|
import sys
|
||||||
from neighbor_genes import read_genome_maps
|
from neighbor_genes import read_genome_maps
|
||||||
from process_data import create_data_homology_ls
|
from process_data import create_data_homology_ls
|
||||||
from threads import Procerssrunner
|
from threads import Procerssrunner
|
||||||
from read_get_gene_seq import read_gene_sequences
|
from read_get_gene_seq import read_gene_sequences
|
||||||
from access_data_rest import update_rest
|
from access_data_rest import update_rest,update_rest_protein
|
||||||
from prepare_synteny_matrix import read_data_synteny
|
from prepare_synteny_matrix import read_data_synteny,write_fasta
|
||||||
from save_data import write_dict_json
|
from save_data import write_dict_json
|
||||||
|
from access_data_rest import update_rest_protein
|
||||||
|
|
||||||
def read_database_txt(filename):
|
def read_database_txt(filename):
|
||||||
df = pd.read_csv(filename, sep="\t", header=None)
|
df=pd.read_csv(filename,sep="\t",header=None)
|
||||||
df = df.drop(0, axis=1)
|
df=df.drop(0,axis=1)
|
||||||
df.columns = [
|
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","wga","goc","homology_type"]
|
||||||
"gene_stable_id",
|
|
||||||
"species",
|
|
||||||
"homology_gene_stable_id",
|
|
||||||
"homology_species",
|
|
||||||
"wga",
|
|
||||||
"goc",
|
|
||||||
"homology_type"]
|
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
arg = sys.argv
|
arg=sys.argv
|
||||||
a, d, ld, ldg, cmap, cimap = read_genome_maps()
|
a,d,ld,ldg,cmap,cimap=read_genome_maps()
|
||||||
print("Genome Maps Loaded.")
|
print("Genome Maps Loaded.")
|
||||||
df = read_database_txt(arg[-2])
|
df=read_database_txt(arg[-2])
|
||||||
nop = int(arg[-1])
|
nop=int(arg[-1])
|
||||||
print("Data Read.")
|
print("Data Read.")
|
||||||
a_h = []
|
a_h=[]
|
||||||
d_h = []
|
d_h=[]
|
||||||
a_h.append(df)
|
a_h.append(df)
|
||||||
d_h.append(arg[-2].split(".")[0])
|
d_h.append(arg[-2].split(".")[0])
|
||||||
n = 3
|
n=3
|
||||||
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
lsy=create_data_homology_ls(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,0)
|
||||||
write_dict_json("neighbor_genes_negative", "processed", lsy)
|
write_dict_json("neighbor_genes_negative","processed",lsy)
|
||||||
print("Neighbor Genes Found and Saved Successfully:)")
|
print("Neighbor Genes Found and Saved Successfully:)")
|
||||||
gene_sequences = read_gene_sequences(
|
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","gene_seq_negative")
|
||||||
a_h, lsy, "geneseq", "gene_seq_negative")
|
gene_sequences=update_rest(gene_sequences,"gene_seq_negative")
|
||||||
gene_sequences = update_rest(gene_sequences, "gene_seq_negative")
|
ndir="processed/synteny_matrices/"
|
||||||
ndir = "processed/synteny_matrices/"
|
nf1="synteny_matrices_global"
|
||||||
nf1 = "synteny_matrices_global"
|
nf2="synteny_matrices_local"
|
||||||
nf2 = "synteny_matrices_local"
|
nf3="indexes"
|
||||||
nf3 = "indexes"
|
|
||||||
for i in range(len(a_h)):
|
for i in range(len(a_h)):
|
||||||
df = a_h[i]
|
df=a_h[i]
|
||||||
part = len(df) // nop
|
part=len(df)//nop
|
||||||
pr = Procerssrunner()
|
pr=Procerssrunner()
|
||||||
pr.start_processes(nop, df, gene_sequences, lsy, part, n, d_h[i])
|
pr.start_processes(nop,df,gene_sequences,lsy,part,n,d_h[i])
|
||||||
smg, sml, indexes = read_data_synteny(nop, d_h[i])
|
smg,sml,indexes=read_data_synteny(nop,d_h[i])
|
||||||
print(len(indexes))
|
print(len(indexes))
|
||||||
np.save(ndir + str(d_h[i]) + "_" + nf1, smg)
|
np.save(ndir+str(d_h[i])+"_"+nf1,smg)
|
||||||
np.save(ndir + str(d_h[i]) + "_" + nf2, sml)
|
np.save(ndir+str(d_h[i])+"_"+nf2,sml)
|
||||||
np.save(ndir + str(d_h[i]) + "_" + nf3, indexes)
|
np.save(ndir+str(d_h[i])+"_"+nf3,indexes)
|
||||||
|
a_h[i]=df.loc[indexes]
|
||||||
print("Synteny Matrices Created Successfully :)")
|
print("Synteny Matrices Created Successfully :)")
|
||||||
|
protein_sequences=read_gene_sequences(a_h,lsy,"pro_seq","pro_seq_negative")
|
||||||
|
protein_sequences=update_rest_protein(protein_sequences,"pro_seq_negative")
|
||||||
|
write_fasta(protein_sequences,"pro_seq_negative")
|
||||||
|
|
||||||
|
if __name__=="__main__":
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
main()
|
||||||
|
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue