mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
fix flank8 errors
This commit is contained in:
parent
07f9f23101
commit
3f885458a4
17 changed files with 791 additions and 819 deletions
|
|
@ -3,14 +3,17 @@ import requests
|
|||
import progressbar
|
||||
import sys
|
||||
|
||||
|
||||
def update_protein(gene_seq, gene):
|
||||
t = 0
|
||||
while(t != 2):
|
||||
try:
|
||||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id/"+str(gene)+"?type=protein;multiple_sequences=1"
|
||||
ext = "/sequence/id/" + \
|
||||
str(gene)+"?type=protein;multiple_sequences=1"
|
||||
|
||||
r = requests.get(server+ext, headers={ "Content-Type" : "application/json"})
|
||||
r = requests.get(
|
||||
server+ext, headers={"Content-Type": "application/json"})
|
||||
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
|
|
@ -31,12 +34,13 @@ def update_protein(gene_seq,gene):
|
|||
r = dict(r[maxi])
|
||||
gene_seq[gene] = str(r["seq"])
|
||||
return
|
||||
except :
|
||||
except Exception:
|
||||
t += 1
|
||||
# print("\nError:",e)
|
||||
continue
|
||||
gene_seq[gene] = ""
|
||||
|
||||
|
||||
def update_rest_protein(data, fname):
|
||||
gids = {}
|
||||
with open("processed/not_found_"+fname+".json", "r") as file:
|
||||
|
|
@ -48,13 +52,15 @@ def update_rest_protein(data,fname):
|
|||
|
||||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id?type=protein"
|
||||
headers={ "Content-Type" : "application/json", "Accept" : "application/json"}
|
||||
headers = {"Content-Type": "application/json",
|
||||
"Accept": "application/json"}
|
||||
|
||||
for i in progressbar.progressbar(range(0, len(gids)-50, 50)):
|
||||
ids = dict(ids=list(gids[i:i+50]))
|
||||
while(1):
|
||||
try:
|
||||
r = requests.post(server+ext, headers=headers, data=str(json.dumps(ids)))
|
||||
r = requests.post(server+ext, headers=headers,
|
||||
data=str(json.dumps(ids)))
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
gs = r.json()
|
||||
|
|
@ -71,13 +77,14 @@ def update_rest_protein(data,fname):
|
|||
for genes in gids:
|
||||
try:
|
||||
_ = data[genes]
|
||||
except:
|
||||
except Exception:
|
||||
print(genes)
|
||||
update_protein(data, genes)
|
||||
|
||||
print("Protein Sequences Updated Successfully")
|
||||
return data
|
||||
|
||||
|
||||
def update(gene_seq, gene):
|
||||
t = 0
|
||||
while(t != 2):
|
||||
|
|
@ -85,7 +92,8 @@ def update(gene_seq,gene):
|
|||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id/"+str(gene)+"?type=cds;multiple_sequences=1"
|
||||
|
||||
r = requests.get(server+ext, headers={ "Content-Type" : "application/json"})
|
||||
r = requests.get(
|
||||
server+ext, headers={"Content-Type": "application/json"})
|
||||
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
|
|
@ -106,12 +114,13 @@ def update(gene_seq,gene):
|
|||
r = dict(r[maxi])
|
||||
gene_seq[gene] = str(r["seq"])
|
||||
return
|
||||
except Exception as e:
|
||||
except Exception:
|
||||
t += 1
|
||||
# print("\nError:",e)
|
||||
continue
|
||||
gene_seq[gene] = ""
|
||||
|
||||
|
||||
def update_rest(data, fname):
|
||||
gids = {}
|
||||
with open("processed/not_found_"+fname+".json", "r") as file:
|
||||
|
|
@ -123,13 +132,15 @@ def update_rest(data,fname):
|
|||
|
||||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id?type=cds"
|
||||
headers={ "Content-Type" : "application/json", "Accept" : "application/json"}
|
||||
headers = {"Content-Type": "application/json",
|
||||
"Accept": "application/json"}
|
||||
|
||||
for i in progressbar.progressbar(range(0, len(gids)-50, 50)):
|
||||
ids = dict(ids=list(gids[i:i+50]))
|
||||
while(1):
|
||||
try:
|
||||
r = requests.post(server+ext, headers=headers, data=str(json.dumps(ids)))
|
||||
r = requests.post(server+ext, headers=headers,
|
||||
data=str(json.dumps(ids)))
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
gs = r.json()
|
||||
|
|
@ -146,7 +157,7 @@ def update_rest(data,fname):
|
|||
for genes in gids:
|
||||
try:
|
||||
_ = data[genes]
|
||||
except:
|
||||
except Exception:
|
||||
print(genes)
|
||||
update(data, genes)
|
||||
|
||||
|
|
|
|||
19
ftpg.py
19
ftpg.py
|
|
@ -3,13 +3,14 @@ import progressbar
|
|||
import os
|
||||
import sys
|
||||
import urllib.request as urllib
|
||||
import requests
|
||||
|
||||
|
||||
def download_data(x, dir_name):
|
||||
fname = x.split("/")[-1]
|
||||
path = os.path.join(dir_name, fname)
|
||||
urllib.urlretrieve(x, path)
|
||||
|
||||
|
||||
def get_data_file(file, dir):
|
||||
if not os.path.isfile(file):
|
||||
print("The specified file does not exist!!!")
|
||||
|
|
@ -24,8 +25,8 @@ def get_data_file(file,dir):
|
|||
download_data(x, dir)
|
||||
|
||||
|
||||
|
||||
#This wil download all the fasta files for the coding sequences. To change the directory, change the argument in the get_data_file argument.
|
||||
# This wil download all the fasta files for the coding sequences.
|
||||
# To change the directory, change the argument in the get_data_file argument.
|
||||
host = "ftp.ensembl.org"
|
||||
user = "anonymous"
|
||||
password = ""
|
||||
|
|
@ -36,9 +37,9 @@ ftp.login(user, password)
|
|||
print("Connected to {}".format(host))
|
||||
base_link = "ftp://ftp.ensembl.org"
|
||||
# find sequences of all the cds files
|
||||
l=ftp.nlst("/pub/release-96/fasta")
|
||||
list_of_files = ftp.nlst("/pub/release-96/fasta")
|
||||
lt = []
|
||||
for x in l:
|
||||
for x in list_of_files:
|
||||
y = ftp.nlst(x+"/cds")
|
||||
for z in y:
|
||||
if z.endswith(".cds.all.fa.gz"):
|
||||
|
|
@ -50,9 +51,9 @@ with open("seq_link.txt","w") as file:
|
|||
file.write("\n")
|
||||
|
||||
# find all the files with protein sequences
|
||||
l=ftp.nlst("/pub/release-96/fasta")
|
||||
list_of_files = ftp.nlst("/pub/release-96/fasta")
|
||||
lt = []
|
||||
for x in progressbar.progressbar(l):
|
||||
for x in progressbar.progressbar(list_of_files):
|
||||
y = ftp.nlst(x+"/pep")
|
||||
for z in y:
|
||||
if z.endswith(".pep.all.fa.gz"):
|
||||
|
|
@ -63,9 +64,9 @@ with open("protein_seq.txt","w") as file:
|
|||
file.write(base_link+x)
|
||||
file.write("\n")
|
||||
# get link of all the gtf files
|
||||
l=ftp.nlst("/pub/release-96/gtf")
|
||||
list_of_files = ftp.nlst("/pub/release-96/gtf")
|
||||
lt = []
|
||||
for x in l:
|
||||
for x in list_of_files:
|
||||
y = ftp.nlst(x)
|
||||
for z in y:
|
||||
if z.endswith(".96.gtf.gz"):
|
||||
|
|
|
|||
85
model.py
85
model.py
|
|
@ -1,70 +1,97 @@
|
|||
import tensorflow as tf
|
||||
dim = 10
|
||||
n = 3
|
||||
fl = 2*n+1
|
||||
maxbl = 29
|
||||
|
||||
import tensorflow as tf
|
||||
|
||||
def create_model():
|
||||
tf.reset_default_graph()
|
||||
g = tf.Graph()
|
||||
with g.as_default():
|
||||
synmg=tf.placeholder(dtype=tf.float64,shape=(None,2*n+1,2*n+1,2),name="Synteny_matrix_placeholder_Global")
|
||||
synml=tf.placeholder(dtype=tf.float64,shape=(None,2*n+1,2*n+1,2),name="Synteny_matrix_placeholder_Local")
|
||||
pfam=tf.placeholder(dtype=tf.float64,shape=(None,2*n+1,2*n+1,1),name="Pfam_matrix_placeholder")
|
||||
bls=tf.placeholder(dtype=tf.float64,shape=(None,maxbl),name="Species_Branch_Length_Placeholder")
|
||||
blhs=tf.placeholder(dtype=tf.float64,shape=(None,maxbl),name="Homology_Species_Branch_Length_Placeholder")
|
||||
synmg = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, 2*n+1, 2*n+1, 2), name="Synteny_matrix_placeholder_Global")
|
||||
synml = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, 2*n+1, 2*n+1, 2), name="Synteny_matrix_placeholder_Local")
|
||||
pfam = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, 2*n+1, 2*n+1, 1), name="Pfam_matrix_placeholder")
|
||||
bls = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, maxbl), name="Species_Branch_Length_Placeholder")
|
||||
blhs = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, maxbl), name="Homology_Species_Branch_Length_Placeholder")
|
||||
# gl=tf.placeholder(dtype=tf.float64,shape=(None,1),name="Mean_gene_length")
|
||||
dps=tf.placeholder(dtype=tf.float64,shape=(None,1),name="mca_species_distance")
|
||||
dphs=tf.placeholder(dtype=tf.float64,shape=(None,1),name="mca_homology_species_distance")
|
||||
dis=tf.placeholder(dtype=tf.float64,shape=(None,1),name="total_distance")
|
||||
dps = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, 1), name="mca_species_distance")
|
||||
dphs = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, 1), name="mca_homology_species_distance")
|
||||
dis = tf.placeholder(dtype=tf.float64, shape=(
|
||||
None, 1), name="total_distance")
|
||||
lr = tf.placeholder(dtype=tf.float64, shape=(), name="learning_rate")
|
||||
y = tf.placeholder(dtype=tf.int32, shape=(None), name="labels")
|
||||
|
||||
lrs=tf.summary.scalar("Learning_Rate",lr)
|
||||
# lrs = tf.summary.scalar("Learning_Rate", lr)
|
||||
|
||||
x = tf.concat([dps, dps-dphs, dis], 1, name="Create_train_vector")
|
||||
|
||||
print(synmg,"\n",synml,"\n",bls,"\n",blhs,"\n",dps,"\n",dphs,"\n",dis,"\n",x)
|
||||
print(synmg, "\n", synml, "\n", bls, "\n", blhs,
|
||||
"\n", dps, "\n", dphs, "\n", dis, "\n", x)
|
||||
|
||||
reg_l2 = tf.contrib.layers.l2_regularizer(0.001)
|
||||
reg_l1 = tf.contrib.layers.l1_regularizer(scale=0.005, scope=None)
|
||||
# reg_l1 = tf.contrib.layers.l1_regularizer(scale=0.005, scope=None)
|
||||
|
||||
def get_variable_by_shape(shape, name):
|
||||
f=tf.get_variable(name,shape=shape,initializer=tf.glorot_uniform_initializer(),dtype=tf.float64,regularizer=reg_l2)
|
||||
f = tf.get_variable(name, shape=shape,
|
||||
initializer=tf.glorot_uniform_initializer(),
|
||||
dtype=tf.float64,
|
||||
regularizer=reg_l2)
|
||||
return f
|
||||
|
||||
def create_synteny_aligner(name, synm):
|
||||
with tf.variable_scope(name+"Synteny_Aligner",reuse=tf.AUTO_REUSE):
|
||||
with tf.variable_scope(
|
||||
name+"Synteny_Aligner", reuse=tf.AUTO_REUSE):
|
||||
|
||||
fconv = get_variable_by_shape((2, 2, 2, dim), "fconv")
|
||||
conv=tf.nn.conv2d(synm,fconv,(1,1,1,1),padding="VALID",name="Conv_aligner")
|
||||
conv = tf.nn.conv2d(synm, fconv, (1, 1, 1, 1),
|
||||
padding="VALID", name="Conv_aligner")
|
||||
|
||||
fconv_1 = get_variable_by_shape((2, 2, dim, dim*2), "fconv_1")
|
||||
conv_1=tf.nn.conv2d(conv,fconv_1,(1,1,1,1),padding="VALID",name="Conv_aligner_1")
|
||||
conv_1 = tf.nn.conv2d(
|
||||
conv, fconv_1, (1, 1, 1, 1), padding="VALID",
|
||||
name="Conv_aligner_1")
|
||||
|
||||
fxconv = get_variable_by_shape((fl, 2, dim*2), "fxconv")
|
||||
x_conv = tf.reshape(synm, (-1, fl*fl, 2))
|
||||
x_conv=tf.nn.conv1d(x_conv,fxconv,stride=fl,padding="SAME",name="row_aligner")
|
||||
x_conv = tf.nn.conv1d(
|
||||
x_conv, fxconv, stride=fl, padding="SAME",
|
||||
name="row_aligner")
|
||||
|
||||
y_conv=tf.reshape(tf.transpose(synm,(0,2,1,3)),(-1,fl*fl,2))
|
||||
y_conv = tf.reshape(tf.transpose(
|
||||
synm, (0, 2, 1, 3)), (-1, fl*fl, 2))
|
||||
fyconv = get_variable_by_shape((fl, 2, dim*2), "fyconv")
|
||||
y_conv=tf.nn.conv1d(y_conv,fyconv,stride=fl,padding="SAME",name="column_aligner")
|
||||
|
||||
y_conv = tf.nn.conv1d(
|
||||
y_conv, fyconv, stride=fl, padding="SAME",
|
||||
name="column_aligner")
|
||||
|
||||
wconv = get_variable_by_shape((fl, fl, 2, dim*2*10), "wconv")
|
||||
w_conv=tf.nn.conv2d(synm,wconv,(1,1,1,1),padding="VALID",name="Global_Aligner_1")
|
||||
w_conv = tf.nn.conv2d(
|
||||
synm, wconv, (1, 1, 1, 1), padding="VALID",
|
||||
name="Global_Aligner_1")
|
||||
|
||||
conv_1 = tf.reshape(conv_1, (-1, 25, dim*2))
|
||||
x_conv = tf.reshape(x_conv, (-1, fl, dim*2))
|
||||
y_conv = tf.reshape(y_conv, (-1, fl, dim*2))
|
||||
w_conv = tf.reshape(w_conv, (-1, 10, dim*2))
|
||||
conv_final=tf.concat([conv_1,x_conv,y_conv,w_conv],1,name="Concatenate_All_Alignments")
|
||||
conv_final = tf.concat(
|
||||
[conv_1, x_conv, y_conv, w_conv], 1,
|
||||
name="Concatenate_All_Alignments")
|
||||
return conv_final
|
||||
|
||||
with tf.variable_scope("Pfam", reuse=tf.AUTO_REUSE):
|
||||
wconv_pfam=get_variable_by_shape((fl,fl,1,dim*2*10),"wconv_pfam")
|
||||
w_conv_pfam=tf.nn.conv2d(pfam,wconv_pfam,(1,1,1,1),padding="VALID",name="Global_Aligner_pfam")
|
||||
wconv_pfam = get_variable_by_shape(
|
||||
(fl, fl, 1, dim*2*10), "wconv_pfam")
|
||||
w_conv_pfam = tf.nn.conv2d(
|
||||
pfam, wconv_pfam, (1, 1, 1, 1), padding="VALID",
|
||||
name="Global_Aligner_pfam")
|
||||
w_conv_pfam = tf.reshape(w_conv_pfam, (-1, 10, dim*2))
|
||||
|
||||
conv_final_g = create_synteny_aligner("Global_", synmg)
|
||||
|
|
@ -99,11 +126,13 @@ def create_model():
|
|||
print(flat)
|
||||
# dense=tf.layers.dense(flat,2048,kernel_regularizer=reg_l2,bias_regularizer=reg_l2)
|
||||
# dense_2=tf.layers.dense(dense,1024,kernel_regularizer=reg_l2,bias_regularizer=reg_l2)
|
||||
dense_3=tf.layers.dense(flat,512,kernel_regularizer=reg_l2,bias_regularizer=reg_l2)
|
||||
dense_3 = tf.layers.dense(
|
||||
flat, 512, kernel_regularizer=reg_l2, bias_regularizer=reg_l2)
|
||||
|
||||
logits_pred = tf.layers.dense(dense_3, 3, name="Predictions")
|
||||
print(logits_pred)
|
||||
entropy=tf.nn.sparse_softmax_cross_entropy_with_logits(logits=logits_pred,labels=y)
|
||||
entropy = tf.nn.sparse_softmax_cross_entropy_with_logits(
|
||||
logits=logits_pred, labels=y)
|
||||
print(entropy)
|
||||
|
||||
# weights = tf.trainable_variables() # all vars of your graph
|
||||
|
|
@ -114,11 +143,11 @@ def create_model():
|
|||
# loss=tf.reduce_mean(entropy)
|
||||
optimizer = tf.train.RMSPropOptimizer(lr)
|
||||
# optimizer=tf.train.AdamOptimizer()
|
||||
losses=tf.summary.scalar("Loss",loss)
|
||||
# losses = tf.summary.scalar("Loss", loss)
|
||||
t_op = optimizer.minimize(loss)
|
||||
acc = tf.math.in_top_k(tf.cast(logits_pred, tf.float32), y, 1)
|
||||
accuracy = tf.reduce_mean(tf.cast(acc, tf.float32))
|
||||
accs=tf.summary.scalar("Accuracy",accuracy)
|
||||
# accs = tf.summary.scalar("Accuracy", accuracy)
|
||||
summary = tf.summary.merge_all()
|
||||
init = tf.global_variables_initializer()
|
||||
saver = tf.train.Saver()
|
||||
|
|
|
|||
|
|
@ -1,17 +1,13 @@
|
|||
import json
|
||||
import gc
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import pickle
|
||||
import sys
|
||||
import progressbar
|
||||
import os
|
||||
from neighbor_genes import read_genome_maps
|
||||
from process_data import create_data_homology_ls
|
||||
from read_get_gene_seq import read_gene_sequences
|
||||
from access_data_rest import update_rest,update_rest_protein
|
||||
from access_data_rest import update_rest_protein
|
||||
from prepare_synteny_matrix import write_fasta
|
||||
from process_data import create_map_list
|
||||
|
||||
|
||||
def read_database(fname, dirname):
|
||||
df = pd.read_csv(dirname+"/"+fname, sep="\t", header=None)
|
||||
|
|
@ -28,9 +24,11 @@ def read_database(fname,dirname):
|
|||
df = df.assign(label=label)
|
||||
df = df.drop(7, axis=1)
|
||||
df = df.drop(0, axis=1)
|
||||
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","goc","wga","label"]
|
||||
df.columns = ["gene_stable_id", "species", "homology_gene_stable_id",
|
||||
"homology_species", "goc", "wga", "label"]
|
||||
return df
|
||||
|
||||
|
||||
def read_prediction_file_folder(dir_name):
|
||||
lf = os.listdir(dir_name)
|
||||
a_h = []
|
||||
|
|
@ -41,13 +39,17 @@ def read_prediction_file_folder(dir_name):
|
|||
d_h.append(x.split(".")[0])
|
||||
return a_h, d_h
|
||||
|
||||
|
||||
def create_synteny_features(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, name):
|
||||
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
||||
protein_sequences=read_gene_sequences(a_h,lsy,"pro_seq","prediction_"+name)
|
||||
protein_sequences=update_rest_protein(protein_sequences,"prediction_"+name)
|
||||
protein_sequences = read_gene_sequences(
|
||||
a_h, lsy, "pro_seq", "prediction_"+name)
|
||||
protein_sequences = update_rest_protein(
|
||||
protein_sequences, "prediction_"+name)
|
||||
write_fasta(protein_sequences, "prediction_"+name)
|
||||
print("Protein Sequences Loaded")
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
dirname = arg[-1]
|
||||
|
|
@ -58,5 +60,7 @@ def main():
|
|||
print("Genome Maps Loaded.")
|
||||
|
||||
create_synteny_features(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, dirname)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -12,18 +12,25 @@ from process_negative import read_database_txt
|
|||
def get_score_overlap(x, y, pfam_db, pfam_map):
|
||||
df_1 = pfam_db.loc[pfam_map[x]]
|
||||
df_2 = pfam_db.loc[pfam_map[y]]
|
||||
l=list(df_2.domain)
|
||||
list_of_domains = list(df_2.domain)
|
||||
c = 0
|
||||
c_1 = 0
|
||||
for _, row in df_1.iterrows():
|
||||
if row.domain in l:#check if the domain exists in the list
|
||||
# check if the domain exists in the list
|
||||
if row.domain in list_of_domains:
|
||||
c_1 += 1
|
||||
st=int(df_2[df_2["domain"]==row.domain].hmm_from)#get the start
|
||||
end=int(df_2[df_2["domain"]==row.domain].hmm_to)#get the end
|
||||
if (int(row.hmm_from)>st and int(row.hmm_from)<end) or (int(row.hmm_to)>st and int(row.hmm_from)<end):#check if the domain is a ovelapping domain
|
||||
# get the start
|
||||
st = int(df_2[df_2["domain"] == row.domain].hmm_from)
|
||||
# get the end
|
||||
end = int(df_2[df_2["domain"] == row.domain].hmm_to)
|
||||
# check if the domain is a ovelapping domain
|
||||
if (int(
|
||||
row.hmm_from) > st and int(row.hmm_from) < end) or (int(
|
||||
row.hmm_to) > st and int(row.hmm_from) < end):
|
||||
c += 1
|
||||
return c/max(len(df_1), len(df_2))
|
||||
|
||||
|
||||
def pfam_matrix(g1, g2, n, pfam_db, gmap, pfam_map):
|
||||
pm = np.zeros((n, n))
|
||||
for i in range(n):
|
||||
|
|
@ -31,18 +38,19 @@ def pfam_matrix(g1,g2,n,pfam_db,gmap,pfam_map):
|
|||
continue
|
||||
try:
|
||||
_ = gmap[g1[i]]
|
||||
except:
|
||||
except Exception:
|
||||
continue
|
||||
for j in range(n):
|
||||
if g2[j] == "NULL_GENE":
|
||||
continue
|
||||
try:
|
||||
_ = gmap[g2[j]]
|
||||
except:
|
||||
except Exception:
|
||||
continue
|
||||
pm[i][j] = get_score_overlap(g1[i], g2[j], pfam_db, pfam_map)
|
||||
return pm
|
||||
|
||||
|
||||
def create_pfam_matrix(df, lsy, pfam_db, pfam_map):
|
||||
n = 3
|
||||
glist = list(pfam_db.gene_stable_id)
|
||||
|
|
@ -57,7 +65,7 @@ def create_pfam_matrix(df,lsy,pfam_db,pfam_map):
|
|||
try:
|
||||
_ = lsy[g1]
|
||||
_ = lsy[g2]
|
||||
except:
|
||||
except Exception:
|
||||
continue
|
||||
for i in range(len(lsy[g1]['b'])-1, -1, -1):
|
||||
x.append(lsy[g1]['b'][i])
|
||||
|
|
@ -78,16 +86,18 @@ def create_pfam_matrix(df,lsy,pfam_db,pfam_map):
|
|||
indexes.append(index)
|
||||
return np.array(pg), np.array(indexes)
|
||||
|
||||
|
||||
def create_pfam_map(pfam_db):
|
||||
pfam_map = {}
|
||||
for index, row in progressbar.progressbar(pfam_db.iterrows()):
|
||||
try:
|
||||
_ = pfam_map[row.gene_stable_id]
|
||||
except:
|
||||
except Exception:
|
||||
pfam_map[row.gene_stable_id] = []
|
||||
pfam_map[row.gene_stable_id].append(index)
|
||||
return pfam_map
|
||||
|
||||
|
||||
def main_positive():
|
||||
if not os.path.isdir("processed/pfam_matrices"):
|
||||
os.mkdir("processed/pfam_matrices")
|
||||
|
|
@ -106,6 +116,7 @@ def main_positive():
|
|||
np.save(ndir+str(d_h[i])+"_"+nf3, indexes)
|
||||
print(len(indexes))
|
||||
|
||||
|
||||
def read_data_negative(arg):
|
||||
df = read_database_txt(arg[-1])
|
||||
name = arg[-1].split(".")[0]
|
||||
|
|
@ -117,6 +128,7 @@ def read_data_negative(arg):
|
|||
lsy = dict(json.load(file))
|
||||
return df, pfam_db, pfam_map, lsy, name
|
||||
|
||||
|
||||
def main_negative(arg):
|
||||
df, pfam_db, pfam_map, lsy, name = read_data_negative(arg)
|
||||
ndir = "processed/pfam_matrices/"
|
||||
|
|
@ -128,10 +140,12 @@ def main_negative(arg):
|
|||
np.save(ndir+name+"_"+nf3, indexes)
|
||||
print(len(indexes))
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
main_positive()
|
||||
main_negative(arg)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import progressbar
|
|||
import gc
|
||||
import sys
|
||||
|
||||
|
||||
def pfam_parse(filename):
|
||||
rlist = []
|
||||
try:
|
||||
|
|
@ -34,6 +35,7 @@ def pfam_parse(filename):
|
|||
tdf = pd.DataFrame(rlist)
|
||||
return tdf
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
fname_1 = arg[-2]
|
||||
|
|
@ -46,5 +48,6 @@ def main():
|
|||
df.to_hdf("pfam_db_negative.h5", key="pfam_db_negative", mode="w")
|
||||
print("Pfam Databases Written Successfully :)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
|
|||
|
|
@ -130,7 +130,8 @@ def write_preds(fname, model_name, name, preds, index_dict, df):
|
|||
"_" +
|
||||
name +
|
||||
"_multiple.txt")
|
||||
with open("prediction_" + fname + "_" + model_name + "_" + name + "_multiple.txt", "w") as file:
|
||||
with open("prediction_" + fname + "_" + model_name + "_" + name +
|
||||
"_multiple.txt", "w") as file:
|
||||
for index, row in progressbar.progressbar(df.iterrows()):
|
||||
file.write(str(row[0]))
|
||||
file.write("\t")
|
||||
|
|
|
|||
|
|
@ -1,211 +1,54 @@
|
|||
import json
|
||||
import gc
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import pickle
|
||||
import tensorflow as tf
|
||||
import sys
|
||||
import os
|
||||
import progressbar
|
||||
from neighbor_genes import read_genome_maps
|
||||
from process_data import create_data_homology_ls
|
||||
from read_get_gene_seq import read_gene_sequences
|
||||
from access_data_rest import update_rest,update_rest_protein
|
||||
from threads import Procerssrunner
|
||||
from prepare_synteny_matrix import read_data_synteny
|
||||
from tree_data import create_tree_data
|
||||
from process_data import create_map_list
|
||||
from pfam_parser import pfam_parse
|
||||
from pfam_matrix import create_pfam_map,create_pfam_matrix
|
||||
import gc
|
||||
import sys
|
||||
|
||||
def read_database(fname):
|
||||
df=pd.read_csv(fname,sep="\t",header=None)
|
||||
label_dict=dict(ortholog_one2one=1,
|
||||
other_paralog=0,
|
||||
non_homolog=2,
|
||||
ortholog_one2many=1,
|
||||
ortholog_many2many=1,
|
||||
within_species_paralog=0,
|
||||
gene_split=4)
|
||||
label=[]
|
||||
for _,row in df.iterrows():
|
||||
label.append(label_dict[row[7]])
|
||||
df=df.assign(label=label)
|
||||
df=df.drop(7,axis=1)
|
||||
df=df.drop(0,axis=1)
|
||||
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","goc","wga","label"]
|
||||
return df
|
||||
|
||||
def select_data_by_length(df,st,end):
|
||||
def pfam_parse(filename):
|
||||
rlist = []
|
||||
try:
|
||||
if end<len(df):
|
||||
if st<end:
|
||||
df=df.loc[df.index.values[st:end]]
|
||||
else:
|
||||
raise ValueError()
|
||||
except:
|
||||
print("Making Predictions for the complete dataframe:)")
|
||||
print(len(df))
|
||||
return df
|
||||
|
||||
def create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,name):
|
||||
lsy=create_data_homology_ls(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,0)
|
||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","prediction_"+name)
|
||||
gene_sequences=update_rest(gene_sequences,"prediction_"+name)
|
||||
print("Gene Sequences Loaded.")
|
||||
return lsy,gene_sequences
|
||||
|
||||
def threadmaker(nop,df,lsy,gene_sequences,n,name):
|
||||
part=len(df)//nop
|
||||
pr=Procerssrunner()
|
||||
pr.start_processes(nop,df,gene_sequences,lsy,part,n,name)
|
||||
smg,sml,indexes=read_data_synteny(nop,name)
|
||||
sml=np.array(sml)
|
||||
smg=np.array(smg)
|
||||
indexes=np.array(indexes)
|
||||
return sml,smg,indexes
|
||||
|
||||
def get_prediction(smg,sml,pfam_matrices,indexes,bls,blhs,dis,dps,dphs,model_name,no_of_model,w):
|
||||
preds=np.zeros((len(smg),no_of_model))
|
||||
pfam_matrices=pfam_matrices.reshape((len(smg),7,7,1))
|
||||
for i in range(1,no_of_model+1):
|
||||
try:
|
||||
model=tf.train.import_meta_graph(model_name+'_v'+str(i)+'/model.ckpt.meta')
|
||||
except:
|
||||
print("Something wrong with the model.")
|
||||
with open(filename) as file:
|
||||
file.seek(0, 0)
|
||||
for line in progressbar.progressbar(file):
|
||||
if line.startswith("#"):
|
||||
continue
|
||||
with tf.Session() as sess:
|
||||
try:
|
||||
model.restore(sess,model_name+'_v'+str(i)+"/model.ckpt")
|
||||
graph = tf.get_default_graph()
|
||||
synmgt,synmlt,pfamt,blst,blhst,dpst,dphst,dist,lrt,yt=graph.get_collection("input_nodes")
|
||||
predictions=graph.get_tensor_by_name("Predictions/BiasAdd:0")
|
||||
print("Model Loaded Successfully :)")
|
||||
except:
|
||||
print(":(")
|
||||
sys.exit()
|
||||
temp_dict = {}
|
||||
x = [y for y in line.split(" ") if y != '']
|
||||
if x[9] != "1":
|
||||
continue
|
||||
temp_dict["gene_stable_id"] = x[3]
|
||||
temp_dict["accession"] = x[1]
|
||||
temp_dict["tlen"] = x[2]
|
||||
temp_dict["qlen"] = x[5]
|
||||
temp_dict["domain"] = x[0]
|
||||
temp_dict["hmm_from"] = x[15]
|
||||
temp_dict["hmm_to"] = x[16]
|
||||
temp_dict["ali_from"] = x[17]
|
||||
temp_dict["ali_to"] = x[18]
|
||||
temp_dict["env_from"] = x[19]
|
||||
temp_dict["env_to"] = x[20]
|
||||
rlist.append(temp_dict)
|
||||
except Exception as e:
|
||||
print(e)
|
||||
return pd.DataFrame()
|
||||
print(len(rlist))
|
||||
tdf = pd.DataFrame(rlist)
|
||||
return tdf
|
||||
|
||||
fd={synmgt:smg,
|
||||
synmlt:sml,
|
||||
pfamt:pfam_matrices,
|
||||
blst:bls,
|
||||
blhst:blhs,
|
||||
dpst:dps.reshape((len(blhs),1)),
|
||||
dist:dis.reshape((len(blhs),1)),
|
||||
dphst:dphs.reshape((len(blhs),1))}
|
||||
preds_t_1=sess.run([predictions],feed_dict=fd)
|
||||
preds_t_1=np.array(preds_t_1)[0]
|
||||
fd={synmgt:smg.transpose((0,2,1,3)),
|
||||
synmlt:sml.transpose((0,2,1,3)),
|
||||
pfamt:pfam_matrices.transpose((0,2,1,3)),
|
||||
blst:blhs,
|
||||
blhst:bls,
|
||||
dpst:dphs.reshape((len(blhs),1)),
|
||||
dist:dis.reshape((len(blhs),1)),
|
||||
dphst:dps.reshape((len(blhs),1))}
|
||||
preds_t_2=sess.run([predictions],feed_dict=fd)
|
||||
preds_t_2=np.array(preds_t_2)[0]
|
||||
preds=preds+w[i-1]*(preds_t_1+preds_t_2)/2
|
||||
tf.reset_default_graph()
|
||||
preds=np.argmax(preds,axis=1)
|
||||
print(preds.shape)
|
||||
return preds
|
||||
|
||||
def write_preds(fname,model_name,name,preds,index_dict,df):
|
||||
print("Writing predcitions to:","prediction_"+fname+"_"+model_name+"_"+name+"_multiple_pfam.txt")
|
||||
with open("prediction_"+fname+"_"+model_name+"_"+name+"_multiple_pfam.txt","w") as file:
|
||||
for index,row in progressbar.progressbar(df.iterrows()):
|
||||
file.write(row["gene_stable_id"])
|
||||
file.write("\t")
|
||||
file.write(row["homology_gene_stable_id"])
|
||||
file.write("\t")
|
||||
file.write(str(row["label"]))
|
||||
file.write("\t")
|
||||
if index in index_dict:
|
||||
file.write(str(preds[index_dict[index]]))
|
||||
file.write("\t")
|
||||
if preds[index_dict[index]]==row["label"]:
|
||||
file.write(str(1))
|
||||
else:
|
||||
file.write(str(0))
|
||||
else:
|
||||
file.write("Error")
|
||||
file.write("\t")
|
||||
file.write("NaN")
|
||||
file.write("\n")
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
arg=arg[1:]
|
||||
fname=arg[0]
|
||||
model_name=arg[1]
|
||||
no_of_model=int(arg[2])
|
||||
nop=int(arg[3])
|
||||
st=int(arg[4])
|
||||
end=int(arg[5])
|
||||
name=arg[6]
|
||||
pfam_fname=arg[7]
|
||||
weight=arg[8]
|
||||
if weight=="e":
|
||||
w=[1]*no_of_model
|
||||
else:
|
||||
w=[]
|
||||
for i in range(no_of_model):
|
||||
w.append(float(arg[9+i]))
|
||||
df=read_database(fname)
|
||||
df=select_data_by_length(df,st,end)
|
||||
n=3
|
||||
|
||||
if os.path.exists("prediction_data/data_"+fname):
|
||||
|
||||
with open("prediction_data/data_"+fname,"rb") as file:
|
||||
save_dict=pickle.load(file)
|
||||
smg=save_dict["smg"]
|
||||
sml=save_dict["sml"]
|
||||
pfam_matrices=save_dict["pfam"]
|
||||
indexes=save_dict["indexes"]
|
||||
bls=save_dict["bls"]
|
||||
blhs=save_dict["blhs"]
|
||||
dis=save_dict["dis"]
|
||||
dps=save_dict["dps"]
|
||||
dphs=save_dict["dphs"]
|
||||
|
||||
else:
|
||||
a,d,ld,ldg,cmap,cimap=read_genome_maps()#read the genome maps
|
||||
print("Genome Maps Loaded.")
|
||||
a_h=[df]
|
||||
d_h=["prediction"]
|
||||
|
||||
lsy,gene_sequences=create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,name)
|
||||
sml,smg,indexes=threadmaker(nop,df,lsy,gene_sequences,n,name)
|
||||
a=""
|
||||
d=""
|
||||
ld=""
|
||||
ldg=""
|
||||
cmap=""
|
||||
cimap=""
|
||||
gene_sequences=""
|
||||
fname_1 = arg[-2]
|
||||
fname_2 = arg[-1]
|
||||
df = pfam_parse(fname_1)
|
||||
df.to_hdf("pfam_db_positive.h5", key="pfam_db_positive", mode="w")
|
||||
df = ""
|
||||
gc.collect()
|
||||
df_temp=df.loc[indexes]
|
||||
pfam_db=pd.read_hdf(pfam_fname.split(".")[0]+"_pfam_db.h5")
|
||||
with open(pfam_fname.split(".")[0]+"_pfam_map","rb") as file:
|
||||
pfam_map=pickle.load(file)
|
||||
pfam_matrices,indexes_pfam=create_pfam_matrix(df_temp,lsy,pfam_db,pfam_map)
|
||||
pfam_db=""
|
||||
gc.collect()
|
||||
bls,blhs,dis,dps,dphs=create_tree_data("species_tree.tree",df_temp)
|
||||
save_dict=dict(smg=smg,sml=sml,pfam=pfam_matrices,indexes=indexes,bls=bls,blhs=blhs,dis=dis,dps=dps,dphs=dphs)
|
||||
df = pfam_parse(fname_2)
|
||||
df.to_hdf("pfam_db_negative.h5", key="pfam_db_negative", mode="w")
|
||||
print("Pfam Databases Written Successfully :)")
|
||||
|
||||
if not os.path.isdir("prediction_data"):
|
||||
os.mkdir("prediction_data")
|
||||
|
||||
with open("prediction_data/data_"+fname,"wb") as file:
|
||||
pickle.dump(save_dict,file)
|
||||
|
||||
index_dict=create_map_list(indexes)
|
||||
preds=get_prediction(smg,sml,pfam_matrices,indexes,bls,blhs,dis,dps,dphs,model_name,no_of_model,w)
|
||||
|
||||
write_preds(fname,model_name,name,preds,index_dict,df)
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
|
|
|
|||
|
|
@ -1,5 +1,4 @@
|
|||
import numpy as np
|
||||
import pandas as pd
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
|
@ -13,6 +12,7 @@ from Bio.Seq import Seq
|
|||
from Bio.SeqRecord import SeqRecord
|
||||
from Bio.Alphabet import IUPAC
|
||||
|
||||
|
||||
def write_fasta(sequences, name):
|
||||
with open(name+".fa", "w") as file:
|
||||
for seq in sequences:
|
||||
|
|
@ -21,17 +21,21 @@ def write_fasta(sequences,name):
|
|||
record = SeqRecord(Seq(sequences[seq], IUPAC.protein), id=seq)
|
||||
SeqIO.write(record, file, "fasta")
|
||||
|
||||
|
||||
def read_data_synteny(nop, name):
|
||||
smg = []
|
||||
sml = []
|
||||
indexes = []
|
||||
for i in range(nop):
|
||||
try:
|
||||
with open("temp_"+name+"/thread_"+str(i+1)+"_smg.temp","rb") as file:
|
||||
with open("temp_"+name+"/thread_"+str(i+1) +
|
||||
"_smg.temp", "rb") as file:
|
||||
smg = smg+pickle.load(file)
|
||||
with open("temp_"+name+"/thread_"+str(i+1)+"_sml.temp","rb") as file:
|
||||
with open("temp_"+name+"/thread_"+str(i+1) +
|
||||
"_sml.temp", "rb") as file:
|
||||
sml = sml+pickle.load(file)
|
||||
with open("temp_"+name+"/thread_"+str(i+1)+"_indexes.temp","rb") as file:
|
||||
with open("temp_"+name+"/thread_"+str(i+1) +
|
||||
"_indexes.temp", "rb") as file:
|
||||
indexes = indexes+pickle.load(file)
|
||||
except Exception as e:
|
||||
print("Problem with thread", i+1, "detected for", name, e)
|
||||
|
|
@ -39,6 +43,7 @@ def read_data_synteny(nop,name):
|
|||
print(len(indexes))
|
||||
return smg, sml, indexes
|
||||
|
||||
|
||||
def load_neighbor_genes():
|
||||
with open("processed/neighbor_genes.json", "r") as file:
|
||||
lsy = dict(json.load(file))
|
||||
|
|
@ -46,6 +51,7 @@ def load_neighbor_genes():
|
|||
print("Neighbor Genes Loaded")
|
||||
return lsy
|
||||
|
||||
|
||||
def read_data_homology(dirname):
|
||||
lf = os.listdir(dirname)
|
||||
if len(lf) == 0:
|
||||
|
|
@ -58,13 +64,14 @@ def read_data_homology(dirname):
|
|||
n = n.split()[0]
|
||||
try:
|
||||
indexes = np.load("processed/"+n+"_selected_indexes.npy")
|
||||
except:
|
||||
except Exception:
|
||||
print("Incomplete data for:", n)
|
||||
df = df.loc[indexes]
|
||||
a_h.append(df)
|
||||
d_h.append(n)
|
||||
return a_h, d_h
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
nop = int(arg[-1])
|
||||
|
|
@ -72,7 +79,8 @@ def main():
|
|||
a_h, d_h = read_data_homology("data_homology")
|
||||
print("Data Read")
|
||||
lsy = load_neighbor_genes()
|
||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","gene_seq_positive")
|
||||
gene_sequences = read_gene_sequences(
|
||||
a_h, lsy, "geneseq", "gene_seq_positive")
|
||||
gene_sequences = update_rest(gene_sequences, "gene_seq_positive")
|
||||
print("Gene Sequences Loaded.")
|
||||
if not os.path.isdir("processed/synteny_matrices"):
|
||||
|
|
@ -93,9 +101,13 @@ def main():
|
|||
np.save(ndir+str(d_h[i])+"_"+nf3, indexes)
|
||||
a_h[i] = df.loc[indexes]
|
||||
print("Synteny Matrices Created Successfully :)")
|
||||
protein_sequences=read_gene_sequences(a_h,lsy,"pro_seq","pro_seq_positive")
|
||||
protein_sequences=update_rest_protein(protein_sequences,"pro_seq_positive")
|
||||
protein_sequences = read_gene_sequences(
|
||||
a_h, lsy, "pro_seq", "pro_seq_positive")
|
||||
protein_sequences = update_rest_protein(
|
||||
protein_sequences, "pro_seq_positive")
|
||||
write_fasta(protein_sequences, "protein_seq_positive")
|
||||
print("Protein Sequences Loaded.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
|
|||
|
|
@ -3,10 +3,10 @@ import progressbar
|
|||
from save_data import write_dict_json
|
||||
|
||||
|
||||
def create_map_list(l): # this function maps the indexes to values
|
||||
def create_map_list(lister): # this function maps the indexes to values
|
||||
t = {}
|
||||
for i in range(len(l)):
|
||||
t[l[i]] = i
|
||||
for i in range(len(lister)):
|
||||
t[lister[i]] = i
|
||||
return t
|
||||
|
||||
|
||||
|
|
@ -58,10 +58,15 @@ def get_nearest_neighbors(g, gs, n, a, d, ld, ldg, cmap, cimap):
|
|||
try:
|
||||
_ = ld[gi] # see if the corresponding gene map exists
|
||||
except BaseException:
|
||||
# print("Length of Dataframes:{} \t Length of Loaded Genes:{} \t Length of Loaded Genomes Dictionaries:{}".format(len(a),len(ld),len(ldg)))
|
||||
# print("Length of Dataframes:{}
|
||||
# Length of Loaded Genes:{}
|
||||
# Length of Loaded Genomes Dictionaries:
|
||||
# {}".format(len(a),len(ld),len(ldg)))
|
||||
return ne, nr
|
||||
sldg = ldg[gi] # select the corresponding map
|
||||
if g not in sldg: # if the gene is not present in the dataframe return empty lists
|
||||
# if the gene is not present in the dataframe
|
||||
# return empty lists
|
||||
if g not in sldg:
|
||||
# print(g,"\t",gs)
|
||||
return ne, nr
|
||||
i = sldg[g] # find the index of the gene
|
||||
|
|
@ -80,9 +85,11 @@ def get_nearest_neighbors(g, gs, n, a, d, ld, ldg, cmap, cimap):
|
|||
end = list(sldf.end)
|
||||
end = np.array(end)
|
||||
assert(len(end) == len(sldf))
|
||||
end = end - start # subtract start from it so as to get relative position
|
||||
# subtract start from it so as to get relative position
|
||||
end = end - start
|
||||
end_s = np.argsort(end) # sort them by the order of distance
|
||||
if end[end_s[0]] >= 0: # if all the genes end ahead of the one in consideration
|
||||
# if all the genes end ahead of the one in consideration
|
||||
if end[end_s[0]] >= 0:
|
||||
flag = 1 # increment the pointer
|
||||
ne.append("NULL_GENE") # append the NULL_GENE value
|
||||
continue
|
||||
|
|
@ -144,7 +151,8 @@ def create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, to_write):
|
|||
xl, xr = get_nearest_neighbors(
|
||||
x, xs, n, a, d, ld, ldg, cmap, cimap)
|
||||
if len(
|
||||
xl) != 0: # check if neighboring genes were successfully found
|
||||
# check if neighboring genes were successfully found
|
||||
xl) != 0:
|
||||
lsy[x] = dict(b=xl, f=xr)
|
||||
lsytemp[x] = dict(b=xl, f=xr)
|
||||
except BaseException:
|
||||
|
|
|
|||
|
|
@ -1,9 +1,5 @@
|
|||
import pandas as pd
|
||||
import numpy as np
|
||||
import os
|
||||
import sys
|
||||
import progressbar
|
||||
import json
|
||||
import sys
|
||||
from neighbor_genes import read_genome_maps
|
||||
from process_data import create_data_homology_ls
|
||||
|
|
@ -12,12 +8,13 @@ from read_get_gene_seq import read_gene_sequences
|
|||
from access_data_rest import update_rest, update_rest_protein
|
||||
from prepare_synteny_matrix import read_data_synteny, write_fasta
|
||||
from save_data import write_dict_json
|
||||
from access_data_rest import update_rest_protein
|
||||
|
||||
|
||||
def read_database_txt(filename):
|
||||
df = pd.read_csv(filename, sep="\t", header=None)
|
||||
df = df.drop(0, axis=1)
|
||||
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","wga","goc","homology_type"]
|
||||
df.columns = ["gene_stable_id", "species", "homology_gene_stable_id",
|
||||
"homology_species", "wga", "goc", "homology_type"]
|
||||
return df
|
||||
|
||||
|
||||
|
|
@ -36,7 +33,8 @@ def main():
|
|||
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
||||
write_dict_json("neighbor_genes_negative", "processed", lsy)
|
||||
print("Neighbor Genes Found and Saved Successfully:)")
|
||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","gene_seq_negative")
|
||||
gene_sequences = read_gene_sequences(
|
||||
a_h, lsy, "geneseq", "gene_seq_negative")
|
||||
gene_sequences = update_rest(gene_sequences, "gene_seq_negative")
|
||||
ndir = "processed/synteny_matrices/"
|
||||
nf1 = "synteny_matrices_global"
|
||||
|
|
@ -54,10 +52,12 @@ def main():
|
|||
np.save(ndir+str(d_h[i])+"_"+nf3, indexes)
|
||||
a_h[i] = df.loc[indexes]
|
||||
print("Synteny Matrices Created Successfully :)")
|
||||
protein_sequences=read_gene_sequences(a_h,lsy,"pro_seq","pro_seq_negative")
|
||||
protein_sequences=update_rest_protein(protein_sequences,"pro_seq_negative")
|
||||
protein_sequences = read_gene_sequences(
|
||||
a_h, lsy, "pro_seq", "pro_seq_negative")
|
||||
protein_sequences = update_rest_protein(
|
||||
protein_sequences, "pro_seq_negative")
|
||||
write_fasta(protein_sequences, "pro_seq_negative")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
|
|
|
|||
|
|
@ -63,7 +63,8 @@ def read_data_genome(dir_name, a, dict_ind_genome):
|
|||
print(e)
|
||||
continue
|
||||
# print(data_gene[0:10])
|
||||
data_gene = data_gene[(data_gene['gene_biotype'] == 'protein_coding') | (
|
||||
data_gene = data_gene[(
|
||||
data_gene['gene_biotype'] == 'protein_coding') | (
|
||||
data_gene['gene_source'] == 'protein_coding')]
|
||||
# print(data_gene[data_gene["gene_id"]=="ENSNGAG00000000407"])
|
||||
a.append(data_gene)
|
||||
|
|
|
|||
|
|
@ -27,7 +27,8 @@ def create_dict(keys, values, dictionary):
|
|||
return dictionary
|
||||
|
||||
# this function maps all the genes to their respective species.
|
||||
# (Function: when finding the species of any gene we do not need to search the entire dataframe)
|
||||
# (Function: when finding the species of any gene
|
||||
# we do not need to search the entire dataframe)
|
||||
|
||||
|
||||
def group_seq_by_species(df, g_to_sp):
|
||||
|
|
@ -64,7 +65,9 @@ def read_gene_seq(dirname, s, genes_by_species):
|
|||
ftr = []
|
||||
for f in lof:
|
||||
if f.split(".")[
|
||||
0] in s: # check whether the species is present in the species to read list. Will skip those species which are not present in the dataframe
|
||||
# check whether the species is present in the species to read list.
|
||||
# Will skip those species which are not present in the dataframe
|
||||
0] in s:
|
||||
ftr.append(f)
|
||||
data = {}
|
||||
for f in progressbar.progressbar(ftr):
|
||||
|
|
@ -73,16 +76,17 @@ def read_gene_seq(dirname, s, genes_by_species):
|
|||
record = SeqIO.parse(file, "fasta")
|
||||
for r in record:
|
||||
gid, gbt = description_cleaner(r.description)
|
||||
if str(gid) not in data and str(
|
||||
gid) in genes_by_species[species] and gbt == "protein_coding":
|
||||
if str(gid) not in data and str(gid) in \
|
||||
genes_by_species[species] and gbt == "protein_coding":
|
||||
data[gid] = str(r.seq)
|
||||
return data
|
||||
|
||||
|
||||
def read_gene_sequences(hdf, lsy, data_dir, fname):
|
||||
"""The basic idea here is to create a list/dictionary of all the genes by their species.
|
||||
Once the mapping is done, all the respective fasta sequence files are read by Species
|
||||
and the CDNA sequences for each gene in the species record are read and stored.
|
||||
"""The basic idea here is to create a list/dictionary of all the genes by
|
||||
their species. Once the mapping is done, all the respective fasta sequence
|
||||
files are read by Species and the CDNA sequences for each gene in the
|
||||
species record are read and stored.
|
||||
Thus we don't have to read the same file multiple times."""
|
||||
grouped_genes = {}
|
||||
gene_by_species_dict = {}
|
||||
|
|
|
|||
|
|
@ -15,9 +15,12 @@ def write_dict_json(name, dir, d):
|
|||
def write_data_synteny(smg, sml, indexes, i, name):
|
||||
if not os.path.exists("temp_" + name):
|
||||
os.mkdir("temp_" + name)
|
||||
with open("temp_" + name + "/thread_" + str(i + 1) + "_smg.temp", "wb") as file:
|
||||
with open("temp_" + name + "/thread_" + str(i + 1) +
|
||||
"_smg.temp", "wb") as file:
|
||||
pickle.dump(smg, file)
|
||||
with open("temp_" + name + "/thread_" + str(i + 1) + "_sml.temp", "wb") as file:
|
||||
with open("temp_" + name + "/thread_" + str(i + 1) +
|
||||
"_sml.temp", "wb") as file:
|
||||
pickle.dump(sml, file)
|
||||
with open("temp_" + name + "/thread_" + str(i + 1) + "_indexes.temp", "wb") as file:
|
||||
with open("temp_" + name + "/thread_" + str(i + 1) +
|
||||
"_indexes.temp", "wb") as file:
|
||||
pickle.dump(indexes, file)
|
||||
|
|
|
|||
|
|
@ -52,10 +52,12 @@ class Thread_objects():
|
|||
norm_len = max(len(gene_seq[g1[i]]), len(gene_seq[g2[j]]))
|
||||
try:
|
||||
result = ed.align(
|
||||
gene_seq[g1[i]], gene_seq[g2[j]], mode="NW", task="distance")
|
||||
gene_seq[g1[i]], gene_seq[g2[j]],
|
||||
mode="NW", task="distance")
|
||||
sm[i][j][0] = result["editDistance"] / (norm_len)
|
||||
result = ed.align(
|
||||
gene_seq[g1[i]], gene_seq[g2[j]][::-1], mode="NW", task="distance")
|
||||
gene_seq[g1[i]], gene_seq[g2[j]][::-1], mode="NW",
|
||||
task="distance")
|
||||
sm[i][j][1] = result["editDistance"] / (norm_len)
|
||||
_, result, _ = local_pairwise_align_ssw(
|
||||
DNA(gene_seq[g1[i]]), DNA(gene_seq[g2[j]]))
|
||||
|
|
@ -120,7 +122,8 @@ class Procerssrunner():
|
|||
|
||||
def start_thread(self, obj, i, thread_alive, n, name):
|
||||
t = Process(target=obj.synteny_matrix, args=(
|
||||
obj.gene_sequences, obj.df, obj.lsy, n), name="Thread_" + str(i + 1))
|
||||
obj.gene_sequences, obj.df, obj.lsy, n),
|
||||
name="Thread_" + str(i + 1))
|
||||
print("Thread ", (i + 1), " started for ", name, ".")
|
||||
thread_alive.append(t)
|
||||
|
||||
|
|
|
|||
79
train.py
79
train.py
|
|
@ -4,6 +4,7 @@ import sys
|
|||
import tensorflow as tf
|
||||
from model import create_model
|
||||
|
||||
|
||||
def create_branch_length_padding(bl):
|
||||
maxlen = 0
|
||||
for x in bl:
|
||||
|
|
@ -17,10 +18,21 @@ def create_branch_length_padding(bl):
|
|||
bl[x] = temp
|
||||
return bl
|
||||
|
||||
def train(train_synteny_matrices_global,train_synteny_matrices_local,train_pfam_matrices,train_branch_length_species,train_branch_length_homology_species,train_dist_p_s,train_dist_p_hs,train_distance,train_labels,v,num_epochs,learning_rate,decay,size_train,batch_size,model_name):
|
||||
print("Going to train model {} for:\n Batch Size:{} \n Learning Rate:{} \n Decay:{}\n On {} Samples".format(v,batch_size,learning_rate,decay,len(train_synteny_matrices_global)))
|
||||
|
||||
def train(train_synteny_matrices_global, train_synteny_matrices_local,
|
||||
train_pfam_matrices, train_branch_length_species,
|
||||
train_branch_length_homology_species, train_dist_p_s,
|
||||
train_dist_p_hs, train_distance, train_labels, v,
|
||||
num_epochs, learning_rate, decay, size_train,
|
||||
batch_size, model_name):
|
||||
print("Going to train model {} for:\n Batch Size:{} \
|
||||
\n Learning Rate:{} \n Decay:{}\n On {} Samples".format(
|
||||
v, batch_size, learning_rate, decay,
|
||||
len(train_synteny_matrices_global)
|
||||
))
|
||||
graph, saver = create_model()
|
||||
synmg,synml,pfam,bls,blhs,dps,dphs,dis,lr,y=graph.get_collection("input_nodes")
|
||||
synmg, synml, pfam, bls, blhs, dps, \
|
||||
dphs, dis, lr, y = graph.get_collection("input_nodes")
|
||||
loss, t_op, accuracy, init, summary = graph.get_collection("output_nodes")
|
||||
with tf.Session(graph=graph) as sess:
|
||||
writer = tf.summary.FileWriter('./'+model_name+'_v'+str(v), sess.graph)
|
||||
|
|
@ -29,14 +41,26 @@ def train(train_synteny_matrices_global,train_synteny_matrices_local,train_pfam_
|
|||
for j in range(num_epochs):
|
||||
for i in range(size_train//batch_size):
|
||||
feed_dict = {
|
||||
synmg:train_synteny_matrices_global[i*batch_size:(i+1)*batch_size],
|
||||
synml:train_synteny_matrices_local[i*batch_size:(i+1)*batch_size],
|
||||
pfam:train_pfam_matrices[i*batch_size:(i+1)*batch_size].reshape((batch_size,7,7,1)),
|
||||
bls:train_branch_length_species[i*batch_size:(i+1)*batch_size],
|
||||
blhs:train_branch_length_homology_species[i*batch_size:(i+1)*batch_size],
|
||||
dps:train_dist_p_s[i*batch_size:(i+1)*batch_size].reshape((batch_size,1)),
|
||||
dphs:train_dist_p_hs[i*batch_size:(i+1)*batch_size].reshape((batch_size,1)),
|
||||
dis:train_distance[i*batch_size:(i+1)*batch_size].reshape((batch_size,1)),
|
||||
synmg: train_synteny_matrices_global
|
||||
[i*batch_size:(i+1)*batch_size],
|
||||
synml: train_synteny_matrices_local
|
||||
[i*batch_size:(i+1)*batch_size],
|
||||
pfam: train_pfam_matrices
|
||||
[i*batch_size:(i+1)*batch_size]
|
||||
.reshape((batch_size, 7, 7, 1)),
|
||||
bls: train_branch_length_species
|
||||
[i*batch_size:(i+1)*batch_size],
|
||||
blhs: train_branch_length_homology_species
|
||||
[i*batch_size:(i+1)*batch_size],
|
||||
dps: train_dist_p_s
|
||||
[i*batch_size:(i+1)*batch_size]
|
||||
.reshape((batch_size, 1)),
|
||||
dphs: train_dist_p_hs
|
||||
[i*batch_size:(i+1)*batch_size]
|
||||
.reshape((batch_size, 1)),
|
||||
dis: train_distance
|
||||
[i*batch_size:(i+1)*batch_size]
|
||||
.reshape((batch_size, 1)),
|
||||
lr: learn,
|
||||
y: train_labels[i*batch_size:(i+1)*batch_size]
|
||||
}
|
||||
|
|
@ -45,22 +69,30 @@ def train(train_synteny_matrices_global,train_synteny_matrices_local,train_pfam_
|
|||
test_dict = {
|
||||
synmg: train_synteny_matrices_global[size_train:],
|
||||
synml: train_synteny_matrices_local[size_train:],
|
||||
pfam:train_pfam_matrices[size_train:].reshape((len(train_synteny_matrices_global)-size_train,7,7,1)),
|
||||
pfam: train_pfam_matrices[size_train:]
|
||||
.reshape((
|
||||
len(train_synteny_matrices_global)-size_train, 7, 7, 1)),
|
||||
bls: train_branch_length_species[size_train:],
|
||||
blhs: train_branch_length_homology_species[size_train:],
|
||||
dps:train_dist_p_s[size_train:].reshape((len(train_synteny_matrices_global)-size_train,1)),
|
||||
dphs:train_dist_p_hs[size_train:].reshape((len(train_synteny_matrices_global)-size_train,1)),
|
||||
dis:train_distance[size_train:].reshape((len(train_synteny_matrices_global)-size_train,1)),
|
||||
dps: train_dist_p_s[size_train:]
|
||||
.reshape((len(train_synteny_matrices_global)-size_train, 1)),
|
||||
dphs: train_dist_p_hs[size_train:]
|
||||
.reshape((len(train_synteny_matrices_global)-size_train, 1)),
|
||||
dis: train_distance[size_train:]
|
||||
.reshape((len(train_synteny_matrices_global)-size_train, 1)),
|
||||
y: train_labels[size_train:],
|
||||
lr: learn
|
||||
}
|
||||
accuracy_test,loss_test,summary_write=sess.run([accuracy,loss,summary],feed_dict=test_dict)
|
||||
accuracy_test, loss_test, summary_write = sess.run(
|
||||
[accuracy, loss, summary], feed_dict=test_dict)
|
||||
writer.add_summary(summary_write, i+1)
|
||||
print("Epoch:{} Test Accuracy:{} Test Loss:{}".format(j+1,accuracy_test*100,loss_test))
|
||||
print("Epoch:{} Test Accuracy:{} Test Loss:{}".format(
|
||||
j+1, accuracy_test*100, loss_test))
|
||||
learn *= decay
|
||||
saver.save(sess, model_name+"_v"+str(v)+"/model.ckpt")
|
||||
writer.close()
|
||||
|
||||
|
||||
def read_positive(len_p, bls, blhs, dis, dps, dphs, sml, smg, pfam, label):
|
||||
rowsh = []
|
||||
with open("dataset", "rb") as file:
|
||||
|
|
@ -123,7 +155,9 @@ def read_negative(len_n,bls,blhs,dis,dps,dphs,sml,smg,pfam,label):
|
|||
label.append(row["label"])
|
||||
pfam.append(row["pfam_matrix"])
|
||||
|
||||
def train_models(model_name,start,end,num_epochs,learn_rate,decay,size_train,batch_size):
|
||||
|
||||
def train_models(model_name, start, end, num_epochs, learn_rate,
|
||||
decay, size_train, batch_size):
|
||||
k = 1
|
||||
for i in range(start//10, end//10+1):
|
||||
bls = []
|
||||
|
|
@ -170,9 +204,11 @@ def train_models(model_name,start,end,num_epochs,learn_rate,decay,size_train,bat
|
|||
sml = sml[shi]
|
||||
smg = smg[shi]
|
||||
pfam = pfam[shi]
|
||||
train(smg,sml,pfam,bls,blhs,dps,dphs,dis,labels,k,num_epochs,learn_rate,decay,int(0.9*size_train),batch_size,model_name)
|
||||
train(smg, sml, pfam, bls, blhs, dps, dphs, dis, labels, k, num_epochs,
|
||||
learn_rate, decay, int(0.9*size_train), batch_size, model_name)
|
||||
k += 1
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
model_name = arg[-8]
|
||||
|
|
@ -183,10 +219,9 @@ def main():
|
|||
decay = float(arg[-3])
|
||||
size_train = float(arg[-2])
|
||||
batch_size = int(arg[-1])
|
||||
train_models(model_name,start_p,end_p,num_epochs,learn_rate,decay,size_train,batch_size)
|
||||
train_models(model_name, start_p, end_p, num_epochs,
|
||||
learn_rate, decay, size_train, batch_size)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Reference in a new issue