mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
Add files via upload
This commit is contained in:
parent
29bd784782
commit
a58c842610
16 changed files with 971 additions and 821 deletions
|
|
@ -3,14 +3,19 @@ import requests
|
|||
import progressbar
|
||||
import sys
|
||||
|
||||
|
||||
def update_protein(gene_seq, gene):
|
||||
t = 0
|
||||
while(t != 2):
|
||||
try:
|
||||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id/"+str(gene)+"?type=protein;multiple_sequences=1"
|
||||
ext = "/sequence/id/" + \
|
||||
str(gene) + "?type=protein;multiple_sequences=1"
|
||||
|
||||
r = requests.get(server+ext, headers={ "Content-Type" : "application/json"})
|
||||
r = requests.get(
|
||||
server + ext,
|
||||
headers={
|
||||
"Content-Type": "application/json"})
|
||||
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
|
|
@ -31,12 +36,13 @@ def update_protein(gene_seq,gene):
|
|||
r = dict(r[maxi])
|
||||
gene_seq[gene] = str(r["seq"])
|
||||
return
|
||||
except :
|
||||
except BaseException:
|
||||
t += 1
|
||||
# print("\nError:",e)
|
||||
continue
|
||||
gene_seq[gene] = ""
|
||||
|
||||
|
||||
def update_rest_protein(data):
|
||||
gids = {}
|
||||
with open("processed/not_found.json", "r") as file:
|
||||
|
|
@ -48,13 +54,19 @@ def update_rest_protein(data):
|
|||
|
||||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id?type=protein"
|
||||
headers={ "Content-Type" : "application/json", "Accept" : "application/json"}
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Accept": "application/json"}
|
||||
|
||||
for i in progressbar.progressbar(range(0, len(gids) - 50, 50)):
|
||||
ids = dict(ids=list(gids[i:i + 50]))
|
||||
while(1):
|
||||
try:
|
||||
r = requests.post(server+ext, headers=headers, data=str(json.dumps(ids)))
|
||||
r = requests.post(
|
||||
server + ext,
|
||||
headers=headers,
|
||||
data=str(
|
||||
json.dumps(ids)))
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
gs = r.json()
|
||||
|
|
@ -71,21 +83,26 @@ def update_rest_protein(data):
|
|||
for genes in gids:
|
||||
try:
|
||||
_ = data[genes]
|
||||
except:
|
||||
except BaseException:
|
||||
print(genes)
|
||||
update_protein(data, genes)
|
||||
|
||||
print("Gene Sequences Updated Successfully")
|
||||
return data
|
||||
|
||||
|
||||
def update(gene_seq, gene):
|
||||
t = 0
|
||||
while(t != 2):
|
||||
try:
|
||||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id/"+str(gene)+"?type=cds;multiple_sequences=1"
|
||||
ext = "/sequence/id/" + \
|
||||
str(gene) + "?type=cds;multiple_sequences=1"
|
||||
|
||||
r = requests.get(server+ext, headers={ "Content-Type" : "application/json"})
|
||||
r = requests.get(
|
||||
server + ext,
|
||||
headers={
|
||||
"Content-Type": "application/json"})
|
||||
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
|
|
@ -106,12 +123,13 @@ def update(gene_seq,gene):
|
|||
r = dict(r[maxi])
|
||||
gene_seq[gene] = str(r["seq"])
|
||||
return
|
||||
except Exception as e:
|
||||
except BaseException:
|
||||
t += 1
|
||||
# print("\nError:",e)
|
||||
continue
|
||||
gene_seq[gene] = ""
|
||||
|
||||
|
||||
def update_rest(data, fname):
|
||||
gids = {}
|
||||
with open("processed/not_found_" + fname + ".json", "r") as file:
|
||||
|
|
@ -123,13 +141,19 @@ def update_rest(data,fname):
|
|||
|
||||
server = "https://rest.ensembl.org"
|
||||
ext = "/sequence/id?type=cds"
|
||||
headers={ "Content-Type" : "application/json", "Accept" : "application/json"}
|
||||
headers = {
|
||||
"Content-Type": "application/json",
|
||||
"Accept": "application/json"}
|
||||
|
||||
for i in progressbar.progressbar(range(0, len(gids) - 50, 50)):
|
||||
ids = dict(ids=list(gids[i:i + 50]))
|
||||
while(1):
|
||||
try:
|
||||
r = requests.post(server+ext, headers=headers, data=str(json.dumps(ids)))
|
||||
r = requests.post(
|
||||
server + ext,
|
||||
headers=headers,
|
||||
data=str(
|
||||
json.dumps(ids)))
|
||||
if not r.ok:
|
||||
r.raise_for_status()
|
||||
gs = r.json()
|
||||
|
|
@ -146,7 +170,7 @@ def update_rest(data,fname):
|
|||
for genes in gids:
|
||||
try:
|
||||
_ = data[genes]
|
||||
except:
|
||||
except BaseException:
|
||||
print(genes)
|
||||
update(data, genes)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,3 @@
|
|||
import pandas as pd
|
||||
import requests
|
||||
import sys
|
||||
import pickle
|
||||
from get_data import get_data_genome
|
||||
|
||||
|
|
|
|||
|
|
@ -1,7 +1,4 @@
|
|||
import pandas as pd
|
||||
import numpy as np
|
||||
import json
|
||||
import gc
|
||||
import pickle
|
||||
import os
|
||||
import sys
|
||||
|
|
@ -9,6 +6,7 @@ from tree_data import create_tree_data
|
|||
from process_negative import read_database_txt
|
||||
from select_data import read_db_homology
|
||||
|
||||
|
||||
def read_data_homology(dirname, nfname):
|
||||
lf = os.listdir(dirname)
|
||||
if len(lf) == 0:
|
||||
|
|
@ -20,20 +18,27 @@ def read_data_homology(dirname,nfname):
|
|||
df, n = read_db_homology(dirname, x)
|
||||
n = n.split()[0]
|
||||
try:
|
||||
indexes=np.load("processed/synteny_matrices/"+n+"_indexes.npy")
|
||||
except:
|
||||
indexes = np.load(
|
||||
"processed/synteny_matrices/" +
|
||||
n +
|
||||
"_indexes.npy")
|
||||
except BaseException:
|
||||
print("Incomplete data for:", n)
|
||||
df = df.loc[indexes]
|
||||
a_h.append(df)
|
||||
d_h.append(n)
|
||||
# read the negative dataset
|
||||
df = read_database_txt(nfname)
|
||||
indexes=np.load("processed/synteny_matrices/"+nfname.split(".")[0]+"_indexes.npy")
|
||||
indexes = np.load(
|
||||
"processed/synteny_matrices/" +
|
||||
nfname.split(".")[0] +
|
||||
"_indexes.npy")
|
||||
df = df.loc[indexes]
|
||||
a_h.append(df)
|
||||
d_h.append(nfname.split(".")[0])
|
||||
return a_h, d_h
|
||||
|
||||
|
||||
def prepare_features(a_h, d_h, sptree, label):
|
||||
rows = []
|
||||
smg_name = "_synteny_matrices_global.npy"
|
||||
|
|
@ -47,12 +52,15 @@ def prepare_features(a_h,d_h,sptree,label):
|
|||
smg = np.load(dir_name + n + smg_name)
|
||||
sml = np.load(dir_name + n + sml_name)
|
||||
indexes = np.load(dir_name + n + smi_name)
|
||||
except:
|
||||
except BaseException:
|
||||
print("Incomplete data for:", n)
|
||||
continue
|
||||
df = df.loc[indexes]
|
||||
|
||||
branch_length_species,branch_length_homology_species,distance,dist_p_s,dist_p_hs=create_tree_data(sptree,df)
|
||||
branch_length_species, \
|
||||
branch_length_homology_species, \
|
||||
distance, dist_p_s, dist_p_hs = create_tree_data(
|
||||
sptree, df)
|
||||
assert(len(branch_length_species) == len(df))
|
||||
assert(len(sml) == len(distance))
|
||||
|
||||
|
|
@ -76,6 +84,7 @@ def prepare_features(a_h,d_h,sptree,label):
|
|||
rows.append(r)
|
||||
return rows
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
nfname = arg[-1]
|
||||
|
|
@ -92,5 +101,6 @@ def main():
|
|||
pickle.dump(rows, file)
|
||||
print("Dataset_Finalized")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
|
|||
12
get_data.py
12
get_data.py
|
|
@ -1,8 +1,7 @@
|
|||
import sys
|
||||
import os
|
||||
from read_data import read_data_genome,read_data_homology
|
||||
from read_data import read_data_genome
|
||||
from process_data import list_dict_genomes, create_chromosome_maps
|
||||
|
||||
|
||||
def get_data_genome(dir):
|
||||
a = []
|
||||
d = {}
|
||||
|
|
@ -17,10 +16,3 @@ def get_data_genome(dir):
|
|||
for i in range(len(ld)):
|
||||
assert(len(ld[i]) == len(ldg[i]))
|
||||
return cmap, cimap, ld, ldg, a, d
|
||||
|
||||
def get_data_homology(dir):
|
||||
a_h=[]
|
||||
d_h={}
|
||||
a_h,d_h=read_data_homology(dir)
|
||||
assert(len(a_h)==len(d_h))
|
||||
return a_h,d_h
|
||||
|
|
|
|||
|
|
@ -1,13 +1,11 @@
|
|||
import sys
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import json
|
||||
import os
|
||||
import gc
|
||||
import pickle
|
||||
from select_data import read_db_homology
|
||||
from process_data import create_data_homology_ls
|
||||
|
||||
|
||||
def read_genome_maps():
|
||||
data = {}
|
||||
with open("genome_maps", "rb") as file:
|
||||
|
|
@ -20,6 +18,7 @@ def read_genome_maps():
|
|||
d = data["d"]
|
||||
return a, d, ld, ldg, cmap, cimap
|
||||
|
||||
|
||||
def read_data_homology(dirname):
|
||||
lf = os.listdir(dirname)
|
||||
if len(lf) == 0:
|
||||
|
|
@ -32,14 +31,14 @@ def read_data_homology(dirname):
|
|||
n = n.split()[0]
|
||||
try:
|
||||
indexes = np.load("processed/" + n + "_selected_indexes.npy")
|
||||
except:
|
||||
except BaseException:
|
||||
print("Incomplete data for:", n)
|
||||
df = df.loc[indexes]
|
||||
print(len(df))
|
||||
a_h.append(df)
|
||||
d_h.append(n)
|
||||
return a_h, d_h
|
||||
|
||||
|
||||
def main():
|
||||
a, d, ld, ldg, cmap, cimap = read_genome_maps()
|
||||
print("Genome Maps Loaded.")
|
||||
|
|
@ -49,5 +48,6 @@ def main():
|
|||
_ = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 1)
|
||||
print("Neighbor Genes Found and Saved Successfully:)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -1,8 +1,5 @@
|
|||
import json
|
||||
import gc
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import pickle
|
||||
import tensorflow as tf
|
||||
import sys
|
||||
import progressbar
|
||||
|
|
@ -15,6 +12,7 @@ from prepare_synteny_matrix import read_data_synteny
|
|||
from tree_data import create_tree_data
|
||||
from process_data import create_map_list
|
||||
|
||||
|
||||
def read_database(fname):
|
||||
df = pd.read_csv(fname, sep="\t", header=None)
|
||||
label_dict = dict(ortholog_one2one=1,
|
||||
|
|
@ -30,9 +28,17 @@ def read_database(fname):
|
|||
df = df.assign(label=label)
|
||||
df = df.drop(7, axis=1)
|
||||
df = df.drop(0, axis=1)
|
||||
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","goc","wga","label"]
|
||||
df.columns = [
|
||||
"gene_stable_id",
|
||||
"species",
|
||||
"homology_gene_stable_id",
|
||||
"homology_species",
|
||||
"goc",
|
||||
"wga",
|
||||
"label"]
|
||||
return df
|
||||
|
||||
|
||||
def select_data_by_length(df, st, end):
|
||||
try:
|
||||
if end < len(df):
|
||||
|
|
@ -40,18 +46,21 @@ def select_data_by_length(df,st,end):
|
|||
df = df.loc[df.index.values[st:end]]
|
||||
else:
|
||||
raise ValueError()
|
||||
except:
|
||||
except BaseException:
|
||||
print("Making Predictions for the complete dataframe:)")
|
||||
print(len(df))
|
||||
return df
|
||||
|
||||
|
||||
def create_synteny_features(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, name):
|
||||
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","prediction_"+name)
|
||||
gene_sequences = read_gene_sequences(
|
||||
a_h, lsy, "geneseq", "prediction_" + name)
|
||||
gene_sequences = update_rest(gene_sequences, "prediction_" + name)
|
||||
print("Gene Sequences Loaded.")
|
||||
return lsy, gene_sequences
|
||||
|
||||
|
||||
def threadmaker(nop, df, lsy, gene_sequences, n, name):
|
||||
part = len(df) // nop
|
||||
pr = Procerssrunner()
|
||||
|
|
@ -62,23 +71,27 @@ def threadmaker(nop,df,lsy,gene_sequences,n,name):
|
|||
indexes = np.array(indexes)
|
||||
return sml, smg, indexes
|
||||
|
||||
|
||||
def get_prediction(smg, sml, indexes, bls, blhs, dis, dps, dphs, model_name):
|
||||
preds = np.zeros((len(smg), 3))
|
||||
w = [0.86, 0.8, 0.06]
|
||||
for i in range(1, 4):
|
||||
try:
|
||||
model=tf.train.import_meta_graph(model_name+'_v'+str(i)+'/model.ckpt.meta')
|
||||
except:
|
||||
model = tf.train.import_meta_graph(
|
||||
model_name + '_v' + str(i) + '/model.ckpt.meta')
|
||||
except BaseException:
|
||||
print("Something wrong with the model.")
|
||||
continue
|
||||
with tf.Session() as sess:
|
||||
try:
|
||||
model.restore(sess, model_name + '_v' + str(i) + "/model.ckpt")
|
||||
graph = tf.get_default_graph()
|
||||
synmgt,synmlt,blst,blhst,dpst,dphst,dist,lrt,yt=graph.get_collection("input_nodes")
|
||||
synmgt, synmlt, blst, \
|
||||
blhst, dpst, dphst, \
|
||||
dist, lrt, yt = graph.get_collection("input_nodes")
|
||||
predictions = graph.get_tensor_by_name("Predictions/BiasAdd:0")
|
||||
print("Model Loaded Successfully :)")
|
||||
except:
|
||||
except BaseException:
|
||||
print(":(")
|
||||
sys.exit()
|
||||
|
||||
|
|
@ -106,8 +119,17 @@ def get_prediction(smg,sml,indexes,bls,blhs,dis,dps,dphs,model_name):
|
|||
print(preds.shape)
|
||||
return preds
|
||||
|
||||
|
||||
def write_preds(fname, model_name, name, preds, index_dict, df):
|
||||
print("Writing predcitions to:","prediction_"+fname+"_"+model_name+"_"+name+"_multiple.txt")
|
||||
print(
|
||||
"Writing predcitions to:",
|
||||
"prediction_" +
|
||||
fname +
|
||||
"_" +
|
||||
model_name +
|
||||
"_" +
|
||||
name +
|
||||
"_multiple.txt")
|
||||
with open("prediction_" + fname + "_" + model_name + "_" + name + "_multiple.txt", "w") as file:
|
||||
for index, row in progressbar.progressbar(df.iterrows()):
|
||||
file.write(str(row[0]))
|
||||
|
|
@ -131,6 +153,7 @@ def write_preds(fname,model_name,name,preds,index_dict,df):
|
|||
file.write("NaN")
|
||||
file.write("\n")
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
fname = arg[-6]
|
||||
|
|
@ -148,14 +171,25 @@ def main():
|
|||
a_h = [df]
|
||||
d_h = ["prediction"]
|
||||
|
||||
lsy,gene_sequences=create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,name)
|
||||
lsy, gene_sequences = create_synteny_features(
|
||||
a_h, d_h, n, a, d, ld, ldg, cmap, cimap, name)
|
||||
sml, smg, indexes = threadmaker(nop, df, lsy, gene_sequences, n, name)
|
||||
df_temp = df.loc[indexes]
|
||||
bls, blhs, dis, dps, dphs = create_tree_data("species_tree.tree", df_temp)
|
||||
index_dict = create_map_list(indexes)
|
||||
preds=get_prediction(smg,sml,indexes,bls,blhs,dis,dps,dphs,model_name)
|
||||
preds = get_prediction(
|
||||
smg,
|
||||
sml,
|
||||
indexes,
|
||||
bls,
|
||||
blhs,
|
||||
dis,
|
||||
dps,
|
||||
dphs,
|
||||
model_name)
|
||||
|
||||
write_preds(fname, model_name, name, preds, index_dict, df)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
@ -1,5 +1,4 @@
|
|||
import numpy as np
|
||||
import pandas as pd
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
|
|
@ -9,6 +8,7 @@ from threads import Procerssrunner
|
|||
from read_get_gene_seq import read_gene_sequences
|
||||
from access_data_rest import update_rest
|
||||
|
||||
|
||||
def read_data_synteny(nop, name):
|
||||
smg = []
|
||||
sml = []
|
||||
|
|
@ -27,6 +27,7 @@ def read_data_synteny(nop,name):
|
|||
print(len(indexes))
|
||||
return smg, sml, indexes
|
||||
|
||||
|
||||
def load_neighbor_genes():
|
||||
with open("processed/neighbor_genes.json", "r") as file:
|
||||
lsy = dict(json.load(file))
|
||||
|
|
@ -34,6 +35,7 @@ def load_neighbor_genes():
|
|||
print("Neighbor Genes Loaded")
|
||||
return lsy
|
||||
|
||||
|
||||
def read_data_homology(dirname):
|
||||
lf = os.listdir(dirname)
|
||||
if len(lf) == 0:
|
||||
|
|
@ -46,13 +48,14 @@ def read_data_homology(dirname):
|
|||
n = n.split()[0]
|
||||
try:
|
||||
indexes = np.load("processed/" + n + "_selected_indexes.npy")
|
||||
except:
|
||||
except BaseException:
|
||||
print("Incomplete data for:", n)
|
||||
df = df.loc[indexes]
|
||||
a_h.append(df)
|
||||
d_h.append(n)
|
||||
return a_h, d_h
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
nop = int(arg[-1])
|
||||
|
|
@ -60,7 +63,8 @@ def main():
|
|||
a_h, d_h = read_data_homology("data_homology")
|
||||
print("Data Read")
|
||||
lsy = load_neighbor_genes()
|
||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","gene_seq_positive")
|
||||
gene_sequences = read_gene_sequences(
|
||||
a_h, lsy, "geneseq", "gene_seq_positive")
|
||||
gene_sequences = update_rest(gene_sequences, "gene_seq_positive")
|
||||
print("Gene Sequences Loaded.")
|
||||
if not os.path.isdir("processed/synteny_matrices"):
|
||||
|
|
@ -81,5 +85,6 @@ def main():
|
|||
np.save(ndir + str(d_h[i]) + "_" + nf3, indexes)
|
||||
print("Synteny Matrices Created Successfully :)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
|
|||
|
|
@ -1,17 +1,15 @@
|
|||
import pandas
|
||||
import gc
|
||||
import numpy as np
|
||||
import json
|
||||
import os
|
||||
import progressbar
|
||||
from save_data import write_dict_json
|
||||
|
||||
|
||||
def create_map_list(l): # this function maps the indexes to values
|
||||
t = {}
|
||||
for i in range(len(l)):
|
||||
t[l[i]] = i
|
||||
return t
|
||||
|
||||
|
||||
def create_chromosome_maps(a, n):
|
||||
cmap = []
|
||||
cimap = []
|
||||
|
|
@ -21,8 +19,8 @@ def create_chromosome_maps(a,n):
|
|||
for index, row in df.iterrows():
|
||||
g = row.gene_id
|
||||
try:
|
||||
temp=chmap[g]
|
||||
except:
|
||||
_ = chmap[g]
|
||||
except BaseException:
|
||||
chmap[g] = str(row.Chr)
|
||||
if str(row.Chr) in chindmap:
|
||||
chindmap[str(row.Chr)].append(index)
|
||||
|
|
@ -33,6 +31,7 @@ def create_chromosome_maps(a,n):
|
|||
cimap.append(chindmap)
|
||||
return cmap, cimap
|
||||
|
||||
|
||||
def list_dict_genomes(a, n):
|
||||
lst = []
|
||||
ldt = []
|
||||
|
|
@ -45,17 +44,20 @@ def list_dict_genomes(a,n):
|
|||
ldt.append(ldgt)
|
||||
return lst, ldt
|
||||
|
||||
|
||||
def get_nearest_neighbors(g, gs, n, a, d, ld, ldg, cmap, cimap):
|
||||
# print("Finding Neighbor Genes")
|
||||
ne = [] # list to store the backward genes
|
||||
nr = [] # list to store the forward genes
|
||||
gi=d[gs.capitalize()] #get the address of the corresponding species to which the gene belongs whose neighbor has to be found
|
||||
# get the address of the corresponding species to which the gene belongs
|
||||
# whose neighbor has to be found
|
||||
gi = d[gs.capitalize()]
|
||||
sldf = a[gi] # select the dataframe
|
||||
scmap = cmap[gi] # select the correct chromosome map
|
||||
scimap = cimap[gi] # select the correct index maps
|
||||
try:
|
||||
sld=ld[gi]#see if the corresponding gene map exists
|
||||
except:
|
||||
_ = ld[gi] # see if the corresponding gene map exists
|
||||
except BaseException:
|
||||
# print("Length of Dataframes:{} \t Length of Loaded Genes:{} \t Length of Loaded Genomes Dictionaries:{}".format(len(a),len(ld),len(ldg)))
|
||||
return ne, nr
|
||||
sldg = ldg[gi] # select the corresponding map
|
||||
|
|
@ -85,12 +87,15 @@ def get_nearest_neighbors(g,gs,n,a,d,ld,ldg,cmap,cimap):
|
|||
ne.append("NULL_GENE") # append the NULL_GENE value
|
||||
continue
|
||||
for k in end_s: # iterate through the sorted array
|
||||
if end[k]<0 and end[k+1]>=0:#find the first value that is negative and the next one is positive to get the nearest gene
|
||||
# find the first value that is negative and the next one is
|
||||
# positive to get the nearest gene
|
||||
if end[k] < 0 and end[k + 1] >= 0:
|
||||
itemp = k
|
||||
break
|
||||
itemp = scimap[itemp]
|
||||
ne.append(sldf.loc[itemp].gene_id)
|
||||
start=int(sldf.loc[itemp].start)#make "start" the start location of the current gene
|
||||
# make "start" the start location of the current gene
|
||||
start = int(sldf.loc[itemp].start)
|
||||
# print(start)
|
||||
# get the +n neighbors
|
||||
flag = 0
|
||||
|
|
@ -117,8 +122,11 @@ def get_nearest_neighbors(g,gs,n,a,d,ld,ldg,cmap,cimap):
|
|||
end = int(sldf.loc[itemp].end)
|
||||
return ne, nr
|
||||
|
||||
|
||||
def create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, to_write):
|
||||
lsy={} #dictionary which stores +/- n genes of the given gene by id. Each key is a gene id which corresponds to the one in center.
|
||||
# dictionary which stores +/- n genes of the given gene by id. Each key is
|
||||
# a gene id which corresponds to the one in center.
|
||||
lsy = {}
|
||||
lsytemp = {}
|
||||
name = "neighbor_genes"
|
||||
for df in a_h:
|
||||
|
|
@ -128,26 +136,30 @@ def create_data_homology_ls(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,to_write):
|
|||
xs = row["species"]
|
||||
ys = row["homology_species"]
|
||||
try:
|
||||
z=lsy[x]
|
||||
except:
|
||||
_ = lsy[x]
|
||||
except BaseException:
|
||||
try:
|
||||
t2=d[xs.capitalize()]#see if the species exist in genomic maps
|
||||
xl,xr=get_nearest_neighbors(x,xs,n,a,d,ld,ldg,cmap,cimap)
|
||||
if len(xl)!=0:#check if neighboring genes were successfully found
|
||||
# see if the species exist in genomic maps
|
||||
_ = d[xs.capitalize()]
|
||||
xl, xr = get_nearest_neighbors(
|
||||
x, xs, n, a, d, ld, ldg, cmap, cimap)
|
||||
if len(
|
||||
xl) != 0: # check if neighboring genes were successfully found
|
||||
lsy[x] = dict(b=xl, f=xr)
|
||||
lsytemp[x] = dict(b=xl, f=xr)
|
||||
except:
|
||||
except BaseException:
|
||||
continue
|
||||
try:
|
||||
z=lsy[y]
|
||||
except:
|
||||
_ = lsy[y]
|
||||
except BaseException:
|
||||
try:
|
||||
t2=d[ys.capitalize()]
|
||||
yl,yr=get_nearest_neighbors(y,ys,n,a,d,ld,ldg,cmap,cimap)
|
||||
_ = d[ys.capitalize()]
|
||||
yl, yr = get_nearest_neighbors(
|
||||
y, ys, n, a, d, ld, ldg, cmap, cimap)
|
||||
if len(yl) != 0:
|
||||
lsy[y] = dict(b=yl, f=yr)
|
||||
lsytemp[y] = dict(b=yl, f=yr)
|
||||
except:
|
||||
except BaseException:
|
||||
continue
|
||||
if to_write == 1:
|
||||
write_dict_json(name, "processed", lsy)
|
||||
|
|
|
|||
|
|
@ -1,9 +1,5 @@
|
|||
import pandas as pd
|
||||
import numpy as np
|
||||
import os
|
||||
import sys
|
||||
import progressbar
|
||||
import json
|
||||
import sys
|
||||
from neighbor_genes import read_genome_maps
|
||||
from process_data import create_data_homology_ls
|
||||
|
|
@ -13,12 +9,21 @@ from access_data_rest import update_rest
|
|||
from prepare_synteny_matrix import read_data_synteny
|
||||
from save_data import write_dict_json
|
||||
|
||||
|
||||
def read_database_txt(filename):
|
||||
df = pd.read_csv(filename, sep="\t", header=None)
|
||||
df = df.drop(0, axis=1)
|
||||
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","wga","goc","homology_type"]
|
||||
df.columns = [
|
||||
"gene_stable_id",
|
||||
"species",
|
||||
"homology_gene_stable_id",
|
||||
"homology_species",
|
||||
"wga",
|
||||
"goc",
|
||||
"homology_type"]
|
||||
return df
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
a, d, ld, ldg, cmap, cimap = read_genome_maps()
|
||||
|
|
@ -34,7 +39,8 @@ def main():
|
|||
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
||||
write_dict_json("neighbor_genes_negative", "processed", lsy)
|
||||
print("Neighbor Genes Found and Saved Successfully:)")
|
||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","gene_seq_negative")
|
||||
gene_sequences = read_gene_sequences(
|
||||
a_h, lsy, "geneseq", "gene_seq_negative")
|
||||
gene_sequences = update_rest(gene_sequences, "gene_seq_negative")
|
||||
ndir = "processed/synteny_matrices/"
|
||||
nf1 = "synteny_matrices_global"
|
||||
|
|
@ -52,6 +58,6 @@ def main():
|
|||
np.save(ndir + str(d_h[i]) + "_" + nf3, indexes)
|
||||
print("Synteny Matrices Created Successfully :)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
|
|
|
|||
42
read_data.py
42
read_data.py
|
|
@ -1,50 +1,72 @@
|
|||
import os
|
||||
import pandas as pd
|
||||
import gzip
|
||||
import sys
|
||||
import progressbar
|
||||
import traceback
|
||||
|
||||
|
||||
def clear_data(x):
|
||||
if x==None:
|
||||
if x is None:
|
||||
return x
|
||||
x = x.split()
|
||||
try:
|
||||
x = x[1]
|
||||
except:
|
||||
c=0
|
||||
except BaseException:
|
||||
_ = 0
|
||||
# print(x)
|
||||
x = x[1:-1]
|
||||
return x
|
||||
|
||||
|
||||
def read_data_genome(dir_name, a, dict_ind_genome):
|
||||
lf = os.listdir(dir_name)
|
||||
if len(lf) == 0:
|
||||
print("No files in the data directory!!!!!!")
|
||||
sys.exit(1)
|
||||
colname=["Chr","source","feature","start","end","score","strand","frame","attribute"]
|
||||
colname = [
|
||||
"Chr",
|
||||
"source",
|
||||
"feature",
|
||||
"start",
|
||||
"end",
|
||||
"score",
|
||||
"strand",
|
||||
"frame",
|
||||
"attribute"]
|
||||
print("Going to read data:")
|
||||
for x in progressbar.progressbar(range(len(lf))):
|
||||
data_gene=pd.read_csv(dir_name+"/"+lf[x],compression='gzip',sep='\t',comment='#',header=None,names=colname)
|
||||
data_gene = pd.read_csv(
|
||||
dir_name + "/" + lf[x],
|
||||
compression='gzip',
|
||||
sep='\t',
|
||||
comment='#',
|
||||
header=None,
|
||||
names=colname)
|
||||
# print(data_gene.head)
|
||||
data_gene = data_gene[data_gene["feature"] == "gene"]
|
||||
tmp = data_gene["attribute"].str.split(";", expand=True)
|
||||
tmp = tmp.iloc[:, :5]
|
||||
data_gene[["gene_id","gene_version","gene_name","gene_source","gene_biotype"]]=tmp
|
||||
data_gene[["gene_id", "gene_version", "gene_name",
|
||||
"gene_source", "gene_biotype"]] = tmp
|
||||
data_gene = data_gene.drop("attribute", axis=1)
|
||||
# print(data_gene[0:10])
|
||||
try:
|
||||
for y in ["gene_version","gene_name","gene_source","gene_biotype","gene_id"]:
|
||||
for y in [
|
||||
"gene_version",
|
||||
"gene_name",
|
||||
"gene_source",
|
||||
"gene_biotype",
|
||||
"gene_id"]:
|
||||
data_gene[y] = data_gene[y].apply(clear_data)
|
||||
except Exception as e:
|
||||
traceback.print_exc()
|
||||
print(e)
|
||||
continue
|
||||
# print(data_gene[0:10])
|
||||
data_gene=data_gene[(data_gene['gene_biotype']=='protein_coding') | (data_gene['gene_source']=='protein_coding')]
|
||||
data_gene = data_gene[(data_gene['gene_biotype'] == 'protein_coding') | (
|
||||
data_gene['gene_source'] == 'protein_coding')]
|
||||
# print(data_gene[data_gene["gene_id"]=="ENSNGAG00000000407"])
|
||||
a.append(data_gene)
|
||||
n = lf[x].split(".")[0]
|
||||
dict_ind_genome[n] = len(a) - 1
|
||||
return a, dict_ind_genome
|
||||
|
||||
|
|
|
|||
|
|
@ -1,11 +1,10 @@
|
|||
import json
|
||||
from Bio import SeqIO
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import os
|
||||
import gzip
|
||||
import progressbar
|
||||
|
||||
|
||||
def read_from_multiple_lsy(lsyfl):
|
||||
lsy = {}
|
||||
for f in lsyfl:
|
||||
|
|
@ -16,7 +15,10 @@ def read_from_multiple_lsy(lsyfl):
|
|||
lsy[t] = d[t]
|
||||
return lsy
|
||||
|
||||
#this function updates the given dictionary with the given keys and values list
|
||||
# this function updates the given dictionary with the given keys and
|
||||
# values list
|
||||
|
||||
|
||||
def create_dict(keys, values, dictionary):
|
||||
for i in range(len(keys)):
|
||||
if keys[i] not in dictionary:
|
||||
|
|
@ -26,6 +28,8 @@ def create_dict(keys,values,dictionary):
|
|||
|
||||
# this function maps all the genes to their respective species.
|
||||
# (Function: when finding the species of any gene we do not need to search the entire dataframe)
|
||||
|
||||
|
||||
def group_seq_by_species(df, g_to_sp):
|
||||
sph = list(df.homology_species)
|
||||
ghsp = list(df.homology_gene_stable_id)
|
||||
|
|
@ -35,7 +39,10 @@ def group_seq_by_species(df,g_to_sp):
|
|||
create_dict(ghsp, sph, g_to_sp)
|
||||
return g_to_sp
|
||||
|
||||
#this function returns the gene-id and gene-biotype from the description in the fasta file record.
|
||||
# this function returns the gene-id and gene-biotype from the description
|
||||
# in the fasta file record.
|
||||
|
||||
|
||||
def description_cleaner(description):
|
||||
description = description.split()
|
||||
t = ""
|
||||
|
|
@ -47,15 +54,17 @@ def description_cleaner(description):
|
|||
t = x[1].split(".")[0]
|
||||
if x[0] == "gene_biotype":
|
||||
gbt = x[1]
|
||||
except:
|
||||
except BaseException:
|
||||
return "aa", "aa"
|
||||
return t, gbt
|
||||
|
||||
|
||||
def read_gene_seq(dirname, s, genes_by_species):
|
||||
lof = os.listdir(dirname) # list all the files in the sequences directory
|
||||
ftr = []
|
||||
for f in lof:
|
||||
if f.split(".")[0] in s:#check whether the species is present in the species to read list. Will skip those species which are not present in the dataframe
|
||||
if f.split(".")[
|
||||
0] in s: # check whether the species is present in the species to read list. Will skip those species which are not present in the dataframe
|
||||
ftr.append(f)
|
||||
data = {}
|
||||
for f in progressbar.progressbar(ftr):
|
||||
|
|
@ -64,12 +73,13 @@ def read_gene_seq(dirname,s,genes_by_species):
|
|||
record = SeqIO.parse(file, "fasta")
|
||||
for r in record:
|
||||
gid, gbt = description_cleaner(r.description)
|
||||
if str(gid) not in data and str(gid) in genes_by_species[species] and gbt=="protein_coding":
|
||||
if str(gid) not in data and str(
|
||||
gid) in genes_by_species[species] and gbt == "protein_coding":
|
||||
data[gid] = str(r.seq)
|
||||
return data
|
||||
|
||||
def read_gene_sequences(hdf,lsy,data_dir,fname):
|
||||
|
||||
def read_gene_sequences(hdf, lsy, data_dir, fname):
|
||||
"""The basic idea here is to create a list/dictionary of all the genes by their species.
|
||||
Once the mapping is done, all the respective fasta sequence files are read by Species
|
||||
and the CDNA sequences for each gene in the species record are read and stored.
|
||||
|
|
@ -86,9 +96,10 @@ def read_gene_sequences(hdf,lsy,data_dir,fname):
|
|||
for x in progressbar.progressbar(lsy):
|
||||
try:
|
||||
species = grouped_genes[x] # get the species
|
||||
except:
|
||||
except BaseException:
|
||||
continue
|
||||
if x not in gene_by_species_dict[species]:#check if the gene already exists in the species dict or not.
|
||||
# check if the gene already exists in the species dict or not.
|
||||
if x not in gene_by_species_dict[species]:
|
||||
gene_by_species_dict[species].append(x)
|
||||
xl = lsy[x]['b']
|
||||
xr = lsy[x]['f']
|
||||
|
|
@ -103,8 +114,8 @@ def read_gene_sequences(hdf,lsy,data_dir,fname):
|
|||
if gxr not in gene_by_species_dict[species]:
|
||||
gene_by_species_dict[species].append(gxr)
|
||||
|
||||
|
||||
s=[x for x in gene_by_species_dict if len(gene_by_species_dict[x])!=0]#select those species only whose gene sequences we have to read.
|
||||
# select those species only whose gene sequences we have to read.
|
||||
s = [x for x in gene_by_species_dict if len(gene_by_species_dict[x]) != 0]
|
||||
s = [x.capitalize() for x in s]
|
||||
|
||||
data = read_gene_seq(data_dir, s, gene_by_species_dict)
|
||||
|
|
@ -113,7 +124,7 @@ def read_gene_sequences(hdf,lsy,data_dir,fname):
|
|||
for gene in gene_by_species_dict[species]:
|
||||
try:
|
||||
_ = data[gene]
|
||||
except:
|
||||
except BaseException:
|
||||
not_found[gene] = 1
|
||||
|
||||
with open("processed/not_found_" + fname + ".json", "w") as file:
|
||||
|
|
|
|||
|
|
@ -1,7 +1,7 @@
|
|||
import os
|
||||
import pickle
|
||||
import json
|
||||
import sys
|
||||
|
||||
|
||||
def write_dict_json(name, dir, d):
|
||||
if not os.path.exists(dir):
|
||||
|
|
@ -11,6 +11,7 @@ def write_dict_json(name,dir,d):
|
|||
with open(path, 'w') as file:
|
||||
json.dump(d, file)
|
||||
|
||||
|
||||
def write_data_synteny(smg, sml, indexes, i, name):
|
||||
if not os.path.exists("temp_" + name):
|
||||
os.mkdir("temp_" + name)
|
||||
|
|
|
|||
|
|
@ -1,18 +1,17 @@
|
|||
import pandas as pd
|
||||
import gc
|
||||
import numpy as np
|
||||
import json
|
||||
import os
|
||||
import progressbar
|
||||
import sys
|
||||
from selector import select, create_map_reverse
|
||||
|
||||
|
||||
def read_db_homology(dir_name, filename):
|
||||
df = pd.read_csv(dir_name + "/" + filename, compression='gzip', sep='\t')
|
||||
n = filename.split(".")[0]
|
||||
n = n.split(" ")[0]
|
||||
return df, n
|
||||
|
||||
|
||||
def get_selection_data():
|
||||
with open("dist_matrix", "r") as file:
|
||||
matrix = file.readlines()
|
||||
|
|
@ -25,6 +24,7 @@ def get_selection_data():
|
|||
spnmap, nspmap = create_map_reverse(dname)
|
||||
return matrix, spnmap, nspmap
|
||||
|
||||
|
||||
def read_select_data(dirname, matrix, spnmap, nspmap, nos):
|
||||
lf = os.listdir(dirname)
|
||||
if len(lf) == 0:
|
||||
|
|
@ -40,11 +40,13 @@ def read_select_data(dirname,matrix,spnmap,nspmap,nos):
|
|||
indexes = np.array(list(df.index.values))
|
||||
np.save("processed/" + n + "_selected_indexes", indexes)
|
||||
|
||||
|
||||
def main():
|
||||
arg = sys.argv
|
||||
nos = int(arg[-1])
|
||||
matrix, spnmap, nspmap = get_selection_data()
|
||||
read_select_data("data_homology", matrix, spnmap, nspmap, nos)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
|
|
|||
39
selector.py
39
selector.py
|
|
@ -1,6 +1,7 @@
|
|||
import pandas as pd
|
||||
import numpy as np
|
||||
|
||||
|
||||
def create_map_reverse(arr):
|
||||
m = {}
|
||||
rm = {}
|
||||
|
|
@ -9,6 +10,7 @@ def create_map_reverse(arr):
|
|||
rm[i] = arr[i]
|
||||
return m, rm
|
||||
|
||||
|
||||
def get_data_prop(df, nspmap, sp, prop, nos):
|
||||
nos = int(nos * prop)
|
||||
sp = [nspmap[x] for x in sp]
|
||||
|
|
@ -20,7 +22,15 @@ def get_data_prop(df,nspmap,sp,prop,nos):
|
|||
else:
|
||||
return data
|
||||
|
||||
def create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos,hom_type,spname):
|
||||
|
||||
def create_balanced_dataset_paralog(
|
||||
df,
|
||||
matrix,
|
||||
spnmap,
|
||||
nspmap,
|
||||
nos,
|
||||
hom_type,
|
||||
spname):
|
||||
df = df[df["homology_type"] == hom_type]
|
||||
dist = matrix[spnmap[spname]]
|
||||
dist_sort = np.argsort(dist)
|
||||
|
|
@ -38,6 +48,7 @@ def create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos,hom_type,spname)
|
|||
df = pd.concat([df_r, df_dist])
|
||||
return df
|
||||
|
||||
|
||||
def select_data_goc(df, prop, nos):
|
||||
df = df[df["goc_score"] == 0.0]
|
||||
nos = int(prop * nos)
|
||||
|
|
@ -47,7 +58,15 @@ def select_data_goc(df,prop,nos):
|
|||
else:
|
||||
return df.loc[df.index.values[rind[:nos]]]
|
||||
|
||||
def create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos,hom_type,spname):
|
||||
|
||||
def create_balanced_dataset_ortholog(
|
||||
df,
|
||||
matrix,
|
||||
spnmap,
|
||||
nspmap,
|
||||
nos,
|
||||
hom_type,
|
||||
spname):
|
||||
df = df[df["homology_type"] == hom_type]
|
||||
dist = matrix[spnmap[spname]]
|
||||
dist_sort = np.argsort(dist)
|
||||
|
|
@ -69,17 +88,23 @@ def create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos,hom_type,spname
|
|||
df = pd.concat([df_r, df_goc, df_dist])
|
||||
return df
|
||||
|
||||
|
||||
def select(df, nos, matrix, spnmap, nspmap, sp):
|
||||
nos_p = int((0.5 * nos) / 2)
|
||||
nos_o = int((0.5 * nos) / 3)
|
||||
|
||||
# get the paralogy data
|
||||
df_p1=create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos_p,"within_species_paralog",sp)
|
||||
df_p2=create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos_p,"other_paralog",sp)
|
||||
df_p1 = create_balanced_dataset_paralog(
|
||||
df, matrix, spnmap, nspmap, nos_p, "within_species_paralog", sp)
|
||||
df_p2 = create_balanced_dataset_paralog(
|
||||
df, matrix, spnmap, nspmap, nos_p, "other_paralog", sp)
|
||||
# get the orthology data
|
||||
df_o1=create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos_o,"ortholog_one2many",sp)
|
||||
df_o2=create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos_o,"ortholog_many2many",sp)
|
||||
df_o3=create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos_o,"ortholog_one2one",sp)
|
||||
df_o1 = create_balanced_dataset_ortholog(
|
||||
df, matrix, spnmap, nspmap, nos_o, "ortholog_one2many", sp)
|
||||
df_o2 = create_balanced_dataset_ortholog(
|
||||
df, matrix, spnmap, nspmap, nos_o, "ortholog_many2many", sp)
|
||||
df_o3 = create_balanced_dataset_ortholog(
|
||||
df, matrix, spnmap, nspmap, nos_o, "ortholog_one2one", sp)
|
||||
|
||||
# concatenate everything
|
||||
df = pd.concat([df_o1, df_o2, df_o3, df_p1, df_p2])
|
||||
|
|
|
|||
52
threads.py
52
threads.py
|
|
@ -1,17 +1,14 @@
|
|||
import pandas as pd
|
||||
from threading import Thread
|
||||
from multiprocessing import Process,Lock,Manager
|
||||
from multiprocessing import Process
|
||||
import numpy as np
|
||||
import edlib as ed
|
||||
import pandas as pd
|
||||
import progressbar
|
||||
import time
|
||||
from skbio.alignment import local_pairwise_align_ssw
|
||||
from skbio import DNA,TabularMSA,RNA
|
||||
from skbio import DNA
|
||||
import copy
|
||||
import time
|
||||
from save_data import write_data_synteny
|
||||
|
||||
|
||||
class Thread_objects():
|
||||
def __init__(self, df_temp, gene_sequences, lsy, i, name):
|
||||
self.gene_sequences = copy.deepcopy(gene_sequences)
|
||||
|
|
@ -30,15 +27,15 @@ class Thread_objects():
|
|||
if gene == "NULL_GENE":
|
||||
continue
|
||||
try:
|
||||
temp=gene_seq[gene]
|
||||
except:
|
||||
_ = gene_seq[gene]
|
||||
except BaseException:
|
||||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||
for gene in g2:
|
||||
if gene == "NULL_GENE":
|
||||
continue
|
||||
try:
|
||||
temp=gene_seq[gene]
|
||||
except:
|
||||
_ = gene_seq[gene]
|
||||
except BaseException:
|
||||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||
sm = np.zeros((n, n, 2))
|
||||
sml = np.zeros((n, n, 2))
|
||||
|
|
@ -54,15 +51,19 @@ class Thread_objects():
|
|||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||
norm_len = max(len(gene_seq[g1[i]]), len(gene_seq[g2[j]]))
|
||||
try:
|
||||
result = ed.align(gene_seq[g1[i]],gene_seq[g2[j]], mode="NW", task="distance")
|
||||
result = ed.align(
|
||||
gene_seq[g1[i]], gene_seq[g2[j]], mode="NW", task="distance")
|
||||
sm[i][j][0] = result["editDistance"] / (norm_len)
|
||||
result = ed.align(gene_seq[g1[i]],gene_seq[g2[j]][::-1], mode="NW", task="distance")
|
||||
result = ed.align(
|
||||
gene_seq[g1[i]], gene_seq[g2[j]][::-1], mode="NW", task="distance")
|
||||
sm[i][j][1] = result["editDistance"] / (norm_len)
|
||||
_,result,_=local_pairwise_align_ssw(DNA(gene_seq[g1[i]]),DNA(gene_seq[g2[j]]))
|
||||
_, result, _ = local_pairwise_align_ssw(
|
||||
DNA(gene_seq[g1[i]]), DNA(gene_seq[g2[j]]))
|
||||
sml[i][j][0] = result / (norm_len)
|
||||
_,result,_=local_pairwise_align_ssw(DNA(gene_seq[g1[i]]),DNA(gene_seq[g2[j]][::-1]))
|
||||
_, result, _ = local_pairwise_align_ssw(
|
||||
DNA(gene_seq[g1[i]]), DNA(gene_seq[g2[j]][::-1]))
|
||||
sml[i][j][1] = result / (norm_len)
|
||||
except:
|
||||
except BaseException:
|
||||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||
return sm, sml
|
||||
|
||||
|
|
@ -76,12 +77,12 @@ class Thread_objects():
|
|||
y = []
|
||||
t += 1
|
||||
try:
|
||||
temp=lsy[g1]
|
||||
except:
|
||||
_ = lsy[g1]
|
||||
except BaseException:
|
||||
continue
|
||||
try:
|
||||
temp=lsy[g2]
|
||||
except:
|
||||
_ = lsy[g2]
|
||||
except BaseException:
|
||||
continue
|
||||
for i in range(len(lsy[g1]['b']) - 1, -1, -1):
|
||||
x.append(lsy[g1]['b'][i])
|
||||
|
|
@ -97,23 +98,29 @@ class Thread_objects():
|
|||
|
||||
assert(len(x) == len(y))
|
||||
assert(len(x) == (2 * n + 1))
|
||||
smgtemp,smltemp=self.create_synteny_matrix_mul(gene_seq,x,y,2*n+1)
|
||||
smgtemp, smltemp = self.create_synteny_matrix_mul(
|
||||
gene_seq, x, y, 2 * n + 1)
|
||||
if np.all(smgtemp == 0):
|
||||
continue
|
||||
self.smg.append(smgtemp)
|
||||
self.sml.append(smltemp)
|
||||
self.indexes.append(index)
|
||||
self.end = time.time()
|
||||
print("Thread {} finished in {}s.".format(self.i+1,self.end-self.start))
|
||||
print(
|
||||
"Thread {} finished in {}s.".format(
|
||||
self.i + 1,
|
||||
self.end - self.start))
|
||||
write_data_synteny(self.smg, self.sml, self.indexes, self.i, self.name)
|
||||
|
||||
|
||||
class Procerssrunner():
|
||||
def __init__(self):
|
||||
self.thread_alive = []
|
||||
self.obj_list = []
|
||||
|
||||
def start_thread(self, obj, i, thread_alive, n, name):
|
||||
t=Process(target=obj.synteny_matrix,args=(obj.gene_sequences,obj.df,obj.lsy,n),name="Thread_"+str(i+1))
|
||||
t = Process(target=obj.synteny_matrix, args=(
|
||||
obj.gene_sequences, obj.df, obj.lsy, n), name="Thread_" + str(i + 1))
|
||||
print("Thread ", (i + 1), " started for ", name, ".")
|
||||
thread_alive.append(t)
|
||||
|
||||
|
|
@ -135,4 +142,3 @@ class Procerssrunner():
|
|||
end = time.time()
|
||||
print("Ending Processes")
|
||||
print("Time taken:{}s".format(end - st))
|
||||
|
||||
|
|
|
|||
|
|
@ -2,12 +2,14 @@ from ete3 import Tree
|
|||
import numpy as np
|
||||
import progressbar
|
||||
|
||||
|
||||
def create_branch_length_padding(bl):
|
||||
maxlen = 29
|
||||
for x in bl:
|
||||
for i in range(len(x), maxlen):
|
||||
x.append(0)
|
||||
|
||||
|
||||
def create_tree_data(treename, df):
|
||||
t = Tree(treename)
|
||||
branch_lengths_s = []
|
||||
|
|
@ -43,4 +45,5 @@ def create_tree_data(treename,df):
|
|||
dist.append(d)
|
||||
create_branch_length_padding(branch_lengths_s)
|
||||
create_branch_length_padding(branch_lengths_hs)
|
||||
return np.array(branch_lengths_s),np.array(branch_lengths_hs),np.array(dist),np.array(ns),np.array(nhs)
|
||||
return np.array(branch_lengths_s), np.array(
|
||||
branch_lengths_hs), np.array(dist), np.array(ns), np.array(nhs)
|
||||
|
|
|
|||
Loading…
Reference in a new issue