mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
Add files via upload
This commit is contained in:
parent
29bd784782
commit
a58c842610
16 changed files with 971 additions and 821 deletions
|
|
@ -3,14 +3,19 @@ import requests
|
||||||
import progressbar
|
import progressbar
|
||||||
import sys
|
import sys
|
||||||
|
|
||||||
|
|
||||||
def update_protein(gene_seq, gene):
|
def update_protein(gene_seq, gene):
|
||||||
t = 0
|
t = 0
|
||||||
while(t != 2):
|
while(t != 2):
|
||||||
try:
|
try:
|
||||||
server = "https://rest.ensembl.org"
|
server = "https://rest.ensembl.org"
|
||||||
ext = "/sequence/id/"+str(gene)+"?type=protein;multiple_sequences=1"
|
ext = "/sequence/id/" + \
|
||||||
|
str(gene) + "?type=protein;multiple_sequences=1"
|
||||||
|
|
||||||
r = requests.get(server+ext, headers={ "Content-Type" : "application/json"})
|
r = requests.get(
|
||||||
|
server + ext,
|
||||||
|
headers={
|
||||||
|
"Content-Type": "application/json"})
|
||||||
|
|
||||||
if not r.ok:
|
if not r.ok:
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
|
|
@ -31,12 +36,13 @@ def update_protein(gene_seq,gene):
|
||||||
r = dict(r[maxi])
|
r = dict(r[maxi])
|
||||||
gene_seq[gene] = str(r["seq"])
|
gene_seq[gene] = str(r["seq"])
|
||||||
return
|
return
|
||||||
except :
|
except BaseException:
|
||||||
t += 1
|
t += 1
|
||||||
# print("\nError:",e)
|
# print("\nError:",e)
|
||||||
continue
|
continue
|
||||||
gene_seq[gene] = ""
|
gene_seq[gene] = ""
|
||||||
|
|
||||||
|
|
||||||
def update_rest_protein(data):
|
def update_rest_protein(data):
|
||||||
gids = {}
|
gids = {}
|
||||||
with open("processed/not_found.json", "r") as file:
|
with open("processed/not_found.json", "r") as file:
|
||||||
|
|
@ -48,13 +54,19 @@ def update_rest_protein(data):
|
||||||
|
|
||||||
server = "https://rest.ensembl.org"
|
server = "https://rest.ensembl.org"
|
||||||
ext = "/sequence/id?type=protein"
|
ext = "/sequence/id?type=protein"
|
||||||
headers={ "Content-Type" : "application/json", "Accept" : "application/json"}
|
headers = {
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"Accept": "application/json"}
|
||||||
|
|
||||||
for i in progressbar.progressbar(range(0, len(gids) - 50, 50)):
|
for i in progressbar.progressbar(range(0, len(gids) - 50, 50)):
|
||||||
ids = dict(ids=list(gids[i:i + 50]))
|
ids = dict(ids=list(gids[i:i + 50]))
|
||||||
while(1):
|
while(1):
|
||||||
try:
|
try:
|
||||||
r = requests.post(server+ext, headers=headers, data=str(json.dumps(ids)))
|
r = requests.post(
|
||||||
|
server + ext,
|
||||||
|
headers=headers,
|
||||||
|
data=str(
|
||||||
|
json.dumps(ids)))
|
||||||
if not r.ok:
|
if not r.ok:
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
gs = r.json()
|
gs = r.json()
|
||||||
|
|
@ -71,21 +83,26 @@ def update_rest_protein(data):
|
||||||
for genes in gids:
|
for genes in gids:
|
||||||
try:
|
try:
|
||||||
_ = data[genes]
|
_ = data[genes]
|
||||||
except:
|
except BaseException:
|
||||||
print(genes)
|
print(genes)
|
||||||
update_protein(data, genes)
|
update_protein(data, genes)
|
||||||
|
|
||||||
print("Gene Sequences Updated Successfully")
|
print("Gene Sequences Updated Successfully")
|
||||||
return data
|
return data
|
||||||
|
|
||||||
|
|
||||||
def update(gene_seq, gene):
|
def update(gene_seq, gene):
|
||||||
t = 0
|
t = 0
|
||||||
while(t != 2):
|
while(t != 2):
|
||||||
try:
|
try:
|
||||||
server = "https://rest.ensembl.org"
|
server = "https://rest.ensembl.org"
|
||||||
ext = "/sequence/id/"+str(gene)+"?type=cds;multiple_sequences=1"
|
ext = "/sequence/id/" + \
|
||||||
|
str(gene) + "?type=cds;multiple_sequences=1"
|
||||||
|
|
||||||
r = requests.get(server+ext, headers={ "Content-Type" : "application/json"})
|
r = requests.get(
|
||||||
|
server + ext,
|
||||||
|
headers={
|
||||||
|
"Content-Type": "application/json"})
|
||||||
|
|
||||||
if not r.ok:
|
if not r.ok:
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
|
|
@ -106,12 +123,13 @@ def update(gene_seq,gene):
|
||||||
r = dict(r[maxi])
|
r = dict(r[maxi])
|
||||||
gene_seq[gene] = str(r["seq"])
|
gene_seq[gene] = str(r["seq"])
|
||||||
return
|
return
|
||||||
except Exception as e:
|
except BaseException:
|
||||||
t += 1
|
t += 1
|
||||||
# print("\nError:",e)
|
# print("\nError:",e)
|
||||||
continue
|
continue
|
||||||
gene_seq[gene] = ""
|
gene_seq[gene] = ""
|
||||||
|
|
||||||
|
|
||||||
def update_rest(data, fname):
|
def update_rest(data, fname):
|
||||||
gids = {}
|
gids = {}
|
||||||
with open("processed/not_found_" + fname + ".json", "r") as file:
|
with open("processed/not_found_" + fname + ".json", "r") as file:
|
||||||
|
|
@ -123,13 +141,19 @@ def update_rest(data,fname):
|
||||||
|
|
||||||
server = "https://rest.ensembl.org"
|
server = "https://rest.ensembl.org"
|
||||||
ext = "/sequence/id?type=cds"
|
ext = "/sequence/id?type=cds"
|
||||||
headers={ "Content-Type" : "application/json", "Accept" : "application/json"}
|
headers = {
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"Accept": "application/json"}
|
||||||
|
|
||||||
for i in progressbar.progressbar(range(0, len(gids) - 50, 50)):
|
for i in progressbar.progressbar(range(0, len(gids) - 50, 50)):
|
||||||
ids = dict(ids=list(gids[i:i + 50]))
|
ids = dict(ids=list(gids[i:i + 50]))
|
||||||
while(1):
|
while(1):
|
||||||
try:
|
try:
|
||||||
r = requests.post(server+ext, headers=headers, data=str(json.dumps(ids)))
|
r = requests.post(
|
||||||
|
server + ext,
|
||||||
|
headers=headers,
|
||||||
|
data=str(
|
||||||
|
json.dumps(ids)))
|
||||||
if not r.ok:
|
if not r.ok:
|
||||||
r.raise_for_status()
|
r.raise_for_status()
|
||||||
gs = r.json()
|
gs = r.json()
|
||||||
|
|
@ -146,7 +170,7 @@ def update_rest(data,fname):
|
||||||
for genes in gids:
|
for genes in gids:
|
||||||
try:
|
try:
|
||||||
_ = data[genes]
|
_ = data[genes]
|
||||||
except:
|
except BaseException:
|
||||||
print(genes)
|
print(genes)
|
||||||
update(data, genes)
|
update(data, genes)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,3 @@
|
||||||
import pandas as pd
|
|
||||||
import requests
|
|
||||||
import sys
|
|
||||||
import pickle
|
import pickle
|
||||||
from get_data import get_data_genome
|
from get_data import get_data_genome
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,4 @@
|
||||||
import pandas as pd
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import json
|
|
||||||
import gc
|
|
||||||
import pickle
|
import pickle
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
|
@ -9,6 +6,7 @@ from tree_data import create_tree_data
|
||||||
from process_negative import read_database_txt
|
from process_negative import read_database_txt
|
||||||
from select_data import read_db_homology
|
from select_data import read_db_homology
|
||||||
|
|
||||||
|
|
||||||
def read_data_homology(dirname, nfname):
|
def read_data_homology(dirname, nfname):
|
||||||
lf = os.listdir(dirname)
|
lf = os.listdir(dirname)
|
||||||
if len(lf) == 0:
|
if len(lf) == 0:
|
||||||
|
|
@ -20,20 +18,27 @@ def read_data_homology(dirname,nfname):
|
||||||
df, n = read_db_homology(dirname, x)
|
df, n = read_db_homology(dirname, x)
|
||||||
n = n.split()[0]
|
n = n.split()[0]
|
||||||
try:
|
try:
|
||||||
indexes=np.load("processed/synteny_matrices/"+n+"_indexes.npy")
|
indexes = np.load(
|
||||||
except:
|
"processed/synteny_matrices/" +
|
||||||
|
n +
|
||||||
|
"_indexes.npy")
|
||||||
|
except BaseException:
|
||||||
print("Incomplete data for:", n)
|
print("Incomplete data for:", n)
|
||||||
df = df.loc[indexes]
|
df = df.loc[indexes]
|
||||||
a_h.append(df)
|
a_h.append(df)
|
||||||
d_h.append(n)
|
d_h.append(n)
|
||||||
# read the negative dataset
|
# read the negative dataset
|
||||||
df = read_database_txt(nfname)
|
df = read_database_txt(nfname)
|
||||||
indexes=np.load("processed/synteny_matrices/"+nfname.split(".")[0]+"_indexes.npy")
|
indexes = np.load(
|
||||||
|
"processed/synteny_matrices/" +
|
||||||
|
nfname.split(".")[0] +
|
||||||
|
"_indexes.npy")
|
||||||
df = df.loc[indexes]
|
df = df.loc[indexes]
|
||||||
a_h.append(df)
|
a_h.append(df)
|
||||||
d_h.append(nfname.split(".")[0])
|
d_h.append(nfname.split(".")[0])
|
||||||
return a_h, d_h
|
return a_h, d_h
|
||||||
|
|
||||||
|
|
||||||
def prepare_features(a_h, d_h, sptree, label):
|
def prepare_features(a_h, d_h, sptree, label):
|
||||||
rows = []
|
rows = []
|
||||||
smg_name = "_synteny_matrices_global.npy"
|
smg_name = "_synteny_matrices_global.npy"
|
||||||
|
|
@ -47,12 +52,15 @@ def prepare_features(a_h,d_h,sptree,label):
|
||||||
smg = np.load(dir_name + n + smg_name)
|
smg = np.load(dir_name + n + smg_name)
|
||||||
sml = np.load(dir_name + n + sml_name)
|
sml = np.load(dir_name + n + sml_name)
|
||||||
indexes = np.load(dir_name + n + smi_name)
|
indexes = np.load(dir_name + n + smi_name)
|
||||||
except:
|
except BaseException:
|
||||||
print("Incomplete data for:", n)
|
print("Incomplete data for:", n)
|
||||||
continue
|
continue
|
||||||
df = df.loc[indexes]
|
df = df.loc[indexes]
|
||||||
|
|
||||||
branch_length_species,branch_length_homology_species,distance,dist_p_s,dist_p_hs=create_tree_data(sptree,df)
|
branch_length_species, \
|
||||||
|
branch_length_homology_species, \
|
||||||
|
distance, dist_p_s, dist_p_hs = create_tree_data(
|
||||||
|
sptree, df)
|
||||||
assert(len(branch_length_species) == len(df))
|
assert(len(branch_length_species) == len(df))
|
||||||
assert(len(sml) == len(distance))
|
assert(len(sml) == len(distance))
|
||||||
|
|
||||||
|
|
@ -76,6 +84,7 @@ def prepare_features(a_h,d_h,sptree,label):
|
||||||
rows.append(r)
|
rows.append(r)
|
||||||
return rows
|
return rows
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
arg = sys.argv
|
arg = sys.argv
|
||||||
nfname = arg[-1]
|
nfname = arg[-1]
|
||||||
|
|
@ -92,5 +101,6 @@ def main():
|
||||||
pickle.dump(rows, file)
|
pickle.dump(rows, file)
|
||||||
print("Dataset_Finalized")
|
print("Dataset_Finalized")
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
|
|
|
||||||
12
get_data.py
12
get_data.py
|
|
@ -1,8 +1,7 @@
|
||||||
import sys
|
from read_data import read_data_genome
|
||||||
import os
|
|
||||||
from read_data import read_data_genome,read_data_homology
|
|
||||||
from process_data import list_dict_genomes, create_chromosome_maps
|
from process_data import list_dict_genomes, create_chromosome_maps
|
||||||
|
|
||||||
|
|
||||||
def get_data_genome(dir):
|
def get_data_genome(dir):
|
||||||
a = []
|
a = []
|
||||||
d = {}
|
d = {}
|
||||||
|
|
@ -17,10 +16,3 @@ def get_data_genome(dir):
|
||||||
for i in range(len(ld)):
|
for i in range(len(ld)):
|
||||||
assert(len(ld[i]) == len(ldg[i]))
|
assert(len(ld[i]) == len(ldg[i]))
|
||||||
return cmap, cimap, ld, ldg, a, d
|
return cmap, cimap, ld, ldg, a, d
|
||||||
|
|
||||||
def get_data_homology(dir):
|
|
||||||
a_h=[]
|
|
||||||
d_h={}
|
|
||||||
a_h,d_h=read_data_homology(dir)
|
|
||||||
assert(len(a_h)==len(d_h))
|
|
||||||
return a_h,d_h
|
|
||||||
|
|
|
||||||
|
|
@ -1,13 +1,11 @@
|
||||||
import sys
|
import sys
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import pandas as pd
|
|
||||||
import json
|
|
||||||
import os
|
import os
|
||||||
import gc
|
|
||||||
import pickle
|
import pickle
|
||||||
from select_data import read_db_homology
|
from select_data import read_db_homology
|
||||||
from process_data import create_data_homology_ls
|
from process_data import create_data_homology_ls
|
||||||
|
|
||||||
|
|
||||||
def read_genome_maps():
|
def read_genome_maps():
|
||||||
data = {}
|
data = {}
|
||||||
with open("genome_maps", "rb") as file:
|
with open("genome_maps", "rb") as file:
|
||||||
|
|
@ -20,6 +18,7 @@ def read_genome_maps():
|
||||||
d = data["d"]
|
d = data["d"]
|
||||||
return a, d, ld, ldg, cmap, cimap
|
return a, d, ld, ldg, cmap, cimap
|
||||||
|
|
||||||
|
|
||||||
def read_data_homology(dirname):
|
def read_data_homology(dirname):
|
||||||
lf = os.listdir(dirname)
|
lf = os.listdir(dirname)
|
||||||
if len(lf) == 0:
|
if len(lf) == 0:
|
||||||
|
|
@ -32,14 +31,14 @@ def read_data_homology(dirname):
|
||||||
n = n.split()[0]
|
n = n.split()[0]
|
||||||
try:
|
try:
|
||||||
indexes = np.load("processed/" + n + "_selected_indexes.npy")
|
indexes = np.load("processed/" + n + "_selected_indexes.npy")
|
||||||
except:
|
except BaseException:
|
||||||
print("Incomplete data for:", n)
|
print("Incomplete data for:", n)
|
||||||
df = df.loc[indexes]
|
df = df.loc[indexes]
|
||||||
print(len(df))
|
|
||||||
a_h.append(df)
|
a_h.append(df)
|
||||||
d_h.append(n)
|
d_h.append(n)
|
||||||
return a_h, d_h
|
return a_h, d_h
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
a, d, ld, ldg, cmap, cimap = read_genome_maps()
|
a, d, ld, ldg, cmap, cimap = read_genome_maps()
|
||||||
print("Genome Maps Loaded.")
|
print("Genome Maps Loaded.")
|
||||||
|
|
@ -49,5 +48,6 @@ def main():
|
||||||
_ = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 1)
|
_ = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 1)
|
||||||
print("Neighbor Genes Found and Saved Successfully:)")
|
print("Neighbor Genes Found and Saved Successfully:)")
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
|
|
@ -1,8 +1,5 @@
|
||||||
import json
|
|
||||||
import gc
|
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import pickle
|
|
||||||
import tensorflow as tf
|
import tensorflow as tf
|
||||||
import sys
|
import sys
|
||||||
import progressbar
|
import progressbar
|
||||||
|
|
@ -15,6 +12,7 @@ from prepare_synteny_matrix import read_data_synteny
|
||||||
from tree_data import create_tree_data
|
from tree_data import create_tree_data
|
||||||
from process_data import create_map_list
|
from process_data import create_map_list
|
||||||
|
|
||||||
|
|
||||||
def read_database(fname):
|
def read_database(fname):
|
||||||
df = pd.read_csv(fname, sep="\t", header=None)
|
df = pd.read_csv(fname, sep="\t", header=None)
|
||||||
label_dict = dict(ortholog_one2one=1,
|
label_dict = dict(ortholog_one2one=1,
|
||||||
|
|
@ -30,9 +28,17 @@ def read_database(fname):
|
||||||
df = df.assign(label=label)
|
df = df.assign(label=label)
|
||||||
df = df.drop(7, axis=1)
|
df = df.drop(7, axis=1)
|
||||||
df = df.drop(0, axis=1)
|
df = df.drop(0, axis=1)
|
||||||
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","goc","wga","label"]
|
df.columns = [
|
||||||
|
"gene_stable_id",
|
||||||
|
"species",
|
||||||
|
"homology_gene_stable_id",
|
||||||
|
"homology_species",
|
||||||
|
"goc",
|
||||||
|
"wga",
|
||||||
|
"label"]
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def select_data_by_length(df, st, end):
|
def select_data_by_length(df, st, end):
|
||||||
try:
|
try:
|
||||||
if end < len(df):
|
if end < len(df):
|
||||||
|
|
@ -40,18 +46,21 @@ def select_data_by_length(df,st,end):
|
||||||
df = df.loc[df.index.values[st:end]]
|
df = df.loc[df.index.values[st:end]]
|
||||||
else:
|
else:
|
||||||
raise ValueError()
|
raise ValueError()
|
||||||
except:
|
except BaseException:
|
||||||
print("Making Predictions for the complete dataframe:)")
|
print("Making Predictions for the complete dataframe:)")
|
||||||
print(len(df))
|
print(len(df))
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def create_synteny_features(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, name):
|
def create_synteny_features(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, name):
|
||||||
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
||||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","prediction_"+name)
|
gene_sequences = read_gene_sequences(
|
||||||
|
a_h, lsy, "geneseq", "prediction_" + name)
|
||||||
gene_sequences = update_rest(gene_sequences, "prediction_" + name)
|
gene_sequences = update_rest(gene_sequences, "prediction_" + name)
|
||||||
print("Gene Sequences Loaded.")
|
print("Gene Sequences Loaded.")
|
||||||
return lsy, gene_sequences
|
return lsy, gene_sequences
|
||||||
|
|
||||||
|
|
||||||
def threadmaker(nop, df, lsy, gene_sequences, n, name):
|
def threadmaker(nop, df, lsy, gene_sequences, n, name):
|
||||||
part = len(df) // nop
|
part = len(df) // nop
|
||||||
pr = Procerssrunner()
|
pr = Procerssrunner()
|
||||||
|
|
@ -62,23 +71,27 @@ def threadmaker(nop,df,lsy,gene_sequences,n,name):
|
||||||
indexes = np.array(indexes)
|
indexes = np.array(indexes)
|
||||||
return sml, smg, indexes
|
return sml, smg, indexes
|
||||||
|
|
||||||
|
|
||||||
def get_prediction(smg, sml, indexes, bls, blhs, dis, dps, dphs, model_name):
|
def get_prediction(smg, sml, indexes, bls, blhs, dis, dps, dphs, model_name):
|
||||||
preds = np.zeros((len(smg), 3))
|
preds = np.zeros((len(smg), 3))
|
||||||
w = [0.86, 0.8, 0.06]
|
w = [0.86, 0.8, 0.06]
|
||||||
for i in range(1, 4):
|
for i in range(1, 4):
|
||||||
try:
|
try:
|
||||||
model=tf.train.import_meta_graph(model_name+'_v'+str(i)+'/model.ckpt.meta')
|
model = tf.train.import_meta_graph(
|
||||||
except:
|
model_name + '_v' + str(i) + '/model.ckpt.meta')
|
||||||
|
except BaseException:
|
||||||
print("Something wrong with the model.")
|
print("Something wrong with the model.")
|
||||||
continue
|
continue
|
||||||
with tf.Session() as sess:
|
with tf.Session() as sess:
|
||||||
try:
|
try:
|
||||||
model.restore(sess, model_name + '_v' + str(i) + "/model.ckpt")
|
model.restore(sess, model_name + '_v' + str(i) + "/model.ckpt")
|
||||||
graph = tf.get_default_graph()
|
graph = tf.get_default_graph()
|
||||||
synmgt,synmlt,blst,blhst,dpst,dphst,dist,lrt,yt=graph.get_collection("input_nodes")
|
synmgt, synmlt, blst, \
|
||||||
|
blhst, dpst, dphst, \
|
||||||
|
dist, lrt, yt = graph.get_collection("input_nodes")
|
||||||
predictions = graph.get_tensor_by_name("Predictions/BiasAdd:0")
|
predictions = graph.get_tensor_by_name("Predictions/BiasAdd:0")
|
||||||
print("Model Loaded Successfully :)")
|
print("Model Loaded Successfully :)")
|
||||||
except:
|
except BaseException:
|
||||||
print(":(")
|
print(":(")
|
||||||
sys.exit()
|
sys.exit()
|
||||||
|
|
||||||
|
|
@ -106,8 +119,17 @@ def get_prediction(smg,sml,indexes,bls,blhs,dis,dps,dphs,model_name):
|
||||||
print(preds.shape)
|
print(preds.shape)
|
||||||
return preds
|
return preds
|
||||||
|
|
||||||
|
|
||||||
def write_preds(fname, model_name, name, preds, index_dict, df):
|
def write_preds(fname, model_name, name, preds, index_dict, df):
|
||||||
print("Writing predcitions to:","prediction_"+fname+"_"+model_name+"_"+name+"_multiple.txt")
|
print(
|
||||||
|
"Writing predcitions to:",
|
||||||
|
"prediction_" +
|
||||||
|
fname +
|
||||||
|
"_" +
|
||||||
|
model_name +
|
||||||
|
"_" +
|
||||||
|
name +
|
||||||
|
"_multiple.txt")
|
||||||
with open("prediction_" + fname + "_" + model_name + "_" + name + "_multiple.txt", "w") as file:
|
with open("prediction_" + fname + "_" + model_name + "_" + name + "_multiple.txt", "w") as file:
|
||||||
for index, row in progressbar.progressbar(df.iterrows()):
|
for index, row in progressbar.progressbar(df.iterrows()):
|
||||||
file.write(str(row[0]))
|
file.write(str(row[0]))
|
||||||
|
|
@ -131,6 +153,7 @@ def write_preds(fname,model_name,name,preds,index_dict,df):
|
||||||
file.write("NaN")
|
file.write("NaN")
|
||||||
file.write("\n")
|
file.write("\n")
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
arg = sys.argv
|
arg = sys.argv
|
||||||
fname = arg[-6]
|
fname = arg[-6]
|
||||||
|
|
@ -148,14 +171,25 @@ def main():
|
||||||
a_h = [df]
|
a_h = [df]
|
||||||
d_h = ["prediction"]
|
d_h = ["prediction"]
|
||||||
|
|
||||||
lsy,gene_sequences=create_synteny_features(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,name)
|
lsy, gene_sequences = create_synteny_features(
|
||||||
|
a_h, d_h, n, a, d, ld, ldg, cmap, cimap, name)
|
||||||
sml, smg, indexes = threadmaker(nop, df, lsy, gene_sequences, n, name)
|
sml, smg, indexes = threadmaker(nop, df, lsy, gene_sequences, n, name)
|
||||||
df_temp = df.loc[indexes]
|
df_temp = df.loc[indexes]
|
||||||
bls, blhs, dis, dps, dphs = create_tree_data("species_tree.tree", df_temp)
|
bls, blhs, dis, dps, dphs = create_tree_data("species_tree.tree", df_temp)
|
||||||
index_dict = create_map_list(indexes)
|
index_dict = create_map_list(indexes)
|
||||||
preds=get_prediction(smg,sml,indexes,bls,blhs,dis,dps,dphs,model_name)
|
preds = get_prediction(
|
||||||
|
smg,
|
||||||
|
sml,
|
||||||
|
indexes,
|
||||||
|
bls,
|
||||||
|
blhs,
|
||||||
|
dis,
|
||||||
|
dps,
|
||||||
|
dphs,
|
||||||
|
model_name)
|
||||||
|
|
||||||
write_preds(fname, model_name, name, preds, index_dict, df)
|
write_preds(fname, model_name, name, preds, index_dict, df)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
|
|
@ -1,5 +1,4 @@
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import pandas as pd
|
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
|
@ -9,6 +8,7 @@ from threads import Procerssrunner
|
||||||
from read_get_gene_seq import read_gene_sequences
|
from read_get_gene_seq import read_gene_sequences
|
||||||
from access_data_rest import update_rest
|
from access_data_rest import update_rest
|
||||||
|
|
||||||
|
|
||||||
def read_data_synteny(nop, name):
|
def read_data_synteny(nop, name):
|
||||||
smg = []
|
smg = []
|
||||||
sml = []
|
sml = []
|
||||||
|
|
@ -27,6 +27,7 @@ def read_data_synteny(nop,name):
|
||||||
print(len(indexes))
|
print(len(indexes))
|
||||||
return smg, sml, indexes
|
return smg, sml, indexes
|
||||||
|
|
||||||
|
|
||||||
def load_neighbor_genes():
|
def load_neighbor_genes():
|
||||||
with open("processed/neighbor_genes.json", "r") as file:
|
with open("processed/neighbor_genes.json", "r") as file:
|
||||||
lsy = dict(json.load(file))
|
lsy = dict(json.load(file))
|
||||||
|
|
@ -34,6 +35,7 @@ def load_neighbor_genes():
|
||||||
print("Neighbor Genes Loaded")
|
print("Neighbor Genes Loaded")
|
||||||
return lsy
|
return lsy
|
||||||
|
|
||||||
|
|
||||||
def read_data_homology(dirname):
|
def read_data_homology(dirname):
|
||||||
lf = os.listdir(dirname)
|
lf = os.listdir(dirname)
|
||||||
if len(lf) == 0:
|
if len(lf) == 0:
|
||||||
|
|
@ -46,13 +48,14 @@ def read_data_homology(dirname):
|
||||||
n = n.split()[0]
|
n = n.split()[0]
|
||||||
try:
|
try:
|
||||||
indexes = np.load("processed/" + n + "_selected_indexes.npy")
|
indexes = np.load("processed/" + n + "_selected_indexes.npy")
|
||||||
except:
|
except BaseException:
|
||||||
print("Incomplete data for:", n)
|
print("Incomplete data for:", n)
|
||||||
df = df.loc[indexes]
|
df = df.loc[indexes]
|
||||||
a_h.append(df)
|
a_h.append(df)
|
||||||
d_h.append(n)
|
d_h.append(n)
|
||||||
return a_h, d_h
|
return a_h, d_h
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
arg = sys.argv
|
arg = sys.argv
|
||||||
nop = int(arg[-1])
|
nop = int(arg[-1])
|
||||||
|
|
@ -60,7 +63,8 @@ def main():
|
||||||
a_h, d_h = read_data_homology("data_homology")
|
a_h, d_h = read_data_homology("data_homology")
|
||||||
print("Data Read")
|
print("Data Read")
|
||||||
lsy = load_neighbor_genes()
|
lsy = load_neighbor_genes()
|
||||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","gene_seq_positive")
|
gene_sequences = read_gene_sequences(
|
||||||
|
a_h, lsy, "geneseq", "gene_seq_positive")
|
||||||
gene_sequences = update_rest(gene_sequences, "gene_seq_positive")
|
gene_sequences = update_rest(gene_sequences, "gene_seq_positive")
|
||||||
print("Gene Sequences Loaded.")
|
print("Gene Sequences Loaded.")
|
||||||
if not os.path.isdir("processed/synteny_matrices"):
|
if not os.path.isdir("processed/synteny_matrices"):
|
||||||
|
|
@ -81,5 +85,6 @@ def main():
|
||||||
np.save(ndir + str(d_h[i]) + "_" + nf3, indexes)
|
np.save(ndir + str(d_h[i]) + "_" + nf3, indexes)
|
||||||
print("Synteny Matrices Created Successfully :)")
|
print("Synteny Matrices Created Successfully :)")
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
|
|
|
||||||
|
|
@ -1,17 +1,15 @@
|
||||||
import pandas
|
|
||||||
import gc
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import json
|
|
||||||
import os
|
|
||||||
import progressbar
|
import progressbar
|
||||||
from save_data import write_dict_json
|
from save_data import write_dict_json
|
||||||
|
|
||||||
|
|
||||||
def create_map_list(l): # this function maps the indexes to values
|
def create_map_list(l): # this function maps the indexes to values
|
||||||
t = {}
|
t = {}
|
||||||
for i in range(len(l)):
|
for i in range(len(l)):
|
||||||
t[l[i]] = i
|
t[l[i]] = i
|
||||||
return t
|
return t
|
||||||
|
|
||||||
|
|
||||||
def create_chromosome_maps(a, n):
|
def create_chromosome_maps(a, n):
|
||||||
cmap = []
|
cmap = []
|
||||||
cimap = []
|
cimap = []
|
||||||
|
|
@ -21,8 +19,8 @@ def create_chromosome_maps(a,n):
|
||||||
for index, row in df.iterrows():
|
for index, row in df.iterrows():
|
||||||
g = row.gene_id
|
g = row.gene_id
|
||||||
try:
|
try:
|
||||||
temp=chmap[g]
|
_ = chmap[g]
|
||||||
except:
|
except BaseException:
|
||||||
chmap[g] = str(row.Chr)
|
chmap[g] = str(row.Chr)
|
||||||
if str(row.Chr) in chindmap:
|
if str(row.Chr) in chindmap:
|
||||||
chindmap[str(row.Chr)].append(index)
|
chindmap[str(row.Chr)].append(index)
|
||||||
|
|
@ -33,6 +31,7 @@ def create_chromosome_maps(a,n):
|
||||||
cimap.append(chindmap)
|
cimap.append(chindmap)
|
||||||
return cmap, cimap
|
return cmap, cimap
|
||||||
|
|
||||||
|
|
||||||
def list_dict_genomes(a, n):
|
def list_dict_genomes(a, n):
|
||||||
lst = []
|
lst = []
|
||||||
ldt = []
|
ldt = []
|
||||||
|
|
@ -45,17 +44,20 @@ def list_dict_genomes(a,n):
|
||||||
ldt.append(ldgt)
|
ldt.append(ldgt)
|
||||||
return lst, ldt
|
return lst, ldt
|
||||||
|
|
||||||
|
|
||||||
def get_nearest_neighbors(g, gs, n, a, d, ld, ldg, cmap, cimap):
|
def get_nearest_neighbors(g, gs, n, a, d, ld, ldg, cmap, cimap):
|
||||||
# print("Finding Neighbor Genes")
|
# print("Finding Neighbor Genes")
|
||||||
ne = [] # list to store the backward genes
|
ne = [] # list to store the backward genes
|
||||||
nr = [] # list to store the forward genes
|
nr = [] # list to store the forward genes
|
||||||
gi=d[gs.capitalize()] #get the address of the corresponding species to which the gene belongs whose neighbor has to be found
|
# get the address of the corresponding species to which the gene belongs
|
||||||
|
# whose neighbor has to be found
|
||||||
|
gi = d[gs.capitalize()]
|
||||||
sldf = a[gi] # select the dataframe
|
sldf = a[gi] # select the dataframe
|
||||||
scmap = cmap[gi] # select the correct chromosome map
|
scmap = cmap[gi] # select the correct chromosome map
|
||||||
scimap = cimap[gi] # select the correct index maps
|
scimap = cimap[gi] # select the correct index maps
|
||||||
try:
|
try:
|
||||||
sld=ld[gi]#see if the corresponding gene map exists
|
_ = ld[gi] # see if the corresponding gene map exists
|
||||||
except:
|
except BaseException:
|
||||||
# print("Length of Dataframes:{} \t Length of Loaded Genes:{} \t Length of Loaded Genomes Dictionaries:{}".format(len(a),len(ld),len(ldg)))
|
# print("Length of Dataframes:{} \t Length of Loaded Genes:{} \t Length of Loaded Genomes Dictionaries:{}".format(len(a),len(ld),len(ldg)))
|
||||||
return ne, nr
|
return ne, nr
|
||||||
sldg = ldg[gi] # select the corresponding map
|
sldg = ldg[gi] # select the corresponding map
|
||||||
|
|
@ -85,12 +87,15 @@ def get_nearest_neighbors(g,gs,n,a,d,ld,ldg,cmap,cimap):
|
||||||
ne.append("NULL_GENE") # append the NULL_GENE value
|
ne.append("NULL_GENE") # append the NULL_GENE value
|
||||||
continue
|
continue
|
||||||
for k in end_s: # iterate through the sorted array
|
for k in end_s: # iterate through the sorted array
|
||||||
if end[k]<0 and end[k+1]>=0:#find the first value that is negative and the next one is positive to get the nearest gene
|
# find the first value that is negative and the next one is
|
||||||
|
# positive to get the nearest gene
|
||||||
|
if end[k] < 0 and end[k + 1] >= 0:
|
||||||
itemp = k
|
itemp = k
|
||||||
break
|
break
|
||||||
itemp = scimap[itemp]
|
itemp = scimap[itemp]
|
||||||
ne.append(sldf.loc[itemp].gene_id)
|
ne.append(sldf.loc[itemp].gene_id)
|
||||||
start=int(sldf.loc[itemp].start)#make "start" the start location of the current gene
|
# make "start" the start location of the current gene
|
||||||
|
start = int(sldf.loc[itemp].start)
|
||||||
# print(start)
|
# print(start)
|
||||||
# get the +n neighbors
|
# get the +n neighbors
|
||||||
flag = 0
|
flag = 0
|
||||||
|
|
@ -117,8 +122,11 @@ def get_nearest_neighbors(g,gs,n,a,d,ld,ldg,cmap,cimap):
|
||||||
end = int(sldf.loc[itemp].end)
|
end = int(sldf.loc[itemp].end)
|
||||||
return ne, nr
|
return ne, nr
|
||||||
|
|
||||||
|
|
||||||
def create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, to_write):
|
def create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, to_write):
|
||||||
lsy={} #dictionary which stores +/- n genes of the given gene by id. Each key is a gene id which corresponds to the one in center.
|
# dictionary which stores +/- n genes of the given gene by id. Each key is
|
||||||
|
# a gene id which corresponds to the one in center.
|
||||||
|
lsy = {}
|
||||||
lsytemp = {}
|
lsytemp = {}
|
||||||
name = "neighbor_genes"
|
name = "neighbor_genes"
|
||||||
for df in a_h:
|
for df in a_h:
|
||||||
|
|
@ -128,26 +136,30 @@ def create_data_homology_ls(a_h,d_h,n,a,d,ld,ldg,cmap,cimap,to_write):
|
||||||
xs = row["species"]
|
xs = row["species"]
|
||||||
ys = row["homology_species"]
|
ys = row["homology_species"]
|
||||||
try:
|
try:
|
||||||
z=lsy[x]
|
_ = lsy[x]
|
||||||
except:
|
except BaseException:
|
||||||
try:
|
try:
|
||||||
t2=d[xs.capitalize()]#see if the species exist in genomic maps
|
# see if the species exist in genomic maps
|
||||||
xl,xr=get_nearest_neighbors(x,xs,n,a,d,ld,ldg,cmap,cimap)
|
_ = d[xs.capitalize()]
|
||||||
if len(xl)!=0:#check if neighboring genes were successfully found
|
xl, xr = get_nearest_neighbors(
|
||||||
|
x, xs, n, a, d, ld, ldg, cmap, cimap)
|
||||||
|
if len(
|
||||||
|
xl) != 0: # check if neighboring genes were successfully found
|
||||||
lsy[x] = dict(b=xl, f=xr)
|
lsy[x] = dict(b=xl, f=xr)
|
||||||
lsytemp[x] = dict(b=xl, f=xr)
|
lsytemp[x] = dict(b=xl, f=xr)
|
||||||
except:
|
except BaseException:
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
z=lsy[y]
|
_ = lsy[y]
|
||||||
except:
|
except BaseException:
|
||||||
try:
|
try:
|
||||||
t2=d[ys.capitalize()]
|
_ = d[ys.capitalize()]
|
||||||
yl,yr=get_nearest_neighbors(y,ys,n,a,d,ld,ldg,cmap,cimap)
|
yl, yr = get_nearest_neighbors(
|
||||||
|
y, ys, n, a, d, ld, ldg, cmap, cimap)
|
||||||
if len(yl) != 0:
|
if len(yl) != 0:
|
||||||
lsy[y] = dict(b=yl, f=yr)
|
lsy[y] = dict(b=yl, f=yr)
|
||||||
lsytemp[y] = dict(b=yl, f=yr)
|
lsytemp[y] = dict(b=yl, f=yr)
|
||||||
except:
|
except BaseException:
|
||||||
continue
|
continue
|
||||||
if to_write == 1:
|
if to_write == 1:
|
||||||
write_dict_json(name, "processed", lsy)
|
write_dict_json(name, "processed", lsy)
|
||||||
|
|
|
||||||
|
|
@ -1,9 +1,5 @@
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import os
|
|
||||||
import sys
|
|
||||||
import progressbar
|
|
||||||
import json
|
|
||||||
import sys
|
import sys
|
||||||
from neighbor_genes import read_genome_maps
|
from neighbor_genes import read_genome_maps
|
||||||
from process_data import create_data_homology_ls
|
from process_data import create_data_homology_ls
|
||||||
|
|
@ -13,12 +9,21 @@ from access_data_rest import update_rest
|
||||||
from prepare_synteny_matrix import read_data_synteny
|
from prepare_synteny_matrix import read_data_synteny
|
||||||
from save_data import write_dict_json
|
from save_data import write_dict_json
|
||||||
|
|
||||||
|
|
||||||
def read_database_txt(filename):
|
def read_database_txt(filename):
|
||||||
df = pd.read_csv(filename, sep="\t", header=None)
|
df = pd.read_csv(filename, sep="\t", header=None)
|
||||||
df = df.drop(0, axis=1)
|
df = df.drop(0, axis=1)
|
||||||
df.columns=["gene_stable_id","species","homology_gene_stable_id","homology_species","wga","goc","homology_type"]
|
df.columns = [
|
||||||
|
"gene_stable_id",
|
||||||
|
"species",
|
||||||
|
"homology_gene_stable_id",
|
||||||
|
"homology_species",
|
||||||
|
"wga",
|
||||||
|
"goc",
|
||||||
|
"homology_type"]
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
arg = sys.argv
|
arg = sys.argv
|
||||||
a, d, ld, ldg, cmap, cimap = read_genome_maps()
|
a, d, ld, ldg, cmap, cimap = read_genome_maps()
|
||||||
|
|
@ -34,7 +39,8 @@ def main():
|
||||||
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
lsy = create_data_homology_ls(a_h, d_h, n, a, d, ld, ldg, cmap, cimap, 0)
|
||||||
write_dict_json("neighbor_genes_negative", "processed", lsy)
|
write_dict_json("neighbor_genes_negative", "processed", lsy)
|
||||||
print("Neighbor Genes Found and Saved Successfully:)")
|
print("Neighbor Genes Found and Saved Successfully:)")
|
||||||
gene_sequences=read_gene_sequences(a_h,lsy,"geneseq","gene_seq_negative")
|
gene_sequences = read_gene_sequences(
|
||||||
|
a_h, lsy, "geneseq", "gene_seq_negative")
|
||||||
gene_sequences = update_rest(gene_sequences, "gene_seq_negative")
|
gene_sequences = update_rest(gene_sequences, "gene_seq_negative")
|
||||||
ndir = "processed/synteny_matrices/"
|
ndir = "processed/synteny_matrices/"
|
||||||
nf1 = "synteny_matrices_global"
|
nf1 = "synteny_matrices_global"
|
||||||
|
|
@ -52,6 +58,6 @@ def main():
|
||||||
np.save(ndir + str(d_h[i]) + "_" + nf3, indexes)
|
np.save(ndir + str(d_h[i]) + "_" + nf3, indexes)
|
||||||
print("Synteny Matrices Created Successfully :)")
|
print("Synteny Matrices Created Successfully :)")
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
|
|
||||||
|
|
|
||||||
42
read_data.py
42
read_data.py
|
|
@ -1,50 +1,72 @@
|
||||||
import os
|
import os
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import gzip
|
|
||||||
import sys
|
import sys
|
||||||
import progressbar
|
import progressbar
|
||||||
import traceback
|
import traceback
|
||||||
|
|
||||||
|
|
||||||
def clear_data(x):
|
def clear_data(x):
|
||||||
if x==None:
|
if x is None:
|
||||||
return x
|
return x
|
||||||
x = x.split()
|
x = x.split()
|
||||||
try:
|
try:
|
||||||
x = x[1]
|
x = x[1]
|
||||||
except:
|
except BaseException:
|
||||||
c=0
|
_ = 0
|
||||||
# print(x)
|
# print(x)
|
||||||
x = x[1:-1]
|
x = x[1:-1]
|
||||||
return x
|
return x
|
||||||
|
|
||||||
|
|
||||||
def read_data_genome(dir_name, a, dict_ind_genome):
|
def read_data_genome(dir_name, a, dict_ind_genome):
|
||||||
lf = os.listdir(dir_name)
|
lf = os.listdir(dir_name)
|
||||||
if len(lf) == 0:
|
if len(lf) == 0:
|
||||||
print("No files in the data directory!!!!!!")
|
print("No files in the data directory!!!!!!")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
colname=["Chr","source","feature","start","end","score","strand","frame","attribute"]
|
colname = [
|
||||||
|
"Chr",
|
||||||
|
"source",
|
||||||
|
"feature",
|
||||||
|
"start",
|
||||||
|
"end",
|
||||||
|
"score",
|
||||||
|
"strand",
|
||||||
|
"frame",
|
||||||
|
"attribute"]
|
||||||
print("Going to read data:")
|
print("Going to read data:")
|
||||||
for x in progressbar.progressbar(range(len(lf))):
|
for x in progressbar.progressbar(range(len(lf))):
|
||||||
data_gene=pd.read_csv(dir_name+"/"+lf[x],compression='gzip',sep='\t',comment='#',header=None,names=colname)
|
data_gene = pd.read_csv(
|
||||||
|
dir_name + "/" + lf[x],
|
||||||
|
compression='gzip',
|
||||||
|
sep='\t',
|
||||||
|
comment='#',
|
||||||
|
header=None,
|
||||||
|
names=colname)
|
||||||
# print(data_gene.head)
|
# print(data_gene.head)
|
||||||
data_gene = data_gene[data_gene["feature"] == "gene"]
|
data_gene = data_gene[data_gene["feature"] == "gene"]
|
||||||
tmp = data_gene["attribute"].str.split(";", expand=True)
|
tmp = data_gene["attribute"].str.split(";", expand=True)
|
||||||
tmp = tmp.iloc[:, :5]
|
tmp = tmp.iloc[:, :5]
|
||||||
data_gene[["gene_id","gene_version","gene_name","gene_source","gene_biotype"]]=tmp
|
data_gene[["gene_id", "gene_version", "gene_name",
|
||||||
|
"gene_source", "gene_biotype"]] = tmp
|
||||||
data_gene = data_gene.drop("attribute", axis=1)
|
data_gene = data_gene.drop("attribute", axis=1)
|
||||||
# print(data_gene[0:10])
|
# print(data_gene[0:10])
|
||||||
try:
|
try:
|
||||||
for y in ["gene_version","gene_name","gene_source","gene_biotype","gene_id"]:
|
for y in [
|
||||||
|
"gene_version",
|
||||||
|
"gene_name",
|
||||||
|
"gene_source",
|
||||||
|
"gene_biotype",
|
||||||
|
"gene_id"]:
|
||||||
data_gene[y] = data_gene[y].apply(clear_data)
|
data_gene[y] = data_gene[y].apply(clear_data)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
traceback.print_exc()
|
traceback.print_exc()
|
||||||
print(e)
|
print(e)
|
||||||
continue
|
continue
|
||||||
# print(data_gene[0:10])
|
# print(data_gene[0:10])
|
||||||
data_gene=data_gene[(data_gene['gene_biotype']=='protein_coding') | (data_gene['gene_source']=='protein_coding')]
|
data_gene = data_gene[(data_gene['gene_biotype'] == 'protein_coding') | (
|
||||||
|
data_gene['gene_source'] == 'protein_coding')]
|
||||||
# print(data_gene[data_gene["gene_id"]=="ENSNGAG00000000407"])
|
# print(data_gene[data_gene["gene_id"]=="ENSNGAG00000000407"])
|
||||||
a.append(data_gene)
|
a.append(data_gene)
|
||||||
n = lf[x].split(".")[0]
|
n = lf[x].split(".")[0]
|
||||||
dict_ind_genome[n] = len(a) - 1
|
dict_ind_genome[n] = len(a) - 1
|
||||||
return a, dict_ind_genome
|
return a, dict_ind_genome
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,11 +1,10 @@
|
||||||
import json
|
import json
|
||||||
from Bio import SeqIO
|
from Bio import SeqIO
|
||||||
import numpy as np
|
|
||||||
import pandas as pd
|
|
||||||
import os
|
import os
|
||||||
import gzip
|
import gzip
|
||||||
import progressbar
|
import progressbar
|
||||||
|
|
||||||
|
|
||||||
def read_from_multiple_lsy(lsyfl):
|
def read_from_multiple_lsy(lsyfl):
|
||||||
lsy = {}
|
lsy = {}
|
||||||
for f in lsyfl:
|
for f in lsyfl:
|
||||||
|
|
@ -16,7 +15,10 @@ def read_from_multiple_lsy(lsyfl):
|
||||||
lsy[t] = d[t]
|
lsy[t] = d[t]
|
||||||
return lsy
|
return lsy
|
||||||
|
|
||||||
#this function updates the given dictionary with the given keys and values list
|
# this function updates the given dictionary with the given keys and
|
||||||
|
# values list
|
||||||
|
|
||||||
|
|
||||||
def create_dict(keys, values, dictionary):
|
def create_dict(keys, values, dictionary):
|
||||||
for i in range(len(keys)):
|
for i in range(len(keys)):
|
||||||
if keys[i] not in dictionary:
|
if keys[i] not in dictionary:
|
||||||
|
|
@ -26,6 +28,8 @@ def create_dict(keys,values,dictionary):
|
||||||
|
|
||||||
# this function maps all the genes to their respective species.
|
# this function maps all the genes to their respective species.
|
||||||
# (Function: when finding the species of any gene we do not need to search the entire dataframe)
|
# (Function: when finding the species of any gene we do not need to search the entire dataframe)
|
||||||
|
|
||||||
|
|
||||||
def group_seq_by_species(df, g_to_sp):
|
def group_seq_by_species(df, g_to_sp):
|
||||||
sph = list(df.homology_species)
|
sph = list(df.homology_species)
|
||||||
ghsp = list(df.homology_gene_stable_id)
|
ghsp = list(df.homology_gene_stable_id)
|
||||||
|
|
@ -35,7 +39,10 @@ def group_seq_by_species(df,g_to_sp):
|
||||||
create_dict(ghsp, sph, g_to_sp)
|
create_dict(ghsp, sph, g_to_sp)
|
||||||
return g_to_sp
|
return g_to_sp
|
||||||
|
|
||||||
#this function returns the gene-id and gene-biotype from the description in the fasta file record.
|
# this function returns the gene-id and gene-biotype from the description
|
||||||
|
# in the fasta file record.
|
||||||
|
|
||||||
|
|
||||||
def description_cleaner(description):
|
def description_cleaner(description):
|
||||||
description = description.split()
|
description = description.split()
|
||||||
t = ""
|
t = ""
|
||||||
|
|
@ -47,15 +54,17 @@ def description_cleaner(description):
|
||||||
t = x[1].split(".")[0]
|
t = x[1].split(".")[0]
|
||||||
if x[0] == "gene_biotype":
|
if x[0] == "gene_biotype":
|
||||||
gbt = x[1]
|
gbt = x[1]
|
||||||
except:
|
except BaseException:
|
||||||
return "aa", "aa"
|
return "aa", "aa"
|
||||||
return t, gbt
|
return t, gbt
|
||||||
|
|
||||||
|
|
||||||
def read_gene_seq(dirname, s, genes_by_species):
|
def read_gene_seq(dirname, s, genes_by_species):
|
||||||
lof = os.listdir(dirname) # list all the files in the sequences directory
|
lof = os.listdir(dirname) # list all the files in the sequences directory
|
||||||
ftr = []
|
ftr = []
|
||||||
for f in lof:
|
for f in lof:
|
||||||
if f.split(".")[0] in s:#check whether the species is present in the species to read list. Will skip those species which are not present in the dataframe
|
if f.split(".")[
|
||||||
|
0] in s: # check whether the species is present in the species to read list. Will skip those species which are not present in the dataframe
|
||||||
ftr.append(f)
|
ftr.append(f)
|
||||||
data = {}
|
data = {}
|
||||||
for f in progressbar.progressbar(ftr):
|
for f in progressbar.progressbar(ftr):
|
||||||
|
|
@ -64,12 +73,13 @@ def read_gene_seq(dirname,s,genes_by_species):
|
||||||
record = SeqIO.parse(file, "fasta")
|
record = SeqIO.parse(file, "fasta")
|
||||||
for r in record:
|
for r in record:
|
||||||
gid, gbt = description_cleaner(r.description)
|
gid, gbt = description_cleaner(r.description)
|
||||||
if str(gid) not in data and str(gid) in genes_by_species[species] and gbt=="protein_coding":
|
if str(gid) not in data and str(
|
||||||
|
gid) in genes_by_species[species] and gbt == "protein_coding":
|
||||||
data[gid] = str(r.seq)
|
data[gid] = str(r.seq)
|
||||||
return data
|
return data
|
||||||
|
|
||||||
def read_gene_sequences(hdf,lsy,data_dir,fname):
|
|
||||||
|
|
||||||
|
def read_gene_sequences(hdf, lsy, data_dir, fname):
|
||||||
"""The basic idea here is to create a list/dictionary of all the genes by their species.
|
"""The basic idea here is to create a list/dictionary of all the genes by their species.
|
||||||
Once the mapping is done, all the respective fasta sequence files are read by Species
|
Once the mapping is done, all the respective fasta sequence files are read by Species
|
||||||
and the CDNA sequences for each gene in the species record are read and stored.
|
and the CDNA sequences for each gene in the species record are read and stored.
|
||||||
|
|
@ -86,9 +96,10 @@ def read_gene_sequences(hdf,lsy,data_dir,fname):
|
||||||
for x in progressbar.progressbar(lsy):
|
for x in progressbar.progressbar(lsy):
|
||||||
try:
|
try:
|
||||||
species = grouped_genes[x] # get the species
|
species = grouped_genes[x] # get the species
|
||||||
except:
|
except BaseException:
|
||||||
continue
|
continue
|
||||||
if x not in gene_by_species_dict[species]:#check if the gene already exists in the species dict or not.
|
# check if the gene already exists in the species dict or not.
|
||||||
|
if x not in gene_by_species_dict[species]:
|
||||||
gene_by_species_dict[species].append(x)
|
gene_by_species_dict[species].append(x)
|
||||||
xl = lsy[x]['b']
|
xl = lsy[x]['b']
|
||||||
xr = lsy[x]['f']
|
xr = lsy[x]['f']
|
||||||
|
|
@ -103,8 +114,8 @@ def read_gene_sequences(hdf,lsy,data_dir,fname):
|
||||||
if gxr not in gene_by_species_dict[species]:
|
if gxr not in gene_by_species_dict[species]:
|
||||||
gene_by_species_dict[species].append(gxr)
|
gene_by_species_dict[species].append(gxr)
|
||||||
|
|
||||||
|
# select those species only whose gene sequences we have to read.
|
||||||
s=[x for x in gene_by_species_dict if len(gene_by_species_dict[x])!=0]#select those species only whose gene sequences we have to read.
|
s = [x for x in gene_by_species_dict if len(gene_by_species_dict[x]) != 0]
|
||||||
s = [x.capitalize() for x in s]
|
s = [x.capitalize() for x in s]
|
||||||
|
|
||||||
data = read_gene_seq(data_dir, s, gene_by_species_dict)
|
data = read_gene_seq(data_dir, s, gene_by_species_dict)
|
||||||
|
|
@ -113,7 +124,7 @@ def read_gene_sequences(hdf,lsy,data_dir,fname):
|
||||||
for gene in gene_by_species_dict[species]:
|
for gene in gene_by_species_dict[species]:
|
||||||
try:
|
try:
|
||||||
_ = data[gene]
|
_ = data[gene]
|
||||||
except:
|
except BaseException:
|
||||||
not_found[gene] = 1
|
not_found[gene] = 1
|
||||||
|
|
||||||
with open("processed/not_found_" + fname + ".json", "w") as file:
|
with open("processed/not_found_" + fname + ".json", "w") as file:
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
import os
|
import os
|
||||||
import pickle
|
import pickle
|
||||||
import json
|
import json
|
||||||
import sys
|
|
||||||
|
|
||||||
def write_dict_json(name, dir, d):
|
def write_dict_json(name, dir, d):
|
||||||
if not os.path.exists(dir):
|
if not os.path.exists(dir):
|
||||||
|
|
@ -11,6 +11,7 @@ def write_dict_json(name,dir,d):
|
||||||
with open(path, 'w') as file:
|
with open(path, 'w') as file:
|
||||||
json.dump(d, file)
|
json.dump(d, file)
|
||||||
|
|
||||||
|
|
||||||
def write_data_synteny(smg, sml, indexes, i, name):
|
def write_data_synteny(smg, sml, indexes, i, name):
|
||||||
if not os.path.exists("temp_" + name):
|
if not os.path.exists("temp_" + name):
|
||||||
os.mkdir("temp_" + name)
|
os.mkdir("temp_" + name)
|
||||||
|
|
|
||||||
|
|
@ -1,18 +1,17 @@
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import gc
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import json
|
|
||||||
import os
|
import os
|
||||||
import progressbar
|
|
||||||
import sys
|
import sys
|
||||||
from selector import select, create_map_reverse
|
from selector import select, create_map_reverse
|
||||||
|
|
||||||
|
|
||||||
def read_db_homology(dir_name, filename):
|
def read_db_homology(dir_name, filename):
|
||||||
df = pd.read_csv(dir_name + "/" + filename, compression='gzip', sep='\t')
|
df = pd.read_csv(dir_name + "/" + filename, compression='gzip', sep='\t')
|
||||||
n = filename.split(".")[0]
|
n = filename.split(".")[0]
|
||||||
n = n.split(" ")[0]
|
n = n.split(" ")[0]
|
||||||
return df, n
|
return df, n
|
||||||
|
|
||||||
|
|
||||||
def get_selection_data():
|
def get_selection_data():
|
||||||
with open("dist_matrix", "r") as file:
|
with open("dist_matrix", "r") as file:
|
||||||
matrix = file.readlines()
|
matrix = file.readlines()
|
||||||
|
|
@ -25,6 +24,7 @@ def get_selection_data():
|
||||||
spnmap, nspmap = create_map_reverse(dname)
|
spnmap, nspmap = create_map_reverse(dname)
|
||||||
return matrix, spnmap, nspmap
|
return matrix, spnmap, nspmap
|
||||||
|
|
||||||
|
|
||||||
def read_select_data(dirname, matrix, spnmap, nspmap, nos):
|
def read_select_data(dirname, matrix, spnmap, nspmap, nos):
|
||||||
lf = os.listdir(dirname)
|
lf = os.listdir(dirname)
|
||||||
if len(lf) == 0:
|
if len(lf) == 0:
|
||||||
|
|
@ -40,11 +40,13 @@ def read_select_data(dirname,matrix,spnmap,nspmap,nos):
|
||||||
indexes = np.array(list(df.index.values))
|
indexes = np.array(list(df.index.values))
|
||||||
np.save("processed/" + n + "_selected_indexes", indexes)
|
np.save("processed/" + n + "_selected_indexes", indexes)
|
||||||
|
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
arg = sys.argv
|
arg = sys.argv
|
||||||
nos = int(arg[-1])
|
nos = int(arg[-1])
|
||||||
matrix, spnmap, nspmap = get_selection_data()
|
matrix, spnmap, nspmap = get_selection_data()
|
||||||
read_select_data("data_homology", matrix, spnmap, nspmap, nos)
|
read_select_data("data_homology", matrix, spnmap, nspmap, nos)
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
if __name__ == "__main__":
|
||||||
main()
|
main()
|
||||||
|
|
|
||||||
39
selector.py
39
selector.py
|
|
@ -1,6 +1,7 @@
|
||||||
import pandas as pd
|
import pandas as pd
|
||||||
import numpy as np
|
import numpy as np
|
||||||
|
|
||||||
|
|
||||||
def create_map_reverse(arr):
|
def create_map_reverse(arr):
|
||||||
m = {}
|
m = {}
|
||||||
rm = {}
|
rm = {}
|
||||||
|
|
@ -9,6 +10,7 @@ def create_map_reverse(arr):
|
||||||
rm[i] = arr[i]
|
rm[i] = arr[i]
|
||||||
return m, rm
|
return m, rm
|
||||||
|
|
||||||
|
|
||||||
def get_data_prop(df, nspmap, sp, prop, nos):
|
def get_data_prop(df, nspmap, sp, prop, nos):
|
||||||
nos = int(nos * prop)
|
nos = int(nos * prop)
|
||||||
sp = [nspmap[x] for x in sp]
|
sp = [nspmap[x] for x in sp]
|
||||||
|
|
@ -20,7 +22,15 @@ def get_data_prop(df,nspmap,sp,prop,nos):
|
||||||
else:
|
else:
|
||||||
return data
|
return data
|
||||||
|
|
||||||
def create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos,hom_type,spname):
|
|
||||||
|
def create_balanced_dataset_paralog(
|
||||||
|
df,
|
||||||
|
matrix,
|
||||||
|
spnmap,
|
||||||
|
nspmap,
|
||||||
|
nos,
|
||||||
|
hom_type,
|
||||||
|
spname):
|
||||||
df = df[df["homology_type"] == hom_type]
|
df = df[df["homology_type"] == hom_type]
|
||||||
dist = matrix[spnmap[spname]]
|
dist = matrix[spnmap[spname]]
|
||||||
dist_sort = np.argsort(dist)
|
dist_sort = np.argsort(dist)
|
||||||
|
|
@ -38,6 +48,7 @@ def create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos,hom_type,spname)
|
||||||
df = pd.concat([df_r, df_dist])
|
df = pd.concat([df_r, df_dist])
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def select_data_goc(df, prop, nos):
|
def select_data_goc(df, prop, nos):
|
||||||
df = df[df["goc_score"] == 0.0]
|
df = df[df["goc_score"] == 0.0]
|
||||||
nos = int(prop * nos)
|
nos = int(prop * nos)
|
||||||
|
|
@ -47,7 +58,15 @@ def select_data_goc(df,prop,nos):
|
||||||
else:
|
else:
|
||||||
return df.loc[df.index.values[rind[:nos]]]
|
return df.loc[df.index.values[rind[:nos]]]
|
||||||
|
|
||||||
def create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos,hom_type,spname):
|
|
||||||
|
def create_balanced_dataset_ortholog(
|
||||||
|
df,
|
||||||
|
matrix,
|
||||||
|
spnmap,
|
||||||
|
nspmap,
|
||||||
|
nos,
|
||||||
|
hom_type,
|
||||||
|
spname):
|
||||||
df = df[df["homology_type"] == hom_type]
|
df = df[df["homology_type"] == hom_type]
|
||||||
dist = matrix[spnmap[spname]]
|
dist = matrix[spnmap[spname]]
|
||||||
dist_sort = np.argsort(dist)
|
dist_sort = np.argsort(dist)
|
||||||
|
|
@ -69,17 +88,23 @@ def create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos,hom_type,spname
|
||||||
df = pd.concat([df_r, df_goc, df_dist])
|
df = pd.concat([df_r, df_goc, df_dist])
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def select(df, nos, matrix, spnmap, nspmap, sp):
|
def select(df, nos, matrix, spnmap, nspmap, sp):
|
||||||
nos_p = int((0.5 * nos) / 2)
|
nos_p = int((0.5 * nos) / 2)
|
||||||
nos_o = int((0.5 * nos) / 3)
|
nos_o = int((0.5 * nos) / 3)
|
||||||
|
|
||||||
# get the paralogy data
|
# get the paralogy data
|
||||||
df_p1=create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos_p,"within_species_paralog",sp)
|
df_p1 = create_balanced_dataset_paralog(
|
||||||
df_p2=create_balanced_dataset_paralog(df,matrix,spnmap,nspmap,nos_p,"other_paralog",sp)
|
df, matrix, spnmap, nspmap, nos_p, "within_species_paralog", sp)
|
||||||
|
df_p2 = create_balanced_dataset_paralog(
|
||||||
|
df, matrix, spnmap, nspmap, nos_p, "other_paralog", sp)
|
||||||
# get the orthology data
|
# get the orthology data
|
||||||
df_o1=create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos_o,"ortholog_one2many",sp)
|
df_o1 = create_balanced_dataset_ortholog(
|
||||||
df_o2=create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos_o,"ortholog_many2many",sp)
|
df, matrix, spnmap, nspmap, nos_o, "ortholog_one2many", sp)
|
||||||
df_o3=create_balanced_dataset_ortholog(df,matrix,spnmap,nspmap,nos_o,"ortholog_one2one",sp)
|
df_o2 = create_balanced_dataset_ortholog(
|
||||||
|
df, matrix, spnmap, nspmap, nos_o, "ortholog_many2many", sp)
|
||||||
|
df_o3 = create_balanced_dataset_ortholog(
|
||||||
|
df, matrix, spnmap, nspmap, nos_o, "ortholog_one2one", sp)
|
||||||
|
|
||||||
# concatenate everything
|
# concatenate everything
|
||||||
df = pd.concat([df_o1, df_o2, df_o3, df_p1, df_p2])
|
df = pd.concat([df_o1, df_o2, df_o3, df_p1, df_p2])
|
||||||
|
|
|
||||||
52
threads.py
52
threads.py
|
|
@ -1,17 +1,14 @@
|
||||||
import pandas as pd
|
from multiprocessing import Process
|
||||||
from threading import Thread
|
|
||||||
from multiprocessing import Process,Lock,Manager
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import edlib as ed
|
import edlib as ed
|
||||||
import pandas as pd
|
|
||||||
import progressbar
|
import progressbar
|
||||||
import time
|
import time
|
||||||
from skbio.alignment import local_pairwise_align_ssw
|
from skbio.alignment import local_pairwise_align_ssw
|
||||||
from skbio import DNA,TabularMSA,RNA
|
from skbio import DNA
|
||||||
import copy
|
import copy
|
||||||
import time
|
|
||||||
from save_data import write_data_synteny
|
from save_data import write_data_synteny
|
||||||
|
|
||||||
|
|
||||||
class Thread_objects():
|
class Thread_objects():
|
||||||
def __init__(self, df_temp, gene_sequences, lsy, i, name):
|
def __init__(self, df_temp, gene_sequences, lsy, i, name):
|
||||||
self.gene_sequences = copy.deepcopy(gene_sequences)
|
self.gene_sequences = copy.deepcopy(gene_sequences)
|
||||||
|
|
@ -30,15 +27,15 @@ class Thread_objects():
|
||||||
if gene == "NULL_GENE":
|
if gene == "NULL_GENE":
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
temp=gene_seq[gene]
|
_ = gene_seq[gene]
|
||||||
except:
|
except BaseException:
|
||||||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||||
for gene in g2:
|
for gene in g2:
|
||||||
if gene == "NULL_GENE":
|
if gene == "NULL_GENE":
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
temp=gene_seq[gene]
|
_ = gene_seq[gene]
|
||||||
except:
|
except BaseException:
|
||||||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||||
sm = np.zeros((n, n, 2))
|
sm = np.zeros((n, n, 2))
|
||||||
sml = np.zeros((n, n, 2))
|
sml = np.zeros((n, n, 2))
|
||||||
|
|
@ -54,15 +51,19 @@ class Thread_objects():
|
||||||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||||
norm_len = max(len(gene_seq[g1[i]]), len(gene_seq[g2[j]]))
|
norm_len = max(len(gene_seq[g1[i]]), len(gene_seq[g2[j]]))
|
||||||
try:
|
try:
|
||||||
result = ed.align(gene_seq[g1[i]],gene_seq[g2[j]], mode="NW", task="distance")
|
result = ed.align(
|
||||||
|
gene_seq[g1[i]], gene_seq[g2[j]], mode="NW", task="distance")
|
||||||
sm[i][j][0] = result["editDistance"] / (norm_len)
|
sm[i][j][0] = result["editDistance"] / (norm_len)
|
||||||
result = ed.align(gene_seq[g1[i]],gene_seq[g2[j]][::-1], mode="NW", task="distance")
|
result = ed.align(
|
||||||
|
gene_seq[g1[i]], gene_seq[g2[j]][::-1], mode="NW", task="distance")
|
||||||
sm[i][j][1] = result["editDistance"] / (norm_len)
|
sm[i][j][1] = result["editDistance"] / (norm_len)
|
||||||
_,result,_=local_pairwise_align_ssw(DNA(gene_seq[g1[i]]),DNA(gene_seq[g2[j]]))
|
_, result, _ = local_pairwise_align_ssw(
|
||||||
|
DNA(gene_seq[g1[i]]), DNA(gene_seq[g2[j]]))
|
||||||
sml[i][j][0] = result / (norm_len)
|
sml[i][j][0] = result / (norm_len)
|
||||||
_,result,_=local_pairwise_align_ssw(DNA(gene_seq[g1[i]]),DNA(gene_seq[g2[j]][::-1]))
|
_, result, _ = local_pairwise_align_ssw(
|
||||||
|
DNA(gene_seq[g1[i]]), DNA(gene_seq[g2[j]][::-1]))
|
||||||
sml[i][j][1] = result / (norm_len)
|
sml[i][j][1] = result / (norm_len)
|
||||||
except:
|
except BaseException:
|
||||||
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
return np.zeros((n, n, 2)), np.zeros((n, n, 2))
|
||||||
return sm, sml
|
return sm, sml
|
||||||
|
|
||||||
|
|
@ -76,12 +77,12 @@ class Thread_objects():
|
||||||
y = []
|
y = []
|
||||||
t += 1
|
t += 1
|
||||||
try:
|
try:
|
||||||
temp=lsy[g1]
|
_ = lsy[g1]
|
||||||
except:
|
except BaseException:
|
||||||
continue
|
continue
|
||||||
try:
|
try:
|
||||||
temp=lsy[g2]
|
_ = lsy[g2]
|
||||||
except:
|
except BaseException:
|
||||||
continue
|
continue
|
||||||
for i in range(len(lsy[g1]['b']) - 1, -1, -1):
|
for i in range(len(lsy[g1]['b']) - 1, -1, -1):
|
||||||
x.append(lsy[g1]['b'][i])
|
x.append(lsy[g1]['b'][i])
|
||||||
|
|
@ -97,23 +98,29 @@ class Thread_objects():
|
||||||
|
|
||||||
assert(len(x) == len(y))
|
assert(len(x) == len(y))
|
||||||
assert(len(x) == (2 * n + 1))
|
assert(len(x) == (2 * n + 1))
|
||||||
smgtemp,smltemp=self.create_synteny_matrix_mul(gene_seq,x,y,2*n+1)
|
smgtemp, smltemp = self.create_synteny_matrix_mul(
|
||||||
|
gene_seq, x, y, 2 * n + 1)
|
||||||
if np.all(smgtemp == 0):
|
if np.all(smgtemp == 0):
|
||||||
continue
|
continue
|
||||||
self.smg.append(smgtemp)
|
self.smg.append(smgtemp)
|
||||||
self.sml.append(smltemp)
|
self.sml.append(smltemp)
|
||||||
self.indexes.append(index)
|
self.indexes.append(index)
|
||||||
self.end = time.time()
|
self.end = time.time()
|
||||||
print("Thread {} finished in {}s.".format(self.i+1,self.end-self.start))
|
print(
|
||||||
|
"Thread {} finished in {}s.".format(
|
||||||
|
self.i + 1,
|
||||||
|
self.end - self.start))
|
||||||
write_data_synteny(self.smg, self.sml, self.indexes, self.i, self.name)
|
write_data_synteny(self.smg, self.sml, self.indexes, self.i, self.name)
|
||||||
|
|
||||||
|
|
||||||
class Procerssrunner():
|
class Procerssrunner():
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
self.thread_alive = []
|
self.thread_alive = []
|
||||||
self.obj_list = []
|
self.obj_list = []
|
||||||
|
|
||||||
def start_thread(self, obj, i, thread_alive, n, name):
|
def start_thread(self, obj, i, thread_alive, n, name):
|
||||||
t=Process(target=obj.synteny_matrix,args=(obj.gene_sequences,obj.df,obj.lsy,n),name="Thread_"+str(i+1))
|
t = Process(target=obj.synteny_matrix, args=(
|
||||||
|
obj.gene_sequences, obj.df, obj.lsy, n), name="Thread_" + str(i + 1))
|
||||||
print("Thread ", (i + 1), " started for ", name, ".")
|
print("Thread ", (i + 1), " started for ", name, ".")
|
||||||
thread_alive.append(t)
|
thread_alive.append(t)
|
||||||
|
|
||||||
|
|
@ -135,4 +142,3 @@ class Procerssrunner():
|
||||||
end = time.time()
|
end = time.time()
|
||||||
print("Ending Processes")
|
print("Ending Processes")
|
||||||
print("Time taken:{}s".format(end - st))
|
print("Time taken:{}s".format(end - st))
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -2,12 +2,14 @@ from ete3 import Tree
|
||||||
import numpy as np
|
import numpy as np
|
||||||
import progressbar
|
import progressbar
|
||||||
|
|
||||||
|
|
||||||
def create_branch_length_padding(bl):
|
def create_branch_length_padding(bl):
|
||||||
maxlen = 29
|
maxlen = 29
|
||||||
for x in bl:
|
for x in bl:
|
||||||
for i in range(len(x), maxlen):
|
for i in range(len(x), maxlen):
|
||||||
x.append(0)
|
x.append(0)
|
||||||
|
|
||||||
|
|
||||||
def create_tree_data(treename, df):
|
def create_tree_data(treename, df):
|
||||||
t = Tree(treename)
|
t = Tree(treename)
|
||||||
branch_lengths_s = []
|
branch_lengths_s = []
|
||||||
|
|
@ -43,4 +45,5 @@ def create_tree_data(treename,df):
|
||||||
dist.append(d)
|
dist.append(d)
|
||||||
create_branch_length_padding(branch_lengths_s)
|
create_branch_length_padding(branch_lengths_s)
|
||||||
create_branch_length_padding(branch_lengths_hs)
|
create_branch_length_padding(branch_lengths_hs)
|
||||||
return np.array(branch_lengths_s),np.array(branch_lengths_hs),np.array(dist),np.array(ns),np.array(nhs)
|
return np.array(branch_lengths_s), np.array(
|
||||||
|
branch_lengths_hs), np.array(dist), np.array(ns), np.array(nhs)
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue