compara-deep-learning/process_data.py

145 lines
4.9 KiB
Python
Raw Normal View History

2019-05-10 10:21:21 -07:00
import pandas
import gc
2019-05-13 10:53:31 -07:00
import numpy as np
2019-06-10 09:37:49 -07:00
import json
import progressbar
from save_data import save_data_json
from save_data import write_dict_json
#process_data.py
def create_map_list(l): #this function maps the indexes to values
t={}
for i in range(len(l)):
t[l[i]]=i
return t
2019-05-10 10:21:21 -07:00
2019-06-10 09:37:49 -07:00
def list_dict_genomes(a,n):
lst=[]
ldt=[]
for x in a:
ldgt={}
uc=list(x["gene_id"])
for i in range(len(uc)):
ldgt[uc[i]]=i
lst.append(uc)
ldt.append(ldgt)
return lst,ldt
2019-05-13 10:53:31 -07:00
def get_nearest_neighbors(g,gs,n,a,d,ld,ldg):
ne=[] #list to store the backward genes
nr=[] #list to store the forward genes
gi=d[gs.capitalize()] #get the address of the corresponding species to which the gene belongs whose neighbor has to be found
sldf=a[gi]#select the dataframe
try:
sld=ld[gi]#see if the corresponding gene map exists
except:
#print("Length of Dataframes:{} \t Length of Loaded Genes:{} \t Length of Loaded Genomes Dictionaries:{}".format(len(a),len(ld),len(ldg)))
return ne,nr
sldg=ldg[gi]#select the corresponding map
if g not in sldg:#if the gene is not present in the dataframe return empty lists
return ne,nr
2019-06-10 09:37:49 -07:00
i=sldg[g]#find the index of the genes
2019-05-13 10:53:31 -07:00
#get the -n neighbors
start=int(sldf.iloc[i,[3]])#get the start location of the gene
flag=0
2019-05-13 10:53:31 -07:00
for j in range(n):
if flag==1:
ne.append("NULL_GENE")
continue
2019-05-13 10:53:31 -07:00
itemp=0
#select the column
2019-05-13 10:53:31 -07:00
end=sldf.iloc[:,4]
end=np.array(end)
assert(len(end)==len(sld))
end=end-start #subtract start from it so as to get relative position
end_s=np.argsort(end)#sort them by the order of distance
2019-06-10 09:37:49 -07:00
if end[end_s[0]]>=0:#if all the genes end ahead of the one in consideration
flag=1#increment the pointer
ne.append("NULL_GENE")#append the NULL_GENE value
continue
for k in end_s:#iterate through the sorted array
if end[k]<0 and end[k+1]>=0:#find the first value that is negative and the next one is positive to get the nearest gene
2019-05-13 10:53:31 -07:00
itemp=k
break
ne.append(sld[itemp])#push the gene in the array
start=int(sldf.iloc[itemp,[3]])#make "start" the start location of the current gene
2019-05-13 10:53:31 -07:00
#print(start)
#get the +n neighbors
flag=0
2019-05-13 10:53:31 -07:00
end=int(sldf.iloc[i,[4]])
for j in range(n):
if flag==1:
nr.append("NULL_GENE")
continue
2019-05-13 10:53:31 -07:00
itemp=0
start=sldf.iloc[:,3]
start=np.array(start)
start=start-end
start_s=np.argsort(start)
if start[start_s[-1]]<0:
flag=1
nr.append("NULL_GENE")
continue
2019-05-13 10:53:31 -07:00
for k in start_s:
if start[k]>0:
itemp=k
break
nr.append(sld[itemp])
end=int(sldf.iloc[itemp,[4]])
return ne,nr
2019-05-10 10:21:21 -07:00
def create_data_homology_ls(a_h,d_h,n,a,d,ld,ldg,save_after,enable_break):
2019-05-13 10:53:31 -07:00
lsy={} #dictionary which stores +/- n genes of the given gene by id. Each key is a gene id which corresponds to the one in center.
t=0
2019-06-10 09:37:49 -07:00
with open("processed/neighbor_genes.json","r") as file:
lsy=dict(json.load(file))
print("Existing neighbor genes read!!")
print(len(lsy))
c=4
lsytemp={}
name="neighbor_genes"
2019-05-13 10:53:31 -07:00
for df in a_h:
2019-06-10 09:37:49 -07:00
for _,row in progressbar.progressbar(df.iterrows()):
2019-05-13 10:53:31 -07:00
x=row["gene_stable_id"]
y=row["homology_gene_stable_id"]
xs=row["species"]
ys=row["homology_species"]
try:
z=lsy[x]
except:
try:
t2=d[xs.capitalize()]#see if the species exist in genomic maps
xl,xr=get_nearest_neighbors(x,xs,n,a,d,ld,ldg)
if len(xl)!=0:#check if neighboring genes were successfully found
lsy[x]=dict(b=xl,f=xr)
lsytemp[x]=dict(b=xl,f=xr)
except:
continue
try:
z=lsy[y]
except:
try:
t2=d[ys.capitalize()]
yl,yr=get_nearest_neighbors(y,ys,n,a,d,ld,ldg)
if len(yl)!=0:
lsy[y]=dict(b=yl,f=yr)
lsytemp[y]=dict(b=yl,f=yr)
except:
continue
t+=1
if t>=save_after:
t=0
c+=1
write_dict_json(name+str(c),"processed",lsytemp)
lsytemp={}
if enable_break==1:
break
c+=1
write_dict_json(name+str(c),"processed",lsytemp)
write_dict_json(name,"processed",lsy)
return lsy