mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
Merge pull request #3 from HarshitGupta11/master
Basic code to download and read files
This commit is contained in:
commit
0e989c02c3
6 changed files with 171 additions and 0 deletions
50
get_data.py
Normal file
50
get_data.py
Normal file
|
|
@ -0,0 +1,50 @@
|
|||
import sys
|
||||
import os
|
||||
from req_data import get_data_file,download_data
|
||||
from read_data import read_data_genome,read_data_homology
|
||||
from process_data import list_dict_genomes
|
||||
|
||||
def get_data_genome(arg,dir,a,d,ld,ldg):
|
||||
if arg[0]=='-d':
|
||||
if arg[4]=="-r":
|
||||
c=0
|
||||
else:
|
||||
return a,d,ld,ldg
|
||||
elif arg[0]=='-f':
|
||||
get_data_file(arg[1],dir)
|
||||
elif arg[0]=="-nd":
|
||||
return ld,ldg,a,d
|
||||
|
||||
if arg[4]=="-r":
|
||||
a,d=read_data_genome(dir,a,d)
|
||||
assert(len(a)==len(d))
|
||||
|
||||
ld,ldg=list_dict_genomes(a,d)
|
||||
|
||||
assert(len(ld)==len(ldg))
|
||||
|
||||
for i in range(len(ld)):
|
||||
assert(len(ld[i])==len(ldg[i]))
|
||||
|
||||
return ld,ldg,a,d
|
||||
|
||||
def get_data_homology(arg,dir,a_h,d_h):
|
||||
if arg[2]=="-l":
|
||||
if not os.path.exists(dir):
|
||||
os.mkdir(dir)
|
||||
download_data(arg[3],dir)
|
||||
elif arg[2]=="-f":
|
||||
get_data_file(arg[3],dir)
|
||||
elif arg[2]=="-d":
|
||||
if arg[4]=="-r":
|
||||
c=0
|
||||
else:
|
||||
return a_h,d_h
|
||||
elif arg[2]=="-nd":
|
||||
return a_h,d_h
|
||||
|
||||
if arg[4]=="-r":
|
||||
a_h,d_h=read_data_homology(dir,a_h,d_h)
|
||||
assert(len(a_h)==len(d_h))
|
||||
|
||||
return a_h,d_h
|
||||
27
main.py
Normal file
27
main.py
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
import sys
|
||||
|
||||
from get_data import get_data_homology,get_data_genome
|
||||
|
||||
arg=sys.argv
|
||||
arg=arg[1:]
|
||||
|
||||
if len(arg)!=5:
|
||||
print("No. of arguments more or less. Please check")
|
||||
sys.exit(1)
|
||||
|
||||
a=[]
|
||||
d={}
|
||||
a_h=[]
|
||||
d_h={}
|
||||
ld=[]
|
||||
ldg=[]
|
||||
dir_g="data"
|
||||
|
||||
ld,ldg,a,d=get_data_genome(arg,dir_g,a,d,ld,ldg)
|
||||
#print(a[0][0:10],"\n",a[2][0:10],"\n",d,"\n",ld[0][0:10],"\n",ld[2][0:10])
|
||||
dir_hom="data_homology"
|
||||
|
||||
a_h,d_h=get_data_homology(arg,dir_hom,a_h,d_h)
|
||||
#print(a_h[0][0:10],"\n",d_h)
|
||||
|
||||
x=input()
|
||||
3
my.txt
Normal file
3
my.txt
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
ftp://ftp.ensembl.org/pub/release-96/gtf/lepisosteus_oculatus/Lepisosteus_oculatus.LepOcu1.96.gtf.gz
|
||||
ftp://ftp.ensembl.org/pub/release-96/gtf/mola_mola/Mola_mola.ASM169857v1.96.gtf.gz
|
||||
ftp://ftp.ensembl.org/pub/release-96/gtf/homo_sapiens/Homo_sapiens.GRCh38.96.gtf.gz
|
||||
15
process_data.py
Normal file
15
process_data.py
Normal file
|
|
@ -0,0 +1,15 @@
|
|||
import pandas
|
||||
import gc
|
||||
|
||||
|
||||
ls=[]
|
||||
ld=[]
|
||||
def list_dict_genomes(a,n):
|
||||
for x in a:
|
||||
ldg={}
|
||||
uc=list(x["gene_id"])
|
||||
for i in range(len(uc)):
|
||||
ldg[uc[i]]=i
|
||||
ls.append(uc)
|
||||
ld.append(ldg)
|
||||
return ls,ld
|
||||
50
read_data.py
Normal file
50
read_data.py
Normal file
|
|
@ -0,0 +1,50 @@
|
|||
import os
|
||||
import pandas as pd
|
||||
import gzip
|
||||
import sys
|
||||
|
||||
def clear_data(x):
|
||||
x=x.split()
|
||||
try:
|
||||
x=x[1]
|
||||
except:
|
||||
c=0
|
||||
#print(x)
|
||||
x=x[1:-1]
|
||||
return x
|
||||
|
||||
def read_data_genome(dir_name,a,dict_ind_genome):
|
||||
lf=os.listdir(dir_name)
|
||||
if len(lf)==0:
|
||||
print("No files in the data directory!!!!!!")
|
||||
sys.exit(1)
|
||||
for x in lf:
|
||||
|
||||
data_gene=pd.read_csv(dir_name+"/"+x,compression='gzip',sep='\t',comment='#',header=None)
|
||||
#print(data_gene.head)
|
||||
data_gene=data_gene[data_gene[2]=="gene"]
|
||||
data_gene=data_gene.sort_values(3)
|
||||
tmp=data_gene[8].str.split(";",expand=True)
|
||||
tmp=tmp.iloc[:,:5]
|
||||
data_gene[["gene_id","gene_version","gene_name","gene_source","gene_biotype"]]=tmp
|
||||
data_gene=data_gene.drop(8,axis=1)
|
||||
#print(data_gene[0:10])
|
||||
for y in ["gene_version","gene_name","gene_source","gene_biotype","gene_id"]:
|
||||
data_gene[y]=data_gene[y].apply(clear_data)
|
||||
#print(data_gene[0:10])
|
||||
a.append(data_gene)
|
||||
n=x.split(".")[0]
|
||||
dict_ind_genome[n]=len(a)-1
|
||||
return a,dict_ind_genome
|
||||
|
||||
def read_data_homology(dir,a_h,d_h):
|
||||
lf=os.listdir(dir)
|
||||
if len(lf)==0:
|
||||
print("No Files in the Directory!!!!!!!")
|
||||
sys.exit(1)
|
||||
for x in lf:
|
||||
data=pd.read_csv(dir+"/"+x,compression='gzip',sep='\t')
|
||||
a_h.append(data)
|
||||
n=x.split(".")[0]
|
||||
d_h[n]=len(a_h)-1
|
||||
return a_h,d_h
|
||||
26
req_data.py
Normal file
26
req_data.py
Normal file
|
|
@ -0,0 +1,26 @@
|
|||
import os
|
||||
import pandas
|
||||
import sys
|
||||
import urllib.request as urllib
|
||||
import pandas as pd
|
||||
|
||||
|
||||
lf=[]
|
||||
|
||||
def download_data(x,dir_name):
|
||||
fname=x.split("/")[-1]
|
||||
path=os.path.join(dir_name,fname)
|
||||
urllib.urlretrieve(x,path)
|
||||
|
||||
def get_data_file(file,dir):
|
||||
if not os.path.isfile(file):
|
||||
print("The specified file does not exist!!!")
|
||||
sys.exit(1)
|
||||
|
||||
with open(file,"r")as f:
|
||||
lf=f.read().splitlines()
|
||||
|
||||
if not os.path.exists(dir):
|
||||
os.mkdir(dir)
|
||||
for x in lf:
|
||||
download_data(x,dir)
|
||||
Loading…
Reference in a new issue