mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
Add files via upload
This commit is contained in:
parent
162ea64342
commit
606b2a041f
5 changed files with 87 additions and 29 deletions
56
get_data.py
56
get_data.py
|
|
@ -1,28 +1,50 @@
|
|||
import sys
|
||||
import os
|
||||
from req_data import get_data_file
|
||||
from read_data import read_data_genome
|
||||
from req_data import get_data_file,download_data
|
||||
from read_data import read_data_genome,read_data_homology
|
||||
from process_data import list_dict_genomes
|
||||
|
||||
arg=sys.argv
|
||||
arg=arg[1:]
|
||||
def get_data_genome(arg,dir,a,d,ld,ldg):
|
||||
if arg[0]=='-d':
|
||||
if arg[4]=="-r":
|
||||
c=0
|
||||
else:
|
||||
return a,d,ld,ldg
|
||||
elif arg[0]=='-f':
|
||||
get_data_file(arg[1],dir)
|
||||
elif arg[0]=="-nd":
|
||||
return ld,ldg,a,d
|
||||
|
||||
a=[]
|
||||
d={}
|
||||
if arg[4]=="-r":
|
||||
a,d=read_data_genome(dir,a,d)
|
||||
assert(len(a)==len(d))
|
||||
|
||||
feature_name='gene'
|
||||
ld,ldg=list_dict_genomes(a,d)
|
||||
|
||||
if arg[0]=='-d':
|
||||
a,d=read_data_genome(arg[1],a,d)
|
||||
elif arg[0]=='-f':
|
||||
dir_name=get_data_file(arg[1])
|
||||
a,d=read_data_genome(dir_name,a,d)
|
||||
assert(len(ld)==len(ldg))
|
||||
|
||||
assert(len(a)==len(d))
|
||||
for i in range(len(ld)):
|
||||
assert(len(ld[i])==len(ldg[i]))
|
||||
|
||||
ld,ldg=list_dict_genomes(a,d)
|
||||
return ld,ldg,a,d
|
||||
|
||||
assert(len(ld)==len(ldg))
|
||||
def get_data_homology(arg,dir,a_h,d_h):
|
||||
if arg[2]=="-l":
|
||||
if not os.path.exists(dir):
|
||||
os.mkdir(dir)
|
||||
download_data(arg[3],dir)
|
||||
elif arg[2]=="-f":
|
||||
get_data_file(arg[3],dir)
|
||||
elif arg[2]=="-d":
|
||||
if arg[4]=="-r":
|
||||
c=0
|
||||
else:
|
||||
return a_h,d_h
|
||||
elif arg[2]=="-nd":
|
||||
return a_h,d_h
|
||||
|
||||
for i in range(len(ld)):
|
||||
assert(len(ld[i])==len(ldg[i]))
|
||||
if arg[4]=="-r":
|
||||
a_h,d_h=read_data_homology(dir,a_h,d_h)
|
||||
assert(len(a_h)==len(d_h))
|
||||
|
||||
return a_h,d_h
|
||||
|
|
|
|||
27
main.py
Normal file
27
main.py
Normal file
|
|
@ -0,0 +1,27 @@
|
|||
import sys
|
||||
|
||||
from get_data import get_data_homology,get_data_genome
|
||||
|
||||
arg=sys.argv
|
||||
arg=arg[1:]
|
||||
|
||||
if len(arg)!=5:
|
||||
print("No. of arguments more or less. Please check")
|
||||
sys.exit(1)
|
||||
|
||||
a=[]
|
||||
d={}
|
||||
a_h=[]
|
||||
d_h={}
|
||||
ld=[]
|
||||
ldg=[]
|
||||
dir_g="data"
|
||||
|
||||
ld,ldg,a,d=get_data_genome(arg,dir_g,a,d,ld,ldg)
|
||||
#print(a[0][0:10],"\n",a[2][0:10],"\n",d,"\n",ld[0][0:10],"\n",ld[2][0:10])
|
||||
dir_hom="data_homology"
|
||||
|
||||
a_h,d_h=get_data_homology(arg,dir_hom,a_h,d_h)
|
||||
#print(a_h[0][0:10],"\n",d_h)
|
||||
|
||||
x=input()
|
||||
1
my.txt
1
my.txt
|
|
@ -1,2 +1,3 @@
|
|||
ftp://ftp.ensembl.org/pub/release-96/gtf/lepisosteus_oculatus/Lepisosteus_oculatus.LepOcu1.96.gtf.gz
|
||||
ftp://ftp.ensembl.org/pub/release-96/gtf/mola_mola/Mola_mola.ASM169857v1.96.gtf.gz
|
||||
ftp://ftp.ensembl.org/pub/release-96/gtf/homo_sapiens/Homo_sapiens.GRCh38.96.gtf.gz
|
||||
|
|
|
|||
12
read_data.py
12
read_data.py
|
|
@ -36,3 +36,15 @@ def read_data_genome(dir_name,a,dict_ind_genome):
|
|||
n=x.split(".")[0]
|
||||
dict_ind_genome[n]=len(a)-1
|
||||
return a,dict_ind_genome
|
||||
|
||||
def read_data_homology(dir,a_h,d_h):
|
||||
lf=os.listdir(dir)
|
||||
if len(lf)==0:
|
||||
print("No Files in the Directory!!!!!!!")
|
||||
sys.exit(1)
|
||||
for x in lf:
|
||||
data=pd.read_csv(dir+"/"+x,compression='gzip',sep='\t')
|
||||
a_h.append(data)
|
||||
n=x.split(".")[0]
|
||||
d_h[n]=len(a_h)-1
|
||||
return a_h,d_h
|
||||
|
|
|
|||
20
req_data.py
20
req_data.py
|
|
@ -7,10 +7,12 @@ import pandas as pd
|
|||
|
||||
lf=[]
|
||||
|
||||
dir_name="data"
|
||||
def download_data(x,dir_name):
|
||||
fname=x.split("/")[-1]
|
||||
path=os.path.join(dir_name,fname)
|
||||
urllib.urlretrieve(x,path)
|
||||
|
||||
def get_data_file(file):
|
||||
print(file)
|
||||
def get_data_file(file,dir):
|
||||
if not os.path.isfile(file):
|
||||
print("The specified file does not exist!!!")
|
||||
sys.exit(1)
|
||||
|
|
@ -18,13 +20,7 @@ def get_data_file(file):
|
|||
with open(file,"r")as f:
|
||||
lf=f.read().splitlines()
|
||||
|
||||
if os.path.exists("data"):
|
||||
print("Data Directory Already Exists!!!")
|
||||
|
||||
os.mkdir(dir_name)
|
||||
if not os.path.exists(dir):
|
||||
os.mkdir(dir)
|
||||
for x in lf:
|
||||
fname=x.split("/")[-1]
|
||||
path=os.path.join(dir_name,fname)
|
||||
urllib.urlretrieve(x,path)
|
||||
|
||||
return dir_name
|
||||
download_data(x,dir)
|
||||
|
|
|
|||
Loading…
Reference in a new issue