mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
Add files via upload
This commit is contained in:
parent
162ea64342
commit
606b2a041f
5 changed files with 87 additions and 29 deletions
48
get_data.py
48
get_data.py
|
|
@ -1,23 +1,22 @@
|
||||||
import sys
|
import sys
|
||||||
import os
|
import os
|
||||||
from req_data import get_data_file
|
from req_data import get_data_file,download_data
|
||||||
from read_data import read_data_genome
|
from read_data import read_data_genome,read_data_homology
|
||||||
from process_data import list_dict_genomes
|
from process_data import list_dict_genomes
|
||||||
|
|
||||||
arg=sys.argv
|
def get_data_genome(arg,dir,a,d,ld,ldg):
|
||||||
arg=arg[1:]
|
|
||||||
|
|
||||||
a=[]
|
|
||||||
d={}
|
|
||||||
|
|
||||||
feature_name='gene'
|
|
||||||
|
|
||||||
if arg[0]=='-d':
|
if arg[0]=='-d':
|
||||||
a,d=read_data_genome(arg[1],a,d)
|
if arg[4]=="-r":
|
||||||
|
c=0
|
||||||
|
else:
|
||||||
|
return a,d,ld,ldg
|
||||||
elif arg[0]=='-f':
|
elif arg[0]=='-f':
|
||||||
dir_name=get_data_file(arg[1])
|
get_data_file(arg[1],dir)
|
||||||
a,d=read_data_genome(dir_name,a,d)
|
elif arg[0]=="-nd":
|
||||||
|
return ld,ldg,a,d
|
||||||
|
|
||||||
|
if arg[4]=="-r":
|
||||||
|
a,d=read_data_genome(dir,a,d)
|
||||||
assert(len(a)==len(d))
|
assert(len(a)==len(d))
|
||||||
|
|
||||||
ld,ldg=list_dict_genomes(a,d)
|
ld,ldg=list_dict_genomes(a,d)
|
||||||
|
|
@ -26,3 +25,26 @@ assert(len(ld)==len(ldg))
|
||||||
|
|
||||||
for i in range(len(ld)):
|
for i in range(len(ld)):
|
||||||
assert(len(ld[i])==len(ldg[i]))
|
assert(len(ld[i])==len(ldg[i]))
|
||||||
|
|
||||||
|
return ld,ldg,a,d
|
||||||
|
|
||||||
|
def get_data_homology(arg,dir,a_h,d_h):
|
||||||
|
if arg[2]=="-l":
|
||||||
|
if not os.path.exists(dir):
|
||||||
|
os.mkdir(dir)
|
||||||
|
download_data(arg[3],dir)
|
||||||
|
elif arg[2]=="-f":
|
||||||
|
get_data_file(arg[3],dir)
|
||||||
|
elif arg[2]=="-d":
|
||||||
|
if arg[4]=="-r":
|
||||||
|
c=0
|
||||||
|
else:
|
||||||
|
return a_h,d_h
|
||||||
|
elif arg[2]=="-nd":
|
||||||
|
return a_h,d_h
|
||||||
|
|
||||||
|
if arg[4]=="-r":
|
||||||
|
a_h,d_h=read_data_homology(dir,a_h,d_h)
|
||||||
|
assert(len(a_h)==len(d_h))
|
||||||
|
|
||||||
|
return a_h,d_h
|
||||||
|
|
|
||||||
27
main.py
Normal file
27
main.py
Normal file
|
|
@ -0,0 +1,27 @@
|
||||||
|
import sys
|
||||||
|
|
||||||
|
from get_data import get_data_homology,get_data_genome
|
||||||
|
|
||||||
|
arg=sys.argv
|
||||||
|
arg=arg[1:]
|
||||||
|
|
||||||
|
if len(arg)!=5:
|
||||||
|
print("No. of arguments more or less. Please check")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
a=[]
|
||||||
|
d={}
|
||||||
|
a_h=[]
|
||||||
|
d_h={}
|
||||||
|
ld=[]
|
||||||
|
ldg=[]
|
||||||
|
dir_g="data"
|
||||||
|
|
||||||
|
ld,ldg,a,d=get_data_genome(arg,dir_g,a,d,ld,ldg)
|
||||||
|
#print(a[0][0:10],"\n",a[2][0:10],"\n",d,"\n",ld[0][0:10],"\n",ld[2][0:10])
|
||||||
|
dir_hom="data_homology"
|
||||||
|
|
||||||
|
a_h,d_h=get_data_homology(arg,dir_hom,a_h,d_h)
|
||||||
|
#print(a_h[0][0:10],"\n",d_h)
|
||||||
|
|
||||||
|
x=input()
|
||||||
1
my.txt
1
my.txt
|
|
@ -1,2 +1,3 @@
|
||||||
ftp://ftp.ensembl.org/pub/release-96/gtf/lepisosteus_oculatus/Lepisosteus_oculatus.LepOcu1.96.gtf.gz
|
ftp://ftp.ensembl.org/pub/release-96/gtf/lepisosteus_oculatus/Lepisosteus_oculatus.LepOcu1.96.gtf.gz
|
||||||
ftp://ftp.ensembl.org/pub/release-96/gtf/mola_mola/Mola_mola.ASM169857v1.96.gtf.gz
|
ftp://ftp.ensembl.org/pub/release-96/gtf/mola_mola/Mola_mola.ASM169857v1.96.gtf.gz
|
||||||
|
ftp://ftp.ensembl.org/pub/release-96/gtf/homo_sapiens/Homo_sapiens.GRCh38.96.gtf.gz
|
||||||
|
|
|
||||||
12
read_data.py
12
read_data.py
|
|
@ -36,3 +36,15 @@ def read_data_genome(dir_name,a,dict_ind_genome):
|
||||||
n=x.split(".")[0]
|
n=x.split(".")[0]
|
||||||
dict_ind_genome[n]=len(a)-1
|
dict_ind_genome[n]=len(a)-1
|
||||||
return a,dict_ind_genome
|
return a,dict_ind_genome
|
||||||
|
|
||||||
|
def read_data_homology(dir,a_h,d_h):
|
||||||
|
lf=os.listdir(dir)
|
||||||
|
if len(lf)==0:
|
||||||
|
print("No Files in the Directory!!!!!!!")
|
||||||
|
sys.exit(1)
|
||||||
|
for x in lf:
|
||||||
|
data=pd.read_csv(dir+"/"+x,compression='gzip',sep='\t')
|
||||||
|
a_h.append(data)
|
||||||
|
n=x.split(".")[0]
|
||||||
|
d_h[n]=len(a_h)-1
|
||||||
|
return a_h,d_h
|
||||||
|
|
|
||||||
20
req_data.py
20
req_data.py
|
|
@ -7,10 +7,12 @@ import pandas as pd
|
||||||
|
|
||||||
lf=[]
|
lf=[]
|
||||||
|
|
||||||
dir_name="data"
|
def download_data(x,dir_name):
|
||||||
|
fname=x.split("/")[-1]
|
||||||
|
path=os.path.join(dir_name,fname)
|
||||||
|
urllib.urlretrieve(x,path)
|
||||||
|
|
||||||
def get_data_file(file):
|
def get_data_file(file,dir):
|
||||||
print(file)
|
|
||||||
if not os.path.isfile(file):
|
if not os.path.isfile(file):
|
||||||
print("The specified file does not exist!!!")
|
print("The specified file does not exist!!!")
|
||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
@ -18,13 +20,7 @@ def get_data_file(file):
|
||||||
with open(file,"r")as f:
|
with open(file,"r")as f:
|
||||||
lf=f.read().splitlines()
|
lf=f.read().splitlines()
|
||||||
|
|
||||||
if os.path.exists("data"):
|
if not os.path.exists(dir):
|
||||||
print("Data Directory Already Exists!!!")
|
os.mkdir(dir)
|
||||||
|
|
||||||
os.mkdir(dir_name)
|
|
||||||
for x in lf:
|
for x in lf:
|
||||||
fname=x.split("/")[-1]
|
download_data(x,dir)
|
||||||
path=os.path.join(dir_name,fname)
|
|
||||||
urllib.urlretrieve(x,path)
|
|
||||||
|
|
||||||
return dir_name
|
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue