Merge pull request #3 from HarshitGupta11/master

Basic code to download and read files
This commit is contained in:
Mateus Patricio 2019-05-13 09:30:30 +01:00 committed by GitHub
commit 0e989c02c3
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23
6 changed files with 171 additions and 0 deletions

50
get_data.py Normal file
View file

@ -0,0 +1,50 @@
import sys
import os
from req_data import get_data_file,download_data
from read_data import read_data_genome,read_data_homology
from process_data import list_dict_genomes
def get_data_genome(arg,dir,a,d,ld,ldg):
if arg[0]=='-d':
if arg[4]=="-r":
c=0
else:
return a,d,ld,ldg
elif arg[0]=='-f':
get_data_file(arg[1],dir)
elif arg[0]=="-nd":
return ld,ldg,a,d
if arg[4]=="-r":
a,d=read_data_genome(dir,a,d)
assert(len(a)==len(d))
ld,ldg=list_dict_genomes(a,d)
assert(len(ld)==len(ldg))
for i in range(len(ld)):
assert(len(ld[i])==len(ldg[i]))
return ld,ldg,a,d
def get_data_homology(arg,dir,a_h,d_h):
if arg[2]=="-l":
if not os.path.exists(dir):
os.mkdir(dir)
download_data(arg[3],dir)
elif arg[2]=="-f":
get_data_file(arg[3],dir)
elif arg[2]=="-d":
if arg[4]=="-r":
c=0
else:
return a_h,d_h
elif arg[2]=="-nd":
return a_h,d_h
if arg[4]=="-r":
a_h,d_h=read_data_homology(dir,a_h,d_h)
assert(len(a_h)==len(d_h))
return a_h,d_h

27
main.py Normal file
View file

@ -0,0 +1,27 @@
import sys
from get_data import get_data_homology,get_data_genome
arg=sys.argv
arg=arg[1:]
if len(arg)!=5:
print("No. of arguments more or less. Please check")
sys.exit(1)
a=[]
d={}
a_h=[]
d_h={}
ld=[]
ldg=[]
dir_g="data"
ld,ldg,a,d=get_data_genome(arg,dir_g,a,d,ld,ldg)
#print(a[0][0:10],"\n",a[2][0:10],"\n",d,"\n",ld[0][0:10],"\n",ld[2][0:10])
dir_hom="data_homology"
a_h,d_h=get_data_homology(arg,dir_hom,a_h,d_h)
#print(a_h[0][0:10],"\n",d_h)
x=input()

3
my.txt Normal file
View file

@ -0,0 +1,3 @@
ftp://ftp.ensembl.org/pub/release-96/gtf/lepisosteus_oculatus/Lepisosteus_oculatus.LepOcu1.96.gtf.gz
ftp://ftp.ensembl.org/pub/release-96/gtf/mola_mola/Mola_mola.ASM169857v1.96.gtf.gz
ftp://ftp.ensembl.org/pub/release-96/gtf/homo_sapiens/Homo_sapiens.GRCh38.96.gtf.gz

15
process_data.py Normal file
View file

@ -0,0 +1,15 @@
import pandas
import gc
ls=[]
ld=[]
def list_dict_genomes(a,n):
for x in a:
ldg={}
uc=list(x["gene_id"])
for i in range(len(uc)):
ldg[uc[i]]=i
ls.append(uc)
ld.append(ldg)
return ls,ld

50
read_data.py Normal file
View file

@ -0,0 +1,50 @@
import os
import pandas as pd
import gzip
import sys
def clear_data(x):
x=x.split()
try:
x=x[1]
except:
c=0
#print(x)
x=x[1:-1]
return x
def read_data_genome(dir_name,a,dict_ind_genome):
lf=os.listdir(dir_name)
if len(lf)==0:
print("No files in the data directory!!!!!!")
sys.exit(1)
for x in lf:
data_gene=pd.read_csv(dir_name+"/"+x,compression='gzip',sep='\t',comment='#',header=None)
#print(data_gene.head)
data_gene=data_gene[data_gene[2]=="gene"]
data_gene=data_gene.sort_values(3)
tmp=data_gene[8].str.split(";",expand=True)
tmp=tmp.iloc[:,:5]
data_gene[["gene_id","gene_version","gene_name","gene_source","gene_biotype"]]=tmp
data_gene=data_gene.drop(8,axis=1)
#print(data_gene[0:10])
for y in ["gene_version","gene_name","gene_source","gene_biotype","gene_id"]:
data_gene[y]=data_gene[y].apply(clear_data)
#print(data_gene[0:10])
a.append(data_gene)
n=x.split(".")[0]
dict_ind_genome[n]=len(a)-1
return a,dict_ind_genome
def read_data_homology(dir,a_h,d_h):
lf=os.listdir(dir)
if len(lf)==0:
print("No Files in the Directory!!!!!!!")
sys.exit(1)
for x in lf:
data=pd.read_csv(dir+"/"+x,compression='gzip',sep='\t')
a_h.append(data)
n=x.split(".")[0]
d_h[n]=len(a_h)-1
return a_h,d_h

26
req_data.py Normal file
View file

@ -0,0 +1,26 @@
import os
import pandas
import sys
import urllib.request as urllib
import pandas as pd
lf=[]
def download_data(x,dir_name):
fname=x.split("/")[-1]
path=os.path.join(dir_name,fname)
urllib.urlretrieve(x,path)
def get_data_file(file,dir):
if not os.path.isfile(file):
print("The specified file does not exist!!!")
sys.exit(1)
with open(file,"r")as f:
lf=f.read().splitlines()
if not os.path.exists(dir):
os.mkdir(dir)
for x in lf:
download_data(x,dir)