Add files via upload

This commit is contained in:
HarshitGupta11 2019-05-10 22:51:21 +05:30 committed by GitHub
parent f416db443d
commit df729fa8dc
No known key found for this signature in database
GPG key ID: 4AEE18F83AFDEB23
5 changed files with 113 additions and 0 deletions

28
get_data.py Normal file
View file

@ -0,0 +1,28 @@
import sys
import os
from req_data import get_data_file
from read_data import read_data_genome
from process_data import list_dict_genomes
arg=sys.argv
arg=arg[1:]
a=[]
d={}
feature_name='gene'
if arg[0]=='-d':
a,d=read_data_genome(arg[1],a,d)
elif arg[0]=='-f':
dir_name=get_data_file(arg[1])
a,d=read_data_genome(dir_name,a,d)
assert(len(a)==len(d))
ld,ldg=list_dict_genomes(a,d)
assert(len(ld)==len(ldg))
for i in range(len(ld)):
assert(len(ld[i])==len(ldg[i]))

2
my.txt Normal file
View file

@ -0,0 +1,2 @@
ftp://ftp.ensembl.org/pub/release-96/gtf/lepisosteus_oculatus/Lepisosteus_oculatus.LepOcu1.96.gtf.gz
ftp://ftp.ensembl.org/pub/release-96/gtf/mola_mola/Mola_mola.ASM169857v1.96.gtf.gz

15
process_data.py Normal file
View file

@ -0,0 +1,15 @@
import pandas
import gc
ls=[]
ld=[]
def list_dict_genomes(a,n):
for x in a:
ldg={}
uc=list(x["gene_id"])
for i in range(len(uc)):
ldg[uc[i]]=i
ls.append(uc)
ld.append(ldg)
return ls,ld

38
read_data.py Normal file
View file

@ -0,0 +1,38 @@
import os
import pandas as pd
import gzip
import sys
def clear_data(x):
x=x.split()
try:
x=x[1]
except:
c=0
#print(x)
x=x[1:-1]
return x
def read_data_genome(dir_name,a,dict_ind_genome):
lf=os.listdir(dir_name)
if len(lf)==0:
print("No files in the data directory!!!!!!")
sys.exit(1)
for x in lf:
data_gene=pd.read_csv(dir_name+"/"+x,compression='gzip',sep='\t',comment='#',header=None)
#print(data_gene.head)
data_gene=data_gene[data_gene[2]=="gene"]
data_gene=data_gene.sort_values(3)
tmp=data_gene[8].str.split(";",expand=True)
tmp=tmp.iloc[:,:5]
data_gene[["gene_id","gene_version","gene_name","gene_source","gene_biotype"]]=tmp
data_gene=data_gene.drop(8,axis=1)
#print(data_gene[0:10])
for y in ["gene_version","gene_name","gene_source","gene_biotype","gene_id"]:
data_gene[y]=data_gene[y].apply(clear_data)
print(data_gene[0:10])
a.append(data_gene)
n=x.split(".")[0]
dict_ind_genome[n]=len(a)-1
return a,dict_ind_genome

30
req_data.py Normal file
View file

@ -0,0 +1,30 @@
import os
import pandas
import sys
import urllib.request as urllib
import pandas as pd
lf=[]
dir_name="data"
def get_data_file(file):
print(file)
if not os.path.isfile(file):
print("The specified file does not exist!!!")
sys.exit(1)
with open(file,"r")as f:
lf=f.read().splitlines()
if os.path.exists("data"):
print("Data Directory Already Exists!!!")
os.mkdir(dir_name)
for x in lf:
fname=x.split("/")[-1]
path=os.path.join(dir_name,fname)
urllib.urlretrieve(x,path)
return dir_name