mirror of
https://github.com/Priyatham-sai-chand/compara-deep-learning.git
synced 2026-10-05 08:11:34 -07:00
Add files via upload
This commit is contained in:
parent
f416db443d
commit
df729fa8dc
5 changed files with 113 additions and 0 deletions
28
get_data.py
Normal file
28
get_data.py
Normal file
|
|
@ -0,0 +1,28 @@
|
||||||
|
import sys
|
||||||
|
import os
|
||||||
|
from req_data import get_data_file
|
||||||
|
from read_data import read_data_genome
|
||||||
|
from process_data import list_dict_genomes
|
||||||
|
|
||||||
|
arg=sys.argv
|
||||||
|
arg=arg[1:]
|
||||||
|
|
||||||
|
a=[]
|
||||||
|
d={}
|
||||||
|
|
||||||
|
feature_name='gene'
|
||||||
|
|
||||||
|
if arg[0]=='-d':
|
||||||
|
a,d=read_data_genome(arg[1],a,d)
|
||||||
|
elif arg[0]=='-f':
|
||||||
|
dir_name=get_data_file(arg[1])
|
||||||
|
a,d=read_data_genome(dir_name,a,d)
|
||||||
|
|
||||||
|
assert(len(a)==len(d))
|
||||||
|
|
||||||
|
ld,ldg=list_dict_genomes(a,d)
|
||||||
|
|
||||||
|
assert(len(ld)==len(ldg))
|
||||||
|
|
||||||
|
for i in range(len(ld)):
|
||||||
|
assert(len(ld[i])==len(ldg[i]))
|
||||||
2
my.txt
Normal file
2
my.txt
Normal file
|
|
@ -0,0 +1,2 @@
|
||||||
|
ftp://ftp.ensembl.org/pub/release-96/gtf/lepisosteus_oculatus/Lepisosteus_oculatus.LepOcu1.96.gtf.gz
|
||||||
|
ftp://ftp.ensembl.org/pub/release-96/gtf/mola_mola/Mola_mola.ASM169857v1.96.gtf.gz
|
||||||
15
process_data.py
Normal file
15
process_data.py
Normal file
|
|
@ -0,0 +1,15 @@
|
||||||
|
import pandas
|
||||||
|
import gc
|
||||||
|
|
||||||
|
|
||||||
|
ls=[]
|
||||||
|
ld=[]
|
||||||
|
def list_dict_genomes(a,n):
|
||||||
|
for x in a:
|
||||||
|
ldg={}
|
||||||
|
uc=list(x["gene_id"])
|
||||||
|
for i in range(len(uc)):
|
||||||
|
ldg[uc[i]]=i
|
||||||
|
ls.append(uc)
|
||||||
|
ld.append(ldg)
|
||||||
|
return ls,ld
|
||||||
38
read_data.py
Normal file
38
read_data.py
Normal file
|
|
@ -0,0 +1,38 @@
|
||||||
|
import os
|
||||||
|
import pandas as pd
|
||||||
|
import gzip
|
||||||
|
import sys
|
||||||
|
|
||||||
|
def clear_data(x):
|
||||||
|
x=x.split()
|
||||||
|
try:
|
||||||
|
x=x[1]
|
||||||
|
except:
|
||||||
|
c=0
|
||||||
|
#print(x)
|
||||||
|
x=x[1:-1]
|
||||||
|
return x
|
||||||
|
|
||||||
|
def read_data_genome(dir_name,a,dict_ind_genome):
|
||||||
|
lf=os.listdir(dir_name)
|
||||||
|
if len(lf)==0:
|
||||||
|
print("No files in the data directory!!!!!!")
|
||||||
|
sys.exit(1)
|
||||||
|
for x in lf:
|
||||||
|
|
||||||
|
data_gene=pd.read_csv(dir_name+"/"+x,compression='gzip',sep='\t',comment='#',header=None)
|
||||||
|
#print(data_gene.head)
|
||||||
|
data_gene=data_gene[data_gene[2]=="gene"]
|
||||||
|
data_gene=data_gene.sort_values(3)
|
||||||
|
tmp=data_gene[8].str.split(";",expand=True)
|
||||||
|
tmp=tmp.iloc[:,:5]
|
||||||
|
data_gene[["gene_id","gene_version","gene_name","gene_source","gene_biotype"]]=tmp
|
||||||
|
data_gene=data_gene.drop(8,axis=1)
|
||||||
|
#print(data_gene[0:10])
|
||||||
|
for y in ["gene_version","gene_name","gene_source","gene_biotype","gene_id"]:
|
||||||
|
data_gene[y]=data_gene[y].apply(clear_data)
|
||||||
|
print(data_gene[0:10])
|
||||||
|
a.append(data_gene)
|
||||||
|
n=x.split(".")[0]
|
||||||
|
dict_ind_genome[n]=len(a)-1
|
||||||
|
return a,dict_ind_genome
|
||||||
30
req_data.py
Normal file
30
req_data.py
Normal file
|
|
@ -0,0 +1,30 @@
|
||||||
|
import os
|
||||||
|
import pandas
|
||||||
|
import sys
|
||||||
|
import urllib.request as urllib
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
|
||||||
|
lf=[]
|
||||||
|
|
||||||
|
dir_name="data"
|
||||||
|
|
||||||
|
def get_data_file(file):
|
||||||
|
print(file)
|
||||||
|
if not os.path.isfile(file):
|
||||||
|
print("The specified file does not exist!!!")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
with open(file,"r")as f:
|
||||||
|
lf=f.read().splitlines()
|
||||||
|
|
||||||
|
if os.path.exists("data"):
|
||||||
|
print("Data Directory Already Exists!!!")
|
||||||
|
|
||||||
|
os.mkdir(dir_name)
|
||||||
|
for x in lf:
|
||||||
|
fname=x.split("/")[-1]
|
||||||
|
path=os.path.join(dir_name,fname)
|
||||||
|
urllib.urlretrieve(x,path)
|
||||||
|
|
||||||
|
return dir_name
|
||||||
Loading…
Reference in a new issue