forked from Reed-CompBio/spras-benchmarking
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathstringdb.py
More file actions
49 lines (38 loc) · 1.56 KB
/
Copy pathstringdb.py
File metadata and controls
49 lines (38 loc) · 1.56 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
import argparse
import os
from pathlib import Path
from databases.util import uncompress
from cache.directory import get_cache_item
# https://stackoverflow.com/a/5137509/7589775
dir_path = os.path.dirname(os.path.realpath(__file__))
string_path = Path(dir_path, "string")
def parse_args():
parser = argparse.ArgumentParser(
prog="STRING DB Fetcher", description="Downloads specified STRING DB background interactomes from a specific organism."
)
parser.add_argument(
"-i",
"--id",
help="""
The specified organism ID to use.
See https://string-db.org/cgi/download for more info.
For example, 9606 is the homo sapiens background interactome.
For an example usage, see datasets/diseases's Snakefile.
""",
type=int,
required=True,
)
return parser.parse_args()
def main():
args = parse_args()
string_path.mkdir(exist_ok=True)
# We download the links file
links_file = string_path / f"{args.id}.protein.links.v12.0.txt.gz"
get_cache_item(["STRING", str(args.id), "links"]).download(links_file)
uncompress(links_file, links_file.with_suffix("")) # an extra call of with_suffix strips the `.gz` prefix
# and its associated aliases
aliases_file = string_path / f"{args.id}.protein.aliases.v12.0.txt.gz"
get_cache_item(["STRING", str(args.id), "aliases"]).download(aliases_file)
uncompress(aliases_file, aliases_file.with_suffix(""))
if __name__ == "__main__":
main()