2017-10-12 21:05:06 +03:00
|
|
|
# coding: utf8
|
2017-10-27 15:38:39 +03:00
|
|
|
from __future__ import unicode_literals, print_function
|
2017-10-12 21:05:06 +03:00
|
|
|
|
|
|
|
import pkg_resources
|
|
|
|
from pathlib import Path
|
2018-01-03 23:20:35 +03:00
|
|
|
import sys
|
2018-05-20 21:26:56 +03:00
|
|
|
import requests
|
2018-11-30 22:16:14 +03:00
|
|
|
from wasabi import Printer
|
2017-10-12 21:05:06 +03:00
|
|
|
|
2018-04-03 16:50:31 +03:00
|
|
|
from ._messages import Messages
|
2018-11-30 22:16:14 +03:00
|
|
|
from ..compat import path2str
|
|
|
|
from ..util import get_data_path, read_json
|
2017-10-12 21:05:06 +03:00
|
|
|
from .. import about
|
|
|
|
|
|
|
|
|
2018-01-04 23:33:47 +03:00
|
|
|
def validate():
|
2018-11-30 22:16:14 +03:00
|
|
|
"""
|
|
|
|
Validate that the currently installed version of spaCy is compatible
|
2017-10-12 21:05:06 +03:00
|
|
|
with the installed models. Should be run after `pip install -U spacy`.
|
|
|
|
"""
|
2018-11-30 22:16:14 +03:00
|
|
|
msg = Printer()
|
|
|
|
with msg.loading("Loading compatibility table..."):
|
|
|
|
r = requests.get(about.__compatibility__)
|
|
|
|
if r.status_code != 200:
|
|
|
|
msg.fail(Messages.M003.format(code=r.status_code), Messages.M021, exits=1)
|
|
|
|
msg.good("Loaded compatibility table")
|
|
|
|
compat = r.json()["spacy"]
|
2018-02-01 00:06:28 +03:00
|
|
|
current_compat = compat.get(about.__version__)
|
|
|
|
if not current_compat:
|
2018-11-30 22:16:14 +03:00
|
|
|
msg.fail(
|
|
|
|
Messages.M022.format(version=about.__version__),
|
|
|
|
about.__compatibility__,
|
|
|
|
exits=1,
|
|
|
|
)
|
2017-10-12 21:05:06 +03:00
|
|
|
all_models = set()
|
|
|
|
for spacy_v, models in dict(compat).items():
|
|
|
|
all_models.update(models.keys())
|
|
|
|
for model, model_vs in models.items():
|
|
|
|
compat[spacy_v][model] = [reformat_version(v) for v in model_vs]
|
|
|
|
model_links = get_model_links(current_compat)
|
|
|
|
model_pkgs = get_model_pkgs(current_compat, all_models)
|
2018-11-30 22:16:14 +03:00
|
|
|
incompat_links = {l for l, d in model_links.items() if not d["compat"]}
|
|
|
|
incompat_models = {d["name"] for _, d in model_pkgs.items() if not d["compat"]}
|
|
|
|
incompat_models.update(
|
|
|
|
[d["name"] for _, d in model_links.items() if not d["compat"]]
|
|
|
|
)
|
2017-10-12 21:05:06 +03:00
|
|
|
na_models = [m for m in incompat_models if m not in current_compat]
|
|
|
|
update_models = [m for m in incompat_models if m in current_compat]
|
2018-11-30 22:16:14 +03:00
|
|
|
spacy_dir = Path(__file__).parent.parent
|
|
|
|
|
|
|
|
msg.divider(Messages.M023.format(version=about.__version__))
|
|
|
|
msg.info("spaCy installation: {}".format(path2str(spacy_dir)))
|
2017-10-12 21:05:06 +03:00
|
|
|
|
|
|
|
if model_links or model_pkgs:
|
2018-11-30 22:16:14 +03:00
|
|
|
header = ("TYPE", "NAME", "MODEL", "VERSION", "")
|
|
|
|
rows = []
|
2017-10-12 21:05:06 +03:00
|
|
|
for name, data in model_pkgs.items():
|
2018-11-30 22:16:14 +03:00
|
|
|
rows.append(get_model_row(current_compat, name, data, msg))
|
2017-10-12 21:05:06 +03:00
|
|
|
for name, data in model_links.items():
|
2018-11-30 22:16:14 +03:00
|
|
|
rows.append(get_model_row(current_compat, name, data, msg, "link"))
|
|
|
|
msg.table(rows, header=header)
|
2017-10-12 21:05:06 +03:00
|
|
|
else:
|
2018-11-30 22:16:14 +03:00
|
|
|
msg.text(Messages.M024, exits=0)
|
2017-10-12 21:05:06 +03:00
|
|
|
if update_models:
|
2018-11-30 22:16:14 +03:00
|
|
|
msg.divider("Install updates")
|
|
|
|
cmd = "python -m spacy download {}"
|
|
|
|
print("\n".join([cmd.format(pkg) for pkg in update_models]) + "\n")
|
2017-10-12 21:05:06 +03:00
|
|
|
if na_models:
|
2018-11-30 22:16:14 +03:00
|
|
|
msg.text(
|
|
|
|
Messages.M025.format(version=about.__version__, models=", ".join(na_models))
|
|
|
|
)
|
2017-10-12 21:05:06 +03:00
|
|
|
if incompat_links:
|
2018-11-30 22:16:14 +03:00
|
|
|
msg.text(Messages.M027.format(path=path2str(get_data_path())))
|
2018-01-03 23:20:35 +03:00
|
|
|
if incompat_models or incompat_links:
|
|
|
|
sys.exit(1)
|
|
|
|
|
2017-10-12 21:05:06 +03:00
|
|
|
|
|
|
|
def get_model_links(compat):
|
|
|
|
links = {}
|
|
|
|
data_path = get_data_path()
|
|
|
|
if data_path:
|
|
|
|
models = [p for p in data_path.iterdir() if is_model_path(p)]
|
|
|
|
for model in models:
|
2018-11-30 22:16:14 +03:00
|
|
|
meta_path = Path(model) / "meta.json"
|
2017-10-12 21:05:06 +03:00
|
|
|
if not meta_path.exists():
|
|
|
|
continue
|
|
|
|
meta = read_json(meta_path)
|
|
|
|
link = model.parts[-1]
|
2018-11-30 22:16:14 +03:00
|
|
|
name = meta["lang"] + "_" + meta["name"]
|
|
|
|
links[link] = {
|
|
|
|
"name": name,
|
|
|
|
"version": meta["version"],
|
|
|
|
"compat": is_compat(compat, name, meta["version"]),
|
|
|
|
}
|
2017-10-12 21:05:06 +03:00
|
|
|
return links
|
|
|
|
|
|
|
|
|
|
|
|
def get_model_pkgs(compat, all_models):
|
|
|
|
pkgs = {}
|
|
|
|
for pkg_name, pkg_data in pkg_resources.working_set.by_key.items():
|
2018-11-30 22:16:14 +03:00
|
|
|
package = pkg_name.replace("-", "_")
|
2017-10-12 21:05:06 +03:00
|
|
|
if package in all_models:
|
|
|
|
version = pkg_data.version
|
2018-11-30 22:16:14 +03:00
|
|
|
pkgs[pkg_name] = {
|
|
|
|
"name": package,
|
|
|
|
"version": version,
|
|
|
|
"compat": is_compat(compat, package, version),
|
|
|
|
}
|
2017-10-12 21:05:06 +03:00
|
|
|
return pkgs
|
|
|
|
|
|
|
|
|
2018-11-30 22:16:14 +03:00
|
|
|
def get_model_row(compat, name, data, msg, model_type="package"):
|
|
|
|
if data["compat"]:
|
|
|
|
comp = msg.text("", color="green", icon="good", no_print=True)
|
|
|
|
version = msg.text(data["version"], color="green", no_print=True)
|
2017-10-12 21:05:06 +03:00
|
|
|
else:
|
2018-11-30 22:16:14 +03:00
|
|
|
version = msg.text(data["version"], color="red", no_print=True)
|
|
|
|
comp = "--> {}".format(compat.get(data["name"], ["n/a"])[0])
|
|
|
|
return (model_type, name, data["name"], version, comp)
|
2017-10-12 21:05:06 +03:00
|
|
|
|
|
|
|
|
|
|
|
def is_model_path(model_path):
|
2018-11-30 22:16:14 +03:00
|
|
|
exclude = ["cache", "pycache", "__pycache__"]
|
2017-10-12 21:05:06 +03:00
|
|
|
name = model_path.parts[-1]
|
2018-11-30 22:16:14 +03:00
|
|
|
return model_path.is_dir() and name not in exclude and not name.startswith(".")
|
2017-10-12 21:05:06 +03:00
|
|
|
|
|
|
|
|
|
|
|
def is_compat(compat, name, version):
|
|
|
|
return name in compat and version in compat[name]
|
|
|
|
|
|
|
|
|
|
|
|
def reformat_version(version):
|
2017-10-27 15:38:39 +03:00
|
|
|
"""Hack to reformat old versions ending on '-alpha' to match pip format."""
|
2018-11-30 22:16:14 +03:00
|
|
|
if version.endswith("-alpha"):
|
|
|
|
return version.replace("-alpha", "a0")
|
|
|
|
return version.replace("-alpha", "a")
|