2020-08-28 15:08:33 +03:00
|
|
|
from typing import Dict, Any, Tuple, Callable, List
|
2020-08-26 16:24:33 +03:00
|
|
|
|
|
|
|
from ..util import registry
|
2020-08-28 14:55:32 +03:00
|
|
|
from .. import util
|
2020-08-26 16:24:33 +03:00
|
|
|
from ..errors import Errors
|
|
|
|
from wasabi import msg
|
|
|
|
|
|
|
|
|
|
|
|
@registry.loggers("spacy.ConsoleLogger.v1")
|
|
|
|
def console_logger():
|
|
|
|
def setup_printer(
|
2020-08-29 14:01:10 +03:00
|
|
|
nlp: "Language",
|
2020-08-26 16:24:33 +03:00
|
|
|
) -> Tuple[Callable[[Dict[str, Any]], None], Callable]:
|
2020-09-23 11:37:12 +03:00
|
|
|
# we assume here that only components are enabled that should be trained & logged
|
|
|
|
logged_pipes = nlp.pipe_names
|
2020-09-24 12:04:35 +03:00
|
|
|
score_weights = nlp.config["training"]["score_weights"]
|
|
|
|
score_cols = [col for col, value in score_weights.items() if value is not None]
|
2020-08-26 16:24:33 +03:00
|
|
|
score_widths = [max(len(col), 6) for col in score_cols]
|
2020-09-23 11:37:12 +03:00
|
|
|
loss_cols = [f"Loss {pipe}" for pipe in logged_pipes]
|
2020-08-26 16:24:33 +03:00
|
|
|
loss_widths = [max(len(col), 8) for col in loss_cols]
|
|
|
|
table_header = ["E", "#"] + loss_cols + score_cols + ["Score"]
|
|
|
|
table_header = [col.upper() for col in table_header]
|
|
|
|
table_widths = [3, 6] + loss_widths + score_widths + [6]
|
|
|
|
table_aligns = ["r" for _ in table_widths]
|
|
|
|
msg.row(table_header, widths=table_widths)
|
|
|
|
msg.row(["-" * width for width in table_widths])
|
|
|
|
|
|
|
|
def log_step(info: Dict[str, Any]):
|
|
|
|
try:
|
|
|
|
losses = [
|
|
|
|
"{0:.2f}".format(float(info["losses"][pipe_name]))
|
2020-09-23 11:37:12 +03:00
|
|
|
for pipe_name in logged_pipes
|
2020-08-26 16:24:33 +03:00
|
|
|
]
|
|
|
|
except KeyError as e:
|
|
|
|
raise KeyError(
|
|
|
|
Errors.E983.format(
|
|
|
|
dict="scores (losses)",
|
|
|
|
key=str(e),
|
|
|
|
keys=list(info["losses"].keys()),
|
|
|
|
)
|
|
|
|
) from None
|
2020-09-13 18:39:31 +03:00
|
|
|
scores = []
|
|
|
|
for col in score_cols:
|
2020-09-24 12:04:35 +03:00
|
|
|
score = info["other_scores"].get(col, 0.0)
|
|
|
|
try:
|
|
|
|
score = float(score)
|
|
|
|
if col != "speed":
|
|
|
|
score *= 100
|
|
|
|
scores.append("{0:.2f}".format(score))
|
|
|
|
except TypeError:
|
|
|
|
err = Errors.E916.format(name=col, score_type=type(score))
|
2020-09-24 12:29:07 +03:00
|
|
|
raise ValueError(err) from None
|
2020-08-26 16:24:33 +03:00
|
|
|
data = (
|
|
|
|
[info["epoch"], info["step"]]
|
|
|
|
+ losses
|
|
|
|
+ scores
|
|
|
|
+ ["{0:.2f}".format(float(info["score"]))]
|
|
|
|
)
|
|
|
|
msg.row(data, widths=table_widths, aligns=table_aligns)
|
|
|
|
|
|
|
|
def finalize():
|
|
|
|
pass
|
|
|
|
|
|
|
|
return log_step, finalize
|
|
|
|
|
|
|
|
return setup_printer
|
|
|
|
|
|
|
|
|
|
|
|
@registry.loggers("spacy.WandbLogger.v1")
|
2020-08-28 15:08:33 +03:00
|
|
|
def wandb_logger(project_name: str, remove_config_values: List[str] = []):
|
2020-08-26 16:24:33 +03:00
|
|
|
import wandb
|
|
|
|
|
|
|
|
console = console_logger()
|
|
|
|
|
|
|
|
def setup_logger(
|
2020-08-29 14:01:10 +03:00
|
|
|
nlp: "Language",
|
2020-08-26 16:24:33 +03:00
|
|
|
) -> Tuple[Callable[[Dict[str, Any]], None], Callable]:
|
|
|
|
config = nlp.config.interpolate()
|
2020-08-28 14:55:32 +03:00
|
|
|
config_dot = util.dict_to_dot(config)
|
2020-08-28 15:06:23 +03:00
|
|
|
for field in remove_config_values:
|
2020-08-28 14:55:32 +03:00
|
|
|
del config_dot[field]
|
|
|
|
config = util.dot_to_dict(config_dot)
|
2020-09-15 13:56:33 +03:00
|
|
|
wandb.init(project=project_name, config=config, reinit=True)
|
2020-08-26 16:24:33 +03:00
|
|
|
console_log_step, console_finalize = console(nlp)
|
|
|
|
|
|
|
|
def log_step(info: Dict[str, Any]):
|
|
|
|
console_log_step(info)
|
|
|
|
score = info["score"]
|
|
|
|
other_scores = info["other_scores"]
|
|
|
|
losses = info["losses"]
|
2020-08-28 14:55:32 +03:00
|
|
|
wandb.log({"score": score})
|
2020-08-26 16:24:33 +03:00
|
|
|
if losses:
|
|
|
|
wandb.log({f"loss_{k}": v for k, v in losses.items()})
|
|
|
|
if isinstance(other_scores, dict):
|
|
|
|
wandb.log(other_scores)
|
|
|
|
|
|
|
|
def finalize():
|
|
|
|
console_finalize()
|
2020-09-15 13:56:33 +03:00
|
|
|
wandb.join()
|
2020-08-26 16:24:33 +03:00
|
|
|
|
|
|
|
return log_step, finalize
|
|
|
|
|
|
|
|
return setup_logger
|