mirror of
https://github.com/explosion/spaCy.git
synced 2024-11-14 05:37:03 +03:00
5ca7dd0f94
* Improve load_language_data helper * WIP: Add Lookups implementation * Start moving lemma data over to JSON * WIP: move data over for more languages * Convert more languages * Fix lemmatizer fixtures in tests * Finish conversion * Auto-format JSON files * Fix test for now * Make sure tables are stored on instance
47 lines
1.3 KiB
Cython
47 lines
1.3 KiB
Cython
from libcpp.vector cimport vector
|
|
|
|
from preshed.maps cimport PreshMap
|
|
from cymem.cymem cimport Pool
|
|
from murmurhash.mrmr cimport hash64
|
|
|
|
from .structs cimport LexemeC, TokenC
|
|
from .typedefs cimport utf8_t, attr_t, hash_t
|
|
from .strings cimport StringStore
|
|
from .morphology cimport Morphology
|
|
|
|
|
|
cdef LexemeC EMPTY_LEXEME
|
|
|
|
|
|
cdef union LexemesOrTokens:
|
|
const LexemeC* const* lexemes
|
|
const TokenC* tokens
|
|
|
|
|
|
cdef struct _Cached:
|
|
LexemesOrTokens data
|
|
bint is_lex
|
|
int length
|
|
|
|
|
|
cdef class Vocab:
|
|
cdef Pool mem
|
|
cpdef readonly StringStore strings
|
|
cpdef public Morphology morphology
|
|
cpdef public object vectors
|
|
cpdef public object lookups
|
|
cdef readonly int length
|
|
cdef public object data_dir
|
|
cdef public object lex_attr_getters
|
|
cdef public object cfg
|
|
|
|
cdef const LexemeC* get(self, Pool mem, unicode string) except NULL
|
|
cdef const LexemeC* get_by_orth(self, Pool mem, attr_t orth) except NULL
|
|
cdef const TokenC* make_fused_token(self, substrings) except NULL
|
|
|
|
cdef const LexemeC* _new_lexeme(self, Pool mem, unicode string) except NULL
|
|
cdef int _add_lex_to_vocab(self, hash_t key, const LexemeC* lex) except -1
|
|
cdef const LexemeC* _new_lexeme(self, Pool mem, unicode string) except NULL
|
|
|
|
cdef PreshMap _by_orth
|