-
Notifications
You must be signed in to change notification settings - Fork 43
OpenConceptLab/ocl_online#247 | Release-level vectorization: vectors for every semantic repo version, reused instead of re-encoded #926
New issue
Have a question about this project? Sign up for a free GitHub account to open an issue and contact its maintainers and the community.
By clicking “Sign up for GitHub”, you agree to our terms of service and privacy statement. We’ll occasionally send you account related emails.
Already on GitHub? Sign in to your account
base: master
Are you sure you want to change the base?
Changes from all commits
47c611f
0391aa3
0ed51ec
2f4b235
74a339d
bd0e9b4
File filter
Filter by extension
Conversations
Jump to
Diff view
Diff view
There are no files selected for viewing
| Original file line number | Diff line number | Diff line change |
|---|---|---|
|
|
@@ -1019,3 +1019,18 @@ def get_embeddings(txt): | |
| from sentence_transformers import SentenceTransformer | ||
| model = SentenceTransformer(settings.LM_MODEL_NAME) | ||
| return model.encode(str(txt)) | ||
|
|
||
|
|
||
| def encode_texts(texts): | ||
| """Embeddings for several texts, from one batched model call (OpenConceptLab/ocl_online#247).""" | ||
| texts = [str(text) for text in texts] | ||
| if not texts: | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. this |
||
| return [] | ||
| if settings.ENV == 'ci': | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. this check should be before
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. should we also add "demo" env check here -- it was overridden in settings.py by making it default constant |
||
| return [None] * len(texts) | ||
|
|
||
| model = settings.LM | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. 1032-1035 are common with |
||
| if not model: | ||
| from sentence_transformers import SentenceTransformer | ||
| model = SentenceTransformer(settings.LM_MODEL_NAME) | ||
| return list(model.encode(texts, batch_size=settings.LM_ENCODE_BATCH_SIZE)) | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. wouldn't want to mutate |
||
| Original file line number | Diff line number | Diff line change |
|---|---|---|
| @@ -1,10 +1,23 @@ | ||
| from itertools import islice | ||
|
|
||
| from django.conf import settings | ||
| from django_elasticsearch_dsl import Document, fields | ||
| from django_elasticsearch_dsl.registries import registry | ||
| from pydash import compact, get | ||
|
|
||
| from core.common.utils import jsonify_safe, flatten_extras, get_embeddings, drop_version | ||
| from core.common.utils import jsonify_safe, flatten_extras, drop_version | ||
| from core.concepts.embeddings import ConceptVectors, needs_vectors | ||
| from core.concepts.models import Concept | ||
|
|
||
| # The text a vector encoded, kept in _source only: reuse reads it back, nothing searches it (ocl_online#247) | ||
| EMBEDDING_TEXT = {"type": "keyword", "index": False, "doc_values": False} | ||
| # What ocl_online#247 adds to an existing concepts index's mapping (the concept_vector_mapping command) | ||
| VECTOR_PROVENANCE_MAPPING = { | ||
| '_embeddings': {'type': 'nested', 'properties': {'text': EMBEDDING_TEXT}}, | ||
| '_synonyms_embeddings': {'type': 'nested', 'properties': {'text': EMBEDDING_TEXT}}, | ||
| '_embeddings_model': {'type': 'keyword'}, | ||
| } | ||
|
|
||
|
|
||
| @registry.register_document | ||
| class ConceptDocument(Document): | ||
|
|
@@ -65,7 +78,8 @@ class Index: | |
| }, | ||
| "type": { | ||
| "type": "text" | ||
| } | ||
| }, | ||
| "text": EMBEDDING_TEXT, | ||
| } | ||
| ) | ||
| _synonyms_embeddings = fields.NestedField( | ||
|
|
@@ -75,9 +89,17 @@ class Index: | |
| }, | ||
| "type": { | ||
| "type": "text" | ||
| } | ||
| }, | ||
| "text": EMBEDDING_TEXT, | ||
| } | ||
| ) | ||
| _embeddings_model = fields.KeywordField() | ||
|
|
||
| VECTOR_CHUNK_SIZE = 100 | ||
| _vectors = None # the ConceptVectors of the chunk being prepared, see _get_actions | ||
| vectors_reused = 0 # over this document instance's _get_actions calls | ||
| texts_encoded = 0 | ||
| _source_versions = None # (version, match_algorithms) of each repo version the row being prepared belongs to | ||
|
|
||
| class Django: | ||
| model = Concept | ||
|
|
@@ -168,9 +190,9 @@ def prepare_numeric_id(instance): | |
| def prepare_locale(instance): | ||
| return compact(set(instance.active_names.values_list('locale', flat=True))) | ||
|
|
||
| @staticmethod | ||
| def prepare_source_version(instance): | ||
| return list(instance.sources.values_list('version', flat=True)) | ||
| def prepare_source_version(self, instance): | ||
| versions = self._source_versions or instance.sources.values_list('version', 'match_algorithms') | ||
| return [version for version, _ in versions] | ||
|
|
||
| @staticmethod | ||
| def prepare_extras(instance): | ||
|
|
@@ -209,8 +231,33 @@ def prepare_description_types(instance): | |
| def prepare_description(instance): | ||
| return '. '.join(compact(set(instance.active_descriptions.values_list('name', flat=True)))) | ||
|
|
||
| def _get_actions(self, object_list, action): | ||
| """ | ||
| As django_elasticsearch_dsl's, but prepares 100 docs at a time, so that their vectors are resolved together | ||
| (ConceptVectors): reused where the stored docs already hold them, and the rest encoded in one batched call. | ||
| """ | ||
| if action == 'delete': | ||
| yield from super()._get_actions(object_list, action) | ||
| return | ||
| objects = iter(object_list) | ||
| while chunk := list(islice(objects, self.VECTOR_CHUNK_SIZE)): | ||
| self._vectors = ConceptVectors(self._index._name) | ||
| try: | ||
| actions = [self._prepare_action(obj, action) for obj in chunk if self.should_index_object(obj)] | ||
|
Contributor
There was a problem hiding this comment. Choose a reason for hiding this commentThe reason will be displayed to describe this comment to others. Learn more. can use |
||
| self._vectors.resolve() | ||
| self.vectors_reused += self._vectors.reused | ||
| self.texts_encoded += self._vectors.encoded | ||
| finally: | ||
| self._vectors = None | ||
| yield from actions | ||
|
|
||
| def prepare(self, instance): | ||
| data = super().prepare(instance) | ||
| self._source_versions = list(instance.sources.values_list('version', 'match_algorithms')) | ||
| try: | ||
| data = super().prepare(instance) | ||
| finally: | ||
| versions_match_algorithms = [match_algorithms for _, match_algorithms in self._source_versions] | ||
| self._source_versions = None | ||
| same_as_mapped_codes, other_mapped_codes, verbose_info = self.get_mapped_codes(instance) | ||
| data['same_as_map_codes'] = same_as_mapped_codes | ||
| data['other_map_codes'] = other_mapped_codes | ||
|
|
@@ -224,19 +271,8 @@ def prepare(self, instance): | |
| data['synonyms'] = compact(set(n.name for n in synonyms)) | ||
| data['_synonyms'] = data['synonyms'] | ||
|
|
||
| if instance.parent.has_semantic_match_algorithm: | ||
| data['_embeddings'] = { | ||
| 'vector': get_embeddings(name), | ||
| 'type': get(preferred_locale, 'type'), | ||
| 'locale': get(preferred_locale, 'locale') | ||
| } | ||
| data['_synonyms_embeddings'] = [ | ||
| { | ||
| 'vector': get_embeddings(s.name), | ||
| 'type': get(s, 'type'), | ||
| 'locale': get(s, 'locale') | ||
| } for s in synonyms | ||
| ] | ||
| if needs_vectors(instance.parent, versions_match_algorithms): | ||
| self.add_vectors(data, instance, name, preferred_locale, synonyms) | ||
|
|
||
| expansions = list(instance.expansion_set.only('mnemonic', 'uri')) | ||
| data['expansion'] = [e.mnemonic for e in expansions] | ||
|
|
@@ -248,6 +284,24 @@ def prepare(self, instance): | |
|
|
||
| return data | ||
|
|
||
| def add_vectors(self, data, instance, name, preferred_locale, synonyms): # pylint: disable=too-many-arguments | ||
| """ | ||
| Vector entries for the display name and each synonym, each with the text it encodes, and the model on the doc. | ||
| Their vectors come from the chunk's ConceptVectors once every doc in it is prepared -- or, for a doc prepared | ||
| on its own, right away. | ||
| """ | ||
| data['_embeddings'] = { | ||
| 'vector': None, 'text': name, 'type': get(preferred_locale, 'type'), 'locale': get(preferred_locale, 'locale') | ||
| } | ||
| data['_synonyms_embeddings'] = [ | ||
| {'vector': None, 'text': s.name, 'type': get(s, 'type'), 'locale': get(s, 'locale')} for s in synonyms | ||
| ] | ||
| data['_embeddings_model'] = settings.LM_MODEL_NAME | ||
| vectors = self._vectors or ConceptVectors(self._index._name) | ||
| vectors.add(instance, [data['_embeddings'], *data['_synonyms_embeddings']]) | ||
| if self._vectors is None: | ||
| vectors.resolve() | ||
|
|
||
| @staticmethod | ||
| def get_mapped_codes(instance): | ||
| mappings = instance.get_unidirectional_mappings() | ||
|
|
||
There was a problem hiding this comment.
Choose a reason for hiding this comment
The reason will be displayed to describe this comment to others. Learn more.
This is probably a defensive check -- a new version is copying concepts from HEAD which should already be indexed/vectorised.