♻️(backend) index the content of a document from updated_content endpoint

the search indexer reads it with `YHubService`, and the indexation of an
edited document is triggered by the `content-updated` call the collaboration
server makes — nothing else sees the content change anymore. It is queued as
a celery task, throttled like the other updates, so no indexation ever runs
in the process serving the request. A document whose content cannot be read
is left out of the batch rather than indexed empty, which would have erased
it from the search backend
This commit is contained in:
Manuel Raynaud
2026-08-13 14:32:10 +02:00
parent 57065e8845
commit e7982328eb
11 changed files with 233 additions and 60 deletions
+14 -1
View File
@@ -9,6 +9,7 @@ import pytest
import responses
from core import factories
from core.services.yhub_services import YHubService
from core.tests.utils.urls import reload_urls
USER = "user"
@@ -35,6 +36,11 @@ def mock_user_teams():
def indexer_settings_fixture(settings):
"""
Setup valid settings for the document indexer. Clear the indexer cache.
The indexer reads the content of a document from the collaboration server,
which is faked here: it serves what the factories wrote in the database, so
a document built with `content=""` is one the collaboration server holds no
content for.
"""
# pylint: disable-next=import-outside-toplevel
@@ -50,7 +56,14 @@ def indexer_settings_fixture(settings):
settings.SEARCH_URL = "http://localhost:8081/api/v1.0/documents/search/"
settings.SEARCH_INDEXER_COUNTDOWN = 1
yield settings
def get_ydoc(_service, document):
"""Answer the raw update the collaboration server would serve."""
return base64.b64decode(document.content) if document.content else None
with mock.patch.object(
YHubService, "get_ydoc", autospec=True, side_effect=get_ydoc
):
yield settings
# clear cache to prevent issues with other tests
get_document_indexer.cache_clear()