♻️(backend) index the content of a document from updated_content endpoint

the search indexer reads it with `YHubService`, and the indexation of an
edited document is triggered by the `content-updated` call the collaboration
server makes — nothing else sees the content change anymore. It is queued as
a celery task, throttled like the other updates, so no indexation ever runs
in the process serving the request. A document whose content cannot be read
is left out of the batch rather than indexed empty, which would have erased
it from the search backend
This commit is contained in:
Manuel Raynaud
2026-08-13 14:32:10 +02:00
parent 57065e8845
commit e7982328eb
11 changed files with 233 additions and 60 deletions
+8 -3
View File
@@ -23,12 +23,17 @@ def base64_yjs_to_xml(base64_string):
return yjs_to_xml(base64.b64decode(base64_string))
def yjs_to_text(update):
"""Extract text from a raw yjs update."""
soup = BeautifulSoup(yjs_to_xml(update), "lxml-xml")
return soup.get_text(separator=" ", strip=True)
def base64_yjs_to_text(base64_string):
"""Extract text from base64 yjs document."""
blocknote_structure = base64_yjs_to_xml(base64_string)
soup = BeautifulSoup(blocknote_structure, "lxml-xml")
return soup.get_text(separator=" ", strip=True)
return yjs_to_text(base64.b64decode(base64_string))
def extract_attachments(content):