🚨(backend) various fixes

I am proofreading myself
This commit is contained in:
charles
2026-01-20 19:07:58 +01:00
parent 40d1d8cc24
commit f88a80d93f
8 changed files with 47 additions and 24 deletions
+4
View File
@@ -8,6 +8,10 @@ and this project adheres to
## [Unreleased]
### Added
- ✨(backend) add FindRagBackend
### Removed
- 🔥(chat) consider PDF documents as other kind of documents #234
+3
View File
@@ -95,6 +95,9 @@ These are the environment variables you can set for the `conversations-backend`
| CACHES_KEY_PREFIX | The prefix used to every cache keys. | conversations |
| THEME_CUSTOMIZATION_FILE_PATH | full path to the file customizing the theme. An example is provided in src/backend/conversations/configuration/theme/default.json | BASE_DIR/conversations/configuration/theme/default.json |
| THEME_CUSTOMIZATION_CACHE_TIMEOUT | Cache duration for the customization settings | 86400 |
| FIND_API_KEY | API key of Find | |
| FIND_API_URL | URL of Find | https://app-find/api |
| FIND_API_TIMEOUT | Find API timeout | 30 |
## conversations-frontend image
+1 -1
View File
@@ -357,7 +357,7 @@ The RAG backend performs semantic search to find the most relevant content:
rag_results = document_store.search(
query,
results_count=settings.BRAVE_RAG_WEB_SEARCH_CHUNK_NUMBER,
**kwargs,
**kwargs, # Additional search parameters like session with access_token
)
```
@@ -38,7 +38,7 @@ class AlbertParser(BaseParser):
endpoint = urljoin(settings.ALBERT_API_URL, "/v1/parse-beta")
def parse_pdf_document(self, name: str, content_type: str, content: BytesIO) -> str:
def parse_pdf_document(self, name: str, content_type: str, content: bytes) -> str:
"""Parse PDF document using Albert API."""
response = requests.post(
self.endpoint,
@@ -57,7 +57,7 @@ class AlbertParser(BaseParser):
document_page["content"] for document_page in response.json().get("data", [])
)
def parse_document(self, name: str, content_type: str, content: BytesIO) -> str:
def parse_document(self, name: str, content_type: str, content: bytes) -> str:
"""Parse document based on content type."""
if content_type == "application/pdf":
return self.parse_pdf_document(name=name, content_type=content_type, content=content)
@@ -26,9 +26,6 @@ class AlbertRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-att
It provides methods to:
- Create a collection for the search operation.
- Parse documents and convert them to Markdown format:
+ Handle PDF parsing using the Albert API.
+ Use the DocumentConverter (markitdown) for other formats.
- Store parsed documents in the Albert collection.
- Perform a search operation using the Albert API.
"""
@@ -213,6 +210,7 @@ class AlbertRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-att
Args:
query (str): The search query.
results_count (int): The number of results to return.
**kwargs: Additional arguments.
Returns:
RAGWebResults: The search results.
@@ -7,28 +7,27 @@ from urllib.parse import urljoin
from uuid import uuid4
from django.conf import settings
from django.core.exceptions import ImproperlyConfigured
from django.utils import timezone
import requests
from chat.agent_rag.constants import RAGWebResult, RAGWebResults, RAGWebUsage
from chat.agent_rag.document_rag_backends.albert_rag_backend import AlbertParser
from chat.agent_rag.document_converter.parser import AlbertParser
from chat.agent_rag.document_rag_backends.base_rag_backend import BaseRagBackend
from utils.oidc import with_fresh_access_token
logger = logging.getLogger(__name__)
class FindRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-attributes
SUPPORTED_LANGUAGE_CODES = ["en", "fr", "de", "nl"]
class FindRagBackend(BaseRagBackend):
"""
This class is a placeholder for the Find API implementation.
It is designed to be used with the RAG (Retrieval-Augmented Generation) document search system.
It provides methods to:
- Parse documents and convert them to Markdown format:
+ Handle PDF parsing using the Albert API.
+ Use the DocumentConverter (markitdown) for other formats.
- Store parsed documents in the Find index.
- Perform a search operation using the Find API.
"""
@@ -43,12 +42,9 @@ class FindRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-attri
self.api_key = settings.FIND_API_KEY
self.search_endpoint = "api/v1.0/documents/search/"
self.indexing_endpoint = "api/v1.0/documents/index/"
self.parser = AlbertParser()
self.parser = AlbertParser() # Find Rag relies on Albert parser
if not self.api_key:
raise ImproperlyConfigured("FIND_API_KEY must be set in Django settings.")
def create_collection(self, name: str, description: Optional[str] = None) -> uuid.UUID:
def create_collection(self, name: str, description: Optional[str] = None) -> str:
"""
init collection_id
"""
@@ -71,6 +67,11 @@ class FindRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-attri
user_sub (str): The user subject identifier for access control.
"""
logger.debug("index document '%s' in Find", name)
user_sub = kwargs.get("user_sub")
if not user_sub:
raise ValueError("user_sub is required to store document in FindRagBackend")
response = requests.post(
urljoin(settings.FIND_API_URL, self.indexing_endpoint),
headers={"Authorization": f"Bearer {self.api_key}"},
@@ -85,7 +86,7 @@ class FindRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-attri
"updated_at": timezone.now().isoformat(),
"tags": [f"collection-{self.collection_id}"],
"size": len(content.encode("utf-8")),
"users": [kwargs["user_sub"]] if "user_sub" in kwargs else [],
"users": [user_sub],
"groups": [],
"reach": "authenticated",
"is_active": True,
@@ -122,14 +123,13 @@ class FindRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-attri
},
timeout=settings.FIND_API_TIMEOUT,
)
logger.debug(response.json())
response.raise_for_status()
return RAGWebResults(
data=[
RAGWebResult(
url=result["_source"]["title.fr"],
content=result["_source"]["content.fr"],
url=get_language_value(result["_source"], "title"),
content=get_language_value(result["_source"], "content"),
score=result["_score"],
)
for result in response.json()
@@ -139,3 +139,15 @@ class FindRagBackend(BaseRagBackend): # pylint: disable=too-many-instance-attri
completion_tokens=0,
),
)
def get_language_value(source, language_field):
"""
extract the value of the language field with the correct language_code extension.
"title" and "content" have extensions like "title.en" or "title.fr".
get_language_value will return the value regardless of the extension.
"""
for language_code in SUPPORTED_LANGUAGE_CODES:
if f"{language_field}.{language_code}" in source:
return source[f"{language_field}.{language_code}"]
raise ValueError(f"No '{language_field}' field with any supported language code in object")
@@ -7,8 +7,8 @@ import json
import logging
from io import BytesIO
from unittest import mock
from unittest.mock import Mock
from django.contrib.sessions.backends.cache import SessionStore
from django.utils import formats, timezone
import httpx
@@ -89,7 +89,9 @@ def mock_process_request():
def mock_refresh_access_token():
"""Mock refresh_access_token to bypass token refresh in tests."""
with mock.patch("utils.oidc.refresh_access_token") as mocked_refresh_access_token:
mocked_refresh_access_token.return_value = Mock(spec=httpx.Client)
session = SessionStore()
session["oidc_access_token"] = "mocked-access-token"
mocked_refresh_access_token.return_value = session
yield mocked_refresh_access_token
+5 -1
View File
@@ -11,13 +11,17 @@ from rest_framework.exceptions import AuthenticationFailed
def refresh_access_token(session):
"""Refresh the OIDC access token using the refresh token."""
refresh_token = get_oidc_refresh_token(session)
if not refresh_token:
raise AuthenticationFailed({"error": "Refresh token is missing from session"})
response = requests.post(
settings.OIDC_OP_TOKEN_ENDPOINT,
data={
"grant_type": "refresh_token",
"client_id": settings.OIDC_RP_CLIENT_ID,
"client_secret": settings.OIDC_RP_CLIENT_SECRET,
"refresh_token": get_oidc_refresh_token(session),
"refresh_token": refresh_token,
},
timeout=5,
)