feat(backend): Use deterministic content in connector ingestion

2026-07-22 23:31:12 +02:00 · 2026-06-04 00:51:38 +05:30 · 2026-06-04 00:51:38 +05:30 · f3866b9e7e
commit f3866b9e7e
parent 81fa219b30
24 changed files with 80 additions and 625 deletions
--- a/surfsense_backend/app/services/confluence/kb_sync_service.py
+++ b/surfsense_backend/app/services/confluence/kb_sync_service.py
@ -9,7 +9,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
    generate_unique_identifier_hash,
 )

@ -65,29 +64,11 @@ class ConfluenceKBSyncService:
            if dup:
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
-            )

-            doc_metadata_for_summary = {
-                "page_title": page_title,
-                "space_id": space_id,
-                "document_type": "Confluence Page",
-                "connector_type": "Confluence",
-            }

-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    page_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                summary_content = f"Confluence Page: {page_title}\n\n{page_content}"
-                summary_embedding = embed_text(summary_content)
+            summary_content = f"Confluence Page: {page_title}\n\n{page_content}"
+            summary_embedding = embed_text(summary_content)

            chunks = await create_document_chunks(page_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
@ -185,25 +166,10 @@ class ConfluenceKBSyncService:

            space_id = (document.document_metadata or {}).get("space_id", "")

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session, user_id, search_space_id, disable_streaming=True
-            )

-            if user_llm:
-                doc_meta = {
-                    "page_title": page_title,
-                    "space_id": space_id,
-                    "document_type": "Confluence Page",
-                    "connector_type": "Confluence",
-                }
-                summary_content, summary_embedding = await generate_document_summary(
-                    page_content, user_llm, doc_meta
-                )
-            else:
-                summary_content = f"Confluence Page: {page_title}\n\n{page_content}"
-                summary_embedding = embed_text(summary_content)
+            summary_content = f"Confluence Page: {page_title}\n\n{page_content}"
+            summary_embedding = embed_text(summary_content)

            chunks = await create_document_chunks(page_content)

--- a/surfsense_backend/app/services/dropbox/kb_sync_service.py
+++ b/surfsense_backend/app/services/dropbox/kb_sync_service.py
@ -9,7 +9,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
 )

 logger = logging.getLogger(__name__)
@ -72,29 +71,11 @@ class DropboxKBSyncService:
                )
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
-            )

-            doc_metadata_for_summary = {
-                "file_name": file_name,
-                "document_type": "Dropbox File",
-                "connector_type": "Dropbox",
-            }

-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    indexable_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                logger.warning("No LLM configured — using fallback summary")
-                summary_content = f"Dropbox File: {file_name}\n\n{indexable_content}"
-                summary_embedding = embed_text(summary_content)
+            summary_content = f"Dropbox File: {file_name}\n\n{indexable_content}"
+            summary_embedding = embed_text(summary_content)

            chunks = await create_document_chunks(indexable_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
--- a/surfsense_backend/app/services/gmail/kb_sync_service.py
+++ b/surfsense_backend/app/services/gmail/kb_sync_service.py
@ -9,7 +9,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
    generate_unique_identifier_hash,
 )

@ -78,30 +77,11 @@ class GmailKBSyncService:
                )
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
-            )

-            doc_metadata_for_summary = {
-                "subject": subject,
-                "sender": sender,
-                "document_type": "Gmail Message",
-                "connector_type": "Gmail",
-            }

-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    indexable_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                logger.warning("No LLM configured -- using fallback summary")
-                summary_content = f"Gmail Message: {subject}\n\n{indexable_content}"
-                summary_embedding = await asyncio.to_thread(embed_text, summary_content)
+            summary_content = f"Gmail Message: {subject}\n\n{indexable_content}"
+            summary_embedding = await asyncio.to_thread(embed_text, summary_content)

            chunks = await create_document_chunks(indexable_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
--- a/surfsense_backend/app/services/google_calendar/kb_sync_service.py
+++ b/surfsense_backend/app/services/google_calendar/kb_sync_service.py
@ -19,7 +19,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
    generate_unique_identifier_hash,
 )

@ -90,33 +89,13 @@ class GoogleCalendarKBSyncService:
                )
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
+
+
+            summary_content = (
+                f"Google Calendar Event: {event_summary}\n\n{indexable_content}"
            )
-
-            doc_metadata_for_summary = {
-                "event_summary": event_summary,
-                "start_time": start_time,
-                "end_time": end_time,
-                "document_type": "Google Calendar Event",
-                "connector_type": "Google Calendar",
-            }
-
-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    indexable_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                logger.warning("No LLM configured -- using fallback summary")
-                summary_content = (
-                    f"Google Calendar Event: {event_summary}\n\n{indexable_content}"
-                )
-                summary_embedding = await asyncio.to_thread(embed_text, summary_content)
+            summary_embedding = await asyncio.to_thread(embed_text, summary_content)

            chunks = await create_document_chunks(indexable_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
@ -273,29 +252,13 @@ class GoogleCalendarKBSyncService:
            if not indexable_content:
                return {"status": "error", "message": "Event produced empty content"}

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session, user_id, search_space_id, disable_streaming=True
+
+
+            summary_content = (
+                f"Google Calendar Event: {event_summary}\n\n{indexable_content}"
            )
-
-            doc_metadata_for_summary = {
-                "event_summary": event_summary,
-                "start_time": start_time,
-                "end_time": end_time,
-                "document_type": "Google Calendar Event",
-                "connector_type": "Google Calendar",
-            }
-
-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    indexable_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                summary_content = (
-                    f"Google Calendar Event: {event_summary}\n\n{indexable_content}"
-                )
-                summary_embedding = await asyncio.to_thread(embed_text, summary_content)
+            summary_embedding = await asyncio.to_thread(embed_text, summary_content)

            chunks = await create_document_chunks(indexable_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
--- a/surfsense_backend/app/services/google_drive/kb_sync_service.py
+++ b/surfsense_backend/app/services/google_drive/kb_sync_service.py
@ -8,7 +8,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
    generate_unique_identifier_hash,
 )

@ -74,32 +73,13 @@ class GoogleDriveKBSyncService:
                )
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
+
+
+            summary_content = (
+                f"Google Drive File: {file_name}\n\n{indexable_content}"
            )
-
-            doc_metadata_for_summary = {
-                "file_name": file_name,
-                "mime_type": mime_type,
-                "document_type": "Google Drive File",
-                "connector_type": "Google Drive",
-            }
-
-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    indexable_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                logger.warning("No LLM configured — using fallback summary")
-                summary_content = (
-                    f"Google Drive File: {file_name}\n\n{indexable_content}"
-                )
-                summary_embedding = embed_text(summary_content)
+            summary_embedding = embed_text(summary_content)

            chunks = await create_document_chunks(indexable_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
--- a/surfsense_backend/app/services/linear/kb_sync_service.py
+++ b/surfsense_backend/app/services/linear/kb_sync_service.py
@ -9,7 +9,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
    generate_unique_identifier_hash,
 )

@ -84,32 +83,13 @@ class LinearKBSyncService:
                )
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
+
+
+            summary_content = (
+                f"Linear Issue {issue_identifier}: {issue_title}\n\n{issue_content}"
            )
-
-            doc_metadata_for_summary = {
-                "issue_id": issue_identifier,
-                "issue_title": issue_title,
-                "document_type": "Linear Issue",
-                "connector_type": "Linear",
-            }
-
-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    issue_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                logger.warning("No LLM configured — using fallback summary")
-                summary_content = (
-                    f"Linear Issue {issue_identifier}: {issue_title}\n\n{issue_content}"
-                )
-                summary_embedding = embed_text(summary_content)
+            summary_embedding = embed_text(summary_content)

            chunks = await create_document_chunks(issue_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
@ -227,30 +207,12 @@ class LinearKBSyncService:
            comment_count = len(formatted_issue.get("comments", []))
            formatted_issue.get("description", "")

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session, user_id, search_space_id, disable_streaming=True
+
+            summary_content = (
+                f"Linear Issue {issue_identifier}: {issue_title}\n\n{issue_content}"
            )
-
-            if user_llm:
-                document_metadata_for_summary = {
-                    "issue_id": issue_identifier,
-                    "issue_title": issue_title,
-                    "state": state,
-                    "priority": priority,
-                    "comment_count": comment_count,
-                    "document_type": "Linear Issue",
-                    "connector_type": "Linear",
-                }
-                summary_content, summary_embedding = await generate_document_summary(
-                    issue_content, user_llm, document_metadata_for_summary
-                )
-            else:
-                summary_content = (
-                    f"Linear Issue {issue_identifier}: {issue_title}\n\n{issue_content}"
-                )
-                summary_embedding = embed_text(summary_content)
+            summary_embedding = embed_text(summary_content)

            chunks = await create_document_chunks(issue_content)

--- a/surfsense_backend/app/services/notion/kb_sync_service.py
+++ b/surfsense_backend/app/services/notion/kb_sync_service.py
@ -8,7 +8,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
    generate_unique_identifier_hash,
 )

@ -73,30 +72,11 @@ class NotionKBSyncService:
                )
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
-            )

-            doc_metadata_for_summary = {
-                "page_title": page_title,
-                "page_id": page_id,
-                "document_type": "Notion Page",
-                "connector_type": "Notion",
-            }

-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    markdown_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                logger.warning("No LLM configured — using fallback summary")
-                summary_content = f"Notion Page: {page_title}\n\n{markdown_content}"
-                summary_embedding = embed_text(summary_content)
+            summary_content = f"Notion Page: {page_title}\n\n{markdown_content}"
+            summary_embedding = embed_text(summary_content)

            chunks = await create_document_chunks(markdown_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
@ -245,31 +225,11 @@ class NotionKBSyncService:
                f"Final content length: {len(full_content)} chars, verified={content_verified}"
            )

-            from app.services.llm_service import get_user_long_context_llm

            logger.debug("Generating summary and embeddings")
-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,  # disable streaming to avoid leaking into the chat
-            )

-            if user_llm:
-                document_metadata_for_summary = {
-                    "page_title": document.document_metadata.get("page_title"),
-                    "page_id": document.document_metadata.get("page_id"),
-                    "document_type": "Notion Page",
-                    "connector_type": "Notion",
-                }
-                summary_content, summary_embedding = await generate_document_summary(
-                    full_content, user_llm, document_metadata_for_summary
-                )
-                logger.debug(f"Generated summary length: {len(summary_content)} chars")
-            else:
-                logger.warning("No LLM configured - using fallback summary")
-                summary_content = f"Notion Page: {document.document_metadata.get('page_title')}\n\n{full_content}"
-                summary_embedding = embed_text(summary_content)
+            summary_content = f"Notion Page: {document.document_metadata.get('page_title')}\n\n{full_content}"
+            summary_embedding = embed_text(summary_content)

            logger.debug("Creating new chunks")
            chunks = await create_document_chunks(full_content)
--- a/surfsense_backend/app/services/onedrive/kb_sync_service.py
+++ b/surfsense_backend/app/services/onedrive/kb_sync_service.py
@ -10,7 +10,6 @@ from app.utils.document_converters import (
    create_document_chunks,
    embed_text,
    generate_content_hash,
-    generate_document_summary,
 )

 logger = logging.getLogger(__name__)
@ -73,30 +72,11 @@ class OneDriveKBSyncService:
                )
                content_hash = unique_hash

-            from app.services.llm_service import get_user_long_context_llm

-            user_llm = await get_user_long_context_llm(
-                self.db_session,
-                user_id,
-                search_space_id,
-                disable_streaming=True,
-            )

-            doc_metadata_for_summary = {
-                "file_name": file_name,
-                "mime_type": mime_type,
-                "document_type": "OneDrive File",
-                "connector_type": "OneDrive",
-            }

-            if user_llm:
-                summary_content, summary_embedding = await generate_document_summary(
-                    indexable_content, user_llm, doc_metadata_for_summary
-                )
-            else:
-                logger.warning("No LLM configured — using fallback summary")
-                summary_content = f"OneDrive File: {file_name}\n\n{indexable_content}"
-                summary_embedding = await asyncio.to_thread(embed_text, summary_content)
+            summary_content = f"OneDrive File: {file_name}\n\n{indexable_content}"
+            summary_embedding = await asyncio.to_thread(embed_text, summary_content)

            chunks = await create_document_chunks(indexable_content)
            now_str = datetime.now().strftime("%Y-%m-%d %H:%M:%S")