mirror of
https://github.com/MODSetter/SurfSense.git
synced 2026-07-22 23:31:12 +02:00
refactor(connectors): remove legacy web-crawler KB indexer paths (keep enums)
This commit is contained in:
parent
e3ed3b2be3
commit
ab6be6cbda
14 changed files with 4 additions and 1456 deletions
|
|
@ -22,7 +22,6 @@ CONNECTOR_TASK_MAP = {
|
|||
SearchSourceConnectorType.GITHUB_CONNECTOR: "index_github_repos",
|
||||
SearchSourceConnectorType.CONFLUENCE_CONNECTOR: "index_confluence_pages",
|
||||
SearchSourceConnectorType.ELASTICSEARCH_CONNECTOR: "index_elasticsearch_documents",
|
||||
SearchSourceConnectorType.WEBCRAWLER_CONNECTOR: "index_crawled_urls",
|
||||
SearchSourceConnectorType.BOOKSTACK_CONNECTOR: "index_bookstack_pages",
|
||||
}
|
||||
|
||||
|
|
@ -54,20 +53,6 @@ def create_periodic_schedule(
|
|||
True if successful, False otherwise
|
||||
"""
|
||||
try:
|
||||
# Special handling for connectors that require config validation
|
||||
if connector_type == SearchSourceConnectorType.WEBCRAWLER_CONNECTOR:
|
||||
from app.utils.webcrawler_utils import parse_webcrawler_urls
|
||||
|
||||
config = connector_config or {}
|
||||
urls = parse_webcrawler_urls(config.get("INITIAL_URLS"))
|
||||
|
||||
if not urls:
|
||||
logger.info(
|
||||
f"Webcrawler connector {connector_id} has no URLs configured, "
|
||||
"skipping first indexing run (will run when URLs are added)"
|
||||
)
|
||||
return True # Return success - schedule is created, just no first run
|
||||
|
||||
logger.info(
|
||||
f"Periodic indexing enabled for connector {connector_id} "
|
||||
f"(frequency: {frequency_minutes} minutes). Triggering first run..."
|
||||
|
|
@ -76,7 +61,6 @@ def create_periodic_schedule(
|
|||
from app.tasks.celery_tasks.connector_tasks import (
|
||||
index_bookstack_pages_task,
|
||||
index_confluence_pages_task,
|
||||
index_crawled_urls_task,
|
||||
index_elasticsearch_documents_task,
|
||||
index_github_repos_task,
|
||||
index_notion_pages_task,
|
||||
|
|
@ -87,7 +71,6 @@ def create_periodic_schedule(
|
|||
SearchSourceConnectorType.GITHUB_CONNECTOR: index_github_repos_task,
|
||||
SearchSourceConnectorType.CONFLUENCE_CONNECTOR: index_confluence_pages_task,
|
||||
SearchSourceConnectorType.ELASTICSEARCH_CONNECTOR: index_elasticsearch_documents_task,
|
||||
SearchSourceConnectorType.WEBCRAWLER_CONNECTOR: index_crawled_urls_task,
|
||||
SearchSourceConnectorType.BOOKSTACK_CONNECTOR: index_bookstack_pages_task,
|
||||
}
|
||||
|
||||
|
|
|
|||
|
|
@ -469,14 +469,6 @@ def validate_connector_config(
|
|||
if not isinstance(value, list) or not value:
|
||||
raise ValueError(f"{field_name} must be a non-empty list of strings")
|
||||
|
||||
def validate_initial_urls() -> None:
|
||||
initial_urls = config.get("INITIAL_URLS", "")
|
||||
if initial_urls and initial_urls.strip():
|
||||
urls = [url.strip() for url in initial_urls.split("\n") if url.strip()]
|
||||
for url in urls:
|
||||
if not validators.url(url):
|
||||
raise ValueError(f"Invalid URL format in INITIAL_URLS: {url}")
|
||||
|
||||
# Lookup table for connector validation rules
|
||||
connector_rules = {
|
||||
"SERPER_API": {"required": ["SERPER_API_KEY"], "validators": {}},
|
||||
|
|
@ -562,13 +554,6 @@ def validate_connector_config(
|
|||
# "validators": {}
|
||||
# },
|
||||
"LUMA_CONNECTOR": {"required": ["LUMA_API_KEY"], "validators": {}},
|
||||
"WEBCRAWLER_CONNECTOR": {
|
||||
"required": [], # No required fields - URLs are optional
|
||||
"optional": ["INITIAL_URLS"],
|
||||
"validators": {
|
||||
"INITIAL_URLS": lambda: validate_initial_urls(),
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
rules = connector_rules.get(connector_type_str)
|
||||
|
|
|
|||
|
|
@ -1,28 +0,0 @@
|
|||
"""
|
||||
Utility functions for webcrawler connector.
|
||||
"""
|
||||
|
||||
|
||||
def parse_webcrawler_urls(initial_urls: str | list | None) -> list[str]:
|
||||
"""
|
||||
Parse URLs from webcrawler INITIAL_URLS value.
|
||||
|
||||
Handles both string (newline-separated) and list formats.
|
||||
|
||||
Args:
|
||||
initial_urls: The INITIAL_URLS value (string, list, or None)
|
||||
|
||||
Returns:
|
||||
List of parsed, stripped, non-empty URLs
|
||||
"""
|
||||
if initial_urls is None:
|
||||
return []
|
||||
|
||||
if isinstance(initial_urls, str):
|
||||
return [url.strip() for url in initial_urls.split("\n") if url.strip()]
|
||||
elif isinstance(initial_urls, list):
|
||||
return [
|
||||
url.strip() for url in initial_urls if isinstance(url, str) and url.strip()
|
||||
]
|
||||
else:
|
||||
return []
|
||||
Loading…
Add table
Add a link
Reference in a new issue