diff --git a/README.es.md b/README.es.md index 8cdbffbd0..cb21d8cf1 100644 --- a/README.es.md +++ b/README.es.md @@ -22,7 +22,7 @@ # SurfSense: La alternativa de código abierto a NotebookLM para la investigación de la web abierta -SurfSense es la **alternativa de código abierto a NotebookLM para agentes de IA**, una plataforma de investigación de la web abierta con conectores de datos en vivo. Tus agentes investigan la web en vivo con datos estructurados de **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search y cualquier página de la web abierta**, a través de una única **API REST** o un **servidor MCP**. Agentes programados y activados por eventos convierten lo que encuentran en informes y alertas, y una base de conocimiento integrada mantiene cada hallazgo disponible para búsqueda con citas. +SurfSense es la **alternativa de código abierto a NotebookLM para agentes de IA**, una plataforma de investigación de la web abierta con conectores de datos en vivo. Tus agentes investigan la web en vivo con datos estructurados de **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed y cualquier página de la web abierta**, a través de una única **API REST** o un **servidor MCP**. Agentes programados y activados por eventos convierten lo que encuentran en informes y alertas, y una base de conocimiento integrada mantiene cada hallazgo disponible para búsqueda con citas. > [!NOTE] > **📢 Una nota para nuestros usuarios de la alternativa a NotebookLM** @@ -60,6 +60,7 @@ Pregúntale a cualquier agente capaz "¿qué está diciendo Reddit sobre este pr | **TikTok** | Videos, comentarios, hashtags y perfiles sin aprobación de la Research API | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | Lugares, calificaciones y reseñas para investigar negocios locales | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | SERPs en vivo para investigación y monitoreo de búsquedas | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | Ofertas de empleo públicas con salarios y descripciones completas, por búsqueda o empresa | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Amazon** | Datos públicos de productos: precios, calificaciones, ofertas, vendedores y rankings de más vendidos | [Amazon Product API](https://www.surfsense.com/amazon) | | **Web Crawl** (rastreo web) | Cualquier página de la web abierta como contenido limpio y estructurado | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **Conectores MCP externos** | Conecta cualquier servidor MCP a tus agentes, con OAuth de un clic para Notion, Slack, Jira y más | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -217,7 +218,7 @@ SurfSense es el único producto de código abierto que combina un espacio de tra | Característica | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **Datos web en vivo para agentes** | No | Conectores de Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search y rastreo web vía API REST y MCP | +| **Datos web en vivo para agentes** | No | Conectores de Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed y rastreo web vía API REST y MCP | | **Servidor MCP** | No | Cada conector expuesto como herramienta nativa de agente, más servidores MCP propios con aplicaciones OAuth de un clic | | **Fuentes por Notebook** | 50 (gratis) a 600 (Ultra, $249.99/mes) | Ilimitadas | | **Número de Notebooks** | 100 (gratis) a 500 (niveles de pago) | Ilimitado | diff --git a/README.hi.md b/README.hi.md index 1630615af..b7e74d239 100644 --- a/README.hi.md +++ b/README.hi.md @@ -22,7 +22,7 @@ # SurfSense: ओपन वेब रिसर्च के लिए ओपन सोर्स NotebookLM विकल्प -SurfSense **AI एजेंट्स के लिए ओपन सोर्स NotebookLM विकल्प** है, लाइव डेटा कनेक्टर्स के साथ एक ओपन वेब रिसर्च प्लेटफ़ॉर्म। आपके एजेंट **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search और ओपन वेब के किसी भी पेज** से स्ट्रक्चर्ड डेटा के साथ लाइव वेब पर रिसर्च करते हैं, वह भी एक ही **REST API** या **MCP सर्वर** के ज़रिए। शेड्यूल्ड और इवेंट-ट्रिगर्ड एजेंट अपनी खोजों को ब्रीफ़ और अलर्ट में बदलते हैं, और एक बिल्ट-इन नॉलेज बेस हर खोज को साइटेशन के साथ खोजने योग्य बनाए रखता है। +SurfSense **AI एजेंट्स के लिए ओपन सोर्स NotebookLM विकल्प** है, लाइव डेटा कनेक्टर्स के साथ एक ओपन वेब रिसर्च प्लेटफ़ॉर्म। आपके एजेंट **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed और ओपन वेब के किसी भी पेज** से स्ट्रक्चर्ड डेटा के साथ लाइव वेब पर रिसर्च करते हैं, वह भी एक ही **REST API** या **MCP सर्वर** के ज़रिए। शेड्यूल्ड और इवेंट-ट्रिगर्ड एजेंट अपनी खोजों को ब्रीफ़ और अलर्ट में बदलते हैं, और एक बिल्ट-इन नॉलेज बेस हर खोज को साइटेशन के साथ खोजने योग्य बनाए रखता है। > [!NOTE] > **📢 हमारे NotebookLM-विकल्प उपयोगकर्ताओं के लिए एक सूचना** @@ -60,6 +60,7 @@ SurfSense **AI एजेंट्स के लिए ओपन सोर्स | **TikTok** | Research API अप्रूवल के बिना वीडियो, कमेंट, हैशटैग और प्रोफ़ाइल | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | स्थानीय बिज़नेस रिसर्च के लिए स्थान, रेटिंग और रिव्यू | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | सर्च रिसर्च और मॉनिटरिंग के लिए लाइव SERP | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | सार्वजनिक नौकरी लिस्टिंग, सैलरी और पूरे विवरण के साथ, सर्च या कंपनी के अनुसार | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Amazon** | सार्वजनिक प्रोडक्ट डेटा: कीमतें, रेटिंग, ऑफ़र, विक्रेता और बेस्ट-सेलर रैंक | [Amazon Product API](https://www.surfsense.com/amazon) | | **Web Crawl** | ओपन वेब का कोई भी पेज साफ़-सुथरे, स्ट्रक्चर्ड कंटेंट के रूप में | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **External MCP Connectors** | कोई भी MCP सर्वर अपने एजेंट्स से जोड़ें, Notion, Slack, Jira और अन्य के लिए वन-क्लिक OAuth के साथ | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -217,7 +218,7 @@ SurfSense एकमात्र ओपन सोर्स प्रोडक् | फ़ीचर | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **एजेंट्स के लिए लाइव वेब डेटा** | नहीं | REST API और MCP के ज़रिए Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search और वेब क्रॉल कनेक्टर | +| **एजेंट्स के लिए लाइव वेब डेटा** | नहीं | REST API और MCP के ज़रिए Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed और वेब क्रॉल कनेक्टर | | **MCP सर्वर** | नहीं | हर कनेक्टर नेटिव एजेंट टूल के रूप में उपलब्ध, साथ ही वन-क्लिक OAuth ऐप्स के साथ अपने MCP सर्वर लाने की सुविधा | | **प्रति नोटबुक स्रोत** | 50 (Free) से 600 (Ultra, $249.99/माह) | असीमित | | **नोटबुक की संख्या** | 100 (Free) से 500 (सशुल्क टियर) | असीमित | diff --git a/README.md b/README.md index f9b0fd2be..3ef8547a0 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ # SurfSense: The Open-Source NotebookLM Alternative for Open Web Research -SurfSense is the **open-source NotebookLM alternative for AI agents**, an open web research platform with live data connectors. Your agents research the live web with structured data from **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, and any page on the open web**, through one **REST API** or **MCP server**. Scheduled and event-triggered agents turn what they find into briefs and alerts, and a built-in knowledge base keeps every finding searchable with citations. +SurfSense is the **open-source NotebookLM alternative for AI agents**, an open web research platform with live data connectors. Your agents research the live web with structured data from **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed, and any page on the open web**, through one **REST API** or **MCP server**. Scheduled and event-triggered agents turn what they find into briefs and alerts, and a built-in knowledge base keeps every finding searchable with citations. > [!NOTE] > **📢 A note for our NotebookLM-alternative users** @@ -60,6 +60,7 @@ Ask any capable agent "what is Reddit saying about this product since launch?" o | **TikTok** | Videos, comments, hashtags, and profiles without Research API approval | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | Places, ratings, and reviews for local business research | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | Live SERPs for search research and monitoring | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | Public job postings with salaries and full descriptions, by search or company | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Amazon** | Public product data: prices, ratings, offers, sellers, and best-seller ranks | [Amazon Product API](https://www.surfsense.com/amazon) | | **Web Crawl** | Any page on the open web as clean, structured content | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **External MCP Connectors** | Bring any MCP server to your agents, with one-click OAuth for Notion, Slack, Jira, and more | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -217,7 +218,7 @@ Still comparing us as a NotebookLM alternative? Here is the honest breakdown. | Feature | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **Live web data for agents** | No | Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, and web crawl connectors via REST API and MCP | +| **Live web data for agents** | No | Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed, and web crawl connectors via REST API and MCP | | **MCP server** | No | Every connector exposed as a native agent tool, plus bring-your-own MCP servers with one-click OAuth apps | | **Sources per Notebook** | 50 (Free) to 600 (Ultra, $249.99/mo) | Unlimited | | **Number of Notebooks** | 100 (Free) to 500 (paid tiers) | Unlimited | diff --git a/README.pt-BR.md b/README.pt-BR.md index 43070f754..a5c453f7a 100644 --- a/README.pt-BR.md +++ b/README.pt-BR.md @@ -22,7 +22,7 @@ # SurfSense: A Alternativa Open Source ao NotebookLM para Pesquisa na Web Aberta -O SurfSense é a **alternativa open source ao NotebookLM para agentes de IA**, uma plataforma de pesquisa na web aberta com conectores de dados ao vivo. Seus agentes pesquisam a web ao vivo com dados estruturados do **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search e de qualquer página da web aberta**, por meio de uma única **API REST** ou de um **servidor MCP**. Agentes agendados ou acionados por eventos transformam o que encontram em relatórios e alertas, e uma base de conhecimento integrada mantém cada descoberta pesquisável, com citações. +O SurfSense é a **alternativa open source ao NotebookLM para agentes de IA**, uma plataforma de pesquisa na web aberta com conectores de dados ao vivo. Seus agentes pesquisam a web ao vivo com dados estruturados do **Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed e de qualquer página da web aberta**, por meio de uma única **API REST** ou de um **servidor MCP**. Agentes agendados ou acionados por eventos transformam o que encontram em relatórios e alertas, e uma base de conhecimento integrada mantém cada descoberta pesquisável, com citações. > [!NOTE] > **📢 Um recado para nossos usuários que buscavam uma alternativa ao NotebookLM** @@ -60,6 +60,7 @@ Pergunte a qualquer agente capaz "o que o Reddit está dizendo sobre este produt | **TikTok** | Vídeos, comentários, hashtags e perfis sem aprovação da Research API | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | Estabelecimentos, avaliações e reviews para pesquisa de negócios locais | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | SERPs ao vivo para pesquisa e monitoramento de buscas | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | Vagas públicas com salários e descrições completas, por busca ou empresa | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Amazon** | Dados públicos de produtos: preços, avaliações, ofertas, vendedores e rankings de mais vendidos | [Amazon Product API](https://www.surfsense.com/amazon) | | **Web Crawl** (rastreamento web) | Qualquer página da web aberta como conteúdo limpo e estruturado | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **Conectores MCP externos** | Traga qualquer servidor MCP para seus agentes, com OAuth em um clique para Notion, Slack, Jira e outros | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -217,7 +218,7 @@ Ainda nos comparando como alternativa ao NotebookLM? Aqui está o comparativo ho | Recurso | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **Dados da web ao vivo para agentes** | Não | Conectores de Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search e rastreamento web via API REST e MCP | +| **Dados da web ao vivo para agentes** | Não | Conectores de Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, Indeed e rastreamento web via API REST e MCP | | **Servidor MCP** | Não | Cada conector exposto como ferramenta nativa de agente, além de servidores MCP próprios com apps OAuth em um clique | | **Fontes por Notebook** | 50 (gratuito) a 600 (Ultra, US$ 249,99/mês) | Ilimitadas | | **Número de Notebooks** | 100 (gratuito) a 500 (planos pagos) | Ilimitado | diff --git a/README.zh-CN.md b/README.zh-CN.md index 84be75918..9b9de2535 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -22,7 +22,7 @@ # SurfSense:面向开放网络研究的开源 NotebookLM 替代品 -SurfSense 是**面向 AI 智能体的开源 NotebookLM 替代品**,一个配备实时数据连接器的开放网络研究平台。你的智能体可以通过一个 **REST API** 或 **MCP 服务器**,利用来自 **Reddit、YouTube、Instagram、TikTok、Amazon、Google Maps、Google Search 以及开放网络上任意页面**的结构化数据研究实时网络。定时和事件触发的智能体会把发现的内容转化为简报和预警,内置的知识库则让每一条发现都可搜索、可引用。 +SurfSense 是**面向 AI 智能体的开源 NotebookLM 替代品**,一个配备实时数据连接器的开放网络研究平台。你的智能体可以通过一个 **REST API** 或 **MCP 服务器**,利用来自 **Reddit、YouTube、Instagram、TikTok、Amazon、Google Maps、Google Search、Indeed 以及开放网络上任意页面**的结构化数据研究实时网络。定时和事件触发的智能体会把发现的内容转化为简报和预警,内置的知识库则让每一条发现都可搜索、可引用。 > [!NOTE] > **📢 致我们的 NotebookLM 替代品用户** @@ -60,6 +60,7 @@ SurfSense 是**面向 AI 智能体的开源 NotebookLM 替代品**,一个配 | **TikTok** | 视频、评论、话题标签和主页,无需 Research API 审批 | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | 地点、评分和评论,用于本地商户研究 | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | 实时搜索结果页,用于搜索研究和监控 | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | 公开职位信息,含薪资与完整职位描述,按搜索或公司抓取 | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Amazon** | 公开商品数据:价格、评分、报价、卖家和畅销榜排名 | [Amazon Product API](https://www.surfsense.com/amazon) | | **Web Crawl** | 把开放网络上的任意页面转为干净、结构化的内容 | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **外部 MCP 连接器** | 将任意 MCP 服务器接入你的智能体,Notion、Slack、Jira 等支持一键 OAuth | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -217,7 +218,7 @@ SurfSense 是唯一一款把面向人的 NotebookLM 式研究工作区与面向 | 功能 | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **面向智能体的实时网络数据** | 无 | 通过 REST API 和 MCP 提供 Reddit、YouTube、Instagram、TikTok、Amazon、Google Maps、Google Search 和网页爬取连接器 | +| **面向智能体的实时网络数据** | 无 | 通过 REST API 和 MCP 提供 Reddit、YouTube、Instagram、TikTok、Amazon、Google Maps、Google Search、Indeed 和网页爬取连接器 | | **MCP 服务器** | 无 | 每个连接器都作为原生智能体工具暴露,还可自带 MCP 服务器并使用一键 OAuth 应用 | | **每个笔记本的来源数** | 50 个(免费版)至 600 个(Ultra 版,249.99 美元/月) | 无限制 | | **笔记本数量** | 100 个(免费版)至 500 个(付费档位) | 无限制 | diff --git a/VERSION b/VERSION index bb951c884..155069a39 100644 --- a/VERSION +++ b/VERSION @@ -1 +1 @@ -0.0.34 +0.0.35 diff --git a/docker/.env.example b/docker/.env.example index cd9789fae..fb7f5ff42 100644 --- a/docker/.env.example +++ b/docker/.env.example @@ -392,6 +392,15 @@ STT_SERVICE=local/base # OTEL_HTTP_PORT=4318 # OTEL_HEALTH_PORT=13133 +# PostHog product analytics (server-side). Opt-in like OTel: leave +# POSTHOG_API_KEY unset for zero telemetry. Passed to backend/worker/beat via +# env_file, so no compose changes are needed. Use the SAME project key as the +# frontend's NEXT_PUBLIC_POSTHOG_KEY so server events merge onto the persons the +# web app already identifies by user id. +# POSTHOG_API_KEY=phc_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx +# POSTHOG_HOST=https://us.i.posthog.com +# POSTHOG_AI_PRIVACY_MODE=true # false ships LLM prompt/completion bodies to PostHog + # ------------------------------------------------------------------------------ # Advanced (optional) # ------------------------------------------------------------------------------ @@ -441,7 +450,7 @@ SURFSENSE_ENABLE_DOOM_LOOP=true # WEB_CRAWL_CAPTCHA_MICROS_PER_SOLVE=3000 # Debit the credit wallet per *item returned* by the platform-native scrapers -# (Reddit, Google Search, Google Maps, Amazon, YouTube). Default FALSE keeps scraping +# (Reddit, Google Search, Google Maps, Amazon, Walmart, YouTube). Default FALSE keeps scraping # effectively free for self-hosted installs. Each rate is micro-USD per item, # config-driven: = round(USD_per_1000_items * 1_000). Defaults sit # at/above Apify's first-party actor rates (we charge no subscription tiers, @@ -458,6 +467,9 @@ SURFSENSE_ENABLE_DOOM_LOOP=true # TIKTOK_MICROS_PER_VIDEO=3500 # TIKTOK_MICROS_PER_USER=2500 # TIKTOK_MICROS_PER_COMMENT=1500 +# INDEED_SCRAPE_MICROS_PER_JOB=3500 +# WALMART_MICROS_PER_PRODUCT=3500 +# WALMART_MICROS_PER_REVIEW=1500 # Safety ceiling on per-call premium reservation, in micro-USD ($1.00 default). # QUOTA_MAX_RESERVE_MICROS=1000000 diff --git a/surfsense_backend/.env.example b/surfsense_backend/.env.example index 404bd3b44..1d60c98c3 100644 --- a/surfsense_backend/.env.example +++ b/surfsense_backend/.env.example @@ -277,8 +277,8 @@ MICROS_PER_PAGE=1000 # WEB_CRAWL_CAPTCHA_MICROS_PER_SOLVE=3000 # Debit the credit wallet per *item returned* by the platform-native scrapers -# (Reddit, Google Search, Google Maps, Amazon, YouTube). Default FALSE keeps -# scraping effectively free for self-hosted/OSS installs; hosted set TRUE. +# (Reddit, Google Search, Google Maps, Amazon, Walmart, YouTube). Default FALSE +# keeps scraping effectively free for self-hosted/OSS installs; hosted set TRUE. # Each rate is micro-USD per item, fully config-driven (no hardcoded rate): # = round(USD_per_1000_items * 1_000) # 3500 == $3.50/1000 | 5000 == $5/1000 | 2000 == $2/1000 @@ -298,9 +298,15 @@ MICROS_PER_PAGE=1000 # TIKTOK_MICROS_PER_VIDEO=3500 # TIKTOK_MICROS_PER_USER=2500 # TIKTOK_MICROS_PER_COMMENT=1500 +# INDEED_SCRAPE_MICROS_PER_JOB=3500 # Browser-listing retries when a feed is empty (profile feed is withheld from # flagged IPs; each retry draws a fresh rotating exit IP). Set to 1 for a static IP. # TIKTOK_LISTING_MAX_ATTEMPTS=3 +# Walmart products (server-rendered JSON behind residential proxies) priced like +# Amazon; reviews are 10/page (many light requests) so priced on the per-review +# market like Google Maps reviews. +# WALMART_MICROS_PER_PRODUCT=3500 +# WALMART_MICROS_PER_REVIEW=1500 # Low-balance warning threshold (micro-USD), surfaced to the UI. Default $0.50. CREDIT_LOW_BALANCE_WARNING_MICROS=500000 @@ -549,6 +555,14 @@ LANGSMITH_PROJECT=surfsense # OTEL_METRIC_EXPORT_INTERVAL=300000 # ms; 5 minutes # OTEL_SDK_DISABLED=true # emergency kill-switch +# Observability - PostHog product analytics (server-side) +# Opt-in like OTel: leave POSTHOG_API_KEY unset for zero telemetry. Use the +# SAME project key as the frontend's NEXT_PUBLIC_POSTHOG_KEY so server events +# merge onto the persons the web app already identifies by user id. +# POSTHOG_API_KEY=phc_xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx +# POSTHOG_HOST=https://us.i.posthog.com +# POSTHOG_AI_PRIVACY_MODE=true # false ships LLM prompt/completion bodies to PostHog + # Skills + subagents # SURFSENSE_ENABLE_SKILLS=false # SURFSENSE_ENABLE_SPECIALIZED_SUBAGENTS=false diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/constants.py b/surfsense_backend/app/agents/chat/multi_agent_chat/constants.py index 333d4d274..0fb2fc336 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/constants.py +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/constants.py @@ -36,9 +36,11 @@ SUBAGENT_TO_REQUIRED_CONNECTOR_MAP: dict[str, frozenset[str]] = { "youtube": frozenset(), "google_maps": frozenset(), "google_search": frozenset(), + "indeed": frozenset(), "reddit": frozenset(), "instagram": frozenset(), "tiktok": frozenset(), + "walmart": frozenset(), "mcp_discovery": frozenset( { "SLACK_CONNECTOR", diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/citations/on.md b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/citations/on.md index 371cd1e57..effe5078c 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/citations/on.md +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/citations/on.md @@ -1,8 +1,9 @@ Cite with one token: the bracket label `[n]`. Every citable result — prose from a `task` knowledge_base/research specialist (including the -knowledge_base specialist's `[n]`-labelled workspace findings) — already -carries `[n]` labels on a single shared count. +knowledge_base specialist's `[n]`-labelled workspace findings) and +scraper specialists' run-backed findings — already carries `[n]` labels +on a single shared count. Those labels are the only citation you write; the server resolves each one back to its source after the turn. diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/private.md b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/private.md index 82444634a..56e933253 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/private.md +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/private.md @@ -6,7 +6,7 @@ are changing, and what is being published across the open web — and to put that research to work alongside their own knowledge base. You do this by dispatching **specialist subagents** via the `task` tool: -- **Live web data** — Reddit, YouTube, Instagram, TikTok, Amazon, Google +- **Live web data** — Reddit, YouTube, Instagram, TikTok, Amazon, Walmart, Google Maps, Google Search, and the web crawler return structured, current platform data (posts, comments, transcripts, videos, products, reviews, SERPs, full page content). diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/team.md b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/team.md index 38cec63dc..a16a8bfd2 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/team.md +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/identity/team.md @@ -6,7 +6,7 @@ are changing, and what is being published across the open web — and to put that research to work alongside the team's shared knowledge base. You do this by dispatching **specialist subagents** via the `task` tool: -- **Live web data** — Reddit, YouTube, Instagram, TikTok, Amazon, Google +- **Live web data** — Reddit, YouTube, Instagram, TikTok, Amazon, Walmart, Google Maps, Google Search, and the web crawler return structured, current platform data (posts, comments, transcripts, videos, products, reviews, SERPs, full page content). diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/kb_first.md b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/kb_first.md index 3bae85262..60192fb39 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/kb_first.md +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/kb_first.md @@ -2,7 +2,8 @@ CRITICAL — ground factual answers in what you actually receive this turn: - **live platform data** via the market specialists — `task(reddit, ...)`, `task(youtube, ...)`, `task(instagram, ...)`, - `task(tiktok, ...)`, `task(amazon, ...)`, `task(google_maps, ...)`, + `task(tiktok, ...)`, `task(amazon, ...)`, `task(walmart, ...)`, + `task(google_maps, ...)`, `task(google_search, ...)`, `task(web_crawler, ...)`. Anything about competitors, markets, rankings, reviews, or audience sentiment is answered from what these return **this turn**, never from your training data: your diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/routing.md b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/routing.md index fb818cabc..7e2a0f983 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/routing.md +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/main_agent/system_prompt/prompts/routing.md @@ -32,7 +32,9 @@ about a brand, product, or topic is answered from the platform where they say it — `task(reddit, …)` for community discussion and threads, `task(youtube, …)` for video content, transcripts, and comment sections, `task(tiktok, …)` for short-form video trends by hashtag or search, -`task(google_maps, …)` for customer reviews of physical businesses. Web +`task(google_maps, …)` for customer reviews of physical businesses, +`task(amazon, …)` / `task(walmart, …)` for product ratings and customer +reviews of retail products (Walmart pages the full review history). Web search only finds articles *about* the conversation; the platform specialists return the conversation itself, structured and current. For competitive questions ("what are people saying about X", "how is Y diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/markers.py b/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/markers.py index 025d364f6..61d1cbd8b 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/markers.py +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/markers.py @@ -1,10 +1,11 @@ """Map a registered citation to the frontend ``[citation:]`` payload. The citation renderer understands a chunk id (``42``), a negative chunk id for -anonymous uploads (``-3``), and a URL. This is the seam that turns a server-side -source into one the renderer can resolve; it grows as more source kinds become -renderable. Kinds with no renderable form yet return ``None`` so the marker is -dropped rather than emitted broken. +anonymous uploads (``-3``), a URL, and a scraper-run handle (``run_``). +This is the seam that turns a server-side source into one the renderer can +resolve; it grows as more source kinds become renderable. Kinds with no +renderable form yet return ``None`` so the marker is dropped rather than +emitted broken. """ from __future__ import annotations @@ -22,6 +23,9 @@ def to_frontend_payload(entry: CitationEntry) -> str | None: case CitationSourceType.WEB_RESULT: url = locator.get("url") return url or None + case CitationSourceType.RUN: + run_id = locator.get("run_id") + return str(run_id) if run_id else None case _: # Connector items and chat turns have no client-side renderer yet # (the frontend resolves only chunk ids and URLs), so they stay diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/models.py b/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/models.py index 1273271af..31a0e3372 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/models.py +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/shared/citations/models.py @@ -17,6 +17,7 @@ class CitationSourceType(StrEnum): WEB_RESULT = "web_result" CHAT_TURN = "chat_turn" ANON_CHUNK = "anon_chunk" + RUN = "run" class CitationEntry(BaseModel): diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/__init__.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/__init__.py new file mode 100644 index 000000000..ddd8a9cd3 --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/__init__.py @@ -0,0 +1 @@ +"""``indeed`` builtin subagent: structured Indeed job postings.""" diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/agent.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/agent.py new file mode 100644 index 000000000..6da835dbd --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/agent.py @@ -0,0 +1,43 @@ +"""``indeed`` route: ``SurfSenseSubagentSpec`` builder for deepagents.""" + +from __future__ import annotations + +from typing import Any + +from langchain_core.language_models import BaseChatModel +from langchain_core.tools import BaseTool + +from app.agents.chat.multi_agent_chat.subagents.shared.md_file_reader import ( + read_md_file, +) +from app.agents.chat.multi_agent_chat.subagents.shared.spec import SurfSenseSubagentSpec +from app.agents.chat.multi_agent_chat.subagents.shared.subagent_builder import ( + pack_subagent, +) + +from .tools.index import NAME, RULESET, load_tools + + +def build_subagent( + *, + dependencies: dict[str, Any], + model: BaseChatModel | None = None, + middleware_stack: dict[str, Any] | None = None, + mcp_tools: list[BaseTool] | None = None, +) -> SurfSenseSubagentSpec: + tools = [*load_tools(dependencies=dependencies), *(mcp_tools or [])] + description = ( + read_md_file(__package__, "description").strip() + or "Pulls structured job postings from Indeed search and company pages." + ) + system_prompt = read_md_file(__package__, "system_prompt").strip() + return pack_subagent( + name=NAME, + description=description, + system_prompt=system_prompt, + tools=tools, + ruleset=RULESET, + dependencies=dependencies, + model=model, + middleware_stack=middleware_stack, + ) diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/description.md b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/description.md new file mode 100644 index 000000000..221c54fe9 --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/description.md @@ -0,0 +1,2 @@ +Indeed jobs specialist: pulls structured job postings — title, company, location, salary (range, currency, period), job types, benefits, remote/hybrid flag, posting age, apply URL, and the full job description. Discovers jobs by search query (with country, location, radius, job type, experience level, remote/hybrid, and date-posted filters), or scrapes a known Indeed search, company jobs, or single job URL as-is. Optionally fetches each job's detail page for the full description. +Use whenever the task is to find or gather job listings from Indeed — openings for a role, hiring at a company, salaries for a title in a location, or remote roles in a field. Triggers include "find jobs for X", "who is hiring X", "data analyst jobs in Y", "remote X roles", and scraping a specific Indeed URL. Not for general web pages (use the web crawling specialist), Google results (use the Google Search specialist), or other job boards. diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/system_prompt.md b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/system_prompt.md new file mode 100644 index 000000000..36eb855d2 --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/system_prompt.md @@ -0,0 +1,65 @@ +You are the SurfSense Indeed sub-agent. +You receive delegated instructions from a supervisor agent and return structured results for supervisor synthesis. + + +Answer the delegated question from live Indeed job data gathered with your verb, comparing against earlier results already in this conversation when the task calls for it. + + + +- `indeed_scrape` +- `read_run` / `search_run` (free readers for stored scrape output) + + + +- Finding jobs for a role: call `indeed_scrape` with `search_queries`; narrow with `location`, `country`, `job_type`, `level`, `remote`, `radius`, and `from_days`. +- Scraping a specific Indeed URL: pass a search, company jobs, or single job URL in `urls`. +- Full descriptions: set `scrape_job_details=true` to fetch each job's detail page (slower: one extra load per job). Leave it false when the listing snippet is enough. +- Cost model: Indeed is latency-bound. A cold session spends minutes solving Cloudflare, and each query returns only its first page (~15 jobs) — anonymous pagination is gated, so `max_items` above ~15/query buys nothing. Every extra phrasing adds wait, not depth. +- Default to ONE focused query. Only add phrasings when the role genuinely needs variety, and then put at most 2–3 in a SINGLE call's `search_queries` (they reuse one warmed session); never make separate `indeed_scrape` calls, which each pay the cold-start cost. +- Prefer returning the first page's on-topic hits promptly over exhaustive coverage. If a single call under-delivers against a large requested N, return `status=partial` noting the ~one-page-per-query ceiling — do not chase it with more calls. + +- Comparison requests: pull the current results, compare against prior values already in this conversation's earlier tool results, and report concrete deltas (added, removed, salary/rank changes). + + + +- Use only tools in ``. +- Report only results present in the tool output. Never invent titles, companies, salaries, locations, or description text. + + + +- Do not read arbitrary web pages — that belongs to the web crawling specialist. +- Do not generate deliverables or perform connector mutations; return findings for the supervisor to act on. +- Google results belong to the Google Search specialist; other job boards are out of scope. + + + +- Report uncertainty explicitly when evidence is incomplete or conflicting. +- Never present unverified claims as facts. + + + +- Underspecified request — no usable query or URL — return `status=blocked` with the missing fields. +- Tool failure: return `status=error` with a concise recovery `next_step`. +- No useful evidence: return `status=blocked` with a narrower query or the scope you still need. + + + +Return **only** one JSON object (no markdown/prose): +{ + "status": "success" | "partial" | "blocked" | "error", + "action_summary": string, + "evidence": { + "findings": string[], + "sources": string[], + "confidence": "high" | "medium" | "low" + }, + "next_step": string | null, + "missing_fields": string[] | null, + "assumptions": string[] | null +} + +Route-specific rules: +- `evidence.findings`: one entry per distinct job or delta — a single sentence each; do not paste raw payloads. Max 10 entries, unless the delegated task asks for N items: then return up to N (each backed by a real scraped result, never padded). +- `evidence.sources`: one Indeed job URL per finding when applicable, same cap as findings. List each URL once. + + diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/tools/__init__.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/tools/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/tools/index.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/tools/index.py new file mode 100644 index 000000000..523579471 --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/indeed/tools/index.py @@ -0,0 +1,27 @@ +"""``indeed`` sub-agent tools: the Indeed scrape capability verb.""" + +from __future__ import annotations + +from typing import Any + +from langchain_core.tools import BaseTool + +from app.agents.chat.multi_agent_chat.shared.permissions import Ruleset +from app.capabilities.core.access.agent import build_capability_tools +from app.capabilities.indeed.scrape.definition import INDEED_SCRAPE + +NAME = "indeed" + +RULESET = Ruleset(origin=NAME, rules=[]) + +_CI_VERBS = [INDEED_SCRAPE] + + +def load_tools( + *, dependencies: dict[str, Any] | None = None, **kwargs: Any +) -> list[BaseTool]: + d = {**(dependencies or {}), **kwargs} + return build_capability_tools( + workspace_id=d.get("workspace_id"), + capabilities=_CI_VERBS, + ) diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/__init__.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/__init__.py new file mode 100644 index 000000000..8126071cf --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/__init__.py @@ -0,0 +1 @@ +"""``walmart`` builtin subagent: structured public Walmart product data and reviews.""" diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/agent.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/agent.py new file mode 100644 index 000000000..794b050a2 --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/agent.py @@ -0,0 +1,43 @@ +"""``walmart`` route: ``SurfSenseSubagentSpec`` builder for deepagents.""" + +from __future__ import annotations + +from typing import Any + +from langchain_core.language_models import BaseChatModel +from langchain_core.tools import BaseTool + +from app.agents.chat.multi_agent_chat.subagents.shared.md_file_reader import ( + read_md_file, +) +from app.agents.chat.multi_agent_chat.subagents.shared.spec import SurfSenseSubagentSpec +from app.agents.chat.multi_agent_chat.subagents.shared.subagent_builder import ( + pack_subagent, +) + +from .tools.index import NAME, RULESET, load_tools + + +def build_subagent( + *, + dependencies: dict[str, Any], + model: BaseChatModel | None = None, + middleware_stack: dict[str, Any] | None = None, + mcp_tools: list[BaseTool] | None = None, +) -> SurfSenseSubagentSpec: + tools = [*load_tools(dependencies=dependencies), *(mcp_tools or [])] + description = ( + read_md_file(__package__, "description").strip() + or "Scrapes public Walmart product data and reviews for a URL or search term." + ) + system_prompt = read_md_file(__package__, "system_prompt").strip() + return pack_subagent( + name=NAME, + description=description, + system_prompt=system_prompt, + tools=tools, + ruleset=RULESET, + dependencies=dependencies, + model=model, + middleware_stack=middleware_stack, + ) diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/description.md b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/description.md new file mode 100644 index 000000000..88c6b7c28 --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/description.md @@ -0,0 +1,2 @@ +Walmart product specialist: scrapes public Walmart listings and returns structured product data — title, item id (usItemId), brand, price and list price, star rating and review count, availability, images, features, seller, and product variants — plus deep paginated customer reviews (rating, text, author, verified-purchase flag, images, and seller responses). Works from a search term (e.g. "air fryer") or from Walmart product (/ip/...), search, category, or browse URLs, and can pull many reviews per product by item id or URL. Only public, anonymous US Walmart data — no login or seller account. +Use it for product research, price tracking, catalog enrichment by item id, and review mining. Triggers include "find X on Walmart", "Walmart price of X", "reviews for this Walmart product", "compare these Walmart products", and "look up this Walmart URL/item id". Not for general web search (use the Google Search specialist), reading an arbitrary non-Walmart page (use the web crawling specialist), or other marketplaces. diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/system_prompt.md b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/system_prompt.md new file mode 100644 index 000000000..bf30fe2bb --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/system_prompt.md @@ -0,0 +1,68 @@ +You are the SurfSense Walmart sub-agent. +You receive delegated instructions from a supervisor agent and return structured results for supervisor synthesis. + + +Answer the delegated question from public Walmart product data and reviews gathered with your verbs, comparing against earlier results already in this conversation when the task calls for it. + + + +- `walmart_scrape` — product details and search/category listings +- `walmart_reviews` — deep paginated reviews for a product +- `read_run` / `search_run` (free readers for stored scrape output) + + + +- Discovering products: call `walmart_scrape` with `search_terms` (e.g. ["air fryer"]). +- Specific products: pass Walmart product URLs (/ip/...) or search/category/browse URLs in `urls`. +- Faster listings: set `include_details=false` to return card-only results without opening each product page. +- Sampled reviews: `walmart_scrape` returns a small on-page review sample by default (`include_reviews_sample=true`); disable it when reviews are irrelevant. +- Deep review mining: use `walmart_reviews` with product `urls` or numeric `item_ids` (usItemId); raise `max_reviews` and set `sort_by` (most-recent, most-helpful, rating-high, rating-low) as the task needs. Reviews are billed per review, so keep `max_reviews` to what the task actually requires. +- Batch multiple URLs or search terms into one call rather than many single-source calls. + +- Comparison requests: pull the current products, compare against prior values already in this conversation's earlier tool results, and report concrete deltas (price up/down, rating change, stock changes). + + + +- Use only tools in ``. +- Report only results present in the tool output. Never invent titles, item ids, prices, ratings, or reviews. +- `walmart_scrape`: provide at least one of `urls` or `search_terms`. +- `walmart_reviews`: provide at least one of `urls` or `item_ids`. + + + +- Do not perform general web search — that is the Google Search specialist's job. +- Do not read or extract an arbitrary non-Walmart page — return the URL for the web crawling specialist. +- Do not generate deliverables or perform connector mutations; return findings for the supervisor to act on. +- Only public, anonymous Walmart data — never anything behind a login or seller account. + + + +- Report uncertainty explicitly when evidence is incomplete or conflicting. +- Never present unverified claims as facts. + + + +- Underspecified request — no usable search term, URL, or item id — return `status=blocked` with the missing fields. +- Tool failure: return `status=error` with a concise recovery `next_step`. +- No useful evidence: return `status=blocked` with a narrower query or the scope you still need. + + + +Return **only** one JSON object (no markdown/prose): +{ + "status": "success" | "partial" | "blocked" | "error", + "action_summary": string, + "evidence": { + "findings": string[], + "sources": string[], + "confidence": "high" | "medium" | "low" + }, + "next_step": string | null, + "missing_fields": string[] | null, + "assumptions": string[] | null +} + +Route-specific rules: +- `evidence.findings`: max 10 entries, each a single sentence stating one distinct product, review theme, or delta. Do not paste raw payloads. +- `evidence.sources`: max 10 URLs, one per finding when applicable. List each URL once. + diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/tools/__init__.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/tools/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/tools/index.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/tools/index.py new file mode 100644 index 000000000..ac3e5ed81 --- /dev/null +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/builtins/walmart/tools/index.py @@ -0,0 +1,28 @@ +"""``walmart`` sub-agent tools: the Walmart product scrape and reviews verbs.""" + +from __future__ import annotations + +from typing import Any + +from langchain_core.tools import BaseTool + +from app.agents.chat.multi_agent_chat.shared.permissions import Ruleset +from app.capabilities.core.access.agent import build_capability_tools +from app.capabilities.walmart.reviews.definition import WALMART_REVIEWS +from app.capabilities.walmart.scrape.definition import WALMART_SCRAPE + +NAME = "walmart" + +RULESET = Ruleset(origin=NAME, rules=[]) + +_CI_VERBS = [WALMART_SCRAPE, WALMART_REVIEWS] + + +def load_tools( + *, dependencies: dict[str, Any] | None = None, **kwargs: Any +) -> list[BaseTool]: + d = {**(dependencies or {}), **kwargs} + return build_capability_tools( + workspace_id=d.get("workspace_id"), + capabilities=_CI_VERBS, + ) diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/middleware_stack.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/middleware_stack.py index 124ccf704..4181180d1 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/middleware_stack.py +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/middleware_stack.py @@ -15,6 +15,9 @@ from __future__ import annotations from typing import Any from app.agents.chat.multi_agent_chat.shared.feature_flags import AgentFeatureFlags +from app.agents.chat.multi_agent_chat.shared.middleware.citation_state import ( + build_citation_state_mw, +) from app.agents.chat.multi_agent_chat.shared.middleware.resilience import ( ResilienceMiddlewares, ) @@ -45,6 +48,8 @@ def build_subagent_middleware_stack( return { "todos": build_todos_mw(), "permission": permission, + # Declares the citation_registry channel so run [n]s merge up from tools. + "citation": build_citation_state_mw(), "retry": resilience.retry, "fallback": resilience.fallback, "model_call_limit": resilience.model_call_limit, diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/registry.py b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/registry.py index 7b8b8a6f1..5ecd2721c 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/registry.py +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/registry.py @@ -24,6 +24,9 @@ from app.agents.chat.multi_agent_chat.subagents.builtins.google_maps.agent impor from app.agents.chat.multi_agent_chat.subagents.builtins.google_search.agent import ( build_subagent as build_google_search_subagent, ) +from app.agents.chat.multi_agent_chat.subagents.builtins.indeed.agent import ( + build_subagent as build_indeed_subagent, +) from app.agents.chat.multi_agent_chat.subagents.builtins.instagram.agent import ( build_subagent as build_instagram_subagent, ) @@ -42,6 +45,9 @@ from app.agents.chat.multi_agent_chat.subagents.builtins.reddit.agent import ( from app.agents.chat.multi_agent_chat.subagents.builtins.tiktok.agent import ( build_subagent as build_tiktok_subagent, ) +from app.agents.chat.multi_agent_chat.subagents.builtins.walmart.agent import ( + build_subagent as build_walmart_subagent, +) from app.agents.chat.multi_agent_chat.subagents.builtins.web_crawler.agent import ( build_subagent as build_web_crawler_subagent, ) @@ -89,6 +95,7 @@ SUBAGENT_BUILDERS_BY_NAME: dict[str, SubagentBuilder] = { "google_drive": build_google_drive_subagent, "google_maps": build_google_maps_subagent, "google_search": build_google_search_subagent, + "indeed": build_indeed_subagent, "instagram": build_instagram_subagent, "knowledge_base": build_knowledge_base_subagent, "mcp_discovery": build_mcp_discovery_subagent, @@ -96,6 +103,7 @@ SUBAGENT_BUILDERS_BY_NAME: dict[str, SubagentBuilder] = { "onedrive": build_onedrive_subagent, "reddit": build_reddit_subagent, "tiktok": build_tiktok_subagent, + "walmart": build_walmart_subagent, "web_crawler": build_web_crawler_subagent, "youtube": build_youtube_subagent, } diff --git a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/shared/snippets/output_contract_base.md b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/shared/snippets/output_contract_base.md index 35fd48814..cd3b94400 100644 --- a/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/shared/snippets/output_contract_base.md +++ b/surfsense_backend/app/agents/chat/multi_agent_chat/subagents/shared/snippets/output_contract_base.md @@ -5,3 +5,4 @@ Rules (universal): - `status=blocked` due to missing required inputs -> `missing_fields` must be non-null. - `assumptions`: any inferences you made about the user's intent; `null` when no inferences were needed. - The `evidence` object's fields are documented in your route-specific `` above; never invent fields the tool did not return. +- When a finding is drawn from a scraper run, append that run's `[n]` (the tool result states `Cite this scraper run as [n]`) to the finding text so the citation survives into the final answer. Copy the label exactly; never invent one. diff --git a/surfsense_backend/app/agents/video_presentation/nodes.py b/surfsense_backend/app/agents/video_presentation/nodes.py index dca89059f..e52d1f4ce 100644 --- a/surfsense_backend/app/agents/video_presentation/nodes.py +++ b/surfsense_backend/app/agents/video_presentation/nodes.py @@ -17,6 +17,7 @@ from app.config import config as app_config from app.services.kokoro_tts_service import get_kokoro_tts_service from app.services.llm_service import get_agent_llm from app.utils.content_utils import extract_text_content, strip_markdown_fences +from app.utils.file_io import write_bytes from .configuration import Configuration from .prompts import ( @@ -137,8 +138,7 @@ async def create_slide_audio(state: State, config: RunnableConfig) -> dict[str, kwargs["api_base"] = app_config.TTS_SERVICE_API_BASE response = await aspeech(**kwargs) - with open(chunk_path, "wb") as f: - f.write(response.content) + await write_bytes(chunk_path, response.content) return chunk_path diff --git a/surfsense_backend/app/app.py b/surfsense_backend/app/app.py index dde2ce7fa..7f1fbfabe 100644 --- a/surfsense_backend/app/app.py +++ b/surfsense_backend/app/app.py @@ -8,6 +8,7 @@ from collections import defaultdict from contextlib import asynccontextmanager from datetime import UTC, datetime from threading import Lock +from typing import Any import redis from fastapi import Depends, FastAPI, HTTPException, Request, Response, status @@ -50,7 +51,7 @@ from app.gateway.inbox_worker import ( start_gateway_inbox_worker, stop_gateway_inbox_worker, ) -from app.observability import metrics as ot_metrics +from app.observability import analytics as ph_analytics, metrics as ot_metrics from app.observability.bootstrap import init_otel, shutdown_otel from app.rate_limiter import get_real_client_ip, limiter from app.routes import router as crud_router @@ -94,17 +95,21 @@ def _build_error_response( code: str = "INTERNAL_ERROR", request_id: str = "", extra_headers: dict[str, str] | None = None, + fields: list[dict[str, Any]] | None = None, ) -> JSONResponse: """Build the standardized error envelope (new ``error`` + legacy ``detail``).""" + error: dict[str, Any] = { + "code": code, + "message": message, + "status": status_code, + "request_id": request_id, + "timestamp": datetime.now(UTC).isoformat(), + "report_url": ISSUES_URL, + } + if fields: + error["fields"] = fields body = { - "error": { - "code": code, - "message": message, - "status": status_code, - "request_id": request_id, - "timestamp": datetime.now(UTC).isoformat(), - "report_url": ISSUES_URL, - }, + "error": error, "detail": message, } headers = {"X-Request-ID": request_id} @@ -222,16 +227,35 @@ def _http_exception_handler(request: Request, exc: HTTPException) -> JSONRespons def _validation_error_handler( request: Request, exc: RequestValidationError ) -> JSONResponse: - """Return 422 with field-level detail in the standard envelope.""" + """Return 422 with field-level detail in the standard envelope. + + ``error.fields`` carries each failure's location path and message so clients + can attach errors to the offending input; ``message`` is the flat summary. + """ rid = _get_request_id(request) - fields = [] - for err in exc.errors(): - loc = " -> ".join(str(part) for part in err.get("loc", [])) - fields.append(f"{loc}: {err.get('msg', 'invalid')}") - message = ( - f"Validation failed: {'; '.join(fields)}" if fields else "Validation failed." + fields = [ + { + "loc": [str(part) for part in err.get("loc", ())], + # Drop pydantic's "Value error, " prefix so messages read for humans. + "msg": str(err.get("msg", "invalid")).removeprefix("Value error, "), + } + for err in exc.errors() + ] + + def _segment(field: dict[str, Any]) -> str: + # Drop the "body" request root so model-level errors read as a plain + # sentence and field errors read as "field -> sub", not "body -> field". + path = field["loc"] + if path and path[0] == "body": + path = path[1:] + loc = " -> ".join(path) + return f"{loc}: {field['msg']}" if loc else field["msg"] + + summary = "; ".join(_segment(f) for f in fields) + message = f"Validation failed: {summary}" if fields else "Validation failed." + return _build_error_response( + 422, message, code="VALIDATION_ERROR", request_id=rid, fields=fields ) - return _build_error_response(422, message, code="VALIDATION_ERROR", request_id=rid) def _unhandled_exception_handler(request: Request, exc: Exception) -> JSONResponse: @@ -690,6 +714,7 @@ async def lifespan(app: FastAPI): await stop_gateway_inbox_worker() _stop_openrouter_background_refresh() await close_checkpointer() + ph_analytics.shutdown() shutdown_otel() @@ -796,6 +821,52 @@ class RequestPerfMiddleware(BaseHTTPMiddleware): app.add_middleware(RequestPerfMiddleware) + +# --------------------------------------------------------------------------- +# PAT / MCP API attribution middleware +# --------------------------------------------------------------------------- +# Emits a PostHog ``pat_api_request`` event for any request authenticated by a +# Personal Access Token, so "documents added via MCP", "searches via MCP" etc. +# are queryable without instrumenting each route. Relies on ``get_auth_context`` +# stashing the resolved principal on ``request.state.auth_context``; requests +# that never resolve a PAT principal are silently skipped. No-op when PostHog +# is unconfigured. + + +class PatApiAnalyticsMiddleware(BaseHTTPMiddleware): + """Capture PAT-authenticated API usage (incl. MCP) after each response.""" + + async def dispatch( + self, request: StarletteRequest, call_next: RequestResponseEndpoint + ) -> StarletteResponse: + response = await call_next(request) + with contextlib.suppress(Exception): + ctx = getattr(request.state, "auth_context", None) + if ctx is not None and ctx.method == "pat" and ph_analytics.is_enabled(): + # Use the route *template* (e.g. /documents/{id}) to keep the + # ``route`` property low-cardinality; fall back to the raw path. + route = request.scope.get("route") + route_path = getattr(route, "path", None) or request.url.path + client = ( + "mcp" + if request.headers.get("X-SurfSense-Client") == "mcp" + else "pat_script" + ) + ph_analytics.capture_for( + ctx, + "pat_api_request", + { + "route": route_path, + "method": request.method, + "status_code": response.status_code, + "client": client, + }, + ) + return response + + +app.add_middleware(PatApiAnalyticsMiddleware) + # Add SlowAPI middleware for automatic rate limiting # Uses Starlette BaseHTTPMiddleware (not the raw ASGI variant) to avoid # corrupting StreamingResponse — SlowAPIASGIMiddleware re-sends diff --git a/surfsense_backend/app/automations/runtime/executor.py b/surfsense_backend/app/automations/runtime/executor.py index d9544cd0c..a515c6563 100644 --- a/surfsense_backend/app/automations/runtime/executor.py +++ b/surfsense_backend/app/automations/runtime/executor.py @@ -15,11 +15,36 @@ from app.automations.schemas.definition.envelope import ( ) from app.automations.schemas.definition.plan_step import PlanStep from app.automations.templating import build_run_context +from app.observability import analytics as ph_analytics from . import repository from .step import execute_step +def _capture_run_outcome(run: AutomationRun, status: str) -> None: + """Emit ``automation_run_completed`` — headless runs the frontend never sees. + + No-op when PostHog is unconfigured or the owning user is unknown. + """ + automation = run.automation + creator_id = getattr(automation, "created_by_user_id", None) + if not creator_id: + return + workspace_id = getattr(automation, "workspace_id", None) + ph_analytics.capture( + "automation_run_completed", + distinct_id=str(creator_id), + properties={ + "automation_id": run.automation_id, + "run_id": run.id, + "workspace_id": workspace_id, + "status": status, + "trigger_type": run.trigger.type.value if run.trigger else None, + }, + groups={"workspace": str(workspace_id)} if workspace_id is not None else None, + ) + + async def execute_run(session: AsyncSession, run_id: int) -> None: """Load run ``run_id`` and execute its snapshot plan to a terminal state.""" run = await repository.load_run(session, run_id) @@ -41,6 +66,7 @@ async def execute_run(session: AsyncSession, run_id: int) -> None: }, ) await session.commit() + _capture_run_outcome(run, "failed") return await repository.mark_running(session, run) @@ -66,6 +92,7 @@ async def execute_run(session: AsyncSession, run_id: int) -> None: await _run_on_failure(session, run, definition) await repository.mark_failed(session, run, result.get("error")) await session.commit() + _capture_run_outcome(run, "failed") return if result["status"] == "succeeded": @@ -73,6 +100,7 @@ async def execute_run(session: AsyncSession, run_id: int) -> None: await repository.mark_succeeded(session, run) await session.commit() + _capture_run_outcome(run, "succeeded") async def _run_on_failure( diff --git a/surfsense_backend/app/automations/services/automation.py b/surfsense_backend/app/automations/services/automation.py index a419327bf..a7bbd3af3 100644 --- a/surfsense_backend/app/automations/services/automation.py +++ b/surfsense_backend/app/automations/services/automation.py @@ -29,6 +29,7 @@ from app.automations.services.model_policy import ( from app.automations.triggers import get_trigger from app.automations.triggers.builtin.schedule import compute_next_fire_at from app.db import Permission, Workspace, get_async_session +from app.observability import analytics as ph_analytics from app.users import get_auth_context from app.utils.rbac import check_permission @@ -75,6 +76,18 @@ class AutomationService: self.session.add(automation) await self.session.commit() + + # Authoritative creation (migrated from automations-mutation.atoms.ts). + ph_analytics.capture_for( + self.auth, + "automation_created", + { + "automation_id": automation.id, + "workspace_id": automation.workspace_id, + "trigger_count": len(payload.triggers), + }, + groups={"workspace": str(automation.workspace_id)}, + ) return await self._get_with_triggers_or_raise(automation.id) async def list( @@ -150,6 +163,31 @@ class AutomationService: automation.version += 1 await self.session.commit() + + # Migrated from automations-mutation.atoms.ts: a status-only change is a + # distinct event; other field edits are ``automation_updated``. + if "status" in data: + ph_analytics.capture_for( + self.auth, + "automation_status_changed", + { + "automation_id": automation.id, + "workspace_id": automation.workspace_id, + "next_status": str(data["status"]), + }, + groups={"workspace": str(automation.workspace_id)}, + ) + if any(k in data for k in ("name", "description", "definition")): + ph_analytics.capture_for( + self.auth, + "automation_updated", + { + "automation_id": automation.id, + "workspace_id": automation.workspace_id, + "has_definition_change": "definition" in data, + }, + groups={"workspace": str(automation.workspace_id)}, + ) return await self._get_with_triggers_or_raise(automation_id) async def delete(self, automation_id: int) -> None: @@ -158,9 +196,18 @@ class AutomationService: await self._authorize( automation.workspace_id, Permission.AUTOMATIONS_DELETE.value ) + workspace_id = automation.workspace_id await self.session.delete(automation) await self.session.commit() + # Authoritative deletion (migrated from automations-mutation.atoms.ts). + ph_analytics.capture_for( + self.auth, + "automation_deleted", + {"automation_id": automation_id, "workspace_id": workspace_id}, + groups={"workspace": str(workspace_id)}, + ) + async def _get_or_raise(self, automation_id: int) -> Automation: automation = await self.session.get(Automation, automation_id) if automation is None: diff --git a/surfsense_backend/app/automations/services/trigger.py b/surfsense_backend/app/automations/services/trigger.py index 5175eb15e..866e66395 100644 --- a/surfsense_backend/app/automations/services/trigger.py +++ b/surfsense_backend/app/automations/services/trigger.py @@ -16,6 +16,7 @@ from app.automations.schemas.api import TriggerCreate, TriggerUpdate from app.automations.triggers import get_trigger from app.automations.triggers.builtin.schedule import compute_next_fire_at from app.db import Permission, get_async_session +from app.observability import analytics as ph_analytics from app.users import get_auth_context from app.utils.rbac import check_permission @@ -48,6 +49,18 @@ class TriggerService: self.session.add(trigger) await self.session.commit() await self.session.refresh(trigger) + + # Migrated from automations-mutation.atoms.ts. + ph_analytics.capture_for( + self.auth, + "automation_trigger_added", + { + "automation_id": automation_id, + "trigger_id": trigger.id, + "trigger_type": getattr(trigger.type, "value", str(trigger.type)), + "enabled": trigger.enabled, + }, + ) return trigger async def update( @@ -82,6 +95,26 @@ class TriggerService: await self.session.commit() await self.session.refresh(trigger) + + # Migrated from automations-mutation.atoms.ts. ``change`` mirrors the + # frontend's coarse categorisation. + _change = ( + "enabled" + if "enabled" in data and "params" not in data + else "params" + if "params" in data + else "other" + ) + ph_analytics.capture_for( + self.auth, + "automation_trigger_updated", + { + "automation_id": automation_id, + "trigger_id": trigger_id, + "change": _change, + "enabled": trigger.enabled, + }, + ) return trigger async def remove(self, *, automation_id: int, trigger_id: int) -> None: @@ -92,6 +125,13 @@ class TriggerService: await self.session.delete(trigger) await self.session.commit() + # Migrated from automations-mutation.atoms.ts. + ph_analytics.capture_for( + self.auth, + "automation_trigger_removed", + {"automation_id": automation_id, "trigger_id": trigger_id}, + ) + async def _authorize_automation( self, automation_id: int, permission: str ) -> Automation: diff --git a/surfsense_backend/app/capabilities/__init__.py b/surfsense_backend/app/capabilities/__init__.py index 9c97ef59a..dfdab630b 100644 --- a/surfsense_backend/app/capabilities/__init__.py +++ b/surfsense_backend/app/capabilities/__init__.py @@ -1,4 +1,4 @@ -"""Scraper capability registry — typed, stateless verbs. See plans/backend/04-capabilities.md.""" +"""Scraper capability registry — typed, stateless verbs.""" from __future__ import annotations diff --git a/surfsense_backend/app/capabilities/amazon/scrape/schemas.py b/surfsense_backend/app/capabilities/amazon/scrape/schemas.py index e510e2eea..b2a847058 100644 --- a/surfsense_backend/app/capabilities/amazon/scrape/schemas.py +++ b/surfsense_backend/app/capabilities/amazon/scrape/schemas.py @@ -4,6 +4,7 @@ from __future__ import annotations from pydantic import BaseModel, Field, model_validator +from app.capabilities.core.validation import HttpUrlStr from app.proprietary.platforms.amazon import ProductItem MAX_AMAZON_SOURCES = 20 @@ -13,7 +14,7 @@ MAX_AMAZON_RESULTS = 1000 class ScrapeInput(BaseModel): """Agent-facing controls for public product discovery and enrichment.""" - urls: list[str] = Field(default_factory=list, max_length=MAX_AMAZON_SOURCES) + urls: list[HttpUrlStr] = Field(default_factory=list, max_length=MAX_AMAZON_SOURCES) search_terms: list[str] = Field(default_factory=list, max_length=MAX_AMAZON_SOURCES) max_items: int = Field(default=10, ge=1, le=100) domain: str = Field( diff --git a/surfsense_backend/app/capabilities/core/access/agent.py b/surfsense_backend/app/capabilities/core/access/agent.py index 4aa569e24..3d3864a53 100644 --- a/surfsense_backend/app/capabilities/core/access/agent.py +++ b/surfsense_backend/app/capabilities/core/access/agent.py @@ -11,9 +11,13 @@ subagent can follow a truncation reference without extra wiring. from __future__ import annotations +import json import time +from langchain.tools import ToolRuntime +from langchain_core.messages import ToolMessage from langchain_core.tools import BaseTool, StructuredTool +from langgraph.types import Command from app.capabilities.core.billing import charge_capability, gate_capability from app.capabilities.core.progress import progress_scope @@ -64,7 +68,7 @@ def _capability_tool(capability: Capability, workspace_id: int) -> BaseTool: executor = capability.executor name = capability.name - async def _run(**kwargs: object) -> dict | str: + async def _run(runtime: ToolRuntime, **kwargs: object) -> dict | str | Command: payload = input_model(**kwargs) input_dump = payload.model_dump(exclude_none=True) thread_id = _current_thread_id() @@ -119,13 +123,42 @@ def _capability_tool(capability: Capability, workspace_id: int) -> BaseTool: progress=reporter.coarse, ) + # No stored run to cite: keep the legacy return shape, no citation. + if run_id is None: + if serialized.char_count <= RUN_OUTPUT_CHAR_CAP: + return output.model_dump(exclude_none=True) + return _build_preview(serialized, run_id) + + run_external_id = f"run_{run_id}" if serialized.char_count <= RUN_OUTPUT_CHAR_CAP: dump = output.model_dump(exclude_none=True) - if run_id is not None: - dump["run_id"] = f"run_{run_id}" - return dump + dump["run_id"] = run_external_id + content = json.dumps(dump, ensure_ascii=False, default=str) + else: + content = _build_preview(serialized, run_id) - return _build_preview(serialized, run_id) + # Deferred import: the citation spine imports from here; lazy avoids a cycle. + from app.agents.chat.multi_agent_chat.shared.citations import load_registry + from app.capabilities.core.access.run_citation import attach_run_citation + + registry = load_registry(getattr(runtime, "state", None)) + _, label = attach_run_citation( + registry, run_external_id=run_external_id, capability=name + ) + return Command( + update={ + "messages": [ + ToolMessage( + content=content + label, + tool_call_id=runtime.tool_call_id, + ) + ], + "citation_registry": registry, + } + ) + + # Un-stringify for StructuredTool's signature-based runtime injection. + _run.__annotations__["runtime"] = ToolRuntime return StructuredTool.from_function( coroutine=_run, diff --git a/surfsense_backend/app/capabilities/core/access/rest.py b/surfsense_backend/app/capabilities/core/access/rest.py index 2047b21ae..26ac46318 100644 --- a/surfsense_backend/app/capabilities/core/access/rest.py +++ b/surfsense_backend/app/capabilities/core/access/rest.py @@ -45,6 +45,7 @@ from app.capabilities.core.store import all_capabilities from app.capabilities.core.types import Capability, CapabilityContext from app.db import Run, async_session_maker, get_async_session from app.exceptions import ExternalServiceError, SurfSenseError +from app.observability import analytics as ph_analytics from app.services.web_crawl_credit_service import InsufficientCreditsError from app.users import get_auth_context from app.utils.rbac import check_workspace_access @@ -107,6 +108,41 @@ def _origin_for(auth: AuthContext) -> str: return "ui" if getattr(auth, "method", None) == "session" else "api" +def _capture_scraper_run( + *, + capability: str, + status: str, + user_id, + origin: str, + duration_ms: int | None = None, + item_count: int | None = None, + cost_micros: int | None = None, +) -> None: + """Emit ``scraper_run_completed`` — scrapers are the highest-value MCP surface. + + Covers both sync and async runs; the async background task never flows + through the PAT middleware, so this is the only place its outcome is + captured. No-op when PostHog is unconfigured. + """ + if not ph_analytics.is_enabled() or not user_id: + return + platform, _, verb = capability.partition(".") + ph_analytics.capture( + "scraper_run_completed", + distinct_id=str(user_id), + properties={ + "platform": platform, + "verb": verb, + "capability": capability, + "status": status, + "origin": origin, + "duration_ms": duration_ms, + "item_count": item_count, + "cost_micros": cost_micros, + }, + ) + + def _now_ms() -> int: return int(time.time() * 1000) @@ -185,6 +221,8 @@ async def _execute_async_run( unit, executor, payload, + user_id=None, + origin: str = "api", ) -> None: """Run a scrape in the background: stream progress, charge, finalize the row. @@ -218,6 +256,13 @@ async def _execute_async_run( progress=reporter.coarse, ) _publish_finished(run_id, "error", error=str(exc)) + _capture_scraper_run( + capability=capability, + status="error", + user_id=user_id, + origin=origin, + duration_ms=int((time.perf_counter() - started) * 1000), + ) return except Exception: logger.exception("async run %s failed with an upstream error", run_id) @@ -229,6 +274,13 @@ async def _execute_async_run( progress=reporter.coarse, ) _publish_finished(run_id, "error", error="upstream error") + _capture_scraper_run( + capability=capability, + status="error", + user_id=user_id, + origin=origin, + duration_ms=int((time.perf_counter() - started) * 1000), + ) return duration_ms = int((time.perf_counter() - started) * 1000) @@ -252,6 +304,15 @@ async def _execute_async_run( progress=reporter.coarse, ) _publish_finished(run_id, "success", item_count=serialized.item_count) + _capture_scraper_run( + capability=capability, + status="success", + user_id=user_id, + origin=origin, + duration_ms=duration_ms, + item_count=serialized.item_count, + cost_micros=cost_micros, + ) async def _finalize_async( @@ -358,6 +419,8 @@ def _register_verb(router: APIRouter, capability: Capability) -> None: unit=unit, executor=executor, payload=payload, + user_id=user_id, + origin=origin, ) ) run_event_bus.register_task(run_id, task) @@ -373,6 +436,7 @@ def _register_verb(router: APIRouter, capability: Capability) -> None: try: output = await executor(payload) except (SurfSenseError, HTTPException) as exc: + _sync_err_duration = int((time.perf_counter() - started) * 1000) await _record_rest_run( workspace_id=workspace_id, capability=name, @@ -381,11 +445,19 @@ def _register_verb(router: APIRouter, capability: Capability) -> None: input=input_dump, user_id=user_id, error=str(exc), - duration_ms=int((time.perf_counter() - started) * 1000), + duration_ms=_sync_err_duration, progress=reporter.coarse, ) + _capture_scraper_run( + capability=name, + status="error", + user_id=user_id, + origin=origin, + duration_ms=_sync_err_duration, + ) raise except Exception as exc: + _sync_err_duration = int((time.perf_counter() - started) * 1000) await _record_rest_run( workspace_id=workspace_id, capability=name, @@ -394,9 +466,16 @@ def _register_verb(router: APIRouter, capability: Capability) -> None: input=input_dump, user_id=user_id, error=str(exc), - duration_ms=int((time.perf_counter() - started) * 1000), + duration_ms=_sync_err_duration, progress=reporter.coarse, ) + _capture_scraper_run( + capability=name, + status="error", + user_id=user_id, + origin=origin, + duration_ms=_sync_err_duration, + ) raise ExternalServiceError( f"The '{name}' capability failed due to an upstream error.", code="CAPABILITY_UPSTREAM_ERROR", @@ -418,6 +497,15 @@ def _register_verb(router: APIRouter, capability: Capability) -> None: cost_micros=cost_micros, progress=reporter.coarse, ) + _capture_scraper_run( + capability=name, + status="success", + user_id=user_id, + origin=origin, + duration_ms=duration_ms, + item_count=serialized.item_count, + cost_micros=cost_micros, + ) if run_id is not None: response.headers["X-Run-Id"] = f"run_{run_id}" return output diff --git a/surfsense_backend/app/capabilities/core/access/run_citation.py b/surfsense_backend/app/capabilities/core/access/run_citation.py new file mode 100644 index 000000000..9b7c3f3db --- /dev/null +++ b/surfsense_backend/app/capabilities/core/access/run_citation.py @@ -0,0 +1,23 @@ +"""Register a recorded scraper run as a citable ``[n]``.""" + +from __future__ import annotations + +from app.agents.chat.multi_agent_chat.shared.citations import ( + CitationRegistry, + CitationSourceType, +) + + +def attach_run_citation( + registry: CitationRegistry, + *, + run_external_id: str, + capability: str, +) -> tuple[int, str]: + """Register the ``run_`` handle; return its ``[n]`` and the label line.""" + n = registry.register( + CitationSourceType.RUN, + {"run_id": run_external_id}, + {"capability": capability}, + ) + return n, f"\n\nCite this scraper run as [{n}] after any claim drawn from its data." diff --git a/surfsense_backend/app/capabilities/core/billing.py b/surfsense_backend/app/capabilities/core/billing.py index c2c7f6b5b..cdaf1eb9d 100644 --- a/surfsense_backend/app/capabilities/core/billing.py +++ b/surfsense_backend/app/capabilities/core/billing.py @@ -41,6 +41,9 @@ _PLATFORM_RATE_KEYS: dict[BillingUnit, str] = { BillingUnit.TIKTOK_VIDEO: "TIKTOK_MICROS_PER_VIDEO", BillingUnit.TIKTOK_USER: "TIKTOK_MICROS_PER_USER", BillingUnit.TIKTOK_COMMENT: "TIKTOK_MICROS_PER_COMMENT", + BillingUnit.INDEED_JOB: "INDEED_SCRAPE_MICROS_PER_JOB", + BillingUnit.WALMART_PRODUCT: "WALMART_MICROS_PER_PRODUCT", + BillingUnit.WALMART_REVIEW: "WALMART_MICROS_PER_REVIEW", } @@ -63,6 +66,9 @@ _UNIT_NOUNS: dict[BillingUnit, str] = { BillingUnit.TIKTOK_VIDEO: "video", BillingUnit.TIKTOK_USER: "profile", BillingUnit.TIKTOK_COMMENT: "comment", + BillingUnit.INDEED_JOB: "job", + BillingUnit.WALMART_PRODUCT: "product", + BillingUnit.WALMART_REVIEW: "review", } diff --git a/surfsense_backend/app/capabilities/core/types.py b/surfsense_backend/app/capabilities/core/types.py index 2e8d3b6cc..8b1d664c1 100644 --- a/surfsense_backend/app/capabilities/core/types.py +++ b/surfsense_backend/app/capabilities/core/types.py @@ -31,6 +31,9 @@ class BillingUnit(StrEnum): TIKTOK_VIDEO = "tiktok_video" TIKTOK_USER = "tiktok_user" TIKTOK_COMMENT = "tiktok_comment" + INDEED_JOB = "indeed_job" + WALMART_PRODUCT = "walmart_product" + WALMART_REVIEW = "walmart_review" class BillableInput(Protocol): diff --git a/surfsense_backend/app/capabilities/core/validation.py b/surfsense_backend/app/capabilities/core/validation.py new file mode 100644 index 000000000..3e0885488 --- /dev/null +++ b/surfsense_backend/app/capabilities/core/validation.py @@ -0,0 +1,24 @@ +"""Shared Pydantic field types for capability I/O schemas.""" + +from __future__ import annotations + +from typing import Annotated +from urllib.parse import urlsplit + +import validators +from pydantic import AfterValidator +from pydantic_core import PydanticCustomError + +_HTTP_SCHEMES = frozenset({"http", "https"}) + + +def _validate_http_url(value: str) -> str: + """Accept only well-formed http(s) URLs, returned trimmed and unchanged.""" + url = value.strip() + if not validators.url(url) or urlsplit(url).scheme.lower() not in _HTTP_SCHEMES: + raise PydanticCustomError("http_url", "must be a valid http(s) URL") + return url + + +HttpUrlStr = Annotated[str, AfterValidator(_validate_http_url)] +"""A request URL validated as http(s) and kept as ``str`` (no normalization).""" diff --git a/surfsense_backend/app/capabilities/google_maps/reviews/schemas.py b/surfsense_backend/app/capabilities/google_maps/reviews/schemas.py index 1868765e9..26cdad29e 100644 --- a/surfsense_backend/app/capabilities/google_maps/reviews/schemas.py +++ b/surfsense_backend/app/capabilities/google_maps/reviews/schemas.py @@ -10,6 +10,7 @@ from typing import Literal from pydantic import BaseModel, Field, model_validator +from app.capabilities.core.validation import HttpUrlStr from app.proprietary.platforms.google_maps import ReviewItem MAX_MAPS_REVIEW_SOURCES = 20 @@ -17,7 +18,7 @@ MAX_MAPS_REVIEW_SOURCES = 20 class ReviewsInput(BaseModel): - urls: list[str] = Field( + urls: list[HttpUrlStr] = Field( default_factory=list, max_length=MAX_MAPS_REVIEW_SOURCES, description=( diff --git a/surfsense_backend/app/capabilities/google_maps/scrape/schemas.py b/surfsense_backend/app/capabilities/google_maps/scrape/schemas.py index 95833c8f7..7cf1ca6a1 100644 --- a/surfsense_backend/app/capabilities/google_maps/scrape/schemas.py +++ b/surfsense_backend/app/capabilities/google_maps/scrape/schemas.py @@ -10,6 +10,7 @@ from __future__ import annotations from pydantic import BaseModel, Field, model_validator +from app.capabilities.core.validation import HttpUrlStr from app.proprietary.platforms.google_maps import PlaceItem MAX_MAPS_SOURCES = 20 @@ -26,7 +27,7 @@ class ScrapeInput(BaseModel): "(at least one is required). Pair with location to scope a search." ), ) - urls: list[str] = Field( + urls: list[HttpUrlStr] = Field( default_factory=list, max_length=MAX_MAPS_SOURCES, description=( diff --git a/surfsense_backend/app/capabilities/indeed/__init__.py b/surfsense_backend/app/capabilities/indeed/__init__.py new file mode 100644 index 000000000..5e053244f --- /dev/null +++ b/surfsense_backend/app/capabilities/indeed/__init__.py @@ -0,0 +1,5 @@ +"""``indeed.*`` namespace: platform-native Indeed data verbs.""" + +from __future__ import annotations + +from app.capabilities.indeed.scrape import definition as _scrape # noqa: F401 diff --git a/surfsense_backend/app/capabilities/indeed/scrape/__init__.py b/surfsense_backend/app/capabilities/indeed/scrape/__init__.py new file mode 100644 index 000000000..6b7fa588e --- /dev/null +++ b/surfsense_backend/app/capabilities/indeed/scrape/__init__.py @@ -0,0 +1,3 @@ +"""``indeed.scrape`` verb: Indeed search / company URLs → job postings.""" + +from __future__ import annotations diff --git a/surfsense_backend/app/capabilities/indeed/scrape/definition.py b/surfsense_backend/app/capabilities/indeed/scrape/definition.py new file mode 100644 index 000000000..b831cd4a7 --- /dev/null +++ b/surfsense_backend/app/capabilities/indeed/scrape/definition.py @@ -0,0 +1,23 @@ +"""``indeed.scrape`` capability registration (billed per job; see config +``INDEED_SCRAPE_MICROS_PER_JOB``).""" + +from __future__ import annotations + +from app.capabilities.core import BillingUnit, Capability, register_capability +from app.capabilities.indeed.scrape.executor import build_scrape_executor +from app.capabilities.indeed.scrape.schemas import ScrapeInput, ScrapeOutput + +INDEED_SCRAPE = Capability( + name="indeed.scrape", + description=( + "Scrape public Indeed job postings, including title, company, location, " + "salary, and description. Use urls or search_queries." + ), + input_schema=ScrapeInput, + output_schema=ScrapeOutput, + executor=build_scrape_executor(), + billing_unit=BillingUnit.INDEED_JOB, + docs_url="/docs/connectors/native/indeed", +) + +register_capability(INDEED_SCRAPE) diff --git a/surfsense_backend/app/capabilities/indeed/scrape/executor.py b/surfsense_backend/app/capabilities/indeed/scrape/executor.py new file mode 100644 index 000000000..85983e0da --- /dev/null +++ b/surfsense_backend/app/capabilities/indeed/scrape/executor.py @@ -0,0 +1,56 @@ +"""``indeed.scrape`` executor: verb input → scraper → Indeed job items.""" + +from __future__ import annotations + +from collections.abc import Awaitable, Callable + +from app.capabilities.core import Executor +from app.capabilities.core.progress import emit_progress +from app.capabilities.indeed.scrape.schemas import ScrapeInput, ScrapeOutput +from app.exceptions import ForbiddenError +from app.proprietary.platforms.indeed_jobs import ( + IndeedAccessBlockedError, + IndeedScrapeInput, + scrape_indeed, +) + +ScrapeFn = Callable[..., Awaitable[list[dict]]] + + +def build_scrape_executor(scrape_fn: ScrapeFn | None = None) -> Executor: + """Bind the executor to a scraper fn (defaults to the proprietary actor).""" + scrape_fn = scrape_fn or scrape_indeed + + async def execute(payload: ScrapeInput) -> ScrapeOutput: + actor_input = IndeedScrapeInput( + startUrls=[{"url": url} for url in payload.urls], + queries=payload.search_queries, + country=payload.country, + location=payload.location, + radius=payload.radius, + jobType=payload.job_type, + level=payload.level, + remote=payload.remote, + fromDays=payload.from_days, + sort=payload.sort, + scrapeJobDetails=payload.scrape_job_details, + maxItems=payload.max_items, + maxItemsPerQuery=payload.max_items_per_query, + ) + emit_progress( + "starting", "Resolving Indeed targets", total=payload.max_items, unit="job" + ) + try: + items = await scrape_fn(actor_input, limit=payload.max_items) + except IndeedAccessBlockedError as exc: + # Anonymous-only scraper; a hard block can't be retried with creds. + raise ForbiddenError( + f"Indeed refused anonymous access: {exc}", + code="INDEED_ACCESS_BLOCKED", + ) from exc + emit_progress( + "done", f"Scraped {len(items)} job(s)", current=len(items), unit="job" + ) + return ScrapeOutput(items=items) + + return execute diff --git a/surfsense_backend/app/capabilities/indeed/scrape/schemas.py b/surfsense_backend/app/capabilities/indeed/scrape/schemas.py new file mode 100644 index 000000000..7c5dacb59 --- /dev/null +++ b/surfsense_backend/app/capabilities/indeed/scrape/schemas.py @@ -0,0 +1,120 @@ +"""``indeed.scrape`` I/O contracts. + +A lean, agent-friendly surface over ``IndeedScrapeInput`` +(``app/proprietary/platforms/indeed_jobs``). The executor maps this to the full +scraper input; the scraper's ``IndeedItem`` is reused verbatim as the output +element. +""" + +from __future__ import annotations + +from pydantic import BaseModel, Field, model_validator + +from app.capabilities.core.validation import HttpUrlStr +from app.proprietary.platforms.indeed_jobs import IndeedItem +from app.proprietary.platforms.indeed_jobs.schemas import ( + IndeedJobType, + IndeedLevel, + IndeedRemote, + IndeedSort, +) + +MAX_INDEED_SOURCES = 20 +"""Per-call cap on urls + search_queries: bounds a synchronous request's fan-out.""" + +MAX_INDEED_ITEMS = 100 +"""Hard ceiling on jobs returned per call, regardless of the per-query caps.""" + + +class ScrapeInput(BaseModel): + urls: list[HttpUrlStr] = Field( + default_factory=list, + max_length=MAX_INDEED_SOURCES, + description=( + "Indeed URLs to scrape: a search page (/jobs?q=&l=) or a company " + "jobs page (/cmp//jobs). Provide these OR search_queries " + "(at least one source is required)." + ), + ) + search_queries: list[str] = Field( + default_factory=list, + max_length=MAX_INDEED_SOURCES, + description=( + "Job search terms; each returns up to max_items_per_query results, " + "shaped by country/location/job_type/etc." + ), + ) + country: str = Field( + default="us", + description="Country code selecting the Indeed domain, e.g. 'us', 'gb', 'de'.", + ) + location: str | None = Field( + default=None, + description="Where to search, e.g. 'Remote', 'New York, NY'.", + ) + radius: int | None = Field( + default=None, + description="Search radius in miles/km around location.", + ) + job_type: IndeedJobType | None = Field( + default=None, + description="Employment type filter: fulltime, parttime, contract, etc.", + ) + level: IndeedLevel | None = Field( + default=None, + description="Experience level filter: entry_level, mid_level, senior_level.", + ) + remote: IndeedRemote | None = Field( + default=None, + description="Work model filter: remote or hybrid.", + ) + from_days: int | None = Field( + default=None, + description="Only return jobs posted within the last N days.", + ) + sort: IndeedSort = Field( + default="relevance", + description="Result ordering: relevance or date.", + ) + scrape_job_details: bool = Field( + default=False, + description=( + "Fetch each job's detail page for the full description (slower: one " + "extra page load per job)." + ), + ) + max_items: int = Field( + default=25, + ge=1, + le=MAX_INDEED_ITEMS, + description="Max total jobs to return across all sources.", + ) + max_items_per_query: int = Field( + default=25, + ge=0, + description="Max jobs to pull per search/company target.", + ) + + @model_validator(mode="after") + def _require_a_source(self) -> ScrapeInput: + if not self.urls and not self.search_queries: + raise ValueError("Provide at least one of 'urls' or 'search_queries'.") + return self + + @property + def estimated_units(self) -> int: + """Worst-case billable jobs for the pre-flight gate: ``max_items`` is a + hard cross-source ceiling (le=100), so no call can exceed it.""" + return self.max_items + + +class ScrapeOutput(BaseModel): + items: list[IndeedItem] = Field( + default_factory=list, + description="One item per job posting, in emission order.", + ) + + @property + def billable_units(self) -> int: + """One returned job = one billable unit.""" + return len(self.items) diff --git a/surfsense_backend/app/capabilities/reddit/scrape/schemas.py b/surfsense_backend/app/capabilities/reddit/scrape/schemas.py index 0c2da536e..9e0143ba7 100644 --- a/surfsense_backend/app/capabilities/reddit/scrape/schemas.py +++ b/surfsense_backend/app/capabilities/reddit/scrape/schemas.py @@ -10,6 +10,7 @@ from __future__ import annotations from pydantic import BaseModel, Field, model_validator +from app.capabilities.core.validation import HttpUrlStr from app.proprietary.platforms.reddit import RedditItem from app.proprietary.platforms.reddit.schemas import RedditSort, RedditTime @@ -21,7 +22,7 @@ MAX_REDDIT_ITEMS = 100 class ScrapeInput(BaseModel): - urls: list[str] = Field( + urls: list[HttpUrlStr] = Field( default_factory=list, max_length=MAX_REDDIT_SOURCES, description=( diff --git a/surfsense_backend/app/capabilities/tiktok/comments/schemas.py b/surfsense_backend/app/capabilities/tiktok/comments/schemas.py index f3c68d445..61ae0c756 100644 --- a/surfsense_backend/app/capabilities/tiktok/comments/schemas.py +++ b/surfsense_backend/app/capabilities/tiktok/comments/schemas.py @@ -10,6 +10,7 @@ from __future__ import annotations from pydantic import BaseModel, Field +from app.capabilities.core.validation import HttpUrlStr from app.capabilities.tiktok.scrape.schemas import ( MAX_TIKTOK_ITEMS, MAX_TIKTOK_SOURCES, @@ -18,7 +19,7 @@ from app.proprietary.platforms.tiktok import CommentItem class CommentsInput(BaseModel): - video_urls: list[str] = Field( + video_urls: list[HttpUrlStr] = Field( min_length=1, max_length=MAX_TIKTOK_SOURCES, description="TikTok video URLs (/@/video/) to pull comments from.", diff --git a/surfsense_backend/app/capabilities/tiktok/scrape/schemas.py b/surfsense_backend/app/capabilities/tiktok/scrape/schemas.py index ac3792712..b139e6e49 100644 --- a/surfsense_backend/app/capabilities/tiktok/scrape/schemas.py +++ b/surfsense_backend/app/capabilities/tiktok/scrape/schemas.py @@ -12,6 +12,7 @@ from __future__ import annotations from pydantic import BaseModel, Field, model_validator +from app.capabilities.core.validation import HttpUrlStr from app.proprietary.platforms.tiktok import TikTokVideoItem MAX_TIKTOK_SOURCES = 20 @@ -22,7 +23,7 @@ MAX_TIKTOK_ITEMS = 100 class ScrapeInput(BaseModel): - urls: list[str] = Field( + urls: list[HttpUrlStr] = Field( default_factory=list, max_length=MAX_TIKTOK_SOURCES, description=( diff --git a/surfsense_backend/app/capabilities/walmart/__init__.py b/surfsense_backend/app/capabilities/walmart/__init__.py new file mode 100644 index 000000000..1031f4230 --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/__init__.py @@ -0,0 +1,6 @@ +"""``walmart.*`` namespace: platform-native Walmart data verbs.""" + +from __future__ import annotations + +from app.capabilities.walmart.reviews import definition as _reviews # noqa: F401 +from app.capabilities.walmart.scrape import definition as _scrape # noqa: F401 diff --git a/surfsense_backend/app/capabilities/walmart/reviews/__init__.py b/surfsense_backend/app/capabilities/walmart/reviews/__init__.py new file mode 100644 index 000000000..473d60402 --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/reviews/__init__.py @@ -0,0 +1,3 @@ +"""Walmart deep review scraping capability.""" + +from __future__ import annotations diff --git a/surfsense_backend/app/capabilities/walmart/reviews/definition.py b/surfsense_backend/app/capabilities/walmart/reviews/definition.py new file mode 100644 index 000000000..6984ee031 --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/reviews/definition.py @@ -0,0 +1,24 @@ +"""``walmart.reviews`` capability registration (billed per review; see config +``WALMART_MICROS_PER_REVIEW``).""" + +from __future__ import annotations + +from app.capabilities.core import BillingUnit, Capability, register_capability +from app.capabilities.walmart.reviews.executor import build_reviews_executor +from app.capabilities.walmart.reviews.schemas import ReviewsInput, ReviewsOutput + +WALMART_REVIEWS = Capability( + name="walmart.reviews", + description=( + "Fetch deep paginated public Walmart product reviews with ratings, text, " + "authors, verified-purchase flags, images, and seller responses. Use " + "product urls or item ids." + ), + input_schema=ReviewsInput, + output_schema=ReviewsOutput, + executor=build_reviews_executor(), + billing_unit=BillingUnit.WALMART_REVIEW, + docs_url="/docs/connectors/native/walmart", +) + +register_capability(WALMART_REVIEWS) diff --git a/surfsense_backend/app/capabilities/walmart/reviews/executor.py b/surfsense_backend/app/capabilities/walmart/reviews/executor.py new file mode 100644 index 000000000..ba1ee0367 --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/reviews/executor.py @@ -0,0 +1,40 @@ +"""``walmart.reviews`` executor: verb input → scraper → review items.""" + +from __future__ import annotations + +from collections.abc import Awaitable, Callable + +from app.capabilities.core import Executor +from app.capabilities.core.progress import emit_progress +from app.capabilities.walmart.reviews.schemas import ReviewsInput, ReviewsOutput +from app.proprietary.platforms.walmart import WalmartReviewsInput, scrape_reviews + +ReviewsFn = Callable[..., Awaitable[list[dict]]] + + +def build_reviews_executor(scrape_fn: ReviewsFn | None = None) -> Executor: + """Bind the capability input mapping to a replaceable reviews scraper function.""" + scrape_fn = scrape_fn or scrape_reviews + + async def execute(payload: ReviewsInput) -> ReviewsOutput: + input_model = WalmartReviewsInput( + itemIds=payload.sources(), + maxReviews=payload.max_reviews, + sort=payload.sort_by, + ) + emit_progress( + "starting", + "Fetching Walmart reviews", + total=payload.estimated_units, + unit="review", + ) + items = await scrape_fn(input_model, limit=payload.estimated_units) + emit_progress( + "done", + f"Scraped {sum('error' not in item for item in items)} review(s)", + current=len(items), + unit="review", + ) + return ReviewsOutput(items=items) + + return execute diff --git a/surfsense_backend/app/capabilities/walmart/reviews/schemas.py b/surfsense_backend/app/capabilities/walmart/reviews/schemas.py new file mode 100644 index 000000000..bd1dc73d9 --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/reviews/schemas.py @@ -0,0 +1,73 @@ +"""``walmart.reviews`` I/O contracts. + +A lean surface over ``WalmartReviewsInput``; the scraper's ``ReviewItem`` is +reused verbatim as the output element. Accepts product URLs or bare item ids — +both resolve to a ``usItemId`` the reviews page is keyed on. +""" + +from __future__ import annotations + +from typing import Literal + +from pydantic import BaseModel, Field, model_validator + +from app.capabilities.core.validation import HttpUrlStr +from app.proprietary.platforms.walmart import ReviewItem + +MAX_WALMART_REVIEW_SOURCES = 20 + + +class ReviewsInput(BaseModel): + urls: list[HttpUrlStr] = Field( + default_factory=list, + max_length=MAX_WALMART_REVIEW_SOURCES, + description=( + "Walmart product URLs (/ip/...) or reviews URLs to fetch reviews for. " + "Provide these OR item_ids (at least one is required)." + ), + ) + item_ids: list[str] = Field( + default_factory=list, + max_length=MAX_WALMART_REVIEW_SOURCES, + description="Walmart numeric item ids (usItemId) to fetch reviews for.", + ) + max_reviews: int = Field( + default=200, + ge=1, + le=5000, + description="Max reviews to return per product (10 per page).", + ) + sort_by: Literal["most-recent", "most-helpful", "rating-high", "rating-low"] = ( + Field(default="most-recent", description="Review ordering.") + ) + + @model_validator(mode="after") + def _require_a_source(self) -> ReviewsInput: + if not (self.urls or self.item_ids): + raise ValueError("Provide at least one of 'urls' or 'item_ids'.") + return self + + def sources(self) -> list[str]: + """URLs and item ids merged; the scraper resolves each to a usItemId.""" + return [*self.urls, *self.item_ids] + + @property + def estimated_units(self) -> int: + """Worst-case billable reviews: up to ``max_reviews`` per source.""" + return (len(self.urls) + len(self.item_ids)) * self.max_reviews + + +class ReviewsOutput(BaseModel): + items: list[ReviewItem] = Field( + default_factory=list, + description="One item per review, in the scraper's emission order.", + ) + + @property + def billable_units(self) -> int: + """One returned review = one billable unit; error items are not billed.""" + return sum( + 1 + for item in self.items + if not (item.model_extra and item.model_extra.get("error")) + ) diff --git a/surfsense_backend/app/capabilities/walmart/scrape/__init__.py b/surfsense_backend/app/capabilities/walmart/scrape/__init__.py new file mode 100644 index 000000000..bb3e8f794 --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/scrape/__init__.py @@ -0,0 +1,3 @@ +"""Walmart product scraping capability.""" + +from __future__ import annotations diff --git a/surfsense_backend/app/capabilities/walmart/scrape/definition.py b/surfsense_backend/app/capabilities/walmart/scrape/definition.py new file mode 100644 index 000000000..6165a8992 --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/scrape/definition.py @@ -0,0 +1,22 @@ +"""Registration for the ``walmart.scrape`` capability.""" + +from __future__ import annotations + +from app.capabilities.core import BillingUnit, Capability, register_capability +from app.capabilities.walmart.scrape.executor import build_scrape_executor +from app.capabilities.walmart.scrape.schemas import ScrapeInput, ScrapeOutput + +WALMART_SCRAPE = Capability( + name="walmart.scrape", + description=( + "Scrape public Walmart product details, search/category listings, " + "prices, sellers, variants, availability, and a sample of on-page reviews." + ), + input_schema=ScrapeInput, + output_schema=ScrapeOutput, + executor=build_scrape_executor(), + billing_unit=BillingUnit.WALMART_PRODUCT, + docs_url="/docs/connectors/native/walmart", +) + +register_capability(WALMART_SCRAPE) diff --git a/surfsense_backend/app/capabilities/walmart/scrape/executor.py b/surfsense_backend/app/capabilities/walmart/scrape/executor.py new file mode 100644 index 000000000..4c4b073ee --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/scrape/executor.py @@ -0,0 +1,46 @@ +"""Executor for the ``walmart.scrape`` capability.""" + +from __future__ import annotations + +from collections.abc import Awaitable, Callable + +from app.capabilities.core import Executor +from app.capabilities.core.progress import emit_progress +from app.capabilities.walmart.scrape.schemas import ( + MAX_WALMART_RESULTS, + ScrapeInput, + ScrapeOutput, +) +from app.proprietary.platforms.walmart import WalmartScrapeInput, scrape_products + +ScrapeFn = Callable[..., Awaitable[list[dict]]] + + +def build_scrape_executor(scrape_fn: ScrapeFn | None = None) -> Executor: + """Bind the capability input mapping to a replaceable scraper function.""" + scrape_fn = scrape_fn or scrape_products + + async def execute(payload: ScrapeInput) -> ScrapeOutput: + input_model = WalmartScrapeInput( + startUrls=payload.start_urls(), + maxItemsPerStartUrl=payload.max_items, + includeDetails=payload.include_details, + includeReviewsSample=payload.include_reviews_sample, + ) + emit_progress( + "starting", + "Scraping Walmart products", + total=payload.estimated_units, + unit="product", + ) + items = await scrape_fn(input_model, limit=MAX_WALMART_RESULTS) + emit_progress( + "done", + f"Scraped {sum('error' not in item for item in items)} product(s)", + current=len(items), + total=payload.estimated_units, + unit="product", + ) + return ScrapeOutput(items=items) + + return execute diff --git a/surfsense_backend/app/capabilities/walmart/scrape/schemas.py b/surfsense_backend/app/capabilities/walmart/scrape/schemas.py new file mode 100644 index 000000000..7faaabd1d --- /dev/null +++ b/surfsense_backend/app/capabilities/walmart/scrape/schemas.py @@ -0,0 +1,65 @@ +"""Input and output contracts for ``walmart.scrape``.""" + +from __future__ import annotations + +from urllib.parse import quote_plus + +from pydantic import BaseModel, Field, model_validator + +from app.capabilities.core.validation import HttpUrlStr +from app.proprietary.platforms.walmart import ProductItem + +MAX_WALMART_SOURCES = 20 +MAX_WALMART_RESULTS = 1000 + + +class ScrapeInput(BaseModel): + """Agent-facing controls for public Walmart product discovery and enrichment.""" + + urls: list[HttpUrlStr] = Field(default_factory=list, max_length=MAX_WALMART_SOURCES) + search_terms: list[str] = Field( + default_factory=list, max_length=MAX_WALMART_SOURCES + ) + max_items: int = Field(default=10, ge=1, le=100) + include_details: bool = True + include_reviews_sample: bool = True + + @model_validator(mode="after") + def _require_source(self) -> ScrapeInput: + if not (self.urls or self.search_terms): + raise ValueError("Provide at least one URL or search term.") + if len(self.urls) + len(self.search_terms) > MAX_WALMART_SOURCES: + raise ValueError( + f"Provide no more than {MAX_WALMART_SOURCES} combined sources." + ) + return self + + def start_urls(self) -> list[str]: + """Direct URLs plus a search URL synthesized per search term.""" + searches = [ + f"https://www.walmart.com/search?q={quote_plus(term)}" + for term in self.search_terms + ] + return [*self.urls, *searches] + + @property + def estimated_units(self) -> int: + """Worst-case returned products within the hard per-run ceiling.""" + search_products = len(self.search_terms) * self.max_items + direct_products = len(self.urls) * self.max_items + return min(search_products + direct_products, MAX_WALMART_RESULTS) + + +class ScrapeOutput(BaseModel): + """Products and structured per-input errors in emission order.""" + + items: list[ProductItem] = Field(default_factory=list) + + @property + def billable_units(self) -> int: + """Count successful products; error items are never billed.""" + return sum( + 1 + for item in self.items + if not (item.model_extra and item.model_extra.get("error")) + ) diff --git a/surfsense_backend/app/capabilities/web/crawl/schemas.py b/surfsense_backend/app/capabilities/web/crawl/schemas.py index d6c9c38a3..30e8ff6ae 100644 --- a/surfsense_backend/app/capabilities/web/crawl/schemas.py +++ b/surfsense_backend/app/capabilities/web/crawl/schemas.py @@ -18,6 +18,8 @@ from typing import Literal from pydantic import BaseModel, Field +from app.capabilities.core.validation import HttpUrlStr + MAX_START_URLS = 20 """Per-call cap on seed URLs: bounds a synchronous request's fan-out (05).""" @@ -29,7 +31,7 @@ MAX_CRAWL_PAGES = 200 class CrawlInput(BaseModel): - startUrls: list[str] = Field( + startUrls: list[HttpUrlStr] = Field( min_length=1, max_length=MAX_START_URLS, description=( diff --git a/surfsense_backend/app/capabilities/youtube/comments/schemas.py b/surfsense_backend/app/capabilities/youtube/comments/schemas.py index a9da5b7a2..ecfa80026 100644 --- a/surfsense_backend/app/capabilities/youtube/comments/schemas.py +++ b/surfsense_backend/app/capabilities/youtube/comments/schemas.py @@ -10,6 +10,7 @@ from typing import Literal from pydantic import BaseModel, Field +from app.capabilities.core.validation import HttpUrlStr from app.proprietary.platforms.youtube import CommentItem MAX_COMMENT_VIDEOS = 20 @@ -17,7 +18,7 @@ MAX_COMMENT_VIDEOS = 20 class CommentsInput(BaseModel): - urls: list[str] = Field( + urls: list[HttpUrlStr] = Field( min_length=1, max_length=MAX_COMMENT_VIDEOS, description="YouTube video URLs to fetch comments (and replies) for (1-20).", diff --git a/surfsense_backend/app/capabilities/youtube/scrape/schemas.py b/surfsense_backend/app/capabilities/youtube/scrape/schemas.py index 4d80769b6..63a1df060 100644 --- a/surfsense_backend/app/capabilities/youtube/scrape/schemas.py +++ b/surfsense_backend/app/capabilities/youtube/scrape/schemas.py @@ -10,6 +10,7 @@ from __future__ import annotations from pydantic import BaseModel, Field, model_validator +from app.capabilities.core.validation import HttpUrlStr from app.proprietary.platforms.youtube import VideoItem MAX_YOUTUBE_SOURCES = 20 @@ -17,7 +18,7 @@ MAX_YOUTUBE_SOURCES = 20 class ScrapeInput(BaseModel): - urls: list[str] = Field( + urls: list[HttpUrlStr] = Field( default_factory=list, max_length=MAX_YOUTUBE_SOURCES, description=( diff --git a/surfsense_backend/app/celery_app.py b/surfsense_backend/app/celery_app.py index 471d80197..2a9cddbd5 100644 --- a/surfsense_backend/app/celery_app.py +++ b/surfsense_backend/app/celery_app.py @@ -10,6 +10,7 @@ from celery.signals import ( task_postrun, task_prerun, worker_process_init, + worker_process_shutdown, ) from dotenv import load_dotenv @@ -123,6 +124,18 @@ def init_worker(**kwargs): initialize_image_gen_router() +@worker_process_shutdown.connect +def shutdown_worker(**kwargs): + """Flush queued PostHog events before a Celery worker process exits. + + The analytics client init is lazy (fork-safe), so there is nothing to + start here — only a flush to avoid dropping events captured by tasks. + """ + from app.observability import analytics as ph_analytics + + ph_analytics.shutdown() + + # Celery configuration, sourced from the central Config singleton CELERY_BROKER_URL = config.CELERY_BROKER_URL CELERY_RESULT_BACKEND = config.CELERY_RESULT_BACKEND diff --git a/surfsense_backend/app/config/__init__.py b/surfsense_backend/app/config/__init__.py index 20497bb15..45ef065df 100644 --- a/surfsense_backend/app/config/__init__.py +++ b/surfsense_backend/app/config/__init__.py @@ -736,6 +736,17 @@ class Config: # Comments are the cheapest per-item TikTok data, matching the per-comment # market (and YouTube's comment meter). TIKTOK_MICROS_PER_COMMENT = int(os.getenv("TIKTOK_MICROS_PER_COMMENT", "1500")) + # Warmed-browser listings put Indeed on par with the other browser-driven + # scrapers (Reddit, Instagram) rather than the cheaper API-backed meters. + INDEED_SCRAPE_MICROS_PER_JOB = int( + os.getenv("INDEED_SCRAPE_MICROS_PER_JOB", "3500") + ) + # Walmart products come from server-rendered JSON behind residential proxies, + # priced alongside Amazon's per-product meter. + WALMART_MICROS_PER_PRODUCT = int(os.getenv("WALMART_MICROS_PER_PRODUCT", "3500")) + # Reviews are 10 per page (many light requests per product), priced on the + # cheaper per-review market like the Google Maps review meter. + WALMART_MICROS_PER_REVIEW = int(os.getenv("WALMART_MICROS_PER_REVIEW", "1500")) # Retry an empty listing draw on a fresh rotating IP. Set to 1 for a static # proxy, where every retry re-hits the same exit. TIKTOK_LISTING_MAX_ATTEMPTS = int(os.getenv("TIKTOK_LISTING_MAX_ATTEMPTS", "3")) @@ -850,6 +861,9 @@ class Config: # Auth AUTH_TYPE = os.getenv("AUTH_TYPE", "LOCAL") REGISTRATION_ENABLED = os.getenv("REGISTRATION_ENABLED", "TRUE").upper() == "TRUE" + # Max workspaces a user may own. The frontend reads this through the + # workspace limits route; do not duplicate this value client-side. + MAX_WORKSPACES_PER_USER = int(os.getenv("MAX_WORKSPACES_PER_USER", "400")) # Google OAuth GOOGLE_OAUTH_CLIENT_ID = os.getenv("GOOGLE_OAUTH_CLIENT_ID") @@ -1165,6 +1179,19 @@ class Config: os.getenv("CRAWL_HEADED_XVFB_ENABLED", "FALSE").upper() == "TRUE" ) + # PostHog server-side product analytics (opt-in, mirrors the OTel pattern: + # no key set => the analytics wrapper is a silent no-op). Use the SAME + # project key as the frontend's NEXT_PUBLIC_POSTHOG_KEY so server events + # merge onto the persons the web app already identifies by user id. + POSTHOG_API_KEY = os.getenv("POSTHOG_API_KEY") + POSTHOG_HOST = os.getenv("POSTHOG_HOST", "https://us.i.posthog.com") + # When true (default), the LLM-analytics LangChain handler suppresses + # prompt/completion bodies ($ai_input / $ai_output_choices) and captures + # only metrics — chat content includes users' private documents. + POSTHOG_AI_PRIVACY_MODE = ( + os.getenv("POSTHOG_AI_PRIVACY_MODE", "TRUE").upper() == "TRUE" + ) + # Litellm TTS Configuration TTS_SERVICE = os.getenv("TTS_SERVICE") TTS_SERVICE_API_BASE = os.getenv("TTS_SERVICE_API_BASE") diff --git a/surfsense_backend/app/observability/__init__.py b/surfsense_backend/app/observability/__init__.py index a675b1dae..4ec9f1b03 100644 --- a/surfsense_backend/app/observability/__init__.py +++ b/surfsense_backend/app/observability/__init__.py @@ -6,4 +6,4 @@ wrapper is a no-op when OTEL is not configured, so importing it from performance-critical paths is safe. """ -__all__ = ["bootstrap", "metrics", "otel"] +__all__ = ["analytics", "bootstrap", "metrics", "otel"] diff --git a/surfsense_backend/app/observability/analytics.py b/surfsense_backend/app/observability/analytics.py new file mode 100644 index 000000000..312335d24 --- /dev/null +++ b/surfsense_backend/app/observability/analytics.py @@ -0,0 +1,202 @@ +"""Server-side PostHog product analytics for SurfSense. + +Opt-in, mirroring the OpenTelemetry bootstrap contract: when +``POSTHOG_API_KEY`` is unset every function here is a silent no-op, so it is +safe to call from hot paths (including async request handlers) and from +self-hosted installs that never configure telemetry. + +Design notes: +- The underlying ``posthog`` client enqueues events onto a background + consumer thread, so ``capture()`` is a non-blocking queue append; the only + network I/O happens off-thread. ``shutdown()`` flushes and joins that thread + and MUST run before a process exits or queued events are lost. +- The client is created lazily on first use, never at import time. This keeps + it fork-safe under Celery's prefork pool: a client (and its consumer thread) + created in the parent would not survive ``fork()``, so each worker process + builds its own on first capture. +- ``distinct_id`` is always ``str(user.id)`` so server events merge onto the + same PostHog persons the web frontend identifies (see + ``surfsense_web/components/providers/PostHogIdentify.tsx``). +- Every event passes ``disable_geoip=True``; without it PostHog would resolve + the *server's* IP and overwrite each person's real (client-derived) location. +""" + +from __future__ import annotations + +import logging +import threading +from typing import TYPE_CHECKING, Any + +from app.config import config + +if TYPE_CHECKING: + from app.auth.context import AuthContext + +logger = logging.getLogger(__name__) + +_client: Any | None = None +_init_attempted = False +_lock = threading.Lock() + +# Stamped on every backend event so client-observed (frontend) and +# server-truth events are always distinguishable in PostHog. +_SOURCE = "backend" + + +def _get_client() -> Any | None: + """Return the process-local PostHog client, or ``None`` when disabled. + + Lazy + fork-safe: built on first use inside whichever process (web worker + or Celery worker) calls it, never at import time. + """ + global _client, _init_attempted + + if _init_attempted: + return _client + + with _lock: + if _init_attempted: + return _client + _init_attempted = True + + api_key = config.POSTHOG_API_KEY + if not api_key: + # ponytail: opt-in like OTel — no key means telemetry is off, not + # a misconfiguration. Stay silent so self-hosters see no noise. + return None + + try: + from posthog import Posthog + + _client = Posthog( + project_api_key=api_key, + host=config.POSTHOG_HOST, + ) + except Exception: + logger.warning("PostHog analytics init failed; disabling", exc_info=True) + _client = None + + return _client + + +def is_enabled() -> bool: + """True when a PostHog client is configured and available.""" + return _get_client() is not None + + +def get_client() -> Any | None: + """Raw PostHog client for integrations that need it (e.g. the LLM handler).""" + return _get_client() + + +def _client_label(auth: AuthContext) -> str: + """Best-effort ``client`` property derived from the auth principal. + + ``session`` can't be split into web vs desktop from auth alone, so callers + that know better may override ``client`` in ``properties``. + """ + if auth.method == "system": + return auth.source or "system" + if auth.method == "pat": + return "pat" + return "web" + + +def capture( + event: str, + *, + distinct_id: str, + properties: dict[str, Any] | None = None, + groups: dict[str, str] | None = None, +) -> None: + """Capture a product event. No-op (and never raises) when disabled. + + Wrapped in try/except like the frontend ``safeCapture`` — analytics must + never break a request. ``posthog`` v6 signature is ``capture(event, + distinct_id=..., properties=...)`` (event first, distinct_id a kwarg). + """ + client = _get_client() + if client is None: + return + + try: + props = {"source": _SOURCE, **(properties or {})} + client.capture( + event, + distinct_id=distinct_id, + properties=props, + groups=groups, + disable_geoip=True, + ) + except Exception: + logger.debug("PostHog capture failed for %s", event, exc_info=True) + + +def capture_for( + auth: AuthContext, + event: str, + properties: dict[str, Any] | None = None, + groups: dict[str, str] | None = None, +) -> None: + """Capture an event attributed to an ``AuthContext`` principal. + + Derives ``distinct_id`` from the user id and stamps ``auth_method`` and a + best-effort ``client`` so events are attributable to their surface + (web/desktop/pat/gateway/automation). + """ + if _get_client() is None: + return + + props = { + "auth_method": auth.method, + "client": _client_label(auth), + **(properties or {}), + } + capture( + event, + distinct_id=str(auth.user.id), + properties=props, + groups=groups, + ) + + +def group_identify( + group_type: str, + group_key: str, + properties: dict[str, Any] | None = None, +) -> None: + """Upsert group properties (e.g. per-workspace metadata). No-op when disabled.""" + client = _get_client() + if client is None: + return + + try: + client.group_identify( + group_type=group_type, + group_key=group_key, + properties=properties or {}, + ) + except Exception: + logger.debug("PostHog group_identify failed for %s", group_type, exc_info=True) + + +def shutdown() -> None: + """Flush queued events and stop the consumer thread. Safe to call always.""" + global _client + client = _client + if client is None: + return + try: + client.shutdown() + except Exception: + logger.debug("PostHog shutdown failed", exc_info=True) + + +__all__ = [ + "capture", + "capture_for", + "get_client", + "group_identify", + "is_enabled", + "shutdown", +] diff --git a/surfsense_backend/app/podcasts/tasks/render.py b/surfsense_backend/app/podcasts/tasks/render.py index 7759691a9..0cd6fbe5a 100644 --- a/surfsense_backend/app/podcasts/tasks/render.py +++ b/surfsense_backend/app/podcasts/tasks/render.py @@ -11,7 +11,10 @@ import logging import tempfile from pathlib import Path +from sqlalchemy import select + from app.celery_app import celery_app +from app.observability import analytics as ph_analytics from app.podcasts.persistence import PodcastRepository from app.podcasts.rendering import PodcastRenderer from app.podcasts.service import ( @@ -76,6 +79,30 @@ async def _render_audio(podcast_id: int) -> dict: podcast, storage_backend=backend_name, storage_key=key ) await session.commit() + + # Credit-consuming deliverable; the frontend never confirms the + # render finished. Owner (workspace.user_id) resolved lazily so + # disabled installs pay nothing for the extra query. + if ph_analytics.is_enabled(): + # Local import: app.db <-> app.podcasts.persistence have a + # module-init cycle; deferring keeps this task importable. + from app.db import Workspace + + owner_id = await session.scalar( + select(Workspace.user_id).where( + Workspace.id == podcast.workspace_id + ) + ) + if owner_id: + ph_analytics.capture( + "podcast_generated", + distinct_id=str(owner_id), + properties={ + "workspace_id": podcast.workspace_id, + "podcast_id": podcast_id, + }, + groups={"workspace": str(podcast.workspace_id)}, + ) except InvalidTransitionError: # A user back-out won the race (e.g. the regeneration was # reverted): drop the stale render and leave the row alone. diff --git a/surfsense_backend/app/proprietary/platforms/google_search/fetch.py b/surfsense_backend/app/proprietary/platforms/google_search/fetch.py index 49c6fcabd..bc9c4402f 100644 --- a/surfsense_backend/app/proprietary/platforms/google_search/fetch.py +++ b/surfsense_backend/app/proprietary/platforms/google_search/fetch.py @@ -35,7 +35,6 @@ import contextlib import logging import os import random -import sys import threading import time from urllib.parse import urlsplit, urlunsplit @@ -46,6 +45,7 @@ from app.proprietary.platforms.google_search import ( captcha as _captcha, pool_store as _store, ) +from app.utils.browser_loop import in_browser_loop as _in_browser_loop from app.utils.captcha import captcha_enabled, get_captcha_config from app.utils.proxy import get_proxy_url @@ -393,41 +393,11 @@ _MOBILE_UA = "Mozilla/5.0 (Android 14; Mobile; rv:132.0) Gecko/132.0 Firefox/132 _MOBILE_VIEWPORT = {"width": 412, "height": 915} -# patchright launches Chromium via asyncio.create_subprocess_exec, which the -# server's main loop cannot do on Windows (main.py pins a SelectorEventLoop -# for psycopg; Selector loops raise NotImplementedError on subprocess_exec). -# All browser work therefore runs on ONE dedicated background loop that is -# explicitly subprocess-capable; callers await it across threads. This also -# keeps the persistent AsyncStealthySession (and its async page_action) intact -# — the sync-fetcher-in-a-thread pattern the other scrapers use would tear -# down the browser on every fetch. -_browser_loop: asyncio.AbstractEventLoop | None = None -_browser_loop_guard = threading.Lock() - - -def _get_browser_loop() -> asyncio.AbstractEventLoop: - """The lazily-started, process-wide event loop the browser lives on.""" - global _browser_loop - with _browser_loop_guard: - if _browser_loop is None: - loop = ( - asyncio.ProactorEventLoop() - if sys.platform == "win32" - else asyncio.new_event_loop() - ) - threading.Thread( - target=loop.run_forever, name="google-search-browser", daemon=True - ).start() - _browser_loop = loop - return _browser_loop - - -async def _in_browser_loop(coro): - """Run ``coro`` on the browser loop and await its result from this loop.""" - return await asyncio.wrap_future( - asyncio.run_coroutine_threadsafe(coro, _get_browser_loop()) - ) - +# All browser work runs on the shared subprocess-capable loop (Windows: the +# server's SelectorEventLoop cannot spawn Chromium; see app.utils.browser_loop). +# This also keeps the persistent AsyncStealthySession (and its async +# page_action) intact — the sync-fetcher-in-a-thread pattern would tear down +# the browser on every fetch. # One live browser per layout (desktop / mobile — the UA and viewport are # session-level context options). Launching Chromium costs ~5 s, so it's paid diff --git a/surfsense_backend/app/proprietary/platforms/indeed_jobs/__init__.py b/surfsense_backend/app/proprietary/platforms/indeed_jobs/__init__.py new file mode 100644 index 000000000..eb59671d7 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/indeed_jobs/__init__.py @@ -0,0 +1,13 @@ +"""Platform-native Indeed jobs scraper (anonymous, warmed browser session).""" + +from .fetch import IndeedAccessBlockedError +from .schemas import IndeedItem, IndeedScrapeInput +from .scraper import iter_indeed, scrape_indeed + +__all__ = [ + "IndeedAccessBlockedError", + "IndeedItem", + "IndeedScrapeInput", + "iter_indeed", + "scrape_indeed", +] diff --git a/surfsense_backend/app/proprietary/platforms/indeed_jobs/fetch.py b/surfsense_backend/app/proprietary/platforms/indeed_jobs/fetch.py new file mode 100644 index 000000000..e29aef0a9 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/indeed_jobs/fetch.py @@ -0,0 +1,200 @@ +"""Browser-session fetch seam for the Indeed scraper. + +Indeed fronts its origin with Cloudflare plus an anonymous-bot check that bounces +cold sessions to ``secure.indeed.com/auth``. The working recipe: a persistent +camoufox session that solves Cloudflare, warms on the domain home page, then +navigates to ``/jobs`` in the same context so the clearance carries. + +:class:`IndeedSession` warms per domain once, retries a blocked page on a fresh +residential IP, and caps each navigation with a hard timeout so a stuck solve +can't stall a run. All egress is through the residential proxy. +""" + +from __future__ import annotations + +import asyncio +import logging +from collections.abc import Awaitable, Callable +from contextlib import asynccontextmanager, suppress +from datetime import UTC, datetime +from typing import Any, Protocol +from urllib.parse import urlparse + +from app.utils.browser_loop import in_browser_loop +from app.utils.proxy import get_proxy_url + +logger = logging.getLogger(__name__) + + +class IndeedAccessBlockedError(RuntimeError): + """Every rotated IP was bounced to Indeed's security wall.""" + + +# Per navigation; a stuck Cloudflare solve otherwise hangs the whole run. +_PAGE_TIMEOUT_S = 75.0 +# Browser-internal timeout; kept above the page timeout so ours fires first. +_SESSION_TIMEOUT_MS = 90_000 +_MAX_ROTATIONS = 3 + +# Markers of a Cloudflare / security-check interstitial served instead of jobs. +_BLOCK_MARKERS = ( + "secure.indeed.com", + "bot-detection", + "security check", + "challenge-platform", + "just a moment", + "verify you are human", + "hcaptcha", +) + + +def now_iso() -> str: + """UTC timestamp in the millisecond ISO shape used by scraper output.""" + return datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z" + + +class _Session(Protocol): + """Minimal browser-session surface used here (real or fake).""" + + async def start(self) -> Any: ... + async def fetch(self, url: str, **kwargs: Any) -> Any: ... + async def close(self) -> Any: ... + + +def _default_session_factory() -> _Session: + """Build a proxied, Cloudflare-solving camoufox session. + + ``disable_resources`` skips images/fonts/media; job data is inline in the + document, so this only trims bandwidth. + """ + from scrapling.fetchers import AsyncStealthySession + + return AsyncStealthySession( + headless=True, + solve_cloudflare=True, + network_idle=True, + block_webrtc=True, + disable_resources=True, + timeout=_SESSION_TIMEOUT_MS, + proxy=get_proxy_url(), + ) + + +def _html(page: Any) -> str: + """Best-effort HTML body across scrapling response shapes.""" + for attr in ("html_content", "body", "text"): + val = getattr(page, attr, None) + if isinstance(val, bytes): + val = val.decode("utf-8", "replace") + if isinstance(val, str) and val: + return val + return "" + + +def _looks_blocked(html: str, final_url: str) -> bool: + """Whether a response is an interstitial rather than a real page.""" + if not html: + return True + haystack = (final_url + " " + html[:6000]).lower() + return any(marker in haystack for marker in _BLOCK_MARKERS) + + +class IndeedSession: + """One warmed browser session that rotates its exit IP when blocked.""" + + def __init__( + self, session_factory: Callable[[], _Session] = _default_session_factory + ) -> None: + self._factory = session_factory + self._session: _Session | None = None + self._warmed: set[str] = set() + self.rotations = 0 + + # The session's whole lifecycle (build, start, fetch, close) is marshalled + # onto the shared browser loop: patchright can't spawn Chromium from the + # server's Windows SelectorEventLoop (see app.utils.browser_loop), and its + # internals are bound to the loop they started on. + + async def start(self) -> None: + async def _build_and_start() -> _Session: + session = self._factory() + await session.start() + return session + + self._session = await in_browser_loop(_build_and_start()) + + async def close(self) -> None: + if self._session is not None: + with suppress(Exception): + await in_browser_loop(self._session.close()) + self._session = None + self._warmed.clear() + + async def _rotate(self) -> None: + """Drop the session for a fresh exit IP; clears warmed domains.""" + await self.close() + self.rotations += 1 + await self.start() + logger.info("[indeed] rotated session (rotation #%d)", self.rotations) + + async def _timed_fetch(self, url: str, **kwargs: Any) -> Any: + assert self._session is not None + coro: Awaitable[Any] = self._session.fetch(url, **kwargs) + # wait_for runs on the browser loop too, so its timeout task lives on + # the same loop as the fetch it cancels. + return await in_browser_loop(asyncio.wait_for(coro, timeout=_PAGE_TIMEOUT_S)) + + async def _ensure_warm(self, domain: str) -> None: + """Land on the domain home with a Google referer before scraping it.""" + if domain in self._warmed: + return + with suppress(Exception): + await self._timed_fetch(f"https://{domain}/", google_search=True) + self._warmed.add(domain) + + async def fetch_html(self, url: str, *, max_rotations: int | None = None) -> str: + """Return a search/company/job page's HTML through the warmed session. + + Rotates the IP and re-warms on a security-wall bounce or timeout; raises + :class:`IndeedAccessBlockedError` once rotations are exhausted. ``max_rotations`` + overrides the default budget: pass ``0`` to fail fast on a systematically + gated page (e.g. anonymous pagination) instead of burning rotations on a + block no fresh IP will clear. + """ + if self._session is None: + await self.start() + budget = _MAX_ROTATIONS if max_rotations is None else max_rotations + domain = urlparse(url).hostname or "www.indeed.com" + attempt = 0 + while True: + try: + await self._ensure_warm(domain) + page = await self._timed_fetch(url) + html = _html(page) + if not _looks_blocked(html, str(getattr(page, "url", "") or "")): + return html + logger.info("[indeed] blocked on %s", url) + except TimeoutError: + logger.warning("[indeed] fetch timed out on %s", url) + except Exception as e: + logger.warning("[indeed] fetch failed on %s: %s", url, e) + + if attempt >= budget: + raise IndeedAccessBlockedError( + f"Indeed refused {url} after {attempt + 1} attempt(s)" + ) + attempt += 1 + await self._rotate() + + +@asynccontextmanager +async def open_session( + session_factory: Callable[[], _Session] = _default_session_factory, +): + """Open an :class:`IndeedSession` and guarantee teardown.""" + session = IndeedSession(session_factory) + await session.start() + try: + yield session + finally: + await session.close() diff --git a/surfsense_backend/app/proprietary/platforms/indeed_jobs/parsers.py b/surfsense_backend/app/proprietary/platforms/indeed_jobs/parsers.py new file mode 100644 index 000000000..87824e8c7 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/indeed_jobs/parsers.py @@ -0,0 +1,327 @@ +"""Pure HTML/JSON -> item mapping for the Indeed scraper. + +I/O-free and deterministic so it can be unit-tested against captured fixtures; +the orchestrator stamps ``scrapedAt``. + +Indeed embeds job data as a JS assignment:: + + window.mosaic.providerData["mosaic-provider-jobcards"]={"metaData":{...}}; + +The literal recurs hundreds of times in the page, so :func:`extract_jobcards_blob` +anchors on the assignment (``...]=``) and brace-matches the balanced object. +""" + +from __future__ import annotations + +from datetime import UTC, datetime +from html import unescape +from re import sub as _re_sub +from typing import Any + +_DEFAULT_BASE = "https://www.indeed.com" + +_JOBCARDS_ANCHOR = 'window.mosaic.providerData["mosaic-provider-jobcards"]=' + +# A /viewjob page carries the posting model in ``window._rootProps`` (JSON, +# under ``preloadedVJData``); older pages inlined it as ``window._initialData``. +_ROOT_PROPS_ANCHOR = "window._rootProps" +_ROOT_PROPS_KEY = "preloadedVJData" +_INITIAL_DATA_ANCHOR = "window._initialData" + +# Indeed's extractedSalary.type -> our SalaryPeriod. +_SALARY_PERIODS = { + "HOURLY": "hour", + "DAILY": "day", + "WEEKLY": "week", + "MONTHLY": "month", + "YEARLY": "year", +} + + +def _brace_match(text: str, start: int) -> str | None: + """Return the balanced ``{...}``/``[...]`` blob at ``text[start]``, quote-aware.""" + open_ch = text[start] if start < len(text) else "" + close_ch = {"[": "]", "{": "}"}.get(open_ch) + if close_ch is None: + return None + depth = 0 + i = start + n = len(text) + while i < n: + ch = text[i] + if ch == open_ch: + depth += 1 + elif ch == close_ch: + depth -= 1 + if depth == 0: + return text[start : i + 1] + elif ch == '"': + i += 1 + while i < n and text[i] != '"': + if text[i] == "\\": + i += 1 + i += 1 + i += 1 + return None + + +def extract_jobcards_blob(html: str) -> dict | None: + """Decode the ``mosaic-provider-jobcards`` assignment, or ``None`` if absent.""" + import json + + idx = html.find(_JOBCARDS_ANCHOR) + if idx == -1: + return None + brace = html.find("{", idx + len(_JOBCARDS_ANCHOR)) + if brace == -1: + return None + blob = _brace_match(html, brace) + if not blob: + return None + try: + data = json.loads(blob) + except ValueError: + return None + return data if isinstance(data, dict) else None + + +def _decode_assignment(html: str, anchor: str) -> dict | None: + """Decode the balanced JSON object assigned after ``anchor``, or ``None``.""" + import json + + idx = html.find(anchor) + if idx == -1: + return None + brace = html.find("{", idx + len(anchor)) + if brace == -1: + return None + blob = _brace_match(html, brace) + if not blob: + return None + try: + data = json.loads(blob) + except ValueError: + return None + return data if isinstance(data, dict) else None + + +def extract_initial_data(html: str) -> dict | None: + """Return a /viewjob posting model rooted at ``jobInfoWrapperModel``. + + Prefers ``window._rootProps`` (JSON) unwrapped at ``preloadedVJData``; falls + back to a legacy inline ``window._initialData`` blob. ``window._initialData`` + is now a JS object literal that references other globals, so it is not JSON + and is skipped when the JSON parse fails. + """ + root = _decode_assignment(html, _ROOT_PROPS_ANCHOR) + if isinstance(root, dict): + vj = root.get(_ROOT_PROPS_KEY) + if isinstance(vj, dict) and vj.get("jobInfoWrapperModel"): + return vj + legacy = _decode_assignment(html, _INITIAL_DATA_ANCHOR) + if isinstance(legacy, dict) and legacy.get("jobInfoWrapperModel"): + return legacy + return None + + +def job_results(blob: dict | None) -> list[dict[str, Any]]: + """Return the raw job records from a decoded blob.""" + if not isinstance(blob, dict): + return [] + results = ( + blob.get("metaData", {}).get("mosaicProviderJobCardsModel", {}).get("results") + ) + if not isinstance(results, list): + return [] + return [r for r in results if isinstance(r, dict)] + + +def _utc_from_ms(value: Any) -> str | None: + """Epoch milliseconds -> millisecond ISO string.""" + if isinstance(value, bool) or not isinstance(value, int | float): + return None + dt = datetime.fromtimestamp(float(value) / 1000.0, tz=UTC) + return dt.strftime("%Y-%m-%dT%H:%M:%S.%f")[:-3] + "Z" + + +def _int(value: Any) -> int | None: + """Coerce to int, dropping bools.""" + if isinstance(value, bool): + return None + if isinstance(value, int): + return value + if isinstance(value, float): + return int(value) + return None + + +def _abs_url(path: Any, base_url: str) -> str | None: + """Resolve an Indeed-relative path against ``base_url``; keep absolute URLs.""" + if not isinstance(path, str) or not path: + return None + if path.startswith("http"): + return path + return f"{base_url}{path if path.startswith('/') else '/' + path}" + + +def _clean_snippet(snippet: Any) -> str | None: + """Strip tags and decode entities into plain text.""" + if not isinstance(snippet, str) or not snippet: + return None + text = _re_sub(r"<[^>]+>", " ", snippet) + text = unescape(text) + return _re_sub(r"\s+", " ", text).strip() or None + + +def _taxonomy(raw: dict[str, Any]) -> dict[str, list[str]]: + """Flatten ``taxonomyAttributes`` into ``{group label: [attribute labels]}``.""" + out: dict[str, list[str]] = {} + for group in raw.get("taxonomyAttributes") or []: + if not isinstance(group, dict): + continue + label = group.get("label") + attrs = group.get("attributes") + if isinstance(label, str) and isinstance(attrs, list): + out[label] = [ + a["label"] + for a in attrs + if isinstance(a, dict) and isinstance(a.get("label"), str) + ] + return out + + +def _job_types(raw: dict[str, Any], taxo: dict[str, list[str]]) -> list[str]: + """Job types from ``jobTypes`` then the taxonomy, deduped and order-stable.""" + seen: dict[str, None] = {} + for jt in raw.get("jobTypes") or []: + if isinstance(jt, str): + seen.setdefault(jt, None) + for label in ("job-types", "job-types-cc"): + for jt in taxo.get(label, []): + seen.setdefault(jt, None) + return list(seen) + + +def _salary(raw: dict[str, Any]) -> dict[str, Any]: + """Flatten salary from ``salarySnippet`` (text) + ``extractedSalary`` (bounds).""" + snippet = raw.get("salarySnippet") or {} + extracted = raw.get("extractedSalary") or {} + estimated = raw.get("estimatedSalary") or {} + source = extracted or estimated + text = snippet.get("text") if isinstance(snippet, dict) else None + return { + "salaryText": text if isinstance(text, str) else None, + "salaryMin": source.get("min") if isinstance(source, dict) else None, + "salaryMax": source.get("max") if isinstance(source, dict) else None, + "currency": snippet.get("currency") if isinstance(snippet, dict) else None, + "period": _SALARY_PERIODS.get( + source.get("type") if isinstance(source, dict) else None + ), + "isEstimated": bool(estimated) and not extracted, + } + + +def _is_remote(raw: dict[str, Any], taxo: dict[str, list[str]]) -> bool: + """Resolve remote/hybrid across Indeed's several signals.""" + if raw.get("remoteLocation") is True: + return True + if isinstance(raw.get("remoteWorkModel"), dict): + return True + return bool(taxo.get("remote")) + + +def parse_job(raw: dict[str, Any], *, base_url: str = _DEFAULT_BASE) -> dict[str, Any]: + """Map one raw ``results[]`` record to a flat item dict. + + ``base_url`` is the country domain the record came from, so job and company + URLs resolve to the right host. + """ + taxo = _taxonomy(raw) + job_key = raw.get("jobkey") + remote_model = raw.get("remoteWorkModel") + return { + "jobKey": job_key if isinstance(job_key, str) else None, + "title": raw.get("displayTitle") or raw.get("title"), + "jobUrl": f"{base_url}/viewjob?jk={job_key}" if job_key else None, + "applyUrl": raw.get("thirdPartyApplyUrl") or None, + "company": raw.get("company") or raw.get("truncatedCompany"), + "companyUrl": _abs_url(raw.get("companyOverviewLink"), base_url), + "companyRating": raw.get("companyRating"), + "companyReviewCount": _int(raw.get("companyReviewCount")), + "formattedLocation": raw.get("formattedLocation"), + "city": raw.get("jobLocationCity"), + "state": raw.get("jobLocationState"), + "postalCode": raw.get("jobLocationPostal"), + "country": raw.get("country"), + "isRemote": _is_remote(raw, taxo), + "remoteType": remote_model.get("type") + if isinstance(remote_model, dict) + else None, + "jobTypes": _job_types(raw, taxo), + "salary": _salary(raw), + "benefits": taxo.get("benefits", []), + "descriptionText": _clean_snippet(raw.get("snippet")), + "descriptionHtml": None, + "sponsored": raw.get("sponsored"), + "isNew": raw.get("newJob"), + "urgentlyHiring": raw.get("urgentlyHiring"), + "expired": raw.get("expired"), + "indeedApplyEnabled": raw.get("indeedApplyEnabled"), + "age": raw.get("formattedRelativeTime"), + "datePublished": _utc_from_ms(raw.get("pubDate")), + "createdAt": _utc_from_ms(raw.get("createDate")), + } + + +def _detail_salary(hdr: dict[str, Any]) -> dict[str, Any] | None: + """Salary from the detail header's flat ``salaryMin/Max/Currency/Type`` fields.""" + smin = hdr.get("salaryMin") + smax = hdr.get("salaryMax") + if smin is None and smax is None: + return None + return { + "salaryText": None, + "salaryMin": smin, + "salaryMax": smax, + "currency": hdr.get("salaryCurrency"), + "period": _SALARY_PERIODS.get(hdr.get("salaryType")), + "isEstimated": False, + } + + +def parse_job_detail(html: str, *, base_url: str = _DEFAULT_BASE) -> dict[str, Any]: + """Map a /viewjob page to enrichment fields (empty dict if not a job page). + + Returns only fields the detail page actually carries, so the caller can merge + it onto a listing item without clobbering known values with blanks. The full + description (``sanitizedJobDescription``) is the field listings never have. + """ + data = extract_initial_data(html) + if not isinstance(data, dict): + return {} + jim = (data.get("jobInfoWrapperModel") or {}).get("jobInfoModel") or {} + if not isinstance(jim, dict): + return {} + hdr = jim.get("jobInfoHeaderModel") + hdr = hdr if isinstance(hdr, dict) else {} + taxo = _taxonomy(hdr) + desc_html = jim.get("sanitizedJobDescription") + desc_html = desc_html if isinstance(desc_html, str) and desc_html else None + remote_model = hdr.get("remoteWorkModel") + out: dict[str, Any] = { + "descriptionHtml": desc_html, + "descriptionText": _clean_snippet(desc_html), + "title": hdr.get("jobTitle"), + "company": hdr.get("companyName"), + "companyUrl": _abs_url(hdr.get("companyOverviewLink"), base_url), + "formattedLocation": hdr.get("formattedLocation") or data.get("jobLocation"), + "remoteType": remote_model.get("type") + if isinstance(remote_model, dict) + else None, + "jobTypes": _job_types(hdr, taxo), + "benefits": taxo.get("benefits", []), + "salary": _detail_salary(hdr), + } + if _is_remote(hdr, taxo): + out["isRemote"] = True + return {k: v for k, v in out.items() if v not in (None, [], {})} diff --git a/surfsense_backend/app/proprietary/platforms/indeed_jobs/schemas.py b/surfsense_backend/app/proprietary/platforms/indeed_jobs/schemas.py new file mode 100644 index 000000000..27be45d9a --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/indeed_jobs/schemas.py @@ -0,0 +1,122 @@ +# ruff: noqa: N815 - field names intentionally use the public camelCase API +"""Input/output models for the Indeed scraper. + +Anonymous scraper: there is no auth field. Fields absent from a listing (full +description, benefits) stay ``None``/``[]`` until a detail fetch fills them. +""" + +from __future__ import annotations + +from typing import Any, Literal + +from pydantic import BaseModel, ConfigDict, Field + +IndeedSort = Literal["relevance", "date"] +IndeedJobType = Literal[ + "fulltime", + "parttime", + "contract", + "internship", + "temporary", + "permanent", + "seasonal", + "freelance", +] +IndeedLevel = Literal["entry_level", "mid_level", "senior_level"] +IndeedRemote = Literal["remote", "hybrid"] +SalaryPeriod = Literal["hour", "day", "week", "month", "year"] + + +class StartUrl(BaseModel): + """A direct URL entry; extra keys ignored.""" + + model_config = ConfigDict(extra="allow") + + url: str + + +class IndeedScrapeInput(BaseModel): + """Indeed scraper input. Caps are collector policy, enforced by ``scrape_indeed``.""" + + model_config = ConfigDict(extra="allow") + + # Discovery: direct URLs and/or search queries. + startUrls: list[StartUrl] = Field(default_factory=list) + queries: list[str] = Field(default_factory=list) + + # Search parameters applied to ``queries``. + country: str = "us" + location: str | None = None + radius: int | None = None + jobType: IndeedJobType | None = None + level: IndeedLevel | None = None + remote: IndeedRemote | None = None + fromDays: int | None = None + sort: IndeedSort = "relevance" + + # Fetch each job's detail page for the full description. + scrapeJobDetails: bool = False + + maxItems: int = Field(default=25, ge=0) + maxItemsPerQuery: int = Field(default=25, ge=0) + + +class Salary(BaseModel): + """Salary block; fields are ``None`` when Indeed omits pay.""" + + model_config = ConfigDict(extra="allow") + + salaryText: str | None = None + salaryMin: float | None = None + salaryMax: float | None = None + currency: str | None = None + period: SalaryPeriod | None = None + isEstimated: bool | None = None + + +class IndeedItem(BaseModel): + """One job posting. ``extra="allow"`` keeps the contract additive.""" + + model_config = ConfigDict(extra="allow") + + jobKey: str | None = None + title: str | None = None + jobUrl: str | None = None + applyUrl: str | None = None + + company: str | None = None + companyUrl: str | None = None + companyRating: float | None = None + companyReviewCount: int | None = None + + formattedLocation: str | None = None + city: str | None = None + state: str | None = None + postalCode: str | None = None + country: str | None = None + isRemote: bool | None = None + remoteType: str | None = None + + jobTypes: list[str] = Field(default_factory=list) + salary: Salary = Field(default_factory=Salary) + benefits: list[str] = Field(default_factory=list) + + descriptionText: str | None = None + descriptionHtml: str | None = None + + sponsored: bool | None = None + isNew: bool | None = None + urgentlyHiring: bool | None = None + expired: bool | None = None + indeedApplyEnabled: bool | None = None + + age: str | None = None + datePublished: str | None = None + createdAt: str | None = None + scrapedAt: str | None = None + + source: str = "indeed" + + def to_output(self) -> dict[str, Any]: + """Serialize to the flat output dict, keeping extras.""" + return self.model_dump(exclude_none=False) diff --git a/surfsense_backend/app/proprietary/platforms/indeed_jobs/scraper.py b/surfsense_backend/app/proprietary/platforms/indeed_jobs/scraper.py new file mode 100644 index 000000000..68b70870a --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/indeed_jobs/scraper.py @@ -0,0 +1,178 @@ +"""Orchestrator for the Indeed scraper. + +:func:`iter_indeed` streams deduped job items from one warmed session; each +search/company target contributes its first page. :func:`scrape_indeed` collects +the stream under a caller ``limit``. Targets run sequentially to reuse the session. +""" + +from __future__ import annotations + +import logging +from collections.abc import AsyncIterator +from typing import Any +from urllib.parse import urlparse + +from .fetch import IndeedSession, now_iso, open_session +from .parsers import ( + extract_jobcards_blob, + job_results, + parse_job, + parse_job_detail, +) +from .schemas import IndeedItem, IndeedScrapeInput +from .url_resolver import build_search_url, resolve_url + +logger = logging.getLogger(__name__) + +__all__ = ["iter_indeed", "scrape_indeed"] + + +def _emit(partial: dict[str, Any]) -> dict[str, Any]: + """Stamp ``scrapedAt`` and normalize through the output model.""" + return IndeedItem(**{**partial, "scrapedAt": now_iso()}).to_output() + + +async def _search_items( + session: IndeedSession, url: str, *, domain: str, max_items: int +) -> AsyncIterator[dict[str, Any]]: + """Yield deduped job cards from one search/company page. + + ponytail: caps a query at its first page (~15 jobs) — anonymous Indeed gates + ``start>=10``; deeper depth needs an authenticated session or Indeed's API. + """ + if max_items <= 0: + return + base_url = f"https://{domain}" + html = await session.fetch_html(url) + seen: set[str] = set() + emitted = 0 + for raw in job_results(extract_jobcards_blob(html)): + item = parse_job(raw, base_url=base_url) + job_key = item.get("jobKey") + if isinstance(job_key, str): + if job_key in seen: + continue + seen.add(job_key) + yield _emit(item) + emitted += 1 + if emitted >= max_items: + return + + +def _targets(input_model: IndeedScrapeInput) -> list[tuple[str, str, str]]: + """Resolve inputs to ``(kind, url, domain)`` targets. + + ``startUrls`` take precedence over ``queries``. ``kind`` is ``search`` for + query-built and search/company URLs, or ``job`` for a ``/viewjob`` URL. + """ + if input_model.startUrls: + out: list[tuple[str, str, str]] = [] + for entry in input_model.startUrls: + resolved = resolve_url(entry.url) + if resolved is None: + logger.warning("[indeed] skipping unrecognized URL: %s", entry.url) + continue + kind = "job" if resolved.kind == "job" else "search" + out.append((kind, resolved.url, resolved.domain)) + return out + + domain = None + urls: list[tuple[str, str, str]] = [] + for query in input_model.queries: + url = build_search_url( + query, + country=input_model.country, + location=input_model.location, + radius=input_model.radius, + job_type=input_model.jobType, + level=input_model.level, + remote=input_model.remote, + from_days=input_model.fromDays, + sort=input_model.sort, + ) + domain = domain or urlparse(url).hostname or "www.indeed.com" + urls.append(("search", url, domain)) + return urls + + +async def _enrich(session: IndeedSession, item: dict[str, Any], base_url: str) -> None: + """Merge a job's /viewjob detail (full description, etc.) onto ``item`` in place. + + Best-effort: a blocked or malformed detail page leaves the listing fields as-is + rather than failing the run. + """ + job_url = item.get("jobUrl") + if not isinstance(job_url, str): + return + try: + # Fail fast: enrichment is best-effort, so a gated detail page must not + # rotate IPs and eat the run's time budget for one job's description. + html = await session.fetch_html(job_url, max_rotations=0) + detail = parse_job_detail(html, base_url=base_url) + except Exception as exc: + logger.warning("[indeed] detail fetch failed for %s: %s", job_url, exc) + return + item.update(detail) + + +async def _job_item( + session: IndeedSession, url: str, base_url: str +) -> dict[str, Any] | None: + """Scrape a single /viewjob URL into an item from its detail page alone.""" + detail = parse_job_detail(await session.fetch_html(url), base_url=base_url) + if not detail: + return None + return _emit({"jobUrl": url, "source": "indeed", **detail}) + + +async def iter_indeed( + input_model: IndeedScrapeInput, session: IndeedSession +) -> AsyncIterator[dict[str, Any]]: + """Stream flat job items for every target, deduped by ``jobKey`` across all.""" + global_seen: set[str] = set() + for kind, url, domain in _targets(input_model): + base_url = f"https://{domain}" + if kind == "job": + item = await _job_item(session, url, base_url) + if item is not None: + yield item + continue + async for item in _search_items( + session, url, domain=domain, max_items=input_model.maxItemsPerQuery + ): + job_key = item.get("jobKey") + if isinstance(job_key, str): + if job_key in global_seen: + continue + global_seen.add(job_key) + if input_model.scrapeJobDetails: + await _enrich(session, item, base_url) + yield item + + +async def scrape_indeed( + input_model: IndeedScrapeInput, + *, + limit: int | None = None, + session: IndeedSession | None = None, +) -> list[dict[str, Any]]: + """Collect :func:`iter_indeed` into a list under an optional ``limit``. + + Opens a warmed session when one is not supplied. ``limit`` is a request-time + guard, not a ceiling baked into the stream. + """ + from app.capabilities.core.progress import emit_progress + + async def _collect(sess: IndeedSession) -> list[dict[str, Any]]: + results: list[dict[str, Any]] = [] + async for item in iter_indeed(input_model, sess): + results.append(item) + emit_progress("scraping", current=len(results), total=limit, unit="item") + if limit is not None and len(results) >= limit: + break + return results + + if session is not None: + return await _collect(session) + async with open_session() as sess: + return await _collect(sess) diff --git a/surfsense_backend/app/proprietary/platforms/indeed_jobs/url_resolver.py b/surfsense_backend/app/proprietary/platforms/indeed_jobs/url_resolver.py new file mode 100644 index 000000000..c69d3fe19 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/indeed_jobs/url_resolver.py @@ -0,0 +1,124 @@ +"""Classify Indeed URLs and build search URLs. + +Recognizes search pages (``/jobs?q=&l=``), company pages (``/cmp//jobs``), +and single jobs (``/viewjob?jk=``); other hosts resolve to ``None``. Also owns +the country->domain map so classification and URL building share one source. +""" + +from __future__ import annotations + +from dataclasses import dataclass, field +from typing import Literal +from urllib.parse import parse_qs, urlencode, urlparse + +ResolvedKind = Literal["search", "company", "job"] + +# Locale subdomains that deviate from the ISO code; others map to .indeed.com. +_DOMAIN_OVERRIDES = {"us": "www", "gb": "uk"} + +_JT_VALUES = frozenset( + { + "fulltime", + "parttime", + "contract", + "internship", + "temporary", + "permanent", + "seasonal", + "freelance", + } +) + + +@dataclass(frozen=True) +class ResolvedUrl: + kind: ResolvedKind + value: str # search query, company slug, or job key + url: str + domain: str + location: str | None = None + params: dict[str, str] = field(default_factory=dict) + + +def _is_indeed_host(hostname: str | None) -> bool: + if not hostname: + return False + h = hostname.lower() + return h == "indeed.com" or h.endswith(".indeed.com") + + +def country_domain(country: str) -> str: + """Country code -> Indeed host, e.g. ``us`` -> ``www.indeed.com``.""" + cc = (country or "us").strip().lower() + return f"{_DOMAIN_OVERRIDES.get(cc, cc)}.indeed.com" + + +def resolve_url(url: str) -> ResolvedUrl | None: + """Classify an Indeed URL into a scrape job, or ``None`` if unrecognized.""" + parsed = urlparse(url) + if not _is_indeed_host(parsed.hostname): + return None + domain = parsed.hostname or "www.indeed.com" + path = (parsed.path or "").rstrip("/") + query = parse_qs(parsed.query) + segments = [s for s in path.split("/") if s] + + # /viewjob?jk= + if path.endswith("/viewjob") or segments[:1] == ["viewjob"]: + jk = query.get("jk", [None])[0] + return ResolvedUrl("job", jk, url, domain) if jk else None + + # /cmp//jobs + if segments[:1] == ["cmp"] and "jobs" in segments and len(segments) >= 2: + return ResolvedUrl("company", segments[1], url, domain) + + # /jobs?q=&l= + if path.endswith("/jobs") or segments[-1:] == ["jobs"]: + q = query.get("q", [""])[0] + loc = query.get("l", [None])[0] + extra = { + k: v[0] + for k, v in query.items() + if k in ("radius", "sort", "fromage", "jt", "explvl") and v + } + return ResolvedUrl("search", q, url, domain, location=loc, params=extra) + + return None + + +def build_search_url( + query: str, + *, + country: str = "us", + location: str | None = None, + radius: int | None = None, + job_type: str | None = None, + level: str | None = None, + remote: str | None = None, + from_days: int | None = None, + sort: str = "relevance", + start: int = 0, +) -> str: + """Build an Indeed ``/jobs`` search URL. + + Remote/hybrid is passed as a query keyword; Indeed's structured ``sc`` + attribute codes rotate and aren't stable to hardcode. + """ + domain = country_domain(country) + q = f"{query} {remote}".strip() if remote else query + params: dict[str, str] = {"q": q} + if location: + params["l"] = location + if radius is not None: + params["radius"] = str(radius) + if job_type in _JT_VALUES: + params["jt"] = job_type # type: ignore[assignment] + if level: + params["explvl"] = level + if from_days is not None: + params["fromage"] = str(from_days) + if sort == "date": + params["sort"] = "date" + if start: + params["start"] = str(start) + return f"https://{domain}/jobs?{urlencode(params)}" diff --git a/surfsense_backend/app/proprietary/platforms/reddit/fetch.py b/surfsense_backend/app/proprietary/platforms/reddit/fetch.py index 50fad389d..a1ac77fc6 100644 --- a/surfsense_backend/app/proprietary/platforms/reddit/fetch.py +++ b/surfsense_backend/app/proprietary/platforms/reddit/fetch.py @@ -43,6 +43,7 @@ from urllib.parse import urlencode from scrapling.fetchers import AsyncFetcher, AsyncStealthySession, FetcherSession +from app.utils.browser_loop import in_browser_loop from app.utils.proxy import get_proxy_url, get_sticky_proxy_url # Shared cross-country rotation walk (also used by the TikTok sibling). Kept under @@ -331,18 +332,24 @@ async def warm_session(proxy: str | None) -> dict[str, str] | None: """ if proxy is None: return None + + # Runs on the shared browser loop: patchright can't spawn Chromium from the + # server's Windows SelectorEventLoop (see app.utils.browser_loop). + async def _warm() -> dict[str, str] | None: + async with AsyncStealthySession( + headless=True, + google_search=True, + network_idle=False, + proxy=proxy, + timeout=_WARM_TIMEOUT_MS, + ) as sess: + page = await sess.fetch(_WARM_HTML_URL) + jar = _browser_cookie_jar(page) + return jar if _LOID_COOKIE in jar else None + async with _warm_slots: try: - async with AsyncStealthySession( - headless=True, - google_search=True, - network_idle=False, - proxy=proxy, - timeout=_WARM_TIMEOUT_MS, - ) as sess: - page = await sess.fetch(_WARM_HTML_URL) - jar = _browser_cookie_jar(page) - return jar if _LOID_COOKIE in jar else None + return await in_browser_loop(_warm()) except Exception as e: # a browser crash must not abort the flow logger.warning("[reddit] browser warm failed: %s", e) return None diff --git a/surfsense_backend/app/proprietary/platforms/reddit/schemas.py b/surfsense_backend/app/proprietary/platforms/reddit/schemas.py index 74560ae97..e3661eb9b 100644 --- a/surfsense_backend/app/proprietary/platforms/reddit/schemas.py +++ b/surfsense_backend/app/proprietary/platforms/reddit/schemas.py @@ -14,7 +14,7 @@ from __future__ import annotations from typing import Any, Literal -from pydantic import BaseModel, ConfigDict, Field +from pydantic import BaseModel, ConfigDict, Field, field_validator RedditSort = Literal["relevance", "hot", "top", "new", "rising", "comments"] RedditTime = Literal["all", "hour", "day", "week", "month", "year"] @@ -44,6 +44,23 @@ class RedditScrapeInput(BaseModel): searches: list[str] = Field(default_factory=list) searchCommunityName: str | None = None + @field_validator("searchCommunityName", mode="after") + @classmethod + def _normalize_community(cls, v: str | None) -> str | None: + """Normalize the bare subreddit name at the trust boundary. + + Both the in-subreddit search path and the community-only listing path + interpolate this straight into ``r/{name}/...``, so a pasted ``r/python`` + or ``/r/python/`` would otherwise become ``r/r/python`` and 404. Strip a + leading ``r/`` and surrounding slashes/whitespace; empty coerces to None. + """ + if v is None: + return None + name = v.strip().strip("/") + if name[:2].lower() == "r/": + name = name[2:].strip("/") + return name or None + # Sort / filter sort: RedditSort = "new" time: RedditTime | None = None diff --git a/surfsense_backend/app/proprietary/platforms/reddit/scraper.py b/surfsense_backend/app/proprietary/platforms/reddit/scraper.py index a14e1c13f..bcbab5673 100644 --- a/surfsense_backend/app/proprietary/platforms/reddit/scraper.py +++ b/surfsense_backend/app/proprietary/platforms/reddit/scraper.py @@ -407,6 +407,18 @@ async def iter_reddit( yield item return + # Community-only: a bare subreddit name with no urls/searches. The capability + # schema promises "with no search_queries, its listing is scraped"; without + # this the searches loop below builds zero jobs and yields nothing. Reuse the + # same flow the URL path dispatches for /r/. (searches present => the + # name stays a search scope, handled below.) + if not input_model.searches and input_model.searchCommunityName: + async for item in _subreddit_flow( + input_model.searchCommunityName, input_model=input_model + ): + yield item + return + # Fair-share the item budget across queries: with a shared cap, the # first-finishing (often broadest/noisiest) search would fill the whole # collector limit and starve the precise queries. diff --git a/surfsense_backend/app/proprietary/platforms/walmart/README.md b/surfsense_backend/app/proprietary/platforms/walmart/README.md new file mode 100644 index 000000000..982ee5400 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/walmart/README.md @@ -0,0 +1,52 @@ +# Walmart Scraper + +Two verbs read Walmart's public, anonymous pages: `walmart.scrape` (products + +search/category/browse listings, per-product billing) and `walmart.reviews` +(deep paginated reviews, per-review billing). + +## Scope + +The scraper reads public pages available to anonymous visitors — no login, no +account cookies. Data is extracted from the Next.js `__NEXT_DATA__` JSON blob +embedded in each page (with an `__APP_DATA__` fallback), not from the rendered +DOM, because Walmart obfuscates CSS classes and A/B-tests layout constantly. + +`walmart.scrape` returns a free sample of on-page reviews (rating distribution, +aspects, top reviews) under `reviewsSample`. `walmart.reviews` fetches the full +review history from the public `/reviews/product/{usItemId}` page, which +robots.txt permits (unlike `/search`). + +## Architecture + +- `schemas.py` defines the stable input, product, review, and error models. +- `url_resolver.py` classifies product (`/ip/`) vs listing (`/search`, `/cp/`, + `/browse/`) URLs and extracts the numeric `usItemId`. +- `next_data.py` extracts and navigates the hidden Next.js JSON state. +- `fetch.py` owns proxy-aware HTTP access (US-pinned), block detection, and + retries. +- `parsers.py` contains pure, defensive JSON parsers. +- `scraper.py` coordinates discovery, enrichment, pagination, concurrency, + limits, and in-stream error items. + +## Anti-bot + +Walmart runs Akamai (edge/TLS) + PerimeterX/HUMAN (behavioral JS). Requests go +through US residential proxies with TLS-impersonated headers; blocked responses +(body markers, `412`/`429`/`503`, or the `200`-OK CAPTCHA body) rotate to a +fresh proxy exit. + +Known ceilings and upgrade paths (see `fetch.py` / `scraper.py` `ponytail:` +notes): reviews page at 10/page; search capped at Walmart's 25-page limit; +session warming (seed `_px3`/`_pxhd` on a sticky exit) is the next lever if +block rates on the SSR pages climb, and the `/orchestra/*` GraphQL API is +deliberately avoided (rotating persisted-query hashes make it brittle). + +## Verification + +Offline fixtures cover the parsers and both flows: + + uv run pytest tests/unit/platforms/walmart/ + +A manual live check (requires network + residential proxy): + + uv run python scripts/e2e_walmart_scraper.py diff --git a/surfsense_backend/app/proprietary/platforms/walmart/__init__.py b/surfsense_backend/app/proprietary/platforms/walmart/__init__.py new file mode 100644 index 000000000..83b7cf277 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/walmart/__init__.py @@ -0,0 +1,27 @@ +"""Platform-native Walmart scraper (products, listings, reviews).""" + +from .schemas import ( + ErrorItem, + ProductItem, + ReviewItem, + WalmartReviewsInput, + WalmartScrapeInput, +) +from .scraper import ( + iter_products, + iter_reviews, + scrape_products, + scrape_reviews, +) + +__all__ = [ + "ErrorItem", + "ProductItem", + "ReviewItem", + "WalmartReviewsInput", + "WalmartScrapeInput", + "iter_products", + "iter_reviews", + "scrape_products", + "scrape_reviews", +] diff --git a/surfsense_backend/app/proprietary/platforms/walmart/fetch.py b/surfsense_backend/app/proprietary/platforms/walmart/fetch.py new file mode 100644 index 000000000..851b6a674 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/walmart/fetch.py @@ -0,0 +1,171 @@ +"""Network access for public Walmart pages. + +Mirrors the Amazon fetch layer: every request goes through the configured +residential proxy (pinned to the US, since Walmart geo-locks inventory), and any +response that looks like an anti-bot interstitial is retried on a fresh proxy +exit. Ordinary HTTP failures are returned to the caller for domain-specific +handling. + +Walmart runs Akamai (edge/TLS) + PerimeterX/HUMAN (behavioral JS). Two Walmart +specifics differ from Amazon: + +* Walmart serves CAPTCHA with an HTTP ``200`` body ("Robot or human?"), so block + detection scans the body, never the status alone. +* ``412`` is PerimeterX's rejection code and is treated as blocked → rotate. + +``ponytail:`` MVP hits only the server-rendered ``__NEXT_DATA__`` pages, which +TLS impersonation + residential proxies clear without seeding PerimeterX cookies. +If block rates on those pages climb, the upgrade path is a warmed sticky session +(seed ``_px3``/``_pxhd``/``ACID`` from a homepage fetch, reuse exit + cookies) — +the same shape as Amazon's ``get_location_session``. +""" + +from __future__ import annotations + +import asyncio +import logging +import time +from collections.abc import Awaitable, Callable +from dataclasses import dataclass, field +from typing import Any + +from scrapling.fetchers import AsyncFetcher + +from app.utils.proxy import get_geo_proxy_url, get_sticky_proxy_url + +logger = logging.getLogger(__name__) + +_MAX_IP_ATTEMPTS = 6 +_REQUEST_TIMEOUT_S = 30 +_HEADERS = { + "Accept-Language": "en-US,en;q=0.9", + "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8", +} +_BLOCK_MARKERS = ( + "robot or human", + "px-captcha", + "/blocked", + "verify you are a human", + "access to this page has been denied", +) + + +@dataclass(frozen=True) +class FetchResult: + """The response details needed by scraper flows.""" + + status: int + html: str + url: str + cookies: dict[str, str] + headers: dict[str, str] = field(default_factory=dict) + + +async def gather_bounded[T]( + factories: list[Callable[[], Awaitable[T]]], *, concurrency: int +) -> list[T]: + """Run async factories concurrently while preserving input order.""" + if not factories: + return [] + semaphore = asyncio.Semaphore(max(1, concurrency)) + + async def run(factory: Callable[[], Awaitable[T]]) -> T: + async with semaphore: + return await factory() + + return await asyncio.gather(*(run(factory) for factory in factories)) + + +def is_blocked( + html: str | None, status: int, headers: dict[str, str] | None = None +) -> bool: + """Return whether a response is a Walmart anti-bot interstitial. + + ``412`` is PerimeterX's rejection; ``429``/``503`` are throttles. Walmart also + serves CAPTCHA with a ``200`` body, so the body is scanned regardless of + status. + """ + if status in {412, 429, 503}: + return True + text = (html or "")[:200_000].lower() + return any(marker in text for marker in _BLOCK_MARKERS) + + +def _response_url(page: Any, fallback: str) -> str: + value = getattr(page, "url", None) + return str(value) if value else fallback + + +def _response_cookies(page: Any) -> dict[str, str]: + cookies = getattr(page, "cookies", None) + return dict(cookies) if isinstance(cookies, dict) else {} + + +def _response_headers(page: Any) -> dict[str, str]: + headers = getattr(page, "headers", None) + return dict(headers) if isinstance(headers, dict) else {} + + +def _selected_proxy( + proxy: str | None, country: str, attempt: int, url: str +) -> str | None: + if proxy is not None: + return proxy + if attempt > 1: + session_id = f"walmart-{attempt}-{abs(hash((url, time.time_ns()))):x}" + return get_sticky_proxy_url(session_id, country) + return get_geo_proxy_url(country) + + +async def fetch_page( + url: str, + *, + cookies: dict[str, str] | None = None, + proxy: str | None = None, + country: str = "us", + rotate_on_block: bool = True, +) -> FetchResult | None: + """Fetch a page and retry blocked responses with fresh proxy exits.""" + attempts = _MAX_IP_ATTEMPTS if rotate_on_block else 1 + for attempt in range(1, attempts + 1): + selected_proxy = _selected_proxy(proxy, country, attempt, url) + started = time.perf_counter() + try: + page = await AsyncFetcher.get( + url, + headers={**_HEADERS}, + cookies=cookies or {}, + proxy=selected_proxy, + stealthy_headers=True, + timeout=_REQUEST_TIMEOUT_S, + ) + except Exception as exc: + logger.warning("Walmart request failed for %s: %s", url, exc) + if proxy is not None: + return None + continue + + status = int(getattr(page, "status", 0) or 0) + html = getattr(page, "html_content", None) or "" + response_headers = _response_headers(page) + logger.info( + "[walmart][perf] status=%s attempt=%s fetch_ms=%.1f url=%s", + status, + attempt, + (time.perf_counter() - started) * 1000, + url, + ) + if rotate_on_block and is_blocked(html, status, response_headers): + logger.info( + "Walmart blocked proxy attempt %s/%s for %s", attempt, attempts, url + ) + continue + return FetchResult( + status=status, + html=html, + url=_response_url(page, url), + cookies=_response_cookies(page), + headers=response_headers, + ) + logger.warning("Walmart exhausted %s proxy attempts for %s", attempts, url) + return None diff --git a/surfsense_backend/app/proprietary/platforms/walmart/next_data.py b/surfsense_backend/app/proprietary/platforms/walmart/next_data.py new file mode 100644 index 000000000..2994fcbf6 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/walmart/next_data.py @@ -0,0 +1,69 @@ +"""Extract Walmart's hidden Next.js JSON state from page HTML. + +Every serious Walmart scraper (Scrapfly, Oxylabs, ScrapeOps, Apify) reads the +``', re.DOTALL) +_APP_DATA_RE = re.compile(r'', re.DOTALL) + + +def extract_next_data(html: str | None) -> dict[str, Any] | None: + """Return the parsed Next.js state object, or ``None`` when absent/invalid. + + Tries ``__NEXT_DATA__`` first, then the ``__APP_DATA__`` fallback so a single + Walmart layout experiment does not blank the whole extractor. + """ + if not html: + return None + for pattern in (_NEXT_DATA_RE, _APP_DATA_RE): + match = pattern.search(html) + if not match: + continue + try: + data = json.loads(match.group(1)) + except (ValueError, TypeError): + logger.warning("Walmart hidden JSON present but did not parse") + continue + if isinstance(data, dict): + return data + return None + + +def dig(obj: Any, *keys: str | int) -> Any: + """Walk nested dict/list keys, returning ``None`` on any miss. + + Tolerates the layout drift between Walmart's ``initialData`` variants without + a cascade of ``if key in ...`` guards at every call site. + """ + current = obj + for key in keys: + if isinstance(key, int): + if not isinstance(current, list) or not -len(current) <= key < len(current): + return None + current = current[key] + else: + if not isinstance(current, dict) or key not in current: + return None + current = current[key] + return current + + +def initial_data(next_data: dict[str, Any]) -> dict[str, Any] | None: + """The ``props.pageProps.initialData`` node shared by every page type.""" + node = dig(next_data, "props", "pageProps", "initialData") + return node if isinstance(node, dict) else None diff --git a/surfsense_backend/app/proprietary/platforms/walmart/parsers.py b/surfsense_backend/app/proprietary/platforms/walmart/parsers.py new file mode 100644 index 000000000..110f05e75 --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/walmart/parsers.py @@ -0,0 +1,248 @@ +"""Pure parsers for Walmart's hidden ``__NEXT_DATA__`` JSON. + +Product, listing, and review data all live in the same Next.js state tree; these +functions navigate it defensively (via :func:`.next_data.dig`) and normalize the +fields into the stable output shape. Missing sections yield ``None``/``[]`` so an +isolated schema change never discards an otherwise usable record. +""" + +from __future__ import annotations + +from typing import Any +from urllib.parse import urljoin + +from .next_data import dig, initial_data + +_WALMART_ORIGIN = "https://www.walmart.com" + + +# --------------------------------------------------------------------------- # +# Shared helpers. # +# --------------------------------------------------------------------------- # + + +def _price(node: Any) -> dict[str, Any] | None: + """Normalize a Walmart price node into ``{value, currency}``.""" + if not isinstance(node, dict): + return None + value = node.get("price") + currency = node.get("currencyUnit") + if value is None and currency is None: + return None + return {"value": value, "currency": currency} + + +def _seller(raw: dict[str, Any]) -> dict[str, Any] | None: + name = raw.get("sellerName") or raw.get("sellerDisplayName") + seller_id = raw.get("sellerId") + if not name and not seller_id: + return None + is_walmart = bool(name) and "walmart" in name.lower() + return { + "id": seller_id, + "name": name, + "type": "WALMART" if is_walmart else "MARKETPLACE", + } + + +def _absolute(url: str | None) -> str | None: + return urljoin(_WALMART_ORIGIN, url) if url else None + + +# --------------------------------------------------------------------------- # +# Listing / search cards. # +# --------------------------------------------------------------------------- # + + +def _listing_card(raw: dict[str, Any]) -> dict[str, Any] | None: + """Normalize one search/category result item into a product card.""" + item_id = raw.get("usItemId") or raw.get("id") + name = raw.get("name") + if not item_id or not name: + return None + price_info = raw.get("priceInfo") or {} + availability = raw.get("availabilityStatusV2") or {} + image = raw.get("imageInfo") or {} + return { + "usItemId": str(item_id), + "name": name, + "brand": raw.get("brand"), + "url": _absolute(raw.get("canonicalUrl")), + "price": _price(price_info.get("currentPrice")) + or ({"value": raw.get("price")} if raw.get("price") is not None else None), + "listPrice": _price(price_info.get("wasPrice")), + "stars": raw.get("averageRating"), + "reviewsCount": raw.get("numberOfReviews"), + "seller": _seller(raw), + "availabilityStatus": availability.get("value") + or ("OUT_OF_STOCK" if raw.get("isOutOfStock") else None), + "inStock": (availability.get("value") == "IN_STOCK") + if availability.get("value") + else (raw.get("isOutOfStock") is False if "isOutOfStock" in raw else None), + "thumbnailImage": image.get("thumbnailUrl") or raw.get("image"), + "sponsored": raw.get("isSponsoredFlag"), + } + + +def parse_listing_page(next_data: dict[str, Any]) -> list[dict[str, Any]]: + """Extract normalized product cards from a search/category/browse page.""" + data = initial_data(next_data) + if data is None: + return [] + stacks = dig(data, "searchResult", "itemStacks") + if not isinstance(stacks, list): + return [] + cards: list[dict[str, Any]] = [] + seen: set[str] = set() + for stack in stacks: + for raw in (stack or {}).get("items") or []: + if not isinstance(raw, dict) or raw.get("__typename") not in ( + None, + "Product", + ): + continue + card = _listing_card(raw) + if card and card["usItemId"] not in seen: + seen.add(card["usItemId"]) + cards.append(card) + return cards + + +# --------------------------------------------------------------------------- # +# Product detail. # +# --------------------------------------------------------------------------- # + + +def _images(image_info: dict[str, Any]) -> list[str]: + images: list[str] = [] + for entry in image_info.get("allImages") or []: + url = entry.get("url") if isinstance(entry, dict) else None + if url and url not in images: + images.append(url) + return images + + +def _breadcrumbs(product: dict[str, Any]) -> list[str]: + """Category breadcrumb names, root→leaf. + + Walmart ships ``category.path`` as a list of ``{name, url}`` nodes, e.g. + ``[{"name": "Home Improvement"}, ..., {"name": "Air Conditioners"}]`` — not + a string. + """ + path = dig(product, "category", "path") + if not isinstance(path, list): + return [] + return [ + node["name"] for node in path if isinstance(node, dict) and node.get("name") + ] + + +def _reviews_sample( + reviews: dict[str, Any] | None, limit: int = 10 +) -> dict[str, Any] | None: + if not isinstance(reviews, dict): + return None + customer = reviews.get("customerReviews") or [] + return { + "averageOverallRating": reviews.get("averageOverallRating"), + "totalReviewCount": reviews.get("totalReviewCount"), + "aspects": reviews.get("aspects") or [], + "topReviews": [ + normalize_review(r) for r in customer[:limit] if isinstance(r, dict) + ], + } + + +def parse_product( + next_data: dict[str, Any], *, url: str, include_reviews_sample: bool = True +) -> dict[str, Any]: + """Extract normalized product fields from a product detail page.""" + data = initial_data(next_data) + product = dig(data, "data", "product") if data else None + if not isinstance(product, dict): + return {} + idml = dig(data, "data", "idml") if data else None + reviews = dig(data, "data", "reviews") if data else None + price_info = product.get("priceInfo") or {} + image_info = product.get("imageInfo") or {} + availability = product.get("availabilityStatus") + breadcrumbs = _breadcrumbs(product) + + fields: dict[str, Any] = { + "usItemId": str(product.get("usItemId") or product.get("id") or "") or None, + "name": product.get("name"), + "brand": product.get("brand"), + "url": url, + "price": _price(price_info.get("currentPrice")), + "listPrice": _price(price_info.get("wasPrice")), + "currency": dig(price_info, "currentPrice", "currencyUnit"), + "availabilityStatus": availability, + "inStock": availability == "IN_STOCK" if availability else None, + "stars": product.get("averageRating"), + "reviewsCount": product.get("numberOfReviews"), + "seller": _seller(product), + "manufacturerName": product.get("manufacturerName"), + "shortDescription": product.get("shortDescription"), + "longDescription": (idml or {}).get("longDescription") + if isinstance(idml, dict) + else None, + "thumbnailImage": image_info.get("thumbnailUrl"), + "images": _images(image_info), + "category": breadcrumbs[-1] if breadcrumbs else None, + "breadCrumbs": breadcrumbs, + "variants": product.get("variantCriteria") or [], + } + if include_reviews_sample: + fields["reviewsSample"] = _reviews_sample(reviews) + return {key: value for key, value in fields.items() if value is not None} + + +# --------------------------------------------------------------------------- # +# Reviews (deep pagination). # +# --------------------------------------------------------------------------- # + + +def normalize_review(raw: dict[str, Any]) -> dict[str, Any]: + """Normalize one Walmart ``customerReviews`` record into a ``ReviewItem`` dict.""" + badges = raw.get("badges") or [] + verified = any( + isinstance(b, dict) and b.get("id") == "VerifiedPurchaser" for b in badges + ) + photos = raw.get("photos") or raw.get("media") or [] + images: list[str] = [] + for photo in photos: + if not isinstance(photo, dict): + continue + url = photo.get("normalUrl") or dig(photo, "sizes", "normal", "url") + if url and url not in images: + images.append(url) + responses = raw.get("clientResponses") or [] + seller_response = None + if responses and isinstance(responses[0], dict): + seller_response = responses[0].get("response") or responses[0].get( + "responseText" + ) + return { + "reviewId": raw.get("reviewId"), + "rating": raw.get("rating"), + "title": raw.get("reviewTitle"), + "text": raw.get("reviewText"), + "submissionTime": raw.get("reviewSubmissionTime"), + "author": raw.get("userNickname"), + "verifiedPurchase": verified, + "positiveFeedback": raw.get("positiveFeedback"), + "negativeFeedback": raw.get("negativeFeedback"), + "images": images, + "syndicated": bool(raw.get("syndicationSource")), + "sellerResponse": seller_response, + } + + +def parse_reviews_page(next_data: dict[str, Any]) -> list[dict[str, Any]]: + """Extract normalized reviews from one ``/reviews/product/{id}`` page.""" + data = initial_data(next_data) + reviews = dig(data, "data", "reviews") if data else None + customer = reviews.get("customerReviews") if isinstance(reviews, dict) else None + if not isinstance(customer, list): + return [] + return [normalize_review(r) for r in customer if isinstance(r, dict)] diff --git a/surfsense_backend/app/proprietary/platforms/walmart/schemas.py b/surfsense_backend/app/proprietary/platforms/walmart/schemas.py new file mode 100644 index 000000000..ea25f2eff --- /dev/null +++ b/surfsense_backend/app/proprietary/platforms/walmart/schemas.py @@ -0,0 +1,188 @@ +# ruff: noqa: N815 +"""Input/output models for the public Walmart scraper. + +Two verbs share this module: ``walmart.scrape`` (products + listings) and +``walmart.reviews`` (deep paginated reviews). Walmart is a Next.js app that +ships its data as JSON in a `` \ No newline at end of file diff --git a/surfsense_backend/tests/unit/platforms/indeed_jobs/test_fetch_resilience.py b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_fetch_resilience.py new file mode 100644 index 000000000..684f7a46c --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_fetch_resilience.py @@ -0,0 +1,138 @@ +"""Offline tests for the rotate-on-block fetch loop (no network, fake session).""" + +from __future__ import annotations + +import asyncio +import sys + +import pytest + +from app.proprietary.platforms.indeed_jobs.fetch import ( + IndeedAccessBlockedError, + IndeedSession, +) + +_OK_HTML = "jobs listing" +_BLOCK_HTML = "secure.indeed.com security check" + + +class _FakePage: + def __init__(self, html: str, url: str) -> None: + self.html_content = html + self.url = url + + +class _Controller: + """Shared state across sessions the factory hands out (survives rotation).""" + + def __init__(self, target_outcomes: list[str]) -> None: + self.target_outcomes = target_outcomes + self.sessions_started = 0 + self.home_fetches: dict[str, int] = {} + self.target_index = 0 + + def factory(self) -> _FakeSession: + return _FakeSession(self) + + +class _FakeSession: + def __init__(self, ctrl: _Controller) -> None: + self._ctrl = ctrl + + async def start(self) -> None: + self._ctrl.sessions_started += 1 + + async def close(self) -> None: + pass + + async def fetch(self, url: str, **_: object) -> _FakePage: + if url.endswith("/") and "/jobs" not in url: # warm-up hit + self._ctrl.home_fetches[url] = self._ctrl.home_fetches.get(url, 0) + 1 + return _FakePage("home", url) + outcome = self._ctrl.target_outcomes[self._ctrl.target_index] + self._ctrl.target_index += 1 + if outcome == "OK": + return _FakePage(_OK_HTML, url) + if outcome == "ERROR": + raise RuntimeError("boom") + return _FakePage(_BLOCK_HTML, "https://secure.indeed.com/auth") + + +_URL = "https://www.indeed.com/jobs?q=dev" + + +@pytest.mark.asyncio +async def test_rotates_past_a_block_then_succeeds(): + ctrl = _Controller(["BLOCK", "OK"]) + session = IndeedSession(ctrl.factory) + html = await session.fetch_html(_URL) + assert html == _OK_HTML + assert session.rotations == 1 + assert ctrl.sessions_started == 2 # initial + one rotation + + +@pytest.mark.asyncio +async def test_recovers_after_a_fetch_error(): + ctrl = _Controller(["ERROR", "OK"]) + session = IndeedSession(ctrl.factory) + assert await session.fetch_html(_URL) == _OK_HTML + assert session.rotations == 1 + + +@pytest.mark.asyncio +async def test_raises_after_exhausting_rotations(): + ctrl = _Controller(["BLOCK"] * 10) + session = IndeedSession(ctrl.factory) + with pytest.raises(IndeedAccessBlockedError): + await session.fetch_html(_URL) + assert session.rotations == 3 + + +@pytest.mark.asyncio +async def test_max_rotations_zero_fails_fast(): + # A gated page (pagination) must raise on the first block without rotating. + ctrl = _Controller(["BLOCK", "OK"]) + session = IndeedSession(ctrl.factory) + with pytest.raises(IndeedAccessBlockedError): + await session.fetch_html(_URL, max_rotations=0) + assert session.rotations == 0 + + +def test_browser_work_marshalled_off_selector_loop(): + """The Windows regression this guards: main.py runs the server on a + SelectorEventLoop (psycopg needs it), and Selector loops cannot spawn + subprocesses — patchright's Chromium launch died with NotImplementedError. + A fake session that really spawns a subprocess proves the whole session + lifecycle now runs on the shared subprocess-capable browser loop. + """ + + class _SpawningSession: + async def start(self) -> None: + # The exact call that used to blow up on the server loop. + proc = await asyncio.create_subprocess_exec( + sys.executable, "-c", "print('ok')", stdout=asyncio.subprocess.PIPE + ) + await proc.communicate() + + async def close(self) -> None: + pass + + async def fetch(self, url: str, **_: object) -> _FakePage: + return _FakePage(_OK_HTML if "/jobs" in url else "home", url) + + async def main() -> None: + session = IndeedSession(_SpawningSession) + assert await session.fetch_html(_URL) == _OK_HTML + await session.close() + + asyncio.run(main(), loop_factory=asyncio.SelectorEventLoop) + + +@pytest.mark.asyncio +async def test_warms_domain_once_without_rotation(): + ctrl = _Controller(["OK", "OK"]) + session = IndeedSession(ctrl.factory) + await session.fetch_html(_URL) + await session.fetch_html(_URL + "&start=10") + assert ctrl.home_fetches["https://www.indeed.com/"] == 1 + await session.close() diff --git a/surfsense_backend/tests/unit/platforms/indeed_jobs/test_parsers.py b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_parsers.py new file mode 100644 index 000000000..43fed6948 --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_parsers.py @@ -0,0 +1,196 @@ +"""Offline parser tests: synthetic mapping plus a real captured blob.""" + +from __future__ import annotations + +import json +from pathlib import Path + +from app.proprietary.platforms.indeed_jobs.parsers import ( + extract_jobcards_blob, + job_results, + parse_job, + parse_job_detail, +) + +_FIXTURE_DIR = Path(__file__).parent / "fixtures" + + +# --- synthetic mapping (always runs) --------------------------------------- + + +def _raw_job() -> dict: + return { + "jobkey": "abc123", + "displayTitle": "Senior Data Analyst", + "title": "Senior Data Analyst (fallback)", + "company": "Acme Corp", + "truncatedCompany": "Acme", + "companyOverviewLink": "/cmp/Acme-Corp", + "companyRating": 4.1, + "companyReviewCount": 320, + "formattedLocation": "New York, NY", + "jobLocationCity": "New York", + "jobLocationState": "NY", + "jobLocationPostal": "10001", + "country": "US", + "remoteLocation": False, + "remoteWorkModel": {"type": "REMOTE_ALWAYS", "text": "Remote"}, + "jobTypes": [], + "salarySnippet": {"currency": "USD", "text": "$90,000 - $120,000 a year"}, + "extractedSalary": {"min": 90000, "max": 120000, "type": "YEARLY"}, + "snippet": "
  • 5+ years SQL & Python
", + "sponsored": True, + "newJob": False, + "urgentlyHiring": True, + "expired": False, + "indeedApplyEnabled": True, + "formattedRelativeTime": "3 days ago", + "pubDate": 1_774_242_000_000, + "createDate": 1_774_276_267_415, + "thirdPartyApplyUrl": "https://ats.example.com/apply/abc123", + "taxonomyAttributes": [ + {"label": "job-types", "attributes": [{"label": "Full-time", "suid": "x"}]}, + {"label": "remote", "attributes": [{"label": "Remote", "suid": "y"}]}, + { + "label": "benefits", + "attributes": [ + {"label": "Health insurance", "suid": "a"}, + {"label": "401(k)", "suid": "b"}, + ], + }, + ], + } + + +def test_parse_job_maps_core_fields(): + item = parse_job(_raw_job()) + assert item["jobKey"] == "abc123" + assert item["title"] == "Senior Data Analyst" # displayTitle wins over title + assert item["jobUrl"] == "https://www.indeed.com/viewjob?jk=abc123" + assert item["applyUrl"] == "https://ats.example.com/apply/abc123" + assert item["company"] == "Acme Corp" + assert item["companyUrl"] == "https://www.indeed.com/cmp/Acme-Corp" + assert item["companyReviewCount"] == 320 + assert item["city"] == "New York" + assert item["isRemote"] is True + assert item["remoteType"] == "REMOTE_ALWAYS" + assert item["jobTypes"] == ["Full-time"] + assert item["benefits"] == ["Health insurance", "401(k)"] + assert item["sponsored"] is True + assert item["urgentlyHiring"] is True + + +def test_parse_job_salary_and_snippet(): + item = parse_job(_raw_job()) + sal = item["salary"] + assert sal["salaryText"] == "$90,000 - $120,000 a year" + assert sal["salaryMin"] == 90000 + assert sal["salaryMax"] == 120000 + assert sal["currency"] == "USD" + assert sal["period"] == "year" + assert sal["isEstimated"] is False + # snippet HTML is stripped + entities decoded into plain text. + assert item["descriptionText"] == "5+ years SQL & Python" + assert item["descriptionHtml"] is None + + +def test_parse_job_dates_from_epoch_ms(): + item = parse_job(_raw_job()) + assert item["datePublished"] == "2026-03-23T05:00:00.000Z" + assert item["age"] == "3 days ago" + + +def test_parse_job_respects_base_url(): + item = parse_job(_raw_job(), base_url="https://uk.indeed.com") + assert item["jobUrl"] == "https://uk.indeed.com/viewjob?jk=abc123" + assert item["companyUrl"] == "https://uk.indeed.com/cmp/Acme-Corp" + + +def test_extract_blob_anchors_on_assignment_not_first_occurrence(): + # Decoy mention precedes the real assignment; the extractor must skip it. + html = ( + '' + '' + ) + blob = extract_jobcards_blob(html) + results = job_results(blob) + assert [r["jobkey"] for r in results] == ["k1", "k2"] + + +def test_extract_blob_missing_returns_none(): + assert extract_jobcards_blob("just a moment...") is None + assert job_results(None) == [] + + +# --- fixture-pinned (real captured blob) ----------------------------------- + + +def test_fixture_blob_parses_into_items(): + fixture = _FIXTURE_DIR / "sample_jobcards.json" + blob = json.loads(fixture.read_text()) + results = job_results(blob) + assert len(results) == 3 + for raw in results: + item = parse_job(raw) + assert isinstance(item["jobKey"], str) and item["jobKey"] + assert item["title"] + assert item["jobUrl"].startswith("https://www.indeed.com/viewjob?jk=") + assert "salaryText" in item["salary"] + + +# --- detail page (parse_job_detail) ---------------------------------------- + + +def test_parse_job_detail_extracts_description_and_fields(): + html = (_FIXTURE_DIR / "sample_viewjob.html").read_text() + detail = parse_job_detail(html) + assert detail["descriptionHtml"].startswith("
") + assert "Data Analyst" in detail["descriptionText"] + assert "
" not in detail["descriptionText"] # tags stripped + assert detail["title"] + assert detail["company"] + assert detail["formattedLocation"] + assert detail["jobTypes"] == ["Full-time"] + assert detail["benefits"] == ["401(k)", "Health insurance"] + assert detail["isRemote"] is True + assert detail["remoteType"] == "HYBRID" + sal = detail["salary"] + assert (sal["salaryMin"], sal["salaryMax"], sal["period"]) == (60000, 90000, "year") + + +def test_parse_job_detail_reads_rootprops_shape(): + # Live /viewjob assigns window._rootProps (JSON) with the model under + # preloadedVJData; window._initialData is now a non-JSON JS literal that + # references other globals and must be skipped, not misparsed. + html = ( + "" + '' + ) + detail = parse_job_detail(html) + assert detail["title"] == "Data Analyst" + assert detail["company"] == "Acme" + assert detail["descriptionHtml"] == "

Full JD text

" + assert detail["descriptionText"] == "Full JD text" + + +def test_parse_job_detail_omits_blank_fields(): + # A page with a header but no salary/description must not emit those keys, + # so a merge won't clobber listing values with blanks. + html = ( + "' + ) + detail = parse_job_detail(html) + assert detail == {"title": "Analyst"} + + +def test_parse_job_detail_not_a_job_page_returns_empty(): + assert parse_job_detail("just a moment...") == {} diff --git a/surfsense_backend/tests/unit/platforms/indeed_jobs/test_scraper.py b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_scraper.py new file mode 100644 index 000000000..1d815ce98 --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_scraper.py @@ -0,0 +1,155 @@ +"""Offline orchestration tests: pagination, dedupe, and caps via a fake session.""" + +from __future__ import annotations + +import json +from urllib.parse import parse_qs, urlparse + +import pytest + +from app.proprietary.platforms.indeed_jobs.fetch import IndeedAccessBlockedError +from app.proprietary.platforms.indeed_jobs.schemas import IndeedScrapeInput +from app.proprietary.platforms.indeed_jobs.scraper import iter_indeed, scrape_indeed + + +def _page_html(job_keys: list[str]) -> str: + """Wrap job keys in the ``mosaic-provider-jobcards`` assignment shape.""" + results = [ + {"jobkey": k, "displayTitle": f"Job {k}", "company": "Acme"} for k in job_keys + ] + model = {"metaData": {"mosaicProviderJobCardsModel": {"results": results}}} + return ( + f'window.mosaic.providerData["mosaic-provider-jobcards"]={json.dumps(model)};' + ) + + +def _detail_html(job_key: str) -> str: + """A minimal /viewjob page carrying a full description for ``job_key``.""" + data = { + "jobInfoWrapperModel": { + "jobInfoModel": { + "sanitizedJobDescription": f"
Full description for {job_key}
", + "jobInfoHeaderModel": {"jobTitle": f"Detailed {job_key}"}, + } + } + } + return f"" + + +class _FakeSession: + """Returns per-``start`` search pages (or a /viewjob detail) and records URLs.""" + + def __init__( + self, pages: dict[int, list[str]], blocked_starts: set[int] | None = None + ) -> None: + self._pages = pages + self._blocked = blocked_starts or set() + self.fetched: list[str] = [] + + async def fetch_html(self, url: str, *, max_rotations: int | None = None) -> str: + self.fetched.append(url) + query = parse_qs(urlparse(url).query) + if "/viewjob" in url: + return _detail_html(query.get("jk", [""])[0]) + start = int(query.get("start", ["0"])[0]) + if start in self._blocked: + raise IndeedAccessBlockedError(f"gated at start={start}") + return _page_html(self._pages.get(start, [])) + + +async def _collect(input_model, session) -> list[dict]: + return [item async for item in iter_indeed(input_model, session)] + + +@pytest.mark.asyncio +async def test_dedupes_within_page(): + session = _FakeSession({0: ["k1", "k2", "k2", "k3"]}) + items = await _collect( + IndeedScrapeInput(queries=["dev"], maxItemsPerQuery=100), session + ) + assert [i["jobKey"] for i in items] == ["k1", "k2", "k3"] + assert all(i["scrapedAt"] for i in items) # stamped by the orchestrator + + +@pytest.mark.asyncio +async def test_does_not_fetch_deeper_pages(): + # First page only; ``start>=10`` must never be requested. + session = _FakeSession({0: ["k1", "k2"], 10: ["k3"]}) + items = await _collect( + IndeedScrapeInput(queries=["dev"], maxItemsPerQuery=100), session + ) + assert [i["jobKey"] for i in items] == ["k1", "k2"] + assert all("start=" not in u for u in session.fetched) + + +@pytest.mark.asyncio +async def test_page_block_propagates(): + # Nothing yielded before the block, so it surfaces as an error. + session = _FakeSession({}, blocked_starts={0}) + with pytest.raises(IndeedAccessBlockedError): + await _collect(IndeedScrapeInput(queries=["dev"]), session) + + +@pytest.mark.asyncio +async def test_respects_max_items_per_query(): + session = _FakeSession({0: ["k1", "k2", "k3", "k4"]}) + items = await _collect( + IndeedScrapeInput(queries=["dev"], maxItemsPerQuery=2), session + ) + assert [i["jobKey"] for i in items] == ["k1", "k2"] + + +@pytest.mark.asyncio +async def test_global_dedupe_across_queries(): + # Both queries hit page 0 (same fake pages) and return the same keys. + session = _FakeSession({0: ["k1", "k2"]}) + items = await _collect( + IndeedScrapeInput(queries=["dev", "engineer"], maxItemsPerQuery=100), session + ) + assert [i["jobKey"] for i in items] == ["k1", "k2"] + + +@pytest.mark.asyncio +async def test_start_urls_scrape_search_and_job_url_detail(): + session = _FakeSession({0: ["k1"]}) + input_model = IndeedScrapeInput( + startUrls=[ + {"url": "https://www.indeed.com/jobs?q=dev"}, + {"url": "https://www.indeed.com/viewjob?jk=abc"}, + ], + maxItemsPerQuery=100, + ) + items = await _collect(input_model, session) + assert len(items) == 2 + search_item, job_item = items + assert search_item["jobKey"] == "k1" + # The /viewjob URL is scraped from its detail page alone. + assert job_item["jobUrl"].endswith("jk=abc") + assert job_item["title"] == "Detailed abc" + assert "Full description for abc" in job_item["descriptionText"] + + +@pytest.mark.asyncio +async def test_scrape_job_details_enriches_listing_items(): + session = _FakeSession({0: ["k1", "k2"]}) + items = await _collect( + IndeedScrapeInput(queries=["dev"], maxItemsPerQuery=100, scrapeJobDetails=True), + session, + ) + assert [i["jobKey"] for i in items] == ["k1", "k2"] + for it in items: + assert it["descriptionHtml"].startswith("
Full description for") + assert "Full description for" in it["descriptionText"] + # One extra /viewjob load per listing item. + assert sum("/viewjob" in u for u in session.fetched) == 2 + + +@pytest.mark.asyncio +async def test_scrape_indeed_limit_with_injected_session(): + session = _FakeSession({0: ["k1", "k2", "k3"]}) + items = await scrape_indeed( + IndeedScrapeInput(queries=["dev"], maxItemsPerQuery=100), + limit=2, + session=session, + ) + assert [i["jobKey"] for i in items] == ["k1", "k2"] diff --git a/surfsense_backend/tests/unit/platforms/indeed_jobs/test_url_resolver.py b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_url_resolver.py new file mode 100644 index 000000000..b995b3d0b --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/indeed_jobs/test_url_resolver.py @@ -0,0 +1,87 @@ +"""Offline tests for Indeed URL classification and search-URL building.""" + +from __future__ import annotations + +from urllib.parse import parse_qs, urlparse + +from app.proprietary.platforms.indeed_jobs.url_resolver import ( + build_search_url, + country_domain, + resolve_url, +) + + +def test_resolve_search_url(): + r = resolve_url( + "https://www.indeed.com/jobs?q=software+engineer&l=Remote&sort=date" + ) + assert r is not None + assert r.kind == "search" + assert r.value == "software engineer" + assert r.location == "Remote" + assert r.domain == "www.indeed.com" + assert r.params.get("sort") == "date" + + +def test_resolve_company_url(): + r = resolve_url("https://www.indeed.com/cmp/Google/jobs") + assert r is not None + assert r.kind == "company" + assert r.value == "Google" + + +def test_resolve_viewjob_url(): + r = resolve_url("https://uk.indeed.com/viewjob?jk=abc123&from=serp") + assert r is not None + assert r.kind == "job" + assert r.value == "abc123" + assert r.domain == "uk.indeed.com" + + +def test_resolve_country_subdomain_host(): + r = resolve_url("https://de.indeed.com/jobs?q=entwickler") + assert r is not None + assert r.kind == "search" + assert r.domain == "de.indeed.com" + + +def test_resolve_rejects_non_indeed(): + assert resolve_url("https://www.linkedin.com/jobs?q=dev") is None + assert resolve_url("https://notindeed.com.evil.com/jobs") is None + + +def test_country_domain_map(): + assert country_domain("us") == "www.indeed.com" + assert country_domain("gb") == "uk.indeed.com" + assert country_domain("de") == "de.indeed.com" + assert country_domain("") == "www.indeed.com" + + +def test_build_search_url_basic(): + url = build_search_url( + "data analyst", + country="us", + location="New York, NY", + sort="date", + start=20, + ) + parsed = urlparse(url) + qs = parse_qs(parsed.query) + assert parsed.netloc == "www.indeed.com" + assert parsed.path == "/jobs" + assert qs["q"] == ["data analyst"] + assert qs["l"] == ["New York, NY"] + assert qs["sort"] == ["date"] + assert qs["start"] == ["20"] + + +def test_build_search_url_remote_keyword_fallback_and_jobtype(): + url = build_search_url( + "developer", country="gb", remote="remote", job_type="fulltime", from_days=7 + ) + parsed = urlparse(url) + qs = parse_qs(parsed.query) + assert parsed.netloc == "uk.indeed.com" + assert qs["q"] == ["developer remote"] + assert qs["jt"] == ["fulltime"] + assert qs["fromage"] == ["7"] diff --git a/surfsense_backend/tests/unit/platforms/reddit/test_community_listing.py b/surfsense_backend/tests/unit/platforms/reddit/test_community_listing.py new file mode 100644 index 000000000..89fa60b64 --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/reddit/test_community_listing.py @@ -0,0 +1,85 @@ +"""Offline tests: a bare ``community`` scrapes its subreddit listing. + +No network: ``_subreddit_flow`` is faked. Guards the capability-schema promise +that a community with no urls/searches scrapes its listing (regression for the +orchestrator dropping community-only input), and that the name is normalized at +the trust boundary ("r/x", "/r/x/", " x " -> "x") so it never becomes ``r/r/x``. +""" + +from __future__ import annotations + +from collections.abc import AsyncIterator + +import pytest + +from app.proprietary.platforms.reddit import scraper +from app.proprietary.platforms.reddit.schemas import RedditScrapeInput + + +def _fake_subreddit_flow(seen: list[str]): + """Record the subreddit it's called with; yield a community + one post.""" + + def flow( + subreddit: str, + *, + input_model: RedditScrapeInput, + sort: str | None = None, + ) -> AsyncIterator[dict]: + seen.append(subreddit) + + async def gen() -> AsyncIterator[dict]: + yield {"dataType": "community", "id": subreddit} + yield {"dataType": "post", "id": f"{subreddit}-p1", "title": "t"} + + return gen() + + return flow + + +async def test_community_only_scrapes_listing(monkeypatch): + seen: list[str] = [] + monkeypatch.setattr(scraper, "_subreddit_flow", _fake_subreddit_flow(seen)) + + model = RedditScrapeInput(searchCommunityName="movies", maxItems=10) + items = await scraper.scrape_reddit(model, limit=10) + + assert seen == ["movies"] + assert [i["id"] for i in items] == ["movies", "movies-p1"] + + +@pytest.mark.parametrize( + "raw", ["movies", "r/movies", "/r/movies/", " movies ", "R/movies"] +) +async def test_community_name_normalized(monkeypatch, raw): + seen: list[str] = [] + monkeypatch.setattr(scraper, "_subreddit_flow", _fake_subreddit_flow(seen)) + + model = RedditScrapeInput(searchCommunityName=raw, maxItems=5) + await scraper.scrape_reddit(model, limit=5) + + assert seen == ["movies"] + + +async def test_searches_present_keeps_community_as_scope(monkeypatch): + """With searches, the name stays a search scope (community-only branch skipped).""" + sub_seen: list[str] = [] + monkeypatch.setattr(scraper, "_subreddit_flow", _fake_subreddit_flow(sub_seen)) + + scoped: list[str | None] = [] + + def fake_search_flow(query, *, input_model, subreddit=None, max_items=None): + scoped.append(subreddit) + + async def gen() -> AsyncIterator[dict]: + yield {"dataType": "post", "id": f"{query}-1", "title": query} + + return gen() + + monkeypatch.setattr(scraper, "_search_flow", fake_search_flow) + + model = RedditScrapeInput(searches=["tempo"], searchCommunityName="r/movies") + items = await scraper.scrape_reddit(model, limit=10) + + assert sub_seen == [] # community-only flow NOT taken + assert scoped == ["movies"] # normalized name used as the search scope + assert [i["id"] for i in items] == ["tempo-1"] diff --git a/surfsense_backend/tests/unit/platforms/walmart/__init__.py b/surfsense_backend/tests/unit/platforms/walmart/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/surfsense_backend/tests/unit/platforms/walmart/fixtures/blocked.html b/surfsense_backend/tests/unit/platforms/walmart/fixtures/blocked.html new file mode 100644 index 000000000..a9e9d606e --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/walmart/fixtures/blocked.html @@ -0,0 +1,4 @@ +Robot or human? +
Activate and hold the button to confirm that you're human.
+

Robot or human?

+ diff --git a/surfsense_backend/tests/unit/platforms/walmart/fixtures/listing.html b/surfsense_backend/tests/unit/platforms/walmart/fixtures/listing.html new file mode 100644 index 000000000..0ebb7e682 --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/walmart/fixtures/listing.html @@ -0,0 +1,3 @@ +laptop - Walmart.com + + diff --git a/surfsense_backend/tests/unit/platforms/walmart/fixtures/product.html b/surfsense_backend/tests/unit/platforms/walmart/fixtures/product.html new file mode 100644 index 000000000..183ddff6b --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/walmart/fixtures/product.html @@ -0,0 +1,3 @@ +Midea AC + + diff --git a/surfsense_backend/tests/unit/platforms/walmart/fixtures/reviews.html b/surfsense_backend/tests/unit/platforms/walmart/fixtures/reviews.html new file mode 100644 index 000000000..b0d27e5b6 --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/walmart/fixtures/reviews.html @@ -0,0 +1,3 @@ +Reviews + + diff --git a/surfsense_backend/tests/unit/platforms/walmart/test_flows.py b/surfsense_backend/tests/unit/platforms/walmart/test_flows.py new file mode 100644 index 000000000..045aa99d5 --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/walmart/test_flows.py @@ -0,0 +1,138 @@ +from __future__ import annotations + +from pathlib import Path + +from app.proprietary.platforms.walmart import ( + WalmartReviewsInput, + WalmartScrapeInput, + scrape_products, + scrape_reviews, + scraper, +) +from app.proprietary.platforms.walmart.fetch import FetchResult + +_FIXTURES = Path(__file__).parent / "fixtures" + + +def _fixture(name: str) -> str: + return (_FIXTURES / name).read_text(encoding="utf-8") + + +def _response(url: str, html: str, status: int = 200) -> FetchResult: + return FetchResult(status=status, html=html, url=url, cookies={}) + + +async def test_product_flow_emits_parsed_item(monkeypatch): + async def fetch_page(url: str, **_kwargs): + return _response(url, _fixture("product.html")) + + monkeypatch.setattr(scraper, "fetch_page", fetch_page) + items = await scrape_products( + WalmartScrapeInput(startUrls=["https://www.walmart.com/ip/212092810"]) + ) + + assert len(items) == 1 + assert items[0]["usItemId"] == "212092810" + assert items[0]["name"].startswith("Midea") + + +async def test_product_flow_maps_not_found_to_error_item(monkeypatch): + async def fetch_page(url: str, **_kwargs): + return _response(url, "", status=404) + + monkeypatch.setattr(scraper, "fetch_page", fetch_page) + items = await scrape_products( + WalmartScrapeInput(startUrls=["https://www.walmart.com/ip/212092810"]) + ) + + assert items[0]["error"] == "product_not_found" + + +async def test_listing_flow_card_only_honors_cap(monkeypatch): + calls: list[str] = [] + + async def fetch_page(url: str, **_kwargs): + calls.append(url) + html = _fixture("listing.html") if "page=1" in url else "" + return _response(url, html) + + monkeypatch.setattr(scraper, "fetch_page", fetch_page) + items = await scrape_products( + WalmartScrapeInput( + startUrls=["https://www.walmart.com/search?q=laptop"], + maxItemsPerStartUrl=1, + includeDetails=False, + ) + ) + + assert len(items) == 1 + assert items[0]["usItemId"] == "791595618" + assert len(calls) == 1 + + +async def test_listing_flow_enriches_with_detail_pages(monkeypatch): + async def fetch_page(url: str, **_kwargs): + if "/ip/" in url: + return _response(url, _fixture("product.html")) + html = _fixture("listing.html") if "page=1" in url else "" + return _response(url, html) + + monkeypatch.setattr(scraper, "fetch_page", fetch_page) + items = await scrape_products( + WalmartScrapeInput( + startUrls=["https://www.walmart.com/search?q=laptop"], + maxItemsPerStartUrl=2, + includeDetails=True, + ) + ) + + # Both cards enrich to the same product fixture (detail fetch wins). + assert all(item["longDescription"] for item in items) + + +async def test_invalid_url_yields_error_item(monkeypatch): + items = await scrape_products( + WalmartScrapeInput(startUrls=["https://example.com/not-walmart"]) + ) + assert items[0]["error"] == "invalid_url" + + +async def test_reviews_flow_paginates_until_empty(monkeypatch): + calls: list[str] = [] + + async def fetch_page(url: str, **_kwargs): + calls.append(url) + html = _fixture("reviews.html") if "page=1" in url else "" + return _response(url, html) + + monkeypatch.setattr(scraper, "fetch_page", fetch_page) + items = await scrape_reviews( + WalmartReviewsInput(itemIds=["212092810"], maxReviews=100) + ) + + assert len(items) == 2 + assert items[0]["reviewId"] == "296013686" + # page=1 returned records, page=2 empty → stop. Two fetches total. + assert len(calls) == 2 + + +async def test_reviews_flow_honors_max_reviews(monkeypatch): + async def fetch_page(url: str, **_kwargs): + return _response(url, _fixture("reviews.html")) + + monkeypatch.setattr(scraper, "fetch_page", fetch_page) + items = await scrape_reviews( + WalmartReviewsInput(itemIds=["212092810"], maxReviews=1) + ) + + assert len(items) == 1 + + +async def test_reviews_flow_maps_empty_to_error_item(monkeypatch): + async def fetch_page(url: str, **_kwargs): + return _response(url, "") + + monkeypatch.setattr(scraper, "fetch_page", fetch_page) + items = await scrape_reviews(WalmartReviewsInput(itemIds=["212092810"])) + + assert items[0]["error"] == "reviews_not_found" diff --git a/surfsense_backend/tests/unit/platforms/walmart/test_parsers.py b/surfsense_backend/tests/unit/platforms/walmart/test_parsers.py new file mode 100644 index 000000000..596131a2e --- /dev/null +++ b/surfsense_backend/tests/unit/platforms/walmart/test_parsers.py @@ -0,0 +1,93 @@ +from __future__ import annotations + +from pathlib import Path + +from app.proprietary.platforms.walmart.fetch import is_blocked +from app.proprietary.platforms.walmart.next_data import extract_next_data +from app.proprietary.platforms.walmart.parsers import ( + parse_listing_page, + parse_product, + parse_reviews_page, +) +from app.proprietary.platforms.walmart.url_resolver import extract_item_id, resolve_url + +_FIXTURES = Path(__file__).parent / "fixtures" + + +def _fixture(name: str) -> str: + return (_FIXTURES / name).read_text(encoding="utf-8") + + +def test_product_parser_extracts_core_fields_and_review_sample(): + data = extract_next_data(_fixture("product.html")) + item = parse_product(data, url="https://www.walmart.com/ip/212092810") + + assert item["usItemId"] == "212092810" + assert item["name"].startswith("Midea") + assert item["price"] == {"value": 149.0, "currency": "USD"} + assert item["listPrice"] == {"value": 199.0, "currency": "USD"} + assert item["stars"] == 4.5 + assert item["reviewsCount"] == 6287 + assert item["inStock"] is True + assert item["seller"] == {"id": "F55CT", "name": "Walmart.com", "type": "WALMART"} + assert item["longDescription"] == "

A powerful window unit.

" + assert item["images"] == [ + "https://i5.walmartimages.com/1.jpg", + "https://i5.walmartimages.com/2.jpg", + ] + # Walmart ships category.path as breadcrumb objects, not a string. + assert item["breadCrumbs"] == ["Home", "Air Conditioners"] + assert item["category"] == "Air Conditioners" + assert item["reviewsSample"]["totalReviewCount"] == 6287 + assert item["reviewsSample"]["topReviews"][0]["verifiedPurchase"] is True + + +def test_listing_parser_normalizes_cards_and_marketplace_seller(): + data = extract_next_data(_fixture("listing.html")) + cards = parse_listing_page(data) + + assert [c["usItemId"] for c in cards] == ["791595618", "999001"] + first = cards[0] + assert first["url"] == "https://www.walmart.com/ip/Acer-Chromebook/791595618" + assert first["price"] == {"value": 79.99, "currency": "USD"} + assert first["seller"]["type"] == "MARKETPLACE" + assert first["inStock"] is True + # Fallback price + out-of-stock derivation for the second card. + assert cards[1]["price"] == {"value": 499.0} + assert cards[1]["inStock"] is False + + +def test_reviews_parser_extracts_all_records(): + data = extract_next_data(_fixture("reviews.html")) + reviews = parse_reviews_page(data) + + assert len(reviews) == 2 + assert reviews[0]["reviewId"] == "296013686" + assert reviews[0]["author"] == "JohnPaul" + assert reviews[0]["verifiedPurchase"] is True + assert reviews[0]["images"] == ["https://i5.walmartimages.com/r.jpg"] + assert reviews[0]["sellerResponse"] == "Thanks for the review!" + assert reviews[1]["verifiedPurchase"] is False + + +def test_block_detection_handles_status_and_body(): + assert is_blocked(_fixture("blocked.html"), 200) + assert is_blocked("", 412) + assert is_blocked("", 429) + assert not is_blocked(_fixture("product.html"), 200) + + +def test_missing_next_data_yields_none(): + assert extract_next_data("no script here") is None + assert parse_product(extract_next_data("") or {}, url="x") == {} + + +def test_url_resolver_classifies_and_extracts_ids(): + assert resolve_url("https://www.walmart.com/ip/Foo/123456789").kind == "product" + assert ( + extract_item_id("https://www.walmart.com/reviews/product/212092810") + == "212092810" + ) + assert resolve_url("https://www.walmart.com/search?q=tv").kind == "listing" + assert resolve_url("https://www.walmart.com/cp/tvs/3944").kind == "listing" + assert resolve_url("https://example.com/ip/1") is None diff --git a/surfsense_backend/tests/unit/platforms/youtube/test_parsers.py b/surfsense_backend/tests/unit/platforms/youtube/test_parsers.py index 55f591c63..6b7b460a9 100644 --- a/surfsense_backend/tests/unit/platforms/youtube/test_parsers.py +++ b/surfsense_backend/tests/unit/platforms/youtube/test_parsers.py @@ -946,8 +946,22 @@ def test_resolve_url(url, kind, value): assert resolved.value == value -def test_resolve_url_unrecognized(): - assert resolve_url("https://example.com/foo") is None +@pytest.mark.parametrize( + "url", + [ + "https://example.com/foo", + # Well-formed non-YouTube URLs whose path mimics a YouTube page must be + # rejected, not misclassified by path alone (host-spoof guard). + "https://evil.com/@Apify", + "https://evil.com/channel/UC123456789abc", + "https://evil.com/shorts/abc123", + "https://evil.com/playlist?list=PL123", + "https://evil.com/hashtag/tech", + "https://evil.com/results?search_query=web+scraping", + ], +) +def test_resolve_url_unrecognized(url): + assert resolve_url(url) is None # --- optional: exercise captured real fixtures if present -------------------- diff --git a/surfsense_backend/tests/unit/routes/test_workspaces_limits.py b/surfsense_backend/tests/unit/routes/test_workspaces_limits.py new file mode 100644 index 000000000..ab89d08bc --- /dev/null +++ b/surfsense_backend/tests/unit/routes/test_workspaces_limits.py @@ -0,0 +1,63 @@ +from __future__ import annotations + +from types import SimpleNamespace + +import pytest +from fastapi import HTTPException + +from app.routes import workspaces_routes +from app.schemas import WorkspaceCreate + +pytestmark = pytest.mark.unit + + +class _CountResult: + def __init__(self, count: int): + self.count = count + + def scalar_one(self) -> int: + return self.count + + +class _FakeSession: + def __init__(self, owned_count: int): + self.owned_count = owned_count + + async def execute(self, _statement): + return _CountResult(self.owned_count) + + +@pytest.mark.asyncio +async def test_read_workspace_limits_uses_backend_config(monkeypatch): + monkeypatch.setattr( + workspaces_routes.config, + "MAX_WORKSPACES_PER_USER", + 37, + raising=False, + ) + + result = await workspaces_routes.read_workspace_limits(_auth=SimpleNamespace()) + + assert result == {"max_workspaces_per_user": 37} + + +@pytest.mark.asyncio +async def test_create_workspace_rejects_when_owned_limit_reached(monkeypatch): + monkeypatch.setattr( + workspaces_routes.config, + "MAX_WORKSPACES_PER_USER", + 2, + raising=False, + ) + auth = SimpleNamespace(user=SimpleNamespace(id="user-1")) + session = _FakeSession(owned_count=2) + + with pytest.raises(HTTPException) as exc_info: + await workspaces_routes.create_workspace( + WorkspaceCreate(name="Extra", description=""), + session=session, + auth=auth, + ) + + assert exc_info.value.status_code == 409 + assert "at most 2 workspaces" in exc_info.value.detail diff --git a/surfsense_backend/tests/unit/services/okf/test_audit.py b/surfsense_backend/tests/unit/services/okf/test_audit.py new file mode 100644 index 000000000..f8303e237 --- /dev/null +++ b/surfsense_backend/tests/unit/services/okf/test_audit.py @@ -0,0 +1,44 @@ +"""A real export bundle - concepts plus reserved ``index.md``/``log.md`` - must +pass the same ``validate_bundle`` a consumer would run. +""" + +from datetime import UTC, datetime + +from app.db import Document, DocumentType +from app.services.okf import ( + LogEntry, + document_to_concept, + folder_to_index, + folder_to_log, + validate_bundle, +) + + +def _sample_bundle() -> dict[str, str]: + note = Document( + title="Weekly Sync", + document_type=DocumentType.NOTE, + document_metadata={"tags": ["team"]}, + updated_at=datetime(2026, 5, 28, tzinfo=UTC), + ) + page = Document(title="Docs Home", document_type=DocumentType.CRAWLED_URL) + return { + "weekly-sync.md": document_to_concept(note, body="# Agenda"), + "docs-home.md": document_to_concept(page, body="content"), + # Reserved files carry no frontmatter and must be exempt from the check. + "index.md": folder_to_index(), + "log.md": folder_to_log( + [LogEntry(title="Weekly Sync", timestamp="2026-05-28T00:00:00+00:00")] + ), + } + + +def test_real_export_bundle_is_conformant() -> None: + assert validate_bundle(_sample_bundle()) == {} + + +def test_audit_flags_a_drifted_concept() -> None: + bundle = _sample_bundle() + bundle["broken.md"] = "no frontmatter at all" + problems = validate_bundle(bundle) + assert list(problems) == ["broken.md"] diff --git a/surfsense_backend/tests/unit/services/okf/test_conformance.py b/surfsense_backend/tests/unit/services/okf/test_conformance.py new file mode 100644 index 000000000..da1988224 --- /dev/null +++ b/surfsense_backend/tests/unit/services/okf/test_conformance.py @@ -0,0 +1,51 @@ +"""Every ``DocumentType`` must serialize to a concept a permissive OKF consumer +can read: parseable frontmatter with a non-empty ``type``. Covers all types plus +missing metadata, empty body, and non-ASCII titles. +""" + +from datetime import UTC, datetime + +import pytest + +from app.db import Document, DocumentType +from app.services.okf import document_to_concept, is_conformant_concept +from app.services.okf.validator import ( + RECOMMENDED_FRONTMATTER_KEYS, + REQUIRED_FRONTMATTER_KEYS, +) + + +@pytest.mark.parametrize("document_type", list(DocumentType)) +def test_every_document_type_serializes_to_conformant_concept( + document_type: DocumentType, +) -> None: + doc = Document( + title="Sample", + document_type=document_type, + document_metadata={"url": "https://example.com/x"}, + updated_at=datetime(2026, 5, 28, tzinfo=UTC), + ) + assert is_conformant_concept(document_to_concept(doc, body="body")) + + +def test_conformant_without_metadata_or_body() -> None: + doc = Document(title="Bare", document_type=DocumentType.NOTE) + assert is_conformant_concept(document_to_concept(doc, body="")) + + +def test_conformant_with_non_ascii_title() -> None: + doc = Document(title="日本語ノート", document_type=DocumentType.NOTE) + concept = document_to_concept(doc, body="本文") + assert is_conformant_concept(concept) + assert "日本語ノート" in concept + + +def test_contract_marks_only_type_as_required() -> None: + assert REQUIRED_FRONTMATTER_KEYS == ("type",) + assert set(RECOMMENDED_FRONTMATTER_KEYS) == { + "title", + "description", + "resource", + "tags", + "timestamp", + } diff --git a/surfsense_backend/tests/unit/services/okf/test_ingestion.py b/surfsense_backend/tests/unit/services/okf/test_ingestion.py new file mode 100644 index 000000000..e7c49bc33 --- /dev/null +++ b/surfsense_backend/tests/unit/services/okf/test_ingestion.py @@ -0,0 +1,31 @@ +"""A ConnectorDocument (the indexing write door) must serialize to a valid OKF +concept. Per-``DocumentType`` conformance lives in ``test_conformance.py``. +""" + +from app.db import Document, DocumentType +from app.indexing_pipeline.connector_document import ConnectorDocument +from app.services.okf import document_to_concept, is_conformant_concept + + +def _document_from(connector_doc: ConnectorDocument) -> Document: + """Mirror how prepare_for_indexing builds a Document from a ConnectorDocument.""" + return Document( + title=connector_doc.title, + document_type=connector_doc.document_type, + source_markdown=connector_doc.source_markdown, + document_metadata=connector_doc.metadata, + ) + + +def test_minimal_connector_document_yields_conformant_concept() -> None: + connector_doc = ConnectorDocument( + title="Bare", + source_markdown="just a body", + unique_id="u1", + document_type=DocumentType.FILE, + workspace_id=1, + created_by_id="user-1", + ) + doc = _document_from(connector_doc) + concept = document_to_concept(doc, body=doc.source_markdown) + assert is_conformant_concept(concept) diff --git a/surfsense_backend/tests/unit/services/okf/test_serializer.py b/surfsense_backend/tests/unit/services/okf/test_serializer.py new file mode 100644 index 000000000..d52addf5b --- /dev/null +++ b/surfsense_backend/tests/unit/services/okf/test_serializer.py @@ -0,0 +1,128 @@ +"""OKF serializer/validator self-checks: emitted concepts stay conformant and the +frontmatter fields consumers rely on (type/title/timestamp) round-trip. +""" + +from datetime import UTC, datetime + +from app.db import Document, DocumentType +from app.services.okf import ( + ConceptRef, + LogEntry, + SubdirRef, + document_to_concept, + folder_to_index, + folder_to_log, + is_conformant_concept, + parse_frontmatter, + validate_concept, +) + + +def _make_document() -> Document: + return Document( + title="Weekly Sync Notes", + document_type=DocumentType.NOTE, + document_metadata={"tags": ["team", "meeting"], "url": "https://example.com/n"}, + updated_at=datetime(2026, 5, 28, 22, 49, 59, tzinfo=UTC), + ) + + +def test_concept_is_conformant_and_roundtrips() -> None: + concept = document_to_concept(_make_document(), body="# Agenda\n\nShip OKF.") + + assert is_conformant_concept(concept) + + frontmatter, error = parse_frontmatter(concept) + assert error is None + assert frontmatter["type"] == "Note" + assert frontmatter["title"] == "Weekly Sync Notes" + assert frontmatter["tags"] == ["team", "meeting"] + assert frontmatter["resource"] == "https://example.com/n" + # timestamp must survive as an ISO-8601 string, not a parsed datetime. + assert frontmatter["timestamp"] == "2026-05-28T22:49:59+00:00" + assert "# Agenda" in concept + + +def test_type_is_always_present_even_without_metadata() -> None: + doc = Document(title="Raw", document_type=DocumentType.CRAWLED_URL) + concept = document_to_concept(doc, body="body") + frontmatter, error = parse_frontmatter(concept) + assert error is None + assert frontmatter["type"] == "Web Page" + # No source URL / tags available -> those recommended keys are omitted. + assert "resource" not in frontmatter + assert "tags" not in frontmatter + + +def test_validator_rejects_non_conformant_documents() -> None: + assert validate_concept("no frontmatter here") + assert validate_concept("---\ntitle: Missing type\n---\nbody") + + +def test_folder_index_groups_by_type_and_lists_subdirs() -> None: + index = folder_to_index( + concepts=[ + ConceptRef( + title="Orders", filename="orders.md", type="Note", description="x" + ), + ], + subdirectories=[SubdirRef(name="tables", description="Table docs")], + ) + assert "# Subdirectories" in index + assert "* [tables](tables/index.md) - Table docs" in index + assert "# Note" in index + assert "* [Orders](orders.md) - x" in index + + +def test_folder_log_lists_concepts_newest_first() -> None: + log = folder_to_log( + [ + LogEntry(title="Older", timestamp="2026-01-01T00:00:00+00:00"), + LogEntry(title="Newer", timestamp="2026-06-01T00:00:00+00:00"), + LogEntry(title="Undated", timestamp=None), + ] + ) + assert "# Change Log" in log + # Newest dated entry precedes the older one; undated sorts last. + assert log.index("Newer") < log.index("Older") < log.index("Undated") + assert "* Newer - 2026-06-01T00:00:00+00:00" in log + + +def test_folder_log_is_empty_when_no_entries() -> None: + assert folder_to_log([]) == "" + + +def test_export_log_files_synthesized_only_where_docs_live() -> None: + from app.services.export_service import _build_log_files + + files = dict( + _build_log_files( + { + "": [LogEntry(title="Root Doc", timestamp="2026-05-01T00:00:00+00:00")], + "Research/AI": [LogEntry(title="Nested", timestamp=None)], + } + ) + ) + assert "# Change Log" in files["log.md"] + assert "Root Doc" in files["log.md"] + assert "Nested" in files["Research/AI/log.md"] + # No empty intermediate log: "Research" holds no concepts of its own. + assert "Research/log.md" not in files + + +def test_export_index_files_include_root_version_and_ancestors() -> None: + from app.services.export_service import _build_index_files + + # A concept nested two levels deep, with no direct docs in the middle dir. + files = dict( + _build_index_files( + {"Research/AI": [ConceptRef(title="Note", filename="note.md", type="Note")]} + ) + ) + + # Root, the empty intermediate dir, and the leaf all get an index.md so the + # hierarchy is fully navigable. + assert files["index.md"].startswith('---\nokf_version: "0.1"\n---') + assert "* [Research](Research/index.md)" in files["index.md"] + assert "* [AI](AI/index.md)" in files["Research/index.md"] + assert "* [Note](note.md)" in files["Research/AI/index.md"] diff --git a/surfsense_backend/tests/unit/services/test_model_connections.py b/surfsense_backend/tests/unit/services/test_model_connections.py index bb5ca318e..296c3fbe2 100644 --- a/surfsense_backend/tests/unit/services/test_model_connections.py +++ b/surfsense_backend/tests/unit/services/test_model_connections.py @@ -64,6 +64,22 @@ def test_openai_compatible_resolver_uses_explicit_api_base() -> None: assert ensure_v1("http://example.com/v1") == "http://example.com/v1" +def test_lm_studio_resolver_supplies_dummy_api_key_when_empty() -> None: + model, kwargs = to_litellm( + { + "provider": "lm_studio", + "base_url": "http://host.docker.internal:1234/v1", + "api_key": None, + "extra": {}, + }, + "tinyllama-1.1b-chat-v0.6", + ) + + assert model == "openai/tinyllama-1.1b-chat-v0.6" + assert kwargs["api_base"] == "http://host.docker.internal:1234/v1" + assert kwargs["api_key"] == "not-needed" + + def test_openai_compatible_raw_resolver_does_not_append_v1() -> None: model, kwargs = to_litellm( { diff --git a/surfsense_backend/tests/unit/test_error_contract.py b/surfsense_backend/tests/unit/test_error_contract.py index ec8021290..e1e160232 100644 --- a/surfsense_backend/tests/unit/test_error_contract.py +++ b/surfsense_backend/tests/unit/test_error_contract.py @@ -14,6 +14,7 @@ import json import pytest from fastapi import HTTPException +from pydantic import BaseModel, model_validator from starlette.testclient import TestClient from app.exceptions import ( @@ -32,6 +33,23 @@ from app.exceptions import ( pytestmark = pytest.mark.unit +class _RequireOne(BaseModel): + """Body model with a cross-field rule, used to test the 422 summary. + + Defined at module scope (not inside the app factory) so FastAPI's ``Body`` + ``TypeAdapter`` can resolve the forward reference. + """ + + a: str | None = None + b: str | None = None + + @model_validator(mode="after") + def _at_least_one(self): + if not self.a and not self.b: + raise ValueError("Provide at least one of 'a' or 'b'.") + return self + + # --------------------------------------------------------------------------- # Helpers - lightweight FastAPI app that re-uses the real global handlers # --------------------------------------------------------------------------- @@ -39,7 +57,7 @@ pytestmark = pytest.mark.unit def _make_test_app(): """Build a minimal FastAPI app with the same handlers as the real one.""" - from fastapi import FastAPI + from fastapi import Body, FastAPI from fastapi.exceptions import RequestValidationError from pydantic import BaseModel @@ -124,6 +142,10 @@ def _make_test_app(): async def validated(item: Item): return item.model_dump() + @app.post("/require-one") + async def require_one(payload: _RequireOne = Body(...)): + return payload.model_dump() + return app @@ -277,6 +299,26 @@ class TestValidationErrorHandler: body = _assert_envelope(resp, 422) assert body["error"]["code"] == "VALIDATION_ERROR" + def test_field_error_drops_body_root(self, client): + # The "body" request root is noise in the human summary; the field name + # is kept so clients can still attach the error inline. + resp = client.post("/require-one", json={"a": 123}) + body = _assert_envelope(resp, 422) + msg = body["error"]["message"] + assert "body" not in msg + assert "a:" in msg + assert body["error"]["fields"][0]["loc"] == ["body", "a"] + + def test_model_level_error_reads_as_sentence(self, client): + # A cross-field rule (loc == ["body"]) must read as a plain sentence, + # not "body: ...", while fields still carry the location for clients. + resp = client.post("/require-one", json={}) + body = _assert_envelope(resp, 422) + assert body["error"]["message"] == ( + "Validation failed: Provide at least one of 'a' or 'b'." + ) + assert body["error"]["fields"][0]["loc"] == ["body"] + # --------------------------------------------------------------------------- # SurfSenseError class hierarchy unit tests diff --git a/surfsense_backend/uv.lock b/surfsense_backend/uv.lock index 3a27770b6..23d98934e 100644 --- a/surfsense_backend/uv.lock +++ b/surfsense_backend/uv.lock @@ -48,6 +48,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'linux' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -94,6 +100,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform == 'linux' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -140,6 +152,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -186,6 +204,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'emscripten' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -232,6 +256,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform != 'emscripten' and sys_platform != 'linux' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -278,6 +308,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -324,6 +360,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'emscripten' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -370,6 +412,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform != 'emscripten' and sys_platform != 'linux' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -416,6 +464,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform != 'linux' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -462,6 +516,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra == 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -523,6 +583,14 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'linux' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -569,6 +637,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'linux' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -615,6 +689,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform == 'linux' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -661,6 +741,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -707,6 +793,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'emscripten' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -753,6 +845,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform != 'emscripten' and sys_platform != 'linux' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -799,6 +897,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -845,6 +949,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'emscripten' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -891,6 +1001,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform != 'emscripten' and sys_platform != 'linux' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -937,6 +1053,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform != 'linux' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -983,6 +1105,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra == 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1044,6 +1172,14 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'linux' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1090,6 +1226,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'linux' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1136,6 +1278,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform == 'linux' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1182,6 +1330,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'win32' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1228,6 +1382,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'emscripten' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1274,6 +1434,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform != 'emscripten' and sys_platform != 'linux' and sys_platform != 'win32' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1320,6 +1486,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'win32' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1366,6 +1538,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'emscripten' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1412,6 +1590,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform != 'emscripten' and sys_platform != 'linux' and sys_platform != 'win32' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1458,6 +1642,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform != 'linux' and sys_platform != 'win32' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1504,6 +1694,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform == 'win32' and extra == 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1565,6 +1761,14 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1611,6 +1815,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform == 'emscripten' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1702,6 +1912,18 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version >= '3.14' and sys_platform != 'emscripten' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1748,6 +1970,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1794,6 +2022,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform == 'emscripten' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1885,6 +2119,18 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version == '3.13.*' and sys_platform != 'emscripten' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -1976,6 +2222,18 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform != 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", "python_version < '0'", "python_version < '0'", @@ -2022,6 +2280,12 @@ resolution-markers = [ "python_version < '0'", "python_version < '0'", "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", + "python_version < '0'", "python_full_version < '3.13' and sys_platform == 'win32' and extra != 'extra-16-surf-new-backend-cpu' and extra != 'extra-16-surf-new-backend-cu126' and extra != 'extra-16-surf-new-backend-cu128'", ] conflicts = [[ @@ -2557,6 +2821,15 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/77/f5/21d2de20e8b8b0408f0681956ca2c69f1320a3848ac50e6e7f39c6159675/babel-2.18.0-py3-none-any.whl", hash = "sha256:e2b422b277c2b9a9630c1d7903c2a00d0830c409c59ac8cae9081c92f1aeba35", size = 10196845 }, ] +[[package]] +name = "backoff" +version = "2.2.1" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/47/d7/5bbeb12c44d7c4f2fb5b56abce497eb5ed9f34d85701de869acedd602619/backoff-2.2.1.tar.gz", hash = "sha256:03f829f5bb1923180821643f8753b0502c3b682293992485b0eef2807afa5cba", size = 17001 } +wheels = [ + { url = "https://files.pythonhosted.org/packages/df/73/b6e24bd22e6720ca8ee9a85a0c4a2971af8497d8f3193fa05390cbd46e09/backoff-2.2.1-py3-none-any.whl", hash = "sha256:63579f9a0628e06278f7e47b7d7d5b6ce20dc65c5e96a6f3ca99a6adca0396e8", size = 15148 }, +] + [[package]] name = "banks" version = "2.4.1" @@ -8650,6 +8923,21 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/4b/a6/38c8e2f318bf67d338f4d629e93b0b4b9af331f455f0390ea8ce4a099b26/portalocker-3.2.0-py3-none-any.whl", hash = "sha256:3cdc5f565312224bc570c49337bd21428bba0ef363bbcf58b9ef4a9f11779968", size = 22424 }, ] +[[package]] +name = "posthog" +version = "7.28.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "backoff" }, + { name = "distro" }, + { name = "requests" }, + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/36/f0/3af875ac3fd5863ed4874c9618d85eab3f7fd24b395827d99da0ef90aca6/posthog-7.28.0.tar.gz", hash = "sha256:9e048dee58f27373db622c0744be30c1a2b7f1df31049956d6341c9646fb3833", size = 356360 } +wheels = [ + { url = "https://files.pythonhosted.org/packages/1c/8b/d41d98e64bd6ce650d45690cffc718274b1eda65e3bfd79575a1cf95b45c/posthog-7.28.0-py3-none-any.whl", hash = "sha256:4cff10062807bbd8ae6ec2804a71584831a75b390ce7ba3e736bf4cc0fb25d28", size = 425147 }, +] + [[package]] name = "preshed" version = "3.0.13" @@ -10869,7 +11157,7 @@ wheels = [ [[package]] name = "surf-new-backend" -version = "0.0.34" +version = "0.0.35" source = { editable = "." } dependencies = [ { name = "alembic" }, @@ -10926,6 +11214,7 @@ dependencies = [ { name = "opentelemetry-sdk" }, { name = "opentelemetry-semantic-conventions" }, { name = "pgvector" }, + { name = "posthog" }, { name = "psycopg", extra = ["binary", "pool"] }, { name = "pyarrow" }, { name = "pyjwt" }, @@ -11041,6 +11330,7 @@ requires-dist = [ { name = "opentelemetry-sdk", specifier = ">=1.40.0" }, { name = "opentelemetry-semantic-conventions", specifier = ">=0.61b0" }, { name = "pgvector", specifier = ">=0.3.6" }, + { name = "posthog", specifier = ">=6.0.0" }, { name = "psycopg", extras = ["binary", "pool"], specifier = ">=3.3.2" }, { name = "pyarrow", specifier = ">=15.0.0,<19.0.0" }, { name = "pyjwt", specifier = ">=2.12.0" }, diff --git a/surfsense_browser_extension/package.json b/surfsense_browser_extension/package.json index 02a5b7b95..f1207373c 100644 --- a/surfsense_browser_extension/package.json +++ b/surfsense_browser_extension/package.json @@ -1,7 +1,7 @@ { "name": "surfsense_browser_extension", "displayName": "Surfsense Browser Extension", - "version": "0.0.34", + "version": "0.0.35", "description": "Extension to collect Browsing History for SurfSense.", "author": "https://github.com/MODSetter", "engines": { diff --git a/surfsense_desktop/package.json b/surfsense_desktop/package.json index 3eb3484f9..20809509f 100644 --- a/surfsense_desktop/package.json +++ b/surfsense_desktop/package.json @@ -1,7 +1,7 @@ { "name": "surfsense-desktop", "productName": "SurfSense", - "version": "0.0.34", + "version": "0.0.35", "description": "SurfSense Desktop App", "main": "dist/main.js", "scripts": { diff --git a/surfsense_desktop/src/modules/window.ts b/surfsense_desktop/src/modules/window.ts index 3ab47fb58..a8886571e 100644 --- a/surfsense_desktop/src/modules/window.ts +++ b/surfsense_desktop/src/modules/window.ts @@ -65,6 +65,7 @@ export function createMainWindow(initialPath = '/dashboard'): BrowserWindow { }); mainWindow.once('ready-to-show', () => { + mainWindow?.maximize(); mainWindow?.show(); }); diff --git a/surfsense_mcp/mcp_server/core/client.py b/surfsense_mcp/mcp_server/core/client.py index 9ae446ae2..2bdf405dc 100644 --- a/surfsense_mcp/mcp_server/core/client.py +++ b/surfsense_mcp/mcp_server/core/client.py @@ -37,7 +37,11 @@ class SurfSenseClient: self._fallback_api_key = fallback_api_key self._http = httpx.AsyncClient( base_url=api_base, - headers={"Accept": "application/json"}, + # ``X-SurfSense-Client`` lets the backend distinguish PAT traffic + # originating from this MCP server vs. raw PAT scripts, so + # "documents added via MCP" / "searches via MCP" are queryable. + # Server-to-server, so no CORS implications. + headers={"Accept": "application/json", "X-SurfSense-Client": "mcp"}, timeout=timeout, ) @@ -61,13 +65,17 @@ class SurfSenseClient: json: Any | None = None, data: dict[str, Any] | None = None, files: Any | None = None, + headers: dict[str, str] | None = None, ) -> Any: - """Send a request and return the parsed body, or raise ``ToolError``.""" + """Send a request and return the parsed body, or raise ``ToolError``. + + ``headers`` overrides the client defaults for this call. + """ # Omit unset query params: sending them empty makes the API parse "" # as a value (e.g. int("") on folder_id) and fail. if params is not None: params = {key: value for key, value in params.items() if value is not None} - headers = self._auth_headers() + headers = {**self._auth_headers(), **(headers or {})} try: response = await self._http.request( method, diff --git a/surfsense_mcp/mcp_server/features/knowledge_base/search_tools.py b/surfsense_mcp/mcp_server/features/knowledge_base/search_tools.py index c0f7f83a9..dc085e00f 100644 --- a/surfsense_mcp/mcp_server/features/knowledge_base/search_tools.py +++ b/surfsense_mcp/mcp_server/features/knowledge_base/search_tools.py @@ -123,11 +123,19 @@ def register(mcp: FastMCP, client: SurfSenseClient, context: WorkspaceContext) - Use this after surfsense_search_knowledge_base or surfsense_list_documents to open a specific document — search results only include the matching passages, this returns the whole text. + The markdown form is an Open Knowledge Format (OKF) concept: a YAML + frontmatter block (type, title, tags, resource, timestamp) followed by + the document body. """ - document = await client.request("GET", f"/documents/{document_id}") if response_format == "json": + document = await client.request("GET", f"/documents/{document_id}") return clip(to_json(document)) - return _render_document(document) + concept = await client.request( + "GET", + f"/documents/{document_id}", + headers={"Accept": "text/markdown"}, + ) + return clip(concept if isinstance(concept, str) else str(concept)) def _join(values: list[str] | None) -> str | None: @@ -169,14 +177,3 @@ def _render_document_list(result: dict | None) -> str: + (" · more available_" if has_more else "_") ) return "\n".join(lines) - - -def _render_document(document: dict) -> str: - content = clip(document.get("content", "") or "(empty)") - return ( - f"# {document.get('title', 'Untitled')} (id {document.get('id')})\n" - f"- type: {document.get('document_type')}\n" - f"- workspace: {document.get('workspace_id')}\n" - f"- updated: {document.get('updated_at')}\n\n" - f"{content}" - ) diff --git a/surfsense_mcp/mcp_server/features/scrapers/__init__.py b/surfsense_mcp/mcp_server/features/scrapers/__init__.py index e6f1055cd..c74626d41 100644 --- a/surfsense_mcp/mcp_server/features/scrapers/__init__.py +++ b/surfsense_mcp/mcp_server/features/scrapers/__init__.py @@ -1,7 +1,8 @@ """Scraper tools: one MCP surface per SurfSense platform capability. -Web crawl, Google Search, Reddit, YouTube, and Google Maps each get a tool that -maps a natural-language request to the workspace's scraper. Two run-history tools +Web crawl, Google Search, Reddit, YouTube, Google Maps, Amazon, Indeed, and +Walmart each get a tool that maps a natural-language request to the workspace's +scraper. Two run-history tools list and fetch past runs, so a large result truncated inline can be retrieved in full later. Each platform lives in its own module under platforms/. """ @@ -17,9 +18,11 @@ from .platforms import ( amazon, google_maps, google_search, + indeed, instagram, reddit, tiktok, + walmart, web, youtube, ) @@ -32,7 +35,9 @@ _REGISTRARS = ( instagram, tiktok, google_maps, + indeed, amazon, + walmart, run_history, ) diff --git a/surfsense_mcp/mcp_server/features/scrapers/platforms/amazon.py b/surfsense_mcp/mcp_server/features/scrapers/platforms/amazon.py index 09b703157..15efa311c 100644 --- a/surfsense_mcp/mcp_server/features/scrapers/platforms/amazon.py +++ b/surfsense_mcp/mcp_server/features/scrapers/platforms/amazon.py @@ -93,8 +93,10 @@ def register(mcp: FastMCP, client: SurfSenseClient, context: WorkspaceContext) - ] = False, country_code: Annotated[ str | None, - Field(description="Two-letter delivery country for localized pricing, " - "e.g. 'us'."), + Field( + description="Two-letter delivery country for localized pricing, " + "e.g. 'us'." + ), ] = None, zip_code: Annotated[ str | None, diff --git a/surfsense_mcp/mcp_server/features/scrapers/platforms/indeed.py b/surfsense_mcp/mcp_server/features/scrapers/platforms/indeed.py new file mode 100644 index 000000000..e9323cd32 --- /dev/null +++ b/surfsense_mcp/mcp_server/features/scrapers/platforms/indeed.py @@ -0,0 +1,137 @@ +"""Indeed scraper tool.""" + +from __future__ import annotations + +from typing import Annotated, Literal + +from mcp.server.fastmcp import FastMCP +from pydantic import Field + +from ....core.client import SurfSenseClient +from ....core.rendering import ResponseFormatParam +from ....core.workspace_context import WorkspaceContext, WorkspaceParam +from ..annotations import SCRAPE +from ..capability import run_scraper + +IndeedSort = Literal["relevance", "date"] +IndeedJobType = Literal[ + "fulltime", + "parttime", + "contract", + "internship", + "temporary", + "permanent", + "seasonal", + "freelance", +] +IndeedLevel = Literal["entry_level", "mid_level", "senior_level"] +IndeedRemote = Literal["remote", "hybrid"] + + +def register(mcp: FastMCP, client: SurfSenseClient, context: WorkspaceContext) -> None: + """Register the Indeed tool.""" + + @mcp.tool( + name="surfsense_indeed_scrape", + title="Search or scrape Indeed jobs", + annotations=SCRAPE, + structured_output=False, + ) + async def indeed_scrape( + urls: Annotated[ + list[str] | None, + Field( + description="Indeed URLs: a search page " + "('https://www.indeed.com/jobs?q=data+analyst'), a company jobs " + "page ('/cmp//jobs'), or a single job ('/viewjob?jk=...'). " + "Provide urls OR search_queries." + ), + ] = None, + search_queries: Annotated[ + list[str] | None, + Field( + description="Job search terms, e.g. ['data analyst', 'ml engineer']. " + "Provide search_queries OR urls." + ), + ] = None, + country: Annotated[ + str, + Field( + description="Country code selecting the Indeed domain, e.g. 'us', 'gb'." + ), + ] = "us", + location: Annotated[ + str | None, + Field(description="Where to search, e.g. 'Remote', 'New York, NY'."), + ] = None, + radius: Annotated[ + int | None, + Field(description="Search radius in miles/km around location."), + ] = None, + job_type: Annotated[ + IndeedJobType | None, + Field(description="Employment type filter."), + ] = None, + level: Annotated[ + IndeedLevel | None, + Field(description="Experience level filter."), + ] = None, + remote: Annotated[ + IndeedRemote | None, + Field(description="Work model filter: remote or hybrid."), + ] = None, + from_days: Annotated[ + int | None, + Field(description="Only return jobs posted within the last N days."), + ] = None, + sort: Annotated[ + IndeedSort, Field(description="Result ordering: relevance or date.") + ] = "relevance", + scrape_job_details: Annotated[ + bool, + Field( + description="True fetches each job's detail page for the full " + "description (slower); False returns the listing snippet only." + ), + ] = False, + max_items: Annotated[ + int, Field(ge=1, description="Maximum jobs to return in total.") + ] = 25, + max_items_per_query: Annotated[ + int, Field(ge=0, description="Max jobs per search/company target.") + ] = 25, + workspace: WorkspaceParam = None, + response_format: ResponseFormatParam = "markdown", + ) -> str: + """Search or scrape public Indeed job postings. + + Use this for ANY Indeed job research — openings for a role, who is hiring + at a company, salaries for a title in a location, or remote roles — + instead of a generic web search. Returns jobs with title, company, + location, salary, job types, and description; set scrape_job_details for + the full description per job. + Example: search_queries=['data analyst'], location='Remote', max_items=30. + """ + return await run_scraper( + client, + context, + platform="indeed", + verb="scrape", + payload={ + "urls": urls, + "search_queries": search_queries, + "country": country, + "location": location, + "radius": radius, + "job_type": job_type, + "level": level, + "remote": remote, + "from_days": from_days, + "sort": sort, + "scrape_job_details": scrape_job_details, + "max_items": max_items, + "max_items_per_query": max_items_per_query, + }, + workspace=workspace, + response_format=response_format, + ) diff --git a/surfsense_mcp/mcp_server/features/scrapers/platforms/walmart.py b/surfsense_mcp/mcp_server/features/scrapers/platforms/walmart.py new file mode 100644 index 000000000..33a7d51e9 --- /dev/null +++ b/surfsense_mcp/mcp_server/features/scrapers/platforms/walmart.py @@ -0,0 +1,142 @@ +"""Walmart scraper tools: products/listings and deep reviews.""" + +from __future__ import annotations + +from typing import Annotated, Literal + +from mcp.server.fastmcp import FastMCP +from pydantic import Field + +from ....core.client import SurfSenseClient +from ....core.rendering import ResponseFormatParam +from ....core.workspace_context import WorkspaceContext, WorkspaceParam +from ..annotations import SCRAPE +from ..capability import run_scraper + +ReviewSort = Literal["most-recent", "most-helpful", "rating-high", "rating-low"] + + +def register(mcp: FastMCP, client: SurfSenseClient, context: WorkspaceContext) -> None: + """Register the Walmart product and review tools.""" + + @mcp.tool( + name="surfsense_walmart_scrape", + title="Scrape Walmart products", + annotations=SCRAPE, + structured_output=False, + ) + async def walmart_scrape( + urls: Annotated[ + list[str] | None, + Field( + description="Walmart product (/ip/), search (/search), category " + "(/cp/), or browse (/browse/) URLs. Provide urls OR search_terms." + ), + ] = None, + search_terms: Annotated[ + list[str] | None, + Field( + description="Search phrases run on walmart.com, e.g. ['air fryer']. " + "Provide search_terms OR urls." + ), + ] = None, + max_items: Annotated[ + int, + Field( + ge=1, le=100, description="Max products per search term or listing URL." + ), + ] = 10, + include_details: Annotated[ + bool, + Field( + description="Fetch full product detail pages. False returns faster " + "card-only results from listings." + ), + ] = True, + include_reviews_sample: Annotated[ + bool, + Field( + description="Include the free on-page review sample (rating " + "distribution, aspects, top reviews) on detail pages." + ), + ] = True, + workspace: WorkspaceParam = None, + response_format: ResponseFormatParam = "markdown", + ) -> str: + """Scrape public Walmart product data by URL or search term. + + Use this for product research: title, price, list price, rating and + review count, availability, seller (Walmart 1P vs marketplace), images, + description, variants, and a sample of on-page reviews. Only public, + anonymous data — no login. For a product's full review history use + surfsense_walmart_reviews instead. + Example: search_terms=['air fryer'], max_items=5. + """ + return await run_scraper( + client, + context, + platform="walmart", + verb="scrape", + payload={ + "urls": urls, + "search_terms": search_terms, + "max_items": max_items, + "include_details": include_details, + "include_reviews_sample": include_reviews_sample, + }, + workspace=workspace, + response_format=response_format, + ) + + @mcp.tool( + name="surfsense_walmart_reviews", + title="Fetch Walmart reviews", + annotations=SCRAPE, + structured_output=False, + ) + async def walmart_reviews( + urls: Annotated[ + list[str] | None, + Field( + description="Walmart product URLs (/ip/...). Provide urls OR item_ids." + ), + ] = None, + item_ids: Annotated[ + list[str] | None, + Field( + description="Walmart numeric item ids (usItemId) from " + "surfsense_walmart_scrape." + ), + ] = None, + max_reviews: Annotated[ + int, + Field(ge=1, le=5000, description="Max reviews per product (10 per page)."), + ] = 200, + sort_by: Annotated[ + ReviewSort, Field(description="Review ordering.") + ] = "most-recent", + workspace: WorkspaceParam = None, + response_format: ResponseFormatParam = "markdown", + ) -> str: + """Fetch deep paginated customer reviews for Walmart products. + + Use this to read the full review history on specific products; get urls + or item_ids from surfsense_walmart_scrape first if you only have a name. + Returns rating, title, text, author, verified-purchase flag, images, and + seller response per review. + Example: item_ids=['212092810'], sort_by='most-helpful', max_reviews=100. + """ + return await run_scraper( + client, + context, + platform="walmart", + verb="reviews", + payload={ + "urls": urls, + "item_ids": item_ids, + "max_reviews": max_reviews, + "sort_by": sort_by, + }, + workspace=workspace, + response_format=response_format, + ) diff --git a/surfsense_mcp/mcp_server/selfcheck.py b/surfsense_mcp/mcp_server/selfcheck.py index ffce29058..986caf9a5 100644 --- a/surfsense_mcp/mcp_server/selfcheck.py +++ b/surfsense_mcp/mcp_server/selfcheck.py @@ -32,6 +32,7 @@ EXPECTED_TOOLS = { "surfsense_amazon_scrape", "surfsense_instagram_scrape", "surfsense_instagram_details", + "surfsense_indeed_scrape", "surfsense_list_scraper_runs", "surfsense_get_scraper_run", # knowledge-base management diff --git a/surfsense_mcp/mcp_server/server.py b/surfsense_mcp/mcp_server/server.py index f801aeb15..ef5f15c8a 100644 --- a/surfsense_mcp/mcp_server/server.py +++ b/surfsense_mcp/mcp_server/server.py @@ -38,7 +38,8 @@ def build_server(settings: Settings) -> tuple[FastMCP, SurfSenseClient]: "task involves Reddit (posts, comments, finding subreddits or " "communities), YouTube (videos, transcripts, comments), Instagram " "(posts, reels, profile details), TikTok (videos by hashtag, " - "search, or URL), Google Maps (places, reviews), Google Search " + "search, or URL), Google Maps (places, reviews), Indeed (job " + "postings by role, company, or location), Google Search " "results, or reading " "specific web pages. Scraper results are persisted as runs; if an " "inline result is truncated, fetch it in full with " diff --git a/surfsense_mcp/tests/test_get_document_okf.py b/surfsense_mcp/tests/test_get_document_okf.py new file mode 100644 index 000000000..8b5cdae2f --- /dev/null +++ b/surfsense_mcp/tests/test_get_document_okf.py @@ -0,0 +1,68 @@ +"""surfsense_get_document round-trips through the real tool registration. + +The markdown form must ask the backend for the OKF concept via content +negotiation (``Accept: text/markdown``) and pass it through untouched; the JSON +form must leave the default ``application/json`` Accept in place. This is the +only coverage of the MCP-side glue that forwards the header. +""" + +from __future__ import annotations + +import asyncio +from unittest.mock import MagicMock + +import httpx +from mcp.server.fastmcp import FastMCP + +from mcp_server.core.client import SurfSenseClient +from mcp_server.features.knowledge_base import search_tools + +_CONCEPT = "---\ntype: Note\ntitle: T\n---\n\nBody." + + +def _client_recording(seen: dict) -> SurfSenseClient: + async def handler(request: httpx.Request) -> httpx.Response: + seen["path"] = request.url.path + seen["accept"] = request.headers.get("accept") + if "text/markdown" in (seen["accept"] or ""): + return httpx.Response( + 200, text=_CONCEPT, headers={"content-type": "text/markdown"} + ) + return httpx.Response(200, json={"id": 1, "title": "T"}) + + client = SurfSenseClient( + api_base="http://test/api/v1", timeout=5, fallback_api_key="ss_pat_x" + ) + client._http = httpx.AsyncClient( + base_url="http://test/api/v1", + headers={"Accept": "application/json"}, + transport=httpx.MockTransport(handler), + ) + return client + + +def _call_get_document(client: SurfSenseClient, **arguments) -> str: + mcp = FastMCP("test") + search_tools.register(mcp, client, MagicMock()) + blocks = asyncio.run(mcp.call_tool("surfsense_get_document", arguments)) + return "".join(block.text for block in blocks) + + +def test_markdown_requests_okf_concept_and_passes_it_through(): + seen: dict = {} + text = _call_get_document(_client_recording(seen), document_id=1) + + assert seen["path"] == "/api/v1/documents/1" + assert "text/markdown" in seen["accept"] + assert text == _CONCEPT + + +def test_json_keeps_default_accept(): + seen: dict = {} + text = _call_get_document( + _client_recording(seen), document_id=1, response_format="json" + ) + + assert seen["path"] == "/api/v1/documents/1" + assert seen["accept"] == "application/json" + assert '"id": 1' in text diff --git a/surfsense_web/app/(home)/login/LocalLoginForm.tsx b/surfsense_web/app/(home)/login/LocalLoginForm.tsx index dd415e10f..e60463b9d 100644 --- a/surfsense_web/app/(home)/login/LocalLoginForm.tsx +++ b/surfsense_web/app/(home)/login/LocalLoginForm.tsx @@ -13,7 +13,7 @@ import { Spinner } from "@/components/ui/spinner"; import { getAuthErrorDetails, isNetworkError } from "@/lib/auth-errors"; import { getPostLoginRedirectPath } from "@/lib/auth-utils"; import { ValidationError } from "@/lib/error"; -import { trackLoginAttempt, trackLoginFailure, trackLoginSuccess } from "@/lib/posthog/events"; +import { trackLoginAttempt, trackLoginFailure } from "@/lib/posthog/events"; export function LocalLoginForm() { const t = useTranslations("auth"); @@ -45,8 +45,8 @@ export function LocalLoginForm() { grant_type: "password", }); - // Track successful login - trackLoginSuccess("local"); + // auth_login_success is now emitted server-side + // (UserManager.on_after_login) — authoritative vs. optimistic. // Small delay to show success message setTimeout(() => { diff --git a/surfsense_web/app/(home)/mcp-server/page.tsx b/surfsense_web/app/(home)/mcp-server/page.tsx index ab7769ee9..439e1ae6f 100644 --- a/surfsense_web/app/(home)/mcp-server/page.tsx +++ b/surfsense_web/app/(home)/mcp-server/page.tsx @@ -16,7 +16,7 @@ import type { FaqItem } from "@/lib/connectors-marketing/types"; const canonicalUrl = "https://www.surfsense.com/mcp-server"; const metaDescription = - "The SurfSense MCP server gives Claude, Cursor, and any MCP client native tools for your workspace: scrape Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, and the web, plus full knowledge base access. One API key."; + "The SurfSense MCP server gives Claude, Cursor, and any MCP client native tools for your workspace: scrape Reddit, YouTube, Instagram, TikTok, Amazon, Walmart, Google Maps, Google Search, and the web, plus full knowledge base access. One API key."; export const metadata: Metadata = { title: "SurfSense MCP Server: Scraper APIs and Knowledge Base as Agent Tools", @@ -103,6 +103,8 @@ const TOOL_GROUPS = [ "surfsense_google_maps_reviews", "surfsense_google_search", "surfsense_amazon_scrape", + "surfsense_walmart_scrape", + "surfsense_walmart_reviews", "surfsense_web_crawl", "surfsense_list_scraper_runs", "surfsense_get_scraper_run", @@ -134,7 +136,7 @@ const FAQ: FaqItem[] = [ { question: "What is the SurfSense MCP server?", answer: - "It is a Model Context Protocol server that exposes your SurfSense workspace to MCP clients like Claude Code, Cursor, and Claude Desktop. Your agents get native tools for every scraper API (Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, web crawl) and for searching, reading, and writing your knowledge base.", + "It is a Model Context Protocol server that exposes your SurfSense workspace to MCP clients like Claude Code, Cursor, and Claude Desktop. Your agents get native tools for every scraper API (Reddit, YouTube, Instagram, TikTok, Amazon, Walmart, Google Maps, Google Search, web crawl) and for searching, reading, and writing your knowledge base.", }, { question: "Which MCP clients does it work with?", @@ -222,9 +224,9 @@ export default function McpServerPage() {

The SurfSense MCP server hands Claude, Cursor, or any MCP client the whole platform: - scrape Reddit, YouTube, Instagram, TikTok, Amazon, Google Maps, Google Search, and - the open web, and search, read, and write your knowledge base. One API key, typed - tools, pay as you go. + scrape Reddit, YouTube, Instagram, TikTok, Amazon, Walmart, Google Maps, Google + Search, and the open web, and search, read, and write your knowledge base. One API + key, typed tools, pay as you go.

+ {isOpen && } +
+ ); + })}
)} diff --git a/surfsense_web/app/dashboard/[workspace_id]/playground/components/schema-form.tsx b/surfsense_web/app/dashboard/[workspace_id]/playground/components/schema-form.tsx index cdd425925..b8eb72c66 100644 --- a/surfsense_web/app/dashboard/[workspace_id]/playground/components/schema-form.tsx +++ b/surfsense_web/app/dashboard/[workspace_id]/playground/components/schema-form.tsx @@ -1,7 +1,7 @@ "use client"; import { ChevronDown } from "lucide-react"; -import { useMemo, useState } from "react"; +import { useEffect, useMemo, useState } from "react"; import { Badge } from "@/components/ui/badge"; import { Input } from "@/components/ui/input"; import { Label } from "@/components/ui/label"; @@ -32,6 +32,8 @@ interface SchemaFormProps { getFieldOptions?: FieldOptionsResolver; /** Field names flagged by a 422 response, shown with error styling. */ fieldErrors?: Record; + /** Client-side, non-blocking URL warnings keyed by field name. */ + fieldWarnings?: Record; } function FieldControl({ @@ -147,6 +149,7 @@ function FieldRow({ onChange, disabled, error, + warning, options, }: { field: FormField; @@ -154,6 +157,7 @@ function FieldRow({ onChange: (value: unknown) => void; disabled?: boolean; error?: string; + warning?: string; options?: FieldOption[]; }) { return ( @@ -179,6 +183,7 @@ function FieldRow({ options={options} /> {error &&

{error}

} + {!error && warning &&

{warning}

}
); } @@ -190,6 +195,7 @@ export function SchemaForm({ disabled, getFieldOptions, fieldErrors, + fieldWarnings, }: SchemaFormProps) { const [showAdvanced, setShowAdvanced] = useState(false); @@ -199,6 +205,25 @@ export function SchemaForm({ return { primary: primaryFields, advanced: advancedFields }; }, [fields]); + // First invalid field in display order; drives reveal + focus below. + const firstErrorName = useMemo( + () => (fieldErrors ? fields.find((f) => fieldErrors[f.name])?.name : undefined), + [fields, fieldErrors] + ); + + // Reveal the section holding an invalid field, then move focus to it, so an + // error is never left hidden inside the collapsed "Advanced" group. + useEffect(() => { + if (!firstErrorName) return; + if (advanced.some((f) => f.name === firstErrorName)) setShowAdvanced(true); + const raf = requestAnimationFrame(() => { + const el = document.getElementById(`field-${firstErrorName}`); + el?.focus(); + el?.scrollIntoView({ block: "center", behavior: "smooth" }); + }); + return () => cancelAnimationFrame(raf); + }, [firstErrorName, advanced]); + return (
{primary.map((field) => ( @@ -209,6 +234,7 @@ export function SchemaForm({ onChange={(value) => onChange(field.name, value)} disabled={disabled} error={fieldErrors?.[field.name]} + warning={fieldWarnings?.[field.name]} options={getFieldOptions?.(field)} /> ))} @@ -239,6 +265,7 @@ export function SchemaForm({ onChange={(value) => onChange(field.name, value)} disabled={disabled} error={fieldErrors?.[field.name]} + warning={fieldWarnings?.[field.name]} options={getFieldOptions?.(field)} /> ))} diff --git a/surfsense_web/app/dashboard/[workspace_id]/team/team-content.tsx b/surfsense_web/app/dashboard/[workspace_id]/team/team-content.tsx index 571512305..dc45462e8 100644 --- a/surfsense_web/app/dashboard/[workspace_id]/team/team-content.tsx +++ b/surfsense_web/app/dashboard/[workspace_id]/team/team-content.tsx @@ -96,7 +96,6 @@ import type { Role } from "@/contracts/types/roles.types"; import { invitesApiService } from "@/lib/apis/invites-api.service"; import { rolesApiService } from "@/lib/apis/roles-api.service"; import { formatRelativeDate } from "@/lib/format-date"; -import { trackWorkspaceInviteSent, trackWorkspaceUsersViewed } from "@/lib/posthog/events"; import { cacheKeys } from "@/lib/query-client/cache-keys"; import { cn } from "@/lib/utils"; @@ -226,12 +225,7 @@ export function TeamContent({ workspaceId }: TeamContentProps) { const canPrev = pageIndex > 0; const canNext = displayEnd < totalItems; - useEffect(() => { - if (members.length > 0 && !membersLoading) { - const ownerCount = members.filter((m) => m.is_owner).length; - trackWorkspaceUsersViewed(workspaceId, members.length, ownerCount); - } - }, [members, membersLoading, workspaceId]); + // workspace_users_viewed removed — redundant with $pageview. if (accessLoading || membersLoading) { return ( @@ -342,11 +336,7 @@ export function TeamContent({ workspaceId }: TeamContentProps) { Invite members ) : ( - + )} {invitesLoading ? ( - - Go to dashboard home - - - - Report Issue - + +
); diff --git a/surfsense_web/app/globals.css b/surfsense_web/app/globals.css index 4a29edfa6..72a3bbfef 100644 --- a/surfsense_web/app/globals.css +++ b/surfsense_web/app/globals.css @@ -54,7 +54,7 @@ --sidebar-ring: oklch(0.708 0 0); --main-panel: var(--panel); --syntax-bg: #f5f5f5; - --brand: oklch(0.623 0.214 259.815); + --brand: oklch(0.546 0.245 262.881); --highlight: oklch(0.852 0.199 91.936); } @@ -103,7 +103,7 @@ html[data-surfsense-auth-type="LOCAL"] .runtime-auth-google { --sidebar-ring: oklch(0.439 0 0); --main-panel: var(--panel); --syntax-bg: #1e1e1e; - --brand: oklch(0.707 0.165 254.624); + --brand: oklch(0.546 0.245 262.881); --highlight: oklch(0.852 0.199 91.936); } diff --git a/surfsense_web/app/invite/[invite_code]/page.tsx b/surfsense_web/app/invite/[invite_code]/page.tsx index 2384ca422..4d61b742c 100644 --- a/surfsense_web/app/invite/[invite_code]/page.tsx +++ b/surfsense_web/app/invite/[invite_code]/page.tsx @@ -16,7 +16,7 @@ import { motion } from "motion/react"; import Image from "next/image"; import Link from "next/link"; import { useParams, useRouter } from "next/navigation"; -import { use, useCallback, useEffect, useState } from "react"; +import { useCallback, useEffect, useState } from "react"; import { toast } from "sonner"; import { acceptInviteMutationAtom } from "@/atoms/invites/invites-mutation.atoms"; import { Button } from "@/components/ui/button"; @@ -33,11 +33,7 @@ import type { AcceptInviteResponse } from "@/contracts/types/invites.types"; import { useSession } from "@/hooks/use-session"; import { invitesApiService } from "@/lib/apis/invites-api.service"; import { setRedirectPath } from "@/lib/auth-utils"; -import { - trackWorkspaceInviteAccepted, - trackWorkspaceInviteDeclined, - trackWorkspaceUserAdded, -} from "@/lib/posthog/events"; +import { trackWorkspaceInviteDeclined } from "@/lib/posthog/events"; import { cacheKeys } from "@/lib/query-client/cache-keys"; export default function InviteAcceptPage() { @@ -96,9 +92,9 @@ export default function InviteAcceptPage() { setAccepted(true); setAcceptedData(result); - // Track invite accepted and user added events - trackWorkspaceInviteAccepted(result.workspace_id, result.workspace_name, result.role_name); - trackWorkspaceUserAdded(result.workspace_id, result.workspace_name, result.role_name); + // workspace_invite_accepted + workspace_user_added are now emitted + // server-side (rbac_routes.accept_invite) — the server redirect is + // the authoritative join point. } } catch (err: any) { setError(err.message || "Failed to accept invite"); diff --git a/surfsense_web/atoms/automations/automations-mutation.atoms.ts b/surfsense_web/atoms/automations/automations-mutation.atoms.ts index f96b2a252..8e8ddbb16 100644 --- a/surfsense_web/atoms/automations/automations-mutation.atoms.ts +++ b/surfsense_web/atoms/automations/automations-mutation.atoms.ts @@ -8,18 +8,11 @@ import type { } from "@/contracts/types/automation.types"; import { automationsApiService } from "@/lib/apis/automations-api.service"; import { - trackAutomationCreated, trackAutomationCreateFailed, - trackAutomationDeleted, trackAutomationDeleteFailed, - trackAutomationStatusChanged, - trackAutomationTriggerAdded, trackAutomationTriggerAddFailed, - trackAutomationTriggerRemoved, trackAutomationTriggerRemoveFailed, - trackAutomationTriggerUpdated, trackAutomationTriggerUpdateFailed, - trackAutomationUpdated, trackAutomationUpdateFailed, } from "@/lib/posthog/events"; import { cacheKeys } from "@/lib/query-client/cache-keys"; @@ -48,20 +41,10 @@ export const createAutomationMutationAtom = atomWithMutation(() => ({ mutationFn: async (request: AutomationCreateRequest) => { return automationsApiService.createAutomation(request); }, - onSuccess: (automation, variables) => { + onSuccess: (_automation, variables) => { invalidateList(variables.workspace_id); toast.success("Automation created"); - trackAutomationCreated({ - workspace_id: variables.workspace_id, - automation_id: automation.id, - task_count: variables.definition.plan.length, - trigger_type: variables.triggers?.[0]?.type ?? "none", - has_schedule: (variables.triggers?.length ?? 0) > 0, - chat_model_id: variables.definition.models?.chat_model_id, - image_gen_model_id: variables.definition.models?.image_gen_model_id, - vision_model_id: variables.definition.models?.vision_model_id, - tags_count: variables.definition.metadata?.tags?.length, - }); + // automation_created is now emitted server-side (AutomationService.create). }, onError: (error: Error, variables) => { console.error("Error creating automation:", error); @@ -82,24 +65,8 @@ export const updateAutomationMutationAtom = atomWithMutation(() => ({ invalidateDetail(vars.automationId); invalidateList(automation.workspace_id); toast.success("Automation updated"); - // A status-only patch (pause/resume/archive) is a distinct action from a - // definition/name edit, so split it into its own event. - if (vars.patch.status && !vars.patch.definition) { - trackAutomationStatusChanged({ - automation_id: vars.automationId, - workspace_id: automation.workspace_id, - next_status: vars.patch.status, - }); - } else { - trackAutomationUpdated({ - automation_id: vars.automationId, - workspace_id: automation.workspace_id, - has_definition_change: !!vars.patch.definition, - has_name_change: vars.patch.name != null, - has_description_change: vars.patch.description !== undefined, - task_count: vars.patch.definition?.plan?.length, - }); - } + // automation_updated / automation_status_changed are now emitted + // server-side (AutomationService.update). }, onError: (error: Error, vars) => { console.error("Error updating automation:", error); @@ -121,10 +88,7 @@ export const deleteAutomationMutationAtom = atomWithMutation(() => ({ invalidateList(vars.workspaceId); invalidateDetail(vars.automationId); toast.success("Automation deleted"); - trackAutomationDeleted({ - automation_id: vars.automationId, - workspace_id: vars.workspaceId, - }); + // automation_deleted is now emitted server-side (AutomationService.delete). }, onError: (error: Error, vars) => { console.error("Error deleting automation:", error); @@ -141,16 +105,10 @@ export const addTriggerMutationAtom = atomWithMutation(() => ({ mutationFn: async (vars: { automationId: number; payload: TriggerCreateRequest }) => { return automationsApiService.addTrigger(vars.automationId, vars.payload); }, - onSuccess: (trigger, vars) => { + onSuccess: (_trigger, vars) => { invalidateDetail(vars.automationId); toast.success("Trigger added"); - trackAutomationTriggerAdded({ - automation_id: vars.automationId, - trigger_id: trigger.id, - trigger_type: trigger.type, - enabled: trigger.enabled, - has_cron: !!trigger.params?.cron, - }); + // automation_trigger_added is now emitted server-side (TriggerService.add). }, onError: (error: Error, vars) => { console.error("Error adding trigger:", error); @@ -174,17 +132,7 @@ export const updateTriggerMutationAtom = atomWithMutation(() => ({ onSuccess: (_, vars) => { invalidateDetail(vars.automationId); toast.success("Trigger updated"); - const change: "enabled" | "params" | "other" = vars.patch.params - ? "params" - : vars.patch.enabled !== undefined && vars.patch.enabled !== null - ? "enabled" - : "other"; - trackAutomationTriggerUpdated({ - automation_id: vars.automationId, - trigger_id: vars.triggerId, - change, - enabled: vars.patch.enabled ?? undefined, - }); + // automation_trigger_updated is now emitted server-side (TriggerService.update). }, onError: (error: Error, vars) => { console.error("Error updating trigger:", error); @@ -206,10 +154,7 @@ export const removeTriggerMutationAtom = atomWithMutation(() => ({ onSuccess: (vars) => { invalidateDetail(vars.automationId); toast.success("Trigger removed"); - trackAutomationTriggerRemoved({ - automation_id: vars.automationId, - trigger_id: vars.triggerId, - }); + // automation_trigger_removed is now emitted server-side (TriggerService.remove). }, onError: (error: Error, vars) => { console.error("Error removing trigger:", error); diff --git a/surfsense_web/atoms/citation/citation-panel.atom.ts b/surfsense_web/atoms/citation/citation-panel.atom.ts index ca7312857..e07131378 100644 --- a/surfsense_web/atoms/citation/citation-panel.atom.ts +++ b/surfsense_web/atoms/citation/citation-panel.atom.ts @@ -1,14 +1,17 @@ -import { atom } from "jotai"; +import { atom, type Getter, type Setter } from "jotai"; import { rightPanelCollapsedAtom, rightPanelTabAtom } from "@/atoms/layout/right-panel.atom"; +/** The source the citation panel is showing: a KB chunk or a scraper run. */ +export type CitationTarget = { kind: "chunk"; chunkId: number } | { kind: "run"; runId: string }; + interface CitationPanelState { isOpen: boolean; - chunkId: number | null; + target: CitationTarget | null; } const initialState: CitationPanelState = { isOpen: false, - chunkId: null, + target: null, }; export const citationPanelAtom = atom(initialState); @@ -17,16 +20,21 @@ export const citationPanelOpenAtom = atom((get) => get(citationPanelAtom).isOpen const preCitationCollapsedAtom = atom(null); -export const openCitationPanelAtom = atom(null, (get, set, payload: { chunkId: number }) => { +function openWithTarget(get: Getter, set: Setter, target: CitationTarget) { if (!get(citationPanelAtom).isOpen) { set(preCitationCollapsedAtom, get(rightPanelCollapsedAtom)); } - set(citationPanelAtom, { - isOpen: true, - chunkId: payload.chunkId, - }); + set(citationPanelAtom, { isOpen: true, target }); set(rightPanelTabAtom, "citation"); set(rightPanelCollapsedAtom, false); +} + +export const openCitationPanelAtom = atom(null, (get, set, payload: { chunkId: number }) => { + openWithTarget(get, set, { kind: "chunk", chunkId: payload.chunkId }); +}); + +export const openRunCitationPanelAtom = atom(null, (get, set, payload: { runId: string }) => { + openWithTarget(get, set, { kind: "run", runId: payload.runId }); }); export const closeCitationPanelAtom = atom(null, (get, set) => { diff --git a/surfsense_web/atoms/documents/folder.atoms.ts b/surfsense_web/atoms/documents/folder.atoms.ts index 2d381557b..161627d5c 100644 --- a/surfsense_web/atoms/documents/folder.atoms.ts +++ b/surfsense_web/atoms/documents/folder.atoms.ts @@ -26,3 +26,10 @@ export const localExpandedFolderKeysAtom = atomWithStorage(null); + +/** + * Bumped whenever a folder is (un)watched outside DocumentsSidebar (e.g. the + * sidebar-footer "Watch Local Folder" button) so the tree re-reads Electron's + * watched-folder list and shows the sync badge without a reload. + */ +export const watchedFoldersRefreshAtom = atom(0); diff --git a/surfsense_web/atoms/model-connections/model-connections-mutation.atoms.ts b/surfsense_web/atoms/model-connections/model-connections-mutation.atoms.ts index 5e7e87b04..2b82c8d02 100644 --- a/surfsense_web/atoms/model-connections/model-connections-mutation.atoms.ts +++ b/surfsense_web/atoms/model-connections/model-connections-mutation.atoms.ts @@ -4,7 +4,6 @@ import type { ConnectionCreateRequest, ConnectionRead, ConnectionUpdateRequest, - ModelCreateRequest, ModelPreviewRead, ModelRead, ModelRoles, @@ -117,13 +116,7 @@ export const verifyModelConnectionMutationAtom = atomWithMutation((get) => { if (result.ok) { toast.success("Connection verified"); } else { - // Non-fatal: many providers lack a /models endpoint yet still serve - // chat. Guide the user to add model IDs manually instead of alarming. - toast.warning( - result.message - ? `${result.message} Chat may still work — add model IDs manually.` - : "Couldn't list models. Chat may still work — add model IDs manually." - ); + toast.warning(result.message || "Couldn't verify this connection."); } invalidateModelConnections(workspaceId); }, @@ -170,20 +163,6 @@ export const testPreviewModelMutationAtom = atomWithMutation(() => { }; }); -export const addManualModelMutationAtom = atomWithMutation((get) => { - const workspaceId = Number(get(activeWorkspaceIdAtom)); - return { - mutationKey: ["models", "add-manual"], - mutationFn: ({ connectionId, data }: { connectionId: number; data: ModelCreateRequest }) => - modelConnectionsApiService.addManualModel(connectionId, data), - onSuccess: () => { - toast.success("Model added"); - invalidateModelConnections(workspaceId); - }, - onError: (error: Error) => toast.error(error.message || "Failed to add model"), - }; -}); - export const updateModelMutationAtom = atomWithMutation((get) => { const workspaceId = Number(get(activeWorkspaceIdAtom)); return { diff --git a/surfsense_web/atoms/tabs/migrate-tabs.test.ts b/surfsense_web/atoms/tabs/migrate-tabs.test.ts deleted file mode 100644 index e0067a4d2..000000000 --- a/surfsense_web/atoms/tabs/migrate-tabs.test.ts +++ /dev/null @@ -1,21 +0,0 @@ -import assert from "node:assert/strict"; -import { test } from "node:test"; -import { migrateLegacyTabs } from "./migrate-tabs"; - -// Run with: pnpm exec tsx --test atoms/tabs/migrate-tabs.test.ts -test("maps legacy searchSpaceId to workspaceId on read", () => { - const migrated = migrateLegacyTabs({ - tabs: [{ id: "chat-new", type: "chat", searchSpaceId: 7 } as never], - activeTabId: "chat-new", - }); - const tab = migrated.tabs[0] as { workspaceId?: number }; - assert.equal(tab.workspaceId, 7); -}); - -test("leaves an already-migrated workspaceId untouched", () => { - const migrated = migrateLegacyTabs({ - tabs: [{ id: "d1", type: "document", workspaceId: 3, searchSpaceId: 9 } as never], - }); - const tab = migrated.tabs[0] as { workspaceId?: number }; - assert.equal(tab.workspaceId, 3); -}); diff --git a/surfsense_web/atoms/tabs/migrate-tabs.ts b/surfsense_web/atoms/tabs/migrate-tabs.ts deleted file mode 100644 index e95a921de..000000000 --- a/surfsense_web/atoms/tabs/migrate-tabs.ts +++ /dev/null @@ -1,19 +0,0 @@ -/** - * One-time read-migration for persisted tabs: legacy state stored the workspace - * as `searchSpaceId`. Map it to `workspaceId` on read so already-open tabs keep - * their workspace association after the rename. Pure + dependency-free so it can - * be unit-checked without loading the atom module. - */ -export function migrateLegacyTabs }>( - state: T -): T { - return { - ...state, - tabs: state.tabs.map((t) => { - const legacy = t as { workspaceId?: number; searchSpaceId?: number }; - return legacy.workspaceId === undefined && legacy.searchSpaceId !== undefined - ? { ...t, workspaceId: legacy.searchSpaceId } - : t; - }), - }; -} diff --git a/surfsense_web/atoms/tabs/tabs.atom.ts b/surfsense_web/atoms/tabs/tabs.atom.ts index 45e0c098c..b709bcd50 100644 --- a/surfsense_web/atoms/tabs/tabs.atom.ts +++ b/surfsense_web/atoms/tabs/tabs.atom.ts @@ -1,22 +1,13 @@ import { atom } from "jotai"; import { atomWithStorage, createJSONStorage } from "jotai/utils"; -import type { ChatVisibility } from "@/lib/chat/thread-persistence"; -import { migrateLegacyTabs } from "./migrate-tabs"; export type TabType = "chat" | "document"; export interface Tab { id: string; type: TabType; - title: string; - /** For chat tabs */ - chatId?: number | null; - chatUrl?: string; - visibility?: ChatVisibility; - hasComments?: boolean; - /** For document tabs */ - documentId?: number; - workspaceId?: number; + entityId: number | null; + workspaceId: number; } interface TabsState { @@ -27,9 +18,8 @@ interface TabsState { const INITIAL_CHAT_TAB: Tab = { id: "chat-new", type: "chat", - title: "New Chat", - chatId: null, - chatUrl: undefined, + entityId: null, + workspaceId: 0, }; const initialState: TabsState = { @@ -37,23 +27,14 @@ const initialState: TabsState = { activeTabId: "chat-new", }; -// Prevent race conditions where route-sync recreates a just-deleted chat tab. -const deletedChatIdsAtom = atom>(new Set()); - // Persist tabs in localStorage so they survive a hard refresh and let the user // keep tabs open across multiple workspaces (browser-like behavior). const localStorageAdapter = createJSONStorage( () => (typeof window !== "undefined" ? localStorage : undefined) as Storage ); -// Wrap getItem in place so the adapter keeps its original (sync) type while -// migrating legacy persisted state on read. -const baseGetItem = localStorageAdapter.getItem.bind(localStorageAdapter); -localStorageAdapter.getItem = (key, initialValue) => - migrateLegacyTabs(baseGetItem(key, initialValue)); - export const tabsStateAtom = atomWithStorage( - "surfsense:tabs", + "surfsense:tabs:v2", initialState, localStorageAdapter, { getOnInit: true } @@ -66,11 +47,11 @@ export const activeTabAtom = atom((get) => { return state.tabs.find((t) => t.id === state.activeTabId) ?? null; }); -function makeChatTabId(chatId: number | null): string { +export function makeChatTabId(chatId: number | null): string { return chatId ? `chat-${chatId}` : "chat-new"; } -function makeDocumentTabId(documentId: number): string { +export function makeDocumentTabId(documentId: number): string { return `doc-${documentId}`; } @@ -86,24 +67,12 @@ export const syncChatTabAtom = atom( set, { chatId, - title, - chatUrl, workspaceId, - visibility, - hasComments, }: { chatId: number | null; - title?: string; - chatUrl?: string; workspaceId: number; - visibility?: ChatVisibility; - hasComments?: boolean; } ) => { - if (chatId && get(deletedChatIdsAtom).has(chatId)) { - return; - } - const state = get(tabsStateAtom); const tabId = makeChatTabId(chatId); const existing = state.tabs.find((t) => t.id === tabId); @@ -116,11 +85,8 @@ export const syncChatTabAtom = atom( t.id === tabId ? { ...t, - title: title || t.title, - chatUrl: chatUrl || t.chatUrl, - workspaceId: workspaceId ?? t.workspaceId, - ...(visibility !== undefined ? { visibility } : {}), - ...(hasComments !== undefined ? { hasComments } : {}), + entityId: chatId, + workspaceId, } : t ), @@ -136,11 +102,13 @@ export const syncChatTabAtom = atom( set(tabsStateAtom, { ...state, activeTabId: "chat-new", - tabs: state.tabs.map((t) => (t.id === "chat-new" ? { ...t, workspaceId, chatUrl } : t)), + tabs: state.tabs.map((t) => + t.id === "chat-new" ? { ...t, entityId: null, workspaceId } : t + ), }); } else { set(tabsStateAtom, { - tabs: [...state.tabs, { ...INITIAL_CHAT_TAB, workspaceId, chatUrl }], + tabs: [...state.tabs, { ...INITIAL_CHAT_TAB, workspaceId }], activeTabId: "chat-new", }); } @@ -152,12 +120,8 @@ export const syncChatTabAtom = atom( const newTab: Tab = { id: tabId, type: "chat", - title: title || "New Chat", - chatId, - chatUrl, + entityId: chatId, workspaceId, - ...(visibility !== undefined ? { visibility } : {}), - ...(hasComments !== undefined ? { hasComments } : {}), }; let updatedTabs: Tab[]; @@ -172,29 +136,26 @@ export const syncChatTabAtom = atom( } ); -/** Update the title of the current chat tab (e.g., when a chat gets its first response). */ +/** Promote the lazy "new chat" tab once the server creates its thread. */ export const updateChatTabTitleAtom = atom( null, - (get, set, { chatId, title }: { chatId: number; title: string }) => { + (get, set, { chatId }: { chatId: number; title?: string }) => { const state = get(tabsStateAtom); const tabId = makeChatTabId(chatId); const hasExactTab = state.tabs.some((t) => t.id === tabId); - // During lazy thread creation, title updates can arrive before "chat-new" - // is swapped to chat-{id}. In that case, promote the active "chat-new" tab. + // During lazy thread creation, title updates can arrive before "chat-new" is + // swapped to chat-{id}. In that case, promote the pointer only; title comes + // from Zero. if (!hasExactTab && state.activeTabId === "chat-new") { set(tabsStateAtom, { ...state, activeTabId: tabId, - tabs: state.tabs.map((t) => (t.id === "chat-new" ? { ...t, id: tabId, chatId, title } : t)), + tabs: state.tabs.map((t) => + t.id === "chat-new" ? { ...t, id: tabId, entityId: chatId } : t + ), }); - return; } - - set(tabsStateAtom, { - ...state, - tabs: state.tabs.map((t) => (t.id === tabId ? { ...t, title } : t)), - }); } ); @@ -204,7 +165,7 @@ export const openDocumentTabAtom = atom( ( get, set, - { documentId, workspaceId, title }: { documentId: number; workspaceId: number; title?: string } + { documentId, workspaceId }: { documentId: number; workspaceId: number; title?: string } ) => { const state = get(tabsStateAtom); const tabId = makeDocumentTabId(documentId); @@ -214,7 +175,7 @@ export const openDocumentTabAtom = atom( set(tabsStateAtom, { ...state, activeTabId: tabId, - tabs: state.tabs.map((t) => (t.id === tabId ? { ...t, title: title || t.title } : t)), + tabs: state.tabs.map((t) => (t.id === tabId ? { ...t, workspaceId } : t)), }); return; } @@ -222,8 +183,7 @@ export const openDocumentTabAtom = atom( const newTab: Tab = { id: tabId, type: "document", - title: title || `Document ${documentId}`, - documentId, + entityId: documentId, workspaceId, }; @@ -254,11 +214,12 @@ export const closeTabAtom = atom(null, (get, set, tabId: string) => { // Don't close the last tab — always keep at least one if (remaining.length === 0) { + const closedTab = state.tabs[idx]; set(tabsStateAtom, { - tabs: [INITIAL_CHAT_TAB], + tabs: [{ ...INITIAL_CHAT_TAB, workspaceId: closedTab.workspaceId }], activeTabId: "chat-new", }); - return INITIAL_CHAT_TAB; + return { ...INITIAL_CHAT_TAB, workspaceId: closedTab.workspaceId }; } let newActiveId = state.activeTabId; @@ -279,18 +240,16 @@ export const removeChatTabAtom = atom(null, (get, set, chatId: number) => { const idx = state.tabs.findIndex((t) => t.id === tabId); if (idx === -1) return null; - const deletedChatIds = get(deletedChatIdsAtom); - set(deletedChatIdsAtom, new Set([...deletedChatIds, chatId])); - const remaining = state.tabs.filter((t) => t.id !== tabId); // Always keep at least one tab available. if (remaining.length === 0) { + const removedTab = state.tabs[idx]; set(tabsStateAtom, { - tabs: [INITIAL_CHAT_TAB], + tabs: [{ ...INITIAL_CHAT_TAB, workspaceId: removedTab.workspaceId }], activeTabId: "chat-new", }); - return INITIAL_CHAT_TAB; + return { ...INITIAL_CHAT_TAB, workspaceId: removedTab.workspaceId }; } let newActiveId = state.activeTabId; @@ -303,8 +262,38 @@ export const removeChatTabAtom = atom(null, (get, set, chatId: number) => { return remaining.find((t) => t.id === newActiveId) ?? null; }); -/** Reset tabs when switching workspaces. */ -export const resetTabsAtom = atom(null, (_get, set) => { - set(tabsStateAtom, { ...initialState }); - set(deletedChatIdsAtom, new Set()); +/** Remove unresolved chat pointers after Zero confirms the queried rows are complete. */ +export const pruneMissingChatTabsAtom = atom(null, (get, set, missingChatIds: Set) => { + if (missingChatIds.size === 0) return; + + const state = get(tabsStateAtom); + const firstMissingIdx = state.tabs.findIndex( + (t) => t.type === "chat" && t.entityId !== null && missingChatIds.has(t.entityId) + ); + if (firstMissingIdx === -1) return; + + const remaining = state.tabs.filter( + (t) => !(t.type === "chat" && t.entityId !== null && missingChatIds.has(t.entityId)) + ); + + if (remaining.length === 0) { + set(tabsStateAtom, { + tabs: [INITIAL_CHAT_TAB], + activeTabId: "chat-new", + }); + return; + } + + const activeWasPruned = state.tabs.some( + (t) => + t.id === state.activeTabId && + t.type === "chat" && + t.entityId !== null && + missingChatIds.has(t.entityId) + ); + const newActiveId = activeWasPruned + ? remaining[Math.min(firstMissingIdx, remaining.length - 1)].id + : state.activeTabId; + + set(tabsStateAtom, { tabs: remaining, activeTabId: newActiveId }); }); diff --git a/surfsense_web/atoms/workspaces/workspace-query.atoms.ts b/surfsense_web/atoms/workspaces/workspace-query.atoms.ts index 85203cc1d..09e2aa290 100644 --- a/surfsense_web/atoms/workspaces/workspace-query.atoms.ts +++ b/surfsense_web/atoms/workspaces/workspace-query.atoms.ts @@ -6,14 +6,23 @@ import { cacheKeys } from "@/lib/query-client/cache-keys"; export const activeWorkspaceIdAtom = atom(null); -export const workspacesQueryParamsAtom = atom({ - skip: 0, - limit: 10, - owned_only: false, +export const workspaceLimitsAtom = atomWithQuery(() => { + return { + queryKey: cacheKeys.workspaces.limits, + staleTime: Infinity, + queryFn: async () => { + return workspacesApiService.getWorkspaceLimits(); + }, + }; }); export const workspacesAtom = atomWithQuery((get) => { - const queryParams = get(workspacesQueryParamsAtom); + const workspaceLimits = get(workspaceLimitsAtom).data; + const queryParams: GetWorkspacesRequest["queryParams"] = { + skip: 0, + ...(workspaceLimits ? { limit: workspaceLimits.max_workspaces_per_user } : {}), + owned_only: false, + }; return { queryKey: cacheKeys.workspaces.withQueryParams(queryParams), diff --git a/surfsense_web/components/assistant-ui/chat-viewport.tsx b/surfsense_web/components/assistant-ui/chat-viewport.tsx index 83308b642..cb0c57442 100644 --- a/surfsense_web/components/assistant-ui/chat-viewport.tsx +++ b/surfsense_web/components/assistant-ui/chat-viewport.tsx @@ -22,9 +22,19 @@ const ChatScrollToBottom: FC = () => ( export interface ChatViewportProps { children: ReactNode; footer?: ReactNode; + /** + * Keep the footer (composer) pinned even when the thread has no messages — + * needed while an existing thread's messages are still loading, so the + * bottom composer stays visible above the loading skeleton. + */ + footerAlwaysVisible?: boolean; } -export const ChatViewport: FC = ({ children, footer }) => ( +export const ChatViewport: FC = ({ + children, + footer, + footerAlwaysVisible = false, +}) => ( = ({ children, footer }) => ( /> {children} {footer ? ( - !thread.isEmpty}> + footerAlwaysVisible || !thread.isEmpty}> void; + /** Connected connectors: one row per type, with live indexing health (`useConnectorRows`). */ + connectorRows: ConnectorRow[]; + /** Open a connector's manage view (deep-links via importConnectorRequestAtom). */ + onSelectConnector: (row: ConnectorRow) => void; + /** Navigate to the full connectors catalog. */ + onBrowseConnectors: () => void; + regularToolGroups: ToolGroupView[]; + connectorToolGroups: ToolGroupView[]; + otherToolGroup?: ToolGroupView; + disabledToolsSet: Set; + onToggleTool: (name: string) => void; + onToggleToolGroup: (names: string[]) => void; + /** True while the tool list is still loading (shows a skeleton). */ + toolsLoading: boolean; +} + +/** + * Mobile "+" menu. A single vaul drawer that behaves like a flat list at the + * root and drills into submenus in place (each level replaces the previous, + * with a back button) rather than nesting overlays — the touch-friendly + * equivalent of the desktop dropdown submenus. Screen state is a small push/pop + * stack; closing the drawer resets it to the root. + */ +type Screen = + | { kind: "root" } + | { kind: "connectors" } + | { kind: "tools" } + | { kind: "toolGroup"; label: string }; + +const ROW = + "flex w-full items-center gap-3 px-4 py-3 text-sm hover:bg-accent hover:text-accent-foreground transition-colors"; + +export function ComposerAddMenuDrawer({ + trigger, + onUploadFiles, + connectorRows, + onSelectConnector, + onBrowseConnectors, + regularToolGroups, + connectorToolGroups, + otherToolGroup, + disabledToolsSet, + onToggleTool, + onToggleToolGroup, + toolsLoading, +}: ComposerAddMenuDrawerProps) { + const [open, setOpen] = useState(false); + const [stack, setStack] = useState([{ kind: "root" }]); + // Slide direction: forward on push, back on pop — drives the enter animation. + const dirRef = useRef<"forward" | "back">("forward"); + const current = stack[stack.length - 1]; + + const push = useCallback((screen: Screen) => { + dirRef.current = "forward"; + setStack((prev) => [...prev, screen]); + }, []); + const pop = useCallback(() => { + dirRef.current = "back"; + setStack((prev) => (prev.length > 1 ? prev.slice(0, -1) : prev)); + }, []); + + const handleOpenChange = useCallback((next: boolean) => { + setOpen(next); + // Reset to root when closed so the next open starts fresh. + if (!next) { + dirRef.current = "forward"; + setStack([{ kind: "root" }]); + } + }, []); + + const close = useCallback(() => handleOpenChange(false), [handleOpenChange]); + + const title = + current.kind === "connectors" + ? "MCP Connectors" + : current.kind === "tools" + ? "Manage Tools" + : current.kind === "toolGroup" + ? current.label + : "Add"; + + const renderToolRow = (name: string) => { + const isDisabled = disabledToolsSet.has(name); + const ToolIcon = getToolIcon(name); + return ( +
+ + {getToolDisplayName(name)} + onToggleTool(name)} + className="shrink-0" + /> +
+ ); + }; + + const renderBody = () => { + if (current.kind === "root") { + return ( + <> + + + + + ); + } + + if (current.kind === "connectors") { + return ( + <> + {connectorRows.length === 0 ? ( +

+ No connectors yet. +

+ ) : ( + connectorRows.map((row) => ( + + )) + )} + + + + ); + } + + if (current.kind === "toolGroup") { + const group = connectorToolGroups.find((g) => g.label === current.label); + return <>{group?.tools.map((t) => renderToolRow(t.name))}; + } + + // current.kind === "tools" + if (toolsLoading) { + return ( +
+ + {["t1", "t2", "t3", "t4"].map((k) => ( +
+ + + +
+ ))} +
+ ); + } + + return ( + <> + {regularToolGroups.map((group) => ( +
+
+ {group.label} +
+ {group.tools.map((t) => renderToolRow(t.name))} +
+ ))} + {connectorToolGroups.length > 0 && ( +
+
+ Connector Actions +
+ {connectorToolGroups.map((group) => { + const iconInfo = CONNECTOR_TOOL_ICON_PATHS[group.connectorIcon ?? ""]; + const toolNames = group.tools.map((t) => t.name); + const allDisabled = toolNames.every((n) => disabledToolsSet.has(n)); + return ( +
+ + onToggleToolGroup(toolNames)} + className="shrink-0" + /> +
+ ); + })} +
+ )} + {otherToolGroup && ( +
+
+ {otherToolGroup.label} +
+ {otherToolGroup.tools.map((t) => renderToolRow(t.name))} +
+ )} + + ); + }; + + return ( + + {trigger} + + + + {stack.length > 1 ? ( + + ) : ( + + )} + {title} + + +
+
+ {renderBody()} +
+
+
+
+ ); +} diff --git a/surfsense_web/components/assistant-ui/connector-popup.tsx b/surfsense_web/components/assistant-ui/connector-popup.tsx deleted file mode 100644 index a8bfc29bd..000000000 --- a/surfsense_web/components/assistant-ui/connector-popup.tsx +++ /dev/null @@ -1,388 +0,0 @@ -"use client"; - -import { useAtomValue } from "jotai"; -import { forwardRef, useEffect, useImperativeHandle, useMemo, useState } from "react"; -import { createPortal } from "react-dom"; -import { statusInboxItemsAtom } from "@/atoms/inbox/status-inbox.atom"; -import { activeWorkspaceIdAtom } from "@/atoms/workspaces/workspace-query.atoms"; -import { Dialog, DialogContent, DialogTitle } from "@/components/ui/dialog"; -import { Tabs, TabsContent } from "@/components/ui/tabs"; -import type { SearchSourceConnector } from "@/contracts/types/connector.types"; -import { useConnectorsSync } from "@/hooks/use-connectors-sync"; -import { PICKER_CLOSE_EVENT, PICKER_OPEN_EVENT } from "@/hooks/use-google-picker"; -import { useZeroDocumentTypeCounts } from "@/hooks/use-zero-document-type-counts"; -import { ConnectorDialogHeader } from "./connector-popup/components/connector-dialog-header"; -import { ConnectorConnectView } from "./connector-popup/connector-configs/views/connector-connect-view"; -import { ConnectorEditView } from "./connector-popup/connector-configs/views/connector-edit-view"; -import { IndexingConfigurationView } from "./connector-popup/connector-configs/views/indexing-configuration-view"; -import { - COMPOSIO_CONNECTORS, - OAUTH_CONNECTORS, -} from "./connector-popup/constants/connector-constants"; -import { useConnectorDialog } from "./connector-popup/hooks/use-connector-dialog"; -import { useIndexingConnectors } from "./connector-popup/hooks/use-indexing-connectors"; -import { ActiveConnectorsTab } from "./connector-popup/tabs/active-connectors-tab"; -import { AllConnectorsTab } from "./connector-popup/tabs/all-connectors-tab"; -import { ConnectorAccountsListView } from "./connector-popup/views/connector-accounts-list-view"; -import { YouTubeCrawlerView } from "./connector-popup/views/youtube-crawler-view"; - -export interface ConnectorIndicatorHandle { - open: () => void; -} - -interface ConnectorIndicatorProps { - showTrigger?: boolean; -} - -export const ConnectorIndicator = forwardRef( - (_props, ref) => { - const workspaceId = useAtomValue(activeWorkspaceIdAtom); - - // Real-time document type counts via Zero (updates instantly as docs are indexed) - const documentTypeCounts = useZeroDocumentTypeCounts(workspaceId); - // Read status inbox items from shared atom (populated by LayoutDataProvider) - // instead of creating a duplicate useInbox("status") hook. - const statusInboxItems = useAtomValue(statusInboxItemsAtom); - const inboxItems = useMemo( - () => statusInboxItems.filter((item) => item.type === "connector_indexing"), - [statusInboxItems] - ); - - // Use the custom hook for dialog state management - const { - isOpen, - activeTab, - connectingId, - isScrolled, - searchQuery, - indexingConfig, - indexingConnector, - indexingConnectorConfig, - editingConnector, - connectingConnectorType, - isCreatingConnector, - startDate, - endDate, - isStartingIndexing, - isSaving, - isDisconnecting, - periodicEnabled, - frequencyMinutes, - enableVisionLlm, - allConnectors, - viewingAccountsType, - viewingMCPList, - isYouTubeView, - isFromOAuth, - setSearchQuery, - setStartDate, - setEndDate, - setPeriodicEnabled, - setFrequencyMinutes, - setEnableVisionLlm, - handleOpenChange, - handleTabChange, - handleScroll, - handleConnectOAuth, - handleConnectNonOAuth, - handleCreateWebcrawler, - handleCreateYouTubeCrawler, - handleSubmitConnectForm, - handleStartIndexing, - handleSkipIndexing, - handleStartEdit, - handleSaveConnector, - handleDisconnectConnector, - handleBackFromEdit, - handleBackFromConnect, - handleBackFromYouTube, - handleViewAccountsList, - handleBackFromAccountsList, - handleBackFromMCPList, - handleAddNewMCPFromList, - handleQuickIndexConnector, - connectorConfig, - setConnectorConfig, - setIndexingConnectorConfig, - setConnectorName, - } = useConnectorDialog(); - - const [pickerOpen, setPickerOpen] = useState(false); - useEffect(() => { - const onOpen = () => setPickerOpen(true); - const onClose = () => setPickerOpen(false); - window.addEventListener(PICKER_OPEN_EVENT, onOpen); - window.addEventListener(PICKER_CLOSE_EVENT, onClose); - return () => { - window.removeEventListener(PICKER_OPEN_EVENT, onOpen); - window.removeEventListener(PICKER_CLOSE_EVENT, onClose); - }; - }, []); - - const { - connectors: connectorsFromSync = [], - loading: connectorsLoading, - error: connectorsError, - refreshConnectors: refreshConnectorsSync, - } = useConnectorsSync(workspaceId); - - const useSyncData = connectorsFromSync.length > 0 || (connectorsLoading && !connectorsError); - const connectors = useSyncData ? connectorsFromSync : allConnectors || []; - - const refreshConnectors = async () => { - if (useSyncData) { - await refreshConnectorsSync(); - } - }; - - // Track indexing state locally - clears automatically when last_indexed_at changes via real-time sync - // Also clears when failed notifications are detected - const { indexingConnectorIds, startIndexing, stopIndexing } = useIndexingConnectors( - connectors as SearchSourceConnector[], - inboxItems - ); - - // Get document types that have documents in the workspace - const activeDocumentTypes = documentTypeCounts - ? Object.entries(documentTypeCounts).filter(([, count]) => count > 0) - : []; - - const hasConnectors = connectors.length > 0; - const hasSources = hasConnectors || activeDocumentTypes.length > 0; - const totalSourceCount = connectors.length + activeDocumentTypes.length; - - const activeConnectorsCount = connectors.length; - - // Check which connectors are already connected - // Real-time connector updates via Zero sync - const connectedTypes = new Set( - (connectors || []).map((c: SearchSourceConnector) => c.connector_type) - ); - - useImperativeHandle(ref, () => ({ - open: () => handleOpenChange(true), - })); - - if (!workspaceId) return null; - - return ( - - {isOpen && - createPortal( - - ); - } -); - -ConnectorIndicator.displayName = "ConnectorIndicator"; diff --git a/surfsense_web/components/assistant-ui/connector-popup/components/connector-card.tsx b/surfsense_web/components/assistant-ui/connector-popup/components/connector-card.tsx index d8e4b174c..dadfe32ab 100644 --- a/surfsense_web/components/assistant-ui/connector-popup/components/connector-card.tsx +++ b/surfsense_web/components/assistant-ui/connector-popup/components/connector-card.tsx @@ -23,7 +23,6 @@ interface ConnectorCardProps { accountCount?: number; connectorCount?: number; isIndexing?: boolean; - deprecated?: boolean; onConnect?: () => void; onManage?: () => void; } @@ -53,15 +52,11 @@ export const ConnectorCard: FC = ({ accountCount, connectorCount, isIndexing = false, - deprecated = false, onConnect, onManage, }) => { const isMCP = connectorType === EnumConnectorName.MCP_CONNECTOR; const isLive = !!connectorType && LIVE_CONNECTOR_TYPES.has(connectorType); - // Deprecated connectors can no longer be connected, but existing rows stay - // manageable (so users can disconnect them). - const isDeprecatedForConnect = deprecated && !isConnected; // Get connector status const { getConnectorStatus, isConnectorEnabled, getConnectorStatusMessage, shouldShowWarnings } = useConnectorStatus(); @@ -78,10 +73,6 @@ export const ConnectorCard: FC = ({ return null; } - if (isDeprecatedForConnect) { - return "Deprecated. No longer available to connect."; - } - return description; }; @@ -102,16 +93,7 @@ export const ConnectorCard: FC = ({ : "bg-slate-400/5 dark:bg-white/5 border-slate-400/5 dark:border-white/5" )} > - {isMCP ? ( -