diff --git a/README.es.md b/README.es.md index 08231b5ab..3d38b8334 100644 --- a/README.es.md +++ b/README.es.md @@ -22,7 +22,7 @@ # SurfSense: NotebookLM para investigación de inteligencia competitiva -SurfSense es la **plataforma de inteligencia competitiva de código abierto para agentes de IA**, como NotebookLM pero con conectores de scraping en vivo. Tus agentes monitorean a la competencia, siguen los rankings y escuchan a tu mercado con datos en vivo de **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search y la web abierta**, a través de una única **API REST** o un **servidor MCP**. Agentes programados y activados por eventos convierten lo que encuentran en informes y alertas, y una base de conocimiento integrada mantiene cada hallazgo disponible para búsqueda con citas. +SurfSense es la **plataforma de inteligencia competitiva de código abierto para agentes de IA**, como NotebookLM pero con conectores de scraping en vivo. Tus agentes monitorean a la competencia, siguen los rankings y escuchan a tu mercado con datos en vivo de **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed y la web abierta**, a través de una única **API REST** o un **servidor MCP**. Agentes programados y activados por eventos convierten lo que encuentran en informes y alertas, y una base de conocimiento integrada mantiene cada hallazgo disponible para búsqueda con citas. > [!NOTE] > **📢 Una nota para nuestros usuarios de la alternativa a NotebookLM** @@ -107,6 +107,7 @@ Las automatizaciones ejecutan turnos completos de agente según un horario o en | **TikTok** | Videos, comentarios, hashtags y perfiles sin aprobación de la Research API | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | Lugares, calificaciones y reseñas para investigar competidores locales y prospectos | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | SERPs en vivo para seguimiento de posiciones y monitoreo de mercado | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | Ofertas de empleo públicas con salarios y descripciones completas, por búsqueda o empresa | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Web Crawl** (rastreo web) | Cualquier página de la web abierta como contenido limpio y estructurado | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **Conectores MCP externos** | Conecta cualquier servidor MCP a tus agentes, con OAuth de un clic para Notion, Slack, Jira y más | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -246,7 +247,7 @@ https://github.com/user-attachments/assets/a0a16566-6967-4374-ac51-9b3e07fbecd7 | Característica | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **Datos de mercado en vivo para agentes** | No | Conectores de Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search y rastreo web vía API REST y MCP | +| **Datos de mercado en vivo para agentes** | No | Conectores de Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed y rastreo web vía API REST y MCP | | **Servidor MCP** | No | Cada conector expuesto como herramienta nativa de agente, más servidores MCP propios con aplicaciones OAuth de un clic | | **Fuentes por Notebook** | 50 (gratis) a 600 (Ultra, $249.99/mes) | Ilimitadas | | **Número de Notebooks** | 100 (gratis) a 500 (niveles de pago) | Ilimitado | diff --git a/README.hi.md b/README.hi.md index eb65ff489..2a26e2146 100644 --- a/README.hi.md +++ b/README.hi.md @@ -22,7 +22,7 @@ # SurfSense: कॉम्पिटिटिव इंटेलिजेंस रिसर्च के लिए NotebookLM -SurfSense **AI एजेंट्स के लिए ओपन सोर्स कॉम्पिटिटिव इंटेलिजेंस प्लेटफ़ॉर्म** है, बिलकुल NotebookLM जैसा, पर लाइव स्क्रैपिंग कनेक्टर्स के साथ। आपके एजेंट प्रतिस्पर्धियों पर नज़र रखते हैं, रैंकिंग ट्रैक करते हैं, और **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search और ओपन वेब** से लाइव डेटा के साथ आपके बाज़ार की बात सुनते हैं, वह भी एक ही **REST API** या **MCP सर्वर** के ज़रिए। शेड्यूल्ड और इवेंट-ट्रिगर्ड एजेंट अपनी खोजों को ब्रीफ़ और अलर्ट में बदलते हैं, और एक बिल्ट-इन नॉलेज बेस हर खोज को साइटेशन के साथ खोजने योग्य बनाए रखता है। +SurfSense **AI एजेंट्स के लिए ओपन सोर्स कॉम्पिटिटिव इंटेलिजेंस प्लेटफ़ॉर्म** है, बिलकुल NotebookLM जैसा, पर लाइव स्क्रैपिंग कनेक्टर्स के साथ। आपके एजेंट प्रतिस्पर्धियों पर नज़र रखते हैं, रैंकिंग ट्रैक करते हैं, और **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed और ओपन वेब** से लाइव डेटा के साथ आपके बाज़ार की बात सुनते हैं, वह भी एक ही **REST API** या **MCP सर्वर** के ज़रिए। शेड्यूल्ड और इवेंट-ट्रिगर्ड एजेंट अपनी खोजों को ब्रीफ़ और अलर्ट में बदलते हैं, और एक बिल्ट-इन नॉलेज बेस हर खोज को साइटेशन के साथ खोजने योग्य बनाए रखता है। > [!NOTE] > **📢 हमारे NotebookLM-विकल्प उपयोगकर्ताओं के लिए एक सूचना** @@ -107,6 +107,7 @@ SurfSense **AI एजेंट्स के लिए ओपन सोर्स | **TikTok** | Research API अप्रूवल के बिना वीडियो, कमेंट, हैशटैग और प्रोफ़ाइल | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | स्थानीय प्रतिस्पर्धी और लीड रिसर्च के लिए स्थान, रेटिंग और रिव्यू | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | रैंक ट्रैकिंग और मार्केट मॉनिटरिंग के लिए लाइव SERP | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | सार्वजनिक नौकरी लिस्टिंग, सैलरी और पूरे विवरण के साथ, सर्च या कंपनी के अनुसार | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Web Crawl** | ओपन वेब का कोई भी पेज साफ़-सुथरे, स्ट्रक्चर्ड कंटेंट के रूप में | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **External MCP Connectors** | कोई भी MCP सर्वर अपने एजेंट्स से जोड़ें, Notion, Slack, Jira और अन्य के लिए वन-क्लिक OAuth के साथ | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -246,7 +247,7 @@ https://github.com/user-attachments/assets/a0a16566-6967-4374-ac51-9b3e07fbecd7 | फ़ीचर | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **एजेंट्स के लिए लाइव मार्केट डेटा** | नहीं | REST API और MCP के ज़रिए Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search और वेब क्रॉल कनेक्टर | +| **एजेंट्स के लिए लाइव मार्केट डेटा** | नहीं | REST API और MCP के ज़रिए Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed और वेब क्रॉल कनेक्टर | | **MCP सर्वर** | नहीं | हर कनेक्टर नेटिव एजेंट टूल के रूप में उपलब्ध, साथ ही वन-क्लिक OAuth ऐप्स के साथ अपने MCP सर्वर लाने की सुविधा | | **प्रति नोटबुक स्रोत** | 50 (Free) से 600 (Ultra, $249.99/माह) | असीमित | | **नोटबुक की संख्या** | 100 (Free) से 500 (सशुल्क टियर) | असीमित | diff --git a/README.md b/README.md index ba289ada4..56af65dd1 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ # SurfSense: NotebookLM for Competitive Intelligence Research -SurfSense is the **open-source competitive intelligence platform for AI agents**, like NotebookLM but with live scraping connectors. Your agents monitor competitors, track rankings, and listen to your market with live data from **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, and the open web**, through one **REST API** or **MCP server**. Scheduled and event-triggered agents turn what they find into briefs and alerts, and a built-in knowledge base keeps every finding searchable with citations. +SurfSense is the **open-source competitive intelligence platform for AI agents**, like NotebookLM but with live scraping connectors. Your agents monitor competitors, track rankings, and listen to your market with live data from **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed, and the open web**, through one **REST API** or **MCP server**. Scheduled and event-triggered agents turn what they find into briefs and alerts, and a built-in knowledge base keeps every finding searchable with citations. > [!NOTE] > **📢 A note for our NotebookLM-alternative users** @@ -107,6 +107,7 @@ Automations run full agent turns on a schedule or in response to events, then wr | **TikTok** | Videos, comments, hashtags, and profiles without Research API approval | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | Places, ratings, and reviews for local competitor and lead research | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | Live SERPs for rank tracking and market monitoring | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | Public job postings with salaries and full descriptions, by search or company | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Web Crawl** | Any page on the open web as clean, structured content | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **External MCP Connectors** | Bring any MCP server to your agents, with one-click OAuth for Notion, Slack, Jira, and more | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -245,7 +246,7 @@ Still comparing us as a NotebookLM alternative? Here is the honest breakdown. | Feature | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **Live market data for agents** | No | Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, and web crawl connectors via REST API and MCP | +| **Live market data for agents** | No | Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed, and web crawl connectors via REST API and MCP | | **MCP server** | No | Every connector exposed as a native agent tool, plus bring-your-own MCP servers with one-click OAuth apps | | **Sources per Notebook** | 50 (Free) to 600 (Ultra, $249.99/mo) | Unlimited | | **Number of Notebooks** | 100 (Free) to 500 (paid tiers) | Unlimited | diff --git a/README.pt-BR.md b/README.pt-BR.md index 327e3373c..496398d19 100644 --- a/README.pt-BR.md +++ b/README.pt-BR.md @@ -22,7 +22,7 @@ # SurfSense: NotebookLM para Pesquisa de Inteligência Competitiva -O SurfSense é a **plataforma open source de inteligência competitiva para agentes de IA**, como o NotebookLM, mas com conectores de scraping ao vivo. Seus agentes monitoram concorrentes, acompanham rankings e escutam o seu mercado com dados ao vivo do **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search e da web aberta**, por meio de uma única **API REST** ou de um **servidor MCP**. Agentes agendados ou acionados por eventos transformam o que encontram em relatórios e alertas, e uma base de conhecimento integrada mantém cada descoberta pesquisável, com citações. +O SurfSense é a **plataforma open source de inteligência competitiva para agentes de IA**, como o NotebookLM, mas com conectores de scraping ao vivo. Seus agentes monitoram concorrentes, acompanham rankings e escutam o seu mercado com dados ao vivo do **Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed e da web aberta**, por meio de uma única **API REST** ou de um **servidor MCP**. Agentes agendados ou acionados por eventos transformam o que encontram em relatórios e alertas, e uma base de conhecimento integrada mantém cada descoberta pesquisável, com citações. > [!NOTE] > **📢 Um recado para nossos usuários que buscavam uma alternativa ao NotebookLM** @@ -107,6 +107,7 @@ As automações executam turnos completos de agente de forma agendada ou em resp | **TikTok** | Vídeos, comentários, hashtags e perfis sem aprovação da Research API | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | Estabelecimentos, avaliações e reviews para pesquisa local de concorrentes e leads | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | SERPs ao vivo para acompanhamento de rankings e monitoramento de mercado | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | Vagas públicas com salários e descrições completas, por busca ou empresa | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Web Crawl** (rastreamento web) | Qualquer página da web aberta como conteúdo limpo e estruturado | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **Conectores MCP externos** | Traga qualquer servidor MCP para seus agentes, com OAuth em um clique para Notion, Slack, Jira e outros | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -246,7 +247,7 @@ Ainda nos comparando como alternativa ao NotebookLM? Aqui está o comparativo ho | Recurso | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **Dados de mercado ao vivo para agentes** | Não | Conectores de Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search e rastreamento web via API REST e MCP | +| **Dados de mercado ao vivo para agentes** | Não | Conectores de Reddit, YouTube, Instagram, TikTok, Google Maps, Google Search, Indeed e rastreamento web via API REST e MCP | | **Servidor MCP** | Não | Cada conector exposto como ferramenta nativa de agente, além de servidores MCP próprios com apps OAuth em um clique | | **Fontes por Notebook** | 50 (gratuito) a 600 (Ultra, US$ 249,99/mês) | Ilimitadas | | **Número de Notebooks** | 100 (gratuito) a 500 (planos pagos) | Ilimitado | diff --git a/README.zh-CN.md b/README.zh-CN.md index 53a95968b..c75e71a81 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -22,7 +22,7 @@ # SurfSense:面向竞争情报研究的 NotebookLM -SurfSense 是**面向 AI 智能体的开源竞争情报平台**,就像 NotebookLM,但配备了实时抓取连接器。你的智能体可以通过一个 **REST API** 或 **MCP 服务器**,利用来自 **Reddit、YouTube、Instagram、TikTok、Google Maps、Google Search 和开放网络**的实时数据,监控竞争对手、追踪排名、倾听市场动态。定时和事件触发的智能体会把发现的内容转化为简报和预警,内置的知识库则让每一条发现都可搜索、可引用。 +SurfSense 是**面向 AI 智能体的开源竞争情报平台**,就像 NotebookLM,但配备了实时抓取连接器。你的智能体可以通过一个 **REST API** 或 **MCP 服务器**,利用来自 **Reddit、YouTube、Instagram、TikTok、Google Maps、Google Search、Indeed 和开放网络**的实时数据,监控竞争对手、追踪排名、倾听市场动态。定时和事件触发的智能体会把发现的内容转化为简报和预警,内置的知识库则让每一条发现都可搜索、可引用。 > [!NOTE] > **📢 致我们的 NotebookLM 替代品用户** @@ -107,6 +107,7 @@ SurfSense 是**面向 AI 智能体的开源竞争情报平台**,就像 Noteboo | **TikTok** | 视频、评论、话题标签和主页,无需 Research API 审批 | [TikTok Scraper API](https://www.surfsense.com/tiktok) | | **Google Maps** | 地点、评分和评论,用于本地竞争对手和潜在客户调研 | [Google Maps Scraper API](https://www.surfsense.com/google-maps) | | **Google Search** | 实时搜索结果页,用于排名追踪和市场监控 | [Google Search API](https://www.surfsense.com/google-search) | +| **Indeed** | 公开职位信息,含薪资与完整职位描述,按搜索或公司抓取 | [Indeed Scraper API](https://www.surfsense.com/indeed) | | **Web Crawl** | 把开放网络上的任意页面转为干净、结构化的内容 | [Web Crawling API](https://www.surfsense.com/web-crawl) | | **外部 MCP 连接器** | 将任意 MCP 服务器接入你的智能体,Notion、Slack、Jira 等支持一键 OAuth | [External MCP Connectors](https://www.surfsense.com/external-mcp-connectors) | @@ -246,7 +247,7 @@ https://github.com/user-attachments/assets/a0a16566-6967-4374-ac51-9b3e07fbecd7 | 功能 | Google NotebookLM | SurfSense | |---------|-------------------|-----------| -| **面向智能体的实时市场数据** | 无 | 通过 REST API 和 MCP 提供 Reddit、YouTube、Instagram、TikTok、Google Maps、Google Search 和网页爬取连接器 | +| **面向智能体的实时市场数据** | 无 | 通过 REST API 和 MCP 提供 Reddit、YouTube、Instagram、TikTok、Google Maps、Google Search、Indeed 和网页爬取连接器 | | **MCP 服务器** | 无 | 每个连接器都作为原生智能体工具暴露,还可自带 MCP 服务器并使用一键 OAuth 应用 | | **每个笔记本的来源数** | 50 个(免费版)至 600 个(Ultra 版,249.99 美元/月) | 无限制 | | **笔记本数量** | 100 个(免费版)至 500 个(付费档位) | 无限制 | diff --git a/docker/.env.example b/docker/.env.example index 2aa0806a8..7f4510320 100644 --- a/docker/.env.example +++ b/docker/.env.example @@ -457,6 +457,7 @@ SURFSENSE_ENABLE_DOOM_LOOP=true # TIKTOK_MICROS_PER_VIDEO=3500 # TIKTOK_MICROS_PER_USER=2500 # TIKTOK_MICROS_PER_COMMENT=1500 +# INDEED_SCRAPE_MICROS_PER_JOB=3500 # Safety ceiling on per-call premium reservation, in micro-USD ($1.00 default). # QUOTA_MAX_RESERVE_MICROS=1000000 diff --git a/surfsense_backend/.env.example b/surfsense_backend/.env.example index 3d2355460..44a3af872 100644 --- a/surfsense_backend/.env.example +++ b/surfsense_backend/.env.example @@ -297,6 +297,7 @@ MICROS_PER_PAGE=1000 # TIKTOK_MICROS_PER_VIDEO=3500 # TIKTOK_MICROS_PER_USER=2500 # TIKTOK_MICROS_PER_COMMENT=1500 +# INDEED_SCRAPE_MICROS_PER_JOB=3500 # Browser-listing retries when a feed is empty (profile feed is withheld from # flagged IPs; each retry draws a fresh rotating exit IP). Set to 1 for a static IP. # TIKTOK_LISTING_MAX_ATTEMPTS=3 diff --git a/surfsense_web/lib/auth-utils.ts b/surfsense_web/lib/auth-utils.ts index 7d9320fb0..837c3eb44 100644 --- a/surfsense_web/lib/auth-utils.ts +++ b/surfsense_web/lib/auth-utils.ts @@ -44,6 +44,7 @@ const PUBLIC_ROUTE_PREFIXES = [ "/youtube", "/google-maps", "/google-search", + "/indeed", "/web-crawl", ]; diff --git a/surfsense_web/lib/connectors-marketing/google-maps.tsx b/surfsense_web/lib/connectors-marketing/google-maps.tsx index 03b1d247b..b09301ca6 100644 --- a/surfsense_web/lib/connectors-marketing/google-maps.tsx +++ b/surfsense_web/lib/connectors-marketing/google-maps.tsx @@ -306,6 +306,7 @@ export const googleMaps: ConnectorPageContent = { { label: "Instagram API", href: "/instagram" }, { label: "SERP API", href: "/google-search" }, { label: "Web Crawl API", href: "/web-crawl" }, + { label: "Indeed API", href: "/indeed" }, { label: "SurfSense MCP Server", href: "/mcp-server" }, { label: "Read the docs", href: "/docs" }, ], diff --git a/surfsense_web/lib/connectors-marketing/google-search.tsx b/surfsense_web/lib/connectors-marketing/google-search.tsx index 7f436f9b0..a7b69c2ad 100644 --- a/surfsense_web/lib/connectors-marketing/google-search.tsx +++ b/surfsense_web/lib/connectors-marketing/google-search.tsx @@ -261,6 +261,7 @@ export const googleSearch: ConnectorPageContent = { { label: "Google Maps API", href: "/google-maps" }, { label: "Reddit API", href: "/reddit" }, { label: "Instagram API", href: "/instagram" }, + { label: "Indeed API", href: "/indeed" }, { label: "SurfSense MCP Server", href: "/mcp-server" }, { label: "Read the docs", href: "/docs" }, ], diff --git a/surfsense_web/lib/connectors-marketing/indeed.tsx b/surfsense_web/lib/connectors-marketing/indeed.tsx new file mode 100644 index 000000000..ce717f3cc --- /dev/null +++ b/surfsense_web/lib/connectors-marketing/indeed.tsx @@ -0,0 +1,327 @@ +import { IconBriefcase } from "@tabler/icons-react"; +import type { ConnectorPageContent } from "./types"; + +export const indeed: ConnectorPageContent = { + slug: "indeed", + name: "Indeed", + icon: IconBriefcase, + + metaTitle: "Indeed Scraper API for Jobs and Hiring Data | SurfSense", + metaDescription: + "Scrape public Indeed job postings with the SurfSense Indeed Scraper API: titles, companies, salaries, and full descriptions by search or company. No Indeed API. Start free.", + keywords: [ + "indeed scraper", + "indeed scraper api", + "indeed api", + "indeed jobs api", + "scrape indeed", + "indeed job scraper", + "job posting scraper", + "salary data api", + "hiring data api", + "indeed mcp", + "labor market data", + "recruiting data tool", + ], + + h1: "Indeed Scraper API for Job Postings and Hiring Data", + heroLede: + "The SurfSense Indeed API extracts public job postings, salaries, companies, and full descriptions by search query, company page, or job URL, without Indeed's official API. Give your AI agents a live feed of who is hiring, for what, at what pay, so you track the labor market as it moves.", + + transcript: { + prompt: "Find remote data analyst roles posted this week and what they pay", + toolCall: + 'indeed.scrape({ search_queries: ["data analyst"], location: "Remote",\n remote: "remote", from_days: 7, sort: "date", max_items: 30 })', + rows: [ + { + primary: "Senior Data Analyst · Acme Corp", + secondary: "Remote (US) · $120k–$145k/year · posted 2 days ago", + tag: "salary listed", + }, + { + primary: "Data Analyst, Growth · Globex", + secondary: "Remote · $95k–$110k/year · Indeed Apply", + tag: "buying signal", + }, + { + primary: "Marketing Data Analyst · Initech", + secondary: "Remote (US) · estimated $88k–$102k · posted today", + tag: "new today", + }, + ], + resultSummary: "30 jobs · 22 with salary · surfaced in 3.4s", + }, + + extractIntro: + "Every call returns structured job items. Point the API at a search query, an Indeed search or company page, or a single job URL, and set scrape_job_details for the full description per job.", + extractFields: [ + { + label: "Job", + description: "Title, job key, listing URL, apply URL, and whether Indeed Apply is enabled.", + }, + { + label: "Company", + description: "Company name, profile URL, star rating, and review count where available.", + }, + { + label: "Location", + description: "Formatted location, city, state, postal code, country, and remote or hybrid flags.", + }, + { + label: "Salary", + description: + "Pay text, min and max bounds, currency, period, and whether the figure is an Indeed estimate.", + }, + { + label: "Description", + description: + "Listing snippet by default; the full text and HTML description with scrape_job_details.", + }, + { + label: "Signals", + description: + "Job types, benefits, sponsored, urgently hiring, new, and expired flags, plus post age.", + }, + ], + + useCasesHeading: "What teams do with the Indeed API", + useCases: [ + { + title: "Competitor hiring intelligence", + description: + "Track what your competitors are hiring for and where. A spike in sales or ML roles is a roadmap signal months before it ships. Feed the stream to an agent that flags the moves that matter.", + }, + { + title: "Salary and compensation benchmarking", + description: + "Pull real posted salaries for a title in a location and benchmark your own bands against the live market, instead of a survey that is a year stale.", + }, + { + title: "Labor market and sector research", + description: + "Measure hiring demand for a role, skill, or sector over time. Turn thousands of postings into a demand index your analysts and clients can act on.", + }, + { + title: "Recruiting and lead sourcing", + description: + "Find companies actively hiring for a role and reach them while the need is hot. Job postings are a public, timely buying signal for staffing and B2B sales.", + }, + ], + + comparison: { + heading: "An Indeed API alternative built for agents", + intro: + "Indeed retired its public Publisher jobs API and gates data behind partner programs. If you cannot get access or need clean structured jobs now, here is how SurfSense compares.", + columnLabel: "DIY Indeed scraping", + rows: [ + { + feature: "Access", + official: "Publisher API retired; partner-gated and approval-only", + surfsense: "One API key; scrape public postings without an approval process", + }, + { + feature: "Anti-bot", + official: "You fight Cloudflare, fingerprinting, and CAPTCHAs yourself", + surfsense: "Warmed, rotated sessions managed for you; no proxy plumbing", + }, + { + feature: "Pricing", + official: "Proxy, browser, and maintenance costs you own", + surfsense: "Pay per job returned, with a free tier to start", + }, + { + feature: "Descriptions", + official: "Extra page fetch and parsing you build and maintain", + surfsense: "Full description per job with one scrape_job_details flag", + }, + { + feature: "Agent-ready", + official: "No; you build the harness yourself", + surfsense: "MCP server exposes indeed.scrape as a native tool", + }, + ], + }, + + api: { + platform: "indeed", + verb: "scrape", + mcpTool: "indeed.scrape", + requestBody: { + search_queries: ["data analyst"], + location: "Remote", + remote: "remote", + from_days: 7, + sort: "date", + max_items: 30, + }, + }, + + schema: { + requestNote: + "Provide at least one source: urls or search_queries. Up to 20 sources per call.", + request: [ + { + name: "urls", + type: "string[]", + defaultValue: "[]", + description: + "Indeed URLs: a search page (/jobs?q=&l=), a company jobs page (/cmp//jobs), or a single job (/viewjob?jk=...). Max 20.", + }, + { + name: "search_queries", + type: "string[]", + defaultValue: "[]", + description: + "Job search terms. Each returns up to max_items_per_query results, shaped by the filters below. Max 20.", + }, + { + name: "country", + type: "string", + defaultValue: '"us"', + description: "Country code selecting the Indeed domain, e.g. 'us', 'gb', 'de'.", + }, + { + name: "location", + type: "string", + description: "Where to search, e.g. 'Remote', 'New York, NY'.", + }, + { + name: "radius", + type: "integer", + description: "Search radius in miles or km around location.", + }, + { + name: "job_type", + type: "string", + description: "Employment type: fulltime, parttime, contract, internship, and more.", + }, + { + name: "level", + type: "string", + description: "Experience level: entry_level, mid_level, or senior_level.", + }, + { + name: "remote", + type: "string", + description: "Work model filter: remote or hybrid.", + }, + { + name: "from_days", + type: "integer", + description: "Only return jobs posted within the last N days.", + }, + { + name: "sort", + type: "string", + defaultValue: '"relevance"', + description: "Result ordering: relevance or date.", + }, + { + name: "scrape_job_details", + type: "boolean", + defaultValue: "false", + description: + "Fetch each job's detail page for the full description. Slower: one extra page load per job.", + }, + { + name: "max_items", + type: "integer", + defaultValue: "25", + description: "Max total jobs to return across all sources. 1 to 100.", + }, + { + name: "max_items_per_query", + type: "integer", + defaultValue: "25", + description: "Max jobs to pull per search or company target.", + }, + ], + responseNote: + "The response is { items: [...] } with one flat item per job. Fields Indeed omits are null. One returned job is one billable unit.", + response: [ + { + name: "jobKey / jobUrl / applyUrl", + type: "string", + description: "Indeed job key, listing URL, and third-party apply URL.", + }, + { + name: "title", + type: "string", + description: "The job title as posted.", + }, + { + name: "company / companyUrl", + type: "string", + description: "Company name and its Indeed profile URL.", + }, + { + name: "companyRating / companyReviewCount", + type: "number / integer", + description: "Employer star rating and number of reviews, where Indeed shows them.", + }, + { + name: "formattedLocation / isRemote / remoteType", + type: "string / boolean", + description: "Location string plus remote and hybrid flags.", + }, + { + name: "salary", + type: "object", + description: + "salaryText, salaryMin, salaryMax, currency, period, and isEstimated when the pay is an Indeed estimate.", + }, + { + name: "jobTypes / benefits", + type: "string[]", + description: "Employment types and listed benefits parsed from the posting.", + }, + { + name: "descriptionText / descriptionHtml", + type: "string", + description: "Snippet by default; the full description when scrape_job_details is set.", + }, + { + name: "sponsored / urgentlyHiring / isNew / expired", + type: "boolean", + description: "Listing flags for ranking and filtering.", + }, + { + name: "age / datePublished / scrapedAt", + type: "string", + description: "Relative post age, ISO publish date, and when the job was scraped.", + }, + ], + }, + + faq: [ + { + question: "Is scraping Indeed legal?", + answer: + "SurfSense reads only public Indeed job postings, the same listings any logged-out visitor can see. It never logs in and cannot access private or applicant data. As always, review Indeed's terms and your own compliance needs before you run at scale.", + }, + { + question: "Does Indeed have an official jobs API?", + answer: + "Indeed retired its public Publisher jobs API and now gates job data behind partner and approval programs. SurfSense is an independent alternative: you call one API, or add the MCP server to your agent, and get structured public postings back.", + }, + { + question: "Can I get the full job description?", + answer: + "Yes. By default each job returns the listing snippet, which is fast. Set scrape_job_details to true and SurfSense fetches each job's detail page for the full description text and HTML, at the cost of one extra page load per job.", + }, + { + question: "What are the rate limits?", + answer: + "Each call returns up to 100 jobs across all sources, with up to 20 URLs or search queries per request. SurfSense manages the anti-bot request budget for you, so you scale reads without running proxies or a headless browser yourself.", + }, + ], + + related: [ + { label: "Reddit API", href: "/reddit" }, + { label: "YouTube API", href: "/youtube" }, + { label: "Google Maps API", href: "/google-maps" }, + { label: "SERP API", href: "/google-search" }, + { label: "Web Crawl API", href: "/web-crawl" }, + { label: "SurfSense MCP Server", href: "/mcp-server" }, + ], +}; diff --git a/surfsense_web/lib/connectors-marketing/index.ts b/surfsense_web/lib/connectors-marketing/index.ts index 826a1f261..be3948398 100644 --- a/surfsense_web/lib/connectors-marketing/index.ts +++ b/surfsense_web/lib/connectors-marketing/index.ts @@ -1,5 +1,6 @@ import { googleMaps } from "./google-maps"; import { googleSearch } from "./google-search"; +import { indeed } from "./indeed"; import { instagram } from "./instagram"; import { reddit } from "./reddit"; import { tiktok } from "./tiktok"; @@ -17,6 +18,7 @@ const CONNECTOR_LIST: ConnectorPageContent[] = [ tiktok, googleMaps, googleSearch, + indeed, webCrawl, ]; diff --git a/surfsense_web/lib/connectors-marketing/instagram.tsx b/surfsense_web/lib/connectors-marketing/instagram.tsx index 5422a0c90..999be0de8 100644 --- a/surfsense_web/lib/connectors-marketing/instagram.tsx +++ b/surfsense_web/lib/connectors-marketing/instagram.tsx @@ -288,6 +288,7 @@ export const instagram: ConnectorPageContent = { { label: "Reddit API", href: "/reddit" }, { label: "Google Maps API", href: "/google-maps" }, { label: "SERP API", href: "/google-search" }, + { label: "Indeed API", href: "/indeed" }, { label: "SurfSense MCP Server", href: "/mcp-server" }, ], }; diff --git a/surfsense_web/lib/connectors-marketing/reddit.tsx b/surfsense_web/lib/connectors-marketing/reddit.tsx index 2a930afe9..c76c9984f 100644 --- a/surfsense_web/lib/connectors-marketing/reddit.tsx +++ b/surfsense_web/lib/connectors-marketing/reddit.tsx @@ -315,6 +315,7 @@ export const reddit: ConnectorPageContent = { { label: "Google Maps API", href: "/google-maps" }, { label: "SERP API", href: "/google-search" }, { label: "Web Crawl API", href: "/web-crawl" }, + { label: "Indeed API", href: "/indeed" }, { label: "SurfSense MCP Server", href: "/mcp-server" }, ], }; diff --git a/surfsense_web/lib/connectors-marketing/tiktok.tsx b/surfsense_web/lib/connectors-marketing/tiktok.tsx index 6063fd1ae..9a765ef0e 100644 --- a/surfsense_web/lib/connectors-marketing/tiktok.tsx +++ b/surfsense_web/lib/connectors-marketing/tiktok.tsx @@ -284,6 +284,7 @@ export const tiktok: ConnectorPageContent = { { label: "Google Maps API", href: "/google-maps" }, { label: "SERP API", href: "/google-search" }, { label: "Web Crawl API", href: "/web-crawl" }, + { label: "Indeed API", href: "/indeed" }, { label: "SurfSense MCP Server", href: "/mcp-server" }, ], }; diff --git a/surfsense_web/lib/connectors-marketing/web-crawl.tsx b/surfsense_web/lib/connectors-marketing/web-crawl.tsx index 1cb4d5dac..1684a61e2 100644 --- a/surfsense_web/lib/connectors-marketing/web-crawl.tsx +++ b/surfsense_web/lib/connectors-marketing/web-crawl.tsx @@ -283,6 +283,7 @@ export const webCrawl: ConnectorPageContent = { { label: "Google Maps API", href: "/google-maps" }, { label: "Reddit API", href: "/reddit" }, { label: "Instagram API", href: "/instagram" }, + { label: "Indeed API", href: "/indeed" }, { label: "SurfSense MCP Server", href: "/mcp-server" }, { label: "Read the docs", href: "/docs" }, ], diff --git a/surfsense_web/lib/connectors-marketing/youtube.tsx b/surfsense_web/lib/connectors-marketing/youtube.tsx index fceaee940..431c78357 100644 --- a/surfsense_web/lib/connectors-marketing/youtube.tsx +++ b/surfsense_web/lib/connectors-marketing/youtube.tsx @@ -275,6 +275,7 @@ export const youtube: ConnectorPageContent = { { label: "Google Maps API", href: "/google-maps" }, { label: "SERP API", href: "/google-search" }, { label: "Web Crawl API", href: "/web-crawl" }, + { label: "Indeed API", href: "/indeed" }, { label: "SurfSense MCP Server", href: "/mcp-server" }, ], }; diff --git a/surfsense_web/lib/playground/catalog.ts b/surfsense_web/lib/playground/catalog.ts index 456e52008..a8825e52a 100644 --- a/surfsense_web/lib/playground/catalog.ts +++ b/surfsense_web/lib/playground/catalog.ts @@ -2,6 +2,7 @@ import type { ComponentType } from "react"; import { GoogleMapsIcon, GoogleSearchIcon, + IndeedIcon, InstagramIcon, RedditIcon, TikTokIcon, @@ -89,6 +90,12 @@ export const PLAYGROUND_PLATFORMS: PlaygroundPlatform[] = [ icon: GoogleSearchIcon, verbs: [{ name: "google_search.scrape", verb: "scrape", label: "Scrape" }], }, + { + id: "indeed", + label: "Indeed", + icon: IndeedIcon, + verbs: [{ name: "indeed.scrape", verb: "scrape", label: "Scrape" }], + }, { id: "web", label: "Web", diff --git a/surfsense_web/lib/playground/platform-icons.tsx b/surfsense_web/lib/playground/platform-icons.tsx index c1f61978b..5bae99d1b 100644 --- a/surfsense_web/lib/playground/platform-icons.tsx +++ b/surfsense_web/lib/playground/platform-icons.tsx @@ -28,4 +28,5 @@ export const InstagramIcon = brandIcon("/connectors/instagram.svg", "Instagram") export const TikTokIcon = brandIcon("/connectors/tiktok.svg", "TikTok"); export const GoogleMapsIcon = brandIcon("/connectors/google-maps.svg", "Google Maps"); export const GoogleSearchIcon = brandIcon("/connectors/google-search.svg", "Google Search"); +export const IndeedIcon = brandIcon("/connectors/indeed.svg", "Indeed"); export const WebIcon = brandIcon("/connectors/web.svg", "Web"); diff --git a/surfsense_web/public/connectors/indeed.svg b/surfsense_web/public/connectors/indeed.svg new file mode 100644 index 000000000..39d792bd6 --- /dev/null +++ b/surfsense_web/public/connectors/indeed.svg @@ -0,0 +1,5 @@ + + + + +