mirror of
https://github.com/MODSetter/SurfSense.git
synced 2026-07-20 23:21:06 +02:00
Merge remote-tracking branch 'upstream/dev' into fix/onboarding
This commit is contained in:
commit
2d837ea18c
185 changed files with 4729 additions and 3630 deletions
|
|
@ -66,6 +66,4 @@ def test_details_wraps_profile_items():
|
|||
|
||||
def test_details_rejects_both_sources():
|
||||
with pytest.raises(ValidationError):
|
||||
DetailsInput(
|
||||
urls=["https://www.instagram.com/natgeo/"], search_queries=["x"]
|
||||
)
|
||||
DetailsInput(urls=["https://www.instagram.com/natgeo/"], search_queries=["x"])
|
||||
|
|
|
|||
|
|
@ -54,7 +54,6 @@ async def test_forwards_typed_sources_and_limit():
|
|||
ScrapeInput(
|
||||
profiles=["nasa"],
|
||||
hashtags=["food"],
|
||||
search_queries=["cats"],
|
||||
results_per_page=7,
|
||||
max_items=25,
|
||||
)
|
||||
|
|
@ -63,7 +62,6 @@ async def test_forwards_typed_sources_and_limit():
|
|||
(actor_input, limit) = scraper.calls[0]
|
||||
assert actor_input.profiles == ["nasa"]
|
||||
assert actor_input.hashtags == ["food"]
|
||||
assert actor_input.searchQueries == ["cats"]
|
||||
assert actor_input.resultsPerPage == 7
|
||||
# The outer collection limit is the caller's total-item cap.
|
||||
assert limit == 25
|
||||
|
|
|
|||
|
|
@ -42,8 +42,7 @@ def _profile_payload(n: int) -> dict:
|
|||
"edge_owner_to_timeline_media": {
|
||||
"count": n,
|
||||
"edges": [
|
||||
{"node": {"id": str(i), "shortcode": f"S{i}"}}
|
||||
for i in range(n)
|
||||
{"node": {"id": str(i), "shortcode": f"S{i}"}} for i in range(n)
|
||||
],
|
||||
},
|
||||
}
|
||||
|
|
|
|||
|
|
@ -33,9 +33,7 @@ async def test_google_discovery_keeps_only_profiles(monkeypatch):
|
|||
"https://example.com/not-instagram",
|
||||
),
|
||||
)
|
||||
targets = await scraper._discover(
|
||||
"nat geo photos", search_type="profile", limit=10
|
||||
)
|
||||
targets = await scraper._discover("nat geo photos", search_type="profile", limit=10)
|
||||
assert [(t.kind, t.value) for t in targets] == [("profile", "natgeo")]
|
||||
|
||||
|
||||
|
|
@ -48,9 +46,7 @@ async def test_google_discovery_dedupes(monkeypatch):
|
|||
"https://www.instagram.com/natgeo/",
|
||||
),
|
||||
)
|
||||
targets = await scraper._discover(
|
||||
"nat geo photos", search_type="profile", limit=10
|
||||
)
|
||||
targets = await scraper._discover("nat geo photos", search_type="profile", limit=10)
|
||||
assert len(targets) == 1
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -102,7 +102,9 @@ async def test_warms_then_returns_json():
|
|||
holder = _FakeHolder([_FakeSession(200, csrftoken=True)])
|
||||
token = _current_session.set(holder)
|
||||
try:
|
||||
result = await fetch_json("api/v1/users/web_profile_info/", {"username": "natgeo"})
|
||||
result = await fetch_json(
|
||||
"api/v1/users/web_profile_info/", {"username": "natgeo"}
|
||||
)
|
||||
finally:
|
||||
_current_session.reset(token)
|
||||
assert result == _PAYLOAD
|
||||
|
|
|
|||
|
|
@ -175,8 +175,14 @@ def test_parse_post_prefers_relay_json():
|
|||
"image_versions2": {"candidates": [{"url": "https://cdn/c2.jpg"}]},
|
||||
},
|
||||
],
|
||||
"usertags": {"in": [{"position": [0.5, 0.5], "user": {"username": "tagged1", "id": "77"}}]},
|
||||
"coauthor_producers": [{"username": "coauthor1", "id": "88", "is_verified": True}],
|
||||
"usertags": {
|
||||
"in": [
|
||||
{"position": [0.5, 0.5], "user": {"username": "tagged1", "id": "77"}}
|
||||
]
|
||||
},
|
||||
"coauthor_producers": [
|
||||
{"username": "coauthor1", "id": "88", "is_verified": True}
|
||||
],
|
||||
"location": {"id": "123", "name": "Bali"},
|
||||
}
|
||||
html = (
|
||||
|
|
|
|||
|
|
@ -84,7 +84,9 @@ async def test_warms_then_returns_html():
|
|||
|
||||
|
||||
async def test_rotates_when_warm_fails_then_succeeds():
|
||||
holder = _FakeHolder([_FakeSession(200, warms=False), _FakeSession(200, warms=True)])
|
||||
holder = _FakeHolder(
|
||||
[_FakeSession(200, warms=False), _FakeSession(200, warms=True)]
|
||||
)
|
||||
token = _current_session.set(holder)
|
||||
try:
|
||||
result = await client.fetch_html("https://www.tiktok.com/@scout2015")
|
||||
|
|
@ -119,9 +121,7 @@ async def test_rotates_and_rewarms_on_403():
|
|||
|
||||
async def test_persistent_403_raises_blocked(monkeypatch):
|
||||
_no_sleep(monkeypatch)
|
||||
holder = _FakeHolder(
|
||||
[_FakeSession(403) for _ in range(client._MAX_ROTATIONS + 1)]
|
||||
)
|
||||
holder = _FakeHolder([_FakeSession(403) for _ in range(client._MAX_ROTATIONS + 1)])
|
||||
token = _current_session.set(holder)
|
||||
try:
|
||||
raised = False
|
||||
|
|
|
|||
|
|
@ -6,7 +6,9 @@ from app.proprietary.platforms.tiktok.targets import resolve_target
|
|||
|
||||
|
||||
def test_resolve_video_carries_username_and_id():
|
||||
target = resolve_target("https://www.tiktok.com/@scout2015/video/6718335390845095173")
|
||||
target = resolve_target(
|
||||
"https://www.tiktok.com/@scout2015/video/6718335390845095173"
|
||||
)
|
||||
assert target is not None
|
||||
assert target.kind == "video"
|
||||
assert target.value == "6718335390845095173"
|
||||
|
|
|
|||
|
|
@ -28,9 +28,7 @@ async def test_user_search_parses_dedupes_and_caps():
|
|||
async def fake_fetch(_url: str, _cap: int) -> list[dict]:
|
||||
return [_user("1", "nasa"), _user("1", "nasa"), _user("2", "nasa2")]
|
||||
|
||||
items = await search_tiktok_users(
|
||||
["nasa"], per_query=2, fetch_users=fake_fetch
|
||||
)
|
||||
items = await search_tiktok_users(["nasa"], per_query=2, fetch_users=fake_fetch)
|
||||
|
||||
assert [i["id"] for i in items] == ["1", "2"]
|
||||
first = items[0]
|
||||
|
|
@ -49,9 +47,7 @@ async def test_user_search_empty_query_emits_error_item():
|
|||
async def fake_fetch(_url: str, _cap: int) -> list[dict]:
|
||||
return []
|
||||
|
||||
items = await search_tiktok_users(
|
||||
["ghost"], per_query=5, fetch_users=fake_fetch
|
||||
)
|
||||
items = await search_tiktok_users(["ghost"], per_query=5, fetch_users=fake_fetch)
|
||||
|
||||
assert len(items) == 1
|
||||
assert items[0]["errorCode"] == "no_users"
|
||||
|
|
|
|||
|
|
@ -64,6 +64,22 @@ def test_openai_compatible_resolver_uses_explicit_api_base() -> None:
|
|||
assert ensure_v1("http://example.com/v1") == "http://example.com/v1"
|
||||
|
||||
|
||||
def test_openai_compatible_raw_resolver_does_not_append_v1() -> None:
|
||||
model, kwargs = to_litellm(
|
||||
{
|
||||
"provider": "openai_compatible_raw",
|
||||
"base_url": "https://ark.cn-beijing.volces.com/api/v3",
|
||||
"api_key": "ark-key",
|
||||
"extra": {},
|
||||
},
|
||||
"ep-20260101000000-test",
|
||||
)
|
||||
|
||||
assert model == "openai/ep-20260101000000-test"
|
||||
assert kwargs["api_base"] == "https://ark.cn-beijing.volces.com/api/v3"
|
||||
assert kwargs["api_key"] == "ark-key"
|
||||
|
||||
|
||||
def test_ollama_resolver_uses_native_api_base() -> None:
|
||||
model, kwargs = to_litellm(
|
||||
{
|
||||
|
|
|
|||
|
|
@ -0,0 +1,103 @@
|
|||
"""Unit tests for Requesty model normalization.
|
||||
|
||||
Mirrors the OpenRouter normalizer coverage but exercises Requesty's flat
|
||||
boolean capability fields (``supports_tool_calling`` / ``supports_vision``)
|
||||
and ``context_window`` sizing.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
from app.services.requesty_model_normalizer import (
|
||||
is_requesty_chat_model,
|
||||
is_requesty_image_model,
|
||||
normalize_requesty_models,
|
||||
supports_image_input,
|
||||
supports_tool_calling,
|
||||
)
|
||||
|
||||
pytestmark = pytest.mark.unit
|
||||
|
||||
|
||||
def _requesty_model(
|
||||
*,
|
||||
model_id: str,
|
||||
context_window: int = 128_000,
|
||||
tools: bool = True,
|
||||
vision: bool = False,
|
||||
image_generation: bool = False,
|
||||
name: str | None = None,
|
||||
) -> dict:
|
||||
"""Return a synthetic Requesty ``/v1/models`` entry.
|
||||
|
||||
Only the fields the normalizer inspects are populated; the live payload
|
||||
carries many more (pricing, ``supports_caching``, ``description``, ...).
|
||||
"""
|
||||
return {
|
||||
"id": model_id,
|
||||
"name": name or model_id,
|
||||
"api": "chat",
|
||||
"object": "model",
|
||||
"context_window": context_window,
|
||||
"supports_tool_calling": tools,
|
||||
"supports_vision": vision,
|
||||
"supports_image_generation": image_generation,
|
||||
}
|
||||
|
||||
|
||||
def test_chat_model_requires_slash_tools_and_context():
|
||||
assert is_requesty_chat_model(_requesty_model(model_id="openai/gpt-4o-mini"))
|
||||
assert not is_requesty_chat_model(
|
||||
_requesty_model(model_id="openai/gpt-4o-mini", tools=False)
|
||||
)
|
||||
assert not is_requesty_chat_model(
|
||||
_requesty_model(model_id="openai/gpt-4o-mini", context_window=8_000)
|
||||
)
|
||||
assert not is_requesty_chat_model(_requesty_model(model_id="bare-model"))
|
||||
|
||||
|
||||
def test_excluded_provider_slug_is_filtered():
|
||||
assert not is_requesty_chat_model(_requesty_model(model_id="amazon/nova-pro-v1"))
|
||||
|
||||
|
||||
def test_image_generation_models_excluded_from_chat_and_flagged():
|
||||
image_model = _requesty_model(
|
||||
model_id="google/gemini-2.5-flash-image", image_generation=True
|
||||
)
|
||||
assert not is_requesty_chat_model(image_model)
|
||||
assert is_requesty_image_model(image_model)
|
||||
|
||||
|
||||
def test_capability_helpers_read_flat_booleans():
|
||||
model = _requesty_model(
|
||||
model_id="anthropic/claude-sonnet-4-5", vision=True, tools=True
|
||||
)
|
||||
assert supports_image_input(model) is True
|
||||
assert supports_tool_calling(model) is True
|
||||
|
||||
|
||||
def test_normalize_maps_context_window_and_capabilities():
|
||||
normalized = normalize_requesty_models(
|
||||
[
|
||||
_requesty_model(
|
||||
model_id="openai/gpt-4o-mini",
|
||||
context_window=128_000,
|
||||
vision=True,
|
||||
name="GPT-4o mini",
|
||||
),
|
||||
_requesty_model(model_id="openai/gpt-4o-mini", tools=False),
|
||||
_requesty_model(model_id="black-forest-labs/flux", image_generation=True),
|
||||
]
|
||||
)
|
||||
|
||||
assert len(normalized) == 1
|
||||
entry = normalized[0]
|
||||
assert entry["model_id"] == "openai/gpt-4o-mini"
|
||||
assert entry["display_name"] == "GPT-4o mini"
|
||||
assert entry["supports_chat"] is True
|
||||
assert entry["max_input_tokens"] == 128_000
|
||||
assert entry["supports_image_input"] is True
|
||||
assert entry["supports_tools"] is True
|
||||
assert entry["supports_image_generation"] is False
|
||||
assert entry["metadata"]["id"] == "openai/gpt-4o-mini"
|
||||
Loading…
Add table
Add a link
Reference in a new issue