mirror of
https://github.com/MODSetter/SurfSense.git
synced 2026-07-24 23:41:10 +02:00
feat(walmart): add url resolver
This commit is contained in:
parent
a72cc3436c
commit
be1fd0491f
1 changed files with 75 additions and 0 deletions
|
|
@ -0,0 +1,75 @@
|
|||
"""Classify a Walmart start URL and extract its identifiers.
|
||||
|
||||
Covers the shapes accepted on input: product pages (``/ip/{slug}/{id}`` or
|
||||
``/ip/{id}``), search (``/search?q=``), category (``/cp/{slug}/{id}``), and
|
||||
browse (``/browse/...``) pages. Category and browse render the same
|
||||
``searchResult`` JSON as search, so they share the ``listing`` kind and parser.
|
||||
|
||||
Pure, no I/O. ``item_id`` is Walmart's ``usItemId`` — the numeric id the review
|
||||
page and detail page are both keyed on.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from dataclasses import dataclass
|
||||
from typing import Literal
|
||||
from urllib.parse import parse_qs, urlparse
|
||||
|
||||
ResolvedKind = Literal["product", "listing"]
|
||||
|
||||
# usItemId: the trailing numeric segment of an /ip/ or /reviews/product/ path,
|
||||
# or a bare numeric id passed directly.
|
||||
_ITEM_ID_PATH_RE = re.compile(r"/(?:ip|reviews/product)/(?:[^/?#]+/)*(\d{4,})")
|
||||
_BARE_ID_RE = re.compile(r"^\d{4,}$")
|
||||
|
||||
_LISTING_QUERY_KEYS = ("q", "cat_id", "browse")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ResolvedUrl:
|
||||
kind: ResolvedKind
|
||||
url: str
|
||||
item_id: str | None = None
|
||||
domain: str | None = None
|
||||
|
||||
|
||||
def extract_item_id(url: str) -> str | None:
|
||||
"""Pull the numeric ``usItemId`` out of a product/review URL or bare id."""
|
||||
candidate = url.strip()
|
||||
if _BARE_ID_RE.match(candidate):
|
||||
return candidate
|
||||
match = _ITEM_ID_PATH_RE.search(candidate)
|
||||
return match.group(1) if match else None
|
||||
|
||||
|
||||
def resolve_url(url: str) -> ResolvedUrl | None:
|
||||
"""Classify a Walmart start URL into a scrape job, or ``None`` if unrecognized.
|
||||
|
||||
A bare numeric id resolves to a product (its ``usItemId``) so callers can
|
||||
pass item ids directly without constructing a full URL.
|
||||
"""
|
||||
if _BARE_ID_RE.match(url.strip()):
|
||||
return ResolvedUrl("product", url.strip(), item_id=url.strip())
|
||||
|
||||
parsed = urlparse(url)
|
||||
host = (parsed.hostname or "").lower()
|
||||
path = parsed.path or ""
|
||||
|
||||
if "walmart." not in host:
|
||||
return None
|
||||
|
||||
if path.startswith("/ip/"):
|
||||
return ResolvedUrl("product", url, item_id=extract_item_id(path), domain=host)
|
||||
|
||||
query = parse_qs(parsed.query)
|
||||
is_listing = (
|
||||
path.startswith("/search")
|
||||
or path.startswith("/cp/")
|
||||
or path.startswith("/browse/")
|
||||
or any(key in query for key in _LISTING_QUERY_KEYS)
|
||||
)
|
||||
if is_listing:
|
||||
return ResolvedUrl("listing", url, domain=host)
|
||||
|
||||
return None
|
||||
Loading…
Add table
Add a link
Reference in a new issue