Files
Medios-Macina/cmdlet/file/search_engines.py
T

995 lines
30 KiB
Python

"""Search engine result parsing — Bing, DuckDuckGo, Yahoo, and site crawling.
Extracted from search.py to keep the cmdlet class focused on orchestration.
"""
from __future__ import annotations
from typing import Any, Dict, List, Optional, Sequence
from collections import deque
from pathlib import Path
import re
import time
import html as _html
from urllib.parse import urlparse, parse_qs, unquote, urljoin
from SYS.logger import debug
from SYS.payload_builders import normalize_file_extension
_WHITESPACE_RE = re.compile(r"\s+")
_SITE_TOKEN_RE = re.compile(r"(?:^|\s)site:([^\s,]+)", flags=re.IGNORECASE)
_FILETYPE_TOKEN_RE = re.compile(
r"(?:^|\s)(?:ext|filetype|type):\.?([a-z0-9]{1,12})\b",
flags=re.IGNORECASE,
)
_SITE_REMOVE_RE = re.compile(r"(?:^|\s)site:[^\s,]+", flags=re.IGNORECASE)
_FILETYPE_REMOVE_RE = re.compile(
r"(?:^|\s)(?:ext|filetype|type):\.?[a-z0-9]{1,12}\b",
flags=re.IGNORECASE,
)
_SCHEME_PREFIX_RE = re.compile(r"^[a-z]+:")
_YAHOO_RU_RE = re.compile(r"/RU=([^/]+)/RK=", flags=re.IGNORECASE)
_HTML_TAG_RE = re.compile(r"<[^>]+>")
_DDG_RESULT_ANCHOR_RE = re.compile(
r'<a[^>]+class="[^"]*result__a[^"]*"[^>]+href="([^"]+)"[^>]*>(.*?)</a>',
flags=re.IGNORECASE | re.DOTALL,
)
_GENERIC_ANCHOR_RE = re.compile(
r'<a[^>]+href=["\']([^"\']+)["\'][^>]*>(.*?)</a>',
flags=re.IGNORECASE | re.DOTALL,
)
_BING_RESULT_ANCHOR_RE = re.compile(
r'<h2[^>]*>\s*<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>',
flags=re.IGNORECASE | re.DOTALL,
)
def _normalize_extension(ext_value: Any) -> str:
"""Sanitize extension strings to alphanumerics and cap at 5 chars."""
return normalize_file_extension(ext_value)
def _normalize_host(value: Any) -> str:
"""Normalize host names for matching/filtering."""
host = str(value or "").strip().lower()
if host.startswith("www."):
host = host[4:]
if ":" in host:
host = host.split(":", 1)[0]
return host
def _normalize_space(text: Any) -> str:
return _WHITESPACE_RE.sub(" ", str(text or "")).strip()
def _url_matches_site(cls, url: str, site_host: str) -> bool:
"""Return True when URL host is the requested site/subdomain."""
try:
parsed = urlparse(str(url or ""))
host = _normalize_host(getattr(parsed, "hostname", "") or "")
except Exception:
return False
target = _normalize_host(site_host)
if not host or not target:
return False
return host == target or host.endswith(f".{target}")
def _itertext_join(node: Any) -> str:
try:
return " ".join([str(text).strip() for text in node.itertext() if str(text).strip()])
except Exception:
return ""
def _html_fragment_to_text(fragment: Any) -> str:
text = _HTML_TAG_RE.sub(" ", str(fragment or ""))
return _html.unescape(text)
def _append_web_result(
cls,
items: List[Dict[str, str]],
seen_urls: set[str],
*,
site_host: str,
url_text: str,
title_text: str,
snippet_text: str,
) -> None:
url_clean = str(url_text or "").strip()
if not url_clean or not url_clean.startswith(("http://", "https://")):
return
if not _url_matches_site(url_clean, site_host):
return
if url_clean in seen_urls:
return
seen_urls.add(url_clean)
items.append(
{
"url": url_clean,
"title": _normalize_space(title_text) or url_clean,
"snippet": _normalize_space(snippet_text),
}
)
def _parse_web_results_with_fallback(
cls,
*,
html_text: str,
limit: int,
lxml_parser: Any,
regex_parser: Any,
fallback_when_empty: bool = False,
) -> List[Dict[str, str]]:
"""Run an lxml-based parser with an optional regex fallback."""
items: List[Dict[str, str]] = []
seen_urls: set[str] = set()
should_run_regex = False
try:
from lxml import html as lxml_html
doc = lxml_html.fromstring(html_text or "")
lxml_parser(doc, items, seen_urls)
should_run_regex = fallback_when_empty and not items
except Exception:
should_run_regex = True
if should_run_regex:
regex_parser(html_text or "", items, seen_urls)
return items[:limit]
def _extract_duckduckgo_target_url(href: Any) -> str:
"""Extract direct target URL from DuckDuckGo result links."""
raw_href = str(href or "").strip()
if not raw_href:
return ""
if raw_href.startswith("//"):
raw_href = f"https:{raw_href}"
if raw_href.startswith("/"):
raw_href = f"https://duckduckgo.com{raw_href}"
parsed = None
try:
parsed = urlparse(raw_href)
except Exception:
parsed = None
try:
host = str(getattr(parsed, "hostname", "") or "").strip().lower()
except Exception:
host = ""
if host.endswith("duckduckgo.com"):
try:
query = parse_qs(str(getattr(parsed, "query", "") or ""))
candidate = (query.get("uddg") or [""])[0]
if candidate:
return str(unquote(candidate)).strip()
except Exception:
pass
return raw_href
def _extract_yahoo_target_url(href: Any) -> str:
"""Extract direct target URL from Yahoo redirect links."""
raw_href = str(href or "").strip()
if not raw_href:
return ""
ru_match = _YAHOO_RU_RE.search(raw_href)
if ru_match:
try:
return str(unquote(ru_match.group(1))).strip()
except Exception:
pass
try:
parsed = urlparse(raw_href)
query = parse_qs(str(getattr(parsed, "query", "") or ""))
candidate = (query.get("RU") or query.get("ru") or [""])[0]
if candidate:
return str(unquote(candidate)).strip()
except Exception:
pass
return raw_href
def parse_duckduckgo_results(
cls,
*,
html_text: str,
site_host: str,
limit: int,
) -> List[Dict[str, str]]:
"""Parse DuckDuckGo HTML results into normalized rows."""
def _parse_lxml(doc: Any, items: List[Dict[str, str]], seen_urls: set[str]) -> None:
result_nodes = doc.xpath("//div[contains(@class, 'result')]")
for node in result_nodes:
links = node.xpath(".//a[contains(@class, 'result__a')]")
if not links:
continue
link = links[0]
href = _extract_duckduckgo_target_url(link.get("href"))
title = _itertext_join(link)
snippet_nodes = node.xpath(".//*[contains(@class, 'result__snippet')]")
snippet = ""
if snippet_nodes:
snippet = _itertext_join(snippet_nodes[0])
_append_web_result(
cls,
items,
seen_urls,
site_host=site_host,
url_text=href,
title_text=title,
snippet_text=snippet,
)
if len(items) >= limit:
break
def _parse_regex(raw_html: str, items: List[Dict[str, str]], seen_urls: set[str]) -> None:
for match in _DDG_RESULT_ANCHOR_RE.finditer(raw_html):
href = _extract_duckduckgo_target_url(match.group(1))
title_html = match.group(2)
title = _html_fragment_to_text(title_html)
_append_web_result(
cls,
items,
seen_urls,
site_host=site_host,
url_text=href,
title_text=title,
snippet_text="",
)
if len(items) >= limit:
break
return _parse_web_results_with_fallback(
cls,
html_text=html_text,
limit=limit,
lxml_parser=_parse_lxml,
regex_parser=_parse_regex,
fallback_when_empty=True,
)
def parse_yahoo_results(
cls,
*,
html_text: str,
site_host: str,
limit: int,
) -> List[Dict[str, str]]:
"""Parse Yahoo HTML search results into normalized rows."""
def _parse_lxml(doc: Any, items: List[Dict[str, str]], seen_urls: set[str]) -> None:
for node in doc.xpath("//a[@href]"):
href = _extract_yahoo_target_url(node.get("href"))
title = _itertext_join(node)
_append_web_result(
cls,
items,
seen_urls,
site_host=site_host,
url_text=href,
title_text=title,
snippet_text="",
)
if len(items) >= limit:
break
def _parse_regex(raw_html: str, items: List[Dict[str, str]], seen_urls: set[str]) -> None:
for match in _GENERIC_ANCHOR_RE.finditer(raw_html):
href = _extract_yahoo_target_url(match.group(1))
title_html = match.group(2)
title = _html_fragment_to_text(title_html)
_append_web_result(
cls,
items,
seen_urls,
site_host=site_host,
url_text=href,
title_text=title,
snippet_text="",
)
if len(items) >= limit:
break
return _parse_web_results_with_fallback(
cls,
html_text=html_text,
limit=limit,
lxml_parser=_parse_lxml,
regex_parser=_parse_regex,
)
def query_yahoo(
cls,
*,
search_query: str,
site_host: str,
limit: int,
session: Any,
deadline: Optional[float] = None,
) -> List[Dict[str, str]]:
"""Fetch results from Yahoo search (robust fallback in bot-protected envs)."""
all_rows: List[Dict[str, str]] = []
seen_urls: set[str] = set()
max_pages = max(1, min((max(1, int(limit or 1)) + 9) // 10, 3))
for page_idx in range(max_pages):
if deadline is not None and time.monotonic() >= deadline:
break
params = {
"p": search_query,
"n": "10",
"b": str((page_idx * 10) + 1),
}
try:
read_timeout = 10.0
if deadline is not None:
remaining = max(0.0, float(deadline - time.monotonic()))
if remaining <= 0.0:
break
read_timeout = max(3.0, min(10.0, remaining))
response = session.get(
"https://search.yahoo.com/search",
params=params,
timeout=(3, read_timeout),
headers={
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/124.0.0.0 Safari/537.36"
),
"Accept-Language": "en-US,en;q=0.9",
},
)
response.raise_for_status()
except Exception:
break
page_rows = parse_yahoo_results(
cls,
html_text=response.text,
site_host=site_host,
limit=max(1, limit - len(all_rows)),
)
new_rows = 0
for row in page_rows:
url_value = str(row.get("url") or "").strip()
if not url_value or url_value in seen_urls:
continue
seen_urls.add(url_value)
all_rows.append(row)
new_rows += 1
if len(all_rows) >= limit:
break
if len(all_rows) >= limit or new_rows == 0:
break
return all_rows[:limit]
def parse_bing_results(
cls,
*,
html_text: str,
site_host: str,
limit: int,
) -> List[Dict[str, str]]:
"""Parse Bing HTML search results into normalized rows."""
def _parse_lxml(doc: Any, items: List[Dict[str, str]], seen_urls: set[str]) -> None:
result_nodes = doc.xpath("//li[contains(@class, 'b_algo')]")
for node in result_nodes:
links = node.xpath(".//h2/a")
if not links:
continue
link = links[0]
href = str(link.get("href") or "").strip()
title = _itertext_join(link)
snippet = ""
for sel in (
".//*[contains(@class,'b_caption')]//p",
".//*[contains(@class,'b_snippet')]",
".//p",
):
snip_nodes = node.xpath(sel)
if snip_nodes:
snippet = _itertext_join(snip_nodes[0])
break
_append_web_result(
cls,
items,
seen_urls,
site_host=site_host,
url_text=href,
title_text=title,
snippet_text=snippet,
)
if len(items) >= limit:
break
def _parse_regex(raw_html: str, items: List[Dict[str, str]], seen_urls: set[str]) -> None:
for match in _BING_RESULT_ANCHOR_RE.finditer(raw_html):
href = match.group(1)
title = _html_fragment_to_text(match.group(2))
_append_web_result(
cls,
items,
seen_urls,
site_host=site_host,
url_text=href,
title_text=title,
snippet_text="",
)
if len(items) >= limit:
break
return _parse_web_results_with_fallback(
cls,
html_text=html_text,
limit=limit,
lxml_parser=_parse_lxml,
regex_parser=_parse_regex,
)
def query_bing(
cls,
*,
search_query: str,
site_host: str,
limit: int,
session: Any,
deadline: Optional[float] = None,
) -> List[Dict[str, str]]:
"""Fetch results from Bing (supports filetype: and site: natively)."""
all_rows: List[Dict[str, str]] = []
seen_urls: set[str] = set()
page_start = 1
pages_checked = 0
max_pages = max(1, min((max(1, int(limit or 1)) + 49) // 50, 3))
while len(all_rows) < limit and pages_checked < max_pages:
if deadline is not None and time.monotonic() >= deadline:
break
params = {"q": search_query, "first": str(page_start), "count": "50"}
try:
read_timeout = 10.0
if deadline is not None:
remaining = max(0.0, float(deadline - time.monotonic()))
if remaining <= 0.0:
break
read_timeout = max(3.0, min(10.0, remaining))
response = session.get(
"https://www.bing.com/search",
params=params,
timeout=(3, read_timeout),
headers={
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/124.0.0.0 Safari/537.36"
),
"Accept-Language": "en-US,en;q=0.9",
},
)
response.raise_for_status()
except Exception:
break
page_rows = parse_bing_results(
cls,
html_text=response.text,
site_host=site_host,
limit=max(1, limit - len(all_rows)),
)
new_rows = 0
for row in page_rows:
url_value = str(row.get("url") or "").strip()
if not url_value or url_value in seen_urls:
continue
seen_urls.add(url_value)
all_rows.append(row)
new_rows += 1
if len(all_rows) >= limit:
break
if new_rows == 0 or len(all_rows) >= limit:
break
page_start += 50
pages_checked += 1
return all_rows
def query_web_search(
cls,
*,
search_query: str,
site_host: str,
limit: int,
) -> List[Dict[str, str]]:
"""Execute web search and return parsed result rows.
Uses Yahoo first (works in environments where Bing/DDG HTML endpoints
are challenge-gated), then Bing, then DuckDuckGo.
"""
from API.requests_client import get_requests_session
session = get_requests_session()
normalized_limit = max(1, min(int(limit or 1), 100))
engine_deadline = time.monotonic() + 12.0
all_rows = query_yahoo(
cls,
search_query=search_query,
site_host=site_host,
limit=normalized_limit,
session=session,
deadline=engine_deadline,
)
if all_rows:
return all_rows[:normalized_limit]
all_rows = query_bing(
cls,
search_query=search_query,
site_host=site_host,
limit=normalized_limit,
session=session,
deadline=engine_deadline,
)
if all_rows:
return all_rows[:normalized_limit]
all_rows_ddg: List[Dict[str, str]] = []
seen_urls: set[str] = set()
endpoints = [
"https://html.duckduckgo.com/html/",
"https://duckduckgo.com/html/",
]
for endpoint in endpoints:
if time.monotonic() >= engine_deadline:
break
max_offsets = min(3, max(1, (normalized_limit + 29) // 30))
for page_idx in range(max_offsets):
if time.monotonic() >= engine_deadline:
break
offset = page_idx * 30
params = {"q": search_query, "s": str(offset)}
remaining = max(0.0, float(engine_deadline - time.monotonic()))
if remaining <= 0.0:
break
read_timeout = max(3.0, min(10.0, remaining))
response = session.get(
endpoint,
params=params,
timeout=(3, read_timeout),
headers={"Referer": "https://duckduckgo.com/"},
)
response.raise_for_status()
page_rows = parse_duckduckgo_results(
cls,
html_text=response.text,
site_host=site_host,
limit=max(1, normalized_limit - len(all_rows_ddg)),
)
new_rows = 0
for row in page_rows:
url_value = str(row.get("url") or "").strip()
if not url_value or url_value in seen_urls:
continue
seen_urls.add(url_value)
all_rows_ddg.append(row)
new_rows += 1
if len(all_rows_ddg) >= normalized_limit:
break
if len(all_rows_ddg) >= normalized_limit or new_rows == 0:
break
if all_rows_ddg:
break
return all_rows_ddg[:normalized_limit]
def _is_probable_html_path(path_value: str) -> bool:
"""Return True when URL path likely points to an HTML page."""
path = str(path_value or "").strip()
if not path:
return True
suffix = Path(path).suffix.lower()
if not suffix:
return True
return suffix in {".html", ".htm", ".php", ".asp", ".aspx", ".jsp", ".shtml", ".xhtml"}
def _extract_html_links(cls, *, html_text: str, base_url: str) -> List[str]:
"""Extract absolute links from an HTML document."""
links: List[str] = []
seen: set[str] = set()
def _add_link(raw_href: Any) -> None:
href = str(raw_href or "").strip()
if not href or href.startswith(("#", "javascript:", "mailto:")):
return
try:
absolute = urljoin(base_url, href)
parsed = urlparse(absolute)
except Exception:
return
if str(getattr(parsed, "scheme", "") or "").lower() not in {"http", "https"}:
return
clean = parsed._replace(fragment="").geturl()
if clean in seen:
return
seen.add(clean)
links.append(clean)
try:
from lxml import html as lxml_html
doc = lxml_html.fromstring(html_text or "")
for node in doc.xpath("//a[@href]"):
_add_link(node.get("href"))
except Exception:
href_pattern = re.compile(r'<a[^>]+href=["\']([^"\']+)["\']', flags=re.IGNORECASE)
for match in href_pattern.finditer(html_text or ""):
_add_link(match.group(1))
return links
def crawl_site_for_extension(
cls,
*,
seed_url: str,
site_host: str,
extension: str,
limit: int,
max_duration_seconds: float = 15.0,
) -> List[Dict[str, str]]:
"""Fallback crawler that discovers in-site file links by extension."""
from API.requests_client import get_requests_session
normalized_ext = _normalize_extension(extension)
if not normalized_ext:
return []
start_url = _normalize_seed_url(cls, seed_url, site_host)
if not start_url:
return []
session = get_requests_session()
headers = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/124.0.0.0 Safari/537.36"
),
"Accept-Language": "en-US,en;q=0.9",
}
queue: deque[str] = deque([start_url])
queued: set[str] = {start_url}
visited_pages: set[str] = set()
seen_files: set[str] = set()
rows: List[Dict[str, str]] = []
normalized_limit = max(1, min(int(limit or 1), 100))
max_pages = max(8, min(normalized_limit * 4, 64))
crawl_deadline = time.monotonic() + max(5.0, float(max_duration_seconds or 0.0))
while (
queue
and len(visited_pages) < max_pages
and len(rows) < normalized_limit
and time.monotonic() < crawl_deadline
):
page_url = queue.popleft()
queued.discard(page_url)
if page_url in visited_pages:
continue
visited_pages.add(page_url)
if time.monotonic() >= crawl_deadline:
break
try:
response = session.get(page_url, timeout=(4, 8), headers=headers)
response.raise_for_status()
except Exception:
continue
final_url = str(getattr(response, "url", "") or page_url)
try:
parsed_final = urlparse(final_url)
except Exception:
continue
final_host = _normalize_host(getattr(parsed_final, "hostname", "") or "")
if not _url_matches_site(final_url, site_host):
continue
final_path = str(getattr(parsed_final, "path", "") or "")
direct_ext = _normalize_extension(Path(final_path).suffix)
if direct_ext == normalized_ext:
file_url = parsed_final._replace(fragment="").geturl()
if file_url not in seen_files:
seen_files.add(file_url)
title = Path(unquote(final_path)).name or file_url
rows.append(
{
"url": file_url,
"title": title,
"snippet": "Discovered via in-site crawl",
}
)
continue
content_type = str((response.headers or {}).get("content-type", "") or "").lower()
if "html" not in content_type and "xhtml" not in content_type:
continue
html_text = str(getattr(response, "text", "") or "")
if not html_text:
continue
if len(html_text) > 2_500_000:
continue
discovered_links = _extract_html_links(cls, html_text=html_text, base_url=final_url)
for idx, target in enumerate(discovered_links):
if len(rows) >= normalized_limit:
break
if idx >= 300:
break
if time.monotonic() >= crawl_deadline:
break
try:
parsed_target = urlparse(target)
except Exception:
continue
target_host = _normalize_host(getattr(parsed_target, "hostname", "") or "")
if not target_host or not (target_host == final_host or target_host.endswith(f".{site_host}")):
if not _url_matches_site(target, site_host):
continue
target_clean = parsed_target._replace(fragment="").geturl()
target_path = str(getattr(parsed_target, "path", "") or "")
target_ext = _normalize_extension(Path(target_path).suffix)
if target_ext == normalized_ext:
if target_clean in seen_files:
continue
seen_files.add(target_clean)
title = Path(unquote(target_path)).name or target_clean
rows.append(
{
"url": target_clean,
"title": title,
"snippet": f"Discovered via crawl from {final_path or '/'}",
}
)
continue
if _is_probable_html_path(target_path):
if target_clean not in visited_pages and target_clean not in queued:
queue.append(target_clean)
queued.add(target_clean)
if time.monotonic() >= crawl_deadline:
debug(
"Web crawl fallback reached time budget",
{
"site": site_host,
"visited_pages": len(visited_pages),
"queued_pages": len(queue),
"results": len(rows),
"time_budget_seconds": max_duration_seconds,
},
)
return rows[:normalized_limit]
def _extract_site_host(cls, candidate: Any) -> Optional[str]:
"""Extract a host/domain from URL-like input."""
raw = str(candidate or "").strip().strip('"').strip("'")
if not raw:
return None
if raw.lower().startswith("site:"):
raw = raw.split(":", 1)[1].strip()
parsed = None
try:
parsed = urlparse(raw)
except Exception:
parsed = None
if parsed is None or not getattr(parsed, "hostname", None):
try:
parsed = urlparse(f"https://{raw}")
except Exception:
parsed = None
host = ""
try:
host = str(getattr(parsed, "hostname", "") or "").strip().lower()
except Exception:
host = ""
host = _normalize_host(host)
if not host or "." not in host:
return None
return host
def _normalize_seed_url(cls, seed_value: Any, site_host: str) -> str:
"""Build a safe crawl starting URL from user input and resolved host."""
raw = str(seed_value or "").strip().strip("'\"")
if not raw:
raw = str(site_host or "").strip()
if raw and not raw.startswith(("http://", "https://")):
raw = f"https://{raw}"
try:
parsed = urlparse(raw)
except Exception:
parsed = urlparse("")
target = _normalize_host(site_host)
host = _normalize_host(getattr(parsed, "hostname", "") or "")
if target and host and not (host == target or host.endswith(f".{target}")):
return f"https://{target}/"
scheme = str(getattr(parsed, "scheme", "") or "https").lower()
if scheme not in {"http", "https"}:
scheme = "https"
netloc = str(getattr(parsed, "netloc", "") or "").strip()
if not netloc:
netloc = target
path = str(getattr(parsed, "path", "") or "").strip()
if not path:
path = "/"
return f"{scheme}://{netloc}{path}"
def build_web_search_plan(
cls,
*,
query: str,
positional_args: List[str],
storage_backend: Optional[str],
store_filter: Optional[str],
hash_query: List[str],
) -> Optional[Dict[str, Any]]:
"""Build web-search plan for URL + ext/filetype query syntax.
Example input:
search-file "example.com/foo" -query "ext:pdf"
Produces:
site:example.com filetype:pdf
"""
if storage_backend or store_filter or hash_query:
return None
text = _normalize_space(query)
if not text:
return None
local_markers = ("url:", "hash:", "tag:", "instance:", "system:")
if any(marker in text.lower() for marker in local_markers):
return None
site_host: Optional[str] = None
site_from_positional = False
site_token_to_strip = ""
seed_url = ""
site_match = _SITE_TOKEN_RE.search(text)
if site_match:
site_host = _extract_site_host(cls, site_match.group(1))
seed_url = str(site_match.group(1) or "").strip()
if not site_host and positional_args:
site_host = _extract_site_host(cls, positional_args[0])
site_from_positional = bool(site_host)
if site_from_positional:
site_token_to_strip = str(positional_args[0] or "").strip()
seed_url = site_token_to_strip
if not site_host:
for token in text.split():
candidate = str(token or "").strip().strip(",")
if not candidate:
continue
lower_candidate = candidate.lower()
if lower_candidate.startswith(("ext:", "filetype:", "type:", "site:")):
continue
if _SCHEME_PREFIX_RE.match(lower_candidate) and not lower_candidate.startswith(
("http://", "https://")
):
continue
guessed = _extract_site_host(cls, candidate)
if guessed:
site_host = guessed
site_token_to_strip = candidate
break
if not site_host:
return None
filetype_match = _FILETYPE_TOKEN_RE.search(text)
filetype = _normalize_extension(filetype_match.group(1)) if filetype_match else ""
has_explicit_site = bool(site_match)
if not filetype and not has_explicit_site:
return None
residual = text
residual = _SITE_REMOVE_RE.sub(" ", residual)
residual = _FILETYPE_REMOVE_RE.sub(" ", residual)
if site_from_positional and positional_args:
first = str(positional_args[0] or "").strip()
if first:
residual = re.sub(rf"(?:^|\s){re.escape(first)}(?:\s|$)", " ", residual, count=1)
elif site_token_to_strip:
residual = re.sub(
rf"(?:^|\s){re.escape(site_token_to_strip)}(?:\s|$)",
" ",
residual,
count=1,
)
residual = _normalize_space(residual)
search_terms: List[str] = [f"site:{site_host}"]
if filetype:
search_terms.append(f"filetype:{filetype}")
if residual:
search_terms.append(residual)
search_query = " ".join(search_terms).strip()
if not search_query:
return None
normalized_seed_url = _normalize_seed_url(cls, seed_url, site_host)
return {
"site_host": site_host,
"filetype": filetype,
"search_query": search_query,
"residual": residual,
"seed_url": normalized_seed_url,
}