diff --git a/tests/tools/test_web_result_cache.py b/tests/tools/test_web_result_cache.py index 0f28132a07..94db38651b 100644 --- a/tests/tools/test_web_result_cache.py +++ b/tests/tools/test_web_result_cache.py @@ -196,6 +196,36 @@ def test_extract_cache_oversized_page_not_indexed(_isolated_cache): assert extract_cache_get("https://big.com") is None +@pytest.mark.parametrize("url", [ + "http://localhost:3000/app", + "http://localhost:5173", # vite dev server + "http://127.0.0.1:8080/preview", + "http://[::1]:3000/", + "http://192.168.1.44/dashboard", + "http://10.0.0.5:8000/api/docs", + "http://172.16.0.9/", + "http://myapp.local/", + "http://devbox/page", # single-label LAN name + "http://preview.localhost/artifact", +]) +def test_extract_cache_never_caches_local_dev_urls(url, _isolated_cache): + """Local/private URLs are dev servers and chat-GUI artifact previews — + they change on every save, so freshness beats dedup. Neither put nor + get may touch the cache for them.""" + extract_cache_put(url, "stale build output") + assert extract_cache_get(url) is None + + +@pytest.mark.parametrize("url", [ + "https://example.com/page", + "https://docs.python.org/3/", +]) +def test_extract_cache_public_urls_still_cache(url, _isolated_cache): + extract_cache_put(url, "public content") + hit = extract_cache_get(url) + assert hit is not None and hit["content"] == "public content" + + def test_extract_cache_tampered_index_path_is_miss(_isolated_cache, tmp_path): """An index entry pointing outside cache/web must never be read.""" outside = tmp_path / "outside.md" diff --git a/tools/web_result_cache.py b/tools/web_result_cache.py index 9799e34612..d5b8371e38 100644 --- a/tools/web_result_cache.py +++ b/tools/web_result_cache.py @@ -275,6 +275,43 @@ def _entry_file_path(url: str, format: Optional[str], provider: str) -> Optional return d / f"{slug}-{_url_digest(url, format, provider)}.cache.md" +def _is_local_dev_url(url: str) -> bool: + """True for loopback/private/LAN URLs — never cached. + + A page on a private address is one the user controls and is typically + changing fast (dev servers, hot reload, chat-GUI artifact previews, + LAN preview apps). Freshness is the point of fetching it, so the cache + declines these entirely rather than serving a stale build for a whole + TTL. Only reachable when ``security.allow_private_urls`` is enabled — + default installs SSRF-block these URLs before extraction anyway. + + Hostname heuristics only (no DNS resolution — this is a freshness + decision, not a security boundary; SSRF enforcement lives in + tools/url_safety.py). + """ + try: + from urllib.parse import urlparse + host = (urlparse(url).hostname or "").strip("[]").lower() + if not host: + return True # unparseable → don't cache + if host == "localhost" or host.endswith(".localhost") or host.endswith(".local"): + return True + # Single-label hostnames (no dot) are LAN names, not public DNS. + if "." not in host and ":" not in host: + return True + import ipaddress + try: + ip = ipaddress.ip_address(host) + except ValueError: + return False # public DNS name + return bool( + ip.is_private or ip.is_loopback or ip.is_link_local + or ip.is_reserved or ip.is_unspecified + ) + except Exception: # noqa: BLE001 — on doubt, don't cache + return True + + def extract_cache_get( url: str, format: Optional[str] = None, @@ -283,6 +320,8 @@ def extract_cache_get( """Return {'url','title','content'} for a fresh cached page, else None.""" if not cache_enabled(): return None + if _is_local_dev_url(url): + return None with _index_lock: index = _load_index() entry = index.get(_url_digest(url, format, provider)) @@ -327,6 +366,8 @@ def extract_cache_put( """ if not cache_enabled() or not content: return + if _is_local_dev_url(url): + return try: from tools.web_tools import MAX_STORED_TEXT_CHARS if len(content) > MAX_STORED_TEXT_CHARS: diff --git a/website/docs/user-guide/features/web-search.md b/website/docs/user-guide/features/web-search.md index ceda417248..a90c69bfab 100644 --- a/website/docs/user-guide/features/web-search.md +++ b/website/docs/user-guide/features/web-search.md @@ -75,6 +75,8 @@ Concurrent identical searches (a parallel subagent fan-out firing the same query Only successful responses are cached. Failures always retry the backend, responses served by the one-shot keyless rescue are never cached (the next call attempts your chosen backend again), and URLs matched by your `security.website_blocklist` are never served from cache. Cached extracts re-run the normal truncation pipeline, so a different `char_limit` on the second call works off the same stored scrape. +**Local development URLs are never cached.** Anything on `localhost`, `127.0.0.1`, `*.local`, single-label LAN hostnames, or private/link-local IP ranges (`192.168.*`, `10.*`, `172.16-31.*`) bypasses the extract cache entirely — dev servers, hot-reload builds, and chat-GUI artifact previews change on every save, and a cached copy would show you a stale build. Every fetch of a local page is live. (These URLs are only reachable at all when `security.allow_private_urls` is enabled.) + ```yaml # ~/.hermes/config.yaml web: