fix(web): never cache local development URLs in the extract cache
Dev servers, hot-reload builds, and chat-GUI artifact previews live on localhost/private addresses and change on every save — a 20-minute cached copy would show a stale build exactly when freshness is the point of fetching. The extract cache now declines loopback, private, link-local, *.local, *.localhost, and single-label LAN hostnames on both put and get. Hostname heuristics only (no DNS) — this is a freshness carveout, not a security boundary; SSRF enforcement is unchanged in tools/url_safety.py. Public URLs keep the full TTL.
This commit is contained in:
@@ -196,6 +196,36 @@ def test_extract_cache_oversized_page_not_indexed(_isolated_cache):
|
||||
assert extract_cache_get("https://big.com") is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("url", [
|
||||
"http://localhost:3000/app",
|
||||
"http://localhost:5173", # vite dev server
|
||||
"http://127.0.0.1:8080/preview",
|
||||
"http://[::1]:3000/",
|
||||
"http://192.168.1.44/dashboard",
|
||||
"http://10.0.0.5:8000/api/docs",
|
||||
"http://172.16.0.9/",
|
||||
"http://myapp.local/",
|
||||
"http://devbox/page", # single-label LAN name
|
||||
"http://preview.localhost/artifact",
|
||||
])
|
||||
def test_extract_cache_never_caches_local_dev_urls(url, _isolated_cache):
|
||||
"""Local/private URLs are dev servers and chat-GUI artifact previews —
|
||||
they change on every save, so freshness beats dedup. Neither put nor
|
||||
get may touch the cache for them."""
|
||||
extract_cache_put(url, "stale build output")
|
||||
assert extract_cache_get(url) is None
|
||||
|
||||
|
||||
@pytest.mark.parametrize("url", [
|
||||
"https://example.com/page",
|
||||
"https://docs.python.org/3/",
|
||||
])
|
||||
def test_extract_cache_public_urls_still_cache(url, _isolated_cache):
|
||||
extract_cache_put(url, "public content")
|
||||
hit = extract_cache_get(url)
|
||||
assert hit is not None and hit["content"] == "public content"
|
||||
|
||||
|
||||
def test_extract_cache_tampered_index_path_is_miss(_isolated_cache, tmp_path):
|
||||
"""An index entry pointing outside cache/web must never be read."""
|
||||
outside = tmp_path / "outside.md"
|
||||
|
||||
@@ -275,6 +275,43 @@ def _entry_file_path(url: str, format: Optional[str], provider: str) -> Optional
|
||||
return d / f"{slug}-{_url_digest(url, format, provider)}.cache.md"
|
||||
|
||||
|
||||
def _is_local_dev_url(url: str) -> bool:
|
||||
"""True for loopback/private/LAN URLs — never cached.
|
||||
|
||||
A page on a private address is one the user controls and is typically
|
||||
changing fast (dev servers, hot reload, chat-GUI artifact previews,
|
||||
LAN preview apps). Freshness is the point of fetching it, so the cache
|
||||
declines these entirely rather than serving a stale build for a whole
|
||||
TTL. Only reachable when ``security.allow_private_urls`` is enabled —
|
||||
default installs SSRF-block these URLs before extraction anyway.
|
||||
|
||||
Hostname heuristics only (no DNS resolution — this is a freshness
|
||||
decision, not a security boundary; SSRF enforcement lives in
|
||||
tools/url_safety.py).
|
||||
"""
|
||||
try:
|
||||
from urllib.parse import urlparse
|
||||
host = (urlparse(url).hostname or "").strip("[]").lower()
|
||||
if not host:
|
||||
return True # unparseable → don't cache
|
||||
if host == "localhost" or host.endswith(".localhost") or host.endswith(".local"):
|
||||
return True
|
||||
# Single-label hostnames (no dot) are LAN names, not public DNS.
|
||||
if "." not in host and ":" not in host:
|
||||
return True
|
||||
import ipaddress
|
||||
try:
|
||||
ip = ipaddress.ip_address(host)
|
||||
except ValueError:
|
||||
return False # public DNS name
|
||||
return bool(
|
||||
ip.is_private or ip.is_loopback or ip.is_link_local
|
||||
or ip.is_reserved or ip.is_unspecified
|
||||
)
|
||||
except Exception: # noqa: BLE001 — on doubt, don't cache
|
||||
return True
|
||||
|
||||
|
||||
def extract_cache_get(
|
||||
url: str,
|
||||
format: Optional[str] = None,
|
||||
@@ -283,6 +320,8 @@ def extract_cache_get(
|
||||
"""Return {'url','title','content'} for a fresh cached page, else None."""
|
||||
if not cache_enabled():
|
||||
return None
|
||||
if _is_local_dev_url(url):
|
||||
return None
|
||||
with _index_lock:
|
||||
index = _load_index()
|
||||
entry = index.get(_url_digest(url, format, provider))
|
||||
@@ -327,6 +366,8 @@ def extract_cache_put(
|
||||
"""
|
||||
if not cache_enabled() or not content:
|
||||
return
|
||||
if _is_local_dev_url(url):
|
||||
return
|
||||
try:
|
||||
from tools.web_tools import MAX_STORED_TEXT_CHARS
|
||||
if len(content) > MAX_STORED_TEXT_CHARS:
|
||||
|
||||
@@ -75,6 +75,8 @@ Concurrent identical searches (a parallel subagent fan-out firing the same query
|
||||
|
||||
Only successful responses are cached. Failures always retry the backend, responses served by the one-shot keyless rescue are never cached (the next call attempts your chosen backend again), and URLs matched by your `security.website_blocklist` are never served from cache. Cached extracts re-run the normal truncation pipeline, so a different `char_limit` on the second call works off the same stored scrape.
|
||||
|
||||
**Local development URLs are never cached.** Anything on `localhost`, `127.0.0.1`, `*.local`, single-label LAN hostnames, or private/link-local IP ranges (`192.168.*`, `10.*`, `172.16-31.*`) bypasses the extract cache entirely — dev servers, hot-reload builds, and chat-GUI artifact previews change on every save, and a cached copy would show you a stale build. Every fetch of a local page is live. (These URLs are only reachable at all when `security.allow_private_urls` is enabled.)
|
||||
|
||||
```yaml
|
||||
# ~/.hermes/config.yaml
|
||||
web:
|
||||
|
||||
Reference in New Issue
Block a user