fix(web): never cache local development URLs in the extract cache

Dev servers, hot-reload builds, and chat-GUI artifact previews live on
localhost/private addresses and change on every save — a 20-minute
cached copy would show a stale build exactly when freshness is the
point of fetching. The extract cache now declines loopback, private,
link-local, *.local, *.localhost, and single-label LAN hostnames on
both put and get. Hostname heuristics only (no DNS) — this is a
freshness carveout, not a security boundary; SSRF enforcement is
unchanged in tools/url_safety.py.

Public URLs keep the full TTL.
This commit is contained in:
Teknium
2026-08-25 03:14:15 -07:00
parent 8adef09be8
commit f0381ee4ae
3 changed files with 73 additions and 0 deletions
+30
View File
@@ -196,6 +196,36 @@ def test_extract_cache_oversized_page_not_indexed(_isolated_cache):
assert extract_cache_get("https://big.com") is None
@pytest.mark.parametrize("url", [
"http://localhost:3000/app",
"http://localhost:5173", # vite dev server
"http://127.0.0.1:8080/preview",
"http://[::1]:3000/",
"http://192.168.1.44/dashboard",
"http://10.0.0.5:8000/api/docs",
"http://172.16.0.9/",
"http://myapp.local/",
"http://devbox/page", # single-label LAN name
"http://preview.localhost/artifact",
])
def test_extract_cache_never_caches_local_dev_urls(url, _isolated_cache):
"""Local/private URLs are dev servers and chat-GUI artifact previews —
they change on every save, so freshness beats dedup. Neither put nor
get may touch the cache for them."""
extract_cache_put(url, "stale build output")
assert extract_cache_get(url) is None
@pytest.mark.parametrize("url", [
"https://example.com/page",
"https://docs.python.org/3/",
])
def test_extract_cache_public_urls_still_cache(url, _isolated_cache):
extract_cache_put(url, "public content")
hit = extract_cache_get(url)
assert hit is not None and hit["content"] == "public content"
def test_extract_cache_tampered_index_path_is_miss(_isolated_cache, tmp_path):
"""An index entry pointing outside cache/web must never be read."""
outside = tmp_path / "outside.md"
+41
View File
@@ -275,6 +275,43 @@ def _entry_file_path(url: str, format: Optional[str], provider: str) -> Optional
return d / f"{slug}-{_url_digest(url, format, provider)}.cache.md"
def _is_local_dev_url(url: str) -> bool:
"""True for loopback/private/LAN URLs — never cached.
A page on a private address is one the user controls and is typically
changing fast (dev servers, hot reload, chat-GUI artifact previews,
LAN preview apps). Freshness is the point of fetching it, so the cache
declines these entirely rather than serving a stale build for a whole
TTL. Only reachable when ``security.allow_private_urls`` is enabled —
default installs SSRF-block these URLs before extraction anyway.
Hostname heuristics only (no DNS resolution — this is a freshness
decision, not a security boundary; SSRF enforcement lives in
tools/url_safety.py).
"""
try:
from urllib.parse import urlparse
host = (urlparse(url).hostname or "").strip("[]").lower()
if not host:
return True # unparseable → don't cache
if host == "localhost" or host.endswith(".localhost") or host.endswith(".local"):
return True
# Single-label hostnames (no dot) are LAN names, not public DNS.
if "." not in host and ":" not in host:
return True
import ipaddress
try:
ip = ipaddress.ip_address(host)
except ValueError:
return False # public DNS name
return bool(
ip.is_private or ip.is_loopback or ip.is_link_local
or ip.is_reserved or ip.is_unspecified
)
except Exception: # noqa: BLE001 — on doubt, don't cache
return True
def extract_cache_get(
url: str,
format: Optional[str] = None,
@@ -283,6 +320,8 @@ def extract_cache_get(
"""Return {'url','title','content'} for a fresh cached page, else None."""
if not cache_enabled():
return None
if _is_local_dev_url(url):
return None
with _index_lock:
index = _load_index()
entry = index.get(_url_digest(url, format, provider))
@@ -327,6 +366,8 @@ def extract_cache_put(
"""
if not cache_enabled() or not content:
return
if _is_local_dev_url(url):
return
try:
from tools.web_tools import MAX_STORED_TEXT_CHARS
if len(content) > MAX_STORED_TEXT_CHARS:
@@ -75,6 +75,8 @@ Concurrent identical searches (a parallel subagent fan-out firing the same query
Only successful responses are cached. Failures always retry the backend, responses served by the one-shot keyless rescue are never cached (the next call attempts your chosen backend again), and URLs matched by your `security.website_blocklist` are never served from cache. Cached extracts re-run the normal truncation pipeline, so a different `char_limit` on the second call works off the same stored scrape.
**Local development URLs are never cached.** Anything on `localhost`, `127.0.0.1`, `*.local`, single-label LAN hostnames, or private/link-local IP ranges (`192.168.*`, `10.*`, `172.16-31.*`) bypasses the extract cache entirely — dev servers, hot-reload builds, and chat-GUI artifact previews change on every save, and a cached copy would show you a stale build. Every fetch of a local page is live. (These URLs are only reachable at all when `security.allow_private_urls` is enabled.)
```yaml
# ~/.hermes/config.yaml
web: