Files
hermes-agent/tests/hermes_cli/test_model_data_policy_guard.py
T
Beto de Paola a06f1d7617 feat(models): warn on data-training tiers at model selection
muse-spark-1.2-contributor is heavily discounted BECAUSE Meta uses your
prompts and completions to train future models. Selecting it for the price
without realising the data trade-off is a footgun.

Add hermes_cli/model_data_policy_guard.py (mirrors model_cost_guard):
data_training_warning(model_id, provider, base_url) -> DataTrainingWarning|None,
driven by a vendor-agnostic rule table. The status is not machine-readable on
/v1/models or models.dev, so the v1 rule keys on the documented '-contributor'
model id (fires regardless of provider, so it also covers custom/gateway
routes). Message mirrors Meta's pricing-doc language and figures
(https://dev.meta.ai/docs/pricing-rate-limits/).

Wire it into the CLI model picker's confirm flow (auth.py) as a [y/N]
disclosure, chained after the expensive-model cost guard. Fires only on the
contributor tier; silent on muse-spark-1.1/1.2 and all other models.
2026-08-14 01:06:13 -07:00

39 lines
1.6 KiB
Python

"""Tests for the data-training-tier selection guard."""
from hermes_cli.model_data_policy_guard import (
DataTrainingWarning,
data_training_warning,
)
def test_fires_on_meta_contributor():
w = data_training_warning("muse-spark-1.2-contributor", provider="meta-ai")
assert isinstance(w, DataTrainingWarning)
assert w.model == "muse-spark-1.2-contributor"
assert "train" in w.message.lower()
assert "muse-spark-1.2" in w.message # points to the no-training alternative
# Aligns with Meta's own pricing doc language + figures.
assert "$0.10" in w.message and "$0.20" in w.message and "$0.002" in w.message
assert "prompts and completions" in w.message.lower()
assert "dev.meta.ai/docs/pricing-rate-limits" in w.message
def test_silent_on_non_contributor_muse():
assert data_training_warning("muse-spark-1.2", provider="meta-ai") is None
assert data_training_warning("muse-spark-1.1", provider="meta-ai") is None
def test_silent_on_unrelated_models():
for m in ("anthropic/claude-opus-4.8", "gpt-5.6-sol", "deepseek-v4-pro", ""):
assert data_training_warning(m, provider="anthropic") is None
def test_fires_regardless_of_provider_string():
# id-keyed: must fire even if selected via custom/gateway (no meta-ai provider)
assert data_training_warning("muse-spark-1.2-contributor", provider="custom") is not None
assert data_training_warning("muse-spark-1.2-contributor") is not None
def test_case_insensitive():
assert data_training_warning("MUSE-SPARK-1.2-CONTRIBUTOR", provider="meta-ai") is not None