feat(models): warn on data-training tiers at model selection

muse-spark-1.2-contributor is heavily discounted BECAUSE Meta uses your
prompts and completions to train future models. Selecting it for the price
without realising the data trade-off is a footgun.

Add hermes_cli/model_data_policy_guard.py (mirrors model_cost_guard):
data_training_warning(model_id, provider, base_url) -> DataTrainingWarning|None,
driven by a vendor-agnostic rule table. The status is not machine-readable on
/v1/models or models.dev, so the v1 rule keys on the documented '-contributor'
model id (fires regardless of provider, so it also covers custom/gateway
routes). Message mirrors Meta's pricing-doc language and figures
(https://dev.meta.ai/docs/pricing-rate-limits/).

Wire it into the CLI model picker's confirm flow (auth.py) as a [y/N]
disclosure, chained after the expensive-model cost guard. Fires only on the
contributor tier; silent on muse-spark-1.1/1.2 and all other models.
This commit is contained in:
Beto de Paola
2026-08-07 16:39:10 -07:00
committed by Teknium
parent 9cb456a9b9
commit a06f1d7617
3 changed files with 192 additions and 0 deletions
+46
View File
@@ -7495,6 +7495,43 @@ def _confirm_expensive_model_selection(
return response in {"y", "yes"}
def _confirm_data_policy_selection(
model_id: str,
*,
provider: str = "",
base_url: str = "",
) -> bool:
"""Prompt before saving a model whose tier trains on your prompts/completions.
Mirrors :func:`_confirm_expensive_model_selection`. Keyed on the model id
(e.g. Meta's ``-contributor`` tier), so it is not gated on a known provider.
Returns True to proceed, False to cancel.
"""
try:
from hermes_cli.model_data_policy_guard import data_training_warning
warning = data_training_warning(
model_id,
provider=provider,
base_url=base_url,
)
except Exception:
warning = None
if warning is None:
return True
print()
print("=" * 72)
print(warning.message)
print("=" * 72)
try:
response = input("Use this data-training tier anyway? [y/N]: ").strip().lower()
except (KeyboardInterrupt, EOFError):
print()
return False
return response in {"y", "yes"}
def _prompt_model_selection(
model_ids: List[str],
current_model: str = "",
@@ -7534,6 +7571,15 @@ def _prompt_model_selection(
api_key=confirm_api_key,
):
return None
# Data-policy guard runs regardless of provider (it keys on the model
# id, e.g. a "-contributor" training tier), so it is NOT gated on
# confirm_provider like the cost guard above.
if not _confirm_data_policy_selection(
mid,
provider=confirm_provider,
base_url=confirm_base_url,
):
return None
return mid
# Reorder: current model first, then the rest (deduplicated)
+108
View File
@@ -0,0 +1,108 @@
"""Data-policy confirmation helpers for model selection surfaces.
Some inference tiers are cheap *because* the vendor trains future models on your
prompts and completions. Selecting one for the low price without realising the
data trade-off is a real footgun. This guard mirrors
``hermes_cli.model_cost_guard`` — it returns a warning payload that the CLI and
web model-selection flows surface as an explicit confirm step.
Why a static table (not a ProviderProfile hook): the guard runs inside core
selection code (``auth.py`` / ``web_server.py``), which never calls into the
active provider profile for a selection-time warning. Keeping the rule set here
also means it renders regardless of which provider plugin happens to be loaded,
and it stays testable without importing arbitrary third-party plugin code into
the selection path.
The status is NOT machine-readable anywhere today: neither models.dev nor the
Meta ``/v1/models`` payload exposes a training/retention flag (verified
2026-08-07). The only reliable signals are the vendor-documented model id and
its anomalously low pricing, so the rule keys on the id.
"""
from __future__ import annotations
from dataclasses import dataclass
from typing import Callable, Optional
@dataclass(frozen=True)
class DataTrainingWarning:
"""Confirmation payload for models whose tier trains on user data."""
model: str
provider: str
message: str
# ── Rule table ────────────────────────────────────────────────────────────
# Each rule: (human label, predicate over (model_lower, provider_lower), message).
# Extensible — new data-collection tiers from other vendors slot in here without
# touching the call sites. Predicates are intentionally conservative: match an
# explicit, vendor-documented id rather than guessing from price alone (price is
# only a corroborating signal and can change).
def _is_meta_contributor(model_lower: str, provider_lower: str) -> bool:
# Meta Model API "contributor" tier (muse-spark-1.2-contributor and any
# future -contributor checkpoints). Match on the id suffix; do not require a
# specific provider id so it fires whether selected via the meta-ai plugin,
# a gateway, or a custom endpoint that serves the same model id.
return model_lower.endswith("-contributor") or "contributor" in model_lower.split("-")
_META_CONTRIBUTOR_MESSAGE = (
"!!! CONTRIBUTOR TIER — TRAINS ON YOUR DATA !!!\n"
"\n"
"muse-spark-1.2-contributor is Meta's contributor tier: heavily discounted\n"
"token pricing in exchange for permission to use your prompts and completions\n"
"to train future Meta models.\n"
"\n"
" Price per 1M tokens: input $0.10 | output $0.20 | cached input $0.002\n"
" (vs. standard muse-spark-1.2: input $1.25 | output $4.25 | cached $0.15)\n"
"\n"
"It lowers the barrier to entry for prototyping, testing integrations, and\n"
"scaling experiments where training on your data is acceptable. Do NOT use it\n"
"for confidential, proprietary, personal, or otherwise sensitive data. For the\n"
"same model at standard pricing with no training on your data, select the\n"
"standard variant, muse-spark-1.2.\n"
"\n"
"Source: https://dev.meta.ai/docs/pricing-rate-limits/\n"
"Confirm only if training on your prompts and completions is acceptable."
)
# (predicate, message) pairs, evaluated in order; first match wins.
_RULES: tuple[tuple[Callable[[str, str], bool], str], ...] = (
(_is_meta_contributor, _META_CONTRIBUTOR_MESSAGE),
)
def data_training_warning(
model_name: str,
*,
provider: Optional[str] = None,
base_url: Optional[str] = None, # noqa: ARG001 — reserved for host-scoped rules
) -> Optional[DataTrainingWarning]:
"""Return a warning payload when *model_name* selects a data-training tier.
Returns ``None`` when no rule matches (the common case). Callers should run
this after model resolution so aliases / provider-specific ids have settled,
and surface ``.message`` as a confirm prompt.
"""
model = (model_name or "").strip()
if not model:
return None
model_lower = model.lower()
provider_lower = (provider or "").strip().lower()
for predicate, message in _RULES:
try:
if predicate(model_lower, provider_lower):
return DataTrainingWarning(
model=model,
provider=(provider or "").strip(),
message=message,
)
except Exception:
# A misbehaving predicate must never break model selection.
continue
return None
@@ -0,0 +1,38 @@
"""Tests for the data-training-tier selection guard."""
from hermes_cli.model_data_policy_guard import (
DataTrainingWarning,
data_training_warning,
)
def test_fires_on_meta_contributor():
w = data_training_warning("muse-spark-1.2-contributor", provider="meta-ai")
assert isinstance(w, DataTrainingWarning)
assert w.model == "muse-spark-1.2-contributor"
assert "train" in w.message.lower()
assert "muse-spark-1.2" in w.message # points to the no-training alternative
# Aligns with Meta's own pricing doc language + figures.
assert "$0.10" in w.message and "$0.20" in w.message and "$0.002" in w.message
assert "prompts and completions" in w.message.lower()
assert "dev.meta.ai/docs/pricing-rate-limits" in w.message
def test_silent_on_non_contributor_muse():
assert data_training_warning("muse-spark-1.2", provider="meta-ai") is None
assert data_training_warning("muse-spark-1.1", provider="meta-ai") is None
def test_silent_on_unrelated_models():
for m in ("anthropic/claude-opus-4.8", "gpt-5.6-sol", "deepseek-v4-pro", ""):
assert data_training_warning(m, provider="anthropic") is None
def test_fires_regardless_of_provider_string():
# id-keyed: must fire even if selected via custom/gateway (no meta-ai provider)
assert data_training_warning("muse-spark-1.2-contributor", provider="custom") is not None
assert data_training_warning("muse-spark-1.2-contributor") is not None
def test_case_insensitive():
assert data_training_warning("MUSE-SPARK-1.2-CONTRIBUTOR", provider="meta-ai") is not None