{"baselines": ["always_zero", "always_one", "base_rate_constant", "predict_yesterday", "country_base_rate", "random_with_base_rate"], "endpoint": "/v1/atlas/baseline-benchmark", "filters": {"barely": "true | false (barely_beats_baseline flag)", "family": "forecast | forecast-multi-horizon | classifier | trajectory | contagion | survival | per-platform | per-domain | per-method", "min_lift_pp": "float; filter to rows with F1 lift vs predict_yesterday >= X pp"}, "honest_caveats": ["Lift vs predict_yesterday is reported in percentage points (pp) of F1.", "Some rows compare a training-time sidecar metric against baselines on a separate last-60d temporal holdout. Treat those lifts as approximate; check `source`.", "barely_beats_baseline=true is a yellow flag, not red - some models add value beyond F1 (per-country thresholds, SHAP, conformal intervals)."], "purpose": "Show that each shipped Atlas ML model adds value beyond trivial baselines. Lift < 5pp vs predict_yesterday is flagged.", "refresh": "weekly Thu 04:45 UTC via cron", "sidecar_last_modified": "2026-10-01T04:45:08Z", "sidecar_path": "[internal path]", "sidecar_size_bytes": 51646}