Files
ForbiddenStarsApp/backend/tests/test_rating_examples.py
T
NotBigGhostandClaude Opus 5 e58b4f6614 Рейтинг: движок Elo с множителем отрыва и тесты на примеры документа
scoring.py получает движок из docs/rating/rating-system.md: ожидание пары,
K 64 → 16 за 20 партий, вес стола G(N), множитель отрыва (темп, цели, миры,
clamp [0.5, 2]) и близость по типу победы, включая last_standing. rate_match
и replay — чистые функции без БД; replay отдаёт рейтинги без округления,
ΔR и результат относительно ожидания по каждой партии.

Тесты: 15 примеров раздела 6 с числами документа, совпадение констант
и всех ΔR сезона с эталоном simulate.py (полные партии и история без деталей),
монотонность и сумма-ноль. League Points пока остаётся в модуле — витрины
переводятся отдельным коммитом. #23

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LqSoRj99iwVEH5U5fnZgsd
2026-09-14 22:19:40 +03:00

198 lines
8.6 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Движок рейтинга: примеры docs/rating/rating-system.md и сверка с эталоном simulate.py.
Числа примеров — те же, что в документе (раздел 6) и в EXPECTED эталона: разъехаться
документ, эталон и приложение не должны. Сверка с simulate.py дополнительно гоняет
синтетический сезон и требует совпадения каждого изменения рейтинга."""
from __future__ import annotations
import importlib.util
import sys
from pathlib import Path
import pytest
from app.services import scoring
from app.services.scoring import RatedMatch, RatedSeat, rate_match, replay
VETERAN = 40 # партий у «опытного» игрока: K = K_MIN
A, B, C, D, E, F = 1, 2, 3, 4, 5, 6
def _vets(*ids: int) -> dict[int, int]:
return dict.fromkeys(ids, VETERAN)
def _duel(first: int, second: int, **kw) -> RatedMatch:
return RatedMatch((RatedSeat(first, 1), RatedSeat(second, 2)), **kw)
def _seat(uid: int, place: int, objectives=None, worlds=None, eliminated=False) -> RatedSeat:
return RatedSeat(uid, place, objectives=objectives, worlds=worlds, eliminated=eliminated)
FIVE = tuple(RatedSeat(uid, i + 1) for i, uid in enumerate((A, B, C, D, E)))
SIX = tuple(RatedSeat(uid, i + 1) for i, uid in enumerate((A, B, C, D, E, F)))
# (ключ, рейтинги, сыграно партий, партия, ожидаемые ΔR с точностью до 0.01)
EXAMPLES = [
("1a", {A: 1600, B: 1400}, _vets(A, B), _duel(A, B, win_reason="objectives"),
{A: 3.84, B: -3.84}),
("1b", {A: 1600, B: 1400}, _vets(A, B), _duel(B, A, win_reason="objectives"),
{B: 12.16, A: -12.16}),
("2a", {A: 1500, B: 1500}, _vets(A, B),
RatedMatch((_seat(A, 1, 2, 5), _seat(B, 2, 1, 4)), "objectives", end_round=3),
{A: 11.18, B: -11.18}),
("2b", {A: 1500, B: 1500}, _vets(A, B),
RatedMatch((_seat(A, 1, 2, 5), _seat(B, 2, 1, 4)), "objectives", end_round=8),
{A: 5.46, B: -5.46}),
("3a", {A: 1500, B: 1500}, _vets(A, B), _duel(A, B, win_reason="objectives"),
{A: 8.0, B: -8.0}),
("3b", dict.fromkeys((A, B, C, D, E), 1500), _vets(A, B, C, D, E),
RatedMatch(FIVE, "objectives"),
{A: 11.0, B: 5.5, C: 0.0, D: -5.5, E: -11.0}),
("4a", {A: 1500, B: 1500}, _vets(A, B), _duel(A, B, win_reason="worlds"),
{A: 6.8, B: -6.8}),
("4b", {A: 1500, B: 1500}, _vets(A, B), _duel(A, B, win_reason="plastic"),
{A: 5.6, B: -5.6}),
("4c", {A: 1500, B: 1500}, _vets(A, B), _duel(A, B, win_reason="resources"),
{A: 4.8, B: -4.8}),
("4d", {A: 1500, B: 1500}, _vets(A, B),
RatedMatch((_seat(A, 1, 2, 6), _seat(B, 2, 2, 5)), "worlds", end_round=8),
{A: 3.4, B: -3.4}),
("4e", {A: 1500, B: 1500}, _vets(A, B),
RatedMatch((_seat(A, 1, 2, 8), _seat(B, 2, 0, 2)), "objectives", end_round=3),
{A: 16.0, B: -16.0}),
("5", {A: 1550, B: 1500, C: 1480, D: 1450}, _vets(A, B, C, D),
RatedMatch(
(_seat(A, 1, 4, 8), _seat(B, 2, 3, 7),
_seat(C, 3, 1, 0, eliminated=True), _seat(D, 3, 0, 0, eliminated=True)),
"objectives", end_round=7,
),
{A: 9.27, B: 5.85, C: -7.78, D: -7.35}),
("6a", dict.fromkeys(range(A, F + 1), 1500), _vets(*range(A, F + 1)),
RatedMatch(SIX, "objectives", end_round=8, nine_rounds_rule=True),
{A: 12.0, B: 7.2, C: 2.4, D: -2.4, E: -7.2, F: -12.0}),
("6b", dict.fromkeys(range(A, F + 1), 1500), _vets(*range(A, F + 1)),
RatedMatch(SIX, "objectives", end_round=8, nine_rounds_rule=False),
{A: 10.29, B: 7.54, C: 2.74, D: -2.06, E: -6.86, F: -11.66}),
("7", {A: 1500, B: 1500}, {A: 0, B: VETERAN}, _duel(A, B, win_reason="objectives"),
{A: 32.0, B: -8.0}),
]
@pytest.mark.parametrize(
"ratings,games,match,want", [e[1:] for e in EXAMPLES], ids=[e[0] for e in EXAMPLES]
)
def test_document_examples(ratings, games, match, want):
delta, _perf = rate_match(ratings, games, match)
assert {uid: round(v, 2) for uid, v in delta.items()} == want
def test_examples_cover_whole_section():
assert len(EXAMPLES) == 15
def test_last_standing_counts_full_objective_gap():
"""Победа last_standing: отрыв победителя по целям = 1, сколько бы маркеров ни было."""
seats = (_seat(A, 1, 1, 6), _seat(B, 2, 1, 0, eliminated=True))
ordinary = rate_match({}, _vets(A, B), RatedMatch(seats, "objectives"))[0][A]
standing = rate_match({}, _vets(A, B), RatedMatch(seats, "last_standing"))[0][A]
# Отрыв по целям 1 вместо 0 → множитель больше на W_OBJ·1 = 0.5; ΔR = K·ΔM·(S − E).
assert standing - ordinary == pytest.approx(16 * 0.5 * 0.5)
# ─── Сверка с эталоном ───────────────────────────────────────────────────────
SIMULATE = Path(__file__).resolve().parents[2] / "docs" / "rating" / "simulate.py"
@pytest.fixture(scope="module")
def sim():
if not SIMULATE.exists():
pytest.skip("docs/rating/simulate.py недоступен")
spec = importlib.util.spec_from_file_location("rating_simulate", SIMULATE)
module = importlib.util.module_from_spec(spec)
sys.modules[spec.name] = module # dataclasses ищут модуль по имени
spec.loader.exec_module(module) # type: ignore[union-attr]
return module
def _convert(m, ids: dict[str, int], match_id: int) -> RatedMatch:
return RatedMatch(
tuple(
RatedSeat(
ids[s.player], s.place, eliminated=s.eliminated,
objectives=s.objectives, worlds=s.worlds,
)
for s in m.seats
),
m.win_reason,
end_round=m.round,
nine_rounds_rule=m.nine_rounds,
id=match_id,
)
def test_constants_match_reference(sim):
p = sim.PROPOSED
assert (p.r0, p.d, p.k_max, p.k_min, p.k_games) == (
scoring.R0, scoring.D, scoring.K_MAX, scoring.K_MIN, scoring.K_GAMES
)
assert (p.w_table, p.w_tempo, p.w_obj, p.w_worlds) == (
scoring.W_TABLE, scoring.W_TEMPO, scoring.W_OBJ, scoring.W_WORLDS
)
assert (p.mu_obj, p.mu_worlds, p.m_min, p.m_max) == (
scoring.MU_OBJ, scoring.MU_WORLDS, scoring.M_MIN, scoring.M_MAX
)
assert dict(p.closeness) == scoring.CLOSENESS
assert sim.BOARD_TILES == scoring.BOARD_TILES
assert not p.autocorr
@pytest.mark.parametrize("scenario", ["сигнал", "клубы", "рост"])
@pytest.mark.parametrize("stripped", [False, True], ids=["full", "history"])
def test_replay_matches_reference_season(sim, scenario, stripped):
"""Весь сезон: каждое изменение рейтинга совпадает с эталоном до 1e-9."""
cfg = sim.SCENARIOS[scenario]
_skill, matches = sim.generate_season(
cfg["seed"], sim.SEASON_MATCHES, cfg["informative"], cfg["clubs"], cfg["learning"]
)
if stripped:
matches = [sim.strip_details(m) for m in matches]
ids: dict[str, int] = {}
for m in matches:
for s in m.seats:
ids.setdefault(s.player, len(ids) + 1)
reference = sim.Elo(sim.PROPOSED)
ours = replay(_convert(m, ids, i) for i, m in enumerate(matches))
for i, m in enumerate(matches):
for player, dv in reference.update(m).items():
assert ours.delta[(i, ids[player])] == pytest.approx(dv, abs=1e-9)
for player, uid in ids.items():
assert ours.ratings[uid] == pytest.approx(reference.rating(player), abs=1e-6)
def test_monotone_and_zero_sum(sim):
"""Победитель без ничьей не теряет, последний без ничьей не получает; при равных K
сумма изменений за партию — ноль."""
cfg = sim.SCENARIOS["сигнал"]
_skill, matches = sim.generate_season(cfg["seed"] + 7, 200, True)
ids: dict[str, int] = {}
for m in matches:
for s in m.seats:
ids.setdefault(s.player, len(ids) + 1)
veterans = dict.fromkeys(ids.values(), VETERAN)
for i, m in enumerate(matches):
rm = _convert(m, ids, i)
delta, _ = rate_match({}, veterans, rm)
places = [s.place for s in rm.seats]
for s in rm.seats:
if places.count(s.place) > 1:
continue
if s.place == 1:
assert delta[s.user_id] > 0
if s.place == max(places):
assert delta[s.user_id] < 0
assert sum(delta.values()) == pytest.approx(0.0, abs=1e-9)