Skip to content
This repository was archived by the owner on Jun 25, 2026. It is now read-only.
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
41 changes: 29 additions & 12 deletions commissioners/common/commissioners.py
Original file line number Diff line number Diff line change
Expand Up @@ -409,6 +409,31 @@ def schedule_episodes(
)
return CommissionerScheduleEpisodes(episodes=episodes)

def _round_scores_by_policy(
self,
entries: list[PolicyPoolEntry],
episode_results: list[EpisodeResult],
) -> tuple[dict[UUID, float], dict[UUID, int]]:
"""Per-policy round score and the number of samples behind each.

Default: the mean of a policy's per-episode scores. Subclasses (e.g. the ruleset
commissioner's rank-by-episode mode) override this to score rounds differently while
reusing the ranking/metadata assembly in ``complete_round``.
"""
score_lists = _score_lists_by_policy(episode_results)
scores = {
entry.policy_version_id: (
sum(score_lists.get(entry.policy_version_id, [])) / len(score_lists.get(entry.policy_version_id, []))
if score_lists.get(entry.policy_version_id)
else 0.0
)
for entry in entries
}
ranked_counts = {
entry.policy_version_id: len(score_lists.get(entry.policy_version_id, [])) for entry in entries
}
return scores, ranked_counts

def complete_round(
self,
*,
Expand All @@ -417,23 +442,15 @@ def complete_round(
entries: list[PolicyPoolEntry],
episode_results: list[EpisodeResult],
) -> CommissionerRoundComplete:
score_lists = _score_lists_by_policy(episode_results)
round_score_by_policy, ranked_score_counts = self._round_scores_by_policy(entries, episode_results)
completed_episode_counts: dict[UUID, int] = defaultdict(int)
for result in episode_results:
for policy_version_id in {score.policy_version_id for score in result.scores}:
completed_episode_counts[policy_version_id] += 1
avg_score_by_policy = {
entry.policy_version_id: (
sum(score_lists.get(entry.policy_version_id, [])) / len(score_lists.get(entry.policy_version_id, []))
if score_lists.get(entry.policy_version_id)
else 0.0
)
for entry in entries
}
ranked_entries = sorted(
entries,
key=lambda entry: (
-avg_score_by_policy[entry.policy_version_id],
-round_score_by_policy[entry.policy_version_id],
entry.seed_order,
str(entry.policy_version_id),
),
Expand All @@ -443,11 +460,11 @@ def complete_round(
policy_version_id=entry.policy_version_id,
player_id=str(entry.player_id) if entry.player_id is not None else None,
rank=rank,
score=avg_score_by_policy[entry.policy_version_id],
score=round_score_by_policy[entry.policy_version_id],
result_metadata={
"seed_order": entry.seed_order,
COMPLETED_EPISODE_COUNT_METADATA_KEY: completed_episode_counts[entry.policy_version_id],
RANKED_SCORE_COUNT_METADATA_KEY: len(score_lists.get(entry.policy_version_id, [])),
RANKED_SCORE_COUNT_METADATA_KEY: ranked_score_counts[entry.policy_version_id],
},
)
for rank, entry in enumerate(ranked_entries, start=1)
Expand Down
25 changes: 25 additions & 0 deletions commissioners/common/ruleset_strategy/commissioner.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,7 @@

from datetime import UTC, datetime, timedelta
from typing import Any
from uuid import UUID

from commissioners.common.commissioners import BaselineCommissioner
from commissioners.common.models import (
Expand Down Expand Up @@ -40,6 +41,7 @@
_duration_text,
_leaderboard_rules_description,
_plural_word,
_rank_points_lists_by_policy,
_round_structure_description,
_schedule_slot_description,
)
Expand Down Expand Up @@ -246,6 +248,29 @@ def complete_round_for_round_start(
]
return complete

def _round_scores_by_policy(
self,
entries: list[PolicyPoolEntry],
episode_results: list[EpisodeResult],
) -> tuple[dict[UUID, float], dict[UUID, int]]:
scoring = self._config().scoring
if scoring is None or scoring.round_score != "rank":
return super()._round_scores_by_policy(entries, episode_results)
points_lists = _rank_points_lists_by_policy(episode_results)
scores = {
entry.policy_version_id: (
sum(points_lists.get(entry.policy_version_id, []))
/ len(points_lists.get(entry.policy_version_id, []))
if points_lists.get(entry.policy_version_id)
else 0.0
)
for entry in entries
}
ranked_counts = {
entry.policy_version_id: len(points_lists.get(entry.policy_version_id, [])) for entry in entries
}
return scores, ranked_counts

def complete_round(
self,
*,
Expand Down
35 changes: 30 additions & 5 deletions commissioners/common/ruleset_strategy/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,12 @@
MembershipSnapshot,
)
from commissioners.common.models import V2StageConfig
from commissioners.common.utils import MEAN_ROUND_SCORE_KIND, MEAN_SCORE_EWMA_SCORING_MECHANICS
from commissioners.common.utils import (
MEAN_ROUND_SCORE_KIND,
MEAN_SCORE_EWMA_SCORING_MECHANICS,
RANK_EPISODE_EWMA_SCORING_MECHANICS,
RANK_EPISODE_ROUND_SCORE_KIND,
)

CONFIG_KEY = "ruleset_strategy"
IMAGE_CONFIG_NAME_ENV = "RULESET_STRATEGY_CONFIG_NAME"
Expand Down Expand Up @@ -155,7 +160,10 @@ class LeaderboardScoringConfig(_ConfigModel):


class ScoringConfig(_ConfigModel):
round_score: Literal["mean"] = "mean"
# "mean": round score is the mean of a policy's per-episode scores.
# "rank": round score is the mean of a policy's per-episode rank points (placement within
# each episode, N..1), so margins of victory are discarded and only placement counts.
round_score: Literal["mean", "rank"] = "mean"
leaderboard: LeaderboardScoringConfig = Field(default_factory=LeaderboardScoringConfig)
mechanics: str | None = None

Expand Down Expand Up @@ -355,13 +363,22 @@ def seating(self) -> SeatingStrategy:
def insufficient_players(self) -> InsufficientPlayersConfig:
return self.defaults.insufficient_players()

@property
def round_score_kind(self) -> str:
if self.scoring is not None and self.scoring.round_score == "rank":
return RANK_EPISODE_ROUND_SCORE_KIND
return MEAN_ROUND_SCORE_KIND

@property
def ranking(self) -> RankingConfig:
if self.scoring is None:
return RankingConfig()
# Tag/filter round results by score kind so that switching round_score (e.g. mean -> rank)
# excludes the now-incomparable prior-regime results from the leaderboard instead of
# blending different score scales.
return RankingConfig(
result_metadata={"score_kind": MEAN_ROUND_SCORE_KIND},
filter_metadata={"score_kind": MEAN_ROUND_SCORE_KIND},
result_metadata={"score_kind": self.round_score_kind},
filter_metadata={"score_kind": self.round_score_kind},
ewma_halflife_hours=self.scoring.leaderboard.half_life_hours,
)

Expand All @@ -372,9 +389,17 @@ def scoring_mechanics(self) -> str | None:
if self.scoring.mechanics is not None:
return self.scoring.mechanics
half_life_hours = self.scoring.leaderboard.half_life_hours
is_rank = self.scoring.round_score == "rank"
if half_life_hours == 2:
return MEAN_SCORE_EWMA_SCORING_MECHANICS
return RANK_EPISODE_EWMA_SCORING_MECHANICS if is_rank else MEAN_SCORE_EWMA_SCORING_MECHANICS
half_life_text = int(half_life_hours) if half_life_hours.is_integer() else half_life_hours
if is_rank:
return (
"Rounds rank policies by placement within each episode (N points for the episode winner of an "
"N-policy game down to 1 for last, ties sharing the better place), averaged across the episodes "
"each policy played. The division leaderboard combines completed rounds with a "
f"{half_life_text}-hour half-life EWMA, so newer rounds count more than older rounds."
)
return (
"Rounds rank policies by the average score reported by the game across each policy's episode slots. "
"The division leaderboard only uses current average-score round results and combines completed rounds "
Expand Down
35 changes: 35 additions & 0 deletions commissioners/common/utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,6 +25,7 @@
DIVISION_LEADERBOARD_SCORE_EWMA_HALFLIFE_HOURS = 2
AMONG_THEM_RESULT_METADATA_VERSION = 2
MEAN_ROUND_SCORE_KIND = "mean_round_score"
RANK_EPISODE_ROUND_SCORE_KIND = "rank_episode_round_score"
COMPLETED_EPISODE_COUNT_METADATA_KEY = "completed_episode_count"
RANKED_SCORE_COUNT_METADATA_KEY = "ranked_score_count"
MEAN_SCORE_EWMA_SCORING_MECHANICS = (
Expand All @@ -34,6 +35,13 @@
"The division leaderboard only uses current average-score round results and combines completed rounds with a "
"2-hour half-life EWMA, so newer rounds count more than older rounds."
)
RANK_EPISODE_EWMA_SCORING_MECHANICS = (
"Rounds rank policies by placement within each episode rather than by raw score: in an episode with N "
"policies the highest-scoring policy earns N points and the lowest earns 1 (ties share the better place), and "
"a policy's round score is the average of those rank points across the episodes it played. Margins of victory "
"are discarded — only who beat whom each game matters. The division leaderboard combines completed rounds with "

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

@KyleHerndon note that right now, with division leaderboard computation and commissioner leaderboard computation split the way we do, and with our UI only reflecting commissioner-reported description, we force the commissioner to abstraction-leak by describing how its roundresults get managed by app-backend

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

sorry more plainly: ideally we wouldnt need commissioners to say that their roundresults get 2h-ewma'd; they shouldn't need to know about or speak about it, and can't enforce it

"a 2-hour half-life EWMA, so newer rounds count more than older rounds."
)
AMONG_THEM_SCORE_KIND = MEAN_ROUND_SCORE_KIND
AMONG_THEM_SCORING_MECHANICS = MEAN_SCORE_EWMA_SCORING_MECHANICS

Expand Down Expand Up @@ -214,6 +222,33 @@ def _score_lists_by_policy(episode_results: list[EpisodeResult]) -> dict[UUID, l
return score_lists


def _episode_rank_points(scores: list[float]) -> list[float]:
"""Convert one episode's per-policy scores into N..1 rank points.

A policy earns N minus the number of policies that strictly outscored it, so the winner of
an N-policy episode gets N and last place gets 1. Ties share the better placement (two
policies tied for first both get N). Margins are discarded — only placement matters.
"""
n = len(scores)
return [float(n - sum(1 for other in scores if other > score)) for score in scores]


def _rank_points_lists_by_policy(episode_results: list[EpisodeResult]) -> dict[UUID, list[float]]:
"""Per-policy lists of per-episode rank points across every episode the policy played.

Unlike ``_score_lists_by_policy`` no scores are dropped: placement is meaningful for every
seat in an episode, including a zero score, so each episode contributes one rank point per
participating policy.
"""
points_lists: dict[UUID, list[float]] = defaultdict(list)
for result in episode_results:
episode_scores = [(score.policy_version_id, score.score) for score in result.scores]
points = _episode_rank_points([score for _, score in episode_scores])
for (policy_version_id, _), point in zip(episode_scores, points, strict=True):
points_lists[policy_version_id].append(point)
return points_lists


def _qualification_round_membership_changes(
ctx: OnRoundCompletedContext,
*,
Expand Down
56 changes: 56 additions & 0 deletions tests/test_commissioner_strategies.py
Original file line number Diff line number Diff line change
Expand Up @@ -340,6 +340,62 @@ def test_default_commissioner_round_robin_generation_and_ranking() -> None:
assert [ranking.score for ranking in rankings] == pytest.approx([6.0, 16.0 / 3.0, 3.0])


def test_ruleset_strategy_rank_round_score_uses_per_episode_placement() -> None:
# Same episode results as the mean test above, but with scoring.round_score = "rank": the
# round score becomes the mean of each policy's per-episode rank points (placement N..1),
# not the mean raw score.
policy_version_ids = [uuid4() for _ in range(3)]
pool = PolicyPool(id=uuid4(), label="Round", pool_type="round", config={"num_episodes": 2})
entries = [
PolicyPoolEntry(pool_id=pool.id, policy_version_id=policy_version_id, seed_order=index)
for index, policy_version_id in enumerate(policy_version_ids)
]
commissioner = RulesetStrategyCommissioner(
{
"scoring": {"round_score": "rank"},
"divisions": {"competition": {"match": {"type": "competition"}, "entrants": "champions"}},
}
)

complete = commissioner.complete_round(
round_row=Round(id=uuid4(), division_id=uuid4(), round_number=1, commissioner_key="ruleset_strategy"),
pool=pool,
entries=entries,
episode_results=[
EpisodeResult(
episode_request_id=uuid4(),
scores=[
RoundPolicyScore(policy_version_id=policy_version_ids[0], score=4.0),
RoundPolicyScore(policy_version_id=policy_version_ids[1], score=2.0),
RoundPolicyScore(policy_version_id=policy_version_ids[2], score=6.0),
RoundPolicyScore(policy_version_id=policy_version_ids[0], score=8.0),
],
),
EpisodeResult(
episode_request_id=uuid4(),
scores=[
RoundPolicyScore(policy_version_id=policy_version_ids[1], score=10.0),
RoundPolicyScore(policy_version_id=policy_version_ids[2], score=0.0),
RoundPolicyScore(policy_version_id=policy_version_ids[0], score=6.0),
RoundPolicyScore(policy_version_id=policy_version_ids[1], score=4.0),
],
),
],
)

rankings = complete.results[0].rankings
by_policy = {ranking.policy_version_id: ranking for ranking in rankings}
# Per-episode rank points (N=4 each episode), averaged across a policy's seats:
# p0: ep1 scores 4->2pts, 8->4pts; ep2 score 6->3pts => (2+4+3)/3 = 3.0
# p1: ep1 score 2->1pt; ep2 10->4pts, 4->2pts => (1+4+2)/3 = 7/3
# p2: ep1 score 6->3pts; ep2 score 0->1pt => (3+1)/2 = 2.0
assert by_policy[policy_version_ids[0]].score == pytest.approx(3.0)
assert by_policy[policy_version_ids[1]].score == pytest.approx(7.0 / 3.0)
assert by_policy[policy_version_ids[2]].score == pytest.approx(2.0)
assert [ranking.policy_version_id for ranking in rankings] == policy_version_ids
assert by_policy[policy_version_ids[0]].result_metadata["score_kind"] == "rank_episode_round_score"


def test_default_commissioner_ignores_neutral_zero_scores_only_when_episode_has_negative_score() -> None:
policy_version_ids = [uuid4() for _ in range(3)]
pool = PolicyPool(
Expand Down
Loading