Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
24 changes: 21 additions & 3 deletions src/benchmark_radar/authors.py
Original file line number Diff line number Diff line change
Expand Up @@ -132,13 +132,31 @@ def popular_repositories(
a week is a real signal; so is a highly starred one seen once.
"""
corpus = build_corpus(snapshots)
observed_repos: dict[str, dict[str, str]] = {}
for observation in corpus["observations"]:
match = _GITHUB_REPO.match(str(observation.get("url") or ""))
if match:
owner, name = match.group(1), match.group(2).removesuffix(".git")
full_name = f"{owner}/{name}"
observed_repos.setdefault(observation["entity_id"], {})[full_name.casefold()] = (
f"https://github.com/{full_name}"
)
ranked = []
for entity in corpus["entities"]:
if entity["type"] != "artifact":
continue
match = _GITHUB_REPO.match(str(entity.get("url") or ""))
repo_url = str(entity.get("url") or "")
match = _GITHUB_REPO.match(repo_url)
if not match:
continue
candidates = observed_repos.get(entity["id"], {})
# Recover only an actually observed, unambiguous repository. A
# paper's links alone must not invent seeds or assign stars to one
# of several distinct repositories joined by the artifact graph.
if len(candidates) != 1:
continue
repo_url = next(iter(candidates.values()))
match = _GITHUB_REPO.match(repo_url)
assert match is not None
owner, name = match.group(1), match.group(2).removesuffix(".git")
if owner.lower() in {"apps", "marketplace", "sponsors", "topics"}:
continue
Expand All @@ -155,7 +173,7 @@ def popular_repositories(
"full_name": f"{owner}/{name}",
"owner": owner,
"name": name,
"url": entity.get("url"),
"url": repo_url,
"title": entity.get("label"),
"stars": float(metrics.get("stars") or 0),
"seen_days": len(entity.get("seen_days") or []),
Expand Down
47 changes: 47 additions & 0 deletions tests/test_authors.py
Original file line number Diff line number Diff line change
Expand Up @@ -163,3 +163,50 @@ def test_contacts_csv_contains_every_contact_and_flattens_lists():
assert "data quality; dataset" in rendered
assert "org/one; org/two" in rendered
assert "other" in rendered


def test_survey_keeps_observed_repository_when_corpus_prefers_linked_paper():
repo = _repo("example/benchmark", stars=100, categories=["benchmark"])
paper = RadarItem(
source="arXiv",
source_id="2608.12345",
title="A benchmark paper",
url="https://arxiv.org/abs/2608.12345",
published_at=repo.published_at,
categories=["benchmark"],
artifact_urls=[repo.url],
summary="A scored benchmark with an observed repository.",
)
current = snapshot_for_run(_run([repo, paper]))
ranked = authors.popular_repositories([current])
assert [row["full_name"] for row in ranked] == ["example/benchmark"]
assert ranked[0]["stars"] == 100
assert ranked[0]["url"] == repo.url


def test_survey_does_not_treat_unobserved_paper_links_as_repository_seeds():
paper = RadarItem(
source="arXiv",
source_id="2608.12345",
title="A benchmark paper",
url="https://arxiv.org/abs/2608.12345",
published_at=datetime(2026, 8, 4, tzinfo=UTC),
categories=["benchmark"],
artifact_urls=["https://github.com/example/unobserved"],
)
assert authors.popular_repositories([snapshot_for_run(_run([paper]))]) == []


def test_survey_does_not_assign_aggregate_stars_to_ambiguous_repository_aliases():
first = _repo("one/benchmark", stars=100, categories=["benchmark"])
second = _repo("two/benchmark", stars=50, categories=["benchmark"])
paper = RadarItem(
source="arXiv",
source_id="2608.12345",
title="A benchmark paper",
url="https://arxiv.org/abs/2608.12345",
published_at=first.published_at,
categories=["benchmark"],
artifact_urls=[first.url, second.url],
)
assert authors.popular_repositories([snapshot_for_run(_run([first, second, paper]))]) == []
Loading