Skip to content
Closed
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
108 changes: 60 additions & 48 deletions src/benchmark_radar/sources.py
Original file line number Diff line number Diff line change
Expand Up @@ -225,6 +225,17 @@ def _request_options(config: dict[str, Any]) -> dict[str, Any]:
}


def _reported_metrics(payload: dict[str, Any], fields: dict[str, str]) -> dict[str, float]:
# An omitted or null upstream counter is unknown, not a measured zero.
# Keep an explicit 0 so downstream adoption and trend calculations can
# distinguish actual zero activity from an unreported measurement.
return {
Comment thread
rudycelekli marked this conversation as resolved.
metric: float(payload[source])
for metric, source in fields.items()
if payload.get(source) is not None and payload[source] != ""
}


def _openreview_value(content: dict[str, Any], key: str, default: Any = None) -> Any:
value = content.get(key, default)
if isinstance(value, dict) and "value" in value:
Expand Down Expand Up @@ -490,10 +501,7 @@ def fetch_huggingface(config: dict[str, Any], since: datetime, limit: int) -> li
event_kind=(
"released" if created is not None and created >= since else "updated"
),
metrics={
"downloads": float(row.get("downloads") or 0),
"likes": float(row.get("likes") or 0),
},
metrics=_reported_metrics(row, {"downloads": "downloads", "likes": "likes"}),
raw=row,
parser_version="huggingface-hub/1",
)
Expand Down Expand Up @@ -619,10 +627,9 @@ def fetch_github(config: dict[str, Any], since: datetime, limit: int) -> list[Ra
event_kind=(
"released" if created is not None and created >= since else "updated"
),
metrics={
"stars": float(row.get("stargazers_count") or 0),
"forks": float(row.get("forks_count") or 0),
},
metrics=_reported_metrics(
row, {"stars": "stargazers_count", "forks": "forks_count"}
),
raw=row,
parser_version="github-search/1",
)
Expand Down Expand Up @@ -739,10 +746,9 @@ def fetch_github_organizations(
summary=github_summary(row),
event_kind="released",
organizations=[organization["display_name"]],
metrics={
"stars": float(row.get("stargazers_count") or 0),
"forks": float(row.get("forks_count") or 0),
},
metrics=_reported_metrics(
row, {"stars": "stargazers_count", "forks": "forks_count"}
),
raw={"repository": row, "organization_tier": organization["tier"]},
parser_version=GITHUB_ORGANIZATIONS_PARSER_VERSION,
)
Expand Down Expand Up @@ -822,11 +828,14 @@ def fetch_kaggle_datasets(
authors=(
[str(row.get("creatorName") or "").strip()] if row.get("creatorName") else []
),
metrics={
"downloads": float(row.get("downloadCount") or 0),
"votes": float(row.get("voteCount") or 0),
"views": float(row.get("viewCount") or 0),
},
metrics=_reported_metrics(
row,
{
"downloads": "downloadCount",
"votes": "voteCount",
"views": "viewCount",
},
),
raw=row,
parser_version=KAGGLE_DATASETS_PARSER_VERSION,
)
Expand Down Expand Up @@ -884,7 +893,7 @@ def fetch_huggingface_papers(
event_kind="discovered",
authors=authors,
artifact_urls=sorted(set(artifact_urls)),
metrics={"upvotes": float(paper.get("upvotes") or 0)},
metrics=_reported_metrics(paper, {"upvotes": "upvotes"}),
raw=row,
parser_version=HUGGINGFACE_PAPERS_PARSER_VERSION,
)
Expand Down Expand Up @@ -953,10 +962,7 @@ def fetch_zenodo_records(
if isinstance(creator, dict) and str(creator.get("name") or "").strip()
],
artifact_urls=artifact_urls,
metrics={
"downloads": float(stats.get("downloads") or 0),
"views": float(stats.get("views") or 0),
},
metrics=_reported_metrics(stats, {"downloads": "downloads", "views": "views"}),
raw=row,
parser_version="zenodo-records/1",
)
Expand Down Expand Up @@ -1118,7 +1124,7 @@ def fetch_crossref(
authors=author_names,
organizations=list(dict.fromkeys(organizations)),
artifact_urls=[doi_url],
metrics={"citations": float(row.get("is-referenced-by-count") or 0)},
metrics=_reported_metrics(row, {"citations": "is-referenced-by-count"}),
raw=row,
parser_version="crossref-works/1",
)
Expand Down Expand Up @@ -1380,9 +1386,8 @@ def fetch_openaire(
),
artifact_urls=artifact_urls,
metrics={
"citations": float(citations.get("citationCount") or 0),
"downloads": float(usage.get("downloads") or 0),
"views": float(usage.get("views") or 0),
**_reported_metrics(citations, {"citations": "citationCount"}),
**_reported_metrics(usage, {"downloads": "downloads", "views": "views"}),
},
raw=row,
parser_version="openaire-graph-v3/1",
Expand Down Expand Up @@ -1646,11 +1651,14 @@ def fetch_datacite(
authors=authors,
organizations=list(dict.fromkeys(organizations)),
artifact_urls=artifact_urls,
metrics={
"citations": float(attributes.get("citationCount") or 0),
"downloads": float(attributes.get("downloadCount") or 0),
"views": float(attributes.get("viewCount") or 0),
},
metrics=_reported_metrics(
attributes,
{
"citations": "citationCount",
"downloads": "downloadCount",
"views": "viewCount",
},
),
raw=row,
parser_version="datacite-dois/1",
)
Expand Down Expand Up @@ -1869,10 +1877,13 @@ def fetch_semantic_scholar(
if isinstance(author, dict) and author.get("name")
],
artifact_urls=sorted(set(artifact_urls)),
metrics={
"citations": float(row.get("citationCount") or 0),
"influential_citations": float(row.get("influentialCitationCount") or 0),
},
metrics=_reported_metrics(
row,
{
"citations": "citationCount",
"influential_citations": "influentialCitationCount",
},
),
raw=row,
parser_version="semantic-scholar-graph/1",
)
Expand Down Expand Up @@ -2000,6 +2011,14 @@ def fetch_github_releases(
isinstance(asset, dict) for asset in assets
):
raise ConnectorPayloadError("GitHub release assets must be an array")
# A release with no assets has zero downloads; an omitted assets
# list or an asset without its counter has unknown downloads.
download_metrics = (
{"downloads": float(sum(int(asset["download_count"]) for asset in assets))}
if row.get("assets") is not None
and all(asset.get("download_count") not in (None, "") for asset in assets)
else {}
)
found[f"{repository}@{tag}"] = RadarItem(
source="GitHub Release",
source_id=f"{repository}@{tag}",
Expand All @@ -2015,15 +2034,7 @@ def fetch_github_releases(
else []
),
artifact_urls=[f"https://github.com/{repository}"],
metrics={
"downloads": float(
sum(
int(asset.get("download_count") or 0)
for asset in assets
if isinstance(asset, dict)
)
)
},
metrics=download_metrics,
raw=row,
parser_version=GITHUB_RELEASE_PARSER_VERSION,
)
Expand Down Expand Up @@ -2061,8 +2072,9 @@ def fetch_github_releases(
)
if not isinstance(repository_payload, dict):
raise ConnectorPayloadError("GitHub repository metadata was not an object")
stars = float(repository_payload.get("stargazers_count") or 0)
forks = float(repository_payload.get("forks_count") or 0)
popularity = _reported_metrics(
repository_payload, {"stars": "stargazers_count", "forks": "forks_count"}
)
except Exception as error:
config.setdefault("_source_warnings", []).append(
f"{repository} metadata: {type(error).__name__}: {error}"
Expand All @@ -2071,7 +2083,7 @@ def fetch_github_releases(
for item in found.values():
if not item.source_id.startswith(f"{repository}@"):
continue
item.metrics.update({"stars": stars, "forks": forks})
item.metrics.update(popularity)
item.raw = {"release": item.raw, "repository": repository_payload}
return sorted(found.values(), key=lambda item: item.published_at, reverse=True)[:limit]

Expand Down Expand Up @@ -2170,7 +2182,7 @@ def fetch_openalex(
event_kind="released",
authors=[author for author in authors if author],
organizations=organizations,
metrics={"citations": float(row.get("cited_by_count") or 0)},
metrics=_reported_metrics(row, {"citations": "cited_by_count"}),
raw=row,
parser_version="openalex-works/1",
)
Expand Down
115 changes: 113 additions & 2 deletions tests/test_sources.py
Original file line number Diff line number Diff line change
Expand Up @@ -636,6 +636,22 @@ def test_github_preserves_creation_and_update_times(monkeypatch):
assert items[0].event_kind == "updated"


def test_github_keeps_missing_forks_unknown(monkeypatch):
row = _github_row(1)
row.pop("forks_count")
row["stargazers_count"] = 0
monkeypatch.setattr(
"benchmark_radar.sources.get_json",
lambda url, params=None, headers=None: {"items": [row]},
)
items = fetch_github(
{"queries": ["benchmark"], "request_delay_seconds": 0},
datetime(2026, 7, 26, tzinfo=UTC),
10,
)
assert items[0].metrics == {"stars": 0.0}


def test_github_config_discovers_and_routes_rsi_exam(monkeypatch):
"""Issue #408: the named benchmark matched no configured GitHub query."""
config = yaml.safe_load(Path("config.yml").read_text(encoding="utf-8"))
Expand Down Expand Up @@ -1118,6 +1134,33 @@ def _openaire_payload(**fields):
return _openaire_rows_payload(_openaire_row(**fields))


@pytest.mark.parametrize(
"state,counts,expected",
[
("missing", (None, None, None), {}),
("null", (None, None, None), {}),
("empty", ("", "", ""), {}),
("zero", (0, 0, 0), {"citations": 0.0, "downloads": 0.0, "views": 0.0}),
("positive", (3, 13, 21), {"citations": 3.0, "downloads": 13.0, "views": 21.0}),
("mixed", (None, 0, 7), {"downloads": 0.0, "views": 7.0}),
],
)
def test_openaire_preserves_unknown_and_reported_counters(monkeypatch, state, counts, expected):
# Review #704: the new connector still converted unknown counters to zero,
# even though the shared rule requires omission and retains measured zero.
citations = {} if state == "missing" else {"citationCount": counts[0]}
usage = {} if state == "missing" else {"downloads": counts[1], "views": counts[2]}
row = _openaire_row(indicators={"citationImpact": citations, "usageCounts": usage})
monkeypatch.setattr(
"benchmark_radar.sources.get_json",
lambda url, **kwargs: _openaire_rows_payload(row),
)
items = fetch_openaire({"searches": ["benchmark"]}, datetime(2026, 7, 26, tzinfo=UTC), 10)
assert len(items) == 1
assert items[0].metrics == expected
assert items[0].source_id == row["id"]


def test_openaire_preserves_upstream_metadata_and_bounds_the_query(monkeypatch):
calls = []

Expand Down Expand Up @@ -1644,7 +1687,7 @@ def test_openaire_preserves_a_product_that_carries_almost_nothing(monkeypatch):
item = items[0]
assert item.url == "https://doi.org/10.1000/sparse"
assert item.artifact_urls == ["https://doi.org/10.1000/sparse"]
assert item.metrics == {"citations": 0.0, "downloads": 0.0, "views": 0.0}
assert item.metrics == {}
Comment thread
rudycelekli marked this conversation as resolved.
assert item.authors == []
assert item.organizations == []
assert item.summary == ""
Expand Down Expand Up @@ -1745,6 +1788,34 @@ def _datacite_rows_payload(*rows):
}


@pytest.mark.parametrize(
"state,counts,expected",
[
("missing", (None, None, None), {}),
("null", (None, None, None), {}),
("empty", ("", "", ""), {}),
("zero", (0, 0, 0), {"citations": 0.0, "downloads": 0.0, "views": 0.0}),
("positive", (3, 13, 21), {"citations": 3.0, "downloads": 13.0, "views": 21.0}),
("mixed", (None, 0, 7), {"downloads": 0.0, "views": 7.0}),
],
)
def test_datacite_preserves_unknown_and_reported_counters(monkeypatch, state, counts, expected):
row = _datacite_row()
for field, count in zip(("citationCount", "downloadCount", "viewCount"), counts, strict=True):
if state == "missing":
row["attributes"].pop(field)
else:
row["attributes"][field] = count
monkeypatch.setattr(
"benchmark_radar.sources.get_json",
lambda url, **kwargs: _datacite_rows_payload(row),
)
items = fetch_datacite({"searches": ["benchmark"]}, datetime(2026, 7, 26, tzinfo=UTC), 10)
assert len(items) == 1
assert items[0].metrics == expected
assert items[0].source_id == row["id"]


def test_datacite_preserves_doi_metadata_and_bounds_the_query(monkeypatch):
calls = []

Expand Down Expand Up @@ -2045,7 +2116,7 @@ def test_datacite_preserves_a_deposit_that_carries_almost_nothing(monkeypatch):

item = items[0]
assert item.artifact_urls == ["https://doi.org/10.5281/zenodo.1"]
assert item.metrics == {"citations": 0.0, "downloads": 0.0, "views": 0.0}
assert item.metrics == {}
assert item.authors == []
assert item.organizations == []
assert item.summary == ""
Expand Down Expand Up @@ -2445,6 +2516,26 @@ def fake_get_json(url, **kwargs):
assert radar_item.adoption_score > 0


def test_github_release_omits_incomplete_download_totals_and_repository_counters(monkeypatch):
release = {
"tag_name": "v2",
"html_url": "https://github.com/example/benchmark/releases/tag/v2",
"published_at": "2026-07-27T12:00:00Z",
"assets": [{"download_count": 4}, {"name": "missing-count.zip"}],
}

def fake_get_json(url, **kwargs):
return [release] if url.endswith("/releases") else {"stargazers_count": 0}

monkeypatch.setattr("benchmark_radar.sources.get_json", fake_get_json)
item = fetch_github_releases(
{"repositories": ["example/benchmark"], "repository_metadata_requests": 1},
datetime(2026, 7, 26, tzinfo=UTC),
10,
)[0]
assert item.metrics == {"stars": 0.0}


def test_github_release_repository_counters_are_covered_by_raw_hash(monkeypatch):
stars = 7_000

Expand Down Expand Up @@ -3165,6 +3256,26 @@ def test_huggingface_preserves_creation_and_update_times(monkeypatch):
assert items[0].event_kind == "updated"


def test_huggingface_does_not_publish_absent_counters_as_zero(monkeypatch):
# A missing API field is unknown, while a reported zero is a real measurement.
monkeypatch.setattr(
"benchmark_radar.sources.get_json",
lambda url, params: [
{
"id": "org/benchmark",
"lastModified": "2026-07-27T12:00:00Z",
"likes": 0,
}
],
)
items = fetch_huggingface(
{"kinds": ["datasets"], "searches": ["benchmark"]},
datetime(2026, 7, 26, tzinfo=UTC),
10,
)
assert items[0].metrics == {"likes": 0.0}


def test_huggingface_filters_future_rows_before_the_local_cap(monkeypatch):
seen_limit = []

Expand Down
Loading