Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 7 additions & 1 deletion src/benchmark_radar/pipeline.py
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,13 @@ def dedupe_keys(item: RadarItem) -> list[str]:
if not exact.startswith("artifact:url:")
]
title_key = normalized_title(item.title)
if len(title_key) >= 24:
# Hub repository titles are their upstream owner/name, not a shared paper
# title. Only the exact kind-aware artifact identity can identify them:
# datasets, models and Spaces may use the same owner/name independently.
hub_repository = item.source == "Hugging Face" and any(
key.startswith("artifact:huggingface:") for key in keys
)
if len(title_key) >= 24 and not hub_repository:
keys.append(f"title:{hashlib.sha256(title_key.encode()).hexdigest()}")
keys.append(f"url:{hashlib.sha256(canonical_url(item.url).encode()).hexdigest()}")
return keys
Expand Down
13 changes: 11 additions & 2 deletions src/benchmark_radar/query.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,7 @@
from typing import TYPE_CHECKING, Any

from .citation import citation_block, required_citations
from .corpus import exact_artifact_key
from .science_domains import science_domains_for_record
from .snapshots import REQUIRED_SOURCES, load_snapshots

Expand Down Expand Up @@ -485,10 +486,18 @@ def _radar_candidates(self) -> list[dict[str, Any]]:
for item in snapshot["evidence_items"]:
source = str(item.get("source") or "")
source_id = str(item.get("source_id") or "")
identity = source_id
if source == "Hugging Face":
# Hub kinds have independent owner/name namespaces. Use
# the primary repository URL, excluding related artifacts,
# to keep each kind's latest observation and public key.
identity = exact_artifact_key(
{"source": source, "source_id": source_id, "url": item.get("url")}
)
urls = [str(item.get("url") or ""), *(item.get("artifact_urls") or [])]
latest_by_identity[(source, source_id)] = {
latest_by_identity[(source, identity)] = {
"kind": "radar",
"key": f"radar:{source.casefold()}:{source_id}",
"key": f"radar:{source.casefold()}:{identity}",
"slug": None,
"name": str(item.get("title") or ""),
"description": str(item.get("summary") or ""),
Expand Down
4 changes: 2 additions & 2 deletions src/benchmark_radar/sources.py
Original file line number Diff line number Diff line change
Expand Up @@ -443,7 +443,7 @@ def fetch_arxiv(config: dict[str, Any], since: datetime, limit: int) -> list[Rad


def fetch_huggingface(config: dict[str, Any], since: datetime, limit: int) -> list[RadarItem]:
found: dict[str, RadarItem] = {}
found: dict[tuple[str, str], RadarItem] = {}
for kind in config.get("kinds", ["datasets"]):
for search in config.get("searches", []):
rows = get_json(
Expand Down Expand Up @@ -476,7 +476,7 @@ def fetch_huggingface(config: dict[str, Any], since: datetime, limit: int) -> li
or _reject_future(config, str(item_id), created, changed)
):
continue
found[item_id] = RadarItem(
found[(kind, item_id)] = RadarItem(
Comment thread
rudycelekli marked this conversation as resolved.
source="Hugging Face",
source_id=item_id,
title=item_id,
Expand Down
94 changes: 93 additions & 1 deletion tests/test_query_surfaces.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,6 +4,7 @@
import threading
import urllib.parse
import urllib.request
from copy import deepcopy
from datetime import UTC, datetime, timedelta
from pathlib import Path

Expand All @@ -18,10 +19,12 @@
latex_citation,
)
from benchmark_radar.models import RadarItem, RadarRun, SourceHealth
from benchmark_radar.pipeline import deduplicate
from benchmark_radar.query import QueryError, QueryPaths, QueryService, _tokens
from benchmark_radar.query_cli import run_query_cli
from benchmark_radar.query_http import create_query_server
from benchmark_radar.snapshots import write_snapshot
from benchmark_radar.snapshots import load_snapshots, write_snapshot
from benchmark_radar.sources import fetch_huggingface


def _catalog(tmp_path: Path) -> QueryPaths:
Expand Down Expand Up @@ -514,6 +517,95 @@ def test_radar_results_carry_derived_science_domains(tmp_path: Path) -> None:
assert radar_hits and radar_hits[0]["science_domains"] == ["neuroscience"]


@pytest.mark.parametrize("repository_id", ["lab/suite", "lab/a-long-benchmark-suite-name", None])
def test_hub_kinds_survive_collection_snapshot_and_public_search(
tmp_path: Path, monkeypatch, capsys, repository_id: str | None
) -> None:
paths = _catalog(tmp_path)
generated_at = datetime(2026, 8, 30, 8, tzinfo=UTC)

def get_hub_rows(url, **_kwargs):
kind = url.rsplit("/", 1)[-1]
return [
{
"id": repository_id or f"lab/{kind}-suite",
"createdAt": "2026-08-30T06:00:00Z",
"lastModified": "2026-08-30T07:00:00Z",
"downloads": 3,
"likes": 2,
}
]

monkeypatch.setattr("benchmark_radar.sources.get_json", get_hub_rows)
items = fetch_huggingface(
{"kinds": ["datasets", "models", "spaces"], "searches": ["suite", "suite"]},
generated_at - timedelta(days=1),
10,
)
assert len(items) == 3
items = deduplicate(items)
assert len(items) == 3
run = RadarRun(
generated_at=generated_at,
since=generated_at - timedelta(days=1),
items=items,
health=[SourceHealth(source="huggingface", ok=True, item_count=3, method="API")],
)
write_snapshot(run, paths.snapshots)
write_snapshot(run, paths.snapshots)
assert len(load_snapshots(paths.snapshots)[-1]["evidence_items"]) == 3

initial = QueryService(paths).search("suite", scope="radar", limit=10)
assert initial["total_matches"] == 3
keys_by_url = {row["url"]: row["key"] for row in initial["results"]}
assert len(set(keys_by_url.values())) == 3
assert {row["source_id"] for row in initial["results"]} == {item.source_id for item in items}

# A later observation of only the dataset replaces that kind, while its
# same-named model and Space remain discoverable with stable public keys.
dataset = deepcopy(next(item for item in items if "/datasets/" in item.url))
dataset.summary = "A newer dataset observation."
dataset.updated_at = generated_at + timedelta(days=1)
write_snapshot(
RadarRun(
generated_at=generated_at + timedelta(days=1),
since=generated_at,
items=[dataset],
health=[SourceHealth(source="huggingface", ok=True, item_count=1, method="API")],
),
paths.snapshots,
)
searched = QueryService(paths).search("suite", scope="radar", limit=10)
assert searched["total_matches"] == 3
assert {row["url"]: row["key"] for row in searched["results"]} == keys_by_url
for row in searched["results"]:
if row["url"] == dataset.url:
assert row["description"] == dataset.summary
assert row["snapshot_date"] == "2026-08-31"
else:
assert row["snapshot_date"] == "2026-08-30"

exit_code = run_query_cli(
[
"search",
"suite",
"--scope",
"radar",
"--limit",
"10",
"--json",
"--index",
str(paths.index),
"--shards",
str(paths.shards),
"--snapshots",
str(paths.snapshots),
]
)
assert exit_code == 0
assert json.loads(capsys.readouterr().out) == searched


def test_status_exposes_incomplete_detail_shards(tmp_path: Path) -> None:
# Regression: counting only the index used to hide absent detail artifacts.
paths = _catalog(tmp_path)
Expand Down
53 changes: 53 additions & 0 deletions tests/test_sources.py
Original file line number Diff line number Diff line change
Expand Up @@ -3377,6 +3377,59 @@ def test_collection_method_falls_back_to_a_static_default_without_items():
assert collection_method("openaire", []) == "API"


@pytest.mark.parametrize("repository_id", ["lab/suite", "lab/a-long-benchmark-suite-name"])
def test_huggingface_preserves_same_named_repositories_of_different_kinds(
monkeypatch, repository_id
):
from benchmark_radar.pipeline import deduplicate

def get_hub_rows(url, **_kwargs):
return [
{
"id": repository_id,
"createdAt": "2026-08-08T12:00:00Z",
"lastModified": "2026-08-08T13:00:00Z",
"downloads": 3,
"likes": 2,
}
]

monkeypatch.setattr("benchmark_radar.sources.get_json", get_hub_rows)
items = fetch_huggingface(
{"kinds": ["datasets", "models", "spaces"], "searches": ["suite", "suite"]},
datetime(2026, 8, 8, tzinfo=UTC),
10,
)
# Hub namespaces are independent: a lab can own a dataset, model and Space
# under the same name. Keying only on owner/name silently keeps the last.
assert len(items) == 3
assert {item.source_id for item in items} == {repository_id}
urls = {item.to_dict()["url"] for item in items}
assert len(urls) == 3
assert {
f"https://huggingface.co/datasets/{repository_id}",
f"https://huggingface.co/spaces/{repository_id}",
} <= urls
assert len(deduplicate(items)) == 3
# Repeated observations of each exact repository still merge; kind-specific
# lineage and measurements must not leak into its same-named siblings.
from copy import deepcopy

duplicates = []
for index, item in enumerate(items):
duplicate = deepcopy(item)
duplicate.metrics["likes"] = 10 + index
duplicate.artifact_urls = [f"https://evidence.example/{index}"]
duplicates.append(duplicate)
merged = deduplicate(items + duplicates)
assert len(merged) == 3
for index, item in enumerate(merged):
assert item.source == "Hugging Face"
assert item.source_id == repository_id
assert item.metrics["likes"] == 10 + index
assert item.artifact_urls == [f"https://evidence.example/{index}"]


def test_zenodo_keeps_identical_descriptions_with_unknown_creators(monkeypatch):
rows = [
{
Expand Down
Loading