From a32c723b8a44690bb66002883058117d585ec034 Mon Sep 17 00:00:00 2001 From: Rudy Celekli Date: Tue, 6 Oct 2026 04:40:15 -0400 Subject: [PATCH 1/3] fix(sources): distinguish Hugging Face repository namespaces Signed-off-by: Rudy Celekli --- src/benchmark_radar/sources.py | 4 ++-- tests/test_sources.py | 33 +++++++++++++++++++++++++++++++++ 2 files changed, 35 insertions(+), 2 deletions(-) diff --git a/src/benchmark_radar/sources.py b/src/benchmark_radar/sources.py index 709ed224..ec3a1f2d 100644 --- a/src/benchmark_radar/sources.py +++ b/src/benchmark_radar/sources.py @@ -443,7 +443,7 @@ def fetch_arxiv(config: dict[str, Any], since: datetime, limit: int) -> list[Rad def fetch_huggingface(config: dict[str, Any], since: datetime, limit: int) -> list[RadarItem]: - found: dict[str, RadarItem] = {} + found: dict[tuple[str, str], RadarItem] = {} for kind in config.get("kinds", ["datasets"]): for search in config.get("searches", []): rows = get_json( @@ -476,7 +476,7 @@ def fetch_huggingface(config: dict[str, Any], since: datetime, limit: int) -> li or _reject_future(config, str(item_id), created, changed) ): continue - found[item_id] = RadarItem( + found[(kind, item_id)] = RadarItem( source="Hugging Face", source_id=item_id, title=item_id, diff --git a/tests/test_sources.py b/tests/test_sources.py index 9dba4b53..0e40ce24 100644 --- a/tests/test_sources.py +++ b/tests/test_sources.py @@ -3375,3 +3375,36 @@ def test_collection_method_falls_back_to_a_static_default_without_items(): assert collection_method("brave", []) == "API" assert collection_method("datacite", []) == "API" assert collection_method("openaire", []) == "API" + + +def test_huggingface_preserves_same_named_repositories_of_different_kinds(monkeypatch): + from benchmark_radar.pipeline import deduplicate + + def get_hub_rows(url, **_kwargs): + return [ + { + "id": "lab/suite", + "createdAt": "2026-08-08T12:00:00Z", + "lastModified": "2026-08-08T13:00:00Z", + "downloads": 3, + "likes": 2, + } + ] + + monkeypatch.setattr("benchmark_radar.sources.get_json", get_hub_rows) + items = fetch_huggingface( + {"kinds": ["datasets", "models", "spaces"], "searches": ["suite", "suite"]}, + datetime(2026, 8, 8, tzinfo=UTC), + 10, + ) + # Hub namespaces are independent: a lab can own a dataset, model and Space + # under the same name. Keying only on owner/name silently keeps the last. + assert len(items) == 3 + assert {item.source_id for item in items} == {"lab/suite"} + urls = {item.to_dict()["url"] for item in items} + assert len(urls) == 3 + assert { + "https://huggingface.co/datasets/lab/suite", + "https://huggingface.co/spaces/lab/suite", + } <= urls + assert len(deduplicate(items)) == 3 From cd92d393108b87f8a636e18b279006a82fc50e5c Mon Sep 17 00:00:00 2001 From: Rudy Celekli Date: Tue, 6 Oct 2026 05:53:44 -0400 Subject: [PATCH 2/3] fix(pipeline): deduplicate Hub repositories by exact kind identity Signed-off-by: Rudy Celekli --- src/benchmark_radar/pipeline.py | 8 +++++++- tests/test_sources.py | 30 +++++++++++++++++++++++++----- 2 files changed, 32 insertions(+), 6 deletions(-) diff --git a/src/benchmark_radar/pipeline.py b/src/benchmark_radar/pipeline.py index e771158e..aac9a64a 100644 --- a/src/benchmark_radar/pipeline.py +++ b/src/benchmark_radar/pipeline.py @@ -66,7 +66,13 @@ def dedupe_keys(item: RadarItem) -> list[str]: if not exact.startswith("artifact:url:") ] title_key = normalized_title(item.title) - if len(title_key) >= 24: + # Hub repository titles are their upstream owner/name, not a shared paper + # title. Only the exact kind-aware artifact identity can identify them: + # datasets, models and Spaces may use the same owner/name independently. + hub_repository = item.source == "Hugging Face" and any( + key.startswith("artifact:huggingface:") for key in keys + ) + if len(title_key) >= 24 and not hub_repository: keys.append(f"title:{hashlib.sha256(title_key.encode()).hexdigest()}") keys.append(f"url:{hashlib.sha256(canonical_url(item.url).encode()).hexdigest()}") return keys diff --git a/tests/test_sources.py b/tests/test_sources.py index 0e40ce24..ab3e6035 100644 --- a/tests/test_sources.py +++ b/tests/test_sources.py @@ -3377,13 +3377,16 @@ def test_collection_method_falls_back_to_a_static_default_without_items(): assert collection_method("openaire", []) == "API" -def test_huggingface_preserves_same_named_repositories_of_different_kinds(monkeypatch): +@pytest.mark.parametrize("repository_id", ["lab/suite", "lab/a-long-benchmark-suite-name"]) +def test_huggingface_preserves_same_named_repositories_of_different_kinds( + monkeypatch, repository_id +): from benchmark_radar.pipeline import deduplicate def get_hub_rows(url, **_kwargs): return [ { - "id": "lab/suite", + "id": repository_id, "createdAt": "2026-08-08T12:00:00Z", "lastModified": "2026-08-08T13:00:00Z", "downloads": 3, @@ -3400,11 +3403,28 @@ def get_hub_rows(url, **_kwargs): # Hub namespaces are independent: a lab can own a dataset, model and Space # under the same name. Keying only on owner/name silently keeps the last. assert len(items) == 3 - assert {item.source_id for item in items} == {"lab/suite"} + assert {item.source_id for item in items} == {repository_id} urls = {item.to_dict()["url"] for item in items} assert len(urls) == 3 assert { - "https://huggingface.co/datasets/lab/suite", - "https://huggingface.co/spaces/lab/suite", + f"https://huggingface.co/datasets/{repository_id}", + f"https://huggingface.co/spaces/{repository_id}", } <= urls assert len(deduplicate(items)) == 3 + # Repeated observations of each exact repository still merge; kind-specific + # lineage and measurements must not leak into its same-named siblings. + from copy import deepcopy + + duplicates = [] + for index, item in enumerate(items): + duplicate = deepcopy(item) + duplicate.metrics["likes"] = 10 + index + duplicate.artifact_urls = [f"https://evidence.example/{index}"] + duplicates.append(duplicate) + merged = deduplicate(items + duplicates) + assert len(merged) == 3 + for index, item in enumerate(merged): + assert item.source == "Hugging Face" + assert item.source_id == repository_id + assert item.metrics["likes"] == 10 + index + assert item.artifact_urls == [f"https://evidence.example/{index}"] From c16bf8f0abb4bc99aa9dec630088d5921aa03c3e Mon Sep 17 00:00:00 2001 From: Rudy Celekli Date: Wed, 7 Oct 2026 09:34:31 -0400 Subject: [PATCH 3/3] fix(query): preserve Hub kinds in public radar search --- src/benchmark_radar/query.py | 13 ++++- tests/test_query_surfaces.py | 94 +++++++++++++++++++++++++++++++++++- 2 files changed, 104 insertions(+), 3 deletions(-) diff --git a/src/benchmark_radar/query.py b/src/benchmark_radar/query.py index 301f922e..e3aab0c7 100644 --- a/src/benchmark_radar/query.py +++ b/src/benchmark_radar/query.py @@ -13,6 +13,7 @@ from typing import TYPE_CHECKING, Any from .citation import citation_block, required_citations +from .corpus import exact_artifact_key from .science_domains import science_domains_for_record from .snapshots import REQUIRED_SOURCES, load_snapshots @@ -485,10 +486,18 @@ def _radar_candidates(self) -> list[dict[str, Any]]: for item in snapshot["evidence_items"]: source = str(item.get("source") or "") source_id = str(item.get("source_id") or "") + identity = source_id + if source == "Hugging Face": + # Hub kinds have independent owner/name namespaces. Use + # the primary repository URL, excluding related artifacts, + # to keep each kind's latest observation and public key. + identity = exact_artifact_key( + {"source": source, "source_id": source_id, "url": item.get("url")} + ) urls = [str(item.get("url") or ""), *(item.get("artifact_urls") or [])] - latest_by_identity[(source, source_id)] = { + latest_by_identity[(source, identity)] = { "kind": "radar", - "key": f"radar:{source.casefold()}:{source_id}", + "key": f"radar:{source.casefold()}:{identity}", "slug": None, "name": str(item.get("title") or ""), "description": str(item.get("summary") or ""), diff --git a/tests/test_query_surfaces.py b/tests/test_query_surfaces.py index f0c0ac5f..795ba06e 100644 --- a/tests/test_query_surfaces.py +++ b/tests/test_query_surfaces.py @@ -4,6 +4,7 @@ import threading import urllib.parse import urllib.request +from copy import deepcopy from datetime import UTC, datetime, timedelta from pathlib import Path @@ -18,10 +19,12 @@ latex_citation, ) from benchmark_radar.models import RadarItem, RadarRun, SourceHealth +from benchmark_radar.pipeline import deduplicate from benchmark_radar.query import QueryError, QueryPaths, QueryService, _tokens from benchmark_radar.query_cli import run_query_cli from benchmark_radar.query_http import create_query_server -from benchmark_radar.snapshots import write_snapshot +from benchmark_radar.snapshots import load_snapshots, write_snapshot +from benchmark_radar.sources import fetch_huggingface def _catalog(tmp_path: Path) -> QueryPaths: @@ -514,6 +517,95 @@ def test_radar_results_carry_derived_science_domains(tmp_path: Path) -> None: assert radar_hits and radar_hits[0]["science_domains"] == ["neuroscience"] +@pytest.mark.parametrize("repository_id", ["lab/suite", "lab/a-long-benchmark-suite-name", None]) +def test_hub_kinds_survive_collection_snapshot_and_public_search( + tmp_path: Path, monkeypatch, capsys, repository_id: str | None +) -> None: + paths = _catalog(tmp_path) + generated_at = datetime(2026, 8, 30, 8, tzinfo=UTC) + + def get_hub_rows(url, **_kwargs): + kind = url.rsplit("/", 1)[-1] + return [ + { + "id": repository_id or f"lab/{kind}-suite", + "createdAt": "2026-08-30T06:00:00Z", + "lastModified": "2026-08-30T07:00:00Z", + "downloads": 3, + "likes": 2, + } + ] + + monkeypatch.setattr("benchmark_radar.sources.get_json", get_hub_rows) + items = fetch_huggingface( + {"kinds": ["datasets", "models", "spaces"], "searches": ["suite", "suite"]}, + generated_at - timedelta(days=1), + 10, + ) + assert len(items) == 3 + items = deduplicate(items) + assert len(items) == 3 + run = RadarRun( + generated_at=generated_at, + since=generated_at - timedelta(days=1), + items=items, + health=[SourceHealth(source="huggingface", ok=True, item_count=3, method="API")], + ) + write_snapshot(run, paths.snapshots) + write_snapshot(run, paths.snapshots) + assert len(load_snapshots(paths.snapshots)[-1]["evidence_items"]) == 3 + + initial = QueryService(paths).search("suite", scope="radar", limit=10) + assert initial["total_matches"] == 3 + keys_by_url = {row["url"]: row["key"] for row in initial["results"]} + assert len(set(keys_by_url.values())) == 3 + assert {row["source_id"] for row in initial["results"]} == {item.source_id for item in items} + + # A later observation of only the dataset replaces that kind, while its + # same-named model and Space remain discoverable with stable public keys. + dataset = deepcopy(next(item for item in items if "/datasets/" in item.url)) + dataset.summary = "A newer dataset observation." + dataset.updated_at = generated_at + timedelta(days=1) + write_snapshot( + RadarRun( + generated_at=generated_at + timedelta(days=1), + since=generated_at, + items=[dataset], + health=[SourceHealth(source="huggingface", ok=True, item_count=1, method="API")], + ), + paths.snapshots, + ) + searched = QueryService(paths).search("suite", scope="radar", limit=10) + assert searched["total_matches"] == 3 + assert {row["url"]: row["key"] for row in searched["results"]} == keys_by_url + for row in searched["results"]: + if row["url"] == dataset.url: + assert row["description"] == dataset.summary + assert row["snapshot_date"] == "2026-08-31" + else: + assert row["snapshot_date"] == "2026-08-30" + + exit_code = run_query_cli( + [ + "search", + "suite", + "--scope", + "radar", + "--limit", + "10", + "--json", + "--index", + str(paths.index), + "--shards", + str(paths.shards), + "--snapshots", + str(paths.snapshots), + ] + ) + assert exit_code == 0 + assert json.loads(capsys.readouterr().out) == searched + + def test_status_exposes_incomplete_detail_shards(tmp_path: Path) -> None: # Regression: counting only the index used to hide absent detail artifacts. paths = _catalog(tmp_path)