diff --git a/src/benchmark_radar/snapshots.py b/src/benchmark_radar/snapshots.py index 3d7af171..48751969 100644 --- a/src/benchmark_radar/snapshots.py +++ b/src/benchmark_radar/snapshots.py @@ -46,7 +46,11 @@ from .site_about import write_about from .site_pages import DEFAULT_SHARD_DIR, benchmark_sitemap_entries from .site_seo import site_lastmod, write_sitemap -from .sources import GITHUB_RELEASE_PARSER_VERSION, github_release_title +from .sources import github_release_title + +# This migration only applies the release-title correction introduced in /3. +# It does not reparse metrics and must not inherit newer fetch semantics. +GITHUB_RELEASE_TITLE_BACKFILL_PARSER_VERSION = "github-releases/3" SCHEMA_VERSION = 2 SUPPORTED_SCHEMA_VERSIONS = {1, SCHEMA_VERSION} @@ -1561,7 +1565,7 @@ def migrate_snapshot_history(config: dict[str, Any], snapshot_dir: Path) -> list def _backfill_github_release_titles(snapshot: dict[str, Any]) -> int: - """Reparse persisted bare-tag release titles with the current connector.""" + """Apply the /3 title correction without reparsing persisted metrics.""" changed = 0 for record in snapshot.get("evidence_items") or []: if record.get("source") != "GitHub Release": @@ -1577,6 +1581,6 @@ def _backfill_github_release_titles(snapshot: dict[str, Any]) -> int: if corrected == current: continue record["title"] = corrected - record["parser_version"] = GITHUB_RELEASE_PARSER_VERSION + record["parser_version"] = GITHUB_RELEASE_TITLE_BACKFILL_PARSER_VERSION changed += 1 return changed diff --git a/src/benchmark_radar/sources.py b/src/benchmark_radar/sources.py index 709ed224..97ad23df 100644 --- a/src/benchmark_radar/sources.py +++ b/src/benchmark_radar/sources.py @@ -29,7 +29,7 @@ class ConnectorPayloadError(ValueError): FUTURE_TIMESTAMP_TOLERANCE = timedelta(minutes=5) -GITHUB_RELEASE_PARSER_VERSION = "github-releases/3" +GITHUB_RELEASE_PARSER_VERSION = "github-releases/4" def _xml_local_name(tag: str) -> str: @@ -225,6 +225,17 @@ def _request_options(config: dict[str, Any]) -> dict[str, Any]: } +def _reported_metrics(payload: dict[str, Any], fields: dict[str, str]) -> dict[str, float]: + # An omitted or null upstream counter is unknown, not a measured zero. + # Keep an explicit 0 so downstream adoption and trend calculations can + # distinguish actual zero activity from an unreported measurement. + return { + metric: float(payload[source]) + for metric, source in fields.items() + if payload.get(source) is not None and payload[source] != "" + } + + def _openreview_value(content: dict[str, Any], key: str, default: Any = None) -> Any: value = content.get(key, default) if isinstance(value, dict) and "value" in value: @@ -490,12 +501,9 @@ def fetch_huggingface(config: dict[str, Any], since: datetime, limit: int) -> li event_kind=( "released" if created is not None and created >= since else "updated" ), - metrics={ - "downloads": float(row.get("downloads") or 0), - "likes": float(row.get("likes") or 0), - }, + metrics=_reported_metrics(row, {"downloads": "downloads", "likes": "likes"}), raw=row, - parser_version="huggingface-hub/1", + parser_version="huggingface-hub/2", ) _clear_inherited_short_descriptions(found.values()) # `limit` is applied per request, and this fetcher issues one per kind per @@ -619,12 +627,11 @@ def fetch_github(config: dict[str, Any], since: datetime, limit: int) -> list[Ra event_kind=( "released" if created is not None and created >= since else "updated" ), - metrics={ - "stars": float(row.get("stargazers_count") or 0), - "forks": float(row.get("forks_count") or 0), - }, + metrics=_reported_metrics( + row, {"stars": "stargazers_count", "forks": "forks_count"} + ), raw=row, - parser_version="github-search/1", + parser_version="github-search/2", ) if len(rows) < min(limit, page_size): exhausted.add(index) @@ -638,9 +645,9 @@ def fetch_github(config: dict[str, Any], since: datetime, limit: int) -> list[Ra )[:limit] -GITHUB_ORGANIZATIONS_PARSER_VERSION = "github-organizations/1" -KAGGLE_DATASETS_PARSER_VERSION = "kaggle-datasets/1" -HUGGINGFACE_PAPERS_PARSER_VERSION = "huggingface-papers/1" +GITHUB_ORGANIZATIONS_PARSER_VERSION = "github-organizations/2" +KAGGLE_DATASETS_PARSER_VERSION = "kaggle-datasets/2" +HUGGINGFACE_PAPERS_PARSER_VERSION = "huggingface-papers/2" def _github_headers() -> dict[str, str]: @@ -739,10 +746,9 @@ def fetch_github_organizations( summary=github_summary(row), event_kind="released", organizations=[organization["display_name"]], - metrics={ - "stars": float(row.get("stargazers_count") or 0), - "forks": float(row.get("forks_count") or 0), - }, + metrics=_reported_metrics( + row, {"stars": "stargazers_count", "forks": "forks_count"} + ), raw={"repository": row, "organization_tier": organization["tier"]}, parser_version=GITHUB_ORGANIZATIONS_PARSER_VERSION, ) @@ -822,11 +828,14 @@ def fetch_kaggle_datasets( authors=( [str(row.get("creatorName") or "").strip()] if row.get("creatorName") else [] ), - metrics={ - "downloads": float(row.get("downloadCount") or 0), - "votes": float(row.get("voteCount") or 0), - "views": float(row.get("viewCount") or 0), - }, + metrics=_reported_metrics( + row, + { + "downloads": "downloadCount", + "votes": "voteCount", + "views": "viewCount", + }, + ), raw=row, parser_version=KAGGLE_DATASETS_PARSER_VERSION, ) @@ -884,7 +893,7 @@ def fetch_huggingface_papers( event_kind="discovered", authors=authors, artifact_urls=sorted(set(artifact_urls)), - metrics={"upvotes": float(paper.get("upvotes") or 0)}, + metrics=_reported_metrics(paper, {"upvotes": "upvotes"}), raw=row, parser_version=HUGGINGFACE_PAPERS_PARSER_VERSION, ) @@ -953,12 +962,9 @@ def fetch_zenodo_records( if isinstance(creator, dict) and str(creator.get("name") or "").strip() ], artifact_urls=artifact_urls, - metrics={ - "downloads": float(stats.get("downloads") or 0), - "views": float(stats.get("views") or 0), - }, + metrics=_reported_metrics(stats, {"downloads": "downloads", "views": "views"}), raw=row, - parser_version="zenodo-records/1", + parser_version="zenodo-records/2", ) return sorted( collapse_batch_deposits(found.values()), @@ -1118,9 +1124,9 @@ def fetch_crossref( authors=author_names, organizations=list(dict.fromkeys(organizations)), artifact_urls=[doi_url], - metrics={"citations": float(row.get("is-referenced-by-count") or 0)}, + metrics=_reported_metrics(row, {"citations": "is-referenced-by-count"}), raw=row, - parser_version="crossref-works/1", + parser_version="crossref-works/2", ) return sorted(found.values(), key=lambda item: item.published_at, reverse=True)[:limit] @@ -1380,12 +1386,11 @@ def fetch_openaire( ), artifact_urls=artifact_urls, metrics={ - "citations": float(citations.get("citationCount") or 0), - "downloads": float(usage.get("downloads") or 0), - "views": float(usage.get("views") or 0), + **_reported_metrics(citations, {"citations": "citationCount"}), + **_reported_metrics(usage, {"downloads": "downloads", "views": "views"}), }, raw=row, - parser_version="openaire-graph-v3/1", + parser_version="openaire-graph-v3/2", ) return sorted(found.values(), key=lambda item: item.published_at, reverse=True)[:limit] @@ -1646,13 +1651,16 @@ def fetch_datacite( authors=authors, organizations=list(dict.fromkeys(organizations)), artifact_urls=artifact_urls, - metrics={ - "citations": float(attributes.get("citationCount") or 0), - "downloads": float(attributes.get("downloadCount") or 0), - "views": float(attributes.get("viewCount") or 0), - }, + metrics=_reported_metrics( + attributes, + { + "citations": "citationCount", + "downloads": "downloadCount", + "views": "viewCount", + }, + ), raw=row, - parser_version="datacite-dois/1", + parser_version="datacite-dois/2", ) return sorted(found.values(), key=lambda item: item.published_at, reverse=True)[:limit] @@ -1869,12 +1877,15 @@ def fetch_semantic_scholar( if isinstance(author, dict) and author.get("name") ], artifact_urls=sorted(set(artifact_urls)), - metrics={ - "citations": float(row.get("citationCount") or 0), - "influential_citations": float(row.get("influentialCitationCount") or 0), - }, + metrics=_reported_metrics( + row, + { + "citations": "citationCount", + "influential_citations": "influentialCitationCount", + }, + ), raw=row, - parser_version="semantic-scholar-graph/1", + parser_version="semantic-scholar-graph/2", ) next_offset = _payload_dict(payload, "Semantic Scholar").get("next") if next_offset is None or len(rows) < min(page_size, limit - len(found)): @@ -2000,6 +2011,14 @@ def fetch_github_releases( isinstance(asset, dict) for asset in assets ): raise ConnectorPayloadError("GitHub release assets must be an array") + # A release with no assets has zero downloads; an omitted assets + # list or an asset without its counter has unknown downloads. + download_metrics = ( + {"downloads": float(sum(int(asset["download_count"]) for asset in assets))} + if row.get("assets") is not None + and all(asset.get("download_count") not in (None, "") for asset in assets) + else {} + ) found[f"{repository}@{tag}"] = RadarItem( source="GitHub Release", source_id=f"{repository}@{tag}", @@ -2015,15 +2034,7 @@ def fetch_github_releases( else [] ), artifact_urls=[f"https://github.com/{repository}"], - metrics={ - "downloads": float( - sum( - int(asset.get("download_count") or 0) - for asset in assets - if isinstance(asset, dict) - ) - ) - }, + metrics=download_metrics, raw=row, parser_version=GITHUB_RELEASE_PARSER_VERSION, ) @@ -2061,8 +2072,9 @@ def fetch_github_releases( ) if not isinstance(repository_payload, dict): raise ConnectorPayloadError("GitHub repository metadata was not an object") - stars = float(repository_payload.get("stargazers_count") or 0) - forks = float(repository_payload.get("forks_count") or 0) + popularity = _reported_metrics( + repository_payload, {"stars": "stargazers_count", "forks": "forks_count"} + ) except Exception as error: config.setdefault("_source_warnings", []).append( f"{repository} metadata: {type(error).__name__}: {error}" @@ -2071,7 +2083,7 @@ def fetch_github_releases( for item in found.values(): if not item.source_id.startswith(f"{repository}@"): continue - item.metrics.update({"stars": stars, "forks": forks}) + item.metrics.update(popularity) item.raw = {"release": item.raw, "repository": repository_payload} return sorted(found.values(), key=lambda item: item.published_at, reverse=True)[:limit] @@ -2170,9 +2182,9 @@ def fetch_openalex( event_kind="released", authors=[author for author in authors if author], organizations=organizations, - metrics={"citations": float(row.get("cited_by_count") or 0)}, + metrics=_reported_metrics(row, {"citations": "cited_by_count"}), raw=row, - parser_version="openalex-works/1", + parser_version="openalex-works/2", ) return list(found.values()) diff --git a/tests/test_snapshots.py b/tests/test_snapshots.py index d2aefa3e..01942481 100644 --- a/tests/test_snapshots.py +++ b/tests/test_snapshots.py @@ -26,7 +26,6 @@ validate_snapshot, write_snapshot, ) -from benchmark_radar.sources import GITHUB_RELEASE_PARSER_VERSION def radar_run(day: int = 27, *, title: str = "A New Evaluation Benchmark") -> RadarRun: @@ -909,6 +908,7 @@ def test_migrate_backfills_bare_github_release_titles_idempotently(tmp_path): "title": "v1.11.0", "url": "https://github.com/modelscope/evalscope/releases/tag/v1.11.0", "parser_version": "github-releases/1", + "metrics": {"downloads": 0.0}, } ) original_hash = record["raw_payload_hash"] @@ -921,7 +921,9 @@ def test_migrate_backfills_bare_github_release_titles_idempotently(tmp_path): migrated = json.loads(first_pass)["evidence_items"][0] assert migrated["title"] == "modelscope/evalscope v1.11.0" - assert migrated["parser_version"] == GITHUB_RELEASE_PARSER_VERSION + assert migrated["parser_version"] == "github-releases/3" + # Title-only backfill must not claim the new optional-counter parser ran. + assert migrated["metrics"] == {"downloads": 0.0} assert migrated["raw_payload_hash"] == original_hash assert dashboard["days"][0]["evidence_items"][0]["title"] == ("modelscope/evalscope v1.11.0") artifact = next( diff --git a/tests/test_sources.py b/tests/test_sources.py index 9dba4b53..35794b83 100644 --- a/tests/test_sources.py +++ b/tests/test_sources.py @@ -395,6 +395,7 @@ def test_openalex_carries_author_institutions(monkeypatch): assert items[0].authors == ["Radar Author"] assert items[0].organizations == ["Example University", "Example Lab"] + assert items[0].parser_version == "openalex-works/2" def test_openalex_accepts_explicitly_null_authorships(monkeypatch): @@ -636,6 +637,23 @@ def test_github_preserves_creation_and_update_times(monkeypatch): assert items[0].event_kind == "updated" +def test_github_keeps_missing_forks_unknown(monkeypatch): + row = _github_row(1) + row.pop("forks_count") + row["stargazers_count"] = 0 + monkeypatch.setattr( + "benchmark_radar.sources.get_json", + lambda url, params=None, headers=None: {"items": [row]}, + ) + items = fetch_github( + {"queries": ["benchmark"], "request_delay_seconds": 0}, + datetime(2026, 7, 26, tzinfo=UTC), + 10, + ) + assert items[0].metrics == {"stars": 0.0} + assert items[0].parser_version == "github-search/2" + + def test_github_config_discovers_and_routes_rsi_exam(monkeypatch): """Issue #408: the named benchmark matched no configured GitHub query.""" config = yaml.safe_load(Path("config.yml").read_text(encoding="utf-8")) @@ -738,6 +756,7 @@ def fake_get_json(url, **kwargs): assert items[0].source == "GitHub Organization" assert items[0].organizations == ["First Lab"] assert items[0].event_kind == "released" + assert items[0].parser_version == "github-organizations/2" def test_github_organizations_isolate_one_failed_organization(monkeypatch): @@ -791,6 +810,7 @@ def test_huggingface_papers_preserves_arxiv_and_project_identifiers(monkeypatch) "https://lab.example/benchmark", ] assert items[0].summary == "An upstream evaluation suite." + assert items[0].parser_version == "huggingface-papers/2" def test_kaggle_datasets_preserves_source_text_and_tags(monkeypatch): @@ -820,6 +840,7 @@ def test_kaggle_datasets_preserves_source_text_and_tags(monkeypatch): assert items[0].source == "Kaggle Dataset" assert items[0].summary == "A public evaluation dataset. | benchmark | llm" assert items[0].metrics == {"downloads": 11.0, "votes": 2.0, "views": 31.0} + assert items[0].parser_version == "kaggle-datasets/2" def test_zenodo_records_preserve_doi_and_upstream_metadata(monkeypatch): @@ -856,6 +877,7 @@ def test_zenodo_records_preserve_doi_and_upstream_metadata(monkeypatch): assert items[0].authors == ["Zenodo Author"] assert items[0].artifact_urls == ["https://doi.org/10.5281/zenodo.12345"] assert items[0].metrics == {"downloads": 13.0, "views": 21.0} + assert items[0].parser_version == "zenodo-records/2" def _deposit(recid: str, title: str, description: str, creators: list[str], day: int) -> dict: @@ -1045,7 +1067,7 @@ def fake_get_json(url, **kwargs): assert items[0].organizations == ["Radar Lab"] assert items[0].artifact_urls == ["https://doi.org/10.1000/radar"] assert items[0].metrics == {"citations": 3.0} - assert items[0].parser_version == "crossref-works/1" + assert items[0].parser_version == "crossref-works/2" assert calls[0][0] == "https://api.crossref.org/works" assert calls[0][1]["params"]["query.title"] == "agent benchmark" assert calls[0][1]["params"]["filter"] == ("from-pub-date:2026-07-26,until-pub-date:2026-07-28") @@ -1118,6 +1140,33 @@ def _openaire_payload(**fields): return _openaire_rows_payload(_openaire_row(**fields)) +@pytest.mark.parametrize( + "state,counts,expected", + [ + ("missing", (None, None, None), {}), + ("null", (None, None, None), {}), + ("empty", ("", "", ""), {}), + ("zero", (0, 0, 0), {"citations": 0.0, "downloads": 0.0, "views": 0.0}), + ("positive", (3, 13, 21), {"citations": 3.0, "downloads": 13.0, "views": 21.0}), + ("mixed", (None, 0, 7), {"downloads": 0.0, "views": 7.0}), + ], +) +def test_openaire_preserves_unknown_and_reported_counters(monkeypatch, state, counts, expected): + # Review #704: the new connector still converted unknown counters to zero, + # even though the shared rule requires omission and retains measured zero. + citations = {} if state == "missing" else {"citationCount": counts[0]} + usage = {} if state == "missing" else {"downloads": counts[1], "views": counts[2]} + row = _openaire_row(indicators={"citationImpact": citations, "usageCounts": usage}) + monkeypatch.setattr( + "benchmark_radar.sources.get_json", + lambda url, **kwargs: _openaire_rows_payload(row), + ) + items = fetch_openaire({"searches": ["benchmark"]}, datetime(2026, 7, 26, tzinfo=UTC), 10) + assert len(items) == 1 + assert items[0].metrics == expected + assert items[0].source_id == row["id"] + + def test_openaire_preserves_upstream_metadata_and_bounds_the_query(monkeypatch): calls = [] @@ -1160,7 +1209,7 @@ def fake_get_json(url, **kwargs): assert item.updated_at == datetime(2026, 7, 27, tzinfo=UTC) assert item.raw["dateOfCollection"] == "2026-08-30T00:00:00Z" assert item.event_kind == "released" - assert item.parser_version == "openaire-graph-v3/1" + assert item.parser_version == "openaire-graph-v3/2" assert calls[0][0] == "https://api.openaire.eu/graph/v3/research-products" assert calls[0][1]["params"] == { "mainTitle": '"agent benchmark"', @@ -1626,7 +1675,7 @@ def fake_get_json(url, **kwargs): def test_openaire_preserves_a_product_that_carries_almost_nothing(monkeypatch): - # An absent public counter is a real zero rather than a missing metric. + # An absent public counter is unknown and must remain omitted. monkeypatch.setattr( "benchmark_radar.sources.get_json", lambda url, **kwargs: _openaire_rows_payload( @@ -1644,7 +1693,7 @@ def test_openaire_preserves_a_product_that_carries_almost_nothing(monkeypatch): item = items[0] assert item.url == "https://doi.org/10.1000/sparse" assert item.artifact_urls == ["https://doi.org/10.1000/sparse"] - assert item.metrics == {"citations": 0.0, "downloads": 0.0, "views": 0.0} + assert item.metrics == {} assert item.authors == [] assert item.organizations == [] assert item.summary == "" @@ -1745,6 +1794,34 @@ def _datacite_rows_payload(*rows): } +@pytest.mark.parametrize( + "state,counts,expected", + [ + ("missing", (None, None, None), {}), + ("null", (None, None, None), {}), + ("empty", ("", "", ""), {}), + ("zero", (0, 0, 0), {"citations": 0.0, "downloads": 0.0, "views": 0.0}), + ("positive", (3, 13, 21), {"citations": 3.0, "downloads": 13.0, "views": 21.0}), + ("mixed", (None, 0, 7), {"downloads": 0.0, "views": 7.0}), + ], +) +def test_datacite_preserves_unknown_and_reported_counters(monkeypatch, state, counts, expected): + row = _datacite_row() + for field, count in zip(("citationCount", "downloadCount", "viewCount"), counts, strict=True): + if state == "missing": + row["attributes"].pop(field) + else: + row["attributes"][field] = count + monkeypatch.setattr( + "benchmark_radar.sources.get_json", + lambda url, **kwargs: _datacite_rows_payload(row), + ) + items = fetch_datacite({"searches": ["benchmark"]}, datetime(2026, 7, 26, tzinfo=UTC), 10) + assert len(items) == 1 + assert items[0].metrics == expected + assert items[0].source_id == row["id"] + + def test_datacite_preserves_doi_metadata_and_bounds_the_query(monkeypatch): calls = [] @@ -1786,7 +1863,7 @@ def fake_get_json(url, **kwargs): assert item.updated_at == datetime(2026, 7, 27, 9, tzinfo=UTC) assert item.raw["attributes"]["updated"] == "2026-07-27T09:00:05.000Z" assert item.event_kind == "released" - assert item.parser_version == "datacite-dois/1" + assert item.parser_version == "datacite-dois/2" assert calls[0][0] == "https://api.datacite.org/dois" params = calls[0][1]["params"] # Both ends of the window travel with the query, second-precise and in UTC. @@ -2024,7 +2101,7 @@ def test_datacite_truncates_to_the_per_source_limit(monkeypatch): def test_datacite_preserves_a_deposit_that_carries_almost_nothing(monkeypatch): # A DOI, a title and a registration date are all DataCite requires. An - # absent public counter is a real zero rather than a missing metric, and a + # absent public counter is unknown and remains omitted, and a # deposit whose landing page is its own DOI must not list that URL twice. monkeypatch.setattr( "benchmark_radar.sources.get_json", @@ -2045,7 +2122,7 @@ def test_datacite_preserves_a_deposit_that_carries_almost_nothing(monkeypatch): item = items[0] assert item.artifact_urls == ["https://doi.org/10.5281/zenodo.1"] - assert item.metrics == {"citations": 0.0, "downloads": 0.0, "views": 0.0} + assert item.metrics == {} assert item.authors == [] assert item.organizations == [] assert item.summary == "" @@ -2342,7 +2419,7 @@ def test_semantic_scholar_success_preserves_external_ids(monkeypatch): assert items[0].summary == "The upstream scholarly abstract." assert "https://doi.org/10.1000/radar" in items[0].artifact_urls assert "https://arxiv.org/abs/2607.12345" in items[0].artifact_urls - assert items[0].parser_version == "semantic-scholar-graph/1" + assert items[0].parser_version == "semantic-scholar-graph/2" def test_semantic_scholar_paces_an_individual_api_key(monkeypatch): @@ -2399,6 +2476,7 @@ def test_github_releases_success_uses_release_notes(monkeypatch): assert items[0].summary == "The upstream release notes." assert items[0].metrics["downloads"] == 7 assert items[0].parser_version == GITHUB_RELEASE_PARSER_VERSION + assert items[0].parser_version == "github-releases/4" def test_github_release_popularity_comes_from_repository_metadata(monkeypatch): @@ -2445,6 +2523,26 @@ def fake_get_json(url, **kwargs): assert radar_item.adoption_score > 0 +def test_github_release_omits_incomplete_download_totals_and_repository_counters(monkeypatch): + release = { + "tag_name": "v2", + "html_url": "https://github.com/example/benchmark/releases/tag/v2", + "published_at": "2026-07-27T12:00:00Z", + "assets": [{"download_count": 4}, {"name": "missing-count.zip"}], + } + + def fake_get_json(url, **kwargs): + return [release] if url.endswith("/releases") else {"stargazers_count": 0} + + monkeypatch.setattr("benchmark_radar.sources.get_json", fake_get_json) + item = fetch_github_releases( + {"repositories": ["example/benchmark"], "repository_metadata_requests": 1}, + datetime(2026, 7, 26, tzinfo=UTC), + 10, + )[0] + assert item.metrics == {"stars": 0.0} + + def test_github_release_repository_counters_are_covered_by_raw_hash(monkeypatch): stars = 7_000 @@ -3165,6 +3263,27 @@ def test_huggingface_preserves_creation_and_update_times(monkeypatch): assert items[0].event_kind == "updated" +def test_huggingface_does_not_publish_absent_counters_as_zero(monkeypatch): + # A missing API field is unknown, while a reported zero is a real measurement. + monkeypatch.setattr( + "benchmark_radar.sources.get_json", + lambda url, params: [ + { + "id": "org/benchmark", + "lastModified": "2026-07-27T12:00:00Z", + "likes": 0, + } + ], + ) + items = fetch_huggingface( + {"kinds": ["datasets"], "searches": ["benchmark"]}, + datetime(2026, 7, 26, tzinfo=UTC), + 10, + ) + assert items[0].metrics == {"likes": 0.0} + assert items[0].parser_version == "huggingface-hub/2" + + def test_huggingface_filters_future_rows_before_the_local_cap(monkeypatch): seen_limit = []