Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
31 changes: 25 additions & 6 deletions src/benchmark_radar/query.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from urllib.parse import urlsplit

from .citation import citation_block
from .snapshots import REQUIRED_SOURCES, load_snapshots
Expand Down Expand Up @@ -162,6 +163,29 @@ def _filter_value(value: Any) -> str:
return str(value or "").strip().casefold()


def _radar_artifact_flags(urls: list[Any]) -> dict[str, bool]:
flags = {"has_paper": False, "has_repo": False, "has_dataset": False}
for url in urls:
if not isinstance(url, str):
continue
try:
parsed = urlsplit(url)
host = (parsed.hostname or "").casefold().removeprefix("www.")
except ValueError:
continue
if parsed.scheme.casefold() not in {"http", "https"}:
continue
# Domain text in a query, path, userinfo, or lookalike host is not
# artifact evidence. These indicators describe links, not page verification.
if host == "arxiv.org":
flags["has_paper"] = True
if host == "github.com":
flags["has_repo"] = True
if host in {"huggingface.co", "kaggle.com"} and parsed.path.startswith("/datasets/"):
flags["has_dataset"] = True
return flags


def _matches_filter(record: dict[str, Any], filters: dict[str, Any]) -> bool:
for flag in ("has_paper", "has_repo", "has_dataset"):
expected = filters.get(flag)
Expand Down Expand Up @@ -497,12 +521,7 @@ def _radar_candidates(self) -> list[dict[str, Any]]:
"score": item.get("total_score"),
"recommended": item.get("recommended", False),
"openness": None,
"has_paper": any("arxiv.org" in value for value in urls),
"has_repo": any("github.com" in value for value in urls),
"has_dataset": any(
"huggingface.co/datasets/" in value or "kaggle.com/datasets/" in value
for value in urls
),
**_radar_artifact_flags(urls),
"has_size": False,
"snapshot_date": snapshot["date"],
}
Expand Down
78 changes: 78 additions & 0 deletions tests/test_query_surfaces.py
Original file line number Diff line number Diff line change
Expand Up @@ -402,6 +402,84 @@ def test_search_filters_before_ranking(tmp_path: Path) -> None:
]


@pytest.mark.parametrize(
"url",
[
"https://example.org/?repo=github.com/a/b&paper=arxiv.org/abs/2601.12345"
"&dataset=huggingface.co/datasets/a/b",
"https://example.org/github.com/arxiv.org/huggingface.co/datasets/a/b",
"https://github.com.evil.example/arxiv.org/huggingface.co/datasets/a/b",
"https://arxiv.org.evil.example/github.com/huggingface.co/datasets/a/b",
"https://huggingface.co.evil.example/datasets/a/b?paper=arxiv.org&repo=github.com",
"https://github.com@evil.example/?paper=arxiv.org&dataset=kaggle.com/datasets/a/b",
],
)
def test_radar_artifact_filters_use_url_authority_instead_of_embedded_text(
tmp_path: Path, url: str
) -> None:
# Regression: domain strings in unrelated URLs falsely satisfied artifact filters.
paths = _catalog(tmp_path)
path = next(paths.snapshots.glob("*.json"))
snapshot = json.loads(path.read_text(encoding="utf-8"))
snapshot["evidence_items"][0]["url"] = url
snapshot["evidence_items"][0]["artifact_urls"] = []
path.write_text(json.dumps(snapshot), encoding="utf-8")
service = QueryService(paths)
candidate = service.search("agent", scope="radar")["results"][0]
for flag in ("has_repo", "has_paper", "has_dataset"):
assert candidate[flag] is False
assert service.search("agent", scope="radar", **{flag: True})["count"] == 0
assert service.search("agent", scope="radar", **{flag: False})["count"] == 1


@pytest.mark.parametrize(
("url", "flag"),
[
("https://GITHUB.COM/example/benchmark/tree/main", "has_repo"),
("https://github.com/", "has_repo"),
("https://arxiv.org/abs/2601.12345", "has_paper"),
("https://www.arxiv.org/pdf/2601.12345.pdf", "has_paper"),
("https://arxiv.org/format/hep-th/9901001v3", "has_paper"),
("https://arxiv.org/src/hep-th/9901001v3", "has_paper"),
("https://arxiv.org/ftp/hep-th/papers/9901/9901001.pdf", "has_paper"),
("https://arxiv.org/html/2601.12345v2", "has_paper"),
("https://HUGGINGFACE.CO/datasets/example/benchmark", "has_dataset"),
("https://huggingface.co/datasets/example", "has_dataset"),
("https://www.kaggle.com/datasets/example/benchmark", "has_dataset"),
],
)
@pytest.mark.parametrize("location", ["main", "artifact"])
def test_radar_artifact_filters_keep_first_party_artifact_links(
tmp_path: Path, url: str, flag: str, location: str
) -> None:
paths = _catalog(tmp_path)
path = next(paths.snapshots.glob("*.json"))
snapshot = json.loads(path.read_text(encoding="utf-8"))
snapshot["evidence_items"][0]["url"] = (
url if location == "main" else "https://example.org/benchmark"
)
snapshot["evidence_items"][0]["artifact_urls"] = [url] if location == "artifact" else []
path.write_text(json.dumps(snapshot), encoding="utf-8")

result = QueryService(paths).search("agent", scope="radar", **{flag: True})
assert result["count"] == 1
assert result["results"][0][flag] is True


def test_radar_non_string_artifact_urls_do_not_break_url_flags(tmp_path: Path) -> None:
# Regression: accepted snapshots may contain an empty artifact object; parsing
# it as a URL introduced an AttributeError that the old flags did not raise.
paths = _catalog(tmp_path)
path = next(paths.snapshots.glob("*.json"))
snapshot = json.loads(path.read_text(encoding="utf-8"))
snapshot["evidence_items"][0]["url"] = "https://example.org/benchmark"
snapshot["evidence_items"][0]["artifact_urls"] = [{}]
path.write_text(json.dumps(snapshot), encoding="utf-8")

candidate = QueryService(paths).search("agent", scope="radar")["results"][0]
assert all(candidate[flag] is False for flag in ("has_paper", "has_repo", "has_dataset"))


def test_all_scope_keeps_catalog_and_radar_identity_separate(tmp_path: Path) -> None:
# Regression: same-looking names from two evidence layers are not proven identities.
service = QueryService(_catalog(tmp_path))
Expand Down
Loading