Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 1 addition & 1 deletion eval/lib/retrieval.py
Original file line number Diff line number Diff line change
Expand Up @@ -900,7 +900,7 @@ def _extract_wiki_title(self, url: str) -> str | None:
# Match patterns like:
# https://en.wikipedia.org/wiki/Python_(programming_language)
# https://zh.wikipedia.org/wiki/Artificial_intelligence
pattern = r"https?://[a-z]{2,3}\.wikipedia\.org/wiki/(.+?)(?:#.*)?$"
pattern = r"https?://[a-z]{2,3}\.wikipedia\.org/wiki/(.+?)(?:[?#].*)?$"
match = re.match(pattern, url)
if match:
title = unquote(match.group(1))
Expand Down
35 changes: 35 additions & 0 deletions tests/test_wiki_title_extraction.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,35 @@
"""Regression tests for Wikipedia title extraction from URLs."""

from pathlib import Path
from runpy import run_path

RETRIEVAL = run_path(str(Path(__file__).parents[1] / "eval" / "lib" / "retrieval.py"))
extract = RETRIEVAL["WikipediaAPIRetriever"]._extract_wiki_title


def test_plain_url_title_unchanged():
assert extract(None, "https://en.wikipedia.org/wiki/Albert_Einstein") == (
"Albert Einstein"
)


def test_query_string_is_stripped():
assert extract(
None, "https://en.wikipedia.org/wiki/Albert_Einstein?wprov=rarw1"
) == ("Albert Einstein")


def test_fragment_is_stripped():
assert extract(
None, "https://en.wikipedia.org/wiki/Albert_Einstein#Early_life"
) == ("Albert Einstein")


def test_percent_encoded_title_is_decoded():
assert extract(
None, "https://en.wikipedia.org/wiki/Python_%28programming_language%29"
) == ("Python (programming language)")


def test_non_wiki_url_returns_none():
assert extract(None, "https://example.com/wiki/Albert_Einstein") is None