fix: don't let the SUPPORTED_IDS branch re-add a rejected username in extract_ids_from_page (#2924)

extract_ids_from_page() has the same bug #2907 fixed in
checking.py::parse_usernames(). "username" is itself in SUPPORTED_IDS, so
the `if k in SUPPORTED_IDS` loop stores a bare `username` value directly,
bypassing the is_plausible_username() guard that extract_usernames()
applies in its own separate pass. A URL or email returned by
socid_extractor under a bare `username` key therefore becomes a recursive
search target via `maigret --parse-url`, reproducing the #1403
false-error cascade.

Skip keys containing "username" in the SUPPORTED_IDS loop; those are
already owned (and validated) by extract_usernames(). Every other
SUPPORTED_ID (gaia_id, vk_id, orcid, ...) is unaffected since none of
those keys contain "username".

Add regression tests for the bare `username` key with a URL value and for
the unchanged handling of other supported IDs.

Closes #2911
This commit is contained in:
Sanjay Santhanam
2026-08-03 08:43:50 +09:00
committed by GitHub
parent d1b8b9fa5c
commit e08310da87
2 changed files with 44 additions and 2 deletions
+3 -2
View File
@@ -84,8 +84,9 @@ def extract_ids_from_page(url, logger, timeout=5) -> dict:
else:
print(get_dict_ascii_tree(info.items(), new_line=False), ' ')
for k, v in info.items():
if k in SUPPORTED_IDS:
# keys containing "username" are owned by extract_usernames() below,
# which validates them; adding them here would bypass that check
if "username" not in k and k in SUPPORTED_IDS:
results[v] = k
for username in extract_usernames(info, logger):
+41
View File
@@ -75,3 +75,44 @@ def test_extract_ids_from_page_username_contract():
)
assert result == {"emily": "username"}
def test_extract_ids_from_page_rejects_bare_username_url():
logger = Mock()
with patch("maigret.maigret.parse") as mock_parse, \
patch("maigret.maigret.extract") as mock_extract:
mock_parse.return_value = ("<html></html>", {})
# socid_extractor returned a URL under the bare `username` key
mock_extract.return_value = {"username": "https://instagram.com/zuck"}
result = extract_ids_from_page(
"https://example.com/profile",
logger,
timeout=5,
)
assert result == {}
def test_extract_ids_from_page_keeps_other_supported_ids():
logger = Mock()
with patch("maigret.maigret.parse") as mock_parse, \
patch("maigret.maigret.extract") as mock_extract:
mock_parse.return_value = ("<html></html>", {})
mock_extract.return_value = {
"gaia_id": "123456789",
"profile_username": "emily",
}
result = extract_ids_from_page(
"https://example.com/profile",
logger,
timeout=5,
)
assert result == {"123456789": "gaia_id", "emily": "username"}