From e08310da874bdff3fbd891028202f853c70feb27 Mon Sep 17 00:00:00 2001 From: Sanjay Santhanam <51058514+Sanjays2402@users.noreply.github.com> Date: Sun, 2 Aug 2026 16:43:50 -0700 Subject: [PATCH] fix: don't let the SUPPORTED_IDS branch re-add a rejected username in extract_ids_from_page (#2924) extract_ids_from_page() has the same bug #2907 fixed in checking.py::parse_usernames(). "username" is itself in SUPPORTED_IDS, so the `if k in SUPPORTED_IDS` loop stores a bare `username` value directly, bypassing the is_plausible_username() guard that extract_usernames() applies in its own separate pass. A URL or email returned by socid_extractor under a bare `username` key therefore becomes a recursive search target via `maigret --parse-url`, reproducing the #1403 false-error cascade. Skip keys containing "username" in the SUPPORTED_IDS loop; those are already owned (and validated) by extract_usernames(). Every other SUPPORTED_ID (gaia_id, vk_id, orcid, ...) is unaffected since none of those keys contain "username". Add regression tests for the bare `username` key with a URL value and for the unchanged handling of other supported IDs. Closes #2911 --- maigret/maigret.py | 5 +++-- tests/test_extractors.py | 41 ++++++++++++++++++++++++++++++++++++++++ 2 files changed, 44 insertions(+), 2 deletions(-) diff --git a/maigret/maigret.py b/maigret/maigret.py index 7604dda3..eb410248 100755 --- a/maigret/maigret.py +++ b/maigret/maigret.py @@ -84,8 +84,9 @@ def extract_ids_from_page(url, logger, timeout=5) -> dict: else: print(get_dict_ascii_tree(info.items(), new_line=False), ' ') for k, v in info.items(): - - if k in SUPPORTED_IDS: + # keys containing "username" are owned by extract_usernames() below, + # which validates them; adding them here would bypass that check + if "username" not in k and k in SUPPORTED_IDS: results[v] = k for username in extract_usernames(info, logger): diff --git a/tests/test_extractors.py b/tests/test_extractors.py index d99845ce..5412014e 100644 --- a/tests/test_extractors.py +++ b/tests/test_extractors.py @@ -75,3 +75,44 @@ def test_extract_ids_from_page_username_contract(): ) assert result == {"emily": "username"} + + +def test_extract_ids_from_page_rejects_bare_username_url(): + logger = Mock() + + with patch("maigret.maigret.parse") as mock_parse, \ + patch("maigret.maigret.extract") as mock_extract: + + mock_parse.return_value = ("", {}) + + # socid_extractor returned a URL under the bare `username` key + mock_extract.return_value = {"username": "https://instagram.com/zuck"} + + result = extract_ids_from_page( + "https://example.com/profile", + logger, + timeout=5, + ) + + assert result == {} + + +def test_extract_ids_from_page_keeps_other_supported_ids(): + logger = Mock() + + with patch("maigret.maigret.parse") as mock_parse, \ + patch("maigret.maigret.extract") as mock_extract: + + mock_parse.return_value = ("", {}) + mock_extract.return_value = { + "gaia_id": "123456789", + "profile_username": "emily", + } + + result = extract_ids_from_page( + "https://example.com/profile", + logger, + timeout=5, + ) + + assert result == {"123456789": "gaia_id", "emily": "username"}