diff --git a/maigret/maigret.py b/maigret/maigret.py index 7604dda3..eb410248 100755 --- a/maigret/maigret.py +++ b/maigret/maigret.py @@ -84,8 +84,9 @@ def extract_ids_from_page(url, logger, timeout=5) -> dict: else: print(get_dict_ascii_tree(info.items(), new_line=False), ' ') for k, v in info.items(): - - if k in SUPPORTED_IDS: + # keys containing "username" are owned by extract_usernames() below, + # which validates them; adding them here would bypass that check + if "username" not in k and k in SUPPORTED_IDS: results[v] = k for username in extract_usernames(info, logger): diff --git a/tests/test_extractors.py b/tests/test_extractors.py index d99845ce..5412014e 100644 --- a/tests/test_extractors.py +++ b/tests/test_extractors.py @@ -75,3 +75,44 @@ def test_extract_ids_from_page_username_contract(): ) assert result == {"emily": "username"} + + +def test_extract_ids_from_page_rejects_bare_username_url(): + logger = Mock() + + with patch("maigret.maigret.parse") as mock_parse, \ + patch("maigret.maigret.extract") as mock_extract: + + mock_parse.return_value = ("", {}) + + # socid_extractor returned a URL under the bare `username` key + mock_extract.return_value = {"username": "https://instagram.com/zuck"} + + result = extract_ids_from_page( + "https://example.com/profile", + logger, + timeout=5, + ) + + assert result == {} + + +def test_extract_ids_from_page_keeps_other_supported_ids(): + logger = Mock() + + with patch("maigret.maigret.parse") as mock_parse, \ + patch("maigret.maigret.extract") as mock_extract: + + mock_parse.return_value = ("", {}) + mock_extract.return_value = { + "gaia_id": "123456789", + "profile_username": "emily", + } + + result = extract_ids_from_page( + "https://example.com/profile", + logger, + timeout=5, + ) + + assert result == {"123456789": "gaia_id", "emily": "username"}