remove '&' from URL patterns

'/?&#' -> '/?#' and '?&#' -> '?#'

According to https://www.ietf.org/rfc/rfc3986.txt, URLs are
"organized hierarchically" by using "the slash ("/"), question
mark ("?"), and number sign ("#") characters to delimit components"
This commit is contained in:
Mike Fährmann
2020-10-22 23:12:59 +02:00
parent 1686dc1757
commit 968d3e8465
74 changed files with 158 additions and 158 deletions

View File

@@ -168,7 +168,7 @@ class SexcomBoardExtractor(SexcomExtractor):
subcategory = "board"
directory_fmt = ("{category}", "{user}", "{board}")
pattern = (r"(?:https?://)?(?:www\.)?sex\.com/user"
r"/([^/?&#]+)/(?!(?:following|pins|repins|likes)/)([^/?&#]+)")
r"/([^/?#]+)/(?!(?:following|pins|repins|likes)/)([^/?#]+)")
test = ("https://www.sex.com/user/ronin17/exciting-hentai/", {
"count": ">= 15",
})
@@ -193,7 +193,7 @@ class SexcomSearchExtractor(SexcomExtractor):
subcategory = "search"
directory_fmt = ("{category}", "search", "{search[query]}")
pattern = (r"(?:https?://)?(?:www\.)?sex\.com/((?:"
r"(pic|gif|video)s/([^/?&#]+)|search/(pic|gif|video)s"
r"(pic|gif|video)s/([^/?#]+)|search/(pic|gif|video)s"
r")/?(?:\?([^#]+))?)")
test = (
("https://www.sex.com/search/pics?query=ecchi", {