From d21e32e4fceb68f0e68dc541c6c955a5d44657cb Mon Sep 17 00:00:00 2001 From: monosans Date: Sun, 15 Jan 2023 13:05:16 +0300 Subject: [PATCH] Add CheckWebsite setting Fixes #16, fixes #39 --- README.md | 6 ++- config.ini | 5 +++ proxy_scraper_checker/constants.py | 5 +++ proxy_scraper_checker/folder.py | 2 +- proxy_scraper_checker/proxy.py | 39 +++++++++++------ .../proxy_scraper_checker.py | 42 ++++++++++++++----- 6 files changed, 72 insertions(+), 27 deletions(-) create mode 100644 proxy_scraper_checker/constants.py diff --git a/README.md b/README.md index 7233a94..21f31cd 100644 --- a/README.md +++ b/README.md @@ -5,9 +5,11 @@ HTTP, SOCKS4, SOCKS5 proxies scraper and checker. - Asynchronous. -- Uses regex to search for proxies (ip:port format) on a web page, which allows you to pull out proxies even from json without making any changes to the code. +- Uses regex to search for proxies (ip:port format) on a web page, allowing proxies to be extracted even from json without making changes to the code. +- It is possible to specify the URL to which to send a request to check the proxy. +- Can sort proxies by speed. - Supports determining the geolocation of the proxy exit node. -- Can determine if a proxy is anonymous. +- Can determine if the proxy is anonymous. You can get proxies obtained using this script in [monosans/proxy-list](https://github.com/monosans/proxy-list). diff --git a/config.ini b/config.ini index eb04318..6939775 100644 --- a/config.ini +++ b/config.ini @@ -15,6 +15,11 @@ SourceTimeout = 15 ; Make sure you have enough RAM first, gradually increasing the default value. MaxConnections = 512 +; URL to which to send a request to check the proxy. +; If not equal to 'default', it will not be possible +; to determine the anonymity and geolocation of the proxies. +CheckWebsite = default + ; Set to no to sort proxies alphabetically. SortBySpeed = yes diff --git a/proxy_scraper_checker/constants.py b/proxy_scraper_checker/constants.py new file mode 100644 index 0000000..5f77f26 --- /dev/null +++ b/proxy_scraper_checker/constants.py @@ -0,0 +1,5 @@ +from __future__ import annotations + +USER_AGENT = ( + "Mozilla/5.0 (Windows NT 10.0; rv:108.0) Gecko/20100101 Firefox/108.0" +) diff --git a/proxy_scraper_checker/folder.py b/proxy_scraper_checker/folder.py index e553ee7..7be42b3 100644 --- a/proxy_scraper_checker/folder.py +++ b/proxy_scraper_checker/folder.py @@ -5,7 +5,7 @@ from pathlib import Path from shutil import rmtree -@dataclass(frozen=True) +@dataclass class Folder: path: Path is_enabled: bool diff --git a/proxy_scraper_checker/proxy.py b/proxy_scraper_checker/proxy.py index 927400c..6af6dee 100644 --- a/proxy_scraper_checker/proxy.py +++ b/proxy_scraper_checker/proxy.py @@ -7,6 +7,8 @@ from aiohttp import ClientSession, ClientTimeout from aiohttp.abc import AbstractCookieJar from aiohttp_socks import ProxyConnector, ProxyType +from .constants import USER_AGENT + class Proxy: __slots__ = ("geolocation", "host", "is_anonymous", "port", "timeout") @@ -15,30 +17,41 @@ class Proxy: self.host = host self.port = port + @property + def default_check_website(self) -> str: + return "http://ip-api.com/json/?fields=8217" + async def check( self, *, + website: str, sem: asyncio.Semaphore, cookie_jar: AbstractCookieJar, proto: ProxyType, timeout: ClientTimeout, ) -> None: + check_website = ( + self.default_check_website if website == "default" else website + ) async with sem: start = perf_counter() - async with self.get_connector(proto) as connector: - async with ClientSession( - connector=connector, cookie_jar=cookie_jar, timeout=timeout - ) as session: - async with session.get( - "http://ip-api.com/json/?fields=8217", - raise_for_status=True, - ) as response: - data = await response.json() + async with self.get_connector(proto) as connector, ClientSession( + connector=connector, + cookie_jar=cookie_jar, + timeout=timeout, + headers={"User-Agent": USER_AGENT}, + ) as session, session.get( + check_website, raise_for_status=True + ) as response: + if website == "default": + await response.read() self.timeout = perf_counter() - start - self.is_anonymous = self.host != data["query"] - self.geolocation = "|{}|{}|{}".format( - data["country"], data["regionName"], data["city"] - ) + if website == "default": + data = await response.json() + self.is_anonymous = self.host != data["query"] + self.geolocation = "|{}|{}|{}".format( + data["country"], data["regionName"], data["city"] + ) def get_connector(self, proto: ProxyType) -> ProxyConnector: return ProxyConnector(proxy_type=proto, host=self.host, port=self.port) diff --git a/proxy_scraper_checker/proxy_scraper_checker.py b/proxy_scraper_checker/proxy_scraper_checker.py index ec05e73..0bbfcd8 100644 --- a/proxy_scraper_checker/proxy_scraper_checker.py +++ b/proxy_scraper_checker/proxy_scraper_checker.py @@ -18,6 +18,7 @@ from typing import ( TypeVar, Union, ) +from urllib.parse import urlparse from aiohttp import ClientSession, ClientTimeout, DummyCookieJar from aiohttp_socks import ProxyType @@ -32,6 +33,7 @@ from rich.progress import ( from rich.table import Table from . import sort +from .constants import USER_AGENT from .folder import Folder from .proxy import Proxy @@ -67,6 +69,7 @@ class ProxyScraperChecker: """HTTP, SOCKS4, SOCKS5 proxies scraper and checker.""" __slots__ = ( + "check_website", "console", "cookie_jar", "folders", @@ -87,6 +90,7 @@ class ProxyScraperChecker: timeout: float, source_timeout: float, max_connections: int, + check_website: str, sort_by_speed: bool, save_path: Path, folders: Tuple[Folder, ...], @@ -108,17 +112,37 @@ class ProxyScraperChecker: Don't be in a hurry to set high values. Make sure you have enough RAM first, gradually increasing the default value. + check_website: URL to which to send a request to check the proxy. + If not equal to 'default', it will not be possible + to determine the anonymity and geolocation of the proxies. sort_by_speed: Set to False to sort proxies alphabetically. save_path: Path to the folder where the proxy folders will be saved. Leave empty to save the proxies to the current directory. """ - self.path = save_path - self.folders = folders - if not any(folder for folder in self.folders if folder.is_enabled): - raise ValueError("all folders are disabled in the config") + self.check_website = check_website + if self.check_website != "default": + parsed_url = urlparse(check_website) + if not parsed_url.scheme or not parsed_url.netloc: + logger.error("Invalid CheckWebsite URL: %s", check_website) + sys.exit(1) + + logger.info( + "CheckWebsite is not 'default', " + + "so it will not be possible to determine " + + "the anonymity and geolocation of the proxies" + ) + for folder in self.folders: + folder.is_enabled = ( + not folder.for_anonymous and not folder.for_geolocation + ) + elif not any(folder for folder in self.folders if folder.is_enabled): + logger.error("All folders are disabled in the config") + sys.exit(1) + + self.path = save_path self.regex = re.compile( r"(?:^|\D)?(" + r"(?:[1-9]|[1-9]\d|1\d{2}|2[0-4]\d|25[0-5])" # 1-255 @@ -167,6 +191,7 @@ class ProxyScraperChecker: timeout=general.getfloat("Timeout", 5), source_timeout=general.getfloat("SourceTimeout", 15), max_connections=general.getint("MaxConnections", 512), + check_website=general.get("CheckWebsite", "default"), sort_by_speed=general.getboolean("SortBySpeed", True), save_path=save_path, folders=( @@ -258,6 +283,7 @@ class ProxyScraperChecker: """Check if proxy is alive.""" try: await proxy.check( + website=self.check_website, sem=self.sem, cookie_jar=self.cookie_jar, proto=proto, @@ -279,14 +305,8 @@ class ProxyScraperChecker: ) for proto, sources in self.sources.items() } - headers = { - "User-Agent": ( - "Mozilla/5.0 (Windows NT 10.0; rv:108.0)" - + " Gecko/20100101 Firefox/108.0" - ) - } async with ClientSession( - headers=headers, + headers={"User-Agent": USER_AGENT}, cookie_jar=self.cookie_jar, timeout=ClientTimeout(total=self.source_timeout), ) as session: