Add CheckWebsite setting

Fixes #16, fixes #39
This commit is contained in:
monosans
2023-01-15 13:05:16 +03:00
parent 7298a5594f
commit d21e32e4fc
6 changed files with 72 additions and 27 deletions
+4 -2
View File
@@ -5,9 +5,11 @@
HTTP, SOCKS4, SOCKS5 proxies scraper and checker.
- Asynchronous.
- Uses regex to search for proxies (ip:port format) on a web page, which allows you to pull out proxies even from json without making any changes to the code.
- Uses regex to search for proxies (ip:port format) on a web page, allowing proxies to be extracted even from json without making changes to the code.
- It is possible to specify the URL to which to send a request to check the proxy.
- Can sort proxies by speed.
- Supports determining the geolocation of the proxy exit node.
- Can determine if a proxy is anonymous.
- Can determine if the proxy is anonymous.
You can get proxies obtained using this script in [monosans/proxy-list](https://github.com/monosans/proxy-list).
+5
View File
@@ -15,6 +15,11 @@ SourceTimeout = 15
; Make sure you have enough RAM first, gradually increasing the default value.
MaxConnections = 512
; URL to which to send a request to check the proxy.
; If not equal to 'default', it will not be possible
; to determine the anonymity and geolocation of the proxies.
CheckWebsite = default
; Set to no to sort proxies alphabetically.
SortBySpeed = yes
+5
View File
@@ -0,0 +1,5 @@
from __future__ import annotations
USER_AGENT = (
"Mozilla/5.0 (Windows NT 10.0; rv:108.0) Gecko/20100101 Firefox/108.0"
)
+1 -1
View File
@@ -5,7 +5,7 @@ from pathlib import Path
from shutil import rmtree
@dataclass(frozen=True)
@dataclass
class Folder:
path: Path
is_enabled: bool
+26 -13
View File
@@ -7,6 +7,8 @@ from aiohttp import ClientSession, ClientTimeout
from aiohttp.abc import AbstractCookieJar
from aiohttp_socks import ProxyConnector, ProxyType
from .constants import USER_AGENT
class Proxy:
__slots__ = ("geolocation", "host", "is_anonymous", "port", "timeout")
@@ -15,30 +17,41 @@ class Proxy:
self.host = host
self.port = port
@property
def default_check_website(self) -> str:
return "http://ip-api.com/json/?fields=8217"
async def check(
self,
*,
website: str,
sem: asyncio.Semaphore,
cookie_jar: AbstractCookieJar,
proto: ProxyType,
timeout: ClientTimeout,
) -> None:
check_website = (
self.default_check_website if website == "default" else website
)
async with sem:
start = perf_counter()
async with self.get_connector(proto) as connector:
async with ClientSession(
connector=connector, cookie_jar=cookie_jar, timeout=timeout
) as session:
async with session.get(
"http://ip-api.com/json/?fields=8217",
raise_for_status=True,
) as response:
data = await response.json()
async with self.get_connector(proto) as connector, ClientSession(
connector=connector,
cookie_jar=cookie_jar,
timeout=timeout,
headers={"User-Agent": USER_AGENT},
) as session, session.get(
check_website, raise_for_status=True
) as response:
if website == "default":
await response.read()
self.timeout = perf_counter() - start
self.is_anonymous = self.host != data["query"]
self.geolocation = "|{}|{}|{}".format(
data["country"], data["regionName"], data["city"]
)
if website == "default":
data = await response.json()
self.is_anonymous = self.host != data["query"]
self.geolocation = "|{}|{}|{}".format(
data["country"], data["regionName"], data["city"]
)
def get_connector(self, proto: ProxyType) -> ProxyConnector:
return ProxyConnector(proxy_type=proto, host=self.host, port=self.port)
+31 -11
View File
@@ -18,6 +18,7 @@ from typing import (
TypeVar,
Union,
)
from urllib.parse import urlparse
from aiohttp import ClientSession, ClientTimeout, DummyCookieJar
from aiohttp_socks import ProxyType
@@ -32,6 +33,7 @@ from rich.progress import (
from rich.table import Table
from . import sort
from .constants import USER_AGENT
from .folder import Folder
from .proxy import Proxy
@@ -67,6 +69,7 @@ class ProxyScraperChecker:
"""HTTP, SOCKS4, SOCKS5 proxies scraper and checker."""
__slots__ = (
"check_website",
"console",
"cookie_jar",
"folders",
@@ -87,6 +90,7 @@ class ProxyScraperChecker:
timeout: float,
source_timeout: float,
max_connections: int,
check_website: str,
sort_by_speed: bool,
save_path: Path,
folders: Tuple[Folder, ...],
@@ -108,17 +112,37 @@ class ProxyScraperChecker:
Don't be in a hurry to set high values.
Make sure you have enough RAM first, gradually
increasing the default value.
check_website: URL to which to send a request to check the proxy.
If not equal to 'default', it will not be possible
to determine the anonymity and geolocation of the proxies.
sort_by_speed: Set to False to sort proxies alphabetically.
save_path: Path to the folder where the proxy folders will
be saved. Leave empty to save the proxies to the current
directory.
"""
self.path = save_path
self.folders = folders
if not any(folder for folder in self.folders if folder.is_enabled):
raise ValueError("all folders are disabled in the config")
self.check_website = check_website
if self.check_website != "default":
parsed_url = urlparse(check_website)
if not parsed_url.scheme or not parsed_url.netloc:
logger.error("Invalid CheckWebsite URL: %s", check_website)
sys.exit(1)
logger.info(
"CheckWebsite is not 'default', "
+ "so it will not be possible to determine "
+ "the anonymity and geolocation of the proxies"
)
for folder in self.folders:
folder.is_enabled = (
not folder.for_anonymous and not folder.for_geolocation
)
elif not any(folder for folder in self.folders if folder.is_enabled):
logger.error("All folders are disabled in the config")
sys.exit(1)
self.path = save_path
self.regex = re.compile(
r"(?:^|\D)?("
+ r"(?:[1-9]|[1-9]\d|1\d{2}|2[0-4]\d|25[0-5])" # 1-255
@@ -167,6 +191,7 @@ class ProxyScraperChecker:
timeout=general.getfloat("Timeout", 5),
source_timeout=general.getfloat("SourceTimeout", 15),
max_connections=general.getint("MaxConnections", 512),
check_website=general.get("CheckWebsite", "default"),
sort_by_speed=general.getboolean("SortBySpeed", True),
save_path=save_path,
folders=(
@@ -258,6 +283,7 @@ class ProxyScraperChecker:
"""Check if proxy is alive."""
try:
await proxy.check(
website=self.check_website,
sem=self.sem,
cookie_jar=self.cookie_jar,
proto=proto,
@@ -279,14 +305,8 @@ class ProxyScraperChecker:
)
for proto, sources in self.sources.items()
}
headers = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; rv:108.0)"
+ " Gecko/20100101 Firefox/108.0"
)
}
async with ClientSession(
headers=headers,
headers={"User-Agent": USER_AGENT},
cookie_jar=self.cookie_jar,
timeout=ClientTimeout(total=self.source_timeout),
) as session: