🎈 perf(github_utils): 支持github url下载遍历 (#1632)

* 🎈 perf(github_utils): 支持github url下载遍历

* 🐞 fix(http_utils): 修复一些下载问题

* 🦄 refactor(http_utils): 部分重构

* chore(version): Update version to v0.2.2-e6f17c4

---------

Co-authored-by: AkashiCoin <AkashiCoin@users.noreply.github.com>
This commit is contained in:
AkashiCoin
2024-09-16 20:08:42 +08:00
committed by GitHub
co-authored by AkashiCoin
parent cd88d805ce
commit 51c010daa8
9 changed files with 197 additions and 126 deletions
+18 -14
View File
@@ -1,23 +1,27 @@
from collections.abc import Generator
from .consts import GITHUB_REPO_URL_PATTERN
from .func import get_fastest_raw_format, get_fastest_archive_format
from .func import get_fastest_raw_formats, get_fastest_archive_formats
from .models import RepoAPI, RepoInfo, GitHubStrategy, JsdelivrStrategy
__all__ = [
"parse_github_url",
"get_fastest_raw_format",
"get_fastest_archive_format",
"api_strategy",
"get_fastest_raw_formats",
"get_fastest_archive_formats",
"GithubUtils",
]
def parse_github_url(github_url: str) -> "RepoInfo":
if matched := GITHUB_REPO_URL_PATTERN.match(github_url):
return RepoInfo(**{k: v for k, v in matched.groupdict().items() if v})
raise ValueError("github地址格式错误")
class GithubUtils:
# 使用
jsdelivr_api = RepoAPI(JsdelivrStrategy()) # type: ignore
github_api = RepoAPI(GitHubStrategy()) # type: ignore
@classmethod
def iter_api_strategies(cls) -> Generator[RepoAPI]:
yield from [cls.github_api, cls.jsdelivr_api]
# 使用
jsdelivr_api = RepoAPI(JsdelivrStrategy()) # type: ignore
github_api = RepoAPI(GitHubStrategy()) # type: ignore
api_strategy = [github_api, jsdelivr_api]
@classmethod
def parse_github_url(cls, github_url: str) -> "RepoInfo":
if matched := GITHUB_REPO_URL_PATTERN.match(github_url):
return RepoInfo(**{k: v for k, v in matched.groupdict().items() if v})
raise ValueError("github地址格式错误")
+10 -10
View File
@@ -9,15 +9,15 @@ from .consts import (
)
async def __get_fastest_format(formats: dict[str, str]) -> str:
async def __get_fastest_formats(formats: dict[str, str]) -> list[str]:
sorted_urls = await AsyncHttpx.get_fastest_mirror(list(formats.keys()))
if not sorted_urls:
raise Exception("无法获取任意GitHub资源加速地址,请检查网络")
return formats[sorted_urls[0]]
return [formats[url] for url in sorted_urls]
@cached()
async def get_fastest_raw_format() -> str:
async def get_fastest_raw_formats() -> list[str]:
"""获取最快的raw下载地址格式"""
formats: dict[str, str] = {
"https://raw.githubusercontent.com/": RAW_CONTENT_FORMAT,
@@ -26,11 +26,11 @@ async def get_fastest_raw_format() -> str:
"https://gh-proxy.com/": f"https://gh-proxy.com/{RAW_CONTENT_FORMAT}",
"https://cdn.jsdelivr.net/": "https://cdn.jsdelivr.net/gh/{owner}/{repo}@{branch}/{path}",
}
return await __get_fastest_format(formats)
return await __get_fastest_formats(formats)
@cached()
async def get_fastest_archive_format() -> str:
async def get_fastest_archive_formats() -> list[str]:
"""获取最快的归档下载地址格式"""
formats: dict[str, str] = {
"https://github.com/": ARCHIVE_URL_FORMAT,
@@ -38,11 +38,11 @@ async def get_fastest_archive_format() -> str:
"https://mirror.ghproxy.com/": f"https://mirror.ghproxy.com/{ARCHIVE_URL_FORMAT}",
"https://gh-proxy.com/": f"https://gh-proxy.com/{ARCHIVE_URL_FORMAT}",
}
return await __get_fastest_format(formats)
return await __get_fastest_formats(formats)
@cached()
async def get_fastest_release_format() -> str:
async def get_fastest_release_formats() -> list[str]:
"""获取最快的发行版资源下载地址格式"""
formats: dict[str, str] = {
"https://objects.githubusercontent.com/": RELEASE_ASSETS_FORMAT,
@@ -50,14 +50,14 @@ async def get_fastest_release_format() -> str:
"https://mirror.ghproxy.com/": f"https://mirror.ghproxy.com/{RELEASE_ASSETS_FORMAT}",
"https://gh-proxy.com/": f"https://gh-proxy.com/{RELEASE_ASSETS_FORMAT}",
}
return await __get_fastest_format(formats)
return await __get_fastest_formats(formats)
@cached()
async def get_fastest_release_source_format() -> str:
async def get_fastest_release_source_formats() -> list[str]:
"""获取最快的发行版源码下载地址格式"""
formats: dict[str, str] = {
"https://codeload.github.com/": RELEASE_SOURCE_FORMAT,
"https://p.102333.xyz/": f"https://p.102333.xyz/{RELEASE_SOURCE_FORMAT}",
}
return await __get_fastest_format(formats)
return await __get_fastest_formats(formats)
+35 -15
View File
@@ -7,9 +7,9 @@ from pydantic import BaseModel
from ..http_utils import AsyncHttpx
from .consts import CACHED_API_TTL, GIT_API_TREES_FORMAT, JSD_PACKAGE_API_FORMAT
from .func import (
get_fastest_raw_format,
get_fastest_archive_format,
get_fastest_release_source_format,
get_fastest_raw_formats,
get_fastest_archive_formats,
get_fastest_release_source_formats,
)
@@ -20,21 +20,41 @@ class RepoInfo(BaseModel):
repo: str
branch: str = "main"
async def get_raw_download_url(self, path: str):
url_format = await get_fastest_raw_format()
return url_format.format(**self.dict(), path=path)
async def get_raw_download_url(self, path: str) -> str:
return (await self.get_raw_download_urls(path))[0]
async def get_archive_download_url(self):
url_format = await get_fastest_archive_format()
return url_format.format(**self.dict())
async def get_archive_download_url(self) -> str:
return (await self.get_archive_download_urls())[0]
async def get_release_source_download_url_tgz(self, version: str):
url_format = await get_fastest_release_source_format()
return url_format.format(**self.dict(), version=version, compress="tar.gz")
async def get_release_source_download_url_tgz(self, version: str) -> str:
return (await self.get_release_source_download_urls_tgz(version))[0]
async def get_release_source_download_url_zip(self, version: str):
url_format = await get_fastest_release_source_format()
return url_format.format(**self.dict(), version=version, compress="zip")
async def get_release_source_download_url_zip(self, version: str) -> str:
return (await self.get_release_source_download_urls_zip(version))[0]
async def get_raw_download_urls(self, path: str) -> list[str]:
url_formats = await get_fastest_raw_formats()
return [
url_format.format(**self.dict(), path=path) for url_format in url_formats
]
async def get_archive_download_urls(self) -> list[str]:
url_formats = await get_fastest_archive_formats()
return [url_format.format(**self.dict()) for url_format in url_formats]
async def get_release_source_download_urls_tgz(self, version: str) -> list[str]:
url_formats = await get_fastest_release_source_formats()
return [
url_format.format(**self.dict(), version=version, compress="tar.gz")
for url_format in url_formats
]
async def get_release_source_download_urls_zip(self, version: str) -> list[str]:
url_formats = await get_fastest_release_source_formats()
return [
url_format.format(**self.dict(), version=version, compress="zip")
for url_format in url_formats
]
class APIStrategy(Protocol):
+112 -65
View File
@@ -11,9 +11,9 @@ import httpx
import aiofiles
from retrying import retry
from playwright.async_api import Page
from httpx import Response, ConnectTimeout
from nonebot_plugin_alconna import UniMessage
from nonebot_plugin_htmlrender import get_browser
from httpx import Response, ConnectTimeout, HTTPStatusError
from zhenxun.services.log import logger
from zhenxun.configs.config import BotConfig
@@ -33,7 +33,7 @@ class AsyncHttpx:
@retry(stop_max_attempt_number=3)
async def get(
cls,
url: str,
url: str | list[str],
*,
params: dict[str, Any] | None = None,
headers: dict[str, str] | None = None,
@@ -56,6 +56,49 @@ class AsyncHttpx:
proxy: 指定代理
timeout: 超时时间
"""
urls = [url] if isinstance(url, str) else url
return await cls._get_first_successful(
urls,
params=params,
headers=headers,
cookies=cookies,
verify=verify,
use_proxy=use_proxy,
proxy=proxy,
timeout=timeout,
**kwargs,
)
@classmethod
async def _get_first_successful(
cls,
urls: list[str],
**kwargs,
) -> Response:
last_exception = None
for url in urls:
try:
return await cls._get_single(url, **kwargs)
except Exception as e:
last_exception = e
if url != urls[-1]:
logger.warning(f"获取 {url} 失败, 尝试下一个")
raise last_exception or Exception("All URLs failed")
@classmethod
async def _get_single(
cls,
url: str,
*,
params: dict[str, Any] | None = None,
headers: dict[str, str] | None = None,
cookies: dict[str, str] | None = None,
verify: bool = True,
use_proxy: bool = True,
proxy: dict[str, str] | None = None,
timeout: int = 30,
**kwargs,
) -> Response:
if not headers:
headers = get_user_agent()
_proxy = proxy if proxy else cls.proxy if use_proxy else None
@@ -162,7 +205,7 @@ class AsyncHttpx:
@classmethod
async def download_file(
cls,
url: str,
url: str | list[str],
path: str | Path,
*,
params: dict[str, str] | None = None,
@@ -195,75 +238,79 @@ class AsyncHttpx:
path.parent.mkdir(parents=True, exist_ok=True)
try:
for _ in range(3):
if not stream:
if not isinstance(url, list):
url = [url]
for u in url:
try:
response = await cls.get(
url,
params=params,
headers=headers,
cookies=cookies,
use_proxy=use_proxy,
proxy=proxy,
timeout=timeout,
follow_redirects=follow_redirects,
**kwargs,
)
response.raise_for_status()
content = response.content
async with aiofiles.open(path, "wb") as wf:
await wf.write(content)
logger.info(f"下载 {url} 成功.. Path:{path.absolute()}")
return True
except (TimeoutError, ConnectTimeout):
pass
else:
if not headers:
headers = get_user_agent()
_proxy = proxy if proxy else cls.proxy if use_proxy else None
try:
async with httpx.AsyncClient(
proxies=_proxy, # type: ignore
verify=verify,
) as client:
async with client.stream(
"GET",
url,
if not stream:
response = await cls.get(
u,
params=params,
headers=headers,
cookies=cookies,
use_proxy=use_proxy,
proxy=proxy,
timeout=timeout,
follow_redirects=True,
follow_redirects=follow_redirects,
**kwargs,
) as response:
response.raise_for_status()
logger.info(
f"开始下载 {path.name}.. Path: {path.absolute()}"
)
async with aiofiles.open(path, "wb") as wf:
total = int(response.headers["Content-Length"])
with rich.progress.Progress( # type: ignore
rich.progress.TextColumn(path.name), # type: ignore
"[progress.percentage]{task.percentage:>3.0f}%", # type: ignore
rich.progress.BarColumn(bar_width=None), # type: ignore
rich.progress.DownloadColumn(), # type: ignore
rich.progress.TransferSpeedColumn(), # type: ignore
) as progress:
download_task = progress.add_task(
"Download", total=total
)
async for chunk in response.aiter_bytes():
await wf.write(chunk)
await wf.flush()
progress.update(
download_task,
completed=response.num_bytes_downloaded,
)
)
response.raise_for_status()
content = response.content
async with aiofiles.open(path, "wb") as wf:
await wf.write(content)
logger.info(f"下载 {u} 成功.. Path:{path.absolute()}")
return True
else:
if not headers:
headers = get_user_agent()
_proxy = (
proxy if proxy else cls.proxy if use_proxy else None
)
async with httpx.AsyncClient(
proxies=_proxy, # type: ignore
verify=verify,
) as client:
async with client.stream(
"GET",
u,
params=params,
headers=headers,
cookies=cookies,
timeout=timeout,
follow_redirects=True,
**kwargs,
) as response:
response.raise_for_status()
logger.info(
f"下载 {url} 成功.. Path:{path.absolute()}"
f"开始下载 {path.name}.. "
f"Path: {path.absolute()}"
)
return True
except (TimeoutError, ConnectTimeout):
pass
async with aiofiles.open(path, "wb") as wf:
total = int(response.headers["Content-Length"])
with rich.progress.Progress( # type: ignore
rich.progress.TextColumn(path.name), # type: ignore
"[progress.percentage]{task.percentage:>3.0f}%", # type: ignore
rich.progress.BarColumn(bar_width=None), # type: ignore
rich.progress.DownloadColumn(), # type: ignore
rich.progress.TransferSpeedColumn(), # type: ignore
) as progress:
download_task = progress.add_task(
"Download", total=total
)
async for chunk in response.aiter_bytes():
await wf.write(chunk)
await wf.flush()
progress.update(
download_task,
completed=response.num_bytes_downloaded,
)
logger.info(
f"下载 {u} 成功.. "
f"Path:{path.absolute()}"
)
return True
except (TimeoutError, ConnectTimeout, HTTPStatusError):
logger.warning(f"下载 {u} 失败.. 尝试下一个地址..")
else:
logger.error(f"下载 {url} 下载超时.. Path:{path.absolute()}")
except Exception as e:
@@ -273,7 +320,7 @@ class AsyncHttpx:
@classmethod
async def gather_download_file(
cls,
url_list: list[str],
url_list: list[str] | list[list[str]],
path_list: list[str | Path],
*,
limit_async_number: int | None = None,