Baseline: pr1 HYC下载站 v2.3 before security/functional fixes

This commit is contained in:
HYC Fixer
2026-08-30 12:12:58 +08:00
commit a8e773839b
77 changed files with 38568 additions and 0 deletions
+583
View File
@@ -0,0 +1,583 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
镜像源处理器包
支持 Docker、APT、YUM、PyPI、npm、Go、Maven、Gradle 等镜像源代理
"""
from .docker import DockerMirror
from .apt import APTMirror
from .yum import YUMMirror
from .pypi import PyPIMirror
from .npm import NpmMirror
from .go import GoProxy
from .http import HttpMirror
# 镜像处理器映射
# 特殊类型使用专用处理器,其他使用通用HTTP处理器
MIRROR_HANDLERS = {
# 专用处理器
'docker': DockerMirror,
'apt': APTMirror,
'yum': YUMMirror,
'pypi': PyPIMirror,
'npm': NpmMirror,
'go': GoProxy,
# ======== 包管理器 ========
# Python
'pip': HttpMirror,
'pipenv': HttpMirror,
'poetry': HttpMirror,
'conda': HttpMirror,
'anaconda': HttpMirror,
'bandersnatch': HttpMirror,
'twine': HttpMirror,
# Node.js
'yarn': HttpMirror,
'pnpm': HttpMirror,
'bower': HttpMirror,
# Java
'maven': HttpMirror,
'gradle': HttpMirror,
# .NET
'nuget': HttpMirror,
# Ruby
'gem': HttpMirror,
'rubygems': HttpMirror,
# Rust
'cargo': HttpMirror,
'rustup': HttpMirror,
# PHP
'composer': HttpMirror,
'packagist': HttpMirror,
# Swift/Apple
'cocoapods': HttpMirror,
# C/C++
'conan': HttpMirror,
'vcpkg': HttpMirror,
# Dart/Flutter
'pub': HttpMirror,
'dart': HttpMirror,
'flutter': HttpMirror,
# Haskell
'hackage': HttpMirror,
'stackage': HttpMirror,
# OCaml
'opam': HttpMirror,
# D语言
'dub': HttpMirror,
# Nim
'nimble': HttpMirror,
# V语言
'v': HttpMirror,
# Julia
'julia': HttpMirror,
# Lua
'lua': HttpMirror,
'luarocks': HttpMirror,
# Elm
'elm': HttpMirror,
# Perl
'cpan': HttpMirror,
'cpanm': HttpMirror,
# R
'cran': HttpMirror,
# LaTeX
'ctan': HttpMirror,
# Haskell
'haskell': HttpMirror,
# ======== 容器/云原生 ========
'helm': HttpMirror,
'kubernetes': HttpMirror,
' Quay.io': HttpMirror,
'quay': HttpMirror,
'ghcr': HttpMirror,
'gcr': HttpMirror,
'harbor': HttpMirror,
# ======== 开发工具 ========
'jetbrains': HttpMirror,
'vscode': HttpMirror,
'cuda': HttpMirror,
# ======== 语言运行时 ========
'node': HttpMirror,
'python': HttpMirror,
'ruby': HttpMirror,
'php': HttpMirror,
'java': HttpMirror,
'dotnet': HttpMirror,
# ======== Linux 发行版 ========
'alpine': HttpMirror,
'arch': HttpMirror,
'aur': HttpMirror,
'centos': HttpMirror,
'debian': HttpMirror,
'fedora': HttpMirror,
'gentoo': HttpMirror,
'opensuse': HttpMirror,
'void': HttpMirror,
'freebsd': HttpMirror,
'netbsd': HttpMirror,
'openbsd': HttpMirror,
'rocky': HttpMirror,
'alma': HttpMirror,
'kali': HttpMirror,
'ubuntu': HttpMirror,
'mint': HttpMirror,
'manjaro': HttpMirror,
'slackware': HttpMirror,
'mageia': HttpMirror,
'openmandriva': HttpMirror,
# ======== 系统包管理器 ========
'homebrew': HttpMirror,
'brew': HttpMirror,
'chocolatey': HttpMirror,
'snap': HttpMirror,
'flatpak': HttpMirror,
'appimage': HttpMirror,
'winget': HttpMirror,
'scoop': HttpMirror,
# ======== 数据库/运维 ========
'postgresql': HttpMirror,
'mysql': HttpMirror,
'mariadb': HttpMirror,
'mongodb': HttpMirror,
'redis': HttpMirror,
'influxdata': HttpMirror,
'grafana': HttpMirror,
'prometheus': HttpMirror,
'elastic': HttpMirror,
'bitnami': HttpMirror,
# ======== 代码托管/源码 ========
'git': HttpMirror,
'github': HttpMirror,
'gitlab': HttpMirror,
'bitbucket': HttpMirror,
'sourceforge': HttpMirror,
# ======== 其他 ========
'pacman': HttpMirror,
'nix': HttpMirror,
'guix': HttpMirror,
'termux': HttpMirror,
'msys2': HttpMirror,
'google-fonts': HttpMirror,
'aurora': HttpMirror,
'cloudflare': HttpMirror,
'fastly': HttpMirror,
# ======== 自定义 ========
'http': HttpMirror,
'custom': HttpMirror,
}
def get_mirror_handler(mirror_type: str):
"""获取镜像处理器类"""
return MIRROR_HANDLERS.get(mirror_type)
# 各镜像类型的默认上游URL
DEFAULT_UPSTREAM_URLS = {
# ======== 包管理器 ========
# Python
'pip': 'https://pypi.org/simple',
'pipenv': 'https://pypi.org/simple',
'poetry': 'https://pypi.org/simple',
'conda': 'https://repo.anaconda.com',
'anaconda': 'https://repo.anaconda.com',
'bandersnatch': 'https://pypi.org/simple',
'twine': 'https://pypi.org/simple',
# Node.js
'yarn': 'https://registry.yarnpkg.com',
'pnpm': 'https://registry.npmjs.org',
'bower': 'https://registry.bower.io',
# Java
'maven': 'https://repo1.maven.org/maven2',
'gradle': 'https://services.gradle.org/distributions',
# .NET
'nuget': 'https://api.nuget.org/v3',
# Ruby
'gem': 'https://rubygems.org',
'rubygems': 'https://rubygems.org',
# Rust
'cargo': 'https://crates.io',
'rustup': 'https://static.rust-lang.org',
# PHP
'composer': 'https://repo.packagist.org',
'packagist': 'https://repo.packagist.org',
# Swift/Apple
'cocoapods': 'https://cdn.cocoapods.org',
# C/C++
'conan': 'https://center.conan.io',
'vcpkg': 'https://github.com/microsoft/vcpkg/archive/refs/heads/main.tar.gz',
# Dart/Flutter
'pub': 'https://pub.dev',
'dart': 'https://storage.googleapis.com/flutter_infra_release',
'flutter': 'https://storage.flutter-io.cn',
# Haskell
'hackage': 'https://hackage.haskell.org',
'stackage': 'https://www.stackage.org',
# OCaml
'opam': 'https://opam.ocaml.org',
# D语言
'dub': 'https://code.dlang.org',
# Nim
'nimble': 'https://nimble.directory',
# V语言
'v': 'https://vpm.itsgriffin.com',
# Julia
'julia': 'https://pkg.julialang.org',
# Lua
'lua': 'https://luarocks.org',
'luarocks': 'https://luarocks.org',
# Elm
'elm': 'https://package.elm-lang.org',
# Perl
'cpan': 'https://www.cpan.org',
'cpanm': 'https://www.cpan.org',
# R
'cran': 'https://cran.r-project.org',
# LaTeX
'ctan': 'https://ctan.math.illinois.edu',
# Haskell
'haskell': 'https://haskell.org',
# ======== 容器/云原生 ========
'helm': 'https://charts.helm.sh',
'kubernetes': 'https://dl.k8s.io/release',
'quay': 'https://quay.io',
'quay.io': 'https://quay.io',
'ghcr': 'https://ghcr.io',
'gcr': 'https://gcr.io',
'harbor': 'https://harbor.io',
# ======== 开发工具 ========
'jetbrains': 'https://www.jetbrains.com',
'vscode': 'https://code.visualstudio.com',
'cuda': 'https://developer.download.nvidia.com/compute/cuda/repos',
# ======== 语言运行时 ========
'node': 'https://nodejs.org/dist',
'python': 'https://www.python.org/ftp/python',
'ruby': 'https://www.ruby-lang.org',
'php': 'https://www.php.net/distributions',
'java': 'https://download.oracle.com/java',
'dotnet': 'https://dotnetcli.azureedge.net',
# ======== Linux 发行版 ========
'alpine': 'https://dl-cdn.alpinelinux.org',
'arch': 'https://mirror.archlinux.org',
'aur': 'https://aur.archlinux.org',
'centos': 'https://mirrors.aliyun.com/centos',
'debian': 'https://deb.debian.org/debian',
'fedora': 'https://mirrors.fedoraproject.org',
'gentoo': 'https://distfiles.gentoo.org',
'opensuse': 'https://download.opensuse.org',
'void': 'https://repo.voidlinux.org',
'freebsd': 'https://pkg.freebsd.org',
'netbsd': 'https://cdn.netbsd.org',
'openbsd': 'https://cdn.openbsd.org',
'rocky': 'https://download.rockylinux.org',
'alma': 'https://repo.almalinux.org',
'kali': 'http://.kali.org',
'ubuntu': 'https://releases.ubuntu.com',
'mint': 'https://packages.linuxmint.com',
'manjaro': 'https://repo.manjaro.org',
'slackware': 'https://mirrors.slackware.com',
'mageia': 'https://www.mageia.org',
'openmandriva': 'https://download.openmandriva.org',
# ======== 系统包管理器 ========
'homebrew': 'https://github.com/Homebrew/brew',
'brew': 'https://github.com/Homebrew/brew',
'chocolatey': 'https://chocolatey.org/api/v2',
'snap': 'https://snapcraft.io',
'flatpak': 'https://flathub.org',
'appimage': 'https://github.com/AppImage/AppImageHub/releases',
'winget': 'https://github.com/microsoft/winget-pkgs',
'scoop': 'https://scoop.sh',
# ======== 数据库/运维 ========
'postgresql': 'https://www.postgresql.org',
'mysql': 'https://dev.mysql.com',
'mariadb': 'https://mariadb.org',
'mongodb': 'https://www.mongodb.org',
'redis': 'https://redis.io',
'influxdata': 'https://portal.influxdata.com',
'grafana': 'https://grafana.com',
'prometheus': 'https://prometheus.io',
'elastic': 'https://www.elastic.co',
'bitnami': 'https://bitnami.com',
# ======== 代码托管/源码 ========
'git': 'https://github.com',
'github': 'https://github.com',
'gitlab': 'https://gitlab.com',
'bitbucket': 'https://bitbucket.org',
'sourceforge': 'https://sourceforge.net',
# ======== 其他 ========
'pacman': 'https://mirror.archlinux.org',
'nix': 'https://nix-community.org',
'guix': 'https://guix.gnu.org',
'termux': 'https://termux.net',
'msys2': 'https://repo.msys2.org',
'google-fonts': 'https://fonts.google.com',
'aurora': 'https://auroralinux.org',
'cloudflare': 'https://cloudflare.com',
'fastly': 'https://fastly.com',
}
def get_default_upstream(mirror_type: str) -> str:
"""获取镜像类型的默认上游URL"""
return DEFAULT_UPSTREAM_URLS.get(mirror_type, 'https://mirror.example.com')
def list_available_mirrors() -> list:
"""列出可用的镜像类型"""
return [
# ======== 专用处理器 ========
{
'type': 'docker',
'name': 'Docker Registry',
'description': 'Docker镜像代理'
},
{
'type': 'apt',
'name': 'APT (Debian/Ubuntu)',
'description': 'APT软件源代理'
},
{
'type': 'yum',
'name': 'YUM/DNF (RHEL/CentOS)',
'description': 'YUM/DNF软件源代理'
},
{
'type': 'pypi',
'name': 'PyPI',
'description': 'Python包索引代理'
},
{
'type': 'npm',
'name': 'npm Registry',
'description': 'Node.js包管理器代理'
},
{
'type': 'go',
'name': 'Go Modules',
'description': 'Go模块代理'
},
# ======== 包管理器 ========
# Python
{'type': 'pip', 'name': 'pip', 'description': 'Python pip包管理器'},
{'type': 'pipenv', 'name': 'Pipenv', 'description': 'Python Pipenv代理'},
{'type': 'poetry', 'name': 'Poetry', 'description': 'Python Poetry包管理器'},
{'type': 'conda', 'name': 'Conda', 'description': 'Python Conda包管理器'},
{'type': 'anaconda', 'name': 'Anaconda', 'description': 'Anaconda Python发行版'},
# Node.js
{'type': 'yarn', 'name': 'Yarn', 'description': 'Node.js Yarn包管理器'},
{'type': 'pnpm', 'name': 'pnpm', 'description': 'Node.js pnpm包管理器'},
{'type': 'bower', 'name': 'Bower', 'description': '前端包管理器'},
# Java
{'type': 'maven', 'name': 'Maven Central', 'description': 'Java/Maven包管理器'},
{'type': 'gradle', 'name': 'Gradle', 'description': 'Gradle构建工具分发'},
# .NET
{'type': 'nuget', 'name': 'NuGet', 'description': '.NET包管理器'},
# Ruby
{'type': 'gem', 'name': 'RubyGems', 'description': 'Ruby包管理器'},
{'type': 'rubygems', 'name': 'RubyGems', 'description': 'Ruby官方仓库'},
# Rust
{'type': 'cargo', 'name': 'Crates.io', 'description': 'Rust语言包管理器'},
{'type': 'rustup', 'name': 'Rustup', 'description': 'Rust工具链'},
# PHP
{'type': 'composer', 'name': 'Composer', 'description': 'PHP包管理器'},
{'type': 'packagist', 'name': 'Packagist', 'description': 'PHP官方包仓库'},
# Swift/Apple
{'type': 'cocoapods', 'name': 'CocoaPods', 'description': 'iOS/macOS包管理器'},
# C/C++
{'type': 'conan', 'name': 'Conan', 'description': 'C/C++包管理器'},
{'type': 'vcpkg', 'name': 'vcpkg', 'description': 'C/C++包管理器'},
# Dart/Flutter
{'type': 'pub', 'name': 'Pub.dev', 'description': 'Dart/Flutter包管理器'},
{'type': 'dart', 'name': 'Dart SDK', 'description': 'Dart SDK'},
{'type': 'flutter', 'name': 'Flutter', 'description': 'Flutter SDK'},
# Haskell
{'type': 'hackage', 'name': 'Hackage', 'description': 'Haskell包管理器'},
{'type': 'stackage', 'name': 'Stackage', 'description': 'Haskell Stackage'},
# OCaml
{'type': 'opam', 'name': 'OPAM', 'description': 'OCaml包管理器'},
# D语言
{'type': 'dub', 'name': 'Dub', 'description': 'D语言包管理器'},
# Nim
{'type': 'nimble', 'name': 'Nimble', 'description': 'Nim语言包管理器'},
# V语言
{'type': 'v', 'name': 'V PM', 'description': 'V语言包管理器'},
# Julia
{'type': 'julia', 'name': 'Julia Packages', 'description': 'Julia语言包管理器'},
# Lua
{'type': 'lua', 'name': 'Lua', 'description': 'Lua语言'},
{'type': 'luarocks', 'name': 'LuaRocks', 'description': 'Lua包管理器'},
# Elm
{'type': 'elm', 'name': 'Elm Packages', 'description': 'Elm语言包管理器'},
# Perl
{'type': 'cpan', 'name': 'CPAN', 'description': 'Perl包管理器'},
{'type': 'cpanm', 'name': 'cpanm', 'description': 'Perl cpanminus'},
# R
{'type': 'cran', 'name': 'CRAN', 'description': 'R语言包仓库'},
# LaTeX
{'type': 'ctan', 'name': 'CTAN', 'description': 'LaTeX包管理器'},
# ======== 容器/云原生 ========
{'type': 'helm', 'name': 'Helm Charts', 'description': 'Kubernetes Helm包'},
{'type': 'kubernetes', 'name': 'Kubernetes', 'description': 'Kubernetes发行版'},
{'type': 'quay', 'name': 'Quay.io', 'description': 'Quay容器仓库'},
{'type': 'ghcr', 'name': 'GitHub Container Registry', 'description': 'GitHub容器仓库'},
{'type': 'gcr', 'name': 'Google Container Registry', 'description': 'Google容器仓库'},
{'type': 'harbor', 'name': 'Harbor', 'description': 'Harbor容器仓库'},
# ======== 开发工具 ========
{'type': 'jetbrains', 'name': 'JetBrains', 'description': 'JetBrains IDE'},
{'type': 'vscode', 'name': 'VS Code', 'description': 'Visual Studio Code'},
{'type': 'cuda', 'name': 'CUDA', 'description': 'NVIDIA CUDA工具包'},
# ======== 语言运行时 ========
{'type': 'node', 'name': 'Node.js', 'description': 'Node.js官方'},
{'type': 'python', 'name': 'Python', 'description': 'Python官方'},
{'type': 'ruby', 'name': 'Ruby', 'description': 'Ruby官方'},
{'type': 'php', 'name': 'PHP', 'description': 'PHP官方'},
{'type': 'java', 'name': 'Java JDK', 'description': 'Oracle/OpenJDK'},
{'type': 'dotnet', 'name': '.NET SDK', 'description': '.NET SDK'},
# ======== Linux 发行版 ========
{'type': 'alpine', 'name': 'Alpine Linux', 'description': 'Alpine Linux'},
{'type': 'arch', 'name': 'Arch Linux', 'description': 'Arch Linux'},
{'type': 'aur', 'name': 'AUR', 'description': 'Arch Linux AUR'},
{'type': 'centos', 'name': 'CentOS', 'description': 'CentOS'},
{'type': 'debian', 'name': 'Debian', 'description': 'Debian'},
{'type': 'fedora', 'name': 'Fedora', 'description': 'Fedora'},
{'type': 'gentoo', 'name': 'Gentoo', 'description': 'Gentoo'},
{'type': 'opensuse', 'name': 'openSUSE', 'description': 'openSUSE'},
{'type': 'void', 'name': 'Void Linux', 'description': 'Void Linux'},
{'type': 'freebsd', 'name': 'FreeBSD', 'description': 'FreeBSD'},
{'type': 'netbsd', 'name': 'NetBSD', 'description': 'NetBSD'},
{'type': 'openbsd', 'name': 'OpenBSD', 'description': 'OpenBSD'},
{'type': 'rocky', 'name': 'Rocky Linux', 'description': 'Rocky Linux'},
{'type': 'alma', 'name': 'AlmaLinux', 'description': 'AlmaLinux'},
{'type': 'kali', 'name': 'Kali Linux', 'description': 'Kali Linux'},
{'type': 'ubuntu', 'name': 'Ubuntu', 'description': 'Ubuntu'},
{'type': 'mint', 'name': 'Linux Mint', 'description': 'Linux Mint'},
{'type': 'manjaro', 'name': 'Manjaro', 'description': 'Manjaro Linux'},
{'type': 'slackware', 'name': 'Slackware', 'description': 'Slackware'},
{'type': 'mageia', 'name': 'Mageia', 'description': 'Mageia Linux'},
{'type': 'openmandriva', 'name': 'OpenMandriva', 'description': 'OpenMandriva'},
# ======== 系统包管理器 ========
{'type': 'homebrew', 'name': 'Homebrew', 'description': 'macOS Homebrew'},
{'type': 'brew', 'name': 'Brew', 'description': 'Homebrew'},
{'type': 'chocolatey', 'name': 'Chocolatey', 'description': 'Windows Chocolatey'},
{'type': 'snap', 'name': 'Snap Store', 'description': 'Snap应用商店'},
{'type': 'flatpak', 'name': 'Flatpak', 'description': 'Flatpak应用'},
{'type': 'appimage', 'name': 'AppImage', 'description': 'AppImage应用'},
{'type': 'winget', 'name': 'winget', 'description': 'Windows winget'},
{'type': 'scoop', 'name': 'Scoop', 'description': 'Windows Scoop'},
# ======== 数据库/运维 ========
{'type': 'postgresql', 'name': 'PostgreSQL', 'description': 'PostgreSQL数据库'},
{'type': 'mysql', 'name': 'MySQL', 'description': 'MySQL数据库'},
{'type': 'mariadb', 'name': 'MariaDB', 'description': 'MariaDB数据库'},
{'type': 'mongodb', 'name': 'MongoDB', 'description': 'MongoDB数据库'},
{'type': 'redis', 'name': 'Redis', 'description': 'Redis数据库'},
{'type': 'influxdata', 'name': 'InfluxData', 'description': 'InfluxDB时序数据库'},
{'type': 'grafana', 'name': 'Grafana', 'description': 'Grafana可视化'},
{'type': 'prometheus', 'name': 'Prometheus', 'description': 'Prometheus监控'},
{'type': 'elastic', 'name': 'Elastic', 'description': 'Elasticsearch'},
# ======== 代码托管 ========
{'type': 'github', 'name': 'GitHub', 'description': 'GitHub代码托管'},
{'type': 'gitlab', 'name': 'GitLab', 'description': 'GitLab代码托管'},
{'type': 'bitbucket', 'name': 'Bitbucket', 'description': 'Bitbucket代码托管'},
{'type': 'sourceforge', 'name': 'SourceForge', 'description': 'SourceForge开源托管'},
# ======== 其他 ========
{'type': 'nix', 'name': 'Nix/NixOS', 'description': 'Nix包管理器'},
{'type': 'guix', 'name': 'Guix', 'description': 'Guix包管理器'},
{'type': 'termux', 'name': 'Termux', 'description': 'Termux包管理器'},
{'type': 'msys2', 'name': 'MSYS2', 'description': 'MSYS2包管理器'},
{'type': 'google-fonts', 'name': 'Google Fonts', 'description': 'Google字体库'},
# ======== 自定义 ========
{'type': 'custom', 'name': '自定义HTTP', 'description': '自定义HTTP镜像源'}
]
+432
View File
@@ -0,0 +1,432 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
APT镜像代理处理器
支持Debian/Ubuntu软件源
"""
import os
import json
import time
import gzip
import re
import urllib.request
from typing import Dict, List, Optional, Tuple
from datetime import datetime
class APTMirror:
"""APT镜像代理"""
def __init__(self, config: dict):
self.config = config
# 配置 - 使用 storage_dir(基于 base_dir)
self.mirrors = config.get('mirrors', [
'http://archive.ubuntu.com/ubuntu',
'http://security.ubuntu.com/ubuntu'
])
self.storage_dir = config.get('storage_dir', './downloads/apt')
self.base_dir = config.get('base_dir', './downloads')
self.default_suite = config.get('suite', 'jammy')
self.default_components = config.get('components', ['main', 'restricted', 'universe', 'multiverse'])
self.default_arch = config.get('arch', 'amd64')
# 确保存储目录存在
os.makedirs(self.storage_dir, exist_ok=True)
def handle_request(self, handler, path: str) -> bool:
"""
处理APT请求
路径格式: /ubuntu/dists/jammy/main/binary-amd64/Packages.gz
"""
try:
# 解析路径
parts = path.strip('/').split('/')
if len(parts) < 5:
# 返回镜像列表或帮助信息
return self._handle_index(handler)
# 提取组件
distro = parts[0] # ubuntu, debian 等
dist_type = parts[1] # dists
suite = parts[2] # jammy, focal 等
component = parts[3] # main, updates 等
rest = '/'.join(parts[4:])
# 确定请求类型
if rest.endswith('Packages.gz'):
return self._handle_packages(handler, distro, suite, component, rest)
elif rest.endswith('Packages'):
return self._handle_packages_uncompressed(handler, distro, suite, component, rest)
elif rest.endswith('Release'):
return self._handle_release(handler, distro, suite, component, rest)
elif rest.endswith('Release.gpg'):
return self._handle_release_gpg(handler, distro, suite, component, rest)
elif rest.endswith('InRelease'):
return self._handle_inrelease(handler, distro, suite, component, rest)
else:
# 其他文件(源码包等)
return self._handle_file(handler, distro, suite, rest)
except Exception as e:
handler.send_error(500, str(e))
return False
def _handle_index(self, handler) -> bool:
"""处理索引请求"""
handler.send_json_response({
'mirrors': self.mirrors,
'default_suite': self.default_suite,
'default_components': self.default_components,
'cache_stats': self.get_cache_stats()
})
return True
def _handle_packages(self, handler, distro: str, suite: str, component: str, path: str) -> bool:
"""处理Packages.gz请求"""
cache_key = f"packages:{distro}:{suite}:{component}:{self.default_arch}"
# 检查架构
if 'binary-' in path:
arch = path.split('binary-')[1].split('/')[0]
else:
arch = self.default_arch
cache_key = f"packages:{distro}:{suite}:{component}:{arch}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
for mirror in self.mirrors:
url = f"{mirror}/{path}"
try:
data = self._fetch(url)
if data:
# 缓存
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
continue
handler.send_error(502, "Failed to fetch from all mirrors")
return False
def _handle_packages_uncompressed(self, handler, distro: str, suite: str, component: str, path: str) -> bool:
"""处理未压缩的Packages文件"""
# 先获取gz版本
gz_path = path + '.gz'
for mirror in self.mirrors:
url = f"{mirror}/{gz_path}"
try:
data = self._fetch(url)
if data:
# 解压
packages_data = gzip.decompress(data)
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain')
handler.send_header('Content-Length', str(len(packages_data)))
handler.end_headers()
handler.wfile.write(packages_data)
return True
except Exception:
continue
handler.send_error(502, "Failed to fetch packages")
return False
def _handle_release(self, handler, distro: str, suite: str, component: str, path: str) -> bool:
"""处理Release文件"""
cache_key = f"release:{distro}:{suite}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 获取Release文件
release_path = f"/{distro}/dists/{suite}/Release"
for mirror in self.mirrors:
url = mirror + release_path
try:
data = self._fetch(url)
if data:
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception:
continue
handler.send_error(502, "Failed to fetch Release")
return False
def _handle_release_gpg(self, handler, distro: str, suite: str, component: str, path: str) -> bool:
"""处理Release.gpg文件"""
cache_key = f"release_gpg:{distro}:{suite}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/pgp-signature')
handler.end_headers()
handler.wfile.write(cached)
return True
# 尝试获取
gpg_path = f"/{distro}/dists/{suite}/Release.gpg"
for mirror in self.mirrors:
url = mirror + gpg_path
try:
data = self._fetch(url)
if data:
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/pgp-signature')
handler.end_headers()
handler.wfile.write(data)
return True
except Exception:
continue
handler.send_error(404, "Release.gpg not found")
return False
def _handle_inrelease(self, handler, distro: str, suite: str, component: str, path: str) -> bool:
"""处理InRelease文件 - 获取或生成签名后的Release信息"""
cache_key = f"inrelease:{distro}:{suite}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 尝试从上游获取 InRelease
inrelease_path = f"/{distro}/dists/{suite}/InRelease"
for mirror in self.mirrors:
url = mirror + inrelease_path
try:
data = self._fetch(url)
if data:
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception:
continue
# 如果没有 InRelease,尝试生成一个(基于 Release + 方括号注释)
# 注意:这不是有效的签名,但可以用于不验证签名的客户端
release_cache_key = f"release:{distro}:{suite}"
release_data = self._get_cache(release_cache_key)
if not release_data:
# 尝试获取 Release
release_path = f"/{distro}/dists/{suite}/Release"
for mirror in self.mirrors:
url = mirror + release_path
try:
release_data = self._fetch(url)
if release_data:
break
except Exception:
continue
if release_data:
# 添加注释说明这是未签名的 Release
comment = f"# Note: This is a synthesized InRelease (original InRelease not available)\n"
inrelease_data = comment + release_data.decode('utf-8', errors='replace')
if self.cache_enabled:
self._set_cache(cache_key, inrelease_data.encode())
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain')
handler.send_header('Content-Length', str(len(inrelease_data)))
handler.end_headers()
handler.wfile.write(inrelease_data.encode())
return True
handler.send_error(502, "Failed to fetch InRelease")
return False
def _handle_file(self, handler, distro: str, suite: str, path: str) -> bool:
"""处理普通文件请求(如源码包)"""
cache_key = f"file:{distro}:{path.replace('/', ':')}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
for mirror in self.mirrors:
url = f"{mirror}/{path}"
try:
data = self._fetch(url)
if data:
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception:
continue
handler.send_error(404, "File not found")
return False
def _fetch(self, url: str) -> Optional[bytes]:
"""从URL获取数据"""
try:
req = urllib.request.Request(url)
req.add_header('User-Agent', 'APT-Mirror/1.0')
with urllib.request.urlopen(req, timeout=30) as response:
return response.read()
except Exception:
return None
def _get_cache(self, cache_key: str) -> Optional[bytes]:
"""获取缓存"""
if not self.cache_enabled:
return None
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
if not os.path.exists(cache_path):
return None
# 检查过期
if os.path.exists(meta_path):
try:
with open(meta_path, 'r') as f:
meta = json.load(f)
if time.time() > meta.get('expires', 0):
return None
except Exception:
pass
try:
with open(cache_path, 'rb') as f:
return f.read()
except Exception:
return None
def _set_cache(self, cache_key: str, data: bytes):
"""设置缓存"""
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
try:
with open(cache_path, 'wb') as f:
f.write(data)
meta = {
'cached_at': time.time(),
'expires': time.time() + self.cache_ttl,
'size': len(data)
}
with open(meta_path, 'w') as f:
json.dump(meta, f)
except Exception as e:
print(f"APT缓存写入失败: {e}")
def _get_cache_path(self, cache_key: str) -> str:
"""获取缓存路径"""
subdir = cache_key[:2]
return os.path.join(self.storage_dir, subdir, cache_key)
def get_cache_stats(self) -> dict:
"""获取缓存统计"""
if not os.path.exists(self.storage_dir):
return {'files': 0, 'size': 0}
total_size = 0
file_count = 0
for root, dirs, files in os.walk(self.storage_dir):
for f in files:
if not f.endswith('.meta'):
file_count += 1
total_size += os.path.getsize(os.path.join(root, f))
return {
'files': file_count,
'size': total_size,
'size_formatted': self._format_size(total_size)
}
def _format_size(self, size_bytes: int) -> str:
"""格式化文件大小"""
if size_bytes == 0:
return "0 B"
units = ["B", "KB", "MB", "GB"]
i = 0
while size_bytes >= 1024 and i < len(units) - 1:
size_bytes /= 1024.0
i += 1
return f"{size_bytes:.2f} {units[i]}"
+359
View File
@@ -0,0 +1,359 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Docker镜像代理处理器
支持Docker Registry API v2
"""
import os
import json
import time
import uuid
import urllib.request
import base64
import hashlib
import hmac
from typing import Dict, List, Optional, Tuple
from datetime import datetime
class DockerMirror:
"""Docker镜像代理"""
def __init__(self, config: dict):
self.config = config
# 配置 - 使用 storage_dir(基于 base_dir)
self.registry_url = config.get('registry_url', 'https://registry-1.docker.io')
self.mirror_url = config.get('mirror_url', '')
self.storage_dir = config.get('storage_dir', './downloads/docker')
self.base_dir = config.get('base_dir', './downloads')
# 认证(可选)
self.username = config.get('username')
self.password = config.get('password')
# 确保存储目录存在
os.makedirs(self.storage_dir, exist_ok=True)
def handle_request(self, handler, path: str) -> bool:
"""
处理Docker镜像请求
路径格式: /v2/library/ubuntu/tags/list 或 /v2/library/ubuntu/manifests/latest
"""
try:
# 解析路径
parts = path.strip('/').split('/')
if len(parts) < 2 or parts[0] != 'v2':
handler.send_error(400, "Invalid Docker API path")
return False
# 提取组件
if parts[1] == 'library':
# 官方镜像
image = 'library/' + '/'.join(parts[2:-2]) if len(parts) > 4 else 'library/' + parts[2]
action = parts[-2] # tags 或 manifests
reference = parts[-1]
else:
# 非官方镜像
image = '/'.join(parts[1:-2])
action = parts[-2]
reference = parts[-1]
# 根据操作类型处理
if action == 'tags' and reference == 'list':
return self._handle_tag_list(handler, image.rstrip('/tags'))
elif action == 'manifests':
return self._handle_manifest(handler, image, reference)
elif action == 'blobs':
return self._handle_blob(handler, image, reference)
elif action == 'token':
return self._handle_token(handler)
else:
handler.send_error(404, "Unknown action")
return False
except Exception as e:
handler.send_error(500, str(e))
return False
def _handle_tag_list(self, handler, image: str) -> bool:
"""处理标签列表请求"""
cache_key = f"tags:{image}"
cached = self._get_cache(handler, cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
url = f"{self.registry_url}/v2/{image}/tags/list"
try:
data = self._fetch_from_upstream(url)
# 缓存
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(500, f"Failed to fetch tags: {str(e)}")
return False
def _handle_manifest(self, handler, image: str, reference: str) -> bool:
"""处理清单请求"""
cache_key = f"manifest:{image}:{reference}"
cached = self._get_cache(handler, cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/vnd.docker.distribution.manifest.v2+json')
handler.send_header('Content-Length', str(len(cached)))
handler.send_header('Docker-Content-Digest', f"sha256:{hashlib.sha256(cached).hexdigest()}")
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
url = f"{self.registry_url}/v2/{image}/manifests/{reference}"
try:
req = urllib.request.Request(url)
req.add_header('Accept', 'application/vnd.docker.distribution.manifest.v2+json')
if self.username and self.password:
auth = base64.b64encode(f"{self.username}:{self.password}".encode()).decode()
req.add_header('Authorization', f"Basic {auth}")
with urllib.request.urlopen(req) as response:
data = response.read()
# 缓存
if self.cache_enabled:
self._set_cache(cache_key, data)
digest = f"sha256:{hashlib.sha256(data).hexdigest()}"
handler.send_response(200)
handler.send_header('Content-Type', 'application/vnd.docker.distribution.manifest.v2+json')
handler.send_header('Content-Length', str(len(data)))
handler.send_header('Docker-Content-Digest', digest)
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(500, f"Failed to fetch manifest: {str(e)}")
return False
def _handle_blob(self, handler, image: str, digest: str) -> bool:
"""处理Blob层下载"""
# 移除 sha256: 前缀
if digest.startswith('sha256:'):
digest = digest[7:]
cache_key = f"blob:{digest}"
cached = self._get_cache(handler, cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(cached)))
handler.send_header('Docker-Content-Digest', f"sha256:{digest}")
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
url = f"{self.registry_url}/v2/{image}/blobs/sha256:{digest}"
try:
req = urllib.request.Request(url)
if self.username and self.password:
auth = base64.b64encode(f"{self.username}:{self.password}".encode()).decode()
req.add_header('Authorization', f"Basic {auth}")
with urllib.request.urlopen(req) as response:
data = response.read()
# 缓存
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(data)))
handler.send_header('Docker-Content-Digest', f"sha256:{digest}")
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(500, f"Failed to fetch blob: {str(e)}")
return False
def _handle_token(self, handler) -> bool:
"""处理Token请求 - 生成真实的访问令牌"""
# 解析认证信息
auth_header = handler.headers.get('Authorization', '')
username = None
password = None
if auth_header.startswith('Basic '):
try:
decoded = base64.b64decode(auth_header[6:]).decode('utf-8')
username, password = decoded.split(':', 1)
except Exception:
pass
# 验证凭据(如果有)
if self.username and self.password:
if username != self.username or password != self.password:
handler.send_error(401, "Invalid credentials")
return False
# 生成唯一的访问令牌
token_id = str(uuid.uuid4())
issued_at = int(time.time())
expires_in = 300 # 5分钟
expires_at = issued_at + expires_in
# 创建令牌信息(简化版 JWT 结构)
token_data = {
"iss": "hyc-mirror",
"sub": username or "anonymous",
"aud": self.registry_url,
"iat": issued_at,
"exp": expires_at,
"access": [
{"type": "repository", "actions": ["pull"]},
{"type": "registry", "actions": ["catalog"]}
]
}
# 使用 HMAC-SHA256 对令牌进行简单签名
secret_key = f"hyc-mirror-{self.registry_url}".encode()
signature = hmac.new(
secret_key,
f"{token_id}:{issued_at}".encode(),
hashlib.sha256
).hexdigest()[:32]
full_token = f"{token_id}-{signature}"
handler.send_json_response({
"token": full_token,
"expires_in": expires_in,
"issued_at": issued_at
})
return True
def _fetch_from_upstream(self, url: str) -> bytes:
"""从上游获取数据"""
req = urllib.request.Request(url)
if self.username and self.password:
auth = base64.b64encode(f"{self.username}:{self.password}".encode()).decode()
req.add_header('Authorization', f"Basic {auth}")
with urllib.request.urlopen(req, timeout=30) as response:
return response.read()
def _get_cache(self, handler, cache_key: str) -> Optional[bytes]:
"""获取缓存"""
if not self.cache_enabled:
return None
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
if not os.path.exists(cache_path):
return None
# 检查是否过期
if os.path.exists(meta_path):
try:
with open(meta_path, 'r') as f:
meta = json.load(f)
if time.time() > meta.get('expires', 0):
return None
except Exception:
pass
try:
with open(cache_path, 'rb') as f:
return f.read()
except Exception:
return None
def _set_cache(self, cache_key: str, data: bytes):
"""设置缓存"""
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
try:
with open(cache_path, 'wb') as f:
f.write(data)
meta = {
'cached_at': time.time(),
'expires': time.time() + self.cache_ttl,
'size': len(data)
}
with open(meta_path, 'w') as f:
json.dump(meta, f)
except Exception as e:
print(f"Docker缓存写入失败: {e}")
def _get_cache_path(self, cache_key: str) -> str:
"""获取缓存路径"""
subdir = cache_key[:2]
return os.path.join(self.storage_dir, subdir, cache_key)
def get_cache_stats(self) -> dict:
"""获取缓存统计"""
if not os.path.exists(self.storage_dir):
return {'files': 0, 'size': 0}
total_size = 0
file_count = 0
for root, dirs, files in os.walk(self.storage_dir):
for f in files:
if not f.endswith('.meta'):
file_count += 1
total_size += os.path.getsize(os.path.join(root, f))
return {
'files': file_count,
'size': total_size,
'size_formatted': self._format_size(total_size)
}
def _format_size(self, size_bytes: int) -> str:
"""格式化文件大小"""
if size_bytes == 0:
return "0 B"
units = ["B", "KB", "MB", "GB"]
i = 0
while size_bytes >= 1024 and i < len(units) - 1:
size_bytes /= 1024.0
i += 1
return f"{size_bytes:.2f} {units[i]}"
+525
View File
@@ -0,0 +1,525 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Go模块代理处理器
支持Go模块代理协议
"""
import os
import json
import time
import urllib.request
import urllib.parse
import urllib.error
from typing import Dict, List, Optional
from datetime import datetime
class GoProxy:
"""Go模块代理"""
def __init__(self, config: dict):
self.config = config
# 配置 - 使用 storage_dir(基于 base_dir)
self.upstream_url = config.get('upstream_url', 'https://proxy.golang.org')
self.storage_dir = config.get('storage_dir', './downloads/go')
self.base_dir = config.get('base_dir', './downloads')
self.mode = config.get('mode', 'proxy') # proxy | direct
# 确保存储目录存在
os.makedirs(self.storage_dir, exist_ok=True)
def handle_request(self, handler, path: str) -> bool:
"""
处理Go模块请求
路径格式: /golang.org/x/net/@v/list
/golang.org/x/net/@v/v1.0.0.info
/golang.org/x/net/@v/v1.0.0.zip
/golang.org/x/net/@latest
"""
try:
parts = path.strip('/').split('/')
if len(parts) < 2:
return self._handle_index(handler)
# 解析模块路径和操作
module_parts = []
for i, part in enumerate(parts):
if part.startswith('@'):
# 找到操作部分
module_path = '/'.join(parts[:i])
action = parts[i:]
break
else:
# 没有找到操作符
module_path = '/'.join(parts)
action = []
if not action:
handler.send_error(400, "Invalid Go module path")
return False
action_type = action[0]
if action_type == '@v':
# 版本相关操作
if len(action) >= 3:
version = action[2]
return self._handle_version(handler, module_path, version)
elif len(action) == 2:
# /@v/list
return self._handle_version_list(handler, module_path)
elif action_type == '@latest':
# /@latest
return self._handle_latest(handler, module_path)
elif action_type == '@all':
# /@all
return self._handle_all(handler, module_path)
elif action_type == '@list':
# /@list
return self._handle_module_list(handler, module_path)
else:
handler.send_error(400, f"Unknown action: {action_type}")
return False
except Exception as e:
handler.send_error(500, str(e))
return False
def _handle_index(self, handler) -> bool:
"""处理索引请求"""
handler.send_json_response({
'proxy_url': self.upstream_url,
'mode': self.mode,
'cache_stats': self.get_cache_stats()
})
return True
def _handle_version_list(self, handler, module: str) -> bool:
"""处理版本列表请求 /@v/list"""
cache_key = f"vlist:{module}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
url = f"{self.upstream_url}/{module}/@v/list"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
if e.code == 404:
handler.send_error(404, f"Module not found: {module}")
else:
handler.send_error(502, f"Failed to fetch: {str(e)}")
return False
def _handle_version_info(self, handler, module: str, version: str) -> bool:
"""处理版本信息请求 /@v/version.info"""
cache_key = f"info:{module}:{version}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.end_headers()
handler.wfile.write(cached)
return True
url = f"{self.upstream_url}/{module}/@v/{version}.info"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
handler.send_error(502, f"Failed to fetch info: {str(e)}")
return False
def _handle_version(self, handler, module: str, suffix: str) -> bool:
"""处理版本相关请求"""
if suffix.endswith('.info'):
version = suffix[:-5]
return self._handle_version_info(handler, module, version)
elif suffix.endswith('.zip'):
version = suffix[:-4]
return self._handle_zip(handler, module, version)
elif suffix.endswith('.mod'):
version = suffix[:-4]
return self._handle_mod(handler, module, version)
elif suffix.endswith('.sum'):
version = suffix[:-4]
return self._handle_sum(handler, module, version)
else:
handler.send_error(400, f"Unknown suffix: {suffix}")
return False
def _handle_latest(self, handler, module: str) -> bool:
"""处理最新版本请求 /@latest"""
cache_key = f"latest:{module}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.end_headers()
handler.wfile.write(cached)
return True
url = f"{self.upstream_url}/{module}/@latest"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
handler.send_error(502, f"Failed to fetch latest: {str(e)}")
return False
def _handle_zip(self, handler, module: str, version: str) -> bool:
"""处理zip下载"""
cache_key = f"zip:{module}:{version}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/zip')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
url = f"{self.upstream_url}/{module}/@v/{version}.zip"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/zip')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
handler.send_error(502, f"Failed to fetch zip: {str(e)}")
return False
def _handle_mod(self, handler, module: str, version: str) -> bool:
"""处理mod文件"""
cache_key = f"mod:{module}:{version}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
url = f"{self.upstream_url}/{module}/@v/{version}.mod"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
handler.send_error(502, f"Failed to fetch mod: {str(e)}")
return False
def _handle_sum(self, handler, module: str, version: str) -> bool:
"""处理sum文件"""
cache_key = f"sum:{module}:{version}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.end_headers()
handler.wfile.write(cached)
return True
url = f"{self.upstream_url}/{module}/@v/{version}.sum"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
if e.code == 404:
# 没有sum文件时返回空
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.end_headers()
handler.wfile.write(b'')
else:
handler.send_error(502, f"Failed to fetch sum: {str(e)}")
return False
def _handle_all(self, handler, module: str) -> bool:
"""处理/@all请求 - 返回模块及其所有依赖的zip包"""
cache_key = f"all:{module}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/zip')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取所有依赖的zip
url = f"{self.upstream_url}/{module}/@all.zip"
try:
data = self._fetch(url)
if not data:
handler.send_error(404, f"Module not found: {module}")
return False
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/zip')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
if e.code == 404:
handler.send_error(404, f"Module not found: {module}")
else:
handler.send_error(502, f"Failed to fetch @all: {str(e)}")
return False
def _handle_module_list(self, handler, module: str) -> bool:
"""处理/@list请求 - 返回模块及其依赖的路径列表"""
cache_key = f"list:{module}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.end_headers()
handler.wfile.write(cached)
return True
# 首先获取模块的 go.mod 文件以提取依赖
mod_url = f"{self.upstream_url}/{module}/@v/{module}.mod"
try:
mod_data = self._fetch(mod_url)
if not mod_data:
handler.send_error(404, f"Module not found: {module}")
return False
# 解析go.mod获取依赖
modules = [module]
mod_content = mod_data.decode('utf-8', errors='replace')
# 提取require语句中的依赖
import re
require_pattern = r'require\s+\(([^\)]+)\)'
inline_require_pattern = r'require\s+([^\s]+)\s+([^\s]+)'
# 处理多行require
matches = re.findall(require_pattern, mod_content, re.DOTALL)
for match in matches:
for line in match.strip().split('\n'):
line = line.strip()
if line and not line.startswith('//'):
parts = line.split()
if parts:
modules.append(parts[0])
# 处理单行require
matches = re.findall(inline_require_pattern, mod_content)
for match in matches:
if match[0] not in modules:
modules.append(match[0])
# 生成列表输出
list_output = '\n'.join(sorted(set(modules))) + '\n'
if self.cache_enabled:
self._set_cache(cache_key, list_output.encode())
handler.send_response(200)
handler.send_header('Content-Type', 'text/plain; charset=utf-8')
handler.end_headers()
handler.wfile.write(list_output.encode())
return True
except urllib.error.HTTPError as e:
if e.code == 404:
handler.send_error(404, f"Module not found: {module}")
else:
handler.send_error(502, f"Failed to fetch @list: {str(e)}")
return False
def _fetch(self, url: str) -> Optional[bytes]:
"""从URL获取数据"""
try:
req = urllib.request.Request(url)
req.add_header('User-Agent', 'Go-Mirror/1.0')
with urllib.request.urlopen(req, timeout=60) as response:
return response.read()
except Exception:
return None
def _get_cache(self, cache_key: str) -> Optional[bytes]:
"""获取缓存"""
if not self.cache_enabled:
return None
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
if not os.path.exists(cache_path):
return None
if os.path.exists(meta_path):
try:
with open(meta_path, 'r') as f:
meta = json.load(f)
if time.time() > meta.get('expires', 0):
return None
except Exception:
pass
try:
with open(cache_path, 'rb') as f:
return f.read()
except Exception:
return None
def _set_cache(self, cache_key: str, data: bytes):
"""设置缓存"""
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
try:
with open(cache_path, 'wb') as f:
f.write(data)
meta = {
'cached_at': time.time(),
'expires': time.time() + self.cache_ttl,
'size': len(data)
}
with open(meta_path, 'w') as f:
json.dump(meta, f)
except Exception as e:
print(f"Go缓存写入失败: {e}")
def _get_cache_path(self, cache_key: str) -> str:
"""获取缓存路径"""
subdir = cache_key[:2]
return os.path.join(self.storage_dir, subdir, cache_key)
def get_cache_stats(self) -> dict:
"""获取缓存统计"""
if not os.path.exists(self.storage_dir):
return {'files': 0, 'size': 0}
total_size = 0
file_count = 0
for root, dirs, files in os.walk(self.storage_dir):
for f in files:
if not f.endswith('.meta'):
file_count += 1
total_size += os.path.getsize(os.path.join(root, f))
return {
'files': file_count,
'size': total_size,
'size_formatted': self._format_size(total_size)
}
def _format_size(self, size_bytes: int) -> str:
"""格式化文件大小"""
if size_bytes == 0:
return "0 B"
units = ["B", "KB", "MB", "GB"]
i = 0
while size_bytes >= 1024 and i < len(units) - 1:
size_bytes /= 1024.0
i += 1
return f"{size_bytes:.2f} {units[i]}"
+485
View File
@@ -0,0 +1,485 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
通用HTTP镜像代理处理器
支持Maven、Gradle、RubyGems、Cargo、NuGet、CocoaPods、CRAN、CTAN、CUDA、Pacman等基于HTTP的镜像
"""
import os
import re
import json
import time
import urllib.request
import urllib.parse
import urllib.error
from typing import Dict, List, Optional
from datetime import datetime
from html import escape
class HttpMirror:
"""通用HTTP镜像代理 - 支持多种包管理器"""
# 不同包管理器的默认上游URL
DEFAULT_UPSTREAM = {
'maven': 'https://repo1.maven.org/maven2',
'gradle': 'https://services.gradle.org/distributions',
'gem': 'https://rubygems.org',
'cargo': 'https://crates.io',
'nuget': 'https://api.nuget.org/v3',
'cocoapods': 'https://cdn.cocoapods.org',
'cran': 'https://cran.r-project.org',
'ctan': 'https://ctan.math.illinois.edu',
'cuda': 'https://developer.download.nvidia.com/compute/cuda/repos',
'pacman': 'https://mirror.archlinux.org',
}
def __init__(self, config: dict):
self.config = config
self.mirror_type = config.get('type', 'http')
# 配置
self.upstream_url = config.get('upstream_url', self.DEFAULT_UPSTREAM.get(self.mirror_type, 'https://mirror.example.com'))
self.storage_dir = config.get('storage_dir', f'./downloads/{self.mirror_type}')
self.base_dir = config.get('base_dir', './downloads')
self.cache_enabled = config.get('cache_enabled', True)
self.cache_ttl = config.get('cache_ttl', 3600) # 默认1小时
# 确保存储目录存在
os.makedirs(self.storage_dir, exist_ok=True)
def handle_request(self, handler, path: str) -> bool:
"""处理HTTP镜像请求"""
try:
# 移除前导斜杠
path = path.strip('/')
if not path or path == self.mirror_type:
return self._handle_index(handler)
# 根据镜像类型分发请求
if self.mirror_type == 'maven':
return self._handle_maven(handler, path)
elif self.mirror_type == 'gradle':
return self._handle_gradle(handler, path)
elif self.mirror_type == 'gem':
return self._handle_gem(handler, path)
elif self.mirror_type == 'cargo':
return self._handle_cargo(handler, path)
elif self.mirror_type == 'nuget':
return self._handle_nuget(handler, path)
elif self.mirror_type == 'cocoapods':
return self._handle_cocoapods(handler, path)
elif self.mirror_type == 'cran':
return self._handle_cran(handler, path)
elif self.mirror_type == 'ctan':
return self._handle_ctan(handler, path)
elif self.mirror_type == 'cuda':
return self._handle_cuda(handler, path)
elif self.mirror_type == 'pacman':
return self._handle_pacman(handler, path)
else:
# 默认作为普通HTTP文件处理
return self._handle_generic(handler, path)
except Exception as e:
handler.send_error(500, str(e))
return False
def _handle_index(self, handler) -> bool:
"""返回镜像索引信息"""
handler.send_json_response({
'type': self.mirror_type,
'upstream_url': self.upstream_url,
'storage_dir': self.storage_dir,
'cache_enabled': self.cache_enabled,
'cache_stats': self.get_cache_stats()
})
return True
# ========== Maven ==========
def _handle_maven(self, handler, path: str) -> bool:
"""处理Maven请求: /groupId/artifactId/version/artifactId-version.jar"""
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== Gradle ==========
def _handle_gradle(self, handler, path: str) -> bool:
"""处理Gradle请求: /gradle-x.x-bin.zip"""
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== RubyGems ==========
def _handle_gem(self, handler, path: str) -> bool:
"""处理RubyGems请求"""
# API请求: /api/v1/gems/xxx.json
# Spec请求: /quick/Marshal.4.8/xxx-x.x.0.gemspec.rz
# 下载请求: /gems/xxx-x.x.0.gem
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== Cargo ==========
def _handle_cargo(self, handler, path: str) -> bool:
"""处理Cargo请求"""
# API: /api/v1/crates
# 下载: /crates/xxx/xxx-x.x.x.crate
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== NuGet ==========
def _handle_nuget(self, handler, path: str) -> bool:
"""处理NuGet请求"""
# API v3: /v3-flatcontainer/xxx.nupkg
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== CocoaPods ==========
def _handle_cocoapods(self, handler, path: str) -> bool:
"""处理CocoaPods请求"""
# Specs: /Specs/xxx.podspec.json
# Pods: /Pods/xxx/xxx.pod
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== CRAN ==========
def _handle_cran(self, handler, path: str) -> bool:
"""处理CRAN请求"""
# /src/contrib/xxx.tar.gz
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== CTAN ==========
def _handle_ctan(self, handler, path: str) -> bool:
"""处理CTAN请求"""
# /macros/latex/xxx.zip
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== CUDA ==========
def _handle_cuda(self, handler, path: str) -> bool:
"""处理CUDA请求"""
# /xxx/xxx.deb
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== Pacman ==========
def _handle_pacman(self, handler, path: str) -> bool:
"""处理Pacman请求"""
# /os/x86_64/xxx.db.tar.gz
# /os/x86_64/xxx-x.x.x-x-x86_64.pkg.tar.zst
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== Generic HTTP ==========
def _handle_generic(self, handler, path: str) -> bool:
"""通用的HTTP文件代理"""
# 先检查本地
local_path = os.path.join(self.storage_dir, path)
if os.path.exists(local_path) and os.path.isfile(local_path):
return self._serve_local_file(handler, local_path)
# 从上游获取
url = f"{self.upstream_url}/{path}"
return self._proxy_request(handler, url)
# ========== 辅助方法 ==========
def _serve_local_file(self, handler, local_path: str) -> bool:
"""服务本地文件"""
try:
file_size = os.path.getsize(local_path)
# 尝试确定内容类型
content_type = self._guess_content_type(local_path)
handler.send_response(200)
handler.send_header('Content-Type', content_type)
handler.send_header('Content-Length', str(file_size))
handler.send_header('X-Local', 'true')
handler.end_headers()
with open(local_path, 'rb') as f:
handler.wfile.write(f.read())
return True
except Exception as e:
handler.send_error(500, f"Error serving file: {str(e)}")
return False
def _proxy_request(self, handler, url: str) -> bool:
"""代理请求到上游"""
try:
# 检查缓存
cache_key = self._get_cache_key(url)
if self.cache_enabled:
cached = self._get_cache(cache_key)
if cached:
return self._serve_cached(handler, cached, url)
# 从上游获取
req = urllib.request.Request(url)
req.add_header('User-Agent', f'HTTP-Mirror/1.0 ({self.mirror_type})')
# 处理Range请求
range_header = handler.headers.get('Range')
if range_header:
req.add_header('Range', range_header)
try:
response = urllib.request.urlopen(req, timeout=60)
except urllib.error.HTTPError as e:
if e.code == 404:
handler.send_error(404, f"File not found: {url}")
else:
handler.send_error(502, f"Upstream error: {str(e)}")
return False
except Exception as e:
handler.send_error(502, f"Failed to connect upstream: {str(e)}")
return False
# 获取响应头
content_type = response.headers.get('Content-Type', 'application/octet-stream')
content_length = response.headers.get('Content-Length')
content_range = response.headers.get('Content-Range')
# 处理部分内容响应
if content_range:
handler.send_response(206)
handler.send_header('Content-Range', content_range)
else:
handler.send_response(200)
handler.send_header('Content-Type', content_type)
if content_length:
handler.send_header('Content-Length', content_length)
handler.send_header('X-Upstream', self.upstream_url)
handler.end_headers()
# 读取并转发内容,同时缓存
data = response.read()
if self.cache_enabled and response.status == 200:
self._set_cache(cache_key, data, content_type)
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(502, f"Proxy error: {str(e)}")
return False
def _serve_cached(self, handler, cached: dict, url: str):
"""服务缓存的文件"""
handler.send_response(200)
handler.send_header('Content-Type', cached.get('content_type', 'application/octet-stream'))
handler.send_header('Content-Length', len(cached.get('data', b'')))
handler.send_header('X-Cached', 'true')
handler.end_headers()
handler.wfile.write(cached.get('data', b''))
return True
def _get_cache_key(self, url: str) -> str:
"""生成缓存键"""
return urllib.parse.quote(url, safe='')
def _get_cache_path(self, cache_key: str) -> str:
"""获取缓存文件路径"""
# 使用前两个字符作为子目录
subdir = cache_key[:2] if len(cache_key) >= 2 else 'cache'
return os.path.join(self.storage_dir, '.cache', subdir, cache_key)
def _get_cache(self, cache_key: str) -> Optional[dict]:
"""获取缓存"""
if not self.cache_enabled:
return None
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
if not os.path.exists(cache_path):
return None
# 检查是否过期
if os.path.exists(meta_path):
try:
with open(meta_path, 'r') as f:
meta = json.load(f)
if time.time() > meta.get('expires', 0):
return None
except Exception:
pass
# 读取缓存数据
try:
with open(cache_path, 'rb') as f:
data = f.read()
with open(meta_path, 'r') as f:
meta = json.load(f)
return {
'data': data,
'content_type': meta.get('content_type', 'application/octet-stream')
}
except Exception:
return None
def _set_cache(self, cache_key: str, data: bytes, content_type: str = 'application/octet-stream'):
"""设置缓存"""
try:
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
# 写入数据
with open(cache_path, 'wb') as f:
f.write(data)
# 写入元数据
meta = {
'cached_at': time.time(),
'expires': time.time() + self.cache_ttl,
'size': len(data),
'content_type': content_type,
'url': cache_key
}
with open(meta_path, 'w') as f:
json.dump(meta, f)
except Exception as e:
print(f"[{self.mirror_type}] Cache write failed: {e}")
def _guess_content_type(self, path: str) -> str:
"""根据文件扩展名猜测内容类型"""
ext = os.path.splitext(path)[1].lower()
content_types = {
'.jar': 'application/java-archive',
'.war': 'application/java-archive',
'.ear': 'application/java-archive',
'.pom': 'application/xml',
'.xml': 'application/xml',
'.json': 'application/json',
'.gem': 'application/octet-stream',
'.crate': 'application/octet-stream',
'.nupkg': 'application/zip',
'.tar.gz': 'application/gzip',
'.tgz': 'application/gzip',
'.zip': 'application/zip',
'.deb': 'application/deb',
'.rpm': 'application/x-rpm',
'.pkg.tar.zst': 'application/zstd',
'.podspec': 'text/plain',
'.tar': 'application/x-tar',
'.pdf': 'application/pdf',
'.html': 'text/html',
'.css': 'text/css',
'.js': 'application/javascript',
}
return content_types.get(ext, 'application/octet-stream')
def get_cache_stats(self) -> dict:
"""获取缓存统计"""
cache_dir = os.path.join(self.storage_dir, '.cache')
if not os.path.exists(cache_dir):
return {'files': 0, 'size': 0, 'size_formatted': '0 B'}
total_size = 0
file_count = 0
try:
for root, dirs, files in os.walk(cache_dir):
for f in files:
if not f.endswith('.meta'):
file_count += 1
total_size += os.path.getsize(os.path.join(root, f))
except Exception:
pass
return {
'files': file_count,
'size': total_size,
'size_formatted': self._format_size(total_size)
}
def _format_size(self, size_bytes: int) -> str:
"""格式化文件大小"""
if size_bytes == 0:
return "0 B"
units = ["B", "KB", "MB", "GB", "TB"]
i = 0
size = float(size_bytes)
while size >= 1024 and i < len(units) - 1:
size /= 1024.0
i += 1
return f"{size:.2f} {units[i]}"
+280
View File
@@ -0,0 +1,280 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
npm镜像代理处理器
支持Node.js包管理器
"""
import os
import json
import time
import urllib.request
import urllib.parse
from typing import Dict, List, Optional
from datetime import datetime
class NpmMirror:
"""npm镜像代理"""
def __init__(self, config: dict):
self.config = config
# 配置 - 使用 storage_dir(基于 base_dir)
self.upstream_url = config.get('upstream_url', 'https://registry.npmjs.org')
self.storage_dir = config.get('storage_dir', './downloads/npm')
self.base_dir = config.get('base_dir', './downloads')
# 确保存储目录存在
os.makedirs(self.storage_dir, exist_ok=True)
def handle_request(self, handler, path: str) -> bool:
"""
处理npm请求
路径格式: /lodash 或 /-/package/lodash/dist
"""
try:
parts = path.strip('/').split('/')
if not parts:
return self._handle_index(handler)
if parts[0] == '-':
# Scoped package 或其他特殊请求
if len(parts) >= 4 and parts[1] == 'package':
return self._handleScopedPackage(handler, parts[2], parts[3] if len(parts) > 3 else None)
elif len(parts) >= 3 and parts[1] == 'package':
return self._handle_package(handler, parts[2], None)
else:
handler.send_error(400, "Invalid npm API path")
return False
elif parts[0] == '@':
# Scoped package
if len(parts) >= 2:
scope = parts[0]
package = '/'.join(parts[1:])
return self._handle_scoped_package(handler, scope, package)
else:
handler.send_error(400, "Invalid scoped package")
return False
elif parts[0] == '-/':
# npm特殊路径
return self._handle_special(handler, '/'.join(parts))
elif len(parts) == 1:
# 单个包名
return self._handle_package(handler, parts[0], None)
else:
# 其他请求
return self._handle_package(handler, parts[0], parts[1] if len(parts) > 1 else None)
except Exception as e:
handler.send_error(500, str(e))
return False
def _handle_index(self, handler) -> bool:
"""处理索引请求"""
handler.send_json_response({
'registry_url': self.upstream_url,
'cache_stats': self.get_cache_stats()
})
return True
def _handle_package(self, handler, package: str, version: str = None) -> bool:
"""处理包元数据请求"""
cache_key = f"package:{package}:{version or 'latest'}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
if version:
url = f"{self.upstream_url}/{package}/{version}"
else:
url = f"{self.upstream_url}/{package}/latest"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.end_headers()
handler.wfile.write(data)
return True
except urllib.error.HTTPError as e:
handler.send_error(404, f"Package not found: {package}")
return False
def _handle_scoped_package(self, handler, scope: str, package: str) -> bool:
"""处理scoped包"""
full_name = f"{scope}/{package}"
return self._handle_package(handler, full_name, None)
def _handleScopedPackage(self, handler, scope: str, package: str) -> bool:
"""处理特殊路径的scoped包"""
full_name = f"{scope}/{package}"
return self._handle_package(handler, full_name, None)
def _handle_special(self, handler, path: str) -> bool:
"""处理特殊npm路径"""
# 简化实现:转发到上游
url = f"{self.upstream_url}/{path}"
try:
data = self._fetch(url)
handler.send_response(200)
handler.send_header('Content-Type', 'application/json')
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(502, f"Failed to fetch: {str(e)}")
return False
def _handle_tarball(self, handler, package: str, filename: str) -> bool:
"""处理tarball下载"""
cache_key = f"tarball:{package}:{filename}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
url = f"{self.upstream_url}/{package}/-/{filename}"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(502, f"Failed to fetch tarball: {str(e)}")
return False
def _fetch(self, url: str) -> Optional[bytes]:
"""从URL获取数据"""
try:
req = urllib.request.Request(url)
req.add_header('User-Agent', 'npm-Mirror/1.0')
req.add_header('Accept', 'application/json')
with urllib.request.urlopen(req, timeout=30) as response:
return response.read()
except Exception:
return None
def _get_cache(self, cache_key: str) -> Optional[bytes]:
"""获取缓存"""
if not self.cache_enabled:
return None
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
if not os.path.exists(cache_path):
return None
if os.path.exists(meta_path):
try:
with open(meta_path, 'r') as f:
meta = json.load(f)
if time.time() > meta.get('expires', 0):
return None
except Exception:
pass
try:
with open(cache_path, 'rb') as f:
return f.read()
except Exception:
return None
def _set_cache(self, cache_key: str, data: bytes):
"""设置缓存"""
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
try:
with open(cache_path, 'wb') as f:
f.write(data)
meta = {
'cached_at': time.time(),
'expires': time.time() + self.cache_ttl,
'size': len(data)
}
with open(meta_path, 'w') as f:
json.dump(meta, f)
except Exception as e:
print(f"npm缓存写入失败: {e}")
def _get_cache_path(self, cache_key: str) -> str:
"""获取缓存路径"""
subdir = cache_key[:2]
return os.path.join(self.storage_dir, subdir, cache_key)
def get_cache_stats(self) -> dict:
"""获取缓存统计"""
if not os.path.exists(self.storage_dir):
return {'files': 0, 'size': 0}
total_size = 0
file_count = 0
for root, dirs, files in os.walk(self.storage_dir):
for f in files:
if not f.endswith('.meta'):
file_count += 1
total_size += os.path.getsize(os.path.join(root, f))
return {
'files': file_count,
'size': total_size,
'size_formatted': self._format_size(total_size)
}
def _format_size(self, size_bytes: int) -> str:
"""格式化文件大小"""
if size_bytes == 0:
return "0 B"
units = ["B", "KB", "MB", "GB"]
i = 0
while size_bytes >= 1024 and i < len(units) - 1:
size_bytes /= 1024.0
i += 1
return f"{size_bytes:.2f} {units[i]}"
+788
View File
@@ -0,0 +1,788 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
PyPI镜像代理处理器
支持Python包索引
"""
import os
import json
import re
import time
import urllib.request
import urllib.parse
import urllib.error
from typing import Dict, List, Optional
from datetime import datetime
class PyPIMirror:
"""PyPI镜像代理"""
def __init__(self, config: dict):
self.config = config
# 配置 - storage_dir 基于 base_dir
self.upstream_url = config.get('upstream_url', 'https://pypi.org')
self.base_dir = config.get('base_dir', './downloads')
storage_subdir = config.get('storage_dir', 'pypi')
self.storage_dir = os.path.join(self.base_dir, storage_subdir)
self.simple_dir = os.path.join(self.storage_dir, 'simple')
self.web_dir = os.path.join(self.storage_dir, 'web')
# 确保存储目录存在
os.makedirs(self.simple_dir, exist_ok=True)
os.makedirs(self.web_dir, exist_ok=True)
def handle_request(self, handler, path: str) -> bool:
"""
处理PyPI请求
路径格式: /simple/requests/ 或 /packages/xxx.tar.gz 或 /pypi/web/package 或 /pypi/packages/hash/file
"""
try:
import sys
# 如果路径以 pypi/ 开头,也需要去掉
path = path.lstrip('/')
if path.startswith('pypi/'):
path = path[5:]
parts = path.strip('/').split('/')
# 过滤空字符串
parts = [p for p in parts if p]
if not parts:
return self._handle_index(handler)
if parts[0] == 'simple':
# Simple API
import sys
if len(parts) == 1:
# /simple/ - 返回根索引
return self._handle_index(handler)
elif len(parts) == 2:
# /simple/package/
return self._handle_simple_index(handler, parts[1])
elif len(parts) >= 3:
# /simple/package/version/ 或 /simple/package/version#egg=...
return self._handle_package_file(handler, parts[1], '/'.join(parts[2:]))
else:
handler.send_error(400, "Invalid simple API path")
return False
elif parts[0] == 'web':
# /web/package/ 或 /web/package/json
# pip sends /pypi/web/<package>/json
package = parts[1] if len(parts) >= 2 else ''
return self._handle_web_api(handler, package)
elif parts[0] == 'packages':
# 包下载
filename = '/'.join(parts[1:])
return self._handle_package_download(handler, filename)
elif parts[0] == 'legacy':
# 旧版PyPI兼容
return self._handle_legacy(handler, '/'.join(parts[1:]))
else:
handler.send_error(404, "Unknown API")
return False
except Exception as e:
handler.send_error(500, str(e))
return False
def _handle_index(self, handler) -> bool:
"""处理索引请求 - 返回所有可用包的列表"""
# 从上游获取包列表
url = self.upstream_url.rstrip('/')
if url.endswith('/simple'):
url = url # 保持 /simple
else:
url = url + '/simple'
try:
req = urllib.request.Request(url)
req.add_header('Accept', 'text/html')
req.add_header('User-Agent', 'PyPI-Mirror/1.0')
with urllib.request.urlopen(req, timeout=30) as response:
data = response.read().decode('utf-8')
# 转换相对链接
# 清华源返回的可能是完整的HTML,需要转换链接
data = self._convert_simple_index_html(data)
data_bytes = data.encode('utf-8')
handler.send_response(200)
handler.send_header('Content-Type', 'text/html; charset=utf-8')
handler.send_header('Content-Length', str(len(data_bytes)))
handler.end_headers()
handler.wfile.write(data_bytes)
return True
except Exception as e:
handler.send_error(502, f"Failed to fetch package index: {str(e)}")
return False
def _convert_simple_index_html(self, html: str) -> str:
"""转换根索引页面的HTML"""
import re
# 替换上游链接
def convert_link(match):
href = match.group(1)
text = match.group(2)
if href.startswith('/simple/'):
return match.group(0) # 已经是相对路径
elif href.startswith('https://pypi.tuna.tsinghua.edu.cn/simple/'):
simple_part = href.split("/simple/")[-1]
return f'<a href="/simple/{simple_part}">{text}</a>'
elif href.startswith('https://'):
# 其他上游链接,提取包名
pkg_name = href.rstrip('/').split('/')[-1]
return f'<a href="/simple/{pkg_name}/">{text}</a>'
return match.group(0)
# 匹配 <a href="...">text</a>
return re.sub(r'<a[^>]+href="([^"]+)"[^>]*>([^<]*)</a>', convert_link, html)
def _handle_simple_index(self, handler, package: str) -> bool:
"""处理Simple API索引请求"""
import json
import time
package = package.lower()
# 调试 - 确保函数被调用
debug_file = '/tmp/pypi_debug.log'
with open(debug_file, 'a') as f:
f.write(f"[HANDLE_SIMPLE] START package={package}\n")
# 检查客户端Accept header
accept = handler.headers.get('Accept', '')
wants_json = 'application/vnd.pypi.simple.v1+json' in accept
# 调试
debug_file = '/tmp/pypi_debug.log'
with open(debug_file, 'a') as f:
f.write(f"[SIMPLE_INDEX] package={package}, wants_json={wants_json}, accept={accept[:50]}\n")
# 根据请求格式选择正确的缓存key,统一使用 simple/ 前缀
cache_key = f"simple/{package}"
cached = self._get_cache(cache_key)
# 调试缓存
debug_file = '/tmp/pypi_debug.log'
with open(debug_file, 'a') as f:
f.write(f"[CACHE_CHECK] cache_key={cache_key}, cached={'YES' if cached else 'NO'}\n")
if cached:
# 返回缓存,使用正确的Content-Type
if wants_json:
handler.send_response(200)
handler.send_header('Content-Type', 'application/vnd.pypi.simple.v1+json; charset=utf-8')
else:
handler.send_response(200)
handler.send_header('Content-Type', 'text/html; charset=utf-8')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取HTML
# upstream_url 已经是完整路径(如 https://pypi.tuna.tsinghua.edu.cn/simple)
# 所以只需要添加 /package/
url = f"{self.upstream_url}/{package}/"
try:
req = urllib.request.Request(url)
req.add_header('Accept', 'text/html')
req.add_header('User-Agent', 'PyPI-Mirror/1.0')
with urllib.request.urlopen(req, timeout=30) as response:
data = response.read().decode('utf-8')
# 根据客户端请求返回不同格式,统一使用 simple/ 路径
if wants_json:
# 转换为JSON格式
json_data = self._convert_to_json(package, data)
proxy_data = json.dumps(json_data)
content_type = 'application/vnd.pypi.simple.v1+json; charset=utf-8'
cache_key = f"simple/{package}"
else:
# 转换为HTML格式
debug_file = '/tmp/pypi_debug.log'
with open(debug_file, 'a') as f:
f.write(f"[CONVERT_CALL] Before conversion\n")
try:
proxy_data = self._convert_simple_html(package, data)
with open(debug_file, 'a') as f:
f.write(f"[CONVERT_CALL] After conversion\n")
except Exception as e:
with open(debug_file, 'a') as f:
f.write(f"[CONVERT_ERROR] {e}\n")
proxy_data = data # fallback to raw data
content_type = 'application/vnd.pypi.simple.v1+html; charset=utf-8'
if True:
self._set_cache(cache_key, proxy_data.encode('utf-8'))
proxy_bytes = proxy_data.encode('utf-8')
handler.send_response(200)
handler.send_header('Content-Type', content_type)
handler.send_header('Content-Length', str(len(proxy_bytes)))
handler.end_headers()
handler.wfile.write(proxy_bytes)
return True
except urllib.error.HTTPError as e:
if e.code == 404:
handler.send_error(404, f"Package not found: {package}")
else:
handler.send_error(502, f"Failed to fetch from upstream: {str(e)}")
return False
def _convert_to_json(self, package: str, html: str) -> dict:
"""将HTML转换为JSON格式"""
import re
# 解析HTML中的链接
links = []
# 匹配 <a href="...">text</a>
pattern = r'<a[^>]+href="([^"]+)"[^>]*>([^<]*)</a>'
matches = re.findall(pattern, html)
for href, text in matches:
# 提取文件名(从链接文本)
filename = text.strip() if text.strip() else ''
# 如果没有链接文本,从URL中提取
if not filename:
if '#' in href:
filename = href.split('#')[0].split('/')[-1]
else:
filename = href.split('/')[-1]
# JSON格式中URL不应该包含fragment
# pip从filename字段提取版本号(如 Flask-1.0.0.tar.gz -> 1.0.0)
# 解析URL并转换为代理路径
url = href
if href.startswith('../'):
# 相对路径 - 需要转换为代理路径
# 格式: ../../packages/hash1/hash2/fullhash/filename#sha256=...
# 或: ../../packages/hash1/hash2/filename
# parts = ['..', '..', 'packages', 'hash1', 'hash2', 'fullhash', 'filename', ...]
parts = href.split('/')
try:
pkg_idx = parts.index('packages')
# 提取从 packages 后面到文件名之前的所有部分作为 hash 路径
# 文件名是最后一个非空部分(可能包含 #fragment)
# 找到文件名的位置(最后一个部分)
filename_idx = len(parts) - 1
while filename_idx > pkg_idx and not parts[filename_idx]:
filename_idx -= 1
# hash_path 是 packages 后面到文件名之前的所有部分
if filename_idx > pkg_idx + 1:
hash_path = '/'.join(parts[pkg_idx+1:filename_idx])
else:
hash_path = parts[pkg_idx+1] if pkg_idx + 1 < len(parts) else ''
# 文件名: 检查 pkg_idx+3 是否存在且不是哈希
if pkg_idx + 3 < len(parts):
fname_full = parts[pkg_idx+3]
# 如果 fname_full 看起来像哈希(包含 sha256= 或长度>=32的十六进制),则使用原始 filename
# 清华源格式: .../hash/filename#sha256=...
# 其中 hash 是 28-30 位十六进制
is_hash_like = ('sha256=' in fname_full or 'sha512=' in fname_full or
(len(fname_full) >= 28 and all(c in '0123456789abcdef' for c in fname_full[:28].lower())))
if is_hash_like:
# 这是哈希,不是文件名
fname = filename if filename else ''
else:
# 这是文件名
fname = fname_full.split('#')[0]
if not filename:
filename = fname
else:
fname = filename if filename else ''
# URL不包含fragment
url = f"/pypi/packages/{hash_path}/{fname}"
except ValueError:
pass
elif href.startswith('/pypi/'):
# 绝对路径(如 /pypi/packages/hash/filename#egg=package-version)
# 去掉fragment
href_clean = href.split('#')[0]
parts = href_clean.split('/')
# parts = ['', 'pypi', 'packages', 'hash', 'filename']
if len(parts) >= 5:
hash_path = parts[3]
fname = parts[4]
if not filename:
filename = fname
url = f"/pypi/packages/{hash_path}/{fname}"
elif href.startswith('http'):
# 绝对URL - 转换为代理路径
if 'files.pythonhosted.org' in href or 'files.pypi.org' in href:
# 格式: https://files.pythonhosted.org/packages/hash1/hash2/完整哈希/filename
# 例如: https://files.pythonhosted.org/packages/ec/f9/7f9263c5695f4bd0023734af91bedb2ff8209e8de6ead162f35d8dc762fd/flask-3.1.2-py3-none-any.whl
parts = href.split('/packages/')
if len(parts) >= 2:
path_after_packages = parts[1]
# 完整路径: hash1/hash2/完整哈希/filename
url = f"/pypi/packages/{path_after_packages}"
fname = path_after_packages.split('/')[-1]
if not filename:
filename = fname
elif 'pypi.tuna.tsinghua.edu.cn' in href or 'mirrors.tuna.tsinghua.edu.cn' in href:
parts = href.rsplit('/', 1)
if len(parts) == 2:
path_part = parts[0]
fname = parts[1]
hash_path = path_part.split('/')[-1]
if not filename:
filename = fname
url = f"/pypi/packages/{hash_path}/{fname}"
link_entry = {
"filename": filename,
"url": url
}
links.append(link_entry)
return {
"meta": {
"api-version": "1.0",
"repository-version": "1.0"
},
"name": package,
"files": links
}
def _handle_package_file(self, handler, package: str, filename: str) -> bool:
"""处理包文件请求"""
import urllib.parse
# 解析文件名
# 新格式: /pypi/packages/hash/filename#pip=package-version
# 或旧格式: /pypi/packages/filename?url=...
parsed = urllib.parse.urlparse(f"/{filename}")
actual_filename = parsed.path.lstrip('/')
query_params = urllib.parse.parse_qs(parsed.query)
fragment = urllib.parse.parse_qs(parsed.fragment) if parsed.fragment else {}
# 从fragment中提取包信息(用于缓存键)
actual_package = package
if 'pip' in fragment:
# 格式: #pip=flask-2.0.0
pip_info = fragment['pip'][0]
if '-' in pip_info:
# 提取版本号
parts = pip_info.split('-', 1)
if len(parts) == 2:
actual_package = parts[0]
cache_key = f"packages/{actual_filename}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 尝试从上游获取
# 格式: /pypi/packages/hash/filename -> 构造上游URL
possible_urls = []
# 获取基础URL(去掉 /simple 后缀)
base_url = self.upstream_url.rstrip('/')
if base_url.endswith('/simple'):
base_url = base_url[:-7]
# 新格式: hash/filename -> 尝试清华源
# 使用 actual_filename(已解析的纯文件路径)
possible_urls.append(f"{base_url}/packages/{actual_filename}")
# 尝试官方源
possible_urls.append(f"https://files.pythonhosted.org/packages/{actual_filename}")
# 如果有查询参数中的URL,也尝试
if 'url' in query_params:
possible_urls.insert(0, urllib.parse.unquote(query_params['url'][0]))
data = None
last_error = None
for url in possible_urls:
try:
import sys
req = urllib.request.Request(url)
req.add_header('User-Agent', 'PyPI-Mirror/1.0')
with urllib.request.urlopen(req, timeout=60) as response:
data = response.read()
break # 成功获取,退出循环
except Exception as e:
import sys
last_error = e
continue
if data is None:
handler.send_error(502, f"Failed to fetch package: {last_error}")
return False
if True:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
def _handle_web_api(self, handler, package: str) -> bool:
"""处理Web API请求"""
import sys
package = package.lower()
cache_key = f"web/{package}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/vnd.pypi.simple.v1+json; charset=utf-8')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取 - 需要去掉 /simple 后缀
base_url = self.upstream_url.rstrip('/')
if base_url.endswith('/simple'):
base_url = base_url[:-7]
url = f"{base_url}/pypi/{package}/json"
try:
req = urllib.request.Request(url)
req.add_header('Accept', 'application/json')
req.add_header('User-Agent', 'PyPI-Mirror/1.0')
with urllib.request.urlopen(req, timeout=30) as response:
data = response.read().decode('utf-8')
data_json = json.loads(data)
# 转换URL
data_json = self._convert_package_json(package, data_json)
proxy_data = json.dumps(data_json)
if True:
self._set_cache(cache_key, proxy_data.encode('utf-8'))
proxy_bytes = proxy_data.encode('utf-8')
handler.send_response(200)
handler.send_header('Content-Type', 'application/vnd.pypi.simple.v1+json; charset=utf-8')
handler.send_header('Content-Length', str(len(proxy_bytes)))
handler.end_headers()
handler.wfile.write(proxy_bytes)
return True
except urllib.error.HTTPError as e:
handler.send_error(502, f"Failed to fetch package info: {str(e)}")
return False
def _handle_package_download(self, handler, filename: str) -> bool:
"""处理包下载请求"""
import sys
# 使用 packages/ 前缀,保持标准 PyPI 目录结构
cache_key = f"packages/{filename}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取 - 注意去掉 /simple 后缀
base_url = self.upstream_url.rstrip('/')
if base_url.endswith('/simple'):
base_url = base_url[:-7] # 去掉 /simple
url = f"{base_url}/packages/{filename}"
try:
req = urllib.request.Request(url)
req.add_header('User-Agent', 'PyPI-Mirror/1.0')
with urllib.request.urlopen(req, timeout=120) as response:
data = response.read()
if True:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(502, f"Failed to download: {str(e)}")
return False
def _handle_legacy(self, handler, path: str) -> bool:
"""处理旧版PyPI兼容"""
handler.send_error(410, "Legacy PyPI API is deprecated")
return False
def _convert_simple_html(self, package: str, html: str) -> str:
"""转换Simple API HTML,替换URL为代理地址"""
import urllib.parse
import sys
import os
# 写入调试文件
debug_file = '/tmp/pypi_debug.log'
with open(debug_file, 'a') as f:
f.write(f"[PyPI] Converting HTML for package: {package}\n")
# 打印前几个链接用于调试
import re
test_matches = re.findall(r'href="([^"]+)"', html)[:3]
with open(debug_file, 'a') as f:
f.write(f"[PyPI] Sample links: {test_matches}\n")
def convert_absolute_url(match):
"""转换绝对URL为代理链接"""
original_url = match.group(1) if match.lastindex else match.group(0)
# 提取文件名和完整的hash路径
# 格式: https://pypi.tuna.tsinghua.edu.cn/packages/hash1/hash2/fullhash/filename
# 我们将其转换为: /pypi/packages/hash1/hash2/fullhash/filename#pip=<package>
parts = original_url.rsplit('/', 1)
if len(parts) == 2:
path_part = parts[0]
filename = parts[1]
# 提取从 packages/ 后面的完整路径(包含完整hash)
try:
pkg_idx = path_part.index('/packages/')
hash_path = path_part[pkg_idx + 10:] # 去掉 /packages/
except ValueError:
hash_path = filename
return f'href="/pypi/packages/{hash_path}/{filename}#pip={package}-{filename.split("-")[1] if "-" in filename else ""}"'
return match.group(0)
# 替换绝对URL - pypi.tuna.tsinghua.edu.cn (清华源)
html = re.sub(
r'(https://pypi\.tuna\.tsinghua\.edu\.cn/packages/[^"\']+)',
convert_absolute_url,
html
)
# 替换绝对URL - mirrors.tuna.tsinghua.edu.cn
html = re.sub(
r'(https://mirrors\.tuna\.tsinghua\.edu\.cn/pypi/packages/[^"\']+)',
convert_absolute_url,
html
)
# 替换绝对URL - files.pypi.org
html = re.sub(
r'(https://files\.pypi\.org/packages/[^"\']+)',
convert_absolute_url,
html
)
# 替换绝对URL - files.pythonhosted.org
html = re.sub(
r'(https://files\.pythonhosted\.org/packages/[^"\']+)',
convert_absolute_url,
html
)
# 替换相对路径链接 - ../../packages/hash1/hash2/fullhash/filename -> /pypi/packages/hash1/hash2/fullhash/filename#pip=...
def convert_relative_match(match):
"""转换相对路径链接"""
href = match.group(1)
debug_file = '/tmp/pypi_debug.log'
with open(debug_file, 'a') as f:
f.write(f"[CONVERT] Input href: {href[:80]}...\n")
# 提取文件名
filename = href.split('/')[-1].split('#')[0]
# 提取完整的hash路径(从 packages/ 后面的所有部分除了文件名)
parts = href.split('/')
with open(debug_file, 'a') as f:
f.write(f"[CONVERT] Parts: {parts}\n")
try:
pkg_idx = parts.index('packages')
# packages 后面到倒数第二个是 hash 路径,最后一个是文件名
hash_parts = parts[pkg_idx+1:-1] # 除了最后一个(文件名)
with open(debug_file, 'a') as f:
f.write(f"[CONVERT] hash_parts: {hash_parts}\n")
hash_path = '/'.join(hash_parts) if hash_parts else filename
with open(debug_file, 'a') as f:
f.write(f"[CONVERT] hash_path: {hash_path}\n")
except ValueError:
hash_path = filename
with open(debug_file, 'a') as f:
f.write(f"[CONVERT] ValueError, hash_path: {hash_path}\n")
# 提取版本号 - 从文件名中提取,如 Flask-0.1.tar.gz -> 0.1
base_name = filename
# 去掉扩展名
for ext in ['.tar.gz', '.whl', '.tar.bz2', '.tar.xz']:
if base_name.endswith(ext):
base_name = base_name[:-len(ext)]
break
# 尝试多种大小写组合来去掉包名前缀
version = base_name
for pkg_name in [package, package.lower(), package.upper(), package.capitalize()]:
if base_name.lower().startswith(pkg_name.lower() + '-'):
version = base_name[len(pkg_name)+1:]
break
return f'href="/pypi/packages/{hash_path}/{filename}#egg={package}-{version}"'
# 匹配相对路径的链接
html = re.sub(
r'href="(\.\./\.\./packages/[^"]+)"',
convert_relative_match,
html
)
html = re.sub(
r"href='(\.\./\.\./packages/[^']+)'",
convert_relative_match,
html
)
return html
def _convert_package_json(self, package: str, data: dict) -> dict:
"""转换Package JSON,替换URL为代理地址"""
# 转换URL函数
def convert_url(url):
# 处理 files.pythonhosted.org 和 files.pypi.org
# 格式: https://files.pythonhosted.org/packages/<hash1>/<hash2>/<完整哈希>/<filename>
# 例如: https://files.pythonhosted.org/packages/ec/f9/7f9263c5695f4bd0023734af91bedb2ff8209e8de6ead162f35d8dc762fd/flask-3.1.2-py3-none-any.whl
if 'files.pythonhosted.org' in url or 'files.pypi.org' in url:
path_parts = url.split('/packages/')
if len(path_parts) >= 2:
# 直接使用 /packages/ 后的完整路径
path_after_packages = path_parts[1]
return f'/pypi/packages/{path_after_packages}'
return url
# 转换urls
if 'urls' in data:
for item in data['urls']:
if 'url' in item:
item['url'] = convert_url(item['url'])
return data
def _fetch(self, url: str) -> Optional[bytes]:
"""从URL获取数据"""
try:
req = urllib.request.Request(url)
req.add_header('User-Agent', 'PyPI-Mirror/1.0')
with urllib.request.urlopen(req, timeout=30) as response:
return response.read()
except Exception:
return None
def _get_cache(self, cache_key: str) -> Optional[bytes]:
"""获取缓存"""
if not True:
return None
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
if not os.path.exists(cache_path):
return None
if os.path.exists(meta_path):
try:
with open(meta_path, 'r') as f:
meta = json.load(f)
if time.time() > meta.get('expires', 0):
return None
except Exception:
pass
try:
with open(cache_path, 'rb') as f:
return f.read()
except Exception:
return None
def _set_cache(self, cache_key: str, data: bytes):
"""设置缓存"""
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
try:
with open(cache_path, 'wb') as f:
f.write(data)
meta = {
'cached_at': time.time(),
'expires': time.time() + 86400,
'size': len(data)
}
with open(meta_path, 'w') as f:
json.dump(meta, f)
except Exception as e:
print(f"PyPI缓存写入失败: {e}")
def _get_cache_path(self, cache_key: str) -> str:
"""获取缓存路径"""
# cache_key 格式: packages/fe/df/88ccbee.../filename
# 存储路径: downloads/pypi-cn/packages/fe/df/88ccbee.../filename
# 确保使用正斜杠
safe_key = cache_key.replace('\\', '/')
return os.path.join(self.storage_dir, safe_key)
def get_cache_stats(self) -> dict:
"""获取缓存统计"""
if not os.path.exists(self.storage_dir):
return {'files': 0, 'size': 0}
total_size = 0
file_count = 0
for root, dirs, files in os.walk(self.storage_dir):
for f in files:
if not f.endswith('.meta'):
file_count += 1
total_size += os.path.getsize(os.path.join(root, f))
return {
'files': file_count,
'size': total_size,
'size_formatted': self._format_size(total_size)
}
def _format_size(self, size_bytes: int) -> str:
"""格式化文件大小"""
if size_bytes == 0:
return "0 B"
units = ["B", "KB", "MB", "GB"]
i = 0
while size_bytes >= 1024 and i < len(units) - 1:
size_bytes /= 1024.0
i += 1
return f"{size_bytes:.2f} {units[i]}"
+384
View File
@@ -0,0 +1,384 @@
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
YUM/DNF镜像代理处理器
支持RHEL/CentOS/Rocky/AlmaLinux软件源
"""
import os
import json
import time
import gzip
import xml.etree.ElementTree as ET
import urllib.request
from typing import Dict, List, Optional
from datetime import datetime
class YUMMirror:
"""YUM/DNF镜像代理"""
def __init__(self, config: dict):
self.config = config
# 配置 - 使用 storage_dir(基于 base_dir)
self.base_url = config.get('base_url', 'http://mirror.centos.org/centos')
self.storage_dir = config.get('storage_dir', './downloads/yum')
self.base_dir = config.get('base_dir', './downloads')
self.repo_id = config.get('repo_id', 'baseos')
self.arch = config.get('arch', 'x86_64')
# 确保存储目录存在
os.makedirs(self.storage_dir, exist_ok=True)
def handle_request(self, handler, path: str) -> bool:
"""
处理YUM请求
路径格式: /centos/7/updates/x86_64/repodata/repomd.xml
"""
try:
parts = path.strip('/').split('/')
if len(parts) < 3:
return self._handle_index(handler)
distro = parts[0] # centos, rocky, alma
version = parts[1] # 7, 8, 9
repo = parts[2] # baseos, appstream, updates
rest = '/'.join(parts[3:])
# 确定文件类型
if 'repomd.xml' in rest:
return self._handle_repomd(handler, distro, version, repo)
elif 'primary.xml.gz' in rest:
return self._handle_primary(handler, distro, version, repo, 'primary')
elif 'filelists.xml.gz' in rest:
return self._handle_filelists(handler, distro, version, repo, 'filelists')
elif 'other.xml.gz' in rest:
return self._handle_other(handler, distro, version, repo, 'other')
else:
return self._handle_repo_file(handler, distro, version, repo, rest)
except Exception as e:
handler.send_error(500, str(e))
return False
def _handle_index(self, handler) -> bool:
"""处理索引请求"""
handler.send_json_response({
'base_url': self.base_url,
'repo_id': self.repo_id,
'arch': self.arch,
'cache_stats': self.get_cache_stats()
})
return True
def _handle_repomd(self, handler, distro: str, version: str, repo: str) -> bool:
"""处理repomd.xml请求"""
cache_key = f"repomd:{distro}:{version}:{repo}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/xml')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 从上游获取
url = f"{self.base_url}/{version}/{repo}/{self.arch}/repodata/repomd.xml"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/xml')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(502, f"Failed to fetch repomd: {str(e)}")
return False
def _handle_primary(self, handler, distro: str, version: str, repo: str, db_type: str) -> bool:
"""处理primary.xml.gz"""
cache_key = f"primary:{distro}:{version}:{repo}:{self.arch}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 先获取repomd.xml找到对应的数据库文件
repomd_url = f"{self.base_url}/{version}/{repo}/{self.arch}/repodata/repomd.xml"
try:
repomd_data = self._fetch(repomd_url)
# 解析repomd.xml找到primary文件
root = ET.fromstring(repomd_data)
ns = {'repomd': 'http://linux.duke.edu/metadata/repo'}
data_location = None
for elem in root.findall('.//repomd:data', ns):
if elem.get('type') == 'primary':
data_location = elem.find('repomd:location', ns).get('href')
break
if data_location:
db_url = f"{self.base_url}/{version}/{repo}/{self.arch}/repodata/{data_location}"
data = self._fetch(db_url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
pass
handler.send_error(502, "Failed to fetch primary database")
return False
def _handle_filelists(self, handler, distro: str, version: str, repo: str, db_type: str) -> bool:
"""处理filelists.xml.gz"""
cache_key = f"filelists:{distro}:{version}:{repo}:{self.arch}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 先获取repomd.xml找到对应的数据库文件
repomd_url = f"{self.base_url}/{version}/{repo}/{self.arch}/repodata/repomd.xml"
try:
repomd_data = self._fetch(repomd_url)
# 解析repomd.xml找到filelists文件
root = ET.fromstring(repomd_data)
ns = {'repomd': 'http://linux.duke.edu/metadata/repo'}
data_location = None
for elem in root.findall('.//repomd:data', ns):
if elem.get('type') == 'filelists':
data_location = elem.find('repomd:location', ns).get('href')
break
if data_location:
db_url = f"{self.base_url}/{version}/{repo}/{self.arch}/repodata/{data_location}"
data = self._fetch(db_url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
pass
handler.send_error(502, "Failed to fetch filelists database")
return False
def _handle_other(self, handler, distro: str, version: str, repo: str, db_type: str) -> bool:
"""处理other.xml.gz"""
cache_key = f"other:{distro}:{version}:{repo}:{self.arch}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(cached)))
handler.end_headers()
handler.wfile.write(cached)
return True
# 先获取repomd.xml找到对应的数据库文件
repomd_url = f"{self.base_url}/{version}/{repo}/{self.arch}/repodata/repomd.xml"
try:
repomd_data = self._fetch(repomd_url)
# 解析repomd.xml找到other文件
root = ET.fromstring(repomd_data)
ns = {'repomd': 'http://linux.duke.edu/metadata/repo'}
data_location = None
for elem in root.findall('.//repomd:data', ns):
if elem.get('type') == 'other':
data_location = elem.find('repomd:location', ns).get('href')
break
if data_location:
db_url = f"{self.base_url}/{version}/{repo}/{self.arch}/repodata/{data_location}"
data = self._fetch(db_url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/x-gzip')
handler.send_header('Content-Length', str(len(data)))
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
pass
handler.send_error(502, "Failed to fetch other database")
return False
def _handle_repo_file(self, handler, distro: str, version: str, repo: str, path: str) -> bool:
"""处理仓库中的其他文件"""
cache_key = f"file:{distro}:{version}:{repo}:{path.replace('/', ':')}"
cached = self._get_cache(cache_key)
if cached:
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.end_headers()
handler.wfile.write(cached)
return True
url = f"{self.base_url}/{version}/{repo}/{self.arch}/{path}"
try:
data = self._fetch(url)
if self.cache_enabled:
self._set_cache(cache_key, data)
handler.send_response(200)
handler.send_header('Content-Type', 'application/octet-stream')
handler.end_headers()
handler.wfile.write(data)
return True
except Exception as e:
handler.send_error(404, f"File not found: {str(e)}")
return False
def _fetch(self, url: str) -> Optional[bytes]:
"""从URL获取数据"""
try:
req = urllib.request.Request(url)
req.add_header('User-Agent', 'YUM-Mirror/1.0')
with urllib.request.urlopen(req, timeout=30) as response:
return response.read()
except Exception:
return None
def _get_cache(self, cache_key: str) -> Optional[bytes]:
"""获取缓存"""
if not self.cache_enabled:
return None
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
if not os.path.exists(cache_path):
return None
if os.path.exists(meta_path):
try:
with open(meta_path, 'r') as f:
meta = json.load(f)
if time.time() > meta.get('expires', 0):
return None
except Exception:
pass
try:
with open(cache_path, 'rb') as f:
return f.read()
except Exception:
return None
def _set_cache(self, cache_key: str, data: bytes):
"""设置缓存"""
cache_path = self._get_cache_path(cache_key)
meta_path = cache_path + '.meta'
os.makedirs(os.path.dirname(cache_path), exist_ok=True)
try:
with open(cache_path, 'wb') as f:
f.write(data)
meta = {
'cached_at': time.time(),
'expires': time.time() + self.cache_ttl,
'size': len(data)
}
with open(meta_path, 'w') as f:
json.dump(meta, f)
except Exception as e:
print(f"YUM缓存写入失败: {e}")
def _get_cache_path(self, cache_key: str) -> str:
"""获取缓存路径"""
subdir = cache_key[:2]
return os.path.join(self.storage_dir, subdir, cache_key)
def get_cache_stats(self) -> dict:
"""获取缓存统计"""
if not os.path.exists(self.storage_dir):
return {'files': 0, 'size': 0}
total_size = 0
file_count = 0
for root, dirs, files in os.walk(self.storage_dir):
for f in files:
if not f.endswith('.meta'):
file_count += 1
total_size += os.path.getsize(os.path.join(root, f))
return {
'files': file_count,
'size': total_size,
'size_formatted': self._format_size(total_size)
}
def _format_size(self, size_bytes: int) -> str:
"""格式化文件大小"""
if size_bytes == 0:
return "0 B"
units = ["B", "KB", "MB", "GB"]
i = 0
while size_bytes >= 1024 and i < len(units) - 1:
size_bytes /= 1024.0
i += 1
return f"{size_bytes:.2f} {units[i]}"