added conopy script

This commit is contained in:
2026-06-24 09:33:09 +03:00
parent 4989face48
commit 6355f8afd1
+53 -7
View File
@@ -4,6 +4,7 @@ from __future__ import annotations
import argparse import argparse
import json import json
import time
from collections.abc import Iterable from collections.abc import Iterable
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
@@ -142,7 +143,43 @@ def destination_for(url: str, output_dir: Path) -> Path:
return output_dir / name return output_dir / name
def download_urls(urls: list[str], output_dir: Path, overwrite: bool) -> tuple[int, int, list[str]]: def _format_mb(value: int) -> str:
return f"{value / (1024 * 1024):.1f} MB"
def _download_one(url: str, destination: Path, timeout_s: int, chunk_size: int) -> None:
partial = destination.with_suffix(f"{destination.suffix}.part")
request = Request(url, headers={"User-Agent": USER_AGENT})
with urlopen(request, timeout=timeout_s) as response, partial.open("wb") as output:
total_header = response.headers.get("Content-Length")
total = int(total_header) if total_header and total_header.isdigit() else None
read = 0
last_report = time.monotonic()
while True:
chunk = response.read(chunk_size)
if not chunk:
break
output.write(chunk)
read += len(chunk)
now = time.monotonic()
if now - last_report >= 5:
if total:
pct = (read / total) * 100
print(f" {_format_mb(read)} / {_format_mb(total)} ({pct:.1f}%)")
else:
print(f" {_format_mb(read)}")
last_report = now
partial.replace(destination)
def download_urls(
urls: list[str],
output_dir: Path,
overwrite: bool,
*,
timeout_s: int = 60,
chunk_size: int = 1024 * 1024,
) -> tuple[int, int, list[str]]:
downloaded = 0 downloaded = 0
skipped = 0 skipped = 0
failed: list[str] = [] failed: list[str] = []
@@ -155,15 +192,17 @@ def download_urls(urls: list[str], output_dir: Path, overwrite: bool) -> tuple[i
print(f"skip existing {destination}") print(f"skip existing {destination}")
continue continue
partial = destination.with_suffix(f"{destination.suffix}.part")
if partial.exists() and not overwrite:
partial.unlink()
print(f"download {url}") print(f"download {url}")
try: try:
request = Request(url, headers={"User-Agent": USER_AGENT}) _download_one(url, destination, timeout_s, chunk_size)
with urlopen(request) as response, destination.open("wb") as output:
output.write(response.read())
except (HTTPError, URLError) as exc: except (HTTPError, URLError) as exc:
failed.append(f"{url}: {exc}") failed.append(f"{url}: {exc}")
if destination.exists(): if partial.exists():
destination.unlink() partial.unlink()
continue continue
downloaded += 1 downloaded += 1
@@ -178,6 +217,7 @@ def parse_args() -> argparse.Namespace:
parser.add_argument("--index-url", default=META_CHM_V2_INDEX_URL) parser.add_argument("--index-url", default=META_CHM_V2_INDEX_URL)
parser.add_argument("--urls-file", type=Path) parser.add_argument("--urls-file", type=Path)
parser.add_argument("--skip-url-check", action="store_true") parser.add_argument("--skip-url-check", action="store_true")
parser.add_argument("--timeout-s", default=60, type=int)
parser.add_argument("--overwrite", action="store_true") parser.add_argument("--overwrite", action="store_true")
return parser.parse_args() return parser.parse_args()
@@ -199,7 +239,13 @@ def main() -> None:
if not args.skip_url_check: if not args.skip_url_check:
validate_urls(urls) validate_urls(urls)
downloaded, skipped, failed = download_urls(urls, args.output_dir, args.overwrite) print(f"selected_tiles={len(urls)}")
downloaded, skipped, failed = download_urls(
urls,
args.output_dir,
args.overwrite,
timeout_s=args.timeout_s,
)
print(f"downloaded={downloaded} skipped={skipped} failed={len(failed)}") print(f"downloaded={downloaded} skipped={skipped} failed={len(failed)}")
if failed: if failed:
for item in failed: for item in failed: