Reference for ultralytics/utils/downloads.py#
This page is sourced from https://github.com/ultralytics/ultralytics/blob/main/ultralytics/utils/downloads.py. Have an improvement or example to add? Open a Pull Request — thank you! 🙏
Function ultralytics.utils.downloads.is_url#
def is_url(url: str | Path, check: bool = False) -> boolValidate if the given string is a URL and optionally check if the URL exists online.
Args
| Name | Type | Description | Default |
|---|---|---|---|
url | str | Path | The string to be validated as a URL. | required |
check | bool, optional | If True, performs an additional check to see if the URL exists online. | False |
Returns
| Type | Description |
|---|---|
bool | True if the string is a valid URL and, when 'check' is True, the URL is also reachable online. |
Examples
>>> valid = is_url("https://www.example.com")
>>> valid_and_exists = is_url("https://www.example.com", check=True)ultralytics/utils/downloads.py
def is_url(url: str | Path, check: bool = False) -> bool:
"""Validate if the given string is a URL and optionally check if the URL exists online.
Args:
url (str | Path): The string to be validated as a URL.
check (bool, optional): If True, performs an additional check to see if the URL exists online.
Returns:
(bool): True if the string is a valid URL and, when 'check' is True, the URL is also reachable online.
Examples:
>>> valid = is_url("https://www.example.com")
>>> valid_and_exists = is_url("https://www.example.com", check=True)
"""
try:
url = str(url)
result = parse.urlparse(url)
if not (result.scheme and result.netloc):
return False
if check:
import requests # scoped as slow import
return requests.head(url, timeout=3, allow_redirects=True).ok
return True
except Exception:
return FalseFunction ultralytics.utils.downloads.delete_dsstore#
def delete_dsstore(path: str | Path, files_to_delete: tuple[str, ...] = (".DS_Store", "__MACOSX")) -> NoneDelete all specified system files and directories in a directory.
Args
| Name | Type | Description | Default |
|---|---|---|---|
path | str | Path | The directory path where the files should be deleted. | required |
files_to_delete | tuple[str, ...] | Names of files and directories to delete. | (".DS_Store", "__MACOSX") |
Examples
>>> from ultralytics.utils.downloads import delete_dsstore
>>> delete_dsstore("path/to/dir")".DS_Store" files are created by the Apple operating system and contain metadata about folders and files. They are hidden system files and can cause issues when transferring files between different operating systems.
ultralytics/utils/downloads.py
def delete_dsstore(path: str | Path, files_to_delete: tuple[str, ...] = (".DS_Store", "__MACOSX")) -> None:
"""Delete all specified system files and directories in a directory.
Args:
path (str | Path): The directory path where the files should be deleted.
files_to_delete (tuple[str, ...]): Names of files and directories to delete.
Examples:
>>> from ultralytics.utils.downloads import delete_dsstore
>>> delete_dsstore("path/to/dir")
Notes:
".DS_Store" files are created by the Apple operating system and contain metadata about folders and files. They
are hidden system files and can cause issues when transferring files between different operating systems.
"""
for file in files_to_delete:
matches = sorted(Path(path).rglob(file), key=lambda x: len(x.parts), reverse=True)
LOGGER.info(f"Deleting {file} files: {matches}")
for f in matches:
if f.is_dir() and not f.is_symlink():
shutil.rmtree(f)
else:
f.unlink()Function ultralytics.utils.downloads.zip_directory#
def zip_directory(
directory: str | Path,
compress: bool = True,
exclude: tuple[str, ...] = (".DS_Store", "__MACOSX"),
progress: bool = True,
) -> PathZip the contents of a directory, excluding specified files.
The resulting zip file is named after the directory and placed alongside it.
Args
| Name | Type | Description | Default |
|---|---|---|---|
directory | str | Path | The path to the directory to be zipped. | required |
compress | bool | Whether to compress the files while zipping. | True |
exclude | tuple[str, ...], optional | A tuple of filename strings to be excluded. | (".DS_Store", "__MACOSX") |
progress | bool, optional | Whether to display a progress bar. | True |
Returns
| Type | Description |
|---|---|
Path | The path to the resulting zip file. |
Examples
>>> from ultralytics.utils.downloads import zip_directory
>>> file = zip_directory("path/to/dir")Raises
| Type | Description |
|---|---|
FileNotFoundError | If the directory does not exist. |
ultralytics/utils/downloads.py
def zip_directory(
directory: str | Path,
compress: bool = True,
exclude: tuple[str, ...] = (".DS_Store", "__MACOSX"),
progress: bool = True,
) -> Path:
"""Zip the contents of a directory, excluding specified files.
The resulting zip file is named after the directory and placed alongside it.
Args:
directory (str | Path): The path to the directory to be zipped.
compress (bool): Whether to compress the files while zipping.
exclude (tuple[str, ...], optional): A tuple of filename strings to be excluded.
progress (bool, optional): Whether to display a progress bar.
Returns:
(Path): The path to the resulting zip file.
Raises:
FileNotFoundError: If the directory does not exist.
Examples:
>>> from ultralytics.utils.downloads import zip_directory
>>> file = zip_directory("path/to/dir")
"""
from zipfile import ZIP_DEFLATED, ZIP_STORED, ZipFile
delete_dsstore(directory)
directory = Path(directory)
if not directory.is_dir():
raise FileNotFoundError(f"Directory '{directory}' does not exist.")
# Zip with progress bar
files = [f for f in directory.rglob("*") if f.is_file() and all(x not in f.name for x in exclude)] # files to zip
zip_file = directory.with_suffix(".zip")
compression = ZIP_DEFLATED if compress else ZIP_STORED
with ZipFile(zip_file, "w", compression) as f:
for file in TQDM(files, desc=f"Zipping {directory} to {zip_file}...", unit="files", disable=not progress):
f.write(file, file.relative_to(directory))
return zip_file # return path to zip fileFunction ultralytics.utils.downloads.unzip_file#
def unzip_file(
file: str | Path,
path: str | Path | None = None,
exclude: tuple[str, ...] = (".DS_Store", "__MACOSX"),
exist_ok: bool = False,
progress: bool = True,
) -> PathUnzip a *.zip file to the specified path, excluding specified files.
If the zipfile does not contain a single top-level directory, the function will create a new directory with the same name as the zipfile (without the extension) to extract its contents. If a path is not provided, the function will use the parent directory of the zipfile as the default path.
Args
| Name | Type | Description | Default |
|---|---|---|---|
file | str | Path | The path to the zipfile to be extracted. | required |
path | str | Path, optional | The path to extract the zipfile to. | None |
exclude | tuple[str, ...], optional | A tuple of filename strings to be excluded. | (".DS_Store", "__MACOSX") |
exist_ok | bool, optional | Whether to overwrite existing contents if they exist. | False |
progress | bool, optional | Whether to display a progress bar. | True |
Returns
| Type | Description |
|---|---|
Path | The path to the directory where the zipfile was extracted. |
Examples
>>> from ultralytics.utils.downloads import unzip_file
>>> directory = unzip_file("path/to/file.zip")Raises
| Type | Description |
|---|---|
BadZipFile | If the provided file does not exist or is not a valid zipfile. |
ultralytics/utils/downloads.py
def unzip_file(
file: str | Path,
path: str | Path | None = None,
exclude: tuple[str, ...] = (".DS_Store", "__MACOSX"),
exist_ok: bool = False,
progress: bool = True,
) -> Path:
"""Unzip a *.zip file to the specified path, excluding specified files.
If the zipfile does not contain a single top-level directory, the function will create a new directory with the same
name as the zipfile (without the extension) to extract its contents. If a path is not provided, the function will
use the parent directory of the zipfile as the default path.
Args:
file (str | Path): The path to the zipfile to be extracted.
path (str | Path, optional): The path to extract the zipfile to.
exclude (tuple[str, ...], optional): A tuple of filename strings to be excluded.
exist_ok (bool, optional): Whether to overwrite existing contents if they exist.
progress (bool, optional): Whether to display a progress bar.
Returns:
(Path): The path to the directory where the zipfile was extracted.
Raises:
BadZipFile: If the provided file does not exist or is not a valid zipfile.
Examples:
>>> from ultralytics.utils.downloads import unzip_file
>>> directory = unzip_file("path/to/file.zip")
"""
from zipfile import BadZipFile, ZipFile, is_zipfile
if not (Path(file).exists() and is_zipfile(file)):
raise BadZipFile(f"File '{file}' does not exist or is a bad zip file.")
if path is None:
path = Path(file).parent # default path
# Unzip the file contents
with ZipFile(file) as zipObj:
files = [f for f in zipObj.namelist() if all(x not in f for x in exclude)]
top_level_dirs = {Path(f).parts[0] for f in files}
# Decide to unzip directly or unzip into a directory
unzip_as_dir = len(top_level_dirs) == 1 # (len(files) > 1 and not files[0].endswith("/"))
if unzip_as_dir:
# Zip has 1 top-level directory
extract_path = path # i.e. ../datasets
path = Path(path) / next(iter(top_level_dirs)) # i.e. extract coco8/ dir to ../datasets/
else:
# Zip has multiple files at top level
path = extract_path = Path(path) / Path(file).stem # i.e. extract multiple files to ../datasets/coco8/
# Skip existing files or non-empty directories unless overwriting
if path.exists() and (path.is_file() or any(path.iterdir())) and not exist_ok:
LOGGER.warning(f"Skipping {file} unzip as destination path {path} already exists.")
return path
extract_path = Path(extract_path).resolve()
for f in TQDM(files, desc=f"Unzipping {file} to {Path(path).resolve()}...", unit="files", disable=not progress):
f_path = Path(f)
target = (extract_path / f_path).resolve()
if (
f_path.is_absolute()
or ".." in f_path.parts
or target.parts[: len(extract_path.parts)] != extract_path.parts
):
LOGGER.warning(f"Potentially insecure file path: {f}, skipping extraction.")
continue
zipObj.extract(f, extract_path)
return path # return unzip dirFunction ultralytics.utils.downloads.check_disk_space#
def check_disk_space(file_bytes: int, path: str | Path | None = None, sf: float = 1.5, hard: bool = True) -> boolCheck if there is sufficient disk space to download and store a file.
Args
| Name | Type | Description | Default |
|---|---|---|---|
file_bytes | int | The file size in bytes. | required |
path | str | Path, optional | The path or drive to check the available free space on. Defaults to the current working directory. | None |
sf | float, optional | Safety factor, the multiplier for the required free space. | 1.5 |
hard | bool, optional | Whether to throw an error or not on insufficient disk space. | True |
Returns
| Type | Description |
|---|---|
bool | True if there is sufficient disk space, False otherwise. |
Raises
| Type | Description |
|---|---|
MemoryError | If there is insufficient disk space and hard is True. |
ultralytics/utils/downloads.py
def check_disk_space(
file_bytes: int,
path: str | Path | None = None,
sf: float = 1.5,
hard: bool = True,
) -> bool:
"""Check if there is sufficient disk space to download and store a file.
Args:
file_bytes (int): The file size in bytes.
path (str | Path, optional): The path or drive to check the available free space on. Defaults to the current
working directory.
sf (float, optional): Safety factor, the multiplier for the required free space.
hard (bool, optional): Whether to throw an error or not on insufficient disk space.
Returns:
(bool): True if there is sufficient disk space, False otherwise.
Raises:
MemoryError: If there is insufficient disk space and `hard` is True.
"""
total, _used, free = shutil.disk_usage(path or Path.cwd()) # bytes
# A filesystem that cannot report usage returns 0 total blocks; free == 0 against a valid total is genuinely
# full and must still be caught, since `free` counts blocks available to an unprivileged process.
if not total or file_bytes * sf < free:
return True # sufficient space
def fmt_bytes(b):
if b < (1 << 20): # without a KB tier every value under 51 KB renders "0.0 MB", hiding how full the disk is
return f"{b / (1 << 10):.1f} KB"
return f"{b / (1 << 20):.1f} MB" if b < (1 << 30) else f"{b / (1 << 30):.3f} GB"
# Insufficient space
text = (
f"Insufficient free disk space {fmt_bytes(free)} < {fmt_bytes(int(file_bytes * sf))} required, "
f"Please free {fmt_bytes(int(file_bytes * sf - free))} additional disk space and try again."
)
if hard:
raise MemoryError(text)
LOGGER.warning(text)
return FalseFunction ultralytics.utils.downloads.get_google_drive_file_info#
def get_google_drive_file_info(link: str) -> tuple[str, str | None]Retrieve the direct download link and filename for a shareable Google Drive file link.
Args
| Name | Type | Description | Default |
|---|---|---|---|
link | str | The shareable link of the Google Drive file. | required |
Returns
| Type | Description |
|---|---|
url (str) | Direct download URL for the Google Drive file. |
filename (str | None) | Original filename of the Google Drive file. If filename extraction fails, returns None. |
Examples
>>> from ultralytics.utils.downloads import get_google_drive_file_info
>>> link = "https://drive.google.com/file/d/1cqT-cJgANNrhIHCrEufUYhQ4RqiWG_lJ/view?usp=drive_link"
>>> url, filename = get_google_drive_file_info(link)Raises
| Type | Description |
|---|---|
ConnectionError | If the Google Drive download quota for the file has been exceeded. |
ultralytics/utils/downloads.py
def get_google_drive_file_info(link: str) -> tuple[str, str | None]:
"""Retrieve the direct download link and filename for a shareable Google Drive file link.
Args:
link (str): The shareable link of the Google Drive file.
Returns:
url (str): Direct download URL for the Google Drive file.
filename (str | None): Original filename of the Google Drive file. If filename extraction fails, returns None.
Raises:
ConnectionError: If the Google Drive download quota for the file has been exceeded.
Examples:
>>> from ultralytics.utils.downloads import get_google_drive_file_info
>>> link = "https://drive.google.com/file/d/1cqT-cJgANNrhIHCrEufUYhQ4RqiWG_lJ/view?usp=drive_link"
>>> url, filename = get_google_drive_file_info(link)
"""
import requests # scoped as slow import
file_id = link.split("/d/")[1].split("/view", 1)[0]
drive_url = f"https://drive.google.com/uc?export=download&id={file_id}"
filename = None
# Start session
with requests.Session() as session:
response = session.get(drive_url, stream=True)
if "quota exceeded" in str(response.content.lower()):
raise ConnectionError(
emojis(
f"❌ Google Drive file download quota exceeded. "
f"Please try again later or download this file manually at {link}."
)
)
for k, v in response.cookies.items():
if k.startswith("download_warning"):
drive_url += f"&confirm={v}" # v is token
if cd := response.headers.get("content-disposition"):
filename = re.findall('filename="(.+)"', cd)[0]
return drive_url, filenameFunction ultralytics.utils.downloads.safe_download#
def safe_download(
url: str | Path,
file: str | Path | None = None,
dir: str | Path | None = None,
unzip: bool = True,
delete: bool = False,
curl: bool = False,
retry: int = 3,
min_bytes: float = 1e0,
exist_ok: bool = False,
progress: bool = True,
) -> PathDownload a file from a URL with options for retrying, unzipping, and deleting the downloaded file.
Partial downloads are detected using Content-Length validation and resumed with HTTP Range requests on retry. If url is an existing local file path, no download occurs and the file is optionally unzipped.
Args
| Name | Type | Description | Default |
|---|---|---|---|
url | str | Path | The URL of the file to be downloaded, or a local file path. | required |
file | str | Path, optional | The filename of the downloaded file. If not provided, the file will be saved with the same name as the URL. | None |
dir | str | Path, optional | The directory to save the downloaded file. If not provided, the file will be saved in the current working directory. | None |
unzip | bool, optional | Whether to unzip the downloaded file. | True |
delete | bool, optional | Whether to delete the downloaded file after unzipping. | False |
curl | bool, optional | Whether to use curl command line tool for downloading. | False |
retry | int, optional | The number of times to retry the download in case of failure. | 3 |
min_bytes | float, optional | The minimum number of bytes that the downloaded file should have, to be considered a successful download. | 1e0 |
exist_ok | bool, optional | Whether to overwrite existing contents during unzipping. | False |
progress | bool, optional | Whether to display a progress bar during the download. | True |
Returns
| Type | Description |
|---|---|
Path | The path to the downloaded file or extracted directory. |
Examples
>>> from ultralytics.utils.downloads import safe_download
>>> link = "https://ultralytics.com/assets/bus.jpg"
>>> path = safe_download(link)Raises
| Type | Description |
|---|---|
ConnectionError | If the download fails after all retries or the environment is offline. |
MemoryError | If there is insufficient disk space for the download. |
ultralytics/utils/downloads.py
def safe_download(
url: str | Path,
file: str | Path | None = None,
dir: str | Path | None = None,
unzip: bool = True,
delete: bool = False,
curl: bool = False,
retry: int = 3,
min_bytes: float = 1e0,
exist_ok: bool = False,
progress: bool = True,
) -> Path:
"""Download a file from a URL with options for retrying, unzipping, and deleting the downloaded file.
Partial downloads are detected using Content-Length validation and resumed with HTTP Range requests on retry. If
`url` is an existing local file path, no download occurs and the file is optionally unzipped.
Args:
url (str | Path): The URL of the file to be downloaded, or a local file path.
file (str | Path, optional): The filename of the downloaded file. If not provided, the file will be saved with
the same name as the URL.
dir (str | Path, optional): The directory to save the downloaded file. If not provided, the file will be saved
in the current working directory.
unzip (bool, optional): Whether to unzip the downloaded file.
delete (bool, optional): Whether to delete the downloaded file after unzipping.
curl (bool, optional): Whether to use curl command line tool for downloading.
retry (int, optional): The number of times to retry the download in case of failure.
min_bytes (float, optional): The minimum number of bytes that the downloaded file should have, to be considered
a successful download.
exist_ok (bool, optional): Whether to overwrite existing contents during unzipping.
progress (bool, optional): Whether to display a progress bar during the download.
Returns:
(Path): The path to the downloaded file or extracted directory.
Raises:
ConnectionError: If the download fails after all retries or the environment is offline.
MemoryError: If there is insufficient disk space for the download.
Examples:
>>> from ultralytics.utils.downloads import safe_download
>>> link = "https://ultralytics.com/assets/bus.jpg"
>>> path = safe_download(link)
"""
url = str(url)
if "://" not in url and Path(url).is_file(): # local file path ('://' check required in Windows Python<3.10)
f = Path(url)
else:
import requests # scoped as slow import
gdrive = url.startswith("https://drive.google.com/") # check if the URL is a Google Drive link
if gdrive:
url, file = get_google_drive_file_info(url)
url = url.replace(" ", "%20") # encode spaces for curl compatibility
f = Path(dir or ".") / (file or url2file(url)) # URL converted to filename
if not f.is_file(): # URL and file do not exist
uri = (url if gdrive else clean_url(url)).replace(ASSETS_URL, "https://ultralytics.com/assets") # clean
desc = f"Downloading {uri} to '{f}'"
f.parent.mkdir(parents=True, exist_ok=True) # make directory if missing
target = f
f = target.with_name(f".{target.name}.{uuid4().hex}.part") # publish only after size validation
curl_installed = shutil.which("curl")
expected_size = 0 # total bytes from Content-Length, kept across retries to validate them
# Both transports save the body as sent, so Content-Length and Range describe the file even when the server
# encodes it despite `Accept-Encoding: identity`, e.g. gzip objects on S3; it is decoded once complete
encoding = ""
for i in range(retry + 1):
try:
resume = f.stat().st_size if f.exists() else 0 # partial bytes kept from a failed attempt
if curl and not resume and curl_installed: # explicit curl download
s = "sS" * (not progress) # silent
# Stall bounds (not a total-transfer cap): abort if <1 B/s for 300 s so a dead connection
# cannot block interpreter shutdown while a non-daemon plot thread waits on a font download
args = ["--retry", "4", "--connect-timeout", "30", "--speed-limit", "1", "--speed-time", "300"]
# -f is required: without it curl writes the server error page as the file and exits 0
r = subprocess.run(
["curl", "-#", f"-{s}fL", url, "-o", f, "-D", "-", *args],
check=False,
stdout=subprocess.PIPE,
)
if r.returncode:
raise ConnectionError(f"Curl return value {r.returncode}")
# Final response, after any redirect, proxy or retry blocks and before any trailer block
final_headers = [h for h in r.stdout.split(b"\r\n\r\n") if h.startswith(b"HTTP/")][-1]
encoding = (
b"".join(re.findall(rb"(?im)^content-encoding:\s*(\S+)", final_headers)).decode().lower()
)
else: # requests download; timeout bounds connect and per-chunk read gaps, not total transfer
headers = {"Accept-Encoding": "identity"}
if resume:
headers["Range"] = f"bytes={resume}-"
with requests.get(url, stream=True, headers=headers, timeout=(30, 300)) as response:
if response.status_code == 416: # nothing left to resume, so the next retry restarts
f.unlink()
response.raise_for_status()
encoding = response.headers.get("Content-Encoding", "").lower()
if response.status_code != 206: # Range ignored, e.g. transcoded GCS objects, so restart
resume = 0
expected_size = int(response.headers.get("Content-Length", 0)) or expected_size
elif not expected_size: # partial left by curl, so take the total from 'bytes 5-9/10'
total = response.headers.get("Content-Range", "").rpartition("/")[2]
expected_size = int(total) if total.isdigit() else 0
if i == 0 and expected_size > 1048576:
check_disk_space(expected_size, path=f.parent)
buffer_size = max(8192, min(1048576, expected_size // 1000)) if expected_size else 8192
with TQDM(
total=expected_size,
desc=desc,
disable=not progress,
unit="B",
unit_scale=True,
unit_divisor=1024,
initial=resume,
) as pbar, open(f, "ab" if resume else "wb") as f_opened:
for data in response.raw.stream(buffer_size, decode_content=False):
f_opened.write(data)
pbar.update(len(data))
if f.exists():
file_size = f.stat().st_size
if expected_size and file_size != expected_size: # only if Content-Length is known
LOGGER.warning(
f"Partial download: {file_size}/{expected_size} bytes ({file_size / expected_size * 100:.1f}%)"
)
else:
if encoding not in {"", "identity"}: # undo the transfer encoding of the complete body
decoded = f.with_name(f"{f.name}.decoded")
try:
with open(f, "rb") as src, open(decoded, "wb") as dst:
wbits = 47 # detects a gzip or zlib header
try:
zlib.decompressobj(wbits).decompress(src.read(1024))
except zlib.error: # some servers send 'deflate' as raw DEFLATE without one
wbits = -15
src.seek(0)
d = zlib.decompressobj(wbits)
for chunk in iter(lambda: src.read(1048576), b""):
while chunk:
if d.eof: # next gzip member, after any zero padding
chunk = chunk.lstrip(b"\0")
if not chunk:
break
d = zlib.decompressobj(wbits)
dst.write(d.decompress(chunk))
chunk = d.unused_data
if not d.eof: # not an assert, which `python -O` removes
raise ConnectionError("Encoded body ended before its end-of-stream marker")
# A gzip encoding under a gzip name means the gzip is the file itself, e.g. a
# .tar.gz object stored with a gzip Content-Encoding, so keep its verified bytes
if encoding != "gzip" or target.suffix not in {".gz", ".tgz"}:
decoded.replace(f)
finally:
decoded.unlink(missing_ok=True)
if f.stat().st_size > min_bytes:
f.replace(target)
f = target
break # success
f.unlink() # remove partial downloads
except MemoryError:
raise # Re-raise immediately - no point retrying if insufficient disk space
except Exception as e:
# Only on the terminal failure: retries resume the partial file via a Range request, but leaving
# one behind makes the `not f.is_file()` guard above serve it as a complete cache hit forever.
if i == 0 and not is_online():
f.unlink(missing_ok=True)
raise ConnectionError(
emojis(f"❌ Download failure for {uri}. Environment may be offline.")
) from e
status = getattr(getattr(e, "response", None), "status_code", None)
# A 4xx other than timeout, range or rate-limit answers will not change on retry, so fail fast
if i >= retry or (status and status < 500 and status not in {408, 416, 429}):
f.unlink(missing_ok=True)
limit = "Retry limit reached. " * (i >= retry)
raise ConnectionError(emojis(f"❌ Download failure for {uri}. {limit}{e}")) from e
delay = 5 * 2**i # 5, 10, 20 s rides out the ~20-40 s HTTP 5xx bursts GitHub Releases returns
LOGGER.warning(f"Download failure, retrying {i + 1}/{retry} in {delay}s {uri}... {e}")
time.sleep(delay)
else: # no attempt reached `break`, so every one failed size validation and unlinked its download
raise ConnectionError(emojis(f"❌ Download failure for {uri}. Retry limit reached."))
if unzip and f.exists() and f.suffix in {"", ".zip", ".tar", ".gz", ".tgz", ".xz", ".bz2", ".txz", ".tbz2"}:
from zipfile import is_zipfile
unzip_dir = Path(dir or f.parent).resolve() # unzip to dir if provided else unzip in place
if is_zipfile(f):
unzip_dir = unzip_file(file=f, path=unzip_dir, exist_ok=exist_ok, progress=progress) # unzip
elif tarfile.is_tarfile(f):
LOGGER.info(f"Unzipping {f} to {unzip_dir}...")
top_level_dirs = set()
with tarfile.open(f, "r:*") as tar:
for m in tar:
if not (m.isfile() or m.isdir()) or m.issym() or m.islnk():
LOGGER.warning(f"Potentially insecure tar member: {m.name}, skipping extraction.")
continue
m_path = Path(m.name)
target = (unzip_dir / m_path).resolve()
if (
m_path.is_absolute()
or ".." in m_path.parts
or target.parts[: len(unzip_dir.parts)] != unzip_dir.parts
):
LOGGER.warning(f"Potentially insecure file path: {m.name}, skipping extraction.")
continue
top_level_dirs.update(m_path.parts[:1]) # slice as './' root entries have no parts
if m.isdir():
target.mkdir(parents=True, exist_ok=True)
elif source := tar.extractfile(m):
target.parent.mkdir(parents=True, exist_ok=True)
with source, open(target, "wb") as out: # 'f' is the archive path, deleted below
shutil.copyfileobj(source, out)
os.utime(target, (m.mtime, m.mtime)) # keep archive mtimes so re-extraction keeps caches valid
if len(top_level_dirs) == 1:
unzip_dir /= next(iter(top_level_dirs)) # return the single extracted file or directory
else:
unzip_dir = f # not a zip or tar, i.e. an HTML error page or plain gzip, return the file
if delete:
f.unlink() # remove archive
return unzip_dir
return fFunction ultralytics.utils.downloads.get_github_assets#
def get_github_assets(
repo: str = "ultralytics/assets",
version: str = "latest",
retry: bool = False,
) -> tuple[str, list[str]]Retrieve the specified version's tag and assets from a GitHub repository.
If the version is not specified, the function fetches the latest release assets.
Args
| Name | Type | Description | Default |
|---|---|---|---|
repo | str, optional | The GitHub repository in the format 'owner/repo'. | "ultralytics/assets" |
version | str, optional | The release version to fetch assets from. | "latest" |
retry | bool, optional | Flag to retry the request once in case of a failure. | False |
Returns
| Type | Description |
|---|---|
tag (str) | The release tag, or an empty string if the request fails. |
assets (list[str]) | A list of asset names, or an empty list if the request fails. |
Examples
>>> tag, assets = get_github_assets(repo="ultralytics/assets", version="latest")ultralytics/utils/downloads.py
def get_github_assets(
repo: str = "ultralytics/assets",
version: str = "latest",
retry: bool = False,
) -> tuple[str, list[str]]:
"""Retrieve the specified version's tag and assets from a GitHub repository.
If the version is not specified, the function fetches the latest release assets.
Args:
repo (str, optional): The GitHub repository in the format 'owner/repo'.
version (str, optional): The release version to fetch assets from.
retry (bool, optional): Flag to retry the request once in case of a failure.
Returns:
tag (str): The release tag, or an empty string if the request fails.
assets (list[str]): A list of asset names, or an empty list if the request fails.
Examples:
>>> tag, assets = get_github_assets(repo="ultralytics/assets", version="latest")
"""
import requests # scoped as slow import
if version != "latest":
version = f"tags/{version}" # i.e. tags/v6.2
url = f"https://api.github.com/repos/{repo}/releases/{version}"
attempts = 2 if retry else 1 # retry once on transient network errors or non-200 responses
for attempt in range(attempts):
try:
r = requests.get(url, timeout=30) # github api
except requests.exceptions.RequestException as e:
if attempt < attempts - 1:
continue # transient network error, try again
LOGGER.warning(f"GitHub assets check failure for {url}: {e}")
return "", []
if r.status_code == 200 or r.reason == "rate limit exceeded": # do not retry 403 rate limits
break
if r.status_code != 200:
LOGGER.warning(f"GitHub assets check failure for {url}: {r.status_code} {r.reason}")
return "", []
data = r.json()
return data["tag_name"], [x["name"] for x in data["assets"]] # tag, assets i.e. ['yolo26n.pt', 'yolo11s.pt', ...]Function ultralytics.utils.downloads.attempt_download_asset#
def attempt_download_asset(file: str | Path, repo: str = "ultralytics/assets", release: str = "v8.4.0", **kwargs) -> strAttempt to download a file from GitHub release assets if it is not found locally.
Args
| Name | Type | Description | Default |
|---|---|---|---|
file | str | Path | The filename or file path to be downloaded. | required |
repo | str, optional | The GitHub repository in the format 'owner/repo'. | "ultralytics/assets" |
release | str, optional | The specific release version to be downloaded. | "v8.4.0" |
**kwargs | Any | Additional keyword arguments passed to safe_download. | required |
Returns
| Type | Description |
|---|---|
str | The path to the local or downloaded file. |
Examples
>>> file_path = attempt_download_asset("yolo26n.pt", repo="ultralytics/assets")ultralytics/utils/downloads.py
def attempt_download_asset(
file: str | Path,
repo: str = "ultralytics/assets",
release: str = "v8.4.0",
**kwargs,
) -> str:
"""Attempt to download a file from GitHub release assets if it is not found locally.
Args:
file (str | Path): The filename or file path to be downloaded.
repo (str, optional): The GitHub repository in the format 'owner/repo'.
release (str, optional): The specific release version to be downloaded.
**kwargs (Any): Additional keyword arguments passed to `safe_download`.
Returns:
(str): The path to the local or downloaded file.
Examples:
>>> file_path = attempt_download_asset("yolo26n.pt", repo="ultralytics/assets")
"""
from ultralytics.utils import SETTINGS # scoped for circular import
# YOLOv3/5u updates
file = Path(checks.check_yolov5u_filename(str(file).strip().replace("'", "")))
if file.exists():
return str(file)
elif (SETTINGS["weights_dir"] / file).exists():
return str(SETTINGS["weights_dir"] / file)
else:
# URL specified
name = Path(parse.unquote(str(file))).name # decode '%2F' to '/' etc.
download_url = f"https://github.com/{repo}/releases/download"
if str(file).startswith(("http:/", "https:/")): # download
url = str(file).replace(":/", "://") # Pathlib turns :// -> :/
file = url2file(name) # parse authentication query strings
if Path(file).is_file():
LOGGER.info(f"Found {clean_url(url)} locally at {file}") # file already exists
else:
safe_download(url=url, file=file, min_bytes=1e5, **kwargs)
elif repo == GITHUB_ASSETS_REPO and name in GITHUB_ASSETS_NAMES:
safe_download(url=f"{download_url}/{release}/{name}", file=file, min_bytes=1e5, **kwargs)
else:
tag, assets = get_github_assets(repo, release)
if not assets:
tag, assets = get_github_assets(repo) # latest release
if name in assets:
safe_download(url=f"{download_url}/{tag}/{name}", file=file, min_bytes=1e5, **kwargs)
return str(file)Function ultralytics.utils.downloads.download#
def download(
url: str | list[str] | Path,
dir: str | Path | None = None,
unzip: bool = True,
delete: bool = False,
curl: bool = False,
threads: int = 1,
retry: int = 3,
exist_ok: bool = False,
) -> NoneDownload files from specified URLs to a given directory.
Supports concurrent downloads if multiple threads are specified.
Args
| Name | Type | Description | Default |
|---|---|---|---|
url | str | list[str] | Path | The URL or list of URLs of the files to be downloaded. | required |
dir | str | Path, optional | The directory where the files will be saved. Defaults to the current working directory. | None |
unzip | bool, optional | Flag to unzip the files after downloading. | True |
delete | bool, optional | Flag to delete the zip files after extraction. | False |
curl | bool, optional | Flag to use curl for downloading. | False |
threads | int, optional | Number of threads to use for concurrent downloads. | 1 |
retry | int, optional | Number of retries in case of download failure. | 3 |
exist_ok | bool, optional | Whether to overwrite existing contents during unzipping. | False |
Examples
>>> download("https://github.com/ultralytics/assets/releases/download/v0.0.0/bus.jpg", dir="path/to/dir")ultralytics/utils/downloads.py
def download(
url: str | list[str] | Path,
dir: str | Path | None = None,
unzip: bool = True,
delete: bool = False,
curl: bool = False,
threads: int = 1,
retry: int = 3,
exist_ok: bool = False,
) -> None:
"""Download files from specified URLs to a given directory.
Supports concurrent downloads if multiple threads are specified.
Args:
url (str | list[str] | Path): The URL or list of URLs of the files to be downloaded.
dir (str | Path, optional): The directory where the files will be saved. Defaults to the current working
directory.
unzip (bool, optional): Flag to unzip the files after downloading.
delete (bool, optional): Flag to delete the zip files after extraction.
curl (bool, optional): Flag to use curl for downloading.
threads (int, optional): Number of threads to use for concurrent downloads.
retry (int, optional): Number of retries in case of download failure.
exist_ok (bool, optional): Whether to overwrite existing contents during unzipping.
Examples:
>>> download("https://github.com/ultralytics/assets/releases/download/v0.0.0/bus.jpg", dir="path/to/dir")
"""
dir = Path(dir or Path.cwd())
dir.mkdir(parents=True, exist_ok=True) # make directory
urls = [url] if isinstance(url, (str, Path)) else url
if threads > 1:
LOGGER.info(f"Downloading {len(urls)} file(s) with {threads} threads to {dir}...")
with ThreadPool(threads) as pool:
pool.map(
lambda x: safe_download(
url=x[0],
dir=x[1],
unzip=unzip,
delete=delete,
curl=curl,
retry=retry,
exist_ok=exist_ok,
progress=True,
),
zip(urls, repeat(dir)),
)
pool.close()
pool.join()
else:
for u in urls:
safe_download(url=u, dir=dir, unzip=unzip, delete=delete, curl=curl, retry=retry, exist_ok=exist_ok)