pip/src/pip/_internal/network/download.py

"""Download files with progress indicators.
"""
import email.message
import logging
import mimetypes
import os
from typing import Iterable, Optional, Tuple

from pip._vendor.requests.models import CONTENT_CHUNK_SIZE, Response

from pip._internal.cli.progress_bars import get_download_progress_renderer
from pip._internal.exceptions import NetworkConnectionError
from pip._internal.models.index import PyPI
from pip._internal.models.link import Link
from pip._internal.network.cache import is_from_cache
from pip._internal.network.session import PipSession
from pip._internal.network.utils import HEADERS, raise_for_status, response_chunks
from pip._internal.utils.misc import format_size, redact_auth_from_url, splitext

logger = logging.getLogger(__name__)


def _get_http_response_size(resp: Response) -> Optional[int]:
    try:
        return int(resp.headers["content-length"])
    except (ValueError, KeyError, TypeError):
        return None


def _prepare_download(
    resp: Response,
    link: Link,
    progress_bar: str,
) -> Iterable[bytes]:
    total_length = _get_http_response_size(resp)

    if link.netloc == PyPI.file_storage_domain:
        url = link.show_url
    else:
        url = link.url_without_fragment

    logged_url = redact_auth_from_url(url)

    if total_length:
        logged_url = f"{logged_url} ({format_size(total_length)})"

    if is_from_cache(resp):
        logger.info("Using cached %s", logged_url)
    else:
        logger.info("Downloading %s", logged_url)

    if logger.getEffectiveLevel() > logging.INFO:
        show_progress = False
    elif is_from_cache(resp):
        show_progress = False
    elif not total_length:
        show_progress = True
    elif total_length > (40 * 1000):
        show_progress = True
    else:
        show_progress = False

    chunks = response_chunks(resp, CONTENT_CHUNK_SIZE)

    if not show_progress:
        return chunks

    renderer = get_download_progress_renderer(bar_type=progress_bar, size=total_length)
    return renderer(chunks)


def sanitize_content_filename(filename: str) -> str:
    """
    Sanitize the "filename" value from a Content-Disposition header.
    """
    return os.path.basename(filename)


def parse_content_disposition(content_disposition: str, default_filename: str) -> str:
    """
    Parse the "filename" value from a Content-Disposition header, and
    return the default filename if the result is empty.
    """
    m = email.message.Message()
    m["content-type"] = content_disposition
    filename = m.get_param("filename")
    if filename:
        # We need to sanitize the filename to prevent directory traversal
        # in case the filename contains ".." path parts.
        filename = sanitize_content_filename(str(filename))
    return filename or default_filename


def _get_http_response_filename(resp: Response, link: Link) -> str:
    """Get an ideal filename from the given HTTP response, falling back to
    the link filename if not provided.
    """
    filename = link.filename  # fallback
    # Have a look at the Content-Disposition header for a better guess
    content_disposition = resp.headers.get("content-disposition")
    if content_disposition:
        filename = parse_content_disposition(content_disposition, filename)
    ext: Optional[str] = splitext(filename)[1]
    if not ext:
        ext = mimetypes.guess_extension(resp.headers.get("content-type", ""))
        if ext:
            filename += ext
    if not ext and link.url != resp.url:
        ext = os.path.splitext(resp.url)[1]
        if ext:
            filename += ext
    return filename


def _http_get_download(session: PipSession, link: Link) -> Response:
    target_url = link.url.split("#", 1)[0]
    resp = session.get(target_url, headers=HEADERS, stream=True)
    raise_for_status(resp)
    return resp


class Downloader:
    def __init__(
        self,
        session: PipSession,
        progress_bar: str,
    ) -> None:
        self._session = session
        self._progress_bar = progress_bar

    def __call__(self, link: Link, location: str) -> Tuple[str, str]:
        """Download the file given by link into location."""
        try:
            resp = _http_get_download(self._session, link)
        except NetworkConnectionError as e:
            assert e.response is not None
            logger.critical(
                "HTTP error %s while getting %s", e.response.status_code, link
            )
            raise

        filename = _get_http_response_filename(resp, link)
        filepath = os.path.join(location, filename)

        chunks = _prepare_download(resp, link, self._progress_bar)
        with open(filepath, "wb") as content_file:
            for chunk in chunks:
                content_file.write(chunk)
        content_type = resp.headers.get("Content-Type", "")
        return filepath, content_type


class BatchDownloader:
    def __init__(
        self,
        session: PipSession,
        progress_bar: str,
    ) -> None:
        self._session = session
        self._progress_bar = progress_bar

    def __call__(
        self, links: Iterable[Link], location: str
    ) -> Iterable[Tuple[Link, Tuple[str, str]]]:
        """Download the files given by links into location."""
        for link in links:
            try:
                resp = _http_get_download(self._session, link)
            except NetworkConnectionError as e:
                assert e.response is not None
                logger.critical(
                    "HTTP error %s while getting %s",
                    e.response.status_code,
                    link,
                )
                raise

            filename = _get_http_response_filename(resp, link)
            filepath = os.path.join(location, filename)

            chunks = _prepare_download(resp, link, self._progress_bar)
            with open(filepath, "wb") as content_file:
                for chunk in chunks:
                    content_file.write(chunk)
            content_type = resp.headers.get("Content-Type", "")
            yield link, (filepath, content_type)
Add network.download module This will be home to Dowloader, Download, and associated helper functions. Since this is an abstraction over PipSession, it makes sense to keep these functions in a separate module. Also move a helper function here from operations.prepare. 2019-11-29 17:24:26 +01:00			`"""Download files with progress indicators.`
			`"""`
Replace `cgi` module with `email.message` (#11098) 2022-05-24 13:36:56 +02:00			`import email.message`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00			`import logging`
Move _get_http_response_filename to network.download 2019-11-29 17:51:42 +01:00			`import mimetypes`
Move sanitize_content_filename to network.download 2019-11-29 17:43:40 +01:00			`import os`
Remove typing.TYPE_CHECKING guards The typing module has been available since Python 3.5. Guarding the import has been unnecessary since dropping Python 2. Some guards remain to either: - Avoid circular imports - Importing objects that are also guarded by typing.TYPE_CHECKING - Avoid mypy_extensions dependency 2021-02-19 13:56:59 +01:00			`from typing import Iterable, Optional, Tuple`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00
Remove typing.TYPE_CHECKING guards The typing module has been available since Python 3.5. Guarding the import has been unnecessary since dropping Python 2. Some guards remain to either: - Avoid circular imports - Importing objects that are also guarded by typing.TYPE_CHECKING - Avoid mypy_extensions dependency 2021-02-19 13:56:59 +01:00			`from pip._vendor.requests.models import CONTENT_CHUNK_SIZE, Response`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00
Use the progress bar from rich by default Utilise rich's progress bar to present download progress. This has a subjectively nicer presentation style and should degrade gracefully without additional effort from our end. 2021-11-06 09:30:37 +01:00			`from pip._internal.cli.progress_bars import get_download_progress_renderer`
feat(pip/_internal/*): Use custom raise_for_status method 2020-05-03 18:48:24 +02:00			`from pip._internal.exceptions import NetworkConnectionError`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00			`from pip._internal.models.index import PyPI`
Remove typing.TYPE_CHECKING guards The typing module has been available since Python 3.5. Guarding the import has been unnecessary since dropping Python 2. Some guards remain to either: - Avoid circular imports - Importing objects that are also guarded by typing.TYPE_CHECKING - Avoid mypy_extensions dependency 2021-02-19 13:56:59 +01:00			`from pip._internal.models.link import Link`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00			`from pip._internal.network.cache import is_from_cache`
Remove typing.TYPE_CHECKING guards The typing module has been available since Python 3.5. Guarding the import has been unnecessary since dropping Python 2. Some guards remain to either: - Avoid circular imports - Importing objects that are also guarded by typing.TYPE_CHECKING - Avoid mypy_extensions dependency 2021-02-19 13:56:59 +01:00			`from pip._internal.network.session import PipSession`
Prepare isort for black 2020-09-23 16:27:09 +02:00			`from pip._internal.network.utils import HEADERS, raise_for_status, response_chunks`
			`from pip._internal.utils.misc import format_size, redact_auth_from_url, splitext`
Add network.download module This will be home to Dowloader, Download, and associated helper functions. Since this is an abstraction over PipSession, it makes sense to keep these functions in a separate module. Also move a helper function here from operations.prepare. 2019-11-29 17:24:26 +01:00
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00			`logger = logging.getLogger(__name__)`

Add network.download module This will be home to Dowloader, Download, and associated helper functions. Since this is an abstraction over PipSession, it makes sense to keep these functions in a separate module. Also move a helper function here from operations.prepare. 2019-11-29 17:24:26 +01:00
Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`def _get_http_response_size(resp: Response) -> Optional[int]:`
Add network.download module This will be home to Dowloader, Download, and associated helper functions. Since this is an abstraction over PipSession, it makes sense to keep these functions in a separate module. Also move a helper function here from operations.prepare. 2019-11-29 17:24:26 +01:00			`try:`
			`return int(resp.headers["content-length"])`
			`except (ValueError, KeyError, TypeError):`
			`return None`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00

			`def _prepare_download(`
Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`resp: Response,`
			`link: Link,`
			`progress_bar: str,`
			`) -> Iterable[bytes]:`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00			`total_length = _get_http_response_size(resp)`

			`if link.netloc == PyPI.file_storage_domain:`
			`url = link.show_url`
			`else:`
			`url = link.url_without_fragment`

			`logged_url = redact_auth_from_url(url)`

			`if total_length:`
Enforce f-strings via Ruff (#12393) 2023-11-07 10:14:56 +01:00			`logged_url = f"{logged_url} ({format_size(total_length)})"`
Move _prepare_download to network.download 2019-11-29 17:41:14 +01:00
			`if is_from_cache(resp):`
			`logger.info("Using cached %s", logged_url)`
			`else:`
			`logger.info("Downloading %s", logged_url)`

			`if logger.getEffectiveLevel() > logging.INFO:`
			`show_progress = False`
			`elif is_from_cache(resp):`
			`show_progress = False`
			`elif not total_length:`
			`show_progress = True`
			`elif total_length > (40 * 1000):`
			`show_progress = True`
			`else:`
			`show_progress = False`

			`chunks = response_chunks(resp, CONTENT_CHUNK_SIZE)`

			`if not show_progress:`
			`return chunks`

Use the progress bar from rich by default Utilise rich's progress bar to present download progress. This has a subjectively nicer presentation style and should degrade gracefully without additional effort from our end. 2021-11-06 09:30:37 +01:00			`renderer = get_download_progress_renderer(bar_type=progress_bar, size=total_length)`
			`return renderer(chunks)`
Move sanitize_content_filename to network.download 2019-11-29 17:43:40 +01:00

Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`def sanitize_content_filename(filename: str) -> str:`
Move sanitize_content_filename to network.download 2019-11-29 17:43:40 +01:00			`"""`
			`Sanitize the "filename" value from a Content-Disposition header.`
			`"""`
			`return os.path.basename(filename)`
Move parse_content_disposition to network.download 2019-11-29 17:48:40 +01:00

Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`def parse_content_disposition(content_disposition: str, default_filename: str) -> str:`
Move parse_content_disposition to network.download 2019-11-29 17:48:40 +01:00			`"""`
			`Parse the "filename" value from a Content-Disposition header, and`
			`return the default filename if the result is empty.`
			`"""`
Replace `cgi` module with `email.message` (#11098) 2022-05-24 13:36:56 +02:00			`m = email.message.Message()`
			`m["content-type"] = content_disposition`
			`filename = m.get_param("filename")`
Move parse_content_disposition to network.download 2019-11-29 17:48:40 +01:00			`if filename:`
			`# We need to sanitize the filename to prevent directory traversal`
			`# in case the filename contains ".." path parts.`
Replace `cgi` module with `email.message` (#11098) 2022-05-24 13:36:56 +02:00			`filename = sanitize_content_filename(str(filename))`
Move parse_content_disposition to network.download 2019-11-29 17:48:40 +01:00			`return filename or default_filename`
Move _get_http_response_filename to network.download 2019-11-29 17:51:42 +01:00

Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`def _get_http_response_filename(resp: Response, link: Link) -> str:`
Move _get_http_response_filename to network.download 2019-11-29 17:51:42 +01:00			`"""Get an ideal filename from the given HTTP response, falling back to`
			`the link filename if not provided.`
			`"""`
			`filename = link.filename # fallback`
			`# Have a look at the Content-Disposition header for a better guess`
			`content_disposition = resp.headers.get("content-disposition")`
			`if content_disposition:`
			`filename = parse_content_disposition(content_disposition, filename)`
Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`ext: Optional[str] = splitext(filename)[1]`
Move _get_http_response_filename to network.download 2019-11-29 17:51:42 +01:00			`if not ext:`
			`ext = mimetypes.guess_extension(resp.headers.get("content-type", ""))`
			`if ext:`
			`filename += ext`
			`if not ext and link.url != resp.url:`
			`ext = os.path.splitext(resp.url)[1]`
			`if ext:`
			`filename += ext`
			`return filename`
Move _http_get_download to network.download 2019-11-29 17:53:59 +01:00

Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`def _http_get_download(session: PipSession, link: Link) -> Response:`
Move _http_get_download to network.download 2019-11-29 17:53:59 +01:00			`target_url = link.url.split("#", 1)[0]`
Draft lazy zip over HTTP 2020-06-18 18:10:00 +02:00			`resp = session.get(target_url, headers=HEADERS, stream=True)`
feat(pip/_internal/*): Use custom raise_for_status method 2020-05-03 18:48:24 +02:00			`raise_for_status(resp)`
Move _http_get_download to network.download 2019-11-29 17:53:59 +01:00			`return resp`
Move Downloader to network.download This will help us move Downloader construction out of RequirementPreparer, reducing its concerns and making it easier to test in isolation. 2019-11-29 17:56:18 +01:00

Remove object from class definitions Unnecessary since dropping Python 2 support. In Python 3, all classes are new style classes. 2020-12-24 22:23:07 +01:00			`class Downloader:`
Move Downloader to network.download This will help us move Downloader construction out of RequirementPreparer, reducing its concerns and making it easier to test in isolation. 2019-11-29 17:56:18 +01:00			`def __init__(`
			`self,`
Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`session: PipSession,`
			`progress_bar: str,`
			`) -> None:`
Move Downloader to network.download This will help us move Downloader construction out of RequirementPreparer, reducing its concerns and making it easier to test in isolation. 2019-11-29 17:56:18 +01:00			`self._session = session`
			`self._progress_bar = progress_bar`

Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`def __call__(self, link: Link, location: str) -> Tuple[str, str]:`
Add memoization mechanism for file download This is intentionally dependent from caching, which relies on cache dir. 2020-08-09 17:44:20 +02:00			`"""Download the file given by link into location."""`
Move Downloader to network.download This will help us move Downloader construction out of RequirementPreparer, reducing its concerns and making it easier to test in isolation. 2019-11-29 17:56:18 +01:00			`try:`
			`resp = _http_get_download(self._session, link)`
feat(pip/_internal/*): Use custom raise_for_status method 2020-05-03 18:48:24 +02:00			`except NetworkConnectionError as e:`
reformat(pip/_internal): Resolve lint errors 2020-07-09 09:14:21 +02:00			`assert e.response is not None`
Move Downloader to network.download This will help us move Downloader construction out of RequirementPreparer, reducing its concerns and making it easier to test in isolation. 2019-11-29 17:56:18 +01:00			`logger.critical(`
			`"HTTP error %s while getting %s", e.response.status_code, link`
			`)`
			`raise`

Make Downloader perform the download 2020-08-09 10:55:33 +02:00			`filename = _get_http_response_filename(resp, link)`
Clean up code style and internal interface Co-Authored-By: Pradyun Gedam <pradyunsg@gmail.com> Co-Authored-By: Chris Hunt <chrahunt@gmail.com> 2020-08-10 17:24:00 +02:00			`filepath = os.path.join(location, filename)`

Make Downloader perform the download 2020-08-09 10:55:33 +02:00			`chunks = _prepare_download(resp, link, self._progress_bar)`
Clean up code style and internal interface Co-Authored-By: Pradyun Gedam <pradyunsg@gmail.com> Co-Authored-By: Chris Hunt <chrahunt@gmail.com> 2020-08-10 17:24:00 +02:00			`with open(filepath, "wb") as content_file:`
Make Downloader perform the download 2020-08-09 10:55:33 +02:00			`for chunk in chunks:`
			`content_file.write(chunk)`
Clean up code style and internal interface Co-Authored-By: Pradyun Gedam <pradyunsg@gmail.com> Co-Authored-By: Chris Hunt <chrahunt@gmail.com> 2020-08-10 17:24:00 +02:00			`content_type = resp.headers.get("Content-Type", "")`
			`return filepath, content_type`
Add memoization mechanism for file download This is intentionally dependent from caching, which relies on cache dir. 2020-08-09 17:44:20 +02:00
Give batch downloader a separate class 2020-08-11 17:56:37 +02:00
Remove object from class definitions Unnecessary since dropping Python 2 support. In Python 3, all classes are new style classes. 2020-12-24 22:23:07 +01:00			`class BatchDownloader:`
Give batch downloader a separate class 2020-08-11 17:56:37 +02:00			`def __init__(`
			`self,`
Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`session: PipSession,`
			`progress_bar: str,`
			`) -> None:`
Give batch downloader a separate class 2020-08-11 17:56:37 +02:00			`self._session = session`
			`self._progress_bar = progress_bar`

Complete type annotations in `pip/_internal/network` (#10184) 2021-07-23 12:27:28 +02:00			`def __call__(`
			`self, links: Iterable[Link], location: str`
			`) -> Iterable[Tuple[Link, Tuple[str, str]]]:`
Add memoization mechanism for file download This is intentionally dependent from caching, which relies on cache dir. 2020-08-09 17:44:20 +02:00			`"""Download the files given by links into location."""`
			`for link in links:`
Give batch downloader a separate class 2020-08-11 17:56:37 +02:00			`try:`
			`resp = _http_get_download(self._session, link)`
			`except NetworkConnectionError as e:`
			`assert e.response is not None`
			`logger.critical(`
			`"HTTP error %s while getting %s",`
			`e.response.status_code,`
			`link,`
			`)`
			`raise`

			`filename = _get_http_response_filename(resp, link)`
			`filepath = os.path.join(location, filename)`

			`chunks = _prepare_download(resp, link, self._progress_bar)`
			`with open(filepath, "wb") as content_file:`
			`for chunk in chunks:`
			`content_file.write(chunk)`
			`content_type = resp.headers.get("Content-Type", "")`
download requirements in the download command, outside of the resolver create PartialRequirementDownloadCompleter, and use in wheel, install, and download add NEWS entry rename NEWS entry rename NEWS entry respond to review comments move the partial requirement download completion to the bottom of the prepare_more method 2020-09-22 10:26:43 +02:00			`yield link, (filepath, content_type)`