summaryrefslogtreecommitdiff
path: root/upstream-layers/bitbake/lib/bb/fetch/gcp.py
blob: dd85b9fd2d1646eab5ccac1a2ca331f22bd52da3 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
"""
BitBake 'Fetch' implementation for Google Cloup Platform Storage.

Class for fetching files from Google Cloud Storage using the
Google Cloud Storage Python Client. The GCS Python Client must
be correctly installed, configured and authenticated prior to use.
Additionally, gsutil must also be installed.

"""

# Copyright (C) 2023, Snap Inc.
#
# Based in part on bb.fetch.s3:
#    Copyright (C) 2017 Andre McCurdy
#
# SPDX-License-Identifier: GPL-2.0-only
#
# Based on functions from the base bb module, Copyright 2003 Holger Schurig

import os
import bb
import urllib.parse, urllib.error
from bb.fetch import FetchMethod
from bb.fetch import FetchError
from bb.fetch import logger

class GCP(FetchMethod):
    """
    Class to fetch urls via GCP's Python API.
    """
    def __init__(self):
        self.gcp_client = None

    def supports(self, ud, d):
        """
        Check to see if a given url can be fetched with GCP.
        """
        return ud.type in ['gs']

    def recommends_checksum(self, urldata):
        return True

    def urldata_init(self, ud, d):
        if 'downloadfilename' in ud.parm:
            ud.basename = ud.parm['downloadfilename']
        else:
            ud.basename = os.path.basename(ud.path)

        ud.localfile = ud.basename

    def get_gcp_client(self):
        from google.cloud import storage
        self.gcp_client = storage.Client(project=None)

    def download(self, ud, d):
        """
        Fetch urls using the GCP API.
        Assumes localpath was called first.
        """
        from google.api_core.exceptions import GatewayTimeout, NotFound
        logger.debug2(f"Trying to download gs://{ud.host}{ud.path} to {ud.localpath}")
        if self.gcp_client is None:
            self.get_gcp_client()

        bb.fetch.check_network_access(d, "blob.download_to_filename", f"gs://{ud.host}{ud.path}")

        # Path sometimes has leading slash, so strip it
        path = ud.path.lstrip("/")
        blob = self.gcp_client.bucket(ud.host).blob(path)
        try:
            blob.download_to_filename(ud.localpath)
        except NotFound:
            raise FetchError("The GCP API threw a NotFound exception")
        except GatewayTimeout as e:
            # The GCS client already retries GatewayTimeout internally.
            # Raise FetchError so mirror fallback can proceed.
            logger.warning(
                f"GCP API GatewayTimeout while downloading gs://{ud.host}{ud.path}: {e}"
            )
            raise FetchError(f"Transient GCP API GatewayTimeout for gs://{ud.host}{ud.path}")

        # Additional sanity checks copied from the wget class (although there
        # are no known issues which mean these are required, treat the GCP API
        # tool with a little healthy suspicion).
        if not os.path.exists(ud.localpath):
            raise FetchError(f"The GCP API returned success for gs://{ud.host}{ud.path} but {ud.localpath} doesn't exist?!")

        if os.path.getsize(ud.localpath) == 0:
            os.remove(ud.localpath)
            raise FetchError(f"The downloaded file for gs://{ud.host}{ud.path} resulted in a zero size file?! Deleting and failing since this isn't right.")

        return True

    def checkstatus(self, fetch, ud, d):
        """
        Check the status of a URL.
        """
        from google.api_core.exceptions import GatewayTimeout

        logger.debug2(f"Checking status of gs://{ud.host}{ud.path}")
        if self.gcp_client is None:
            self.get_gcp_client()

        bb.fetch.check_network_access(d, "gcp_client.bucket(ud.host).blob(path).exists()", f"gs://{ud.host}{ud.path}")

        # Path sometimes has leading slash, so strip it
        path = ud.path.lstrip("/")
        try:
            exists = self.gcp_client.bucket(ud.host).blob(path).exists()
        except GatewayTimeout as e:
            # The GCS client already retries GatewayTimeout internally.
            # Surface a normal checkstatus failure and warn so the timeout
            # is visible to operators.
            logger.warning(
                f"GCP API GatewayTimeout while checking gs://{ud.host}{ud.path}; treating as unavailable: {e}"
            )
            raise FetchError(f"Transient GCP API GatewayTimeout for gs://{ud.host}{ud.path}")

        if exists == False:
            raise FetchError(f"The GCP API reported that gs://{ud.host}{ud.path} does not exist")
        else:
            return True