From 33ce7da4d524358c30824ce6197de22d347971bb Mon Sep 17 00:00:00 2001 From: Mingfei Shao Date: Tue, 18 Aug 2026 11:40:26 -0500 Subject: [PATCH 1/3] fix drs-pull progress bar display behavior --- gen3/cli/drs_pull.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/gen3/cli/drs_pull.py b/gen3/cli/drs_pull.py index 69bab4e0..31372a5a 100644 --- a/gen3/cli/drs_pull.py +++ b/gen3/cli/drs_pull.py @@ -142,7 +142,7 @@ def download_object( ctx.obj["auth_factory"].get(), [object_id], output_dir, - no_progress, + not no_progress, not no_unpack_packages, delete_unpacked_packages, ctx.obj["commons_url"], @@ -226,7 +226,7 @@ def download_objects( ctx.obj["auth_factory"].get(), object_ids, output_dir, - no_progress, + not no_progress, not no_unpack_packages, delete_unpacked_packages, ctx.obj["commons_url"], From 7d6aa637197f0b791fe1c14c4b3f8a5a223ad6a7 Mon Sep 17 00:00:00 2001 From: mfshao Date: Tue, 18 Aug 2026 16:41:44 +0000 Subject: [PATCH 2/3] Apply automatic documentation changes --- docs/_build/html/_modules/gen3/auth.html | 348 ++++++------ docs/_build/html/_modules/gen3/file.html | 104 ++-- docs/_build/html/_modules/gen3/index.html | 426 +++++++-------- docs/_build/html/_modules/gen3/jobs.html | 120 ++-- docs/_build/html/_modules/gen3/metadata.html | 428 +++++++-------- docs/_build/html/_modules/gen3/object.html | 32 +- docs/_build/html/_modules/gen3/query.html | 96 ++-- .../_build/html/_modules/gen3/submission.html | 378 ++++++------- .../gen3/tools/download/drs_download.html | 512 +++++++++--------- .../tools/indexing/download_manifest.html | 242 ++++----- .../gen3/tools/indexing/index_manifest.html | 258 ++++----- .../gen3/tools/indexing/verify_manifest.html | 238 ++++---- .../gen3/tools/metadata/ingest_manifest.html | 206 +++---- docs/_build/html/_modules/gen3/wss.html | 98 ++-- docs/_build/html/auth.html | 8 +- docs/_build/html/file.html | 2 +- docs/_build/html/indexing.html | 2 +- docs/_build/html/jobs.html | 2 +- docs/_build/html/metadata.html | 2 +- docs/_build/html/object.html | 2 +- docs/_build/html/query.html | 28 +- docs/_build/html/submission.html | 36 +- docs/_build/html/tools/drs_pull.html | 8 +- docs/_build/html/wss.html | 2 +- 24 files changed, 1789 insertions(+), 1789 deletions(-) diff --git a/docs/_build/html/_modules/gen3/auth.html b/docs/_build/html/_modules/gen3/auth.html index bac0cdd7..7f88fdc8 100644 --- a/docs/_build/html/_modules/gen3/auth.html +++ b/docs/_build/html/_modules/gen3/auth.html @@ -58,35 +58,35 @@

Source code for gen3.auth

 
 
 def decode_token(token_str):
-    """
-    jq -r '.api_key' < ~/.gen3/qa-covid19.planx-pla.net.json | awk -F . '{ print $2 }' | base64 --decode | jq -r .
-    """
-    tokenParts = token_str.split(".")
+    """
+    jq -r '.api_key' < ~/.gen3/qa-covid19.planx-pla.net.json | awk -F . '{ print $2 }' | base64 --decode | jq -r .
+    """
+    tokenParts = token_str.split(".")
     if len(tokenParts) < 3:
-        raise Exception("Invalid JWT. Could not split into parts.")
-    padding = "===="
+        raise Exception("Invalid JWT. Could not split into parts.")
+    padding = "===="
     infoStr = tokenParts[1] + padding[0 : len(tokenParts[1]) % 4]
     jsonStr = base64.urlsafe_b64decode(infoStr)
     return json.loads(jsonStr)
 
 
 def endpoint_from_token(token_str):
-    """
-    Extract the endpoint from a JWT issue ("iss" property)
-    """
+    """
+    Extract the endpoint from a JWT issue ("iss" property)
+    """
     info = decode_token(token_str)
-    urlparts = urlparse(info["iss"])
-    endpoint = urlparts.scheme + "://" + urlparts.hostname
+    urlparts = urlparse(info["iss"])
+    endpoint = urlparts.scheme + "://" + urlparts.hostname
     if urlparts.port:
-        endpoint += ":" + str(urlparts.port)
+        endpoint += ":" + str(urlparts.port)
     return remove_trailing_whitespace_and_slashes_in_url(endpoint)
 
 
 def _handle_access_token_response(resp, token_key):
-    """
+    """
     Shared helper for both get_access_token_with_key and get_access_token_from_wts
-    """
-    err_msg = "Failed to get an access token from {}:\n{}"
+    """
+    err_msg = "Failed to get an access token from {}:\n{}"
     if resp.status_code != 200:
         raise Gen3AuthError(err_msg.format(resp.url, resp.text))
     try:
@@ -99,64 +99,64 @@ 

Source code for gen3.auth

 
 
 def get_access_token_with_key(api_key):
-    """
+    """
     Try to fetch an access token given the api key
-    """
-    endpoint = endpoint_from_token(api_key["api_key"])
+    """
+    endpoint = endpoint_from_token(api_key["api_key"])
     # attempt to get a token from Fence
-    auth_url = "{}/user/credentials/cdis/access_token".format(endpoint)
+    auth_url = "{}/user/credentials/cdis/access_token".format(endpoint)
     resp = requests.post(auth_url, json=api_key)
-    token_key = "access_token"
+    token_key = "access_token"
     return _handle_access_token_response(resp, token_key)
 
 
 def get_access_token_with_client_credentials(endpoint, client_credentials, scopes):
-    """
+    """
     Try to get an access token from Fence using client credentials
 
     Args:
         endpoint (str): URL of the Gen3 instance to get an access token for
         client_credentials ((str, str) tuple): (client ID, client secret) tuple
         scopes (str): space-delimited list of scopes to request
-    """
+    """
     if not endpoint:
-        raise ValueError("'endpoint' must be specified when using client credentials")
-    url = f"{endpoint}/user/oauth2/token?grant_type=client_credentials&scope={scopes}"
+        raise ValueError("'endpoint' must be specified when using client credentials")
+    url = f"{endpoint}/user/oauth2/token?grant_type=client_credentials&scope={scopes}"
     resp = requests.post(url, auth=client_credentials)
-    return _handle_access_token_response(resp, "access_token")
+    return _handle_access_token_response(resp, "access_token")
 
 
-def get_wts_endpoint(namespace=os.getenv("NAMESPACE", "default")):
-    return "http://workspace-token-service.{}.svc.cluster.local".format(namespace)
+def get_wts_endpoint(namespace=os.getenv("NAMESPACE", "default")):
+    return "http://workspace-token-service.{}.svc.cluster.local".format(namespace)
 
 
-def get_wts_idps(namespace=os.getenv("NAMESPACE", "default"), external_wts_host=None):
+def get_wts_idps(namespace=os.getenv("NAMESPACE", "default"), external_wts_host=None):
     wts_url = None
     if external_wts_host == None:
         wts_url = get_wts_endpoint(namespace)
     else:
         wts_url = external_wts_host
-    url = wts_url.rstrip("/") + "/external_oidc/"
+    url = wts_url.rstrip("/") + "/external_oidc/"
     resp = requests.get(url)
     raise_for_status_and_print_error(resp)
     return resp.json()
 
 
 def get_token_cache_file_name(key):
-    """Compute the path to the access-token cache file"""
-    cache_folder = "{}/.cache/gen3/".format(os.path.expanduser("~"))
+    """Compute the path to the access-token cache file"""
+    cache_folder = "{}/.cache/gen3/".format(os.path.expanduser("~"))
     os.makedirs(cache_folder, exist_ok=True)
 
-    cache_prefix = cache_folder + "token_cache_"
+    cache_prefix = cache_folder + "token_cache_"
     s = hashlib.sha256()
-    s.update(key.encode("utf-8"))
+    s.update(key.encode("utf-8"))
     return cache_prefix + s.hexdigest()
 
 
 
[docs] class Gen3Auth(AuthBase): - """Gen3 auth helper class for use with requests auth. + """Gen3 auth helper class for use with requests auth. Implements requests.auth.AuthBase in order to support JWT authentication. Generates access tokens from the provided refresh token file or string. @@ -164,17 +164,17 @@

Source code for gen3.auth

 
     Args:
         refresh_file (str, opt): The file containing the downloaded JSON web token. Optional if working in a Gen3 Workspace.
-                Defaults to (env["GEN3_API_KEY"] || "credentials") if refresh_token and idp not set.
+                Defaults to (env["GEN3_API_KEY"] || "credentials") if refresh_token and idp not set.
                 Includes ~/.gen3/ in search path if value does not include /.
-                Interprets "idp://wts/<idp>" as an idp.
-                Interprets "accesstoken:///<token>" as an access token
+                Interprets "idp://wts/<idp>" as an idp.
+                Interprets "accesstoken:///<token>" as an access token
         refresh_token (str, opt): The JSON web token. Optional if working in a Gen3 Workspace.
         idp (str, opt): If working in a Gen3 Workspace, the IDP to use can be specified -
-                "local" indicates the local environment fence idp
+                "local" indicates the local environment fence idp
         client_credentials (tuple, opt): The (client_id, client_secret) credentials for an OIDC client
-                that has the 'client_credentials' grant, allowing it to obtain access tokens.
+                that has the 'client_credentials' grant, allowing it to obtain access tokens.
         client_scopes (str, opt): Space-separated list of scopes requested for access tokens obtained from client
-                credentials. Default: "user data openid"
+                credentials. Default: "user data openid"
         access_token (str, opt): provide an access token to override the use of any
                 API key/refresh token. This is intended for cases where you may want to
                 pass a token that was issued to a particular OIDC client (rather than acting on
@@ -189,30 +189,30 @@ 

Source code for gen3.auth

 
         or use ~/.gen3/crdc.json:
 
-        >>> auth = Gen3Auth(refresh_file="crdc")
+        >>> auth = Gen3Auth(refresh_file="crdc")
 
         or use some arbitrary file:
 
-        >>> auth = Gen3Auth(refresh_file="./key.json")
+        >>> auth = Gen3Auth(refresh_file="./key.json")
 
         or set the GEN3_API_KEY environment variable rather
         than pass the refresh_file argument to the Gen3Auth
         constructor.
 
-        If working with an OIDC client that has the 'client_credentials' grant, allowing it to obtain
+        If working with an OIDC client that has the 'client_credentials' grant, allowing it to obtain
         access tokens, provide the client ID and secret:
 
         Note: client secrets should never be hardcoded!
 
         >>> auth = Gen3Auth(
-            endpoint="https://datacommons.example",
-            client_credentials=("client ID", os.environ["GEN3_OIDC_CLIENT_CREDS_SECRET"])
+            endpoint="https://datacommons.example",
+            client_credentials=("client ID", os.environ["GEN3_OIDC_CLIENT_CREDS_SECRET"])
         )
 
         If working in a Gen3 Workspace, initialize as follows:
 
         >>> auth = Gen3Auth()
-    """
+    """
 
     def __init__(
         self,
@@ -224,40 +224,40 @@ 

Source code for gen3.auth

         client_scopes=None,
         access_token=None,
     ):
-        logging.debug("Initializing auth..")
+        logging.debug("Initializing auth..")
         self.endpoint = remove_trailing_whitespace_and_slashes_in_url(endpoint)
-        # note - `_refresh_token` is not actually a JWT refresh token - it's a
-        #  gen3 api key with a token as the "api_key" property
+        # note - `_refresh_token` is not actually a JWT refresh token - it's a
+        #  gen3 api key with a token as the "api_key" property
         self._refresh_token = refresh_token
         self._access_token = access_token
         self._access_token_info = None
-        self._wts_idp = idp or "local"
-        self._wts_namespace = os.environ.get("NAMESPACE", "default")
+        self._wts_idp = idp or "local"
+        self._wts_namespace = os.environ.get("NAMESPACE", "default")
         self._use_wts = False
         self._external_wts_host = None
         self._refresh_file = refresh_file
         self._client_credentials = client_credentials
         if self._client_credentials:
-            self._client_scopes = client_scopes or "user data openid"
+            self._client_scopes = client_scopes or "user data openid"
         elif client_scopes:
             raise ValueError(
-                "'client_scopes' cannot be specified without 'client_credentials'"
+                "'client_scopes' cannot be specified without 'client_credentials'"
             )
 
         if refresh_file and refresh_token:
             raise ValueError(
-                "Only one of 'refresh_file' and 'refresh_token' can be specified."
+                "Only one of 'refresh_file' and 'refresh_token' can be specified."
             )
 
         if endpoint and idp:
-            raise ValueError("Only one of 'endpoint' and 'idp' can be specified.")
+            raise ValueError("Only one of 'endpoint' and 'idp' can be specified.")
 
         if not refresh_file and not refresh_token and not idp:
-            refresh_file = os.getenv("GEN3_API_KEY", "credentials")
+            refresh_file = os.getenv("GEN3_API_KEY", "credentials")
 
         if refresh_file and not idp:
-            idp_prefix = "idp://wts/"
-            access_token_prefix = "accesstoken:///"
+            idp_prefix = "idp://wts/"
+            access_token_prefix = "accesstoken:///"
             if refresh_file[0 : len(idp_prefix)] == idp_prefix:
                 idp = refresh_file[len(idp_prefix) :]
                 refresh_file = None
@@ -267,22 +267,22 @@ 

Source code for gen3.auth

                 refresh_file = None
             elif (
                 not os.path.isfile(refresh_file)
-                and "/" not in refresh_file
-                and "\\" not in refresh_file
+                and "/" not in refresh_file
+                and "\\" not in refresh_file
             ):
-                refresh_file = "{}/.gen3/{}".format(
-                    os.path.expanduser("~"), refresh_file
+                refresh_file = "{}/.gen3/{}".format(
+                    os.path.expanduser("~"), refresh_file
                 )
-                if not os.path.isfile(refresh_file) and refresh_file[-5:] != ".json":
-                    refresh_file += ".json"
+                if not os.path.isfile(refresh_file) and refresh_file[-5:] != ".json":
+                    refresh_file += ".json"
                 if not os.path.isfile(refresh_file):
-                    logging.warning("Unable to find refresh_file")
+                    logging.warning("Unable to find refresh_file")
                     refresh_file = None
 
         if self._client_credentials:
             if not endpoint:
                 raise ValueError(
-                    "'endpoint' must be specified when '_client_credentials' is specified"
+                    "'endpoint' must be specified when '_client_credentials' is specified"
                 )
             self._access_token = get_access_token_with_client_credentials(
                 endpoint, self._client_credentials, self._client_scopes
@@ -292,7 +292,7 @@ 

Source code for gen3.auth

             # at this point - refresh_file either exists or is None
             if not refresh_file and not refresh_token:
                 # check if this is a Gen3 workspace environment
-                # most production environments are in the "default" namespace
+                # most production environments are in the "default" namespace
                 # attempt to get a token from the workspace-token-service
                 self._use_wts = True
                 # hate calling a method from the constructor, but avoids copying code
@@ -304,34 +304,34 @@ 

Source code for gen3.auth

                     self._refresh_token = json.loads(file_data)
                 except Exception as e:
                     raise ValueError(
-                        "Couldn't load your refresh token file: {}\n{}".format(
+                        "Couldn't load your refresh token file: {}\n{}".format(
                             refresh_file, str(e)
                         )
                     )
 
-                assert "api_key" in self._refresh_token
+                assert "api_key" in self._refresh_token
                 # if both endpoint and refresh file are provided, compare endpoint with iss in refresh file
                 # May need to use network wts endpoint
                 if idp or (
                     endpoint
                     and (
-                        not endpoint.rstrip("/")
-                        == endpoint_from_token(self._refresh_token["api_key"])
+                        not endpoint.rstrip("/")
+                        == endpoint_from_token(self._refresh_token["api_key"])
                     )
                 ):
                     try:
                         logging.debug(
-                            "Switch to using WTS and set external WTS host url.."
+                            "Switch to using WTS and set external WTS host url.."
                         )
                         self._use_wts = True
                         self._external_wts_host = (
-                            endpoint_from_token(self._refresh_token["api_key"])
-                            + "/wts/"
+                            endpoint_from_token(self._refresh_token["api_key"])
+                            + "/wts/"
                         )
                         self.get_access_token()
                     except Gen3AuthError as g:
                         logging.warning(
-                            "Could not obtain access token from WTS service."
+                            "Could not obtain access token from WTS service."
                         )
                         raise g
 
@@ -339,20 +339,20 @@ 

Source code for gen3.auth

             if self._access_token:
                 self.endpoint = endpoint_from_token(self._access_token)
             else:
-                self.endpoint = endpoint_from_token(self._refresh_token["api_key"])
+                self.endpoint = endpoint_from_token(self._refresh_token["api_key"])
 
     @property
     def _token_info(self):
-        """
+        """
         Wrapper to fix intermittent errors when the token is being refreshed
         and `_access_token_info` == None
-        """
+        """
         if not self._access_token_info:
             self.refresh_access_token()
         return self._access_token_info
 
     def __call__(self, request):
-        """Adds authorization header to the request
+        """Adds authorization header to the request
 
         This gets called by the python.requests package on outbound requests
         so that authentication can be added.
@@ -360,13 +360,13 @@ 

Source code for gen3.auth

         Args:
             request (object): The incoming request object
 
-        """
-        request.headers["Authorization"] = self._get_auth_value()
-        request.register_hook("response", self._handle_401)
+        """
+        request.headers["Authorization"] = self._get_auth_value()
+        request.register_hook("response", self._handle_401)
         return request
 
     def _handle_401(self, response, **kwargs):
-        """Handles failed requests when authorization failed.
+        """Handles failed requests when authorization failed.
 
         This gets called after a failed request when an HTTP 401 error
         occurs. This then tries to refresh the access token in the event
@@ -375,7 +375,7 @@ 

Source code for gen3.auth

         Args:
             request (object): The failed request object
 
-        """
+        """
         if not response.status_code == 401 and not response.status_code == 403:
             return response
 
@@ -386,7 +386,7 @@ 

Source code for gen3.auth

         # copy the request to resend
         newreq = response.request.copy()
 
-        newreq.headers["Authorization"] = self._get_auth_value()
+        newreq.headers["Authorization"] = self._get_auth_value()
 
         _response = response.connection.send(newreq, **kwargs)
         _response.history.append(response)
@@ -397,7 +397,7 @@ 

Source code for gen3.auth

 
[docs] def refresh_access_token(self, endpoint=None): - """Get a new access token""" + """Get a new access token""" if self._use_wts: self._access_token = self.get_access_token_from_wts(endpoint) elif self._client_credentials: @@ -408,8 +408,8 @@

Source code for gen3.auth

             self._access_token = get_access_token_with_key(self._refresh_token)
         else:
             logging.warning(
-                f"Unable to refresh access token. "
-                f"Authorized API calls will stop working when this token expires."
+                f"Unable to refresh access token. "
+                f"Authorized API calls will stop working when this token expires."
             )
 
         self._access_token_info = decode_token(self._access_token)
@@ -418,14 +418,14 @@ 

Source code for gen3.auth

         if self._use_wts:
             cache_file = get_token_cache_file_name(self._wts_idp)
         elif self._refresh_file:
-            cache_file = get_token_cache_file_name(self._refresh_token["api_key"])
+            cache_file = get_token_cache_file_name(self._refresh_token["api_key"])
 
         if cache_file:
             try:
                 self._write_to_file(cache_file, self._access_token)
             except Exception as e:
                 logging.warning(
-                    f"Unable to write access token to cache file. Exceeded number of retries. Details: {e}"
+                    f"Unable to write access token to cache file. Exceeded number of retries. Details: {e}"
                 )
 
         return self._access_token
@@ -438,37 +438,37 @@

Source code for gen3.auth

         # write a temp file, then rename - to avoid
         # simultaneous writes to same file race condition
         temp = cache_file + (
-            ".tmp_eraseme_%d_%d" % (random.randrange(100000), time.time())
+            ".tmp_eraseme_%d_%d" % (random.randrange(100000), time.time())
         )
         try:
-            with open(temp, "w") as f:
+            with open(temp, "w") as f:
                 f.write(content)
             os.rename(temp, cache_file)
             return True
         except Exception as e:
-            logging.warning("failed to write token cache file: " + cache_file)
+            logging.warning("failed to write token cache file: " + cache_file)
             logging.warning(str(e))
             raise e
 
 
[docs] def get_access_token(self): - """Get the access token - auto refresh if within 5 minutes of expiration""" + """Get the access token - auto refresh if within 5 minutes of expiration""" if not self._access_token: if self._use_wts == True: cache_file = get_token_cache_file_name(self._wts_idp) else: if self._refresh_token: cache_file = get_token_cache_file_name( - self._refresh_token["api_key"] + self._refresh_token["api_key"] ) if cache_file and os.path.isfile(cache_file): - try: # don't freak out on invalid cache + try: # don't freak out on invalid cache with open(cache_file) as f: self._access_token = f.read() self._access_token_info = decode_token(self._access_token) except Exception as e: - logging.warning("ignoring invalid token cache: " + cache_file) + logging.warning("ignoring invalid token cache: " + cache_file) self._access_token = None self._access_token_info = None logging.warning(str(e)) @@ -476,108 +476,108 @@

Source code for gen3.auth

         need_new_token = (
             not self._access_token
             or not self._access_token_info
-            or time.time() + 300 > self._access_token_info["exp"]
+            or time.time() + 300 > self._access_token_info["exp"]
         )
         if need_new_token:
             return self.refresh_access_token(
-                self.endpoint if hasattr(self, "endpoint") else None
+                self.endpoint if hasattr(self, "endpoint") else None
             )
         # use cache
         return self._access_token
def _get_auth_value(self): - """Returns the Authorization header value for the request + """Returns the Authorization header value for the request This gets called when added the Authorization header to the request. This fetches the access token from the refresh token if the access token is missing. - """ - return "bearer " + self.get_access_token() + """ + return "bearer " + self.get_access_token()
[docs] def curl(self, path, request=None, data=None): - """ + """ Curl the given endpoint - ex: gen3 curl /user/user. Return requests.Response Args: path (str): path under the commons to curl (/user/user, /index/index, /authz/mapping, ...) request (str in GET|POST|PUT|DELETE): default to GET if data is not set, else default to POST - data (str): json string or "@filename" of a json file - """ + data (str): json string or "@filename" of a json file + """ if not request: - request = "GET" + request = "GET" if data: - request = "POST" + request = "POST" json_data = data output = None - if data and data[0] == "@": + if data and data[0] == "@": with open(data[1:]) as f: json_data = f.read() - if request == "GET": - output = requests.get(self.endpoint + "/" + path, auth=self) - elif request == "POST": + if request == "GET": + output = requests.get(self.endpoint + "/" + path, auth=self) + elif request == "POST": output = requests.post( - self.endpoint + "/" + path, json=json_data, auth=self + self.endpoint + "/" + path, json=json_data, auth=self ) - elif request == "PUT": - output = requests.put(self.endpoint + "/" + path, json=json_data, auth=self) - elif request == "DELETE": - output = requests.delete(self.endpoint + "/" + path, auth=self) + elif request == "PUT": + output = requests.put(self.endpoint + "/" + path, json=json_data, auth=self) + elif request == "DELETE": + output = requests.delete(self.endpoint + "/" + path, auth=self) else: - raise Exception("Invalid request type: " + request) + raise Exception("Invalid request type: " + request) return output
[docs] def get_access_token_from_wts(self, endpoint=None): - """ + """ Try to fetch an access token for the given idp from the wts - in the given namespace. If idp is not set, then default to "local" - """ + in the given namespace. If idp is not set, then default to "local" + """ # attempt to get a token from the workspace-token-service - logging.debug("getting access token from wts..") - auth_url = get_wts_endpoint(self._wts_namespace) + "/token/" + logging.debug("getting access token from wts..") + auth_url = get_wts_endpoint(self._wts_namespace) + "/token/" - # If non "local" idp value exists, append to auth url + # If non "local" idp value exists, append to auth url # If user specified endpoint value, then first attempt to determine idp value. - if self.endpoint or (self._wts_idp and self._wts_idp != "local"): + if self.endpoint or (self._wts_idp and self._wts_idp != "local"): # If user supplied endpoint value and not idp, figure out the idp value if self.endpoint: logging.debug( - "First try to use the local WTS to figure out idp name for the supplied endpoint.." + "First try to use the local WTS to figure out idp name for the supplied endpoint.." ) try: provider_List = get_wts_idps(self._wts_namespace) matchProviders = list( filter( - lambda provider: provider["base_url"] == endpoint, - provider_List["providers"], + lambda provider: provider["base_url"] == endpoint, + provider_List["providers"], ) ) if len(matchProviders) == 1: - logging.debug("Found matching idp from local WTS.") - self._wts_idp = matchProviders[0]["idp"] + logging.debug("Found matching idp from local WTS.") + self._wts_idp = matchProviders[0]["idp"] elif len(matchProviders) > 1: raise ValueError( - "Multiple idps matched with endpoint value provided." + "Multiple idps matched with endpoint value provided." ) else: - logging.debug("Could not find matching idp from local WTS.") + logging.debug("Could not find matching idp from local WTS.") except Exception as e: logging.debug( - "Exception occured when making network call to local WTS." + "Exception occured when making network call to local WTS." ) if not self._external_wts_host: raise e else: - logging.debug("Since external WTS host exists, continuing on..") + logging.debug("Since external WTS host exists, continuing on..") pass - if self._wts_idp and self._wts_idp != "local": - auth_url += "?idp={}".format(self._wts_idp) + if self._wts_idp and self._wts_idp != "local": + auth_url += "?idp={}".format(self._wts_idp) # If endpoint value exists, only get WTS token if idp value has been successfully determined # Otherwise skip to querying external WTS @@ -585,97 +585,97 @@

Source code for gen3.auth

         if (
             not self._external_wts_host
             or not self.endpoint
-            or (self.endpoint and self._wts_idp != "local")
+            or (self.endpoint and self._wts_idp != "local")
         ):
             try:
-                logging.debug("Try to get access token from local WTS..")
-                logging.debug(f"{auth_url=}")
+                logging.debug("Try to get access token from local WTS..")
+                logging.debug(f"{auth_url=}")
                 resp = requests.get(auth_url)
                 if (resp and resp.status_code == 200) or (not self._external_wts_host):
-                    return _handle_access_token_response(resp, "token")
+                    return _handle_access_token_response(resp, "token")
             except Exception as e:
                 if not self._external_wts_host:
                     raise e
                 else:
                     # Try to obtain token from external wts
-                    logging.debug("Could get obtain token from Local WTS.")
+                    logging.debug("Could get obtain token from Local WTS.")
                     pass
 
         # local workspace wts call failed, try using a network call
         # First get access token with WTS host
-        logging.debug("Trying to get access token from external WTS Host..")
+        logging.debug("Trying to get access token from external WTS Host..")
         wts_token = get_access_token_with_key(self._refresh_token)
-        auth_url = self._external_wts_host + "token/"
+        auth_url = self._external_wts_host + "token/"
 
         provider_List = get_wts_idps(self._wts_namespace, self._external_wts_host)
 
         # if user already supplied idp, use that
-        if self._wts_idp and self._wts_idp != "local":
+        if self._wts_idp and self._wts_idp != "local":
             matchProviders = list(
                 filter(
-                    lambda provider: provider["idp"] == self._wts_idp,
-                    provider_List["providers"],
+                    lambda provider: provider["idp"] == self._wts_idp,
+                    provider_List["providers"],
                 )
             )
         elif endpoint:
             matchProviders = list(
                 filter(
-                    lambda provider: provider["base_url"] == endpoint,
-                    provider_List["providers"],
+                    lambda provider: provider["base_url"] == endpoint,
+                    provider_List["providers"],
                 )
             )
         else:
             raise Exception(
-                "Unable to generate matching identity providers (no IdP or endpoint provided)"
+                "Unable to generate matching identity providers (no IdP or endpoint provided)"
             )
 
         if len(matchProviders) == 1:
-            self._wts_idp = matchProviders[0]["idp"]
-            logging.debug("Succesfully determined idp value: {}".format(self._wts_idp))
+            self._wts_idp = matchProviders[0]["idp"]
+            logging.debug("Succesfully determined idp value: {}".format(self._wts_idp))
         else:
-            idp_list = "\n "
+            idp_list = "\n "
 
             if len(matchProviders) > 1:
                 for idp in matchProviders:
                     idp_list = (
                         idp_list
-                        + "idp name: "
-                        + idp["idp"]
-                        + " url: "
-                        + idp["base_url"]
-                        + "\n "
+                        + "idp name: "
+                        + idp["idp"]
+                        + " url: "
+                        + idp["base_url"]
+                        + "\n "
                     )
                 raise ValueError(
-                    "Multiple idps matched with endpoint value provided."
+                    "Multiple idps matched with endpoint value provided."
                     + idp_list
-                    + "Query /wts/external_oidc/ for more information."
+                    + "Query /wts/external_oidc/ for more information."
                 )
             else:
-                for idp in provider_List["providers"]:
+                for idp in provider_List["providers"]:
                     idp_list = (
                         idp_list
-                        + "idp name: "
-                        + idp["idp"]
-                        + "  Endpoint url: "
-                        + idp["base_url"]
-                        + "\n "
+                        + "idp name: "
+                        + idp["idp"]
+                        + "  Endpoint url: "
+                        + idp["base_url"]
+                        + "\n "
                     )
                 raise ValueError(
-                    "No idp matched with the endpoint or idp value provided.\n"
-                    + "Please make sure your endpoint or idp value matches exactly with the output below.\n"
-                    + "i.e. check trailing '/' character for the endpoint url\n"
-                    + "Available Idps:"
+                    "No idp matched with the endpoint or idp value provided.\n"
+                    + "Please make sure your endpoint or idp value matches exactly with the output below.\n"
+                    + "i.e. check trailing '/' character for the endpoint url\n"
+                    + "Available Idps:"
                     + idp_list
-                    + "Query /wts/external_oidc/ for more information."
+                    + "Query /wts/external_oidc/ for more information."
                 )
-        logging.debug("Finally getting access token..")
-        auth_url += "?idp={}".format(self._wts_idp)
-        header = {"Authorization": "Bearer " + wts_token}
+        logging.debug("Finally getting access token..")
+        auth_url += "?idp={}".format(self._wts_idp)
+        header = {"Authorization": "Bearer " + wts_token}
         resp = requests.get(auth_url, headers=header)
-        err_msg = "Please make sure the target commons is connected on your profile page and that connection has not expired."
+        err_msg = "Please make sure the target commons is connected on your profile page and that connection has not expired."
         if resp.status_code != 200:
             logging.warning(err_msg)
-        return _handle_access_token_response(resp, "token")
+ return _handle_access_token_response(resp, "token")
diff --git a/docs/_build/html/_modules/gen3/file.html b/docs/_build/html/_modules/gen3/file.html index 53916adc..1cd1cbfa 100644 --- a/docs/_build/html/_modules/gen3/file.html +++ b/docs/_build/html/_modules/gen3/file.html @@ -59,7 +59,7 @@

Source code for gen3.file

 
[docs] class Gen3File: - """For interacting with Gen3 file management features. + """For interacting with Gen3 file management features. A class for interacting with the Gen3 file download services. Supports getting presigned urls right now. @@ -71,10 +71,10 @@

Source code for gen3.file

         This generates the Gen3File class pointed at the sandbox commons while
         using the credentials.json downloaded from the commons profile page.
 
-        >>> auth = Gen3Auth(refresh_file="credentials.json")
+        >>> auth = Gen3Auth(refresh_file="credentials.json")
         ... file = Gen3File(auth)
 
-    """
+    """
 
     def __init__(self, endpoint=None, auth_provider=None):
         # auth_provider legacy interface required endpoint as 1st arg
@@ -85,7 +85,7 @@ 

Source code for gen3.file

 
[docs] def get_presigned_url(self, guid, protocol=None): - """Generates a presigned URL for a file. + """Generates a presigned URL for a file. Retrieves a presigned url for a file giving access to a file for a limited time. @@ -97,10 +97,10 @@

Source code for gen3.file

 
             >>> Gen3File.get_presigned_url(query)
 
-        """
-        api_url = "{}/user/data/download/{}".format(self._endpoint, guid)
+        """
+        api_url = "{}/user/data/download/{}".format(self._endpoint, guid)
         if protocol:
-            api_url += "?protocol={}".format(protocol)
+            api_url += "?protocol={}".format(protocol)
         resp = requests.get(api_url, auth=self._auth_provider)
         raise_for_status_and_print_error(resp)
 
@@ -113,7 +113,7 @@ 

Source code for gen3.file

 
[docs] def delete_file(self, guid): - """ + """ This method is DEPRECATED. Use delete_file_locations() instead. Delete all locations of a stored data file and remove its record from indexd @@ -121,9 +121,9 @@

Source code for gen3.file

             guid (str): provide a UUID for file id to delete
         Returns:
             text: requests.delete text result
-        """
-        print("This method is DEPRECATED. Use delete_file_locations() instead.")
-        api_url = "{}/user/data/{}".format(self._endpoint, guid)
+        """
+        print("This method is DEPRECATED. Use delete_file_locations() instead.")
+        api_url = "{}/user/data/{}".format(self._endpoint, guid)
         output = requests.delete(api_url, auth=self._auth_provider).text
 
         return output
@@ -132,15 +132,15 @@

Source code for gen3.file

 
[docs] def delete_file_locations(self, guid): - """ + """ Delete all locations of a stored data file and remove its record from indexd Args: guid (str): provide a UUID for file id to delete Returns: requests.Response : requests.delete result - """ - api_url = "{}/user/data/{}".format(self._endpoint, guid) + """ + api_url = "{}/user/data/{}".format(self._endpoint, guid) output = requests.delete(api_url, auth=self._auth_provider) return output
@@ -151,37 +151,37 @@

Source code for gen3.file

     def upload_file(
         self, file_name, authz=None, protocol=None, expires_in=None, bucket=None
     ):
-        """
+        """
         Get a presigned url for a file to upload
 
         Args:
             file_name (str): file_name to use for upload
             authz (list): authorization scope for the file as list of paths, optional.
-            protocol (str): Storage protocol to use for upload: "s3", "az".
-                If this isn't set, the default will be "s3"
+            protocol (str): Storage protocol to use for upload: "s3", "az".
+                If this isn't set, the default will be "s3"
             expires_in (int): Amount in seconds that the signed url will expire from datetime.utcnow().
                 Be sure to use a positive integer.
                 This value will also be treated as <= MAX_PRESIGNED_URL_TTL in the fence configuration.
-            bucket (str): Bucket to upload to. The bucket must be configured in the Fence instance's
+            bucket (str): Bucket to upload to. The bucket must be configured in the Fence instance's
                 `ALLOWED_DATA_UPLOAD_BUCKETS` setting. If not specified, Fence defaults to the
                 `DATA_UPLOAD_BUCKET` setting.
         Returns:
             Document: json representation for the file upload
-        """
-        api_url = f"{self._endpoint}/user/data/upload"
+        """
+        api_url = f"{self._endpoint}/user/data/upload"
         body = {}
         if protocol:
-            body["protocol"] = protocol
+            body["protocol"] = protocol
         if authz:
-            body["authz"] = authz
+            body["authz"] = authz
         if expires_in:
-            body["expires_in"] = expires_in
+            body["expires_in"] = expires_in
         if file_name:
-            body["file_name"] = file_name
+            body["file_name"] = file_name
         if bucket:
-            body["bucket"] = bucket
+            body["bucket"] = bucket
 
-        headers = {"Content-Type": "application/json"}
+        headers = {"Content-Type": "application/json"}
         resp = requests.post(
             api_url, auth=self._auth_provider, json=body, headers=headers
         )
@@ -195,13 +195,13 @@ 

Source code for gen3.file

 
 
     def _ensure_dirpath_exists(path: Path) -> Path:
-        """Utility to create a directory if missing.
+        """Utility to create a directory if missing.
         Returns the path so that the call can be inlined in another call
         Args:
             path (Path): path to create
         Returns
             path of created directory
-        """
+        """
         assert path
         out_path: Path = path
 
@@ -213,58 +213,58 @@ 

Source code for gen3.file

 
[docs] def download_single(self, object_id, path): - """ + """ Download a single file using its GUID. Args: - object_id (str): The file's unique ID + object_id (str): The file's unique ID path (str): Path to store the downloaded file at - """ + """ try: url = self.get_presigned_url(object_id) except Exception as e: - logging.critical(f"Unable to get a presigned URL for download: {e}") + logging.critical(f"Unable to get a presigned URL for download: {e}") return False - response = requests.get(url["url"], stream=True) + response = requests.get(url["url"], stream=True) if response.status_code != 200: - logging.error(f"Response code: {response.status_code}") + logging.error(f"Response code: {response.status_code}") if response.status_code >= 500: for _ in range(MAX_RETRIES): - logging.info("Retrying now...") + logging.info("Retrying now...") # NOTE could be updated with exponential backoff time.sleep(1) - response = requests.get(url["url"], stream=True) + response = requests.get(url["url"], stream=True) if response.status == 200: break if response.status != 200: - logging.critical("Response status not 200, try again later") + logging.critical("Response status not 200, try again later") return False else: return False response.raise_for_status() - total_size_in_bytes = int(response.headers.get("content-length")) + total_size_in_bytes = int(response.headers.get("content-length")) total_downloaded = 0 index = Gen3Index(self._auth_provider) record = index.get_record(object_id) - filename = record["file_name"] + filename = record["file_name"] out_path = Gen3File._ensure_dirpath_exists(Path(path)) - with open(os.path.join(out_path, filename), "wb") as f: + with open(os.path.join(out_path, filename), "wb") as f: for data in response.iter_content(4096): total_downloaded += len(data) f.write(data) if total_size_in_bytes == total_downloaded: - logging.info(f"File {filename} downloaded successfully") + logging.info(f"File {filename} downloaded successfully") else: - logging.error(f"File {filename} not downloaded successfully") + logging.error(f"File {filename} not downloaded successfully") return False return True
@@ -275,32 +275,32 @@

Source code for gen3.file

     def upload_file_to_guid(
         self, guid, file_name, protocol=None, expires_in=None, bucket=None
     ):
-        """
+        """
         Get a presigned url for a file to upload to the specified existing GUID
 
         Args:
             file_name (str): file_name to use for upload
-            protocol (str): Storage protocol to use for upload: "s3", "az".
-                If this isn't set, the default will be "s3"
+            protocol (str): Storage protocol to use for upload: "s3", "az".
+                If this isn't set, the default will be "s3"
             expires_in (int): Amount in seconds that the signed url will expire from datetime.utcnow().
                 Be sure to use a positive integer.
                 This value will also be treated as <= MAX_PRESIGNED_URL_TTL in the fence configuration.
-            bucket (str): Bucket to upload to. The bucket must be configured in the Fence instance's
+            bucket (str): Bucket to upload to. The bucket must be configured in the Fence instance's
                 `ALLOWED_DATA_UPLOAD_BUCKETS` setting. If not specified, Fence defaults to the
                 `DATA_UPLOAD_BUCKET` setting.
         Returns:
             Document: json representation for the file upload
-        """
-        url = f"{self._endpoint}/user/data/upload/{guid}"
+        """
+        url = f"{self._endpoint}/user/data/upload/{guid}"
         params = {}
         if protocol:
-            params["protocol"] = protocol
+            params["protocol"] = protocol
         if expires_in:
-            params["expires_in"] = expires_in
+            params["expires_in"] = expires_in
         if file_name:
-            params["file_name"] = file_name
+            params["file_name"] = file_name
         if bucket:
-            params["bucket"] = bucket
+            params["bucket"] = bucket
 
         url_parts = list(urlparse(url))
         query = dict(parse_qsl(url_parts[4]))
diff --git a/docs/_build/html/_modules/gen3/index.html b/docs/_build/html/_modules/gen3/index.html
index b393f29c..694009f6 100644
--- a/docs/_build/html/_modules/gen3/index.html
+++ b/docs/_build/html/_modules/gen3/index.html
@@ -50,7 +50,7 @@ 

Source code for gen3.index

 
[docs] class Gen3Index: - """ + """ A class for interacting with the Gen3 Index services. @@ -62,26 +62,26 @@

Source code for gen3.index

         This generates the Gen3Index class pointed at the sandbox commons while
         using the credentials.json downloaded from the commons profile page.
 
-        >>> auth = Gen3Auth(refresh_file="credentials.json")
+        >>> auth = Gen3Auth(refresh_file="credentials.json")
         ... index = Gen3Index(auth)
 
-    """
+    """
 
-    def __init__(self, endpoint=None, auth_provider=None, service_location="index"):
+    def __init__(self, endpoint=None, auth_provider=None, service_location="index"):
         # legacy interface required endpoint as 1st arg
         if endpoint and isinstance(endpoint, Gen3Auth):
             auth_provider = endpoint
             endpoint = None
         if auth_provider and isinstance(auth_provider, Gen3Auth):
             endpoint = auth_provider.endpoint
-        endpoint = endpoint.strip("/")
+        endpoint = endpoint.strip("/")
         # if running locally, indexd is deployed by itself without a location relative
         # to the commons
-        if "http://localhost" in endpoint:
-            service_location = ""
+        if "http://localhost" in endpoint:
+            service_location = ""
 
         if not endpoint.endswith(service_location):
-            endpoint += "/" + service_location
+            endpoint += "/" + service_location
 
         self.endpoint = endpoint
         self.client = client.IndexClient(endpoint, auth=auth_provider)
@@ -90,29 +90,29 @@ 

Source code for gen3.index

 
[docs] def is_healthy(self): - """ + """ Return if indexd is healthy or not - """ + """ try: - response = self.client._get("_status") + response = self.client._get("_status") response.raise_for_status() except Exception: return False - return response.text == "Healthy"
+ return response.text == "Healthy"
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) def get_version(self): - """ + """ Return the version of indexd - """ - response = self.client._get("_version") + """ + response = self.client._get("_version") raise_for_status_and_print_error(response) return response.json()
@@ -121,12 +121,12 @@

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_stats(self):
-        """
+        """
 
         Return basic info about the records in indexd
 
-        """
-        response = self.client._get("_stats")
+        """
+        response = self.client._get("_stats")
         raise_for_status_and_print_error(response)
         return response.json()
@@ -135,31 +135,31 @@

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_all_records(self, limit=None, paginate=False):
-        """
+        """
 
         Get a list of all records
 
-        """
+        """
         all_records = []
-        url = "index/"
+        url = "index/"
 
         if limit:
-            url += f"?limit={limit}"
+            url += f"?limit={limit}"
 
         response = self.client._get(url)
         raise_for_status_and_print_error(response)
 
-        records = response.json().get("records")
+        records = response.json().get("records")
         all_records.extend(records)
 
         if paginate and records:
             previous_did = None
-            start_did = records[-1].get("did")
+            start_did = records[-1].get("did")
 
             while start_did != previous_did:
                 previous_did = start_did
 
-                params = {"start": f"{start_did}"}
+                params = {"start": f"{start_did}"}
                 url_parts = list(urllib.parse.urlparse(url))
                 query = dict(urllib.parse.parse_qsl(url_parts[4]))
                 query.update(params)
@@ -170,11 +170,11 @@ 

Source code for gen3.index

                 response = self.client._get(url)
                 raise_for_status_and_print_error(response)
 
-                records = response.json().get("records")
+                records = response.json().get("records")
                 all_records.extend(records)
 
                 if records:
-                    start_did = response.json().get("records")[-1].get("did")
+                    start_did = response.json().get("records")[-1].get("did")
 
         return all_records
@@ -183,33 +183,33 @@

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_records_on_page(self, limit=None, page=None):
-        """
+        """
 
         Get a list of all records given the page and page size limit
 
-        """
+        """
         params = {}
-        url = "index/"
+        url = "index/"
 
         if limit is not None:
-            params["limit"] = limit
+            params["limit"] = limit
 
         if page is not None:
-            params["page"] = page
+            params["page"] = page
 
         query = urllib.parse.urlencode(params)
 
-        response = self.client._get(url + "?" + query)
+        response = self.client._get(url + "?" + query)
         raise_for_status_and_print_error(response)
 
-        return response.json().get("records")
+ return response.json().get("records")
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) async def async_get_record(self, guid=None, _ssl=None): - """ + """ Asynchronous function to request a record from indexd. Args: @@ -217,8 +217,8 @@

Source code for gen3.index

 
         Returns:
             dict: indexd record
-        """
-        url = f"{self.client.url}/index/{guid}"
+        """
+        url = f"{self.client.url}/index/{guid}"
         async with aiohttp.ClientSession() as session:
             async with session.get(url, ssl=_ssl) as response:
                 raise_for_status_and_print_error(response)
@@ -231,7 +231,7 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     async def async_get_records_on_page(self, limit=None, page=None, _ssl=None):
-        """
+        """
         Asynchronous function to request a page from indexd.
 
         Args:
@@ -239,33 +239,33 @@ 

Source code for gen3.index

 
         Returns:
             List[dict]: List of indexd records from the page
-        """
+        """
         all_records = []
         params = {}
 
         if limit is not None:
-            params["limit"] = limit
+            params["limit"] = limit
 
         if page is not None:
-            params["page"] = page
+            params["page"] = page
 
         query = urllib.parse.urlencode(params)
 
-        url = f"{self.client.url}/index" + "?" + query
+        url = f"{self.client.url}/index" + "?" + query
         async with aiohttp.ClientSession() as session:
             async with session.get(url, ssl=_ssl) as response:
                 response = await response.json()
 
-        return response.get("records")
+ return response.get("records")
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) async def async_get_records_from_checksum( - self, checksum, checksum_type="md5", _ssl=None + self, checksum, checksum_type="md5", _ssl=None ): - """ + """ Asynchronous function to request records from indexd matching checksum. Args: @@ -274,27 +274,27 @@

Source code for gen3.index

 
         Returns:
             List[dict]: List of indexd records
-        """
+        """
         all_records = []
         params = {}
 
-        params["hash"] = f"{checksum_type}:{checksum}"
+        params["hash"] = f"{checksum_type}:{checksum}"
 
         query = urllib.parse.urlencode(params)
 
-        url = f"{self.client.url}/index" + "?" + query
+        url = f"{self.client.url}/index" + "?" + query
         async with aiohttp.ClientSession() as session:
             async with session.get(url, ssl=_ssl) as response:
                 response = await response.json()
 
-        return response.get("records")
+ return response.get("records")
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) def get(self, guid, dist_resolution=True): - """ + """ Get the metadata associated with the given id, alias, or distributed identifier @@ -305,7 +305,7 @@

Source code for gen3.index

             dist_resolution: boolean
             - *optional* Specify if we want distributed dist_resolution or not
 
-        """
+        """
         rec = self.client.global_get(guid, dist_resolution)
 
         if not rec:
@@ -318,7 +318,7 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_urls(self, size=None, hashes=None, guids=None):
-        """
+        """
 
         Get a list of urls that match query params
 
@@ -330,11 +330,11 @@ 

Source code for gen3.index

             guids: list
                 - list of ids
 
-        """
+        """
         if guids:
-            guids = ",".join(guids)
-        p = {"size": size, "hash": hashes, "ids": guids}
-        urls = self.client._get("urls", params=p).json()
+            guids = ",".join(guids)
+        p = {"size": size, "hash": hashes, "ids": guids}
+        urls = self.client._get("urls", params=p).json()
         return [url for _, url in urls.items()]
@@ -342,11 +342,11 @@

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_record(self, guid):
-        """
+        """
 
         Get the metadata associated with a given id
 
-        """
+        """
         rec = self.client.get(guid)
 
         if not rec:
@@ -359,11 +359,11 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_record_doc(self, guid):
-        """
+        """
 
         Get the metadata associated with a given id
 
-        """
+        """
         return self.client.get(guid)
@@ -371,16 +371,16 @@

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_with_params(self, params=None):
-        """
+        """
 
         Return a document object corresponding to the supplied parameters, such
-        as ``{'hashes': {'md5': '...'}, 'size': '...', 'metadata': {'file_state': '...'}}``.
+        as ``{'hashes': {'md5': '...'}, 'size': '...', 'metadata': {'file_state': '...'}}``.
 
             - need to include all the hashes in the request
             - index client like signpost or indexd will need to handle the
-              query param `'hash': 'hash_type:hash'`
+              query param `'hash': 'hash_type:hash'`
 
-        """
+        """
         rec = self.client.get_with_params(params)
 
         if not rec:
@@ -393,12 +393,12 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     async def async_get_with_params(self, params, _ssl=None):
-        """
+        """
 
         Return a document object corresponding to the supplied parameter
 
         - need to include all the hashes in the request
-        - need to handle the query param `'hash': 'hash_type:hash'`
+        - need to handle the query param `'hash': 'hash_type:hash'`
 
         Args:
             params (dict): params to search with
@@ -407,9 +407,9 @@ 

Source code for gen3.index

         Returns:
             Document: json representation of an entry in indexd
 
-        """
+        """
         query_params = urllib.parse.urlencode(params)
-        url = f"{self.client.url}/index/?{query_params}"
+        url = f"{self.client.url}/index/?{query_params}"
         async with aiohttp.ClientSession() as session:
             async with session.get(url, ssl=_ssl) as response:
                 await response.raise_for_status()
@@ -422,7 +422,7 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_latest_version(self, guid, has_version=False):
-        """
+        """
 
         Get the metadata of the latest index record version associated
         with the given id
@@ -433,7 +433,7 @@ 

Source code for gen3.index

             has_version: boolean
                 - *optional* exclude entries without a version
 
-        """
+        """
         rec = self.client.get_latest_version(guid, has_version)
 
         if not rec:
@@ -446,7 +446,7 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_versions(self, guid):
-        """
+        """
 
         Get the metadata of index record version associated with the
         given id
@@ -455,8 +455,8 @@ 

Source code for gen3.index

             guid: string
                 - record id
 
-        """
-        response = self.client._get(f"/index/{guid}/versions")
+        """
+        response = self.client._get(f"/index/{guid}/versions")
         raise_for_status_and_print_error(response)
         versions = response.json()
 
@@ -485,13 +485,13 @@ 

Source code for gen3.index

         content_created_date=None,
         content_updated_date=None,
     ):
-        """
+        """
 
         Create a new record and add it to the index
 
         Args:
             hashes (dict): {hash type: hash value,}
-                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
+                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
             size (int): file size metadata associated with a given uuid
             did (str): provide a UUID for the new indexd to be made
             urls (list): list of URLs where you can download the UUID
@@ -508,26 +508,26 @@ 

Source code for gen3.index

         Returns:
             Document: json representation of an entry in indexd
 
-        """
+        """
         if urls is None:
             urls = []
         json = {
-            "urls": urls,
-            "hashes": hashes,
-            "size": size,
-            "file_name": file_name,
-            "metadata": metadata,
-            "urls_metadata": urls_metadata,
-            "baseid": baseid,
-            "acl": acl,
-            "authz": authz,
-            "version": version,
-            "description": description,
-            "content_created_date": content_created_date,
-            "content_updated_date": content_updated_date,
+            "urls": urls,
+            "hashes": hashes,
+            "size": size,
+            "file_name": file_name,
+            "metadata": metadata,
+            "urls_metadata": urls_metadata,
+            "baseid": baseid,
+            "acl": acl,
+            "authz": authz,
+            "version": version,
+            "description": description,
+            "content_created_date": content_created_date,
+            "content_updated_date": content_updated_date,
         }
         if did:
-            json["did"] = did
+            json["did"] = did
         rec = self.client.create(**json)
 
         return rec.to_json()
@@ -554,12 +554,12 @@

Source code for gen3.index

         content_created_date=None,
         content_updated_date=None,
     ):
-        """
+        """
         Asynchronous function to create a record in indexd.
 
         Args:
             hashes (dict): {hash type: hash value,}
-                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
+                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
             size (int): file size metadata associated with a given uuid
             did (str): provide a UUID for the new indexd to be made
             urls (list): list of URLs where you can download the UUID
@@ -576,45 +576,45 @@ 

Source code for gen3.index

 
         Returns:
             Document: json representation of an entry in indexd
-        """
+        """
         async with aiohttp.ClientSession() as session:
             if urls is None:
                 urls = []
 
             json = {
-                "form": "object",
-                "hashes": hashes,
-                "size": size,
-                "urls": urls or [],
+                "form": "object",
+                "hashes": hashes,
+                "size": size,
+                "urls": urls or [],
             }
             if did:
-                json["did"] = did
+                json["did"] = did
             if file_name:
-                json["file_name"] = file_name
+                json["file_name"] = file_name
             if metadata:
-                json["metadata"] = metadata
+                json["metadata"] = metadata
             if baseid:
-                json["baseid"] = baseid
+                json["baseid"] = baseid
             if acl:
-                json["acl"] = acl
+                json["acl"] = acl
             if urls_metadata:
-                json["urls_metadata"] = urls_metadata
+                json["urls_metadata"] = urls_metadata
             if version:
-                json["version"] = version
+                json["version"] = version
             if authz:
-                json["authz"] = authz
+                json["authz"] = authz
             if description:
-                json["description"] = description
+                json["description"] = description
             if content_created_date:
-                json["content_created_date"] = content_created_date
+                json["content_created_date"] = content_created_date
             if content_updated_date:
-                json["content_updated_date"] = content_updated_date
+                json["content_updated_date"] = content_updated_date
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self.client.auth._get_auth_value()}
+            headers = {"Authorization": self.client.auth._get_auth_value()}
 
             async with session.post(
-                f"{self.client.url}/index/",
+                f"{self.client.url}/index/",
                 json=json,
                 headers=headers,
                 ssl=_ssl,
@@ -629,29 +629,29 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def create_blank(self, uploader, file_name=None):
-        """
+        """
 
         Create a blank record
 
         Args:
             json - json in the format:
             {
-                'uploader': type(string)
-                'file_name': type(string) (optional*)
+                'uploader': type(string)
+                'file_name': type(string) (optional*)
             }
 
-        """
-        json = {"uploader": uploader, "file_name": file_name}
+        """
+        json = {"uploader": uploader, "file_name": file_name}
         response = self.client._post(
-            "index/blank",
-            headers={"content-type": "application/json"},
+            "index/blank",
+            headers={"content-type": "application/json"},
             auth=self.client.auth,
             data=client.json_dumps(json),
         )
         raise_for_status_and_print_error(response)
         rec = response.json()
 
-        return self.get_record(rec["did"])
+ return self.get_record(rec["did"])
@@ -674,7 +674,7 @@

Source code for gen3.index

         content_created_date=None,
         content_updated_date=None,
     ):
-        """
+        """
 
         Add new version for the document associated to the provided uuid
 
@@ -686,7 +686,7 @@ 

Source code for gen3.index

         Args:
             guid: (string): record id
             hashes (dict): {hash type: hash value,}
-                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
+                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
             size (int): file size metadata associated with a given uuid
             did (str): provide a UUID for the new indexd to be made
             urls (list): list of URLs where you can download the UUID
@@ -706,38 +706,38 @@ 

Source code for gen3.index

               sufficient. Note: it is a good idea to add a version
               number
 
-        """
+        """
         if urls is None:
             urls = []
         json = {
-            "urls": urls,
-            "form": "object",
-            "hashes": hashes,
-            "size": size,
-            "file_name": file_name,
-            "metadata": metadata,
-            "urls_metadata": urls_metadata,
-            "acl": acl,
-            "authz": authz,
-            "version": version,
-            "description": description,
-            "content_created_date": content_created_date,
-            "content_updated_date": content_updated_date,
+            "urls": urls,
+            "form": "object",
+            "hashes": hashes,
+            "size": size,
+            "file_name": file_name,
+            "metadata": metadata,
+            "urls_metadata": urls_metadata,
+            "acl": acl,
+            "authz": authz,
+            "version": version,
+            "description": description,
+            "content_created_date": content_created_date,
+            "content_updated_date": content_updated_date,
         }
         if did:
-            json["did"] = did
+            json["did"] = did
         response = self.client._post(
-            "index",
+            "index",
             guid,
-            headers={"content-type": "application/json"},
+            headers={"content-type": "application/json"},
             data=client.json_dumps(json),
             auth=self.client.auth,
         )
         raise_for_status_and_print_error(response)
         rec = response.json()
 
-        if rec and "did" in rec:
-            return self.get_record(rec["did"])
+        if rec and "did" in rec:
+            return self.get_record(rec["did"])
         return None
@@ -745,7 +745,7 @@

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_records(self, dids):
-        """
+        """
 
         Get a list of documents given a list of dids
 
@@ -756,10 +756,10 @@ 

Source code for gen3.index

         Returns:
             list: json representing index records
 
-        """
+        """
         try:
             response = self.client._post(
-                "bulk/documents", json=dids, auth=self.client.auth
+                "bulk/documents", json=dids, auth=self.client.auth
             )
         except requests.HTTPError as exception:
             if exception.response.status_code == 404:
@@ -776,7 +776,7 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def update_blank(self, guid, rev, hashes, size, urls=None, authz=None):
-        """
+        """
 
         Update only hashes and size for a blank index
 
@@ -784,21 +784,21 @@ 

Source code for gen3.index

             guid (string): record id
             rev (string): data revision - simple consistency mechanism
             hashes (dict): {hash type: hash value,}
-                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
+                eg ``hashes={'md5': ab167e49d25b488939b1ede42752458b'}``
             size (int): file size metadata associated with a given uuid
 
-        """
-        params = {"rev": rev}
-        json = {"hashes": hashes, "size": size}
+        """
+        params = {"rev": rev}
+        json = {"hashes": hashes, "size": size}
         if urls:
-            json["urls"] = urls
+            json["urls"] = urls
         if authz:
-            json["authz"] = authz
+            json["authz"] = authz
 
         response = self.client._put(
-            "index/blank",
+            "index/blank",
             guid,
-            headers={"content-type": "application/json"},
+            headers={"content-type": "application/json"},
             params=params,
             auth=self.client.auth,
             data=client.json_dumps(json),
@@ -806,7 +806,7 @@ 

Source code for gen3.index

         raise_for_status_and_print_error(response)
         rec = response.json()
 
-        return self.get_record(rec["did"])
+ return self.get_record(rec["did"])
@@ -826,7 +826,7 @@

Source code for gen3.index

         content_created_date=None,
         content_updated_date=None,
     ):
-        """
+        """
 
         Update an existing entry in the index
 
@@ -837,27 +837,27 @@ 

Source code for gen3.index

                  - index record information that needs to be updated.
                  - can not update size or hash, use new version for that
 
-        """
+        """
         updatable_attrs = {
-            "file_name": file_name,
-            "urls": urls,
-            "version": version,
-            "metadata": metadata,
-            "acl": acl,
-            "authz": authz,
-            "urls_metadata": urls_metadata,
-            "description": description,
-            "content_created_date": content_created_date,
-            "content_updated_date": content_updated_date,
+            "file_name": file_name,
+            "urls": urls,
+            "version": version,
+            "metadata": metadata,
+            "acl": acl,
+            "authz": authz,
+            "urls_metadata": urls_metadata,
+            "description": description,
+            "content_created_date": content_created_date,
+            "content_updated_date": content_updated_date,
         }
         rec = self.client.get(guid)
         if not rec:
             raise ValueError(
-                f"No indexd record found for GUID '{guid}' at '{self.endpoint}'"
+                f"No indexd record found for GUID '{guid}' at '{self.endpoint}'"
             )
         for k, v in updatable_attrs.items():
             if v is not None:
-                exec(f"rec.{k} = v")
+                exec(f"rec.{k} = v")
         rec.patch()
         return rec.to_json()
@@ -881,7 +881,7 @@

Source code for gen3.index

         content_updated_date=None,
         **kwargs,
     ):
-        """
+        """
         Asynchronous function to update a record in indexd.
 
         Args:
@@ -890,47 +890,47 @@ 

Source code for gen3.index

              body: json/dictionary format
                  - index record information that needs to be updated.
                  - can not update size or hash, use new version for that
-        """
+        """
         async with aiohttp.ClientSession() as session:
             updatable_attrs = {
-                "file_name": file_name,
-                "urls": urls,
-                "version": version,
-                "metadata": metadata,
-                "acl": acl,
-                "authz": authz,
-                "urls_metadata": urls_metadata,
-                "description": description,
-                "content_created_date": content_created_date,
-                "content_updated_date": content_updated_date,
+                "file_name": file_name,
+                "urls": urls,
+                "version": version,
+                "metadata": metadata,
+                "acl": acl,
+                "authz": authz,
+                "urls_metadata": urls_metadata,
+                "description": description,
+                "content_created_date": content_created_date,
+                "content_updated_date": content_updated_date,
             }
             record = await self.async_get_record(guid)
-            revision = record.get("rev")
+            revision = record.get("rev")
 
             for key, value in updatable_attrs.items():
                 if value is not None:
                     record[key] = value
 
-            del record["created_date"]
-            del record["rev"]
-            del record["updated_date"]
-            del record["version"]
-            del record["uploader"]
-            del record["form"]
-            del record["urls_metadata"]
-            del record["baseid"]
-            del record["size"]
-            del record["hashes"]
-            del record["did"]
+            del record["created_date"]
+            del record["rev"]
+            del record["updated_date"]
+            del record["version"]
+            del record["uploader"]
+            del record["form"]
+            del record["urls_metadata"]
+            del record["baseid"]
+            del record["size"]
+            del record["hashes"]
+            del record["did"]
 
-            logging.info(f"PUT-ing record: {record}")
+            logging.info(f"PUT-ing record: {record}")
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self.client.auth._get_auth_value()}
+            headers = {"Authorization": self.client.auth._get_auth_value()}
 
             async with session.put(
-                f"{self.client.url}/index/{guid}?rev={revision}",
+                f"{self.client.url}/index/{guid}?rev={revision}",
                 json=record,
                 headers=headers,
                 ssl=_ssl,
@@ -947,7 +947,7 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def delete_record(self, guid):
-        """
+        """
 
         Delete an entry from the index
 
@@ -957,7 +957,7 @@ 

Source code for gen3.index

 
         Returns: Nothing
 
-        """
+        """
         rec = self.client.get(guid)
         if rec:
             rec.delete()
@@ -970,7 +970,7 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def query_urls(self, pattern):
-        """
+        """
 
         Query all record URLs for given pattern
 
@@ -979,8 +979,8 @@ 

Source code for gen3.index

 
         Returns:
             List[records]: indexd records with urls matching pattern
-        """
-        response = self.client._get(f"/_query/urls/q?include={pattern}")
+        """
+        response = self.client._get(f"/_query/urls/q?include={pattern}")
         raise_for_status_and_print_error(response)
         return response.json()
@@ -989,7 +989,7 @@

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     async def async_query_urls(self, pattern, _ssl=None):
-        """
+        """
         Asynchronous function to query urls from indexd.
 
         Args:
@@ -997,10 +997,10 @@ 

Source code for gen3.index

 
         Returns:
             List[records]: indexd records with urls matching pattern
-        """
-        url = f"{self.client.url}/_query/urls/q?include={pattern}"
+        """
+        url = f"{self.client.url}/_query/urls/q?include={pattern}"
         async with aiohttp.ClientSession() as session:
-            logging.debug(f"request: {url}")
+            logging.debug(f"request: {url}")
             async with session.get(url, ssl=_ssl) as response:
                 raise_for_status_and_print_error(response)
                 response = await response.json()
@@ -1014,44 +1014,44 @@ 

Source code for gen3.index

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_valid_guids(self, count=None):
-        """
+        """
         Get a list of valid GUIDs without indexing
         Args:
             count (int): number of GUIDs to request
         Returns:
             List[str]: list of valid indexd GUIDs
-        """
-        url = "/guid/mint"
+        """
+        url = "/guid/mint"
         if count:
-            url += f"?count={count}"
+            url += f"?count={count}"
 
         response = self.client._get(url)
         response.raise_for_status()
-        return response.json().get("guids", [])
+ return response.json().get("guids", [])
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) def get_guids_prefix(self): - """ + """ Get the prefix for GUIDs if there is one Returns: str: prefix for this instance - """ - response = self.client._get("/guid/prefix") + """ + response = self.client._get("/guid/prefix") response.raise_for_status() - return response.json().get("prefix")
+ return response.json().get("prefix")
def _print_func_name(function): - return "{}.{}".format(function.__module__, function.__name__) + return "{}.{}".format(function.__module__, function.__name__) def _print_kwargs(kwargs): - return ", ".join("{}={}".format(k, repr(v)) for k, v in list(kwargs.items())) + return ", ".join("{}={}".format(k, repr(v)) for k, v in list(kwargs.items()))
diff --git a/docs/_build/html/_modules/gen3/jobs.html b/docs/_build/html/_modules/gen3/jobs.html index ed3448ca..a81656f8 100644 --- a/docs/_build/html/_modules/gen3/jobs.html +++ b/docs/_build/html/_modules/gen3/jobs.html @@ -31,9 +31,9 @@

Source code for gen3.jobs

-"""
-Contains class for interacting with Gen3's Job Dispatching Service(s).
-"""
+"""
+Contains class for interacting with Gen3's Job Dispatching Service(s).
+"""
 import aiohttp
 import asyncio
 import backoff
@@ -51,12 +51,12 @@ 

Source code for gen3.jobs

     raise_for_status_and_print_error,
 )
 
-# sower's "action" mapping to the relevant job
-INGEST_METADATA_JOB = "ingest-metadata-manifest"
-DBGAP_METADATA_JOB = "get-dbgap-metadata"
-INDEX_MANIFEST_JOB = "index-object-manifest"
-DOWNLOAD_MANIFEST_JOB = "download-indexd-manifest"
-MERGE_MANIFEST_JOB = "merge-manifests"
+# sower's "action" mapping to the relevant job
+INGEST_METADATA_JOB = "ingest-metadata-manifest"
+DBGAP_METADATA_JOB = "get-dbgap-metadata"
+INDEX_MANIFEST_JOB = "index-object-manifest"
+DOWNLOAD_MANIFEST_JOB = "download-indexd-manifest"
+MERGE_MANIFEST_JOB = "merge-manifests"
 
 logging = get_logger(__name__)
 
@@ -64,19 +64,19 @@ 

Source code for gen3.jobs

 
[docs] class Gen3Jobs: - """ - A class for interacting with the Gen3's Job Dispatching Service(s). + """ + A class for interacting with the Gen3's Job Dispatching Service(s). Examples: This generates the Gen3Jobs class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page. - >>> auth = Gen3Auth(refresh_file="credentials.json") + >>> auth = Gen3Auth(refresh_file="credentials.json") ... jobs = Gen3Jobs(auth) - """ + """ - def __init__(self, endpoint=None, auth_provider=None, service_location="job"): - """ + def __init__(self, endpoint=None, auth_provider=None, service_location="job"): + """ Initialization for instance of the class to setup basic endpoint info. Args: @@ -84,25 +84,25 @@

Source code for gen3.jobs

                 token, required for admin endpoints
             service_location (str, optional): deployment location relative to the
                 endpoint provided
-        """
+        """
         # auth_provider legacy interface required endpoint as 1st arg
         auth_provider = auth_provider or endpoint
-        endpoint = auth_provider.endpoint.strip("/")
+        endpoint = auth_provider.endpoint.strip("/")
         # if running locally, mds is deployed by itself without a location relative
         # to the commons
-        if "http://localhost" in endpoint:
-            service_location = ""
+        if "http://localhost" in endpoint:
+            service_location = ""
 
         if not endpoint.endswith(service_location):
-            endpoint += "/" + service_location
+            endpoint += "/" + service_location
 
-        self.endpoint = endpoint.rstrip("/")
+        self.endpoint = endpoint.rstrip("/")
         self._auth_provider = auth_provider
 
 
[docs] async def async_run_job_and_wait(self, job_name, job_input, _ssl=None, **kwargs): - """ + """ Asynchronous function to create a job, wait for output, and return. Will sleep in a linear delay until the job is done, starting with 1 second. @@ -113,71 +113,71 @@

Source code for gen3.jobs

 
         Returns:
             Dict: Response from the endpoint
-        """
+        """
         job_create_response = await self.async_create_job(job_name, job_input)
 
-        status = {"status": "Running"}
+        status = {"status": "Running"}
         sleep_time = 3
-        while status.get("status") == "Running":
-            logging.info(f"job still running, waiting for {sleep_time} seconds...")
+        while status.get("status") == "Running":
+            logging.info(f"job still running, waiting for {sleep_time} seconds...")
             time.sleep(sleep_time)
             sleep_time *= 1.5
-            status = await self.async_get_status(job_create_response.get("uid"))
-            logging.info(f"{status}")
+            status = await self.async_get_status(job_create_response.get("uid"))
+            logging.info(f"{status}")
 
-        logging.info(f"Job is finished!")
+        logging.info(f"Job is finished!")
 
-        if status.get("status") != "Completed":
-            raise Exception(f"Job status not complete: {status.get('status')}.")
+        if status.get("status") != "Completed":
+            raise Exception(f"Job status not complete: {status.get('status')}.")
 
-        response = await self.async_get_output(job_create_response.get("uid"))
+        response = await self.async_get_output(job_create_response.get("uid"))
         return response
[docs] def is_healthy(self): - """ + """ Return if is healthy or not Returns: bool: True if healthy - """ + """ try: response = requests.get( - self.endpoint + "/_status", auth=self._auth_provider + self.endpoint + "/_status", auth=self._auth_provider ) raise_for_status_and_print_error(response) except Exception as exc: logging.error(exc) return False - return response.text == "Healthy"
+ return response.text == "Healthy"
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) def get_version(self): - """ + """ Return the version Returns: str: the version - """ - response = requests.get(self.endpoint + "/_version", auth=self._auth_provider) + """ + response = requests.get(self.endpoint + "/_version", auth=self._auth_provider) raise_for_status_and_print_error(response) - return response.json().get("version")
+ return response.json().get("version")
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) def list_jobs(self): - """ + """ List all jobs - """ - response = requests.get(self.endpoint + "/list", auth=self._auth_provider) + """ + response = requests.get(self.endpoint + "/list", auth=self._auth_provider) raise_for_status_and_print_error(response) return response.json()
@@ -186,7 +186,7 @@

Source code for gen3.jobs

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def create_job(self, job_name, job_input):
-        """
+        """
         Create a job with given name and input
 
         Args:
@@ -195,10 +195,10 @@ 

Source code for gen3.jobs

 
         Returns:
             Dict: Response from the endpoint
-        """
-        data = {"action": job_name, "input": job_input}
+        """
+        data = {"action": job_name, "input": job_input}
         response = requests.post(
-            self.endpoint + "/dispatch", json=data, auth=self._auth_provider
+            self.endpoint + "/dispatch", json=data, auth=self._auth_provider
         )
         raise_for_status_and_print_error(response)
         return response.json()
@@ -207,14 +207,14 @@

Source code for gen3.jobs

     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     async def async_create_job(self, job_name, job_input, _ssl=None, **kwargs):
         async with aiohttp.ClientSession() as session:
-            url = self.endpoint + f"/dispatch"
+            url = self.endpoint + f"/dispatch"
             url_with_params = append_query_params(url, **kwargs)
 
-            data = json.dumps({"action": job_name, "input": job_input})
+            data = json.dumps({"action": job_name, "input": job_input})
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
             async with session.post(
                 url_with_params, data=data, headers=headers, ssl=_ssl
@@ -227,11 +227,11 @@ 

Source code for gen3.jobs

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_status(self, job_id):
-        """
+        """
         Get the status of a previously created job
-        """
+        """
         response = requests.get(
-            self.endpoint + f"/status?UID={job_id}", auth=self._auth_provider
+            self.endpoint + f"/status?UID={job_id}", auth=self._auth_provider
         )
         raise_for_status_and_print_error(response)
         return response.json()
@@ -240,12 +240,12 @@

Source code for gen3.jobs

     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     async def async_get_status(self, job_id, _ssl=None, **kwargs):
         async with aiohttp.ClientSession() as session:
-            url = self.endpoint + f"/status?UID={job_id}"
+            url = self.endpoint + f"/status?UID={job_id}"
             url_with_params = append_query_params(url, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
             async with session.get(
                 url_with_params, headers=headers, ssl=_ssl
@@ -258,11 +258,11 @@ 

Source code for gen3.jobs

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_output(self, job_id):
-        """
+        """
         Get the output of a previously completed job
-        """
+        """
         response = requests.get(
-            self.endpoint + f"/output?UID={job_id}", auth=self._auth_provider
+            self.endpoint + f"/output?UID={job_id}", auth=self._auth_provider
         )
         raise_for_status_and_print_error(response)
         return response.json()
@@ -271,12 +271,12 @@

Source code for gen3.jobs

     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     async def async_get_output(self, job_id, _ssl=None, **kwargs):
         async with aiohttp.ClientSession() as session:
-            url = self.endpoint + f"/output?UID={job_id}"
+            url = self.endpoint + f"/output?UID={job_id}"
             url_with_params = append_query_params(url, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
             async with session.get(
                 url_with_params, headers=headers, ssl=_ssl
diff --git a/docs/_build/html/_modules/gen3/metadata.html b/docs/_build/html/_modules/gen3/metadata.html
index cb785a92..7e1dc353 100644
--- a/docs/_build/html/_modules/gen3/metadata.html
+++ b/docs/_build/html/_modules/gen3/metadata.html
@@ -31,9 +31,9 @@
           

Source code for gen3.metadata

-"""
-Contains class for interacting with Gen3's Metadata Service.
-"""
+"""
+Contains class for interacting with Gen3's Metadata Service.
+"""
 import aiohttp
 import backoff
 from datetime import datetime
@@ -66,24 +66,24 @@ 

Source code for gen3.metadata

 logging = get_logger(__name__)
 
 
-PACKAGE_CONTENTS_STANDARD_KEY = "package_contents"
+PACKAGE_CONTENTS_STANDARD_KEY = "package_contents"
 PACKAGE_CONTENTS_SCHEMA = {
-    "type": "array",
-    "items": {
-        "type": "object",
-        "properties": {
-            "file_name": {
-                "type": "string",
+    "type": "array",
+    "items": {
+        "type": "object",
+        "properties": {
+            "file_name": {
+                "type": "string",
             },
-            "size": {
-                "type": "integer",
+            "size": {
+                "type": "integer",
             },
-            "hashes": {
-                "type": "object",
+            "hashes": {
+                "type": "object",
             },
         },
-        "required": ["file_name"],
-        "additionalProperties": True,
+        "required": ["file_name"],
+        "additionalProperties": True,
     },
 }
 
@@ -91,29 +91,29 @@ 

Source code for gen3.metadata

 
[docs] class Gen3Metadata: - """ + """ A class for interacting with the Gen3 Metadata services. Examples: This generates the Gen3Metadata class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page. - >>> auth = Gen3Auth(refresh_file="credentials.json") + >>> auth = Gen3Auth(refresh_file="credentials.json") ... metadata = Gen3Metadata(auth) Attributes: endpoint (str): public endpoint for reading/querying metadata - only necessary if auth_provider not provided auth_provider (Gen3Auth): auth manager - """ + """ def __init__( self, endpoint=None, auth_provider=None, - service_location="mds", - admin_endpoint_suffix="-admin", + service_location="mds", + admin_endpoint_suffix="-admin", ): - """ + """ Initialization for instance of the class to setup basic endpoint info. Args: @@ -122,59 +122,59 @@

Source code for gen3.metadata

                 token, required for admin endpoints
             service_location (str, optional): deployment location relative to the
                 endpoint provided
-        """
+        """
         # legacy interface required endpoint as 1st arg
         if endpoint and isinstance(endpoint, Gen3Auth):
             auth_provider = endpoint
             endpoint = None
         if auth_provider and isinstance(auth_provider, Gen3Auth):
             endpoint = auth_provider.endpoint
-        endpoint = endpoint.strip("/")
+        endpoint = endpoint.strip("/")
         # if running locally, mds is deployed by itself without a location relative
         # to the commons
-        if "http://localhost" in endpoint:
-            service_location = ""
-            admin_endpoint_suffix = ""
+        if "http://localhost" in endpoint:
+            service_location = ""
+            admin_endpoint_suffix = ""
 
         if not endpoint.endswith(service_location):
-            endpoint += "/" + service_location
+            endpoint += "/" + service_location
 
-        self.endpoint = endpoint.rstrip("/")
-        self.admin_endpoint = endpoint.rstrip("/") + admin_endpoint_suffix
+        self.endpoint = endpoint.rstrip("/")
+        self.admin_endpoint = endpoint.rstrip("/") + admin_endpoint_suffix
         self._auth_provider = auth_provider
 
 
[docs] def is_healthy(self): - """ + """ Return if is healthy or not Returns: bool: True if healthy - """ + """ try: response = requests.get( - self.endpoint + "/_status", auth=self._auth_provider + self.endpoint + "/_status", auth=self._auth_provider ) response.raise_for_status() except Exception as exc: logging.error(exc) return False - return response.json().get("status") == "OK"
+ return response.json().get("status") == "OK"
[docs] @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS) def get_version(self): - """ + """ Return the version Returns: str: the version - """ - response = requests.get(self.endpoint + "/version", auth=self._auth_provider) + """ + response = requests.get(self.endpoint + "/version", auth=self._auth_provider) response.raise_for_status() return response.text
@@ -183,14 +183,14 @@

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def get_index_key_paths(self):
-        """
+        """
         List all the metadata key paths indexed in the database.
 
         Returns:
             List: list of metadata key paths
-        """
+        """
         response = requests.get(
-            self.admin_endpoint + "/metadata_index", auth=self._auth_provider
+            self.admin_endpoint + "/metadata_index", auth=self._auth_provider
         )
         response.raise_for_status()
         return response.json()
@@ -200,14 +200,14 @@

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def create_index_key_path(self, path):
-        """
+        """
         Create a metadata key path indexed in the database.
 
         Args:
             path (str): metadata key path
-        """
+        """
         response = requests.post(
-            self.admin_endpoint + f"/metadata_index/{path}", auth=self._auth_provider
+            self.admin_endpoint + f"/metadata_index/{path}", auth=self._auth_provider
         )
         response.raise_for_status()
         return response.json()
@@ -217,14 +217,14 @@

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def delete_index_key_path(self, path):
-        """
+        """
         List all the metadata key paths indexed in the database.
 
         Args:
             path (str): metadata key path
-        """
+        """
         response = requests.delete(
-            self.admin_endpoint + f"/metadata_index/{path}", auth=self._auth_provider
+            self.admin_endpoint + f"/metadata_index/{path}", auth=self._auth_provider
         )
         response.raise_for_status()
         return response
@@ -242,38 +242,38 @@

Source code for gen3.metadata

         use_agg_mds=False,
         **kwargs,
     ):
-        """
+        """
         Query the metadata given a query.
 
         Query format is based off the logic used in the service:
-            '''
+            '''
             Without filters, this will return all data. Add filters as query strings like this:
 
             GET /metadata?a=1&b=2
 
         This will match all records that have metadata containing all of:
-            {"a": 1, "b": 2}
+            {"a": 1, "b": 2}
             The values are always treated as strings for filtering. Nesting is supported:
 
             GET /metadata?a.b.c=3
 
         Matching records containing:
-            {"a": {"b": {"c": 3}}}
+            {"a": {"b": {"c": 3}}}
             Providing the same key with more than one value filters records whose value of the given key matches any of the given values. But values of different keys must all match. For example:
 
             GET /metadata?a.b.c=3&a.b.c=33&a.b.d=4
 
         Matches these:
-            {"a": {"b": {"c": 3, "d": 4}}}
-            {"a": {"b": {"c": 33, "d": 4}}}
-            {"a": {"b": {"c": "3", "d": 4, "e": 5}}}
-            But won't match these:
+            {"a": {"b": {"c": 3, "d": 4}}}
+            {"a": {"b": {"c": 33, "d": 4}}}
+            {"a": {"b": {"c": "3", "d": 4, "e": 5}}}
+            But won't match these:
 
-            {"a": {"b": {"c": 3}}}
-            {"a": {"b": {"c": 3, "d": 5}}}
-            {"a": {"b": {"d": 5}}}
-            {"a": {"b": {"c": "333", "d": 4}}}
-            '''
+            {"a": {"b": {"c": 3}}}
+            {"a": {"b": {"c": 3, "d": 5}}}
+            {"a": {"b": {"d": 5}}}
+            {"a": {"b": {"c": "333", "d": 4}}}
+            '''
 
         Args:
             query (str): mds query as defined by the metadata api
@@ -286,14 +286,14 @@ 

Source code for gen3.metadata

                 OR if return_full_metadata=True
             Dict{guid: {metadata}}: Dictionary with GUIDs as keys and associated
                 metadata JSON blobs as values
-        """
+        """
 
-        url = self.endpoint + f"/metadata?{query}"
+        url = self.endpoint + f"/metadata?{query}"
 
         url_with_params = append_query_params(
             url, data=return_full_metadata, limit=limit, offset=offset, **kwargs
         )
-        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"hitting: {url_with_params}")
         response = requests.get(url_with_params, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -304,7 +304,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     async def async_get(self, guid, _ssl=None, **kwargs):
-        """
+        """
         Asynchronous function to get metadata
 
         Args:
@@ -313,12 +313,12 @@ 

Source code for gen3.metadata

 
         Returns:
             Dict: metadata for given guid
-        """
+        """
         async with aiohttp.ClientSession() as session:
-            url = self.endpoint + f"/metadata/{guid}"
+            url = self.endpoint + f"/metadata/{guid}"
             url_with_params = append_query_params(url, **kwargs)
 
-            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"hitting: {url_with_params}")
 
             async with session.get(url_with_params, ssl=_ssl) as response:
                 response.raise_for_status()
@@ -331,16 +331,16 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def get(self, guid, **kwargs):
-        """
+        """
         Get the metadata associated with the guid
         Args:
             guid (str): guid to use
         Returns:
             Dict: metadata for given guid
-        """
-        url = self.endpoint + f"/metadata/{guid}"
+        """
+        url = self.endpoint + f"/metadata/{guid}"
         url_with_params = append_query_params(url, **kwargs)
-        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"hitting: {url_with_params}")
         response = requests.get(url_with_params, auth=self._auth_provider)
 
         response.raise_for_status()
@@ -352,30 +352,30 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def batch_create(self, metadata_list, overwrite=True, **kwargs):
-        """
+        """
         Create the list of metadata associated with the list of guids
 
         Args:
-            metadata_list (List[Dict{"guid": "", "data": {}}]): list of metadata
-                objects in a specific format. Expects a dict with "guid" and "data"
-                fields where "data" is another JSON blob to add to the mds
+            metadata_list (List[Dict{"guid": "", "data": {}}]): list of metadata
+                objects in a specific format. Expects a dict with "guid" and "data"
+                fields where "data" is another JSON blob to add to the mds
             overwrite (bool, optional): whether or not to overwrite existing data
-        """
-        url = self.admin_endpoint + f"/metadata"
+        """
+        url = self.admin_endpoint + f"/metadata"
 
         if len(metadata_list) > 1 and (
-            "guid" not in metadata_list[0] and "data" not in metadata_list[0]
+            "guid" not in metadata_list[0] and "data" not in metadata_list[0]
         ):
             logging.warning(
-                "it looks like your metadata list for bulk create is malformed. "
-                "the expected format is a list of dicts that have 2 keys: 'guid' "
-                "and 'data', where 'guid' is a string and 'data' is another dict. "
-                f"The first element doesn't match that pattern: {metadata_list[0]}"
+                "it looks like your metadata list for bulk create is malformed. "
+                "the expected format is a list of dicts that have 2 keys: 'guid' "
+                "and 'data', where 'guid' is a string and 'data' is another dict. "
+                f"The first element doesn't match that pattern: {metadata_list[0]}"
             )
 
         url_with_params = append_query_params(url, overwrite=overwrite, **kwargs)
-        logging.debug(f"hitting: {url_with_params}")
-        logging.debug(f"data: {metadata_list}")
+        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"data: {metadata_list}")
         response = requests.post(
             url_with_params, json=metadata_list, auth=self._auth_provider
         )
@@ -388,7 +388,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def create(self, guid, metadata, aliases=None, overwrite=False, **kwargs):
-        """
+        """
         Create the metadata associated with the guid
 
         Args:
@@ -396,14 +396,14 @@ 

Source code for gen3.metadata

             metadata (Dict): dictionary representing what will end up a JSON blob
                 attached to the provided GUID as metadata
             overwrite (bool, optional): whether or not to overwrite existing data
-        """
+        """
         aliases = aliases or []
 
-        url = self.admin_endpoint + f"/metadata/{guid}"
+        url = self.admin_endpoint + f"/metadata/{guid}"
 
         url_with_params = append_query_params(url, overwrite=overwrite, **kwargs)
-        logging.debug(f"hitting: {url_with_params}")
-        logging.debug(f"data: {metadata}")
+        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"data: {metadata}")
         response = requests.post(
             url_with_params, json=metadata, auth=self._auth_provider
         )
@@ -414,10 +414,10 @@ 

Source code for gen3.metadata

                 self.create_aliases(guid=guid, aliases=aliases, merge=overwrite)
             except Exception:
                 logging.error(
-                    "Error while attempting to create aliases: "
-                    f"'{aliases}' to GUID: '{guid}' with merge={overwrite}. "
-                    "GUID metadata record was created successfully and "
-                    "will NOT be deleted."
+                    "Error while attempting to create aliases: "
+                    f"'{aliases}' to GUID: '{guid}' with merge={overwrite}. "
+                    "GUID metadata record was created successfully and "
+                    "will NOT be deleted."
                 )
 
         return response.json()
@@ -435,7 +435,7 @@

Source code for gen3.metadata

         _ssl=None,
         **kwargs,
     ):
-        """
+        """
         Asynchronous function to create metadata
 
         Args:
@@ -444,19 +444,19 @@ 

Source code for gen3.metadata

                 attached to the provided GUID as metadata
             overwrite (bool, optional): whether or not to overwrite existing data
             _ssl (None, optional): whether or not to use ssl
-        """
+        """
         aliases = aliases or []
 
         async with aiohttp.ClientSession() as session:
-            url = self.admin_endpoint + f"/metadata/{guid}"
+            url = self.admin_endpoint + f"/metadata/{guid}"
             url_with_params = append_query_params(url, overwrite=overwrite, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
-            logging.debug(f"hitting: {url_with_params}")
-            logging.debug(f"data: {metadata}")
+            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"data: {metadata}")
             async with session.post(
                 url_with_params, json=metadata, headers=headers, ssl=_ssl
             ) as response:
@@ -464,17 +464,17 @@ 

Source code for gen3.metadata

                 response = await response.json()
 
             if aliases:
-                logging.info(f"creating aliases: {aliases}")
+                logging.info(f"creating aliases: {aliases}")
                 try:
                     await self.async_create_aliases(
                         guid=guid, aliases=aliases, _ssl=_ssl
                     )
                 except Exception:
                     logging.error(
-                        "Error while attempting to create aliases: "
-                        f"'{aliases}' to GUID: '{guid}'. "
-                        "GUID metadata record was created successfully and "
-                        "will NOT be deleted."
+                        "Error while attempting to create aliases: "
+                        f"'{aliases}' to GUID: '{guid}'. "
+                        "GUID metadata record was created successfully and "
+                        "will NOT be deleted."
                     )
 
         return response
@@ -484,21 +484,21 @@

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def update(self, guid, metadata, aliases=None, merge=False, **kwargs):
-        """
+        """
         Update the metadata associated with the guid
 
         Args:
             guid (str): guid to use
             metadata (Dict): dictionary representing what will end up a JSON blob
                 attached to the provided GUID as metadata
-        """
+        """
         aliases = aliases or []
 
-        url = self.admin_endpoint + f"/metadata/{guid}"
+        url = self.admin_endpoint + f"/metadata/{guid}"
 
         url_with_params = append_query_params(url, **kwargs)
-        logging.debug(f"hitting: {url_with_params}")
-        logging.debug(f"data: {metadata}")
+        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"data: {metadata}")
         response = requests.put(
             url_with_params, json=metadata, auth=self._auth_provider
         )
@@ -509,10 +509,10 @@ 

Source code for gen3.metadata

                 self.update_aliases(guid=guid, aliases=aliases, merge=merge)
             except Exception:
                 logging.error(
-                    "Error while attempting to update aliases: "
-                    f"'{aliases}' to GUID: '{guid}'. "
-                    "GUID metadata record was created successfully and "
-                    "will NOT be deleted."
+                    "Error while attempting to update aliases: "
+                    f"'{aliases}' to GUID: '{guid}'. "
+                    "GUID metadata record was created successfully and "
+                    "will NOT be deleted."
                 )
 
         return response.json()
@@ -524,7 +524,7 @@

Source code for gen3.metadata

     async def async_update(
         self, guid, metadata, aliases=None, merge=False, _ssl=None, **kwargs
     ):
-        """
+        """
         Asynchronous function to update metadata
 
         Args:
@@ -536,16 +536,16 @@ 

Source code for gen3.metadata

               with existing values
             _ssl (None, optional): whether or not to use ssl
             **kwargs: Description
-        """
+        """
         aliases = aliases or []
 
         async with aiohttp.ClientSession() as session:
-            url = self.admin_endpoint + f"/metadata/{guid}"
+            url = self.admin_endpoint + f"/metadata/{guid}"
             url_with_params = append_query_params(url, merge=merge, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
             async with session.put(
                 url_with_params, json=metadata, headers=headers, ssl=_ssl
@@ -560,10 +560,10 @@ 

Source code for gen3.metadata

                     )
                 except Exception:
                     logging.error(
-                        "Error while attempting to update aliases: "
-                        f"'{aliases}' to GUID: '{guid}' with merge={merge}. "
-                        "GUID metadata record was created successfully and "
-                        "will NOT be deleted."
+                        "Error while attempting to update aliases: "
+                        f"'{aliases}' to GUID: '{guid}' with merge={merge}. "
+                        "GUID metadata record was created successfully and "
+                        "will NOT be deleted."
                     )
 
         return response
@@ -573,16 +573,16 @@

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **DEFAULT_BACKOFF_SETTINGS)
     def delete(self, guid, **kwargs):
-        """
+        """
         Delete the metadata associated with the guid
 
         Args:
             guid (str): guid to use
-        """
-        url = self.admin_endpoint + f"/metadata/{guid}"
+        """
+        url = self.admin_endpoint + f"/metadata/{guid}"
 
         url_with_params = append_query_params(url, **kwargs)
-        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"hitting: {url_with_params}")
         response = requests.delete(url_with_params, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -597,7 +597,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def get_aliases(self, guid, **kwargs):
-        """
+        """
         Get Aliases for the given guid
 
         Args:
@@ -606,11 +606,11 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to get aliases
-        """
-        url = self.endpoint + f"/metadata/{guid}/aliases"
+        """
+        url = self.endpoint + f"/metadata/{guid}/aliases"
         url_with_params = append_query_params(url, **kwargs)
 
-        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"hitting: {url_with_params}")
         response = requests.get(url_with_params, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -621,7 +621,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     async def async_get_aliases(self, guid, _ssl=None, **kwargs):
-        """
+        """
         Asyncronously get Aliases for the given guid
 
         Args:
@@ -631,16 +631,16 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to get aliases
-        """
+        """
         async with aiohttp.ClientSession() as session:
-            url = self.endpoint + f"/metadata/{guid}/aliases"
+            url = self.endpoint + f"/metadata/{guid}/aliases"
             url_with_params = append_query_params(url, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
-            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"hitting: {url_with_params}")
             async with session.get(
                 url_with_params, headers=headers, ssl=_ssl
             ) as response:
@@ -651,7 +651,7 @@ 

Source code for gen3.metadata

 
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def delete_alias(self, guid, alias, **kwargs):
-        """
+        """
         Delete single Alias for the given guid
 
         Args:
@@ -660,11 +660,11 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to delete aliases
-        """
-        url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
+        """
+        url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
         url_with_params = append_query_params(url, **kwargs)
 
-        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"hitting: {url_with_params}")
         response = requests.delete(url_with_params, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -672,7 +672,7 @@ 

Source code for gen3.metadata

 
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     async def async_delete_alias(self, guid, alias, _ssl=None, **kwargs):
-        """
+        """
         Asyncronously delete single Aliases for the given guid
 
         Args:
@@ -682,16 +682,16 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to delete aliases
-        """
+        """
         async with aiohttp.ClientSession() as session:
-            url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
+            url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
             url_with_params = append_query_params(url, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
-            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"hitting: {url_with_params}")
             async with session.delete(
                 url_with_params, headers=headers, ssl=_ssl
             ) as response:
@@ -703,7 +703,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def create_aliases(self, guid, aliases, **kwargs):
-        """
+        """
         Create Aliases for the given guid
 
         Args:
@@ -713,14 +713,14 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to create aliases
-        """
-        url = self.admin_endpoint + f"/metadata/{guid}/aliases"
+        """
+        url = self.admin_endpoint + f"/metadata/{guid}/aliases"
         url_with_params = append_query_params(url, **kwargs)
 
-        data = {"aliases": aliases}
+        data = {"aliases": aliases}
 
-        logging.debug(f"hitting: {url_with_params}")
-        logging.debug(f"data: {data}")
+        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"data: {data}")
         response = requests.post(url_with_params, json=data, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -731,7 +731,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     async def async_create_aliases(self, guid, aliases, _ssl=None, **kwargs):
-        """
+        """
         Asyncronously create Aliases for the given guid
 
         Args:
@@ -742,19 +742,19 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to create aliases
-        """
+        """
         async with aiohttp.ClientSession() as session:
-            url = self.admin_endpoint + f"/metadata/{guid}/aliases"
+            url = self.admin_endpoint + f"/metadata/{guid}/aliases"
             url_with_params = append_query_params(url, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
-            data = {"aliases": aliases}
+            data = {"aliases": aliases}
 
-            logging.debug(f"hitting: {url_with_params}")
-            logging.debug(f"data: {data}")
+            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"data: {data}")
             async with session.post(
                 url_with_params, json=data, headers=headers, ssl=_ssl
             ) as response:
@@ -766,7 +766,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def update_aliases(self, guid, aliases, merge=False, **kwargs):
-        """
+        """
         Update Aliases for the given guid
 
         Args:
@@ -777,14 +777,14 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to update aliases
-        """
-        url = self.admin_endpoint + f"/metadata/{guid}/aliases"
+        """
+        url = self.admin_endpoint + f"/metadata/{guid}/aliases"
         url_with_params = append_query_params(url, merge=merge, **kwargs)
 
-        data = {"aliases": aliases}
+        data = {"aliases": aliases}
 
-        logging.debug(f"hitting: {url_with_params}")
-        logging.debug(f"data: {data}")
+        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"data: {data}")
         response = requests.put(url_with_params, json=data, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -797,7 +797,7 @@ 

Source code for gen3.metadata

     async def async_update_aliases(
         self, guid, aliases, merge=False, _ssl=None, **kwargs
     ):
-        """
+        """
         Asyncronously update Aliases for the given guid
 
         Args:
@@ -809,19 +809,19 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to update aliases
-        """
+        """
         async with aiohttp.ClientSession() as session:
-            url = self.admin_endpoint + f"/metadata/{guid}/aliases"
+            url = self.admin_endpoint + f"/metadata/{guid}/aliases"
             url_with_params = append_query_params(url, merge=merge, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
-            data = {"aliases": aliases}
+            data = {"aliases": aliases}
 
-            logging.debug(f"hitting: {url_with_params}")
-            logging.debug(f"data: {data}")
+            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"data: {data}")
             async with session.put(
                 url_with_params, json=data, headers=headers, ssl=_ssl
             ) as response:
@@ -834,7 +834,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def delete_aliases(self, guid, **kwargs):
-        """
+        """
         Delete all Aliases for the given guid
 
         Args:
@@ -843,11 +843,11 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to delete aliases
-        """
-        url = self.admin_endpoint + f"/metadata/{guid}/aliases"
+        """
+        url = self.admin_endpoint + f"/metadata/{guid}/aliases"
         url_with_params = append_query_params(url, **kwargs)
 
-        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"hitting: {url_with_params}")
         response = requests.delete(url_with_params, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -858,7 +858,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     async def async_delete_aliases(self, guid, _ssl=None, **kwargs):
-        """
+        """
         Asyncronously delete all Aliases for the given guid
 
         Args:
@@ -868,16 +868,16 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to delete aliases
-        """
+        """
         async with aiohttp.ClientSession() as session:
-            url = self.admin_endpoint + f"/metadata/{guid}/aliases"
+            url = self.admin_endpoint + f"/metadata/{guid}/aliases"
             url_with_params = append_query_params(url, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
-            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"hitting: {url_with_params}")
             async with session.delete(
                 url_with_params, headers=headers, ssl=_ssl
             ) as response:
@@ -890,7 +890,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     def delete_alias(self, guid, alias, **kwargs):
-        """
+        """
         Delete single Alias for the given guid
 
         Args:
@@ -900,11 +900,11 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to delete aliases
-        """
-        url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
+        """
+        url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
         url_with_params = append_query_params(url, **kwargs)
 
-        logging.debug(f"hitting: {url_with_params}")
+        logging.debug(f"hitting: {url_with_params}")
         response = requests.delete(url_with_params, auth=self._auth_provider)
         response.raise_for_status()
 
@@ -915,7 +915,7 @@ 

Source code for gen3.metadata

 [docs]
     @backoff.on_exception(backoff.expo, Exception, **BACKOFF_NO_LOG_IF_NOT_RETRIED)
     async def async_delete_alias(self, guid, alias, _ssl=None, **kwargs):
-        """
+        """
         Asyncronously delete single Aliases for the given guid
 
         Args:
@@ -926,16 +926,16 @@ 

Source code for gen3.metadata

 
         Returns:
             requests.Response: response from the request to delete aliases
-        """
+        """
         async with aiohttp.ClientSession() as session:
-            url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
+            url = self.admin_endpoint + f"/metadata/{guid}/aliases/{alias}"
             url_with_params = append_query_params(url, **kwargs)
 
             # aiohttp only allows basic auth with their built in auth, so we
             # need to manually add JWT auth header
-            headers = {"Authorization": self._auth_provider._get_auth_value()}
+            headers = {"Authorization": self._auth_provider._get_auth_value()}
 
-            logging.debug(f"hitting: {url_with_params}")
+            logging.debug(f"hitting: {url_with_params}")
             async with session.delete(
                 url_with_params, headers=headers, ssl=_ssl
             ) as response:
@@ -947,11 +947,11 @@ 

Source code for gen3.metadata

     def _prepare_metadata(
         self, metadata, indexd_doc, force_metadata_columns_even_if_empty
     ):
-        """
+        """
         Validate and generate the provided metadata for submission to the metadata
         service.
 
-        If the record is of type "package", also prepare package metadata.
+        If the record is of type "package", also prepare package metadata.
 
         Args:
             metadata (dict): metadata provided by the submitter
@@ -960,13 +960,13 @@ 

Source code for gen3.metadata

 
         Returns:
             dict: metadata ready to be submitted to the metadata service
-        """
+        """
 
         def _extract_non_indexd_metadata(metadata):
-            """
-            Get the "additional metadata": metadata that was provided but is
+            """
+            Get the "additional metadata": metadata that was provided but is
             not stored in indexd, so should be stored in the metadata service.
-            """
+            """
             return {
                 k: v
                 for k, v in metadata.items()
@@ -987,14 +987,14 @@ 

Source code for gen3.metadata

         valid = True
 
         # validate package columns
-        record_type = to_submit.pop(RECORD_TYPE_STANDARD_KEY, "").strip().lower()
+        record_type = to_submit.pop(RECORD_TYPE_STANDARD_KEY, "").strip().lower()
         package_contents = to_submit.pop(PACKAGE_CONTENTS_STANDARD_KEY, None)
-        if record_type == "package":
+        if record_type == "package":
             if package_contents:
                 package_contents = json.loads(package_contents)
                 if not _verify_schema(package_contents, PACKAGE_CONTENTS_SCHEMA):
                     logging.error(
-                        f"ERROR: {package_contents} is not in package contents format"
+                        f"ERROR: {package_contents} is not in package contents format"
                     )
                     valid = False
             # generate package metadata
@@ -1009,19 +1009,19 @@ 

Source code for gen3.metadata

             to_submit.update(package_metadata)
         elif package_contents:
             logging.error(
-                f"ERROR: tried to set '{PACKAGE_CONTENTS_STANDARD_KEY}' for a non-package row. Ignoring '{PACKAGE_CONTENTS_STANDARD_KEY}'. Set '{RECORD_TYPE_STANDARD_KEY}' to 'package' to create packages."
+                f"ERROR: tried to set '{PACKAGE_CONTENTS_STANDARD_KEY}' for a non-package row. Ignoring '{PACKAGE_CONTENTS_STANDARD_KEY}'. Set '{RECORD_TYPE_STANDARD_KEY}' to 'package' to create packages."
             )
             valid = False
 
         if not valid:
-            raise Exception(f"Metadata is not valid: {metadata}")
+            raise Exception(f"Metadata is not valid: {metadata}")
 
         if not force_metadata_columns_even_if_empty:
-            # remove any empty columns if we're not being forced to include them
+            # remove any empty columns if we're not being forced to include them
             to_submit = {
                 key: value
                 for key, value in to_submit.items()
-                if value is not None and value != ""
+                if value is not None and value != ""
             }
 
         return to_submit
@@ -1029,18 +1029,18 @@ 

Source code for gen3.metadata

     def _get_package_metadata(
         self, submitted_metadata, file_name, file_size, hashes, urls, contents
     ):
-        """
+        """
         The MDS Objects API currently expects files that have not been
         uploaded yet. For files we only needs to index, not upload, create
         object records manually by generating the expected object fields.
         TODO: update the MDS objects API to not create upload URLs if the
         relevant data is provided.
-        """
+        """
 
         def _get_filename_from_urls(submitted_metadata, urls):
-            file_name = ""
+            file_name = ""
             if not urls:
-                logging.warning(f"No URLs provided for: {submitted_metadata}")
+                logging.warning(f"No URLs provided for: {submitted_metadata}")
             for url in urls:
                 _file_name = os.path.basename(url)
                 if not file_name:
@@ -1048,7 +1048,7 @@ 

Source code for gen3.metadata

                 else:
                     if file_name != _file_name:
                         logging.warning(
-                            f"Received multiple URLs with different file names; will use the first URL (file name '{file_name}'): {submitted_metadata}"
+                            f"Received multiple URLs with different file names; will use the first URL (file name '{file_name}'): {submitted_metadata}"
                         )
             return file_name
 
@@ -1058,17 +1058,17 @@ 

Source code for gen3.metadata

 
         now = str(datetime.utcnow())
         metadata = {
-            "type": "package",
-            "package": {
-                "version": "0.1",
-                "file_name": file_name,
-                "created_time": now,
-                "updated_time": now,
-                "size": file_size,
-                "hashes": hashes,
-                "contents": contents or None,
+            "type": "package",
+            "package": {
+                "version": "0.1",
+                "file_name": file_name,
+                "created_time": now,
+                "updated_time": now,
+                "size": file_size,
+                "hashes": hashes,
+                "contents": contents or None,
             },
-            "_upload_status": "uploaded",
+            "_upload_status": "uploaded",
         }
         return metadata
diff --git a/docs/_build/html/_modules/gen3/object.html b/docs/_build/html/_modules/gen3/object.html index 19cc6697..ac26f96c 100644 --- a/docs/_build/html/_modules/gen3/object.html +++ b/docs/_build/html/_modules/gen3/object.html @@ -42,7 +42,7 @@

Source code for gen3.object

 
[docs] class Gen3Object: - """For interacting with Gen3 object level features. + """For interacting with Gen3 object level features. A class for interacting with the Gen3 object services. Currently allows creating and deleting of an object from the Gen3 System. @@ -54,36 +54,36 @@

Source code for gen3.object

         This generates the Gen3Object class pointed at the sandbox commons while
         using the credentials.json downloaded from the commons profile page.
 
-        >>> auth = Gen3Auth(refresh_file="credentials.json")
+        >>> auth = Gen3Auth(refresh_file="credentials.json")
         ... object = Gen3Object(auth)
 
-    """
+    """
 
     def __init__(self, auth_provider=None):
         self._auth_provider = auth_provider
-        self.service_endpoint = "/mds"
+        self.service_endpoint = "/mds"
 
     def create_object(self, file_name, authz, metadata=None, aliases=None):
         url = (
-            self._auth_provider.endpoint.rstrip("/")
+            self._auth_provider.endpoint.rstrip("/")
             + self.service_endpoint
-            + "/objects"
+            + "/objects"
         )
         body = {
-            "file_name": file_name,
-            "authz": authz,
-            "metadata": metadata,
-            "aliases": aliases,
+            "file_name": file_name,
+            "authz": authz,
+            "metadata": metadata,
+            "aliases": aliases,
         }
         response = requests.post(url, json=body, auth=self._auth_provider)
         raise_for_status_and_print_error(response)
         data = response.json()
-        return data["guid"], data["upload_url"]
+        return data["guid"], data["upload_url"]
 
 
[docs] def delete_object(self, guid, delete_file_locations=False): - """ + """ Delete the object from indexd, metadata service and optionally all storage locations Args: @@ -91,12 +91,12 @@

Source code for gen3.object

             `delete_file_locations` -- if True, removes the object from existing bucket location(s) through fence
         Returns:
             Nothing
-        """
-        delete_param = "?delete_file_locations" if delete_file_locations else ""
+        """
+        delete_param = "?delete_file_locations" if delete_file_locations else ""
         url = (
-            self._auth_provider.endpoint.rstrip("/")
+            self._auth_provider.endpoint.rstrip("/")
             + self.service_endpoint
-            + "/objects/"
+            + "/objects/"
             + guid
             + delete_param
         )
diff --git a/docs/_build/html/_modules/gen3/query.html b/docs/_build/html/_modules/gen3/query.html
index af258709..295750ef 100644
--- a/docs/_build/html/_modules/gen3/query.html
+++ b/docs/_build/html/_modules/gen3/query.html
@@ -39,7 +39,7 @@ 

Source code for gen3.query

 
[docs] class Gen3Query: - """ + """ Query ElasticSearch data from a Gen3 system. Args: @@ -49,9 +49,9 @@

Source code for gen3.query

         This generates the Gen3Query class pointed at the sandbox commons while
         using the credentials.json downloaded from the commons profile page.
 
-        >>> auth = Gen3Auth(endpoint, refresh_file="credentials.json")
+        >>> auth = Gen3Auth(endpoint, refresh_file="credentials.json")
         ... query = Gen3Query(auth)
-    """
+    """
 
     def __init__(self, auth_provider):
         self._auth_provider = auth_provider
@@ -70,7 +70,7 @@ 

Source code for gen3.query

         accessibility=None,
         verbose=True,
     ):
-        """
+        """
         Execute a query against a Data Commons.
 
         Args:
@@ -81,23 +81,23 @@ 

Source code for gen3.query

             filters: (object, optional): { field: sort method } object. Will filter data with ALL fields EQUAL to the provided respective value. If more complex filters are needed, use the `filter_object` parameter instead.
             filter_object (object, optional): Filter to apply. For syntax details, see https://github.com/uc-cdis/guppy/blob/master/doc/queries.md#filter.
             sort_object (object, optional): { field: sort method } object.
-            accessibility (list, optional): One of ["accessible" (default), "unaccessible", "all"]. Only valid when querying a data type in "regular" tier access mode.
+            accessibility (list, optional): One of ["accessible" (default), "unaccessible", "all"]. Only valid when querying a data type in "regular" tier access mode.
 
         Returns:
-            Object: {"data": {<data_type>: [<record>, <record>, ...]}}
+            Object: {"data": {<data_type>: [<record>, <record>, ...]}}
 
         Examples:
             >>> Gen3Query.query(
-                data_type="subject",
+                data_type="subject",
                 first=50,
                 fields=[
-                    "vital_status",
-                    "submitter_id",
+                    "vital_status",
+                    "submitter_id",
                 ],
-                filters={"vital_status": "Alive"},
-                sort_object={"submitter_id": "asc"},
+                filters={"vital_status": "Alive"},
+                sort_object={"submitter_id": "asc"},
             )
-        """
+        """
         if not first:
             first = 10
         if not offset:
@@ -105,14 +105,14 @@ 

Source code for gen3.query

         if not sort_object:
             sort_object = {}
         if not accessibility:
-            accessibility = "accessible"
+            accessibility = "accessible"
         if filters and filter_object:
             raise Exception(
-                "Only one of `filters` and `filter_object` can be used at a time."
+                "Only one of `filters` and `filter_object` can be used at a time."
             )
         if filters:
             filter_object = {
-                "AND": [{"=": {field: val}} for field, val in filters.items()]
+                "AND": [{"=": {field: val}} for field, val in filters.items()]
             }
 
         if first + offset > 10000:  # ElasticSearch limitation
@@ -126,13 +126,13 @@ 

Source code for gen3.query

                 first=first,
                 offset=offset,
             )
-            return {"data": {data_type: data}}
+            return {"data": {data_type: data}}
 
-        # convert sort_object to graphql: [ { field_name: "sort_method" } ]
-        sorts = [f'{{{field}: "{val}"}}' for field, val in sort_object.items()]
-        sort_string = f'[{", ".join(sorts)}]'
+        # convert sort_object to graphql: [ { field_name: "sort_method" } ]
+        sorts = [f'{{{field}: "{val}"}}' for field, val in sort_object.items()]
+        sort_string = f'[{", ".join(sorts)}]'
 
-        query_string = f"""query($filter: JSON) {{
+        query_string = f"""query($filter: JSON) {{
             {data_type}(
                 first: {first},
                 offset: {offset},
@@ -140,17 +140,17 @@ 

Source code for gen3.query

                 accessibility: {accessibility},
                 filter: $filter
             ) {{
-                {" ".join(fields)}
+                {" ".join(fields)}
             }}
-        }}"""
-        variables = {"filter": filter_object}
+        }}"""
+        variables = {"filter": filter_object}
         return self.graphql_query(query_string=query_string, variables=variables)
[docs] def graphql_query(self, query_string, variables=None): - """ + """ Execute a GraphQL query against a Data Commons. Args: @@ -158,29 +158,29 @@

Source code for gen3.query

             variables (:obj:`object`, optional): Dictionary of variables to pass with the query.
 
         Returns:
-            Object: {"data": {<data_type>: [<record>, <record>, ...]}}
+            Object: {"data": {<data_type>: [<record>, <record>, ...]}}
 
         Examples:
-            >>> query_string = "{ my_index { my_field } }"
+            >>> query_string = "{ my_index { my_field } }"
             ... Gen3Query.graphql_query(query_string)
-        """
-        url = f"{self._auth_provider.endpoint}/guppy/graphql"
+        """
+        url = f"{self._auth_provider.endpoint}/guppy/graphql"
         response = requests.post(
             url,
-            json={"query": query_string, "variables": variables},
+            json={"query": query_string, "variables": variables},
             auth=self._auth_provider,
         )
         try:
             raise_for_status_and_print_error(response)
         except Exception:
             print(
-                f"Unable to query.\nQuery: {query_string}\nVariables: {variables}\n{response.text}"
+                f"Unable to query.\nQuery: {query_string}\nVariables: {variables}\n{response.text}"
             )
             raise
         try:
             return response.json()
         except Exception:
-            print(f"Did not receive JSON: {response.text}")
+            print(f"Did not receive JSON: {response.text}")
             raise
@@ -196,7 +196,7 @@

Source code for gen3.query

         first=None,
         offset=None,
     ):
-        """
+        """
         Execute a raw data download against a Data Commons.
 
         Args:
@@ -204,7 +204,7 @@ 

Source code for gen3.query

             fields (list): List of fields to return.
             filter_object (object, optional): Filter to apply. For syntax details, see https://github.com/uc-cdis/guppy/blob/master/doc/queries.md#filter.
             sort_fields (list, optional): List of { field: sort method } objects.
-            accessibility (list, optional): One of ["accessible" (default), "unaccessible", "all"]. Only valid when downloading from a data type in "regular" tier access mode.
+            accessibility (list, optional): One of ["accessible" (default), "unaccessible", "all"]. Only valid when downloading from a data type in "regular" tier access mode.
             first (int, optional): Number of rows to return (default: all rows).
             offset (int, optional): Starting position (default: 0).
 
@@ -213,29 +213,29 @@ 

Source code for gen3.query

 
         Examples:
             >>> Gen3Query.raw_data_download(
-                    data_type="subject",
+                    data_type="subject",
                     fields=[
-                        "vital_status",
-                        "submitter_id",
-                        "project_id"
+                        "vital_status",
+                        "submitter_id",
+                        "project_id"
                     ],
-                    filter_object={"=": {"project_id": "my_program-my_project"}},
-                    sort_fields=[{"submitter_id": "asc"}],
-                    accessibility="accessible"
+                    filter_object={"=": {"project_id": "my_program-my_project"}},
+                    sort_fields=[{"submitter_id": "asc"}],
+                    accessibility="accessible"
                 )
-        """
+        """
         if not accessibility:
-            accessibility = "accessible"
+            accessibility = "accessible"
         if not offset:
             offset = 0
 
-        body = {"type": data_type, "fields": fields, "accessibility": accessibility}
+        body = {"type": data_type, "fields": fields, "accessibility": accessibility}
         if filter_object:
-            body["filter"] = filter_object
+            body["filter"] = filter_object
         if sort_fields:
-            body["sort"] = sort_fields
+            body["sort"] = sort_fields
 
-        url = f"{self._auth_provider.endpoint}/guppy/download"
+        url = f"{self._auth_provider.endpoint}/guppy/download"
         response = requests.post(
             url,
             json=body,
@@ -244,12 +244,12 @@ 

Source code for gen3.query

         try:
             raise_for_status_and_print_error(response)
         except Exception:
-            print(f"Unable to download.\nBody: {body}\n{response.text}")
+            print(f"Unable to download.\nBody: {body}\n{response.text}")
             raise
         try:
             data = response.json()
         except Exception:
-            print(f"Did not receive JSON: {response.text}")
+            print(f"Did not receive JSON: {response.text}")
             raise
 
         if offset:
diff --git a/docs/_build/html/_modules/gen3/submission.html b/docs/_build/html/_modules/gen3/submission.html
index 04223a5f..b82fd483 100644
--- a/docs/_build/html/_modules/gen3/submission.html
+++ b/docs/_build/html/_modules/gen3/submission.html
@@ -58,7 +58,7 @@ 

Source code for gen3.submission

 
[docs] class Gen3Submission: - """Submit/Export/Query data from a Gen3 Submission system. + """Submit/Export/Query data from a Gen3 Submission system. A class for interacting with the Gen3 submission services. Supports submitting and exporting from Sheepdog. @@ -71,10 +71,10 @@

Source code for gen3.submission

         This generates the Gen3Submission class pointed at the sandbox commons while
         using the credentials.json downloaded from the commons profile page.
 
-        >>> auth = Gen3Auth(refresh_file="credentials.json")
+        >>> auth = Gen3Auth(refresh_file="credentials.json")
         ... sub = Gen3Submission(auth)
 
-    """
+    """
 
     def __init__(self, endpoint=None, auth_provider=None):
         # auth_provider legacy interface required endpoint as 1st arg
@@ -82,18 +82,18 @@ 

Source code for gen3.submission

         self._endpoint = self._auth_provider.endpoint
 
     def __export_file(self, filename, output):
-        """Writes an API response to a file."""
-        with open(filename, "w") as outfile:
+        """Writes an API response to a file."""
+        with open(filename, "w") as outfile:
             outfile.write(output)
-        print("\nOutput written to file: " + filename)
+        print("\nOutput written to file: " + filename)
 
     ### Program functions
 
 
[docs] def get_programs(self): - """List registered programs""" - api_url = f"{self._endpoint}/api/v0/submission/" + """List registered programs""" + api_url = f"{self._endpoint}/api/v0/submission/" output = requests.get(api_url, auth=self._auth_provider) raise_for_status_and_print_error(output) return output.json()
@@ -102,7 +102,7 @@

Source code for gen3.submission

 
[docs] def create_program(self, json): - """Create a program. + """Create a program. Args: json (object): The json of the program to create @@ -110,8 +110,8 @@

Source code for gen3.submission

             This creates a program in the sandbox commons.
 
             >>> Gen3Submission.create_program(json)
-        """
-        api_url = "{}/api/v0/submission/".format(self._endpoint)
+        """
+        api_url = "{}/api/v0/submission/".format(self._endpoint)
         output = requests.post(api_url, auth=self._auth_provider, json=json)
         raise_for_status_and_print_error(output)
         return output.json()
@@ -120,7 +120,7 @@

Source code for gen3.submission

 
[docs] def delete_program(self, program): - """Delete a program. + """Delete a program. This deletes an empty program from the commons. @@ -128,12 +128,12 @@

Source code for gen3.submission

             program (str): The program to delete.
 
         Examples:
-            This deletes the "DCF" program.
+            This deletes the "DCF" program.
 
-            >>> Gen3Submission.delete_program("DCF")
+            >>> Gen3Submission.delete_program("DCF")
 
-        """
-        api_url = "{}/api/v0/submission/{}".format(self._endpoint, program)
+        """
+        api_url = "{}/api/v0/submission/{}".format(self._endpoint, program)
         output = requests.delete(api_url, auth=self._auth_provider)
         raise_for_status_and_print_error(output)
         return output
@@ -144,7 +144,7 @@

Source code for gen3.submission

 
[docs] def get_projects(self, program): - """List registered projects for a given program + """List registered projects for a given program Args: program: the name of the program you want the projects from @@ -152,10 +152,10 @@

Source code for gen3.submission

         Example:
             This lists all the projects under the DCF program
 
-            >>> Gen3Submission.get_projects("DCF")
+            >>> Gen3Submission.get_projects("DCF")
 
-        """
-        api_url = f"{self._endpoint}/api/v0/submission/{program}"
+        """
+        api_url = f"{self._endpoint}/api/v0/submission/{program}"
         output = requests.get(api_url, auth=self._auth_provider)
         raise_for_status_and_print_error(output)
         return output.json()
@@ -164,7 +164,7 @@

Source code for gen3.submission

 
[docs] def create_project(self, program, json): - """Create a project. + """Create a project. Args: program (str): The program to create a project on json (object): The json of the project to create @@ -172,9 +172,9 @@

Source code for gen3.submission

         Examples:
             This creates a project on the DCF program in the sandbox commons.
 
-            >>> Gen3Submission.create_project("DCF", json)
-        """
-        api_url = "{}/api/v0/submission/{}".format(self._endpoint, program)
+            >>> Gen3Submission.create_project("DCF", json)
+        """
+        api_url = "{}/api/v0/submission/{}".format(self._endpoint, program)
         output = requests.put(api_url, auth=self._auth_provider, json=json)
         raise_for_status_and_print_error(output)
         return output.json()
@@ -183,7 +183,7 @@

Source code for gen3.submission

 
[docs] def delete_project(self, program, project): - """Delete a project. + """Delete a project. This deletes an empty project from the commons. @@ -192,12 +192,12 @@

Source code for gen3.submission

             project (str): The project to delete.
 
         Examples:
-            This deletes the "CCLE" project from the "DCF" program.
+            This deletes the "CCLE" project from the "DCF" program.
 
-            >>> Gen3Submission.delete_project("DCF", "CCLE")
+            >>> Gen3Submission.delete_project("DCF", "CCLE")
 
-        """
-        api_url = "{}/api/v0/submission/{}/{}".format(self._endpoint, program, project)
+        """
+        api_url = "{}/api/v0/submission/{}/{}".format(self._endpoint, program, project)
         output = requests.delete(api_url, auth=self._auth_provider)
         raise_for_status_and_print_error(output)
         return output
@@ -206,7 +206,7 @@

Source code for gen3.submission

 
[docs] def get_project_dictionary(self, program, project): - """Get dictionary schema for a given project + """Get dictionary schema for a given project Args: program: the name of the program the project is from @@ -214,10 +214,10 @@

Source code for gen3.submission

 
         Example:
 
-            >>> Gen3Submission.get_project_dictionary("DCF", "CCLE")
+            >>> Gen3Submission.get_project_dictionary("DCF", "CCLE")
 
-        """
-        api_url = f"{self._endpoint}/api/v0/submission/{program}/{project}/_dictionary"
+        """
+        api_url = f"{self._endpoint}/api/v0/submission/{program}/{project}/_dictionary"
         output = requests.get(api_url, auth=self._auth_provider)
         raise_for_status_and_print_error(output)
         return output.json()
@@ -226,18 +226,18 @@

Source code for gen3.submission

 
[docs] def open_project(self, program, project): - """Mark a project ``open``. Opening a project means uploads, deletions, etc. are allowed. + """Mark a project ``open``. Opening a project means uploads, deletions, etc. are allowed. Args: program: the name of the program the project is from - project: the name of the project you want to 'open' + project: the name of the project you want to 'open' Example: - >>> Gen3Submission.get_project_manifest("DCF", "CCLE") + >>> Gen3Submission.get_project_manifest("DCF", "CCLE") - """ - api_url = f"{self._endpoint}/api/v0/submission/{program}/{project}/open" + """ + api_url = f"{self._endpoint}/api/v0/submission/{program}/{project}/open" output = requests.put(api_url, auth=self._auth_provider) raise_for_status_and_print_error(output) return output.json()
@@ -248,7 +248,7 @@

Source code for gen3.submission

 
[docs] def submit_record(self, program, project, json): - """Submit record(s) to a project as json. + """Submit record(s) to a project as json. Args: program (str): The program to submit to. @@ -258,11 +258,11 @@

Source code for gen3.submission

         Examples:
             This submits records to the CCLE project in the sandbox commons.
 
-            >>> Gen3Submission.submit_record("DCF", "CCLE", json)
+            >>> Gen3Submission.submit_record("DCF", "CCLE", json)
 
-        """
-        api_url = "{}/api/v0/submission/{}/{}".format(self._endpoint, program, project)
-        logging.debug("Using the Sheepdog API URL: {}".format(api_url))
+        """
+        api_url = "{}/api/v0/submission/{}/{}".format(self._endpoint, program, project)
+        logging.debug("Using the Sheepdog API URL: {}".format(api_url))
 
         output = requests.put(api_url, auth=self._auth_provider, json=json)
         raise_for_status_and_print_error(output)
@@ -272,7 +272,7 @@ 

Source code for gen3.submission

 
[docs] def delete_record(self, program, project, uuid): - """ + """ Delete a record from a project. Args: @@ -283,15 +283,15 @@

Source code for gen3.submission

         Examples:
             This deletes a record from the CCLE project in the sandbox commons.
 
-            >>> Gen3Submission.delete_record("DCF", "CCLE", uuid)
-        """
+            >>> Gen3Submission.delete_record("DCF", "CCLE", uuid)
+        """
         return self.delete_records(program, project, [uuid])
[docs] def delete_records(self, program, project, uuids, batch_size=100): - """ + """ Delete a list of records from a project. Args: @@ -303,11 +303,11 @@

Source code for gen3.submission

         Examples:
             This deletes a list of records from the CCLE project in the sandbox commons.
 
-            >>> Gen3Submission.delete_records("DCF", "CCLE", ["uuid1", "uuid2"])
-        """
+            >>> Gen3Submission.delete_records("DCF", "CCLE", ["uuid1", "uuid2"])
+        """
         if not uuids:
             return
-        api_url = "{}/api/v0/submission/{}/{}/entities".format(
+        api_url = "{}/api/v0/submission/{}/{}/entities".format(
             self._endpoint, program, project
         )
         for i in itertools.count():
@@ -315,14 +315,14 @@ 

Source code for gen3.submission

             if len(uuids_to_delete) == 0:
                 break
             output = requests.delete(
-                "{}/{}".format(api_url, ",".join(uuids_to_delete)),
+                "{}/{}".format(api_url, ",".join(uuids_to_delete)),
                 auth=self._auth_provider,
             )
             try:
                 raise_for_status_and_print_error(output)
             except requests.exceptions.HTTPError:
                 print(
-                    "\n{}\nFailed to delete uuids: {}".format(
+                    "\n{}\nFailed to delete uuids: {}".format(
                         output.text, uuids_to_delete
                     )
                 )
@@ -333,7 +333,7 @@ 

Source code for gen3.submission

 
[docs] def delete_node(self, program, project, node_name, batch_size=100, verbose=True): - """ + """ Delete all records for a node from a project. Args: @@ -346,8 +346,8 @@

Source code for gen3.submission

         Examples:
             This deletes a node from the CCLE project in the sandbox commons.
 
-            >>> Gen3Submission.delete_node("DCF", "CCLE", "demographic")
-        """
+            >>> Gen3Submission.delete_node("DCF", "CCLE", "demographic")
+        """
         return self.delete_nodes(
             program, project, [node_name], batch_size, verbose=verbose
         )
@@ -358,7 +358,7 @@

Source code for gen3.submission

     def delete_nodes(
         self, program, project, ordered_node_list, batch_size=100, verbose=True
     ):
-        """
+        """
         Delete all records for a list of nodes from a project.
 
         Args:
@@ -371,28 +371,28 @@ 

Source code for gen3.submission

         Examples:
             This deletes a list of nodes from the CCLE project in the sandbox commons.
 
-            >>> Gen3Submission.delete_nodes("DCF", "CCLE", ["demographic", "subject", "experiment"])
-        """
-        project_id = f"{program}-{project}"
+            >>> Gen3Submission.delete_nodes("DCF", "CCLE", ["demographic", "subject", "experiment"])
+        """
+        project_id = f"{program}-{project}"
         for node in ordered_node_list:
             if verbose:
-                print(node, end="", flush=True)
-            first_uuid = ""
+                print(node, end="", flush=True)
+            first_uuid = ""
             while True:
-                query_string = f"""{{
-                    {node} (first: {batch_size}, project_id: "{project_id}") {{
+                query_string = f"""{{
+                    {node} (first: {batch_size}, project_id: "{project_id}") {{
                         id
                     }}
-                }}"""
+                }}"""
                 res = self.query(query_string)
-                uuids = [x["id"] for x in res["data"][node]]
+                uuids = [x["id"] for x in res["data"][node]]
                 if len(uuids) == 0:
                     break  # all done
                 if first_uuid == uuids[0]:
-                    raise Exception("Failed to delete. Exiting")
+                    raise Exception("Failed to delete. Exiting")
                 first_uuid = uuids[0]
                 if verbose:
-                    print(".", end="", flush=True)
+                    print(".", end="", flush=True)
                 self.delete_records(program, project, uuids, batch_size)
             if verbose:
                 print()
@@ -401,35 +401,35 @@

Source code for gen3.submission

 
[docs] def export_record(self, program, project, uuid, fileformat, filename=None): - """Export a single record into json. + """Export a single record into json. Args: program (str): The program the record is under. project (str): The project the record is under. uuid (str): The UUID of the record to export. - fileformat (str): Export data as either 'json' or 'tsv' + fileformat (str): Export data as either 'json' or 'tsv' filename (str): Name of the file to export to; if no filename is provided, prints data to screen Examples: This exports a single record from the sandbox commons. - >>> Gen3Submission.export_record("DCF", "CCLE", "d70b41b9-6f90-4714-8420-e043ab8b77b9", "json", filename="DCF-CCLE_one_record.json") + >>> Gen3Submission.export_record("DCF", "CCLE", "d70b41b9-6f90-4714-8420-e043ab8b77b9", "json", filename="DCF-CCLE_one_record.json") - """ + """ assert fileformat in [ - "json", - "tsv", - ], "File format must be either 'json' or 'tsv'" - api_url = "{}/api/v0/submission/{}/{}/export?ids={}&format={}".format( + "json", + "tsv", + ], "File format must be either 'json' or 'tsv'" + api_url = "{}/api/v0/submission/{}/{}/export?ids={}&format={}".format( self._endpoint, program, project, uuid, fileformat ) output = requests.get(api_url, auth=self._auth_provider).text if filename is None: - if fileformat == "json": + if fileformat == "json": try: output = json.loads(output) except ValueError as e: - print(f"Output: {output}\nUnable to parse JSON: {e}") + print(f"Output: {output}\nUnable to parse JSON: {e}") raise return output else: @@ -440,35 +440,35 @@

Source code for gen3.submission

 
[docs] def export_node(self, program, project, node_type, fileformat, filename=None): - """Export all records in a single node type of a project. + """Export all records in a single node type of a project. Args: program (str): The program to which records belong. project (str): The project to which records belong. node_type (str): The name of the node to export. - fileformat (str): Export data as either 'json' or 'tsv' + fileformat (str): Export data as either 'json' or 'tsv' filename (str): Name of the file to export to; if no filename is provided, prints data to screen Examples: - This exports all records in the "sample" node from the CCLE project in the sandbox commons. + This exports all records in the "sample" node from the CCLE project in the sandbox commons. - >>> Gen3Submission.export_node("DCF", "CCLE", "sample", "tsv", filename="DCF-CCLE_sample_node.tsv") + >>> Gen3Submission.export_node("DCF", "CCLE", "sample", "tsv", filename="DCF-CCLE_sample_node.tsv") - """ + """ assert fileformat in [ - "json", - "tsv", - ], "File format must be either 'json' or 'tsv'" - api_url = "{}/api/v0/submission/{}/{}/export/?node_label={}&format={}".format( + "json", + "tsv", + ], "File format must be either 'json' or 'tsv'" + api_url = "{}/api/v0/submission/{}/{}/export/?node_label={}&format={}".format( self._endpoint, program, project, node_type, fileformat ) output = requests.get(api_url, auth=self._auth_provider).text if filename is None: - if fileformat == "json": + if fileformat == "json": try: output = json.loads(output) except ValueError as e: - print(f"Output: {output}\nUnable to parse JSON: {e}") + print(f"Output: {output}\nUnable to parse JSON: {e}") raise return output else: @@ -481,7 +481,7 @@

Source code for gen3.submission

 
[docs] def query(self, query_txt, variables=None, max_tries=1): - """Execute a GraphQL query against a Data Commons. + """Execute a GraphQL query against a Data Commons. Args: query_txt (str): Query text. @@ -492,25 +492,25 @@

Source code for gen3.submission

             This executes a query to get the list of all the project codes for all the projects
             in the Data Commons.
 
-            >>> query = "{ project(first:0) { code } }"
+            >>> query = "{ project(first:0) { code } }"
             ... Gen3Submission.query(query)
 
-        """
-        api_url = "{}/api/v0/submission/graphql".format(self._endpoint)
+        """
+        api_url = "{}/api/v0/submission/graphql".format(self._endpoint)
         if variables == None:
-            query = {"query": query_txt}
+            query = {"query": query_txt}
         else:
-            query = {"query": query_txt, "variables": variables}
+            query = {"query": query_txt, "variables": variables}
 
         tries = 0
         while tries < max_tries:
             output = requests.post(api_url, auth=self._auth_provider, json=query).text
             data = json.loads(output)
 
-            if "errors" in data:
-                raise Gen3SubmissionQueryError(data["errors"])
+            if "errors" in data:
+                raise Gen3SubmissionQueryError(data["errors"])
 
-            if not "data" in data:
+            if not "data" in data:
                 print(query_txt)
                 print(data)
 
@@ -522,7 +522,7 @@ 

Source code for gen3.submission

 
[docs] def get_graphql_schema(self): - """Returns the GraphQL schema for a commons. + """Returns the GraphQL schema for a commons. This runs the GraphQL introspection query against a commons and returns the results. @@ -531,8 +531,8 @@

Source code for gen3.submission

 
             >>> Gen3Submission.get_graphql_schema()
 
-        """
-        api_url = "{}/api/v0/submission/getschema".format(self._endpoint)
+        """
+        api_url = "{}/api/v0/submission/getschema".format(self._endpoint)
         output = requests.get(api_url).text
         data = json.loads(output)
         return data
@@ -543,7 +543,7 @@

Source code for gen3.submission

 
[docs] def get_dictionary_node(self, node_type): - """Returns the dictionary schema for a specific node. + """Returns the dictionary schema for a specific node. This gets the current json dictionary schema for a specific node type in a commons. @@ -551,12 +551,12 @@

Source code for gen3.submission

             node_type (str): The node_type (or name of the node) to retrieve.
 
         Examples:
-            This returns the dictionary schema the "subject" node.
+            This returns the dictionary schema the "subject" node.
 
-            >>> Gen3Submission.get_dictionary_node("subject")
+            >>> Gen3Submission.get_dictionary_node("subject")
 
-        """
-        api_url = "{}/api/v0/submission/_dictionary/{}".format(
+        """
+        api_url = "{}/api/v0/submission/_dictionary/{}".format(
             self._endpoint, node_type
         )
         output = requests.get(api_url).text
@@ -567,7 +567,7 @@ 

Source code for gen3.submission

 
[docs] def get_dictionary_all(self): - """Returns the entire dictionary object for a commons. + """Returns the entire dictionary object for a commons. This gets a json of the current dictionary schema for a commons. @@ -576,8 +576,8 @@

Source code for gen3.submission

 
             >>> Gen3Submission.get_dictionary_all()
 
-        """
-        return self.get_dictionary_node("_all")
+ """ + return self.get_dictionary_node("_all")
### File functions @@ -585,7 +585,7 @@

Source code for gen3.submission

 
[docs] def get_project_manifest(self, program, project): - """Get a projects file manifest + """Get a projects file manifest Args: program: the name of the program the project is from @@ -593,10 +593,10 @@

Source code for gen3.submission

 
         Example:
 
-            >>> Gen3Submission.get_project_manifest("DCF", "CCLE")
+            >>> Gen3Submission.get_project_manifest("DCF", "CCLE")
 
-        """
-        api_url = f"{self._endpoint}/api/v0/submission/{program}/{project}/manifest"
+        """
+        api_url = f"{self._endpoint}/api/v0/submission/{program}/{project}/manifest"
         output = requests.get(api_url, auth=self._auth_provider)
         return output
@@ -604,51 +604,51 @@

Source code for gen3.submission

 
[docs] def submit_file(self, project_id, filename, chunk_size=30, row_offset=0): - """Submit data in a spreadsheet file containing multiple records in rows to a Gen3 Data Commons. + """Submit data in a spreadsheet file containing multiple records in rows to a Gen3 Data Commons. Args: project_id (str): The project_id to submit to. filename (str): The file containing data to submit. The format can be TSV, CSV or XLSX (first worksheet only for now). chunk_size (integer): The number of rows of data to submit for each request to the API. - row_offset (integer): The number of rows of data to skip; '0' starts submission from the first row and submits all data. + row_offset (integer): The number of rows of data to skip; '0' starts submission from the first row and submits all data. Examples: This submits a spreadsheet file containing multiple records in rows to the CCLE project in the sandbox commons. - >>> Gen3Submission.submit_file("DCF-CCLE","data_spreadsheet.tsv") + >>> Gen3Submission.submit_file("DCF-CCLE","data_spreadsheet.tsv") - """ + """ # Read the file in as a pandas DataFrame f = os.path.basename(filename) - if f.lower().endswith(".csv"): - df = pd.read_csv(filename, header=0, sep=",", dtype=str).fillna("") - elif f.lower().endswith(".xlsx"): + if f.lower().endswith(".csv"): + df = pd.read_csv(filename, header=0, sep=",", dtype=str).fillna("") + elif f.lower().endswith(".xlsx"): xl = pd.ExcelFile(filename) # load excel file sheet = xl.sheet_names[0] # sheetname df = xl.parse(sheet) # save sheet as dataframe converters = { col: str for col in list(df) - } # make sure int isn't converted to float - df = pd.read_excel(filename, converters=converters).fillna("") # remove nan - elif filename.lower().endswith((".tsv", ".txt")): - df = pd.read_csv(filename, header=0, sep="\t", dtype=str).fillna("") + } # make sure int isn't converted to float + df = pd.read_excel(filename, converters=converters).fillna("") # remove nan + elif filename.lower().endswith((".tsv", ".txt")): + df = pd.read_csv(filename, header=0, sep="\t", dtype=str).fillna("") else: - raise Gen3UserError("Please upload a file in CSV, TSV, or XLSX format.") + raise Gen3UserError("Please upload a file in CSV, TSV, or XLSX format.") df.rename( - columns={c: c.lstrip("*") for c in df.columns}, inplace=True + columns={c: c.lstrip("*") for c in df.columns}, inplace=True ) # remove any leading asterisks in the DataFrame column names # Check uniqueness of submitter_ids: if len(list(df.submitter_id)) != len(list(df.submitter_id.unique())): raise Gen3Error( - "Warning: file contains duplicate submitter_ids. \nNote: submitter_ids must be unique within a node!" + "Warning: file contains duplicate submitter_ids. \nNote: submitter_ids must be unique within a node!" ) # Chunk the file - print("\nSubmitting {} with {} records.".format(filename, len(df))) - program, project = project_id.split("-", 1) - api_url = "{}/api/v0/submission/{}/{}".format(self._endpoint, program, project) - headers = {"content-type": "text/tab-separated-values"} + print("\nSubmitting {} with {} records.".format(filename, len(df))) + program, project = project_id.split("-", 1) + api_url = "{}/api/v0/submission/{}/{}".format(self._endpoint, program, project) + headers = {"content-type": "text/tab-separated-values"} start = row_offset end = row_offset + chunk_size @@ -657,11 +657,11 @@

Source code for gen3.submission

         count = 0
 
         results = {
-            "invalid": {},  # these are invalid records
-            "other": [],  # any unhandled API responses
-            "details": [],  # entire API response details
-            "succeeded": [],  # list of submitter_ids that were successfully updated/created
-            "responses": [],  # list of API response codes
+            "invalid": {},  # these are invalid records
+            "other": [],  # any unhandled API responses
+            "details": [],  # entire API response details
+            "succeeded": [],  # list of submitter_ids that were successfully updated/created
+            "responses": [],  # list of API response codes
         }
 
         # Start the chunking loop:
@@ -671,10 +671,10 @@ 

Source code for gen3.submission

             invalid = []
             count += 1
             print(
-                "Chunk {} (chunk size: {}, submitted: {} of {})".format(
+                "Chunk {} (chunk size: {}, submitted: {} of {})".format(
                     count,
                     chunk_size,
-                    len(results["succeeded"]) + len(results["invalid"]),
+                    len(results["succeeded"]) + len(results["invalid"]),
                     len(df),
                 )
             )
@@ -683,22 +683,22 @@ 

Source code for gen3.submission

                 response = requests.put(
                     api_url,
                     auth=self._auth_provider,
-                    data=chunk.to_csv(sep="\t", index=False),
+                    data=chunk.to_csv(sep="\t", index=False),
                     headers=headers,
                 ).text
             except requests.exceptions.ConnectionError as e:
-                results["details"].append(e.message)
+                results["details"].append(e.message)
                 continue
 
             # Handle the API response
             if (
-                "Request Timeout" in response
-                or "413 Request Entity Too Large" in response
-                or "Connection aborted." in response
-                or "service failure - try again later" in response
+                "Request Timeout" in response
+                or "413 Request Entity Too Large" in response
+                or "Connection aborted." in response
+                or "service failure - try again later" in response
             ):  # time-out, response is not valid JSON at the moment
-                print("\t Reducing Chunk Size: {}".format(response))
-                results["responses"].append("Reducing Chunk Size: {}".format(response))
+                print("\t Reducing Chunk Size: {}".format(response))
+                results["responses"].append("Reducing Chunk Size: {}".format(response))
                 timeout = True
 
             else:
@@ -707,72 +707,72 @@ 

Source code for gen3.submission

                 except ValueError as e:
                     print(response)
                     print(str(e))
-                    raise Gen3Error("Unable to parse API response as JSON!")
+                    raise Gen3Error("Unable to parse API response as JSON!")
 
-                if "message" in json_res and "code" not in json_res:
+                if "message" in json_res and "code" not in json_res:
                     print(
-                        "\t No code in the API response for Chunk {}: {}".format(
-                            count, json_res.get("message")
+                        "\t No code in the API response for Chunk {}: {}".format(
+                            count, json_res.get("message")
                         )
                     )
-                    print("\t {}".format(json_res.get("transactional_errors")))
-                    results["responses"].append(
-                        "Error Chunk {}: {}".format(count, json_res.get("message"))
+                    print("\t {}".format(json_res.get("transactional_errors")))
+                    results["responses"].append(
+                        "Error Chunk {}: {}".format(count, json_res.get("message"))
                     )
-                    results["other"].append(json_res.get("transactional_errors"))
+                    results["other"].append(json_res.get("transactional_errors"))
 
-                elif "code" not in json_res:
-                    print("\t Unhandled API-response: {}".format(response))
-                    results["responses"].append(
-                        "Unhandled API response: {}".format(response)
+                elif "code" not in json_res:
+                    print("\t Unhandled API-response: {}".format(response))
+                    results["responses"].append(
+                        "Unhandled API response: {}".format(response)
                     )
 
-                elif json_res["code"] == 200:  # success
-                    entities = json_res.get("entities", [])
-                    print("\t Succeeded: {} entities.".format(len(entities)))
-                    results["responses"].append(
-                        "Chunk {} Succeeded: {} entities.".format(count, len(entities))
+                elif json_res["code"] == 200:  # success
+                    entities = json_res.get("entities", [])
+                    print("\t Succeeded: {} entities.".format(len(entities)))
+                    results["responses"].append(
+                        "Chunk {} Succeeded: {} entities.".format(count, len(entities))
                     )
 
                     for entity in entities:
-                        sid = entity["unique_keys"][0]["submitter_id"]
-                        results["succeeded"].append(sid)
+                        sid = entity["unique_keys"][0]["submitter_id"]
+                        results["succeeded"].append(sid)
 
-                elif json_res["code"] == 500:  # internal server error
-                    print("\t Internal Server Error: {}".format(response))
-                    results["responses"].append(
-                        "Internal Server Error: {}".format(response)
+                elif json_res["code"] == 500:  # internal server error
+                    print("\t Internal Server Error: {}".format(response))
+                    results["responses"].append(
+                        "Internal Server Error: {}".format(response)
                     )
 
                 else:  # failure (400, 401, 403, 404...)
-                    entities = json_res.get("entities", [])
+                    entities = json_res.get("entities", [])
                     print(
-                        "\tChunk Failed (status code {}): {} entities.".format(
-                            json_res.get("code"), len(entities)
+                        "\tChunk Failed (status code {}): {} entities.".format(
+                            json_res.get("code"), len(entities)
                         )
                     )
-                    results["responses"].append(
-                        "Chunk {} Failed: {} entities.".format(count, len(entities))
+                    results["responses"].append(
+                        "Chunk {} Failed: {} entities.".format(count, len(entities))
                     )
 
                     for entity in entities:
-                        sid = entity["unique_keys"][0]["submitter_id"]
-                        if entity["valid"]:  # valid but failed
+                        sid = entity["unique_keys"][0]["submitter_id"]
+                        if entity["valid"]:  # valid but failed
                             valid_but_failed.append(sid)
                         else:  # invalid and failed
-                            message = str(entity["errors"])
-                            results["invalid"][sid] = message
+                            message = str(entity["errors"])
+                            results["invalid"][sid] = message
                             invalid.append(sid)
-                    print("\tInvalid records in this chunk: {}".format(len(invalid)))
+                    print("\tInvalid records in this chunk: {}".format(len(invalid)))
 
             if (
                 len(valid_but_failed) > 0 and len(invalid) > 0
             ):  # if valid entities failed bc grouped with invalid, retry submission
                 chunk = chunk.loc[
-                    df["submitter_id"].isin(valid_but_failed)
-                ]  # these are records that weren't successful because they were part of a chunk that failed, but are valid and can be resubmitted without changes
+                    df["submitter_id"].isin(valid_but_failed)
+                ]  # these are records that weren't successful because they were part of a chunk that failed, but are valid and can be resubmitted without changes
                 print(
-                    "Retrying submission of valid entities from failed chunk: {} valid entities.".format(
+                    "Retrying submission of valid entities from failed chunk: {} valid entities.".format(
                         len(chunk)
                     )
                 )
@@ -781,10 +781,10 @@ 

Source code for gen3.submission

                 len(valid_but_failed) > 0 and len(invalid) == 0
             ):  # if all entities are valid but submission still failed, probably due to duplicate submitter_ids. Can remove this section once the API response is fixed: https://ctds-planx.atlassian.net/browse/PXP-3065
                 raise Gen3Error(
-                    "Please check your data for correct file encoding, special characters, or duplicate submitter_ids or ids."
+                    "Please check your data for correct file encoding, special characters, or duplicate submitter_ids or ids."
                 )
 
-            elif timeout is False:  # get new chunk if didn't timeout
+            elif timeout is False:  # get new chunk if didn't timeout
                 start += chunk_size
                 end = start + chunk_size
                 chunk = df[start:end]
@@ -795,18 +795,18 @@ 

Source code for gen3.submission

                     end = start + chunk_size
                     chunk = df[start:end]
                     print(
-                        "Retrying Chunk with reduced chunk_size: {}".format(chunk_size)
+                        "Retrying Chunk with reduced chunk_size: {}".format(chunk_size)
                     )
                     timeout = False
                 else:
-                    print("Last chunk:\n{}".format(chunk))
+                    print("Last chunk:\n{}".format(chunk))
                     raise Gen3Error(
-                        "Submission is timing out. Please contact the Helpdesk."
+                        "Submission is timing out. Please contact the Helpdesk."
                     )
 
-        print("Finished data submission.")
-        print("Successful records: {}".format(len(set(results["succeeded"]))))
-        print("Failed invalid records: {}".format(len(results["invalid"])))
+        print("Finished data submission.")
+        print("Successful records: {}".format(len(set(results["succeeded"]))))
+        print("Failed invalid records: {}".format(len(results["invalid"])))
 
         return results
diff --git a/docs/_build/html/_modules/gen3/tools/download/drs_download.html b/docs/_build/html/_modules/gen3/tools/download/drs_download.html index aa0e4de3..29009144 100644 --- a/docs/_build/html/_modules/gen3/tools/download/drs_download.html +++ b/docs/_build/html/_modules/gen3/tools/download/drs_download.html @@ -31,7 +31,7 @@

Source code for gen3.tools.download.drs_download

-"""
+"""
 Module for downloading and listing JSON DRS manifest and DRS objects. The main classes in
 this module for downloading DRS objects are DownloadManager and Manifest.
 
@@ -39,16 +39,16 @@ 

Source code for gen3.tools.download.drs_download

This generates the Gen3Jobs class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page. - >>> datafiles = Manifest.load('sample/manifest_1.json') - downloadManager = DownloadManager("source.my_commons.org", - Gen3Auth(refresh_file="~.gen3/my_credentials.json"), datafiles) + >>> datafiles = Manifest.load('sample/manifest_1.json') + downloadManager = DownloadManager("source.my_commons.org", + Gen3Auth(refresh_file="~.gen3/my_credentials.json"), datafiles) for i in datafiles: print(i) - downloadManager.download(datafiles, ".") + downloadManager.download(datafiles, ".") See docs/howto/drsDownloading.md for more details -""" +""" import re @@ -78,7 +78,7 @@

Source code for gen3.tools.download.drs_download

DEFAULT_EXPIRE: timedelta = timedelta(hours=1) # package formats we handle for unpacking -PACKAGE_EXTENSIONS = [".zip"] +PACKAGE_EXTENSIONS = [".zip"] logger = get_logger(__name__) @@ -89,7 +89,7 @@

Source code for gen3.tools.download.drs_download

@dataclass_json(letter_case=LetterCase.SNAKE, undefined=Undefined.EXCLUDE) @dataclass class Manifest: - """Data class representing a Gen3 JSON manifest typically exported from a Gen3 discovery page. + """Data class representing a Gen3 JSON manifest typically exported from a Gen3 discovery page. The class is passed to the DownloadManager to download or list all of the files in the manifest. The Download manager will cache additional information (if available) @@ -100,7 +100,7 @@

Source code for gen3.tools.download.drs_download

file_name(Optional[str]): the name of the file pointed to by the DRS object id md5sum(Optional[str]): the checksum of the object commons_url(Optional[str]): url of the indexd server to retrieve file/bundle from - """ + """ object_id: str # only required member file_size: Optional[int] = -1 # -1 indicated not set @@ -112,8 +112,8 @@

Source code for gen3.tools.download.drs_download

[docs] @staticmethod def load_manifest(path: Path): - """Loads a json manifest""" - with open(path, "rt") as fin: + """Loads a json manifest""" + with open(path, "rt") as fin: data = json_load(fin) return Manifest.schema().load(data, many=True)
@@ -121,14 +121,14 @@

Source code for gen3.tools.download.drs_download

[docs] @staticmethod - def create_object_list(manifest) -> List["Downloadable"]: - """Create a list of Downloadable instances from the manifest + def create_object_list(manifest) -> List["Downloadable"]: + """Create a list of Downloadable instances from the manifest Args: manifest(list): list of manifest objects Returns: List of Downloadable instances - """ + """ results = [] for entry in manifest: results.append( @@ -145,8 +145,8 @@

Source code for gen3.tools.download.drs_download

[docs] @staticmethod - def load(filename: Path) -> Optional[List["Downloadable"]]: - """ + def load(filename: Path) -> Optional[List["Downloadable"]]: + """ Method to load a json manifest and return a list of Bownloadable object. This list is passed to the DownloadManager methods of download, and list @@ -154,14 +154,14 @@

Source code for gen3.tools.download.drs_download

filename(Path): path to manifest file Returns: list of Downloadable objects if successfully opened/parsed None otherwise - """ + """ try: manifest = Manifest.load_manifest(filename) return Manifest.create_object_list(manifest) except FileNotFoundError as ex: - logger.critical(f"Error file not found: {ex.filename}") + logger.critical(f"Error file not found: {ex.filename}") except JSONDecodeError as ex: - logger.critical(f"format of manifest file is valid JSON: {ex.msg}") + logger.critical(f"format of manifest file is valid JSON: {ex.msg}") return None
@@ -170,7 +170,7 @@

Source code for gen3.tools.download.drs_download

@dataclass class KnownDRSEndpoint: - """ + """ Dataclass used internally by the DownloadManager class to cache hostnames and tokens for Gen3 commons possibly accessed using the Workspace Token Service (WTS). The endpoint is assumed to support DRS and therefore caches additional DRS information. @@ -183,7 +183,7 @@

Source code for gen3.tools.download.drs_download

identifier (Optional[str]): DRS prefix (if used) use_wts (bool): if True use WTS to create tokens - """ + """ hostname: str expire: datetime = None @@ -194,27 +194,27 @@

Source code for gen3.tools.download.drs_download

@property def available(self): - """If endpoint has access token it is available""" + """If endpoint has access token it is available""" return self.access_token is not None def expired(self) -> bool: - """check if WTS token is older than the default expiration date + """check if WTS token is older than the default expiration date If not using the WTS return false and use standard Gen3 Auth - """ + """ if not self.use_wts: return False return datetime.now() > self.expire def renew_token(self, wts_server_name: str, server_access_token): - """Gets a new token from the WTS and updates the token and refresh time + """Gets a new token from the WTS and updates the token and refresh time Args: wts_server_name (str): hostname of WTS server server_access_token (str): token used to authenticate use of WTS - """ + """ token = wts_get_token( hostname=wts_server_name, idp=self.idp, @@ -224,22 +224,22 @@

Source code for gen3.tools.download.drs_download

# TODO: this would break if user is trying to download object from different commons # keep BRH token and wts sparate self.access_token = token - self.expire = datetime.fromtimestamp(token_info["exp"]) + self.expire = datetime.fromtimestamp(token_info["exp"]) class DRSObjectType(str, Enum): - """Enum defining the 3 possible DRS object types.""" + """Enum defining the 3 possible DRS object types.""" - unknown = "unknown" - object = "object" - bundle = "bundle" + unknown = "unknown" + object = "object" + bundle = "bundle"
[docs] @dataclass class Downloadable: - """ + """ Class handling the information for a DRS object. The information is populated from the manifest or by retrieving the information from a DRS server. @@ -254,7 +254,7 @@

Source code for gen3.tools.download.drs_download

access_methods (List[Dict[str, Any]]): list of access methods (e.g. s3) for DRS object children (List[Downloadable]): list of child objects (in the case of DRS bundles) _manager (DownloadManager): manager for this Downloadable - """ + """ object_id: str object_type: Optional[DRSObjectType] = DRSObjectType.unknown @@ -264,52 +264,52 @@

Source code for gen3.tools.download.drs_download

updated_time: Optional[datetime] = None created_time: Optional[datetime] = None access_methods: List[Dict[str, Any]] = field(default_factory=list) - children: List["Downloadable"] = field(default_factory=list) + children: List["Downloadable"] = field(default_factory=list) _manager = None def __str__(self): return ( - f'{self.file_name if self.file_name is not None else "not available" : >45}; ' - f"{humanfriendly.format_size(self.file_size) :>12}; " - f'{self.hostname if self.hostname is not None else "not resolved"}; ' - f'{self.created_time.strftime("%m/%d/%Y, %H:%M:%S") if self.created_time is not None else "not available"}' + f'{self.file_name if self.file_name is not None else "not available" : >45}; ' + f"{humanfriendly.format_size(self.file_size) :>12}; " + f'{self.hostname if self.hostname is not None else "not resolved"}; ' + f'{self.created_time.strftime("%m/%d/%Y, %H:%M:%S") if self.created_time is not None else "not available"}' ) def __repr__(self): return ( - f'(Downloadable: {self.file_name if self.file_name is not None else "not available"}; ' - f"{humanfriendly.format_size(self.file_size)}; " - f'{self.hostname if self.hostname is not None else "not resolved"}; ' - f'{self.created_time.strftime("%m/%d/%Y, %H:%M:%S") if self.created_time is not None else "not available"})' + f'(Downloadable: {self.file_name if self.file_name is not None else "not available"}; ' + f"{humanfriendly.format_size(self.file_size)}; " + f'{self.hostname if self.hostname is not None else "not resolved"}; ' + f'{self.created_time.strftime("%m/%d/%Y, %H:%M:%S") if self.created_time is not None else "not available"})' )
[docs] def download(self): - """calls the manager to download this object. Allows Downloadables to be self downloading""" + """calls the manager to download this object. Allows Downloadables to be self downloading""" self._manager.download([self])
[docs] - def pprint(self, indent: str = ""): - """ + def pprint(self, indent: str = ""): + """ Pretty prints the object information. This is used for listing an object. In the case of a DRS bundle the child objects are listed similar to the linux tree command - """ + """ from os import linesep res = self.__str__() + linesep - child_indent = f"{indent} " + child_indent = f"{indent} " pos = -1 for x in self.children: pos += 1 if pos == len(self.children) - 1: - res += f"{child_indent}└── {x.pprint(child_indent)}" + res += f"{child_indent}└── {x.pprint(child_indent)}" else: - res += f"{child_indent}├── {x.pprint(child_indent)}" + res += f"{child_indent}├── {x.pprint(child_indent)}" return res
@@ -319,31 +319,31 @@

Source code for gen3.tools.download.drs_download

[docs] @dataclass class DownloadStatus: - """Stores the download status of objectIDs. + """Stores the download status of objectIDs. The DataManager will return a list of DownloadStatus as a result of calling the download method - Status is "pending" until it is downloaded or an error occurs. + Status is "pending" until it is downloaded or an error occurs. Attributes: filename (str): the name of the file to download - status (str): status of file download initially "pending" + status (str): status of file download initially "pending" start_time (Optional[datetime]): start time of download as datetime initially None end_time (Optional[datetime]): end time of download as datetime initially None - """ + """ filename: str - status: str = "pending" + status: str = "pending" start_time: Optional[datetime] = None end_time: Optional[datetime] = None status_code: Optional[int] = None def __str__(self): return ( - f'filename: {self.filename if self.filename is not None else "not available"}; ' - f"status: {self.status}; " - f"status_code: {self.status_code}; " - f'start_time: {self.start_time.strftime("%m/%d/%Y, %H:%M:%S") if self.start_time is not None else "n/a"}; ' - f'end_time: {self.end_time.strftime("%m/%d/%Y, %H:%M:%S") if self.start_time is not None else "n/a"}' + f'filename: {self.filename if self.filename is not None else "not available"}; ' + f"status: {self.status}; " + f"status_code: {self.status_code}; " + f'start_time: {self.start_time.strftime("%m/%d/%Y, %H:%M:%S") if self.start_time is not None else "n/a"}; ' + f'end_time: {self.end_time.strftime("%m/%d/%Y, %H:%M:%S") if self.start_time is not None else "n/a"}' ) def __repr__(self): @@ -352,7 +352,7 @@

Source code for gen3.tools.download.drs_download

def wts_external_oidc(hostname: str) -> Dict[str, Any]: - """ + """ Get the external_oidc from a connected WTS. Will report if WTS service is missing. Note that in some cases this can be considered a warning not a error. @@ -361,80 +361,80 @@

Source code for gen3.tools.download.drs_download

Returns: dict containing the oidc information - """ + """ oidc = {} if not hostname: return oidc - url = f"https://{hostname}/wts/external_oidc/" - err_msg = "Likely no WTS service running on this Commons. Proceeding, but certain commands might fail." + url = f"https://{hostname}/wts/external_oidc/" + err_msg = "Likely no WTS service running on this Commons. Proceeding, but certain commands might fail." try: response = requests.get(url) response.raise_for_status() except requests.exceptions.HTTPError as exc: resp_msg = json_loads(exc.response.text) - if "message" in resp_msg: - resp_msg = resp_msg["message"] + if "message" in resp_msg: + resp_msg = resp_msg["message"] logger.warning( - f"HTTP Error ({exc.response.status_code}) from '{url}': {resp_msg}. {err_msg}" + f"HTTP Error ({exc.response.status_code}) from '{url}': {resp_msg}. {err_msg}" ) return oidc try: data = response.json() - if "providers" not in data: - logger.warning(f'No "providers" field in WTS response: {data}. {err_msg}') + if "providers" not in data: + logger.warning(f'No "providers" field in WTS response: {data}. {err_msg}') return oidc - for item in data["providers"]: - oidc[urlparse(item["base_url"]).netloc] = item + for item in data["providers"]: + oidc[urlparse(item["base_url"]).netloc] = item except JSONDecodeError as ex: - logger.warning(f"Unable to process WTS response: {response.text}. {err_msg}") + logger.warning(f"Unable to process WTS response: {response.text}. {err_msg}") return oidc def wts_get_token(hostname: str, idp: str, access_token: str): - """ + """ Gets a auth token from a Gen3 WTS server for the supplied idp Args: - hostname (str): Gen3 common's WTS service + hostname (str): Gen3 common's WTS service idp: identity provider to use access_token: Gen3 Auth to use to with WTS Returns: Token for idp if successful, None if failure - """ + """ headers = { - "Content-Type": "application/json", - "Accept": "application/json", - "Connection": "keep-alive", - "Authorization": "bearer " + access_token, + "Content-Type": "application/json", + "Accept": "application/json", + "Connection": "keep-alive", + "Authorization": "bearer " + access_token, } try: - url = f"https://{hostname}/wts/token/?idp={idp}" + url = f"https://{hostname}/wts/token/?idp={idp}" try: response = requests.get(url=url, headers=headers) response.raise_for_status() except requests.exceptions.HTTPError as exc: logger.critical( - f"HTTP Error ({exc.response.status_code}): getting WTS token: {exc.response.text}" + f"HTTP Error ({exc.response.status_code}): getting WTS token: {exc.response.text}" ) logger.critical( - "Please make sure the target commons is connected on your profile page and that connection has not expired." + "Please make sure the target commons is connected on your profile page and that connection has not expired." ) return None - return _handle_access_token_response(response, "token") + return _handle_access_token_response(response, "token") except Gen3AuthError: - logger.critical(f"Unable to authenticate your credentials with {hostname}") + logger.critical(f"Unable to authenticate your credentials with {hostname}") return None def get_drs_object_info(hostname: str, object_id: str) -> Optional[dict]: - """ + """ Retrieves information for a DRS object residing on the hostname Args: hostname (str): hostname of DRS object @@ -442,9 +442,9 @@

Source code for gen3.tools.download.drs_download

Returns: GAG4H DRS object information if sucessful otherwise None - """ + """ try: - response = requests.get(f"https://{hostname}/ga4gh/drs/v1/objects/{object_id}") + response = requests.get(f"https://{hostname}/ga4gh/drs/v1/objects/{object_id}") response.raise_for_status() data = response.json() return data @@ -452,33 +452,33 @@

Source code for gen3.tools.download.drs_download

except requests.HTTPError as exc: if exc.response.status_code == 404: logger.critical( - f"HTTP Error ({exc.response.status_code}): {object_id} not found at {hostname}" + f"HTTP Error ({exc.response.status_code}): {object_id} not found at {hostname}" ) else: logger.critical( - f"HTTP Error ({exc.response.status_code}): accessing object: {object_id}" + f"HTTP Error ({exc.response.status_code}): accessing object: {object_id}" ) return None except ConnectionError as exc: - logger.critical(f"Connection Error {exc} when accessing object: {object_id}") + logger.critical(f"Connection Error {exc} when accessing object: {object_id}") return None def extract_filename_from_object_info(object_info: dict) -> Optional[str]: - """Extracts the filename from the object_info. + """Extracts the filename from the object_info. if filename is in object_info use that, otherwise try to extract it from the one of the access methods. Returns filename if found, else return None Args object_info (dict): DRS object dictionary - """ - if "name" in object_info and object_info["name"]: - return object_info["name"] + """ + if "name" in object_info and object_info["name"]: + return object_info["name"] - for access_method in object_info["access_methods"]: - url = access_method["access_url"]["url"] - parts = url.split("/") + for access_method in object_info["access_methods"]: + url = access_method["access_url"]["url"] + parts = url.split("/") if parts: return parts[-1] @@ -486,7 +486,7 @@

Source code for gen3.tools.download.drs_download

def get_access_methods(object_info: dict) -> List[str]: - """ + """ Returns the DRS GA4GH access methods from the object_info. Args: @@ -494,46 +494,46 @@

Source code for gen3.tools.download.drs_download

Returns: List of access methods - """ + """ if object_info is None: - logger.critical("no access methods defined for this file") + logger.critical("no access methods defined for this file") return [] - return object_info["access_methods"] + return object_info["access_methods"] def get_drs_object_type(object_info: dict) -> DRSObjectType: - """From the object info determine the type of object. + """From the object info determine the type of object. Args: object_info (dict): DRS object dictionary Returns: type of object: either bundle or object - """ - if "form" in object_info: - if object_info["form"] is None: + """ + if "form" in object_info: + if object_info["form"] is None: return DRSObjectType.object - return DRSObjectType(object_info["form"]) + return DRSObjectType(object_info["form"]) - if "contents" in object_info and len(object_info["contents"]) > 0: + if "contents" in object_info and len(object_info["contents"]) > 0: return DRSObjectType.bundle else: return DRSObjectType.object def get_drs_object_timestamp(s: Optional[str]) -> Optional[datetime]: - """returns the timestamp in datetime if not none otherwise returns None + """returns the timestamp in datetime if not none otherwise returns None Args: s (Optional[str]): string to parse Returns: datetime if not None - """ + """ return date_parser.parse(s) if s is not None else None def add_drs_object_info(info: Downloadable) -> bool: - """ + """ Given a downloader object fill in the required fields from the resolved hostname. In the case of a bundle, try to resolve all object_ids contained in the bundle including other objects and bundles. @@ -542,7 +542,7 @@

Source code for gen3.tools.download.drs_download

info (Downloadable): Downloadable to add information to Returns: True if object is valid and resolvable. - """ + """ if info.hostname is None: return False @@ -552,17 +552,17 @@

Source code for gen3.tools.download.drs_download

# Get common information we want info.file_name = extract_filename_from_object_info(object_info) - info.file_size = object_info.get("size", -1) - info.updated_time = get_drs_object_timestamp(object_info.get("updated_time", None)) - info.created_time = get_drs_object_timestamp(object_info.get("created_time", None)) + info.file_size = object_info.get("size", -1) + info.updated_time = get_drs_object_timestamp(object_info.get("updated_time", None)) + info.created_time = get_drs_object_timestamp(object_info.get("created_time", None)) info.object_type = get_drs_object_type(object_info) if info.object_type == DRSObjectType.object: info.access_methods = get_access_methods(object_info) return True else: # a bundle,get everything else - for item in object_info["contents"]: - child_id = item.get("id", None) + for item in object_info["contents"]: + child_id = item.get("id", None) if child_id is None: continue child_object = Downloadable(hostname=info.hostname, object_id=child_id) @@ -573,9 +573,9 @@

Source code for gen3.tools.download.drs_download

class InvisibleProgress: - """ + """ Invisible progress bar which stubs a tqdm progress bar - """ + """ def update(self, value): # pragma: no cover pass @@ -584,7 +584,7 @@

Source code for gen3.tools.download.drs_download

def download_file_from_url( url: str, filename: Path, show_progress: bool = True ) -> bool: - """ + """ Downloads a file using the URL. The URL is a pre-signed url created by the download manager from the access method of the DRS object. @@ -595,56 +595,56 @@

Source code for gen3.tools.download.drs_download

Returns: True if object has been downloaded - """ + """ try: response = requests.get(url, stream=True) response.raise_for_status() except requests.exceptions.Timeout: - logger.critical(f"Was unable to get the download url: {url}. Timeout Error.") + logger.critical(f"Was unable to get the download url: {url}. Timeout Error.") return False except requests.exceptions.HTTPError as exc: logger.critical( - f"HTTP Error ({exc.response.status_code}): downloading file from {url}" + f"HTTP Error ({exc.response.status_code}): downloading file from {url}" ) return False - total_size_in_bytes = int(response.headers.get("content-length", 0)) + total_size_in_bytes = int(response.headers.get("content-length", 0)) if total_size_in_bytes == 0: - logger.warning(f"content-length is 0") + logger.warning(f"content-length is 0") total_downloaded = 0 block_size = 8092 # 8K blocks might want to tune this. progress_bar = ( tqdm( - desc=f"{str(filename) : <45}", + desc=f"{str(filename) : <45}", total=total_size_in_bytes, - unit="iB", + unit="iB", unit_scale=True, - bar_format="{l_bar:45}{bar:35}{r_bar}{bar:-10b}", + bar_format="{l_bar:45}{bar:35}{r_bar}{bar:-10b}", ) if show_progress else InvisibleProgress() ) - # if the file name contains '/', create subdirectories and download there + # if the file name contains '/', create subdirectories and download there ensure_dirpath_exists(Path(os.path.dirname(filename))) try: - with open(filename, "wb") as file: + with open(filename, "wb") as file: for data in response.iter_content(block_size): progress_bar.update(len(data)) total_downloaded += len(data) file.write(data) except IOError as ex: - logger.critical(f"IOError opening {filename} for writing: {ex}") + logger.critical(f"IOError opening {filename} for writing: {ex}") return False if total_downloaded != total_size_in_bytes: logger.critical( - f"Error in downloading {filename}: expected {total_size_in_bytes} bytes, downloaded {total_downloaded} bytes" + f"Error in downloading {filename}: expected {total_size_in_bytes} bytes, downloaded {total_downloaded} bytes" ) return False return True @@ -652,16 +652,16 @@

Source code for gen3.tools.download.drs_download

def unpackage_object(filepath: str): # allowed formats are set in PACKAGE_EXTENSIONS - with zipfile.ZipFile(filepath, "r") as package: + with zipfile.ZipFile(filepath, "r") as package: package.extractall(os.path.dirname(filepath)) def parse_drs_identifier(drs_candidate: str) -> Tuple[str, str, str]: - """ + """ Parses a DRS identifier to extract a hostname in the case of hostname based DRS otherwise it look for a DRS compact identifier. - If neither one is recognized return an empty string and a type of 'unknown' + If neither one is recognized return an empty string and a type of 'unknown' Note: The regex expressions used to extract hostname or identifier has a potential for a false positive. @@ -669,30 +669,30 @@

Source code for gen3.tools.download.drs_download

drs_candidate (str): a drs object identifier Returns: - Tuple (str): tuple of hostname/drs prefix, guid, string: one of "hostname", "compact", "unknown" - """ + Tuple (str): tuple of hostname/drs prefix, guid, string: one of "hostname", "compact", "unknown" + """ # determine if hostname or compact identifier or unknown - drs_regex = r"drs://([A-Za-z0-9\.\-\~]+)/([A-Za-z0-9\.\-\_\~\/]+)" + drs_regex = r"drs://([A-Za-z0-9\.\-\~]+)/([A-Za-z0-9\.\-\_\~\/]+)" # either a drs prefix: matches = re.findall(drs_regex, drs_candidate, re.UNICODE) if len(matches) == 1: # this could be a hostname DRS id hostname_regex = ( - r"^(([a-zA-Z0-9]|[a-zA-Z0-9][a-zA-Z0-9\-]*[a-zA-Z0-9])\.)*" - r"([A-Za-z0-9]|[A-Za-z0-9][A-Za-z0-9\-]*[A-Za-z0-9])$" + r"^(([a-zA-Z0-9]|[a-zA-Z0-9][a-zA-Z0-9\-]*[a-zA-Z0-9])\.)*" + r"([A-Za-z0-9]|[A-Za-z0-9][A-Za-z0-9\-]*[A-Za-z0-9])$" ) hostname_matches = re.findall(hostname_regex, matches[0][0], re.UNICODE) if len(hostname_matches) == 1: - return matches[0][0], matches[0][1], "hostname" + return matches[0][0], matches[0][1], "hostname" # possible compact rep - compact_regex = r"([A-Za-z0-9\.\-\~]+)/([A-Za-z0-9\.\-\_\~\/]+)" + compact_regex = r"([A-Za-z0-9\.\-\~]+)/([A-Za-z0-9\.\-\_\~\/]+)" matches = re.findall(compact_regex, drs_candidate, re.UNICODE) if len(matches) == 1 and len(matches[0]) == 2: - return matches[0][0], matches[0][1], "compact" + return matches[0][0], matches[0][1], "compact" - # can't figure out a this identifier - return "", "", "unknown" + # can't figure out a this identifier + return "", "", "unknown" def resolve_drs_hostname_from_id( @@ -700,7 +700,7 @@

Source code for gen3.tools.download.drs_download

resolved_drs_prefix_cache: dict, mds_url: str, ) -> Optional[Tuple[str, str, str]]: - """Resolves and returns a DRS identifier + """Resolves and returns a DRS identifier The resolved_drs_prefix_cache is updated if needed and is a potential side effect of this call Args: @@ -710,13 +710,13 @@

Source code for gen3.tools.download.drs_download

Returns: the hostname of the DRS server if resolved, otherwise it returns None - """ + """ hostname = None prefix, identifier, identifier_type = parse_drs_identifier(object_id) - if identifier_type == "hostname": + if identifier_type == "hostname": return prefix, identifier, identifier_type - if identifier_type == "compact": + if identifier_type == "compact": if prefix not in resolved_drs_prefix_cache: hostname = resolve_drs(prefix, object_id, metadata_service_url=mds_url) if hostname is not None: @@ -733,14 +733,14 @@

Source code for gen3.tools.download.drs_download

mds_url: str, commons_url: str = None, ) -> None: - """Given a list of object_ids go through list and resolve + cache any unknown hosts + """Given a list of object_ids go through list and resolve + cache any unknown hosts Args: object_ids (List[Downloadable]): list of object to resolve resolved_drs_prefix_cache (dict): cache of resolved DRS prefixes mds_url (str): Gen3 metadata service to resolve DRS prefixes hostname (str): Hostname to main Gen3 environment - """ + """ for entry in object_ids: if commons_url is not None: entry.hostname = commons_url @@ -750,20 +750,20 @@

Source code for gen3.tools.download.drs_download

entry.object_id, resolved_drs_prefix_cache, mds_url ) if ( - drs_type == "hostname" + drs_type == "hostname" ): # drs_type is a hostname so object id will be the GUID entry.object_id = nid def ensure_dirpath_exists(path: Path) -> Path: - """Utility to create a directory if missing. + """Utility to create a directory if missing. Returns the path so that the call can be inlined in a another call Args: path (Path): path to create Returns path of created directory - """ + """ assert path out_path: Path = path @@ -776,7 +776,7 @@

Source code for gen3.tools.download.drs_download

def get_download_url_using_drs( drs_hostname: str, object_id: str, access_method: str, access_token: str ) -> Tuple[Optional[int], Optional[str]]: - """ + """ Returns the presigned URL for a DRS object, from a DRS hostname, via the access method Args: drs_hostname (str): hostname of DRS server @@ -787,86 +787,86 @@

Source code for gen3.tools.download.drs_download

Returns: presigned url to object status code - """ + """ headers = { - "Content-Type": "application/json", - "Accept": "application/json", - "Authorization": "bearer " + access_token, + "Content-Type": "application/json", + "Accept": "application/json", + "Authorization": "bearer " + access_token, } try: response = requests.get( - url=f"https://{drs_hostname}/ga4gh/drs/v1/objects/{object_id}/access/{access_method}", + url=f"https://{drs_hostname}/ga4gh/drs/v1/objects/{object_id}/access/{access_method}", headers=headers, ) response.raise_for_status() data = response.json() - return data.get("url", None), response.status_code + return data.get("url", None), response.status_code except requests.exceptions.Timeout: - logger.critical(f"Was unable to download: {object_id}. Timeout Error.") + logger.critical(f"Was unable to download: {object_id}. Timeout Error.") except requests.exceptions.HTTPError as exc: logger.critical( - f"HTTP Error ({exc.response.status_code}) when requesting download url from {access_method}" + f"HTTP Error ({exc.response.status_code}) when requesting download url from {access_method}" ) return None, exc.response.status_code return None, None def get_user_auth(commons_url: str, access_token: str) -> Optional[List[str]]: - """ - Retrieves a user's authz for the commons based on the access token. + """ + Retrieves a user's authz for the commons based on the access token. Any error will be logged and None is returned Args: commons_url (str): hostname of Gen3 indexd - access_token (str): user's auth token + access_token (str): user's auth token Returns: The authz object from the user endpoint - """ + """ headers = { - "Content-Type": "application/json", - "Accept": "application/json", - "Authorization": "bearer " + access_token, + "Content-Type": "application/json", + "Accept": "application/json", + "Authorization": "bearer " + access_token, } try: - response = requests.get(url=f"https://{commons_url}/user/user", headers=headers) + response = requests.get(url=f"https://{commons_url}/user/user", headers=headers) response.raise_for_status() data = response.json() - authz = data["authz"] + authz = data["authz"] return authz except requests.exceptions.HTTPError as exc: - logger.critical(f"HTTP Error ({exc.response.status_code}): getting user access") + logger.critical(f"HTTP Error ({exc.response.status_code}): getting user access") return None def list_auth(hostname: str, authz: dict): - """ + """ Prints the authz for a DRS hostname Args: hostname (str): hostname to list access authz (str): dictionary of authz stringts - """ + """ print( - "───────────────────────────────────────────────────────────────────────────────────────────────────────" + "───────────────────────────────────────────────────────────────────────────────────────────────────────" ) - print(f"Access for {hostname}:") + print(f"Access for {hostname}:") if authz is not None and len(authz) > 0: for access, methods in authz.items(): print( - f" {access : <55}: {' '.join(dict.fromkeys(x['method'] for x in methods).keys()):>40}" + f" {access : <55}: {' '.join(dict.fromkeys(x['method'] for x in methods).keys()):>40}" ) else: - print(" No access") + print(" No access") def get_hostname_from_endpoint(endpoint: str): - """ + """ Get hostname from an Gen3Auth endpoint value Args: endpoint (str): endpoint value form Gen3Auth - """ + """ if not endpoint: return None urlparts = urlparse(endpoint) @@ -876,10 +876,10 @@

Source code for gen3.tools.download.drs_download

[docs] class DownloadManager: - """ + """ Class to assist in downloading a list of Downloadable object which at a minimum is a json manifest of DRS object ids. The methods of interest are download and user_access. - """ + """ def __init__( self, @@ -889,7 +889,7 @@

Source code for gen3.tools.download.drs_download

show_progress: bool = False, commons_url: str = None, ): - """ + """ Initialize the DownloadManager so that is ready to start downloading. Note the downloadable objects are required so that all tokens are available to support the download. @@ -898,7 +898,7 @@

Source code for gen3.tools.download.drs_download

hostname (str): Gen3 commons home commons auth (Gen3Auth) : Gen3 authentication download_list (List[Downloadable]): list of objects to download - """ + """ self.hostname = ( hostname if hostname else get_hostname_from_endpoint(auth.endpoint) @@ -914,7 +914,7 @@

Source code for gen3.tools.download.drs_download

hostname=self.hostname, access_token=self.access_token, use_wts=False, - expire=datetime.fromtimestamp(decode_token(self.access_token)["exp"]), + expire=datetime.fromtimestamp(decode_token(self.access_token)["exp"]), ) } self.download_list = download_list @@ -923,22 +923,22 @@

Source code for gen3.tools.download.drs_download

[docs] def resolve_objects(self, object_list: List[Downloadable], show_progress: bool): - """ + """ Given an Downloadable object list, resolve the DRS hostnames and update each Downloadable Args: object_list (List[Downloadable]): list of Downloadable objects to resolve - """ + """ resolve_objects_drs_hostname( object_list, self.resolved_compact_drs, - mds_url=f"http://{self.hostname}/mds/aggregate/info" + mds_url=f"http://{self.hostname}/mds/aggregate/info" if self.hostname else None, commons_url=self.commons_url, ) progress_bar = ( - tqdm(desc=f"Resolving objects", total=len(object_list)) + tqdm(desc=f"Resolving objects", total=len(object_list)) if show_progress else InvisibleProgress() ) @@ -952,9 +952,9 @@

Source code for gen3.tools.download.drs_download

[docs] def cache_hosts_wts_tokens(self, object_list): - """ - Using the list of DRS host obtain a WTS token for all DRS hosts in the list. It's is possible - """ + """ + Using the list of DRS host obtain a WTS token for all DRS hosts in the list. It's is possible + """ # create two sets: one of the know WTS host and the other of the host in the manifest wts_endpoint_set = set(self.wts_endpoints.keys()) object_id_hostnames = { @@ -970,12 +970,12 @@

Source code for gen3.tools.download.drs_download

for drs_hostname in drs_in_wts: endpoint = KnownDRSEndpoint( hostname=drs_hostname, - idp=self.wts_endpoints[drs_hostname]["idp"], + idp=self.wts_endpoints[drs_hostname]["idp"], ) endpoint.renew_token(self.hostname, self.access_token) self.known_hosts[drs_hostname] = endpoint for drs_hostname in drs_not_in_wts: - # if we already know the host then we don't need to reset the host + # if we already know the host then we don't need to reset the host if drs_hostname in self.known_hosts: continue # mark hostname as unavailable @@ -983,22 +983,22 @@

Source code for gen3.tools.download.drs_download

hostname=drs_hostname, ) logger.critical( - f"Could not retrieve a token for {drs_hostname}: it is not available as a WTS endpoint." + f"Could not retrieve a token for {drs_hostname}: it is not available as a WTS endpoint." )
[docs] def get_fresh_token(self, drs_hostname: str) -> Optional[str]: - """Will return and/or refresh and return a WTS token if hostname is known otherwise returns None. + """Will return and/or refresh and return a WTS token if hostname is known otherwise returns None. Args: drs_hostname (str): hostname to get token for Returns: access token if successful otherwise None - """ + """ if drs_hostname not in self.known_hosts: - logger.critical(f"Could not find {drs_hostname} in cache.") + logger.critical(f"Could not find {drs_hostname} in cache.") return None if self.known_hosts[drs_hostname].available: if not self.known_hosts[drs_hostname].expired(): @@ -1018,12 +1018,12 @@

Source code for gen3.tools.download.drs_download

def download( self, object_list: List[Downloadable], - save_directory: str = ".", + save_directory: str = ".", show_progress: bool = False, unpack_packages: bool = True, delete_unpacked_packages: bool = False, ) -> Dict[str, Any]: - """ + """ Downloads objects to the directory or current working directory. The input is an list of Downloadable object created by loading a manifest using the Manifest class or a call to Manifest.load(... @@ -1043,7 +1043,7 @@

Source code for gen3.tools.download.drs_download

Returns: List of DownloadStatus objects for each object id in object_list - """ + """ self.cache_hosts_wts_tokens(object_list) output_dir = Path(save_directory) @@ -1072,23 +1072,23 @@

Source code for gen3.tools.download.drs_download

if entry.hostname is None: logger.critical( - f"Unable to resolve, skipping {entry.object_id}. Skipping" + f"Unable to resolve, skipping {entry.object_id}. Skipping" ) - completed[entry.object_id].status = "error (resolving DRS host)" + completed[entry.object_id].status = "error (resolving DRS host)" continue # check to see if we have tokens if entry.hostname not in self.known_hosts: logger.critical( - f"{entry.hostname} is not present in this commons remote user access. Skipping {entry.file_name}" + f"{entry.hostname} is not present in this commons remote user access. Skipping {entry.file_name}" ) - completed[entry.object_id].status = "error (resolving DRS host)" + completed[entry.object_id].status = "error (resolving DRS host)" continue if self.known_hosts[entry.hostname].available is False: logger.critical( - f"Was unable to get user authorization from {entry.hostname}. Skipping {entry.file_name}" + f"Was unable to get user authorization from {entry.hostname}. Skipping {entry.file_name}" ) - completed[entry.object_id].status = "error (no auth)" + completed[entry.object_id].status = "error (no auth)" continue drs_hostname = entry.hostname @@ -1096,18 +1096,18 @@

Source code for gen3.tools.download.drs_download

if access_token is None: logger.critical( - f"No access token defined for {entry.object_id}. Skipping" + f"No access token defined for {entry.object_id}. Skipping" ) - completed[entry.object_id].status = "error (no access token)" + completed[entry.object_id].status = "error (no access token)" continue # TODO refine the selection of access_method if len(entry.access_methods) == 0: logger.critical( - f"No access methods defined for {entry.object_id}. Skipping" + f"No access methods defined for {entry.object_id}. Skipping" ) - completed[entry.object_id].status = "error (no access methods)" + completed[entry.object_id].status = "error (no access methods)" continue - access_method = entry.access_methods[0]["access_id"] + access_method = entry.access_methods[0]["access_id"] download_url, status_code = get_download_url_using_drs( drs_hostname, @@ -1119,7 +1119,7 @@

Source code for gen3.tools.download.drs_download

if download_url is None: if status_code != 200: completed[entry.object_id].status_code = status_code - completed[entry.object_id].status = "error" + completed[entry.object_id].status = "error" continue completed[entry.object_id].start_time = datetime.now(timezone.utc) @@ -1136,29 +1136,29 @@

Source code for gen3.tools.download.drs_download

except Exception: mds_entry = {} # no MDS or object not in MDS logger.debug( - f"{entry.file_name} is not a package and will not be expanded" + f"{entry.file_name} is not a package and will not be expanded" ) - # if the metadata type is "package", then unpack - if mds_entry.get("type") == "package": + # if the metadata type is "package", then unpack + if mds_entry.get("type") == "package": try: unpackage_object(filepath) except Exception as e: logger.critical( - f"{entry.file_name} had an issue while being unpackaged: {e}" + f"{entry.file_name} had an issue while being unpackaged: {e}" ) res = False if delete_unpacked_packages: filepath.unlink() if res: - completed[entry.object_id].status = "downloaded" + completed[entry.object_id].status = "downloaded" logger.debug( - f"object {entry.object_id} has been successfully downloaded." + f"object {entry.object_id} has been successfully downloaded." ) else: - completed[entry.object_id].status = "error" - logger.debug(f"object {entry.object_id} has failed to be downloaded.") + completed[entry.object_id].status = "error" + logger.debug(f"object {entry.object_id} has failed to be downloaded.") completed[entry.object_id].end_time = datetime.now(timezone.utc) return completed
@@ -1167,20 +1167,20 @@

Source code for gen3.tools.download.drs_download

[docs] def user_access(self): - """ - List the user's access permissions on each host needed to download + """ + List the user's access permissions on each host needed to download DRS objects in the manifest. A useful way to determine if access permissions are one reason a download failed. Returns: list of authz for each DRS host - """ + """ results = {} self.cache_hosts_wts_tokens(self.download_list) for hostname in self.known_hosts.keys(): if self.known_hosts[hostname].available is False: logger.critical( - f"Was unable to get user authorization from {hostname}." + f"Was unable to get user authorization from {hostname}." ) continue access_token = self.known_hosts[hostname].access_token @@ -1196,13 +1196,13 @@

Source code for gen3.tools.download.drs_download

hostname, auth, infile, - output_dir=".", + output_dir=".", show_progress=False, unpack_packages=True, delete_unpacked_packages=False, commons_url=None, ) -> Optional[Dict[str, Any]]: - """ + """ A convenience function used to download a json manifest. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1215,15 +1215,15 @@

Source code for gen3.tools.download.drs_download

Returns: List of DownloadStatus objects for each object id in object_list - """ + """ object_list = Manifest.load(Path(infile)) if object_list is None: - logger.critical(f"Error loading {infile}") + logger.critical(f"Error loading {infile}") return None try: auth.get_access_token() except Gen3AuthError: - logger.critical(f"Unable to authenticate your credentials with {hostname}") + logger.critical(f"Unable to authenticate your credentials with {hostname}") return downloader = DownloadManager( @@ -1248,13 +1248,13 @@

Source code for gen3.tools.download.drs_download

hostname, auth, object_ids, - output_dir=".", + output_dir=".", show_progress=False, unpack_packages=True, delete_unpacked_packages=False, commons_url=None, ) -> Optional[Dict[str, Any]]: - """ + """ A convenience function used to download a single DRS object. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1267,11 +1267,11 @@

Source code for gen3.tools.download.drs_download

Returns: List of DownloadStatus objects for the DRS object - """ + """ try: auth.get_access_token() except Gen3AuthError: - logger.critical(f"Unable to authenticate your credentials with {hostname}") + logger.critical(f"Unable to authenticate your credentials with {hostname}") return None object_list = [Downloadable(object_id=object_id) for object_id in object_ids] @@ -1294,7 +1294,7 @@

Source code for gen3.tools.download.drs_download

def _listfiles(hostname, auth, infile: str) -> bool: - """ + """ A wrapper function used by the cli to list files in a manifest. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1303,7 +1303,7 @@

Source code for gen3.tools.download.drs_download

Returns: True if successfully listed - """ + """ object_list = Manifest.load(Path(infile)) if object_list is None: return False @@ -1311,11 +1311,11 @@

Source code for gen3.tools.download.drs_download

try: auth.get_access_token() except Gen3AuthError: - logger.critical(f"Unable to authenticate your credentials with {hostname}") + logger.critical(f"Unable to authenticate your credentials with {hostname}") return False except requests.exceptions.RequestException as ex: logger.critical( - f"Unable to authenticate your credentials with {hostname}: {str(ex)}" + f"Unable to authenticate your credentials with {hostname}: {str(ex)}" ) return False @@ -1330,7 +1330,7 @@

Source code for gen3.tools.download.drs_download

def _list_object(hostname, auth, object_id: str) -> bool: - """ + """ Lists a DRS object. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1339,15 +1339,15 @@

Source code for gen3.tools.download.drs_download

Returns: True if successfully listed - """ + """ try: auth.get_access_token() except Gen3AuthError: - logger.critical(f"Unable to authenticate your credentials with {hostname}") + logger.critical(f"Unable to authenticate your credentials with {hostname}") return False except requests.exceptions.RequestException as ex: logger.critical( - f"Unable to authenticate your credentials with {hostname}: {ex}" + f"Unable to authenticate your credentials with {hostname}: {ex}" ) return False @@ -1366,7 +1366,7 @@

Source code for gen3.tools.download.drs_download

def _list_access(hostname, auth, infile: str) -> bool: - """ + """ A convenience function to list a users access for all DRS hostname in a manifest. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1375,7 +1375,7 @@

Source code for gen3.tools.download.drs_download

Returns: True if successfully listed - """ + """ object_list = Manifest.load(Path(infile)) if object_list is None: return False @@ -1383,11 +1383,11 @@

Source code for gen3.tools.download.drs_download

try: auth.get_access_token() except Gen3AuthError: - logger.critical(f"Unable to authenticate your credentials with {hostname}") + logger.critical(f"Unable to authenticate your credentials with {hostname}") return False except requests.exceptions.RequestException as ex: logger.critical( - f"Unable to authenticate your credentials with {hostname}: {ex}" + f"Unable to authenticate your credentials with {hostname}: {ex}" ) return False @@ -1401,11 +1401,11 @@

Source code for gen3.tools.download.drs_download

return True -# These functions are exposed to the SDK's cli under the drs-pull subcommand +# These functions are exposed to the SDK's cli under the drs-pull subcommand
[docs] def list_files_in_drs_manifest(hostname, auth, infile: str) -> bool: - """ + """ A wrapper function used by the cli to list files in a manifest. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1414,7 +1414,7 @@

Source code for gen3.tools.download.drs_download

Returns: True if successfully listed - """ + """ return _listfiles(hostname, auth, infile)
@@ -1422,7 +1422,7 @@

Source code for gen3.tools.download.drs_download

[docs] def list_drs_object(hostname, auth, object_id: str) -> bool: - """ + """ A convenience function used to list a DRS object. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1431,7 +1431,7 @@

Source code for gen3.tools.download.drs_download

Returns: True if successfully listed - """ + """ return _list_object(hostname, auth, object_id) # pragma: no cover
@@ -1448,7 +1448,7 @@

Source code for gen3.tools.download.drs_download

delete_unpacked_packages=False, commons_url=None, ) -> None: - """ + """ A convenience function used to download a json manifest. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1459,7 +1459,7 @@

Source code for gen3.tools.download.drs_download

delete_unpacked_packages (bool): set to True to delete package files after unpacking them Returns: - """ + """ _download( hostname, auth, @@ -1483,7 +1483,7 @@

Source code for gen3.tools.download.drs_download

delete_unpacked_packages=False, commons_url=None, ) -> None: - """ + """ A convenience function used to download a single DRS object. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1495,7 +1495,7 @@

Source code for gen3.tools.download.drs_download

Returns: List of DownloadStatus objects for the DRS object - """ + """ return _download_obj( hostname, auth, @@ -1511,7 +1511,7 @@

Source code for gen3.tools.download.drs_download

[docs] def list_access_in_drs_manifest(hostname, auth, infile) -> bool: - """ + """ A convenience function to list a users access for all DRS hostname in a manifest. Args: hostname (str): hostname of Gen3 commons to use for access and WTS @@ -1520,7 +1520,7 @@

Source code for gen3.tools.download.drs_download

Returns: True if successfully listed - """ + """ return _list_access(hostname, auth, infile)
diff --git a/docs/_build/html/_modules/gen3/tools/indexing/download_manifest.html b/docs/_build/html/_modules/gen3/tools/indexing/download_manifest.html index ad9bd1fb..a9e34fb9 100644 --- a/docs/_build/html/_modules/gen3/tools/indexing/download_manifest.html +++ b/docs/_build/html/_modules/gen3/tools/indexing/download_manifest.html @@ -31,10 +31,10 @@

Source code for gen3.tools.indexing.download_manifest

-"""
+"""
 Module for indexing actions for downloading a manifest of
-indexed file objects (against indexd's API). Supports
-multiple processes and coroutines using Python's asyncio library.
+indexed file objects (against indexd's API). Supports
+multiple processes and coroutines using Python's asyncio library.
 
 The default manifest format created is a Comma-Separated Value file (csv)
 with rows for every record. A header row is created with field names:
@@ -48,11 +48,11 @@ 

Source code for gen3.tools.indexing.download_manifest

MAX_CONCURRENT_REQUESTS (int): maximum number of desired concurrent requests across processes/threads TMP_FOLDER (str): Folder directory for placing temporary files - NOTE - We have to use a temporary folder b/c Python's file writing is not - thread-safe so we can't have all processes writing to the same file. + NOTE - We have to use a temporary folder b/c Python's file writing is not + thread-safe so we can't have all processes writing to the same file. To workaround this, we have each process write to a file and concat them all post-processing. -""" +""" import asyncio import aiofiles import click @@ -86,7 +86,7 @@

Source code for gen3.tools.indexing.download_manifest

INDEXD_RECORD_PAGE_SIZE = 1024 MAX_CONCURRENT_REQUESTS = 24 CURRENT_DIR = os.path.dirname(os.path.realpath(__file__)) -TMP_FOLDER = os.path.abspath(CURRENT_DIR + "/tmp") + "/" +TMP_FOLDER = os.path.abspath(CURRENT_DIR + "/tmp") + "/" logging = get_logger(__name__) @@ -95,13 +95,13 @@

Source code for gen3.tools.indexing.download_manifest

[docs] async def async_download_object_manifest( commons_url, - output_filename="object-manifest.csv", + output_filename="object-manifest.csv", num_processes=4, max_concurrent_requests=MAX_CONCURRENT_REQUESTS, input_manifest=None, - python_subprocess_command="python", + python_subprocess_command="python", ): - """ + """ Download all file object records into a manifest csv Args: @@ -118,27 +118,27 @@

Source code for gen3.tools.indexing.download_manifest

python_subprocess_command (str, optional): Command used to execute a python process. By default you should not need to change this, but if you are running something like MacOS and only installed Python 3.x - you may need to specify "python3". - """ + you may need to specify "python3". + """ start_time = time.perf_counter() - logging.info(f"start time: {start_time}") + logging.info(f"start time: {start_time}") # ensure tmp directories exists and are empty os.makedirs(TMP_FOLDER, exist_ok=True) - os.makedirs(TMP_FOLDER + "input", exist_ok=True) - os.makedirs(TMP_FOLDER + "output", exist_ok=True) + os.makedirs(TMP_FOLDER + "input", exist_ok=True) + os.makedirs(TMP_FOLDER + "output", exist_ok=True) for file in os.listdir(TMP_FOLDER): file_path = os.path.join(TMP_FOLDER, file) if os.path.isfile(file_path): os.unlink(file_path) - for file in os.listdir(TMP_FOLDER + "input"): - file_path = os.path.join(TMP_FOLDER + "input", file) + for file in os.listdir(TMP_FOLDER + "input"): + file_path = os.path.join(TMP_FOLDER + "input", file) if os.path.isfile(file_path): os.unlink(file_path) - for file in os.listdir(TMP_FOLDER + "output"): - file_path = os.path.join(TMP_FOLDER + "output", file) + for file in os.listdir(TMP_FOLDER + "output"): + file_path = os.path.join(TMP_FOLDER + "output", file) if os.path.isfile(file_path): os.unlink(file_path) @@ -152,8 +152,8 @@

Source code for gen3.tools.indexing.download_manifest

) end_time = time.perf_counter() - logging.info(f"end time: {end_time}") - logging.info(f"run time: {end_time-start_time}")
+ logging.info(f"end time: {end_time}") + logging.info(f"run time: {end_time-start_time}")
@@ -165,7 +165,7 @@

Source code for gen3.tools.indexing.download_manifest

input_manifest, python_subprocess_command, ): - """ + """ Spins up number of processes provided to parse indexd records and eventually write to a single output file manifest. @@ -180,7 +180,7 @@

Source code for gen3.tools.indexing.download_manifest

input_manifest (str): Input file. Read available object data from objects in this file instead of reading everything in indexd. This will attempt to query indexd for only the records identified in this manifest. - """ + """ # used when requesting all records page_chunks = [] @@ -190,40 +190,40 @@

Source code for gen3.tools.indexing.download_manifest

if input_manifest: # create chunks of checksums - logging.debug(f"parsing input file {input_manifest}") + logging.debug(f"parsing input file {input_manifest}") input_records, headers = get_and_verify_fileinfos_from_manifest(input_manifest) num_input_records = len(input_records) if not num_input_records: raise AttributeError( - f"No valid records found in provided input file: {input_manifest}. " - "Please check previous logs." + f"No valid records found in provided input file: {input_manifest}. " + "Please check previous logs." ) - logging.debug(f"number input_records: {num_input_records}") - logging.debug(f"num processes: {num_processes}") + logging.debug(f"number input_records: {num_input_records}") + logging.debug(f"num processes: {num_processes}") # batch records into subprocesses chunks chunk_size = int(math.ceil(float(num_input_records) / num_processes)) - logging.debug(f"records chunk size: {chunk_size}") + logging.debug(f"records chunk size: {chunk_size}") record_chunks = list(yield_chunks(input_records, chunk_size)) else: index = Gen3Index(commons_url) - logging.debug(f"requesting indexd stats...") - num_files = int(index.get_stats().get("fileCount")) - logging.debug(f"number files: {num_files}") + logging.debug(f"requesting indexd stats...") + num_files = int(index.get_stats().get("fileCount")) + logging.debug(f"number files: {num_files}") # paging is 0-based, so subtract 1 from ceiling # note: float() is necessary to force Python 3 to not floor the result max_page = int(math.ceil(float(num_files) / INDEXD_RECORD_PAGE_SIZE)) - 1 - logging.debug(f"max page: {max_page}") - logging.debug(f"num processes: {num_processes}") + logging.debug(f"max page: {max_page}") + logging.debug(f"num processes: {num_processes}") pages = [x for x in range(max_page + 1)] # batch pages into subprocesses chunk_size = int(math.ceil(float(len(pages)) / num_processes)) - logging.debug(f"page chunk size: {chunk_size}") + logging.debug(f"page chunk size: {chunk_size}") if chunk_size: page_chunks = [ @@ -232,31 +232,31 @@

Source code for gen3.tools.indexing.download_manifest

processes = [] for x in range(max(len(page_chunks), len(record_chunks))): - pages = ",".join(map(str, page_chunks[x])) if page_chunks else "," + pages = ",".join(map(str, page_chunks[x])) if page_chunks else "," input_record_chunks = ( - "|||".join(map(json.dumps, record_chunks[x])) if record_chunks else "|||" + "|||".join(map(json.dumps, record_chunks[x])) if record_chunks else "|||" ) # write record_checksum chunks to temporary files since the size can overload # command line arguments - input_records_chunk_filename = TMP_FOLDER + f"input/input_records_chunk_{x}.txt" + input_records_chunk_filename = TMP_FOLDER + f"input/input_records_chunk_{x}.txt" logging.info( - f"writing input_record_chunks chunk {x} into {input_records_chunk_filename}" + f"writing input_record_chunks chunk {x} into {input_records_chunk_filename}" ) - with open(input_records_chunk_filename, "wb") as outfile: - outfile.write(input_record_chunks.encode("utf8")) + with open(input_records_chunk_filename, "wb") as outfile: + outfile.write(input_record_chunks.encode("utf8")) # call the cli function below and pass in chunks of pages for each process command = ( - f"{python_subprocess_command} {CURRENT_DIR}/download_manifest.py --commons_url " - f"{commons_url} --pages {pages} --input-record-chunks-file {input_records_chunk_filename} " - f"--num_processes {num_processes} --max_concurrent_requests {max_concurrent_requests}" + f"{python_subprocess_command} {CURRENT_DIR}/download_manifest.py --commons_url " + f"{commons_url} --pages {pages} --input-record-chunks-file {input_records_chunk_filename} " + f"--num_processes {num_processes} --max_concurrent_requests {max_concurrent_requests}" ) logging.info(command) process = await asyncio.create_subprocess_shell(command) - logging.info(f"Process_{process.pid} - Started w/: {command}") + logging.info(f"Process_{process.pid} - Started w/: {command}") processes.append(process) for process in processes: @@ -264,59 +264,59 @@

Source code for gen3.tools.indexing.download_manifest

stdout, stderr = await process.communicate() if process.returncode == 0: - logging.info(f"Process_{process.pid} - Done") + logging.info(f"Process_{process.pid} - Done") else: - logging.info(f"Process_{process.pid} - FAILED") + logging.info(f"Process_{process.pid} - FAILED") - logging.info(f"done processing, combining outputs to single file {output_filename}") + logging.info(f"done processing, combining outputs to single file {output_filename}") # remove existing output if it exists if os.path.isfile(output_filename): os.unlink(output_filename) - with open(output_filename, "wb") as outfile: - outfile.write("guid,urls,authz,acl,md5,file_size,file_name\n".encode("utf8")) - for filename in glob.glob(TMP_FOLDER + "output/*"): + with open(output_filename, "wb") as outfile: + outfile.write("guid,urls,authz,acl,md5,file_size,file_name\n".encode("utf8")) + for filename in glob.glob(TMP_FOLDER + "output/*"): if output_filename == filename: - # don't want to copy the output into the output + # don't want to copy the output into the output continue - logging.info(f"combining {filename} into {output_filename}") - with open(filename, "rb") as readfile: + logging.info(f"combining {filename} into {output_filename}") + with open(filename, "rb") as readfile: shutil.copyfileobj(readfile, outfile) - logging.info(f"done writing output to file {output_filename}") + logging.info(f"done writing output to file {output_filename}") @click.command() @click.option( - "--commons_url", help="Root domain (url) for a commons containing indexd." + "--commons_url", help="Root domain (url) for a commons containing indexd." ) @click.option( - "--pages", - help='Comma-Separated string of integers representing pages. ex: "2,4,5,6"', + "--pages", + help='Comma-Separated string of integers representing pages. ex: "2,4,5,6"', ) @click.option( - "--input-record-chunks-file", + "--input-record-chunks-file", help=( - "File containing delimited string of records to retrieve." "ex: /foo/bar.txt" + "File containing delimited string of records to retrieve." "ex: /foo/bar.txt" ), ) @click.option( - "--num_processes", + "--num_processes", type=int, - help="number of processes you are running so we can make sure we don't open " - 'too many http connections. ex: "4"', + help="number of processes you are running so we can make sure we don't open " + 'too many http connections. ex: "4"', ) @click.option( - "--max_concurrent_requests", + "--max_concurrent_requests", type=int, - help="number of processes you are running so we can make sure we don't open " - 'too many http connections. ex: "4"', + help="number of processes you are running so we can make sure we don't open " + 'too many http connections. ex: "4"', ) def write_page_records_to_files( commons_url, pages, input_record_chunks_file, num_processes, max_concurrent_requests ): - """ + """ Command line interface function for requesting a number of records from indexd and writing to a file in that process. num_processes is only used to calculate how many open connections this process should request. @@ -335,34 +335,34 @@

Source code for gen3.tools.indexing.download_manifest

Raises: AttributeError: No pages specified to get records from - """ + """ if not pages and not input_record_chunks_file: raise AttributeError( - "No info specified to get records with. " - "Supply either pages or input-record-chunks-file" + "No info specified to get records with. " + "Supply either pages or input-record-chunks-file" ) - pages = [item for item in pages.strip().strip(",").split(",") if item] + pages = [item for item in pages.strip().strip(",").split(",") if item] input_record_chunks = [] if input_record_chunks_file: - with open(input_record_chunks_file, "r", encoding="utf8") as file: - input_record_chunks_from_file = "|||".join(file.readlines()) + with open(input_record_chunks_file, "r", encoding="utf8") as file: + input_record_chunks_from_file = "|||".join(file.readlines()) input_record_chunks = [ json.loads(item) - for item in input_record_chunks_from_file.strip().split("|||") + for item in input_record_chunks_from_file.strip().split("|||") if item ] if not pages and not input_record_chunks: raise AttributeError( - "No info specified to get records with. " - "Supply either pages or input-record-chunks-file with records in the file. " + "No info specified to get records with. " + "Supply either pages or input-record-chunks-file with records in the file. " ) if pages and input_record_chunks: raise AttributeError( - "YOU MUST USE EITHER `pages` OR `input-record-chunks-file`, YOU CANNOT USE BOTH. " - f"You provided pages={pages} and input-record-chunks-file={input_record_chunks_file}." + "YOU MUST USE EITHER `pages` OR `input-record-chunks-file`, YOU CANNOT USE BOTH. " + f"You provided pages={pages} and input-record-chunks-file={input_record_chunks_file}." ) loop = get_or_create_event_loop_for_thread() @@ -382,14 +382,14 @@

Source code for gen3.tools.indexing.download_manifest

async def _get_records_and_write_to_file( commons_url, pages, input_record_chunks, num_processes, max_concurrent_requests ): - """ + """ Getting indexd records and writing to a file. This function creates semaphores to limit the number of concurrent http connections that get opened to send requests to indexd. It then uses asyncio to start a number of coroutines. Steps: 1) requests to indexd to get records (writes resulting records to a queue) - 2) puts a final "DONE" in the queue to stop coroutine that will read from queue + 2) puts a final "DONE" in the queue to stop coroutine that will read from queue 3) reading those records from the queue and writing to a file Args: @@ -398,15 +398,15 @@

Source code for gen3.tools.indexing.download_manifest

input_record_chunks (List[dict]): List of indexd records to request num_processes (int): number of concurrent processes being requested (including this one) - """ + """ max_requests = int(max_concurrent_requests / num_processes) - logging.debug(f"max concurrent requests per process: {max_requests}") + logging.debug(f"max concurrent requests per process: {max_requests}") lock = asyncio.Semaphore(max_requests) queue = asyncio.Queue() write_to_file_task = asyncio.ensure_future(_parse_from_queue(queue)) if pages: - logging.debug("putting records from page into queue") + logging.debug("putting records from page into queue") await asyncio.gather( *( _put_records_from_page_in_queue(page, commons_url, lock, queue) @@ -414,7 +414,7 @@

Source code for gen3.tools.indexing.download_manifest

) ) else: - logging.debug("putting records from input manifest into queue") + logging.debug("putting records from input manifest into queue") await asyncio.gather( *( _put_records_from_input_manifest_in_queue( @@ -424,14 +424,14 @@

Source code for gen3.tools.indexing.download_manifest

) ) - await queue.put("DONE") + await queue.put("DONE") await write_to_file_task async def _put_records_from_input_manifest_in_queue( input_record, commons_url, lock, queue ): - """ + """ Gets a semaphore then requests records for the given input_record and puts them in a queue. @@ -441,14 +441,14 @@

Source code for gen3.tools.indexing.download_manifest

lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections queue (asyncio.Queue): queue to put indexd records in - """ + """ checksum = input_record.get(MD5_STANDARD_KEY) index = Gen3Index(commons_url) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False records = await index.async_get_records_from_checksum( @@ -463,7 +463,7 @@

Source code for gen3.tools.indexing.download_manifest

async def _put_records_from_page_in_queue(page, commons_url, lock, queue): - """ + """ Gets a semaphore then requests records for the given page and puts them in a queue. @@ -473,12 +473,12 @@

Source code for gen3.tools.indexing.download_manifest

lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections queue (asyncio.Queue): queue to put indexd records in - """ + """ index = Gen3Index(commons_url) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False records = await index.async_get_records_on_page( @@ -488,69 +488,69 @@

Source code for gen3.tools.indexing.download_manifest

async def _parse_from_queue(queue): - """ + """ Read from the queue and write to a file Args: queue (asyncio.Queue): queue to read indexd records from - """ + """ loop = get_or_create_event_loop_for_thread() - file_name = TMP_FOLDER + f"output/{os.getpid()}.csv" - async with aiofiles.open(file_name, "w+", encoding="utf8") as file: - logging.info(f"Writing to {file_name}") + file_name = TMP_FOLDER + f"output/{os.getpid()}.csv" + async with aiofiles.open(file_name, "w+", encoding="utf8") as file: + logging.info(f"Writing to {file_name}") csv_writer = csv.writer(file) records = await queue.get() - while records != "DONE": + while records != "DONE": if records: for record in list(records): # we want to represent records that are found correctly - # (e.g. ones with did's), but records that are directly from an input + # (e.g. ones with did's), but records that are directly from an input # manifest (e.g. no did) we do NOT want to modify, so # treat these cases separately - if record.get("did"): - urls = " ".join( + if record.get("did"): + urls = " ".join( sorted( [ - url.replace(" ", "%20") - for url in record.get("urls") + url.replace(" ", "%20") + for url in record.get("urls") if url ] ) ) - authz = " ".join( + authz = " ".join( sorted( [ - authz_resource.replace(" ", "%20") - for authz_resource in record.get("authz") + authz_resource.replace(" ", "%20") + for authz_resource in record.get("authz") if authz_resource ] ) ) - acl = " ".join( + acl = " ".join( sorted( - [a.replace(" ", "%20") for a in record.get("acl") if a] + [a.replace(" ", "%20") for a in record.get("acl") if a] ) ) manifest_row = [ - record.get("did", ""), + record.get("did", ""), urls, authz, acl, - record.get("hashes", {}).get("md5", ""), - record.get("size", ""), - record.get("file_name", ""), + record.get("hashes", {}).get("md5", ""), + record.get("size", ""), + record.get("file_name", ""), ] else: manifest_row = [ - record.get(GUID_STANDARD_KEY, ""), - record.get(URLS_STANDARD_KEY, ""), - record.get(AUTHZ_STANDARD_KEY, ""), - record.get(ACL_STANDARD_KEY, ""), - record.get(MD5_STANDARD_KEY, ""), - record.get(SIZE_STANDARD_KEY, ""), - record.get(FILENAME_STANDARD_KEY, ""), + record.get(GUID_STANDARD_KEY, ""), + record.get(URLS_STANDARD_KEY, ""), + record.get(AUTHZ_STANDARD_KEY, ""), + record.get(ACL_STANDARD_KEY, ""), + record.get(MD5_STANDARD_KEY, ""), + record.get(SIZE_STANDARD_KEY, ""), + record.get(FILENAME_STANDARD_KEY, ""), ] await csv_writer.writerow(manifest_row) @@ -559,7 +559,7 @@

Source code for gen3.tools.indexing.download_manifest

file.flush() -if __name__ == "__main__": +if __name__ == "__main__": write_page_records_to_files()
diff --git a/docs/_build/html/_modules/gen3/tools/indexing/index_manifest.html b/docs/_build/html/_modules/gen3/tools/indexing/index_manifest.html index 142ed2ab..d859b083 100644 --- a/docs/_build/html/_modules/gen3/tools/indexing/index_manifest.html +++ b/docs/_build/html/_modules/gen3/tools/indexing/index_manifest.html @@ -31,8 +31,8 @@

Source code for gen3.tools.indexing.index_manifest

-"""
-Module for indexing object files in a manifest (against indexd's API).
+"""
+Module for indexing object files in a manifest (against indexd's API).
 
 The default manifest format created is a Tab-Separated Value file (tsv)
 with rows for every record.
@@ -43,11 +43,11 @@ 

Source code for gen3.tools.indexing.index_manifest

All supported formats of acl, authz and url fields are shown in the below example. guid md5 size acl authz url -255e396f-f1f8-11e9-9a07-0a80fada099c 473d83400bc1bc9dc635e334faddf33c 363455714 ['Open'] [s3://pdcdatastore/test1.raw] +255e396f-f1f8-11e9-9a07-0a80fada099c 473d83400bc1bc9dc635e334faddf33c 363455714 ['Open'] [s3://pdcdatastore/test1.raw] 255e396f-f1f8-11e9-9a07-0a80fada098c 473d83400bc1bc9dc635e334faddd33c 343434344 Open s3://pdcdatastore/test2.raw 255e396f-f1f8-11e9-9a07-0a80fada097c 473d83400bc1bc9dc635e334fadd433c 543434443 phs0001 phs0002 s3://pdcdatastore/test3.raw -255e396f-f1f8-11e9-9a07-0a80fada096c 473d83400bc1bc9dc635e334fadd433c 363455714 ['phs0001', 'phs0002'] ['s3://pdcdatastore/test4.raw'] -255e396f-f1f8-11e9-9a07-0a80fada010c 473d83400bc1bc9dc635e334fadde33c 363455714 ['Open'] s3://pdcdatastore/test5.raw +255e396f-f1f8-11e9-9a07-0a80fada096c 473d83400bc1bc9dc635e334fadd433c 363455714 ['phs0001', 'phs0002'] ['s3://pdcdatastore/test4.raw'] +255e396f-f1f8-11e9-9a07-0a80fada010c 473d83400bc1bc9dc635e334fadde33c 363455714 ['Open'] s3://pdcdatastore/test5.raw Attributes: CURRENT_DIR (str): directory this file is in @@ -60,9 +60,9 @@

Source code for gen3.tools.indexing.index_manifest

PREV_GUID (list(string)): supported previous guid column names Usages: - python index_manifest.py --commons_url https://giangb.planx-pla.net --manifest_file path_to_manifest --auth "admin,admin" --replace_urls False --thread_num 10 + python index_manifest.py --commons_url https://giangb.planx-pla.net --manifest_file path_to_manifest --auth "admin,admin" --replace_urls False --thread_num 10 python index_manifest.py --commons_url https://giangb.planx-pla.net --manifest_file path_to_manifest --api_key ./credentials.json --replace_urls False --thread_num 10 -""" +""" import os import csv import click @@ -104,9 +104,9 @@

Source code for gen3.tools.indexing.index_manifest

[docs] class ThreadControl(object): - """ + """ Class for thread synchronization - """ + """ def __init__(self, processed_files=0, num_total_files=0): self.mutexLock = threading.Lock() @@ -116,7 +116,7 @@

Source code for gen3.tools.indexing.index_manifest

def _write_csv(filename, files, fieldnames=None): - """ + """ write to csv file Args: @@ -124,24 +124,24 @@

Source code for gen3.tools.indexing.index_manifest

files(list(dict)): list of file info [ { - "guid": "guid_example", - "filename": "example", - "size": 100, - "acl": "['open']", - "md5": "md5_hash", + "guid": "guid_example", + "filename": "example", + "size": 100, + "acl": "['open']", + "md5": "md5_hash", }, ] fieldnames(list(str)): list of column names Returns: None - """ + """ if not files: return None fieldnames = fieldnames or files[0].keys() - with open(filename, mode="w") as outfile: - writer = csv.DictWriter(outfile, delimiter="\t", fieldnames=fieldnames) + with open(filename, mode="w") as outfile: + writer = csv.DictWriter(outfile, delimiter="\t", fieldnames=fieldnames) writer.writeheader() for f in files: @@ -159,7 +159,7 @@

Source code for gen3.tools.indexing.index_manifest

force_metadata_columns_even_if_empty, fi, ): - """ + """ Index a single file, and submit additional metadata to the metadata service if provided Args: @@ -174,49 +174,49 @@

Source code for gen3.tools.indexing.index_manifest

Returns: None - """ + """ index_success = True try: urls = ( get_urls(fi[URLS_STANDARD_KEY]) if URLS_STANDARD_KEY in fi - and fi[URLS_STANDARD_KEY] != "[]" + and fi[URLS_STANDARD_KEY] != "[]" and fi[URLS_STANDARD_KEY] else [] ) authz = ( [ - element.strip().replace("'", "").replace('"', "").replace("%20", " ") + element.strip().replace("'", "").replace('"', "").replace("%20", " ") for element in _standardize_str(fi[AUTHZ_STANDARD_KEY]) .strip() - .lstrip("[") - .rstrip("]") - .split(" ") + .lstrip("[") + .rstrip("]") + .split(" ") ] if AUTHZ_STANDARD_KEY in fi - and fi[AUTHZ_STANDARD_KEY] != "[]" + and fi[AUTHZ_STANDARD_KEY] != "[]" and fi[AUTHZ_STANDARD_KEY] else [] ) if ACL_STANDARD_KEY in fi: - if fi[ACL_STANDARD_KEY].strip().lower() in {"[u'open']", "['open']"}: - acl = ["*"] + if fi[ACL_STANDARD_KEY].strip().lower() in {"[u'open']", "['open']"}: + acl = ["*"] else: acl = ( [ element.strip() - .replace("'", "") - .replace('"', "") - .replace("%20", " ") + .replace("'", "") + .replace('"', "") + .replace("%20", " ") for element in _standardize_str(fi[ACL_STANDARD_KEY]) .strip() - .lstrip("[") - .rstrip("]") - .split(" ") + .lstrip("[") + .rstrip("]") + .split(" ") ] if ACL_STANDARD_KEY in fi - and fi[ACL_STANDARD_KEY] != "[]" + and fi[ACL_STANDARD_KEY] != "[]" and fi[ACL_STANDARD_KEY] else [] ) @@ -226,7 +226,7 @@

Source code for gen3.tools.indexing.index_manifest

if FILENAME_STANDARD_KEY in fi: file_name = _standardize_str(fi[FILENAME_STANDARD_KEY]) else: - file_name = "" + file_name = "" if fi.get(PREV_GUID_STANDARD_KEY): prev_guid = fi[PREV_GUID_STANDARD_KEY] @@ -243,7 +243,7 @@

Source code for gen3.tools.indexing.index_manifest

MD5_STANDARD_KEY ) != fi.get(MD5_STANDARD_KEY): logging.error( - "The guid {} with different size/hash already exists. Can not index it without getting a new guid".format( + "The guid {} with different size/hash already exists. Can not index it without getting a new guid".format( fi.get(GUID_STANDARD_KEY) ) ) @@ -254,7 +254,7 @@

Source code for gen3.tools.indexing.index_manifest

doc.urls = urls need_update = True - # indexd doesn't like when records have metadata for non-existing + # indexd doesn't like when records have metadata for non-existing # urls new_urls_metadata = copy.deepcopy(doc.urls_metadata) for url, metadata in doc.urls_metadata.items(): @@ -281,17 +281,17 @@

Source code for gen3.tools.indexing.index_manifest

need_update = True if need_update: - logging.info(f"updating {doc.did} to: {doc.to_json()}") + logging.info(f"updating {doc.did} to: {doc.to_json()}") doc.patch() else: if fi.get(GUID_STANDARD_KEY): - guid = fi.get(GUID_STANDARD_KEY, "").strip() + guid = fi.get(GUID_STANDARD_KEY, "").strip() else: guid = None record = { - "did": guid, - "hashes": {MD5_STANDARD_KEY: fi.get(MD5_STANDARD_KEY, "").strip()}, + "did": guid, + "hashes": {MD5_STANDARD_KEY: fi.get(MD5_STANDARD_KEY, "").strip()}, SIZE_STANDARD_KEY: fi.get(SIZE_STANDARD_KEY, 0), ACL_STANDARD_KEY: acl, AUTHZ_STANDARD_KEY: authz, @@ -300,41 +300,41 @@

Source code for gen3.tools.indexing.index_manifest

} if prev_guid: - logging.info(f"creating new version of {prev_guid}: {record}") + logging.info(f"creating new version of {prev_guid}: {record}") - # indexd exports a "form" field that gets populated in indexclient.create, + # indexd exports a "form" field that gets populated in indexclient.create, # but not indexclient.add_version, need to add manually here - record.update({"form": "object"}) + record.update({"form": "object"}) # to generate new GUID, new version indexd API expects body to not - # contain "did", rather than have it be None or "" - new_guid = record["did"] - if not record["did"]: - del record["did"] + # contain "did", rather than have it be None or "" + new_guid = record["did"] + if not record["did"]: + del record["did"] new_guid = None new_doc = Document(client=None, did=new_guid, json=record) # TODO: in the case where a new GUID *is* provided AND a version with that - # guid already exists AND you run this, it's gonna throw an error. + # guid already exists AND you run this, it's gonna throw an error. # We need to gracefully handle the error from indexd. there will # be a conflict where a version with this GUID already exists... # but if we can verify that the version has the correct values # for all the fields, we can effectively ignore this error and continue doc = indexclient.add_version(current_did=prev_guid, new_doc=new_doc) else: - logging.info(f"creating: {record}") + logging.info(f"creating: {record}") doc = indexclient.create(**record) fi[GUID_STANDARD_KEY] = doc.did except Exception as e: - # Don't break for any reason + # Don't break for any reason index_success = False exc_info = sys.exc_info() traceback.print_exception(*exc_info) logging.error( - "Can not update/create an indexd record with guid {}. Detail: {}".format( + "Can not update/create an indexd record with guid {}. Detail: {}".format( fi.get(GUID_STANDARD_KEY), e ) ) @@ -344,7 +344,7 @@

Source code for gen3.tools.indexing.index_manifest

try: if not mds: raise Exception( - "Can not submit to the metadata service when using indexd basic auth" + "Can not submit to the metadata service when using indexd basic auth" ) metadata = mds._prepare_metadata( fi, @@ -354,11 +354,11 @@

Source code for gen3.tools.indexing.index_manifest

if metadata: mds.create(guid=doc.did, metadata=metadata, overwrite=True) except Exception as e: - # Don't break, but delete indexd record + # Don't break, but delete indexd record exc_info = sys.exc_info() traceback.print_exception(*exc_info) logging.error( - "Can not create package metadata for guid {}. Deleting indexd record. Detail: {}".format( + "Can not create package metadata for guid {}. Deleting indexd record. Detail: {}".format( fi[GUID_STANDARD_KEY], e ) ) @@ -368,7 +368,7 @@

Source code for gen3.tools.indexing.index_manifest

exc_info = sys.exc_info() traceback.print_exception(*exc_info) logging.error( - "Cannot delete indexd record with {}. Detail: {}".format( + "Cannot delete indexd record with {}. Detail: {}".format( fi[GUID_STANDARD_KEY], e ) ) @@ -377,7 +377,7 @@

Source code for gen3.tools.indexing.index_manifest

thread_control.num_processed_files += 1 if (thread_control.num_processed_files * 10) % thread_control.num_total_files == 0: logging.info( - "Progress: {}%".format( + "Progress: {}%".format( thread_control.num_processed_files * 100.0 / thread_control.num_total_files @@ -395,11 +395,11 @@

Source code for gen3.tools.indexing.index_manifest

auth=None, replace_urls=True, manifest_file_delimiter=None, - output_filename="indexing-output-manifest.csv", + output_filename="indexing-output-manifest.csv", submit_additional_metadata_columns=False, force_metadata_columns_even_if_empty=True, ): - """ + """ Loop through all the files in the manifest, update/create records in indexd update indexd if the url is not in the record url list or acl has changed @@ -409,7 +409,7 @@

Source code for gen3.tools.indexing.index_manifest

thread_num(int): number of threads for indexing auth(Gen3Auth): Gen3 auth or tuple with basic auth name and password replace_urls(bool): flag to indicate if replace urls or not - manifest_file_delimiter(str): manifest's delimiter + manifest_file_delimiter(str): manifest's delimiter output_filename(str): output file name for manifest submit_additional_metadata_columns(bool): whether to submit additional metadata to the metadata service force_metadata_columns_even_if_empty(bool): force the creation of a metadata column @@ -423,58 +423,58 @@

Source code for gen3.tools.indexing.index_manifest

2, ..., , dataB, Resulting metadata if force_metadata_columns_even_if_empty=True : - "1": { - "columnA": "dataA", - "columnB": "", - "ColumnC": "", + "1": { + "columnA": "dataA", + "columnB": "", + "ColumnC": "", }, - "2": { - "columnA": "", - "columnB": "dataB", - "ColumnC": "", + "2": { + "columnA": "", + "columnB": "dataB", + "ColumnC": "", }, Resulting metadata if force_metadata_columns_even_if_empty=False : - "1": { - "columnA": "dataA", + "1": { + "columnA": "dataA", }, - "2": { - "columnB": "dataB", + "2": { + "columnB": "dataB", }, Returns: files(list(dict)): list of file info [ { - "guid": "guid_example", - "filename": "example", - "size": 100, - "acl": "['open']", - "md5": "md5_hash", + "guid": "guid_example", + "filename": "example", + "size": 100, + "acl": "['open']", + "md5": "md5_hash", }, ] headers(list(str)): list of fieldnames - """ - logging.info("Start the process ...") - service_location = "index" - commons_url = commons_url.strip("/") + """ + logging.info("Start the process ...") + service_location = "index" + commons_url = commons_url.strip("/") # if running locally, indexd is deployed by itself without a location relative # to the commons - if "http://localhost" in commons_url: - service_location = "" + if "http://localhost" in commons_url: + service_location = "" if not commons_url.endswith(service_location): - commons_url += "/" + service_location + commons_url += "/" + service_location - logging.info("\nUsing URL {}\n".format(commons_url)) + logging.info("\nUsing URL {}\n".format(commons_url)) - indexclient = client.IndexClient(commons_url, "v0", auth=auth) + indexclient = client.IndexClient(commons_url, "v0", auth=auth) if isinstance(auth, tuple): # basic auth if submit_additional_metadata_columns: logging.warning( - f"'submit_additional_metadata_columns' is on, but using indexd basic auth. Will not be able to submit to the metadata service. To create metadata, use Gen3Auth instance instead." + f"'submit_additional_metadata_columns' is on, but using indexd basic auth. Will not be able to submit to the metadata service. To create metadata, use Gen3Auth instance instead." ) mds = None else: # Gen3Auth @@ -487,7 +487,7 @@

Source code for gen3.tools.indexing.index_manifest

except Exception as e: exc_info = sys.exc_info() traceback.print_exception(*exc_info) - logging.error("Can not read {}. Detail: {}".format(manifest_file, e)) + logging.error("Can not read {}. Detail: {}".format(manifest_file, e)) return None, None # Early terminate @@ -522,7 +522,7 @@

Source code for gen3.tools.indexing.index_manifest

pool.join() output_filename = os.path.abspath(output_filename) - logging.info(f"Writing output to {output_filename}") + logging.info(f"Writing output to {output_filename}") # remove existing output if it exists if os.path.isfile(output_filename): @@ -536,42 +536,42 @@

Source code for gen3.tools.indexing.index_manifest

@click.command() @click.option( - "--commons-url", - "commons_url", - help="Root domain (url) for a commons containing indexd.", + "--commons-url", + "commons_url", + help="Root domain (url) for a commons containing indexd.", required=True, ) -@click.option("--manifest_file", help="The path to input manifest") +@click.option("--manifest_file", help="The path to input manifest") @click.option( - "--thread-num", - "thread_num", + "--thread-num", + "thread_num", type=int, - help="Number of threads", + help="Number of threads", default=1, show_default=True, ) -@click.option("--api-key", "api_key", help="path to api key") -@click.option("--auth", help="basic auth") +@click.option("--api-key", "api_key", help="path to api key") +@click.option("--auth", help="basic auth") @click.option( - "--replace-urls", - "replace_urls", + "--replace-urls", + "replace_urls", type=bool, - help="If supplied, will replace urls for existing records. e.g. existing urls will be overwritten by the new ones", + help="If supplied, will replace urls for existing records. e.g. existing urls will be overwritten by the new ones", default=False, show_default=True, ) @click.option( - "--manifest-file-delimiter", - "manifest_file_delimiter", - help="string character that delimites the file (tab or comma). Defaults to tab.", - default="\t", + "--manifest-file-delimiter", + "manifest_file_delimiter", + help="string character that delimites the file (tab or comma). Defaults to tab.", + default="\t", show_default=True, ) @click.option( - "--out-manifest-file", - "out_manifest_file", - help="The path to output manifest", - default="indexing-output-manifest.csv", + "--out-manifest-file", + "out_manifest_file", + help="The path to output manifest", + default="indexing-output-manifest.csv", show_default=True, ) def index_object_manifest_cli( @@ -584,7 +584,7 @@

Source code for gen3.tools.indexing.index_manifest

manifest_file_delimiter, out_manifest_file, ): - """ + """ Commandline interface for indexing a given manifest to indexd Args: @@ -596,18 +596,18 @@

Source code for gen3.tools.indexing.index_manifest

replace_urls(bool): Replace urls or not NOTE: if both api_key and auth are specified, it will ignore the later and take the former as a default - manifest_file_delimiter(str): manifest's delimiter + manifest_file_delimiter(str): manifest's delimiter out_manifest_file(str): path to the output manifest - """ + """ if api_key: auth = Gen3Auth(commons_url, refresh_file=api_key) else: - auth = tuple(auth.split(",")) if auth else None + auth = tuple(auth.split(",")) if auth else None files, headers = index_object_manifest( - commons_url + "/index", + commons_url + "/index", manifest_file, int(thread_num), auth, @@ -617,8 +617,8 @@

Source code for gen3.tools.indexing.index_manifest

) -if __name__ == "__main__": - logging.basicConfig(filename="index_object_manifest.log", level=logging.DEBUG) +if __name__ == "__main__": + logging.basicConfig(filename="index_object_manifest.log", level=logging.DEBUG) logging.getLogger().addHandler(logging.StreamHandler(sys.stdout)) index_object_manifest_cli() @@ -626,47 +626,47 @@

Source code for gen3.tools.indexing.index_manifest

[docs] def delete_all_guids(auth, file): - """ + """ Delete all GUIDs specified in the object manifest. WARNING: THIS COMPLETELY REMOVES INDEX RECORDS. USE THIS ONLY IF YOU KNOW THE IMPLICATIONS. - """ + """ index = Gen3Index(auth.endpoint, auth_provider=auth) if not index.is_healthy(): logging.debug( - f"uh oh! The indexing service is not healthy in the commons {auth.endpoint}" + f"uh oh! The indexing service is not healthy in the commons {auth.endpoint}" ) exit() # try to get delimeter based on file ext file_ext = os.path.splitext(file) - if file_ext[-1].lower() == ".tsv": - manifest_file_delimiter = "\t" + if file_ext[-1].lower() == ".tsv": + manifest_file_delimiter = "\t" else: # default, assume CSV - manifest_file_delimiter = "," + manifest_file_delimiter = "," - with open(file, "r", encoding="utf-8-sig") as input_file: + with open(file, "r", encoding="utf-8-sig") as input_file: csvReader = csv.DictReader(input_file, delimiter=manifest_file_delimiter) fieldnames = csvReader.fieldnames - logging.debug(f"got fieldnames from {file}: {fieldnames}") + logging.debug(f"got fieldnames from {file}: {fieldnames}") # figure out which permutation of the name GUID is being used 1 time # then use it for all future rows - guid_name = "guid" - for name in ["guid", "GUID", "did", "DID"]: + guid_name = "guid" + for name in ["guid", "GUID", "did", "DID"]: if name in fieldnames: guid_name = name - logging.debug(f"using {guid_name} to retrieve GUID to delete...") + logging.debug(f"using {guid_name} to retrieve GUID to delete...") for row in csvReader: guid = row.get(guid_name) if guid: - logging.debug(f"deleting GUID record:{guid}") + logging.debug(f"deleting GUID record:{guid}") logging.debug(index.delete_record(guid=guid))
diff --git a/docs/_build/html/_modules/gen3/tools/indexing/verify_manifest.html b/docs/_build/html/_modules/gen3/tools/indexing/verify_manifest.html index 5d9a9974..c5f59de5 100644 --- a/docs/_build/html/_modules/gen3/tools/indexing/verify_manifest.html +++ b/docs/_build/html/_modules/gen3/tools/indexing/verify_manifest.html @@ -31,10 +31,10 @@

Source code for gen3.tools.indexing.verify_manifest

-"""
+"""
 Module for indexing actions for verifying a manifest of
-indexed file objects (against indexd's API). Supports
-multiple processes and coroutines using Python's asyncio library.
+indexed file objects (against indexd's API). Supports
+multiple processes and coroutines using Python's asyncio library.
 
 The default manifest format created is a Comma-Separated Value file (csv)
 with rows for every record. A header row is created with field names:
@@ -53,10 +53,10 @@ 

Source code for gen3.tools.indexing.verify_manifest

from gen3.tools.indexing.verify_manifest import manifest_row_parsers def _get_authz_from_row(row): - return [row.get("authz").strip().strip("[").strip("]").strip("'")] + return [row.get("authz").strip().strip("[").strip("]").strip("'")] # override default parsers -manifest_row_parsers["authz"] = _get_authz_from_row +manifest_row_parsers["authz"] = _get_authz_from_row indexing.verify_object_manifest(COMMONS) ``` @@ -65,13 +65,13 @@

Source code for gen3.tools.indexing.verify_manifest

format: {guid}|{error_name}|expected {value_from_manifest}|actual {value_from_indexd} -ex: 93d9af72-b0f1-450c-a5c6-7d3d8d2083b4|authz|expected ['']|actual ['/programs/DEV/projects/test'] +ex: 93d9af72-b0f1-450c-a5c6-7d3d8d2083b4|authz|expected ['']|actual ['/programs/DEV/projects/test'] Attributes: CURRENT_DIR (str): directory this file is in MAX_CONCURRENT_REQUESTS (int): maximum number of desired concurrent requests across processes/threads -""" +""" import aiohttp import asyncio import csv @@ -90,7 +90,7 @@

Source code for gen3.tools.indexing.verify_manifest

def _get_guid_from_row(row): - """ + """ Given a row from the manifest, return the field representing expected indexd guid. Args: @@ -98,69 +98,69 @@

Source code for gen3.tools.indexing.verify_manifest

Returns: str: guid - """ - guid = row.get("guid") + """ + guid = row.get("guid") if not guid: - guid = row.get("GUID") + guid = row.get("GUID") return guid def _get_md5_from_row(row): - """ - Given a row from the manifest, return the field representing file's md5 sum. + """ + Given a row from the manifest, return the field representing file's md5 sum. Args: row (dict): column_name:row_value Returns: str: md5 sum for file - """ - if "md5" in row: - return row["md5"] - elif "md5sum" in row: - return row["md5sum"] + """ + if "md5" in row: + return row["md5"] + elif "md5sum" in row: + return row["md5sum"] else: return None def _get_file_size_from_row(row): - """ - Given a row from the manifest, return the field representing file's size in bytes. + """ + Given a row from the manifest, return the field representing file's size in bytes. Args: row (dict): column_name:row_value Returns: int: integer representing file size in bytes - """ + """ try: - if "file_size" in row: - return int(row["file_size"]) - elif "size" in row: - return int(row["size"]) + if "file_size" in row: + return int(row["file_size"]) + elif "size" in row: + return int(row["size"]) else: return None except Exception: - logging.warning(f"could not convert this to an int: {row.get('file_size')}") - return row.get("file_size") + logging.warning(f"could not convert this to an int: {row.get('file_size')}") + return row.get("file_size") def _get_acl_from_row(row): - """ - Given a row from the manifest, return the field representing file's expected acls. + """ + Given a row from the manifest, return the field representing file's expected acls. Args: row (dict): column_name:row_value Returns: List[str]: acls for the indexd record - """ - return [item for item in row.get("acl", "").strip().split(" ") if item] + """ + return [item for item in row.get("acl", "").strip().split(" ") if item] def _get_authz_from_row(row): - """ - Given a row from the manifest, return the field representing file's expected authz + """ + Given a row from the manifest, return the field representing file's expected authz resources. Args: @@ -168,56 +168,56 @@

Source code for gen3.tools.indexing.verify_manifest

Returns: List[str]: authz resources for the indexd record - """ - return [item for item in row.get("authz", "").strip().split(" ") if item] + """ + return [item for item in row.get("authz", "").strip().split(" ") if item] def _get_urls_from_row(row): - """ - Given a row from the manifest, return the field representing file's expected urls. + """ + Given a row from the manifest, return the field representing file's expected urls. Args: row (dict): column_name:row_value Returns: List[str]: urls for indexd record file location(s) - """ - if "urls" in row: - return [item for item in row.get("urls", "").strip().split(" ") if item] - elif "url" in row: - return [item for item in row.get("urls", "").strip().split(" ") if item] + """ + if "urls" in row: + return [item for item in row.get("urls", "").strip().split(" ") if item] + elif "url" in row: + return [item for item in row.get("urls", "").strip().split(" ") if item] else: return [] def _get_file_name_from_row(row): - """ - Given a row from the manifest, return the field representing file's expected file_name. + """ + Given a row from the manifest, return the field representing file's expected file_name. Args: row (dict): column_name:row_value Returns: List[str]: file_name for indexd record file location(s) - """ - if "file_name" in row: - return row["file_name"] - elif "filename" in row: - return row["filename"] - elif "name" in row: - return row["name"] + """ + if "file_name" in row: + return row["file_name"] + elif "filename" in row: + return row["filename"] + elif "name" in row: + return row["name"] else: return None manifest_row_parsers = { - "guid": _get_guid_from_row, - "md5": _get_md5_from_row, - "file_size": _get_file_size_from_row, - "acl": _get_acl_from_row, - "authz": _get_authz_from_row, - "urls": _get_urls_from_row, - "file_name": _get_file_name_from_row, + "guid": _get_guid_from_row, + "md5": _get_md5_from_row, + "file_size": _get_file_size_from_row, + "acl": _get_acl_from_row, + "authz": _get_authz_from_row, + "urls": _get_urls_from_row, + "file_name": _get_file_name_from_row, } @@ -231,7 +231,7 @@

Source code for gen3.tools.indexing.verify_manifest

manifest_file_delimiter=None, output_filename=None, ): - """ + """ Verify all file object records into a manifest csv Args: @@ -241,23 +241,23 @@

Source code for gen3.tools.indexing.verify_manifest

manifest_row_parsers (Dict{indexd_field:func_to_parse_row}): Row parsers manifest_file_delimiter (str): delimeter in manifest_file output_filename (str): filename for output logs - """ + """ if not output_filename: - output_filename = f"verify-manifest-errors-{time.time()}.log" + output_filename = f"verify-manifest-errors-{time.time()}.log" start_time = time.perf_counter() - logging.info(f"start time: {start_time}") + logging.info(f"start time: {start_time}") # if delimiter not specified, try to get based on file ext if not manifest_file_delimiter: file_ext = os.path.splitext(manifest_file) - if file_ext[-1].lower() == ".tsv": - manifest_file_delimiter = "\t" + if file_ext[-1].lower() == ".tsv": + manifest_file_delimiter = "\t" else: # default, assume CSV - manifest_file_delimiter = "," + manifest_file_delimiter = "," - logging.debug(f"detected {manifest_file_delimiter} as delimiter between columns") + logging.debug(f"detected {manifest_file_delimiter} as delimiter between columns") await _verify_all_index_records_in_file( commons_url, @@ -268,8 +268,8 @@

Source code for gen3.tools.indexing.verify_manifest

) end_time = time.perf_counter() - logging.info(f"end time: {end_time}") - logging.info(f"run time: {end_time-start_time}")
+ logging.info(f"end time: {end_time}") + logging.info(f"run time: {end_time-start_time}")
@@ -280,14 +280,14 @@

Source code for gen3.tools.indexing.verify_manifest

max_concurrent_requests, output_filename, ): - """ + """ Getting indexd records and writing to a file. This function creates semaphores to limit the number of concurrent http connections that get opened to send requests to indexd. It then uses asyncio to start a number of coroutines. Steps: 1) requests to indexd to get records (writes resulting records to a queue) - 2) puts a final "DONE" in the queue to stop coroutine that will read from queue + 2) puts a final "DONE" in the queue to stop coroutine that will read from queue 3) reading those records from the queue and writing to a file Args: @@ -296,14 +296,14 @@

Source code for gen3.tools.indexing.verify_manifest

manifest_file_delimiter (str): delimeter in manifest_file output_filename (str, optional): filename for output max_concurrent_requests (int): the maximum number of concurrent requests allowed - """ + """ max_requests = int(max_concurrent_requests) - logging.debug(f"max concurrent requests: {max_requests}") + logging.debug(f"max concurrent requests: {max_requests}") lock = asyncio.Semaphore(max_requests) queue = asyncio.Queue() output_queue = asyncio.Queue() - with open(manifest_file, encoding="utf-8-sig") as manifest: + with open(manifest_file, encoding="utf-8-sig") as manifest: reader = csv.DictReader(manifest, delimiter=manifest_file_delimiter) for row in reader: @@ -316,16 +316,16 @@

Source code for gen3.tools.indexing.verify_manifest

await queue.put(new_row) for _ in range(0, int(max_concurrent_requests + (max_concurrent_requests / 4))): - await queue.put("DONE") + await queue.put("DONE") await asyncio.gather( *( _parse_from_queue(queue, lock, commons_url, output_queue) - # why "+ (max_concurrent_requests / 4)"? + # why "+ (max_concurrent_requests / 4)"? # This is because the max requests at any given time could be - # waiting for metadata responses all at once and there's processing done + # waiting for metadata responses all at once and there's processing done # before that semaphore, so this just adds a few extra processes to get - # through the queue up to that point of metadata requests so it's ready + # through the queue up to that point of metadata requests so it's ready # right away when a lock is released. Not entirely necessary but speeds # things up a tiny bit to always ensure something is waiting for that lock for x in range( @@ -336,23 +336,23 @@

Source code for gen3.tools.indexing.verify_manifest

output_filename = os.path.abspath(output_filename) logging.info( - f"done processing, writing output queue to single file {output_filename}" + f"done processing, writing output queue to single file {output_filename}" ) # remove existing output if it exists if os.path.isfile(output_filename): os.unlink(output_filename) - with open(output_filename, "w") as outfile: + with open(output_filename, "w") as outfile: while not output_queue.empty(): line = await output_queue.get() outfile.write(line) - logging.info(f"done writing output to file {output_filename}") + logging.info(f"done writing output to file {output_filename}") async def _parse_from_queue(queue, lock, commons_url, output_queue): - """ + """ Keep getting items from the queue and verifying that indexd contains the expected fields from that row. If there are any issues, log errors into a file. Return when nothing is left in the queue. @@ -363,81 +363,81 @@

Source code for gen3.tools.indexing.verify_manifest

connections commons_url (str): root domain for commons where indexd lives output_queue (asyncio.Queue): queue for output - """ + """ loop = get_or_create_event_loop_for_thread() row = await queue.get() - while row != "DONE": - guid = manifest_row_parsers["guid"](row) - authz = manifest_row_parsers["authz"](row) - acl = manifest_row_parsers["acl"](row) - file_size = manifest_row_parsers["file_size"](row) - md5 = manifest_row_parsers["md5"](row) - urls = manifest_row_parsers["urls"](row) - file_name = manifest_row_parsers["file_name"](row) + while row != "DONE": + guid = manifest_row_parsers["guid"](row) + authz = manifest_row_parsers["authz"](row) + acl = manifest_row_parsers["acl"](row) + file_size = manifest_row_parsers["file_size"](row) + md5 = manifest_row_parsers["md5"](row) + urls = manifest_row_parsers["urls"](row) + file_name = manifest_row_parsers["file_name"](row) actual_record = await _get_record_from_indexd(guid, commons_url, lock) if not actual_record: - output = f"{guid}|no_record|expected {row}|actual None\n" + output = f"{guid}|no_record|expected {row}|actual None\n" await output_queue.put(output) logging.error(output) else: - logging.info(f"verifying {guid}...") + logging.info(f"verifying {guid}...") - if sorted(authz) != sorted(actual_record["authz"]): + if sorted(authz) != sorted(actual_record["authz"]): output = ( - f"{guid}|authz|expected {authz}|actual {actual_record['authz']}\n" + f"{guid}|authz|expected {authz}|actual {actual_record['authz']}\n" ) await output_queue.put(output) logging.error(output) - if sorted(acl) != sorted(actual_record["acl"]): - output = f"{guid}|acl|expected {acl}|actual {actual_record['acl']}\n" + if sorted(acl) != sorted(actual_record["acl"]): + output = f"{guid}|acl|expected {acl}|actual {actual_record['acl']}\n" await output_queue.put(output) logging.error(output) - if file_size != actual_record["size"]: + if file_size != actual_record["size"]: if ( not file_size and file_size != 0 - and not actual_record["size"] - and actual_record["size"] != 0 + and not actual_record["size"] + and actual_record["size"] != 0 ): # actual and expected are both either empty string or None - # so even though they're not equal, they represent null value so - # we don't need to consider this an error in validation + # so even though they're not equal, they represent null value so + # we don't need to consider this an error in validation pass else: - output = f"{guid}|file_size|expected {file_size}|actual {actual_record['size']}\n" + output = f"{guid}|file_size|expected {file_size}|actual {actual_record['size']}\n" await output_queue.put(output) logging.error(output) - if md5 != actual_record["hashes"].get("md5"): + if md5 != actual_record["hashes"].get("md5"): if ( not md5 and md5 != 0 - and not actual_record["hashes"].get("md5") - and actual_record["hashes"].get("md5") != 0 + and not actual_record["hashes"].get("md5") + and actual_record["hashes"].get("md5") != 0 ): # actual and expected are both either empty string or None - # so even though they're not equal, they represent null value so - # we don't need to consider this an error in validation + # so even though they're not equal, they represent null value so + # we don't need to consider this an error in validation pass else: - output = f"{guid}|md5|expected {md5}|actual {actual_record['hashes'].get('md5')}\n" + output = f"{guid}|md5|expected {md5}|actual {actual_record['hashes'].get('md5')}\n" await output_queue.put(output) logging.error(output) - urls = [url.replace("%20", " ") for url in urls] - if sorted(urls) != sorted(actual_record["urls"]): - output = f"{guid}|urls|expected {urls}|actual {actual_record['urls']}\n" + urls = [url.replace("%20", " ") for url in urls] + if sorted(urls) != sorted(actual_record["urls"]): + output = f"{guid}|urls|expected {urls}|actual {actual_record['urls']}\n" await output_queue.put(output) logging.error(output) - if not actual_record["file_name"] and file_name: - # if the actual record name is "" or None but something was specified + if not actual_record["file_name"] and file_name: + # if the actual record name is "" or None but something was specified # in the manifest, we have a problem - output = f"{guid}|file_name|expected {file_name}|actual {actual_record['file_name']}\n" + output = f"{guid}|file_name|expected {file_name}|actual {actual_record['file_name']}\n" await output_queue.put(output) logging.error(output) @@ -445,7 +445,7 @@

Source code for gen3.tools.indexing.verify_manifest

async def _get_record_from_indexd(guid, commons_url, lock): - """ + """ Gets a semaphore then requests a record for the given guid Args: @@ -453,12 +453,12 @@

Source code for gen3.tools.indexing.verify_manifest

commons_url (str): root domain for commons where indexd lives lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections - """ + """ index = Gen3Index(commons_url) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False record = None @@ -467,7 +467,7 @@

Source code for gen3.tools.indexing.verify_manifest

return await index.async_get_record(guid, _ssl=ssl) except aiohttp.client_exceptions.ClientResponseError as exc: - logging.warning(f"couldn't get record. error: {exc}") + logging.warning(f"couldn't get record. error: {exc}") return record
diff --git a/docs/_build/html/_modules/gen3/tools/metadata/ingest_manifest.html b/docs/_build/html/_modules/gen3/tools/metadata/ingest_manifest.html index 64446862..830a16dd 100644 --- a/docs/_build/html/_modules/gen3/tools/metadata/ingest_manifest.html +++ b/docs/_build/html/_modules/gen3/tools/metadata/ingest_manifest.html @@ -31,7 +31,7 @@

Source code for gen3.tools.metadata.ingest_manifest

-"""
+"""
 Tools for ingesting a CSV/TSV metadata manifest into the Metdata Service.
 
 Attributes:
@@ -42,16 +42,16 @@ 

Source code for gen3.tools.metadata.ingest_manifest

NOT exist in indexd manifest_row_parsers (Dict{str: function}): functions for parsing, users can override manifest_row_parsers = { - "guid_from_file": _get_guid_for_row, - "indexed_file_object_guid": _query_for_associated_indexd_record_guid, + "guid_from_file": _get_guid_for_row, + "indexed_file_object_guid": _query_for_associated_indexd_record_guid, } - "guid_for_row" is the function to retrieve the guid from the given file - "indexed_file_object_guid" is the function to retrieve the guid from elsewhere, + "guid_for_row" is the function to retrieve the guid from the given file + "indexed_file_object_guid" is the function to retrieve the guid from elsewhere, like indexd (by querying) MAX_CONCURRENT_REQUESTS (int): Maximum concurrent requests to mds for ingestion -""" +""" import aiohttp import asyncio import csv @@ -65,19 +65,19 @@

Source code for gen3.tools.metadata.ingest_manifest

from gen3.index import Gen3Index from gen3.metadata import Gen3Metadata -TMP_FOLDER = os.path.abspath("./tmp") + "/" +TMP_FOLDER = os.path.abspath("./tmp") + "/" CURRENT_DIR = os.path.dirname(os.path.realpath(__file__)) MAX_CONCURRENT_REQUESTS = 24 -COLUMN_TO_USE_AS_GUID = "guid" -GUID_TYPE_FOR_INDEXED_FILE_OBJECT = "indexed_file_object" -GUID_TYPE_FOR_NON_INDEXED_FILE_OBJECT = "metadata_object" +COLUMN_TO_USE_AS_GUID = "guid" +GUID_TYPE_FOR_INDEXED_FILE_OBJECT = "indexed_file_object" +GUID_TYPE_FOR_NON_INDEXED_FILE_OBJECT = "metadata_object" logging = get_logger(__name__) def _get_guid_for_row(commons_url, row, lock): - """ + """ Given a row from the manifest, return the guid to use for the metadata object. Args: @@ -88,21 +88,21 @@

Source code for gen3.tools.metadata.ingest_manifest

Returns: str: guid - """ + """ return row.get(COLUMN_TO_USE_AS_GUID) async def _query_for_associated_indexd_record_guid( commons_url, row, lock, output_queue ): - """ + """ Given a row from the manifest, return the guid for the related indexd record. - By default attempts to use a column "submitted_sample_id" to pattern match + By default attempts to use a column "submitted_sample_id" to pattern match URLs in indexd records to find a single match. For example: - "NWD12345" would match a record with url: "s3://some-bucket/file_NWD12345.cram" + "NWD12345" would match a record with url: "s3://some-bucket/file_NWD12345.cram" WARNING: The query endpoint this uses in indexd is incredibly slow when there are lots of indexd records. @@ -116,21 +116,21 @@

Source code for gen3.tools.metadata.ingest_manifest

Returns: str: guid or None - """ - mapping = {"urls": "submitted_sample_id"} + """ + mapping = {"urls": "submitted_sample_id"} # Alternate example: # # mapping = { - # "acl": "study_with_consent", - # "size": "file_size", + # "acl": "study_with_consent", + # "size": "file_size", # } - if "urls" in mapping and len(mapping.items()) > 1: + if "urls" in mapping and len(mapping.items()) > 1: msg = ( - "You cannot pattern match 'urls' and exact match other fields for mapping " - "from indexd record to metadata columns. You can match by URL pattern " - "*OR* match exact record fields like size, hash, uploader, url, acl, authz." - f"\nYou provided mapping: {mapping}" + "You cannot pattern match 'urls' and exact match other fields for mapping " + "from indexd record to metadata columns. You can match by URL pattern " + "*OR* match exact record fields like size, hash, uploader, url, acl, authz." + f"\nYou provided mapping: {mapping}" ) logging.error(msg) await output_queue.put(msg) @@ -138,25 +138,25 @@

Source code for gen3.tools.metadata.ingest_manifest

# special query endpoint for matching url patterns, other fields # just use get with params - if "urls" in mapping: - pattern = row.get(mapping["urls"]) - logging.debug(f"trying to find matching record matching url pattern: {pattern}") + if "urls" in mapping: + pattern = row.get(mapping["urls"]) + logging.debug(f"trying to find matching record matching url pattern: {pattern}") records = await async_query_urls_from_indexd(pattern, commons_url, lock) else: params = { mapping_key: row.get(mapping_value) for mapping_key, mapping_value in mapping.items() } - logging.debug(f"trying to find matching record matching params: {params}") + logging.debug(f"trying to find matching record matching params: {params}") records = await _get_with_params_from_indexd(params, commons_url, lock) - logging.debug(f"matching record(s): {records}") + logging.debug(f"matching record(s): {records}") if len(records) > 1: msg = ( - "Multiple records were found with the given search criteria, this is assumed " - "to be unintentional so the metadata will NOT be linked to these records:\n" - f"{records}" + "Multiple records were found with the given search criteria, this is assumed " + "to be unintentional so the metadata will NOT be linked to these records:\n" + f"{records}" ) logging.warning(msg) await output_queue.put(msg) @@ -164,14 +164,14 @@

Source code for gen3.tools.metadata.ingest_manifest

guid = None if len(records) == 1: - guid = records[0].get("did") + guid = records[0].get("did") return guid manifest_row_parsers = { - "guid_for_row": _get_guid_for_row, - "indexed_file_object_guid": _query_for_associated_indexd_record_guid, + "guid_for_row": _get_guid_for_row, + "indexed_file_object_guid": _query_for_associated_indexd_record_guid, } @@ -189,7 +189,7 @@

Source code for gen3.tools.metadata.ingest_manifest

get_guid_from_file=True, metadata_type=None, ): - """ + """ Ingest all metadata records into a manifest csv Args: @@ -204,24 +204,24 @@

Source code for gen3.tools.metadata.ingest_manifest

output_filename (str): filename for output logs get_guid_from_file (bool): whether or not to get the guid for metadata from file NOTE: When this is True, will use the function in - manifest_row_parsers["guid_for_row"] to determine the GUID - (usually just a specific column in the file row like "guid") + manifest_row_parsers["guid_for_row"] to determine the GUID + (usually just a specific column in the file row like "guid") metadata_type (str): the type of metadata to be filled into the _guid_type field. If provided, will override the default logic per GUID: (GUID_TYPE_FOR_INDEXED_FILE_OBJECT if is_indexed_file_object else GUID_TYPE_FOR_NON_INDEXED_FILE_OBJECT) - """ + """ if not output_filename: - output_filename = f"ingest-metadata-manifest-errors-{time.time()}.log" + output_filename = f"ingest-metadata-manifest-errors-{time.time()}.log" # if delimiter not specified, try to get based on file ext if not manifest_file_delimiter: file_ext = os.path.splitext(manifest_file) - if file_ext[-1].lower() == ".tsv": - manifest_file_delimiter = "\t" + if file_ext[-1].lower() == ".tsv": + manifest_file_delimiter = "\t" else: # default, assume CSV - manifest_file_delimiter = "," + manifest_file_delimiter = "," await _ingest_all_metadata_in_file( commons_url, @@ -230,7 +230,7 @@

Source code for gen3.tools.metadata.ingest_manifest

auth, manifest_file_delimiter, max_concurrent_requests, - output_filename.split("/")[-1], + output_filename.split("/")[-1], get_guid_from_file, metadata_type, )
@@ -248,7 +248,7 @@

Source code for gen3.tools.metadata.ingest_manifest

get_guid_from_file, metadata_type, ): - """ + """ Ingest metadata from file into metadata service. This function creates semaphores to limit the number of concurrent http connections that get opened to send requests to mds. @@ -268,36 +268,36 @@

Source code for gen3.tools.metadata.ingest_manifest

output_filename (str): filename for output logs get_guid_from_file (bool): whether or not to get the guid for metadata from file NOTE: When this is True, will use the function in - manifest_row_parsers["guid_for_row"] to determine the GUID - (usually just a specific column in the file row like "guid") + manifest_row_parsers["guid_for_row"] to determine the GUID + (usually just a specific column in the file row like "guid") metadata_type (str): the type of metadata to be filled into the _guid_type field. If provided, will override the default logic per GUID: (GUID_TYPE_FOR_INDEXED_FILE_OBJECT if is_indexed_file_object else GUID_TYPE_FOR_NON_INDEXED_FILE_OBJECT) - """ + """ max_requests = int(max_concurrent_requests) - logging.debug(f"max concurrent requests: {max_requests}") + logging.debug(f"max concurrent requests: {max_requests}") lock = asyncio.Semaphore(max_requests) queue = asyncio.Queue() output_queue = asyncio.Queue() start_time = time.perf_counter() - msg = f"start time: {start_time}" + msg = f"start time: {start_time}" logging.info(msg) await output_queue.put(msg) - with open(manifest_file, encoding="utf-8-sig") as manifest: + with open(manifest_file, encoding="utf-8-sig") as manifest: reader = csv.DictReader(manifest, delimiter=manifest_file_delimiter) for row in reader: new_row = {} for key, value in row.items(): # I know this looks crazy, DictReader is doing goofy things when - # column contains a JSON-like string so we're trying to fix it here + # column contains a JSON-like string so we're trying to fix it here # Basically make sure the resulting column is something that we can # later json.loads(). # remove redudant quoting if value: - value = value.strip().strip("'").strip('"').replace("''", "'") + value = value.strip().strip("'").strip('"').replace("''", "'") new_row[key.strip()] = value await queue.put(new_row) @@ -313,11 +313,11 @@

Source code for gen3.tools.metadata.ingest_manifest

metadata_source, metadata_type, ) - # why "+ (max_concurrent_requests / 4)"? + # why "+ (max_concurrent_requests / 4)"? # This is because the max requests at any given time could be - # waiting for metadata responses all at once and there's processing done + # waiting for metadata responses all at once and there's processing done # before that semaphore, so this just adds a few extra processes to get - # through the queue up to that point of metadata requests so it's ready + # through the queue up to that point of metadata requests so it's ready # right away when a lock is released. Not entirely necessary but speeds # things up a tiny bit to always ensure something is waiting for that lock for x in range( @@ -327,29 +327,29 @@

Source code for gen3.tools.metadata.ingest_manifest

) end_time = time.perf_counter() - msg = f"end time: {end_time}" + msg = f"end time: {end_time}" logging.info(msg) await output_queue.put(msg) - msg = f"run time: {end_time-start_time}" + msg = f"run time: {end_time-start_time}" logging.info(msg) await output_queue.put(msg) output_filename = os.path.abspath(output_filename) logging.info( - f"done processing, writing output queue to single file {output_filename}" + f"done processing, writing output queue to single file {output_filename}" ) # remove existing output if it exists if os.path.isfile(output_filename): os.unlink(output_filename) - with open(output_filename, "w") as outfile: + with open(output_filename, "w") as outfile: while not output_queue.empty(): line = await output_queue.get() - outfile.write(line + "\n") + outfile.write(line + "\n") - logging.info(f"done writing output to file {output_filename}") + logging.info(f"done writing output to file {output_filename}") async def _parse_from_queue( @@ -362,7 +362,7 @@

Source code for gen3.tools.metadata.ingest_manifest

metadata_source, metadata_type, ): - """ + """ Keep getting items from the queue and checking if indexd contains a record with that guid. Then create/update metadta for that GUID in the metadata service. Also log to output queue. Return when nothing is left in the queue. @@ -376,42 +376,42 @@

Source code for gen3.tools.metadata.ingest_manifest

auth (Gen3Auth): Gen3 auth or tuple with basic auth name and password get_guid_from_file (bool): whether or not to get the guid for metadata from file NOTE: When this is True, will use the function in - manifest_row_parsers["guid_for_row"] to determine the GUID - (usually just a specific column in the file row like "guid") + manifest_row_parsers["guid_for_row"] to determine the GUID + (usually just a specific column in the file row like "guid") metadata_source (str): the name of the source of metadata (used to namespace in the metadata service) ex: dbgap metadata_type (str): the type of metadata to be filled into the _guid_type field. If provided, will override the default logic per GUID: (GUID_TYPE_FOR_INDEXED_FILE_OBJECT if is_indexed_file_object else GUID_TYPE_FOR_NON_INDEXED_FILE_OBJECT) - """ + """ while not queue.empty(): row = await queue.get() if get_guid_from_file: - guid = manifest_row_parsers["guid_for_row"](commons_url, row, lock) + guid = manifest_row_parsers["guid_for_row"](commons_url, row, lock) is_indexed_file_object = await _is_indexed_file_object( guid, commons_url, lock ) else: - guid = await manifest_row_parsers["indexed_file_object_guid"]( + guid = await manifest_row_parsers["indexed_file_object_guid"]( commons_url, row, lock, output_queue ) is_indexed_file_object = True if guid: - # construct metadata from rows, don't include redundant guid column - logging.debug(f"row: {row}") + # construct metadata from rows, don't include redundant guid column + logging.debug(f"row: {row}") metadata_from_file = {} for key, value in row.items(): try: new_value = json.loads(value) except json.decoder.JSONDecodeError as exc: - if "}" in value or "{" in value or "[" in value or "]" in value: + if "}" in value or "{" in value or "[" in value or "]" in value: msg = ( - f"Unable to json.loads a string that looks like json: {value}. " - f"adding as a string instead of nested json. Exception: {exc}" + f"Unable to json.loads a string that looks like json: {value}. " + f"adding as a string instead of nested json. Exception: {exc}" ) logging.warning(msg) await output_queue.put(msg) @@ -422,47 +422,47 @@

Source code for gen3.tools.metadata.ingest_manifest

if COLUMN_TO_USE_AS_GUID in metadata_from_file.keys(): del metadata_from_file[COLUMN_TO_USE_AS_GUID] - logging.debug(f"metadata from file: {metadata_from_file}") + logging.debug(f"metadata from file: {metadata_from_file}") # namespace by metadata source metadata = {metadata_source: metadata_from_file} if metadata_type: - metadata["_guid_type"] = metadata_type + metadata["_guid_type"] = metadata_type else: - metadata["_guid_type"] = ( + metadata["_guid_type"] = ( GUID_TYPE_FOR_INDEXED_FILE_OBJECT if is_indexed_file_object else GUID_TYPE_FOR_NON_INDEXED_FILE_OBJECT ) - logging.debug(f"metadata: {metadata}") + logging.debug(f"metadata: {metadata}") try: await _create_metadata(guid, metadata, auth, commons_url, lock) - msg = f"Successfully created {guid}" + msg = f"Successfully created {guid}" logging.info(msg) await output_queue.put(msg) except Exception as exc: logging.debug( - f"Got conflict for {guid}. Let's update instead of create..." + f"Got conflict for {guid}. Let's update instead of create..." ) await _update_metadata(guid, metadata, auth, commons_url, lock) - msg = f"Successfully updated {guid}" + msg = f"Successfully updated {guid}" logging.info(msg) await output_queue.put(msg) else: msg = ( - f"Did not add a metadata object for row because an invalid " - f"GUID was parsed or no record with this GUID was found in " - f"indexd: {guid}.\nRow: {row}" + f"Did not add a metadata object for row because an invalid " + f"GUID was parsed or no record with this GUID was found in " + f"indexd: {guid}.\nRow: {row}" ) logging.warning(msg) await output_queue.put(msg) async def _create_metadata(guid, metadata, auth, commons_url, lock): - """ + """ Gets a semaphore then creates metadata for guid Args: @@ -472,12 +472,12 @@

Source code for gen3.tools.metadata.ingest_manifest

commons_url (str): root domain for commons where metadata service lives lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections - """ + """ mds = Gen3Metadata(commons_url, auth_provider=auth) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False response = await mds.async_create(guid, metadata, _ssl=ssl) @@ -485,7 +485,7 @@

Source code for gen3.tools.metadata.ingest_manifest

async def _update_metadata(guid, metadata, auth, commons_url, lock): - """ + """ Gets a semaphore then updates metadata for guid Args: @@ -495,12 +495,12 @@

Source code for gen3.tools.metadata.ingest_manifest

commons_url (str): root domain for commons where metadata service lives lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections - """ + """ mds = Gen3Metadata(commons_url, auth_provider=auth) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False response = await mds.async_update(guid, metadata, _ssl=ssl) @@ -508,7 +508,7 @@

Source code for gen3.tools.metadata.ingest_manifest

async def _is_indexed_file_object(guid, commons_url, lock): - """ + """ Gets a semaphore then requests a record for the given guid Args: @@ -516,12 +516,12 @@

Source code for gen3.tools.metadata.ingest_manifest

commons_url (str): root domain for commons where mds lives lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections - """ + """ index = Gen3Index(commons_url) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False try: @@ -536,7 +536,7 @@

Source code for gen3.tools.metadata.ingest_manifest

[docs] async def async_query_urls_from_indexd(pattern, commons_url, lock): - """ + """ Gets a semaphore then requests a record for the given pattern Args: @@ -544,12 +544,12 @@

Source code for gen3.tools.metadata.ingest_manifest

commons_url (str): root domain for commons where mds lives lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections - """ + """ index = Gen3Index(commons_url) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False return await index.async_query_urls(pattern, _ssl=ssl)
@@ -557,7 +557,7 @@

Source code for gen3.tools.metadata.ingest_manifest

async def _get_with_params_from_indexd(params, commons_url, lock): - """ + """ Gets a semaphore then requests a record for the given params Args: @@ -565,12 +565,12 @@

Source code for gen3.tools.metadata.ingest_manifest

commons_url (str): root domain for commons where mds lives lock (asyncio.Semaphore): semaphones used to limit ammount of concurrent http connections - """ + """ index = Gen3Index(commons_url) async with lock: - # default ssl handling unless it's explicitly http:// + # default ssl handling unless it's explicitly http:// ssl = None - if "https" not in commons_url: + if "https" not in commons_url: ssl = False return await index.async_get_with_params(params, _ssl=ssl) diff --git a/docs/_build/html/_modules/gen3/wss.html b/docs/_build/html/_modules/gen3/wss.html index c39cfb4f..393b1936 100644 --- a/docs/_build/html/_modules/gen3/wss.html +++ b/docs/_build/html/_modules/gen3/wss.html @@ -49,32 +49,32 @@

Source code for gen3.wss

 
 
 def wsurl_to_tokens(ws_urlstr):
-    """Tokenize ws:/// paths - so ws:///@user/bla/foo returns ("@user", "bla/foo")"""
+    """Tokenize ws:/// paths - so ws:///@user/bla/foo returns ("@user", "bla/foo")"""
     urlparts = urlparse(ws_urlstr)
-    if urlparts.scheme != "ws":
-        raise Exception("invalid path {}".format(ws_urlstr))
-    pathparts = [part for part in urlparts.path.split("/") if part]
+    if urlparts.scheme != "ws":
+        raise Exception("invalid path {}".format(ws_urlstr))
+    pathparts = [part for part in urlparts.path.split("/") if part]
     if len(pathparts) < 1:
-        raise Exception("invalid path {}".format(ws_urlstr))
-    return (pathparts[0], "/".join(pathparts[1:]))
+        raise Exception("invalid path {}".format(ws_urlstr))
+    return (pathparts[0], "/".join(pathparts[1:]))
 
 
 @backoff.on_exception(backoff.expo, requests.HTTPError, **DEFAULT_BACKOFF_SETTINGS)
 def get_url(urlstr, dest_path):
-    """Simple url fetch to dest_path with backoff"""
+    """Simple url fetch to dest_path with backoff"""
     res = requests.get(urlstr)
     raise_for_status_and_print_error(res)
-    if dest_path == "-":
+    if dest_path == "-":
         sys.stdout.write(res.text)
     else:
-        with open(dest_path, "wb") as f:
+        with open(dest_path, "wb") as f:
             f.write(res.content)
 
 
 @backoff.on_exception(backoff.expo, requests.HTTPError, **DEFAULT_BACKOFF_SETTINGS)
 def put_url(urlstr, src_path):
-    """Simple put src_path to url with backoff"""
-    with open(src_path, "rb") as f:
+    """Simple put src_path to url with backoff"""
+    with open(src_path, "rb") as f:
         res = requests.put(urlstr, data=f)
     raise_for_status_and_print_error(res)
 
@@ -82,39 +82,39 @@ 

Source code for gen3.wss

 
[docs] class Gen3WsStorage: - """A class for interacting with the Gen3 workspace storage service. + """A class for interacting with the Gen3 workspace storage service. Examples: This generates the Gen3WsStorage class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page. - >>> auth = Gen3Auth(endpoint, refresh_file="credentials.json") + >>> auth = Gen3Auth(endpoint, refresh_file="credentials.json") ... wss = Gen3WsStorage(auth) - """ + """ def __init__(self, auth_provider=None): - """ + """ Initialization for instance of the class to setup basic endpoint info. Args: auth_provider (Gen3Auth, optional): Gen3Auth class to handle passing your token, required for admin endpoints - """ + """ self._auth_provider = auth_provider
[docs] @backoff.on_exception(backoff.expo, requests.HTTPError, **DEFAULT_BACKOFF_SETTINGS) def upload_url(self, ws, wskey): - """ + """ Get a upload url for the given workspace key Args: ws (string): name of the workspace wskey (string): key of the object in the workspace - """ - wskey = wskey.lstrip("/") - res = self._auth_provider.curl("/ws-storage/upload/{}/{}".format(ws, wskey)) + """ + wskey = wskey.lstrip("/") + res = self._auth_provider.curl("/ws-storage/upload/{}/{}".format(ws, wskey)) raise_for_status_and_print_error(res) return res.json()
@@ -123,10 +123,10 @@

Source code for gen3.wss

 [docs]
     @backoff.on_exception(backoff.expo, requests.HTTPError, **DEFAULT_BACKOFF_SETTINGS)
     def upload(self, src_path, dest_ws, dest_wskey):
-        """
+        """
         Upload a local file to the specified workspace path
-        """
-        url = self.upload_url(dest_ws, dest_wskey)["Data"]
+        """
+        url = self.upload_url(dest_ws, dest_wskey)["Data"]
         put_url(url, src_path)
@@ -134,15 +134,15 @@

Source code for gen3.wss

 [docs]
     @backoff.on_exception(backoff.expo, requests.HTTPError, **DEFAULT_BACKOFF_SETTINGS)
     def download_url(self, ws, wskey):
-        """
+        """
         Get a download url for the given workspace key
 
         Args:
           ws (string): name of the workspace
           wskey (string): key of the object in the workspace
-        """
-        wskey = wskey.lstrip("/")
-        res = self._auth_provider.curl("/ws-storage/download/{}/{}".format(ws, wskey))
+        """
+        wskey = wskey.lstrip("/")
+        res = self._auth_provider.curl("/ws-storage/download/{}/{}".format(ws, wskey))
         raise_for_status_and_print_error(res)
         return res.json()
@@ -150,51 +150,51 @@

Source code for gen3.wss

 
[docs] def download(self, src_ws, src_wskey, dest_path): - """ + """ Download a file from the workspace to local disk Args: src_ws (string): name of the workspace src_wskey (string): key of the object in the workspace dest_path (string): to download the file to - """ - durl = self.download_url(src_ws, src_wskey)["Data"] + """ + durl = self.download_url(src_ws, src_wskey)["Data"] get_url(durl, dest_path)
[docs] def copy(self, src_urlstr, dest_urlstr): - """ + """ Parse src_urlstr and dest_urlstr, and call upload or download as appropriate - """ - if src_urlstr[0:3] == "ws:": - if dest_urlstr[0:3] == "ws:": + """ + if src_urlstr[0:3] == "ws:": + if dest_urlstr[0:3] == "ws:": raise Exception( - "source and destination may not both reference a workspace" + "source and destination may not both reference a workspace" ) pathparts = wsurl_to_tokens(src_urlstr) return self.download(pathparts[0], pathparts[1], dest_urlstr) - if dest_urlstr[0:3] == "ws:": + if dest_urlstr[0:3] == "ws:": pathparts = wsurl_to_tokens(dest_urlstr) return self.upload(src_urlstr, pathparts[0], pathparts[1]) - raise Exception("source and destination may not both be local")
+ raise Exception("source and destination may not both be local")
[docs] @backoff.on_exception(backoff.expo, requests.HTTPError, **DEFAULT_BACKOFF_SETTINGS) def ls(self, ws, wskey): - """ + """ List the contents under the given workspace path Args: ws (string): name of the workspace wskey (string): key of the object in the workspace - """ - wskey = wskey.lstrip("/") - res = self._auth_provider.curl("/ws-storage/list/{}/{}".format(ws, wskey)) + """ + wskey = wskey.lstrip("/") + res = self._auth_provider.curl("/ws-storage/list/{}/{}".format(ws, wskey)) raise_for_status_and_print_error(res) return res.json()
@@ -202,14 +202,14 @@

Source code for gen3.wss

 
[docs] def ls_path(self, ws_urlstr): - """ + """ Same as ls - but parses ws_urlstr argument of form: ws:///workspace/key Args: ws (string): name of the workspace wskey (string): key of the object in the workspace - """ + """ pathparts = wsurl_to_tokens(ws_urlstr) return self.ls(pathparts[0], pathparts[1])
@@ -218,16 +218,16 @@

Source code for gen3.wss

 [docs]
     @backoff.on_exception(backoff.expo, requests.HTTPError, **DEFAULT_BACKOFF_SETTINGS)
     def rm(self, ws, wskey):
-        """
+        """
         Remove the given workspace key
 
         Args:
           ws (string): name of the workspace
           wskey (string): key of the object in the workspace
-        """
-        wskey = wskey.lstrip("/")
+        """
+        wskey = wskey.lstrip("/")
         res = self._auth_provider.curl(
-            "/ws-storage/list/{}/{}".format(ws, wskey), request="DELETE"
+            "/ws-storage/list/{}/{}".format(ws, wskey), request="DELETE"
         )
         raise_for_status_and_print_error(res)
         return res.json()
@@ -236,9 +236,9 @@

Source code for gen3.wss

 
[docs] def rm_path(self, ws_urlstr): - """ + """ Same as rm - but parses the ws_urlstr argument - """ + """ pathparts = wsurl_to_tokens(ws_urlstr) return self.rm(pathparts[0], pathparts[1])
diff --git a/docs/_build/html/auth.html b/docs/_build/html/auth.html index 4c24f392..ccccddc2 100644 --- a/docs/_build/html/auth.html +++ b/docs/_build/html/auth.html @@ -73,11 +73,11 @@

Gen3 Auth Helper
>>> auth = Gen3Auth(refresh_file="crdc")
+
>>> auth = Gen3Auth(refresh_file="crdc")
 

or use some arbitrary file:

-
diff --git a/docs/_build/html/file.html b/docs/_build/html/file.html index cb970d07..11b4ace1 100644 --- a/docs/_build/html/file.html +++ b/docs/_build/html/file.html @@ -50,7 +50,7 @@

Gen3 File ClassExamples

This generates the Gen3File class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(refresh_file="credentials.json")
+
>>> auth = Gen3Auth(refresh_file="credentials.json")
 ... file = Gen3File(auth)
 
diff --git a/docs/_build/html/indexing.html b/docs/_build/html/indexing.html index 4448e460..3b08eb29 100644 --- a/docs/_build/html/indexing.html +++ b/docs/_build/html/indexing.html @@ -51,7 +51,7 @@

Gen3 Index ClassExamples

This generates the Gen3Index class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(refresh_file="credentials.json")
+
>>> auth = Gen3Auth(refresh_file="credentials.json")
 ... index = Gen3Index(auth)
 
diff --git a/docs/_build/html/jobs.html b/docs/_build/html/jobs.html index ef4d7609..2184a792 100644 --- a/docs/_build/html/jobs.html +++ b/docs/_build/html/jobs.html @@ -43,7 +43,7 @@

Gen3 Jobs ClassExamples

This generates the Gen3Jobs class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(refresh_file="credentials.json")
+
>>> auth = Gen3Auth(refresh_file="credentials.json")
 ... jobs = Gen3Jobs(auth)
 
diff --git a/docs/_build/html/metadata.html b/docs/_build/html/metadata.html index 182b95cb..868e4e43 100644 --- a/docs/_build/html/metadata.html +++ b/docs/_build/html/metadata.html @@ -43,7 +43,7 @@

Gen3 Metadata ClassExamples

This generates the Gen3Metadata class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(refresh_file="credentials.json")
+
>>> auth = Gen3Auth(refresh_file="credentials.json")
 ... metadata = Gen3Metadata(auth)
 
diff --git a/docs/_build/html/object.html b/docs/_build/html/object.html index cc648045..1bbada43 100644 --- a/docs/_build/html/object.html +++ b/docs/_build/html/object.html @@ -50,7 +50,7 @@

Gen3 Object ClassExamples

This generates the Gen3Object class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(refresh_file="credentials.json")
+
>>> auth = Gen3Auth(refresh_file="credentials.json")
 ... object = Gen3Object(auth)
 
diff --git a/docs/_build/html/query.html b/docs/_build/html/query.html index 94d535d1..9d60a0b4 100644 --- a/docs/_build/html/query.html +++ b/docs/_build/html/query.html @@ -48,7 +48,7 @@

Gen3 Query ClassExamples

This generates the Gen3Query class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(endpoint, refresh_file="credentials.json")
+
>>> auth = Gen3Auth(endpoint, refresh_file="credentials.json")
 ... query = Gen3Query(auth)
 
@@ -71,7 +71,7 @@

Gen3 Query ClassExamples

-
>>> query_string = "{ my_index { my_field } }"
+
>>> query_string = "{ my_index { my_field } }"
 ... Gen3Query.graphql_query(query_string)
 
@@ -103,14 +103,14 @@

Gen3 Query ClassExamples

>>> Gen3Query.query(
-    data_type="subject",
+    data_type="subject",
     first=50,
     fields=[
-        "vital_status",
-        "submitter_id",
+        "vital_status",
+        "submitter_id",
     ],
-    filters={"vital_status": "Alive"},
-    sort_object={"submitter_id": "asc"},
+    filters={"vital_status": "Alive"},
+    sort_object={"submitter_id": "asc"},
 )
 
@@ -141,15 +141,15 @@

Gen3 Query ClassExamples

>>> Gen3Query.raw_data_download(
-        data_type="subject",
+        data_type="subject",
         fields=[
-            "vital_status",
-            "submitter_id",
-            "project_id"
+            "vital_status",
+            "submitter_id",
+            "project_id"
         ],
-        filter_object={"=": {"project_id": "my_program-my_project"}},
-        sort_fields=[{"submitter_id": "asc"}],
-        accessibility="accessible"
+        filter_object={"=": {"project_id": "my_program-my_project"}},
+        sort_fields=[{"submitter_id": "asc"}],
+        accessibility="accessible"
     )
 
diff --git a/docs/_build/html/submission.html b/docs/_build/html/submission.html index b9c4f829..485ffbcd 100644 --- a/docs/_build/html/submission.html +++ b/docs/_build/html/submission.html @@ -51,7 +51,7 @@

Gen3 Submission ClassExamples

This generates the Gen3Submission class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(refresh_file="credentials.json")
+
>>> auth = Gen3Auth(refresh_file="credentials.json")
 ... sub = Gen3Submission(auth)
 
@@ -78,7 +78,7 @@

Gen3 Submission Class

Examples

This creates a project on the DCF program in the sandbox commons.

-
>>> Gen3Submission.create_project("DCF", json)
+
>>> Gen3Submission.create_project("DCF", json)
 
@@ -104,7 +104,7 @@

Gen3 Submission Class

Examples

This deletes a node from the CCLE project in the sandbox commons.

-
>>> Gen3Submission.delete_node("DCF", "CCLE", "demographic")
+
>>> Gen3Submission.delete_node("DCF", "CCLE", "demographic")
 
@@ -130,7 +130,7 @@

Gen3 Submission Class

Examples

This deletes a list of nodes from the CCLE project in the sandbox commons.

-
>>> Gen3Submission.delete_nodes("DCF", "CCLE", ["demographic", "subject", "experiment"])
+
>>> Gen3Submission.delete_nodes("DCF", "CCLE", ["demographic", "subject", "experiment"])
 
@@ -147,7 +147,7 @@

Gen3 Submission Class

Examples

This deletes the “DCF” program.

-
>>> Gen3Submission.delete_program("DCF")
+
>>> Gen3Submission.delete_program("DCF")
 
@@ -167,7 +167,7 @@

Gen3 Submission Class

Examples

This deletes the “CCLE” project from the “DCF” program.

-
>>> Gen3Submission.delete_project("DCF", "CCLE")
+
>>> Gen3Submission.delete_project("DCF", "CCLE")
 
@@ -187,7 +187,7 @@

Gen3 Submission Class

Examples

This deletes a record from the CCLE project in the sandbox commons.

-
>>> Gen3Submission.delete_record("DCF", "CCLE", uuid)
+
>>> Gen3Submission.delete_record("DCF", "CCLE", uuid)
 
@@ -210,7 +210,7 @@

Gen3 Submission Class

Examples

This deletes a list of records from the CCLE project in the sandbox commons.

-
>>> Gen3Submission.delete_records("DCF", "CCLE", ["uuid1", "uuid2"])
+
>>> Gen3Submission.delete_records("DCF", "CCLE", ["uuid1", "uuid2"])
 
@@ -232,7 +232,7 @@

Gen3 Submission Class

Examples

This exports all records in the “sample” node from the CCLE project in the sandbox commons.

-
>>> Gen3Submission.export_node("DCF", "CCLE", "sample", "tsv", filename="DCF-CCLE_sample_node.tsv")
+
>>> Gen3Submission.export_node("DCF", "CCLE", "sample", "tsv", filename="DCF-CCLE_sample_node.tsv")
 
@@ -254,7 +254,7 @@

Gen3 Submission Class

Examples

This exports a single record from the sandbox commons.

-
>>> Gen3Submission.export_record("DCF", "CCLE", "d70b41b9-6f90-4714-8420-e043ab8b77b9", "json", filename="DCF-CCLE_one_record.json")
+
>>> Gen3Submission.export_record("DCF", "CCLE", "d70b41b9-6f90-4714-8420-e043ab8b77b9", "json", filename="DCF-CCLE_one_record.json")
 
@@ -283,7 +283,7 @@

Gen3 Submission Class

Examples

This returns the dictionary schema the “subject” node.

-
>>> Gen3Submission.get_dictionary_node("subject")
+
>>> Gen3Submission.get_dictionary_node("subject")
 
@@ -319,7 +319,7 @@

Gen3 Submission Class

Example

-
>>> Gen3Submission.get_project_dictionary("DCF", "CCLE")
+
>>> Gen3Submission.get_project_dictionary("DCF", "CCLE")
 
@@ -337,7 +337,7 @@

Gen3 Submission Class

Example

-
>>> Gen3Submission.get_project_manifest("DCF", "CCLE")
+
>>> Gen3Submission.get_project_manifest("DCF", "CCLE")
 
@@ -353,7 +353,7 @@

Gen3 Submission Class

Example

This lists all the projects under the DCF program

-
>>> Gen3Submission.get_projects("DCF")
+
>>> Gen3Submission.get_projects("DCF")
 
@@ -371,7 +371,7 @@

Gen3 Submission Class

Example

-
>>> Gen3Submission.get_project_manifest("DCF", "CCLE")
+
>>> Gen3Submission.get_project_manifest("DCF", "CCLE")
 
@@ -392,7 +392,7 @@

Gen3 Submission ClassExamples

This executes a query to get the list of all the project codes for all the projects in the Data Commons.

-
>>> query = "{ project(first:0) { code } }"
+
>>> query = "{ project(first:0) { code } }"
 ... Gen3Submission.query(query)
 
@@ -414,7 +414,7 @@

Gen3 Submission Class

Examples

This submits a spreadsheet file containing multiple records in rows to the CCLE project in the sandbox commons.

-
>>> Gen3Submission.submit_file("DCF-CCLE","data_spreadsheet.tsv")
+
>>> Gen3Submission.submit_file("DCF-CCLE","data_spreadsheet.tsv")
 
@@ -434,7 +434,7 @@

Gen3 Submission Class

Examples

This submits records to the CCLE project in the sandbox commons.

-
>>> Gen3Submission.submit_record("DCF", "CCLE", json)
+
>>> Gen3Submission.submit_record("DCF", "CCLE", json)
 
diff --git a/docs/_build/html/tools/drs_pull.html b/docs/_build/html/tools/drs_pull.html index 84cb5793..1c5e3288 100644 --- a/docs/_build/html/tools/drs_pull.html +++ b/docs/_build/html/tools/drs_pull.html @@ -41,12 +41,12 @@
Examples:

This generates the Gen3Jobs class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> datafiles = Manifest.load('sample/manifest_1.json')
-    downloadManager = DownloadManager("source.my_commons.org",
-                      Gen3Auth(refresh_file="~.gen3/my_credentials.json"), datafiles)
+
>>> datafiles = Manifest.load('sample/manifest_1.json')
+    downloadManager = DownloadManager("source.my_commons.org",
+                      Gen3Auth(refresh_file="~.gen3/my_credentials.json"), datafiles)
     for i in datafiles:
         print(i)
-    downloadManager.download(datafiles, ".")
+    downloadManager.download(datafiles, ".")
 

See docs/howto/drsDownloading.md for more details

diff --git a/docs/_build/html/wss.html b/docs/_build/html/wss.html index d373f5e2..2e2d1759 100644 --- a/docs/_build/html/wss.html +++ b/docs/_build/html/wss.html @@ -42,7 +42,7 @@

Gen3 Workspace StorageExamples

This generates the Gen3WsStorage class pointed at the sandbox commons while using the credentials.json downloaded from the commons profile page.

-
>>> auth = Gen3Auth(endpoint, refresh_file="credentials.json")
+
>>> auth = Gen3Auth(endpoint, refresh_file="credentials.json")
 ... wss = Gen3WsStorage(auth)
 
From 6720febb30278fc0db7fd4c7cbe5cf28789ced74 Mon Sep 17 00:00:00 2001 From: Mingfei Shao Date: Tue, 18 Aug 2026 11:41:55 -0500 Subject: [PATCH 3/3] update version --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index 17982f82..85368130 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,7 +1,7 @@ [tool.poetry] name = "gen3" homepage = "https://gen3.org/" -version = "4.28.2" +version = "4.28.3" description = "Gen3 CLI and Python SDK" authors = ["Center for Translational Data Science at the University of Chicago "] license = "Apache-2.0"