flake8 reported 49 findings and pyright 10, and two of them were real bugs rather than matters of style. The rest were unused imports and four functions over the complexity limit. - Unlock a DOI folder with the id the DOI response carries, rather than an attribute the model does not have, which raised `AttributeError` on every locked DOI folder. - Size the `ls` Size column from the sub-folder on the row rather than from its parent, which pushed the later columns out of line. - Share the path resolution and the destination checks between `cp` and `mv`, and split the recursive download and the `ls` row printing, bringing all four functions under the complexity limit. - Accept a client built without a connection, which `config` and `version` rely on, and report the reason if one is then asked for. - Declare the config protocol's constructor for the type checker alone, so the protocol keeps its guard against being instantiated. - Remove 41 unused imports, and let flake8 accept black's spacing. - Point pyright at the project's own environment, without which it resolved no dependency and reported 124 findings that were not real. - Cover the two fixes and the shared `cp`/`mv` paths with new tests.
322 lines
14 KiB
Python
322 lines
14 KiB
Python
import os
|
|
from concurrent.futures import ThreadPoolExecutor
|
|
from typing import Any
|
|
from unicodedata import normalize
|
|
|
|
from pydantic.dataclasses import dataclass
|
|
|
|
from mdrsclient.api import FilesApi, FoldersApi
|
|
from mdrsclient.exceptions import IllegalArgumentException, UnexpectedException
|
|
from mdrsclient.models import File, Folder, Laboratory
|
|
from mdrsclient.models.file import find_file
|
|
from mdrsclient.settings import CONCURRENT
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class UploadFileInfo:
|
|
folder: Folder
|
|
files: list[File]
|
|
path: str
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class DownloadFileInfo:
|
|
file: File
|
|
path: str
|
|
|
|
|
|
@dataclass
|
|
class DownloadContext:
|
|
isSkipIfExists: bool
|
|
files: list[DownloadFileInfo]
|
|
|
|
|
|
class Uploader:
|
|
def __init__(self, client: Any) -> None:
|
|
self.client = client
|
|
|
|
def upload(
|
|
self, local_path: str, remote_path: str, is_recursive: bool = False, is_skip_if_exists: bool = False
|
|
) -> None:
|
|
remote, laboratory_name, r_path = self.client.parse_remote_host_with_path(remote_path)
|
|
l_path = os.path.abspath(local_path)
|
|
if not os.path.exists(l_path):
|
|
raise IllegalArgumentException(f"File or directory `{local_path}` not found.")
|
|
|
|
laboratory = self.client.find_laboratory(laboratory_name)
|
|
folder = self.client.find_folder(laboratory, r_path)
|
|
files = self.client.find_files(folder.id)
|
|
if os.path.isdir(l_path):
|
|
if not is_recursive:
|
|
raise IllegalArgumentException(f"Cannot upload `{local_path}`: Is a directory.")
|
|
infos = self.__collect_directory_uploads(laboratory, r_path, l_path, folder, files)
|
|
else:
|
|
infos = [UploadFileInfo(folder, files, l_path)]
|
|
if not self.__multiple_upload(infos, is_skip_if_exists):
|
|
# One file failing is worth reporting on its own line, and worth the caller
|
|
# hearing about: a batch that lost files is not a batch that succeeded.
|
|
raise UnexpectedException("Some files failed to upload.")
|
|
|
|
def __collect_directory_uploads(
|
|
self, laboratory: Laboratory, r_path: str, l_path: str, folder: Folder, files: list[File]
|
|
) -> list[UploadFileInfo]:
|
|
"""Mirror a local directory tree on the remote, and list the files to send into it."""
|
|
infos: list[UploadFileInfo] = []
|
|
folder_api = FoldersApi(self.client.connection)
|
|
folder_map: dict[str, Folder] = {}
|
|
folder_map[r_path] = folder
|
|
files_map: dict[str, list[File]] = {}
|
|
files_map[r_path] = files
|
|
l_basename = os.path.basename(l_path)
|
|
for dirpath, _, filenames in os.walk(l_path, followlinks=True):
|
|
sub = l_basename if dirpath == l_path else os.path.join(l_basename, os.path.relpath(dirpath, l_path))
|
|
d_dirname = os.path.join(r_path, sub)
|
|
d_basename = os.path.basename(d_dirname)
|
|
# prepare destination parent path
|
|
d_parent_dirname = os.path.dirname(d_dirname)
|
|
if folder_map.get(d_parent_dirname) is None:
|
|
parent_folder = self.client.find_folder(laboratory, d_parent_dirname)
|
|
folder_map[d_parent_dirname] = parent_folder
|
|
parent_files = self.client.find_files(parent_folder.id)
|
|
files_map[d_parent_dirname] = parent_files
|
|
# prepare destination path
|
|
if folder_map.get(d_dirname) is None:
|
|
d_folder = folder_map[d_parent_dirname].find_sub_folder(d_basename)
|
|
if d_folder is None:
|
|
d_folder_id = folder_api.create(normalize("NFC", d_basename), folder_map[d_parent_dirname].id)
|
|
else:
|
|
d_folder_id = d_folder.id
|
|
print(d_dirname)
|
|
folder_map[d_dirname] = folder_api.retrieve(d_folder_id)
|
|
files_map[d_dirname] = self.client.find_files(d_folder_id)
|
|
if d_folder is None:
|
|
folder_map[d_parent_dirname].sub_folders.append(folder_map[d_dirname])
|
|
# register upload file list
|
|
for filename in filenames:
|
|
infos.append(
|
|
UploadFileInfo(folder_map[d_dirname], files_map[d_dirname], os.path.join(dirpath, filename))
|
|
)
|
|
return infos
|
|
|
|
def __multiple_upload(self, infos: list[UploadFileInfo], is_skip_if_exists: bool) -> bool:
|
|
"""Send every file, and report whether all of them arrived."""
|
|
file_api = FilesApi(self.client.connection)
|
|
with ThreadPoolExecutor(max_workers=CONCURRENT) as pool:
|
|
results = pool.map(lambda x: self.__multiple_upload_worker(file_api, x, is_skip_if_exists), infos)
|
|
# Consumed inside the block: the results are what carry each worker's verdict.
|
|
return all(list(results))
|
|
|
|
def __multiple_upload_worker(self, file_api: FilesApi, info: UploadFileInfo, is_skip_if_exists: bool) -> bool:
|
|
basename = os.path.basename(info.path)
|
|
file = find_file(info.files, basename)
|
|
try:
|
|
if file is None:
|
|
file_api.create(info.folder.id, info.path)
|
|
elif not is_skip_if_exists or file.size != os.path.getsize(info.path):
|
|
file_api.update(file, info.path)
|
|
print(os.path.join(info.folder.path, basename))
|
|
except Exception as e:
|
|
# Everything, not just the exceptions the API layer raises: the batch verdict
|
|
# is read now that the results are consumed, and a file vanishing between the
|
|
# walk and the upload would otherwise end the whole run with a traceback and
|
|
# throw away what every other file did.
|
|
print(f"Failed: {info.path}: {e}")
|
|
return False
|
|
return True
|
|
|
|
|
|
class Downloader:
|
|
def __init__(self, client: Any) -> None:
|
|
self.client = client
|
|
|
|
def download(
|
|
self,
|
|
remote_path: str,
|
|
local_path: str,
|
|
is_recursive: bool = False,
|
|
is_skip_if_exists: bool = False,
|
|
password: str | None = None,
|
|
excludes: list[str] | None = None,
|
|
) -> None:
|
|
if not self.__download(remote_path, local_path, is_recursive, is_skip_if_exists, password, excludes):
|
|
# Every failure has already been printed against the file it belongs to.
|
|
# This is what makes the command as a whole end in failure.
|
|
raise UnexpectedException("Some files failed to download.")
|
|
|
|
def __download(
|
|
self,
|
|
remote_path: str,
|
|
local_path: str,
|
|
is_recursive: bool,
|
|
is_skip_if_exists: bool,
|
|
password: str | None,
|
|
excludes: list[str] | None,
|
|
) -> bool:
|
|
"""Fetch what the remote path names, and report whether every file arrived."""
|
|
excludes_clean = excludes or []
|
|
l_dirname = os.path.realpath(local_path)
|
|
if not os.path.isdir(l_dirname):
|
|
raise IllegalArgumentException(f"Local directory `{local_path}` not found.")
|
|
|
|
# "remote:10.xxxx/prefix.ID[/optional/sub/path]" names a published dataset rather
|
|
# than a path within a laboratory, and is resolved through the DOI instead.
|
|
path_component = remote_path.split(":", 1)[1] if ":" in remote_path else ""
|
|
if self.client.is_doi(path_component):
|
|
return self.__download_doi(
|
|
remote_path, l_dirname, is_recursive, is_skip_if_exists, password, excludes_clean
|
|
)
|
|
|
|
remote, laboratory_name, r_path = self.client.parse_remote_host_with_path(remote_path)
|
|
r_path = r_path.rstrip("/")
|
|
r_basename = os.path.basename(r_path)
|
|
laboratory = self.client.find_laboratory(laboratory_name)
|
|
r_parent_folder = self.client.find_folder(laboratory, os.path.dirname(r_path), password)
|
|
r_parent_files = self.client.find_files(r_parent_folder.id)
|
|
file = find_file(r_parent_files, r_basename)
|
|
if file is not None:
|
|
return self.__download_one(
|
|
excludes_clean, laboratory, r_parent_folder, file, l_dirname, r_basename, is_skip_if_exists
|
|
)
|
|
folder = r_parent_folder.find_sub_folder(r_basename)
|
|
if folder is None:
|
|
raise IllegalArgumentException(f"File or folder `{r_path}` not found.")
|
|
if not is_recursive:
|
|
raise IllegalArgumentException(f"Cannot download `{r_path}`: Is a folder.")
|
|
return self.__multiple_download_pickup_recursive_files(
|
|
FoldersApi(self.client.connection), laboratory, folder.id, l_dirname, excludes_clean, is_skip_if_exists
|
|
)
|
|
|
|
def __download_doi(
|
|
self,
|
|
remote_path: str,
|
|
l_dirname: str,
|
|
is_recursive: bool,
|
|
is_skip_if_exists: bool,
|
|
password: str | None,
|
|
excludes: list[str],
|
|
) -> bool:
|
|
"""Fetch what a DOI names: the dataset's folder, or something inside it."""
|
|
remote, doi, subpath = self.client.parse_doi_remote_host(remote_path)
|
|
doi_folder, laboratory = self.client.find_folder_by_doi(doi, password)
|
|
|
|
subpath_clean = subpath.rstrip("/")
|
|
if not subpath_clean:
|
|
folder = doi_folder
|
|
else:
|
|
r_basename = os.path.basename(subpath_clean)
|
|
abs_path = doi_folder.path.rstrip("/") + os.path.dirname(subpath_clean)
|
|
r_parent_folder = self.client.find_folder(laboratory, abs_path, password)
|
|
file = find_file(self.client.find_files(r_parent_folder.id), r_basename)
|
|
if file is not None:
|
|
return self.__download_one(
|
|
excludes, laboratory, r_parent_folder, file, l_dirname, r_basename, is_skip_if_exists
|
|
)
|
|
folder_simple = r_parent_folder.find_sub_folder(r_basename)
|
|
if folder_simple is None:
|
|
raise IllegalArgumentException(f"File or folder `{subpath_clean}` not found.")
|
|
folder = FoldersApi(self.client.connection).retrieve(folder_simple.id)
|
|
|
|
if is_recursive:
|
|
return self.__multiple_download_pickup_recursive_files(
|
|
FoldersApi(self.client.connection), laboratory, folder.id, l_dirname, excludes, is_skip_if_exists
|
|
)
|
|
# Without -r the dataset's own files are fetched, and its sub-folders are not.
|
|
context = DownloadContext(is_skip_if_exists, [])
|
|
for file in self.client.find_files(folder.id):
|
|
if self.__check_excludes(excludes, laboratory, folder, file):
|
|
continue
|
|
context.files.append(DownloadFileInfo(file, os.path.join(l_dirname, file.name)))
|
|
return self.__multiple_download(context)
|
|
|
|
def __download_one(
|
|
self,
|
|
excludes: list[str],
|
|
laboratory: Laboratory,
|
|
folder: Folder,
|
|
file: File,
|
|
l_dirname: str,
|
|
local_name: str,
|
|
is_skip_if_exists: bool,
|
|
) -> bool:
|
|
"""
|
|
Fetch a single named file into the local directory.
|
|
|
|
Saved under the name the caller asked for rather than the one the server holds:
|
|
the two are matched case-insensitively, so they need not be spelled alike.
|
|
"""
|
|
if self.__check_excludes(excludes, laboratory, folder, file):
|
|
return True
|
|
context = DownloadContext(is_skip_if_exists, [])
|
|
context.files.append(DownloadFileInfo(file, os.path.join(l_dirname, local_name)))
|
|
return self.__multiple_download(context)
|
|
|
|
def __multiple_download_pickup_recursive_files(
|
|
self,
|
|
folder_api: FoldersApi,
|
|
laboratory: Laboratory,
|
|
folder_id: str,
|
|
basedir: str,
|
|
excludes: list[str],
|
|
is_skip_if_exists: bool,
|
|
) -> bool:
|
|
context = DownloadContext(is_skip_if_exists, [])
|
|
try:
|
|
folder = folder_api.retrieve(folder_id)
|
|
files = self.client.find_files(folder.id)
|
|
except Exception as e:
|
|
print(f"Failed: {basedir}: {e}")
|
|
return False
|
|
dirname = os.path.join(basedir, folder.name)
|
|
if self.__check_excludes(excludes, laboratory, folder, None):
|
|
return True
|
|
try:
|
|
# `exist_ok` rather than a prior check: two workers can reach the same parent.
|
|
os.makedirs(dirname, exist_ok=True)
|
|
except OSError as e:
|
|
# One folder the client cannot make locally is not a reason to abandon its
|
|
# siblings, which is what this walk now promises.
|
|
print(f"Failed: {dirname}: {e}")
|
|
return False
|
|
print(dirname)
|
|
for file in files:
|
|
if self.__check_excludes(excludes, laboratory, folder, file):
|
|
continue
|
|
path = os.path.join(dirname, file.name)
|
|
context.files.append(DownloadFileInfo(file, path))
|
|
succeeded = self.__multiple_download(context)
|
|
# A folder that lost a file is still a folder whose sub-folders the user asked
|
|
# for, so the walk carries on and the verdict is collected for the caller.
|
|
for sub_folder in folder.sub_folders:
|
|
if not self.__multiple_download_pickup_recursive_files(
|
|
folder_api, laboratory, sub_folder.id, dirname, excludes, is_skip_if_exists
|
|
):
|
|
succeeded = False
|
|
return succeeded
|
|
|
|
def __multiple_download(self, context: DownloadContext) -> bool:
|
|
"""Fetch every file in the batch, and report whether all of them arrived."""
|
|
file_api = FilesApi(self.client.connection)
|
|
with ThreadPoolExecutor(max_workers=CONCURRENT) as pool:
|
|
results = pool.map(
|
|
lambda x: self.__multiple_download_worker(file_api, x, context.isSkipIfExists), context.files
|
|
)
|
|
# Consumed inside the block, and in full: every worker's verdict counts, not
|
|
# just the first refusal.
|
|
return all(list(results))
|
|
|
|
def __multiple_download_worker(self, file_api: FilesApi, info: DownloadFileInfo, is_skip_if_exists: bool) -> bool:
|
|
if not is_skip_if_exists or not os.path.exists(info.path) or info.file.size != os.path.getsize(info.path):
|
|
try:
|
|
file_api.download(info.file, info.path)
|
|
except Exception as e:
|
|
# Nothing to clear up: a failed transfer writes only to its own scratch
|
|
# file beside the destination, and removes that itself.
|
|
print(f"Failed: {info.path}: {e}")
|
|
return False
|
|
print(info.path)
|
|
return True
|
|
|
|
def __check_excludes(self, excludes: list[str], laboratory: Laboratory, folder: Folder, file: File | None) -> bool:
|
|
path = f"/{laboratory.name}{folder.path}{file.name if file is not None else ''}".rstrip("/").lower()
|
|
return path in excludes
|