diff --git a/scripts/storage_image.py b/scripts/storage_image.py new file mode 100644 index 0000000..02e7283 --- /dev/null +++ b/scripts/storage_image.py @@ -0,0 +1,53 @@ +import subprocess +import os +import pandas as pd +import concurrent.futures + +def download_image(bucket_url, directory_path, destination_folder, destination_filename): + """Download images from a specific directory in Google Cloud Storage if they don't exist locally.""" + os.makedirs(destination_folder, exist_ok=True) + + # List images in the GCS directory + gsutil_ls_command = f"gsutil ls '{bucket_url}/{directory_path}/*.jpg'" + result = subprocess.run(gsutil_ls_command, shell=True, capture_output=True, text=True) + + # Extract image filenames from the gsutil ls command output + image_filenames = result.stdout.strip().split('\n') + + for image_filename in image_filenames: + image_filename = os.path.basename(image_filename) + # local_path = os.path.join(destination_folder, image_filename) + local_path = os.path.join(destination_folder, destination_filename) + + # Check if the image already exists locally + if not os.path.exists(local_path): + # Download the image only if it doesn't exist locally + gsutil_cp_command = f"gsutil cp '{bucket_url}/{directory_path}/{image_filename}' '{local_path}'" + subprocess.run(gsutil_cp_command, shell=True, check=True) + # print(f"Downloaded: {local_path}") + # else: + # print(f"Skipped (Already Exists): {local_path}") + + # print(f"All images downloaded to: {destination_folder}") + +def user_image(sample_id): + bucket_url = "gs://hpos-prod.appspot.com" + directory_path = f"{sample_id}" + destination_folder = "data/images" + destination_filename = f"{sample_id}.jpg" + + local_path = os.path.join(destination_folder, destination_filename) + + if not os.path.exists(local_path): + download_image(bucket_url, directory_path, destination_folder, destination_filename) + + +if __name__ == "__main__": + curdir = os.getcwd() + path_delim = '/' + df = pd.read_excel(curdir + path_delim + "data/CardPrint.xlsx", sheet_name="op") + + # Using ThreadPoolExecutor for parallel processing + with concurrent.futures.ThreadPoolExecutor() as executor: + # Map the user_image function to the list of sample IDs, allowing parallel execution + executor.map(user_image, df['Sample ID']) \ No newline at end of file