From bef91020c84373f9beb6f18c9602945c38c44d61 Mon Sep 17 00:00:00 2001 From: Mariya Varghese Date: Wed, 16 Aug 2023 09:58:54 +0000 Subject: [PATCH] Add new file for downloading images and data, --- DownloadImagesandData | 181 ++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 181 insertions(+) create mode 100644 DownloadImagesandData diff --git a/DownloadImagesandData b/DownloadImagesandData new file mode 100644 index 0000000..f868bc0 --- /dev/null +++ b/DownloadImagesandData @@ -0,0 +1,181 @@ +import os +import firebase_admin +from firebase_admin import credentials, firestore +import pandas as pd +from urllib.request import urlopen +import datetime +import tkinter as tk +from tkinter import filedialog +import openpyxl +import xlsxwriter + +# Initialize Firebase Admin SDK +file_path = os.getcwd() + "\hpos-prod-firebase-adminsdk-bionp-5486f7becd.json" +cred = credentials.Certificate(fr'{file_path}') # Replace with your own service account key path + # Replace with your own service account key path +firebase_admin.initialize_app(cred) + +# Get a reference to the Firestore database +db = firestore.client() + +# Specify the collections +patient_collection = db.collection("patientData") +test_collection = db.collection("testData") + +# Create a tkinter root window (hidden) +root = tk.Tk() +root.withdraw() + +# Ask the user to select the base folder where data will be saved +base_output_folder = filedialog.askdirectory(title="Select Base Output Folder") + +# Get user-defined date ranges +start_date = "2023-08-01" # input("Please enter the start date (yyyy-mm-dd): ") +end_date = "2023-08-02" # input("Please enter the end date (yyyy-mm-dd): ") + +# Parse the date range to get a list of days +start_datetime = datetime.datetime.strptime(start_date, "%Y-%m-%d") +end_datetime = datetime.datetime.strptime(end_date, "%Y-%m-%d") +date_range = [start_datetime + datetime.timedelta(days=x) for x in range((end_datetime - start_datetime).days + 1)] + +# Iterate through each day +for date in date_range: + current_date = date.strftime("%Y-%m-%d") + formatted_date = date.strftime("%d%b%Y") # Format as "10Aug2023" + output_folder = os.path.join(base_output_folder, formatted_date) + os.makedirs(output_folder, exist_ok=True) + + # Create a subfolder for images + images_folder = os.path.join(output_folder, "images") + os.makedirs(images_folder, exist_ok=True) + + print(f"Processing data for {current_date}...") + + # Query Firestore for patient data + query = patient_collection.where("createdAt", ">=", current_date).where("createdAt", "<", (date + datetime.timedelta(days=1)).strftime("%Y-%m-%d")) + patient_docs = query.stream() + + # Initialize a list to hold the combined data + data = [] + + # Iterate through patient documents + for patient_doc in patient_docs: + patient_data = patient_doc.to_dict() + patient_id = patient_data["_id"] + + # Query the document from testData collection based on the common _id + test_docs = test_collection.where("_id", "==", patient_id).stream() + + # Iterate through test documents + for test_doc in test_docs: + test_data = test_doc.to_dict() + + # Combine the data from both collections into a single dictionary + combined_data = {**patient_data, **test_data} + + # Access specific attributes from the combined data + csv_path = combined_data.get("csvPath", "") + if csv_path and not csv_path.startswith("http"): + # Perform the desired action + print("CSV path does not start with 'http'.") + + # Append the combined_data dictionary to the list + data.append(combined_data) + + # Download images separately and store them in the images folder + user_image_url = combined_data.get("userImageURL", "") + if user_image_url and user_image_url.startswith("http"): + image_data = urlopen(user_image_url).read() + image_filename = f"{patient_id}.jpg" + image_path = os.path.join(images_folder, image_filename) + with open(image_path, "wb") as image_file: + image_file.write(image_data) + + # Add the image URL to the DataFrame + combined_data["userImageUrl"] = os.path.relpath(image_path, output_folder) + + # Convert the data to a DataFrame + df = pd.DataFrame(data) + + # Rename columns and select specific columns + df.rename(columns={ + "name": "Name", + "abhaId": "ABHA ID", + "birthYear": "Age", + "gender": "Gender", + "category": "Category", + "maritalStatus": "Marital Status", + "careOf": "Care-of", + "city": "Address", + "district": "District", + "state": "State", + "pinCode": "Pincode", + "phoneNumber": "Mobile Number", + "bloodGroup": "Blood group", + "result": "Test Result", + "userImageUrl": "Image URL" # Rename userImageUrl column + }, inplace=True) + + selected_columns = [ + "Name", "ABHA ID", "Age", "Gender", "Category", "Marital Status", + "Care-of", "Address", "District", "State", "Pincode", "Mobile Number", + "Blood group", "Test Result", "Image URL" + ] + + df = df[selected_columns] + + # Save the DataFrame to a CSV file + csv_output_filename = f"data_{current_date}.csv" + csv_output_path = os.path.join(output_folder, csv_output_filename) + # df.to_csv(csv_output_path, index=False) + + print(f"CSV file created: '{csv_output_path}'") + + # Save the DataFrame to an Excel file with hyperlinks + excel_output_filename = f"data_{current_date}.xlsx" + excel_output_path = os.path.join(output_folder, excel_output_filename) + + + # Create an Excel writer using XlsxWriter + excel_writer = pd.ExcelWriter(excel_output_path, engine='xlsxwriter') + + df.to_excel(excel_writer, sheet_name='Data', index=False) + + workbook = excel_writer.book + worksheet = excel_writer.sheets['Data'] + + for idx, row in df.iterrows(): + image_url = f"external:{row['Image URL']}" + worksheet.write_url(idx + 1, df.columns.get_loc('Image URL'), image_url, string=row['Image URL']) + + # Lock the worksheet with a password + worksheet.protect('your_password_here') + + # Adjust column width to fit the longest value in each column + for i, col in enumerate(df.columns): + max_len = max(df[col].astype(str).apply(len).max(), len(col)) # max length in column + worksheet.set_column(i, i, max_len) # set column width + + # Close the Excel writer + excel_writer._save() + + print(f"Excel file with hyperlinks and locked sheet created: '{excel_output_path}'") + + +# ... (previous code) + +# Unlock the Excel sheet using openpyxl +with pd.ExcelWriter(excel_output_path, engine='openpyxl') as writer: + writer.book = openpyxl.load_workbook(excel_output_path) + writer.sheets = dict((ws.title, ws) for ws in writer.book.worksheets) + sheet = writer.sheets['Data'] + sheet.protection.sheet = False # Disable protection + + # Remove the password from the worksheet protection + sheet.protection.password = None + +print(f"Excel sheet unlocked: '{excel_output_path}'") + +# Close the Firebase Admin SDK +firebase_admin.delete_app(firebase_admin.get_app()) +