import boto3 import pandas as pd # AWS credentials aws_access_key_id = "YOUR_S3_ACCESS_KEY" aws_secret_access_key = "YOUR_SECRET_KEY" region_name = "ap-south-1" # Excel output output_file = "s3bucket_size.xlsx" # Connect to S3 s3 = boto3.client( "s3", aws_access_key_id=aws_access_key_id, aws_secret_access_key=aws_secret_access_key, region_name=region_name ) # Get all S3 buckets response = s3.list_buckets() buckets = response.get("Buckets", []) print(f"Found {len(buckets)} S3 buckets.\n") report = [] # Process each bucket for bucket in buckets: bucket_name = bucket["Name"] print(f"Scanning: {bucket_name}") file_count = 0 total_size = 0 try: paginator = s3.get_paginator("list_objects_v2") for page in paginator.paginate(Bucket=bucket_name): for obj in page.get("Contents", []): key = obj["Key"] # Ignore folder placeholders if key.endswith("/"): continue file_count += 1 total_size += obj["Size"] total_size_gb = total_size / (1024 ** 3) total_size_tb = total_size / (1024 ** 4) report.append({ "Bucket": bucket_name, "File Count": file_count, "Total Size (Bytes)": total_size, "Total Size (GB)": round(total_size_gb, 2), "Total Size (TB)": round(total_size_tb, 4) }) print( f" Files: {file_count:,} | " f"Size: {total_size_gb:.2f} GB" ) except Exception as e: print(f" ERROR: {e}") report.append({ "Bucket": bucket_name, "File Count": "ERROR", "Total Size (Bytes)": "ERROR", "Total Size (GB)": "ERROR", "Total Size (TB)": "ERROR" }) # Create Excel report df = pd.DataFrame(report) df = df.sort_values(by="Bucket") df.to_excel( output_file, index=False ) print("\n===================================") print("S3 BUCKET REPORT COMPLETED") print("===================================") print(f"Report saved to: {output_file}")