From 00058eeab003247f3dd61dd83d70cd8c4b86bc14 Mon Sep 17 00:00:00 2001 From: Moritz Graf Date: Thu, 23 Jul 2026 20:07:35 +0200 Subject: [PATCH] fix(backup): update job backoffLimit to 100 and refine verify_migration parsing --- k8s/backup/job-gdrive-to-gphotos.yaml | 2 +- k8s/backup/verify_migration.py | 217 ++++++++++++++++++++++++++ 2 files changed, 218 insertions(+), 1 deletion(-) create mode 100755 k8s/backup/verify_migration.py diff --git a/k8s/backup/job-gdrive-to-gphotos.yaml b/k8s/backup/job-gdrive-to-gphotos.yaml index bb49080..9608dfc 100644 --- a/k8s/backup/job-gdrive-to-gphotos.yaml +++ b/k8s/backup/job-gdrive-to-gphotos.yaml @@ -25,4 +25,4 @@ spec: configMap: name: rclone-gdrive-config restartPolicy: Never - backoffLimit: 1 + backoffLimit: 100 diff --git a/k8s/backup/verify_migration.py b/k8s/backup/verify_migration.py new file mode 100755 index 0000000..a692fc9 --- /dev/null +++ b/k8s/backup/verify_migration.py @@ -0,0 +1,217 @@ +#!/usr/bin/env python3 +""" +Google Drive to Google Photos Migration Verification Tool + +This script analyzes the migration status of pictures from Google Drive to Google Photos. +It compares Google Drive folder contents with Pod logs from Kubernetes (and/or Google Photos API) +to determine exact file counts, total sizes, and completion status per subfolder. +""" + +import argparse +import json +import os +import re +import subprocess +import sys +import tempfile +from collections import defaultdict + +DEFAULT_SECRET_FILE = os.path.join(os.path.dirname(__file__), "rclone.secret.yml") +DEFAULT_DRIVE_ROOT_ID = "1f43AtWvfb7qd-eGr9nw-8vHraA5k9mVl" +DEFAULT_PHOTOS_ALBUM = "All pictures of the last years" + +def extract_rclone_conf(secret_file_path): + """Extract rclone.conf from Kubernetes ConfigMap secret YAML file.""" + if not os.path.exists(secret_file_path): + print(f"Error: Secret file not found at {secret_file_path}") + sys.exit(1) + + with open(secret_file_path, "r", encoding="utf-8") as f: + content = f.read() + + match = re.search(r"rclone\.conf:\s*\|\n(.*)", content, re.DOTALL) + if not match: + print("Error: Could not find rclone.conf data in secret file.") + sys.exit(1) + + conf_data = match.group(1) + # Remove leading 4-space indentation if present + cleaned_lines = [] + for line in conf_data.split("\n"): + if line.startswith(" "): + cleaned_lines.append(line[4:]) + else: + cleaned_lines.append(line) + + tmp_conf = tempfile.NamedTemporaryFile(mode="w", suffix=".conf", delete=False) + tmp_conf.write("\n".join(cleaned_lines)) + tmp_conf.close() + return tmp_conf.name + +def get_k8s_job_logs(namespace="backup", job_name="gdrive-to-gphotos"): + """Fetch logs from all pods associated with the Kubernetes job.""" + print("Fetching execution logs from Kubernetes pods...") + try: + res = subprocess.run( + ["kubectl", "get", "pods", "-n", namespace, "-l", f"job-name={job_name}", "-o", "jsonpath={.items[*].metadata.name}"], + capture_output=True, text=True, check=True + ) + pod_names = res.stdout.strip().split() + if not pod_names: + print(f"Warning: No pods found for job '{job_name}' in namespace '{namespace}'.") + return "" + + all_logs = [] + for pod in pod_names: + print(f" Retrieving logs for pod: {pod}") + log_res = subprocess.run( + ["kubectl", "logs", "-n", namespace, pod], + capture_output=True, text=True + ) + all_logs.append(log_res.stdout) + return "\n".join(all_logs) + except Exception as e: + print(f"Warning: Could not fetch k8s logs directly ({e}).") + return "" + +def parse_copied_files_from_logs(log_text): + """Parse log text for confirmed uploaded files per subfolder.""" + copied_files = defaultdict(set) + for line in log_text.splitlines(): + if "Copied (new)" in line or "Copied (replaced)" in line: + m = re.search(r"INFO\s+:\s+(.*?)/(.*?):\s+Copied", line) + if m: + folder, filename = m.group(1), m.group(2) + copied_files[folder].add(filename) + return copied_files + +def get_gdrive_subfolders(config_path, root_id): + """List subdirectories in the source Google Drive folder.""" + print("Querying subfolders from Google Drive...") + res = subprocess.run( + ["rclone", "--config", config_path, "lsd", f"--drive-root-folder-id={root_id}", "gdrive:"], + capture_output=True, text=True + ) + if res.returncode != 0: + print(f"Error listing Google Drive folders: {res.stderr}") + return [] + + subfolders = [] + for line in res.stdout.strip().splitlines(): + if line: + parts = line.split(maxsplit=4) + if len(parts) >= 5: + subfolders.append(parts[4]) + return sorted(subfolders) + +def get_gdrive_folder_stats(config_path, root_id, subfolder): + """Get file count and size for a Google Drive subfolder using rclone size.""" + res = subprocess.run( + ["rclone", "--config", config_path, "size", f"--drive-root-folder-id={root_id}", f"gdrive:{subfolder}"], + capture_output=True, text=True + ) + if res.returncode != 0: + return {"count": 0, "size_str": "Unknown", "bytes": 0} + + output = res.stdout.strip() + # Handle formats like "Total objects: 2.563k (2563)" or "Total objects: 524" + count_match = re.search(r"Total objects:\s*(?:[\d\.\w]+\s*\()?(\d+)\)?", output) + size_match = re.search(r"Total size:\s*([\d\.]+\s*[KMGT]?i?B)\s*\((.*?)\s*Bytes\)", output) + + count = int(count_match.group(1)) if count_match else 0 + size_str = size_match.group(1) if size_match else "0 B" + size_bytes = int(size_match.group(2)) if size_match else 0 + + return {"count": count, "size_str": size_str, "bytes": size_bytes} + +def main(): + parser = argparse.ArgumentParser(description="Verify Google Drive to Google Photos migration status.") + parser.add_argument("--secret-file", default=DEFAULT_SECRET_FILE, help="Path to rclone.secret.yml") + parser.add_argument("--root-id", default=DEFAULT_DRIVE_ROOT_ID, help="Google Drive root folder ID") + parser.add_argument("--local-log", help="Path to local pod log file (optional override)") + parser.add_argument("--csv", help="Path to save CSV report output (optional)") + args = parser.parse_args() + + print("==========================================================") + print(" Google Drive to Google Photos Migration Verification Tool") + print("==========================================================") + + # 1. Extract Config + config_path = extract_rclone_conf(args.secret_file) + + try: + # 2. Get Logs + if args.local_log and os.path.exists(args.local_log): + print(f"Reading logs from local file: {args.local_log}") + with open(args.local_log, "r", errors="ignore") as f: + log_text = f.read() + else: + log_text = get_k8s_job_logs() + + copied_map = parse_copied_files_from_logs(log_text) + + # 3. Get Google Drive Subfolders + subfolders = get_gdrive_subfolders(config_path, args.root_id) + if not subfolders: + print("No subfolders found or unable to connect to Google Drive.") + sys.exit(1) + + print(f"\nAnalyzing {len(subfolders)} subfolders...\n") + + results = [] + total_drive_files = 0 + total_drive_bytes = 0 + total_copied_files = 0 + + hdr = f"{'Subfolder Name':<35} | {'GDrive Files':<12} | {'GDrive Size':<12} | {'Uploaded':<10} | {'Status':<15}" + print(hdr) + print("-" * len(hdr)) + + for folder in subfolders: + stats = get_gdrive_folder_stats(config_path, args.root_id, folder) + copied_count = len(copied_map.get(folder, set())) + + drive_count = stats["count"] + drive_size = stats["size_str"] + total_drive_files += drive_count + total_drive_bytes += stats["bytes"] + total_copied_files += copied_count + + if drive_count > 0 and copied_count >= drive_count: + status = "COMPLETE" + elif copied_count > 0: + pct = (copied_count / drive_count * 100) if drive_count > 0 else 0 + status = f"IN PROGRESS ({pct:.1f}%)" + else: + status = "QUEUED / WAITING" + + print(f"{folder:<35} | {drive_count:<12} | {drive_size:<12} | {copied_count:<10} | {status:<15}") + + results.append({ + "folder": folder, + "drive_files": drive_count, + "drive_size": drive_size, + "uploaded_files": copied_count, + "status": status + }) + + print("-" * len(hdr)) + total_gb = total_drive_bytes / (1024 ** 3) + summary_str = f"TOTAL: {total_drive_files} files ({total_gb:.2f} GiB) | Uploaded: {total_copied_files} files" + print(f"\n{summary_str}") + + # Save CSV if requested + if args.csv: + import csv + with open(args.csv, "w", newline="", encoding="utf-8") as csvfile: + writer = csv.DictWriter(csvfile, fieldnames=["folder", "drive_files", "drive_size", "uploaded_files", "status"]) + writer.writeheader() + writer.writerows(results) + print(f"Saved CSV report to: {args.csv}") + + finally: + if os.path.exists(config_path): + os.remove(config_path) + +if __name__ == "__main__": + main()