Merge pull request #58 from SELab-3-2026/feat/wandb_download_project
feat: added script to download all runs in a given project
This commit is contained in:
commit
7040764107
4 changed files with 175 additions and 21 deletions
|
|
@ -182,7 +182,7 @@ def plot_grouped_bar(
|
||||||
color="w",
|
color="w",
|
||||||
markerfacecolor=BEST_PERFORMER_COLOR,
|
markerfacecolor=BEST_PERFORMER_COLOR,
|
||||||
markersize=15,
|
markersize=15,
|
||||||
label="Best Performance",
|
# label="Best Performance",
|
||||||
ls="",
|
ls="",
|
||||||
)
|
)
|
||||||
ax.legend(**LEGEND_KWARGS, ncol=len(architectures) + 1)
|
ax.legend(**LEGEND_KWARGS, ncol=len(architectures) + 1)
|
||||||
|
|
@ -317,7 +317,7 @@ def plot_grouped_bar_alt(
|
||||||
color="w",
|
color="w",
|
||||||
markerfacecolor=BEST_PERFORMER_COLOR,
|
markerfacecolor=BEST_PERFORMER_COLOR,
|
||||||
markersize=15,
|
markersize=15,
|
||||||
label="Best Performance",
|
# label="Best Performance",
|
||||||
ls="",
|
ls="",
|
||||||
)
|
)
|
||||||
ax.legend(**LEGEND_KWARGS, ncol=len(morphologies) + 1)
|
ax.legend(**LEGEND_KWARGS, ncol=len(morphologies) + 1)
|
||||||
|
|
|
||||||
|
|
@ -45,19 +45,22 @@ class Columns(str, Enum):
|
||||||
"""Column names expected in every evaluation CSV."""
|
"""Column names expected in every evaluation CSV."""
|
||||||
|
|
||||||
ARCH = "architecture"
|
ARCH = "architecture"
|
||||||
TIMESTEPS = "total_trained_timesteps"
|
TIMESTEPS = "trained_timesteps"
|
||||||
REWARD = "accumulated_reward"
|
REWARD = "eval_return"
|
||||||
VELOCITY = "velocity"
|
VELOCITY = "velocity"
|
||||||
|
EVAL_STEPS = "eval_steps"
|
||||||
|
FINAL_XY_DIST = "final_xy_dist"
|
||||||
|
INITIAL_XY_DIST = "initial_xy_dist"
|
||||||
|
REACHED_TARGET = "reached_target"
|
||||||
|
|
||||||
|
|
||||||
# Maps architecture display names to the path of their evaluation CSV.
|
# Maps architecture display names to the path of their evaluation CSV.
|
||||||
# Update these paths once real evaluation data is available.
|
# Update these paths once real evaluation data is available.
|
||||||
FILE_MAPPING: dict[str, str] = {
|
FILE_MAPPING: dict[str, str] = {
|
||||||
"centralized 2 arms": "runs/dummy/dummy_centralized_2_arms.csv",
|
# "centralized 2 arms": "runs/dummy/dummy_centralized_2_arms.csv",
|
||||||
"centralized 5 arms": "runs/dummy/dummy_centralized_5_arms.csv",
|
"centralized 5 arms": "runs/final-v2-centralized/checkpoint_evaluation.csv",
|
||||||
"decentralized fully connected": "runs/dummy/dummy_decentralized_fully_connected.csv",
|
"decentralized fully connected": "runs/final-v2-fully-conn/checkpoint_evaluation.csv",
|
||||||
"decentralized ring-level": "runs/dummy/dummy_decentralized_ring-level.csv",
|
"decentralized ring-level": "runs/final-v2-ring/checkpoint_evaluation.csv",
|
||||||
"decentralized segment-level": "runs/dummy/dummy_decentralized_segment-level.csv",
|
|
||||||
}
|
}
|
||||||
|
|
||||||
# Architecture profiles for dummy data generation: (max_reward, max_velocity, sigmoid_speed)
|
# Architecture profiles for dummy data generation: (max_reward, max_velocity, sigmoid_speed)
|
||||||
|
|
@ -108,7 +111,13 @@ def load_metrics(file_mapping: dict[str, str]) -> pd.DataFrame:
|
||||||
Loads one CSV per architecture, injects the architecture name as a column,
|
Loads one CSV per architecture, injects the architecture name as a column,
|
||||||
and returns the combined DataFrame with only the required columns.
|
and returns the combined DataFrame with only the required columns.
|
||||||
"""
|
"""
|
||||||
required = [Columns.TIMESTEPS, Columns.REWARD, Columns.VELOCITY]
|
required = [
|
||||||
|
Columns.TIMESTEPS,
|
||||||
|
Columns.REWARD,
|
||||||
|
Columns.INITIAL_XY_DIST,
|
||||||
|
Columns.FINAL_XY_DIST,
|
||||||
|
Columns.EVAL_STEPS,
|
||||||
|
]
|
||||||
dfs = []
|
dfs = []
|
||||||
|
|
||||||
for arch_name, filepath in file_mapping.items():
|
for arch_name, filepath in file_mapping.items():
|
||||||
|
|
@ -125,6 +134,10 @@ def load_metrics(file_mapping: dict[str, str]) -> pd.DataFrame:
|
||||||
|
|
||||||
df = df[required].copy()
|
df = df[required].copy()
|
||||||
df[Columns.ARCH] = arch_name
|
df[Columns.ARCH] = arch_name
|
||||||
|
df[Columns.VELOCITY] = (df[Columns.INITIAL_XY_DIST] - df[Columns.FINAL_XY_DIST]) / df[
|
||||||
|
Columns.EVAL_STEPS
|
||||||
|
]
|
||||||
|
|
||||||
dfs.append(df)
|
dfs.append(df)
|
||||||
|
|
||||||
return pd.concat(dfs, ignore_index=True) if dfs else pd.DataFrame()
|
return pd.concat(dfs, ignore_index=True) if dfs else pd.DataFrame()
|
||||||
|
|
@ -303,6 +316,7 @@ def plot_results(df: pd.DataFrame, results: pd.DataFrame, output_dir: str, **kwa
|
||||||
def obtain_data() -> pd.DataFrame:
|
def obtain_data() -> pd.DataFrame:
|
||||||
"""Resolves the file mapping, falling back to generated dummy CSVs if needed."""
|
"""Resolves the file mapping, falling back to generated dummy CSVs if needed."""
|
||||||
global USING_DUMMY_DATA
|
global USING_DUMMY_DATA
|
||||||
|
|
||||||
if not any(os.path.exists(p) for p in FILE_MAPPING.values()):
|
if not any(os.path.exists(p) for p in FILE_MAPPING.values()):
|
||||||
logger.info("No real evaluation files found. Generating dummy CSVs at expected locations.")
|
logger.info("No real evaluation files found. Generating dummy CSVs at expected locations.")
|
||||||
generate_dummy_csvs(FILE_MAPPING)
|
generate_dummy_csvs(FILE_MAPPING)
|
||||||
|
|
|
||||||
|
|
@ -4,26 +4,24 @@ import matplotlib.pyplot as plt
|
||||||
# Shared Color Palette (Colorblind friendly, high contrast)
|
# Shared Color Palette (Colorblind friendly, high contrast)
|
||||||
# Matches poster design
|
# Matches poster design
|
||||||
COLORS = {
|
COLORS = {
|
||||||
"CENTRALIZED": "#2B4162", # Deep Slate Blue
|
"CENTRALIZED": "#0D567C", # Blue
|
||||||
"FULLY_CONNECTED": "#FA9F42", # Vibrant Orange
|
"FULLY_CONNECTED": "#8C0E0F", # Reddish
|
||||||
"RING_LEVEL": "#4E937A", # Muted Teal
|
"RING_LEVEL": "#E1BA6D", # Pale Yellow
|
||||||
"SEGMENT_LEVEL": "#B4436C", # Soft Red
|
|
||||||
"DECENTRALIZED": "#4E937A", # Default decentralized fallback
|
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
def apply_style(font_size=28):
|
def apply_style(font_size=36):
|
||||||
"""
|
"""
|
||||||
Applies the shared typography and aesthetic settings to Matplotlib.
|
Applies the shared typography and aesthetic settings to Matplotlib.
|
||||||
"""
|
"""
|
||||||
plt.rcParams.update(
|
plt.rcParams.update(
|
||||||
{
|
{
|
||||||
"font.size": font_size,
|
"font.size": font_size,
|
||||||
"axes.labelsize": font_size + 4,
|
"axes.labelsize": font_size,
|
||||||
"axes.titlesize": font_size + 8,
|
"axes.titlesize": font_size,
|
||||||
"xtick.labelsize": font_size - 4,
|
"xtick.labelsize": font_size,
|
||||||
"ytick.labelsize": font_size - 4,
|
"ytick.labelsize": font_size,
|
||||||
"legend.fontsize": font_size - 6,
|
"legend.fontsize": font_size,
|
||||||
"axes.linewidth": 2,
|
"axes.linewidth": 2,
|
||||||
"axes.spines.top": False,
|
"axes.spines.top": False,
|
||||||
"axes.spines.right": False,
|
"axes.spines.right": False,
|
||||||
|
|
|
||||||
142
scripts/tools/download_wandb_project.py
Normal file
142
scripts/tools/download_wandb_project.py
Normal file
|
|
@ -0,0 +1,142 @@
|
||||||
|
from pathlib import Path
|
||||||
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
|
import threading
|
||||||
|
import wandb
|
||||||
|
import argparse
|
||||||
|
|
||||||
|
# tune these depending on network / W&B limits
|
||||||
|
MAX_RUN_WORKERS = 8
|
||||||
|
MAX_FILE_WORKERS = 16
|
||||||
|
MAX_ARTIFACT_WORKERS = 8
|
||||||
|
|
||||||
|
api = wandb.Api()
|
||||||
|
|
||||||
|
print_lock = threading.Lock()
|
||||||
|
|
||||||
|
|
||||||
|
def safe_print(*args, **kwargs):
|
||||||
|
with print_lock:
|
||||||
|
print(*args, **kwargs)
|
||||||
|
|
||||||
|
|
||||||
|
def download_file(file, run_dir):
|
||||||
|
target = run_dir / file.name
|
||||||
|
|
||||||
|
try:
|
||||||
|
# skip existing files
|
||||||
|
if target.exists():
|
||||||
|
return f"SKIP FILE {target}"
|
||||||
|
|
||||||
|
target.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
file.download(root=run_dir, replace=False)
|
||||||
|
|
||||||
|
return f"DONE FILE {target}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
return f"FAIL FILE {target}: {e}"
|
||||||
|
|
||||||
|
|
||||||
|
def sanitize_artifact_name(name: str):
|
||||||
|
return name.replace(":", "_")
|
||||||
|
|
||||||
|
|
||||||
|
def download_artifact(artifact, artifact_root):
|
||||||
|
try:
|
||||||
|
artifact_name = sanitize_artifact_name(artifact.name)
|
||||||
|
artifact_dir = artifact_root / artifact_name
|
||||||
|
|
||||||
|
if artifact_dir.exists() and any(artifact_dir.iterdir()):
|
||||||
|
return f"SKIP ARTIFACT {artifact.name}"
|
||||||
|
|
||||||
|
artifact_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
artifact.download(root=artifact_dir)
|
||||||
|
|
||||||
|
return f"DONE ARTIFACT {artifact.name}"
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
return f"FAIL ARTIFACT {artifact.name}: {e}"
|
||||||
|
|
||||||
|
|
||||||
|
def download_run(run, root):
|
||||||
|
run_dir = root / f"{run.name}"
|
||||||
|
run_dir.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
safe_print(f"\n=== {run.name} ({run.id}) ===")
|
||||||
|
|
||||||
|
# -------------------------
|
||||||
|
# Download regular run files
|
||||||
|
# -------------------------
|
||||||
|
files = list(run.files())
|
||||||
|
|
||||||
|
with ThreadPoolExecutor(max_workers=MAX_FILE_WORKERS) as executor:
|
||||||
|
futures = [executor.submit(download_file, file, run_dir) for file in files]
|
||||||
|
|
||||||
|
for future in as_completed(futures):
|
||||||
|
safe_print(future.result())
|
||||||
|
|
||||||
|
# -------------------------
|
||||||
|
# Download logged artifacts
|
||||||
|
# -------------------------
|
||||||
|
artifact_root = run_dir / "artifacts"
|
||||||
|
|
||||||
|
try:
|
||||||
|
artifacts = list(run.logged_artifacts())
|
||||||
|
safe_print(f"Found {len(artifacts)} artifacts for {run.name}")
|
||||||
|
|
||||||
|
with ThreadPoolExecutor(max_workers=MAX_ARTIFACT_WORKERS) as executor:
|
||||||
|
futures = [
|
||||||
|
executor.submit(download_artifact, artifact, artifact_root)
|
||||||
|
for artifact in artifacts
|
||||||
|
]
|
||||||
|
|
||||||
|
for future in as_completed(futures):
|
||||||
|
safe_print(future.result())
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
safe_print(f"Artifact download failed for {run.name}: {e}")
|
||||||
|
|
||||||
|
# -------------------------
|
||||||
|
# OPTIONAL: download used/input artifacts
|
||||||
|
# -------------------------
|
||||||
|
# try:
|
||||||
|
# used_artifacts = list(run.used_artifacts())
|
||||||
|
# used_root = run_dir / "used_artifacts"
|
||||||
|
#
|
||||||
|
# for artifact in used_artifacts:
|
||||||
|
# download_artifact(artifact, used_root)
|
||||||
|
# except Exception as e:
|
||||||
|
# safe_print(f"Used artifact download failed: {e}")
|
||||||
|
|
||||||
|
safe_print(f"Finished {run.name}")
|
||||||
|
|
||||||
|
|
||||||
|
def main(entity: str, project: str, root: Path):
|
||||||
|
root.mkdir(exist_ok=True)
|
||||||
|
|
||||||
|
runs = list(api.runs(f"{entity}/{project}"))
|
||||||
|
|
||||||
|
safe_print(f"Found {len(runs)} runs")
|
||||||
|
|
||||||
|
with ThreadPoolExecutor(max_workers=MAX_RUN_WORKERS) as executor:
|
||||||
|
futures = [executor.submit(download_run, run, root) for run in runs]
|
||||||
|
|
||||||
|
for future in as_completed(futures):
|
||||||
|
try:
|
||||||
|
future.result()
|
||||||
|
except Exception as e:
|
||||||
|
safe_print("RUN FAILED:", e)
|
||||||
|
|
||||||
|
safe_print("\nAll downloads complete.")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--entity", type=str, default="SEL3-2026-Groep-4")
|
||||||
|
parser.add_argument("--project", type=str, required=True)
|
||||||
|
parser.add_argument("--root", type=str, default="runs")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
root = Path(args.root)
|
||||||
|
main(entity=args.entity, project=args.project, root=root)
|
||||||
Reference in a new issue