mirror of
https://github.com/NVIDIA/Model-Optimizer.git
synced 2026-10-02 03:14:52 +08:00
``` # Install cd Model-Optimizer/launcher curl -LsSf https://astral.sh/uv/install.sh | sh git submodule update --init --recursive # Run locally with Docker (single GPU) uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml hf_local=/mnt/hf-local --yes # Run on Slurm cluster (no need to export the follow SLURM_XXX envs if used in sandbox) export SLURM_HOST=login-node.example.com export SLURM_ACCOUNT=my_account export SLURM_HF_LOCAL=/shared/hf-local export SLURM_JOB_DIR=/shared/experiments uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --yes # Preview config without running uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --dryrun --yes -v # Override parameters uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml \ pipeline.task_0.slurm_config.nodes=2 --yes # Dump resolved config for reproducibility (single YAML for reproducibility, great for QA, Eng, and agent to triage) uv run launch.py --yaml Qwen/Qwen3-8B/megatron_lm_ptq.yaml --to-yaml resolved.yaml # Run tests uv pip install -e . pytest uv run pytest -v ``` ## Summary Add `launcher/` module for submitting quantization, training, and evaluation jobs to Slurm clusters or running them locally with Docker via `nemo-run`. `nemo-run` is used in all `NVIDIA-NeMo/*` projects. It supports modern YAML factory (superset of the `OmegaConf` and `Hydra`) and it support multiple executor backends (here we use docker and slurm mainly). A sample YAML config `launcher/Qwen/Qwen3-8B/megatron_lm_ptq.yaml`: ``` job_name: Qwen3-8B_NVFP4_DEFAULT_CFG pipeline: # hf_local: path prefix for model weights and datasets. # # This should be a self-managed directory that mirrors the HuggingFace Hub # hierarchy (e.g., /hf-local/Qwen/Qwen3-8B/, /hf-local/cais/mmlu/). Using # a dedicated folder is preferred over the HuggingFace cache (~/.cache/huggingface) # to avoid cache corruption issues with concurrent jobs. # # Override on CLI: # pipeline.global_vars.hf_local=/mnt/my-models/ # use a different path # pipeline.global_vars.hf_local="" # download from HuggingFace Hub global_vars: hf_local: /hf-local/ task_0: script: common/megatron-lm/quantize/quantize.sh args: - --calib-dataset-path-or-name <<global_vars.hf_local>>abisee/cnn_dailymail - --calib-size 32 environment: - MLM_MODEL_CFG: Qwen/Qwen3-8B - QUANT_CFG: NVFP4_DEFAULT_CFG - HF_MODEL_CKPT: <<global_vars.hf_local>>Qwen/Qwen3-8B - MMLU_DATASET: <<global_vars.hf_local>>cais/mmlu - TP: 4 slurm_config: _factory_: "slurm_factory" nodes: 1 ntasks_per_node: 4 gpus_per_node: 4 ``` ### Key features - **`launch.py`** — public entrypoint accepting `--yaml` config format - **`core.py`** — shared logic (dataclasses, executor builders, run loop) also used by nmm-sandbox's `slurm.py` - **Factory system** — env-var-driven `slurm_factory` with `register_factory()` registry - **`<<global_vars.X>>`** interpolation for sharing values across pipeline tasks - **`hf_local`** global var for configurable model/dataset storage path - **Version reporting** — git commit/branch printed at job start for reproducibility - **`--to-yaml`** — dump resolved config for bug reports and reproducibility - **Model-Optimizer symlink** — `modules/Model-Optimizer -> ../..` (auto-created, avoids recursive submodule) ### Files | Path | Description | |------|-------------| | `launcher/launch.py` | Public entrypoint | | `launcher/core.py` | Shared dataclasses, executors, run loop | | `launcher/slurm_config.py` | SlurmConfig + env-var factory | | `launcher/common/` | Shell scripts (quantize, query, eagle3, specdec_bench) | | `launcher/Qwen/Qwen3-8B/` | Example configs (PTQ, EAGLE3 pipeline) | | `launcher/tests/` | 64 unit tests | | `launcher/README.md` | User guide | | `launcher/ADVANCED.md` | Architecture, mount mechanism, Claude Code workflows | | `launcher/CLAUDE.md` | Claude Code project instructions | | `.github/workflows/unit_tests.yml` | CI job for launcher tests | ### Verified - Same YAML produces identical MMLU results via both `slurm.py` and `launch.py`: - Local Docker (TP=1): 0.719 (128/178) - OCI-HSG Slurm (TP=4): 0.730 (130/178) ## Test plan - [x] 64 unit tests (core, factory, YAML, Docker executor, Slurm executor, Docker launch) - [x] CI workflow added to `.github/workflows/unit_tests.yml` - [x] Local Docker end-to-end with `python:3.12-slim` - [x] Qwen3-8B PTQ on OCI-HSG via both launchers - [ ] Reviewer runs: `cd launcher && uv pip install -e . pytest && uv run pytest -v` ### Before your PR is "*Ready for review*" Make sure you read and follow [Contributor guidelines](https://github.com/NVIDIA/Model-Optimizer/blob/main/CONTRIBUTING.md) and your commits are signed (`git commit -s -S`). Make sure you read and follow the [Security Best Practices](https://github.com/NVIDIA/Model-Optimizer/blob/main/SECURITY.md#security-coding-practices-for-contributors) (e.g. avoiding hardcoded `trust_remote_code=True`, `torch.load(..., weights_only=False)`, `pickle`, etc.). - Is this change backward compatible?: ✅ / ❌ / N/A <!--- If ❌, explain why. --> - If you copied code from any other sources or added a new PIP dependency, did you follow guidance in `CONTRIBUTING.md`: ✅ / ❌ / N/A <!--- Mandatory --> - Did you write any new necessary tests?: ✅ / ❌ / N/A <!--- Mandatory for new features or examples. --> - Did you update [Changelog](https://github.com/NVIDIA/Model-Optimizer/blob/main/CHANGELOG.rst)?: ✅ / ❌ / N/A <!--- Only for new features, API changes, critical bug fixes or backward incompatible changes. --> ### Additional Information <!-- E.g. related issue. --> <!-- This is an auto-generated comment: release notes by coderabbit.ai --> ## Summary by CodeRabbit ## Release Notes * **New Features** * Introduced ModelOpt Launcher for submitting quantization, training, and evaluation jobs to Slurm clusters or running locally via Docker. * Added YAML-based job configuration with multi-task pipeline support and global variable interpolation. * Included example workflows for Qwen3-8B quantization and EAGLE3 speculative decoding. * Provided configurable Slurm and execution environment defaults. * **Documentation** * Added comprehensive README with quick start, environment setup, and configuration guidance. * Added advanced guide detailing launcher architecture and integration patterns. <!-- end of auto-generated comment: release notes by coderabbit.ai --> --------- Signed-off-by: Chenhan Yu <chenhany@nvidia.com> Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
486 lines
15 KiB
Python
486 lines
15 KiB
Python
# SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
|
# SPDX-License-Identifier: Apache-2.0
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
"""Core logic for the ModelOpt Launcher.
|
|
|
|
Dataclasses, executor builders, and the job run loop used by launch.py.
|
|
"""
|
|
|
|
import dataclasses
|
|
import getpass
|
|
import json
|
|
import os
|
|
import re
|
|
from dataclasses import dataclass
|
|
|
|
import nemo_run as run
|
|
import yaml
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Default environment variables injected into every job
|
|
# ---------------------------------------------------------------------------
|
|
|
|
DEFAULT_EXPERIMENT_TITLE = "cicd"
|
|
|
|
|
|
def get_default_env(experiment_title=None):
|
|
"""Return (slurm_env, local_env) dicts for the given experiment title."""
|
|
title = experiment_title or DEFAULT_EXPERIMENT_TITLE
|
|
slurm_env = {
|
|
"TRITON_CACHE_DIR": f"/{title}/triton-cache",
|
|
"HF_HOME": f"/{title}/hf-cache",
|
|
"HF_TOKEN": os.getenv("HF_TOKEN", ""),
|
|
"MLM_SKIP_INSTALL": "1",
|
|
"LAUNCH_SCRIPT": "python",
|
|
}
|
|
local_env = {
|
|
"TRITON_CACHE_DIR": f"/{title}/triton-cache",
|
|
"HF_HOME": f"/{title}/hf-cache",
|
|
"HF_TOKEN": os.getenv("HF_TOKEN", ""),
|
|
"MLM_SKIP_INSTALL": "1",
|
|
}
|
|
return slurm_env, local_env
|
|
|
|
|
|
# SlurmConfig type — set by the caller via set_slurm_config_type() before use.
|
|
# This allows both slurm.py and launch.py to use their own SlurmConfig class.
|
|
_SLURM_CONFIG_TYPE = None
|
|
_FACTORY_REGISTRY = {}
|
|
|
|
|
|
def set_slurm_config_type(cls):
|
|
"""Register the SlurmConfig dataclass type used by SandboxTask."""
|
|
global _SLURM_CONFIG_TYPE
|
|
_SLURM_CONFIG_TYPE = cls
|
|
# Patch SandboxTask's type annotation so nemo-run's CLI parser can resolve factories
|
|
SandboxTask.__dataclass_fields__["slurm_config"].type = cls
|
|
SandboxTask.__annotations__["slurm_config"] = cls
|
|
|
|
|
|
def register_factory(name, fn):
|
|
"""Register a factory function by name for task_configs YAML resolution."""
|
|
_FACTORY_REGISTRY[name] = fn
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Task and pipeline dataclasses
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
@dataclass
|
|
class SandboxTask:
|
|
"""A single task with a script, slurm config, args, and environment."""
|
|
|
|
script: str = None
|
|
slurm_config: object = None # Patched at runtime by set_slurm_config_type()
|
|
args: list[str] = None
|
|
environment: list[dict[str, str]] = None
|
|
yaml_file: str = None
|
|
skip: bool = False
|
|
|
|
|
|
@dataclass
|
|
class SandboxTask0(SandboxTask):
|
|
"""Task slot 0 in a pipeline."""
|
|
|
|
|
|
@dataclass
|
|
class SandboxTask1(SandboxTask):
|
|
"""Task slot 1 in a pipeline."""
|
|
|
|
|
|
@dataclass
|
|
class SandboxTask2(SandboxTask):
|
|
"""Task slot 2 in a pipeline."""
|
|
|
|
|
|
@dataclass
|
|
class SandboxTask3(SandboxTask):
|
|
"""Task slot 3 in a pipeline."""
|
|
|
|
|
|
@dataclass
|
|
class SandboxTask4(SandboxTask):
|
|
"""Task slot 4 in a pipeline."""
|
|
|
|
|
|
def create_task_from_yaml(yaml_file, factory_lookup):
|
|
"""Create a SandboxTask from a YAML config file.
|
|
|
|
Args:
|
|
yaml_file: Path to the YAML config.
|
|
factory_lookup: Dict mapping factory names to callable factory functions.
|
|
"""
|
|
with open(yaml_file) as file:
|
|
config_from_yaml = yaml.safe_load(file)
|
|
|
|
script = config_from_yaml["script"]
|
|
function_name = config_from_yaml["slurm_config"].pop("_factory_")
|
|
slurm_config = factory_lookup[function_name](**config_from_yaml["slurm_config"])
|
|
args = config_from_yaml.get("args", None)
|
|
environment = config_from_yaml.get("environment", None)
|
|
|
|
return SandboxTask(script=script, slurm_config=slurm_config, args=args, environment=environment)
|
|
|
|
|
|
@dataclass
|
|
class GlobalVariables:
|
|
"""Shared variables for <<global_vars.X>> interpolation in pipeline YAMLs."""
|
|
|
|
hf_model: str = None
|
|
hf_data: str = None
|
|
hf_local: str = None
|
|
|
|
|
|
@dataclass
|
|
class SandboxPipeline:
|
|
"""A multi-task pipeline with shared global variables and task dependencies."""
|
|
|
|
global_vars: GlobalVariables = None
|
|
|
|
task_0: SandboxTask0 = None
|
|
task_1: SandboxTask1 = None
|
|
task_2: SandboxTask2 = None
|
|
task_3: SandboxTask3 = None
|
|
task_4: SandboxTask4 = None
|
|
tasks: list[SandboxTask] = None
|
|
|
|
test_level: int = 0
|
|
allow_to_fail: bool = False
|
|
skip: bool = False
|
|
note: str = ""
|
|
task_configs: list[str] = None
|
|
experiment = None
|
|
|
|
# Set by caller — used by create_task_from_yaml
|
|
_factory_lookup: dict = None
|
|
|
|
def __post_init__(self):
|
|
"""Collect tasks from slots/configs and resolve <<global_vars.X>> references."""
|
|
if self.tasks is None:
|
|
self.tasks = []
|
|
for i in range(5):
|
|
task = getattr(self, f"task_{i}", None)
|
|
if task is not None:
|
|
self.tasks += [task]
|
|
if self.task_configs is not None:
|
|
lookup = self._factory_lookup or _FACTORY_REGISTRY
|
|
if lookup:
|
|
self.tasks += [
|
|
create_task_from_yaml(yaml_file=yf, factory_lookup=lookup)
|
|
for yf in self.task_configs
|
|
]
|
|
|
|
if self.global_vars is not None:
|
|
global_vars_dict = {
|
|
k: v for k, v in dataclasses.asdict(self.global_vars).items() if v is not None
|
|
}
|
|
|
|
def _resolve(s):
|
|
"""Replace <<global_vars.X>> with the corresponding value."""
|
|
if not isinstance(s, str):
|
|
return s
|
|
return re.sub(
|
|
r"<<global_vars\.(\w+)>>",
|
|
lambda m: global_vars_dict.get(m.group(1), m.group(0)),
|
|
s,
|
|
)
|
|
|
|
for task in self.tasks:
|
|
if task.environment:
|
|
if isinstance(task.environment, list):
|
|
task.environment = [
|
|
{k: _resolve(v) for k, v in item.items()} for item in task.environment
|
|
]
|
|
else:
|
|
task.environment = {k: _resolve(v) for k, v in task.environment.items()}
|
|
if task.args:
|
|
task.args = [_resolve(a) for a in task.args]
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Executor builders
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def build_slurm_executor(
|
|
user,
|
|
identity,
|
|
slurm_config,
|
|
experiment_id,
|
|
job_dir,
|
|
task_name,
|
|
packager,
|
|
experiment_title="cicd",
|
|
):
|
|
"""Build a SlurmExecutor for remote job submission."""
|
|
container_mounts = list(slurm_config.container_mounts or [])
|
|
|
|
scratch_dst = "/scratchspace"
|
|
scratch_src = f"{job_dir}/{experiment_title}/{experiment_id}"
|
|
modelopt_dst = slurm_config.modelopt_install_path
|
|
modelopt_src = (
|
|
f"{job_dir}/{experiment_title}/{experiment_id}"
|
|
f"/{task_name}/code/modules/Model-Optimizer/modelopt"
|
|
)
|
|
container_mounts += [
|
|
f"{scratch_src}:{scratch_dst}",
|
|
f"{modelopt_src}:{modelopt_dst}",
|
|
f"{job_dir}/{experiment_title}:/{experiment_title}",
|
|
]
|
|
|
|
tunnel = run.SSHTunnel(
|
|
host=slurm_config.host,
|
|
user=getpass.getuser() if user is None else user,
|
|
port=slurm_config.port,
|
|
job_dir=job_dir,
|
|
identity=identity,
|
|
)
|
|
|
|
executor = run.SlurmExecutor(
|
|
account=slurm_config.account,
|
|
partition=slurm_config.partition,
|
|
ntasks_per_node=slurm_config.ntasks_per_node,
|
|
gpus_per_node=slurm_config.gpus_per_node,
|
|
nodes=slurm_config.nodes,
|
|
tunnel=tunnel,
|
|
container_image=slurm_config.container,
|
|
container_mounts=container_mounts,
|
|
array=slurm_config.array,
|
|
time="04:00:00",
|
|
mem="0",
|
|
retries=0,
|
|
packager=packager,
|
|
srun_args=slurm_config.srun_args,
|
|
)
|
|
return executor
|
|
|
|
|
|
def build_docker_executor(
|
|
hf_local,
|
|
slurm_config,
|
|
experiment_id,
|
|
job_dir,
|
|
task_name,
|
|
packager,
|
|
modelopt_src_path=None,
|
|
experiment_title="cicd",
|
|
):
|
|
"""Build a DockerExecutor for local GPU jobs."""
|
|
if slurm_config.local:
|
|
container_mounts = list(slurm_config.container_mounts or [])
|
|
else:
|
|
container_mounts = []
|
|
container_mounts += [f"{hf_local}:/hf-local"]
|
|
|
|
scratch_dst = "/scratchspace"
|
|
scratch_src = os.path.join(job_dir, experiment_title, experiment_id, task_name)
|
|
os.makedirs(scratch_src, exist_ok=True)
|
|
modelopt_dst = slurm_config.modelopt_install_path
|
|
if modelopt_src_path is None:
|
|
modelopt_src_path = os.path.join(os.getcwd(), "modules/Model-Optimizer/modelopt")
|
|
exp_title_src = os.path.join(job_dir, experiment_title)
|
|
os.makedirs(exp_title_src, exist_ok=True)
|
|
container_mounts += [
|
|
f"{scratch_src}:{scratch_dst}",
|
|
f"{modelopt_src_path}:{modelopt_dst}",
|
|
f"{exp_title_src}:/{experiment_title}",
|
|
]
|
|
|
|
executor = run.DockerExecutor(
|
|
num_gpus=-1,
|
|
runtime="nvidia",
|
|
ipc_mode="host",
|
|
container_image=slurm_config.container,
|
|
volumes=container_mounts,
|
|
additional_kwargs={"user": f"{os.getuid()}:{os.getgid()}"},
|
|
packager=packager,
|
|
)
|
|
return executor
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Version reporting
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def _git_info(path):
|
|
"""Get git commit hash and branch for a directory."""
|
|
import subprocess # nosec B404
|
|
|
|
try:
|
|
commit = subprocess.run( # nosec B603 B607
|
|
["git", "rev-parse", "--short", "HEAD"],
|
|
cwd=path,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=5,
|
|
).stdout.strip()
|
|
branch = subprocess.run( # nosec B603 B607
|
|
["git", "rev-parse", "--abbrev-ref", "HEAD"],
|
|
cwd=path,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=5,
|
|
).stdout.strip()
|
|
return commit, branch
|
|
except Exception:
|
|
return "unknown", "unknown"
|
|
|
|
|
|
def report_versions(base_dir):
|
|
"""Print git commit and branch for the launcher and all submodules."""
|
|
print("=" * 60)
|
|
print("Version Report")
|
|
print("=" * 60)
|
|
|
|
# Launcher / repo root
|
|
commit, branch = _git_info(base_dir)
|
|
print(f" {'Launcher':<30} {commit:<12} ({branch})")
|
|
|
|
# Submodules
|
|
modules_dir = os.path.join(base_dir, "modules")
|
|
if os.path.isdir(modules_dir):
|
|
for name in sorted(os.listdir(modules_dir)):
|
|
sub_path = os.path.join(modules_dir, name)
|
|
if os.path.exists(os.path.join(sub_path, ".git")):
|
|
commit, branch = _git_info(sub_path)
|
|
print(f" {name:<30} {commit:<12} ({branch})")
|
|
|
|
print("=" * 60)
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Shared job run loop
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
def run_jobs(
|
|
job_table,
|
|
hf_local,
|
|
user,
|
|
identity,
|
|
job_dir,
|
|
packager,
|
|
default_slurm_env,
|
|
default_local_env,
|
|
experiment_title="cicd",
|
|
detach=False,
|
|
test_level=0,
|
|
modelopt_src_path=None,
|
|
base_dir=None,
|
|
):
|
|
"""Run all jobs in job_table.
|
|
|
|
Args:
|
|
job_table: Dict mapping job_name -> SandboxPipeline.
|
|
hf_local: Path to local HF cache (None for remote Slurm).
|
|
user: SSH user.
|
|
identity: SSH identity file.
|
|
job_dir: Base directory for job artifacts.
|
|
packager: PatternPackager instance.
|
|
default_slurm_env: Default env vars for Slurm jobs.
|
|
default_local_env: Default env vars for local Docker jobs.
|
|
experiment_title: Experiment title (e.g., "cicd" or "modelopt").
|
|
detach: Whether to detach from the experiment.
|
|
test_level: Only run jobs with test_level <= this value.
|
|
modelopt_src_path: Path to modelopt source for Docker mounts.
|
|
base_dir: Base directory for version reporting (default: cwd).
|
|
"""
|
|
report_versions(base_dir or os.getcwd())
|
|
|
|
for job_name, job in job_table.items():
|
|
if job.test_level > test_level:
|
|
job.skip = True
|
|
if job.skip:
|
|
continue
|
|
|
|
dependency = None
|
|
exp = run.Experiment(experiment_title, log_level="INFO")
|
|
job.experiment = exp
|
|
|
|
with exp:
|
|
for task_id, task in enumerate(job.tasks):
|
|
if task.skip:
|
|
print(f"job {job_name} task {task_id}: skipped")
|
|
continue
|
|
task_name = f"{job_name}_{task_id}"
|
|
task_args = [] if task.args is None else task.args
|
|
|
|
task_env = {}
|
|
if task.environment is not None:
|
|
if isinstance(task.environment, list):
|
|
for item in task.environment:
|
|
task_env.update(item.items())
|
|
else:
|
|
task_env = task.environment
|
|
for k, v in task_env.items():
|
|
task_env[k] = "" if v is None else str(v)
|
|
|
|
if hf_local is not None:
|
|
executor = build_docker_executor(
|
|
hf_local,
|
|
task.slurm_config,
|
|
exp._id,
|
|
job_dir,
|
|
task_name,
|
|
packager,
|
|
modelopt_src_path,
|
|
experiment_title,
|
|
)
|
|
task_env.update(default_local_env)
|
|
else:
|
|
executor = build_slurm_executor(
|
|
user,
|
|
identity,
|
|
task.slurm_config,
|
|
exp._id,
|
|
job_dir,
|
|
task_name,
|
|
packager,
|
|
experiment_title,
|
|
)
|
|
task_env.update(default_slurm_env)
|
|
|
|
task_instance = run.Script(task.script, args=task_args, env=task_env)
|
|
print(f"job {job_name} task {task_id} slurm_config: {task.slurm_config}")
|
|
|
|
if dependency is None:
|
|
dependency = exp.add(
|
|
task_instance, tail_logs=True, name=task_name, executor=executor
|
|
)
|
|
else:
|
|
dependency = exp.add(
|
|
task_instance,
|
|
tail_logs=True,
|
|
name=task_name,
|
|
executor=executor,
|
|
dependencies=[dependency],
|
|
)
|
|
|
|
exp.run(detach=detach)
|
|
|
|
# Write metadata for downstream tools
|
|
metadata = {
|
|
"experiment_id": exp._id,
|
|
"job_name": job_name,
|
|
"allow_to_fail": job.allow_to_fail,
|
|
"note": job.note,
|
|
}
|
|
metadata_path = os.path.join("experiments", experiment_title, exp._id, "metadata.json")
|
|
os.makedirs(os.path.dirname(metadata_path), exist_ok=True)
|
|
with open(metadata_path, "w") as f:
|
|
json.dump(metadata, f)
|