feat: add nsys model layer name scope and benchmark support (with nsys) in app (#951)

ZhiyuLi-Nvidia · web-flow · commit ab56f2fd97cf · 2025-12-16T10:03:19.000-05:00
Signed-off-by: Zhiyu Li &lt;zhiyul@NVIDIA.com&gt;
diff --git a/nemo_automodel/_cli/app.py b/nemo_automodel/_cli/app.py
@@ -42,6 +42,24 @@
 #         ├── ...
 #         └── qwen2_5_vl_3b_rdr.yaml
 
+COMMAND_ALIASES = {"finetune": "train_ft", "pretrain": "train_ft", "benchmark": "benchmark"}
+
+
+def get_recipe_script_path(command: str, domain: str, repo_root: str | Path) -> str:
+    """
+    Get the script path for a given command and domain.
+
+    Args:
+        command: The command name (e.g., 'finetune', 'benchmark', 'pretrain')
+        domain: The domain (e.g., 'llm', 'vlm')
+        repo_root: The repository root path
+
+    Returns:
+        str: Full path to the recipe script
+    """
+    recipe_name = COMMAND_ALIASES.get(command, command)
+    return f"{repo_root}/nemo_automodel/recipes/{domain}/{recipe_name}.py"
+
 
 def load_function(file_path: str | Path, func_name: str):
     """
@@ -135,16 +153,36 @@ def launch_with_slurm(args, job_conf_path, job_dir, slurm_config, extra_args=Non
     if slurm_config.get("job_name", "") == "":
         slurm_config["job_name"] = f"{args.domain}_{args.command}"
 
+    # Get the recipe script path
+    script_path = get_recipe_script_path(args.command, args.domain, repo_root)
+
+    # Build nsys profile command if enabled
+    if slurm_config.get("nsys_enabled", False):
+        profile_cmd = (
+            f"nsys profile -s none "
+            f"--trace=cuda,cudnn,nvtx "
+            f"--cudabacktrace=all "
+            f"--cuda-graph-trace=node "
+            f"--python-backtrace=cuda "
+            f"--wait all "
+            f"-o {job_dir}/automodel_profile_%p.nsys-rep "
+            f"--force-overwrite true "
+            f"--capture-range=cudaProfilerApi "
+            f"--capture-range-end=stop "
+        )
+    else:
+        profile_cmd = ""
+
     # create the command
     command_parts = [
         f"PYTHONPATH={repo_root}:$PYTHONPATH",
         # Use torchrun to launch multiple processes instead
-        "uv sync --inexact --frozen $(cat /opt/uv_args.txt) && uv run --no-sync torchrun ",
+        f"uv sync --inexact --frozen $(cat /opt/uv_args.txt) && {profile_cmd}uv run --no-sync torchrun ",
         f"--nproc_per_node={slurm_config['ntasks_per_node']} ",
         f"--nnodes={slurm_config['nodes']} ",
         "--rdzv_backend=c10d ",
         f"--rdzv_endpoint=${{MASTER_ADDR}}:${{MASTER_PORT}}",  # noqa: F541
-        f"{repo_root}/examples/{args.domain}_{args.command}/{args.command}.py",
+        script_path,
         "-c",
         f"{job_conf_path}",
     ]
@@ -174,8 +212,8 @@ def build_parser() -> argparse.ArgumentParser:
     parser.add_argument(
         "command",
         metavar="<command>",
-        choices=["finetune", "pretrain", "kd"],
-        help="Command within the domain (e.g., finetune, pretrain, kd, etc)",
+        choices=["finetune", "pretrain", "kd", "benchmark"],
+        help="Command within the domain (e.g., finetune, pretrain, kd, benchmark, etc)",
     )
     parser.add_argument(
         "domain",
@@ -229,12 +267,9 @@ def run_interactive(args):
     from torch.distributed.run import determine_local_world_size, get_args_parser
     from torch.distributed.run import run as thrun
 
-    COMMAND_ALIASES = {"finetune": "train_ft", "pretrain": "train_ft"}
-    # remap commands: finetune -> train_ft
-    command = COMMAND_ALIASES.get(args.command, args.command)
     config_path = args.config.resolve()
     repo_root = get_repo_root()
-    script_path = repo_root / "nemo_automodel" / "recipes" / args.domain / f"{command}.py"
+    script_path = Path(get_recipe_script_path(args.command, args.domain, repo_root))
 
     # launch job on this node
     num_devices = determine_local_world_size(nproc_per_node="gpu")
diff --git a/nemo_automodel/autonvtx/__init__.py b/nemo_automodel/autonvtx/__init__.py
@@ -0,0 +1,97 @@
+# Copyright (c) 2020, NVIDIA CORPORATION.  All rights reserved.
+#
+# Licensed under the Apache License, Version 2.0 (the "License");
+# you may not use this file except in compliance with the License.
+# You may obtain a copy of the License at
+#
+#     http://www.apache.org/licenses/LICENSE-2.0
+#
+# Unless required by applicable law or agreed to in writing, software
+# distributed under the License is distributed on an "AS IS" BASIS,
+# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
+# See the License for the specific language governing permissions and
+# limitations under the License.
+
+
+from threading import local
+
+import torch
+
+# inspired by https://github.com/zasdfgbnm/autonvtx
+
+# Thread-local storage to track active NVTX ranges and prevent recursion
+_thread_local = local()
+
+
+def _get_active_ranges():
+    """Get the set of currently active NVTX ranges for this thread."""
+    if not hasattr(_thread_local, "active_ranges"):
+        _thread_local.active_ranges = set()
+    return _thread_local.active_ranges
+
+
+def _add_nvtx_hooks(model, name, add_backward_hooks=True):
+    """Add NVTX range hooks to a model's forward and optionally backward passes."""
+    if hasattr(model, "_nvtx_patched"):
+        return
+
+    def push_fwd(module, *args, **kwargs):
+        if name in _get_active_ranges():
+            module._nvtx_skipped = True
+            return
+        module._nvtx_skipped = False
+        _get_active_ranges().add(name)
+        torch.cuda.nvtx.range_push(name)
+
+    def pop_fwd(module, *args, **kwargs):
+        if getattr(module, "_nvtx_skipped", False):
+            return
+        torch.cuda.nvtx.range_pop()
+        _get_active_ranges().discard(name)
+
+    model.register_forward_pre_hook(push_fwd)
+    model.register_forward_hook(pop_fwd)
+
+    if add_backward_hooks:
+
+        def push_bwd(module, grad_input):
+            if name in _get_active_ranges():
+                module._nvtx_skipped = True
+                return
+            module._nvtx_skipped = False
+            _get_active_ranges().add(name)
+            torch.cuda.nvtx.range_push(name)
+
+        def pop_bwd(module, grad_input, grad_output):
+            if getattr(module, "_nvtx_skipped", False):
+                return
+            torch.cuda.nvtx.range_pop()
+            _get_active_ranges().discard(name)
+
+        model.register_full_backward_pre_hook(push_bwd)
+        model.register_full_backward_hook(pop_bwd)
+
+    model._nvtx_patched = True
+
+
+def patch(model, name=None, add_backward_hooks=True):
+    """
+    Recursively patch a model with NVTX profiling annotations.
+
+    Prevents duplicate scopes when activation checkpointing reruns forward passes.
+    """
+    if hasattr(model, "_nvtx_patched"):
+        return model
+
+    name = type(model).__name__ if name is None else f"{name}: {type(model).__name__}"
+    _add_nvtx_hooks(model, name, add_backward_hooks=add_backward_hooks)
+
+    # Recursively patch all children
+    for child_name, child in model.named_children():
+        patch(child, child_name, add_backward_hooks)
+
+    return model
+
+
+# Export the functions properly
+__all__ = ["patch"]
diff --git a/nemo_automodel/components/_peft/lora.py b/nemo_automodel/components/_peft/lora.py
@@ -268,6 +268,7 @@ def patch_linear_module(
     lora_A_init_method="xavier",
     lora_dtype=None,
     use_triton=True,
+    layer_name=None,
 ):
     """
     Monkey-patches a nn.Linear (orig_linear param) to be a LinearLoRA.
@@ -321,6 +322,8 @@ def patch_linear_module(
         orig_linear.super_fwd = orig_linear.forward
 
     orig_linear.__class__ = new_cls
+    if layer_name is not None:
+        orig_linear._layer_name = layer_name
     return orig_linear
 
 
@@ -382,6 +385,7 @@ def apply_lora_to_linear_modules(
                 lora_A_init_method=peft_config.lora_A_init,
                 lora_dtype=lora_dtype,
                 use_triton=peft_config.use_triton,
+                layer_name=name,
             )
 
     return num_modules_matched
diff --git a/nemo_automodel/components/launcher/slurm/config.py b/nemo_automodel/components/launcher/slurm/config.py
@@ -71,6 +71,7 @@ class SlurmConfig:
     # User command
     command: str = field(default="", metadata=dict(help="Shell command(s) to run inside container"))
     chdir: str = field(default=None, metadata=dict(help="Working directory of the job"))
+    nsys_enabled: bool = field(default=False, metadata=dict(help="Enable nsys profiling"))
 
     def __post_init__(self):
         if isinstance(self.extra_mounts, list):
diff --git a/nemo_automodel/recipes/llm/train_ft.py b/nemo_automodel/recipes/llm/train_ft.py
@@ -1006,7 +1006,16 @@ def setup(self):
         if isinstance(model, AutoPipeline):
             self.model_parts = model.parts
             self.pp = model
+            import nemo_automodel.autonvtx as autonvtx
+
+            # Patch each pipeline stage with NVTX profiling
+            for i, part in enumerate(self.model_parts):
+                autonvtx.patch(part, name=f"PipelineStage_{i}")
         else:
+            import nemo_automodel.autonvtx as autonvtx
+
+            # Patch model with NVTX profiling
+            autonvtx.patch(model, name=model.__class__.__name__)
             self.model_parts = [model]
             self.pp = None