From 8451c47e1a1a9ed161651797e476f2e079cc6a27 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Mon, 14 Sep 2026 23:48:55 +0000 Subject: [PATCH 01/28] Add PyTorch backend interval filter - libPyTorchInterval.so: babeltrace2 filter turning lttng_ust_pytorch op_entry/op_exit into generic interval lttng:host messages - btx_pytorch_model.yaml: hand-written upstream model (cxi-style, no --matching needed for 2 fixed events) - btx_pytorchinterval_callbacks.cpp: pairs entry/exit via EntryState, keyed by {hostname, vpid, vtid} - Adds BACKEND_PYTORCH to backend_e (utils/xprof_utils.hpp) Verified against a real trace (854 op_entry/op_exit pairs): to_interval produces 854 correctly-named, correctly-timed lttng:host messages. Pending: tally name/level registration, default --backends wiring (next commit); timeline needs no changes (backend-agnostic, verified). Co-Authored-By: Claude Sonnet 5 --- backends/pytorch/Makefile.am | 37 +++++++++++++++ backends/pytorch/btx_pytorch_model.yaml | 45 +++++++++++++++++++ .../pytorch/btx_pytorchinterval_callbacks.cpp | 44 ++++++++++++++++++ utils/xprof_utils.hpp | 1 + 4 files changed, 127 insertions(+) create mode 100644 backends/pytorch/btx_pytorch_model.yaml create mode 100644 backends/pytorch/btx_pytorchinterval_callbacks.cpp diff --git a/backends/pytorch/Makefile.am b/backends/pytorch/Makefile.am index 44ccfceea..01448432f 100644 --- a/backends/pytorch/Makefile.am +++ b/backends/pytorch/Makefile.am @@ -84,3 +84,40 @@ libTracerPytorch_la_CXXFLAGS = -std=c++17 -Wall -Wextra -Wno-unused-parameter -W libTracerPytorch_la_LDFLAGS = $(LTTNG_UST_LIBS) -Ldummy_libs -ltorch_cpu -avoid-version -module libTracerPytorch_la_LIBADD = libpytorchtracepoints.la EXTRA_libTracerPytorch_la_DEPENDENCIES = $(DUMMY_TORCH_LIBS) + +BTX_PYTORCH_GENERATED = \ + btx_filter_pytorch/metababel/metababel.h \ + btx_filter_pytorch/metababel/btx_component.h \ + btx_filter_pytorch/metababel/btx_component.c \ + btx_filter_pytorch/metababel/btx_upstream.h \ + btx_filter_pytorch/metababel/btx_upstream.c \ + btx_filter_pytorch/metababel/btx_downstream.h \ + btx_filter_pytorch/metababel/btx_downstream.c \ + btx_filter_pytorch/btx_main.c + +EXTRA_DIST += \ + $(top_srcdir)/xprof/btx_interval_model.yaml \ + $(srcdir)/btx_pytorch_model.yaml + +$(BTX_PYTORCH_GENERATED) &: $(top_srcdir)/xprof/btx_interval_model.yaml $(srcdir)/btx_pytorch_model.yaml + $(METABABEL) -u $(srcdir)/btx_pytorch_model.yaml -d $(top_srcdir)/xprof/btx_interval_model.yaml -t FILTER -o btx_filter_pytorch -p pytorchinterval -c interval + +CLEANFILES += \ + $(BTX_PYTORCH_GENERATED) + +BUILT_SOURCES += \ + $(BTX_PYTORCH_GENERATED) + +nodist_libPyTorchInterval_la_SOURCES = \ + $(BTX_PYTORCH_GENERATED) + +libPyTorchInterval_la_SOURCES = \ + btx_pytorchinterval_callbacks.cpp + +libPyTorchInterval_la_CPPFLAGS = -I$(top_srcdir)/utils -I$(top_srcdir)/utils/include -I./ -I./btx_filter_pytorch +libPyTorchInterval_la_CFLAGS = -Wall -Wextra -Wno-unused-parameter $(WERROR) $(BABELTRACE2_CFLAGS) +libPyTorchInterval_la_CXXFLAGS = -std=c++17 -Wall -Wextra -Wno-unused-parameter $(WERROR) $(BABELTRACE2_CFLAGS) +libPyTorchInterval_la_LDFLAGS = $(BABELTRACE2_LIBS) -avoid-version -module + +bt2dir = $(pkglibdir)/bt2 +bt2_LTLIBRARIES = libPyTorchInterval.la diff --git a/backends/pytorch/btx_pytorch_model.yaml b/backends/pytorch/btx_pytorch_model.yaml new file mode 100644 index 000000000..8f9996c19 --- /dev/null +++ b/backends/pytorch/btx_pytorch_model.yaml @@ -0,0 +1,45 @@ +:environment: + :entries: + - :name: hostname + :type: string +:stream_classes: +- :name: thapi_pytorch + :default_clock_class: {} + :packet_context_field_class: + :type: structure + :members: + - :name: cpu_id + :field_class: + :type: integer_unsigned + :cast_type: uint64_t + :field_value_range: 32 + :event_common_context_field_class: + :type: structure + :members: + - :name: vpid + :field_class: + :type: integer_signed + :cast_type: int64_t + :field_value_range: 64 + - :name: vtid + :field_class: + :type: integer_unsigned + :cast_type: uint64_t + :field_value_range: 64 + :event_classes: + - :name: lttng_ust_pytorch:op_entry + :payload_field_class: + :type: structure + :members: + - :name: name + :field_class: + :cast_type: char * + :type: string + - :name: lttng_ust_pytorch:op_exit + :payload_field_class: + :type: structure + :members: + - :name: name + :field_class: + :cast_type: char * + :type: string diff --git a/backends/pytorch/btx_pytorchinterval_callbacks.cpp b/backends/pytorch/btx_pytorchinterval_callbacks.cpp new file mode 100644 index 000000000..72e481b61 --- /dev/null +++ b/backends/pytorch/btx_pytorchinterval_callbacks.cpp @@ -0,0 +1,44 @@ +#include "xprof_utils.hpp" +#include + +struct data_s { + EntryState entry_state; +}; +typedef struct data_s data_t; + +static void btx_initialize_component(void **usr_data) { *usr_data = new data_t; } + +static void btx_finalize_component(void *usr_data) { delete static_cast(usr_data); } + +static void lttng_ust_pytorch_op_entry_callback(void *btx_handle, + void *usr_data, + int64_t ts, + const char *hostname, + int64_t vpid, + uint64_t vtid, + char * /*name*/) { + static_cast(usr_data)->entry_state.set_ts({hostname, vpid, vtid}, ts); +} + +static void lttng_ust_pytorch_op_exit_callback(void *btx_handle, + void *usr_data, + int64_t ts, + const char *hostname, + int64_t vpid, + uint64_t vtid, + char *name) { + auto *state = static_cast(usr_data); + const int64_t entry_ts = state->entry_state.get_ts({hostname, vpid, vtid}); + + const bool err = false; + btx_push_message_lttng_host(btx_handle, hostname, vpid, vtid, entry_ts, BACKEND_PYTORCH, name, + (ts - entry_ts), err); +} + +void btx_register_usr_callbacks(void *btx_handle) { + btx_register_callbacks_initialize_component(btx_handle, &btx_initialize_component); + btx_register_callbacks_finalize_component(btx_handle, &btx_finalize_component); + + btx_register_callbacks_lttng_ust_pytorch_op_entry(btx_handle, <tng_ust_pytorch_op_entry_callback); + btx_register_callbacks_lttng_ust_pytorch_op_exit(btx_handle, <tng_ust_pytorch_op_exit_callback); +} diff --git a/utils/xprof_utils.hpp b/utils/xprof_utils.hpp index 26c99aee7..63b3c0de4 100644 --- a/utils/xprof_utils.hpp +++ b/utils/xprof_utils.hpp @@ -23,6 +23,7 @@ enum backend_e { BACKEND_MPI = 7, BACKEND_CXI = 8, BACKEND_ITT = 9, + BACKEND_PYTORCH = 10, }; typedef enum backend_e backend_t; typedef unsigned backend_level_t; From a625d1f744d4b3cea5963af053b626e53eebdb62 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 15 Sep 2026 14:12:55 +0000 Subject: [PATCH 02/28] Register pytorch backend with tally and default --backends lists - utils/xprof_utils.hpp: adds pytorch to pretty_backend_name_g and backend_levels_g at level 6 (its own tier, above itt(5), since RecordFunction ops wrap cuda/ze/omp/mpi calls beneath them) - xprof/xprof.rb.in, utils/babeltrace_thapi.in: add pytorch:6 to the default --backends list so tally/timeline pick it up without an explicit --backends flag Verified against the same real trace (854 op_entry/op_exit pairs): `tally` now prints a dedicated BACKEND_PYTORCH section (854 calls, 27.75ms total) instead of "Wrong Backend passed" warnings; other backends' tally output (ze/cl) is unaffected. Timeline needs no changes -- confirmed backend-agnostic in the prior commit. Co-Authored-By: Claude Sonnet 5 --- utils/babeltrace_thapi.in | 2 +- utils/xprof_utils.hpp | 2 ++ xprof/xprof.rb.in | 2 +- 3 files changed, 4 insertions(+), 2 deletions(-) diff --git a/utils/babeltrace_thapi.in b/utils/babeltrace_thapi.in index 0685ab128..801086d33 100755 --- a/utils/babeltrace_thapi.in +++ b/utils/babeltrace_thapi.in @@ -395,7 +395,7 @@ class BabeltraceParserThapi < OptionParserWithDefaultAndValidation on('-h', '--help', 'Prints this help') { print_help_and_exit(self, exit_code: 0) } on('-b', '--backends BACKENDS', Array, "Select which and how backends' need to handled.", 'Format: backend_name[:backend_level],...', - default: ['mpi:3', 'omp:2', 'cl:1', 'ze:1', 'cuda:1', 'hip:1', 'cxi:4', 'itt:5']) + default: ['mpi:3', 'omp:2', 'cl:1', 'ze:1', 'cuda:1', 'hip:1', 'cxi:4', 'itt:5', 'pytorch:6']) on('--debug', default: false) on('--archive SESSION-NAME') on('--archive-session-found-file-path PATH') diff --git a/utils/xprof_utils.hpp b/utils/xprof_utils.hpp index 63b3c0de4..bb256d51d 100644 --- a/utils/xprof_utils.hpp +++ b/utils/xprof_utils.hpp @@ -40,6 +40,7 @@ const std::unordered_map pretty_backend_name_g = { {"mpi", BACKEND_MPI}, {"cxi", BACKEND_CXI}, {"itt", BACKEND_ITT}, + {"pytorch", BACKEND_PYTORCH}, }; const std::unordered_map backend_levels_g = { @@ -53,6 +54,7 @@ const std::unordered_map backend_levels_g = { {BACKEND_MPI, 3}, {BACKEND_CXI, 4}, {BACKEND_ITT, 5}, + {BACKEND_PYTORCH, 6}, }; typedef std::string thapi_metadata_t; diff --git a/xprof/xprof.rb.in b/xprof/xprof.rb.in index 35cf9b9cd..0441a8d63 100755 --- a/xprof/xprof.rb.in +++ b/xprof/xprof.rb.in @@ -1075,7 +1075,7 @@ if $thapi_launch || __FILE__ == $PROGRAM_NAME # General Options parser.on('-b', '--backends BACKENDS', Array, 'Select which backends to use and their grouping level.', 'Format: backend_name[:backend_level],...', - default: ['mpi:3', 'omp:2', 'cl:1', 'ze:1', 'cuda:1', 'hip:1', 'cxi:4', 'itt:5']) + default: ['mpi:3', 'omp:2', 'cl:1', 'ze:1', 'cuda:1', 'hip:1', 'cxi:4', 'itt:5', 'pytorch:6']) parser.on('--[no-]archive', 'Enable or disable archive support.', default: false) # Analysis From a4d774c56b9b031d9f89b01dc824b043905d103d Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 15 Sep 2026 21:07:40 +0000 Subject: [PATCH 03/28] pytorch: fix entry/exit duration corruption on nested/reentrant ops Co-Authored-By: Claude Sonnet 5 --- .../pytorch/btx_pytorchinterval_callbacks.cpp | 24 +++++++++++++++---- .../pytorch/include/ATen/record_function.h | 14 ++++++----- backends/pytorch/tracer_pytorch.cpp | 22 +++++++++++++++-- 3 files changed, 47 insertions(+), 13 deletions(-) diff --git a/backends/pytorch/btx_pytorchinterval_callbacks.cpp b/backends/pytorch/btx_pytorchinterval_callbacks.cpp index 72e481b61..fbfd6a770 100644 --- a/backends/pytorch/btx_pytorchinterval_callbacks.cpp +++ b/backends/pytorch/btx_pytorchinterval_callbacks.cpp @@ -1,8 +1,17 @@ #include "xprof_utils.hpp" #include - +#include +#include + +// A per-thread LIFO stack of entry timestamps, not a single scalar, is +// required because PyTorch ops can re-enter themselves before returning +// (reentrance): torch.isfinite() on a complex tensor calls at::isfinite() +// again, once on the real part and once on the imaginary part, while its +// own outer call is still open (see aten/src/ATen/native/TensorCompare.cpp). +// A scalar slot would be overwritten by the inner call and the outer +// call's exit would then read the wrong (inner) entry timestamp. struct data_s { - EntryState entry_state; + std::unordered_map> entry_stack; }; typedef struct data_s data_t; @@ -17,7 +26,7 @@ static void lttng_ust_pytorch_op_entry_callback(void *btx_handle, int64_t vpid, uint64_t vtid, char * /*name*/) { - static_cast(usr_data)->entry_state.set_ts({hostname, vpid, vtid}, ts); + static_cast(usr_data)->entry_stack[{hostname, vpid, vtid}].push_back(ts); } static void lttng_ust_pytorch_op_exit_callback(void *btx_handle, @@ -28,9 +37,14 @@ static void lttng_ust_pytorch_op_exit_callback(void *btx_handle, uint64_t vtid, char *name) { auto *state = static_cast(usr_data); - const int64_t entry_ts = state->entry_state.get_ts({hostname, vpid, vtid}); + auto &stack = state->entry_stack[{hostname, vpid, vtid}]; + // Empty means an exit arrived with no matching entry (e.g. a trace + // truncated mid-call); report it via the existing err flag instead of + // reading undefined data. + const bool err = stack.empty(); + const int64_t entry_ts = err ? ts : stack.back(); + if (!err) stack.pop_back(); - const bool err = false; btx_push_message_lttng_host(btx_handle, hostname, vpid, vtid, entry_ts, BACKEND_PYTORCH, name, (ts - entry_ts), err); } diff --git a/backends/pytorch/include/ATen/record_function.h b/backends/pytorch/include/ATen/record_function.h index 6003887ed..40dc6d369 100644 --- a/backends/pytorch/include/ATen/record_function.h +++ b/backends/pytorch/include/ATen/record_function.h @@ -1,13 +1,14 @@ // Minimal hand-written stand-in for PyTorch's . // -// Declares only the five names tracer_pytorch.cpp actually uses: +// Declares only the names tracer_pytorch.cpp actually uses: // at::RecordScope - FUNCTION / BACKWARD_FUNCTION values // at::ObserverContext - empty base; we only ever return nullptr -// at::RecordFunction - only .name(); we never construct one -// ourselves, only receive a reference from -// the real library, so no field layout is -// needed here -- name() resolves against the -// real out-of-line symbol in libtorch_cpu. +// at::RecordFunction - only .name() and .overload_name(); we +// never construct one ourselves, only +// receive a reference from the real library, +// so no field layout is needed here -- both +// resolve against their real out-of-line +// symbols in libtorch_cpu. // at::RecordFunctionCallback - constructed BY US and passed BY VALUE into // addGlobalCallback. Its field layout below // (order, types, and the scopes_ array sized @@ -55,6 +56,7 @@ struct ObserverContext { struct RecordFunction { const char *name() const; + const char *overload_name() const; }; class RecordFunctionCallback { diff --git a/backends/pytorch/tracer_pytorch.cpp b/backends/pytorch/tracer_pytorch.cpp index ac95849ae..b886d31cb 100644 --- a/backends/pytorch/tracer_pytorch.cpp +++ b/backends/pytorch/tracer_pytorch.cpp @@ -1,16 +1,34 @@ #include #include "pytorch.h" +#include + +// PyTorch identifies an operator by TWO strings: a schema name (e.g. +// "aten::abs") and an overload name (e.g. "" for the default overload, +// "out" for the variant that writes into a caller-supplied output tensor). +// fn.name() alone returns only the schema name, so two different overloads +// of the same op are indistinguishable in the trace. This matters because +// PyTorch's own operators frequently call one overload from another: the +// default abs(Tensor) allocates an output tensor and then calls abs.out() +// to actually compute the result, so a trace keyed on fn.name() alone shows +// "aten::abs" entering, then "aten::abs" entering AGAIN before the first +// one exits -- indistinguishable from the op reentering itself, even though +// it is really two different overloads, one nested inside the other. +// Emitting fn.name() + "." + fn.overload_name() keeps the two distinguishable. +static std::string qualified_name(const at::RecordFunction &fn) { + const char *overload = fn.overload_name(); + return (overload[0] == '\0') ? fn.name() : std::string(fn.name()) + "." + overload; +} // ENTRY: fires BEFORE the op runs. LTTng adds time + vpid/vtid via context. static std::unique_ptr on_entry(const at::RecordFunction &fn) { - tracepoint(lttng_ust_pytorch, op_entry, fn.name()); + tracepoint(lttng_ust_pytorch, op_entry, qualified_name(fn).c_str()); return nullptr; } // EXIT: fires AFTER the op returns. static void on_exit(const at::RecordFunction &fn, at::ObserverContext *) { - tracepoint(lttng_ust_pytorch, op_exit, fn.name()); + tracepoint(lttng_ust_pytorch, op_exit, qualified_name(fn).c_str()); } // Auto-register at library load (works under LD_PRELOAD, no python changes). From f56ba9428ce7e80434af0c494aca0cf6583c3bf6 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Wed, 16 Sep 2026 15:30:48 +0000 Subject: [PATCH 04/28] validate pytorch version --- xprof/xprof.rb.in | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/xprof/xprof.rb.in b/xprof/xprof.rb.in index 0441a8d63..6ac06a152 100755 --- a/xprof/xprof.rb.in +++ b/xprof/xprof.rb.in @@ -969,11 +969,21 @@ def all_env_tracers(usr_binary) # Supported calling: iprof -- python3 model.py (torch env is active). # Unsupported calling: iprof -- venv/bin/python3 module.py (non-active torch env) # TODO: We need env vars to support the second calling convention. - torch_file = exec('python3 -c "import torch; print(torch.__file__)"', debug: false).strip - torch_lib = torch_file.empty? ? '' : File.join(File.dirname(torch_file), 'lib') + torch_info = exec('python3 -c "import torch; print(torch.__file__); print(torch.__version__)"', debug: false) + torch_file, torch_version = torch_info.strip.split("\n") + torch_lib = torch_file.nil? ? '' : File.join(File.dirname(torch_file), 'lib') if torch_lib.empty? LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') else + torch_version_parts = torch_version.split(/[.+ab-]/).reject(&:empty?).first(3).map(&:to_i) + # 1.10.0 and older fail this check: RecordScope had fewer members + # (6 vs. our 10), shrinking RecordFunctionCallback's scopes_ bit-array. + # This only catches size/count changes -- a same-size reordering of + # members (e.g. swapping two fields of equal size) would pass + # undetected, breaking ABI compatibility. + unless ([1, 11, 0]..[2, 14, 0]).cover?(torch_version_parts) + LOGGER.warn("THAPI: untested PyTorch version #{torch_version}") + end backends << 'pytorch' # LibTracerPytorch.so calls at::addGlobalCallback() and RecordFunction::name() # both implemented in libtorch_cpu.so. The lib is reachable via LD_LIBRARY_PATH. From 020785b2844dbe521176ca6e7024a25a65af23dd Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Wed, 16 Sep 2026 19:30:26 +0000 Subject: [PATCH 05/28] LTTNG_UST_PYTORCH_LIBRARY_PATH --- xprof/xprof.rb.in | 43 ++++++++++++++++++++++++++----------------- 1 file changed, 26 insertions(+), 17 deletions(-) diff --git a/xprof/xprof.rb.in b/xprof/xprof.rb.in index 6ac06a152..6080175f0 100755 --- a/xprof/xprof.rb.in +++ b/xprof/xprof.rb.in @@ -966,24 +966,33 @@ def all_env_tracers(usr_binary) end if OPTIONS[:'backend-names'].include?('pytorch') - # Supported calling: iprof -- python3 model.py (torch env is active). - # Unsupported calling: iprof -- venv/bin/python3 module.py (non-active torch env) - # TODO: We need env vars to support the second calling convention. - torch_info = exec('python3 -c "import torch; print(torch.__file__); print(torch.__version__)"', debug: false) - torch_file, torch_version = torch_info.strip.split("\n") - torch_lib = torch_file.nil? ? '' : File.join(File.dirname(torch_file), 'lib') - if torch_lib.empty? - LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') - else - torch_version_parts = torch_version.split(/[.+ab-]/).reject(&:empty?).first(3).map(&:to_i) - # 1.10.0 and older fail this check: RecordScope had fewer members - # (6 vs. our 10), shrinking RecordFunctionCallback's scopes_ bit-array. - # This only catches size/count changes -- a same-size reordering of - # members (e.g. swapping two fields of equal size) would pass - # undetected, breaking ABI compatibility. - unless ([1, 11, 0]..[2, 14, 0]).cover?(torch_version_parts) - LOGGER.warn("THAPI: untested PyTorch version #{torch_version}") + # Supported patterns: + # * iprof -- python3 model.py (torch env is active). + # * iprof -- venv/bin/python3 module.py (non-active torch env, with LTTNG_UST_PYTORCH_LIBRARY_PATH) + + torch_lib = env_fetch_first('LTTNG_UST_PYTORCH_LIBRARY_PATH') + + if torch_lib.nil? + torch_info = exec('python3 -c "import torch; print(torch.__file__); print(torch.__version__)"', debug: false) + torch_file, torch_version = torch_info.strip.split("\n") + torch_lib = torch_file && File.join(File.dirname(torch_file), 'lib') + + if torch_lib.nil? + LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') + else + torch_version_parts = torch_version.split(/[.+ab-]/).reject(&:empty?).first(3).map(&:to_i) + # 1.10.0 and older fail this check: RecordScope had fewer members + # (6 vs. our 10), shrinking RecordFunctionCallback's scopes_ bit-array. + # This only catches size/count changes -- a same-size reordering of + # members (e.g. swapping two fields of equal size) would pass + # undetected, breaking ABI compatibility. + unless ([1, 11, 0]..[2, 14, 0]).cover?(torch_version_parts) + LOGGER.warn("THAPI: untested PyTorch version #{torch_version}") + end end + end + + unless torch_lib.nil? backends << 'pytorch' # LibTracerPytorch.so calls at::addGlobalCallback() and RecordFunction::name() # both implemented in libtorch_cpu.so. The lib is reachable via LD_LIBRARY_PATH. From 5c02d90e11644ed309db54aff9c7873278d9685c Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Wed, 16 Sep 2026 20:00:16 +0000 Subject: [PATCH 06/28] Formatting --- backends/pytorch/btx_pytorchinterval_callbacks.cpp | 14 +++++--------- 1 file changed, 5 insertions(+), 9 deletions(-) diff --git a/backends/pytorch/btx_pytorchinterval_callbacks.cpp b/backends/pytorch/btx_pytorchinterval_callbacks.cpp index fbfd6a770..3950e796e 100644 --- a/backends/pytorch/btx_pytorchinterval_callbacks.cpp +++ b/backends/pytorch/btx_pytorchinterval_callbacks.cpp @@ -3,13 +3,7 @@ #include #include -// A per-thread LIFO stack of entry timestamps, not a single scalar, is -// required because PyTorch ops can re-enter themselves before returning -// (reentrance): torch.isfinite() on a complex tensor calls at::isfinite() -// again, once on the real part and once on the imaginary part, while its -// own outer call is still open (see aten/src/ATen/native/TensorCompare.cpp). -// A scalar slot would be overwritten by the inner call and the outer -// call's exit would then read the wrong (inner) entry timestamp. +// PyTorch requies a per-thread LIFO stack since there is reentrace and nesting calls. struct data_s { std::unordered_map> entry_stack; }; @@ -43,7 +37,8 @@ static void lttng_ust_pytorch_op_exit_callback(void *btx_handle, // reading undefined data. const bool err = stack.empty(); const int64_t entry_ts = err ? ts : stack.back(); - if (!err) stack.pop_back(); + if (!err) + stack.pop_back(); btx_push_message_lttng_host(btx_handle, hostname, vpid, vtid, entry_ts, BACKEND_PYTORCH, name, (ts - entry_ts), err); @@ -53,6 +48,7 @@ void btx_register_usr_callbacks(void *btx_handle) { btx_register_callbacks_initialize_component(btx_handle, &btx_initialize_component); btx_register_callbacks_finalize_component(btx_handle, &btx_finalize_component); - btx_register_callbacks_lttng_ust_pytorch_op_entry(btx_handle, <tng_ust_pytorch_op_entry_callback); + btx_register_callbacks_lttng_ust_pytorch_op_entry(btx_handle, + <tng_ust_pytorch_op_entry_callback); btx_register_callbacks_lttng_ust_pytorch_op_exit(btx_handle, <tng_ust_pytorch_op_exit_callback); } From adbf879b699389cd6ed0f3d9ce76d6506f03200f Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 22 Sep 2026 18:34:57 +0000 Subject: [PATCH 07/28] Rename to pytorch_tracepoints.h, add overload_name support, re-write pytorch env variable and version validation --- backends/pytorch/Makefile.am | 19 +++-------- backends/pytorch/btx_pytorch_model.yaml | 8 +++++ .../pytorch/btx_pytorchinterval_callbacks.cpp | 31 ++++++++++++----- .../pytorch/include/ATen/record_function.h | 28 ++-------------- backends/pytorch/pytorch_events.yaml | 4 +++ backends/pytorch/tracer_pytorch.cpp | 27 +++------------ xprof/xprof.rb.in | 33 +++++++------------ 7 files changed, 57 insertions(+), 93 deletions(-) diff --git a/backends/pytorch/Makefile.am b/backends/pytorch/Makefile.am index 01448432f..33f70c516 100644 --- a/backends/pytorch/Makefile.am +++ b/backends/pytorch/Makefile.am @@ -4,7 +4,7 @@ else WERROR = endif -PYTORCH_STATIC_PROBES = pytorch +PYTORCH_STATIC_PROBES = pytorch_tracepoints PYTORCH_STATIC_PROBES_TP = $(PYTORCH_STATIC_PROBES:=.tp) @@ -13,7 +13,7 @@ PYTORCH_STATIC_PROBES_INCL = $(PYTORCH_STATIC_PROBES:=.h) PYTORCH_STATIC_PROBES_SRC = $(PYTORCH_STATIC_PROBES:=.c) $(PYTORCH_STATIC_PROBES_TP): %.tp: $(top_srcdir)/utils/gen_custom_probes.rb $(srcdir)/pytorch_events.yaml $(top_srcdir)/utils/gen_probe_base.rb - $(RUBY) $< $(srcdir)/pytorch_events.yaml lttng_ust_$* > $@ + $(RUBY) $< $(srcdir)/pytorch_events.yaml lttng_ust_pytorch > $@ %.h %.c: %.tp $(LTTNG_GEN_TP) $< -o $*.c -o $*.h @@ -41,8 +41,9 @@ CLEANFILES = \ BUILT_SOURCES = \ $(PYTORCH_STATIC_PROBES_INCL) -# Only -ltorch_cpu is needed: at::addGlobalCallback and at::RecordFunction::name -# are both defined in libtorch_cpu.so. +# Only -ltorch_cpu is needed: at::addGlobalCallback, at::RecordFunction::name +# and at::RecordFuntion::overload_name symbols must be resolved to register +# entry/exit callbacks at load time. DUMMY_TORCH_LIBS = \ dummy_libs/libtorch_cpu.so @@ -63,9 +64,6 @@ bin_SCRIPTS = \ lib_LTLIBRARIES = libTracerPytorch.la -# pytorch.c is only built once, into libpytorchtracepoints.la above; pull it -# in via LIBADD rather than compiling it again here (a second copy would -# clash with the first at link time: duplicate tracepoint symbols). nodist_libTracerPytorch_la_SOURCES = \ $(PYTORCH_STATIC_PROBES_INCL) @@ -73,13 +71,6 @@ libTracerPytorch_la_SOURCES = \ tracer_pytorch.cpp libTracerPytorch_la_CPPFLAGS = -I$(top_srcdir)/utils -I$(top_srcdir)/utils/include -I$(srcdir)/include -I./ -# -Wno-missing-field-initializers: LTTng's own tracepoint.h (not our code) -# aggregate-initializes structs without naming every field; clang is -# stricter than gcc about this under -Wextra. -# -Wno-unused-private-field: RecordFunctionCallback's fields exist only to -# match PyTorch's real class layout byte-for-byte (see the comment in -# include/ATen/record_function.h) -- the real libtorch reads them, we -# never do, by design. libTracerPytorch_la_CXXFLAGS = -std=c++17 -Wall -Wextra -Wno-unused-parameter -Wno-missing-field-initializers -Wno-unused-private-field $(WERROR) $(LTTNG_UST_CFLAGS) libTracerPytorch_la_LDFLAGS = $(LTTNG_UST_LIBS) -Ldummy_libs -ltorch_cpu -avoid-version -module libTracerPytorch_la_LIBADD = libpytorchtracepoints.la diff --git a/backends/pytorch/btx_pytorch_model.yaml b/backends/pytorch/btx_pytorch_model.yaml index 8f9996c19..3bcbe5346 100644 --- a/backends/pytorch/btx_pytorch_model.yaml +++ b/backends/pytorch/btx_pytorch_model.yaml @@ -35,6 +35,10 @@ :field_class: :cast_type: char * :type: string + - :name: overload_name + :field_class: + :cast_type: char * + :type: string - :name: lttng_ust_pytorch:op_exit :payload_field_class: :type: structure @@ -43,3 +47,7 @@ :field_class: :cast_type: char * :type: string + - :name: overload_name + :field_class: + :cast_type: char * + :type: string diff --git a/backends/pytorch/btx_pytorchinterval_callbacks.cpp b/backends/pytorch/btx_pytorchinterval_callbacks.cpp index 3950e796e..dd86900d9 100644 --- a/backends/pytorch/btx_pytorchinterval_callbacks.cpp +++ b/backends/pytorch/btx_pytorchinterval_callbacks.cpp @@ -1,5 +1,6 @@ #include "xprof_utils.hpp" #include +#include #include #include @@ -13,13 +14,22 @@ static void btx_initialize_component(void **usr_data) { *usr_data = new data_t; static void btx_finalize_component(void *usr_data) { delete static_cast(usr_data); } +// PyTorch identifies an operator by TWO strings: a schema name (e.g. +// "aten::abs") and an overload name (e.g. "" for the default overload, "out" +// Joining them here keeps e.g. "aten::abs" and "aten::abs.out" distinguishable +// in the trace instead of both showing up as plain "aten::abs". +static std::string qualified_name(const char *name, const char *overload_name) { + return (overload_name[0] == '\0') ? name : std::string(name) + "." + overload_name; +} + static void lttng_ust_pytorch_op_entry_callback(void *btx_handle, void *usr_data, int64_t ts, const char *hostname, int64_t vpid, uint64_t vtid, - char * /*name*/) { + char * /*name*/, + char * /*overload_name*/) { static_cast(usr_data)->entry_stack[{hostname, vpid, vtid}].push_back(ts); } @@ -29,19 +39,22 @@ static void lttng_ust_pytorch_op_exit_callback(void *btx_handle, const char *hostname, int64_t vpid, uint64_t vtid, - char *name) { + char *name, + char *overload_name) { auto *state = static_cast(usr_data); auto &stack = state->entry_stack[{hostname, vpid, vtid}]; - // Empty means an exit arrived with no matching entry (e.g. a trace - // truncated mid-call); report it via the existing err flag instead of - // reading undefined data. + + // Empty means an exit arrived with no matching entry const bool err = stack.empty(); - const int64_t entry_ts = err ? ts : stack.back(); - if (!err) + int64_t entry_ts = ts; + if (!err) { + entry_ts = stack.back(); stack.pop_back(); + } - btx_push_message_lttng_host(btx_handle, hostname, vpid, vtid, entry_ts, BACKEND_PYTORCH, name, - (ts - entry_ts), err); + const std::string full_name = qualified_name(name, overload_name); + btx_push_message_lttng_host(btx_handle, hostname, vpid, vtid, entry_ts, BACKEND_PYTORCH, + full_name.c_str(), (ts - entry_ts), err); } void btx_register_usr_callbacks(void *btx_handle) { diff --git a/backends/pytorch/include/ATen/record_function.h b/backends/pytorch/include/ATen/record_function.h index 40dc6d369..d6d538057 100644 --- a/backends/pytorch/include/ATen/record_function.h +++ b/backends/pytorch/include/ATen/record_function.h @@ -1,29 +1,5 @@ -// Minimal hand-written stand-in for PyTorch's . -// -// Declares only the names tracer_pytorch.cpp actually uses: -// at::RecordScope - FUNCTION / BACKWARD_FUNCTION values -// at::ObserverContext - empty base; we only ever return nullptr -// at::RecordFunction - only .name() and .overload_name(); we -// never construct one ourselves, only -// receive a reference from the real library, -// so no field layout is needed here -- both -// resolve against their real out-of-line -// symbols in libtorch_cpu. -// at::RecordFunctionCallback - constructed BY US and passed BY VALUE into -// addGlobalCallback. Its field layout below -// (order, types, and the scopes_ array sized -// by NUM_SCOPES) must match PyTorch's real -// class byte-for-byte, copied verbatim from -// the upstream header. If a future PyTorch -// release reorders/adds a field or changes -// RecordScope's member count, this header -// will still compile cleanly but -// addGlobalCallback will read the wrong -// bytes back -- silent corruption, not a -// build failure. Re-verify this layout -// against ATen/record_function.h whenever -// upstream PyTorch changes. -// at::addGlobalCallback - registers the callback pair +// Minimal hand-written PyTorch's . + #pragma once #include diff --git a/backends/pytorch/pytorch_events.yaml b/backends/pytorch/pytorch_events.yaml index 1d6807f42..0abce9e8c 100644 --- a/backends/pytorch/pytorch_events.yaml +++ b/backends/pytorch/pytorch_events.yaml @@ -3,10 +3,14 @@ lttng_ust_pytorch: - name: op_entry args: - ["const char *", name] + - ["const char *", overload_name] fields: - [ctf_string, name, name] + - [ctf_string, overload_name, overload_name] - name: op_exit args: - ["const char *", name] + - ["const char *", overload_name] fields: - [ctf_string, name, name] + - [ctf_string, overload_name, overload_name] diff --git a/backends/pytorch/tracer_pytorch.cpp b/backends/pytorch/tracer_pytorch.cpp index b886d31cb..3ad1647f5 100644 --- a/backends/pytorch/tracer_pytorch.cpp +++ b/backends/pytorch/tracer_pytorch.cpp @@ -1,39 +1,20 @@ #include -#include "pytorch.h" +#include "pytorch_tracepoints.h" #include -// PyTorch identifies an operator by TWO strings: a schema name (e.g. -// "aten::abs") and an overload name (e.g. "" for the default overload, -// "out" for the variant that writes into a caller-supplied output tensor). -// fn.name() alone returns only the schema name, so two different overloads -// of the same op are indistinguishable in the trace. This matters because -// PyTorch's own operators frequently call one overload from another: the -// default abs(Tensor) allocates an output tensor and then calls abs.out() -// to actually compute the result, so a trace keyed on fn.name() alone shows -// "aten::abs" entering, then "aten::abs" entering AGAIN before the first -// one exits -- indistinguishable from the op reentering itself, even though -// it is really two different overloads, one nested inside the other. -// Emitting fn.name() + "." + fn.overload_name() keeps the two distinguishable. -static std::string qualified_name(const at::RecordFunction &fn) { - const char *overload = fn.overload_name(); - return (overload[0] == '\0') ? fn.name() : std::string(fn.name()) + "." + overload; -} - -// ENTRY: fires BEFORE the op runs. LTTng adds time + vpid/vtid via context. static std::unique_ptr on_entry(const at::RecordFunction &fn) { - tracepoint(lttng_ust_pytorch, op_entry, qualified_name(fn).c_str()); + tracepoint(lttng_ust_pytorch, op_entry, fn.name(), fn.overload_name()); return nullptr; } -// EXIT: fires AFTER the op returns. static void on_exit(const at::RecordFunction &fn, at::ObserverContext *) { - tracepoint(lttng_ust_pytorch, op_exit, qualified_name(fn).c_str()); + tracepoint(lttng_ust_pytorch, op_exit, fn.name(), fn.overload_name()); } -// Auto-register at library load (works under LD_PRELOAD, no python changes). __attribute__((constructor)) static void tracer_pytorch_init() { at::addGlobalCallback( at::RecordFunctionCallback(&on_entry, &on_exit) .scopes({at::RecordScope::FUNCTION, at::RecordScope::BACKWARD_FUNCTION})); } + diff --git a/xprof/xprof.rb.in b/xprof/xprof.rb.in index 6080175f0..72abccecd 100755 --- a/xprof/xprof.rb.in +++ b/xprof/xprof.rb.in @@ -967,35 +967,26 @@ def all_env_tracers(usr_binary) if OPTIONS[:'backend-names'].include?('pytorch') # Supported patterns: - # * iprof -- python3 model.py (torch env is active). - # * iprof -- venv/bin/python3 module.py (non-active torch env, with LTTNG_UST_PYTORCH_LIBRARY_PATH) + # - iprof -- python3 model.py (torch env is active). + # - iprof -- venv/bin/python3 model.py (non-active torch env, with LTTNG_UST_PYTORCH_LIBRARY_PATH) + # - LTTNG_UST_PYTORCH_LIBRARY_PATH=${LIBTORCH_USED_BY_VENV} iprof -- venv/bin/python3 model.py torch_lib = env_fetch_first('LTTNG_UST_PYTORCH_LIBRARY_PATH') if torch_lib.nil? torch_info = exec('python3 -c "import torch; print(torch.__file__); print(torch.__version__)"', debug: false) - torch_file, torch_version = torch_info.strip.split("\n") - torch_lib = torch_file && File.join(File.dirname(torch_file), 'lib') - - if torch_lib.nil? - LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') - else - torch_version_parts = torch_version.split(/[.+ab-]/).reject(&:empty?).first(3).map(&:to_i) - # 1.10.0 and older fail this check: RecordScope had fewer members - # (6 vs. our 10), shrinking RecordFunctionCallback's scopes_ bit-array. - # This only catches size/count changes -- a same-size reordering of - # members (e.g. swapping two fields of equal size) would pass - # undetected, breaking ABI compatibility. - unless ([1, 11, 0]..[2, 14, 0]).cover?(torch_version_parts) - LOGGER.warn("THAPI: untested PyTorch version #{torch_version}") - end - end + torch_file, torch_version = torch_info.split("\n") + torch_lib = File.join(File.dirname(torch_file), 'lib') unless torch_file.nil? end - unless torch_lib.nil? + if torch_lib.nil? + LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') + else + tested = Gem::Version.new('1.11.0')..Gem::Version.new('2.14.0') + version = torch_version && Gem::Version.new(torch_version[/\d+(\.\d+)*/] || '0') + LOGGER.warn("THAPI: untested PyTorch version #{torch_version}") if version && !tested.cover?(version) + backends << 'pytorch' - # LibTracerPytorch.so calls at::addGlobalCallback() and RecordFunction::name() - # both implemented in libtorch_cpu.so. The lib is reachable via LD_LIBRARY_PATH. h[%w[LD_LIBRARY_PATH prepend]] << torch_lib h[%w[LD_PRELOAD prepend]] << File.join(LIBDIR, 'libTracerPytorch.so') end From d5473496a432cc4b43c2f61fc85acf4b3519b0ef Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 22 Sep 2026 19:28:49 +0000 Subject: [PATCH 08/28] Delete trailing blank line --- backends/pytorch/tracer_pytorch.cpp | 1 - 1 file changed, 1 deletion(-) diff --git a/backends/pytorch/tracer_pytorch.cpp b/backends/pytorch/tracer_pytorch.cpp index 3ad1647f5..7e967b93b 100644 --- a/backends/pytorch/tracer_pytorch.cpp +++ b/backends/pytorch/tracer_pytorch.cpp @@ -17,4 +17,3 @@ __attribute__((constructor)) static void tracer_pytorch_init() { at::RecordFunctionCallback(&on_entry, &on_exit) .scopes({at::RecordScope::FUNCTION, at::RecordScope::BACKWARD_FUNCTION})); } - From be64fae2fc0ae0aaee467874c13649b31a066448 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Thu, 24 Sep 2026 11:55:53 +0000 Subject: [PATCH 09/28] remove unnecesary import --- backends/pytorch/tracer_pytorch.cpp | 2 -- 1 file changed, 2 deletions(-) diff --git a/backends/pytorch/tracer_pytorch.cpp b/backends/pytorch/tracer_pytorch.cpp index 7e967b93b..0b6ecc3dc 100644 --- a/backends/pytorch/tracer_pytorch.cpp +++ b/backends/pytorch/tracer_pytorch.cpp @@ -1,7 +1,5 @@ #include - #include "pytorch_tracepoints.h" -#include static std::unique_ptr on_entry(const at::RecordFunction &fn) { tracepoint(lttng_ust_pytorch, op_entry, fn.name(), fn.overload_name()); From 275b21e8a846b27990a34e21ee4272e15b5626a9 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Thu, 24 Sep 2026 11:56:31 +0000 Subject: [PATCH 10/28] Update get_pytorch_library_path --- xprof/xprof.rb.in | 42 +++++++++++++++++++++--------------------- 1 file changed, 21 insertions(+), 21 deletions(-) diff --git a/xprof/xprof.rb.in b/xprof/xprof.rb.in index 72abccecd..4f61b3c1d 100755 --- a/xprof/xprof.rb.in +++ b/xprof/xprof.rb.in @@ -886,6 +886,25 @@ end # _| # +def get_pytorch_library_path + # Supported patterns: + # - iprof -- python3 model.py (torch env is active). + # - iprof -- venv/bin/python3 model.py (non-active torch env, with LTTNG_UST_PYTORCH_LIBRARY_PATH) + # - LTTNG_UST_PYTORCH_LIBRARY_PATH=${LIBTORCH_USED_BY_VENV} iprof -- venv/bin/python3 model.py + + torch_tested = Gem::Version.new('1.11.0')..Gem::Version.new('2.14.0') + torch_probe = 'python3 -c "import torch; print(torch.__file__); print(torch.__version__)"' + + lib = env_fetch_first('LTTNG_UST_PYTORCH_LIBRARY_PATH') || begin + file, version = exec(torch_probe, debug: false).split + version.slice(/\d+(\.\d+)*/).then { |v| torch_tested.cover?(Gem::Version.new(v)) || LOGGER.warn("THAPI: untested PyTorch version #{version}") } + file && File.join(File.dirname(file), 'lib') + end + + lib || LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') + lib +end + def all_env_tracers(usr_binary) # Return the list of backends (used by local master to enable lttng events) # and the ENV used by any traced-ranks to preload THAPI tracers @@ -966,28 +985,9 @@ def all_env_tracers(usr_binary) end if OPTIONS[:'backend-names'].include?('pytorch') - # Supported patterns: - # - iprof -- python3 model.py (torch env is active). - # - iprof -- venv/bin/python3 model.py (non-active torch env, with LTTNG_UST_PYTORCH_LIBRARY_PATH) - # - LTTNG_UST_PYTORCH_LIBRARY_PATH=${LIBTORCH_USED_BY_VENV} iprof -- venv/bin/python3 model.py - - torch_lib = env_fetch_first('LTTNG_UST_PYTORCH_LIBRARY_PATH') - - if torch_lib.nil? - torch_info = exec('python3 -c "import torch; print(torch.__file__); print(torch.__version__)"', debug: false) - torch_file, torch_version = torch_info.split("\n") - torch_lib = File.join(File.dirname(torch_file), 'lib') unless torch_file.nil? - end - - if torch_lib.nil? - LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') - else - tested = Gem::Version.new('1.11.0')..Gem::Version.new('2.14.0') - version = torch_version && Gem::Version.new(torch_version[/\d+(\.\d+)*/] || '0') - LOGGER.warn("THAPI: untested PyTorch version #{torch_version}") if version && !tested.cover?(version) - + get_pytorch_library_path&.then do |lib| backends << 'pytorch' - h[%w[LD_LIBRARY_PATH prepend]] << torch_lib + h[%w[LD_LIBRARY_PATH prepend]] << lib h[%w[LD_PRELOAD prepend]] << File.join(LIBDIR, 'libTracerPytorch.so') end end From ea9f8139358d300341d57ac79ea2dd5c03261295 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Thu, 24 Sep 2026 19:06:18 +0000 Subject: [PATCH 11/28] Do not raise on failure but report as warning --- xprof/xprof.rb.in | 28 ++++++++++++++-------------- 1 file changed, 14 insertions(+), 14 deletions(-) diff --git a/xprof/xprof.rb.in b/xprof/xprof.rb.in index 4f61b3c1d..ad47b27ea 100755 --- a/xprof/xprof.rb.in +++ b/xprof/xprof.rb.in @@ -76,12 +76,11 @@ end # \/ (_| | | (_) |_| _> |_ >< (/_ (_ # def exec(cmd, opts: {}, debug: true, ignore_exit_codes: []) - return Open3.capture3(opts, cmd).first unless debug + stdout_str, stderr_str, status = Open3.capture3(opts, cmd) + return [stdout_str, stderr_str, status] unless debug LOGGER.info { cmd } LOGGER.debug { opts } unless opts.empty? - - stdout_str, stderr_str, status = Open3.capture3(opts, cmd) LOGGER.debug { stdout_str.strip } unless stdout_str.empty? unless status.success? || ignore_exit_codes.include?(status.exitstatus) @@ -91,7 +90,8 @@ def exec(cmd, opts: {}, debug: true, ignore_exit_codes: []) end LOGGER.warn { stderr_str.strip } unless stderr_str.empty? - stdout_str + + [stdout_str, stderr_str, status] end def launch_usr_bin(env, cmd) @@ -153,7 +153,7 @@ def whichlib64(binary, *libs) whichlib64_bin = File.join(BINDIR, 'whichlib64') ld_path = [env_fetch_first('LD_LIBRARY_PATH'), env_fetch_first('CRAY_LD_LIBRARY_PATH')].compact.join(':') - stdout_str = exec("#{whichlib64_bin} #{binary} #{libs.join(' ')}", + stdout_str, _, _ = exec("#{whichlib64_bin} #{binary} #{libs.join(' ')}", opts: { 'LD_LIBRARY_PATH' => ld_path }, ignore_exit_codes: [1, 2]) @@ -887,21 +887,21 @@ end # def get_pytorch_library_path - # Supported patterns: - # - iprof -- python3 model.py (torch env is active). - # - iprof -- venv/bin/python3 model.py (non-active torch env, with LTTNG_UST_PYTORCH_LIBRARY_PATH) - # - LTTNG_UST_PYTORCH_LIBRARY_PATH=${LIBTORCH_USED_BY_VENV} iprof -- venv/bin/python3 model.py - torch_tested = Gem::Version.new('1.11.0')..Gem::Version.new('2.14.0') torch_probe = 'python3 -c "import torch; print(torch.__file__); print(torch.__version__)"' lib = env_fetch_first('LTTNG_UST_PYTORCH_LIBRARY_PATH') || begin - file, version = exec(torch_probe, debug: false).split - version.slice(/\d+(\.\d+)*/).then { |v| torch_tested.cover?(Gem::Version.new(v)) || LOGGER.warn("THAPI: untested PyTorch version #{version}") } - file && File.join(File.dirname(file), 'lib') + # Raises for other error codes, but only warns for 127 (python3 command not found) and 1 (ModuleNotFoundError) + stdout_str, _, status = exec(torch_probe, ignore_exit_codes: [1, 127]) + if status.success? + file, version = stdout_str.split + version.slice(/\d+(\.\d+)*/).then do |v| + torch_tested.cover?(Gem::Version.new(v)) || LOGGER.warn("Untested PyTorch version #{version}") + end + File.join(File.dirname(file), 'lib') + end end - lib || LOGGER.warn('No torch module found for python3, pytorch backend will not be enabled') lib end From 6adfc70e5e6213aea27bed668b84397bb41b65fd Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Mon, 28 Sep 2026 10:10:49 -0500 Subject: [PATCH 12/28] integration test --- .github/workflows/presubmit.yml | 5 +++++ integration_tests/backend_pytorch.bats | 7 +++++++ integration_tests/pytorch_example.py | 2 ++ 3 files changed, 14 insertions(+) create mode 100644 integration_tests/backend_pytorch.bats create mode 100644 integration_tests/pytorch_example.py diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index ebacd5766..eedec1be1 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -119,6 +119,11 @@ jobs: python3 -m pip install --user -U pip python3 -m pip install --user -U ittapi echo "$HOME/.local/bin" >> "$GITHUB_PATH" + - name: Install PyTorch (CPU) + run: | + python3 -m pip install --user -U pip + python3 -m pip install --user -U torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu + echo "$HOME/.local/bin" >> "$GITHUB_PATH" - name: Setup tmate session uses: mxschmitt/action-tmate@v3 if: ${{ github.event_name == 'workflow_dispatch' && inputs.debug_enabled }} diff --git a/integration_tests/backend_pytorch.bats b/integration_tests/backend_pytorch.bats new file mode 100644 index 000000000..fe86d31aa --- /dev/null +++ b/integration_tests/backend_pytorch.bats @@ -0,0 +1,7 @@ +bats_require_minimum_version 1.5.0 + +@test "PyTorch: trace contains aten::empty.memory_format" { + iprof --backends pytorch --analysis-output ./pytorch_out.txt -- \ + python3 ./integration_tests/pytorch_example.py + grep "aten::empty.memory_format" ./pytorch_out.txt +} diff --git a/integration_tests/pytorch_example.py b/integration_tests/pytorch_example.py new file mode 100644 index 000000000..2cb34934b --- /dev/null +++ b/integration_tests/pytorch_example.py @@ -0,0 +1,2 @@ +import torch +torch.empty(3) From c32177af96b037ac2d93f93a27b84c9221978e3c Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Mon, 28 Sep 2026 18:51:29 +0000 Subject: [PATCH 13/28] formating --- backends/pytorch/tracer_pytorch.cpp | 2 +- utils/xprof_utils.hpp | 10 +++++++--- xprof/xprof.rb.in | 8 ++++---- 3 files changed, 12 insertions(+), 8 deletions(-) diff --git a/backends/pytorch/tracer_pytorch.cpp b/backends/pytorch/tracer_pytorch.cpp index 0b6ecc3dc..ebbb659a0 100644 --- a/backends/pytorch/tracer_pytorch.cpp +++ b/backends/pytorch/tracer_pytorch.cpp @@ -1,5 +1,5 @@ -#include #include "pytorch_tracepoints.h" +#include static std::unique_ptr on_entry(const at::RecordFunction &fn) { tracepoint(lttng_ust_pytorch, op_entry, fn.name(), fn.overload_name()); diff --git a/utils/xprof_utils.hpp b/utils/xprof_utils.hpp index bb256d51d..daec319d1 100644 --- a/utils/xprof_utils.hpp +++ b/utils/xprof_utils.hpp @@ -77,14 +77,18 @@ typedef std::tuple hpt_t; typedef std::tuple hpt_function_name_t; typedef std::tuple t_function_name_t; -typedef std::tuple hpt_device_function_name_t; typedef std::tuple hp_device_t; typedef std::tuple hp_dsd_t; -typedef std::tuple hi_t; // host + NIC interface -typedef std::tuple hic_t; // host + NIC interface + counter +typedef std::tuple hi_t; // host + NIC interface +typedef std::tuple hic_t; // host + NIC interface + counter typedef std::tuple sd_t; typedef std::tuple tfn_ts_t; typedef std::tuple fn_ts_t; diff --git a/xprof/xprof.rb.in b/xprof/xprof.rb.in index ad47b27ea..ebc16acde 100755 --- a/xprof/xprof.rb.in +++ b/xprof/xprof.rb.in @@ -153,9 +153,9 @@ def whichlib64(binary, *libs) whichlib64_bin = File.join(BINDIR, 'whichlib64') ld_path = [env_fetch_first('LD_LIBRARY_PATH'), env_fetch_first('CRAY_LD_LIBRARY_PATH')].compact.join(':') - stdout_str, _, _ = exec("#{whichlib64_bin} #{binary} #{libs.join(' ')}", - opts: { 'LD_LIBRARY_PATH' => ld_path }, - ignore_exit_codes: [1, 2]) + stdout_str, = exec("#{whichlib64_bin} #{binary} #{libs.join(' ')}", + opts: { 'LD_LIBRARY_PATH' => ld_path }, + ignore_exit_codes: [1, 2]) libs.zip(stdout_str.lines.map(&:strip)).filter_map do |lib, path| [lib, path] unless path.end_with?('not found') @@ -892,7 +892,7 @@ def get_pytorch_library_path lib = env_fetch_first('LTTNG_UST_PYTORCH_LIBRARY_PATH') || begin # Raises for other error codes, but only warns for 127 (python3 command not found) and 1 (ModuleNotFoundError) - stdout_str, _, status = exec(torch_probe, ignore_exit_codes: [1, 127]) + stdout_str, _, status = exec(torch_probe, ignore_exit_codes: [1, 127]) if status.success? file, version = stdout_str.split version.slice(/\d+(\.\d+)*/).then do |v| From 02a08e1044559f2217d0c9941d4a85329e1af4c5 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 13:36:25 +0000 Subject: [PATCH 14/28] debug --- .github/workflows/presubmit.yml | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index eedec1be1..39b6a07f3 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -124,15 +124,25 @@ jobs: python3 -m pip install --user -U pip python3 -m pip install --user -U torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu echo "$HOME/.local/bin" >> "$GITHUB_PATH" - - name: Setup tmate session - uses: mxschmitt/action-tmate@v3 - if: ${{ github.event_name == 'workflow_dispatch' && inputs.debug_enabled }} + # --- diagnostics: which python3 the iprof probe uses, and whether the tracer can resolve against it --- + echo "== which python3 =="; which -a python3 + echo "== python3 --version =="; python3 --version + echo "== probe as iprof runs it =="; python3 -c "import torch; print(torch.__file__); print(torch.__version__)" + TORCH_LIB="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib"))')" + echo "TORCH_LIB=$TORCH_LIB"; ls -la "$TORCH_LIB"/libtorch_cpu.so + echo "== does libtorch_cpu.so export at::addGlobalCallback? ==" + nm -D "$TORCH_LIB"/libtorch_cpu.so | grep 17addGlobalCallback || echo "NOT EXPORTED / not found in dynsym" - name: Integration test run: | export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} bats ${TEST_EXCLUDES} integration_tests/ # Run the tests twice to make sure we do a proper cleanup bats ${TEST_EXCLUDES} integration_tests/ + - name: Debug over SSH on failure (tmate) + uses: mxschmitt/action-tmate@v3 + if: ${{ failure() && github.event_name == 'workflow_dispatch' && inputs.debug_enabled }} + with: + limit-access-to-actor: true build-in-tree-and-check: needs: efficios_dep name: Build in Tree ubuntu-24.04 From c13d4e85100dc5dbe1f19e449bf83d65e97bc51b Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 14:00:33 +0000 Subject: [PATCH 15/28] ci: add pytorch tracer load diagnostics on failure Add a failure-gated step that reproduces the failing iprof command, traces loader binding/relocation with LD_DEBUG, checks libtorch_cpu.so dep resolution, and runs an isolated preload without iprof env-building. Co-Authored-By: Claude Opus 4.8 --- .github/workflows/presubmit.yml | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 39b6a07f3..e1c1ba273 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -138,6 +138,22 @@ jobs: bats ${TEST_EXCLUDES} integration_tests/ # Run the tests twice to make sure we do a proper cleanup bats ${TEST_EXCLUDES} integration_tests/ + - name: Diagnose pytorch tracer load on failure + if: ${{ failure() }} + run: | + export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} + LT="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib","libtorch_cpu.so"))')" + echo "== reproduce failing command ==" + iprof -- bash -c "exit 55" && rc=$? || rc=$?; echo "iprof exit=$rc" + echo "== which libtorch_cpu.so binds + any undefined symbol at load ==" + LD_DEBUG=libs,bindings,reloc iprof -- bash -c "exit 55" 2>&1 \ + | grep -iE "libtorch_cpu|addGlobalCallback|error|not found|undefined" | head -60 || true + echo "== do libtorch_cpu.so's own deps resolve? ==" + ldd "$LT" | grep -i "not found" || echo "all deps of libtorch_cpu.so resolve" + echo "== isolated preload (no iprof env-building) ==" + LD_LIBRARY_PATH="$(dirname "$LT"):$LD_LIBRARY_PATH" \ + LD_PRELOAD="$(pkg-config --variable=libdir thapi)/libTracerPytorch.so" \ + bash -c "exit 55" && rc=$? || rc=$?; echo "isolated exit=$rc" - name: Debug over SSH on failure (tmate) uses: mxschmitt/action-tmate@v3 if: ${{ failure() && github.event_name == 'workflow_dispatch' && inputs.debug_enabled }} From 4a1a3f381e489d203b7a7894798f20f9223c24f9 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 14:01:49 +0000 Subject: [PATCH 16/28] ci: document how to trigger the tmate SSH debug step Co-Authored-By: Claude Opus 4.8 --- .github/workflows/presubmit.yml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index e1c1ba273..5614b9502 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -154,6 +154,10 @@ jobs: LD_LIBRARY_PATH="$(dirname "$LT"):$LD_LIBRARY_PATH" \ LD_PRELOAD="$(pkg-config --variable=libdir thapi)/libTracerPytorch.so" \ bash -c "exit 55" && rc=$? || rc=$?; echo "isolated exit=$rc" + # To get an SSH shell on the runner in the failed state: + # GitHub -> Actions -> Presubmit -> "Run workflow" -> tick "debug_enabled" -> Run. + # On failure this pauses the job and prints an "ssh ...@nyc1.tmate.io" line in the log. + # Connect with that command; access is limited to the actor who launched the run. - name: Debug over SSH on failure (tmate) uses: mxschmitt/action-tmate@v3 if: ${{ failure() && github.event_name == 'workflow_dispatch' && inputs.debug_enabled }} From d6621602b761d43d64a7b081729459df8db1d98a Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 14:13:39 +0000 Subject: [PATCH 17/28] ci: time-box integration test so hangs trigger diagnostics A hung tracee is not a step failure, so the failure-gated diagnostic step never ran. Add timeout-minutes to the integration test to convert a hang into a failure, and wrap the diagnostic iprof calls with timeout so they cannot hang the diagnostic step either. Co-Authored-By: Claude Opus 4.8 --- .github/workflows/presubmit.yml | 14 +++++++++----- 1 file changed, 9 insertions(+), 5 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 5614b9502..ca80d22b5 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -132,7 +132,10 @@ jobs: echo "TORCH_LIB=$TORCH_LIB"; ls -la "$TORCH_LIB"/libtorch_cpu.so echo "== does libtorch_cpu.so export at::addGlobalCallback? ==" nm -D "$TORCH_LIB"/libtorch_cpu.so | grep 17addGlobalCallback || echo "NOT EXPORTED / not found in dynsym" + # timeout-minutes converts a hang (e.g. the pytorch tracer deadlocking the + # tracee) into a step failure, so the diagnostic step below actually runs. - name: Integration test + timeout-minutes: 10 run: | export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} bats ${TEST_EXCLUDES} integration_tests/ @@ -140,20 +143,21 @@ jobs: bats ${TEST_EXCLUDES} integration_tests/ - name: Diagnose pytorch tracer load on failure if: ${{ failure() }} + timeout-minutes: 5 run: | export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} LT="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib","libtorch_cpu.so"))')" - echo "== reproduce failing command ==" - iprof -- bash -c "exit 55" && rc=$? || rc=$?; echo "iprof exit=$rc" + echo "== reproduce failing command (30s cap; 124 = timed out/hung) ==" + timeout 30 iprof -- bash -c "exit 55" && rc=$? || rc=$?; echo "iprof exit=$rc" echo "== which libtorch_cpu.so binds + any undefined symbol at load ==" - LD_DEBUG=libs,bindings,reloc iprof -- bash -c "exit 55" 2>&1 \ + LD_DEBUG=libs,bindings,reloc timeout 30 iprof -- bash -c "exit 55" 2>&1 \ | grep -iE "libtorch_cpu|addGlobalCallback|error|not found|undefined" | head -60 || true echo "== do libtorch_cpu.so's own deps resolve? ==" ldd "$LT" | grep -i "not found" || echo "all deps of libtorch_cpu.so resolve" - echo "== isolated preload (no iprof env-building) ==" + echo "== isolated preload (no iprof env-building; 30s cap) ==" LD_LIBRARY_PATH="$(dirname "$LT"):$LD_LIBRARY_PATH" \ LD_PRELOAD="$(pkg-config --variable=libdir thapi)/libTracerPytorch.so" \ - bash -c "exit 55" && rc=$? || rc=$?; echo "isolated exit=$rc" + timeout 30 bash -c "exit 55" && rc=$? || rc=$?; echo "isolated exit=$rc" # To get an SSH shell on the runner in the failed state: # GitHub -> Actions -> Presubmit -> "Run workflow" -> tick "debug_enabled" -> Run. # On failure this pauses the job and prints an "ssh ...@nyc1.tmate.io" line in the log. From 6130340ea80672253ec0c0f26b884d5b4d1e2870 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 14:14:36 +0000 Subject: [PATCH 18/28] ci: shorten debug timeouts (integration test 2m, probes 15s) Co-Authored-By: Claude Opus 4.8 --- .github/workflows/presubmit.yml | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index ca80d22b5..07da9c6b1 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -135,7 +135,7 @@ jobs: # timeout-minutes converts a hang (e.g. the pytorch tracer deadlocking the # tracee) into a step failure, so the diagnostic step below actually runs. - name: Integration test - timeout-minutes: 10 + timeout-minutes: 2 run: | export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} bats ${TEST_EXCLUDES} integration_tests/ @@ -143,21 +143,21 @@ jobs: bats ${TEST_EXCLUDES} integration_tests/ - name: Diagnose pytorch tracer load on failure if: ${{ failure() }} - timeout-minutes: 5 + timeout-minutes: 2 run: | export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} LT="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib","libtorch_cpu.so"))')" - echo "== reproduce failing command (30s cap; 124 = timed out/hung) ==" - timeout 30 iprof -- bash -c "exit 55" && rc=$? || rc=$?; echo "iprof exit=$rc" + echo "== reproduce failing command (15s cap; 124 = timed out/hung) ==" + timeout 15 iprof -- bash -c "exit 55" && rc=$? || rc=$?; echo "iprof exit=$rc" echo "== which libtorch_cpu.so binds + any undefined symbol at load ==" - LD_DEBUG=libs,bindings,reloc timeout 30 iprof -- bash -c "exit 55" 2>&1 \ + LD_DEBUG=libs,bindings,reloc timeout 15 iprof -- bash -c "exit 55" 2>&1 \ | grep -iE "libtorch_cpu|addGlobalCallback|error|not found|undefined" | head -60 || true echo "== do libtorch_cpu.so's own deps resolve? ==" ldd "$LT" | grep -i "not found" || echo "all deps of libtorch_cpu.so resolve" - echo "== isolated preload (no iprof env-building; 30s cap) ==" + echo "== isolated preload (no iprof env-building; 15s cap) ==" LD_LIBRARY_PATH="$(dirname "$LT"):$LD_LIBRARY_PATH" \ LD_PRELOAD="$(pkg-config --variable=libdir thapi)/libTracerPytorch.so" \ - timeout 30 bash -c "exit 55" && rc=$? || rc=$?; echo "isolated exit=$rc" + timeout 15 bash -c "exit 55" && rc=$? || rc=$?; echo "isolated exit=$rc" # To get an SSH shell on the runner in the failed state: # GitHub -> Actions -> Presubmit -> "Run workflow" -> tick "debug_enabled" -> Run. # On failure this pauses the job and prints an "ssh ...@nyc1.tmate.io" line in the log. From f5cef5c706e5c036c10dc7c94f0b6c80fefe9bb4 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 14:38:03 +0000 Subject: [PATCH 19/28] ci: make pytorch tracer diagnostics hang-proof Detach each probe (setsid + timeout -s KILL) and redirect its stdio to a file so lingering lttng children cannot hold the step pipe open, which previously left the diagnostic step blocked until its timeout with no output. Run under plain bash so a non-zero probe does not abort the step. Co-Authored-By: Claude Opus 4.8 --- .github/workflows/presubmit.yml | 32 +++++++++++++++++++++++--------- 1 file changed, 23 insertions(+), 9 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 07da9c6b1..2f6e162d0 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -144,20 +144,34 @@ jobs: - name: Diagnose pytorch tracer load on failure if: ${{ failure() }} timeout-minutes: 2 + shell: bash # plain bash, no -e, so a non-zero probe does not abort the step run: | export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} - LT="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib","libtorch_cpu.so"))')" - echo "== reproduce failing command (15s cap; 124 = timed out/hung) ==" - timeout 15 iprof -- bash -c "exit 55" && rc=$? || rc=$?; echo "iprof exit=$rc" - echo "== which libtorch_cpu.so binds + any undefined symbol at load ==" - LD_DEBUG=libs,bindings,reloc timeout 15 iprof -- bash -c "exit 55" 2>&1 \ - | grep -iE "libtorch_cpu|addGlobalCallback|error|not found|undefined" | head -60 || true + LT="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib","libtorch_cpu.so"))' 2>/dev/null)" + # Run a command fully detached: its own session, all stdio to a file (so + # lingering lttng children can never hold the step pipe open), SIGKILL on + # timeout (-s KILL, since iprof ignores SIGTERM during cleanup). + run() { # $1=logfile rest=command + local log="$1"; shift + setsid timeout -s KILL 15 "$@" "$log" 2>&1 + local rc=$? + echo "[exit=$rc] (124=timed out/hung, 127=undefined symbol, 55=ok)" + return 0 + } + echo "== reproduce failing command ==" + run a.log iprof -- bash -c "exit 55"; sed -n '1,40p' a.log + echo "== loader trace: which libtorch_cpu.so binds + undefined symbols ==" + LD_DEBUG=libs,bindings,reloc run b.log iprof -- bash -c "exit 55" + grep -iE "libtorch_cpu|addGlobalCallback|not found|undefined|symbol" b.log | head -60 || echo "(no matching loader lines)" echo "== do libtorch_cpu.so's own deps resolve? ==" - ldd "$LT" | grep -i "not found" || echo "all deps of libtorch_cpu.so resolve" - echo "== isolated preload (no iprof env-building; 15s cap) ==" + ldd "$LT" 2>&1 | grep -i "not found" || echo "all deps of libtorch_cpu.so resolve" + echo "== isolated preload (no iprof env-building) ==" LD_LIBRARY_PATH="$(dirname "$LT"):$LD_LIBRARY_PATH" \ LD_PRELOAD="$(pkg-config --variable=libdir thapi)/libTracerPytorch.so" \ - timeout 15 bash -c "exit 55" && rc=$? || rc=$?; echo "isolated exit=$rc" + run c.log bash -c "exit 55"; sed -n '1,40p' c.log + # reap any tracer daemon still alive so the step can end promptly + pkill -KILL -f lttng-sessiond 2>/dev/null || true + echo "== diagnostics done ==" # To get an SSH shell on the runner in the failed state: # GitHub -> Actions -> Presubmit -> "Run workflow" -> tick "debug_enabled" -> Run. # On failure this pauses the job and prints an "ssh ...@nyc1.tmate.io" line in the log. From 579bb97d46b252430260cc35392ce015495cc28f Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 14:58:20 +0000 Subject: [PATCH 20/28] ci: disable errexit in diagnostics step so probes actually print GitHub's `shell: bash` keeps `-e -o pipefail`, so the first probe exiting 127 aborted the step before any diagnostic output. Set +e +o pipefail at the top of the script instead. Co-Authored-By: Claude Opus 4.8 --- .github/workflows/presubmit.yml | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 2f6e162d0..1885e1e26 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -144,8 +144,11 @@ jobs: - name: Diagnose pytorch tracer load on failure if: ${{ failure() }} timeout-minutes: 2 - shell: bash # plain bash, no -e, so a non-zero probe does not abort the step run: | + # GitHub's default shell keeps `-e -o pipefail`; turn it off so a + # non-zero probe (e.g. iprof exiting 127) does not abort the step + # before the diagnostics below print. + set +e +o pipefail export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} LT="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib","libtorch_cpu.so"))' 2>/dev/null)" # Run a command fully detached: its own session, all stdio to a file (so From 544170a0f58d5e85e35c18f0efad80d9f67641a7 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 10:53:39 -0500 Subject: [PATCH 21/28] Refactor presubmit workflow for debugging Removed integration test step and added tmate session setup for debugging. --- .github/workflows/presubmit.yml | 70 +++++++++++---------------------- 1 file changed, 22 insertions(+), 48 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 1885e1e26..7e1d9583e 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -134,56 +134,30 @@ jobs: nm -D "$TORCH_LIB"/libtorch_cpu.so | grep 17addGlobalCallback || echo "NOT EXPORTED / not found in dynsym" # timeout-minutes converts a hang (e.g. the pytorch tracer deadlocking the # tracee) into a step failure, so the diagnostic step below actually runs. - - name: Integration test - timeout-minutes: 2 - run: | - export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} - bats ${TEST_EXCLUDES} integration_tests/ - # Run the tests twice to make sure we do a proper cleanup - bats ${TEST_EXCLUDES} integration_tests/ - - name: Diagnose pytorch tracer load on failure - if: ${{ failure() }} - timeout-minutes: 2 - run: | - # GitHub's default shell keeps `-e -o pipefail`; turn it off so a - # non-zero probe (e.g. iprof exiting 127) does not abort the step - # before the diagnostics below print. - set +e +o pipefail - export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} - LT="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib","libtorch_cpu.so"))' 2>/dev/null)" - # Run a command fully detached: its own session, all stdio to a file (so - # lingering lttng children can never hold the step pipe open), SIGKILL on - # timeout (-s KILL, since iprof ignores SIGTERM during cleanup). - run() { # $1=logfile rest=command - local log="$1"; shift - setsid timeout -s KILL 15 "$@" "$log" 2>&1 - local rc=$? - echo "[exit=$rc] (124=timed out/hung, 127=undefined symbol, 55=ok)" - return 0 - } - echo "== reproduce failing command ==" - run a.log iprof -- bash -c "exit 55"; sed -n '1,40p' a.log - echo "== loader trace: which libtorch_cpu.so binds + undefined symbols ==" - LD_DEBUG=libs,bindings,reloc run b.log iprof -- bash -c "exit 55" - grep -iE "libtorch_cpu|addGlobalCallback|not found|undefined|symbol" b.log | head -60 || echo "(no matching loader lines)" - echo "== do libtorch_cpu.so's own deps resolve? ==" - ldd "$LT" 2>&1 | grep -i "not found" || echo "all deps of libtorch_cpu.so resolve" - echo "== isolated preload (no iprof env-building) ==" - LD_LIBRARY_PATH="$(dirname "$LT"):$LD_LIBRARY_PATH" \ - LD_PRELOAD="$(pkg-config --variable=libdir thapi)/libTracerPytorch.so" \ - run c.log bash -c "exit 55"; sed -n '1,40p' c.log - # reap any tracer daemon still alive so the step can end promptly - pkill -KILL -f lttng-sessiond 2>/dev/null || true - echo "== diagnostics done ==" - # To get an SSH shell on the runner in the failed state: - # GitHub -> Actions -> Presubmit -> "Run workflow" -> tick "debug_enabled" -> Run. - # On failure this pauses the job and prints an "ssh ...@nyc1.tmate.io" line in the log. - # Connect with that command; access is limited to the actor who launched the run. - - name: Debug over SSH on failure (tmate) + # - name: Integration test + # timeout-minutes: 2 + # run: | + # export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} + # bats ${TEST_EXCLUDES} integration_tests/ + # # Run the tests twice to make sure we do a proper cleanup + # bats ${TEST_EXCLUDES} integration_tests/ + + - name: Setup tmate session + id: tmate uses: mxschmitt/action-tmate@v3 - if: ${{ failure() && github.event_name == 'workflow_dispatch' && inputs.debug_enabled }} with: - limit-access-to-actor: true + detached: true + + # This step prints the credentials cleanly using standard echo + - name: Print tmate connection details + run: | + echo "SSH: ${{ steps.tmate.outputs.ssh-command }}" + echo "Web URL: ${{ steps.tmate.outputs.web-url }}" + + # Manually pause the workflow so it doesn't instantly close + - name: Keep alive + run: sleep 1800 # 30 minutes + build-in-tree-and-check: needs: efficios_dep name: Build in Tree ubuntu-24.04 From 398dc705945b33506006f01bf5a143ce2e733959 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 11:08:18 -0500 Subject: [PATCH 22/28] Remove tmate session setup from presubmit workflow Removed tmate session setup and related steps from the workflow. --- .github/workflows/presubmit.yml | 22 +++++----------------- 1 file changed, 5 insertions(+), 17 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 7e1d9583e..8ac507658 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -132,6 +132,11 @@ jobs: echo "TORCH_LIB=$TORCH_LIB"; ls -la "$TORCH_LIB"/libtorch_cpu.so echo "== does libtorch_cpu.so export at::addGlobalCallback? ==" nm -D "$TORCH_LIB"/libtorch_cpu.so | grep 17addGlobalCallback || echo "NOT EXPORTED / not found in dynsym" + - name: Run iprof debug + run: | + run -55 iprof --debug 0 -- bash -c "exit 55" + - name: Setup upterm session + uses: owenthereal/action-upterm@v2 # timeout-minutes converts a hang (e.g. the pytorch tracer deadlocking the # tracee) into a step failure, so the diagnostic step below actually runs. # - name: Integration test @@ -141,23 +146,6 @@ jobs: # bats ${TEST_EXCLUDES} integration_tests/ # # Run the tests twice to make sure we do a proper cleanup # bats ${TEST_EXCLUDES} integration_tests/ - - - name: Setup tmate session - id: tmate - uses: mxschmitt/action-tmate@v3 - with: - detached: true - - # This step prints the credentials cleanly using standard echo - - name: Print tmate connection details - run: | - echo "SSH: ${{ steps.tmate.outputs.ssh-command }}" - echo "Web URL: ${{ steps.tmate.outputs.web-url }}" - - # Manually pause the workflow so it doesn't instantly close - - name: Keep alive - run: sleep 1800 # 30 minutes - build-in-tree-and-check: needs: efficios_dep name: Build in Tree ubuntu-24.04 From 195bf3f6fc72c41b493e9bc78aebc786d4c46282 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 11:18:38 -0500 Subject: [PATCH 23/28] Remove iprof debug step from presubmit workflow Removed the iprof debug step from the workflow. --- .github/workflows/presubmit.yml | 3 --- 1 file changed, 3 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 8ac507658..9f16bec9e 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -132,9 +132,6 @@ jobs: echo "TORCH_LIB=$TORCH_LIB"; ls -la "$TORCH_LIB"/libtorch_cpu.so echo "== does libtorch_cpu.so export at::addGlobalCallback? ==" nm -D "$TORCH_LIB"/libtorch_cpu.so | grep 17addGlobalCallback || echo "NOT EXPORTED / not found in dynsym" - - name: Run iprof debug - run: | - run -55 iprof --debug 0 -- bash -c "exit 55" - name: Setup upterm session uses: owenthereal/action-upterm@v2 # timeout-minutes converts a hang (e.g. the pytorch tracer deadlocking the From 17d93608ce5efe8966ea05a16f62a7cf3e7e3c88 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 12:01:14 -0500 Subject: [PATCH 24/28] Update LDFLAGS for libTracerPytorch --- backends/pytorch/Makefile.am | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backends/pytorch/Makefile.am b/backends/pytorch/Makefile.am index 33f70c516..633c57471 100644 --- a/backends/pytorch/Makefile.am +++ b/backends/pytorch/Makefile.am @@ -72,7 +72,7 @@ libTracerPytorch_la_SOURCES = \ libTracerPytorch_la_CPPFLAGS = -I$(top_srcdir)/utils -I$(top_srcdir)/utils/include -I$(srcdir)/include -I./ libTracerPytorch_la_CXXFLAGS = -std=c++17 -Wall -Wextra -Wno-unused-parameter -Wno-missing-field-initializers -Wno-unused-private-field $(WERROR) $(LTTNG_UST_CFLAGS) -libTracerPytorch_la_LDFLAGS = $(LTTNG_UST_LIBS) -Ldummy_libs -ltorch_cpu -avoid-version -module +libTracerPytorch_la_LDFLAGS = $(LTTNG_UST_LIBS) -Ldummy_libs -Wl,--no-as-needed -ltorch_cpu -Wl,--as-needed -avoid-version -module libTracerPytorch_la_LIBADD = libpytorchtracepoints.la EXTRA_libTracerPytorch_la_DEPENDENCIES = $(DUMMY_TORCH_LIBS) From 5c5271a97d5e83c453fed58961ff20bea1daf4cb Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 13:43:53 -0500 Subject: [PATCH 25/28] Fix LDFLAGS syntax in Makefile.am --- backends/pytorch/Makefile.am | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/backends/pytorch/Makefile.am b/backends/pytorch/Makefile.am index 633c57471..96192d5f2 100644 --- a/backends/pytorch/Makefile.am +++ b/backends/pytorch/Makefile.am @@ -72,7 +72,7 @@ libTracerPytorch_la_SOURCES = \ libTracerPytorch_la_CPPFLAGS = -I$(top_srcdir)/utils -I$(top_srcdir)/utils/include -I$(srcdir)/include -I./ libTracerPytorch_la_CXXFLAGS = -std=c++17 -Wall -Wextra -Wno-unused-parameter -Wno-missing-field-initializers -Wno-unused-private-field $(WERROR) $(LTTNG_UST_CFLAGS) -libTracerPytorch_la_LDFLAGS = $(LTTNG_UST_LIBS) -Ldummy_libs -Wl,--no-as-needed -ltorch_cpu -Wl,--as-needed -avoid-version -module +libTracerPytorch_la_LDFLAGS = $(LTTNG_UST_LIBS) -Ldummy_libs -Wl,--no-as-needed,-ltorch_cpu,--as-needed -avoid-version -module libTracerPytorch_la_LIBADD = libpytorchtracepoints.la EXTRA_libTracerPytorch_la_DEPENDENCIES = $(DUMMY_TORCH_LIBS) From e19f8c264648b9483c7c06b92db1cdea64490933 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 20:14:29 +0000 Subject: [PATCH 26/28] remove debugging --- .github/workflows/presubmit.yml | 28 +++++++++------------------- 1 file changed, 9 insertions(+), 19 deletions(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 9f16bec9e..eedec1be1 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -124,25 +124,15 @@ jobs: python3 -m pip install --user -U pip python3 -m pip install --user -U torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu echo "$HOME/.local/bin" >> "$GITHUB_PATH" - # --- diagnostics: which python3 the iprof probe uses, and whether the tracer can resolve against it --- - echo "== which python3 =="; which -a python3 - echo "== python3 --version =="; python3 --version - echo "== probe as iprof runs it =="; python3 -c "import torch; print(torch.__file__); print(torch.__version__)" - TORCH_LIB="$(python3 -c 'import torch,os; print(os.path.join(os.path.dirname(torch.__file__),"lib"))')" - echo "TORCH_LIB=$TORCH_LIB"; ls -la "$TORCH_LIB"/libtorch_cpu.so - echo "== does libtorch_cpu.so export at::addGlobalCallback? ==" - nm -D "$TORCH_LIB"/libtorch_cpu.so | grep 17addGlobalCallback || echo "NOT EXPORTED / not found in dynsym" - - name: Setup upterm session - uses: owenthereal/action-upterm@v2 - # timeout-minutes converts a hang (e.g. the pytorch tracer deadlocking the - # tracee) into a step failure, so the diagnostic step below actually runs. - # - name: Integration test - # timeout-minutes: 2 - # run: | - # export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} - # bats ${TEST_EXCLUDES} integration_tests/ - # # Run the tests twice to make sure we do a proper cleanup - # bats ${TEST_EXCLUDES} integration_tests/ + - name: Setup tmate session + uses: mxschmitt/action-tmate@v3 + if: ${{ github.event_name == 'workflow_dispatch' && inputs.debug_enabled }} + - name: Integration test + run: | + export PKG_CONFIG_PATH=./build/ici/lib/pkgconfig/:${PKG_CONFIG_PATH} + bats ${TEST_EXCLUDES} integration_tests/ + # Run the tests twice to make sure we do a proper cleanup + bats ${TEST_EXCLUDES} integration_tests/ build-in-tree-and-check: needs: efficios_dep name: Build in Tree ubuntu-24.04 From 1afdaef626c23576d274a20b81e2a40b7bf8bc21 Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 15:23:44 -0500 Subject: [PATCH 27/28] Add PyTorch installation to presubmit workflow Added steps to install PyTorch and update GITHUB_PATH. --- .github/workflows/presubmit.yml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index abf42ce13..898c8c594 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -118,6 +118,12 @@ jobs: run: | python3 -m pip install --user -U pip python3 -m pip install --user -U ittapi + - name: Install PyTorch (CPU) + run: | + python3 -m pip install --user -U pip + python3 -m pip install --user -U torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu + - name: Add $HOME/.local/bin to the GITHUB_PATH + run: | echo "$HOME/.local/bin" >> "$GITHUB_PATH" - name: Setup remote-ssh session uses: owenthereal/action-upterm@v2 From 22e97c80ed1e043c7712b4db5a23946a978a00dd Mon Sep 17 00:00:00 2001 From: Aurelio Vivas Date: Tue, 29 Sep 2026 15:52:31 -0500 Subject: [PATCH 28/28] Update PyTorch installation in presubmit workflow Updated PyTorch installation to use the latest version. --- .github/workflows/presubmit.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/presubmit.yml b/.github/workflows/presubmit.yml index 898c8c594..419ae6fda 100644 --- a/.github/workflows/presubmit.yml +++ b/.github/workflows/presubmit.yml @@ -121,7 +121,7 @@ jobs: - name: Install PyTorch (CPU) run: | python3 -m pip install --user -U pip - python3 -m pip install --user -U torch==2.14.0 --index-url https://download.pytorch.org/whl/cpu + python3 -m pip install --user -U torch - name: Add $HOME/.local/bin to the GITHUB_PATH run: | echo "$HOME/.local/bin" >> "$GITHUB_PATH"