diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index cf2b6278..21ce8190 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -17,6 +17,15 @@ jobs: go-version-file: go.mod check-latest: true + - name: Check formatting + run: | + unformatted=$(gofmt -l $(git ls-files '*.go')) + if [ -n "$unformatted" ]; then + echo "These files are not gofmt-formatted; run 'make fmt':" + echo "$unformatted" + exit 1 + fi + - name: Vet run: go vet ./... diff --git a/.gitignore b/.gitignore index b0eee921..cc8ff868 100644 --- a/.gitignore +++ b/.gitignore @@ -52,3 +52,10 @@ gputrace_bin # Local watch/debug artifacts watch_* beads.svg +brain/ + +# Capture artifacts +*.pprof +*.pftrace +*.gpucapture/ +.hermes/ diff --git a/Makefile b/Makefile index 4fba7deb..8ba21896 100644 --- a/Makefile +++ b/Makefile @@ -6,19 +6,51 @@ AXPERMS_BIN := $(HOME)/go/bin/axperms BUNDLE_ID := com.tmc.gputrace AXPERMS_BUNDLE_ID := com.github.tmc.gputrace.axperms -.PHONY: all build test vet install reinstall clean sign-bundle setup-permissions reset-permissions fullreinstall reset test-permissions axperms setup-axperms help +.PHONY: all build test vet fmt checkfmt check install reinstall clean sign-bundle setup-permissions reset-permissions fullreinstall reset test-permissions axperms setup-axperms help + +GO_FILES = $(shell git ls-files '*.go') all: build build: go install ./cmd/gputrace +# test reports how many cases it skipped, because "ok" and a skip are the same +# line in go test's default output. Most of the skips are opt-in integration +# tests waiting on an environment variable (see docs/TESTING.md); a suite that +# says ok while a third of it sat out is telling the truth and meaning less +# than it reads. test: - go test ./... + @go test -json ./... | go run ./internal/cmd/testcensus + +# test-gated reports which opt-in variables are set and which are not, so the +# skip count above is attributable rather than merely known. +test-gated: + @echo "Opt-in test variables (see docs/TESTING.md):" + @for v in TRACE_PROCESSOR_SHELL GPUTRACE_TEST_TRACE GPUTRACE_MIO_SETUP_DATA_PATH \ + GPUTRACE_PERF_FIXTURE GPUTRACE_TEST_METALLIB GPUTRACE_PARITY_TRACE; do \ + eval val=\$$$$v; \ + if [ -n "$$val" ]; then echo " set $$v=$$val"; else echo " unset $$v"; fi; \ + done + @echo "Unset variables leave their tests skipped; go test reports that as ok." vet: go vet ./... +fmt: + gofmt -w $(GO_FILES) + +# checkfmt fails instead of rewriting, so CI and pre-push can use it. +checkfmt: + @unformatted=$$(gofmt -l $(GO_FILES)); \ + if [ -n "$$unformatted" ]; then \ + echo "These files are not gofmt-formatted; run 'make fmt':"; \ + echo "$$unformatted"; \ + exit 1; \ + fi + +check: checkfmt vet test + install: clean build setup-permissions @echo "Reinstall complete with fresh permissions" @@ -149,6 +181,9 @@ help: @echo " build - Build gputrace" @echo " test - Run Go tests" @echo " vet - Run go vet" + @echo " fmt - Rewrite tracked Go files with gofmt" + @echo " checkfmt - Fail if any tracked Go file is unformatted" + @echo " check - checkfmt + vet + test" @echo " reinstall - Rebuild binary and refresh signed app bundle" @echo " fullreinstall - Clean + rebuild + fresh permissions (resets TCC)" @echo " clean - Remove app bundle (forces macgo to recreate)" diff --git a/README.md b/README.md index d932ff1f..117f4327 100644 --- a/README.md +++ b/README.md @@ -27,13 +27,285 @@ gputrace profiler trace.gputrace gputrace pprof trace.gputrace -o trace.pb go tool pprof -http=:8080 trace.pb -# View text timeline or export Chrome/Perfetto timeline -gputrace timeline trace.gputrace --format perfetto -o trace.json +# Export the readable, cumulative-GPU-busy native Perfetto timeline (default) +gputrace timeline trace.gputrace --format perfetto -o trace.pftrace + +# Inspect command-buffer scheduling on its separate wall-clock axis +gputrace timeline trace.gputrace --format perfetto --clock wall -o command-buffers.pftrace # Compare two traces gputrace diff A.gputrace B.gputrace --explain + +# Permit descriptive deltas when exact environment evidence is unavailable +gputrace diff A.gputrace B.gputrace --allow-cross-environment + +# Serve a native trace through the hosted Perfetto UI without uploading it +gputrace timeline trace.gputrace --format perfetto --open --remote-ui + +# Reproducible mode with a pinned local Perfetto UI build +# The directory must contain index.html and perfetto-ui.json; see below. +gputrace timeline trace.gputrace --format perfetto --open \ + --ui-dir /path/to/perfetto-ui + +# Focus one exact occurrence; repeated names require the occurrence flag +gputrace timeline trace.gputrace --format perfetto --open --remote-ui \ + --kernel rmsbfloat16 --kernel-occurrence 0 + +# Write stable PerfettoSQL views for trace_processor_shell +gputrace timeline trace.gputrace --format perfetto \ + --sql-out gputrace.sql -o trace.pftrace ``` +A local Perfetto UI directory must identify the upstream build in +`perfetto-ui.json`: + +```json +{"schema":"gputrace.perfetto-ui/v1","revision":"UPSTREAM_REVISION"} +``` + +Perfetto has one global time axis. `--clock busy` therefore contains encoders, +dispatches, and only counter series whose timestamps are proven in that +domain; `--clock wall` contains APSTimelineData command buffers and wall-clock +profiler events. Per-encoder APS GPU cycles and derived cost remain selectable +encoder details because their counter clock is not joined to the busy clock. +Lossless busy exports default to an Xcode-like `Shaders / pipelines` group, +with one measured dispatch lane per recorded pipeline and extra lanes only +when uses overlap. Compact encoder spans remain above it. A secondary +`Dispatch sequence by encoder` group contains only strictly contained +dispatches; unproven associations appear separately rather than as track +parentage. All are presentation duplicates: native `gpu_slice` rows remain the +accounting source, and constrained exports omit the duplicate rows. +When profiler timing is unavailable, capture launch records instead appear as +generic instant events with pipeline identity and dispatch geometry. They do +not enter `gpu_slice`, and CS/debug labels are reported separately as observed +annotations rather than encoder or dispatch instances. +Dispatch details and the `gputrace_pipeline` SQL view include every static +compiler statistic carried by the attributed pipeline record, including +register, spill, instruction-family, threadgroup, and compilation-time facts. +The `metrics_source` argument identifies the backing trace section. +The `gputrace_dispatch` view normalizes timing provenance, attribution, +geometry, source location, and profiler-sample coverage across measured GPU +events and capture-only launch records; unavailable fields remain `NULL`. +GPRWCNTR sample counts include their scaled-window attribution basis and raw +mach-absolute tick bounds; they are not presented as measured dispatch timing. +It also exposes capture command-buffer membership and byte offset as structural +identity. Those fields do not imply that wall-clock command-buffer spans +contain busy-clock dispatch intervals. +Per-launch SIMD groups remain dispatch facts. The `gputrace_function` view +holds one row per function for aggregate duration, total SIMD work, and work +share, avoiding repeated aggregates that produce misleading sums. Profiled +function names retain their `gpuCommandInfoData` attribution; capture-only +names retain their capture attribution. Source-reported aggregate SIMD work is +kept in separate `source_aggregate_*` columns and remains `NULL` when absent. +The `gputrace_encoder` view exposes profiled encoder timing and archive-backed +cycle aggregates, including their derivation, coverage, and unjoined counter +clock status. Capture-only traces do not manufacture encoder rows. +The `gputrace_command_buffer` view exposes measured APSTimelineData wall spans +when present. Capture-only command buffers retain their record index and byte +offset but leave wall timing `NULL`. Encoder detail tracks are ordered by their +first event, so multi-digit encoder names do not sort ahead of earlier work. +gputrace does not invent a mapping between these domains. + +See [MLX GPU Trace Rendering in Perfetto](docs/MLX_PERFETTO_RENDERING_SPEC.md) +for the native Perfetto roadmap and proposed MLX semantic view. +`--format perfetto` writes binary protobuf; `--format chrome` retains Chrome +Trace JSON compatibility. +The optional SQL file defines `gputrace_capture`, `gputrace_command_buffer`, `gputrace_dispatch`, +`gputrace_dispatch_arg`, `gputrace_encoder`, `gputrace_encoder_arg`, `gputrace_function`, `gputrace_pipeline`, `gputrace_semantic_node`, `gputrace_semantic_link`, +`gputrace_counter_series`, `gputrace_counter_encoder_sample`, +`gputrace_counter_encoder_aggregate`, +`gputrace_unattributed_counter`, +`gputrace_evidence_gap`, `gputrace_aps_data_blob`, +`gputrace_aps_data_key`, `gputrace_stream_data_archive_blob`, +`gputrace_stream_data_archive_key`, `gputrace_stream_data_table`, +`gputrace_stream_data_string`, `gputrace_pipeline_compiler`, +`gputrace_pipeline_compiler_remark`, +`gputrace_pipeline_compiler_remark_arg`, +`gputrace_semantic_label_conflict`, and +`gputrace_unmatched` views over the native trace. The file's header splits +every view into two tiers: the stable views named there are part of the v1 +contract, and the rest project recorded private archive structure for +inspection and may change with the decoder. The same split applies to the +evidence manifest; see +[the v1 contract surface](docs/MLX_PERFETTO_RENDERING_SPEC.md#the-v1-contract-surface). +`gputrace_capture` provides typed trace identity, environment, clock, coverage, +timing-summary, and loss-receipt columns. Timing columns keep encoder span, +dispatch span, command-buffer active time, command-buffer wall span, restore +timing, display duration, and optional Xcode Effective GPU Time distinct. +Profiled traces also carry the archive's exact Metal device name, Metal plugin +name, and GPU generation into canonical JSON, native GPU metadata, and typed +SQL columns. GPU generation is nullable so a recorded zero remains distinct +from absence. Capture-only traces report streamData identity as unavailable. +The same projection retains streamData archive version, source trace name, +timestamp, profiling-mode scalars, capture-range scalars, completeness flags, +and blit-call count. Private enum and range meanings remain explicitly +uninterpreted; recorded zero and false values remain distinct from absence. +For each fixed-record streamData table, the manifest and SQL expose byte +length, declared record size, computed record count, trailing remainder, and an +integrity status. The untimed `gputrace_stream_data_table` view also retains +each complete table as exact hexadecimal bytes with its whole-table SHA-256 +digest. Recorded size and count delimit rows; any trailing bytes remain in the +same payload. This makes truncation and record-layout mismatches visible while +leaving unknown words and cross-table relationships uninterpreted. +A missing table key and a malformed recorded reference are distinct: the +former reports absent, while the latter carries a decode error and emits no raw +table payload. A valid empty table retains the SHA-256 digest of empty input. +The untimed `gputrace_stream_data_string` view retains the archive's complete +ordered strings array, including empty entries and source paths. Its index is +only the source NSArray position; the view does not classify values or infer a +pipeline, function, source-file, clock, or timing relationship. +`gputrace_pipeline_compiler` retains one untimed static compiler-diagnostic row +per pipeline. It includes exact optimization remarks and nullable compile-stage +timings and cache status from streamData or capture store sections. Compiler +remarks may name source lines, but they are not measured instruction samples or +source-line GPU cost and are never repeated per dispatch. +The raw YAML remains authoritative. `gputrace_pipeline_compiler_remark` adds +one searchable row per YAML document with its kind, pass, name, function, and +source coordinate when recorded. A `Line: 0` sentinel is retained as +`unresolved_source_location`, not presented as a usable source line. The +manifest distinguishes recorded, resolved, unresolved, and malformed location +counts. +`gputrace_pipeline_compiler_remark_arg` expands the ordered scalar entries +beneath each remark's `Args` key. It preserves duplicate names, empty values, +the raw source line and scalar spelling, and a decoded string value. Values +remain strings: the view does not infer that a quoted number is a metric or +assign pass-specific meaning. Under a constrained export, argument rows are +retained with their parent remark; compare the source argument count in +`gputrace_capture` with the projected view count. +The manifest count is the decoded source count; under an explicit output budget +the SQL row count may be lower, with the difference covered by the export loss +receipt. +Each pipeline also carries the sorted exact names of all recorded top-level +statistics. SQL and dispatch details omit an absent metric, while preserving a +recorded zero or false. Opaque keys such as `ComputeBufferPrefetch` remain +presence-only evidence and are not assigned a meaning. +Archive-family inventory separately reports top-level APS, timeline, counter, +shader-profiler, GPU-timeline, and batch-filtered array entry counts. These are +presence counts, not decoded samples; an explicit zero differs from absence. +Source inventory counts remain stable across clock selection; separate +projected counts report what was placed on the selected axis. +When APSTimelineData supplies them, the same view exposes `absolute_time`, +`timebase_numer`, and `timebase_denom` with an explicit wall-domain source and +conversion formula. These fields convert source ticks within the wall domain; +they do not align the wall and cumulative GPU-busy timelines. Missing inputs +remain `NULL` with a `clock_conversion_availability` reason. +The raw `continuous_time` field is retained separately with an availability +receipt and an explicitly unverified clock relationship. gputrace does not use +it to move or align events. +The APSTimelineData `pstate` value is likewise retained as a raw replay +performance-state scalar. Its unit and operating-point mapping are not assumed; +the nullable representation preserves a recorded zero without mistaking it for +missing evidence. +Wall-clock exports retain each APSTimelineData `Restore Timestamps` range on a +separate replay-restore track. These intervals describe replay restore +activity, not GPU execution, and are queryable through the +`gputrace_restore_interval` PerfettoSQL view. +Busy-clock encoder rows retain APSCounterData batch and sample-index identities +when the TraceId tables cover that execution ordinal. The relationship is +positional only: TraceId values are not equated with GRC encoder or kick IDs, +and these identities do not join the counter and busy clocks. +`gputrace_manifest_arg` exposes every manifest field +as a key/value row, including per-class loss fields added by constrained +exports and fields introduced by newer exporters. +With `--clock wall --include-raw-samples`, `gputrace_profiler_stream` exposes +raw stream aggregates and `gputrace_raw_profiler_sample` exposes GPRWCNTR +source record ordinals, original mach-absolute ticks, the seven fixed GRC +fields, exact ShaderProfilerData source and ring-buffer identity, variable +record stride, and hardware-counter column count. Hardware +counter columns remain uninterpreted and are not exported as named metrics. +`gputrace_raw_profiler_sample_arg` retains each payload value by its recorded +zero-based ordinal without assigning a counter name, unit, or meaning. Its +decimal `raw_value_uint64` text preserves the full unsigned range; the +companion `raw_value_int64` column is Perfetto's signed integer projection. +The untimed `gputrace_counter_catalog` view preserves every recorded +APSCounterData pass-column name with its group and column ordinal. Names beyond +the seven fixed GRC fields remain opaque; the catalog supplies no unit, decoded +value series, encoder attribution, or clock mapping. +`gputrace_counter_trace_id` preserves each recorded APSCounterData TraceId, +batch ID, and sample index as untimed source evidence. Only the row ordinal has +a positional relation to encoder execution order; TraceId itself is not a GRC +encoder or kick ID and carries no timing relationship. `trace_id_uint64` +preserves its full unsigned decimal value; `trace_id_int64` is Perfetto's +signed SQL projection. +`gputrace_counter_encoder_aggregate` retains every capture-attributed +APSCounterData aggregate row, rather than collapsing all pass groups onto the +visible encoder list. Rows include the exact encoder and optional kick IDs, +pass group, execution ordinal, optional TraceId-derived batch and sample +index, sample and end-record counts, GPU cycles, and raw counter timestamp +range. Unsigned identifiers and counters have exact decimal columns alongside +Perfetto's signed projections. These are untimed evidence rows: their raw +counter clock has no verified mapping to busy or wall time, and the ordinal is +not a Metal encoder foreign key. A constrained export may retain fewer rows; +the manifest source count and loss receipt remain explicit. +`gputrace_counter_encoder_sample` retains every capture-attributed GPRWCNTR +source record before aggregation. Rows include the source blob and record +ordinals, fixed GRC fields, encoder-placement evidence, and the exact opaque +hardware-counter vector as JSON in recorded order. The archive does not prove +which `passList` names belong to each sample blob, so the view does not assign +counter names, units, derived meaning, a Metal encoder foreign key, or a +timeline coordinate. Full-range unsigned timestamps, cycles, and identifiers +have exact decimal columns alongside Perfetto's signed projections. +`gputrace_aps_data_blob` content-identifies every raw APSData archive entry and +reports its byte count, dictionary status, and root-key count. +`gputrace_aps_data_key` exposes each root key and its decoded structural value +kind in stable lexical order. These views make capture-shape differences +queryable without interpreting private values or duplicating the roughly +megabyte-scale raw blobs into the Perfetto trace. The manifest retains source +blob and key counts when an output budget samples the optional detail rows. +The generic `gputrace_stream_data_archive_blob` and +`gputrace_stream_data_archive_key` views apply the same contract to APSData, +APSTimelineData, and APSCounterData. Their `family` column makes source archive +counts, bytes, digests, and root shapes comparable even when higher-level +decoders produce different coverage. `gputrace_aps_data_*` are filtered views +of this shared evidence model. +Archive key rows also retain exact non-object value descriptors: canonical JSON +plus the recorded scalar type, NSData byte count and digest, or array/dictionary +cardinality. Values with an opaque representation carry `descriptor_error` +instead of a guessed value. Manifest source counts distinguish retained scalar, +data, container, and refused descriptors from rows sampled by an output budget. +`gputrace_track_event_arg` retains every argument for low-volume generic events +such as command buffers and profiler streams; its `event_id` is a trace-local +join key, not a persistent source identity. These are raw profiler input, +not decoded counter values or GPU encoder intervals. They are never joined to +busy-domain dispatches by comparing their displayed timestamps. +Original-execution timing attached with `--clock live --live-timing` appears in +`gputrace_live_command_buffer`, including the command buffer's GPU interval and +the separately reported kernel start and duration; its run, sidecar digest, +and match counts are also in `gputrace_capture`. Verified `--host-correlation` events appear in +`gputrace_host_signpost` with both artifact digests, clocks, bridge identity, +and declared maximum error. +MLX semantic nodes expose parent identity, and links expose their sidecar link +id and exact target index. `gputrace_semantic_arg` retains arbitrary node +attributes such as dtype and shape as key/value rows; filter `event_kind` to +distinguish untimed declarations from timed target projections. +`gputrace_dispatch` includes Xcode's workload type and view classifications. +`gputrace_dispatch_arg` and `gputrace_encoder_arg` expose every event argument +as key/value rows, including fields also available through typed columns. +Pipeline identity includes a numeric address, its capture-local scope, and the +archive record that supplied it. `gputrace_pipeline` groups by that identity +and reports total, measured, and recorded-only dispatch counts plus measured +duration. Pipeline addresses and IDs are not stable cross-trace identifiers. +Pipeline counter rows that lack a capture-backed encoder identity remain +untimed and appear in `gputrace_unattributed_counter`; arbitrary metric values +remain available through `gputrace_unattributed_counter_arg`. Evidence families +that cannot be placed on the selected clock appear in `gputrace_evidence_gap`. +Native Perfetto and timeline JSON exports also content-identify regular files +in the resolved profiler directory. `gputrace_raw_profiler_artifact` exposes +the basename, family, optional numeric index, byte size, and SHA-256 as untimed +evidence; `gputrace_capture` carries the deterministic inventory digest and +aggregate size. The exporter does not retain host directory paths or follow +symlinks. Hashing a large profiler directory may add several seconds to export. +For `Timeline_f_*.raw`, `gputrace_raw_profiler_timeline` also exposes the +fixed header's raw identity, counter count, data-section byte offset, entry +count, and profiler-sampling timestamp. That timestamp remains raw and +unaligned; it is not command-buffer time or cumulative GPU-busy time. + +`diff` fails closed when workload, device/driver, runtime, capture mode, or +timing-source gates differ or are unavailable. The explicit +`--allow-cross-environment` override labels the result +`cross-environment, not causally attributable`; it does not turn the result +into a controlled regression. + ## Commands | Group | Command | Description | @@ -53,20 +325,287 @@ gputrace diff A.gputrace B.gputrace --explain | **Buffer Analysis** | `buffers` | Buffer listing and properties | | | `buffer-access` | Buffer access patterns | | | `buffer-timeline` | Buffer allocation timeline | +| | `residency` | Allocated footprint by storage mode, and whether residency is explicit | | **Visualization** | `timeline` | Text timeline and Chrome/Perfetto export | | | `graph` | Graph visualization | | | `tree` | Execution tree view | | | `diff` | Compare two traces | | | `insights` | Actionable performance insights | -| **Capture** | `xcode-profile` | Xcode GPU profiler automation | +| **Capture** | `capture` | Run a Metal workload under the capture interposer | +| | `profile-replay` | Replay a capture under the profiler to add timing | +| | `xcode-profile` | Xcode GPU profiler automation | | | `xcode-bindings` | Inspect private Xcode GTShaderProfiler bindings | | | `xcode-parity` | Audit Xcode metric parity for a trace | | **Utilities** | `mtlb` | Metal Library Binary inspection | | | `clear-buffers` | Zero out buffers to reduce trace size | +| | `nvidia` | Report NVIDIA GPU status via NVML (Linux) | +| | `doctor` | Diagnose the GPU profiling environment and print fixes | +| | `cupti` | Convert CUPTI activity captures to Perfetto traces (Linux) | +| | `ncu` | Escalate a capture's hottest kernels to Nsight Compute counters (Linux) | +| | `overhead` | Measure how much the capture shim perturbs a workload (Linux) | +| | `devices` | List GPUs and capture backend capabilities | +| | `analyze` | Report kernel metrics and optimization findings (Linux) | +| | `optimize` | Run, compare, and iterate on workload performance | | | `version` | Print build version | Run `gputrace [command] --help` for details on any command. +### Storage modes and residency + + gputrace residency trace.gputrace + +Reports the allocated footprint per `MTLStorageMode` alongside the counts of +`newResidencySet`, `requestResidency`, and `addResidencySet`. The two belong in +one report because they are one finding seen from two directions: an all-shared +allocation profile and an uncommitted residency set both mean the process is +leaving placement and residency to the driver. Read either alone and the +default looks like a decision. + +Two limits are printed with the numbers rather than left to be discovered. + +The first is that these counts are **not** bounded by the decoded-dispatch +fraction `api-calls` reports. Buffer and residency records are found by scanning +the whole capture for record markers, which is independent of dispatch +decoding, so a capture reporting `Decoded API subset: 0 of 39014` can still have +a complete buffer and residency picture. The narrower real limit is that the +scan finds the record shapes it knows; a shape it does not know is absent rather +than counted. + +The second is that residency-set *membership* is not decoded, so there is no +wired-bytes figure separate from the allocated one. When no residency set is +committed, every allocation is under the driver's automatic residency and the +allocated total is the working upper bound on what can be made resident. A +number labelled "wired" that was really "allocated" would be worse than no +number. + +`gputrace gate` carries the same observation in `staging.allocated_bytes` and +`staging.residency_notes`. + +## Linux / NVIDIA + +gputrace is developed against Apple Metal; on Linux the trace-analysis +commands work unchanged, while capture, replay, and Xcode automation report +that they are darwin-only. The `nvidia` command reports local NVIDIA GPU +status through NVML: + +``` +$ gputrace nvidia +NVIDIA driver: 580.95.05 (1 device) + +GPU 0: NVIDIA GB10 + UUID: GPU-... + Memory: 2.3 GiB used / 121.5 GiB total + Util: gpu 7%, memory 0% + Power: 7103 mW +``` + +It uses `github.com/tmc/lib/nvidia/nvml` (purego, no cgo) and loads +`libnvidia-ml.so.1` from the standard driver search paths at runtime. + +### CUPTI kernel tracing + +`gputrace cupti` converts CUPTI activity captures into native Perfetto +traces. Any tracer that enables `CUPTI_ACTIVITY_KIND_CONCURRENT_KERNEL` (and +optionally memcpy/memset) and emits newline-delimited JSON records of the +form: + +```json +{"kind":"kernel","name":"_ZN3mlx...","start_ns":...,"end_ns":...,"grid":"112x1x1","block":"32x8x1","registers":40} +{"kind":"memcpy","start_ns":...,"end_ns":...,"bytes":9216} +``` + +can feed it. `gputrace capture` produces exactly this format natively: it +compiles a small CUPTI shim on demand, preloads it into the target, and +writes a `.gpucapture` bundle (events plus concurrent NVML samples and +provenance metadata). + +```console +$ gputrace capture -o run.gpucapture --samples -- ./workload +wrote run.gpucapture +$ gputrace analyze run.gpucapture # findings with evidence +$ gputrace cupti run.gpucapture --stats +CUPTI capture: 27550 kernels, 630 memory transfers +Total kernel time: 150.85 ms across 21 distinct kernels + +Top kernels by total GPU time: + 111.70 ms 9506x void mlx::core::cu::qmv_kernel<8, 16, 64, ...> + 11.68 ms 6680x void mlx::core::cu::binary_vv +$ gputrace cupti cupti_events.jsonl --samples nvml_samples.jsonl \ + --per-kernel-tracks -o trace.pftrace +Wrote 28180 events -> trace.pftrace +``` + +Open `trace.pftrace` at [ui.perfetto.dev](https://ui.perfetto.dev): kernel +launches appear as GPU compute slices (one track per distinct kernel with +`--per-kernel-tracks`), memory transfers on their own track, and the NVML +power/utilization/temperature series as counter tracks aligned to the same +normalized clock. + +### Diagnosing the environment + +Two ways a CUDA profiling setup fails silently rather than loudly: an +`nsys` whose default `-t cuda` routes to hardware event tracing drops +every kernel record while still writing a healthy-looking report, and a +CUPTI older than the running driver records nothing at all. Both read as +"the workload launched no kernels". + +```console +$ gputrace doctor +ok nvidia driver NVIDIA GB10: driver 580.95.05, CUDA 13.0 +warn libcupti 1 match the CUDA 13 driver, 2 predate it +warn nsight systems 4 nsys installs found + * 2026.1.1 /opt/nvidia/nsight-systems/2026.1.1/.../nsys + -t cuda enables hardware tracing, which drops every kernel record + on GB10-class parts while the capture still looks healthy + 2025.3.2 /opt/nvidia/nsight-systems/2025.3.2/.../nsys + -t cuda falls back to software tracing correctly + fix: always pass -t cuda-sw with this version +warn nsight compute GPU performance counters refused for this user +ok capture shim built with gcc -> ~/.cache/gputrace/libcupticapture-....so +``` + +Pass a workload binary to also check it for capturability: static linkage +defeats `LD_PRELOAD` entirely, and a Go target needs an in-process +`cuptiActivityFlushAll` because it crosses no interposed synchronization +point and exits via `exit_group`. + +### Budget, latency, and comparison + +`analyze` and `summary` report where a capture's wall time went, not just +which kernels ran: + +```console +$ gputrace summary run.gpucapture +GPU budget: 634.6us busy of 36.49ms wall span (1.7% occupancy), 35.85ms idle across 51 gaps + idle gaps: mean 703.0us, p95 1.46ms, max 20.73ms + 20.73ms after scale -> saxpy + +CUDA graphs: 1 graph, 54.4% of kernel time (32 graph kernels vs 17 direct launches) + graph 2: 4 launches x 8 nodes, 225.9us (54.4% of kernel time) + node #0 14.5% of graph 4x mean 8.2us saxpy + +Launch latency (queue -> submit -> start, 17 of 49 launches timed): + queued -> submitted: mean 93.8us, p50 93.2us, p95 139.5us + coverage: computed over the 34.7% of kernels with latency + timestamps; the rest (CUDA-graph launches) report none +``` + +Launch latency is reported as a coverage-gated metric rather than a bare +number. CUPTI leaves its queued/submitted timestamps unset for launches it +cannot time — CUDA-graph nodes among them — and a stale value in a reused +record buffer once made 45,943 of 46,138 kernels share one queued +timestamp, implying 1.16 s of launch latency that never happened. When too +few launches carry a consistent triple, the analysis says so instead of +averaging the rest. + +Two bundles compare kernel by kernel, ordered by GPU time moved: + +```console +$ gputrace diff base.gpucapture variant.gpucapture +verdict: regressed — slower overall (13.2%) +kernel time: 415.5us -> 470.5us (+13.2%) +occupancy: 1.7% -> 2.8% (+1.0 points) +idle budget: 35.85ms across 51 gaps -> 25.25ms across 51 gaps +``` + +### Counters and capture cost + +`gputrace ncu` re-runs the workload a bundle recorded, profiling only the +kernels the capture ranked highest and merging the counters into the +bundle as `ncu.json`. `gputrace overhead` measures what the capture itself +costs, against the effect size under study: + +```console +$ gputrace overhead --effect-size 5 -- ./workload +baseline: median 372.82ms IQR [369.68ms..374.88ms] +instrumented: median 391.84ms IQR [389.72ms..395.39ms] +overhead: +5.1% (+19.02ms) + +NOT usable for this effect size: the shim moves wall time by 5.1%, at or +beyond the 5.0% effect under study +``` + +### Agentic optimization loop + +`analyze`, `optimize run`, and `optimize compare` close the loop from +capture to verified improvement: + +```console +$ gputrace analyze events.jsonl # findings with evidence + hypotheses +$ gputrace analyze events.jsonl --suggest # playbook actions, one per finding +Suggested actions (apply ONE, then re-measure): +1. tile for cache/tensor-core reuse before micro-optimizing inner loops + verify: dominant kernel share drops +... +$ gputrace optimize run --iterations 7 -o base.json -- +median 10.92ms q1 10.88ms q3 10.99ms +# ... apply the one action ... +$ gputrace optimize run --iterations 7 -o variant.json -- +$ gputrace optimize compare base.json variant.json +verdict: improved +base: median 10.92ms +variant: median 5.81ms (-46.8%) +variant IQR [5794749..5819517] sits entirely below baseline IQR [10883246..10987607] +``` + +Compare verdicts are noise-aware: when interquartile ranges overlap, the +verdict is `noisy-change` and the delta is unproven regardless of how good +the medians look — the only sound response is more iterations. The loop, +its guardrails, and the findings-to-actions catalog are documented in +[docs/OPTIMIZATION_PLAYBOOK.md](docs/OPTIMIZATION_PLAYBOOK.md). + +## Headless timing + +A capture records what a Metal workload did and carries no timing. `profile-replay` +replays it on the GPU under Apple's MTLReplayer with the profiler attached, which +takes seconds and opens no window: + +``` +gputrace capture -o run.gputrace -- python3 bench.py +gputrace profile-replay run.gputrace # writes run-perfdata.gputrace +gputrace profiler run-perfdata.gputrace +``` + +`capture --timing-sidecar timing.jsonl --run-id ID` records command-buffer +intervals from the original execution. If Metal writes resources but no +replayable command stream, capture still fails and the sidecar ends with a +`capture_attempt` record whose status is `timing_only`. That record identifies +the attempted bundle; it does not make the intervals attributable to trace +commands or eligible for host-to-GPU projection. + +Replay processes are exclusive. A second invocation normally returns a busy +error; pass `--wait` to queue it and guarantee non-overlapping replay. Go +programs can use `github.com/tmc/gputrace/capture` and +`github.com/tmc/gputrace/profilereplay` for the same operations. + +This serializes separate MTLReplayer jobs. It does not force command buffers or +encoders inside a captured workload to execute without overlap. The replay is +headless: MTLReplayer is an agent process, no Xcode window opens, and the +frontmost application does not change. + +The default output is self-contained: it preserves the capture and adds the +profiler payload, so Xcode and capture-dependent commands can open it. Use +`--profiler-only` to write the smaller `.gpuprofiler_raw` payload when only +`profiler`, `timing`, `timeline`, or `pprof` is needed. Profiler-only output +cannot be opened by Xcode. + +This produces no derived counters. Utilization, limiter and occupancy values are +unavailable on recent GPU generations; see `docs/research/` for why. + +Commands that need performance data say so on stderr when a trace lacks it, +and name the command that would add it. + +To reproduce Xcode's All Shaders `Cost` column, use its processed pipeline +timing rather than the default SIMD-group share: + +```bash +gputrace shaders run-perfdata.gputrace --xcode-cost +``` + +This runs Xcode's private stream-data processor and can take several seconds. +It follows `DEVELOPER_DIR` or `xcode-select`; `GPUTRACE_XCODE_APP` pins a +different Xcode and causes the command to restart itself with that framework. + ## Trace Diff Compare two profiled traces and explain performance deltas at dispatch, kernel, encoder, and timeline-window levels: @@ -94,6 +633,60 @@ gputrace diff A.gputrace B.gputrace --md-out /tmp/report.md See [docs/TRACE_DIFF_WORKFLOW.md](./docs/TRACE_DIFF_WORKFLOW.md) for the full workflow and sample output. +## Go benchmark output + +`bench`, `stats`, `profiler`, and `timing` can write Go benchmark format for direct use +with `benchstat`: + +```bash +gputrace profiler trace.gputrace --benchfmt \ + --bench-config runtime=go \ + --bench-config model=Qwen2.5-0.5B > go.txt +benchstat -ignore trace-uuid go.txt python.txt +``` + +For new integrations, `gputrace bench` emits trace-scoped totals by default and +normalizes only when given `--bench-work` and `--bench-work-unit`. Go programs +can use `github.com/tmc/gputrace/tracebench` to obtain the same sectioned report +and report values directly through `testing.B.ReportMetric`. + +```bash +gputrace bench run-perfdata.gputrace \ + --format benchfmt \ + --bench-name BenchmarkDecode \ + --bench-work 32 --bench-work-unit token \ + --bench-config arm=candidate > gpu.bench +benchstat gpu.bench +``` + +The Go-facing packages are deliberately separate: + +- `capture` runs an eligible workload under the Metal capture interposer. +- `profilereplay` adds measured profiler data headlessly and supports queued, + non-overlapping replay with `Options.Wait`. +- `tracebench` analyzes retained artifacts, writes JSON or benchfmt, and reports + metrics directly to `testing.B` without parsing CLI prose. + +Benchmark suites that do not want the parent module's dependencies can instead +require `github.com/tmc/gputrace/gpubench`. It is a nested, standard-library-only +module that invokes an installed `gputrace` binary and consumes the stable JSON +report: + +```go +client := gpubench.Client{} +report, err := client.Report(ctx, tracePath, gpubench.ReportOptions{ + Work: &gpubench.Work{Count: 32, Unit: "token"}, +}) +if err != nil { + b.Fatal(err) +} +if err := report.ReportMetrics(b); err != nil { + b.Fatal(err) +} +``` + +See [docs/BENCHFMT.md](./docs/BENCHFMT.md) for the unit and provenance mapping. + ## Testing ```bash @@ -124,6 +717,7 @@ Detailed format and workflow documentation lives in `docs/`: - [README.md](./docs/README.md) -- docs index - [ENVIRONMENT.md](./docs/ENVIRONMENT.md) -- environment variables +- [BENCHFMT.md](./docs/BENCHFMT.md) -- Go benchmark and benchstat output - [TESTING.md](./docs/TESTING.md) -- test fixtures and opt-in integration tests - [TRACE_DIFF_WORKFLOW.md](./docs/TRACE_DIFF_WORKFLOW.md) -- trace diff workflow and output interpretation - [STREAMDATA_FORMAT.md](./docs/STREAMDATA_FORMAT.md) -- streamData plist format diff --git a/api_surface_test.go b/api_surface_test.go index b46ff87a..3851f646 100644 --- a/api_surface_test.go +++ b/api_surface_test.go @@ -19,7 +19,6 @@ var ( _ func(*gputrace.Trace) ([]*gputrace.EncoderTiming, error) = gputrace.ExtractTimingData _ func(*gputrace.Trace) (*timing.Store0TimingData, error) = gputrace.ExtractStore0Timing _ func(*gputrace.Trace, *timing.Store0TimingData) []*gputrace.EncoderTiming = gputrace.ConvertStore0ToEncoderTimings - _ func(*gputrace.Trace) []*gputrace.EncoderTiming = gputrace.GenerateSyntheticTiming _ func(*gputrace.Trace) (*gputrace.ShaderMetricsReport, error) = gputrace.ExtractShaderMetrics _ func(...string) *gputrace.ShaderSourceMapper = gputrace.NewShaderSourceMapper _ func(io.Writer, *gputrace.ShaderMetricsReport) error = gputrace.FormatShadersSimple @@ -71,9 +70,6 @@ func TestFacadeCalls(t *testing.T) { if _, err := gputrace.ExtractStatistics(trace); err != nil { t.Fatalf("ExtractStatistics: %v", err) } - if timings := gputrace.GenerateSyntheticTiming(trace); len(timings) == 0 { - t.Fatal("GenerateSyntheticTiming returned no timings") - } if extractor := gputrace.NewTimingMetricsExtractor(trace); extractor == nil { t.Fatal("NewTimingMetricsExtractor returned nil") } diff --git a/api_timing.go b/api_timing.go index e1d0bea9..a6340b3a 100644 --- a/api_timing.go +++ b/api_timing.go @@ -21,11 +21,6 @@ func ConvertStore0ToEncoderTimings(t *Trace, store0Data *timing.Store0TimingData return timing.ConvertStore0ToEncoderTimings(t, store0Data) } -// GenerateSyntheticTiming generates synthetic timing data for t. -func GenerateSyntheticTiming(t *Trace) []*EncoderTiming { - return timing.GenerateSyntheticTiming(t) -} - // NewTimingMetricsExtractor returns a timing metrics extractor for t. func NewTimingMetricsExtractor(t *Trace) *TimingMetricsExtractor { return timing.NewTimingMetricsExtractor(t) @@ -36,6 +31,29 @@ func FormatTimingMetrics(metrics *TimingMetrics) string { return timing.FormatTimingMetrics(metrics) } +// LowSampleMarker follows a row measured from a single dispatch. +const LowSampleMarker = timing.LowSampleMarker + +// TimingSourceUnavailable marks a trace that carries no timing measurement at +// all, as distinct from one measured approximately. +const TimingSourceUnavailable = timing.TimingSourceUnavailable + +// LowSampleFootnote explains LowSampleMarker, or returns "" when unused. +func LowSampleFootnote(timings []*KernelTiming) string { + return timing.LowSampleFootnote(timings) +} + +// FilterMinCalls keeps rows measured from at least min dispatches and reports +// how many it dropped. +func FilterMinCalls(timings []*KernelTiming, min int) ([]*KernelTiming, int) { + return timing.FilterMinCalls(timings, min) +} + +// MinCallsNote states what a --min-calls filter removed. +func MinCallsNote(min, dropped, total int) string { + return timing.MinCallsNote(min, dropped, total) +} + // ExportTimingMetricsJSON writes timing metrics as JSON. func ExportTimingMetricsJSON(w io.Writer, metrics *TimingMetrics) error { return timing.ExportTimingMetricsJSON(w, metrics) diff --git a/api_trace.go b/api_trace.go index f5b14efa..5e7d1dbb 100644 --- a/api_trace.go +++ b/api_trace.go @@ -4,13 +4,42 @@ import ( "io" "github.com/tmc/gputrace/internal/command" + "github.com/tmc/gputrace/internal/trace" ) +// IsLibraryUUID reports whether label identifies a Metal library rather than a +// function. +func IsLibraryUUID(label string) bool { + return trace.IsLibraryUUID(label) +} + +// IsArchiveFunctionName reports whether name identifies a function only by the +// shader archive it came from. A capture records an archive's content id where +// it records a function name for a library the capture describes, so such a +// kernel has a distinct, stable identity but no readable name. Only the +// profiler's streamData carries the name. +func IsArchiveFunctionName(name string) bool { + return trace.IsArchiveFunctionName(name) +} + // ParseDetailedCommandBuffer parses command buffer cbIndex from t. +// +// It reads and rescans the whole capture file on every call. Use OpenCapture +// when walking more than one command buffer. func ParseDetailedCommandBuffer(t *Trace, cbIndex int) (*command.DetailedCommandBuffer, error) { return command.ParseDetailedCommandBuffer(t, cbIndex) } +// Capture holds a capture file and its command-buffer index for repeated +// detailed parses. +type Capture = command.Capture + +// OpenCapture reads t's capture file and command-buffer index once, so that +// walking every command buffer does not reread the file per buffer. +func OpenCapture(t *Trace) (*Capture, error) { + return command.OpenCapture(t) +} + // DumpCommandBuffer writes command buffer cbIndex from t to w. func DumpCommandBuffer(t *Trace, w io.Writer, cbIndex int) error { return command.DumpCommandBuffer(t, w, cbIndex) diff --git a/capture/capture.go b/capture/capture.go new file mode 100644 index 00000000..b364b63b --- /dev/null +++ b/capture/capture.go @@ -0,0 +1,40 @@ +// Package capture runs a Metal workload under the GPUToolsCapture interposer. +package capture + +import ( + "context" + "io" + + internal "github.com/tmc/gputrace/internal/capture" +) + +// ErrNotInterposable reports that dyld will not load the capture interposer +// into the target executable. +var ErrNotInterposable = internal.ErrNotInterposable + +// Options configure a capture run. +type Options struct { + // Output is the .gputrace bundle to create. + Output string + // Dir is the workload's working directory. Empty inherits the caller's. + Dir string + // Env adds environment entries in KEY=value form. + Env []string + // Stdout and Stderr receive workload output. Nil discards it. + Stdout io.Writer + Stderr io.Writer +} + +// Eligible reports whether dyld will honor the capture interposer for path. +func Eligible(path string) error { return internal.Eligible(path) } + +// Run executes argv under the capture interposer and returns the trace path. +func Run(ctx context.Context, opts Options, argv ...string) (string, error) { + return internal.Run(ctx, internal.Options{ + Output: opts.Output, + Dir: opts.Dir, + Env: opts.Env, + Stdout: opts.Stdout, + Stderr: opts.Stderr, + }, argv...) +} diff --git a/cmd/axperms/main.go b/cmd/axperms/main.go index ffd93e39..c78de9ad 100644 --- a/cmd/axperms/main.go +++ b/cmd/axperms/main.go @@ -21,6 +21,16 @@ import ( // +// openSettingsPane opens a System Settings pane with `open`. A failure is not +// fatal: every caller also prints instructions or falls back to polling. It +// must not be silent though, or a headless run sleeps and then reports a +// confusing downstream error instead of the actual cause. +func openSettingsPane(url string) { + if err := exec.Command("open", url).Run(); err != nil { + fmt.Fprintf(os.Stderr, "warning: could not open %s: %v\n", url, err) + } +} + var ( axIsProcessTrusted func() bool axIsProcessTrustedWithOptions func(uintptr) bool @@ -172,17 +182,17 @@ func main() { } if *openSettings { - exec.Command("open", paneURLs[PaneAccessibility]).Run() + openSettingsPane(paneURLs[PaneAccessibility]) return } if *openScreenRecording { - exec.Command("open", paneURLs[PaneScreenRecording]).Run() + openSettingsPane(paneURLs[PaneScreenRecording]) return } if *openFDA { - exec.Command("open", paneURLs[PaneFullDiskAccess]).Run() + openSettingsPane(paneURLs[PaneFullDiskAccess]) return } @@ -261,7 +271,7 @@ func main() { if !axIsProcessTrusted() { fmt.Fprintf(os.Stderr, "\nPlease grant Accessibility permission to axperms in System Settings,\n") fmt.Fprintf(os.Stderr, "then run this command again.\n") - exec.Command("open", paneURLs[PaneAccessibility]).Run() + openSettingsPane(paneURLs[PaneAccessibility]) os.Exit(1) } } @@ -438,7 +448,7 @@ func findAppInPrivacyList(appName string, paneURL string) (found bool, enabled b if debugMode { fmt.Printf("[DEBUG] Opening pane: %s\n", paneURL) } - exec.Command("open", paneURL).Run() + openSettingsPane(paneURL) time.Sleep(1 * time.Second) pid := findSystemSettingsPID() @@ -559,7 +569,7 @@ func listPrivacyApps(paneName string, paneURL string) { pid := findSystemSettingsPID() if pid == 0 { fmt.Println("System Settings not running. Opening...") - exec.Command("open", paneURL).Run() + openSettingsPane(paneURL) time.Sleep(2 * time.Second) pid = findSystemSettingsPID() if pid == 0 { @@ -759,7 +769,7 @@ func findRowForApp(appName string, paneURL string) (found bool, row uintptr) { pid := findSystemSettingsPID() if pid == 0 { fmt.Println("System Settings not running. Opening...") - exec.Command("open", paneURL).Run() + openSettingsPane(paneURL) time.Sleep(2 * time.Second) pid = findSystemSettingsPID() if pid == 0 { @@ -922,7 +932,7 @@ func watchPermission() { fmt.Println("Please grant Accessibility permission in System Settings.") // Open settings - exec.Command("open", "x-apple.systempreferences:com.apple.settings.PrivacySecurity.extension?Privacy_Accessibility").Run() + openSettingsPane("x-apple.systempreferences:com.apple.settings.PrivacySecurity.extension?Privacy_Accessibility") // Trigger initial prompt triggerPrompt() diff --git a/cmd/extract_xcode_metrics/main.go b/cmd/extract_xcode_metrics/main.go new file mode 100644 index 00000000..304a6033 --- /dev/null +++ b/cmd/extract_xcode_metrics/main.go @@ -0,0 +1,256 @@ +//go:build darwin + +// Package main provides a CLI tool to extract high-level Xcode GPU profiler workload metrics and timeline counters safely. +package main + +import ( + "fmt" + "os" + "path/filepath" + "reflect" + "runtime" + "sort" + "unsafe" + + puregoobjc "github.com/ebitengine/purego/objc" + "github.com/tmc/apple/foundation" + "github.com/tmc/apple/objc" + "github.com/tmc/apple/objc/objcinspect" + "github.com/tmc/apple/private/xcode/gtshaderprofiler" + + "github.com/tmc/gputrace/internal/counter" +) + +func check(id objc.ID, selector string, want reflect.Type, args ...any) { + if err := objcinspect.Check(puregoobjc.ID(uintptr(id)), puregoobjc.RegisterName(selector), want, args...); err != nil { + fmt.Fprintf(os.Stderr, "Error: type check failed for selector %s: %v\n", selector, err) + os.Exit(1) + } +} + +type CounterStreamSummary struct { + Name string + MaxVal float64 + AvgVal float64 + Count uint64 + FirstTimestamp uint64 + LastTimestamp uint64 + SampleInterval uint64 + Scope uint16 + ScopeIndex uint64 + IsHex bool + IsUnread bool + UnreadNote string +} + +func main() { + if len(os.Args) < 2 { + fmt.Fprintf(os.Stderr, "Usage: extract_xcode_metrics \n") + os.Exit(1) + } + + inputPath := os.Args[1] + absPath, err := filepath.Abs(inputPath) + if err != nil { + fmt.Fprintf(os.Stderr, "Error: resolving absolute path for %s: %v\n", inputPath, err) + os.Exit(1) + } + if resolvedPath, err := filepath.EvalSymlinks(absPath); err == nil { + absPath = resolvedPath + } + + // Resolve directory vs streamData file path + var archiveDir string + if fi, err := os.Stat(absPath); err == nil && !fi.IsDir() { + if filepath.Base(absPath) == "streamData" { + archiveDir = filepath.Dir(filepath.Dir(absPath)) + } else { + archiveDir = filepath.Dir(absPath) + } + } else { + archiveDir = absPath + } + + // Resolve raw streamData file path inside directory + var streamDataFile string + err = filepath.Walk(archiveDir, func(p string, info os.FileInfo, err error) error { + if err == nil && !info.IsDir() && filepath.Base(p) == "streamData" { + streamDataFile = p + return filepath.SkipAll + } + return nil + }) + + if streamDataFile == "" { + fmt.Fprintf(os.Stderr, "Error: could not resolve .gpuprofiler_raw/streamData under %s\n", archiveDir) + os.Exit(1) + } + + runtime.LockOSThread() + defer runtime.UnlockOSThread() + + os.Setenv("GPUTRACE_XCODE_APP", "/Applications/Xcode-rc.app") + + rawBytes, err := os.ReadFile(streamDataFile) + if err != nil { + fmt.Fprintf(os.Stderr, "Error: reading streamData at %s: %v\n", streamDataFile, err) + os.Exit(1) + } + + dataObj := foundation.NewDataWithBytesLength(rawBytes) + targetCls := gtshaderprofiler.GetGTShaderProfilerStreamDataClass().Class() + + unarchived, err := foundation.GetNSKeyedUnarchiverClass(). + UnarchivedObjectOfClassFromDataError(targetCls, dataObj) + if err != nil || unarchived.GetID() == 0 { + fmt.Fprintf(os.Stderr, "Error: unarchiving streamData: %v\n", err) + os.Exit(1) + } + + stream := gtshaderprofiler.GTShaderProfilerStreamDataFromID(unarchived.GetID()) + + // Call _setupDataPath on streamData object to bind directory dependencies + check(stream.GetID(), "_setupDataPath", reflect.TypeOf(objc.ID(0))) + objc.Send[objc.ID](stream.GetID(), objc.Sel("_setupDataPath")) + + procClassObj := objc.GetClass("GTShaderProfilerStreamDataProcessor") + procAlloc := objc.Send[objc.ID](objc.ID(procClassObj), objc.Sel("alloc")) + + check(procAlloc, "initWithStreamData:llvmHelperPath:", reflect.TypeOf(objc.ID(0)), stream.GetID(), objc.ID(0)) + procObjID := objc.Send[objc.ID](procAlloc, objc.Sel("initWithStreamData:llvmHelperPath:"), stream.GetID(), 0) + procObj := gtshaderprofiler.GTShaderProfilerStreamDataProcessorFromID(procObjID) + + check(procObj.GetID(), "processStreamData", nil) + procObj.ProcessStreamData() + + mID := procObj.MioData() + check(mID.GetID(), "gpuTime", reflect.TypeOf(uint64(0))) + check(mID.GetID(), "encoderCount", reflect.TypeOf(uint64(0))) + check(mID.GetID(), "drawCount", reflect.TypeOf(uint64(0))) + check(mID.GetID(), "pipelineStateCount", reflect.TypeOf(uint64(0))) + + gpuTimeNS := mID.GpuTime() + gpuTimeMS := float64(gpuTimeNS) / 1e6 + + // Extract timeline counters off nonOverlappingTimeline + var summaries []CounterStreamSummary + var withheldCounters int + nonOverlappingPtr := mID.NonOverlappingTimeline() + if nonOverlappingPtr != nil { + nonOverlappingID := objc.ID(uintptr(nonOverlappingPtr)) + check(nonOverlappingID, "timelineCounters", reflect.TypeOf(objc.ID(0))) + countersObj := gtshaderprofiler.GTMioTraceTimelineDataFromID(nonOverlappingID).TimelineCounters() + + if countersObj.GetID() != 0 { + check(countersObj.GetID(), "counters", reflect.TypeOf(objc.ID(0))) + dictObj := countersObj.Counters() + if dictObj.GetID() != 0 { + check(dictObj.GetID(), "allKeys", reflect.TypeOf(objc.ID(0))) + keys := dictObj.AllKeys() + for _, key := range keys { + kStr := foundation.NSStringFromID(key.GetID()).String() + cntObj := dictObj.ObjectForKey(key) + cnt := gtshaderprofiler.GTMioCounterDataFromID(cntObj.GetID()) + check(cnt.GetID(), "sampleCount", reflect.TypeOf(uint64(0))) + check(cnt.GetID(), "values", reflect.TypeOf(unsafe.Pointer(nil))) + check(cnt.GetID(), "timestamps", reflect.TypeOf(unsafe.Pointer(nil))) + check(cnt.GetID(), "sampleInterval", reflect.TypeOf(uint64(0))) + check(cnt.GetID(), "scope", reflect.TypeOf(uint16(0))) + check(cnt.GetID(), "scopeIndex", reflect.TypeOf(uint64(0))) + + // Seed the extremes from the data, not from zero: a series + // that never rises above zero would otherwise report a max + // of zero regardless of what it holds. + vals, stamps, err := counter.CounterSeries(cnt) + if err != nil { + fmt.Fprintf(os.Stderr, "Withholding %s: %v\n", kStr, err) + withheldCounters++ + continue + } + var minV, maxV, sumV float64 + for i, v := range vals { + if i == 0 { + minV, maxV = v, v + } + minV = min(minV, v) + maxV = max(maxV, v) + sumV += v + } + avgV := 0.0 + if len(vals) > 0 { + avgV = sumV / float64(len(vals)) + } + if len(stamps) != len(vals) { + fmt.Fprintf(os.Stderr, "Error: %s timestamps=%d values=%d\n", kStr, len(stamps), len(vals)) + os.Exit(1) + } + var first, last uint64 + if len(stamps) > 0 { + first, last = stamps[0], stamps[len(stamps)-1] + } + + isHex := false + if len(kStr) == 16 || len(kStr) == 64 { + isHex = true + } + + // Two different reasons to withhold a column, and only one of + // them is a property of the name. Whether a series is all + // zero is a property of the data, so measure it rather than + // listing the three counters that happened to be empty here; + // a hardcoded list keeps suppressing them once they carry + // values, and stays quiet when a fourth goes empty. + isUnread := false + unreadNote := "" + switch { + case kStr == "Texture Read Limiter": + isUnread = true + unreadNote = " (Unread: unestablished encoding, max 8.99e10 vs Xcode oracle 0.00%)" + case len(vals) > 0 && maxV == 0 && minV == 0: + isUnread = true + unreadNote = " (Unread: every sample zero)" + } + + summaries = append(summaries, CounterStreamSummary{ + Name: kStr, + MaxVal: maxV, + AvgVal: avgV, + Count: cnt.SampleCount(), + FirstTimestamp: first, + LastTimestamp: last, + SampleInterval: cnt.SampleInterval(), + Scope: cnt.Scope(), + ScopeIndex: cnt.ScopeIndex(), + IsHex: isHex, + IsUnread: isUnread, + UnreadNote: unreadNote, + }) + } + } + } + } + + sort.Slice(summaries, func(i, j int) bool { + return summaries[i].Name < summaries[j].Name + }) + + fmt.Println("==========================================================================") + fmt.Printf(" RESOLVED ARCHIVE PATH: %s\n", archiveDir) + fmt.Println("==========================================================================") + fmt.Printf("[1. Workload Summary]\n") + fmt.Printf(" Compute Encoder Count: %d\n", mID.EncoderCount()) + fmt.Printf(" Compute Dispatch Count: %d\n", mID.DrawCount()) + fmt.Printf(" Pipeline State Count: %d\n", mID.PipelineStateCount()) + fmt.Printf(" GPU Time: %.3f ms\n", gpuTimeMS) + + fmt.Printf("\n[2. Memory Timeline Counters (%d Channels Extracted, %d Withheld)]\n", len(summaries), withheldCounters) + for _, s := range summaries { + if s.IsUnread { + fmt.Printf(" %-36s -> [Unread / Encoding Unestablished]%s (samples: %d, scope: %d/%d, interval: %d, ticks: %d..%d)\n", s.Name, s.UnreadNote, s.Count, s.Scope, s.ScopeIndex, s.SampleInterval, s.FirstTimestamp, s.LastTimestamp) + } else { + fmt.Printf(" %-36s -> Peak: %-10.2f Avg: %-10.2f (samples: %d, scope: %d/%d, interval: %d, ticks: %d..%d)\n", s.Name, s.MaxVal, s.AvgVal, s.Count, s.Scope, s.ScopeIndex, s.SampleInterval, s.FirstTimestamp, s.LastTimestamp) + } + } + + fmt.Println("==========================================================================") +} diff --git a/cmd/gputrace/cmd/admit.go b/cmd/gputrace/cmd/admit.go new file mode 100644 index 00000000..ab36b7ef --- /dev/null +++ b/cmd/gputrace/cmd/admit.go @@ -0,0 +1,87 @@ +package cmd + +import ( + "encoding/json" + "fmt" + + "github.com/spf13/cobra" + + "github.com/tmc/gputrace/internal/admit" +) + +type admitOptions struct { + json bool +} + +var admitCmd = newAdmitCommand(&admitOptions{}) + +func newAdmitCommand(opts *admitOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "admit ", + Short: "Check whether a profiled export supports a measured-timing claim", + Long: `Check a profiled export against the raw capture it claims to measure. + +Every criterion must pass for the export to be admitted: + + exported UUID matches raw the export is of this capture + streamData present and non-empty profiler data was written + payload self-contained the bundle carries its own evidence + dispatch counts match raw the replay ran the same work + timing provenance is measured not synthetic or capture-derived + +A criterion that cannot be evaluated is reported as UNKNOWN and withholds +admission: an unanswerable question leaves the claim unsupported. + +Exits non-zero when the export is not admitted, so it can gate a pipeline.`, + Args: cobra.ExactArgs(2), + RunE: func(cmd *cobra.Command, args []string) error { + return runAdmit(cmd, args, opts) + }, + } + cmd.Flags().BoolVar(&opts.json, "json", false, "Output the verdict as JSON") + return cmd +} + +func init() { + rootCmd.AddCommand(admitCmd) +} + +func runAdmit(cmd *cobra.Command, args []string, opts *admitOptions) error { + rawPath, profiledPath := args[0], args[1] + if err := checkTraceFile(rawPath); err != nil { + return err + } + if err := checkTraceFile(profiledPath); err != nil { + return err + } + + result := admit.Check(rawPath, profiledPath) + out := cmd.OutOrStdout() + + if opts.json { + encoder := json.NewEncoder(out) + encoder.SetIndent("", " ") + if err := encoder.Encode(result); err != nil { + return fmt.Errorf("write verdict: %w", err) + } + } else if err := admit.WriteReport(out, result); err != nil { + return err + } + + if !result.Admitted() { + // The report already says which criteria failed and why, so the + // error only carries the exit status. + return errNotAdmitted + } + return nil +} + +// notAdmittedError carries the exit status for a rejected trace. The report +// has already named the failing criteria, so the entry point must not print +// the error again underneath them. +type notAdmittedError struct{} + +func (notAdmittedError) Error() string { return "trace not admitted" } +func (notAdmittedError) alreadyReported() {} + +var errNotAdmitted = notAdmittedError{} diff --git a/cmd/gputrace/cmd/analyze.go b/cmd/gputrace/cmd/analyze.go new file mode 100644 index 00000000..117692e7 --- /dev/null +++ b/cmd/gputrace/cmd/analyze.go @@ -0,0 +1,271 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "io" + "strings" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/cuptitrace" + "github.com/tmc/gputrace/internal/gpuevent" + "github.com/tmc/gputrace/internal/optimize" +) + +var analyzeOpts = struct { + json bool + suggest bool + samples string + demangle bool + limit int +}{demangle: true, limit: 10} + +var analyzeCmd = &cobra.Command{ + Use: "analyze ", + Short: "Report per-kernel metrics and optimization findings (Linux/NVIDIA)", + Long: `Analyze a CUPTI activity capture and report optimization findings. + +Aggregates kernel launches into per-kernel statistics (count, total, +mean/p50/p95 duration, share of GPU time, launch geometry) and classifies +each kernel as compute-, memory-, or latency-bound from launch shape. +Findings pair measured evidence with a concrete hypothesis an agent or +developer can act on. Classifications are heuristics from launch geometry, +not hardware-counter measurements. + +--samples overlays concurrent NVML samples so findings can cite device +state during the capture window.`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + f, closers, err := cupticapture.OpenEvents(args[0]) + if err != nil { + return err + } + defer closers() + samplesPath := cupticapture.ResolveSamples(args[0], analyzeOpts.samples) + cap, err := readCapture(f, samplesPath) + if err != nil { + return err + } + if len(cap.Events) == 0 { + return fmt.Errorf("no events in %s", args[0]) + } + // Decode names for readable reports when demangling is available; + // raw symbols always stay in RawSymbol for evidence. + if analyzeOpts.demangle { + for i := range cap.Events { + if cap.Events[i].Kind == gpuevent.KindKernel && cap.Events[i].Name == "" { + cap.Events[i].Name = cuptitrace.Demangle(cap.Events[i].RawSymbol) + } + } + } + rep := gpuevent.Analyze(cap.Events, cap.Samples) + // Launch-overhead analysis needs API records; when present, attach + // the host-vs-device split to the report. + if len(cap.APIs) > 0 { + rep.LaunchOverhead = gpuevent.LaunchOverheadAnalysis(cap) + } + if analyzeOpts.json { + return json.NewEncoder(cmd.OutOrStdout()).Encode(rep) + } + if analyzeOpts.suggest { + entries := optimize.Suggest(rep.Findings) + out := cmd.OutOrStdout() + if len(entries) == 0 { + fmt.Fprintln(out, "No findings to act on.") + return nil + } + fmt.Fprintln(out, "Suggested actions (apply ONE, then re-measure):") + fmt.Fprint(out, optimize.RenderSuggestions(entries)) + fmt.Fprintln(out, "\nLoop: gputrace optimize run -o base.json -- ") + fmt.Fprintln(out, " ... apply one action ...") + fmt.Fprintln(out, " gputrace optimize run -o variant.json -- ") + fmt.Fprintln(out, " gputrace optimize compare base.json variant.json") + return nil + } + return printAnalysis(cmd, rep) + }, +} + +func printAnalysis(cmd *cobra.Command, rep *gpuevent.Report) error { + out := cmd.OutOrStdout() + fmt.Fprintf(out, "Capture: %d launches, %d kernels, %.2f ms total kernel time\n", + rep.KernelLaunches, len(rep.Kernels), float64(rep.TotalKernelNS)/1e6) + if rep.MemcpyCount > 0 { + fmt.Fprintf(out, "Transfers: %d memcpys (%.2f ms), %d memsets\n", + rep.MemcpyCount, float64(rep.MemcpyNS)/1e6, rep.MemsetCount) + } else { + fmt.Fprintf(out, "Transfers: 0 memcpys, %d memsets\n", rep.MemsetCount) + } + + fmt.Fprintln(out, "\nKernels by GPU time:") + limit := analyzeOpts.limit + if limit == 0 || limit > len(rep.Kernels) { + limit = len(rep.Kernels) + } + for _, k := range rep.Kernels[:limit] { + occ := "" + if k.TheoreticalOccupancyPct > 0 { + occ = fmt.Sprintf(" occ ~%.0f%% (%s)", k.TheoreticalOccupancyPct, k.OccupancyLimiter) + } + fmt.Fprintf(out, " %6.1f%% %5dx mean %-9s p95 %-9s [%s] %s%s\n", + k.SharePct, k.Count, dur(k.MeanNS), dur(k.P95NS), k.Bound, shortKernel(k.Name), occ) + } + + printUtilization(out, rep.Utilization) + printGraphs(out, rep.Graphs) + printLaunchLatency(out, rep.LaunchLatency) + + if rep.LaunchOverhead != nil && rep.LaunchOverhead.Joins > 0 { + lo := rep.LaunchOverhead + fmt.Fprintf(out, "\nLaunch overhead (host API vs GPU execution, %d joined launches):\n", lo.Joins) + fmt.Fprintf(out, " host cost/launch: mean %s, p50 %s, p95 %s\n", + dur(lo.MeanHostCostNS), dur(lo.P50HostCostNS), dur(lo.P95HostCostNS)) + if lo.TotalGPUNS > 0 { + ratio := float64(lo.TotalHostNS) / float64(lo.TotalGPUNS) * 100 + fmt.Fprintf(out, " host/GPU ratio: %.1f%% (%s host vs %s GPU)\n", + ratio, dur(lo.TotalHostNS), dur(lo.TotalGPUNS)) + if ratio >= 25 { + fmt.Fprintf(out, " => launch-bound: host submission is a material share of GPU time;\n reduce launch count or batch work before tuning kernels\n") + } + } + } + + if len(rep.Findings) == 0 { + fmt.Fprintln(out, "\nNo optimization findings; the capture has no dominant pattern.") + return nil + } + fmt.Fprintf(out, "\nFindings:\n") + for _, f := range rep.Findings { + fmt.Fprintf(out, "\n[%s] %s: %s\n", strings.ToUpper(string(f.Severity)), f.Kind, shortKernel(f.Subject)) + for _, ev := range f.Evidence { + fmt.Fprintf(out, " evidence: %s\n", ev) + } + wrapped := wrapWords(f.Hypothesis, 76) + for i, line := range wrapped { + if i == 0 { + fmt.Fprintf(out, " hypothesis: %s\n", line) + } else { + fmt.Fprintf(out, " %s\n", line) + } + } + } + return nil +} + +// printUtilization reports the busy/idle budget: how much of the capture's +// wall span the device spent executing anything. +func printUtilization(out io.Writer, u gpuevent.Utilization) { + if u.WallSpanNS == 0 { + return + } + fmt.Fprintf(out, "\nGPU budget: %s busy of %s wall span (%.1f%% occupancy), %s idle across %d gaps\n", + dur(u.BusyNS), dur(u.WallSpanNS), u.OccupancyPct, dur(u.IdleNS), u.GapCount) + if u.Concurrency > 1.05 { + fmt.Fprintf(out, " stream overlap: %.2fx (summed activity time over busy wall time)\n", u.Concurrency) + } + if u.GapCount > 0 { + fmt.Fprintf(out, " idle gaps: mean %s, p95 %s, max %s\n", + dur(u.MeanGapNS), dur(u.P95GapNS), dur(u.MaxGapNS)) + for _, g := range u.TopGaps { + fmt.Fprintf(out, " %-9s after %s -> %s\n", dur(g.DurationNS), + shortKernel(orUnknown(g.AfterName)), shortKernel(orUnknown(g.BeforeName))) + } + } +} + +// printGraphs reports how much of the capture ran through CUDA graphs and +// which node of each graph owns its time. +func printGraphs(out io.Writer, g *gpuevent.GraphAnalysis) { + if g == nil || len(g.Graphs) == 0 { + return + } + fmt.Fprintf(out, "\nCUDA graphs: %d graph%s, %.1f%% of kernel time (%d graph kernels vs %d direct launches)\n", + len(g.Graphs), plural(len(g.Graphs)), g.GraphSharePct, g.GraphKernels, g.DirectKernels) + for _, gr := range g.Graphs { + fmt.Fprintf(out, " graph %d: %d launches x %d nodes, %s (%.1f%% of kernel time)\n", + gr.GraphID, gr.Launches, gr.Nodes, dur(gr.TotalNS), gr.SharePct) + for _, n := range gr.TopNodes { + fmt.Fprintf(out, " node #%-3d %5.1f%% of graph %5dx mean %-9s %s\n", + n.Index, n.SharePct, n.Count, dur(n.MeanNS), shortKernel(n.Name)) + } + } +} + +// printLaunchLatency reports the queue -> submit -> start decomposition, or +// says why the capture cannot support one. It never prints a figure the +// analysis declared unusable. +func printLaunchLatency(out io.Writer, l *gpuevent.LaunchLatency) { + if l == nil || l.Kernels == 0 { + return + } + if !l.Usable { + fmt.Fprintf(out, "\nLaunch latency: unavailable — %s\n", l.Reason) + return + } + fmt.Fprintf(out, "\nLaunch latency (queue -> submit -> start, %d of %d launches timed):\n", l.Timed, l.Kernels) + fmt.Fprintf(out, " queued -> submitted: mean %s, p50 %s, p95 %s\n", + dur(l.QueueToSubmitNS.MeanNS), dur(l.QueueToSubmitNS.P50NS), dur(l.QueueToSubmitNS.P95NS)) + fmt.Fprintf(out, " submitted -> start: mean %s, p50 %s, p95 %s\n", + dur(l.SubmitToStartNS.MeanNS), dur(l.SubmitToStartNS.P50NS), dur(l.SubmitToStartNS.P95NS)) + fmt.Fprintf(out, " queued -> start: mean %s, p50 %s, p95 %s\n", + dur(l.QueueToStartNS.MeanNS), dur(l.QueueToStartNS.P50NS), dur(l.QueueToStartNS.P95NS)) + if l.Reason != "" { + fmt.Fprintf(out, " coverage: %s\n", l.Reason) + } +} + +func orUnknown(s string) string { + if s == "" { + return "(unknown)" + } + return s +} + +func shortKernel(name string) string { + const maxLen = 88 + if len(name) <= maxLen { + return name + } + return name[:maxLen-3] + "..." +} + +func wrapWords(s string, width int) []string { + var lines []string + var cur strings.Builder + for _, word := range strings.Fields(s) { + if cur.Len() > 0 && cur.Len()+1+len(word) > width { + lines = append(lines, cur.String()) + cur.Reset() + } + if cur.Len() > 0 { + cur.WriteByte(' ') + } + cur.WriteString(word) + } + if cur.Len() > 0 { + lines = append(lines, cur.String()) + } + return lines +} + +// dur formats nanoseconds at human scale; shared with other command files. +func dur(ns uint64) string { + switch { + case ns >= 1e6: + return fmt.Sprintf("%.2fms", float64(ns)/1e6) + case ns >= 1e3: + return fmt.Sprintf("%.1fus", float64(ns)/1e3) + default: + return fmt.Sprintf("%dns", ns) + } +} + +func init() { + analyzeCmd.Flags().BoolVar(&analyzeOpts.json, "json", false, "Output machine-readable JSON report") + analyzeCmd.Flags().BoolVar(&analyzeOpts.suggest, "suggest", false, "Print playbook actions for each finding instead of the report") + analyzeCmd.Flags().StringVar(&analyzeOpts.samples, "samples", "", "Concurrent NVML sample file to join") + analyzeCmd.Flags().BoolVar(&analyzeOpts.demangle, "demangle", true, "Demangle kernel symbols with c++filt when available") + analyzeCmd.Flags().IntVar(&analyzeOpts.limit, "limit", 10, "Kernels listed in the text report") + rootCmd.AddCommand(analyzeCmd) +} diff --git a/cmd/gputrace/cmd/api_calls.go b/cmd/gputrace/cmd/api_calls.go index 3ce1fb9e..5e310a9d 100644 --- a/cmd/gputrace/cmd/api_calls.go +++ b/cmd/gputrace/cmd/api_calls.go @@ -1,6 +1,7 @@ package cmd import ( + "bytes" "encoding/json" "fmt" "io" @@ -13,17 +14,22 @@ import ( type apiCallsOptions struct { kernelFilter string json bool + limit int + all bool } -var apiCallsCmd = newAPICallsCommand(&apiCallsOptions{}) +var apiCallsCmd = newAPICallsCommand(&apiCallsOptions{limit: defaultHumanLimit}) func newAPICallsCommand(opts *apiCallsOptions) *cobra.Command { + if opts.limit == 0 { + opts.limit = defaultHumanLimit + } cmd := &cobra.Command{ Use: "api-calls ", Short: "Display API call sequences from a GPU trace", - Long: `Display the sequence of Metal API calls captured in a GPU trace. + Long: `Display the decoded subset of Metal API calls captured in a GPU trace. -Shows the full API call sequence including: +Shows decoded API calls including: - Command buffer creation - Encoder creation and configuration - Compute pipeline state setup @@ -31,7 +37,9 @@ Shows the full API call sequence including: - Dispatch calls - Encoder completion -Each call is numbered and indented to show the command buffer hierarchy. +Each call is numbered and indented to show the decoded command buffer hierarchy. +Human output reports how many trace dispatches are represented; a low count +means the decoded API list is incomplete, not that the trace did no GPU work. Examples: # Show all API calls @@ -52,6 +60,8 @@ Examples: } cmd.Flags().StringVarP(&opts.kernelFilter, "kernel", "k", "", "Filter output to show only calls related to kernels matching this pattern (case-insensitive)") cmd.Flags().BoolVar(&opts.json, "json", false, "Output in JSON format") + cmd.Flags().IntVar(&opts.limit, "limit", opts.limit, "Maximum calls in human output") + cmd.Flags().BoolVar(&opts.all, "all", opts.all, "Show all calls in human output") return cmd } @@ -69,28 +79,46 @@ func runAPICalls(cmd *cobra.Command, args []string, opts *apiCallsOptions) error if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } + apiList, err := trace.ParseAPICallList() + if err != nil { + return fmt.Errorf("parse API calls: %w", err) + } if opts.json { - apiList, err := trace.ParseAPICallList() - if err != nil { - return fmt.Errorf("parse API calls: %w", err) - } return writeAPICallsJSON(cmd.OutOrStdout(), apiList) } + limit, err := resolveHumanLimit(opts.limit, opts.all) + if err != nil { + return err + } + var rendered bytes.Buffer if opts.kernelFilter != "" { // Use filtered output - if err := formatAPICallsFiltered(cmd.OutOrStdout(), trace, opts.kernelFilter); err != nil { + if err := formatAPICallsFiltered(&rendered, trace, opts.kernelFilter); err != nil { return fmt.Errorf("failed to format API calls: %w", err) } } else { - // Use FormatAPICallList which prints to stdout - if err := trace.FormatAPICallList(cmd.OutOrStdout()); err != nil { + if err := trace.FormatAPICallList(&rendered); err != nil { return fmt.Errorf("failed to format API calls: %w", err) } } - return nil + decodedDispatches := 0 + for _, cb := range apiList.CommandBuffers { + for _, call := range cb.Calls { + if call.Type == "dispatch" { + decodedDispatches++ + } + } + } + traceDispatches, _ := trace.CountDispatchCalls() + fmt.Fprintf(cmd.OutOrStdout(), "Decoded API subset: %d of %d trace dispatches represented\n\n", + decodedDispatches, traceDispatches) + return writeLimitedLines(cmd.OutOrStdout(), rendered.String(), limit, "calls") } func writeAPICallsJSON(w io.Writer, apiList *gputrace.APICallList) error { diff --git a/cmd/gputrace/cmd/api_calls_test.go b/cmd/gputrace/cmd/api_calls_test.go index d58fda8f..fcd74962 100644 --- a/cmd/gputrace/cmd/api_calls_test.go +++ b/cmd/gputrace/cmd/api_calls_test.go @@ -6,6 +6,7 @@ import ( "strings" "testing" + "github.com/spf13/cobra" "github.com/tmc/gputrace" ) @@ -45,3 +46,23 @@ func TestWriteAPICallsJSON(t *testing.T) { t.Fatalf("decoded command buffers = %+v", got.CommandBuffers) } } + +func TestRunAPICallsTextIsBoundedAndUsesCommandOutput(t *testing.T) { + tracePath := testCommandBuffersTracePath(t) + var out bytes.Buffer + command := &cobra.Command{} + command.SetOut(&out) + + stdout, err := captureStdout(t, func() error { + return runAPICalls(command, []string{tracePath}, &apiCallsOptions{limit: 1}) + }) + if err != nil { + t.Fatalf("runAPICalls: %v", err) + } + if stdout != "" { + t.Fatalf("os stdout = %q, want empty", stdout) + } + if !strings.Contains(out.String(), "Decoded API subset:") { + t.Fatalf("command output missing coverage:\n%s", out.String()) + } +} diff --git a/cmd/gputrace/cmd/automation_cancel.go b/cmd/gputrace/cmd/automation_cancel.go index e6b1deb6..9ac8612f 100644 --- a/cmd/gputrace/cmd/automation_cancel.go +++ b/cmd/gputrace/cmd/automation_cancel.go @@ -7,6 +7,7 @@ import ( "os" "os/exec" "os/signal" + "strings" "sync" "syscall" "time" @@ -76,10 +77,34 @@ func waitForAutomation(ctx context.Context, delay time.Duration) error { } } +// appleScriptString renders s as an AppleScript string literal, including the +// surrounding quotes. Go's %q is not a substitute: it escapes non-ASCII as +// \uXXXX, which AppleScript reads literally rather than as an escape. +// AppleScript recognizes only \" and \\ inside a quoted string, and accepts +// raw UTF-8 for everything else. +func appleScriptString(s string) string { + var b strings.Builder + b.Grow(len(s) + 2) + b.WriteByte('"') + for _, r := range s { + switch r { + case '"', '\\': + b.WriteByte('\\') + b.WriteRune(r) + case '\n': + b.WriteString(`" & return & "`) + default: + b.WriteRune(r) + } + } + b.WriteByte('"') + return b.String() +} + // ShowAutomationOverlay shows a notification that automation is running. // Uses AppleScript to show a system notification. func ShowAutomationOverlay(parent context.Context, message string) error { - script := fmt.Sprintf(`display notification %q with title "gputrace" subtitle "Press Ctrl+C to cancel"`, message) + script := fmt.Sprintf(`display notification %s with title "gputrace" subtitle "Press Ctrl+C to cancel"`, appleScriptString(message)) ctx, cancel := context.WithTimeout(parent, 2*time.Second) defer cancel() cmd := exec.CommandContext(ctx, "osascript", "-e", script) diff --git a/cmd/gputrace/cmd/automation_cancel_quote_test.go b/cmd/gputrace/cmd/automation_cancel_quote_test.go new file mode 100644 index 00000000..3e4934fc --- /dev/null +++ b/cmd/gputrace/cmd/automation_cancel_quote_test.go @@ -0,0 +1,25 @@ +package cmd + +import "testing" + +func TestAppleScriptString(t *testing.T) { + tests := []struct { + name string + in string + want string + }{ + {"plain", `hello`, `"hello"`}, + {"quote", `say "hi"`, `"say \"hi\""`}, + {"backslash", `a\b`, `"a\\b"`}, + {"non-ascii kept raw", "café ✅", `"café ✅"`}, + {"newline becomes return", "a\nb", `"a" & return & "b"`}, + {"empty", ``, `""`}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if got := appleScriptString(tt.in); got != tt.want { + t.Errorf("appleScriptString(%q) = %s, want %s", tt.in, got, tt.want) + } + }) + } +} diff --git a/cmd/gputrace/cmd/bench.go b/cmd/gputrace/cmd/bench.go new file mode 100644 index 00000000..cf104a66 --- /dev/null +++ b/cmd/gputrace/cmd/bench.go @@ -0,0 +1,94 @@ +package cmd + +import ( + "fmt" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/tracebench" +) + +var benchCmd = newBenchCommand(new(benchOptions)) + +type benchOptions struct { + format string + name string + work uint64 + workUnit string + benchConfig benchfmtConfigFlags +} + +func newBenchCommand(opts *benchOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "bench ", + Short: "Export trace evidence for Go benchmark tools", + Long: `Export a stable, sectioned GPU trace report as JSON or Go benchmark text. + +Without --bench-work, measurements are honest trace totals with units such as +dispatches/trace and dispatch_span_ns/trace. Per-work units require both a +positive --bench-work count and --bench-work-unit. + +The JSON report keeps structural counts and measured profiler timing in +separate sections with source, status, and refusal details. Benchfmt output +records observer, payload, trace UUID, timing source, and declared work. It is +accepted directly by golang.org/x/perf/benchfmt and benchstat. + +Go programs can use github.com/tmc/gputrace/tracebench instead of parsing this +command's output. Its ReportMetrics method writes the same values through +testing.B.ReportMetric. The nested github.com/tmc/gputrace/gpubench module is a +standard-library-only client for projects that should not depend on the parent +module's parser and private-framework dependencies. + +Examples: + gputrace bench run.gputrace --format json + gputrace bench run-perfdata.gputrace --format benchfmt + gputrace bench run-perfdata.gputrace --format benchfmt \ + --bench-name BenchmarkDecode --bench-work 32 --bench-work-unit token \ + --bench-config arm=candidate`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runBench(cmd, args[0], opts) + }, + } + f := cmd.Flags() + f.StringVar(&opts.format, "format", "json", "Output format: json or benchfmt") + f.StringVar(&opts.name, "bench-name", "BenchmarkGPUTrace", "Benchmark name for benchfmt output") + f.Uint64Var(&opts.work, "bench-work", 0, "Logical work represented by the trace") + f.StringVar(&opts.workUnit, "bench-work-unit", "", "Logical work unit: op, token, step, or byte") + f.Var(&opts.benchConfig, "bench-config", "Set benchfmt configuration key=value (repeatable)") + return cmd +} + +func runBench(cmd *cobra.Command, path string, opts *benchOptions) error { + var work *tracebench.Work + switch { + case opts.work == 0 && opts.workUnit != "": + return fmt.Errorf("--bench-work-unit requires --bench-work") + case opts.work > 0 && opts.workUnit == "": + return fmt.Errorf("--bench-work requires --bench-work-unit") + case opts.work > 0: + work = &tracebench.Work{Count: opts.work, Unit: opts.workUnit} + } + report, err := tracebench.Analyze(path, tracebench.Options{Work: work}) + if err != nil { + return err + } + switch opts.format { + case "json": + return tracebench.WriteJSON(cmd.OutOrStdout(), report) + case "benchfmt": + config := make([]tracebench.Config, len(opts.benchConfig)) + for i, item := range opts.benchConfig { + config[i] = tracebench.Config{Key: item.Key, Value: item.Value} + } + return tracebench.WriteBenchfmt(cmd.OutOrStdout(), report, tracebench.BenchfmtOptions{ + Name: opts.name, + Config: config, + }) + default: + return fmt.Errorf("unsupported format %q", opts.format) + } +} + +func init() { + rootCmd.AddCommand(benchCmd) +} diff --git a/cmd/gputrace/cmd/bench_test.go b/cmd/gputrace/cmd/bench_test.go new file mode 100644 index 00000000..6f49a4e0 --- /dev/null +++ b/cmd/gputrace/cmd/bench_test.go @@ -0,0 +1,47 @@ +package cmd + +import ( + "bytes" + "path/filepath" + "strings" + "testing" + + "github.com/spf13/cobra" +) + +func TestBenchRequiresCompleteWorkDenominator(t *testing.T) { + tests := []struct { + name string + opts benchOptions + want string + }{ + {"count only", benchOptions{work: 1}, "requires --bench-work-unit"}, + {"unit only", benchOptions{workUnit: "op"}, "requires --bench-work"}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + cmd := new(cobra.Command) + err := runBench(cmd, "not-opened", &test.opts) + if err == nil || !strings.Contains(err.Error(), test.want) { + t.Fatalf("error = %v, want %q", err, test.want) + } + }) + } +} + +func TestBenchStructuralTraceUsesTraceUnits(t *testing.T) { + path := filepath.Join("..", "..", "..", "testdata", "traces", "01-single-encoder", "01-single-encoder-run1.gputrace") + var out bytes.Buffer + cmd := new(cobra.Command) + cmd.SetOut(&out) + if err := runBench(cmd, path, &benchOptions{format: "benchfmt", name: "BenchmarkFixture"}); err != nil { + t.Fatal(err) + } + text := out.String() + if !strings.Contains(text, "dispatches/trace") { + t.Fatalf("trace-scoped dispatch unit missing:\n%s", text) + } + if strings.Contains(text, "/op") { + t.Fatalf("undeclared per-operation unit present:\n%s", text) + } +} diff --git a/cmd/gputrace/cmd/benchfmt.go b/cmd/gputrace/cmd/benchfmt.go new file mode 100644 index 00000000..6db1f3f4 --- /dev/null +++ b/cmd/gputrace/cmd/benchfmt.go @@ -0,0 +1,300 @@ +package cmd + +import ( + "fmt" + "io" + "math" + "sort" + "strconv" + "strings" + "unicode" + + "github.com/spf13/cobra" +) + +const ( + benchfmtDispatchSpanUnit = "dispatch_span_ns/trace" + benchfmtCBActiveUnit = "cb_active_ns/trace" + benchfmtCBWallUnit = "cb_wall_ns/trace" + benchfmtEffectiveGPUUnit = "effective_gpu_ns/trace" + benchfmtProfilerSampleCostUnit = "profiler_sample_cost_percent" + benchfmtProfilerCostSamplesUnit = "profiler_cost_samples/trace" + benchfmtGPRWCNTRSamplesUnit = "gprwcntr_samples/trace" + benchfmtDispatchesUnit = "dispatches/trace" + benchfmtCommandBuffersUnit = "command-buffers/trace" + benchfmtEncodersUnit = "encoders/trace" +) + +var benchfmtConfigOrder = []string{ + "goos", + "goarch", + "pkg", + "cpu", + "runtime", + "model", + "prompt-tokens", + "capture-range", + "compile-mode", + "cache-mode", + "trace-uuid", + "mlx-version", + "payload", + "timing-source", +} + +var benchfmtUnitOrder = []string{ + benchfmtDispatchSpanUnit, + benchfmtCBActiveUnit, + benchfmtCBWallUnit, + benchfmtEffectiveGPUUnit, + benchfmtProfilerSampleCostUnit, + benchfmtProfilerCostSamplesUnit, + benchfmtGPRWCNTRSamplesUnit, + benchfmtDispatchesUnit, + benchfmtCommandBuffersUnit, + benchfmtEncodersUnit, +} + +type benchfmtConfig struct { + Key string + Value string +} + +type benchfmtValue struct { + Value float64 + Unit string +} + +type benchfmtRecord struct { + Suffix string + NameConfig []benchfmtConfig + Iters int + Config []benchfmtConfig + Values []benchfmtValue +} + +type benchfmtConfigFlags []benchfmtConfig + +func (f *benchfmtConfigFlags) String() string { + if f == nil { + return "" + } + values := make([]string, len(*f)) + for i, item := range *f { + values[i] = item.Key + "=" + item.Value + } + return strings.Join(values, ",") +} + +func (f *benchfmtConfigFlags) Set(value string) error { + key, configValue, ok := strings.Cut(value, "=") + if !ok { + return fmt.Errorf("bench config must be key=value") + } + item := benchfmtConfig{Key: key, Value: configValue} + if _, err := validateBenchfmtConfig([]benchfmtConfig{item}); err != nil { + return err + } + *f = append(*f, item) + return nil +} + +func (f *benchfmtConfigFlags) Type() string { + return "key=value" +} + +func addBenchfmtFlags(cmd *cobra.Command, enabled *bool, config *benchfmtConfigFlags) { + cmd.Flags().BoolVar(enabled, "benchfmt", false, "Output Go benchmark format") + cmd.Flags().Var(config, "bench-config", "Set benchmark configuration key=value (repeatable)") +} + +func validateBenchfmtFlags(enabled bool, config benchfmtConfigFlags) error { + if len(config) > 0 && !enabled { + return fmt.Errorf("--bench-config requires --benchfmt") + } + return nil +} + +func mergeBenchfmtConfig(defaults []benchfmtConfig, flags benchfmtConfigFlags) ([]benchfmtConfig, error) { + base, err := validateBenchfmtConfig(defaults) + if err != nil { + return nil, err + } + overrides := make(map[string]string, len(flags)) + for _, item := range flags { + if _, ok := overrides[item.Key]; ok { + return nil, fmt.Errorf("duplicate --bench-config key %q", item.Key) + } + if _, err := validateBenchfmtConfig([]benchfmtConfig{item}); err != nil { + return nil, err + } + overrides[item.Key] = item.Value + } + for key, value := range overrides { + base[key] = value + } + + merged := make([]benchfmtConfig, 0, len(base)) + for _, key := range benchfmtConfigOrder { + if value, ok := base[key]; ok { + merged = append(merged, benchfmtConfig{Key: key, Value: value}) + delete(base, key) + } + } + keys := make([]string, 0, len(base)) + for key := range base { + keys = append(keys, key) + } + sort.Strings(keys) + for _, key := range keys { + merged = append(merged, benchfmtConfig{Key: key, Value: base[key]}) + } + return merged, nil +} + +func writeBenchfmt(w io.Writer, record benchfmtRecord) error { + iters := record.Iters + if iters == 0 { + iters = 1 + } + if iters < 0 { + return fmt.Errorf("benchfmt iteration count must be positive") + } + if len(record.Values) == 0 { + return fmt.Errorf("benchfmt record has no measurements") + } + + config, err := validateBenchfmtConfig(record.Config) + if err != nil { + return err + } + values, err := validateBenchfmtValues(record.Values) + if err != nil { + return err + } + + var out strings.Builder + hasConfig := len(config) > 0 + for _, key := range benchfmtConfigOrder { + if value, ok := config[key]; ok { + fmt.Fprintf(&out, "%s: %s\n", key, value) + delete(config, key) + } + } + keys := make([]string, 0, len(config)) + for key := range config { + keys = append(keys, key) + } + sort.Strings(keys) + for _, key := range keys { + fmt.Fprintf(&out, "%s: %s\n", key, config[key]) + } + if hasConfig { + out.WriteByte('\n') + } + + out.WriteString("BenchmarkGPUTrace") + if suffix := sanitizeBenchfmtSuffix(record.Suffix); suffix != "" { + out.WriteByte('/') + out.WriteString(suffix) + } + for _, item := range record.NameConfig { + if !validBenchfmtConfigKey(item.Key) { + return fmt.Errorf("invalid benchfmt name config key %q", item.Key) + } + value := sanitizeBenchfmtSuffix(item.Value) + if value == "" { + return fmt.Errorf("invalid benchfmt name config value for %q", item.Key) + } + out.WriteByte('/') + out.WriteString(item.Key) + out.WriteByte('=') + out.WriteString(value) + } + fmt.Fprintf(&out, "-1 %d", iters) + for _, value := range values { + out.WriteByte(' ') + out.WriteString(strconv.FormatFloat(value.Value, 'g', -1, 64)) + out.WriteByte(' ') + out.WriteString(value.Unit) + } + out.WriteByte('\n') + + if _, err := io.WriteString(w, out.String()); err != nil { + return fmt.Errorf("write benchfmt: %w", err) + } + return nil +} + +func validateBenchfmtConfig(values []benchfmtConfig) (map[string]string, error) { + config := make(map[string]string, len(values)) + for _, item := range values { + if !validBenchfmtConfigKey(item.Key) { + return nil, fmt.Errorf("invalid benchfmt config key %q", item.Key) + } + if _, ok := config[item.Key]; ok { + return nil, fmt.Errorf("duplicate benchfmt config key %q", item.Key) + } + if item.Value == "" || strings.TrimSpace(item.Value) != item.Value || + strings.ContainsAny(item.Value, "\r\n") { + return nil, fmt.Errorf("invalid benchfmt config value for %q", item.Key) + } + config[item.Key] = item.Value + } + return config, nil +} + +func validBenchfmtConfigKey(key string) bool { + for i, r := range key { + if (i == 0 && !unicode.IsLower(r)) || unicode.IsSpace(r) || + unicode.IsUpper(r) || r == ':' { + return false + } + } + return key != "" +} + +func validateBenchfmtValues(values []benchfmtValue) ([]benchfmtValue, error) { + allowed := make(map[string]bool, len(benchfmtUnitOrder)) + for _, unit := range benchfmtUnitOrder { + allowed[unit] = true + } + seen := make(map[string]benchfmtValue, len(values)) + for _, value := range values { + if !allowed[value.Unit] { + return nil, fmt.Errorf("invalid benchfmt unit %q", value.Unit) + } + if _, ok := seen[value.Unit]; ok { + return nil, fmt.Errorf("duplicate benchfmt unit %q", value.Unit) + } + if math.IsNaN(value.Value) || math.IsInf(value.Value, 0) || value.Value < 0 { + return nil, fmt.Errorf("invalid benchfmt value for %q", value.Unit) + } + seen[value.Unit] = value + } + out := make([]benchfmtValue, 0, len(values)) + for _, unit := range benchfmtUnitOrder { + if value, ok := seen[unit]; ok { + out = append(out, value) + } + } + return out, nil +} + +func sanitizeBenchfmtSuffix(s string) string { + s = strings.TrimSpace(s) + var out strings.Builder + separator := false + for _, r := range s { + if unicode.IsLetter(r) || unicode.IsDigit(r) { + if separator && out.Len() > 0 { + out.WriteByte('_') + } + out.WriteRune(r) + separator = false + continue + } + separator = true + } + return out.String() +} diff --git a/cmd/gputrace/cmd/benchfmt_command_test.go b/cmd/gputrace/cmd/benchfmt_command_test.go new file mode 100644 index 00000000..5ed3d536 --- /dev/null +++ b/cmd/gputrace/cmd/benchfmt_command_test.go @@ -0,0 +1,149 @@ +package cmd + +import ( + "bytes" + "os" + "strings" + "testing" + + "github.com/spf13/cobra" + "golang.org/x/perf/benchfmt" +) + +func TestRawTraceBenchfmtAdmission(t *testing.T) { + tracePath := "../../../testdata/traces/01-single-encoder/01-single-encoder-run1.gputrace" + if _, err := os.Stat(tracePath); os.IsNotExist(err) { + t.Skipf("trace fixture not available: %s", tracePath) + } + + tests := []struct { + name string + run func(*cobra.Command) error + }{ + { + name: "profiler", + run: func(cmd *cobra.Command) error { + return runProfiler(cmd, []string{tracePath}, &profilerOptions{ + benchfmt: true, + limit: 20, + }) + }, + }, + { + name: "timing", + run: func(cmd *cobra.Command) error { + return runTiming(cmd, []string{tracePath}, &timingOptions{ + benchfmt: true, + }) + }, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + var out bytes.Buffer + cmd := &cobra.Command{} + cmd.SetOut(&out) + + if err := test.run(cmd); err == nil { + t.Fatal("command returned nil error for raw trace") + } + if out.Len() != 0 { + t.Fatalf("command wrote stdout on error:\n%s", out.String()) + } + }) + } +} + +func TestStatsRawTraceBenchfmtIsStructuralOnly(t *testing.T) { + tracePath := "../../../testdata/traces/01-single-encoder/01-single-encoder-run1.gputrace" + if _, err := os.Stat(tracePath); os.IsNotExist(err) { + t.Skipf("trace fixture not available: %s", tracePath) + } + + var out bytes.Buffer + cmd := &cobra.Command{} + cmd.SetOut(&out) + if err := runStats(cmd, []string{tracePath}, &statsOptions{benchfmt: true}); err != nil { + t.Fatal(err) + } + + result := readSingleBenchfmtResult(t, out.String()) + for _, unit := range []string{ + benchfmtDispatchesUnit, + benchfmtCommandBuffersUnit, + } { + if value, ok := result.Value(unit); !ok || value <= 0 { + t.Errorf("%s = %v, %v, want positive structural value", unit, value, ok) + } + } + for _, unit := range []string{ + benchfmtDispatchSpanUnit, + benchfmtCBActiveUnit, + benchfmtCBWallUnit, + benchfmtEffectiveGPUUnit, + benchfmtProfilerCostSamplesUnit, + benchfmtProfilerSampleCostUnit, + } { + if value, ok := result.Value(unit); ok { + t.Errorf("%s = %v, want omitted for raw trace", unit, value) + } + } +} + +func TestStatsBenchfmtRepeatedResults(t *testing.T) { + tracePath := "../../../testdata/traces/01-single-encoder/01-single-encoder-run1.gputrace" + if _, err := os.Stat(tracePath); os.IsNotExist(err) { + t.Skipf("trace fixture not available: %s", tracePath) + } + + var out bytes.Buffer + cmd := &cobra.Command{} + cmd.SetOut(&out) + for range 2 { + if err := runStats(cmd, []string{tracePath}, &statsOptions{benchfmt: true}); err != nil { + t.Fatal(err) + } + } + + reader := benchfmt.NewReader(strings.NewReader(out.String()), "test.bench") + var results []*benchfmt.Result + for reader.Scan() { + result, ok := reader.Result().(*benchfmt.Result) + if !ok { + t.Fatalf("record type = %T, want *benchfmt.Result", reader.Result()) + } + results = append(results, result) + } + if err := reader.Err(); err != nil { + t.Fatal(err) + } + if got, want := len(results), 2; got != want { + t.Fatalf("results = %d, want %d\n%s", got, want, out.String()) + } + if string(results[0].Name) != string(results[1].Name) { + t.Fatalf("benchmark names = %q and %q, want equal", results[0].Name, results[1].Name) + } + if results[0].Iters != 1 || results[1].Iters != 1 { + t.Fatalf("iterations = %d and %d, want 1 and 1", results[0].Iters, results[1].Iters) + } +} + +func readSingleBenchfmtResult(t *testing.T, output string) *benchfmt.Result { + t.Helper() + + reader := benchfmt.NewReader(strings.NewReader(output), "test.bench") + if !reader.Scan() { + t.Fatalf("read benchmark result: %v\n%s", reader.Err(), output) + } + result, ok := reader.Result().(*benchfmt.Result) + if !ok { + t.Fatalf("record type = %T, want *benchfmt.Result", reader.Result()) + } + if reader.Scan() { + t.Fatalf("unexpected second benchmark result:\n%s", output) + } + if err := reader.Err(); err != nil { + t.Fatal(err) + } + return result +} diff --git a/cmd/gputrace/cmd/benchfmt_metrics.go b/cmd/gputrace/cmd/benchfmt_metrics.go new file mode 100644 index 00000000..9882031a --- /dev/null +++ b/cmd/gputrace/cmd/benchfmt_metrics.go @@ -0,0 +1,229 @@ +package cmd + +import ( + "bytes" + "crypto/sha256" + "encoding/hex" + "fmt" + "io" + "os/exec" + "path/filepath" + "runtime" + "strings" + + "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/counter" + gputraceTrace "github.com/tmc/gputrace/internal/trace" + "github.com/tmc/gputrace/internal/tracebundle" +) + +func benchfmtDefaults(tracePath, timingSource string) []benchfmtConfig { + config := []benchfmtConfig{ + {Key: "goos", Value: runtime.GOOS}, + {Key: "goarch", Value: runtime.GOARCH}, + {Key: "pkg", Value: "github.com/tmc/gputrace"}, + {Key: "cpu", Value: benchfmtCPU()}, + {Key: "runtime", Value: inferBenchfmtRuntime(tracePath)}, + {Key: "model", Value: inferBenchfmtModel(tracePath)}, + {Key: "prompt-tokens", Value: "unknown"}, + {Key: "capture-range", Value: inferBenchfmtCaptureRange(tracePath)}, + {Key: "compile-mode", Value: "unknown"}, + {Key: "cache-mode", Value: inferBenchfmtCacheMode(tracePath)}, + {Key: "trace-uuid", Value: "unknown"}, + {Key: "mlx-version", Value: "unknown"}, + {Key: "payload", Value: "unknown"}, + } + if metadata, err := gputraceTrace.ReadMetadata(tracePath); err == nil && metadata.UUID != "" { + setBenchfmtConfig(config, "trace-uuid", metadata.UUID) + } + if payload, err := tracebundle.InspectPayload(tracePath); err == nil { + setBenchfmtConfig(config, "payload", string(payload.Class)) + } + if timingSource != "" { + config = append(config, benchfmtConfig{Key: "timing-source", Value: timingSource}) + } + return config +} + +func setBenchfmtConfig(config []benchfmtConfig, key, value string) { + for i := range config { + if config[i].Key == key { + config[i].Value = value + return + } + } +} + +func benchfmtCPU() string { + if runtime.GOOS == "darwin" { + for _, key := range []string{"machdep.cpu.brand_string", "hw.model"} { + out, err := exec.Command("sysctl", "-n", key).Output() + if err == nil { + if value := strings.TrimSpace(string(out)); value != "" { + return value + } + } + } + } + return "unknown" +} + +func inferBenchfmtRuntime(path string) string { + lower := strings.ToLower(filepath.ToSlash(path)) + for _, name := range []string{"go", "python", "swift"} { + if strings.Contains(lower, "/"+name+"/") || + strings.Contains(filepath.Base(lower), "-"+name+"-") { + return name + } + } + return "unknown" +} + +func inferBenchfmtModel(path string) string { + name := strings.TrimSuffix(filepath.Base(path), filepath.Ext(path)) + lower := strings.ToLower(name) + end := len(name) + for _, marker := range []string{ + "_tokens", "-tokens", "-staticmask", "-staticfix", "-warm", + "-producer", "-addmm", "-perfdata", + } { + if i := strings.Index(lower, marker); i >= 0 && i < end { + end = i + } + } + if end == 0 { + return "unknown" + } + return name[:end] +} + +func inferBenchfmtCaptureRange(path string) string { + name := strings.ToLower(filepath.Base(path)) + for _, marker := range []string{"tokens_", "tokens-", "tokens"} { + start := strings.Index(name, marker) + if start < 0 { + continue + } + s := name[start+len(marker):] + var left, right string + switch { + case strings.Contains(s, "_to_"): + left, right, _ = strings.Cut(s, "_to_") + case strings.Contains(s, "-to-"): + left, right, _ = strings.Cut(s, "-to-") + default: + left, right, _ = strings.Cut(s, "-") + } + left = leadingDigits(left) + right = leadingDigits(right) + if left != "" && right != "" { + return left + ":" + right + } + } + return "unknown" +} + +func leadingDigits(s string) string { + for i, r := range s { + if r < '0' || r > '9' { + return s[:i] + } + } + return s +} + +func inferBenchfmtCacheMode(path string) string { + if strings.Contains(strings.ToLower(filepath.Base(path)), "warm") { + return "warm" + } + return "unknown" +} + +func benchfmtProfilerValues(stats *counter.StreamDataStats, executionCost []counter.ExecutionCostByFunction) []benchfmtValue { + values := make([]benchfmtValue, 0, 9) + if stats.TotalDispatchTimeUs > 0 { + values = append(values, benchfmtValue{Value: float64(stats.TotalDispatchTimeUs) * 1000, Unit: benchfmtDispatchSpanUnit}) + } + if stats.CommandBufferActiveNs > 0 { + values = append(values, benchfmtValue{Value: float64(stats.CommandBufferActiveNs), Unit: benchfmtCBActiveUnit}) + } + if stats.CommandBufferWallNs > 0 { + values = append(values, benchfmtValue{Value: float64(stats.CommandBufferWallNs), Unit: benchfmtCBWallUnit}) + } + if stats.EffectiveGPUTimeNs != nil { + values = append(values, benchfmtValue{Value: float64(*stats.EffectiveGPUTimeNs), Unit: benchfmtEffectiveGPUUnit}) + } + samples := 0 + for _, dispatch := range stats.Dispatches { + samples += dispatch.SampleCount + } + if samples > 0 { + values = append(values, benchfmtValue{Value: float64(samples), Unit: benchfmtGPRWCNTRSamplesUnit}) + } + costSamples := 0 + for _, cost := range executionCost { + costSamples += cost.SampleCount + } + if costSamples > 0 { + values = append(values, benchfmtValue{Value: float64(costSamples), Unit: benchfmtProfilerCostSamplesUnit}) + } + values = append(values, benchfmtValue{Value: float64(stats.NumGPUCommands), Unit: benchfmtDispatchesUnit}) + if stats.Timeline != nil { + values = append(values, benchfmtValue{Value: float64(len(stats.Timeline.CommandBufferTimestamps)), Unit: benchfmtCommandBuffersUnit}) + } + values = append(values, benchfmtValue{Value: float64(stats.NumEncoders), Unit: benchfmtEncodersUnit}) + return values +} + +func benchfmtStructuralValues(stats *gputrace.TraceStatistics) []benchfmtValue { + values := []benchfmtValue{ + {Value: float64(stats.DispatchCalls), Unit: benchfmtDispatchesUnit}, + {Value: float64(stats.CommandBuffers), Unit: benchfmtCommandBuffersUnit}, + } + if stats.ComputeEncodersAvailable { + values = append(values, benchfmtValue{Value: float64(stats.ComputeEncoders), Unit: benchfmtEncodersUnit}) + } + return values +} + +func writeProfilerBenchfmt(w io.Writer, tracePath string, stats *counter.StreamDataStats, executionCost []counter.ExecutionCostByFunction, flags benchfmtConfigFlags) error { + config, err := mergeBenchfmtConfig(benchfmtDefaults(tracePath, stats.TimingSource), flags) + if err != nil { + return err + } + var out bytes.Buffer + if err := writeBenchfmt(&out, benchfmtRecord{ + Config: config, + Values: benchfmtProfilerValues(stats, executionCost), + }); err != nil { + return err + } + for _, cost := range executionCost { + if err := writeBenchfmt(&out, benchfmtRecord{ + NameConfig: []benchfmtConfig{{ + Key: "function", + Value: benchfmtSampleCostName(cost.FunctionName), + }}, + Config: config, + Values: []benchfmtValue{{ + Value: cost.CostPercent, + Unit: benchfmtProfilerSampleCostUnit, + }}, + }); err != nil { + return err + } + } + if _, err := w.Write(out.Bytes()); err != nil { + return fmt.Errorf("write benchfmt: %w", err) + } + return nil +} + +func benchfmtSampleCostName(function string) string { + name := []rune(sanitizeBenchfmtSuffix(function)) + if len(name) > 80 { + name = name[:80] + } + sum := sha256.Sum256([]byte(function)) + return string(name) + "_" + hex.EncodeToString(sum[:4]) +} diff --git a/cmd/gputrace/cmd/benchfmt_metrics_test.go b/cmd/gputrace/cmd/benchfmt_metrics_test.go new file mode 100644 index 00000000..47a20aa3 --- /dev/null +++ b/cmd/gputrace/cmd/benchfmt_metrics_test.go @@ -0,0 +1,222 @@ +package cmd + +import ( + "bytes" + "math" + "strings" + "testing" + + "github.com/tmc/gputrace/internal/counter" + "golang.org/x/perf/benchfmt" +) + +func TestInferBenchfmtProvenance(t *testing.T) { + tests := []struct { + path string + runtime string + model string + captureRange string + cacheMode string + }{ + { + path: "/traces/go/qwen25-05b-staticmask-warm-tokens2-4-rep1-perfdata.gputrace", + runtime: "go", + model: "qwen25-05b", + captureRange: "2:4", + cacheMode: "warm", + }, + { + path: "/traces/python/qwen25-05b-warm_tokens_2_to_4.gputrace", + runtime: "python", + model: "qwen25-05b", + captureRange: "2:4", + cacheMode: "warm", + }, + { + path: "/traces/swift/laguna-xs21_tokens_1_to_2_layer2.gputrace", + runtime: "swift", + model: "laguna-xs21", + captureRange: "1:2", + cacheMode: "unknown", + }, + } + for _, test := range tests { + t.Run(test.runtime, func(t *testing.T) { + if got := inferBenchfmtRuntime(test.path); got != test.runtime { + t.Fatalf("runtime = %q, want %q", got, test.runtime) + } + if got := inferBenchfmtModel(test.path); got != test.model { + t.Fatalf("model = %q, want %q", got, test.model) + } + if got := inferBenchfmtCaptureRange(test.path); got != test.captureRange { + t.Fatalf("capture range = %q, want %q", got, test.captureRange) + } + if got := inferBenchfmtCacheMode(test.path); got != test.cacheMode { + t.Fatalf("cache mode = %q, want %q", got, test.cacheMode) + } + }) + } +} + +func TestBenchfmtDefaultsKeepUnavailableProvenance(t *testing.T) { + config := benchfmtDefaults("/missing/go/model_tokens_2_to_4.gputrace", "") + values := make(map[string]string, len(config)) + for _, item := range config { + values[item.Key] = item.Value + } + for _, key := range []string{"trace-uuid", "payload"} { + if got := values[key]; got != "unknown" { + t.Fatalf("%s = %q, want unknown", key, got) + } + } +} + +func TestWriteProfilerBenchfmtOmitsUnavailableTiming(t *testing.T) { + stats := &counter.StreamDataStats{ + NumEncoders: 3, + NumGPUCommands: 7, + TotalDispatchTimeUs: 42, + TimingSource: "gpuCommandInfoData cumulative offsets", + Dispatches: []counter.DispatchInfo{ + {SampleCount: 2}, + {SampleCount: 3}, + }, + } + var out bytes.Buffer + if err := writeProfilerBenchfmt(&out, "/traces/go/model_tokens2-4.gputrace", stats, nil, nil); err != nil { + t.Fatal(err) + } + if strings.Contains(out.String(), benchfmtCBActiveUnit) || + strings.Contains(out.String(), benchfmtCBWallUnit) || + strings.Contains(out.String(), benchfmtEffectiveGPUUnit) { + t.Fatalf("output contains unavailable timing:\n%s", out.String()) + } + + reader := benchfmt.NewReader(strings.NewReader(out.String()), "test.bench") + if !reader.Scan() { + t.Fatalf("Scan: %v", reader.Err()) + } + result := reader.Result().(*benchfmt.Result) + for unit, want := range map[string]float64{ + benchfmtDispatchSpanUnit: 42000, + benchfmtGPRWCNTRSamplesUnit: 5, + benchfmtDispatchesUnit: 7, + benchfmtEncodersUnit: 3, + } { + if got, ok := result.Value(unit); !ok || got != want { + t.Fatalf("%s = %v, %v, want %v, true", unit, got, ok, want) + } + } +} + +func TestWriteProfilerBenchfmtExecutionCost(t *testing.T) { + stats := &counter.StreamDataStats{ + NumEncoders: 3, + NumGPUCommands: 7, + TimingSource: "streamData", + } + costs := []counter.ExecutionCostByFunction{ + { + FunctionName: "steel/gemm", + CostPercent: 20.25, + SampleCount: 4, + }, + { + FunctionName: "steel gemm", + CostPercent: 7.5, + SampleCount: 2, + }, + } + flags := benchfmtConfigFlags{{Key: "experiment", Value: "cost-test"}} + + var out bytes.Buffer + if err := writeProfilerBenchfmt(&out, "/traces/go/model_tokens2-4.gputrace", stats, costs, flags); err != nil { + t.Fatal(err) + } + + type resultSnapshot struct { + name string + experiment string + timingSource string + values map[string]float64 + } + reader := benchfmt.NewReader(strings.NewReader(out.String()), "test.bench") + var results []resultSnapshot + for reader.Scan() { + result, ok := reader.Result().(*benchfmt.Result) + if !ok { + t.Fatalf("record type = %T, want *benchfmt.Result", reader.Result()) + } + values := make(map[string]float64) + for _, value := range result.Values { + values[value.Unit] = value.Value + } + results = append(results, resultSnapshot{ + name: string(result.Name), + experiment: result.GetConfig("experiment"), + timingSource: result.GetConfig("timing-source"), + values: values, + }) + } + if err := reader.Err(); err != nil { + t.Fatal(err) + } + if got, want := len(results), 3; got != want { + t.Fatalf("results = %d, want %d\n%s", got, want, out.String()) + } + + if got := results[0].experiment; got != "cost-test" { + t.Fatalf("main experiment config = %q, want %q", got, "cost-test") + } + if got, ok := results[0].values[benchfmtProfilerCostSamplesUnit]; !ok || got != 6 { + t.Fatalf("main cost samples = %v, %v, want 6, true", got, ok) + } + + names := make(map[string]bool) + for i, result := range results[1:] { + if got := result.experiment; got != "cost-test" { + t.Fatalf("cost %d experiment config = %q, want %q", i, got, "cost-test") + } + if got := result.timingSource; got != "streamData" { + t.Fatalf("cost %d timing-source config = %q, want %q", i, got, "streamData") + } + got, ok := result.values[benchfmtProfilerSampleCostUnit] + if !ok || got != costs[i].CostPercent { + t.Fatalf("cost %d value = %v, %v, want %v, true", i, got, ok, costs[i].CostPercent) + } + name := result.name + if names[name] { + t.Fatalf("duplicate benchmark name %q for colliding sanitized function names", name) + } + names[name] = true + base, parts := benchfmt.Name(result.name).Parts() + if got := string(base); got != "GPUTrace" { + t.Fatalf("cost %d base name = %q, want GPUTrace", i, got) + } + if len(parts) != 2 || !strings.HasPrefix(string(parts[0]), "/function=steel_gemm_") { + t.Fatalf("cost %d name parts = %q, want function field and GOMAXPROCS", i, parts) + } + } +} + +func TestWriteProfilerBenchfmtInvalidExecutionCostIsAtomic(t *testing.T) { + stats := &counter.StreamDataStats{ + NumEncoders: 1, + NumGPUCommands: 1, + TimingSource: "streamData", + } + costs := []counter.ExecutionCostByFunction{{ + FunctionName: "invalid", + CostPercent: math.NaN(), + SampleCount: 1, + }} + + var out bytes.Buffer + err := writeProfilerBenchfmt(&out, "/traces/go/model_tokens2-4.gputrace", stats, costs, nil) + if err == nil || !strings.Contains(err.Error(), "invalid benchfmt value") { + t.Fatalf("error = %v, want invalid benchfmt value", err) + } + if out.Len() != 0 { + t.Fatalf("partial output on error:\n%s", out.String()) + } +} diff --git a/cmd/gputrace/cmd/benchfmt_test.go b/cmd/gputrace/cmd/benchfmt_test.go new file mode 100644 index 00000000..23050d02 --- /dev/null +++ b/cmd/gputrace/cmd/benchfmt_test.go @@ -0,0 +1,276 @@ +package cmd + +import ( + "bytes" + "math" + "strings" + "testing" + + "github.com/spf13/cobra" + "golang.org/x/perf/benchfmt" +) + +func TestWriteBenchfmt(t *testing.T) { + record := benchfmtRecord{ + Suffix: "Qwen 2.5/0.5B", + Config: []benchfmtConfig{ + {Key: "timing-source", Value: "streamData"}, + {Key: "goarch", Value: "arm64"}, + {Key: "goos", Value: "darwin"}, + {Key: "trace-uuid", Value: "ABC-123"}, + {Key: "pkg", Value: "github.com/tmc/gputrace"}, + }, + Values: []benchfmtValue{ + {Value: 23170000, Unit: benchfmtDispatchSpanUnit}, + {Value: 869, Unit: benchfmtDispatchesUnit}, + {Value: 30, Unit: benchfmtCommandBuffersUnit}, + }, + } + var out bytes.Buffer + if err := writeBenchfmt(&out, record); err != nil { + t.Fatal(err) + } + want := `goos: darwin +goarch: arm64 +pkg: github.com/tmc/gputrace +trace-uuid: ABC-123 +timing-source: streamData + +BenchmarkGPUTrace/Qwen_2_5_0_5B-1 1 2.317e+07 dispatch_span_ns/trace 869 dispatches/trace 30 command-buffers/trace +` + if got := out.String(); got != want { + t.Fatalf("output:\n%s\nwant:\n%s", got, want) + } + + reader := benchfmt.NewReader(strings.NewReader(out.String()), "test.bench") + if !reader.Scan() { + t.Fatalf("Scan: %v", reader.Err()) + } + result, ok := reader.Result().(*benchfmt.Result) + if !ok { + t.Fatalf("record type = %T", reader.Result()) + } + if got, want := string(result.Name), "GPUTrace/Qwen_2_5_0_5B-1"; got != want { + t.Fatalf("name = %q, want %q", got, want) + } + if result.Iters != 1 { + t.Fatalf("iters = %d, want 1", result.Iters) + } + if len(result.Values) != 3 { + t.Fatalf("values = %d, want 3", len(result.Values)) + } + for _, want := range record.Config { + if got := result.GetConfig(want.Key); got != want.Value { + t.Fatalf("config %q = %q, want %q", want.Key, got, want.Value) + } + } + for _, want := range record.Values { + got, ok := result.Value(want.Unit) + if !ok || got != want.Value { + t.Fatalf("value %q = %v, %v, want %v, true", want.Unit, got, ok, want.Value) + } + } + if reader.Scan() { + t.Fatalf("unexpected record %T", reader.Result()) + } + if err := reader.Err(); err != nil { + t.Fatal(err) + } +} + +func TestWriteBenchfmtDefaultName(t *testing.T) { + var out bytes.Buffer + err := writeBenchfmt(&out, benchfmtRecord{ + Values: []benchfmtValue{{Value: 1, Unit: benchfmtEncodersUnit}}, + }) + if err != nil { + t.Fatal(err) + } + if got, want := out.String(), "BenchmarkGPUTrace-1 1 1 encoders/trace\n"; got != want { + t.Fatalf("output = %q, want %q", got, want) + } +} + +func TestWriteBenchfmtRejectsInvalidRecords(t *testing.T) { + tests := []struct { + name string + record benchfmtRecord + want string + }{ + { + name: "no values", + record: benchfmtRecord{}, + want: "no measurements", + }, + { + name: "negative iterations", + record: benchfmtRecord{ + Iters: -1, + Values: []benchfmtValue{{Value: 1, Unit: benchfmtEncodersUnit}}, + }, + want: "iteration count", + }, + { + name: "duplicate unit", + record: benchfmtRecord{Values: []benchfmtValue{ + {Value: 1, Unit: benchfmtDispatchesUnit}, + {Value: 2, Unit: benchfmtDispatchesUnit}, + }}, + want: "duplicate benchfmt unit", + }, + { + name: "nonfinite", + record: benchfmtRecord{Values: []benchfmtValue{ + {Value: math.Inf(1), Unit: benchfmtDispatchesUnit}, + }}, + want: "invalid benchfmt value", + }, + { + name: "negative value", + record: benchfmtRecord{Values: []benchfmtValue{ + {Value: -1, Unit: benchfmtDispatchesUnit}, + }}, + want: "invalid benchfmt value", + }, + { + name: "unknown unit", + record: benchfmtRecord{Values: []benchfmtValue{ + {Value: 1, Unit: "ns/op"}, + }}, + want: "invalid benchfmt unit", + }, + { + name: "uppercase config", + record: benchfmtRecord{ + Config: []benchfmtConfig{{Key: "GoOS", Value: "darwin"}}, + Values: []benchfmtValue{{Value: 1, Unit: benchfmtDispatchesUnit}}, + }, + want: "invalid benchfmt config key", + }, + { + name: "config newline", + record: benchfmtRecord{ + Config: []benchfmtConfig{{Key: "model", Value: "qwen\nbad: value"}}, + Values: []benchfmtValue{{Value: 1, Unit: benchfmtDispatchesUnit}}, + }, + want: "invalid benchfmt config value", + }, + { + name: "duplicate config", + record: benchfmtRecord{ + Config: []benchfmtConfig{ + {Key: "model", Value: "a"}, + {Key: "model", Value: "b"}, + }, + Values: []benchfmtValue{{Value: 1, Unit: benchfmtDispatchesUnit}}, + }, + want: "duplicate benchfmt config key", + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + var out bytes.Buffer + err := writeBenchfmt(&out, test.record) + if err == nil || !strings.Contains(err.Error(), test.want) { + t.Fatalf("error = %v, want substring %q", err, test.want) + } + if out.Len() != 0 { + t.Fatalf("partial output = %q", out.String()) + } + }) + } +} + +func TestBenchfmtConfigFlagsAndMerge(t *testing.T) { + var enabled bool + var flags benchfmtConfigFlags + cmd := &cobra.Command{Use: "test"} + addBenchfmtFlags(cmd, &enabled, &flags) + if err := cmd.ParseFlags([]string{ + "--benchfmt", + "--bench-config", "model=qwen2.5", + "--bench-config=goos=darwin", + }); err != nil { + t.Fatal(err) + } + if !enabled { + t.Fatal("benchfmt flag is false") + } + if err := validateBenchfmtFlags(enabled, flags); err != nil { + t.Fatal(err) + } + + got, err := mergeBenchfmtConfig([]benchfmtConfig{ + {Key: "goos", Value: "linux"}, + {Key: "goarch", Value: "arm64"}, + }, flags) + if err != nil { + t.Fatal(err) + } + want := []benchfmtConfig{ + {Key: "goos", Value: "darwin"}, + {Key: "goarch", Value: "arm64"}, + {Key: "model", Value: "qwen2.5"}, + } + if len(got) != len(want) { + t.Fatalf("merged config = %#v, want %#v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Fatalf("merged config[%d] = %#v, want %#v", i, got[i], want[i]) + } + } +} + +func TestBenchfmtConfigFlagsRejectInvalid(t *testing.T) { + tests := []struct { + name string + args []string + }{ + {name: "missing equals", args: []string{"--bench-config", "model"}}, + {name: "uppercase key", args: []string{"--bench-config", "Goos=darwin"}}, + {name: "newline", args: []string{"--bench-config", "model=qwen\nbad: value"}}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + var enabled bool + var flags benchfmtConfigFlags + cmd := &cobra.Command{Use: "test"} + addBenchfmtFlags(cmd, &enabled, &flags) + if err := cmd.ParseFlags(test.args); err == nil { + t.Fatal("ParseFlags succeeded") + } + }) + } +} + +func TestBenchfmtConfigFlagsAllowExperimentKeys(t *testing.T) { + var flags benchfmtConfigFlags + if err := flags.Set("kv-layout=static"); err != nil { + t.Fatal(err) + } + got, err := mergeBenchfmtConfig(nil, flags) + if err != nil { + t.Fatal(err) + } + if len(got) != 1 || got[0] != (benchfmtConfig{Key: "kv-layout", Value: "static"}) { + t.Fatalf("config = %#v", got) + } +} + +func TestBenchfmtConfigRequiresBenchfmt(t *testing.T) { + flags := benchfmtConfigFlags{{Key: "model", Value: "qwen"}} + if err := validateBenchfmtFlags(false, flags); err == nil { + t.Fatal("validateBenchfmtFlags succeeded") + } +} + +func TestMergeBenchfmtConfigRejectsDuplicateOverride(t *testing.T) { + _, err := mergeBenchfmtConfig(nil, benchfmtConfigFlags{ + {Key: "model", Value: "a"}, + {Key: "model", Value: "b"}, + }) + if err == nil || !strings.Contains(err.Error(), "duplicate --bench-config") { + t.Fatalf("error = %v", err) + } +} diff --git a/cmd/gputrace/cmd/brief.go b/cmd/gputrace/cmd/brief.go index 31babc69..a0e586c9 100644 --- a/cmd/gputrace/cmd/brief.go +++ b/cmd/gputrace/cmd/brief.go @@ -117,13 +117,21 @@ type briefPayload struct { } type briefTraceSummary struct { - Label string `json:"label"` - TotalGPUUs int `json:"total_gpu_us"` - Dispatches int `json:"dispatches"` - ProfilerEncoders int `json:"profiler_encoders"` - RawComputeEncoders int `json:"raw_compute_encoders"` - Buffers int `json:"buffers"` - BufferBytes uint64 `json:"buffer_bytes"` + Label string `json:"label"` + Path string `json:"path"` + TotalGPUUs int `json:"total_gpu_us"` + Dispatches int `json:"dispatches"` + ProfilerEncoders int `json:"profiler_encoders"` + RawComputeEncoders *int `json:"raw_compute_encoders"` + RawEncodersAvailable bool `json:"raw_compute_encoders_available"` + RawEncodersSource string `json:"raw_compute_encoders_source"` + Buffers int `json:"buffers"` + BufferBytes uint64 `json:"buffer_bytes"` + CommandBufferActiveUs int `json:"command_buffer_active_us,omitempty"` + EffectiveGPUTimeUs *int `json:"effective_gpu_time_us,omitempty"` + TimingSource string `json:"timing_source,omitempty"` + AttributionLimited bool `json:"attribution_limited,omitempty"` + Warnings []string `json:"warnings,omitempty"` } func newBriefHeader() briefHeader { @@ -240,7 +248,7 @@ func loadBriefTrace(path, label string) (briefTraceData, error) { if err != nil { return briefTraceData{}, fmt.Errorf("summarize buffers: %w", err) } - rawComputeEncoders := countRawComputeEncoders(trace) + rawComputeEncoders, rawEncodersAvailable, rawEncodersSource := inspectRawComputeEncoders(trace) insights, err := gputrace.GenerateInsights(trace) if err != nil { return briefTraceData{}, fmt.Errorf("generate insights: %w", err) @@ -249,24 +257,33 @@ func loadBriefTrace(path, label string) (briefTraceData, error) { return briefTraceData{ data: data, summary: briefTraceSummary{ - Label: label, - TotalGPUUs: totalBriefGPUUs(data.Dispatches), - Dispatches: len(data.Dispatches), - ProfilerEncoders: len(data.Encoders), - RawComputeEncoders: rawComputeEncoders, - Buffers: buffers, - BufferBytes: bytes, + Label: label, + Path: path, + TotalGPUUs: totalBriefGPUUs(data.Dispatches), + Dispatches: len(data.Dispatches), + ProfilerEncoders: len(data.Encoders), + RawComputeEncoders: rawComputeEncoders, + RawEncodersAvailable: rawEncodersAvailable, + RawEncodersSource: rawEncodersSource, + Buffers: buffers, + BufferBytes: bytes, + CommandBufferActiveUs: data.CommandBufferActiveUs, + EffectiveGPUTimeUs: data.EffectiveGPUTimeUs, + TimingSource: data.TimingSource, + AttributionLimited: data.AttributionLimited, + Warnings: append([]string(nil), data.Warnings...), }, insights: insights.Insights, }, nil } -func countRawComputeEncoders(trace *gputrace.Trace) int { - n, err := trace.CountComputeEncoders() - if err != nil || n == 0 { - return 0 +func inspectRawComputeEncoders(trace *gputrace.Trace) (*int, bool, string) { + count := trace.InspectComputeEncoderCount() + if !count.Available { + return nil, false, count.Source } - return n + n := count.Count + return &n, true, count.Source } func briefBufferSummary(path string, trace *gputrace.Trace) (int, uint64, error) { @@ -317,17 +334,83 @@ func writeBriefJSON(w io.Writer, brief briefDocument) error { } func writeBriefMarkdown(w io.Writer, brief briefDocument) error { - _, err := fmt.Fprintf(w, "# gputrace brief\n\nA `%s`: %dus, %d dispatches\n\nB `%s`: %dus, %d dispatches\n\nDelta: %+dus\n\nTop outliers:\n", - brief.Payload.TraceA.Label, brief.Payload.TraceA.TotalGPUUs, brief.Payload.TraceA.Dispatches, - brief.Payload.TraceB.Label, brief.Payload.TraceB.TotalGPUUs, brief.Payload.TraceB.Dispatches, - brief.Payload.TotalDeltaUs) + a, b := brief.Payload.TraceA, brief.Payload.TraceB + _, err := fmt.Fprintf(w, "# gputrace brief\n\n%s.\n\n", brief.Header.Contract) if err != nil { return fmt.Errorf("write brief markdown: %w", err) } + if _, err := fmt.Fprintf(w, "| | Trace A | Trace B |\n|---|---:|---:|\n"+ + "| Label | `%s` | `%s` |\n"+ + "| Path | `%s` | `%s` |\n"+ + "| Dispatch span | %d µs | %d µs |\n"+ + "| Dispatches | %d | %d |\n"+ + "| Profiler encoders | %d | %d |\n"+ + "| Command-buffer active time | %s | %s |\n"+ + "| Xcode Effective GPU Time | %s | %s |\n\n", + a.Label, b.Label, a.Path, b.Path, + a.TotalGPUUs, b.TotalGPUUs, a.Dispatches, b.Dispatches, + a.ProfilerEncoders, b.ProfilerEncoders, + formatBriefMicros(a.CommandBufferActiveUs, false), formatBriefMicros(b.CommandBufferActiveUs, false), + formatBriefOptionalMicros(a.EffectiveGPUTimeUs), formatBriefOptionalMicros(b.EffectiveGPUTimeUs)); err != nil { + return fmt.Errorf("write brief markdown summary: %w", err) + } + if a.TimingSource != "" || b.TimingSource != "" { + fmt.Fprintf(w, "Timing sources: A `%s`; B `%s`.\n\n", briefValueOrUnavailable(a.TimingSource), briefValueOrUnavailable(b.TimingSource)) + } + if a.AttributionLimited || b.AttributionLimited { + fmt.Fprintln(w, "> Attribution warning: per-dispatch values are cumulative-offset deltas and may include boundary or gap time. Function outliers are hypotheses, not direct device-duration measurements.") + fmt.Fprintln(w) + } + for _, warning := range uniqueBriefWarnings(a.Warnings, b.Warnings) { + fmt.Fprintf(w, "- Warning: %s\n", warning) + } + if len(a.Warnings)+len(b.Warnings) > 0 { + fmt.Fprintln(w) + } + fmt.Fprintf(w, "Dispatch-span delta (A-B): **%+d µs**\n\n## Top attributed outliers\n\n", brief.Payload.TotalDeltaUs) for _, outlier := range brief.Payload.Outliers { - if _, err := fmt.Fprintf(w, "- `%s` `%s`: %dus vs %dus, abs delta %dus\n", outlier.FunctionName, outlier.ThreadgroupSig, outlier.AUs, outlier.BUs, outlier.AbsDeltaUs); err != nil { + if _, err := fmt.Fprintf(w, "- `%s` `%s`: %d µs vs %d µs, absolute delta %d µs\n", outlier.FunctionName, outlier.ThreadgroupSig, outlier.AUs, outlier.BUs, outlier.AbsDeltaUs); err != nil { return fmt.Errorf("write brief markdown: %w", err) } } + if brief.Payload.Truncated { + fmt.Fprintf(w, "\n_Outliers truncated: %d additional rows omitted by `--token-budget`._\n", brief.Payload.DroppedCount) + } return nil } + +func formatBriefMicros(value int, zeroAvailable bool) string { + if value == 0 && !zeroAvailable { + return "unavailable" + } + return fmt.Sprintf("%d µs", value) +} + +func formatBriefOptionalMicros(value *int) string { + if value == nil { + return "unavailable" + } + return formatBriefMicros(*value, true) +} + +func briefValueOrUnavailable(value string) string { + if value == "" { + return "unavailable" + } + return value +} + +func uniqueBriefWarnings(groups ...[]string) []string { + seen := make(map[string]bool) + var out []string + for _, group := range groups { + for _, warning := range group { + if warning == "" || seen[warning] { + continue + } + seen[warning] = true + out = append(out, warning) + } + } + return out +} diff --git a/cmd/gputrace/cmd/brief_test.go b/cmd/gputrace/cmd/brief_test.go index ff66864e..41452cef 100644 --- a/cmd/gputrace/cmd/brief_test.go +++ b/cmd/gputrace/cmd/brief_test.go @@ -3,6 +3,7 @@ package cmd import ( "bytes" "encoding/json" + "strings" "testing" "github.com/tmc/gputrace/internal/difftrace" @@ -35,21 +36,26 @@ func TestBriefTokenBudgetTruncatesOutliers(t *testing.T) { } func testBriefDocument(label string, total int) briefDocument { + rawA, rawB := 74, 18 return briefDocument{ SchemaVersion: "1", Header: newBriefHeader(), Payload: briefPayload{ TraceA: briefTraceSummary{ - Label: label, - TotalGPUUs: total, - ProfilerEncoders: 9, - RawComputeEncoders: 74, + Label: label, + TotalGPUUs: total, + ProfilerEncoders: 9, + RawComputeEncoders: &rawA, + RawEncodersAvailable: true, + RawEncodersSource: "test", }, TraceB: briefTraceSummary{ - Label: "right", - TotalGPUUs: 5, - ProfilerEncoders: 9, - RawComputeEncoders: 18, + Label: "right", + TotalGPUUs: 5, + ProfilerEncoders: 9, + RawComputeEncoders: &rawB, + RawEncodersAvailable: true, + RawEncodersSource: "test", }, }, } @@ -64,9 +70,11 @@ func TestBriefTraceSummaryEncoderFields(t *testing.T) { var got struct { Payload struct { TraceA struct { - ProfilerEncoders *int `json:"profiler_encoders"` - RawComputeEncoders *int `json:"raw_compute_encoders"` - ComputeEncoders *int `json:"compute_encoders"` + ProfilerEncoders *int `json:"profiler_encoders"` + RawComputeEncoders *int `json:"raw_compute_encoders"` + RawEncodersAvailable bool `json:"raw_compute_encoders_available"` + RawEncodersSource string `json:"raw_compute_encoders_source"` + ComputeEncoders *int `json:"compute_encoders"` } `json:"trace_a"` TraceB struct { ProfilerEncoders *int `json:"profiler_encoders"` @@ -84,6 +92,9 @@ func TestBriefTraceSummaryEncoderFields(t *testing.T) { if got.Payload.TraceA.RawComputeEncoders == nil || *got.Payload.TraceA.RawComputeEncoders != 74 { t.Fatalf("trace_a raw_compute_encoders = %v, want 74", got.Payload.TraceA.RawComputeEncoders) } + if !got.Payload.TraceA.RawEncodersAvailable || got.Payload.TraceA.RawEncodersSource != "test" { + t.Fatalf("trace_a raw encoder provenance = %t %q", got.Payload.TraceA.RawEncodersAvailable, got.Payload.TraceA.RawEncodersSource) + } if got.Payload.TraceB.ProfilerEncoders == nil || *got.Payload.TraceB.ProfilerEncoders != 9 { t.Fatalf("trace_b profiler_encoders = %v, want 9", got.Payload.TraceB.ProfilerEncoders) } @@ -105,3 +116,64 @@ func marshalBriefHeader(t *testing.T, brief briefDocument) []byte { } return data } + +func TestWriteBriefMarkdownCarriesComparisonProvenance(t *testing.T) { + effective := 3900 + brief := briefDocument{ + Header: newBriefHeader(), + Payload: briefPayload{ + TraceA: briefTraceSummary{ + Label: "go", + Path: "/traces/go.gputrace", + TotalGPUUs: 12_150, + Dispatches: 488, + ProfilerEncoders: 2, + CommandBufferActiveUs: 3_828, + TimingSource: "streamData cumulative offsets", + AttributionLimited: true, + Warnings: []string{"encoder attribution unavailable"}, + }, + TraceB: briefTraceSummary{ + Label: "python", + Path: "/traces/python.gputrace", + TotalGPUUs: 10_769, + Dispatches: 413, + ProfilerEncoders: 2, + CommandBufferActiveUs: 3_913, + EffectiveGPUTimeUs: &effective, + TimingSource: "streamData cumulative offsets", + }, + TotalDeltaUs: 1_381, + Outliers: []difftrace.PipelinePair{{ + FunctionName: "kernel", + ThreadgroupSig: "1x1x1/1x1x1", + AUs: 10, + BUs: 5, + AbsDeltaUs: 5, + }}, + Truncated: true, + DroppedCount: 7, + }, + } + + var out bytes.Buffer + if err := writeBriefMarkdown(&out, brief); err != nil { + t.Fatalf("writeBriefMarkdown: %v", err) + } + got := out.String() + for _, want := range []string{ + "`/traces/go.gputrace`", + "`/traces/python.gputrace`", + "Dispatch span", + "Command-buffer active time", + "Xcode Effective GPU Time", + "Attribution warning:", + "encoder attribution unavailable", + "Dispatch-span delta (A-B): **+1381 µs**", + "7 additional rows omitted", + } { + if !strings.Contains(got, want) { + t.Fatalf("markdown missing %q:\n%s", want, got) + } + } +} diff --git a/cmd/gputrace/cmd/buffer_access.go b/cmd/gputrace/cmd/buffer_access.go index 2c5c20b6..9c66c019 100644 --- a/cmd/gputrace/cmd/buffer_access.go +++ b/cmd/gputrace/cmd/buffer_access.go @@ -21,20 +21,18 @@ func newBufferAccessCommand(opts *bufferAccessOptions) *cobra.Command { cmd := &cobra.Command{ Use: "buffer-access ", Short: "Analyze buffer access patterns", - Long: `Analyze buffer access patterns to identify optimization opportunities. + Long: `Analyze decoded buffer references and report attribution coverage. -This command analyzes Ct and Cul records to track: +This command currently analyzes structured Ct records to track: - Which encoders access which buffers - Buffer reuse frequency across encoders - Memory aliasing (multiple buffer names for same address) - Unused buffers (allocated but never accessed) - Read-only vs read-write buffers (future enhancement) -The analysis helps identify: -- Buffers that could be reused to reduce memory usage -- Unused buffers that waste memory -- Memory aliasing issues that could cause bugs -- Access patterns for optimization +Cul and other resource records are not yet attributed. Human and JSON output +report this limitation; optimization advice is withheld while attribution is +incomplete. Examples: # Analyze buffer access patterns @@ -69,6 +67,9 @@ func runBufferAccess(cmd *cobra.Command, args []string, opts *bufferAccessOption if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } // Analyze buffer access patterns analysis, err := gputrace.AnalyzeBufferAccess(trace) diff --git a/cmd/gputrace/cmd/buffer_access_test.go b/cmd/gputrace/cmd/buffer_access_test.go index 95350bdf..e168d23f 100644 --- a/cmd/gputrace/cmd/buffer_access_test.go +++ b/cmd/gputrace/cmd/buffer_access_test.go @@ -62,7 +62,7 @@ func TestRunBufferAccessJSONUsesCommandOutput(t *testing.T) { if err := json.Unmarshal(out.Bytes(), &got); err != nil { t.Fatalf("command output is invalid json: %v\n%s", err, out.String()) } - if got.BufferAccesses == nil || got.EncoderAccesses == nil { + if got.BufferAccesses == nil || got.BindingGroups == nil { t.Fatalf("json analysis maps were nil: %+v", got) } } @@ -71,17 +71,17 @@ func testBufferAccessAnalysis() *gputrace.BufferAccessAnalysis { return &gputrace.BufferAccessAnalysis{ BufferAccesses: map[uint64]*gputrace.BufferAccessInfo{ 0x1000: { - Address: 0x1000, - AccessCount: 2, - EncoderIDs: []int{1, 2}, - FirstAccess: 4, - LastAccess: 8, - IsShared: true, + Address: 0x1000, + AccessCount: 2, + GroupOrdinals: []int{1, 2}, + FirstAccess: 4, + LastAccess: 8, + IsShared: true, }, }, - EncoderAccesses: map[int]*gputrace.EncoderAccessInfo{ + BindingGroups: map[int]*gputrace.BindingGroupInfo{ 1: { - EncoderID: 1, + CSOrdinal: 1, BufferCount: 1, UniqueBuffers: []uint64{0x1000}, RecordIndices: []int{4}, @@ -92,9 +92,9 @@ func testBufferAccessAnalysis() *gputrace.BufferAccessAnalysis { AliasingDetected: true, AliasingInstances: []gputrace.BufferAlias{ { - Address: 0x1000, - Encoders: []int{1, 2}, - Indices: []int{4, 8}, + Address: 0x1000, + Groups: []int{1, 2}, + Indices: []int{4, 8}, }, }, } diff --git a/cmd/gputrace/cmd/buffer_timeline.go b/cmd/gputrace/cmd/buffer_timeline.go index 96aabfc9..ef87ac1a 100644 --- a/cmd/gputrace/cmd/buffer_timeline.go +++ b/cmd/gputrace/cmd/buffer_timeline.go @@ -27,21 +27,22 @@ type bufferTimelineOptions struct { func newBufferTimelineCommand(opts *bufferTimelineOptions) *cobra.Command { cmd := &cobra.Command{ Use: "buffer-timeline ", - Short: "Visualize buffer allocation and usage timeline", - Long: `Analyze and visualize buffer lifecycle events across the trace. + Short: "Visualize observed buffer-access spans", + Long: `Analyze and visualize observed buffer accesses across trace record order. -This command extracts buffer allocation, usage, and deallocation patterns -and presents them in various formats: +The first and last references are access observations, not allocation or +deallocation events. Memory totals are approximate upper bounds because the +capture does not provide complete lifetime attribution. - - ASCII: Terminal-based bar chart showing buffer lifetimes + - ASCII: Terminal-based bar chart showing observed access spans - summary: Text summary with statistics and top buffers - chrome: Chrome tracing format for ui.perfetto.dev - json: Raw JSON data The timeline shows: - - Buffer allocation/deallocation times - - Memory usage over time - - Peak memory usage + - First and last observed access + - Approximate memory upper bounds over record order + - Approximate peak memory upper bound - Buffer sizes and usage patterns Examples: @@ -92,6 +93,9 @@ func runBufferTimeline(cmd *cobra.Command, args []string, opts *bufferTimelineOp if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } // Extract buffer timeline timeline, err := gputrace.ExtractBufferTimeline(trace) @@ -143,14 +147,21 @@ func writeBufferTimelineOutput(outputPath, output string) error { } type bufferTimelineJSON struct { - TotalBuffers int `json:"total_buffers"` - PeakMemoryBytes uint64 `json:"peak_memory_bytes"` - PeakMemoryMB float64 `json:"peak_memory_mb"` - TotalAllocations int `json:"total_allocations"` - AverageLifetime float64 `json:"average_lifetime_records"` - MinRecordIndex int `json:"min_record_index"` - MaxRecordIndex int `json:"max_record_index"` - Buffers []bufferTimelineJSONBuffer `json:"buffers"` + TotalBuffers int `json:"total_buffers"` + PeakMemoryBytes uint64 `json:"peak_memory_bytes"` + PeakMemoryMB float64 `json:"peak_memory_mb"` + MemorySemantics string `json:"memory_semantics"` + TotalAllocations int `json:"total_allocations"` // Deprecated: use BuffersFirstSeen. + AverageLifetime float64 `json:"average_lifetime_records"` // Deprecated: use AverageAccessSpan. + BuffersFirstSeen int `json:"buffers_first_seen"` + AverageAccessSpan float64 `json:"average_observed_access_span_records"` + MinRecordIndex int `json:"min_record_index"` + MaxRecordIndex int `json:"max_record_index"` + ExpectedEncoders int `json:"expected_encoders"` + AttributedGroups int `json:"attributed_encoders"` + AttributionComplete bool `json:"attribution_complete"` + AttributionNote string `json:"attribution_note,omitempty"` + Buffers []bufferTimelineJSONBuffer `json:"buffers"` } type bufferTimelineJSONBuffer struct { @@ -171,13 +182,16 @@ type bufferTimelineChromeTrace struct { } type bufferTimelineChromeTraceHeader struct { - TotalBuffers int `json:"total_buffers"` - PeakMemoryBytes uint64 `json:"peak_memory_bytes"` - TotalAllocations int `json:"total_allocations"` - AverageLifetime float64 `json:"average_lifetime_records"` - MinRecordIndex int `json:"min_record_index"` - MaxRecordIndex int `json:"max_record_index"` - TimeUnit string `json:"time_unit"` + TotalBuffers int `json:"total_buffers"` + PeakMemoryBytes uint64 `json:"peak_memory_bytes"` + TotalAllocations int `json:"total_allocations"` + AverageLifetime float64 `json:"average_lifetime_records"` + MinRecordIndex int `json:"min_record_index"` + MaxRecordIndex int `json:"max_record_index"` + TimeUnit string `json:"time_unit"` + MemorySemantics string `json:"memory_semantics"` + AttributionComplete bool `json:"attribution_complete"` + AttributionNote string `json:"attribution_note,omitempty"` } type bufferTimelineCounterDelta struct { @@ -188,13 +202,20 @@ type bufferTimelineCounterDelta struct { func formatBufferTimelineJSON(timeline *gputrace.BufferTimelineAnalysis) (string, error) { doc := bufferTimelineJSON{ - TotalBuffers: timeline.TotalBuffers, - PeakMemoryBytes: timeline.PeakMemoryBytes, - PeakMemoryMB: timeline.PeakMemoryMB, - TotalAllocations: timeline.TotalAllocations, - AverageLifetime: timeline.AverageLifetime, - MinRecordIndex: timeline.MinRecordIndex, - MaxRecordIndex: timeline.MaxRecordIndex, + TotalBuffers: timeline.TotalBuffers, + PeakMemoryBytes: timeline.PeakMemoryBytes, + PeakMemoryMB: timeline.PeakMemoryMB, + MemorySemantics: "approximate upper bound from buffers referenced in decoded records", + TotalAllocations: timeline.TotalAllocations, + AverageLifetime: timeline.AverageLifetime, + BuffersFirstSeen: timeline.TotalAllocations, + AverageAccessSpan: timeline.AverageLifetime, + MinRecordIndex: timeline.MinRecordIndex, + MaxRecordIndex: timeline.MaxRecordIndex, + ExpectedEncoders: timeline.ExpectedEncoders, + AttributedGroups: timeline.AttributedGroups, + AttributionComplete: timeline.AttributionComplete, + AttributionNote: timeline.AttributionNote, } lifecycles := sortedBufferLifecycles(timeline) @@ -223,13 +244,16 @@ func formatBufferTimelineJSON(timeline *gputrace.BufferTimelineAnalysis) (string func formatBufferTimelineChrome(timeline *gputrace.BufferTimelineAnalysis) (string, error) { doc := bufferTimelineChromeTrace{ Metadata: bufferTimelineChromeTraceHeader{ - TotalBuffers: timeline.TotalBuffers, - PeakMemoryBytes: timeline.PeakMemoryBytes, - TotalAllocations: timeline.TotalAllocations, - AverageLifetime: timeline.AverageLifetime, - MinRecordIndex: timeline.MinRecordIndex, - MaxRecordIndex: timeline.MaxRecordIndex, - TimeUnit: "record_index_as_microseconds", + TotalBuffers: timeline.TotalBuffers, + PeakMemoryBytes: timeline.PeakMemoryBytes, + TotalAllocations: timeline.TotalAllocations, + AverageLifetime: timeline.AverageLifetime, + MinRecordIndex: timeline.MinRecordIndex, + MaxRecordIndex: timeline.MaxRecordIndex, + TimeUnit: "record_index_as_microseconds", + MemorySemantics: "approximate upper bound from buffers referenced in decoded records", + AttributionComplete: timeline.AttributionComplete, + AttributionNote: timeline.AttributionNote, }, } diff --git a/cmd/gputrace/cmd/buffers.go b/cmd/gputrace/cmd/buffers.go index 9219e51e..8181ebd1 100644 --- a/cmd/gputrace/cmd/buffers.go +++ b/cmd/gputrace/cmd/buffers.go @@ -6,6 +6,7 @@ import ( "encoding/csv" "encoding/json" "fmt" + "io" "math" "os" "path/filepath" @@ -15,6 +16,7 @@ import ( "github.com/spf13/cobra" "github.com/tmc/gputrace" "github.com/tmc/gputrace/internal/fmtutil" + tracepkg "github.com/tmc/gputrace/internal/trace" ) var buffersCmd = newBuffersCommand(&buffersCommandOptions{ @@ -22,6 +24,7 @@ var buffersCmd = newBuffersCommand(&buffersCommandOptions{ format: "table", inspectBytes: 256, inspectFormat: "hex", + limit: defaultHumanLimit, }) type buffersCommandOptions struct { @@ -33,9 +36,14 @@ type buffersCommandOptions struct { inspectBytes int inspectFormat string resources bool + limit int + all bool } func newBuffersCommand(opts *buffersCommandOptions) *cobra.Command { + if opts.limit == 0 { + opts.limit = defaultHumanLimit + } cmd := &cobra.Command{ Use: "buffers ", Short: "List buffers in a GPU trace", @@ -46,7 +54,7 @@ This command shows: - Buffer sizes - Buffer usage (total/unique) - Aliasing information (symlinks) - - Buffer bindings to encoders (with --verbose) + - Buffer bindings to encoders (with --bindings) The output can be sorted by size, ID, or name, and filtered by minimum size. @@ -69,6 +77,8 @@ Examples: cmd.Flags().IntVar(&opts.inspectBytes, "bytes", opts.inspectBytes, "Number of bytes to show in inspection") cmd.Flags().StringVar(&opts.inspectFormat, "inspect-format", opts.inspectFormat, "Inspection format: hex, float32, int32, uint32, float16") cmd.Flags().BoolVar(&opts.resources, "resources", opts.resources, "Show device-resource buffer inventory") + cmd.Flags().IntVar(&opts.limit, "limit", opts.limit, "Maximum buffers in human table output") + cmd.Flags().BoolVar(&opts.all, "all", opts.all, "Show all buffers in human table output") return cmd } @@ -94,13 +104,16 @@ func runBuffers(cmd *cobra.Command, args []string, cmdOpts *buffersCommandOption if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } // If --inspect is specified, handle buffer inspection if cmdOpts.inspect != "" { return inspectBuffer(tracePath, cmdOpts.inspect, opts.inspectBytes, opts.inspectFormat) } if cmdOpts.resources { - return formatBufferResourceInventory(tracePath, opts.format, trace) + return formatBufferResourceInventory(cmd.OutOrStdout(), tracePath, opts.format, trace) } // Extract buffer information @@ -126,11 +139,15 @@ func runBuffers(cmd *cobra.Command, args []string, cmdOpts *buffersCommandOption // Format and display switch opts.format { case "json": - return formatBuffersJSON(buffers) + return formatBuffersJSON(cmd.OutOrStdout(), buffers) case "csv": - return formatBuffersCSV(buffers) + return formatBuffersCSV(cmd.OutOrStdout(), buffers) default: - return formatBuffersTable(buffers, trace) + limit, err := resolveHumanLimit(cmdOpts.limit, cmdOpts.all) + if err != nil { + return err + } + return formatBuffersTable(cmd.OutOrStdout(), buffers, trace, limit) } } @@ -378,43 +395,11 @@ func extractBufferBindings(trace *gputrace.Trace, bufferMap map[string]*BufferIn return fmt.Errorf("read capture: %w", err) } - // Parse CtUulul records from capture - // Marker: CtUulul = 43 74 55 3c 62 3e 75 6c 75 6c - marker := []byte{0x43, 0x74, 0x55, 0x3c, 0x62, 0x3e, 0x75, 0x6c, 0x75, 0x6c} - offset := 0 - matchCount := 0 - for { - pos := bytes.Index(captureData[offset:], marker) - if pos == -1 { - break - } - matchCount++ - absolutePos := offset + pos - - // Structure based on hexdump analysis: - // +0x00: "CtUulul" (10 bytes) - // +0x0a: padding (2 bytes of 0x00) - // +0x0c: first address (8 bytes, little-endian) - // +0x14: buffer address (8 bytes, little-endian) - // +0x1c: buffer name "MTLBuffer-XX-Y" or "MTLHeap-X-Y" - - // Read buffer address at +0x14 (corrected offset) - if absolutePos+0x24 <= len(captureData) { - bufAddr := binary.LittleEndian.Uint64(captureData[absolutePos+0x14 : absolutePos+0x1c]) - - // Read buffer name at +0x1c (corrected offset) - nameStart := absolutePos + 0x1c - if bytes.HasPrefix(captureData[nameStart:], []byte("MTLBuffer-")) || - bytes.HasPrefix(captureData[nameStart:], []byte("MTLHeap-")) { - nameEnd := bytes.IndexByte(captureData[nameStart:], 0) - if nameEnd > 0 && nameEnd < 100 { - name := string(captureData[nameStart : nameStart+nameEnd]) - addrToName[bufAddr] = name - } - } + // Only Metal's own resource names resolve to a file in the bundle. + for bufAddr, name := range tracepkg.ScanBufferNames(captureData) { + if strings.HasPrefix(name, "MTLBuffer-") || strings.HasPrefix(name, "MTLHeap-") { + addrToName[bufAddr] = name } - - offset += pos + 10 } // Build map from buffer name to BufferInfo @@ -471,7 +456,7 @@ func extractBufferBindings(trace *gputrace.Trace, bufferMap map[string]*BufferIn } // Count encoders in this command buffer (number of dispatch calls) - dispatches, _ := trace.ParseDispatchInRegion(cbData, cb.Offset) + dispatches := trace.ParseDispatchInRegion(cbData, cb.Offset) numEncoders := len(dispatches) if numEncoders == 0 { numEncoders = 1 @@ -583,8 +568,13 @@ type CommandBufferBinding struct { func sortBuffers(buffers []BufferInfo, sortBy string) { switch sortBy { case "size": + // Buffers of equal size break by ID, so the listing does not + // reshuffle between runs on the same trace. sort.Slice(buffers, func(i, j int) bool { - return buffers[i].Size > buffers[j].Size // Descending + if buffers[i].Size != buffers[j].Size { + return buffers[i].Size > buffers[j].Size // Descending + } + return buffers[i].ID < buffers[j].ID }) case "id": sort.Slice(buffers, func(i, j int) bool { @@ -598,7 +588,7 @@ func sortBuffers(buffers []BufferInfo, sortBy string) { } // formatBuffersTable formats buffers as a human-readable table. -func formatBuffersTable(buffers []BufferInfo, trace *gputrace.Trace) error { +func formatBuffersTable(w io.Writer, buffers []BufferInfo, trace *gputrace.Trace, limit int) error { // Calculate totals var totalSize uint64 totalAliases := 0 @@ -608,21 +598,27 @@ func formatBuffersTable(buffers []BufferInfo, trace *gputrace.Trace) error { } // Print summary line - fmt.Printf("%d %s, %s", len(buffers), Pluralize(len(buffers), "buffer", "buffers"), FormatBytes(totalSize)) + fmt.Fprintf(w, "%d %s, %s", len(buffers), Pluralize(len(buffers), "buffer", "buffers"), FormatBytes(totalSize)) if totalAliases > 0 { - fmt.Printf(", %d %s", totalAliases, Pluralize(totalAliases, "alias", "aliases")) + fmt.Fprintf(w, ", %d %s", totalAliases, Pluralize(totalAliases, "alias", "aliases")) } - fmt.Println() - fmt.Println() + fmt.Fprintln(w) + // "gputrace residency" also prints a buffer count, from creation records + // rather than from distinct resources, and is roughly 10x larger on the + // same capture with reconciling bytes. Both are right about different + // populations; the word "buffers" is what collides. + fmt.Fprintln(w, "(distinct resources; \"gputrace residency\" counts creation records instead)") + fmt.Fprintln(w) // Print table header - fmt.Println(Colorize("Buffers", ColorBold)) - fmt.Println(TableSeparator(80)) - fmt.Printf("%-8s %-25s %12s %s\n", "ID", "Filename", "Size", "Aliases") - fmt.Println(TableSeparator(80)) + fmt.Fprintln(w, Colorize("Buffers", ColorBold)) + fmt.Fprintln(w, TableSeparator(80)) + fmt.Fprintf(w, "%-8s %-25s %12s %s\n", "ID", "Filename", "Size", "Aliases") + fmt.Fprintln(w, TableSeparator(80)) // Print each buffer - for _, buf := range buffers { + shown := limitedCount(len(buffers), limit) + for _, buf := range buffers[:shown] { aliasInfo := "" if len(buf.Aliases) > 0 { if len(buf.Aliases) == 1 { @@ -632,7 +628,7 @@ func formatBuffersTable(buffers []BufferInfo, trace *gputrace.Trace) error { } } - fmt.Printf("%-8s %-25s %12s %s\n", + fmt.Fprintf(w, "%-8s %-25s %12s %s\n", buf.ID, buf.Filename, FormatBytes(buf.Size), @@ -642,27 +638,30 @@ func formatBuffersTable(buffers []BufferInfo, trace *gputrace.Trace) error { // Show all aliases if more than 1 if len(buf.Aliases) > 1 { for _, alias := range buf.Aliases { - fmt.Printf("%-8s → %s\n", "", alias) + fmt.Fprintf(w, "%-8s → %s\n", "", alias) } } // Show buffer bindings if present if len(buf.Bindings) > 0 { - fmt.Printf("%-8s Used by:\n", "") + fmt.Fprintf(w, "%-8s Used by:\n", "") for _, binding := range buf.Bindings { - fmt.Printf("%-8s - %s (index %d", "", binding.EncoderLabel, binding.Index) + fmt.Fprintf(w, "%-8s - %s (index %d", "", binding.EncoderLabel, binding.Index) if binding.Offset > 0 { - fmt.Printf(", offset %d", binding.Offset) + fmt.Fprintf(w, ", offset %d", binding.Offset) } - fmt.Printf(")\n") + fmt.Fprintln(w, ")") } } } + if shown < len(buffers) { + fmt.Fprintf(w, "... %d more buffers omitted (use --all)\n", len(buffers)-shown) + } return nil } -func formatBufferResourceInventory(tracePath, format string, trace *gputrace.Trace) error { +func formatBufferResourceInventory(w io.Writer, tracePath, format string, trace *gputrace.Trace) error { inventory, err := extractBufferResourceInventory(tracePath, trace) if err != nil { return err @@ -670,13 +669,13 @@ func formatBufferResourceInventory(tracePath, format string, trace *gputrace.Tra switch format { case "json": - enc := json.NewEncoder(os.Stdout) + enc := json.NewEncoder(w) enc.SetIndent("", " ") return enc.Encode(inventory) case "csv": - fmt.Println("Filename,Kind,Records,FinalNameRecords,SizeMatched,SizeBad,NoFinalFile") + fmt.Fprintln(w, "Filename,Kind,Records,FinalNameRecords,SizeMatched,SizeBad,NoFinalFile") for _, file := range inventory.Files { - fmt.Printf("%s,%s,%d,%d,%d,%d,%d\n", + fmt.Fprintf(w, "%s,%s,%d,%d,%d,%d,%d\n", file.Filename, file.Kind, file.Records, @@ -688,18 +687,18 @@ func formatBufferResourceInventory(tracePath, format string, trace *gputrace.Tra } return nil default: - fmt.Printf("%d final %s, %s\n\n", + fmt.Fprintf(w, "%d final %s, %s\n\n", inventory.FinalBuffers, Pluralize(inventory.FinalBuffers, "buffer", "buffers"), FormatBytes(inventory.FinalBytes), ) - fmt.Println(Colorize("Device Resource Buffers", ColorBold)) - fmt.Println(TableSeparator(110)) - fmt.Printf("%-36s %-10s %8s %12s %12s %8s %12s\n", + fmt.Fprintln(w, Colorize("Device Resource Buffers", ColorBold)) + fmt.Fprintln(w, TableSeparator(110)) + fmt.Fprintf(w, "%-36s %-10s %8s %12s %12s %8s %12s\n", "File", "Kind", "Records", "FinalNames", "SizeMatched", "SizeBad", "NoFinalFile") - fmt.Println(TableSeparator(110)) + fmt.Fprintln(w, TableSeparator(110)) for _, file := range inventory.Files { - fmt.Printf("%-36s %-10s %8d %12d %12d %8d %12d\n", + fmt.Fprintf(w, "%-36s %-10s %8d %12d %12d %8d %12d\n", file.Filename, file.Kind, file.Records, @@ -709,12 +708,12 @@ func formatBufferResourceInventory(tracePath, format string, trace *gputrace.Tra file.NoFinalFile, ) } - printBufferResourceSizeBins(inventory.Files) + printBufferResourceSizeBins(w, inventory.Files) return nil } } -func printBufferResourceSizeBins(files []BufferResourceFile) { +func printBufferResourceSizeBins(w io.Writer, files []BufferResourceFile) { type row struct { filename string kind string @@ -746,15 +745,15 @@ func printBufferResourceSizeBins(files []BufferResourceFile) { if len(rows) < limit { limit = len(rows) } - fmt.Println() - fmt.Println(Colorize("Top Matched Resource Size Bins", ColorBold)) - fmt.Println(TableSeparator(100)) - fmt.Printf("%-36s %-10s %12s %8s %7s %8s %8s %12s %8s %10s %8s\n", + fmt.Fprintln(w) + fmt.Fprintln(w, Colorize("Top Matched Resource Size Bins", ColorBold)) + fmt.Fprintln(w, TableSeparator(100)) + fmt.Fprintf(w, "%-36s %-10s %12s %8s %7s %8s %8s %12s %8s %10s %8s\n", "File", "Kind", "Size", "Records", "Names", "First", "Last", "Bytes", "CmdNames", "CmdRecords", "CmdEnc") - fmt.Println(TableSeparator(100)) + fmt.Fprintln(w, TableSeparator(100)) for i := 0; i < limit; i++ { row := rows[i] - fmt.Printf("%-36s %-10s %12d %8d %7d %8d %8d %12s %8d %10d %8d\n", + fmt.Fprintf(w, "%-36s %-10s %12d %8d %7d %8d %8d %12s %8d %10d %8d\n", row.filename, row.kind, row.bin.Size, @@ -978,7 +977,7 @@ func resourceRecordMarkerAt(data []byte, nameStart int) (string, int) { window := data[windowStart:nameStart] markers := []string{ "CUulul", - "CtUulul", + tracepkg.CtUMarker, "Cuw", } bestMarker := "" @@ -1075,7 +1074,7 @@ func extractBufferCommandUses(trace *gputrace.Trace, finalSizes map[string]uint6 if err != nil { continue } - dispatches, _ := trace.ParseDispatchInRegion(cbData, cb.Offset) + dispatches := trace.ParseDispatchInRegion(cbData, cb.Offset) numEncoders := len(dispatches) if numEncoders == 0 { numEncoders = 1 @@ -1117,27 +1116,14 @@ func extractBufferCommandUses(trace *gputrace.Trace, finalSizes map[string]uint6 return uses, nil } +// extractBufferAddressNames maps buffer addresses to the buffer files they +// name. Only MTLBuffer- names resolve to a file in the bundle. func extractBufferAddressNames(captureData []byte) map[uint64]string { addrToName := make(map[uint64]string) - marker := []byte{0x43, 0x74, 0x55, 0x3c, 0x62, 0x3e, 0x75, 0x6c, 0x75, 0x6c} - offset := 0 - for { - pos := bytes.Index(captureData[offset:], marker) - if pos == -1 { - break - } - absolutePos := offset + pos - if absolutePos+0x24 <= len(captureData) { - bufAddr := binary.LittleEndian.Uint64(captureData[absolutePos+0x14 : absolutePos+0x1c]) - nameStart := absolutePos + 0x1c - if bytes.HasPrefix(captureData[nameStart:], []byte("MTLBuffer-")) { - nameEnd := bytes.IndexByte(captureData[nameStart:], 0) - if nameEnd > 0 && nameEnd < 100 { - addrToName[bufAddr] = string(captureData[nameStart : nameStart+nameEnd]) - } - } + for addr, name := range tracepkg.ScanBufferNames(captureData) { + if strings.HasPrefix(name, "MTLBuffer-") { + addrToName[addr] = name } - offset += pos + len(marker) } return addrToName } @@ -1171,7 +1157,7 @@ func resourceRecordSizeOffset(data []byte, nameEnd int, want uint64) int { } // formatBuffersJSON formats buffers as JSON. -func formatBuffersJSON(buffers []BufferInfo) error { +func formatBuffersJSON(w io.Writer, buffers []BufferInfo) error { output := make([]bufferJSONInfo, 0, len(buffers)) for _, buf := range buffers { output = append(output, bufferJSONInfo{ @@ -1181,7 +1167,7 @@ func formatBuffersJSON(buffers []BufferInfo) error { Aliases: len(buf.Aliases), }) } - enc := json.NewEncoder(os.Stdout) + enc := json.NewEncoder(w) enc.SetIndent("", " ") return enc.Encode(output) } @@ -1194,8 +1180,8 @@ type bufferJSONInfo struct { } // formatBuffersCSV formats buffers as CSV. -func formatBuffersCSV(buffers []BufferInfo) error { - w := csv.NewWriter(os.Stdout) +func formatBuffersCSV(out io.Writer, buffers []BufferInfo) error { + w := csv.NewWriter(out) if err := w.Write([]string{"ID", "Filename", "Size", "Aliases"}); err != nil { return fmt.Errorf("write csv header: %w", err) } diff --git a/cmd/gputrace/cmd/buffers_sort_test.go b/cmd/gputrace/cmd/buffers_sort_test.go new file mode 100644 index 00000000..55121777 --- /dev/null +++ b/cmd/gputrace/cmd/buffers_sort_test.go @@ -0,0 +1,44 @@ +package cmd + +import ( + "reflect" + "testing" +) + +func TestSortBuffers(t *testing.T) { + // Equal sizes on purpose: the order of those rows is what used to depend + // on how the buffers happened to be collected. + input := []BufferInfo{ + {ID: "3", Filename: "c", Size: 100}, + {ID: "1", Filename: "a", Size: 100}, + {ID: "2", Filename: "b", Size: 900}, + } + + tests := []struct { + sortBy string + want []string + }{ + {sortBy: "size", want: []string{"2", "1", "3"}}, + {sortBy: "id", want: []string{"1", "2", "3"}}, + {sortBy: "name", want: []string{"1", "2", "3"}}, + } + + for _, tt := range tests { + t.Run(tt.sortBy, func(t *testing.T) { + // Feed a different starting permutation each time to prove the + // result does not depend on input order. + for _, start := range [][]BufferInfo{input, {input[2], input[0], input[1]}} { + buffers := append([]BufferInfo(nil), start...) + sortBuffers(buffers, tt.sortBy) + + got := make([]string, len(buffers)) + for i, b := range buffers { + got[i] = b.ID + } + if !reflect.DeepEqual(got, tt.want) { + t.Errorf("sortBuffers(%q) = %v, want %v", tt.sortBy, got, tt.want) + } + } + }) + } +} diff --git a/cmd/gputrace/cmd/buffers_test.go b/cmd/gputrace/cmd/buffers_test.go index 0b62feac..8da8606e 100644 --- a/cmd/gputrace/cmd/buffers_test.go +++ b/cmd/gputrace/cmd/buffers_test.go @@ -1,6 +1,7 @@ package cmd import ( + "bytes" "encoding/csv" "encoding/json" "path/filepath" @@ -190,12 +191,11 @@ func TestFormatBuffersJSONEscapesFilenames(t *testing.T) { }, } - out, err := captureStdout(t, func() error { - return formatBuffersJSON(buffers) - }) - if err != nil { + var buf bytes.Buffer + if err := formatBuffersJSON(&buf, buffers); err != nil { t.Fatalf("formatBuffersJSON: %v", err) } + out := buf.String() var got []bufferJSONInfo if err := json.Unmarshal([]byte(out), &got); err != nil { @@ -219,13 +219,12 @@ func TestFormatBuffersCSVEscapesFilenames(t *testing.T) { }, } - out, err := captureStdout(t, func() error { - return formatBuffersCSV(buffers) - }) - if err != nil { + var buf bytes.Buffer + if err := formatBuffersCSV(&buf, buffers); err != nil { t.Fatalf("formatBuffersCSV: %v", err) } + out := buf.String() records, err := csv.NewReader(strings.NewReader(out)).ReadAll() if err != nil { t.Fatalf("CSV output did not decode: %v\n%s", err, out) diff --git a/cmd/gputrace/cmd/capture.go b/cmd/gputrace/cmd/capture.go new file mode 100644 index 00000000..11b1a755 --- /dev/null +++ b/cmd/gputrace/cmd/capture.go @@ -0,0 +1,115 @@ +package cmd + +import ( + "bytes" + "errors" + "fmt" + "os" + "os/exec" + + "github.com/spf13/cobra" + + "github.com/tmc/gputrace/internal/capture" +) + +var captureCmd = newCaptureCommand(&captureOptions{}) + +type captureOptions struct { + output string + timingOutput string + runID string + dir string + check bool +} + +func newCaptureCommand(opts *captureOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "capture [flags] -- [args...]", + Short: "Run a Metal workload under the GPU capture interposer", + Long: `Run a command with Apple's GPUToolsCapture interposer loaded and write a +.gputrace bundle. The target does not need to be recompiled or to call +MTLCaptureManager itself. + +The interposer is loaded through DYLD_INSERT_LIBRARIES, which dyld ignores for +hardened-runtime binaries with library validation and for Apple platform +binaries. Those targets run normally and produce no trace. Use --check to test a +target without running it. + +Interposable in practice: adhoc-signed and developer-signed binaries, unsigned +builds, and Homebrew interpreters such as python3. Not interposable: App Store +and notarized applications with the hardened runtime, and anything under +/System. + +Examples: + gputrace capture -o run.gputrace -- python3 bench.py + gputrace capture --check python3`, + Args: cobra.MinimumNArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + if opts.check { + return runCaptureCheck(cmd, args[0]) + } + if opts.output == "" { + return errors.New("capture: -o is required") + } + var stderr bytes.Buffer + out, err := capture.Run(cmd.Context(), capture.Options{ + Output: opts.output, + TimingOutput: opts.timingOutput, + RunID: opts.runID, + Dir: opts.dir, + Stdin: cmd.InOrStdin(), + Stdout: cmd.OutOrStdout(), + Stderr: &stderr, + }, args...) + if err != nil { + if stderr.Len() > 0 { + fmt.Fprint(os.Stderr, stderr.String()) + } + return err + } + fmt.Fprintf(cmd.OutOrStdout(), "wrote %s\n", out) + return nil + }, + } + f := cmd.Flags() + f.StringVarP(&opts.output, "output", "o", "", "path of the .gputrace bundle to write") + f.StringVar(&opts.timingOutput, "timing-sidecar", "", "write live command-buffer timing and clock samples as JSON lines (requires --run-id)") + f.StringVar(&opts.runID, "run-id", "", "run identity shared by timing sidecar and host signposts") + f.StringVar(&opts.dir, "dir", "", "working directory for the target") + f.BoolVar(&opts.check, "check", false, "report whether the target accepts the interposer, then exit") + // --api binds a Linux-only option and is registered beside the other + // Linux-only capture flags; this file builds on every platform. + return cmd +} + +func runCaptureCheck(cmd *cobra.Command, target string) error { + // Resolve through PATH exactly as capture.Run does. Handing the bare argv[0] + // to codesign checks a file of that name in the working directory instead, + // so the verdict would describe a different binary than the one a capture + // would launch. + target, err := exec.LookPath(target) + if err != nil { + return fmt.Errorf("capture: %w", err) + } + if err := capture.Eligible(target); err != nil { + if errors.Is(err, capture.ErrNotInterposable) { + // An ineligible target is a verdict, not a malfunction: report it on + // stdout and exit non-zero, without restating it on stderr. + fmt.Fprintf(cmd.OutOrStdout(), "not capturable: %s: %v\n", target, err) + return reportedCaptureError{err} + } + return err + } + fmt.Fprintf(cmd.OutOrStdout(), "capturable: %s accepts the interposer\n", target) + return nil +} + +// reportedCaptureError carries a verdict already written to stdout, so the +// entry point exits non-zero without printing it a second time. +type reportedCaptureError struct{ error } + +func (reportedCaptureError) alreadyReported() {} + +func init() { + rootCmd.AddCommand(captureCmd) +} diff --git a/cmd/gputrace/cmd/capture_input.go b/cmd/gputrace/cmd/capture_input.go new file mode 100644 index 00000000..8295797a --- /dev/null +++ b/cmd/gputrace/cmd/capture_input.go @@ -0,0 +1,30 @@ +package cmd + +import ( + "io" + "os" + + "github.com/tmc/gputrace/internal/gpuevent" +) + +// readCapture decodes events from r and, when samplesPath is non-empty, +// joins the sample file into the same capture. +func readCapture(r io.Reader, samplesPath string) (gpuevent.Capture, error) { + cap, err := gpuevent.DecodeJSONL(r) + if err != nil { + return cap, err + } + if samplesPath != "" { + sf, err := os.Open(samplesPath) + if err == nil { + defer sf.Close() + sc, err := gpuevent.DecodeJSONL(sf) + if err == nil { + cap.Samples = append(cap.Samples, sc.Samples...) + } + } else if !os.IsNotExist(err) { + return cap, err + } + } + return cap, nil +} diff --git a/cmd/gputrace/cmd/capture_linux.go b/cmd/gputrace/cmd/capture_linux.go new file mode 100644 index 00000000..c6c943e5 --- /dev/null +++ b/cmd/gputrace/cmd/capture_linux.go @@ -0,0 +1,246 @@ +//go:build linux + +package cmd + +import ( + "fmt" + "io" + "os" + "os/exec" + "path/filepath" + "strings" + "syscall" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/buildinfo" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/gpudoctor" + "github.com/tmc/gputrace/internal/gpuevent" + "github.com/tmc/gputrace/internal/nvidia" +) + +// captureLinuxOptions mirrors the darwin captureOptions subset meaningful on +// Linux. It is registered as additional behavior of the same `capture` +// command so the CLI surface stays identical across platforms. +type captureLinuxOptions struct { + samples bool + api bool + nvtx bool + sampleInterval string +} + +var captureLinuxOpts = captureLinuxOptions{sampleInterval: "25ms"} + +// runCaptureLinux executes the workload with the CUPTI shim preloaded and +// writes a .gpucapture bundle. +func runCaptureLinux(cmd *cobra.Command, opts *captureOptions, args []string) error { + out := opts.output + if out == "" { + return fmt.Errorf("capture: -o is required (e.g. -o run.gpucapture)") + } + if !strings.HasSuffix(out, ".gpucapture") { + out += ".gpucapture" + } + if abs, err := filepath.Abs(out); err == nil { + out = abs // the target may run with a different working directory + } + if fi, err := os.Stat(out); err == nil && fi.IsDir() && len(args) > 0 { + return fmt.Errorf("capture: output bundle %s already exists", out) + } + + eventsPath := filepath.Join(out, cupticapture.EventsFileName) + meta := cupticapture.Meta{ + Command: args, + Dir: opts.dir, + GPUTRACEVersion: buildinfo.EffectiveVersion(), + } + if devices, err := nvidia.Devices(); err == nil && len(devices) > 0 { + d := devices[0] + meta.GPUName = d.Name + meta.GPUUUID = d.UUID + meta.DriverVersion = d.DriverVer + } + if err := cupticapture.CreateBundle(out, meta); err != nil { + return fmt.Errorf("capture: create bundle: %w", err) + } + + preloadEnv, err := cupticapture.PreloadEnv(cupticapture.Options{ + OutputPath: eventsPath, + APIRecords: captureLinuxOpts.api, + NVTX: captureLinuxOpts.nvtx, + }) + if err != nil { + return err + } + + // App-events sidecar: any process in the target tree may append + // span/instant JSONL records here; they are merged into the bundle's + // events stream when the target exits. See docs/ENVIRONMENT.md. + sidecarPath := filepath.Join(out, "app_events.jsonl") + preloadEnv = append(preloadEnv, fmt.Sprintf("GPUTRACE_APP_EVENTS=%s", sidecarPath)) + + // Optional NVML sampler runs alongside the target and writes into the + // same bundle; it is stopped when the target exits. + var sampler *exec.Cmd + if captureLinuxOpts.samples { + samplerPath, samplerErr := exec.LookPath("nvml_sampler") + if samplerErr != nil { + fmt.Fprintln(os.Stderr, "capture: --samples requested but nvml_sampler not on PATH; continuing without device samples") + } else { + sampler = exec.Command(samplerPath, "-out", out, "-interval", captureLinuxOpts.sampleInterval) + sampler.Stdout = nil + sampler.Stderr = os.Stderr + if err := sampler.Start(); err != nil { + fmt.Fprintf(os.Stderr, "capture: start sampler: %v; continuing without samples\n", err) + sampler = nil + } + } + } + + target := exec.Command(args[0], args[1:]...) + if opts.dir != "" { + target.Dir = opts.dir + } + target.Env = append(os.Environ(), preloadEnv...) + target.Stdin = cmd.InOrStdin() + target.Stdout = cmd.OutOrStdout() + // Streamed, not buffered. Buffering it hid the target's own stderr + // on every successful run, which is where GPUTRACE_CAPTURE_DEBUG + // writes: the one channel for diagnosing the shim was invisible + // through the command that installs it. + target.Stderr = cmd.ErrOrStderr() + + err = target.Run() + // Stop the sampler before reporting so its file is complete. + if sampler != nil { + if sampler.Process != nil { + _ = sampler.Process.Signal(syscall.SIGTERM) + } + _ = sampler.Wait() + } + // Merge the app-events sidecar into the bundle's events stream so span + // records ride the same decode path as shim-emitted records. The + // sidecar is optional: an absent file means the target declared nothing. + if sidecarData, err := os.ReadFile(sidecarPath); err == nil && len(sidecarData) > 0 { + // Append into an events-glob-matching shard so OpenEvents picks it up. + mergedPath := filepath.Join(out, fmt.Sprintf("events.app.%d.jsonl", os.Getpid())) + if err := os.WriteFile(mergedPath, sidecarData, 0o644); err != nil { + fmt.Fprintf(os.Stderr, "capture: merge app-events: %v\n", err) + } + } + // A failing workload still produced a valid (possibly empty) bundle; + // report both facts rather than discarding evidence. + fmt.Fprintf(cmd.OutOrStdout(), "wrote %s\n", out) + reportCaptureContents(cmd.OutOrStdout(), out, captureLinuxOpts) + if err != nil { + fmt.Fprintf(os.Stderr, "capture: target exited with error; bundle retained for inspection\n") + return reportedCaptureError{err} + } + return nil +} + +// reportCaptureContents states what the bundle actually holds, at the end +// of the run that produced it. +// +// Two failures this closes are the same failure: an arm that never ran and +// an arm that ran and found nothing return identical output. A --nvtx that +// recorded no markers is indistinguishable from a workload that emits +// none, and a capture that lost activity records is indistinguishable from +// a workload that launched less. Both are counts the tool already has and +// was not printing. +func reportCaptureContents(w io.Writer, bundle string, opts captureLinuxOptions) { + r, closers, err := cupticapture.OpenEvents(bundle) + if err != nil { + return // the bundle stands on its own; this line is a courtesy + } + cap, _ := gpuevent.DecodeJSONL(r) + closers() + + var kernels, transfers, markers int + for _, e := range cap.Events { + switch e.Kind { + case gpuevent.KindKernel: + kernels++ + case gpuevent.KindMemcpy, gpuevent.KindMemset: + transfers++ + } + } + for _, sp := range cap.Spans { + if sp.Source == gpuevent.SourceNVTX { + markers++ + } + } + fmt.Fprintf(w, " %d kernels, %d transfers, %d spans\n", kernels, transfers, len(cap.Spans)) + if opts.nvtx && markers == 0 { + fmt.Fprintf(w, " --nvtx: 0 NVTX ranges recorded — either the target emits none, or nothing routed them into CUPTI\n") + } + health := gpuevent.MeasureCompleteness(cap) + if !health.Complete() { + fmt.Fprintf(w, " %s\n", health.Summary()) + for _, line := range strings.Split(health.Remedy(), "\n") { + fmt.Fprintf(w, " %s\n", line) + } + } +} + +func init() { + // Let `gputrace doctor` prove the shim builds without capturing. + gpudoctor.SetShimBuilder(cupticapture.EnsureShim) + + // Extend the existing capture command with Linux-specific flags and + // rerouting. The darwin implementation compiles an Objective-C interposer; + // here we compile a C CUPTI shim and LD_PRELOAD it instead. + captureCmd.Long = `Run a command under the gputrace GPU capture tracer. + +On Linux/NVIDIA this preloads a CUPTI activity shim into the target, +recording every kernel launch, memcpy, and memset as newline-delimited +JSON inside a .gpucapture bundle directory: + + run.gpucapture/ + events.jsonl activity records (kind, timing, geometry) + nvml_samples.jsonl concurrent device samples (--samples) + meta.json provenance (command, time, versions) + +The bundle feeds directly into analysis and export: + + gputrace analyze run.gpucapture + gputrace analyze run.gpucapture --suggest + gputrace cupti run.gpucapture --per-kernel-tracks -o trace.pftrace + +Targets must link CUDA dynamically (-cudart=shared for nvcc builds). +Statically-linked CUDA runtimes bypass the interposition points. + +Examples: + gputrace capture -o run.gpucapture -- python3 bench.py + gputrace capture -o run.gpucapture --samples -- ./matmul` + captureCmd.Short = "Run a workload under the GPU capture tracer" + captureCmd.Flags().BoolVar(&captureLinuxOpts.api, "api", captureLinuxOpts.api, "record host-side CUDA runtime/driver API calls (multiplies record volume)") + captureCmd.Flags().BoolVar(&captureLinuxOpts.nvtx, "nvtx", captureLinuxOpts.nvtx, "record NVTX ranges emitted by the target or the libraries it links") + captureCmd.Flags().BoolVar(&captureLinuxOpts.samples, "samples", captureLinuxOpts.samples, "Sample NVML device counters during the run") + captureCmd.Flags().StringVar(&captureLinuxOpts.sampleInterval, "sample-interval", captureLinuxOpts.sampleInterval, "NVML sampling interval") + + origRunE := captureCmd.RunE + captureCmd.RunE = func(cmd *cobra.Command, args []string) error { + opts := captureOptions{ + output: cmd.Flag("output").Value.String(), + dir: cmd.Flag("dir").Value.String(), + check: false, + } + // --check is a darwin concept (codesign probe); on Linux report + // whether injection prerequisites hold instead. + if len(args) > 0 && args[0] == "--check" || cmd.Flag("check") != nil && cmd.Flag("check").Value.String() == "true" { + shim, err := cupticapture.PreloadEnv(cupticapture.Options{OutputPath: "/dev/null"}) + if err != nil { + return err + } + fmt.Fprintf(cmd.OutOrStdout(), "capturable: shim ready (%s)\n", shim[1]) + return nil + } + if origRunE != nil && opts.output != "" && strings.HasSuffix(opts.output, ".gputrace") { + // Explicit .gputrace output on Linux means the user wants the + // darwin flow; let it fail with its own message. + return origRunE(cmd, args) + } + return runCaptureLinux(cmd, &opts, args) + } +} diff --git a/cmd/gputrace/cmd/capture_test.go b/cmd/gputrace/cmd/capture_test.go new file mode 100644 index 00000000..2b1eb6d2 --- /dev/null +++ b/cmd/gputrace/cmd/capture_test.go @@ -0,0 +1,49 @@ +package cmd + +import ( + "bytes" + "os" + "os/exec" + "path/filepath" + "runtime" + "strings" + "testing" +) + +// TestCaptureCheckResolvesThroughPath pins the resolution rule: --check must +// name the binary a capture would actually launch. Passing the bare argv[0] to +// codesign silently checks a same-named file in the working directory instead, +// so the verdict tracked the caller's cwd rather than PATH. +func TestCaptureCheckResolvesThroughPath(t *testing.T) { + if runtime.GOOS != "darwin" { + t.Skip("codesign-based capture check is darwin-only") + } + want, err := exec.LookPath("python3") + if err != nil { + t.Skip("no python3 on PATH") + } + + // A decoy of the same name in the working directory must not be consulted. + dir := t.TempDir() + if err := os.WriteFile(filepath.Join(dir, "python3"), []byte("#!/bin/sh\n"), 0o755); err != nil { + t.Fatal(err) + } + t.Chdir(dir) + + var out bytes.Buffer + cmd := newCaptureCommand(&captureOptions{}) + cmd.SetOut(&out) + cmd.SetErr(&out) + cmd.SetArgs([]string{"--check", "python3"}) + // The verdict itself depends on the host's python3 and is not asserted; + // only that the command names the PATH-resolved binary and does not fail + // with a codesign lookup error against the decoy. + err = cmd.Execute() + got := out.String() + if err != nil && !strings.Contains(got, "not capturable") { + t.Fatalf("--check python3 = %v, output %q; want a verdict", err, got) + } + if !strings.Contains(got, want) { + t.Errorf("--check python3 output %q does not name the PATH-resolved %q", got, want) + } +} diff --git a/cmd/gputrace/cmd/clear_buffers.go b/cmd/gputrace/cmd/clear_buffers.go index bc31f2a4..2e7b94f6 100644 --- a/cmd/gputrace/cmd/clear_buffers.go +++ b/cmd/gputrace/cmd/clear_buffers.go @@ -21,17 +21,19 @@ var clearBuffersCmd = newClearBuffersCommand(&clearBuffersOptions{}) func newClearBuffersCommand(opts *clearBuffersOptions) *cobra.Command { cmd := &cobra.Command{ Use: "clear-buffers ", - Short: "Zero out MTLBuffer files to reduce trace size", + Short: "Destructively replace captured buffer contents with zeros", Long: `Zero out all MTLBuffer-* files in a GPU trace directory. -This is useful for reducing trace size when buffer contents are not needed, -such as when sharing traces or storing them for later analysis of structure -without the actual data. +This removes captured buffer contents when they are not needed, such as before +sharing a trace for structural analysis. It preserves each file's logical size. +Zero-filled files usually compress well, but this command does not itself +shrink the bundle or reclaim filesystem blocks. The original contents cannot +be recovered from the modified bundle. The command will: - Find all MTLBuffer-* files (skipping symlinks) - Zero out their contents while preserving file size - - Report total space that could be saved + - Report the total number of bytes overwritten Examples: gputrace clear-buffers trace.gputrace # Zero all buffers (prompts for confirmation) @@ -106,7 +108,7 @@ func runClearBuffers(cmd *cobra.Command, args []string, opts *clearBuffersOption } // Show summary and prompt for confirmation - fmt.Fprintf(w, "Found %d buffer files (%s total)\n", fileCount, fmtutil.FormatBytes(totalSize, 2)) + fmt.Fprintf(w, "Found %d buffer files (%s to overwrite; logical size will be preserved)\n", fileCount, fmtutil.FormatBytes(totalSize, 2)) if skippedSymlinks > 0 { fmt.Fprintf(w, "Will skip %d symlinks\n", skippedSymlinks) } @@ -118,7 +120,7 @@ func runClearBuffers(cmd *cobra.Command, args []string, opts *clearBuffersOption // Prompt for confirmation unless -y flag is set if !opts.yes { - fmt.Fprint(w, "\nZero out all buffer files? [y/N]: ") + fmt.Fprint(w, "\nThis permanently destroys the captured buffer contents. Continue? [y/N]: ") reader := bufio.NewReader(os.Stdin) response, err := reader.ReadString('\n') if err != nil { diff --git a/cmd/gputrace/cmd/collect_xcode_profile.go b/cmd/gputrace/cmd/collect_xcode_profile.go index 7c862acb..0bb08cd0 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile.go +++ b/cmd/gputrace/cmd/collect_xcode_profile.go @@ -22,16 +22,103 @@ import ( ) type xcodeProfileActionOutput struct { - Success bool `json:"success"` - Action string `json:"action"` - Target string `json:"target,omitempty"` - Method string `json:"method,omitempty"` - Input string `json:"input,omitempty"` - Output string `json:"output,omitempty"` - Source string `json:"source,omitempty"` - RequestedOutput string `json:"requested_output,omitempty"` - Copied bool `json:"copied,omitempty"` - Warning string `json:"warning,omitempty"` + Success bool `json:"success"` + Action string `json:"action"` + Target string `json:"target,omitempty"` + Method string `json:"method,omitempty"` + Input string `json:"input,omitempty"` + Output string `json:"output,omitempty"` + Source string `json:"source,omitempty"` + SourceUUID string `json:"source_uuid,omitempty"` + XcodePID int `json:"xcode_pid,omitempty"` + XcodeApp string `json:"xcode_app,omitempty"` + RequestedOutput string `json:"requested_output,omitempty"` + Copied bool `json:"copied,omitempty"` + Reused bool `json:"reused,omitempty"` + RequestedTrace string `json:"requested_trace,omitempty"` + SelectedTitle string `json:"selected_title,omitempty"` + SelectedDocument string `json:"selected_document,omitempty"` + Phase string `json:"phase,omitempty"` + Evidence string `json:"evidence,omitempty"` + TargetBound *bool `json:"target_bound,omitempty"` + PayloadClass string `json:"payload_class,omitempty"` + SelfContained *bool `json:"self_contained,omitempty"` + ProfilerTimingAvailable *bool `json:"profiler_timing_available,omitempty"` + StructuralAnalysisAvailable *bool `json:"structural_analysis_available,omitempty"` + Warning string `json:"warning,omitempty"` +} + +type xcodeWindowSelection struct { + RequestedTrace string + Title string + Document string + Bound bool + Evidence string +} + +func newXcodeWindowSelection(requestedTrace, title, document string) xcodeWindowSelection { + selection := xcodeWindowSelection{ + RequestedTrace: requestedTrace, + Title: title, + Document: document, + } + if requestedTrace == "" { + selection.Bound = true + selection.Evidence = "no trace was requested; selected the available GPU trace window" + return selection + } + + requestedClean := strings.ToLower(filepath.Clean(requestedTrace)) + requestedBase := strings.ToLower(filepath.Base(requestedTrace)) + documentClean := strings.ToLower(filepath.Clean(document)) + titleLower := strings.ToLower(title) + switch { + case document != "" && documentClean == requestedClean: + selection.Bound = true + selection.Evidence = "AXDocument exactly matches the requested trace" + case document != "" && requestedBase != "" && strings.Contains(documentClean, requestedBase): + selection.Bound = true + selection.Evidence = "AXDocument contains the requested trace filename" + case title != "" && requestedBase != "" && strings.Contains(titleLower, requestedBase): + selection.Bound = true + selection.Evidence = "window title contains the requested trace filename" + default: + selection.Evidence = "selected GPU trace window has no title or AXDocument match for the requested trace" + } + return selection +} + +func selectionForWindow(requestedTrace string, window uintptr) xcodeWindowSelection { + selection := newXcodeWindowSelection( + requestedTrace, + axString(window, "AXTitle"), + axString(window, "AXDocument"), + ) + return admitUIIdentifiedWindow(selection, window, uiIdentifiedTraceWindow) +} + +// admitUIIdentifiedWindow accepts a window that getPreferredTraceWindow matched +// by its GPU trace UI alone. Xcode clears a trace window's title and AXDocument +// while it replays, which is exactly when that fallback runs, so uniqueness +// within the bound process is the only remaining evidence of binding. +func admitUIIdentifiedWindow(selection xcodeWindowSelection, window, uiIdentified uintptr) xcodeWindowSelection { + if selection.Bound || window == 0 || window != uiIdentified { + return selection + } + selection.Bound = true + selection.Evidence = "sole window with GPU trace UI in the bound Xcode process" + return selection +} + +func boolPointer(value bool) *bool { + return &value +} + +func requireBoundSelection(selection xcodeWindowSelection) error { + if selection.RequestedTrace != "" && !selection.Bound { + return fmt.Errorf("selected Xcode window is not bound to requested trace %q: %s", selection.RequestedTrace, selection.Evidence) + } + return nil } var collectProfileOpts = collectProfileOptions{ @@ -115,8 +202,10 @@ func collectXcodeProfilePreRun(cmd *cobra.Command, args []string) error { if exe, err := os.Executable(); err == nil && strings.Contains(exe, ".app/") { port = "6061" } - addr := ":" + port - fmt.Fprintf(os.Stderr, "[pprof] starting debug server on http://localhost%s/debug/pprof/\n", addr) + // Bind the loopback interface only: /debug/pprof/ exposes heap, + // goroutine, and cmdline data that should not leave the host. + addr := "127.0.0.1:" + port + fmt.Fprintf(os.Stderr, "[pprof] starting debug server on http://%s/debug/pprof/\n", addr) go func() { if err := http.ListenAndServe(addr, nil); err != nil { fmt.Fprintf(os.Stderr, "[pprof] server error: %v\n", err) @@ -268,23 +357,7 @@ func setupMacgo() error { os.Setenv("MACGO_SERVICES_VERSION", "1") - cfg := &macgo.Config{ - AppName: "gputrace", - BundleID: "com.tmc.gputrace", - Permissions: []macgo.Permission{ - macgo.Accessibility, - }, - Custom: []string{ - "com.apple.security.automation.apple-events", - }, - AdHocSign: true, - DevMode: true, - UIMode: macgo.UIModeAccessory, - Info: map[string]interface{}{ - "NSAppleEventsUsageDescription": "gputrace needs to control Xcode to automate GPU trace operations.", - "NSAccessibilityUsageDescription": "gputrace needs Accessibility access to control Xcode's UI for GPU trace automation.", - }, - } + cfg := xcodeProfileMacgoConfig() verboseLog("setupMacgo: calling macgo.Start with BundleID=%s, UIMode=Accessory, DevMode=true", cfg.BundleID) @@ -302,6 +375,31 @@ func setupMacgo() error { return nil } +func xcodeProfileMacgoConfig() *macgo.Config { + return &macgo.Config{ + AppName: "gputrace", + BundleID: "com.tmc.gputrace", + Permissions: []macgo.Permission{ + macgo.Accessibility, + }, + Custom: []string{ + "com.apple.security.automation.apple-events", + }, + AdHocSign: true, + DevMode: true, + // Xcode automation is a synchronous CLI operation. LaunchServices + // returns after launching the wrapper and loses the child command's + // exit status. Direct bundle execution waits for the child and forwards + // its nonzero status while retaining the signed bundle identity. + ForceDirectExecution: true, + UIMode: macgo.UIModeAccessory, + Info: map[string]interface{}{ + "NSAppleEventsUsageDescription": "gputrace needs to control Xcode to automate GPU trace operations.", + "NSAccessibilityUsageDescription": "gputrace needs Accessibility access to control Xcode's UI for GPU trace automation.", + }, + } +} + // logProcessIdentity prints diagnostic info about the current process's TCC identity. // This helps debug cases where check-status passes but runtime fails (different process identities). func logProcessIdentity(phase string) { @@ -365,7 +463,10 @@ func accessibilityPermissionError() error { fmt.Fprintln(os.Stderr, "\nPlease grant Accessibility permission to gputrace in:") fmt.Fprintln(os.Stderr, " System Settings > Privacy & Security > Accessibility") fmt.Fprintln(os.Stderr, "\nThen re-run the command.") - exec.Command("open", "x-apple.systempreferences:com.apple.settings.PrivacySecurity.extension?Privacy_Accessibility").Run() + const pane = "x-apple.systempreferences:com.apple.settings.PrivacySecurity.extension?Privacy_Accessibility" + if err := exec.Command("open", pane).Run(); err != nil { + fmt.Fprintf(os.Stderr, "warning: could not open Settings (%v); open the pane above manually\n", err) + } return fmt.Errorf("accessibility permission required") } diff --git a/cmd/gputrace/cmd/collect_xcode_profile_checkbox.go b/cmd/gputrace/cmd/collect_xcode_profile_checkbox.go index 3ced086c..55e15fb1 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_checkbox.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_checkbox.go @@ -144,14 +144,13 @@ func runToggleCheckbox(cmd *cobra.Command, args []string, opts *checkboxOptions) // If traceFile is empty, it looks for the first .gputrace window. func findTargetWindow(ctx context.Context, appAX uintptr, traceFile string) (uintptr, error) { if traceFile != "" { - // Extract just the filename for matching baseName := filepath.Base(traceFile) - windowAX := GetWindowByTitle(appAX, baseName) - if windowAX == 0 { - if diagnostic := xcodeWindowVisibilityDiagnostic(appAX); diagnostic != "" { - return 0, fmt.Errorf("no AX-visible Xcode window found for trace %q (%s)", baseName, diagnostic) - } - return 0, fmt.Errorf("no Xcode window found for trace %q", baseName) + if windowAX := getPreferredTraceWindow(appAX, traceFile); windowAX != 0 { + return windowAX, nil + } + windowAX, err := waitForWindow(ctx, appAX, traceFile, 10*time.Second) + if err != nil { + return 0, fmt.Errorf("find Xcode GPU trace window for %q: %w", baseName, err) } return windowAX, nil } diff --git a/cmd/gputrace/cmd/collect_xcode_profile_close.go b/cmd/gputrace/cmd/collect_xcode_profile_close.go index 512a0a99..c3c263a3 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_close.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_close.go @@ -3,7 +3,11 @@ package cmd import ( + "context" "fmt" + "path/filepath" + "strings" + "time" "github.com/spf13/cobra" ) @@ -25,11 +29,17 @@ func runCloseTrace(cmd *cobra.Command, args []string) error { defer cfRelease(appAX) var windowAX uintptr - windowAX, err = findTargetWindow(cmd.Context(), appAX, traceFile) + if traceFile != "" { + windowAX, err = waitForExactTraceWindow(cmd.Context(), appAX, traceFile, 10*time.Second) + } else { + windowAX, err = findTargetWindow(cmd.Context(), appAX, "") + } if err != nil { return err } title := axString(windowAX, "AXTitle") + document := axString(windowAX, "AXDocument") + initialWindows := deduplicateAXWindows(GetAllWindows(appAX)) if traceFile != "" { fmt.Fprintf(xcodeProfileStatusWriter(), "Closing window for: %s\n", traceFile) } else if title != "" { @@ -47,14 +57,92 @@ func runCloseTrace(cmd *cobra.Command, args []string) error { if err := axAction(closeBtn, "AXPress"); err != nil { return fmt.Errorf("failed to click close button: %w", err) } + if err := waitForClosedTraceWindow(cmd.Context(), appAX, title, document, len(initialWindows), 5*time.Second); err != nil { + return err + } - fmt.Fprintln(xcodeProfileStatusWriter(), "Done") + fmt.Fprintln(xcodeProfileStatusWriter(), "Trace window closed (verified absent from Xcode window list).") return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "close", - Target: traceFile, + Action: "close", + Target: traceFile, + SelectedTitle: title, + SelectedDocument: document, + Phase: "closed", + Evidence: "selected trace window is absent from the Xcode AX window list", }) } +// waitForExactTraceWindow finds the one Xcode window whose AXDocument names +// traceFile. It does not use the GPU UI fallback: close is destructive, and an +// untitled replay window cannot be safely attributed to a requested trace. +func waitForExactTraceWindow(ctx context.Context, appAX uintptr, traceFile string, timeout time.Duration) (uintptr, error) { + traceIdentity := strings.ToLower(filepath.Clean(traceFile)) + deadline := time.Now().Add(timeout) + for { + windows := deduplicateAXWindows(GetAllWindows(appAX)) + window, err := exactTraceWindow(windows, traceIdentity) + if err == nil { + return window, nil + } + if len(exactTraceWindows(windows, traceIdentity)) == 0 { + if time.Now().After(deadline) { + return 0, fmt.Errorf("find exact Xcode trace window for %q: no AXDocument match", filepath.Base(traceFile)) + } + } else { + return 0, fmt.Errorf("find exact Xcode trace window for %q: %w", filepath.Base(traceFile), err) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return 0, err + } + } +} + +func exactTraceWindow(windows []xcodeAXWindow, traceIdentity string) (uintptr, error) { + matches := exactTraceWindows(windows, traceIdentity) + if len(matches) != 1 { + return 0, fmt.Errorf("found %d AXDocument matches", len(matches)) + } + return matches[0], nil +} + +func waitForClosedTraceWindow(ctx context.Context, appAX uintptr, title, document string, initialCount int, timeout time.Duration) error { + deadline := time.Now().Add(timeout) + for { + windows := deduplicateAXWindows(GetAllWindows(appAX)) + if !windowSnapshotContainsTarget(windows, title, document, initialCount) { + return nil + } + if time.Now().After(deadline) { + return fmt.Errorf("close trace window: selected window is still present (title %q, document %q)", title, document) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return err + } + } +} + +func windowSnapshotContainsTarget(windows []xcodeAXWindow, title, document string, initialCount int) bool { + if document != "" { + want := strings.ToLower(filepath.Clean(document)) + for _, window := range windows { + if strings.ToLower(filepath.Clean(window.Document)) == want { + return true + } + } + return false + } + if title != "" { + want := strings.ToLower(strings.TrimSpace(title)) + for _, window := range windows { + if strings.ToLower(strings.TrimSpace(window.Title)) == want { + return true + } + } + return false + } + return len(windows) >= initialCount +} + // findCloseButton finds the close button in a window. func findCloseButton(window uintptr) uintptr { return findElement(window, func(el uintptr) bool { diff --git a/cmd/gputrace/cmd/collect_xcode_profile_close_test.go b/cmd/gputrace/cmd/collect_xcode_profile_close_test.go new file mode 100644 index 00000000..b3235e2f --- /dev/null +++ b/cmd/gputrace/cmd/collect_xcode_profile_close_test.go @@ -0,0 +1,51 @@ +//go:build darwin + +package cmd + +import ( + "path/filepath" + "strings" + "testing" +) + +func TestExactTraceWindow(t *testing.T) { + windows := []xcodeAXWindow{ + {Element: 1, Document: "/tmp/other.gputrace"}, + {Element: 2, Document: "/tmp/target.gputrace"}, + } + + window, err := exactTraceWindow(windows, strings.ToLower(filepath.Clean("/tmp/target.gputrace"))) + if err != nil || window != 2 { + t.Fatalf("exactTraceWindow(target) = %d, %v, want 2, nil", window, err) + } + + window, err = exactTraceWindow(windows, strings.ToLower(filepath.Clean("/tmp/missing.gputrace"))) + if err == nil || window != 0 { + t.Fatalf("exactTraceWindow(missing) = %d, %v, want 0, error", window, err) + } +} + +func TestWindowSnapshotContainsTarget(t *testing.T) { + windows := []xcodeAXWindow{ + {Title: "Other", Document: "/tmp/other.gputrace"}, + {Title: "Target", Document: "/tmp/target.gputrace"}, + } + if !windowSnapshotContainsTarget(windows, "Target", "/tmp/target.gputrace", 2) { + t.Fatal("document-bound target reported absent") + } + if windowSnapshotContainsTarget(windows[:1], "Target", "/tmp/target.gputrace", 2) { + t.Fatal("closed document-bound target reported present") + } + if !windowSnapshotContainsTarget(windows, "target", "", 2) { + t.Fatal("title-bound target reported absent") + } + if windowSnapshotContainsTarget(windows[:1], "target", "", 2) { + t.Fatal("closed title-bound target reported present") + } + if !windowSnapshotContainsTarget(windows, "", "", 2) { + t.Fatal("untitled target should remain present while window count is unchanged") + } + if windowSnapshotContainsTarget(windows[:1], "", "", 2) { + t.Fatal("untitled target should be absent after window count decreases") + } +} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_export.go b/cmd/gputrace/cmd/collect_xcode_profile_export.go index f6058994..0b8e7bc7 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_export.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_export.go @@ -3,20 +3,68 @@ package cmd import ( + "context" + "errors" "fmt" + "io" + "net/url" "os" "path/filepath" "strings" "time" "github.com/spf13/cobra" + gputraceTrace "github.com/tmc/gputrace/internal/trace" + "github.com/tmc/gputrace/internal/tracebundle" ) -func runExport(cmd *cobra.Command, args []string) error { +type standaloneExportRecovery struct { + Enabled bool + CheckOnly bool + Finalize bool + SourcePath string + SourceUUID string + Identity xcodeProcessIdentity +} + +type standaloneRecoveryWindow struct { + xcodeAXWindow + PID int + PerformanceView bool + SummaryView bool + NewEditorView bool + Finished bool + Debugging bool + Progress95 bool + SheetOpen bool + StopCount int + StopEnabled bool + ShowCount int + ShowEnabled bool +} + +type depthElement struct { + Element uintptr + Depth int +} + +type recoveryFinalizeSnapshot struct { + Identity xcodeProcessIdentity + WindowKey string + Performance bool + SheetOpen bool + StopCount int + StopEnabled bool + StopElement uintptr + ExportFound bool + ExportEnabled bool +} + +func runExport(cmd *cobra.Command, args []string) (retErr error) { status := xcodeProfileStatusWriter() var outputPath string + var err error if len(args) > 0 { - var err error outputPath, err = resolveXcodeProfileTraceOutputPath(args[0]) if err != nil { return err @@ -27,24 +75,107 @@ func runExport(cmd *cobra.Command, args []string) error { return err } - // Try AX-based approach first - appAX, err := FindXcodeApp() + var crashReportDir string + var crashBaseline map[string]crashReportState + recoveryRequested, _ := cmd.Flags().GetBool("recover-untitled") + if recoveryRequested { + crashReportDir = diagnosticReportDirectory() + crashBaseline, err = snapshotXcodeCrashReports(crashReportDir) + if err != nil { + return fmt.Errorf("snapshot Xcode crash reports: %w", err) + } + } + recovery, err := standaloneExportRecoveryFromFlags(cmd) if err != nil { - return fmt.Errorf("AX not available: %w", err) + return err + } + ctx := cmd.Context() + var crashScope *xcodeCrashScope + if recovery.Enabled { + crashScope = newXcodeCrashScope(recovery.Identity.AppPath, time.Now()) + crashScope.allowRebind = false + crashScope.bind(recovery.Identity) + var stopCrashMonitor func() + ctx, stopCrashMonitor = startXcodeCrashMonitor(ctx, crashReportDir, crashBaseline, crashScope) + defer stopCrashMonitor() + defer func() { + if retErr != nil { + retErr = normalizeStandaloneRecoveryFailure(ctx, crashScope, recovery, retErr) + } + }() + if err := validateStandaloneRecoveryIdentity(recovery.Identity, xcodeProcessPath(recovery.Identity.PID)); err != nil { + return err + } + } + + var appAX uintptr + var identity xcodeProcessIdentity + if recovery.Enabled { + identity = recovery.Identity + appAX, err = reacquireXcodeApp(identity) + if err != nil { + return fmt.Errorf("cannot establish recovery Xcode selection: %w", err) + } + fmt.Fprintf(status, "Recovering source: %s\n", recovery.SourcePath) + fmt.Fprintf(status, "Source trace UUID: %s\n", recovery.SourceUUID) + fmt.Fprintf(status, "Bound Xcode: PID %d app %s\n", identity.PID, identity.AppPath) + fmt.Fprintln(status, "Recovery accepts exact Summary, Performance, or source-bound Finished states; replay is never restarted") + } else { + requestedApp := requestedXcodeAppPath() + appAX, identity, err = findSingleXcodeApp(cmd.Context(), requestedApp, 10*time.Second) + if err != nil { + return fmt.Errorf("cannot establish exact Xcode selection: %w", err) + } } defer cfRelease(appAX) - windowAX, err := waitForWindow(cmd.Context(), appAX, "", 10*time.Second) - if err != nil { - return fmt.Errorf("Xcode window not found: %w", err) + var windowAX uintptr + var doc string + if recovery.Enabled { + if recovery.Finalize || recovery.CheckOnly { + windowAX, err = waitForStandaloneFinalizeWindow(ctx, appAX, recovery, 10*time.Second) + } else { + windowAX, err = waitForStandaloneRecoveryWindow(ctx, appAX, recovery, 10*time.Second) + } + if err != nil { + return err + } + doc = recovery.SourcePath + } else { + windowAX, doc, err = waitForStandaloneExportWindow(cmd.Context(), appAX, identity, 10*time.Second) + if err != nil { + return err + } } - doc := axString(windowAX, "AXDocument") + if err := requireStandaloneExportTarget(doc); err != nil { + return err + } + if recovery.CheckOnly { + fmt.Fprintln(status, "Recovery target verified; no UI action performed") + return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + Action: "check_recovery", + Target: recovery.SourcePath, + Source: recovery.SourcePath, + SourceUUID: recovery.SourceUUID, + XcodePID: recovery.Identity.PID, + XcodeApp: recovery.Identity.AppPath, + Phase: "recovery state verified", + Evidence: "exact PID/app and supported Summary, Performance, or source-bound Finished state stable across two samples", + TargetBound: boolPointer(true), + SelectedTitle: "", + SelectedDocument: "", + }) + } + if recovery.Finalize { + windowAX, err = finalizeRecoveredWorkload(ctx, appAX, windowAX, recovery, 2*time.Minute) + if err != nil { + return fmt.Errorf("finalize recovered workload: %w", err) + } + fmt.Fprintln(status, "Recovered workload finalized; source restored, Performance reopened, and Export is enabled") + } // If no output path specified, try to infer from window document if outputPath == "" { - if doc == "" { - return fmt.Errorf("output path not specified and could not be inferred from Xcode window (AXDocument empty)") - } // e.g. /path/to/trace.gputrace -> /path/to/trace-perfdata.gputrace ext := filepath.Ext(doc) // .gputrace if ext == "" { @@ -60,25 +191,1167 @@ func runExport(cmd *cobra.Command, args []string) error { verboseLog("runExport: window AXDocument=%q", doc) } - if err := exportTrace(cmd.Context(), appAX, windowAX, outputPath); err != nil { + if err := exportTrace(ctx, appAX, windowAX, outputPath); err != nil { return fmt.Errorf("export failed: %w", err) } - warning := "" - if _, err := os.Stat(outputPath); err == nil { - fmt.Fprintf(status, Colorize("Exported to: %s\n", ColorGreen), outputPath) - } else { - warning = "output file not found at expected location" - fmt.Fprint(status, Colorize("Note: Output file not found at expected location.\n", ColorYellow)) + candidates := exportCandidatePaths(doc, outputPath) + finalPath, err := waitForExportedTrace(ctx, []string{outputPath}, exportWaitTimeout()) + if err != nil { + if alternates := existingExportCandidates(candidates, outputPath); len(alternates) > 0 { + return fmt.Errorf("export did not appear at requested location %s; Xcode wrote candidate output at %s; preserving it for recovery: %w", + outputPath, strings.Join(alternates, ", "), err) + } + return err } - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "export", - Target: doc, - Output: outputPath, - Warning: warning, + payload, err := finalizeStandaloneExport(status, doc, finalPath) + if err != nil { + return err + } + if err := requireExportedTrace(outputPath); err != nil { + return err + } + fmt.Fprintf(status, Colorize("Exported to: %s\n", ColorGreen), outputPath) + actionOutput := xcodeProfileActionOutput{ + Action: "export", + Target: doc, + Output: outputPath, + } + if recovery.Enabled { + actionOutput.Source = recovery.SourcePath + actionOutput.SourceUUID = recovery.SourceUUID + actionOutput.XcodePID = recovery.Identity.PID + actionOutput.XcodeApp = recovery.Identity.AppPath + actionOutput.Evidence = "explicit untitled-window recovery; exported UUID verified against source" + actionOutput.TargetBound = boolPointer(true) + } + applyXcodePayload(&actionOutput, payload) + return writeXcodeProfileActionOutput(actionOutput) +} + +func standaloneExportRecoveryFromFlags(cmd *cobra.Command) (standaloneExportRecovery, error) { + enabled, _ := cmd.Flags().GetBool("recover-untitled") + checkOnly, _ := cmd.Flags().GetBool("check-recovery") + finalize, _ := cmd.Flags().GetBool("finalize-workload") + source, _ := cmd.Flags().GetString("source") + pid, _ := cmd.Flags().GetInt("xcode-pid") + app, _ := cmd.Flags().GetString("xcode-app") + + any := enabled || checkOnly || finalize || source != "" || pid != 0 || app != "" + if !any { + return standaloneExportRecovery{}, nil + } + // --source on its own declares which trace the selected window holds, for + // the case where Xcode has cleared the window's AXDocument after replay. + // Identity is still verified against that trace after the export is written. + if source != "" && !enabled && !checkOnly && !finalize && pid == 0 && app == "" { + absolute, err := filepath.Abs(source) + if err != nil { + return standaloneExportRecovery{}, fmt.Errorf("resolve declared source: %w", err) + } + declaredExportSource = absolute + return standaloneExportRecovery{}, nil + } + if !enabled || source == "" || pid <= 0 || app == "" { + return standaloneExportRecovery{}, fmt.Errorf( + "untitled recovery requires --recover-untitled, --source, --xcode-pid, and --xcode-app", + ) + } + if checkOnly && finalize { + return standaloneExportRecovery{}, fmt.Errorf("--check-recovery and --finalize-workload are mutually exclusive") + } + if !filepath.IsAbs(app) { + return standaloneExportRecovery{}, fmt.Errorf("--xcode-app must be an absolute .app path") + } + app = filepath.Clean(app) + if !strings.HasSuffix(strings.ToLower(app), ".app") { + return standaloneExportRecovery{}, fmt.Errorf("--xcode-app must name an .app bundle") + } + source, err := filepath.Abs(source) + if err != nil { + return standaloneExportRecovery{}, fmt.Errorf("resolve recovery source: %w", err) + } + source = filepath.Clean(source) + if !strings.HasSuffix(strings.ToLower(source), ".gputrace") { + return standaloneExportRecovery{}, fmt.Errorf("--source must name a .gputrace bundle") + } + payload, err := tracebundle.InspectPayload(source) + if err != nil { + return standaloneExportRecovery{}, fmt.Errorf("inspect recovery source: %w", err) + } + if payload.Class != tracebundle.PayloadFull { + return standaloneExportRecovery{}, fmt.Errorf("recovery source is not self-contained: %s", source) + } + metadata, err := gputraceTrace.ReadMetadata(source) + if err != nil { + return standaloneExportRecovery{}, fmt.Errorf("read recovery source metadata: %w", err) + } + if metadata.UUID == "" { + return standaloneExportRecovery{}, fmt.Errorf("recovery source has no trace UUID: %s", source) + } + identity := xcodeProcessIdentity{PID: pid, AppPath: app, BundleID: "com.apple.dt.Xcode"} + return standaloneExportRecovery{ + Enabled: true, + CheckOnly: checkOnly, + Finalize: finalize, + SourcePath: source, + SourceUUID: metadata.UUID, + Identity: identity, + }, nil +} + +func validateStandaloneRecoveryIdentity(identity xcodeProcessIdentity, actualApp string) error { + if identity.PID <= 0 { + return fmt.Errorf("invalid Xcode PID %d", identity.PID) + } + actualApp = strings.TrimSpace(actualApp) + if actualApp == "" { + return fmt.Errorf("Xcode PID %d is not running", identity.PID) + } + actualApp = filepath.Clean(actualApp) + if actualApp != filepath.Clean(identity.AppPath) { + return fmt.Errorf("Xcode PID %d runs from %s, not requested app %s", + identity.PID, actualApp, identity.AppPath) + } + return nil +} + +func normalizeStandaloneRecoveryFailure(ctx context.Context, scope *xcodeCrashScope, recovery standaloneExportRecovery, original error) error { + return normalizeStandaloneRecoveryFailureWithGrace(ctx, scope, recovery, original, xcodeCrashReportGrace) +} + +func normalizeStandaloneRecoveryFailureWithGrace(ctx context.Context, scope *xcodeCrashScope, recovery standaloneExportRecovery, original error, grace time.Duration) error { + if scope == nil { + return original + } + if xcodeProcessPath(recovery.Identity.PID) != "" { + return original + } + scope.refreshProcesses() + if !scope.crashSuspected() { + return original + } + var report xcodeCrashReport + if cause := context.Cause(ctx); errors.As(cause, &report) { + return cause + } + if err := waitForXcodeCrashReport(ctx, grace); err != nil { + if errors.As(err, &report) { + return err + } + return fmt.Errorf("bound Xcode PID %d exited while waiting for a crash report (File menu state unavailable after process exit): %w", + recovery.Identity.PID, err) + } + return fmt.Errorf("bound Xcode PID %d exited; File menu state unavailable after process exit; no matching DiagnosticReport appeared within %s: %w", + recovery.Identity.PID, grace, original) +} + +// declaredExportSource is the trace path given by a bare --source. It supplies +// the identity that Xcode dropped from the window's AXDocument during replay, +// so verifyExportTraceIdentity still runs against a known trace. +var declaredExportSource string + +func standaloneExportTarget(windows []xcodeAXWindow) (uintptr, string, error) { + var matches []xcodeAXWindow + for _, window := range windows { + doc := filepath.Clean(strings.TrimSpace(window.Document)) + if doc == "." || !filepath.IsAbs(doc) || !strings.HasSuffix(strings.ToLower(doc), ".gputrace") { + continue + } + matches = append(matches, window) + } + switch len(matches) { + case 0: + // Xcode clears a trace window's AXDocument while it replays and does not + // always restore it. Fall back to the window carrying GPU trace UI when + // exactly one does; uniqueness within the bound process is then the only + // available evidence of identity. The caller still verifies the exported + // bundle against the requested source. + var ui []xcodeAXWindow + for _, window := range windows { + if hasGPUTraceUI(window.Element) { + ui = append(ui, window) + } + } + if len(ui) == 1 { + return ui[0].Element, declaredExportSource, nil + } + return 0, "", fmt.Errorf("cannot establish standalone export target: no AXDocument-bound .gputrace window") + case 1: + return matches[0].Element, matches[0].Document, nil + default: + var docs []string + for _, match := range matches { + docs = append(docs, match.Document) + } + return 0, "", fmt.Errorf("cannot establish standalone export target: multiple .gputrace windows are open: %s", + strings.Join(docs, ", ")) + } +} + +func waitForStandaloneExportWindow(ctx context.Context, appAX uintptr, identity xcodeProcessIdentity, timeout time.Duration) (uintptr, string, error) { + deadline := time.Now().Add(timeout) + var lastErr error + for { + bound, err := xcodeIdentityForAX(appAX) + if err != nil || bound.PID != identity.PID || filepath.Clean(bound.AppPath) != filepath.Clean(identity.AppPath) { + return 0, "", fmt.Errorf("standalone export lost exact Xcode PID/app binding: want PID %d app %s", + identity.PID, identity.AppPath) + } + window, doc, err := standaloneExportTarget(deduplicateAXWindows(GetAllWindows(appAX))) + if err == nil { + var pid int32 + if axUIElementGetPid(window, &pid) != kAXErrorSuccess || int(pid) != identity.PID { + return 0, "", fmt.Errorf("standalone export target window is not owned by bound Xcode PID %d", identity.PID) + } + return window, doc, nil + } + lastErr = err + if time.Now().After(deadline) { + return 0, "", fmt.Errorf("standalone export target not established for PID %d app %s: %w", + identity.PID, identity.AppPath, lastErr) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return 0, "", err + } + } +} + +func standaloneRecoveryTarget(windows []standaloneRecoveryWindow, recovery standaloneExportRecovery) (standaloneRecoveryWindow, error) { + var matches []standaloneRecoveryWindow + seen := make(map[uintptr]bool) + for _, window := range windows { + if window.PID != recovery.Identity.PID || window.Document != "" || + strings.TrimSpace(window.Title) != "" || !window.PerformanceView { + continue + } + if seen[window.Element] { + continue + } + seen[window.Element] = true + matches = append(matches, window) + } + switch len(matches) { + case 0: + return standaloneRecoveryWindow{}, fmt.Errorf("no untitled Performance window is bound to Xcode PID %d app %s", + recovery.Identity.PID, recovery.Identity.AppPath) + case 1: + return matches[0], nil + default: + var elements []string + for _, match := range matches { + elements = append(elements, fmt.Sprintf("%d", match.Element)) + } + return standaloneRecoveryWindow{}, fmt.Errorf("multiple untitled Performance windows are ambiguous for Xcode PID %d app %s: AX elements %s", + recovery.Identity.PID, recovery.Identity.AppPath, strings.Join(elements, ", ")) + } +} + +func standaloneRecoveryWindowKey(window standaloneRecoveryWindow) string { + return fmt.Sprintf("%d\x00%s\x00%s\x00%d,%d,%d,%d", + window.PID, + strings.TrimSpace(window.Title), + filepath.Clean(window.Document), + window.X, + window.Y, + window.Width, + window.Height, + ) +} + +func standaloneRecoveryGeometryKey(window standaloneRecoveryWindow) string { + return fmt.Sprintf("%d\x00%d,%d,%d,%d", + window.PID, window.X, window.Y, window.Width, window.Height) +} + +func recoveryGeometryKeyForElement(element uintptr, pid int) string { + x, y := axPosition(element) + width, height := axSize(element) + return standaloneRecoveryGeometryKey(standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{ + Element: element, + X: x, + Y: y, + Width: width, + Height: height, + }, + PID: pid, }) } +func recoveryWindows(appAX uintptr) []standaloneRecoveryWindow { + windows := deduplicateAXWindows(GetAllWindows(appAX)) + out := make([]standaloneRecoveryWindow, 0, len(windows)) + for _, window := range windows { + var pid int32 + if axUIElementGetPid(window.Element, &pid) != kAXErrorSuccess { + continue + } + stops := shallowStopButtons(window.Element) + shows := shallowShowPerformanceButtons(window.Element) + out = append(out, standaloneRecoveryWindow{ + xcodeAXWindow: window, + PID: int(pid), + PerformanceView: hasPerformanceView(window.Element), + SummaryView: hasShallowNamedGroup(window.Element, "Summary"), + NewEditorView: hasShallowNamedGroup(window.Element, "New Editor"), + Finished: hasShallowFinishedActivity(window.Element), + Debugging: hasShallowActivityText(window.Element, "macOS App - Debugging GPU Workload", false), + Progress95: hasShallowActivityText(window.Element, "95% completed", true), + SheetOpen: shallowSheetOpen(window.Element), + StopCount: len(stops), + StopEnabled: len(stops) == 1 && IsElementEnabled(stops[0]), + ShowCount: len(shows), + ShowEnabled: len(shows) == 1 && IsElementEnabled(shows[0]), + }) + } + return out +} + +func hasPerformanceView(root uintptr) bool { + return hasShallowNamedGroup(root, "Performance") || hasPerformanceData(root) +} + +func hasShallowNamedGroup(root uintptr, name string) bool { + return findElementAtDepth( + root, + 4, + 128, + axChildren, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXOutline" + }, + func(element uintptr) bool { + role := axString(element, "AXRole") + description := strings.TrimSpace(axString(element, "AXDescription")) + return (role == "AXGroup" || role == "AXSplitGroup") && description == name + }, + ) != 0 +} + +func hasShallowFinishedActivity(root uintptr) bool { + return hasShallowActivityText(root, "Finished running macOS App", false) +} + +func hasShallowActivityText(root uintptr, text string, contains bool) bool { + return findElementAtDepth( + root, + 5, + 256, + axChildren, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXOutline" + }, + func(element uintptr) bool { + for _, attribute := range []string{"AXValue", "AXTitle", "AXDescription"} { + value := strings.TrimSpace(axString(element, attribute)) + if value == text || contains && strings.Contains(value, text) { + return true + } + } + return false + }, + ) != 0 +} + +func findElementAtDepth( + root uintptr, + maxDepth, maxVisit int, + children func(uintptr) []uintptr, + prune, match func(uintptr) bool, +) uintptr { + if root == 0 || maxDepth < 0 || maxVisit <= 0 { + return 0 + } + queue := []depthElement{{Element: root}} + seen := make(map[uintptr]bool) + visited := 0 + for len(queue) > 0 && visited < maxVisit { + item := queue[0] + queue = queue[1:] + if item.Element == 0 || seen[item.Element] { + continue + } + seen[item.Element] = true + visited++ + if match(item.Element) { + return item.Element + } + if item.Depth >= maxDepth || prune(item.Element) { + continue + } + for _, child := range children(item.Element) { + queue = append(queue, depthElement{Element: child, Depth: item.Depth + 1}) + } + } + return 0 +} + +func findElementsAtDepth( + root uintptr, + maxDepth, maxVisit, maxMatches int, + children func(uintptr) []uintptr, + prune, match func(uintptr) bool, +) []uintptr { + if root == 0 || maxDepth < 0 || maxVisit <= 0 || maxMatches <= 0 { + return nil + } + queue := []depthElement{{Element: root}} + seen := make(map[uintptr]bool) + var matches []uintptr + visited := 0 + for len(queue) > 0 && visited < maxVisit && len(matches) < maxMatches { + item := queue[0] + queue = queue[1:] + if item.Element == 0 || seen[item.Element] { + continue + } + seen[item.Element] = true + visited++ + if match(item.Element) { + matches = append(matches, item.Element) + } + if item.Depth >= maxDepth || prune(item.Element) { + continue + } + for _, child := range children(item.Element) { + queue = append(queue, depthElement{Element: child, Depth: item.Depth + 1}) + } + } + return matches +} + +func shallowStopButtons(root uintptr) []uintptr { + return findElementsAtDepth( + root, + 4, + 128, + 2, + axChildren, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXOutline" + }, + func(element uintptr) bool { + if axString(element, "AXRole") != "AXButton" { + return false + } + title := axString(element, "AXTitle") + description := axString(element, "AXDescription") + return title == "Stop GPU workload" || description == "Stop GPU workload" + }, + ) +} + +func shallowShowPerformanceButtons(root uintptr) []uintptr { + return findElementsAtDepth( + root, + 6, + 512, + 2, + axChildren, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXOutline" + }, + func(element uintptr) bool { + if axString(element, "AXRole") != "AXButton" { + return false + } + title := axString(element, "AXTitle") + description := axString(element, "AXDescription") + return title == "Show Performance" || description == "Show Performance" || + title == "Open Performance" || description == "Open Performance" + }, + ) +} + +func readRecoveryFinalizeSnapshot(appAX uintptr, recovery standaloneExportRecovery) (recoveryFinalizeSnapshot, error) { + identity, err := xcodeIdentityForAX(appAX) + if err != nil { + return recoveryFinalizeSnapshot{}, err + } + if identity.PID != recovery.Identity.PID || + filepath.Clean(identity.AppPath) != filepath.Clean(recovery.Identity.AppPath) { + return recoveryFinalizeSnapshot{}, fmt.Errorf("recovery Xcode identity changed: got PID %d app %s", + identity.PID, identity.AppPath) + } + window, err := standaloneRecoveryTarget(recoveryWindows(appAX), recovery) + if err != nil { + return recoveryFinalizeSnapshot{}, err + } + stops := shallowStopButtons(window.Element) + snapshot := recoveryFinalizeSnapshot{ + Identity: identity, + WindowKey: standaloneRecoveryWindowKey(window), + Performance: window.PerformanceView, + StopCount: len(stops), + } + if len(stops) == 1 { + snapshot.StopElement = stops[0] + snapshot.StopEnabled = IsElementEnabled(stops[0]) + } + snapshot.SheetOpen = findElementAtDepth( + window.Element, + 3, + 64, + axChildren, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXOutline" + }, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXSheet" + }, + ) != 0 + snapshot.ExportFound, snapshot.ExportEnabled, err = fileExportMenuState(appAX, window.Element) + if err != nil { + return recoveryFinalizeSnapshot{}, err + } + afterIdentity, err := xcodeIdentityForAX(appAX) + if err != nil || afterIdentity.PID != identity.PID || + filepath.Clean(afterIdentity.AppPath) != filepath.Clean(identity.AppPath) { + return recoveryFinalizeSnapshot{}, fmt.Errorf("recovery Xcode identity changed while probing File > Export") + } + afterWindow, err := standaloneRecoveryTarget(recoveryWindows(appAX), recovery) + if err != nil { + return recoveryFinalizeSnapshot{}, err + } + if standaloneRecoveryWindowKey(afterWindow) != snapshot.WindowKey { + return recoveryFinalizeSnapshot{}, fmt.Errorf("recovery window identity changed while probing File > Export") + } + return snapshot, nil +} + +func validateRecoveryFinalizePrecondition(snapshot recoveryFinalizeSnapshot, recovery standaloneExportRecovery, windowKey string) error { + if snapshot.Identity.PID != recovery.Identity.PID || + filepath.Clean(snapshot.Identity.AppPath) != filepath.Clean(recovery.Identity.AppPath) { + return fmt.Errorf("recovery finalize identity mismatch") + } + if snapshot.WindowKey != windowKey { + return fmt.Errorf("recovery finalize window identity changed") + } + if !snapshot.Performance { + return fmt.Errorf("recovery finalize requires a populated Performance group") + } + if snapshot.SheetOpen { + return fmt.Errorf("recovery finalize refuses a window with an open sheet") + } + if snapshot.StopCount != 1 || !snapshot.StopEnabled { + return fmt.Errorf("recovery finalize requires exactly one enabled Stop GPU workload control") + } + if !snapshot.ExportFound { + return fmt.Errorf("recovery finalize could not find File > Export") + } + if snapshot.ExportEnabled { + return fmt.Errorf("recovery window is already export-ready; omit --finalize-workload") + } + return nil +} + +func finalizeRecoveredWorkload(ctx context.Context, appAX, windowAX uintptr, recovery standaloneExportRecovery, timeout time.Duration) (uintptr, error) { + axAction(windowAX, "AXRaise") + geometryKey := recoveryGeometryKeyForElement(windowAX, recovery.Identity.PID) + + if _, summaryErr := summaryRecoveryTarget(recoveryWindows(appAX), recovery, geometryKey); summaryErr == nil { + transitioned, err := transitionSummaryToPerformance(ctx, appAX, recovery, geometryKey, time.Now().Add(timeout)) + if err != nil { + return 0, err + } + windowAX = transitioned + } + + if _, err := restoredRecoverySourceTarget(recoveryWindows(appAX), recovery, geometryKey); err != nil { + before, err := readRecoveryFinalizeSnapshot(appAX, recovery) + if err != nil { + return 0, err + } + windowKey := before.WindowKey + if err := validateRecoveryFinalizePrecondition(before, recovery, windowKey); err != nil { + return 0, err + } + + // Re-read after probing File > Export so the exact window and Stop + // control are current at the only mutating action in this phase. + before, err = readRecoveryFinalizeSnapshot(appAX, recovery) + if err != nil { + return 0, err + } + if err := validateRecoveryFinalizePrecondition(before, recovery, windowKey); err != nil { + return 0, err + } + var stopPID int32 + if axUIElementGetPid(before.StopElement, &stopPID) != kAXErrorSuccess || + int(stopPID) != recovery.Identity.PID { + return 0, fmt.Errorf("Stop GPU workload is not owned by bound Xcode PID %d", recovery.Identity.PID) + } + if err := axPressWithFallbackWindow(before.StopElement, windowAX); err != nil { + return 0, fmt.Errorf("press Stop GPU workload: %w", err) + } + } + + deadline := time.Now().Add(timeout) + sourceWindow, err := waitForRestoredRecoverySource(ctx, appAX, recovery, geometryKey, deadline) + if err != nil { + return 0, err + } + shows := shallowShowPerformanceButtons(sourceWindow.Element) + switch len(shows) { + case 0: + if err := clickFinishedPerformanceOCR(ctx, appAX, sourceWindow, recovery, geometryKey); err != nil { + return 0, err + } + case 1: + if !IsElementEnabled(shows[0]) { + return 0, fmt.Errorf("Finished Show Performance control is disabled") + } + var showPID int32 + if axUIElementGetPid(shows[0], &showPID) != kAXErrorSuccess || + int(showPID) != recovery.Identity.PID { + return 0, fmt.Errorf("Show Performance is not owned by bound Xcode PID %d", recovery.Identity.PID) + } + if err := axPressWithFallbackWindow(shows[0], sourceWindow.Element); err != nil { + return 0, fmt.Errorf("press Show Performance: %w", err) + } + default: + return 0, fmt.Errorf("multiple AX Show Performance controls are ambiguous") + } + + return waitForFinalizedRecoveryPerformance(ctx, appAX, recovery, geometryKey, deadline) +} + +func transitionSummaryToPerformance(ctx context.Context, appAX uintptr, recovery standaloneExportRecovery, geometryKey string, deadline time.Time) (uintptr, error) { + stable := 0 + var lastKey string + var summary standaloneRecoveryWindow + for { + if err := checkAutomationCanceled(ctx); err != nil { + return 0, err + } + if err := requireRecoveryIdentity(appAX, recovery); err != nil { + return 0, err + } + window, err := summaryRecoveryTarget(recoveryWindows(appAX), recovery, geometryKey) + if err == nil { + key := standaloneRecoveryWindowKey(window) + if key == lastKey { + stable++ + } else { + lastKey = key + stable = 1 + } + summary = window + } else { + lastKey = "" + stable = 0 + } + if stable >= 2 { + break + } + if time.Now().After(deadline) { + return 0, recoveryTimeoutError("timed out waiting for stable 95% Summary recovery state", err) + } + if err := waitForAutomation(ctx, 250*time.Millisecond); err != nil { + return 0, err + } + } + + // Re-read the complete Summary precondition at the only mutating action. + summary, err := summaryRecoveryTarget(recoveryWindows(appAX), recovery, geometryKey) + if err != nil { + return 0, err + } + shows := shallowShowPerformanceButtons(summary.Element) + switch len(shows) { + case 0: + if err := clickSummaryPerformanceOCR(ctx, appAX, summary, recovery, geometryKey); err != nil { + return 0, err + } + case 1: + if !IsElementEnabled(shows[0]) { + return 0, fmt.Errorf("Summary Show Performance control is disabled") + } + var showPID int32 + if axUIElementGetPid(shows[0], &showPID) != kAXErrorSuccess || + int(showPID) != recovery.Identity.PID { + return 0, fmt.Errorf("Show Performance is not owned by bound Xcode PID %d", recovery.Identity.PID) + } + if err := axPressWithFallbackWindow(shows[0], summary.Element); err != nil { + return 0, fmt.Errorf("press Show Performance from Summary: %w", err) + } + default: + return 0, fmt.Errorf("multiple AX Show Performance controls are ambiguous") + } + + stable = 0 + lastKey = "" + for { + if err := checkAutomationCanceled(ctx); err != nil { + return 0, err + } + if err := requireRecoveryIdentity(appAX, recovery); err != nil { + return 0, err + } + window, err := runningRecoveryPerformanceTarget(recoveryWindows(appAX), recovery, geometryKey) + if err == nil { + key := standaloneRecoveryWindowKey(window) + if key == lastKey { + stable++ + } else { + lastKey = key + stable = 1 + } + } else { + lastKey = "" + stable = 0 + } + if stable >= 2 { + return window.Element, nil + } + if time.Now().After(deadline) { + return 0, recoveryTimeoutError("timed out waiting for Performance after Summary Show Performance", err) + } + if err := waitForAutomation(ctx, 250*time.Millisecond); err != nil { + return 0, err + } + } +} + +func waitForRestoredRecoverySource(ctx context.Context, appAX uintptr, recovery standaloneExportRecovery, geometryKey string, deadline time.Time) (standaloneRecoveryWindow, error) { + stable := 0 + var lastKey string + for { + if err := checkAutomationCanceled(ctx); err != nil { + return standaloneRecoveryWindow{}, err + } + if err := requireRecoveryIdentity(appAX, recovery); err != nil { + return standaloneRecoveryWindow{}, err + } + window, err := restoredRecoverySourceTarget(recoveryWindows(appAX), recovery, geometryKey) + if err == nil { + if shallowSheetOpen(window.Element) { + err = fmt.Errorf("restored source window has an open sheet") + } + } + if err == nil { + key := standaloneRecoveryWindowKey(window) + if key == lastKey { + stable++ + } else { + lastKey = key + stable = 1 + } + } else { + lastKey = "" + stable = 0 + } + if stable >= 2 { + return window, nil + } + if time.Now().After(deadline) { + return standaloneRecoveryWindow{}, recoveryTimeoutError("timed out waiting for exact source-bound Finished state after Stop", err) + } + if err := waitForAutomation(ctx, 500*time.Millisecond); err != nil { + return standaloneRecoveryWindow{}, err + } + } +} + +func waitForFinalizedRecoveryPerformance(ctx context.Context, appAX uintptr, recovery standaloneExportRecovery, geometryKey string, deadline time.Time) (uintptr, error) { + stable := 0 + probes := 0 + var lastKey string + var lastErr error + for { + if err := checkAutomationCanceled(ctx); err != nil { + return 0, err + } + if err := requireRecoveryIdentity(appAX, recovery); err != nil { + return 0, err + } + window, err := finalizedRecoveryPerformanceTarget(recoveryWindows(appAX), recovery, geometryKey) + if err == nil { + if shallowSheetOpen(window.Element) { + err = fmt.Errorf("unexpected sheet appeared after Show Performance") + } else if stops := shallowStopButtons(window.Element); len(stops) != 0 { + err = fmt.Errorf("Stop GPU workload reappeared after Show Performance") + } + } + if err == nil { + key := standaloneRecoveryWindowKey(window) + if key == lastKey { + stable++ + } else { + lastKey = key + stable = 1 + } + } else { + lastKey = "" + stable = 0 + lastErr = err + } + // The File > Export probe opens the menu bar, so it runs only once the + // cheap non-mutating signals above have gone stable. Probing it on every + // poll opened and closed the File menu twice a second for the whole + // wait, which takes key focus away from whatever else is running. + if stable >= 2 { + if probes >= maxFileExportProbes { + return 0, recoveryTimeoutError( + fmt.Sprintf("File > Export never became export-ready in %d probes", probes), lastErr) + } + probes++ + found, enabled, menuErr := fileExportMenuState(appAX, window.Element) + switch { + case menuErr != nil: + err = menuErr + case !found: + err = fmt.Errorf("File > Export disappeared after Show Performance") + case !enabled: + err = fmt.Errorf("File > Export remains disabled after Show Performance") + } + if err == nil { + return window.Element, nil + } + lastErr = err + lastKey = "" + stable = 0 + } + if time.Now().After(deadline) { + return 0, recoveryTimeoutError("timed out waiting for export-ready Performance after Show Performance", lastErr) + } + if err := waitForAutomation(ctx, 500*time.Millisecond); err != nil { + return 0, err + } + } +} + +func recoveryTimeoutError(message string, lastErr error) error { + if lastErr == nil { + return errors.New(message) + } + return fmt.Errorf("%s: %w", message, lastErr) +} + +func requireRecoveryIdentity(appAX uintptr, recovery standaloneExportRecovery) error { + identity, err := xcodeIdentityForAX(appAX) + if err != nil { + return err + } + if identity.PID != recovery.Identity.PID || + filepath.Clean(identity.AppPath) != filepath.Clean(recovery.Identity.AppPath) { + return fmt.Errorf("recovery Xcode identity changed: got PID %d app %s", identity.PID, identity.AppPath) + } + return nil +} + +func summaryRecoveryTarget(windows []standaloneRecoveryWindow, recovery standaloneExportRecovery, geometryKey string) (standaloneRecoveryWindow, error) { + var matches []standaloneRecoveryWindow + seen := make(map[string]bool) + for _, window := range windows { + key := standaloneRecoveryGeometryKey(window) + if window.PID != recovery.Identity.PID || + geometryKey != "" && key != geometryKey || + strings.TrimSpace(window.Title) != "" || + normalizedTraceDocument(window.Document) != "" || + !window.SummaryView || !window.Debugging || !window.Progress95 || + window.SheetOpen || window.StopCount != 1 || !window.StopEnabled || + seen[key] { + continue + } + seen[key] = true + matches = append(matches, window) + } + if len(matches) != 1 { + return standaloneRecoveryWindow{}, fmt.Errorf("want one exact untitled 95%% Summary window, found %d", len(matches)) + } + return matches[0], nil +} + +func runningRecoveryPerformanceTarget(windows []standaloneRecoveryWindow, recovery standaloneExportRecovery, geometryKey string) (standaloneRecoveryWindow, error) { + window, err := transitionedRecoveryPerformanceTarget(windows, recovery, geometryKey) + if err != nil { + return standaloneRecoveryWindow{}, err + } + if window.StopCount != 1 || !window.StopEnabled { + return standaloneRecoveryWindow{}, fmt.Errorf("transitioned Performance window is not running") + } + return window, nil +} + +func transitionedRecoveryPerformanceTarget(windows []standaloneRecoveryWindow, recovery standaloneExportRecovery, geometryKey string) (standaloneRecoveryWindow, error) { + var matches []standaloneRecoveryWindow + seen := make(map[string]bool) + for _, window := range windows { + key := standaloneRecoveryGeometryKey(window) + if window.PID != recovery.Identity.PID || + key != geometryKey || + !window.PerformanceView || window.SheetOpen || + seen[key] { + continue + } + doc := normalizedTraceDocument(window.Document) + title := strings.TrimSpace(window.Title) + if doc != "" && !traceDocumentMatches(window.Document, recovery.SourcePath) { + continue + } + if title != "" && title != filepath.Base(recovery.SourcePath) { + continue + } + seen[key] = true + matches = append(matches, window) + } + if len(matches) != 1 { + return standaloneRecoveryWindow{}, fmt.Errorf("want one Performance window with exact transition provenance, found %d", len(matches)) + } + return matches[0], nil +} + +func restoredRecoverySourceTarget(windows []standaloneRecoveryWindow, recovery standaloneExportRecovery, geometryKey string) (standaloneRecoveryWindow, error) { + var matches []standaloneRecoveryWindow + for _, window := range windows { + if window.PID != recovery.Identity.PID || + standaloneRecoveryGeometryKey(window) != geometryKey || + !isRestoredRecoverySource(window, recovery) { + continue + } + matches = append(matches, window) + } + if len(matches) != 1 { + return standaloneRecoveryWindow{}, fmt.Errorf("want one exact source-bound Finished New Editor window, found %d", len(matches)) + } + return matches[0], nil +} + +func restoredRecoverySourceAnyGeometry(windows []standaloneRecoveryWindow, recovery standaloneExportRecovery) (standaloneRecoveryWindow, error) { + var matches []standaloneRecoveryWindow + for _, window := range windows { + if window.PID == recovery.Identity.PID && isRestoredRecoverySource(window, recovery) { + matches = append(matches, window) + } + } + if len(matches) != 1 { + return standaloneRecoveryWindow{}, fmt.Errorf("want one exact source-bound Finished New Editor window, found %d", len(matches)) + } + return matches[0], nil +} + +func isRestoredRecoverySource(window standaloneRecoveryWindow, recovery standaloneExportRecovery) bool { + return traceDocumentMatches(window.Document, recovery.SourcePath) && + strings.TrimSpace(window.Title) == filepath.Base(recovery.SourcePath) && + window.NewEditorView && window.Finished +} + +func finalizedRecoveryPerformanceTarget(windows []standaloneRecoveryWindow, recovery standaloneExportRecovery, geometryKey string) (standaloneRecoveryWindow, error) { + var matches []standaloneRecoveryWindow + for _, window := range windows { + if window.PID != recovery.Identity.PID || + standaloneRecoveryGeometryKey(window) != geometryKey || + !window.PerformanceView { + continue + } + doc := normalizedTraceDocument(window.Document) + title := strings.TrimSpace(window.Title) + if doc != "" && !traceDocumentMatches(window.Document, recovery.SourcePath) { + continue + } + if title != "" && title != filepath.Base(recovery.SourcePath) { + continue + } + matches = append(matches, window) + } + if len(matches) != 1 { + return standaloneRecoveryWindow{}, fmt.Errorf("want one transitioned Performance window with exact source provenance, found %d", len(matches)) + } + return matches[0], nil +} + +func normalizedTraceDocument(document string) string { + document = strings.TrimSpace(document) + if document == "" { + return "" + } + if parsed, err := url.Parse(document); err == nil && parsed.Scheme == "file" { + if path, err := url.PathUnescape(parsed.Path); err == nil { + document = path + } + } + return filepath.Clean(document) +} + +func traceDocumentMatches(document, source string) bool { + for _, value := range []string{document, source} { + parsed, err := url.Parse(strings.TrimSpace(value)) + if err != nil || parsed.Scheme != "" && parsed.Scheme != "file" { + return false + } + } + document = normalizedTraceDocument(document) + source = normalizedTraceDocument(source) + if document == "" || source == "" { + return false + } + document, err := filepath.Abs(document) + if err != nil { + return false + } + source, err = filepath.Abs(source) + if err != nil { + return false + } + if document == source { + return true + } + resolvedDocument, documentErr := filepath.EvalSymlinks(document) + resolvedSource, sourceErr := filepath.EvalSymlinks(source) + if documentErr == nil && sourceErr == nil && resolvedDocument == resolvedSource { + return true + } + documentInfo, documentErr := os.Stat(document) + sourceInfo, sourceErr := os.Stat(source) + return documentErr == nil && sourceErr == nil && os.SameFile(documentInfo, sourceInfo) +} + +func shallowSheetOpen(window uintptr) bool { + return findElementAtDepth( + window, + 3, + 64, + axChildren, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXOutline" + }, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXSheet" + }, + ) != 0 +} + +func waitForStandaloneRecoveryWindow(ctx context.Context, appAX uintptr, recovery standaloneExportRecovery, timeout time.Duration) (uintptr, error) { + deadline := time.Now().Add(timeout) + var lastKey string + stable := 0 + var lastErr error + for { + bound, err := xcodeIdentityForAX(appAX) + if err != nil || bound.PID != recovery.Identity.PID || + filepath.Clean(bound.AppPath) != filepath.Clean(recovery.Identity.AppPath) { + return 0, fmt.Errorf("untitled recovery lost exact Xcode PID/app binding: want PID %d app %s", + recovery.Identity.PID, recovery.Identity.AppPath) + } + window, err := standaloneRecoveryTarget(recoveryWindows(appAX), recovery) + if err == nil { + key := standaloneRecoveryWindowKey(window) + if key == lastKey { + stable++ + } else { + lastKey = key + stable = 1 + lastErr = fmt.Errorf("untitled Performance window identity is not yet stable") + } + if stable >= 2 { + return window.Element, nil + } + } else { + lastKey = "" + stable = 0 + lastErr = err + } + if time.Now().After(deadline) { + return 0, fmt.Errorf("untitled recovery target not established: %w", lastErr) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return 0, err + } + } +} + +func waitForStandaloneFinalizeWindow(ctx context.Context, appAX uintptr, recovery standaloneExportRecovery, timeout time.Duration) (uintptr, error) { + deadline := time.Now().Add(timeout) + var lastKey string + stable := 0 + var lastErr error + for { + if err := requireRecoveryIdentity(appAX, recovery); err != nil { + return 0, err + } + windows := recoveryWindows(appAX) + window, err := standaloneRecoveryTarget(windows, recovery) + if err != nil { + window, err = restoredRecoverySourceAnyGeometry(windows, recovery) + } + if err != nil { + window, err = summaryRecoveryTarget(windows, recovery, "") + } + if err == nil { + key := standaloneRecoveryWindowKey(window) + if key == lastKey { + stable++ + } else { + lastKey = key + stable = 1 + } + if stable >= 2 { + return window.Element, nil + } + } else { + lastKey = "" + stable = 0 + lastErr = err + } + if time.Now().After(deadline) { + return 0, fmt.Errorf("finalize recovery target not established: %w", lastErr) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return 0, err + } + } +} + +func finalizeStandaloneExport(w io.Writer, targetPath, outputPath string) (tracebundle.Payload, error) { + if err := requireStandaloneExportTarget(targetPath); err != nil { + return tracebundle.Payload{}, err + } + if err := verifyExportTraceIdentity(targetPath, outputPath); err != nil { + return tracebundle.Payload{}, err + } + payload, err := tracebundle.InspectPayload(outputPath) + if err != nil { + return tracebundle.Payload{}, fmt.Errorf("inspect exported trace payload: %w", err) + } + writeXcodePayloadStatus(w, payload) + if err := requireSelfContainedExport(outputPath, payload); err != nil { + return payload, err + } + return payload, nil +} + +func requireStandaloneExportTarget(targetPath string) error { + if targetPath != "" { + return nil + } + return fmt.Errorf( + "cannot verify standalone export identity: selected Xcode window has no AXDocument binding; use a combined xp run or explicitly bind the source trace before export", + ) +} + +func existingExportCandidates(candidates []string, requested string) []string { + var found []string + requested = filepath.Clean(requested) + for _, candidate := range uniquePaths(candidates) { + if filepath.Clean(candidate) == requested { + continue + } + if _, err := os.Stat(candidate); err == nil { + found = append(found, candidate) + } + } + return found +} + +func requireExportedTrace(path string) error { + if _, err := os.Stat(path); err != nil { + return fmt.Errorf("export completed but output not found at expected location %s: %w", path, err) + } + return nil +} + // isExportDialogOpen checks if an export/save dialog is already open on the window. func isExportDialogOpen(window uintptr) bool { saveBtn := findButtonBFS(window, "Save", 500) // Export sheet is shallow @@ -142,7 +1415,7 @@ func runOpenExport(cmd *cobra.Command, args []string) error { } else { // Fall back to menu fmt.Fprintln(status, " Using File > Export menu...") - if err := ClickMenuItem(appAX, []string{"File", "Export..."}); err != nil { + if err := clickMenuItemForWindow(appAX, windowAX, []string{"File", "Export..."}); err != nil { return fmt.Errorf("failed to click Export menu: %w", err) } } diff --git a/cmd/gputrace/cmd/collect_xcode_profile_export_backoff_test.go b/cmd/gputrace/cmd/collect_xcode_profile_export_backoff_test.go new file mode 100644 index 00000000..0f6846db --- /dev/null +++ b/cmd/gputrace/cmd/collect_xcode_profile_export_backoff_test.go @@ -0,0 +1,56 @@ +//go:build darwin + +package cmd + +import ( + "testing" + "time" +) + +// TestFileExportProbeBudget bounds how many times the automation opens Xcode's +// File menu while waiting for Export to become enabled. +// +// Every probe is a UI action, not an observation: the menu bar opens and takes +// key focus. At the previous fixed 500ms interval a two-minute wait flashed the +// menu about 240 times and made the machine unusable. This test fails if that +// budget creeps back up. +func TestFileExportProbeBudget(t *testing.T) { + const window = 2 * time.Minute + + var elapsed time.Duration + probes := 1 // the first probe happens before any delay + for probes < maxFileExportProbes { + elapsed += fileExportProbeDelay(probes) + if elapsed > window { + break + } + probes++ + } + if probes > 20 { + t.Errorf("%v of waiting costs %d File-menu opens, want <= 20", window, probes) + } + if elapsed < window { + t.Errorf("the probe cap is reached after only %v; it must not cut a %v wait short", elapsed, window) + } +} + +// TestFileExportProbeDelayBackoff checks the schedule grows and then holds, so +// a long wait neither hammers the menu bar nor stops probing altogether. +func TestFileExportProbeDelayBackoff(t *testing.T) { + for _, test := range []struct { + attempt int + want time.Duration + }{ + {1, 500 * time.Millisecond}, + {2, time.Second}, + {3, 2 * time.Second}, + {4, 4 * time.Second}, + {5, 8 * time.Second}, + {6, 8 * time.Second}, + {50, 8 * time.Second}, + } { + if got := fileExportProbeDelay(test.attempt); got != test.want { + t.Errorf("fileExportProbeDelay(%d) = %v, want %v", test.attempt, got, test.want) + } + } +} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_export_counters.go b/cmd/gputrace/cmd/collect_xcode_profile_export_counters.go index fc0e81f7..9a73382e 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_export_counters.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_export_counters.go @@ -50,6 +50,10 @@ func runXcodeExportCounters(cmd *cobra.Command, args []string, opts *xcodeExport if err != nil { return fmt.Errorf("could not find trace window: %w", err) } + selection := selectionForWindow(traceFile, windowAX) + if err := requireBoundSelection(selection); err != nil { + return err + } // Raise the window axAction(windowAX, "AXRaise") @@ -128,6 +132,7 @@ func runXcodeExportCounters(cmd *cobra.Command, args []string, opts *xcodeExport // Find and click Save button with retries (AX references can go stale) var clickErr error + var expectedPath string for attempt := 0; attempt < 3; attempt++ { if attempt > 0 { time.Sleep(200 * time.Millisecond) @@ -137,6 +142,11 @@ func runXcodeExportCounters(cmd *cobra.Command, args []string, opts *xcodeExport for _, w := range windows { // Try Save button (export sheet is shallow) if btn := findButtonBFS(w, "Save", 500); btn != 0 { + sheetState := readExportSheetState(w) + expectedPath, err = exportPathFromSheetState(sheetState) + if err != nil { + return fmt.Errorf("cannot verify GPU counter export destination before Save: %w; sheet state: %s", err, formatExportSheetState(sheetState)) + } fmt.Fprintln(status, "Clicking Save...") if err := axAction(btn, "AXPress"); err != nil { clickErr = err @@ -148,6 +158,11 @@ func runXcodeExportCounters(cmd *cobra.Command, args []string, opts *xcodeExport } // Try Export button (export sheet is shallow) if btn := findButtonBFS(w, "Export", 500); btn != 0 { + sheetState := readExportSheetState(w) + expectedPath, err = exportPathFromSheetState(sheetState) + if err != nil { + return fmt.Errorf("cannot verify GPU counter export destination before Export: %w; sheet state: %s", err, formatExportSheetState(sheetState)) + } fmt.Fprintln(status, "Clicking Export...") if err := axAction(btn, "AXPress"); err != nil { clickErr = err @@ -190,11 +205,20 @@ saveClicked: } } - time.Sleep(500 * time.Millisecond) - fmt.Fprintln(status, "Export complete") + if err := waitForExportFile(cmd.Context(), expectedPath, 10*time.Second); err != nil { + return fmt.Errorf("verify GPU counter export: %w", err) + } + fmt.Fprintf(status, "GPU counter export verified: %s\n", expectedPath) return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "xcode-export-counters", - Target: traceFile, + Action: "xcode-export-counters", + Target: traceFile, + Output: expectedPath, + RequestedTrace: traceFile, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "export verified", + Evidence: "saved file exists, is non-empty, and stabilized", + TargetBound: boolPointer(selection.Bound), }) } diff --git a/cmd/gputrace/cmd/collect_xcode_profile_export_memory.go b/cmd/gputrace/cmd/collect_xcode_profile_export_memory.go index 777091b8..7c0df9f3 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_export_memory.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_export_memory.go @@ -46,6 +46,10 @@ func runXcodeExportMemory(cmd *cobra.Command, args []string, opts *xcodeExportMe if err != nil { return fmt.Errorf("could not find trace window: %w", err) } + selection := selectionForWindow(traceFile, windowAX) + if err := requireBoundSelection(selection); err != nil { + return err + } // Raise the window axAction(windowAX, "AXRaise") @@ -119,6 +123,7 @@ func runXcodeExportMemory(cmd *cobra.Command, args []string, opts *xcodeExportMe // Find and click Save button with retries (AX references can go stale) var clickErr error + var expectedPath string for attempt := 0; attempt < 3; attempt++ { if attempt > 0 { time.Sleep(200 * time.Millisecond) @@ -128,6 +133,11 @@ func runXcodeExportMemory(cmd *cobra.Command, args []string, opts *xcodeExportMe for _, w := range windows { // Try Save button (export sheet is shallow) if btn := findButtonBFS(w, "Save", 500); btn != 0 { + sheetState := readExportSheetState(w) + expectedPath, err = exportPathFromSheetState(sheetState) + if err != nil { + return fmt.Errorf("cannot verify memory export destination before Save: %w; sheet state: %s", err, formatExportSheetState(sheetState)) + } fmt.Fprintln(status, "Clicking Save...") if err := axAction(btn, "AXPress"); err != nil { clickErr = err @@ -139,6 +149,11 @@ func runXcodeExportMemory(cmd *cobra.Command, args []string, opts *xcodeExportMe } // Try Export button (export sheet is shallow) if btn := findButtonBFS(w, "Export", 500); btn != 0 { + sheetState := readExportSheetState(w) + expectedPath, err = exportPathFromSheetState(sheetState) + if err != nil { + return fmt.Errorf("cannot verify memory export destination before Export: %w; sheet state: %s", err, formatExportSheetState(sheetState)) + } fmt.Fprintln(status, "Clicking Export...") if err := axAction(btn, "AXPress"); err != nil { clickErr = err @@ -181,10 +196,19 @@ saveClicked: } } - time.Sleep(500 * time.Millisecond) - fmt.Fprintln(status, "Export complete") + if err := waitForExportFile(cmd.Context(), expectedPath, 10*time.Second); err != nil { + return fmt.Errorf("verify memory export: %w", err) + } + fmt.Fprintf(status, "Memory export verified: %s\n", expectedPath) return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "xcode-export-memory", - Target: traceFile, + Action: "xcode-export-memory", + Target: traceFile, + Output: expectedPath, + RequestedTrace: traceFile, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "export verified", + Evidence: "saved file exists, is non-empty, and stabilized", + TargetBound: boolPointer(selection.Bound), }) } diff --git a/cmd/gputrace/cmd/collect_xcode_profile_export_test.go b/cmd/gputrace/cmd/collect_xcode_profile_export_test.go new file mode 100644 index 00000000..dfb3328c --- /dev/null +++ b/cmd/gputrace/cmd/collect_xcode_profile_export_test.go @@ -0,0 +1,830 @@ +//go:build darwin + +package cmd + +import ( + "bytes" + "context" + "encoding/json" + "errors" + "fmt" + "os" + "path/filepath" + "strings" + "testing" + "time" + + "github.com/spf13/cobra" +) + +func writeStandaloneExportFixture(t *testing.T, name, uuid string, full bool) string { + t.Helper() + bundle := filepath.Join(t.TempDir(), name+".gputrace") + profilerDir := filepath.Join(bundle, name+".gputrace.gpuprofiler_raw") + if err := os.MkdirAll(profilerDir, 0o755); err != nil { + t.Fatal(err) + } + metadata := ` +(uuid)` + uuid + `` + files := map[string]string{ + "metadata": metadata, + filepath.Join(filepath.Base(profilerDir), "streamData"): "profiler", + } + if full { + files["capture"] = "capture" + files["MTLBuffer-1-0"] = "raw resource" + } + for path, data := range files { + if err := os.WriteFile(filepath.Join(bundle, path), []byte(data), 0o644); err != nil { + t.Fatal(err) + } + } + return bundle +} + +func TestFinalizeStandaloneExportRequiresBoundIdentity(t *testing.T) { + output := writeStandaloneExportFixture(t, "output", "same", true) + var status bytes.Buffer + _, err := finalizeStandaloneExport(&status, "", output) + if err == nil || !strings.Contains(err.Error(), "no AXDocument binding") { + t.Fatalf("error = %v, want unbound identity error", err) + } + if strings.Contains(status.String(), "Exported to:") { + t.Fatalf("unbound export printed success:\n%s", status.String()) + } +} + +func TestFinalizeStandaloneExportRejectsAndPreservesProfilerOnly(t *testing.T) { + input := writeStandaloneExportFixture(t, "input", "same", true) + output := writeStandaloneExportFixture(t, "output", "same", false) + var status bytes.Buffer + payload, err := finalizeStandaloneExport(&status, input, output) + if err == nil || !strings.Contains(err.Error(), "not self-contained") { + t.Fatalf("error = %v, want self-contained rejection", err) + } + if payload.Class != "profiler-only" || !payload.HasProfilerStream { + t.Fatalf("payload = %+v, want usable profiler-only", payload) + } + if !strings.Contains(status.String(), "profiler-only (not self-contained)") { + t.Fatalf("status missing payload classification:\n%s", status.String()) + } + if strings.Contains(status.String(), "Exported to:") { + t.Fatalf("rejected export printed success:\n%s", status.String()) + } + if _, err := os.Stat(output); err != nil { + t.Fatalf("rejected profiler-only output was not preserved: %v", err) + } +} + +func TestFinalizeStandaloneExportAcceptsFullPayloadFields(t *testing.T) { + input := writeStandaloneExportFixture(t, "input", "same", true) + output := writeStandaloneExportFixture(t, "output", "same", true) + var status bytes.Buffer + payload, err := finalizeStandaloneExport(&status, input, output) + if err != nil { + t.Fatalf("finalizeStandaloneExport: %v", err) + } + if !strings.Contains(status.String(), "full and self-contained") { + t.Fatalf("status missing full classification:\n%s", status.String()) + } + + action := xcodeProfileActionOutput{Action: "export", Target: input, Output: output} + applyXcodePayload(&action, payload) + if action.PayloadClass != "full" || + action.SelfContained == nil || !*action.SelfContained || + action.ProfilerTimingAvailable == nil || !*action.ProfilerTimingAvailable || + action.StructuralAnalysisAvailable == nil || !*action.StructuralAnalysisAvailable { + t.Fatalf("payload action fields = %+v", action) + } + data, err := json.Marshal(action) + if err != nil { + t.Fatal(err) + } + for _, field := range []string{ + `"payload_class":"full"`, + `"self_contained":true`, + `"profiler_timing_available":true`, + `"structural_analysis_available":true`, + } { + if !bytes.Contains(data, []byte(field)) { + t.Fatalf("action JSON missing %s: %s", field, data) + } + } +} + +func TestStandaloneExportTargetRequiresUniqueDocumentBinding(t *testing.T) { + trace := "/Users/tmc/tmp/trace.gputrace" + window, doc, err := standaloneExportTarget([]xcodeAXWindow{ + {Element: 1, Title: "Source", Document: "/Users/tmc/project/main.swift"}, + {Element: 2, Title: "Performance", Document: trace}, + }) + if err != nil { + t.Fatal(err) + } + if window != 2 || doc != trace { + t.Fatalf("target = (%d, %q)", window, doc) + } + + _, _, err = standaloneExportTarget([]xcodeAXWindow{ + {Element: 2, Document: trace}, + {Element: 3, Document: "/Users/tmc/tmp/other.gputrace"}, + }) + if err == nil || !strings.Contains(err.Error(), "multiple .gputrace windows") { + t.Fatalf("ambiguous target error = %v", err) + } +} + +func TestStandaloneExportRecoveryFlagsRequireCompleteIdentity(t *testing.T) { + tests := []struct { + name string + args []string + }{ + {name: "mode only", args: []string{"--recover-untitled"}}, + {name: "check only", args: []string{"--check-recovery"}}, + {name: "finalize only", args: []string{"--finalize-workload"}}, + {name: "missing app", args: []string{"--recover-untitled", "--source", "/trace.gputrace", "--xcode-pid", "81051"}}, + {name: "missing pid", args: []string{"--recover-untitled", "--source", "/trace.gputrace", "--xcode-app", "/Applications/Xcode.app"}}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + cmd := &cobra.Command{} + standaloneExportFlags(cmd) + if err := cmd.ParseFlags(test.args); err != nil { + t.Fatal(err) + } + _, err := standaloneExportRecoveryFromFlags(cmd) + if err == nil || !strings.Contains(err.Error(), "requires --recover-untitled") { + t.Fatalf("error = %v, want incomplete recovery flags", err) + } + }) + } +} + +func TestStandaloneExportSourceOnlyDeclaresIdentity(t *testing.T) { + previous := declaredExportSource + t.Cleanup(func() { declaredExportSource = previous }) + declaredExportSource = "" + + cmd := &cobra.Command{} + standaloneExportFlags(cmd) + if err := cmd.ParseFlags([]string{"--source", "/trace.gputrace"}); err != nil { + t.Fatal(err) + } + recovery, err := standaloneExportRecoveryFromFlags(cmd) + if err != nil { + t.Fatalf("source alone should declare identity, not start recovery: %v", err) + } + if recovery.Enabled { + t.Errorf("recovery.Enabled = true, want false") + } + if declaredExportSource != "/trace.gputrace" { + t.Errorf("declaredExportSource = %q, want %q", declaredExportSource, "/trace.gputrace") + } +} + +func TestStandaloneExportRecoveryFlagsRejectCheckAndFinalize(t *testing.T) { + cmd := &cobra.Command{} + standaloneExportFlags(cmd) + err := cmd.ParseFlags([]string{ + "--recover-untitled", + "--check-recovery", + "--finalize-workload", + "--source", "/trace.gputrace", + "--xcode-pid", "81051", + "--xcode-app", "/Applications/Xcode.app", + }) + if err != nil { + t.Fatal(err) + } + _, err = standaloneExportRecoveryFromFlags(cmd) + if err == nil || !strings.Contains(err.Error(), "mutually exclusive") { + t.Fatalf("error = %v, want mutually exclusive flags", err) + } +} + +func TestStandaloneExportRecoveryFlagsPreserveDeadPIDForSentinel(t *testing.T) { + source := writeStandaloneExportFixture(t, "source", "SOURCE-UUID", true) + cmd := &cobra.Command{} + standaloneExportFlags(cmd) + err := cmd.ParseFlags([]string{ + "--recover-untitled", + "--source", source, + "--xcode-pid", "987654", + "--xcode-app", "/Applications/Xcode.app", + }) + if err != nil { + t.Fatal(err) + } + recovery, err := standaloneExportRecoveryFromFlags(cmd) + if err != nil { + t.Fatalf("parse recovery flags: %v", err) + } + if recovery.Identity.PID != 987654 || recovery.Identity.AppPath != "/Applications/Xcode.app" { + t.Fatalf("identity = %+v", recovery.Identity) + } +} + +func TestStandaloneRecoveryTarget(t *testing.T) { + recovery := standaloneExportRecovery{ + Enabled: true, + SourcePath: "/Users/tmc/tmp/raw.gputrace", + SourceUUID: "RAW-UUID", + Identity: xcodeProcessIdentity{ + PID: 81051, + AppPath: "/Applications/Xcode.app", + }, + } + eligible := standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{Element: 11}, + PID: 81051, + PerformanceView: true, + } + tests := []struct { + name string + windows []standaloneRecoveryWindow + want uintptr + wantErr string + }{ + {name: "unique", windows: []standaloneRecoveryWindow{eligible}, want: 11}, + {name: "duplicate AX representation", windows: []standaloneRecoveryWindow{eligible, eligible}, want: 11}, + {name: "none", wantErr: "no untitled Performance window"}, + { + name: "wrong pid", + windows: []standaloneRecoveryWindow{{ + xcodeAXWindow: xcodeAXWindow{Element: 12}, + PID: 74001, + PerformanceView: true, + }}, + wantErr: "no untitled Performance window", + }, + { + name: "document bound", + windows: []standaloneRecoveryWindow{{ + xcodeAXWindow: xcodeAXWindow{Element: 13, Document: "/Users/tmc/tmp/other.gputrace"}, + PID: 81051, + PerformanceView: true, + }}, + wantErr: "no untitled Performance window", + }, + { + name: "titled", + windows: []standaloneRecoveryWindow{{ + xcodeAXWindow: xcodeAXWindow{Element: 14, Title: "Other"}, + PID: 81051, + PerformanceView: true, + }}, + wantErr: "no untitled Performance window", + }, + { + name: "no performance evidence", + windows: []standaloneRecoveryWindow{{ + xcodeAXWindow: xcodeAXWindow{Element: 15}, + PID: 81051, + }}, + wantErr: "no untitled Performance window", + }, + { + name: "ambiguous", + windows: []standaloneRecoveryWindow{ + eligible, + { + xcodeAXWindow: xcodeAXWindow{Element: 16}, + PID: 81051, + PerformanceView: true, + }, + }, + wantErr: "multiple untitled Performance windows", + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + got, err := standaloneRecoveryTarget(test.windows, recovery) + if test.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), test.wantErr) { + t.Fatalf("error = %v, want %q", err, test.wantErr) + } + return + } + if err != nil { + t.Fatal(err) + } + if got.Element != test.want { + t.Fatalf("window = %d, want %d", got.Element, test.want) + } + }) + } +} + +func TestValidateStandaloneRecoveryIdentity(t *testing.T) { + identity := xcodeProcessIdentity{PID: 81051, AppPath: "/Applications/Xcode.app"} + if err := validateStandaloneRecoveryIdentity(identity, "/Applications/Xcode.app"); err != nil { + t.Fatal(err) + } + err := validateStandaloneRecoveryIdentity(identity, "/Applications/Xcode-rc.app") + if err == nil || !strings.Contains(err.Error(), "not requested app") { + t.Fatalf("error = %v, want cross-app rejection", err) + } + for _, actual := range []string{"", " \t"} { + err := validateStandaloneRecoveryIdentity(identity, actual) + if err == nil || !strings.Contains(err.Error(), "PID 81051 is not running") { + t.Fatalf("validate dead PID with %q: %v", actual, err) + } + if strings.Contains(err.Error(), "runs from .") { + t.Fatalf("dead PID rendered cleaned empty path: %v", err) + } + } +} + +func TestNormalizeStandaloneRecoveryFailureReportsExitWithoutIPS(t *testing.T) { + recovery := standaloneExportRecovery{ + Identity: xcodeProcessIdentity{PID: 987654, AppPath: "/Applications/Xcode.app"}, + } + scope := newXcodeCrashScope(recovery.Identity.AppPath, time.Now()) + scope.allowRebind = false + scope.bind(recovery.Identity) + original := fmt.Errorf("reacquire recovery Xcode: process not running") + err := normalizeStandaloneRecoveryFailureWithGrace( + context.Background(), scope, recovery, original, time.Millisecond, + ) + if err == nil || !strings.Contains(err.Error(), "PID 987654 exited") || + !strings.Contains(err.Error(), "no matching DiagnosticReport") { + t.Fatalf("error = %v, want explicit exit without report", err) + } + if !errors.Is(err, original) { + t.Fatalf("error does not wrap original: %v", err) + } +} + +func TestNormalizeStandaloneRecoveryFailureReturnsCrashReport(t *testing.T) { + recovery := standaloneExportRecovery{ + Identity: xcodeProcessIdentity{PID: 987654, AppPath: "/Applications/Xcode.app"}, + } + scope := newXcodeCrashScope(recovery.Identity.AppPath, time.Now()) + scope.allowRebind = false + scope.bind(recovery.Identity) + report := xcodeCrashReport{ + Path: "/Users/tmc/Library/Logs/DiagnosticReports/Xcode.ips", + PID: recovery.Identity.PID, + AppPath: recovery.Identity.AppPath, + Exception: "EXC_BAD_ACCESS", + Signal: "SIGBUS", + } + ctx, cancel := context.WithCancelCause(context.Background()) + cancel(report) + err := normalizeStandaloneRecoveryFailureWithGrace( + ctx, scope, recovery, fmt.Errorf("window disappeared"), time.Second, + ) + var got xcodeCrashReport + if !errors.As(err, &got) || got.Path != report.Path { + t.Fatalf("error = %T %v, want crash report", err, err) + } +} + +func TestFindElementAtDepth(t *testing.T) { + tests := []struct { + name string + tree map[uintptr][]uintptr + pruned map[uintptr]bool + target uintptr + maxDepth int + want uintptr + }{ + { + name: "root", + target: 1, + maxDepth: 4, + want: 1, + }, + { + name: "depth four", + tree: map[uintptr][]uintptr{ + 1: {2}, + 2: {3}, + 3: {4}, + 4: {5}, + }, + target: 5, + maxDepth: 4, + want: 5, + }, + { + name: "reject depth five", + tree: map[uintptr][]uintptr{ + 1: {2}, + 2: {3}, + 3: {4}, + 4: {5}, + 5: {6}, + }, + target: 6, + maxDepth: 4, + }, + { + name: "prune outline", + tree: map[uintptr][]uintptr{ + 1: {2}, + 2: {3}, + }, + pruned: map[uintptr]bool{2: true}, + target: 3, + maxDepth: 4, + }, + { + name: "cycle", + tree: map[uintptr][]uintptr{ + 1: {2}, + 2: {1, 3}, + }, + target: 3, + maxDepth: 4, + want: 3, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + childCalls := make(map[uintptr]int) + got := findElementAtDepth( + 1, + test.maxDepth, + 32, + func(element uintptr) []uintptr { + childCalls[element]++ + return test.tree[element] + }, + func(element uintptr) bool { + return test.pruned[element] + }, + func(element uintptr) bool { + return element == test.target + }, + ) + if got != test.want { + t.Fatalf("element = %d, want %d", got, test.want) + } + for element := range test.pruned { + if childCalls[element] != 0 { + t.Fatalf("children called for pruned element %d", element) + } + } + }) + } +} + +func TestStandaloneRecoveryWindowKeyIgnoresAXHandle(t *testing.T) { + left := standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{Element: 11, X: 229, Y: 320, Width: 1376, Height: 900}, + PID: 81051, + } + right := left + right.Element = 22 + if standaloneRecoveryWindowKey(left) != standaloneRecoveryWindowKey(right) { + t.Fatal("logical window key depends on transient AX element handle") + } +} + +func TestValidateRecoveryFinalizePrecondition(t *testing.T) { + recovery := standaloneExportRecovery{ + Identity: xcodeProcessIdentity{PID: 81051, AppPath: "/Applications/Xcode.app"}, + } + const key = "window" + valid := recoveryFinalizeSnapshot{ + Identity: recovery.Identity, + WindowKey: key, + Performance: true, + StopCount: 1, + StopEnabled: true, + StopElement: 7, + ExportFound: true, + ExportEnabled: false, + } + tests := []struct { + name string + edit func(*recoveryFinalizeSnapshot) + want string + }{ + {name: "valid"}, + {name: "wrong pid", edit: func(s *recoveryFinalizeSnapshot) { s.Identity.PID++ }, want: "identity mismatch"}, + {name: "wrong app", edit: func(s *recoveryFinalizeSnapshot) { s.Identity.AppPath = "/Applications/Xcode-rc.app" }, want: "identity mismatch"}, + {name: "window changed", edit: func(s *recoveryFinalizeSnapshot) { s.WindowKey = "other" }, want: "window identity changed"}, + {name: "performance missing", edit: func(s *recoveryFinalizeSnapshot) { s.Performance = false }, want: "Performance group"}, + {name: "sheet open", edit: func(s *recoveryFinalizeSnapshot) { s.SheetOpen = true }, want: "open sheet"}, + {name: "stop absent", edit: func(s *recoveryFinalizeSnapshot) { s.StopCount = 0 }, want: "exactly one enabled"}, + {name: "stop duplicate", edit: func(s *recoveryFinalizeSnapshot) { s.StopCount = 2 }, want: "exactly one enabled"}, + {name: "stop disabled", edit: func(s *recoveryFinalizeSnapshot) { s.StopEnabled = false }, want: "exactly one enabled"}, + {name: "export missing", edit: func(s *recoveryFinalizeSnapshot) { s.ExportFound = false }, want: "could not find"}, + {name: "already ready", edit: func(s *recoveryFinalizeSnapshot) { s.ExportEnabled = true }, want: "already export-ready"}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + snapshot := valid + if test.edit != nil { + test.edit(&snapshot) + } + err := validateRecoveryFinalizePrecondition(snapshot, recovery, key) + if test.want == "" { + if err != nil { + t.Fatal(err) + } + return + } + if err == nil || !strings.Contains(err.Error(), test.want) { + t.Fatalf("error = %v, want %q", err, test.want) + } + }) + } +} + +func TestRestoredRecoverySourceTarget(t *testing.T) { + recovery := standaloneExportRecovery{ + SourcePath: "/Users/tmc/tmp/raw.gputrace", + Identity: xcodeProcessIdentity{PID: 81051, AppPath: "/Applications/Xcode.app"}, + } + base := standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{ + Element: 21, + Title: "raw.gputrace", + Document: "file:///Users/tmc/tmp/raw.gputrace", + X: 229, + Y: 320, + Width: 1376, + Height: 900, + }, + PID: 81051, + NewEditorView: true, + Finished: true, + } + key := standaloneRecoveryGeometryKey(base) + tests := []struct { + name string + edit func(*standaloneRecoveryWindow) + wantErr string + }{ + {name: "exact restored source"}, + {name: "wrong pid", edit: func(w *standaloneRecoveryWindow) { w.PID++ }, wantErr: "found 0"}, + {name: "window drift", edit: func(w *standaloneRecoveryWindow) { w.X++ }, wantErr: "found 0"}, + {name: "wrong document", edit: func(w *standaloneRecoveryWindow) { w.Document = "/Users/tmc/tmp/other.gputrace" }, wantErr: "found 0"}, + {name: "empty document", edit: func(w *standaloneRecoveryWindow) { w.Document = "" }, wantErr: "found 0"}, + {name: "wrong title", edit: func(w *standaloneRecoveryWindow) { w.Title = "other.gputrace" }, wantErr: "found 0"}, + {name: "new editor missing", edit: func(w *standaloneRecoveryWindow) { w.NewEditorView = false }, wantErr: "found 0"}, + {name: "finished missing", edit: func(w *standaloneRecoveryWindow) { w.Finished = false }, wantErr: "found 0"}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + window := base + if test.edit != nil { + test.edit(&window) + } + got, err := restoredRecoverySourceTarget([]standaloneRecoveryWindow{window}, recovery, key) + if test.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), test.wantErr) { + t.Fatalf("error = %v, want %q", err, test.wantErr) + } + return + } + if err != nil { + t.Fatal(err) + } + if got.Element != base.Element { + t.Fatalf("element = %d, want %d", got.Element, base.Element) + } + }) + } +} + +func TestSummaryRecoveryTarget(t *testing.T) { + recovery := standaloneExportRecovery{ + SourcePath: "/Users/tmc/tmp/raw.gputrace", + Identity: xcodeProcessIdentity{PID: 13556, AppPath: "/Applications/Xcode.app"}, + } + base := standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{ + Element: 31, + X: 0, + Y: 100, + Width: 1376, + Height: 900, + }, + PID: 13556, + SummaryView: true, + Debugging: true, + Progress95: true, + StopCount: 1, + StopEnabled: true, + ShowCount: 1, + ShowEnabled: true, + } + key := standaloneRecoveryGeometryKey(base) + tests := []struct { + name string + edit func(*standaloneRecoveryWindow) + wantErr string + }{ + {name: "exact summary"}, + {name: "wrong pid", edit: func(w *standaloneRecoveryWindow) { w.PID++ }, wantErr: "found 0"}, + {name: "wrong geometry", edit: func(w *standaloneRecoveryWindow) { w.X++ }, wantErr: "found 0"}, + {name: "titled", edit: func(w *standaloneRecoveryWindow) { w.Title = "other.gputrace" }, wantErr: "found 0"}, + {name: "document bound", edit: func(w *standaloneRecoveryWindow) { w.Document = recovery.SourcePath }, wantErr: "found 0"}, + {name: "summary missing", edit: func(w *standaloneRecoveryWindow) { w.SummaryView = false }, wantErr: "found 0"}, + {name: "debugging missing", edit: func(w *standaloneRecoveryWindow) { w.Debugging = false }, wantErr: "found 0"}, + {name: "wrong progress", edit: func(w *standaloneRecoveryWindow) { w.Progress95 = false }, wantErr: "found 0"}, + {name: "sheet", edit: func(w *standaloneRecoveryWindow) { w.SheetOpen = true }, wantErr: "found 0"}, + {name: "stop absent", edit: func(w *standaloneRecoveryWindow) { w.StopCount = 0 }, wantErr: "found 0"}, + {name: "stop disabled", edit: func(w *standaloneRecoveryWindow) { w.StopEnabled = false }, wantErr: "found 0"}, + {name: "AX show absent uses OCR", edit: func(w *standaloneRecoveryWindow) { w.ShowCount = 0; w.ShowEnabled = false }}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + window := base + if test.edit != nil { + test.edit(&window) + } + got, err := summaryRecoveryTarget([]standaloneRecoveryWindow{window}, recovery, key) + if test.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), test.wantErr) { + t.Fatalf("error = %v, want %q", err, test.wantErr) + } + return + } + if err != nil { + t.Fatal(err) + } + if got.Element != base.Element { + t.Fatalf("element = %d, want %d", got.Element, base.Element) + } + }) + } + + duplicate := base + duplicate.Element = 32 + if _, err := summaryRecoveryTarget([]standaloneRecoveryWindow{base, duplicate}, recovery, key); err != nil { + t.Fatalf("duplicate AX representation: %v", err) + } + other := base + other.Element = 33 + other.X++ + if _, err := summaryRecoveryTarget([]standaloneRecoveryWindow{base, other}, recovery, ""); err == nil { + t.Fatal("distinct Summary windows were not rejected") + } +} + +func TestRunningRecoveryPerformanceTarget(t *testing.T) { + recovery := standaloneExportRecovery{ + SourcePath: "/Users/tmc/tmp/raw.gputrace", + Identity: xcodeProcessIdentity{PID: 13556, AppPath: "/Applications/Xcode.app"}, + } + base := standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{Element: 41, X: 0, Y: 100, Width: 1376, Height: 900}, + PID: 13556, + PerformanceView: true, + StopCount: 1, + StopEnabled: true, + } + key := standaloneRecoveryGeometryKey(base) + if _, err := runningRecoveryPerformanceTarget([]standaloneRecoveryWindow{base}, recovery, key); err != nil { + t.Fatal(err) + } + for _, edit := range []func(*standaloneRecoveryWindow){ + func(w *standaloneRecoveryWindow) { w.PID++ }, + func(w *standaloneRecoveryWindow) { w.Width++ }, + func(w *standaloneRecoveryWindow) { w.PerformanceView = false }, + func(w *standaloneRecoveryWindow) { w.SheetOpen = true }, + func(w *standaloneRecoveryWindow) { w.StopCount = 0 }, + func(w *standaloneRecoveryWindow) { w.StopEnabled = false }, + func(w *standaloneRecoveryWindow) { w.Document = "/Users/tmc/tmp/other.gputrace" }, + } { + window := base + edit(&window) + if _, err := runningRecoveryPerformanceTarget([]standaloneRecoveryWindow{window}, recovery, key); err == nil { + t.Fatalf("invalid running Performance accepted: %+v", window) + } + } +} + +func TestFinalizedRecoveryPerformanceTarget(t *testing.T) { + recovery := standaloneExportRecovery{ + SourcePath: "/Users/tmc/tmp/raw.gputrace", + Identity: xcodeProcessIdentity{PID: 81051, AppPath: "/Applications/Xcode.app"}, + } + base := standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{ + Element: 22, + X: 229, + Y: 320, + Width: 1376, + Height: 900, + }, + PID: 81051, + PerformanceView: true, + } + key := standaloneRecoveryGeometryKey(base) + tests := []struct { + name string + edit func(*standaloneRecoveryWindow) + wantErr string + }{ + {name: "untitled transitioned performance"}, + {name: "source-bound transitioned performance", edit: func(w *standaloneRecoveryWindow) { + w.Title = "raw.gputrace" + w.Document = "file:///Users/tmc/tmp/raw.gputrace" + }}, + {name: "wrong pid", edit: func(w *standaloneRecoveryWindow) { w.PID++ }, wantErr: "found 0"}, + {name: "window drift", edit: func(w *standaloneRecoveryWindow) { w.Width++ }, wantErr: "found 0"}, + {name: "performance missing", edit: func(w *standaloneRecoveryWindow) { w.PerformanceView = false }, wantErr: "found 0"}, + {name: "wrong document", edit: func(w *standaloneRecoveryWindow) { w.Document = "/Users/tmc/tmp/other.gputrace" }, wantErr: "found 0"}, + {name: "wrong title", edit: func(w *standaloneRecoveryWindow) { w.Title = "other.gputrace" }, wantErr: "found 0"}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + window := base + if test.edit != nil { + test.edit(&window) + } + got, err := finalizedRecoveryPerformanceTarget([]standaloneRecoveryWindow{window}, recovery, key) + if test.wantErr != "" { + if err == nil || !strings.Contains(err.Error(), test.wantErr) { + t.Fatalf("error = %v, want %q", err, test.wantErr) + } + return + } + if err != nil { + t.Fatal(err) + } + if got.Element != base.Element { + t.Fatalf("element = %d, want %d", got.Element, base.Element) + } + }) + } +} + +func TestNormalizedTraceDocument(t *testing.T) { + const want = "/Users/tmc/tmp/raw trace.gputrace" + for _, document := range []string{ + want, + "file:///Users/tmc/tmp/raw%20trace.gputrace", + } { + if got := normalizedTraceDocument(document); got != want { + t.Fatalf("normalizedTraceDocument(%q) = %q, want %q", document, got, want) + } + } +} + +func TestTraceDocumentMatchesFilesystemIdentity(t *testing.T) { + root := t.TempDir() + real := filepath.Join(root, "real", "raw trace.gputrace") + if err := os.MkdirAll(real, 0o755); err != nil { + t.Fatal(err) + } + aliasRoot := filepath.Join(root, "alias") + if err := os.Symlink(filepath.Join(root, "real"), aliasRoot); err != nil { + t.Fatal(err) + } + alias := filepath.Join(aliasRoot, "raw trace.gputrace") + fileURL := "file://" + strings.ReplaceAll(real, " ", "%20") + for _, test := range []struct { + name string + document string + source string + want bool + }{ + {name: "alias to real", document: alias, source: real, want: true}, + {name: "real to alias", document: real, source: alias, want: true}, + {name: "escaped file URL", document: fileURL, source: alias, want: true}, + {name: "empty", document: "", source: real}, + {name: "non-file URL", document: "https://example.com/raw.gputrace", source: real}, + {name: "missing unequal", document: filepath.Join(root, "a", "raw.gputrace"), source: filepath.Join(root, "b", "raw.gputrace")}, + } { + t.Run(test.name, func(t *testing.T) { + if got := traceDocumentMatches(test.document, test.source); got != test.want { + t.Fatalf("traceDocumentMatches(%q, %q) = %t, want %t", + test.document, test.source, got, test.want) + } + }) + } +} + +func TestRecoveryTimeoutErrorDoesNotWrapNil(t *testing.T) { + err := recoveryTimeoutError("timed out", nil) + if got := err.Error(); got != "timed out" || strings.Contains(got, "%!w") { + t.Fatalf("error = %q", got) + } + cause := errors.New("last state") + if err := recoveryTimeoutError("timed out", cause); !errors.Is(err, cause) { + t.Fatalf("error = %v, want wrapped cause", err) + } +} + +func TestFinalizeStandaloneExportRejectsUUIDMismatchAndPreservesOutput(t *testing.T) { + input := writeStandaloneExportFixture(t, "input", "wanted", true) + output := writeStandaloneExportFixture(t, "output", "other", true) + var status bytes.Buffer + _, err := finalizeStandaloneExport(&status, input, output) + if err == nil || !strings.Contains(err.Error(), "UUID") { + t.Fatalf("error = %v, want UUID mismatch", err) + } + if strings.Contains(status.String(), "Exported to:") { + t.Fatalf("mismatched export printed success:\n%s", status.String()) + } + if _, err := os.Stat(output); err != nil { + t.Fatalf("mismatched output was not preserved: %v", err) + } +} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_list.go b/cmd/gputrace/cmd/collect_xcode_profile_list.go index 2d7bf269..537745d6 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_list.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_list.go @@ -267,17 +267,31 @@ type JSONError struct { Suggestion string `json:"suggestion,omitempty"` } -// outputJSONError outputs a JSON error and returns nil (to avoid duplicate error output). +type reportedJSONError struct { + JSONError +} + +func (err reportedJSONError) Error() string { + return err.Message +} + +func (reportedJSONError) alreadyReported() {} + +// outputJSONError writes a JSON error and returns a non-nil marker error. +// The command entry point recognizes the marker and does not print it again. func outputJSONError(code, message, suggestion string) error { - enc := json.NewEncoder(os.Stdout) - enc.SetIndent("", " ") - enc.Encode(JSONError{ + report := JSONError{ Error: true, Code: code, Message: message, Suggestion: suggestion, - }) - return nil + } + enc := json.NewEncoder(os.Stdout) + enc.SetIndent("", " ") + if err := enc.Encode(report); err != nil { + return fmt.Errorf("write JSON error: %w", err) + } + return reportedJSONError{JSONError: report} } // keyButtons are the button names we care about for automation. diff --git a/cmd/gputrace/cmd/collect_xcode_profile_open.go b/cmd/gputrace/cmd/collect_xcode_profile_open.go index 94830e04..5a22e193 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_open.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_open.go @@ -10,6 +10,7 @@ import ( "time" "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/tracebundle" ) type openTraceOptions struct { @@ -25,6 +26,9 @@ func runOpenTrace(cmd *cobra.Command, args []string, opts *openTraceOptions) err if _, err := os.Stat(inputPath); os.IsNotExist(err) { return fmt.Errorf("trace file does not exist: %s", inputPath) } + if err := requireXcodeOpenableTrace(inputPath); err != nil { + return err + } status := xcodeProfileStatusWriter() fmt.Fprintf(status, "Opening trace in Xcode: %s\n", inputPath) @@ -43,18 +47,14 @@ func runOpenTrace(cmd *cobra.Command, args []string, opts *openTraceOptions) err fmt.Fprintln(status, "Waiting for Xcode window...") - // Wait for window using AX polling (doesn't steal focus) + // Wait for Xcode using AX polling (doesn't steal focus). deadline := time.Now().Add(30 * time.Second) var appAX uintptr var axErr error for time.Now().Before(deadline) { appAX, axErr = FindXcodeApp() if axErr == nil { - windows := GetAllWindows(appAX) - if len(windows) > 0 { - break - } - cfRelease(appAX) + break } time.Sleep(500 * time.Millisecond) } @@ -64,26 +64,32 @@ func runOpenTrace(cmd *cobra.Command, args []string, opts *openTraceOptions) err } defer cfRelease(appAX) - // Handle startup dialogs (Reopen, etc.) + // Handle startup dialogs (Reopen, etc.) before binding the requested trace + // window; a modal dialog can hide the document's AX attributes. if err := dismissStartupDialogs(); err != nil { verboseLog("dismissStartupDialogs: %v", err) } - // Ensure window is on-screen (may be restored to disconnected monitor position) - windows := GetAllWindows(appAX) - for _, w := range windows { - x, y := axPosition(w) - _, h := axSize(w) - // If window is off-screen (negative Y or very far), move it - if y < 0 || y > 2000 || x < -500 { - verboseLog("Window at (%d,%d) appears off-screen, repositioning", x, y) - setWindowPosition(w, 100, 100) - time.Sleep(200 * time.Millisecond) - } - // Also ensure window has reasonable height (not minimized) - if h < 100 { - verboseLog("Window height %d too small, may be minimized", h) - } + windowAX, err := waitForWindow(cmd.Context(), appAX, inputPath, 30*time.Second) + if err != nil { + return fmt.Errorf("find requested trace window: %w", err) + } + selection := selectionForWindow(inputPath, windowAX) + if err := requireBoundSelection(selection); err != nil { + return fmt.Errorf("Xcode opened a GPU trace window, but %w", err) + } + + // Ensure the selected window is on-screen (it may have been restored to a + // disconnected monitor). + x, y := axPosition(windowAX) + _, h := axSize(windowAX) + if y < 0 || y > 2000 || x < -500 { + verboseLog("Window at (%d,%d) appears off-screen, repositioning", x, y) + setWindowPosition(windowAX, 100, 100) + time.Sleep(200 * time.Millisecond) + } + if h < 100 { + verboseLog("Window height %d too small, may be minimized", h) } // Ensure the Debug navigator is shown using AX menu click @@ -93,13 +99,37 @@ func runOpenTrace(cmd *cobra.Command, args []string, opts *openTraceOptions) err } } - fmt.Fprint(status, Colorize("Trace opened successfully in Xcode\n", ColorGreen)) + fmt.Fprint(status, Colorize("Xcode opened the requested trace window.\n", ColorGreen)) + if selection.Document != "" { + fmt.Fprintf(status, " Selected document: %s\n", selection.Document) + } + if selection.Title != "" { + fmt.Fprintf(status, " Selected window: %s\n", selection.Title) + } + fmt.Fprintf(status, " Evidence: %s\n", selection.Evidence) return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "open", - Input: inputPath, + Action: "open", + Input: inputPath, + RequestedTrace: inputPath, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "trace window ready", + Evidence: selection.Evidence, + TargetBound: boolPointer(selection.Bound), }) } +func requireXcodeOpenableTrace(path string) error { + payload, err := tracebundle.InspectPayload(path) + if err != nil { + return fmt.Errorf("inspect trace before opening in Xcode: %w", err) + } + if payload.Class == tracebundle.PayloadProfilerOnly { + return fmt.Errorf("cannot open %s in Xcode: profiler-only .gpuprofiler_raw data has no capture or index; use gputrace profiler/timing, or rerun profile-replay without --profiler-only", path) + } + return nil +} + func xcodeOpenArgs() []string { if app := os.Getenv("GPUTRACE_XCODE_APP"); app != "" { return []string{"-a", app} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_open_test.go b/cmd/gputrace/cmd/collect_xcode_profile_open_test.go new file mode 100644 index 00000000..7da23da0 --- /dev/null +++ b/cmd/gputrace/cmd/collect_xcode_profile_open_test.go @@ -0,0 +1,43 @@ +//go:build darwin + +package cmd + +import ( + "os" + "path/filepath" + "strings" + "testing" +) + +func TestRequireXcodeOpenableTraceRejectsProfilerOnly(t *testing.T) { + bundle := filepath.Join(t.TempDir(), "profile.gputrace") + raw := filepath.Join(bundle, "profile.gpuprofiler_raw") + if err := os.MkdirAll(raw, 0755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(raw, "streamData"), []byte("data"), 0644); err != nil { + t.Fatal(err) + } + err := requireXcodeOpenableTrace(bundle) + if err == nil { + t.Fatal("profiler-only input accepted") + } + for _, text := range []string{"profiler-only", "no capture or index", "without --profiler-only"} { + if !strings.Contains(err.Error(), text) { + t.Errorf("error %q does not contain %q", err, text) + } + } +} + +func TestRequireXcodeOpenableTraceAcceptsCapture(t *testing.T) { + bundle := filepath.Join(t.TempDir(), "capture.gputrace") + if err := os.Mkdir(bundle, 0755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(bundle, "capture"), []byte("data"), 0644); err != nil { + t.Fatal(err) + } + if err := requireXcodeOpenableTrace(bundle); err != nil { + t.Fatal(err) + } +} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_output_test.go b/cmd/gputrace/cmd/collect_xcode_profile_output_test.go index ceaaceed..81e226bf 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_output_test.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_output_test.go @@ -6,6 +6,7 @@ import ( "bytes" "encoding/json" "errors" + "io" "os" "path/filepath" "strings" @@ -99,6 +100,232 @@ func TestWriteXcodeProfileActionOutputPlainNoop(t *testing.T) { } } +func TestXcodeProfileJSONErrorsAreReportedOnceAndReturnError(t *testing.T) { + tests := []string{ + "check-status", + "list-windows", + "performance-show", + "performance-summary", + } + for _, name := range tests { + t.Run(name, func(t *testing.T) { + command := &cobra.Command{ + Use: name, + SilenceErrors: true, + SilenceUsage: true, + RunE: func(cmd *cobra.Command, args []string) error { + return outputJSONError("NOT_AVAILABLE", name+" failed", "try again") + }, + } + + stdout, err := captureStdout(t, command.Execute) + if err == nil { + t.Fatal("Execute returned nil error") + } + if !ErrorAlreadyReported(err) { + t.Fatalf("ErrorAlreadyReported(%T) = false, want true", err) + } + + dec := json.NewDecoder(strings.NewReader(stdout)) + var got JSONError + if err := dec.Decode(&got); err != nil { + t.Fatalf("decode JSON error: %v\n%s", err, stdout) + } + var extra interface{} + if err := dec.Decode(&extra); !errors.Is(err, io.EOF) { + t.Fatalf("stdout contains more than one JSON value: %s", stdout) + } + if !got.Error || got.Code != "NOT_AVAILABLE" || got.Message != name+" failed" || got.Suggestion != "try again" { + t.Fatalf("JSON error = %+v", got) + } + }) + } +} + +func TestOrdinaryErrorIsNotAlreadyReported(t *testing.T) { + if ErrorAlreadyReported(errors.New("ordinary failure")) { + t.Fatal("ordinary error reported as already written") + } +} + +func TestXcodeProfileMacgoForwardsChildExitStatus(t *testing.T) { + config := xcodeProfileMacgoConfig() + if !config.ForceDirectExecution { + t.Fatal("ForceDirectExecution = false; LaunchServices would lose the child command exit status") + } +} + +func TestXcodeProfileCommandErrorsReachExecute(t *testing.T) { + oldPreRunE := collectXcodeProfileCmd.PersistentPreRunE + oldSilenceErrors := rootCmd.SilenceErrors + oldSilenceUsage := rootCmd.SilenceUsage + t.Cleanup(func() { + collectXcodeProfileCmd.PersistentPreRunE = oldPreRunE + rootCmd.SilenceErrors = oldSilenceErrors + rootCmd.SilenceUsage = oldSilenceUsage + rootCmd.SetArgs(nil) + }) + collectXcodeProfileCmd.PersistentPreRunE = func(cmd *cobra.Command, args []string) error { + return nil + } + rootCmd.SilenceErrors = true + rootCmd.SilenceUsage = true + + tests := []struct { + name string + args []string + want string + }{ + { + name: "wait-profile unbound target", + args: []string{"xcode-profile", "wait-profile", "trace.gputrace"}, + want: `selected Xcode window is not bound to requested trace "trace.gputrace"`, + }, + { + name: "show-performance unavailable verbose", + args: []string{"xcode-profile", "show-performance", "--verbose"}, + want: "Show Performance button not found", + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + command, _, err := rootCmd.Find(tt.args) + if err != nil { + t.Fatal(err) + } + oldRunE := command.RunE + command.RunE = func(cmd *cobra.Command, args []string) error { + return errors.New(tt.want) + } + defer func() { + command.RunE = oldRunE + }() + + rootCmd.SetArgs(tt.args) + err = rootCmd.Execute() + if err == nil { + t.Fatal("Execute returned nil error") + } + if got := err.Error(); got != tt.want { + t.Fatalf("error = %q, want %q", got, tt.want) + } + if ErrorAlreadyReported(err) { + t.Fatal("ordinary command error marked as already reported") + } + }) + } +} + +func TestXcodeWindowSelectionBinding(t *testing.T) { + tests := []struct { + name string + requested string + title string + document string + wantBound bool + wantText string + }{ + { + name: "exact document", + requested: "/Users/test/trace.gputrace", + title: "Summary", + document: "/Users/test/trace.gputrace", + wantBound: true, + wantText: "exactly matches", + }, + { + name: "document basename", + requested: "trace.gputrace", + title: "Summary", + document: "/Users/test/trace.gputrace", + wantBound: true, + wantText: "trace filename", + }, + { + name: "title basename", + requested: "/Users/test/trace.gputrace", + title: "trace.gputrace — Summary", + wantBound: true, + wantText: "window title", + }, + { + name: "untitled unbound", + requested: "/Users/test/trace.gputrace", + title: "Summary", + wantBound: false, + wantText: "no title or AXDocument match", + }, + { + name: "unspecified target", + title: "Summary", + wantBound: true, + wantText: "no trace was requested", + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := newXcodeWindowSelection(tt.requested, tt.title, tt.document) + if got.Bound != tt.wantBound { + t.Fatalf("Bound = %t, want %t: %+v", got.Bound, tt.wantBound, got) + } + if !strings.Contains(got.Evidence, tt.wantText) { + t.Fatalf("Evidence = %q, want %q", got.Evidence, tt.wantText) + } + }) + } +} + +func TestUnboundCompletionIsNotReportedAsComplete(t *testing.T) { + output := StatusOutput{ + Status: "complete", + Phase: profilingPhase("complete"), + Evidence: profilingStatusEvidence("complete"), + } + selection := newXcodeWindowSelection( + "/Users/test/python.gputrace", + "Summary", + "/Users/test/go.gputrace", + ) + applyStatusSelection(&output, selection) + + if output.Status != "unknown" || output.Phase != "unbound" || output.TargetBound { + t.Fatalf("status output = %+v, want unknown unbound target", output) + } + if !strings.Contains(output.Evidence, `refusing to attribute detected "complete"`) { + t.Fatalf("Evidence = %q, want refusal context", output.Evidence) + } + if err := requireBoundSelection(selection); err == nil { + t.Fatal("requireBoundSelection accepted an unbound requested trace") + } +} + +func TestStatusTextIncludesTargetAndEvidence(t *testing.T) { + output := StatusOutput{ + Status: "running", + Phase: profilingPhase("running"), + Evidence: "AXDocument exactly matches; profiling indicator detected", + RequestedTrace: "trace.gputrace", + SelectedTitle: "Summary", + SelectedDocument: "/Users/test/trace.gputrace", + TargetBound: true, + } + var text strings.Builder + writeStatusText(&text, output) + for _, want := range []string{ + "Status: running", + "Phase: performance profiling running", + "Requested trace: trace.gputrace", + "Selected document: /Users/test/trace.gputrace", + "Selected window: Summary", + "Target bound: true", + "Evidence:", + } { + if !strings.Contains(text.String(), want) { + t.Fatalf("text lacks %q:\n%s", want, text.String()) + } + } +} + func TestHiddenXcodeProfileUtilityCommandsRejectJSONBeforeRunE(t *testing.T) { oldJSON := collectProfileOpts.json oldPreRunE := collectXcodeProfileCmd.PersistentPreRunE @@ -207,3 +434,36 @@ func TestDefaultXcodeProfileOutputPath(t *testing.T) { t.Fatalf("default path = %q, want %q", got, want) } } + +func TestRequireExportedTrace(t *testing.T) { + dir := t.TempDir() + if err := requireExportedTrace(dir); err != nil { + t.Fatalf("existing output: %v", err) + } + + missing := filepath.Join(dir, "missing.gputrace") + err := requireExportedTrace(missing) + if err == nil { + t.Fatal("missing output returned nil error") + } + if got := err.Error(); !strings.Contains(got, "output not found at expected location") || !strings.Contains(got, missing) { + t.Fatalf("error = %q, want missing output path and context", got) + } +} + +func TestExistingExportCandidatesReportsAlternateWithoutMovingIt(t *testing.T) { + dir := t.TempDir() + requested := filepath.Join(dir, "requested.gputrace") + alternate := filepath.Join(dir, "raw-basename.gputrace") + if err := os.Mkdir(alternate, 0o755); err != nil { + t.Fatalf("mkdir alternate: %v", err) + } + + got := existingExportCandidates([]string{requested, alternate, alternate}, requested) + if len(got) != 1 || got[0] != alternate { + t.Fatalf("existingExportCandidates = %q, want [%q]", got, alternate) + } + if _, err := os.Stat(alternate); err != nil { + t.Fatalf("alternate output was not preserved: %v", err) + } +} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_performance.go b/cmd/gputrace/cmd/collect_xcode_profile_performance.go index f420c897..10970e27 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_performance.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_performance.go @@ -99,6 +99,7 @@ func runPerformanceShow(cmd *cobra.Command, args []string) error { } return err } + selection := selectionForWindow("", windowAX) btn := findShowPerformanceButton(windowAX) if btn == 0 { @@ -123,18 +124,36 @@ func runPerformanceShow(cmd *cobra.Command, args []string) error { } return fmt.Errorf("failed to click: %w", err) } - - if collectProfileOpts.json { - enc := json.NewEncoder(os.Stdout) - enc.SetIndent("", " ") - return enc.Encode(map[string]interface{}{ - "success": true, - "action": "show_performance", - }) + if err := waitForPerformanceView(cmd.Context(), windowAX, 3*time.Second); err != nil { + return err + } + fmt.Fprintln(status, "Performance view verified") + return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + Action: "show_performance", + Target: selection.Document, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "performance view visible", + Evidence: "Xcode performance navigation controls are present after Show Performance", + TargetBound: boolPointer(selection.Bound), + }) +} + +func waitForPerformanceView(ctx context.Context, window uintptr, timeout time.Duration) error { + deadline := time.Now().Add(timeout) + for { + for _, name := range []string{"Overview", "Timeline", "Shaders", "Counters", "Encoders"} { + if findButtonBFS(window, name, 1000) != 0 { + return nil + } + } + if time.Now().After(deadline) { + return fmt.Errorf("Show Performance was pressed, but the Performance view did not become visible within %s", timeout.Round(time.Second)) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return err + } } - - fmt.Fprintln(status, "Done") - return nil } func runPerformanceStatus(cmd *cobra.Command, args []string) error { @@ -788,6 +807,7 @@ func runPerformanceView(ctx context.Context, viewName string) error { } return err } + selection := selectionForWindow("", windowAX) // Map view names to button names in Xcode UI buttonNames := map[string]string{ @@ -830,16 +850,35 @@ func runPerformanceView(ctx context.Context, viewName string) error { } return fmt.Errorf("failed to click: %w", err) } - - if collectProfileOpts.json { - enc := json.NewEncoder(os.Stdout) - enc.SetIndent("", " ") - return enc.Encode(map[string]interface{}{ - "success": true, - "view": viewName, - }) + if err := waitForSelectedControl(ctx, windowAX, btn, buttonName, 2*time.Second); err != nil { + return err + } + fmt.Fprintf(status, "%s view selected and verified\n", buttonName) + return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + Action: "select_performance_view", + Target: viewName, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "performance view selected", + Evidence: fmt.Sprintf("%s control reports selected", buttonName), + TargetBound: boolPointer(selection.Bound), + }) +} + +func waitForSelectedControl(ctx context.Context, window, control uintptr, name string, timeout time.Duration) error { + deadline := time.Now().Add(timeout) + for { + if isElementSelected(control) || isTabSelected(control) || strings.EqualFold(getCurrentTab(window), name) { + return nil + } + if tab := findTabByName(window, name); tab != 0 && isTabSelected(tab) { + return nil + } + if time.Now().After(deadline) { + return fmt.Errorf("%s was pressed, but Xcode did not expose a selected-state postcondition within %s", name, timeout.Round(time.Second)) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return err + } } - - fmt.Fprintln(status, "Done") - return nil } diff --git a/cmd/gputrace/cmd/collect_xcode_profile_replay.go b/cmd/gputrace/cmd/collect_xcode_profile_replay.go index 05513187..fc1f20aa 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_replay.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_replay.go @@ -4,7 +4,6 @@ package cmd import ( "fmt" - "path/filepath" "github.com/spf13/cobra" ) @@ -55,7 +54,7 @@ func runWaitReplay(cmd *cobra.Command, args []string) error { } status := xcodeProfileStatusWriter() - fmt.Fprintln(status, "Waiting for replay to complete...") + fmt.Fprintln(status, "Waiting for GPU replay and performance profiling...") appAX, err := FindXcodeApp() if err != nil { @@ -67,15 +66,32 @@ func runWaitReplay(cmd *cobra.Command, args []string) error { if err != nil { return fmt.Errorf("window not found: %w", err) } + selection := selectionForWindow(traceFile, windowAX) + if err := requireBoundSelection(selection); err != nil { + return err + } + fmt.Fprintln(status, " Phase: performance profiling running or pending") + if selection.Document != "" { + fmt.Fprintf(status, " Selected document: %s\n", selection.Document) + } + if selection.Title != "" { + fmt.Fprintf(status, " Selected window: %s\n", selection.Title) + } + fmt.Fprintf(status, " Evidence: %s\n", selection.Evidence) - traceFileName := filepath.Base(traceFile) - if err := waitForReplayComplete(cmd.Context(), appAX, traceFileName, windowAX, collectProfileOpts.timeout); err != nil { - return fmt.Errorf("wait failed: %w", err) + if err := waitForReplayComplete(cmd.Context(), appAX, traceFile, windowAX, collectProfileOpts.timeout); err != nil { + return fmt.Errorf("wait for performance profiling: %w", err) } - fmt.Fprint(status, Colorize("Replay completed\n", ColorGreen)) + fmt.Fprint(status, Colorize("Performance data became available after GPU replay; export identity is not yet verified.\n", ColorGreen)) return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "wait-profile", - Target: traceFile, + Action: "wait-profile", + Target: traceFile, + RequestedTrace: traceFile, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "performance data available", + Evidence: "Xcode exposed a completion-ready performance control for the bound trace window; export identity is not yet verified", + TargetBound: boolPointer(selection.Bound), }) } diff --git a/cmd/gputrace/cmd/collect_xcode_profile_run.go b/cmd/gputrace/cmd/collect_xcode_profile_run.go index 48a221c1..cc47bdbf 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_run.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_run.go @@ -5,6 +5,7 @@ package cmd import ( "context" "fmt" + "net/url" "os" "os/exec" "path/filepath" @@ -12,18 +13,83 @@ import ( "time" "github.com/spf13/cobra" + + gputraceTrace "github.com/tmc/gputrace/internal/trace" + "github.com/tmc/gputrace/internal/tracebundle" ) +var xcodeProfileAutomationStartHook = func() {} + func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { + inputPath, err := filepath.Abs(args[0]) + if err != nil { + return fmt.Errorf("invalid input path: %w", err) + } + if _, err := os.Stat(inputPath); os.IsNotExist(err) { + return fmt.Errorf("trace file does not exist: %s", inputPath) + } + payload, err := tracebundle.InspectPayload(inputPath) + if err != nil { + return err + } + + status := xcodeProfileStatusWriter() + if payload.HasProfilerStream { + profilerDir := findProfilerDir(inputPath) + if profilerDir == "" && filepath.Ext(inputPath) == ".gpuprofiler_raw" { + profilerDir = inputPath + } + if _, err := readExportTraceSignature(inputPath, profilerDir); err != nil { + return fmt.Errorf("verify embedded performance data: %w", err) + } + fmt.Fprintln(status, "Performance data already embedded; verified non-empty streamData.") + fmt.Fprintf(status, "Using existing trace: %s\n", inputPath) + writeXcodePayloadStatus(status, payload) + output := xcodeProfileActionOutput{ + Action: "run", + Input: inputPath, + Output: inputPath, + Source: inputPath, + Reused: true, + } + applyXcodePayload(&output, payload) + return writeXcodeProfileActionOutput(output) + } + if err := validateTraceBundle(inputPath); err != nil { + return err + } + + xcodeProfileAutomationStartHook() automationCtx, cleanupCancel := StartAutomationCancelListener(cmd.Context(), true) defer cleanupCancel() ctx, cancel := context.WithTimeout(automationCtx, collectProfileOpts.timeout) defer cancel() - inputPath, err := filepath.Abs(args[0]) - if err != nil { - return fmt.Errorf("invalid input path: %w", err) - } + var activeWindowAX uintptr + var replayCompleted bool + defer func() { + if ctx.Err() == nil { + return + } + status := xcodeProfileStatusWriter() + // Once the replay has finished, the window holds the profile and is the + // only copy of it: a later step timing out is a reason to stop driving + // Xcode, not a reason to destroy the result. Closing here discarded a + // completed 53s profile when a downstream reacquire failed. + if replayCompleted { + fmt.Fprintf(status, " Interrupted after the replay completed (%v); "+ + "leaving the Xcode window open so the performance data survives.\n", ctx.Err()) + return + } + fmt.Fprintf(status, " Cancelling Xcode GPU workload due to CLI interrupt/timeout (%v)...\n", ctx.Err()) + if activeWindowAX != 0 { + _ = stopWorkloadInWindow(activeWindowAX) + closeXcodeWindow(activeWindowAX) + } else { + _ = stopAllXcodeWorkloads(context.Background()) + _ = closeAllXcodeWindows(context.Background()) + } + }() output := collectProfileOpts.output if output == "" { @@ -48,19 +114,27 @@ func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { return err } - status := xcodeProfileStatusWriter() + crashReportDir := diagnosticReportDirectory() + crashBaseline, err := snapshotXcodeCrashReports(crashReportDir) + if err != nil { + return fmt.Errorf("snapshot Xcode crash reports: %w", err) + } + requestedXcode := requestedXcodeAppPath() + crashScope := newXcodeCrashScope(requestedXcode, time.Now()) + // A normal "open" request is delivered to the sole existing instance. + // Observe it before opening so a crash during document loading is still + // attributable even if LaunchServices immediately relaunches Xcode. + if existing := xcodeProcessesForApp(requestedXcode); len(existing) == 1 { + crashScope.observe(existing[0]) + } + crashContext, stopCrashMonitor := startXcodeCrashMonitor(ctx, crashReportDir, crashBaseline, crashScope) + defer stopCrashMonitor() + ctx = crashContext + fmt.Fprint(status, Colorize("Collect Profile: Automating Xcode GPU trace...\n", ColorBold)) fmt.Fprintf(status, " Input: %s\n", inputPath) fmt.Fprintf(status, " Output: %s\n", outputPath) - // Validate trace bundle before opening in Xcode - if _, err := os.Stat(inputPath); os.IsNotExist(err) { - return fmt.Errorf("trace file does not exist: %s", inputPath) - } - if err := validateTraceBundle(inputPath); err != nil { - return err - } - // Step 1: Open File in Xcode fmt.Fprintln(status, " Step 1: Opening trace in Xcode...") @@ -83,24 +157,34 @@ func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { // Step 2: Wait for Xcode window via AX fmt.Fprintln(status, " Step 2: Waiting for Xcode window...") - appAX, err := FindXcodeApp() + appAX, xcodeIdentity, err := findSelectedXcodeApp(ctx, requestedXcode) if err != nil { - return fmt.Errorf("Xcode not found via AX: %w", err) + return fmt.Errorf("selected Xcode app not found via AX: %w", err) } defer cfRelease(appAX) + crashScope.bind(xcodeIdentity) + fmt.Fprintf(status, " Xcode process: PID %d, app %s, bundle %s\n", + xcodeIdentity.PID, xcodeIdentity.AppPath, xcodeIdentity.BundleID) - traceFileName := filepath.Base(inputPath) - windowAX, err := waitForWindow(ctx, appAX, traceFileName, 30*time.Second) + windowAX, err := waitForWindow(ctx, appAX, inputPath, 30*time.Second) if err != nil { + crashScope.refreshProcesses() + if crashScope.crashSuspected() { + if crashErr := waitForXcodeCrashReport(ctx, xcodeCrashReportGrace); crashErr != nil { + return crashErr + } + } return fmt.Errorf("Xcode window not found: %w", err) } + activeWindowAX = windowAX + traceGeometryKey := recoveryGeometryKeyForElement(windowAX, xcodeIdentity.PID) if err := checkAutomationCanceled(ctx); err != nil { return err } - // Check if trace already has performance data (Show Performance button visible) - alreadyHasPerfData := hasShowPerformance(windowAX) + // Check if trace already has performance data. + alreadyHasPerfData := hasPerformanceData(windowAX) // Check if profiling is actually in progress. In Xcode's "Profile after // replay" flow the Replay button can disappear while profiler data is still // being prepared, so Stop alone is enough to mean "keep waiting" here. @@ -120,10 +204,11 @@ func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { } else if profilingInProgress { // Profiling already running (e.g., from a prior attempt or --force) — just wait for it fmt.Fprintln(status, " Profiling already in progress, waiting for completion...") - if err := waitForReplayComplete(ctx, appAX, traceFileName, windowAX, collectProfileOpts.timeout); err != nil { + if err := waitForReplayComplete(ctx, appAX, inputPath, windowAX, collectProfileOpts.timeout); err != nil { return fmt.Errorf("replay wait failed: %w", err) } fmt.Fprintln(status, " Profiling completed") + replayCompleted = true } else { // Step 3: Start replay fmt.Fprintln(status, " Step 3: Starting replay...") @@ -133,10 +218,11 @@ func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { // Step 4: Wait for replay fmt.Fprintln(status, " Step 4: Waiting for replay to complete...") - if err := waitForReplayComplete(ctx, appAX, traceFileName, windowAX, collectProfileOpts.timeout); err != nil { + if err := waitForReplayComplete(ctx, appAX, inputPath, windowAX, collectProfileOpts.timeout); err != nil { return fmt.Errorf("replay wait failed: %w", err) } fmt.Fprintln(status, " Replay completed") + replayCompleted = true } if err := checkAutomationCanceled(ctx); err != nil { @@ -145,39 +231,49 @@ func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { // Verify performance data is actually available after replay. if !alreadyHasPerfData { - if freshWindow := getPreferredTraceWindow(appAX, traceFileName); freshWindow != 0 { - windowAX = freshWindow - } else if freshWindow := findTraceWindowByButtons(appAX); freshWindow != 0 { - windowAX = freshWindow + freshWindow, err := waitForBoundTraceWindowAfterReplay( + ctx, appAX, xcodeIdentity, inputPath, traceGeometryKey, true, false, 10*time.Second, + ) + if err != nil { + return fmt.Errorf("reacquire completed trace window: %w", err) } - if !hasShowPerformance(windowAX) { + windowAX = freshWindow + if !hasPerformanceData(windowAX) { return fmt.Errorf("replay completed but performance data is not available — the trace may not contain enough GPU work to profile") } } - if freshWindow := getPreferredTraceWindow(appAX, traceFileName); freshWindow != 0 { - windowAX = freshWindow - } else if freshWindow := findTraceWindowByButtons(appAX); freshWindow != 0 { - windowAX = freshWindow + freshWindow, err := waitForBoundTraceWindowAfterReplay( + ctx, appAX, xcodeIdentity, inputPath, traceGeometryKey, true, false, 10*time.Second, + ) + if err != nil { + return fmt.Errorf("reacquire trace window before Show Performance: %w", err) } + windowAX = freshWindow if shown, err := showPerformanceBeforeExport(windowAX); err != nil { return fmt.Errorf("show performance before export: %w", err) } else if shown { - // Xcode only enables "Embed performance data" after the Performance view - // has been opened. Give the view time to settle before opening Export. - if err := waitForAutomation(ctx, time.Second); err != nil { + freshWindow, err = waitForBoundTraceWindowAfterReplay( + ctx, appAX, xcodeIdentity, inputPath, traceGeometryKey, false, true, 15*time.Second, + ) + if err != nil { + return fmt.Errorf("reacquire trace window after Show Performance: %w", err) + } + windowAX = freshWindow + if err := startPerformanceProfile(ctx, windowAX); err != nil { return err } } // Export step fmt.Fprintln(status, " Exporting trace...") - if freshWindow := getPreferredTraceWindow(appAX, traceFileName); freshWindow != 0 { - windowAX = freshWindow - } else if freshWindow := findTraceWindowByButtons(appAX); freshWindow != 0 { - windowAX = freshWindow + freshWindow, err = waitForPerformanceExportReady( + ctx, appAX, xcodeIdentity, inputPath, collectProfileOpts.timeout, + ) + if err != nil { + return fmt.Errorf("wait for performance export: %w", err) } - activateXcodeQuick(ctx) + windowAX = freshWindow axAction(windowAX, "AXRaise") time.Sleep(300 * time.Millisecond) @@ -203,13 +299,28 @@ func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { if err != nil { return err } + if err := verifyExportTraceIdentity(inputPath, finalPath); err != nil { + if removeErr := os.RemoveAll(finalPath); removeErr != nil { + return fmt.Errorf("%w; remove mismatched export: %v", err, removeErr) + } + return err + } + exportedPayload, err := tracebundle.InspectPayload(finalPath) + if err != nil { + return fmt.Errorf("inspect exported trace payload: %w", err) + } + if err := requireSelfContainedExport(finalPath, exportedPayload); err != nil { + writeXcodePayloadStatus(status, exportedPayload) + return err + } + writeXcodePayloadStatus(status, exportedPayload) // Close the Xcode window after export completes // Re-fetch window reference since it may have become stale during export // (window title may change or become empty after profiling) if freshWindow := findTraceWindowByButtons(appAX); freshWindow != 0 { closeXcodeWindow(freshWindow) - } else if freshWindow := getPreferredTraceWindow(appAX, traceFileName); freshWindow != 0 { + } else if freshWindow := getPreferredTraceWindow(appAX, inputPath); freshWindow != 0 { closeXcodeWindow(freshWindow) } else { closeXcodeWindow(windowAX) // Try original reference as fallback @@ -221,29 +332,35 @@ func runCollectXcodeProfileFull(cmd *cobra.Command, args []string) error { if err := copyPath(finalPath, outputPath); err != nil { warning := fmt.Sprintf("file saved to %s; copy to %s failed: %v", finalPath, outputPath, err) fmt.Fprintf(status, Colorize("\nNote: File saved to %s (copy to %s failed: %v)\n", ColorYellow), finalPath, outputPath, err) - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + actionOutput := xcodeProfileActionOutput{ Action: "run", Input: inputPath, Output: finalPath, RequestedOutput: outputPath, Warning: warning, - }) + } + applyXcodePayload(&actionOutput, exportedPayload) + return writeXcodeProfileActionOutput(actionOutput) } fmt.Fprintf(status, Colorize("\nDone! Output saved to: %s (copied from %s)\n", ColorGreen), outputPath, finalPath) - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + actionOutput := xcodeProfileActionOutput{ Action: "run", Input: inputPath, Output: outputPath, Source: finalPath, Copied: true, - }) + } + applyXcodePayload(&actionOutput, exportedPayload) + return writeXcodeProfileActionOutput(actionOutput) } fmt.Fprintf(status, Colorize("\nDone! Output saved to: %s\n", ColorGreen), outputPath) - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + actionOutput := xcodeProfileActionOutput{ Action: "run", Input: inputPath, Output: outputPath, - }) + } + applyXcodePayload(&actionOutput, exportedPayload) + return writeXcodeProfileActionOutput(actionOutput) } // findTraceWindowByButtons finds an Xcode window with trace-related buttons @@ -261,6 +378,40 @@ func findTraceWindowByButtons(appAX uintptr) uintptr { return 0 } +// stopWorkloadInWindow stops any active GPU profiling/replay workload in the window by clicking the Stop button if enabled. +func stopWorkloadInWindow(windowAX uintptr) error { + if windowAX == 0 { + return nil + } + stopBtn := FindStopButton(windowAX) + if stopBtn != 0 && IsElementEnabled(stopBtn) { + verboseLog("stopWorkloadInWindow: stopping active GPU workload in window %q", axString(windowAX, "AXTitle")) + if err := axAction(stopBtn, "AXPress"); err != nil { + verboseLog("stopWorkloadInWindow: AXPress failed: %v, trying fallback", err) + if err := axPressWithFallback(stopBtn); err != nil { + return fmt.Errorf("failed to click Stop GPU workload button: %w", err) + } + } + time.Sleep(300 * time.Millisecond) + } + return nil +} + +// stopAllXcodeWorkloads stops any active GPU workloads across all open Xcode windows. +func stopAllXcodeWorkloads(ctx context.Context) error { + appAX, err := FindXcodeApp() + if err != nil { + return nil + } + defer cfRelease(appAX) + + windows := GetAllWindows(appAX) + for _, w := range windows { + _ = stopWorkloadInWindow(w) + } + return nil +} + // closeXcodeWindow closes the specified Xcode window // closeAllXcodeWindows closes all open Xcode windows to clear stale GPU trace sessions. func closeAllXcodeWindows(ctx context.Context) error { @@ -286,6 +437,9 @@ func closeXcodeWindow(windowAX uintptr) { return } + // Stop any active GPU workload before closing the window + _ = stopWorkloadInWindow(windowAX) + // Try AXCloseButton attribute (standard macOS window close button) var closeBtn uintptr key := mkString("AXCloseButton") @@ -371,45 +525,180 @@ func waitForWindow(ctx context.Context, appAX uintptr, traceFileName string, tim return 0, fmt.Errorf("could not find Xcode window for %s (no Xcode windows found - check Accessibility permissions)", traceFileName) } +func waitForBoundTraceWindow(ctx context.Context, appAX uintptr, identity xcodeProcessIdentity, traceFileName string, timeout time.Duration) (uintptr, error) { + deadline := time.Now().Add(timeout) + var candidate uintptr + stable := 0 + for { + bound, err := xcodeIdentityForAX(appAX) + if err != nil || bound.PID != identity.PID || + filepath.Clean(bound.AppPath) != filepath.Clean(identity.AppPath) { + return 0, fmt.Errorf("lost Xcode binding: want PID %d app %s", identity.PID, identity.AppPath) + } + if window := getPreferredTraceWindow(appAX, traceFileName); window != 0 { + var pid int32 + if axUIElementGetPid(window, &pid) == kAXErrorSuccess && int(pid) == identity.PID { + if window == candidate { + stable++ + } else { + candidate = window + stable = 1 + } + if stable >= 2 { + return window, nil + } + } + } else { + candidate = 0 + stable = 0 + } + if time.Now().After(deadline) { + return 0, fmt.Errorf("bound Xcode PID %d app %s did not expose the trace window for %s within %s", + identity.PID, identity.AppPath, traceFileName, timeout.Round(time.Second)) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return 0, err + } + } +} + +func waitForBoundTraceWindowAfterReplay( + ctx context.Context, + appAX uintptr, + identity xcodeProcessIdentity, + traceFileName, geometryKey string, + allowSummary, allowPerformance bool, + timeout time.Duration, +) (uintptr, error) { + deadline := time.Now().Add(timeout) + recovery := standaloneExportRecovery{ + Enabled: true, + SourcePath: filepath.Clean(traceFileName), + Identity: identity, + } + var candidateKey string + stable := 0 + var lastErr error + for { + bound, err := xcodeIdentityForAX(appAX) + if err != nil || bound.PID != identity.PID || + filepath.Clean(bound.AppPath) != filepath.Clean(identity.AppPath) { + return 0, fmt.Errorf("lost Xcode binding: want PID %d app %s", identity.PID, identity.AppPath) + } + + var window standaloneRecoveryWindow + element := getPreferredTraceWindow(appAX, traceFileName) + if element != 0 && !selectionForWindow(traceFileName, element).Bound { + lastErr = fmt.Errorf("GPU window lacks exact title or AXDocument source binding") + element = 0 + } + if element != 0 && allowPerformance && !hasPerformanceView(element) { + lastErr = fmt.Errorf("source-bound trace window has not entered Performance") + element = 0 + } + if element != 0 { + window = standaloneRecoveryWindow{ + xcodeAXWindow: xcodeAXWindow{ + Element: element, + Title: axString(element, "AXTitle"), + Document: axString(element, "AXDocument"), + }, + PID: identity.PID, + } + window.X, window.Y = axPosition(element) + window.Width, window.Height = axSize(element) + } else { + windows := recoveryWindows(appAX) + switch { + case allowSummary: + window, err = summaryRecoveryTarget(windows, recovery, geometryKey) + case allowPerformance: + window, err = transitionedRecoveryPerformanceTarget(windows, recovery, geometryKey) + default: + err = fmt.Errorf("no post-replay transition state is allowed") + } + if err != nil { + lastErr = err + } + } + + if window.Element != 0 { + key := standaloneRecoveryWindowKey(window) + if key == candidateKey { + stable++ + } else { + candidateKey = key + stable = 1 + } + if stable >= 2 { + return window.Element, nil + } + } else { + candidateKey = "" + stable = 0 + } + if time.Now().After(deadline) { + return 0, fmt.Errorf("bound Xcode PID %d app %s did not expose the trace or allowed post-replay state for %s within %s: %w", + identity.PID, identity.AppPath, traceFileName, timeout.Round(time.Second), lastErr) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return 0, err + } + } +} + // getPreferredTraceWindow finds the best matching window for a trace filename. // When multiple windows match (e.g., document window + trace viewer), prefer the one // with GPU trace UI elements (Replay button, profiling status). +// uiIdentifiedTraceWindow is the window that getPreferredTraceWindow last +// accepted solely because it was the only one carrying GPU trace UI. Xcode +// clears a trace window's title and AXDocument during replay, so that +// uniqueness is the only remaining evidence that the window is the requested +// trace. selectionForWindow consults it. The Xcode automation is single +// threaded, so a package-level value is sufficient. +var uiIdentifiedTraceWindow uintptr + func getPreferredTraceWindow(appAX uintptr, traceFileName string) uintptr { - titleLower := strings.ToLower(traceFileName) - allWindows := GetAllWindows(appAX) - for _, child := range allWindows { - title := axString(child, "AXTitle") - doc := axString(child, "AXDocument") - verboseLog("getPreferredTraceWindow: visible window: title=%q doc=%q", title, doc) + uiIdentifiedTraceWindow = 0 + traceIdentity := strings.ToLower(filepath.Clean(traceFileName)) + traceBase := strings.ToLower(filepath.Base(traceFileName)) + allWindows := deduplicateAXWindows(GetAllWindows(appAX)) + for _, window := range allWindows { + verboseLog("getPreferredTraceWindow: visible window: title=%q doc=%q", window.Title, window.Document) } verboseLog("getPreferredTraceWindow: %d total Xcode windows, looking for %q", len(allWindows), traceFileName) + exactWindows := exactTraceWindows(allWindows, traceIdentity) var matchingWindows []uintptr - for _, child := range allWindows { - // Check AXTitle - windowTitle := strings.ToLower(axString(child, "AXTitle")) - if strings.Contains(windowTitle, titleLower) { - matchingWindows = append(matchingWindows, child) - continue - } - // Check AXDocument (file path) - windowDoc := strings.ToLower(axString(child, "AXDocument")) - if strings.Contains(windowDoc, titleLower) { - matchingWindows = append(matchingWindows, child) + if len(exactWindows) == 0 { + for _, window := range allWindows { + child := window.Element + windowTitle := strings.ToLower(window.Title) + windowDoc := strings.ToLower(filepath.Clean(window.Document)) + if traceBase != "" && strings.Contains(windowTitle, traceBase) { + matchingWindows = append(matchingWindows, child) + continue + } + if traceBase != "" && strings.Contains(windowDoc, traceBase) { + matchingWindows = append(matchingWindows, child) + } } + } else { + matchingWindows = exactWindows } // Second pass: try matching without extension (Xcode sometimes strips it) if len(matchingWindows) == 0 { - baseName := strings.ToLower(strings.TrimSuffix(traceFileName, filepath.Ext(traceFileName))) - if baseName != titleLower { - for _, child := range allWindows { - windowTitle := strings.ToLower(axString(child, "AXTitle")) + baseName := strings.TrimSuffix(traceBase, filepath.Ext(traceBase)) + if baseName != traceBase { + for _, window := range allWindows { + child := window.Element + windowTitle := strings.ToLower(window.Title) if strings.Contains(windowTitle, baseName) { matchingWindows = append(matchingWindows, child) continue } - windowDoc := strings.ToLower(axString(child, "AXDocument")) + windowDoc := strings.ToLower(window.Document) if strings.Contains(windowDoc, baseName) { matchingWindows = append(matchingWindows, child) } @@ -426,8 +715,9 @@ func getPreferredTraceWindow(appAX uintptr, traceFileName string) uintptr { // buttons is almost certainly our trace window. if len(matchingWindows) == 0 { verboseLog("getPreferredTraceWindow: no title/doc match, scanning for windows with GPU trace UI elements") - for _, child := range allWindows { - title := axString(child, "AXTitle") + for _, window := range allWindows { + child := window.Element + title := window.Title // Skip windows that are clearly source editors (common extensions) titleLow := strings.ToLower(title) if isSourceEditorWindow(titleLow) { @@ -442,6 +732,9 @@ func getPreferredTraceWindow(appAX uintptr, traceFileName string) uintptr { if len(matchingWindows) > 0 { verboseLog("getPreferredTraceWindow: matched %d windows by GPU trace UI heuristic", len(matchingWindows)) } + if len(matchingWindows) == 1 { + uiIdentifiedTraceWindow = matchingWindows[0] + } } verboseLog("getPreferredTraceWindow: found %d windows matching %q", len(matchingWindows), traceFileName) @@ -455,32 +748,100 @@ func getPreferredTraceWindow(appAX uintptr, traceFileName string) uintptr { return matchingWindows[0] } - // Multiple matches - prefer windows with GPU trace UI (Replay button) + // Multiple matches - prefer a uniquely active profiling window. Do not + // choose an arbitrary untitled Summary window: it may belong to another + // trace and carry a stale Show Performance sentinel. + var activeWindows []uintptr for _, w := range matchingWindows { - title := axString(w, "AXTitle") - // Check for Replay button (fast shallow search) - replayBtn := findButtonBFS(w, "Replay", 500) - if replayBtn != 0 { - verboseLog("getPreferredTraceWindow: selected window %q (has Replay button)", title) - return w - } - // Check for Export button (indicates profiling data ready) - exportBtn := findButtonBFS(w, "Export", 500) - if exportBtn != 0 { - verboseLog("getPreferredTraceWindow: selected window %q (has Export button)", title) - return w - } - // Check for Show Performance button - showPerfBtn := findButtonBFS(w, "Show Performance", 500) - if showPerfBtn != 0 { - verboseLog("getPreferredTraceWindow: selected window %q (has Show Performance button)", title) - return w + if stopBtn := findButtonBFS(w, "Stop GPU workload", 500); stopBtn != 0 && IsElementEnabled(stopBtn) { + activeWindows = append(activeWindows, w) } } + if len(activeWindows) == 1 { + verboseLog("getPreferredTraceWindow: selected unique active GPU window %q", axString(activeWindows[0], "AXTitle")) + return activeWindows[0] + } + if len(activeWindows) > 1 { + verboseLog("getPreferredTraceWindow: %d active GPU windows are ambiguous", len(activeWindows)) + return 0 + } - // No window with trace UI found - return first match - verboseLog("getPreferredTraceWindow: no window with trace UI, using first match") - return matchingWindows[0] + verboseLog("getPreferredTraceWindow: multiple inactive GPU windows are ambiguous") + return 0 +} + +type xcodeAXWindow struct { + Element uintptr + Title string + Document string + X int + Y int + Width int + Height int +} + +func deduplicateAXWindows(elements []uintptr) []xcodeAXWindow { + windows := make([]xcodeAXWindow, 0, len(elements)) + for _, element := range elements { + x, y := axPosition(element) + width, height := axSize(element) + windows = append(windows, xcodeAXWindow{ + Element: element, + Title: axString(element, "AXTitle"), + Document: axString(element, "AXDocument"), + X: x, + Y: y, + Width: width, + Height: height, + }) + } + return deduplicateXcodeWindows(windows) +} + +func deduplicateXcodeWindows(windows []xcodeAXWindow) []xcodeAXWindow { + seenElements := make(map[uintptr]bool) + seenLogical := make(map[string]bool) + out := make([]xcodeAXWindow, 0, len(windows)) + for _, window := range windows { + if window.Element != 0 && seenElements[window.Element] { + continue + } + if window.Element != 0 { + seenElements[window.Element] = true + } + + title := strings.ToLower(strings.TrimSpace(window.Title)) + document := strings.ToLower(filepath.Clean(window.Document)) + if window.Document == "" { + document = "" + } + logical := fmt.Sprintf("%s\x00%s\x00%d,%d,%d,%d", + title, document, window.X, window.Y, window.Width, window.Height) + hasLogicalIdentity := title != "" || document != "" || + window.X != 0 || window.Y != 0 || window.Width != 0 || window.Height != 0 + if hasLogicalIdentity && seenLogical[logical] { + continue + } + if hasLogicalIdentity { + seenLogical[logical] = true + } + out = append(out, window) + } + return out +} + +func exactTraceWindows(windows []xcodeAXWindow, traceIdentity string) []uintptr { + if traceIdentity == "" || traceIdentity == "." { + return nil + } + var matches []uintptr + for _, window := range windows { + document := strings.ToLower(filepath.Clean(window.Document)) + if document != "." && (document == traceIdentity || strings.Contains(document, traceIdentity)) { + matches = append(matches, window.Element) + } + } + return matches } // isSourceEditorWindow returns true if the window title looks like a source code editor @@ -495,10 +856,9 @@ func isSourceEditorWindow(titleLower string) bool { return false } -// hasGPUTraceUI checks whether a window contains GPU trace UI elements -// (Replay, Profile, Export, or Show Performance buttons). +// hasGPUTraceUI checks whether a window contains GPU trace UI elements. func hasGPUTraceUI(windowAX uintptr) bool { - for _, name := range []string{"Replay", "Profile", "Export", "Show Performance"} { + for _, name := range gpuTraceStateButtonNames() { if btn := findButtonBFS(windowAX, name, 500); btn != 0 { return true } @@ -506,6 +866,17 @@ func hasGPUTraceUI(windowAX uintptr) bool { return false } +func gpuTraceStateButtonNames() []string { + return []string{ + "Stop GPU workload", + "Capture GPU workload", + "Replay", + "Profile", + "Export", + "Show Performance", + } +} + // validateTraceBundle checks whether a .gputrace bundle contains enough data // to be worth profiling. An empty capture (header-only MTSP file, ≤8 bytes) // means the original Metal capture recorded no GPU commands. @@ -591,50 +962,319 @@ func uniquePaths(paths []string) []string { } func waitForExportedTrace(ctx context.Context, candidatePaths []string, timeout time.Duration) (string, error) { + return waitForExportedTraceWithReader(ctx, candidatePaths, timeout, readExportTraceSignature) +} + +type exportCandidate struct { + Path string + Identity string + info os.FileInfo +} + +func canonicalExportCandidates(paths []string) []exportCandidate { + var candidates []exportCandidate + for _, path := range uniquePaths(paths) { + path = filepath.Clean(path) + resolved := path + if target, err := filepath.EvalSymlinks(path); err == nil { + resolved = filepath.Clean(target) + } + info, _ := os.Stat(path) + + duplicate := false + for _, candidate := range candidates { + if resolved == candidate.Identity || + (info != nil && candidate.info != nil && os.SameFile(info, candidate.info)) { + duplicate = true + break + } + } + if duplicate { + continue + } + candidates = append(candidates, exportCandidate{ + Path: path, + Identity: resolved, + info: info, + }) + } + return candidates +} + +type exportCandidateStability struct { + signature exportTraceSignature + samples int + set bool +} + +func waitForExportedTraceWithReader( + ctx context.Context, + candidatePaths []string, + timeout time.Duration, + readSignature func(string, string) (exportTraceSignature, error), +) (string, error) { deadline := time.Now().Add(timeout) var foundWithoutProfiler []string + var foundIncomplete []string + stability := make(map[string]exportCandidateStability) for { if err := checkAutomationCanceled(ctx); err != nil { return "", err } - for _, p := range candidatePaths { + for _, candidate := range canonicalExportCandidates(candidatePaths) { + p := candidate.Path info, err := os.Stat(p) if err != nil { continue } if !info.IsDir() { + delete(stability, candidate.Identity) + continue + } + profilerDir := findProfilerDir(p) + if profilerDir == "" { + foundWithoutProfiler = append(foundWithoutProfiler, p) + delete(stability, candidate.Identity) continue } - if findProfilerDir(p) != "" { - return p, nil + signature, err := readSignature(p, profilerDir) + if err != nil { + foundIncomplete = append(foundIncomplete, p) + delete(stability, candidate.Identity) + continue } - foundWithoutProfiler = append(foundWithoutProfiler, p) + state := stability[candidate.Identity] + if state.set && signature == state.signature { + state.samples++ + if state.samples >= 2 { + return p, nil + } + } else { + state.samples = 0 + } + state.signature = signature + state.set = true + stability[candidate.Identity] = state } if time.Now().After(deadline) { break } - if err := waitForAutomation(ctx, time.Second); err != nil { + if err := waitForAutomation(ctx, 250*time.Millisecond); err != nil { return "", err } } + if len(foundIncomplete) > 0 { + return "", fmt.Errorf("export profiler data did not stabilize with non-empty streamData: %s", strings.Join(uniquePaths(foundIncomplete), ", ")) + } if len(foundWithoutProfiler) > 0 { return "", fmt.Errorf("export wrote a bundle without .gpuprofiler_raw: %s; Xcode did not embed performance data", strings.Join(uniquePaths(foundWithoutProfiler), ", ")) } return "", fmt.Errorf("export did not write a perfdata bundle within %s; checked: %s", timeout.Round(time.Second), strings.Join(candidatePaths, ", ")) } -func windowMatchesTraceFile(window uintptr, traceFileName string) bool { - if traceFileName == "" { - return true +type exportTraceSignature struct { + Files int + Bytes int64 + StreamDataSize int64 +} + +type exportSheetState struct { + Filename string + DirectoryCandidates []string + SaveEnabled bool + GoToFolderSheetOpen bool + GoToFolderPath string +} + +func readExportSheetState(window uintptr) exportSheetState { + var state exportSheetState + if field := FindSaveAsTextField(window); field != 0 { + state.Filename = axString(field, "AXValue") + } + if save := findButtonBFS(window, "Save", 500); save != 0 { + state.SaveEnabled = IsElementEnabled(save) } - name := strings.ToLower(traceFileName) - title := strings.ToLower(axString(window, "AXTitle")) - if strings.Contains(title, name) { + goToSheet := findElementBounded(window, 600, func(element uintptr) bool { + return axString(element, "AXRole") == "AXSheet" && + axString(element, "AXIdentifier") == "GoToWindow" + }) + state.GoToFolderSheetOpen = goToSheet != 0 + if goToSheet != 0 { + if pathField := findElementBounded(goToSheet, 200, func(element uintptr) bool { + return axString(element, "AXRole") == "AXTextField" && + axString(element, "AXIdentifier") == "PathTextField" + }); pathField != 0 { + state.GoToFolderPath = strings.TrimSpace(axString(pathField, "AXValue")) + } + } + findElementBounded(window, 1000, func(element uintptr) bool { + role := axString(element, "AXRole") + subrole := axString(element, "AXSubrole") + identifier := strings.ToLower(axString(element, "AXIdentifier")) + description := strings.ToLower(axString(element, "AXDescription")) + // The GoToWindow input is a requested path, not evidence that the + // parent save panel committed that directory. + if identifier == "pathtextfield" { + return false + } + isLocation := role == "AXPopUpButton" || subrole == "AXPathButton" || + strings.Contains(identifier, "path") || strings.Contains(identifier, "location") || + strings.Contains(description, "where") || strings.Contains(description, "location") + // AXURL and AXDocument are useful full-path evidence even when Xcode + // does not identify the owning element as a location control. + for _, attribute := range []string{"AXURL", "AXDocument"} { + value := strings.TrimSpace(axString(element, attribute)) + if value != "" { + state.DirectoryCandidates = append(state.DirectoryCandidates, value) + } + } + if !isLocation { + return false + } + for _, attribute := range []string{"AXValue", "AXTitle"} { + value := strings.TrimSpace(axString(element, attribute)) + if value != "" { + state.DirectoryCandidates = append(state.DirectoryCandidates, value) + } + } + return false + }) + state.DirectoryCandidates = uniquePaths(state.DirectoryCandidates) + return state +} + +func exportSheetDirectoryMatches(state exportSheetState, directory string) bool { + want := filepath.Clean(directory) + for _, candidate := range state.DirectoryCandidates { + value := candidate + if strings.HasPrefix(value, "file://") { + if parsed, err := url.Parse(value); err == nil { + value = parsed.Path + } + } + if decoded, err := url.PathUnescape(value); err == nil { + value = decoded + } + if filepath.IsAbs(value) && filepath.Clean(value) == want { + return true + } + } + return false +} + +func needsDirectExportLocation(remainingPath string, state exportSheetState, directory string) bool { + return remainingPath != "" || !exportSheetDirectoryMatches(state, directory) +} + +func goToFolderNavigationComplete(state exportSheetState, directory string) bool { + return !state.GoToFolderSheetOpen && state.SaveEnabled && + exportSheetDirectoryMatches(state, directory) +} + +// goToFolderNavigationCompleteAfterExactEntry accepts a basename-only save +// panel location only after the caller observed the exact absolute path in the +// open Go To Folder field and then committed it. The ordered proof +// distinguishes identical basenames such as /Users/tmc/tmp and /private/tmp. +func goToFolderNavigationCompleteAfterExactEntry(state exportSheetState, directory string) bool { + if goToFolderNavigationComplete(state, directory) { return true } - doc := strings.ToLower(axString(window, "AXDocument")) - return strings.Contains(doc, name) + if state.GoToFolderSheetOpen || !state.SaveEnabled { + return false + } + wantBase := filepath.Base(filepath.Clean(directory)) + for _, candidate := range state.DirectoryCandidates { + if !filepath.IsAbs(candidate) && filepath.Clean(candidate) == wantBase { + return true + } + } + return false +} + +func goToFolderConfirmationReady(state exportSheetState, directory string) bool { + if !state.GoToFolderSheetOpen { + return exportSheetDirectoryMatches(state, directory) + } + return filepath.IsAbs(state.GoToFolderPath) && + filepath.Clean(state.GoToFolderPath) == filepath.Clean(directory) +} + +func waitForExportDirectoryState(ctx context.Context, window uintptr, directory string, timeout time.Duration) (exportSheetState, error) { + deadline := time.Now().Add(timeout) + var state exportSheetState + for { + state = readExportSheetState(window) + if exportSheetDirectoryMatches(state, directory) { + return state, nil + } + if time.Now().After(deadline) { + return state, fmt.Errorf("save sheet did not expose requested directory %q", directory) + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return state, err + } + } +} + +func formatExportSheetState(state exportSheetState) string { + return fmt.Sprintf("filename=%q directory_candidates=%q save_enabled=%t go_to_folder_open=%t go_to_folder_path=%q", + state.Filename, state.DirectoryCandidates, state.SaveEnabled, + state.GoToFolderSheetOpen, state.GoToFolderPath) +} + +func readExportTraceSignature(bundle, profilerDir string) (exportTraceSignature, error) { + streamInfo, err := os.Stat(filepath.Join(profilerDir, "streamData")) + if err != nil { + return exportTraceSignature{}, err + } + if streamInfo.Size() == 0 { + return exportTraceSignature{}, fmt.Errorf("streamData is empty") + } + + var signature exportTraceSignature + signature.StreamDataSize = streamInfo.Size() + err = filepath.WalkDir(bundle, func(path string, entry os.DirEntry, walkErr error) error { + if walkErr != nil { + return walkErr + } + if entry.IsDir() { + return nil + } + info, err := entry.Info() + if err != nil { + return err + } + signature.Files++ + signature.Bytes += info.Size() + return nil + }) + if err != nil { + return exportTraceSignature{}, err + } + return signature, nil +} + +func verifyExportTraceIdentity(inputPath, outputPath string) error { + input, err := gputraceTrace.ReadMetadata(inputPath) + if err != nil { + return fmt.Errorf("read input trace identity: %w", err) + } + output, err := gputraceTrace.ReadMetadata(outputPath) + if err != nil { + return fmt.Errorf("read exported trace identity: %w", err) + } + if input.UUID == "" || output.UUID == "" { + return fmt.Errorf("trace identity is missing (input UUID %q, exported UUID %q)", input.UUID, output.UUID) + } + if input.UUID != output.UUID { + return fmt.Errorf("exported trace UUID %s does not match requested trace UUID %s", output.UUID, input.UUID) + } + return nil +} + +func windowMatchesTraceFile(window uintptr, traceFileName string) bool { + return selectionForWindow(traceFileName, window).Bound } func clickReplayButton(windowAX uintptr) error { @@ -770,6 +1410,74 @@ func showPerformanceBeforeExport(windowAX uintptr) (bool, error) { return true, nil } +// startPerformanceProfile presses Profile when Show Performance leaves a +// popover open. Some Xcode versions transition straight to the populated +// Performance view instead, in which case there is no Profile button to +// press. Both routes remain bound to the requested trace window. +func startPerformanceProfile(ctx context.Context, window uintptr) error { + deadline := time.Now().Add(10 * time.Second) + for { + profile := findButtonBFS(window, "Profile", 5000) + if profile != 0 { + if !IsElementEnabled(profile) { + return fmt.Errorf("Profile button is disabled") + } + var pid int32 + if axUIElementGetPid(profile, &pid) != kAXErrorSuccess || pid == 0 { + return fmt.Errorf("read Profile button owner") + } + var windowPID int32 + if axUIElementGetPid(window, &windowPID) != kAXErrorSuccess || pid != windowPID { + return fmt.Errorf("Profile button is not owned by the bound trace window") + } + fmt.Fprintln(xcodeProfileStatusWriter(), " Starting performance profile...") + if err := axPressWithFallbackWindow(profile, window); err != nil { + return fmt.Errorf("press Profile: %w", err) + } + return nil + } + if hasPerformanceView(window) { + verboseLog("startPerformanceProfile: Show Performance transitioned directly to Performance") + return nil + } + if time.Now().After(deadline) { + return fmt.Errorf("Show Performance exposed neither Profile nor a populated Performance view") + } + if err := waitForAutomation(ctx, 100*time.Millisecond); err != nil { + return err + } + } +} + +// waitForPerformanceExportReady waits for the bound Performance view after +// Show Performance. Export itself opens File exactly once: probing that +// stateful menu and then reopening it can change the observed state. +func waitForPerformanceExportReady(ctx context.Context, appAX uintptr, identity xcodeProcessIdentity, traceFile string, timeout time.Duration) (uintptr, error) { + deadline := time.Now().Add(timeout) + var lastErr error + for { + if err := checkAutomationCanceled(ctx); err != nil { + return 0, err + } + bound, err := xcodeIdentityForAX(appAX) + if err != nil || bound.PID != identity.PID || filepath.Clean(bound.AppPath) != filepath.Clean(identity.AppPath) { + return 0, fmt.Errorf("lost Xcode binding while waiting for Profile: want PID %d app %s", identity.PID, identity.AppPath) + } + window := getPreferredTraceWindow(appAX, traceFile) + if window != 0 && selectionForWindow(traceFile, window).Bound && hasPerformanceView(window) { + return window, nil + } else { + lastErr = fmt.Errorf("bound Performance window is unavailable") + } + if time.Now().After(deadline) { + return 0, fmt.Errorf("timed out waiting for Performance view: %w", lastErr) + } + if err := waitForAutomation(ctx, 250*time.Millisecond); err != nil { + return 0, err + } + } +} + // targetedShowPerformanceFound is a found-only marker for hasShowPerformance. // That traversal confirms the button is present but does not return an AX // element handle, so callers must not pass this value to IsElementEnabled or @@ -780,6 +1488,37 @@ func isTargetedShowPerformanceFound(button uintptr) bool { return button == targetedShowPerformanceFound } +// selectSummaryAfterReplay selects the Summary row once Xcode has expanded the +// Debug Navigator for a replay. The navigator is not present before replay, +// so callers must retry until this helper finds it. The window is already +// bound to the requested trace; no global window search is performed here. +func selectSummaryAfterReplay(ctx context.Context, window uintptr) (bool, error) { + row := findOutlineRowByName(window, "Summary") + if row == 0 { + return false, nil + } + if isElementSelected(row) || isTabSelected(row) || strings.EqualFold(getCurrentTab(window), "Summary") { + return true, nil + } + + try := func(action string) bool { + if err := axAction(row, action); err != nil { + return false + } + return waitForSelectedControl(ctx, window, row, "Summary", 2*time.Second) == nil + } + if try("AXOpen") || try("AXPress") { + return true, nil + } + if selectElement(row) && doubleClickElement(row) == nil && waitForSelectedControl(ctx, window, row, "Summary", 2*time.Second) == nil { + return true, nil + } + if doubleClickElement(row) == nil && waitForSelectedControl(ctx, window, row, "Summary", 2*time.Second) == nil { + return true, nil + } + return true, fmt.Errorf("select Summary after replay: no selectable Summary row") +} + func waitForReplayComplete(ctx context.Context, appAX uintptr, traceFileName string, initialWindowAX uintptr, timeout time.Duration) error { start := time.Now() currentWindow := initialWindowAX @@ -794,6 +1533,8 @@ func waitForReplayComplete(ctx context.Context, appAX uintptr, traceFileName str // Returns (button, xcodeRunning) // Note: depth of 2000 required for deep UI hierarchies (e.g., Show Performance in summary panel) const buttonSearchDepth = 5000 + var targetPID int32 + _ = axUIElementGetPid(appAX, &targetPID) // tryWindowForButton checks a single window for a button (or Show Performance via targeted traversal). tryWindowForButton := func(w uintptr, name string) uintptr { @@ -822,16 +1563,54 @@ func waitForReplayComplete(ctx context.Context, appAX uintptr, traceFileName str } } // 3. Re-fetch Xcode app and search all windows (handles stale appAX and title changes) - freshApp, err := FindXcodeApp() - if err != nil { - verboseLog("waitForReplayComplete: failed to re-fetch Xcode app: %v", err) - return 0, false + freshApp := uintptr(0) + crashScope := xcodeCrashScopeFromContext(ctx) + targetAppPath := "" + if targetPID != 0 { + targetAppPath = xcodeProcessPath(int(targetPID)) + } + if targetAppPath != "" && + (crashScope == nil || filepath.Clean(targetAppPath) == crashScope.appPath) { + freshApp = axCreateApplication(targetPID) + } + if freshApp == 0 { + verboseLog("waitForReplayComplete: failed to re-fetch target Xcode PID %d; checking exact-app replacements", targetPID) + if crashScope != nil { + crashScope.refreshProcesses() + for _, identity := range xcodeProcessesForApp(crashScope.appPath) { + replacementApp := axCreateApplication(int32(identity.PID)) + if replacementApp == 0 { + continue + } + replacementWindow := getPreferredTraceWindow(replacementApp, traceFileName) + if replacementWindow == 0 { + cfRelease(replacementApp) + continue + } + verboseLog("waitForReplayComplete: rebound exact trace window to Xcode PID %d", identity.PID) + crashScope.observe(identity) + targetPID = int32(identity.PID) + currentWindow = replacementWindow + freshApp = replacementApp + break + } + } + if freshApp == 0 { + return 0, false + } } + defer cfRelease(freshApp) consecutiveXcodeFailures = 0 allWindows := GetAllWindows(freshApp) - // First pass: title-matched windows. Second pass: all windows. - for pass := range 2 { + // When a trace identity was supplied, never fall through to an + // unrelated untitled GPU window. A stale completed Summary window may + // expose the same app-global controls and Show Performance sentinel. + passes := 1 + if traceFileName == "" { + passes = 2 + } + for pass := range passes { for _, w := range allWindows { if pass == 0 && !windowMatchesTraceFile(w, traceFileName) { continue @@ -853,6 +1632,14 @@ func waitForReplayComplete(ctx context.Context, appAX uintptr, traceFileName str if !xcodeRunning { consecutiveXcodeFailures++ if consecutiveXcodeFailures >= maxXcodeFailures { + if crashScope := xcodeCrashScopeFromContext(ctx); crashScope != nil { + crashScope.refreshProcesses() + if crashScope.crashSuspected() { + if crashErr := waitForXcodeCrashReport(ctx, xcodeCrashReportGrace); crashErr != nil { + return 0, crashErr + } + } + } return 0, fmt.Errorf("Xcode exited while waiting for replay completion") } } @@ -943,18 +1730,30 @@ func waitForReplayComplete(ctx context.Context, appAX uintptr, traceFileName str // Now wait for profiling to complete lastStatus := "" + summarySelected := false for time.Since(start) < timeout { if err := checkAutomationCanceled(ctx); err != nil { return err } // Check for completion indicators (only in target window): - // 1. Show Performance button appears (most reliable - profiling complete, ready to view) - // Use targeted traversal via hasShowPerformance (same as check-status) for reliability - if currentWindow != 0 && hasShowPerformance(currentWindow) { - verboseLog("waitForReplayComplete: Show Performance button found (targeted traversal) - complete") + // 1. The Summary view's Show Performance button or the loaded + // Performance view's controls appear. Either means profiling completed. + if currentWindow != 0 && hasPerformanceData(currentWindow) { + verboseLog("waitForReplayComplete: Performance data controls found - complete") return nil } + if !summarySelected && currentWindow != 0 { + found, err := selectSummaryAfterReplay(ctx, currentWindow) + if err != nil { + return err + } + if found { + summarySelected = true + verboseLog("waitForReplayComplete: selected Summary in bound trace window") + continue + } + } // Also try findButtonOrFail as fallback (searches all windows with deeper BFS) // findButton can return targetedShowPerformanceFound for this button. // That sentinel is not an AX element, so skip IsElementEnabled here. @@ -992,8 +1791,8 @@ func waitForReplayComplete(ctx context.Context, appAX uintptr, traceFileName str return err } // Use targeted traversal first - if currentWindow != 0 && hasShowPerformance(currentWindow) { - verboseLog("waitForReplayComplete: Replay enabled, Show Performance available (targeted) - complete") + if currentWindow != 0 && hasPerformanceData(currentWindow) { + verboseLog("waitForReplayComplete: Replay enabled, Performance data controls found - complete") return nil } showPerfBtn, err = findButtonOrFail("Show Performance") @@ -1147,30 +1946,41 @@ func exportTrace(ctx context.Context, appAX, windowAX uintptr, outputPath string if err := checkAutomationCanceled(ctx); err != nil { return err } + identity, err := xcodeIdentityForAX(appAX) + if err != nil { + return fmt.Errorf("establish export Xcode identity: %w", err) + } + var windowPID int32 + if axUIElementGetPid(windowAX, &windowPID) != kAXErrorSuccess || int(windowPID) != identity.PID { + return fmt.Errorf("export window is not owned by bound Xcode PID %d app %s", identity.PID, identity.AppPath) + } status := xcodeProfileStatusWriter() - activateXcodeQuick(ctx) axAction(windowAX, "AXRaise") time.Sleep(300 * time.Millisecond) + if sheet := findElement(windowAX, func(el uintptr) bool { + return axString(el, "AXRole") == "AXSheet" + }); sheet != 0 { + return fmt.Errorf("selected export window already has an open sheet; refusing to reuse stale UI") + } // Try clicking Export button in Summary panel first exportBtn := FindExportButton(windowAX) if exportBtn != 0 { + if !IsElementEnabled(exportBtn) { + return fmt.Errorf("Export button is disabled; Xcode workload is not finalized") + } fmt.Fprintln(status, " Found Export button in Summary panel") if err := axPressWithFallback(exportBtn); err != nil { fmt.Fprintf(status, " Warning: Failed to click Export button: %v\n", err) } } else { - // Fall back to menu - if freshApp, err := FindXcodeApp(); err == nil && freshApp != 0 { - appAX = freshApp + bound, err := xcodeIdentityForAX(appAX) + if err != nil || bound.PID != identity.PID || + filepath.Clean(bound.AppPath) != filepath.Clean(identity.AppPath) { + return fmt.Errorf("bound Xcode identity changed while checking File > Export") } - if collectProfileOpts.debug || collectProfileOpts.verbose { - if err := debugCheckExportMenu(appAX); err != nil { - fmt.Fprintf(os.Stderr, " Debug: Export menu check failed: %v\n", err) - } - } - if err := ClickMenuItem(appAX, []string{"File", "Export..."}); err != nil { - return fmt.Errorf("failed to click Export menu: %w", err) + if err := clickFileExportWhenEnabled(ctx, appAX, windowAX, 2*time.Minute); err != nil { + return fmt.Errorf("click Export menu: %w", err) } } @@ -1180,32 +1990,24 @@ func exportTrace(ctx context.Context, appAX, windowAX uintptr, outputPath string } // Refresh app reference since the UI might have changed - freshApp, err := FindXcodeApp() + freshApp, err := reacquireXcodeApp(identity) if err != nil { - return fmt.Errorf("Xcode not accessible after clicking Export: %w", err) + return fmt.Errorf("bound Xcode not accessible after clicking Export: %w", err) } defer cfRelease(freshApp) - // Search ALL windows for Save button (sheet might be in any window) - var saveWindow uintptr + // The export sheet must descend from the selected trace window. Searching + // every Xcode window can bind a stale sheet from another trace. sheetFound := false for i := 0; i < 30; i++ { if err := checkAutomationCanceled(ctx); err != nil { return err } - windows := GetAllWindows(freshApp) - for _, w := range windows { - // Detect export sheet by looking for Save button or AXSheet role - sheet := findElement(w, func(el uintptr) bool { - return axString(el, "AXRole") == "AXSheet" - }) - if sheet != 0 { - sheetFound = true - saveWindow = w - break - } - } - if sheetFound { + sheet := findElement(windowAX, func(el uintptr) bool { + return axString(el, "AXRole") == "AXSheet" + }) + if sheet != 0 { + sheetFound = true break } if err := waitForAutomation(ctx, 500*time.Millisecond); err != nil { @@ -1215,29 +2017,16 @@ func exportTrace(ctx context.Context, appAX, windowAX uintptr, outputPath string if !sheetFound { if collectProfileOpts.debug { - windows := GetAllWindows(freshApp) - fmt.Fprintf(os.Stderr, " Debug: Found %d windows\n", len(windows)) - for i, w := range windows { - title := axString(w, "AXTitle") - fmt.Fprintf(os.Stderr, " Debug: Window %d: %q\n", i+1, title) - } + fmt.Fprintf(os.Stderr, " Debug: selected window title=%q document=%q\n", + axString(windowAX, "AXTitle"), axString(windowAX, "AXDocument")) } - return fmt.Errorf("export sheet did not appear (Save button not found)") + return fmt.Errorf("export sheet did not appear under the selected trace window") } fmt.Fprintln(status, " Export sheet detected") - // Use the window containing the Save button for subsequent operations - windowAX = saveWindow - // Helper to find element across all windows (using freshApp from above) - findInAllWindows := func(finder func(uintptr) uintptr) uintptr { - windows := GetAllWindows(freshApp) - for _, w := range windows { - if el := finder(w); el != 0 { - return el - } - } - return 0 + findInExportWindow := func(finder func(uintptr) uintptr) uintptr { + return finder(windowAX) } // Check "Embed performance data" checkbox if available and enabled @@ -1271,65 +2060,71 @@ func exportTrace(ctx context.Context, appAX, windowAX uintptr, outputPath string DebugTextFields(windowAX) } - // Try to navigate to the output directory using the path popup button - navigatedToDir := false + // Try the shallow path popup first. Nested file-browser crawling is + // intentionally avoided because large AX trees can stall for minutes. remainingPath := "" if outputDir != "" && outputDir != "." { fmt.Fprintf(status, " Navigating to directory: %s\n", outputDir) - // First try via path popup (more reliable than Cmd+Shift+G) var popupErr error remainingPath, popupErr = navigateViaPathPopup(windowAX, outputDir) if popupErr != nil { verboseLog("exportTrace: path popup navigation failed: %v", popupErr) - // Fall back to Cmd+Shift+G - if err := NavigateToFolderInSaveDialog(windowAX, outputDir); err != nil { - verboseLog("exportTrace: Cmd+Shift+G navigation failed: %v", err) - fmt.Fprintln(status, " Note: Directory navigation failed, using default location") - } else { - navigatedToDir = true - } - } else { - navigatedToDir = true - if remainingPath != "" { - verboseLog("exportTrace: navigated partially, remaining path: %s", remainingPath) - } else { - verboseLog("exportTrace: navigated to directory successfully") - } + } else if remainingPath != "" { + verboseLog("exportTrace: navigated partially, remaining path: %s", remainingPath) } } - // If there's a remaining path (couldn't fully navigate), try Cmd+Shift+G as final fallback - // Note: putting "/" in filename creates ":"-named files due to macOS HFS legacy behavior - if remainingPath != "" { - fmt.Fprintf(status, " Partial navigation, using Cmd+Shift+G to navigate to: %s\n", outputDir) - // Ensure directory exists before trying to navigate - if err := os.MkdirAll(outputDir, 0755); err != nil { - verboseLog("exportTrace: failed to create output directory: %v", err) - } - // Try Cmd+Shift+G to navigate to full path + // A popup result is not proof of the destination: the control may display + // only "tmp" after partial navigation. Unless the sheet exposes the exact + // absolute directory, use the bounded direct-location fallback. + directoryState := readExportSheetState(windowAX) + directoryVerifiedByExactEntry := false + if needsDirectExportLocation(remainingPath, directoryState, outputDir) { + fmt.Fprintf(status, " Using direct location for: %s\n", outputDir) if err := NavigateToFolderInSaveDialog(windowAX, outputDir); err != nil { - verboseLog("exportTrace: Cmd+Shift+G fallback also failed: %v", err) - fmt.Fprintf(status, " Warning: Could not navigate to %s, file may save to wrong location\n", outputDir) - } else { - navigatedToDir = true - remainingPath = "" - fmt.Fprintln(status, " Successfully navigated via Cmd+Shift+G") + return fmt.Errorf("establish export directory %s: %w; sheet state: %s", + outputDir, err, formatExportSheetState(readExportSheetState(windowAX))) } + directoryVerifiedByExactEntry = true } + if directoryVerifiedByExactEntry { + if !goToFolderNavigationCompleteAfterExactEntry(readExportSheetState(windowAX), outputDir) { + return fmt.Errorf("export directory lost exact-entry proof; sheet state: %s", + formatExportSheetState(readExportSheetState(windowAX))) + } + } else { + directoryState, err = waitForExportDirectoryState(ctx, windowAX, outputDir, 2*time.Second) + if err != nil { + return fmt.Errorf("export directory was not established: %w; sheet state: %s", + err, formatExportSheetState(directoryState)) + } + } + fmt.Fprintf(status, " Verified export directory: %s\n", outputDir) // Set just the filename (never include path prefix - macOS converts "/" to ":") fmt.Fprintf(status, " Setting filename: %s\n", outputName) - saveNameField := findInAllWindows(FindSaveAsTextField) + saveNameField := findInExportWindow(FindSaveAsTextField) if saveNameField != 0 { - if err := axSetValue(saveNameField, outputName); err != nil { - fmt.Fprintf(status, " Warning: SetValue failed: %v (using default filename)\n", err) - } else if collectProfileOpts.debug { - fmt.Fprintln(os.Stderr, " [DEBUG] Set filename via AX (saveAsNameTextField)") + if err := setSaveName(saveNameField, outputName); err != nil { + return err + } + if collectProfileOpts.debug { + fmt.Fprintln(os.Stderr, " [DEBUG] Set and verified filename via AX (saveAsNameTextField)") } } else { - fmt.Fprintln(status, " Warning: saveAsNameTextField not found (using default filename)") + return fmt.Errorf("saveAsNameTextField not found") } time.Sleep(300 * time.Millisecond) + finalSheetState := readExportSheetState(windowAX) + directoryVerified := exportSheetDirectoryMatches(finalSheetState, outputDir) + if directoryVerifiedByExactEntry { + directoryVerified = goToFolderNavigationCompleteAfterExactEntry(finalSheetState, outputDir) + } + if finalSheetState.Filename != outputName || !directoryVerified { + return fmt.Errorf("export destination verification failed: requested_directory=%q requested_filename=%q; sheet state: %s", + outputDir, outputName, formatExportSheetState(finalSheetState)) + } + fmt.Fprintf(status, " Verified export filename: %s\n", outputName) // Debug: dump the export sheet state so we can see exactly what's happening if collectProfileOpts.debug { @@ -1346,19 +2141,12 @@ func exportTrace(ctx context.Context, appAX, windowAX uintptr, outputPath string } if !IsElementEnabled(saveBtn) { - // Save disabled — usually means a child sheet (e.g. Go to Folder) is still - // open. Try dismissing any lingering sheets and re-querying. - verboseLog("exportTrace: Save disabled, checking for lingering child sheets") - dismissGoToFolderSheet(windowAX) - time.Sleep(300 * time.Millisecond) - saveBtn = findSaveButtonInSheet() - if saveBtn == 0 || !IsElementEnabled(saveBtn) { - if collectProfileOpts.debug { - fmt.Fprintln(os.Stderr, " [DEBUG] Export sheet state (Save disabled):") - dumpExportSheetState(windowAX) - } - return fmt.Errorf("Save button disabled in export sheet") + if collectProfileOpts.debug { + fmt.Fprintln(os.Stderr, " [DEBUG] Export sheet state (Save disabled):") + dumpExportSheetState(windowAX) } + return fmt.Errorf("Save button disabled in export sheet: %s", + formatExportSheetState(readExportSheetState(windowAX))) } // Click Save button @@ -1374,28 +2162,136 @@ func exportTrace(ctx context.Context, appAX, windowAX uintptr, outputPath string fmt.Fprintln(status, " Confirmed replacement") } - // Wait for export to complete — GPU trace exports can be large and slow - fmt.Fprintln(status, " Waiting for export to write...") - if err := waitForAutomation(ctx, 5*time.Second); err != nil { + if err := waitForExportSheetDismissed(ctx, windowAX, 5*time.Second); err != nil { return err } + fmt.Fprintln(status, " Export accepted; assembling bundle...") // Check if file was saved to expected location if _, err := os.Stat(outputPath); err == nil { return nil // File found at expected path } - // If we didn't navigate, the file is likely in an alternate location - // The caller will check alternate locations and copy if needed - if !navigatedToDir { - verboseLog("exportTrace: file not at %s, may be in Xcode's default export location", outputPath) - } - // Return nil to let caller handle searching alternate locations // Caller is responsible for finding and copying the file return nil } +// maxFileExportProbes bounds how many times File is opened while waiting for +// Export to become enabled, independent of the caller's timeout. +const maxFileExportProbes = 20 + +// fileExportProbeDelay returns how long to wait before probe attempt+1, given +// that attempt has just failed with Export disabled. Attempts are 1-based. +// +// The delay doubles from 500ms to a ceiling of 8s. Over a two-minute wait that +// is 19 probes rather than the 240 a fixed 500ms interval produced. +func fileExportProbeDelay(attempt int) time.Duration { + const ( + base = 500 * time.Millisecond + max = 8 * time.Second + ) + delay := base + for range attempt - 1 { + delay *= 2 + if delay >= max { + return max + } + } + return delay +} + +// clickFileExportWhenEnabled retries a single File > Export action while +// Xcode finishes preparing performance data. Each attempt opens and closes +// File once; the successful attempt presses Export exactly once. +// +// Opening the menu is the only way to read whether Export is enabled, so every +// probe is also a UI action: the menu bar opens, takes key focus, and closes. +// This polled at a fixed 500ms, so a large trace that kept Export disabled for +// the full two-minute window flashed the File menu about 240 times and made the +// machine unusable -- keystrokes went to the menu instead of to the user's +// other applications. Waiting is not free when the wait is performed by +// touching the UI, so back off and cap the attempts. +func clickFileExportWhenEnabled(ctx context.Context, appAX, windowAX uintptr, timeout time.Duration) error { + // Xcode's menu bar is only actionable while Xcode is the frontmost + // application. Without this the loop probed a menu that could not open, + // Export stayed disabled for the whole window, and the wait ended only when + // the user focused Xcode by hand -- the automation appeared to be working + // and was in fact waiting on a person. Activating is itself a focus change, + // so it happens once per backed-off attempt, not on a fast timer. + var windowPID int32 + if axUIElementGetPid(windowAX, &windowPID) != kAXErrorSuccess || windowPID == 0 { + return fmt.Errorf("read owning process of the export window") + } + + deadline := time.Now().Add(timeout) + var lastErr error + for attempt := 1; ; attempt++ { + if !collectProfileOpts.background { + if err := activateProcessPID(windowPID); err != nil { + verboseLog("clickFileExportWhenEnabled: activate PID %d: %v", windowPID, err) + } + } + err := clickMenuItemForWindow(appAX, windowAX, []string{"File", "Export..."}) + if err == nil { + return nil + } + if !strings.Contains(err.Error(), "menu item 'Export") || !strings.Contains(err.Error(), "is disabled") { + return err + } + lastErr = err + if attempt >= maxFileExportProbes { + return fmt.Errorf("File > Export still disabled after %d probes: %w", attempt, lastErr) + } + if time.Now().After(deadline) { + return fmt.Errorf("timed out waiting for File > Export: %w", lastErr) + } + if err := waitForAutomation(ctx, fileExportProbeDelay(attempt)); err != nil { + return err + } + } +} + +func setSaveName(field uintptr, name string) error { + for attempt := 0; attempt < 3; attempt++ { + if err := axSetValue(field, name); err != nil { + if attempt == 2 { + return fmt.Errorf("set export filename: %w", err) + } + continue + } + axAction(field, "AXConfirm") + time.Sleep(150 * time.Millisecond) + if got := axString(field, "AXValue"); got == name { + return nil + } + } + return fmt.Errorf("export filename did not update to %q (still %q)", name, axString(field, "AXValue")) +} + +func waitForExportSheetDismissed(ctx context.Context, window uintptr, timeout time.Duration) error { + deadline := time.Now().Add(timeout) + for { + sheet := findElement(window, func(el uintptr) bool { + return axString(el, "AXRole") == "AXSheet" && + (axString(el, "AXIdentifier") == "save-panel" || axString(el, "AXDescription") == "export") + }) + if sheet == 0 { + return nil + } + if time.Now().After(deadline) { + save := findButtonBFS(sheet, "Save", 500) + if save != 0 && IsElementEnabled(save) { + return fmt.Errorf("export save sheet is still open with Save enabled") + } + return fmt.Errorf("export save sheet did not dismiss") + } + if err := waitForAutomation(ctx, 200*time.Millisecond); err != nil { + return err + } + } +} + func pressReplaceIfPresent(ctx context.Context, windowAX uintptr, timeout time.Duration) (bool, error) { deadline := time.Now().Add(timeout) for { @@ -1431,7 +2327,7 @@ func pressReplaceIfPresent(ctx context.Context, windowAX uintptr, timeout time.D func navigateViaPathPopup(windowAX uintptr, targetPath string) (remainingPath string, err error) { // Look for a path control or popup button that shows the current location // Common identifiers: "Where:" popup, path bar, location dropdown - pathPopup := findElement(windowAX, func(el uintptr) bool { + pathPopup := findElementBounded(windowAX, 600, func(el uintptr) bool { role := axString(el, "AXRole") if role == "AXPopUpButton" { // Check if this is the "Where:" location popup @@ -1448,7 +2344,7 @@ func navigateViaPathPopup(windowAX uintptr, targetPath string) (remainingPath st if pathPopup == 0 { // Try to find any popup button that might be the location selector - pathPopup = findElement(windowAX, func(el uintptr) bool { + pathPopup = findElementBounded(windowAX, 600, func(el uintptr) bool { role := axString(el, "AXRole") subrole := axString(el, "AXSubrole") return role == "AXPopUpButton" && subrole == "AXPathButton" @@ -1510,7 +2406,7 @@ func navigateViaPathPopup(windowAX uintptr, targetPath string) (remainingPath st } var allMenuItems []menuItemRef if popupMenu != 0 { - findElement(popupMenu, func(el uintptr) bool { + findElementBounded(popupMenu, 300, func(el uintptr) bool { role := axString(el, "AXRole") if role == "AXMenuItem" { title := axString(el, "AXTitle") @@ -1567,16 +2463,12 @@ func navigateViaPathPopup(windowAX uintptr, targetPath string) (remainingPath st remainingParts := pathParts[i+1:] if len(remainingParts) > 0 { verboseLog("navigateViaPathPopup: remaining path components: %v", remainingParts) - // Try file browser navigation first (may work for some dialogs) - if err := navigateThroughFileBrowser(windowAX, remainingParts); err != nil { - verboseLog("navigateViaPathPopup: file browser navigation failed: %v", err) - // Return the remaining path - caller will try Cmd+Shift+G as fallback - remaining := strings.Join(remainingParts, "/") - verboseLog("navigateViaPathPopup: returning remaining path %q for caller fallback", remaining) - return remaining, nil - } - // File browser navigation succeeded - return "", nil + // Do not crawl the file browser for nested components. Large + // save-panel AX trees can make that search take minutes, and a + // double-click does not prove the location changed. Return the + // remainder so the caller uses the bounded direct-location + // fallback. + return strings.Join(remainingParts, "/"), nil } return "", nil // We clicked something and no remaining parts } @@ -1612,6 +2504,19 @@ func navigateViaPathPopup(windowAX uintptr, targetPath string) (remainingPath st return "", fmt.Errorf("could not find 'Other...' option in path popup (available: %v)", menuItemTitles) } +func findElementBounded(root uintptr, maxVisit int, match func(uintptr) bool) uintptr { + queue := []uintptr{root} + for visited := 0; len(queue) > 0 && visited < maxVisit; visited++ { + element := queue[0] + queue = queue[1:] + if match(element) { + return element + } + queue = append(queue, axChildren(element)...) + } + return 0 +} + // navigateThroughFileBrowser navigates through folders in a save dialog's file browser. // It finds folders by name in the file list (table/outline view) and double-clicks to open them. func navigateThroughFileBrowser(windowAX uintptr, folders []string) error { diff --git a/cmd/gputrace/cmd/collect_xcode_profile_run_test.go b/cmd/gputrace/cmd/collect_xcode_profile_run_test.go index 7877f189..b1f31401 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_run_test.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_run_test.go @@ -4,14 +4,101 @@ package cmd import ( "context" + "encoding/json" "errors" "os" "path/filepath" "strings" "testing" "time" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/tracebundle" ) +func writeProfiledTraceBundle(t *testing.T) string { + t.Helper() + bundle := filepath.Join(t.TempDir(), "trace-perfdata.gputrace") + profilerDir := filepath.Join(bundle, "trace.gputrace.gpuprofiler_raw") + if err := os.MkdirAll(profilerDir, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(bundle, "capture"), []byte("MTSP capture data"), 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(profilerDir, "streamData"), []byte("profiler data"), 0o644); err != nil { + t.Fatal(err) + } + return bundle +} + +func TestRunCollectXcodeProfileReusesEmbeddedPerformanceData(t *testing.T) { + oldJSON := collectProfileOpts.json + oldOutput := collectProfileOpts.output + oldHook := xcodeProfileAutomationStartHook + t.Cleanup(func() { + collectProfileOpts.json = oldJSON + collectProfileOpts.output = oldOutput + xcodeProfileAutomationStartHook = oldHook + }) + + bundle := writeProfiledTraceBundle(t) + automationStarted := false + xcodeProfileAutomationStartHook = func() { + automationStarted = true + } + collectProfileOpts.json = false + collectProfileOpts.output = "" + + stdout, err := captureStdout(t, func() error { + return runCollectXcodeProfileFull(&cobra.Command{}, []string{bundle}) + }) + if err != nil { + t.Fatalf("runCollectXcodeProfileFull: %v", err) + } + if automationStarted { + t.Fatal("Xcode automation started for a profiled trace") + } + if !strings.Contains(stdout, "Performance data already embedded; verified non-empty streamData.") { + t.Fatalf("stdout lacks verification result:\n%s", stdout) + } + if !strings.Contains(stdout, "Using existing trace: "+bundle) { + t.Fatalf("stdout lacks reused path:\n%s", stdout) + } +} + +func TestRunCollectXcodeProfileReusedJSON(t *testing.T) { + oldJSON := collectProfileOpts.json + oldOutput := collectProfileOpts.output + oldHook := xcodeProfileAutomationStartHook + t.Cleanup(func() { + collectProfileOpts.json = oldJSON + collectProfileOpts.output = oldOutput + xcodeProfileAutomationStartHook = oldHook + }) + + bundle := writeProfiledTraceBundle(t) + xcodeProfileAutomationStartHook = func() { + t.Fatal("Xcode automation started for a profiled trace") + } + collectProfileOpts.json = true + collectProfileOpts.output = "" + + stdout, err := captureStdout(t, func() error { + return runCollectXcodeProfileFull(&cobra.Command{}, []string{bundle}) + }) + if err != nil { + t.Fatalf("runCollectXcodeProfileFull: %v", err) + } + var got xcodeProfileActionOutput + if err := json.Unmarshal([]byte(stdout), &got); err != nil { + t.Fatalf("decode JSON: %v\n%s", err, stdout) + } + if !got.Success || !got.Reused || got.Action != "run" || got.Input != bundle || got.Output != bundle { + t.Fatalf("JSON output = %+v", got) + } +} + func TestWaitForExportedTraceRequiresProfilerData(t *testing.T) { dir := t.TempDir() bundle := filepath.Join(dir, "trace-perfdata.gputrace") @@ -31,11 +118,11 @@ func TestWaitForExportedTraceRequiresProfilerData(t *testing.T) { if err := os.Mkdir(profilerDir, 0755); err != nil { t.Fatal(err) } - if err := os.WriteFile(filepath.Join(profilerDir, "streamData"), nil, 0644); err != nil { + if err := os.WriteFile(filepath.Join(profilerDir, "streamData"), []byte("stream"), 0644); err != nil { t.Fatal(err) } - got, err := waitForExportedTrace(context.Background(), []string{bundle}, 0) + got, err := waitForExportedTrace(context.Background(), []string{bundle}, time.Second) if err != nil { t.Fatalf("waitForExportedTrace failed: %v", err) } @@ -44,6 +131,23 @@ func TestWaitForExportedTraceRequiresProfilerData(t *testing.T) { } } +func TestWaitForExportedTraceRejectsEmptyStreamData(t *testing.T) { + dir := t.TempDir() + bundle := filepath.Join(dir, "trace-perfdata.gputrace") + profilerDir := filepath.Join(bundle, "trace.gputrace.gpuprofiler_raw") + if err := os.MkdirAll(profilerDir, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(profilerDir, "streamData"), nil, 0o644); err != nil { + t.Fatal(err) + } + + _, err := waitForExportedTrace(context.Background(), []string{bundle}, 0) + if err == nil || !strings.Contains(err.Error(), "non-empty streamData") { + t.Fatalf("error = %v, want incomplete streamData error", err) + } +} + func TestWaitForExportedTraceStopsOnCancellation(t *testing.T) { want := errors.New("stop export wait") ctx, cancel := context.WithCancelCause(context.Background()) @@ -55,6 +159,74 @@ func TestWaitForExportedTraceStopsOnCancellation(t *testing.T) { } } +func TestWaitForExportedTraceDeduplicatesSymlinkAliases(t *testing.T) { + physicalRoot := t.TempDir() + bundle := filepath.Join(physicalRoot, "trace-perfdata.gputrace") + profilerDir := filepath.Join(bundle, "trace.gputrace.gpuprofiler_raw") + if err := os.MkdirAll(profilerDir, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(profilerDir, "streamData"), []byte("stream"), 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(bundle, "capture"), []byte("capture"), 0o644); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(bundle, "MTLBuffer-1-0"), []byte("raw resource"), 0o644); err != nil { + t.Fatal(err) + } + metadata := ` +(uuid)same` + if err := os.WriteFile(filepath.Join(bundle, "metadata"), []byte(metadata), 0o644); err != nil { + t.Fatal(err) + } + + aliasRoot := filepath.Join(t.TempDir(), "alias") + if err := os.Symlink(physicalRoot, aliasRoot); err != nil { + t.Fatal(err) + } + requested := filepath.Join(aliasRoot, filepath.Base(bundle)) + + scans := 0 + readSignature := func(path, profilerDir string) (exportTraceSignature, error) { + scans++ + return readExportTraceSignature(path, profilerDir) + } + got, err := waitForExportedTraceWithReader( + context.Background(), + []string{requested, bundle}, + time.Second, + readSignature, + ) + if err != nil { + t.Fatalf("waitForExportedTraceWithReader: %v", err) + } + if got != requested { + t.Fatalf("path = %q, want requested spelling %q", got, requested) + } + if scans != 3 { + t.Fatalf("signature scans = %d, want 3 for one physical bundle", scans) + } + + input := filepath.Join(t.TempDir(), "input.gputrace") + if err := os.Mkdir(input, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(input, "metadata"), []byte(metadata), 0o644); err != nil { + t.Fatal(err) + } + if err := verifyExportTraceIdentity(input, got); err != nil { + t.Fatalf("identity gate after stable alias: %v", err) + } + payload, err := tracebundle.InspectPayload(got) + if err != nil { + t.Fatalf("inspect payload after stable alias: %v", err) + } + if err := requireSelfContainedExport(got, payload); err != nil { + t.Fatalf("payload gate after stable alias: %v", err) + } +} + func TestTargetedShowPerformanceFoundSentinel(t *testing.T) { if targetedShowPerformanceFound == 0 { t.Fatal("targetedShowPerformanceFound must be non-zero") @@ -66,3 +238,403 @@ func TestTargetedShowPerformanceFoundSentinel(t *testing.T) { t.Fatal("zero should not be recognized as targeted Show Performance sentinel") } } + +func TestGPUTraceStateButtonsIncludeRunningState(t *testing.T) { + for _, name := range gpuTraceStateButtonNames() { + if name == "Stop GPU workload" { + return + } + } + t.Fatal("GPU trace state buttons do not include Stop GPU workload") +} + +func TestDuplicateAXWindowsProduceOneExactTraceMatch(t *testing.T) { + const tracePath = "/Users/test/trace.gputrace" + windows := []xcodeAXWindow{ + { + Element: 100, + Title: "Summary", + Document: tracePath, + X: 20, + Y: 30, + Width: 1200, + Height: 800, + }, + // Same AX element returned twice. + { + Element: 100, + Title: "Summary", + Document: tracePath, + X: 20, + Y: 30, + Width: 1200, + Height: 800, + }, + // A distinct AX reference for the same logical window. + { + Element: 101, + Title: "Summary", + Document: tracePath, + X: 20, + Y: 30, + Width: 1200, + Height: 800, + }, + { + Element: 200, + Title: "Other", + Document: "/Users/test/other.gputrace", + X: 80, + Y: 90, + Width: 1000, + Height: 700, + }, + } + + logical := deduplicateXcodeWindows(windows) + if got, want := len(logical), 2; got != want { + t.Fatalf("logical windows = %d, want %d: %+v", got, want, logical) + } + matches := exactTraceWindows(logical, strings.ToLower(filepath.Clean(tracePath))) + if got, want := len(matches), 1; got != want { + t.Fatalf("exact matches = %d, want %d: %v", got, want, matches) + } + if matches[0] != 100 { + t.Fatalf("selected AX element = %d, want first stable element 100", matches[0]) + } +} + +func TestExportSheetDestinationVerification(t *testing.T) { + const target = "/Users/tmc/tmp/gputrace-language-matrix-20260730/traces/go" + tests := []struct { + name string + remaining string + candidates []string + wantDirect bool + }{ + { + name: "basename is not exact destination", + candidates: []string{"tmp"}, + wantDirect: true, + }, + { + name: "parent directory is not nested destination", + candidates: []string{"/Users/tmc/tmp"}, + wantDirect: true, + }, + { + name: "private tmp is not user tmp", + candidates: []string{"/private/tmp"}, + wantDirect: true, + }, + { + name: "partial popup navigation requires direct location", + remaining: "gputrace-language-matrix-20260730/traces/go", + candidates: []string{target}, + wantDirect: true, + }, + { + name: "exact path verified", + candidates: []string{target}, + }, + { + name: "file URL verified", + candidates: []string{"file:///Users/tmc/tmp/gputrace-language-matrix-20260730/traces/go"}, + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + state := exportSheetState{DirectoryCandidates: tt.candidates} + if got := needsDirectExportLocation(tt.remaining, state, target); got != tt.wantDirect { + t.Fatalf("needsDirectExportLocation = %t, want %t", got, tt.wantDirect) + } + }) + } +} + +func TestGoToFolderNavigationComplete(t *testing.T) { + const target = "/Users/tmc/tmp/gputrace-language-matrix-20260730/traces/go" + tests := []struct { + name string + state exportSheetState + want bool + }{ + { + name: "closed at exact path", + state: exportSheetState{ + DirectoryCandidates: []string{target}, + SaveEnabled: true, + }, + want: true, + }, + { + name: "exact path but sheet remains open", + state: exportSheetState{ + DirectoryCandidates: []string{target}, + GoToFolderSheetOpen: true, + SaveEnabled: true, + }, + }, + { + name: "closed at exact path but save disabled", + state: exportSheetState{ + DirectoryCandidates: []string{target}, + }, + }, + { + name: "closed at basename only", + state: exportSheetState{ + DirectoryCandidates: []string{"tmp"}, + }, + }, + { + name: "closed at private tmp", + state: exportSheetState{ + DirectoryCandidates: []string{"/private/tmp"}, + }, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + if got := goToFolderNavigationComplete(test.state, target); got != test.want { + t.Fatalf("goToFolderNavigationComplete() = %t, want %t", got, test.want) + } + }) + } +} + +func TestGoToFolderConfirmationReady(t *testing.T) { + const target = "/Users/tmc/tmp/gputrace-language-matrix-20260730/traces/go" + tests := []struct { + name string + state exportSheetState + want bool + }{ + { + name: "open with exact Go to path", + state: exportSheetState{ + GoToFolderSheetOpen: true, + GoToFolderPath: target, + }, + want: true, + }, + { + name: "open with exact candidate but stale Go to field", + state: exportSheetState{ + DirectoryCandidates: []string{target}, + GoToFolderSheetOpen: true, + GoToFolderPath: "/private/tmp", + }, + }, + { + name: "closed with committed parent path", + state: exportSheetState{ + DirectoryCandidates: []string{target}, + }, + want: true, + }, + { + name: "closed with basename only", + state: exportSheetState{ + DirectoryCandidates: []string{"tmp"}, + }, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + if got := goToFolderConfirmationReady(test.state, target); got != test.want { + t.Fatalf("goToFolderConfirmationReady() = %t, want %t", got, test.want) + } + }) + } +} + +func TestGoToFolderNavigationCompleteAfterExactEntry(t *testing.T) { + const target = "/Users/tmc/tmp/gputrace-language-matrix-20260730/traces/go" + tests := []struct { + name string + state exportSheetState + want bool + }{ + { + name: "exact absolute candidate", + state: exportSheetState{ + DirectoryCandidates: []string{target}, + SaveEnabled: true, + }, + want: true, + }, + { + name: "committed basename", + state: exportSheetState{ + DirectoryCandidates: []string{"go"}, + SaveEnabled: true, + }, + want: true, + }, + { + name: "wrong basename", + state: exportSheetState{ + DirectoryCandidates: []string{"tmp"}, + SaveEnabled: true, + }, + }, + { + name: "sheet still open", + state: exportSheetState{ + DirectoryCandidates: []string{"go"}, + SaveEnabled: true, + GoToFolderSheetOpen: true, + }, + }, + { + name: "save disabled", + state: exportSheetState{ + DirectoryCandidates: []string{"go"}, + }, + }, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + if got := goToFolderNavigationCompleteAfterExactEntry(test.state, target); got != test.want { + t.Fatalf("goToFolderNavigationCompleteAfterExactEntry() = %t, want %t", got, test.want) + } + }) + } +} + +// TestGoToFolderNativeEntryReleasesCommandBeforePath pins the entry order: +// select all, delete, wait for System Events to release Command, then type the +// the path as ordinary key events. An earlier version typed the path body and +// then moved the cursor back to insert the leading slash; under host load that +// final insert was dropped and the field committed a relative path such as "tmp". +func TestGoToFolderNativeEntryReleasesCommandBeforePath(t *testing.T) { + activateIndex := strings.Index(typeGoToFolderPathScript, `tell application id "com.apple.dt.Xcode" to activate`) + selectIndex := strings.Index(typeGoToFolderPathScript, `keystroke "a" using command down`) + clearIndex := strings.Index(typeGoToFolderPathScript, "key code 51") + delayIndex := strings.Index(typeGoToFolderPathScript, "delay 0.4") + typeIndex := strings.Index(typeGoToFolderPathScript, "repeat with pathCharacter in characters of (item 1 of argv)") + if activateIndex < 0 || selectIndex <= activateIndex || clearIndex <= selectIndex || delayIndex <= clearIndex || typeIndex <= delayIndex { + t.Fatalf("native entry script does not clear and release before typing the path:\n%s", + typeGoToFolderPathScript) + } + if strings.Contains(typeGoToFolderPathScript, "key code 123 using command down") { + t.Error("script still moves the cursor to insert a separate leading slash") + } + if strings.Contains(typeGoToFolderPathScript, `keystroke "/"`) { + t.Error("script still types the leading slash separately") + } + if strings.Contains(typeGoToFolderPathScript, "keystroke (item 1 of argv)") { + t.Error("script still types the full path as one truncation-prone event") + } +} + +// TestTypeGoToFolderPathSendsAbsolutePath guards that the whole absolute path, +// leading separator included, is what gets typed. +func TestTypeGoToFolderPathSendsAbsolutePath(t *testing.T) { + if _, err := goToFolderPathBody("tmp"); err == nil { + t.Error("goToFolderPathBody accepted a relative path") + } + if _, err := goToFolderPathBody("/tmp"); err != nil { + t.Errorf("goToFolderPathBody rejected an absolute path: %v", err) + } +} + +func TestGoToFolderPathBody(t *testing.T) { + tests := []struct { + name string + path string + want string + wantErr bool + }{ + {"absolute", "/Users/tmc/tmp", "Users/tmc/tmp", false}, + {"root", "/", "", false}, + {"relative", "Users/tmc/tmp", "", true}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + got, err := goToFolderPathBody(test.path) + if (err != nil) != test.wantErr { + t.Fatalf("goToFolderPathBody(%q) error = %v, wantErr %v", test.path, err, test.wantErr) + } + if got != test.want { + t.Fatalf("goToFolderPathBody(%q) = %q, want %q", test.path, got, test.want) + } + }) + } +} + +func TestStableExportSheetWaitResult(t *testing.T) { + tests := []struct { + name string + stable int + deadlineReached bool + wantDone bool + wantOK bool + }{ + {"first slow match earns confirmation", 1, true, false, false}, + {"second consecutive match succeeds", 2, true, true, true}, + {"first slow miss fails", 0, true, true, false}, + {"before deadline continues", 0, false, false, false}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + gotDone, gotOK := stableExportSheetWaitResult(test.stable, test.deadlineReached) + if gotDone != test.wantDone || gotOK != test.wantOK { + t.Fatalf("stableExportSheetWaitResult(%d, %t) = (%t, %t), want (%t, %t)", + test.stable, test.deadlineReached, + gotDone, gotOK, test.wantDone, test.wantOK) + } + }) + } +} + +func TestFormatExportSheetStateIncludesBlockingEvidence(t *testing.T) { + state := exportSheetState{ + Filename: "raw-basename.gputrace", + DirectoryCandidates: []string{"tmp"}, + SaveEnabled: true, + GoToFolderSheetOpen: false, + } + got := formatExportSheetState(state) + for _, want := range []string{ + `filename="raw-basename.gputrace"`, + `directory_candidates=["tmp"]`, + "save_enabled=true", + "go_to_folder_open=false", + } { + if !strings.Contains(got, want) { + t.Fatalf("sheet state lacks %q: %s", want, got) + } + } +} + +func TestVerifyExportTraceIdentity(t *testing.T) { + writeBundle := func(name, uuid string) string { + t.Helper() + path := filepath.Join(t.TempDir(), name+".gputrace") + if err := os.Mkdir(path, 0o755); err != nil { + t.Fatal(err) + } + metadata := ` +(uuid)` + uuid + `` + if err := os.WriteFile(filepath.Join(path, "metadata"), []byte(metadata), 0o644); err != nil { + t.Fatal(err) + } + return path + } + + input := writeBundle("input", "same") + if err := verifyExportTraceIdentity(input, writeBundle("matching", "same")); err != nil { + t.Fatalf("matching identity: %v", err) + } + if err := verifyExportTraceIdentity(input, writeBundle("wrong", "different")); err == nil { + t.Fatal("mismatched identity succeeded") + } +} + +func TestStopWorkloadInWindow(t *testing.T) { + if err := stopWorkloadInWindow(0); err != nil { + t.Fatalf("stopWorkloadInWindow(0) failed: %v", err) + } +} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_screenshot.go b/cmd/gputrace/cmd/collect_xcode_profile_screenshot.go index 9e1f0612..cd6616cc 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_screenshot.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_screenshot.go @@ -3,7 +3,9 @@ package cmd import ( + "bytes" "fmt" + "io" "os" "path/filepath" "time" @@ -32,6 +34,9 @@ func runScreenshot(cmd *cobra.Command, args []string, opts *screenshotOptions) e if err != nil { return err } + if err := os.MkdirAll(filepath.Dir(outputPath), 0o755); err != nil { + return fmt.Errorf("create screenshot output directory: %w", err) + } // Get Xcode window info using AX if err := setupMacgo(); err != nil { @@ -49,6 +54,10 @@ func runScreenshot(cmd *cobra.Command, args []string, opts *screenshotOptions) e if err != nil { return err } + selection := selectionForWindow(traceFile, windowAX) + if err := requireBoundSelection(selection); err != nil { + return err + } // Get window title for feedback title := axString(windowAX, "AXTitle") @@ -64,22 +73,31 @@ func runScreenshot(cmd *cobra.Command, args []string, opts *screenshotOptions) e return fmt.Errorf("capture failed: %w", err) } - // Verify file was created - if _, err := os.Stat(outputPath); err != nil { - return fmt.Errorf("screenshot file not created") + if err := verifyScreenshotFile(outputPath); err != nil { + return err } - fmt.Fprintf(status, "Screenshot saved to: %s\n", outputPath) + fmt.Fprintf(status, "Screenshot verified: %s\n", outputPath) return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "screenshot", - Target: traceFile, - Output: outputPath, + Action: "screenshot", + Target: traceFile, + Output: outputPath, + RequestedTrace: traceFile, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "screenshot verified", + Evidence: "output is a non-empty PNG file", + TargetBound: boolPointer(selection.Bound), }) } func resolveScreenshotOutputPath(output string, now time.Time) (string, error) { if output == "" { - output = fmt.Sprintf("/tmp/xcode-screenshot-%s.png", now.Format("20060102-150405")) + home, err := os.UserHomeDir() + if err != nil { + return "", fmt.Errorf("find home directory: %w", err) + } + output = filepath.Join(home, "tmp", fmt.Sprintf("xcode-screenshot-%s.png", now.Format("20060102-150405"))) } if commandOutputPathIsStdout(output) { return "", fmt.Errorf("screenshot output must be a file path, not stdout") @@ -91,6 +109,23 @@ func resolveScreenshotOutputPath(output string, now time.Time) (string, error) { return outputPath, nil } +func verifyScreenshotFile(path string) error { + file, err := os.Open(path) + if err != nil { + return fmt.Errorf("open screenshot output: %w", err) + } + defer file.Close() + header := make([]byte, 8) + if _, err := io.ReadFull(file, header); err != nil { + return fmt.Errorf("screenshot output is incomplete: %w", err) + } + want := []byte{0x89, 'P', 'N', 'G', '\r', '\n', 0x1a, '\n'} + if !bytes.Equal(header, want) { + return fmt.Errorf("screenshot output is not a PNG file: %s", path) + } + return nil +} + // triggerScreenRecordingTCC calls CGDisplayCreateImage to create a TCC // database entry for Screen Recording permission without prompting the user. func triggerScreenRecordingTCC() error { diff --git a/cmd/gputrace/cmd/collect_xcode_profile_screenshot_test.go b/cmd/gputrace/cmd/collect_xcode_profile_screenshot_test.go index 69cdcd1c..c3ec13bb 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_screenshot_test.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_screenshot_test.go @@ -4,6 +4,7 @@ package cmd import ( "encoding/json" + "os" "path/filepath" "strings" "testing" @@ -31,7 +32,11 @@ func TestResolveScreenshotOutputPath(t *testing.T) { if err != nil { t.Fatalf("default output path: %v", err) } - if want := "/tmp/xcode-screenshot-20260531-010203.png"; got != want { + home, err := os.UserHomeDir() + if err != nil { + t.Fatal(err) + } + if want := filepath.Join(home, "tmp", "xcode-screenshot-20260531-010203.png"); got != want { t.Fatalf("default path = %q, want %q", got, want) } @@ -47,6 +52,23 @@ func TestResolveScreenshotOutputPath(t *testing.T) { } } +func TestVerifyScreenshotFile(t *testing.T) { + path := filepath.Join(t.TempDir(), "window.png") + png := []byte{0x89, 'P', 'N', 'G', '\r', '\n', 0x1a, '\n', 0} + if err := os.WriteFile(path, png, 0o644); err != nil { + t.Fatal(err) + } + if err := verifyScreenshotFile(path); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(path, []byte("not png"), 0o644); err != nil { + t.Fatal(err) + } + if err := verifyScreenshotFile(path); err == nil { + t.Fatal("non-PNG screenshot returned nil error") + } +} + func TestTriggerScreenRecordingTCCJSONOutput(t *testing.T) { oldJSON := collectProfileOpts.json t.Cleanup(func() { diff --git a/cmd/gputrace/cmd/collect_xcode_profile_status.go b/cmd/gputrace/cmd/collect_xcode_profile_status.go index f800372b..e25c8d81 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_status.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_status.go @@ -5,6 +5,7 @@ package cmd import ( "encoding/json" "fmt" + "io" "os" "strings" "time" @@ -19,6 +20,12 @@ type checkStatusOptions struct { // StatusOutput represents the JSON output for check-status. type StatusOutput struct { Status string `json:"status"` + Phase string `json:"phase"` + Evidence string `json:"evidence"` + RequestedTrace string `json:"requested_trace,omitempty"` + SelectedTitle string `json:"selected_title,omitempty"` + SelectedDocument string `json:"selected_document,omitempty"` + TargetBound bool `json:"target_bound"` ReplayAvailable bool `json:"replay_available"` ExportAvailable bool `json:"export_available"` ShowPerformanceAvailable bool `json:"show_performance_available"` @@ -62,12 +69,14 @@ func runCheckStatus(cmd *cobra.Command, args []string, opts *checkStatusOptions) if debug { fmt.Fprintf(os.Stderr, "[check-status] got window: %v (title=%q)\n", windowAX, axString(windowAX, "AXTitle")) } + selection := selectionForWindow(traceFile, windowAX) if collectProfileOpts.json { if debug { fmt.Fprintln(os.Stderr, "[check-status] getting status output (JSON)...") } output := getStatusOutput(windowAX, debug) + applyStatusSelection(&output, selection) enc := json.NewEncoder(os.Stdout) enc.SetIndent("", " ") return enc.Encode(output) @@ -76,8 +85,9 @@ func runCheckStatus(cmd *cobra.Command, args []string, opts *checkStatusOptions) if debug { fmt.Fprintln(os.Stderr, "[check-status] getting profiling status...") } - status := getProfilingStatusWithDebug(windowAX, debug) - fmt.Println(status) + output := getStatusOutput(windowAX, debug) + applyStatusSelection(&output, selection) + writeStatusText(os.Stdout, output) return nil } @@ -103,6 +113,8 @@ func getStatusOutput(window uintptr, debug bool) StatusOutput { return StatusOutput{ Status: status, + Phase: profilingPhase(status), + Evidence: profilingStatusEvidence(status), ReplayAvailable: replayAvailable, ExportAvailable: exportAvailable, ShowPerformanceAvailable: showPerfAvailable, @@ -110,6 +122,67 @@ func getStatusOutput(window uintptr, debug bool) StatusOutput { } } +func applyStatusSelection(output *StatusOutput, selection xcodeWindowSelection) { + output.RequestedTrace = selection.RequestedTrace + output.SelectedTitle = selection.Title + output.SelectedDocument = selection.Document + output.TargetBound = selection.Bound + if selection.RequestedTrace != "" && !selection.Bound { + detected := output.Status + output.Status = "unknown" + output.Phase = "unbound" + output.Evidence = fmt.Sprintf("%s; refusing to attribute detected %q state to the requested trace", selection.Evidence, detected) + return + } + output.Evidence = selection.Evidence + "; " + output.Evidence +} + +func profilingPhase(status string) string { + switch status { + case "initializing": + return "trace loading" + case "replay-ready": + return "GPU replay ready" + case "running": + return "performance profiling running" + case "complete": + return "performance data available" + default: + return "state unknown" + } +} + +func profilingStatusEvidence(status string) string { + switch status { + case "initializing": + return "a replay or profile control is present but disabled" + case "replay-ready": + return "an enabled replay/profile control or performance-data-unavailable label was detected" + case "running": + return "Xcode reports GPU trace profiling in progress" + case "complete": + return "a Show Performance or performance-navigation control was detected" + default: + return "no recognized replay or performance control state was detected" + } +} + +func writeStatusText(w io.Writer, output StatusOutput) { + fmt.Fprintf(w, "Status: %s\n", output.Status) + fmt.Fprintf(w, "Phase: %s\n", output.Phase) + if output.RequestedTrace != "" { + fmt.Fprintf(w, "Requested trace: %s\n", output.RequestedTrace) + } + if output.SelectedDocument != "" { + fmt.Fprintf(w, "Selected document: %s\n", output.SelectedDocument) + } + if output.SelectedTitle != "" { + fmt.Fprintf(w, "Selected window: %s\n", output.SelectedTitle) + } + fmt.Fprintf(w, "Target bound: %t\n", output.TargetBound) + fmt.Fprintf(w, "Evidence: %s\n", output.Evidence) +} + // getCurrentTab tries to determine the currently selected tab. func getCurrentTab(window uintptr) string { tabs := findAllTabs(window, 500) @@ -225,12 +298,10 @@ func getProfilingStatusWithDebug(window uintptr, debug bool) string { return "running" } - // Now do targeted traversal for "Show Performance" - if hasShowPerformanceDebug(window, debug) { - return "complete" - } - // Also check for "Timeline" or "Encoders" which indicate the trace is loaded and interactive - if findButtonByNameInsensitive(window, "Timeline") != 0 || findButtonByNameInsensitive(window, "Encoders") != 0 { + // "Show Performance" is present while the summary is ready to enter the + // Performance view. Once Xcode has already entered that view, the button + // is gone and its tabs are the completion signal instead. + if hasPerformanceDataDebug(window, debug) { return "complete" } @@ -272,6 +343,28 @@ func hasShowPerformance(window uintptr) bool { return hasShowPerformanceDebug(window, false) } +// hasPerformanceData reports whether Xcode has completed profiling the trace. +// Xcode exposes either the Summary view's Show Performance button or the +// Performance view's Timeline and Encoders controls. Both states are safe to +// advance to export, provided the caller has already bound the window to the +// requested trace. +func hasPerformanceData(window uintptr) bool { + return hasPerformanceDataDebug(window, false) +} + +func hasPerformanceDataDebug(window uintptr, debug bool) bool { + showPerformance := hasShowPerformanceDebug(window, debug) + performanceControls := findButtonByNameInsensitive(window, "Timeline") != 0 || findButtonByNameInsensitive(window, "Encoders") != 0 + if performanceControls && debug { + fmt.Fprintln(os.Stderr, "[DEBUG] Performance view controls found") + } + return performanceDataReady(showPerformance, performanceControls) +} + +func performanceDataReady(showPerformance, performanceControls bool) bool { + return showPerformance || performanceControls +} + func hasShowPerformanceDebug(window uintptr, debug bool) bool { // Find "editor area" group by title (BFS with visit limit) editorArea := findGroupByTitleDebug(window, "editor area", 100, debug) diff --git a/cmd/gputrace/cmd/collect_xcode_profile_status_test.go b/cmd/gputrace/cmd/collect_xcode_profile_status_test.go new file mode 100644 index 00000000..8f2d1f6d --- /dev/null +++ b/cmd/gputrace/cmd/collect_xcode_profile_status_test.go @@ -0,0 +1,27 @@ +//go:build darwin + +package cmd + +import "testing" + +func TestPerformanceDataReady(t *testing.T) { + tests := []struct { + name string + showPerformance bool + performanceControls bool + want bool + }{ + {name: "summary", showPerformance: true, want: true}, + {name: "performance view", performanceControls: true, want: true}, + {name: "not profiled", want: false}, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + got := performanceDataReady(test.showPerformance, test.performanceControls) + if got != test.want { + t.Fatalf("performanceDataReady(%t, %t) = %t, want %t", test.showPerformance, test.performanceControls, got, test.want) + } + }) + } +} diff --git a/cmd/gputrace/cmd/collect_xcode_profile_tabs.go b/cmd/gputrace/cmd/collect_xcode_profile_tabs.go index cab8a599..42ff46ac 100644 --- a/cmd/gputrace/cmd/collect_xcode_profile_tabs.go +++ b/cmd/gputrace/cmd/collect_xcode_profile_tabs.go @@ -6,8 +6,10 @@ import ( "context" "encoding/json" "fmt" + "io" "os" "strings" + "time" "github.com/spf13/cobra" ) @@ -43,6 +45,7 @@ func runSelectTab(cmd *cobra.Command, args []string) error { if err != nil { return err } + selection := selectionForWindow("", windowAX) // Find and click the tab tab := findTabByName(windowAX, tabName) @@ -51,12 +54,7 @@ func runSelectTab(cmd *cobra.Command, args []string) error { if err := axAction(tab, "AXPress"); err != nil { return fmt.Errorf("failed to click tab: %w", err) } - fmt.Fprintln(status, "Done") - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "select_tab", - Target: tabName, - Method: "tab", - }) + return finishVerifiedSelection(cmd.Context(), status, windowAX, tab, tabName, "tab", selection) } // Try as an outline row (navigator items like Summary, Dependencies, etc.) @@ -66,12 +64,7 @@ func runSelectTab(cmd *cobra.Command, args []string) error { if err := axAction(row, "AXPress"); err != nil { return fmt.Errorf("failed to select: %w", err) } - fmt.Fprintln(status, "Done") - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "select_tab", - Target: tabName, - Method: "navigator", - }) + return finishVerifiedSelection(cmd.Context(), status, windowAX, row, tabName, "navigator", selection) } // Try as a button (some tabs appear as buttons) @@ -81,17 +74,29 @@ func runSelectTab(cmd *cobra.Command, args []string) error { if err := axAction(btn, "AXPress"); err != nil { return fmt.Errorf("failed to click: %w", err) } - fmt.Fprintln(status, "Done") - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "select_tab", - Target: tabName, - Method: "button", - }) + return finishVerifiedSelection(cmd.Context(), status, windowAX, btn, tabName, "button", selection) } return fmt.Errorf("tab %q not found", tabName) } +func finishVerifiedSelection(ctx context.Context, status io.Writer, window, control uintptr, name, method string, selection xcodeWindowSelection) error { + if err := waitForSelectedControl(ctx, window, control, name, 2*time.Second); err != nil { + return err + } + fmt.Fprintf(status, "%s selected and verified\n", name) + return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + Action: "select_tab", + Target: name, + Method: method, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "view selected", + Evidence: fmt.Sprintf("%s control reports selected", name), + TargetBound: boolPointer(selection.Bound), + }) +} + // runSelectNavigatorItem selects an item in the Debug navigator by name. func runSelectNavigatorItem(ctx context.Context, name string) error { status := xcodeProfileStatusWriter() @@ -110,6 +115,7 @@ func runSelectNavigatorItem(ctx context.Context, name string) error { if err != nil { return err } + selection := selectionForWindow("", windowAX) // The navigator items have specific capitalization displayName := strings.Title(name) @@ -148,50 +154,47 @@ func runSelectNavigatorItem(ctx context.Context, name string) error { // Try AXOpen first (double-click to open) if err := axAction(targetEl, "AXOpen"); err == nil { - fmt.Fprintln(status, "Done") - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "select_navigator", - Target: displayName, - Method: "AXOpen", - }) + return finishVerifiedNavigatorSelection(ctx, status, windowAX, targetEl, displayName, "AXOpen", selection) } // Try AXPress if err := axAction(targetEl, "AXPress"); err == nil { - fmt.Fprintln(status, "Done") - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "select_navigator", - Target: displayName, - Method: "AXPress", - }) + return finishVerifiedNavigatorSelection(ctx, status, windowAX, targetEl, displayName, "AXPress", selection) } // Try setting AXSelected on the element, then double-click if selectElement(targetEl) { // Also try double-click via CGEvent if err := doubleClickElement(targetEl); err == nil { - fmt.Fprintln(status, "Done") - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "select_navigator", - Target: displayName, - Method: "select_double_click", - }) + return finishVerifiedNavigatorSelection(ctx, status, windowAX, targetEl, displayName, "select_double_click", selection) } } // Last resort: just double-click on the element if err := doubleClickElement(targetEl); err == nil { - fmt.Fprintln(status, "Done") - return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ - Action: "select_navigator", - Target: displayName, - Method: "double_click", - }) + return finishVerifiedNavigatorSelection(ctx, status, windowAX, targetEl, displayName, "double_click", selection) } return fmt.Errorf("could not select %s (element found but selection failed)", displayName) } +func finishVerifiedNavigatorSelection(ctx context.Context, status io.Writer, window, control uintptr, name, method string, selection xcodeWindowSelection) error { + if err := waitForSelectedControl(ctx, window, control, name, 2*time.Second); err != nil { + return err + } + fmt.Fprintf(status, "%s navigator item selected and verified\n", name) + return writeXcodeProfileActionOutput(xcodeProfileActionOutput{ + Action: "select_navigator", + Target: name, + Method: method, + SelectedTitle: selection.Title, + SelectedDocument: selection.Document, + Phase: "navigator item selected", + Evidence: fmt.Sprintf("%s control reports selected", name), + TargetBound: boolPointer(selection.Bound), + }) +} + // findCellByName finds a cell or static text element by name. func findCellByName(root uintptr, name string) uintptr { nameLower := strings.ToLower(name) @@ -410,7 +413,7 @@ func findButtonByNameInsensitive(root uintptr, name string) uintptr { // are typically AXOutlineRow, AXRow, or AXCell elements. func findOutlineRowByName(root uintptr, name string) uintptr { nameLower := strings.ToLower(name) - return findElement(root, func(el uintptr) bool { + el := findElement(root, func(el uintptr) bool { role := axString(el, "AXRole") // Check various row/cell types used in outline views if role == "AXOutlineRow" || role == "AXRow" || role == "AXCell" || role == "AXStaticText" { @@ -430,6 +433,16 @@ func findOutlineRowByName(root uintptr, name string) uintptr { } return false }) + if el == 0 { + return 0 + } + role := axString(el, "AXRole") + if role == "AXStaticText" || role == "AXCell" { + if row := findParentOutlineRow(el); row != 0 { + return row + } + } + return el } // findAllTabs finds all tab elements in the tree. diff --git a/cmd/gputrace/cmd/collect_xcode_profile_uiwindow_test.go b/cmd/gputrace/cmd/collect_xcode_profile_uiwindow_test.go new file mode 100644 index 00000000..1262c27e --- /dev/null +++ b/cmd/gputrace/cmd/collect_xcode_profile_uiwindow_test.go @@ -0,0 +1,70 @@ +//go:build darwin + +package cmd + +import "testing" + +func TestAdmitUIIdentifiedWindow(t *testing.T) { + const unbound = "selected GPU trace window has no title or AXDocument match for the requested trace" + tests := []struct { + name string + selection xcodeWindowSelection + window uintptr + uiIdentified uintptr + want bool + wantEvidence string + }{ + { + name: "untitled window matched by GPU trace UI is admitted", + selection: xcodeWindowSelection{Evidence: unbound}, + window: 0x40, + uiIdentified: 0x40, + want: true, + wantEvidence: "sole window with GPU trace UI in the bound Xcode process", + }, + { + name: "a different window is not admitted", + selection: xcodeWindowSelection{Evidence: unbound}, + window: 0x40, + uiIdentified: 0x50, + want: false, + wantEvidence: unbound, + }, + { + name: "no UI-identified window means no admission", + selection: xcodeWindowSelection{Evidence: unbound}, + window: 0x40, + uiIdentified: 0, + want: false, + wantEvidence: unbound, + }, + { + name: "a zero window is never admitted", + selection: xcodeWindowSelection{Evidence: unbound}, + window: 0, + uiIdentified: 0, + want: false, + wantEvidence: unbound, + }, + { + name: "an already bound selection keeps its own evidence", + selection: xcodeWindowSelection{Bound: true, Evidence: "AXDocument exactly matches the requested trace"}, + window: 0x40, + uiIdentified: 0x40, + want: true, + wantEvidence: "AXDocument exactly matches the requested trace", + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := admitUIIdentifiedWindow(tt.selection, tt.window, tt.uiIdentified) + if got.Bound != tt.want { + t.Errorf("Bound = %v, want %v", got.Bound, tt.want) + } + if got.Evidence != tt.wantEvidence { + t.Errorf("Evidence = %q, want %q", got.Evidence, tt.wantEvidence) + } + }) + } +} diff --git a/cmd/gputrace/cmd/command_buffers.go b/cmd/gputrace/cmd/command_buffers.go index 1df3b9dc..0dc09fa2 100644 --- a/cmd/gputrace/cmd/command_buffers.go +++ b/cmd/gputrace/cmd/command_buffers.go @@ -4,6 +4,7 @@ import ( "encoding/json" "fmt" "io" + "strings" "github.com/spf13/cobra" @@ -14,6 +15,8 @@ type commandBuffersOptions struct { verbose bool detailed bool json bool + limit int + all bool } type commandBufferEncoderJSON struct { @@ -31,9 +34,12 @@ type commandBufferJSON struct { Dispatches int `json:"dispatches"` } -var commandBuffersCmd = newCommandBuffersCommand(&commandBuffersOptions{}) +var commandBuffersCmd = newCommandBuffersCommand(&commandBuffersOptions{limit: defaultHumanLimit}) func newCommandBuffersCommand(opts *commandBuffersOptions) *cobra.Command { + if opts.limit == 0 { + opts.limit = defaultHumanLimit + } cmd := &cobra.Command{ Use: "command-buffers ", Short: "List and analyze command buffers in a GPU trace", @@ -57,6 +63,8 @@ Examples: cmd.Flags().BoolVarP(&opts.verbose, "verbose", "v", false, "Show verbose output with encoder and API call counts") cmd.Flags().BoolVarP(&opts.detailed, "detailed", "d", false, "Show detailed analysis of each command buffer") cmd.Flags().BoolVar(&opts.json, "json", false, "Output in JSON format") + cmd.Flags().IntVar(&opts.limit, "limit", opts.limit, "Maximum human-output rows (and detail lines per buffer)") + cmd.Flags().BoolVar(&opts.all, "all", opts.all, "Show all human-output rows") return cmd } @@ -77,32 +85,42 @@ func runCommandBuffers(cmd *cobra.Command, args []string, opts *commandBuffersOp if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } - // Parse command buffers - commandBuffers, err := trace.ParseCommandBuffers() + // Parse command buffers. The capture handle keeps the file and the + // command-buffer index so the loops below do not reread per buffer. + capture, err := gputrace.OpenCapture(trace) if err != nil { return fmt.Errorf("failed to parse command buffers: %w", err) } + commandBuffers := capture.CommandBuffers() if opts.json { - out, err := commandBuffersJSONOutput(trace, commandBuffers) + out, err := commandBuffersJSONOutput(capture, commandBuffers) if err != nil { return err } return writeCommandBuffersJSON(cmd.OutOrStdout(), out) } + limit, err := resolveHumanLimit(opts.limit, opts.all) + if err != nil { + return err + } w := cmd.OutOrStdout() // Compact one-line-per-buffer output fmt.Fprintf(w, "%d command buffers:\n", len(commandBuffers)) - for _, cb := range commandBuffers { + shown := limitedCount(len(commandBuffers), limit) + for _, cb := range commandBuffers[:shown] { label := "" if cb.Label != "" { label = fmt.Sprintf(" label=%q", cb.Label) } if opts.verbose || opts.detailed { - dcb, err := gputrace.ParseDetailedCommandBuffer(trace, cb.Index) + dcb, err := capture.Detailed(cb.Index) if err != nil { fmt.Fprintf(w, " %3d: offset=0x%08x%s (error: %v)\n", cb.Index, cb.Offset, label, err) } else { @@ -113,13 +131,19 @@ func runCommandBuffers(cmd *cobra.Command, args []string, opts *commandBuffersOp fmt.Fprintf(w, " %3d: offset=0x%08x%s\n", cb.Index, cb.Offset, label) } } + if shown < len(commandBuffers) { + fmt.Fprintf(w, " ... %d more command buffers omitted (use --all)\n", len(commandBuffers)-shown) + } // Show detailed analysis if requested if opts.detailed { fmt.Fprintf(w, "\n=== Detailed Analysis ===\n\n") - for _, cb := range commandBuffers { - if err := gputrace.DumpCommandBuffer(trace, w, cb.Index); err != nil { + for _, cb := range commandBuffers[:shown] { + var detail strings.Builder + if err := gputrace.DumpCommandBuffer(trace, &detail, cb.Index); err != nil { fmt.Fprintf(w, "Error dumping command buffer #%d: %v\n", cb.Index, err) + } else if err := writeLimitedLines(w, detail.String(), limit, "detail lines"); err != nil { + return fmt.Errorf("write command buffer details: %w", err) } } } @@ -130,7 +154,7 @@ func runCommandBuffers(cmd *cobra.Command, args []string, opts *commandBuffersOp totalAPICalls := 0 totalDispatches := 0 for _, cb := range commandBuffers { - dcb, err := gputrace.ParseDetailedCommandBuffer(trace, cb.Index) + dcb, err := capture.Detailed(cb.Index) if err == nil { totalEncoders += len(dcb.Encoders) totalAPICalls += len(dcb.Calls) @@ -149,7 +173,7 @@ func runCommandBuffers(cmd *cobra.Command, args []string, opts *commandBuffersOp return nil } -func commandBuffersJSONOutput(trace *gputrace.Trace, commandBuffers []*gputrace.CommandBuffer) ([]commandBufferJSON, error) { +func commandBuffersJSONOutput(capture *gputrace.Capture, commandBuffers []*gputrace.CommandBuffer) ([]commandBufferJSON, error) { out := make([]commandBufferJSON, len(commandBuffers)) for i, cb := range commandBuffers { entry := commandBufferJSON{ @@ -157,7 +181,7 @@ func commandBuffersJSONOutput(trace *gputrace.Trace, commandBuffers []*gputrace. Label: cb.Label, Offset: fmt.Sprintf("0x%08x", cb.Offset), } - dcb, err := gputrace.ParseDetailedCommandBuffer(trace, cb.Index) + dcb, err := capture.Detailed(cb.Index) if err == nil { entry.Calls = len(dcb.Calls) entry.PipelineRecords = len(dcb.Calls) diff --git a/cmd/gputrace/cmd/command_buffers_test.go b/cmd/gputrace/cmd/command_buffers_test.go index e7e9f549..8971ed58 100644 --- a/cmd/gputrace/cmd/command_buffers_test.go +++ b/cmd/gputrace/cmd/command_buffers_test.go @@ -69,6 +69,26 @@ func TestRunCommandBuffersJSONUsesCommandOutput(t *testing.T) { } } +func TestRunCommandBuffersTextUsesCommandOutput(t *testing.T) { + tracePath := testCommandBuffersTracePath(t) + var out bytes.Buffer + command := &cobra.Command{} + command.SetOut(&out) + + stdout, err := captureStdout(t, func() error { + return runCommandBuffers(command, []string{tracePath}, &commandBuffersOptions{limit: 1}) + }) + if err != nil { + t.Fatalf("runCommandBuffers: %v", err) + } + if stdout != "" { + t.Fatalf("os stdout = %q, want empty", stdout) + } + if !strings.Contains(out.String(), "command buffers:") { + t.Fatalf("command output missing summary:\n%s", out.String()) + } +} + func testCommandBuffersTracePath(t *testing.T) string { t.Helper() diff --git a/cmd/gputrace/cmd/correlate.go b/cmd/gputrace/cmd/correlate.go index 58e09354..5f540d3c 100644 --- a/cmd/gputrace/cmd/correlate.go +++ b/cmd/gputrace/cmd/correlate.go @@ -26,7 +26,7 @@ This command combines timing information from the trace with hardware metrics from the profiler data (.gpuprofiler_raw), providing a comprehensive view of shader performance including: - Execution timing (count, duration, min/max/avg, source, approximation flag) - - Hardware metrics (ALU utilization, kernel occupancy) + - Hardware metrics (ALU utilization) - Memory metrics (bandwidth, total cycles) - Derived metrics (cycles per invocation, GPU frequency) @@ -71,6 +71,9 @@ func runCorrelate(cmd *cobra.Command, args []string, opts *correlateOptions) err if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } defer trace.Close() // Correlate shader metrics @@ -122,7 +125,6 @@ func runCorrelate(cmd *cobra.Command, args []string, opts *correlateOptions) err if shader.ALUUtilization > 0 { fmt.Fprintf(out, " Hardware:\n") fmt.Fprintf(out, " ALU Util: %.1f%%\n", shader.ALUUtilization) - fmt.Fprintf(out, " Occupancy: %.1f%%\n", shader.KernelOccupancy) fmt.Fprintf(out, " SIMD Groups: %d\n", shader.SIMDGroups) fmt.Fprintf(out, " Registers: %d allocated, %d spilled bytes\n", shader.AllocatedRegs, shader.SpilledBytes) diff --git a/cmd/gputrace/cmd/counter_metadata_darwin.go b/cmd/gputrace/cmd/counter_metadata_darwin.go new file mode 100644 index 00000000..025c4ce7 --- /dev/null +++ b/cmd/gputrace/cmd/counter_metadata_darwin.go @@ -0,0 +1,52 @@ +//go:build darwin + +package cmd + +import "github.com/tmc/gputrace/internal/counter" + +// applyXcodeCounterMetadata annotates tracks whose names exactly match Xcode's +// counter dictionary. The dictionary is vocabulary, not a way to name opaque +// raw-counter ids, so unmatched archive-derived tracks retain their own names +// and units. +// +// Reading the dictionary means running plutil, so it is darwin only. The +// enrichment is optional by construction: see the stub in +// counter_metadata_other.go. +func applyXcodeCounterMetadata(tracks []CounterTrack) []CounterTrack { + graph, err := counter.LoadGPUCounterGraph() + if err != nil || graph == nil { + return tracks + } + return applyXcodeCounterMetadataFromGraph(tracks, graph) +} + +// applyXcodeCounterMetadataFromGraph applies one already-loaded dictionary. +// It is separate from applyXcodeCounterMetadata so tests need not depend on an +// installed Xcode bundle. +func applyXcodeCounterMetadataFromGraph(tracks []CounterTrack, graph *counter.GPUCounterGraph) []CounterTrack { + if graph == nil { + return tracks + } + groups := make(map[string][]string) + for _, group := range graph.TimelineGroups { + for _, name := range group.Counters { + groups[name] = append(groups[name], group.Name) + } + } + for i := range tracks { + track := &tracks[i] + metadata, ok := graph.Counters[track.Name] + if !ok { + continue + } + if metadata.Unit != "" { + track.Unit = metadata.Unit + } + if metadata.Description != "" { + track.Description = metadata.Description + } + track.XcodeGroups = append([]string(nil), groups[track.Name]...) + track.XcodeCatalogPath = graph.Path + } + return tracks +} diff --git a/cmd/gputrace/cmd/counter_metadata_darwin_test.go b/cmd/gputrace/cmd/counter_metadata_darwin_test.go new file mode 100644 index 00000000..25a0fb2a --- /dev/null +++ b/cmd/gputrace/cmd/counter_metadata_darwin_test.go @@ -0,0 +1,41 @@ +//go:build darwin + +package cmd + +import ( + "slices" + "testing" + + "github.com/tmc/gputrace/internal/counter" +) + +func TestApplyXcodeCounterMetadata(t *testing.T) { + graph := &counter.GPUCounterGraph{ + Path: "/Applications/Xcode-rc.app/GPUCounterGraph.plist", + Counters: map[string]counter.CounterMetadata{ + "ALU Utilization": {Unit: "Percentage of Peak ALU Performance"}, + }, + TimelineGroups: []counter.TimelineGroup{ + {Name: "ALU", Counters: []string{"ALU Utilization"}}, + {Name: "Secondary", Counters: []string{"ALU Utilization"}}, + }, + } + tracks := []CounterTrack{ + {Name: "ALU Utilization", Unit: "%"}, + {Name: "GPU Cycles", Unit: "cycles"}, + } + + got := applyXcodeCounterMetadataFromGraph(tracks, graph) + if got[0].Unit != "Percentage of Peak ALU Performance" { + t.Fatalf("ALU unit = %q", got[0].Unit) + } + if got[0].XcodeCatalogPath != graph.Path { + t.Fatalf("ALU catalog path = %q, want %q", got[0].XcodeCatalogPath, graph.Path) + } + if want := []string{"ALU", "Secondary"}; !slices.Equal(got[0].XcodeGroups, want) { + t.Fatalf("ALU groups = %q, want %q", got[0].XcodeGroups, want) + } + if got[1].Unit != "cycles" || got[1].XcodeCatalogPath != "" || len(got[1].XcodeGroups) != 0 { + t.Fatalf("unmatched archive track changed: %+v", got[1]) + } +} diff --git a/cmd/gputrace/cmd/counter_metadata_other.go b/cmd/gputrace/cmd/counter_metadata_other.go new file mode 100644 index 00000000..a5c430fc --- /dev/null +++ b/cmd/gputrace/cmd/counter_metadata_other.go @@ -0,0 +1,11 @@ +//go:build !darwin + +package cmd + +// applyXcodeCounterMetadata leaves tracks unannotated off darwin. +// +// The counter dictionary lives inside an installed Xcode and is read with +// plutil, neither of which exists here. Tracks keep the name and unit their +// own source gave them, which is what the enrichment falls back to on darwin +// when no Xcode is installed. +func applyXcodeCounterMetadata(tracks []CounterTrack) []CounterTrack { return tracks } diff --git a/cmd/gputrace/cmd/counters.go b/cmd/gputrace/cmd/counters.go index 215a1a86..d6857a9a 100644 --- a/cmd/gputrace/cmd/counters.go +++ b/cmd/gputrace/cmd/counters.go @@ -29,6 +29,7 @@ func init() { } func runCounters(cmd *cobra.Command, args []string) error { + w := cmd.OutOrStdout() // 1. Get Default Device device := metal.MTLCreateSystemDefaultDevice() if device.GetID() == 0 { @@ -39,7 +40,7 @@ func runCounters(cmd *cobra.Command, args []string) error { if nameID != 0 { cstr := objc.Send[*byte](nameID, objc.Sel("UTF8String")) if cstr != nil { - fmt.Printf("Device: %s\n", objc.GoString(cstr)) + fmt.Fprintf(w, "Device: %s\n", objc.GoString(cstr)) } } @@ -53,7 +54,7 @@ func runCounters(cmd *cobra.Command, args []string) error { if counterSetCount == 0 { return fmt.Errorf("device returned no counter sets") } - fmt.Printf("Found %d counter sets:\n", counterSetCount) + fmt.Fprintf(w, "Counter sets: %d\n", counterSetCount) for i := uint(0); i < counterSetCount; i++ { setID := objc.Send[objc.ID](counterSetsID, objc.Sel("objectAtIndex:"), i) if setID == 0 { @@ -61,7 +62,7 @@ func runCounters(cmd *cobra.Command, args []string) error { } cs := metal.MTLCounterSetObjectFromID(setID) csName := cs.Name() - fmt.Printf(" - %s\n", csName) + fmt.Fprintf(w, " %s\n", csName) if csName == "timestamp" { timestampCounterSet = cs } @@ -93,7 +94,7 @@ func runCounters(cmd *cobra.Command, args []string) error { return fmt.Errorf("failed to create counter sample buffer: unknown error") } sampleBuffer := metal.MTLCounterSampleBufferObjectFromID(sampleBufferID) - fmt.Println("Created Sample Buffer") + fmt.Fprintln(w, "Counter sample buffer: ready (2 timestamp samples)") // 4. Create Library and Pipeline // 2. Load Kernel @@ -253,15 +254,17 @@ func runCounters(cmd *cobra.Command, args []string) error { // Get timestamp frequency for conversion freq := objc.Send[uint64](device.GetID(), objc.Sel("queryTimestampFrequency")) - fmt.Printf("Timestamp 0: %d\n", t0) - fmt.Printf("Timestamp 1: %d\n", t1) - fmt.Printf("Duration: %d ticks\n", durationTicks) + fmt.Fprintf(w, "Timestamp start: %d ticks\n", t0) + fmt.Fprintf(w, "Timestamp end: %d ticks\n", t1) + fmt.Fprintf(w, "Tick delta: %d ticks\n", durationTicks) if freq > 0 { durationNs := float64(durationTicks) * 1e9 / float64(freq) durationUs := durationNs / 1000 - fmt.Printf("Timestamp Frequency: %d Hz (%.1f MHz)\n", freq, float64(freq)/1e6) - fmt.Printf("Duration: %.2f ns (%.2f µs)\n", durationNs, durationUs) + fmt.Fprintf(w, "Timestamp frequency: %d Hz (%.1f MHz)\n", freq, float64(freq)/1e6) + fmt.Fprintf(w, "Elapsed: %.2f ns (%.2f µs)\n", durationNs, durationUs) + } else { + fmt.Fprintln(w, "Elapsed: unavailable (device did not report a timestamp frequency)") } return nil diff --git a/cmd/gputrace/cmd/cupti.go b/cmd/gputrace/cmd/cupti.go new file mode 100644 index 00000000..50a2c22e --- /dev/null +++ b/cmd/gputrace/cmd/cupti.go @@ -0,0 +1,195 @@ +package cmd + +import ( + "fmt" + "os" + "sort" + "strings" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/cuptitrace" + "github.com/tmc/gputrace/internal/gpuevent" +) + +type cuptiOptions struct { + output string + stats bool + spans bool + spansJSON bool + top int + perKernel bool + samples string +} + +var cuptiOpts = &cuptiOptions{} + +var cuptiCmd = newCuptiCommand(cuptiOpts) + +func newCuptiCommand(opts *cuptiOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "cupti ", + Short: "Convert CUPTI activity captures to Perfetto traces (Linux/NVIDIA)", + Long: `Convert CUPTI activity captures to Perfetto traces. + +Reads newline-delimited JSON CUPTI activity records (kernels, memory copies) +as produced by a CUPTI activity tracer, summarizes them, and writes a native +Perfetto protobuf trace viewable at ui.perfetto.dev. Kernel symbols are +demangled with c++filt when available. + +With --samples, a parallel newline-delimited NVML sample file (timestamp_ns, +power_mw, gpu_util_pct, mem_util_pct, temp_c, mem_used_bytes) is overlaid as +native counter tracks on the same normalized clock.`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + capData, err := cuptitrace.ReadCapture(args[0]) + if err != nil { + return err + } + events := capData.Events + if len(events) == 0 { + return fmt.Errorf("no CUPTI events in %s", args[0]) + } + if opts.spans { + return printSpanTable(cmd, capData, opts.spansJSON) + } + if opts.stats { + return printCuptiStats(cmd, events, gpuevent.MeasureCompleteness(capData)) + } + if opts.top > 0 { + return printCuptiTop(cmd, events, opts.top) + } + + samplesPath := cupticapture.ResolveSamples(args[0], opts.samples) + samples, err := cuptitrace.ReadSamples(samplesPath) + if err != nil { + return fmt.Errorf("read NVML samples: %w", err) + } + // Build from the decoded capture rather than its unpacked + // fields: the fields drop ClockSync, and without it the trace + // declares its own clock anchored to nothing, so two captures + // cannot be placed on a shared timeline. + capData.Samples = samples + trace, err := cuptitrace.BuildCapture(capData, args[0], cuptitrace.Options{ + PerKernelTracks: opts.perKernel, + }) + if err != nil { + return err + } + + outPath := opts.output + if outPath == "" { + outPath = stripExt(args[0]) + ".pftrace" + } + f, err := os.Create(outPath) + if err != nil { + return err + } + if err := cuptitrace.Write(trace, f); err != nil { + f.Close() + return err + } + if err := f.Close(); err != nil { + return err + } + fmt.Fprintf(cmd.OutOrStdout(), "Wrote %d events -> %s\n", len(trace.Events), outPath) + fmt.Fprintf(cmd.ErrOrStderr(), "View with: open %s at https://ui.perfetto.dev\n", outPath) + return nil + }, + } + cmd.Flags().StringVarP(&opts.output, "output", "o", opts.output, "Output Perfetto trace path") + cmd.Flags().BoolVar(&opts.stats, "stats", opts.stats, "Print summary statistics instead of writing a trace") + cmd.Flags().BoolVar(&opts.spans, "spans", opts.spans, "Print per-span setup/launch-latency/GPU/tail decomposition") + cmd.Flags().BoolVar(&opts.spansJSON, "json", opts.spansJSON, "Output --spans table as JSON") + cmd.Flags().IntVar(&opts.top, "top", opts.top, "Print the N slowest kernel launches instead of writing a trace") + cmd.Flags().BoolVar(&opts.perKernel, "per-kernel-tracks", opts.perKernel, "Give each distinct kernel its own track") + cmd.Flags().StringVar(&opts.samples, "samples", opts.samples, "Newline-delimited NVML sample file to overlay as counter tracks") + return cmd +} + +func printCuptiStats(cmd *cobra.Command, events []cuptitrace.Event, health gpuevent.Completeness) error { + var kernels, memcpies int + var totalNS uint64 + kernelTime := map[string]uint64{} + counts := map[string]int{} + for _, e := range events { + switch e.Kind { + case "kernel": + kernels++ + d := e.EndNS - e.StartNS + totalNS += d + // Through DisplayName, not Name: the shim writes the mangled + // symbol and leaves Name empty, so grouping on Name alone + // collapses every kernel into one unnamed bucket and reports + // it as "1 distinct kernels". + name := cuptitrace.Demangle(cuptitrace.DisplayName(e)) + kernelTime[name] += d + counts[name]++ + case "memcpy", "memset": + memcpies++ + } + } + out := cmd.OutOrStdout() + fmt.Fprintf(out, "CUPTI capture: %d kernels, %d memory transfers\n", kernels, memcpies) + if !health.Complete() { + fmt.Fprintf(out, "%s\n", health.Summary()) + fmt.Fprintf(out, "The totals below are a share of the run; %s\n", firstLine(health.Remedy())) + + } + fmt.Fprintf(out, "Total kernel time: %.2f ms across %d distinct kernels\n\n", float64(totalNS)/1e6, len(kernelTime)) + + type row struct { + name string + count int + ns uint64 + } + rows := make([]row, 0, len(kernelTime)) + for name, ns := range kernelTime { + rows = append(rows, row{name, counts[name], ns}) + } + sort.Slice(rows, func(i, j int) bool { return rows[i].ns > rows[j].ns }) + fmt.Fprintln(out, "Top kernels by total GPU time:") + limit := 10 + if limit > len(rows) { + limit = len(rows) + } + for _, r := range rows[:limit] { + fmt.Fprintf(out, " %8.2f ms %5dx %s\n", float64(r.ns)/1e6, r.count, r.name) + } + return nil +} + +func printCuptiTop(cmd *cobra.Command, events []cuptitrace.Event, n int) error { + sorted := make([]cuptitrace.Event, len(events)) + copy(sorted, events) + sort.Slice(sorted, func(i, j int) bool { + return sorted[i].EndNS-sorted[i].StartNS > sorted[j].EndNS-sorted[j].StartNS + }) + out := cmd.OutOrStdout() + if n > len(sorted) { + n = len(sorted) + } + for _, e := range sorted[:n] { + fmt.Fprintf(out, "%9.3f us %-48s grid=%-12s block=%-10s regs=%d\n", + float64(e.EndNS-e.StartNS)/1e3, + cuptitrace.ShortName(cuptitrace.Demangle(cuptitrace.DisplayName(e))), + e.Grid, e.Block, e.Registers) + } + return nil +} + +// firstLine keeps a multi-line remedy to one line where the surrounding +// output is a single status line. +func firstLine(s string) string { + if i := strings.IndexByte(s, '\n'); i >= 0 { + return s[:i] + } + return s +} + +func stripExt(path string) string { + if i := strings.LastIndexByte(path, '.'); i > strings.LastIndexByte(path, '/') { + return path[:i] + } + return path +} diff --git a/cmd/gputrace/cmd/cupti_test.go b/cmd/gputrace/cmd/cupti_test.go new file mode 100644 index 00000000..462e4952 --- /dev/null +++ b/cmd/gputrace/cmd/cupti_test.go @@ -0,0 +1,100 @@ +package cmd + +import ( + "bytes" + "strings" + "testing" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cuptitrace" + "github.com/tmc/gputrace/internal/gpuevent" +) + +// The shim writes the mangled symbol into raw_symbol and leaves name +// empty, which is what every real CUDA capture looks like. +const shimRecords = `{"kind":"kernel","raw_symbol":"_Z5saxpyifPfS_","start_ns":100,"end_ns":300,"stream_id":7} +{"kind":"kernel","raw_symbol":"_Z5saxpyifPfS_","start_ns":400,"end_ns":500,"stream_id":7} +{"kind":"kernel","raw_symbol":"_Z4gemvifPfS_","start_ns":600,"end_ns":900,"stream_id":7} +` + +func readRecords(t *testing.T, records string) gpuevent.Capture { + t.Helper() + cap, err := gpuevent.DecodeJSONL(strings.NewReader(records)) + if err != nil { + t.Fatal(err) + } + return cap +} + +func captureOutput(t *testing.T, run func(*cobra.Command) error) string { + t.Helper() + var buf bytes.Buffer + cmd := &cobra.Command{} + cmd.SetOut(&buf) + cmd.SetErr(&buf) + if err := run(cmd); err != nil { + t.Fatal(err) + } + return buf.String() +} + +// TestCuptiSummariesResolveNamesLikeEveryOtherReader pins the fix for two +// readers of one bundle disagreeing. --stats and --top grouped on Event.Name +// alone, which the shim never sets, so a capture of twenty named kernels +// summarized as "1 distinct kernels" with a blank name column — while pprof +// and the Perfetto writer, reading the same bundle, resolved them all. +func TestCuptiSummariesResolveNamesLikeEveryOtherReader(t *testing.T) { + cap := readRecords(t, shimRecords) + health := gpuevent.MeasureCompleteness(cap) + + stats := captureOutput(t, func(c *cobra.Command) error { + return printCuptiStats(c, cap.Events, health) + }) + if strings.Contains(stats, "1 distinct kernels") { + t.Errorf("--stats collapsed two kernels into one bucket:\n%s", stats) + } + if !strings.Contains(stats, "2 distinct kernels") { + t.Errorf("--stats does not report 2 distinct kernels:\n%s", stats) + } + + top := captureOutput(t, func(c *cobra.Command) error { + return printCuptiTop(c, cap.Events, 3) + }) + // The same name each reader shows, so a --top row can be matched + // against a pprof frame or a Perfetto slice by eye. + for _, e := range cap.Events { + want := cuptitrace.ShortName(cuptitrace.Demangle(cuptitrace.DisplayName(e))) + if !strings.Contains(top, want) { + t.Errorf("--top does not name %q:\n%s", want, top) + } + if !strings.Contains(stats, want) { + t.Errorf("--stats does not name %q:\n%s", want, stats) + } + } +} + +// TestCuptiStatsDeclaresAnIncompleteCapture: the totals a partial capture +// produces are well formed and wrong, so the reader has to say so before +// printing them. +func TestCuptiStatsDeclaresAnIncompleteCapture(t *testing.T) { + cap := readRecords(t, `{"kind":"dropped","records":900} +`+shimRecords) + out := captureOutput(t, func(c *cobra.Command) error { + return printCuptiStats(c, cap.Events, gpuevent.MeasureCompleteness(cap)) + }) + for _, want := range []string{"INCOMPLETE", "900"} { + if !strings.Contains(out, want) { + t.Errorf("--stats output does not mention %q:\n%s", want, out) + } + } + + // A complete capture stays quiet: a warning on every run is a warning + // nobody reads on the run that matters. + clean := readRecords(t, shimRecords) + out = captureOutput(t, func(c *cobra.Command) error { + return printCuptiStats(c, clean.Events, gpuevent.MeasureCompleteness(clean)) + }) + if strings.Contains(out, "INCOMPLETE") { + t.Errorf("a complete capture was reported incomplete:\n%s", out) + } +} diff --git a/cmd/gputrace/cmd/dependencies.go b/cmd/gputrace/cmd/dependencies.go index d768afba..0c8e72db 100644 --- a/cmd/gputrace/cmd/dependencies.go +++ b/cmd/gputrace/cmd/dependencies.go @@ -12,17 +12,23 @@ import ( type dependenciesOptions struct { verbose bool + limit int + all bool } -var dependenciesCmd = newDependenciesCommand(&dependenciesOptions{}) +var dependenciesCmd = newDependenciesCommand(&dependenciesOptions{limit: defaultHumanLimit}) func newDependenciesCommand(opts *dependenciesOptions) *cobra.Command { + if opts.limit == 0 { + opts.limit = defaultHumanLimit + } cmd := &cobra.Command{ Use: "dependencies ", Short: "Generate a dependency graph of operations", Hidden: true, - Long: `Analyze buffer usage to generate a dependency graph of operations/encoders. -The output is in Graphviz DOT format. + Long: `Generate a Graphviz DOT graph from decoded buffer dependency events. +Missing record types can make the graph incomplete. Human-readable DOT output +is bounded by default; use --all for the complete decoded graph. Example: gputrace dependencies trace.gputrace | dot -Tpng -o graph.png`, @@ -32,6 +38,8 @@ Example: }, } cmd.Flags().BoolVarP(&opts.verbose, "verbose", "v", false, "Show detailed parsing information") + cmd.Flags().IntVar(&opts.limit, "limit", opts.limit, "Maximum nodes and edges in DOT output") + cmd.Flags().BoolVar(&opts.all, "all", opts.all, "Show the complete dependency graph") return cmd } @@ -74,17 +82,29 @@ func runDependencies(cmd *cobra.Command, args []string, opts *dependenciesOption len(graph.Nodes), len(graph.Edges)) } - return writeDependencyGraphDOT(cmd.OutOrStdout(), graph) + limit, err := resolveHumanLimit(opts.limit, opts.all) + if err != nil { + return err + } + return writeDependencyGraphDOTLimited(cmd.OutOrStdout(), graph, limit) } func writeDependencyGraphDOT(w io.Writer, graph *trace.DependencyGraph) error { + return writeDependencyGraphDOTLimited(w, graph, -1) +} + +func writeDependencyGraphDOTLimited(w io.Writer, graph *trace.DependencyGraph, limit int) error { var buf bytes.Buffer fmt.Fprintln(&buf, "digraph G {") fmt.Fprintln(&buf, " rankdir=LR;") fmt.Fprintln(&buf, " node [shape=box, style=filled, fontname=\"Helvetica\"];") fmt.Fprintln(&buf, " edge [fontname=\"Helvetica\", fontsize=10];") + fmt.Fprintln(&buf, " // Decoded dependency events; missing record types can make this graph incomplete.") - for _, node := range graph.Nodes { + nodeCount := limitedCount(len(graph.Nodes), limit) + included := make(map[int]bool, nodeCount) + for _, node := range graph.Nodes[:nodeCount] { + included[node.ID] = true label := node.Label if len(label) > 50 { label = label[:47] + "..." @@ -92,9 +112,25 @@ func writeDependencyGraphDOT(w io.Writer, graph *trace.DependencyGraph) error { fmt.Fprintf(&buf, " n%d [label=%q];\n", node.ID, label) } + edgeCount := 0 + eligibleEdges := 0 for _, edge := range graph.Edges { + if !included[edge.From] || !included[edge.To] { + continue + } + eligibleEdges++ + if limit >= 0 && edgeCount >= limit { + continue + } label := fmt.Sprintf("%s (%s)", edge.Buffer, edge.Hazard) fmt.Fprintf(&buf, " n%d -> n%d [label=%q];\n", edge.From, edge.To, label) + edgeCount++ + } + if nodeCount < len(graph.Nodes) { + fmt.Fprintf(&buf, " // %d nodes and their incident edges omitted; use --all for the complete decoded graph.\n", len(graph.Nodes)-nodeCount) + } + if edgeCount < eligibleEdges { + fmt.Fprintf(&buf, " // %d additional edges between shown nodes omitted; use --all for the complete decoded graph.\n", eligibleEdges-edgeCount) } fmt.Fprintln(&buf, "}") diff --git a/cmd/gputrace/cmd/dependencies_test.go b/cmd/gputrace/cmd/dependencies_test.go index f5624410..1e43b48b 100644 --- a/cmd/gputrace/cmd/dependencies_test.go +++ b/cmd/gputrace/cmd/dependencies_test.go @@ -34,3 +34,29 @@ func TestWriteDependencyGraphDOTEscapesLabels(t *testing.T) { } } } + +func TestWriteDependencyGraphDOTLimit(t *testing.T) { + graph := &trace.DependencyGraph{ + Nodes: []trace.DependencyNode{ + {ID: 0, Label: "first"}, + {ID: 1, Label: "second"}, + {ID: 2, Label: "third"}, + }, + Edges: []trace.DependencyEdge{ + {From: 0, To: 1, Buffer: "a", Hazard: trace.HazardRAW}, + {From: 1, To: 2, Buffer: "b", Hazard: trace.HazardRAW}, + }, + } + + var out bytes.Buffer + if err := writeDependencyGraphDOTLimited(&out, graph, 2); err != nil { + t.Fatalf("writeDependencyGraphDOTLimited: %v", err) + } + got := out.String() + if !strings.Contains(got, "1 nodes and their incident edges omitted") { + t.Fatalf("limited DOT missing omission notice:\n%s", got) + } + if strings.Contains(got, "n2 [") || strings.Contains(got, "n1 -> n2") { + t.Fatalf("limited DOT references omitted node:\n%s", got) + } +} diff --git a/cmd/gputrace/cmd/devices.go b/cmd/gputrace/cmd/devices.go new file mode 100644 index 00000000..13e0444f --- /dev/null +++ b/cmd/gputrace/cmd/devices.go @@ -0,0 +1,70 @@ +package cmd + +import ( + "encoding/json" + "fmt" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/gpuevent" + "github.com/tmc/gputrace/internal/nvidia" +) + +var devicesOpts = struct{ json bool }{} + +var devicesCmd = &cobra.Command{ + Use: "devices", + Short: "List GPUs and capture backend capabilities on this host", + Long: `List GPUs and capture backend capabilities on this host. + +Probes every known capture backend (CUDA/NVIDIA, Metal/Apple) and reports +availability, device count, and whether kernel tracing and device counters +are usable. This is the entry point for deciding how to trace a workload +on the current machine.`, + Args: cobra.NoArgs, + RunE: func(cmd *cobra.Command, args []string) error { + backends := gpuevent.Registry() + out := cmd.OutOrStdout() + if devicesOpts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(backends) + } + for _, b := range backends { + status := "unavailable" + switch { + case !b.Available: + status = "unavailable" + case b.Tracing && b.Counters: + status = "capture + counters" + case b.Tracing: + status = "capture" + case b.Counters: + status = "counters only" + default: + status = "enumerable only" + } + line := fmt.Sprintf("%-8s %-8s %-16s", b.Name, b.Vendor, status) + if b.Devices > 0 { + line += fmt.Sprintf(" %d device(s)", b.Devices) + } + fmt.Fprintln(out, line) + if b.Detail != "" { + fmt.Fprintf(out, " %s\n", b.Detail) + } + } + // Device detail for available NVIDIA hardware. + if devices, err := nvidia.Devices(); err == nil && len(devices) > 0 { + fmt.Fprintln(out, "\nNVIDIA devices:") + for _, d := range devices { + fmt.Fprintf(out, " GPU %d: %s (%.1f GiB)\n", + d.Index, d.Name, float64(d.MemoryTotal)/(1<<30)) + } + } + return nil + }, +} + +func init() { + devicesCmd.Flags().BoolVar(&devicesOpts.json, "json", false, "Output in JSON format") + rootCmd.AddCommand(devicesCmd) +} diff --git a/cmd/gputrace/cmd/diff.go b/cmd/gputrace/cmd/diff.go index fd581c8d..5d3f14cf 100644 --- a/cmd/gputrace/cmd/diff.go +++ b/cmd/gputrace/cmd/diff.go @@ -9,29 +9,31 @@ import ( "github.com/spf13/cobra" "github.com/tmc/gputrace/internal/difftrace" + "github.com/tmc/gputrace/internal/environment" ) type diffOptions struct { - JSON bool - CSV bool - By string - Limit int - MinDeltaUs int - OnlyEncoder int - OnlyFunction string - ShowMatches bool - ShowUnmatched bool - ShowOccur bool - Explain bool - Quick bool - Divergence bool - DivergenceUs int - ByEncoder bool - MDOut string - PerfettoOut string - BenchDir string - Left string - Right string + JSON bool + CSV bool + By string + Limit int + MinDeltaUs int + OnlyEncoder int + OnlyFunction string + ShowMatches bool + ShowUnmatched bool + ShowOccur bool + Explain bool + Quick bool + Divergence bool + DivergenceUs int + ByEncoder bool + MDOut string + PerfettoOut string + BenchDir string + Left string + Right string + AllowCrossEnvironment bool } var diffCmd = newDiffCommand(&diffOptions{Limit: 20, OnlyEncoder: -1}) @@ -46,9 +48,15 @@ This command supports .gputrace bundles and -perfdata.gputrace bundles. It reports total deltas, function-level contributors, encoder/pipeline deltas, spike windows, unnamed dispatch impact, and matched/unmatched dispatches. +Given two .gpucapture bundles (Linux/NVIDIA), it compares them kernel by +kernel instead: GPU time moved per kernel, launch counts, theoretical +occupancy, the busy/idle budget, and the kernels present in only one of +the two captures. + Examples: + gputrace diff base.gpucapture variant.gpucapture gputrace diff go-perfdata.gputrace py-perfdata.gputrace - gputrace diff --bench-dir ~/bench-traces --quick --by-encoder + gputrace diff --bench-dir ~/bench-traces --quick --explain --by-encoder gputrace diff --bench-dir ~/bench-traces --left go.gputrace --right py.gputrace gputrace diff a.gputrace b.gputrace --by function --limit 25 --explain gputrace diff a.gputrace b.gputrace --by encoder --only-encoder 2 @@ -80,6 +88,7 @@ Examples: cmd.Flags().StringVar(&opts.BenchDir, "bench-dir", "", "Auto-discover newest Go/Python perfdata pair from benchmark directory") cmd.Flags().StringVar(&opts.Left, "left", "", "Explicit left trace path (overrides auto-discovery)") cmd.Flags().StringVar(&opts.Right, "right", "", "Explicit right trace path (overrides auto-discovery)") + cmd.Flags().BoolVar(&opts.AllowCrossEnvironment, "allow-cross-environment", false, "Show descriptive deltas when exact environment gates differ or are unavailable") return cmd } @@ -88,6 +97,12 @@ func init() { } func runDiff(cmd *cobra.Command, args []string, opts diffOptions) error { + // CUDA captures carry no encoder structure to align dispatches + // against, so they take the kernel-name comparison instead of the + // Metal dispatch-alignment one. + if len(args) == 2 && isCaptureInput(args[0]) && isCaptureInput(args[1]) { + return runCaptureDiff(cmd, args[0], args[1], opts) + } if err := opts.validate(args); err != nil { return err } @@ -114,6 +129,15 @@ func runDiff(cmd *cobra.Command, args []string, opts diffOptions) error { if err != nil { return fmt.Errorf("load trace B: %w", err) } + environmentComparison, err := environment.Compare(a.Environment, b.Environment, opts.AllowCrossEnvironment) + if err != nil { + return err + } + if environmentComparison.Label == "incompatible" { + mismatches := append([]string(nil), environmentComparison.ExactMismatches...) + mismatches = append(mismatches, environmentComparison.CapabilityMismatches...) + return fmt.Errorf("compare traces: incompatible or unavailable environment evidence (%s); use --allow-cross-environment for descriptive, non-causal deltas", strings.Join(mismatches, ", ")) + } aligned := difftrace.AlignDispatches(a, b, difftrace.AlignOptions{ OnlyEncoder: opts.OnlyEncoder, @@ -121,6 +145,10 @@ func runDiff(cmd *cobra.Command, args []string, opts diffOptions) error { MinDeltaUs: opts.MinDeltaUs, }) report := difftrace.BuildReport(a, b, aligned, difftrace.ReportOptions{Limit: opts.Limit, MinDeltaUs: opts.MinDeltaUs}) + report.Environment = &environmentComparison + if environmentComparison.Label == "cross-environment, not causally attributable" { + report.Warnings = append(report.Warnings, "cross-environment comparison: deltas are descriptive and not causally attributable") + } if diffByIncludes(opts.By, "pipeline-pairs") { report.PipelinePairs = difftrace.BuildPipelinePairs(a, b) } @@ -167,7 +195,7 @@ func runDiff(cmd *cobra.Command, args []string, opts diffOptions) error { var text string if opts.Quick { - text = difftrace.RenderQuick(report, 10) + text = difftrace.RenderQuick(report, 10, opts.Explain) if opts.ByEncoder { text += "\n" + difftrace.RenderEncoderFocus(report, opts.Limit) } @@ -242,8 +270,8 @@ func (o diffOptions) validate(args []string) error { if strings.TrimSpace(o.By) != "" { return fmt.Errorf("--quick cannot be combined with --by") } - if o.ShowMatches || o.ShowUnmatched || o.ShowOccur || o.Explain { - return fmt.Errorf("--quick cannot be combined with --show-matches/--show-unmatched/--show-occurrences/--explain") + if o.ShowMatches || o.ShowUnmatched || o.ShowOccur { + return fmt.Errorf("--quick cannot be combined with --show-matches/--show-unmatched/--show-occurrences") } } if o.ByEncoder && strings.TrimSpace(o.By) != "" { diff --git a/cmd/gputrace/cmd/diff_capture.go b/cmd/gputrace/cmd/diff_capture.go new file mode 100644 index 00000000..2a6357bc --- /dev/null +++ b/cmd/gputrace/cmd/diff_capture.go @@ -0,0 +1,235 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "io" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/gate" + "github.com/tmc/gputrace/internal/gpuevent" +) + +// warnCrossHost labels a two-bundle comparison with its host provenance. +// Timing deltas between bundles from different hosts (or unverifiable +// sessions) are noise; saying so belongs to the tool, not the reader. +func warnCrossHost(out io.Writer, base, variant string) { + a := gate.ReadHostProvenance(base) + b := gate.ReadHostProvenance(variant) + switch { + case a.Recorded && b.Recorded && a.Hostname != b.Hostname: + fmt.Fprintf(out, "warning: CROSS-HOST comparison: %s (%s) vs %s (%s) — timing deltas are noise; structural counts remain comparable\n\n", + a.Hostname, a.Device, b.Hostname, b.Device) + case !a.Recorded || !b.Recorded: + fmt.Fprint(out, "warning: host provenance absent from at least one bundle: cross-session comparison cannot be verified\n\n") + } +} + +// isCaptureInput reports whether a diff argument names a CUDA capture: a +// .gpucapture bundle, or a bare JSONL activity file. +func isCaptureInput(path string) bool { + if cupticapture.IsBundle(path) { + return true + } + return hasSuffixFold(path, ".jsonl") +} + +func hasSuffixFold(s, suffix string) bool { + if len(s) < len(suffix) { + return false + } + return equalFold(s[len(s)-len(suffix):], suffix) +} + +func equalFold(a, b string) bool { + if len(a) != len(b) { + return false + } + for i := 0; i < len(a); i++ { + ca, cb := a[i], b[i] + if 'A' <= ca && ca <= 'Z' { + ca += 'a' - 'A' + } + if 'A' <= cb && cb <= 'Z' { + cb += 'a' - 'A' + } + if ca != cb { + return false + } + } + return true +} + +// runCaptureDiff compares two CUDA captures kernel by kernel. It is the +// CUPTI counterpart of the Metal dispatch-alignment diff: matching is by +// kernel name rather than by dispatch position, because CUDA captures have +// no encoder structure to align against. +func runCaptureDiff(cmd *cobra.Command, base, variant string, opts diffOptions) error { + baseReport, err := loadCaptureReport(base) + if err != nil { + return fmt.Errorf("load base capture: %w", err) + } + variantReport, err := loadCaptureReport(variant) + if err != nil { + return fmt.Errorf("load variant capture: %w", err) + } + cmp := gpuevent.CompareCaptures(baseReport, variantReport) + out := cmd.OutOrStdout() + if opts.JSON { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(cmp) + } + warnCrossHost(out, base, variant) + writeCaptureDiff(out, cmp, base, variant, opts.Limit) + return nil +} + +func writeCaptureDiff(out io.Writer, c *gpuevent.CaptureComparison, base, variant string, limit int) { + fmt.Fprintf(out, "base: %s\nvariant: %s\n\n", base, variant) + fmt.Fprintf(out, "verdict: %s — %s\n", c.Verdict, c.Summary) + if c.Verdict == gpuevent.CaptureInconclusive { + return + } + fmt.Fprintf(out, "kernel time: %s -> %s (%+.1f%%)\n", + dur(c.BaseTotalNS), dur(c.VariantTotalNS), c.TotalDeltaPct) + + u := c.Utilization + if u.BaseWallSpanNS > 0 || u.VariantWallSpanNS > 0 { + fmt.Fprintf(out, "wall span: %s -> %s\n", dur(u.BaseWallSpanNS), dur(u.VariantWallSpanNS)) + fmt.Fprintf(out, "occupancy: %.1f%% -> %.1f%% (%+.1f points)\n", + u.BaseOccupancyPct, u.VariantOccupancyPct, u.VariantOccupancyPct-u.BaseOccupancyPct) + fmt.Fprintf(out, "idle budget: %s across %d gaps -> %s across %d gaps (mean gap %s -> %s)\n", + dur(u.BaseIdleNS), u.BaseGapCount, dur(u.VariantIdleNS), u.VariantGapCount, + dur(u.BaseMeanGapNS), dur(u.VariantMeanGapNS)) + } + + rows := c.KernelDeltas + if limit > 0 && len(rows) > limit { + rows = rows[:limit] + } + fmt.Fprintf(out, "\nPer-kernel deltas (by GPU time moved):\n") + fmt.Fprintf(out, " %9s %-12s %-22s %-14s %s\n", "TOTAL Δ", "COUNT", "MEAN", "OCCUPANCY", "KERNEL") + for _, d := range rows { + mark := " " + switch d.OnlyIn { + case "base": + mark = "- " + case "variant": + mark = "+ " + } + mean := fmt.Sprintf("%9s -> %-9s", dur(d.BaseMeanNS), dur(d.VariantMeanNS)) + if d.Heterogeneous() { + // The mean is an average over launch geometries that are not + // comparable. Withhold it rather than print a number the reader + // would have no reason to distrust. + mean = fmt.Sprintf("%-22s", fmt.Sprintf("(%d shapes)", d.ShapeCount)) + } + fmt.Fprintf(out, "%s%9s %5d->%-5d %s %5s -> %-5s %s\n", + mark, signedDur(d.TotalDeltaNS), + d.BaseCount, d.VariantCount, + mean, + occupancyOrDash(d.BaseOccupancy), occupancyOrDash(d.VarOccupancy), + shortKernel(d.Name)) + } + if n := len(c.KernelDeltas) - len(rows); n > 0 { + fmt.Fprintf(out, " ... %d more (raise --limit)\n", n) + } + if n := len(c.HeterogeneousKernels); n > 0 { + fmt.Fprintf(out, "\n%d kernel%s launched at more than one geometry; those means are\n"+ + "withheld above because a mean across geometries describes no launch that\n"+ + "occurred. Per-geometry rows follow.\n", n, plural(n)) + } + writeShapeDeltas(out, c, limit) + if len(c.OnlyInBase) > 0 { + fmt.Fprintf(out, "\nOnly in base (%d): %s\n", len(c.OnlyInBase), joinShort(c.OnlyInBase, 3)) + } + if len(c.OnlyInVariant) > 0 { + fmt.Fprintf(out, "Only in variant (%d): %s\n", len(c.OnlyInVariant), joinShort(c.OnlyInVariant, 3)) + } +} + +// signedDur renders a delta with its direction; negative means the +// variant spent less GPU time. +func signedDur(ns int64) string { + if ns < 0 { + return "-" + dur(uint64(-ns)) + } + return "+" + dur(uint64(ns)) +} + +func occupancyOrDash(pct float64) string { + if pct <= 0 { + return "-" + } + return fmt.Sprintf("%.0f%%", pct) +} + +func joinShort(names []string, max int) string { + shown := names + if len(shown) > max { + shown = shown[:max] + } + out := "" + for i, n := range shown { + if i > 0 { + out += ", " + } + out += shortKernel(n) + } + if len(names) > len(shown) { + out += fmt.Sprintf(", and %d more", len(names)-len(shown)) + } + return out +} + +// writeShapeDeltas prints the comparison keyed on launch geometry. This is +// the table to read when asking whether a kernel got slower: every row is a +// single population on each side, and blocks-per-launch is printed beside +// the duration because it, not duration, distinguishes a launch that fills +// the device from one that leaves it idle. +func writeShapeDeltas(out io.Writer, c *gpuevent.CaptureComparison, limit int) { + rows := c.ShapeDeltas + if len(rows) == 0 { + return + } + if limit > 0 && len(rows) > limit { + rows = rows[:limit] + } + fmt.Fprintf(out, "\nPer-launch-geometry deltas (by GPU time moved):\n") + fmt.Fprintf(out, " %9s %8s %-12s %-22s %s\n", "TOTAL Δ", "BLOCKS", "COUNT", "MEAN", "KERNEL / GRID / BLOCK") + for _, d := range rows { + mark := " " + switch d.OnlyIn { + case "base": + mark = "- " + case "variant": + mark = "+ " + } + // An unequal launch count means the totals differ partly because the + // work differs. Mark it so the total column is not read as a speed + // statement. + count := fmt.Sprintf("%5d->%-5d", d.BaseCount, d.VariantCount) + if !d.CountsMatch && d.OnlyIn == "" { + count = fmt.Sprintf("%5d->%-5d!", d.BaseCount, d.VariantCount) + } + fmt.Fprintf(out, "%s%9s %8s %-12s %9s -> %-9s %s %s/%s\n", + mark, signedDur(d.TotalDeltaNS), + blocksOrDash(d.Blocks), + count, + dur(d.BaseMeanNS), dur(d.VariantMeanNS), + shortKernel(d.Name), d.Grid, d.Block) + } + if n := len(c.ShapeDeltas) - len(rows); n > 0 { + fmt.Fprintf(out, " ... %d more (raise --limit)\n", n) + } + fmt.Fprintf(out, " ! = launch counts differ, so the total mixes per-launch cost with work done\n") +} + +func blocksOrDash(n uint64) string { + if n == 0 { + return "-" + } + return fmt.Sprintf("%d", n) +} diff --git a/cmd/gputrace/cmd/diff_capture_test.go b/cmd/gputrace/cmd/diff_capture_test.go new file mode 100644 index 00000000..bb7d996a --- /dev/null +++ b/cmd/gputrace/cmd/diff_capture_test.go @@ -0,0 +1,69 @@ +package cmd + +import ( + "bytes" + "strings" + "testing" + + "github.com/tmc/gputrace/internal/gpuevent" +) + +func TestIsCaptureInput(t *testing.T) { + tests := []struct { + path string + want bool + }{ + {"events.jsonl", true}, + {"run.JSONL", true}, + {"trace.gputrace", false}, + {"", false}, + } + for _, tt := range tests { + if got := isCaptureInput(tt.path); got != tt.want { + t.Errorf("isCaptureInput(%q) = %v, want %v", tt.path, got, tt.want) + } + } +} + +func TestWriteCaptureDiff(t *testing.T) { + cmp := &gpuevent.CaptureComparison{ + Verdict: gpuevent.CaptureImproved, + Summary: "faster overall", + BaseTotalNS: 2_000_000, + VariantTotalNS: 1_000_000, + TotalDeltaPct: -50, + KernelDeltas: []gpuevent.KernelDelta{ + {Name: "hot", BaseCount: 10, VariantCount: 10, BaseMeanNS: 200_000, VariantMeanNS: 100_000, TotalDeltaNS: -1_000_000, BaseOccupancy: 50, VarOccupancy: 75}, + {Name: "gone", BaseCount: 4, VariantCount: 0, BaseMeanNS: 1000, TotalDeltaNS: -4000, OnlyIn: "base"}, + }, + OnlyInBase: []string{"gone"}, + Utilization: gpuevent.UtilizationDelta{BaseWallSpanNS: 10_000_000, VariantWallSpanNS: 4_000_000, BaseOccupancyPct: 20, VariantOccupancyPct: 25, BaseIdleNS: 8_000_000, VariantIdleNS: 3_000_000, BaseGapCount: 9, VariantGapCount: 4}, + } + var buf bytes.Buffer + writeCaptureDiff(&buf, cmp, "base.gpucapture", "variant.gpucapture", 20) + got := buf.String() + for _, want := range []string{ + "verdict: improved", + "-1.00ms", // signed delta on the hot kernel + "50% -> 75%", // occupancy movement + "20.0% -> 25.0%", // occupancy budget + "Only in base (1)", // the kernel one side stopped running + "- ", // the row marker for it + } { + if !strings.Contains(got, want) { + t.Errorf("output missing %q:\n%s", want, got) + } + } +} + +func TestSignedDur(t *testing.T) { + tests := []struct { + in int64 + want string + }{{-1_500_000, "-1.50ms"}, {2000, "+2.0us"}, {0, "+0ns"}} + for _, tt := range tests { + if got := signedDur(tt.in); got != tt.want { + t.Errorf("signedDur(%d) = %q, want %q", tt.in, got, tt.want) + } + } +} diff --git a/cmd/gputrace/cmd/diff_flags_test.go b/cmd/gputrace/cmd/diff_flags_test.go index 480800e7..9f841cbf 100644 --- a/cmd/gputrace/cmd/diff_flags_test.go +++ b/cmd/gputrace/cmd/diff_flags_test.go @@ -170,6 +170,15 @@ func TestDiffOptionsValidate(t *testing.T) { return o }(), }, + { + name: "quick explain allowed", + opts: func() diffOptions { + o := base + o.Quick = true + o.Explain = true + return o + }(), + }, { name: "divergence with encoder by allowed", opts: func() diffOptions { @@ -206,7 +215,7 @@ func TestDiffOptionsValidate(t *testing.T) { o.ShowUnmatched = true return o }(), - wantErr: "--quick cannot be combined with --show-matches/--show-unmatched/--show-occurrences/--explain", + wantErr: "--quick cannot be combined with --show-matches/--show-unmatched/--show-occurrences", }, { name: "by encoder with by rejected", diff --git a/cmd/gputrace/cmd/doctor.go b/cmd/gputrace/cmd/doctor.go new file mode 100644 index 00000000..b5b03d9e --- /dev/null +++ b/cmd/gputrace/cmd/doctor.go @@ -0,0 +1,114 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "io" + "strings" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/gpudoctor" +) + +type doctorOptions struct { + json bool + target string +} + +var doctorCmd = newDoctorCommand(&doctorOptions{}) + +func newDoctorCommand(opts *doctorOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "doctor [target]", + Short: "Diagnose the GPU profiling environment", + Long: `Diagnose the GPU profiling environment and print what to do about it. + +Reports the NVIDIA driver, the CUDA toolkits and CUPTI libraries +installed, every nsys on the system with a verdict on whether its default +CUDA tracing works here, whether GPU performance counters are restricted +to admin users, and whether the capture shim builds. + +Two of these failures are silent rather than loud: an nsys whose hardware +tracing drops every kernel record still writes a healthy-looking report, +and a CUPTI older than the running driver records nothing at all. Both +read as "the workload launched no kernels". + +Pass a workload binary to also diagnose it for capturability: dynamic +CUDA linkage, and whether it is a Go binary needing an in-process flush. + +Pass a .gpucapture bundle instead to diagnose the capture itself. Empty is +not the only way a capture goes wrong: one that lost activity records comes +back half, and half renders, summarizes, and diffs into confident numbers. +The bundle check reports the dropped-record count and cross-checks the +CUDA-graph executions against what the recorded launch counts imply. + +Examples: + gputrace doctor + gputrace doctor ./my-workload + gputrace doctor run.gpucapture + gputrace doctor --json`, + Args: cobra.MaximumNArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + if len(args) == 1 { + opts.target = args[0] + } + rep := gpudoctor.Run(gpudoctor.Options{Target: opts.target}) + if opts.json { + enc := json.NewEncoder(cmd.OutOrStdout()) + enc.SetIndent("", " ") + return enc.Encode(rep) + } + writeDoctorReport(cmd.OutOrStdout(), rep) + return nil + }, + } + cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output in JSON format") + return cmd +} + +// writeDoctorReport renders the checks as a status column plus detail, so +// a failing row is findable without reading the whole page. +func writeDoctorReport(out io.Writer, rep *gpudoctor.Report) { + for _, c := range rep.Checks { + fmt.Fprintf(out, "%-6s %-16s %s\n", doctorMark(c.Status), c.Name, c.Detail) + // Notes and remedies print one source line each: they carry paths + // and shell commands that reflowing would corrupt. + for _, n := range c.Notes { + fmt.Fprintf(out, " %s\n", n) + } + if c.Remedy != "" { + for i, line := range strings.Split(c.Remedy, "\n") { + label := " fix: " + if i > 0 { + label = " " + } + fmt.Fprintf(out, "%s%s\n", label, line) + } + } + } + switch rep.Worst() { + case gpudoctor.StatusFail: + fmt.Fprintln(out, "\nSomething here will not profile correctly; see the fix lines above.") + case gpudoctor.StatusWarn: + fmt.Fprintln(out, "\nUsable, with caveats noted above.") + default: + fmt.Fprintln(out, "\nEnvironment looks good for capture.") + } +} + +func doctorMark(s gpudoctor.Status) string { + switch s { + case gpudoctor.StatusOK: + return "ok" + case gpudoctor.StatusWarn: + return "warn" + case gpudoctor.StatusFail: + return "FAIL" + default: + return "skip" + } +} + +func init() { + rootCmd.AddCommand(doctorCmd) +} diff --git a/cmd/gputrace/cmd/dot_pprof.go b/cmd/gputrace/cmd/dot_pprof.go new file mode 100644 index 00000000..2c5821c3 --- /dev/null +++ b/cmd/gputrace/cmd/dot_pprof.go @@ -0,0 +1,110 @@ +package cmd + +import ( + "fmt" + "strings" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cudagraphdot" + "github.com/tmc/gputrace/internal/cuptiprofile" + "github.com/tmc/gputrace/internal/cuptitrace" +) + +type dotPprofOptions struct { + output string +} + +var dotPprofOpts = &dotPprofOptions{} + +var dotPprofCmd = newDotPprofCommand(dotPprofOpts) + +func newDotPprofCommand(opts *dotPprofOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "dot-pprof ...", + Short: "Convert CUDA-graph DOT dumps to a pprof structure profile", + Long: `Convert CUDA-graph DOT dumps to a pprof structure profile. + +Reads the dumps MLX writes when MLX_SAVE_CUDA_GRAPHS_DOT_FILE is set (one +file per graph commit) and emits a profile of what the graphs commit, with +no timing attached: + + sample_type: kernel_count (count), graph_commits (count) + stack: graph_ -> [child graph...] -> + +The dumps are written by the same libmlx code whichever language binding +drove it, so diffing two of them is a same-instrument comparison: no +cross-stack calibration, no GPU hold, and no run-to-run variance to average +away. One dump per side is the whole measurement, unlike timing. + + gputrace dot-pprof py-dots -o py.pb.gz + gputrace dot-pprof go-dots -o go.pb.gz + go tool pprof -top -diff_base=py.pb.gz go.pb.gz + +That last line is a signed multiset difference over kernel signatures: it +names which kernels one side commits and the other does not, which an +aggregate count comparison cannot. Diff at -top granularity — graph ids are +assigned per run and do not match across two runs, so stack-level diffs +compare labels, not structure. + +Counting rule: a node drawn as a rectangle whose label names another graph +is a child-graph node, not a kernel. It is flattened, and its kernels are +counted once per instantiation. + +Join the structure to measured cost with: + + gputrace pprof .gpucapture --dot -o joined.pb.gz`, + Args: cobra.MinimumNArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runDotPprof(cmd, args, opts) + }, + } + cmd.Flags().StringVarP(&opts.output, "output", "o", opts.output, "Output pprof file path (default: .pprof)") + return cmd +} + +func runDotPprof(cmd *cobra.Command, args []string, opts *dotPprofOptions) error { + var files []*cudagraphdot.File + for _, arg := range args { + parsed, err := loadGraphDumps(arg) + if err != nil { + return err + } + files = append(files, parsed...) + } + + var nodes []cuptiprofile.StructureNode + var commits []string + mangled := map[string]string{} + for _, f := range files { + for _, k := range f.Kernels() { + name := cuptitrace.Demangle(k.Symbol) + mangled[name] = k.Symbol + nodes = append(nodes, cuptiprofile.StructureNode{GraphPath: k.Path, Symbol: name}) + } + commits = append(commits, f.Roots...) + } + + prof, stats, err := cuptiprofile.BuildStructure(nodes, commits) + if err != nil { + return fmt.Errorf("%s: %w", strings.Join(args, " "), err) + } + cuptiprofile.SetSystemNames(prof, mangled) + + outPath := opts.output + if outPath == "" { + outPath = stripExt(args[0]) + ".pprof" + } + if err := writeProfile(prof, outPath); err != nil { + return err + } + + status := pprofStatusWriter(outPath) + fmt.Fprintf(status, "Wrote %d kernel nodes from %d dumps (%d graph commits) -> %s\n", + stats.StructureNodes, len(files), stats.Commits, outPath) + fmt.Fprintf(cmd.ErrOrStderr(), "View with: go tool pprof -top %s\n", outPath) + return nil +} + +func init() { + rootCmd.AddCommand(dotPprofCmd) +} diff --git a/cmd/gputrace/cmd/dump.go b/cmd/gputrace/cmd/dump.go index b0f96f5d..dbe044c1 100644 --- a/cmd/gputrace/cmd/dump.go +++ b/cmd/gputrace/cmd/dump.go @@ -88,6 +88,9 @@ func runDump(cmd *cobra.Command, args []string, opts dumpOptions) error { if err != nil { return fmt.Errorf("open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } apiList, err := trace.ParseAPICallList() if err != nil { @@ -111,10 +114,27 @@ func runDump(cmd *cobra.Command, args []string, opts dumpOptions) error { if err != nil { return fmt.Errorf("format api calls: %w", err) } + if opts.dispatchOnly && dumpFormattedCallCount(apiList) == 0 { + if statistics, statErr := gputrace.ExtractStatistics(trace); statErr == nil && statistics.DispatchCalls > 0 { + fmt.Fprintf(w, "\nDecoded dispatch API calls: 0/%d\n", statistics.DispatchCalls) + fmt.Fprintln(w, "The trace contains dispatch work, but this API-call decoder did not recover its dispatch records.") + } + } return nil } +func dumpFormattedCallCount(apiList *gputrace.APICallList) int { + if apiList == nil { + return 0 + } + total := 0 + for _, cb := range apiList.CommandBuffers { + total += len(cb.Calls) + } + return total +} + func validateDumpOptions(opts dumpOptions) error { if opts.commandBufferIndex < -1 { return fmt.Errorf("--command-buffer must be >= -1") diff --git a/cmd/gputrace/cmd/encoders.go b/cmd/gputrace/cmd/encoders.go index 1b52631d..3f627813 100644 --- a/cmd/gputrace/cmd/encoders.go +++ b/cmd/gputrace/cmd/encoders.go @@ -13,6 +13,8 @@ import ( type encodersOptions struct { verbose bool json bool + limit int + all bool } var encodersCmd = newEncodersCommand(&encodersOptions{}) @@ -20,12 +22,12 @@ var encodersCmd = newEncodersCommand(&encodersOptions{}) func newEncodersCommand(opts *encodersOptions) *cobra.Command { cmd := &cobra.Command{ Use: "encoders ", - Short: "List compute command encoders in a GPU trace", - Long: `List all Metal compute command encoders found in a GPU trace. + Short: "Report compute-encoder counts and observed CS labels", + Long: `Report the best available compute-encoder count and list observed CS labels. -This command parses Cul records to identify compute command encoder -creation and usage. Compute encoders are used to encode compute -commands (kernel dispatches) into command buffers. +The count uses decoded compute-encoder records and available profiler metadata. +The listed CS records are submission/debug labels; they often name kernels and +must not be interpreted as one compute encoder per row. Examples: gputrace encoders trace.gputrace @@ -37,6 +39,8 @@ Examples: } cmd.Flags().BoolVarP(&opts.verbose, "verbose", "v", false, "Show verbose output with encoder details") cmd.Flags().BoolVar(&opts.json, "json", false, "Output in JSON format") + cmd.Flags().IntVar(&opts.limit, "limit", defaultHumanLimit, "Maximum CS-label rows in human output") + cmd.Flags().BoolVar(&opts.all, "all", false, "Show every CS-label row in human output") return cmd } @@ -62,25 +66,35 @@ func runEncoders(cmd *cobra.Command, args []string, opts *encodersOptions) error if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } // Parse compute encoders - encoders, err := trace.ParseComputeEncoders() - if err != nil { - return fmt.Errorf("failed to parse compute encoders: %w", err) - } + encoders := trace.ParseComputeEncoders() if opts.json { return writeEncodersJSON(cmd.OutOrStdout(), encoders) } + limit, err := resolveHumanLimit(opts.limit, opts.all) + if err != nil { + return err + } + statistics, err := gputrace.ExtractStatistics(trace) + if err != nil { + return fmt.Errorf("extract encoder statistics: %w", err) + } + computeEncoderCount := statistics.ComputeEncoders commandBufferCount := 0 var commandBuffers []encodersCommandBufferSummary if opts.verbose { - cbs, err := trace.ParseCommandBuffers() - if err == nil && len(cbs) > 0 { + capture, err := gputrace.OpenCapture(trace) + if err == nil && len(capture.CommandBuffers()) > 0 { + cbs := capture.CommandBuffers() commandBufferCount = len(cbs) for _, cb := range cbs { - dcb, err := gputrace.ParseDetailedCommandBuffer(trace, cb.Index) + dcb, err := capture.Detailed(cb.Index) if err != nil { continue } @@ -92,7 +106,7 @@ func runEncoders(cmd *cobra.Command, args []string, opts *encodersOptions) error } } - return writeEncodersText(cmd.OutOrStdout(), encoders, commandBufferCount, commandBuffers) + return writeEncodersText(cmd.OutOrStdout(), computeEncoderCount, encoders, commandBufferCount, commandBuffers, limit) } func writeEncodersJSON(w io.Writer, encoders []*gputrace.ComputeEncoder) error { @@ -119,11 +133,15 @@ func writeEncodersJSON(w io.Writer, encoders []*gputrace.ComputeEncoder) error { return nil } -func writeEncodersText(w io.Writer, encoders []*gputrace.ComputeEncoder, commandBufferCount int, commandBuffers []encodersCommandBufferSummary) error { - if _, err := fmt.Fprintf(w, "%d encoders:\n", len(encoders)); err != nil { +func writeEncodersText(w io.Writer, computeEncoderCount int, encoders []*gputrace.ComputeEncoder, commandBufferCount int, commandBuffers []encodersCommandBufferSummary, limit int) error { + if _, err := fmt.Fprintf(w, "Compute encoders: %d\n", computeEncoderCount); err != nil { + return fmt.Errorf("write encoders: %w", err) + } + if _, err := fmt.Fprintf(w, "Observed CS labels: %d (submission/debug labels; not encoder instances)\n", len(encoders)); err != nil { return fmt.Errorf("write encoders: %w", err) } - for _, encoder := range encoders { + shown := limitedCount(len(encoders), limit) + for _, encoder := range encoders[:shown] { var err error if encoder.Label != "" { _, err = fmt.Fprintf(w, " %3d: %s\n", encoder.Index, encoder.Label) @@ -134,10 +152,15 @@ func writeEncodersText(w io.Writer, encoders []*gputrace.ComputeEncoder, command return fmt.Errorf("write encoders: %w", err) } } + if shown < len(encoders) { + if _, err := fmt.Fprintf(w, "... %d more CS labels omitted (use --all)\n", len(encoders)-shown); err != nil { + return fmt.Errorf("write encoders: %w", err) + } + } if commandBufferCount > 0 { - if _, err := fmt.Fprintf(w, "\n%d command buffers (%.1f encoders/buffer avg)\n", - commandBufferCount, float64(len(encoders))/float64(commandBufferCount)); err != nil { + if _, err := fmt.Fprintf(w, "\nExplicit encoder markers decoded per command buffer (%d buffers):\n", + commandBufferCount); err != nil { return fmt.Errorf("write encoders: %w", err) } for _, cb := range commandBuffers { diff --git a/cmd/gputrace/cmd/encoders_test.go b/cmd/gputrace/cmd/encoders_test.go index 05d8587d..3b885084 100644 --- a/cmd/gputrace/cmd/encoders_test.go +++ b/cmd/gputrace/cmd/encoders_test.go @@ -45,7 +45,8 @@ func TestWriteEncodersText(t *testing.T) { }{ { name: "normal", - want: "2 encoders:\n" + + want: "Compute encoders: 2\n" + + "Observed CS labels: 2 (submission/debug labels; not encoder instances)\n" + " 0: kernel_a\n" + " 7: (unlabeled) 0x20\n", }, @@ -56,11 +57,12 @@ func TestWriteEncodersText(t *testing.T) { {index: 0, encoderCount: 1}, {index: 1, encoderCount: 2}, }, - want: "2 encoders:\n" + + want: "Compute encoders: 2\n" + + "Observed CS labels: 2 (submission/debug labels; not encoder instances)\n" + " 0: kernel_a\n" + " 7: (unlabeled) 0x20\n" + "\n" + - "2 command buffers (1.0 encoders/buffer avg)\n" + + "Explicit encoder markers decoded per command buffer (2 buffers):\n" + " CB 0: 1 encoders\n" + " CB 1: 2 encoders\n", }, @@ -69,7 +71,7 @@ func TestWriteEncodersText(t *testing.T) { for _, tt := range tests { t.Run(tt.name, func(t *testing.T) { var out bytes.Buffer - err := writeEncodersText(&out, testEncoders(), tt.commandBufferCount, tt.commandBuffers) + err := writeEncodersText(&out, 2, testEncoders(), tt.commandBufferCount, tt.commandBuffers, -1) if err != nil { t.Fatalf("writeEncodersText: %v", err) } @@ -84,7 +86,7 @@ func TestRunEncodersJSONUsesCommandOutput(t *testing.T) { var out bytes.Buffer command := &cobra.Command{} command.SetOut(&out) - opts := &encodersOptions{json: true} + opts := &encodersOptions{json: true, limit: defaultHumanLimit} stdout, err := captureStdout(t, func() error { return runEncoders(command, []string{testEncodersTracePath(t)}, opts) @@ -116,7 +118,7 @@ func TestRunEncodersTextUsesCommandOutput(t *testing.T) { var out bytes.Buffer command := &cobra.Command{} command.SetOut(&out) - opts := &encodersOptions{} + opts := &encodersOptions{limit: defaultHumanLimit} stdout, err := captureStdout(t, func() error { return runEncoders(command, []string{testEncodersTracePath(t)}, opts) @@ -127,7 +129,7 @@ func TestRunEncodersTextUsesCommandOutput(t *testing.T) { if stdout != "" { t.Fatalf("os stdout = %q, want empty", stdout) } - if got := out.String(); !strings.Contains(got, " encoders:\n") { + if got := out.String(); !strings.Contains(got, "Compute encoders:") { t.Fatalf("command output = %q, want encoder header", got) } } diff --git a/cmd/gputrace/cmd/export_counters.go b/cmd/gputrace/cmd/export_counters.go index b732823a..451d0193 100644 --- a/cmd/gputrace/cmd/export_counters.go +++ b/cmd/gputrace/cmd/export_counters.go @@ -6,7 +6,6 @@ import ( "github.com/spf13/cobra" "github.com/tmc/gputrace" - "github.com/tmc/gputrace/internal/counter" ) var exportCountersCmd = newExportCountersCommand(&exportCountersOptions{}) @@ -22,8 +21,9 @@ func newExportCountersCommand(opts *exportCountersOptions) *cobra.Command { Hidden: true, Long: `Export performance counter data in Xcode Instruments Counters.csv format. -Generates a 246-column CSV file matching the exact format used by Xcode -Instruments when exporting GPU performance counter data. This includes: +Generates a 246-column CSV with the same column schema used by an Xcode +Instruments counter export. Schema compatibility does not mean that every row +contains source-backed Xcode measurements. This includes: Metadata Columns (1-5): - Index: Sequential row number @@ -34,7 +34,7 @@ Metadata Columns (1-5): Performance Metrics (6-246): 241 performance counter metrics including: - - ALU Utilization, Kernel Occupancy + - ALU Utilization - Memory bandwidth (Buffer/Texture Device Memory Bytes) - Cache miss rates (L1, Texture Cache) - Shader-specific metrics (VS/FS/Compute) @@ -42,16 +42,16 @@ Performance Metrics (6-246): - Invocation counts and statistics Data Source: - Exports parsed counter rows from .gpuprofiler_raw data when available. - Any encoder row without parsed metrics is emitted with SYNTHETIC FALLBACK - values. The command reports the row source counts on stderr, so stdout - remains valid CSV when exporting there. + This exporter writes encoder identity and leaves metric columns blank. Parsed + .gpuprofiler_raw counter rows are pipeline-scoped, not encoder-scoped, and + are withheld until a stable join exists. The command reports that source + state on stderr, so stdout remains valid CSV when exporting there. - As Metal replay support with MTLCounterSampleBuffer matures, replay-collected - rows can replace remaining fallback rows with hardware measurements. + A capture-backed encoder join or replay-collected measurements can populate + the metric columns in a future export. Output Format: - Standard CSV with quoted strings, matching Xcode's export format exactly. + Standard CSV with quoted strings and an Xcode-compatible column schema. Can be imported into spreadsheet tools or compared with Xcode's output. Examples: @@ -126,46 +126,26 @@ func runExportCounters(cmd *cobra.Command, args []string, opts *exportCountersOp // Print success message to stderr (not stdout which has CSV data) if opts.output != "" { - fmt.Fprintf(cmd.ErrOrStderr(), "✓ Exported counters to: %s\n", opts.output) + fmt.Fprintf(cmd.ErrOrStderr(), "Counter CSV written: %s\n", opts.output) } return nil } type exportCounterSourceSummary struct { - totalRows int - parsedCounterRows int - syntheticFallbackRows int - perfCountersPresent bool + totalRows int + metadataOnlyRows int + perfCountersPresent bool } func summarizeExportCounterSources(trace *gputrace.Trace) (exportCounterSourceSummary, error) { - encoders, err := trace.ParseComputeEncoders() - if err != nil { - return exportCounterSourceSummary{}, err - } + encoders := trace.ParseComputeEncoders() summary := exportCounterSourceSummary{ - totalRows: len(encoders), - syntheticFallbackRows: len(encoders), - perfCountersPresent: trace.HasPerfCounters(), + totalRows: len(encoders), + metadataOnlyRows: len(encoders), + perfCountersPresent: trace.HasPerfCounters(), } - - if !summary.perfCountersPresent { - return summary, nil - } - - metrics, err := counter.PopulateEncoderMetricsFromBinaryParsing(trace) - if err != nil || len(metrics) == 0 { - return summary, nil - } - - summary.parsedCounterRows = len(metrics) - if summary.parsedCounterRows > summary.totalRows { - summary.parsedCounterRows = summary.totalRows - } - summary.syntheticFallbackRows = summary.totalRows - summary.parsedCounterRows - return summary, nil } @@ -173,16 +153,10 @@ func formatExportCounterSourceNotice(summary exportCounterSourceSummary) string switch { case summary.totalRows == 0: return "counter export data source: no encoder rows exported\n" - case summary.syntheticFallbackRows == 0: - return fmt.Sprintf("counter export data source: parsed counter data (%s)\n", formatRows(summary.parsedCounterRows)) - case summary.parsedCounterRows == 0 && summary.perfCountersPresent: - return fmt.Sprintf("counter export data source: synthetic fallback (%s); performance counter files were present but no parsed row metrics were available\n", formatRows(summary.syntheticFallbackRows)) - case summary.parsedCounterRows == 0: - return fmt.Sprintf("counter export data source: synthetic fallback (%s); no parsed .gpuprofiler_raw counter data found\n", formatRows(summary.syntheticFallbackRows)) + case summary.perfCountersPresent: + return fmt.Sprintf("counter export data source: metadata only (%s); performance-counter rows are pipeline-scoped and lack an encoder join\n", formatRows(summary.metadataOnlyRows)) default: - return fmt.Sprintf("counter export data source: parsed counter data (%s), synthetic fallback (%s)\n", - formatRows(summary.parsedCounterRows), - formatRows(summary.syntheticFallbackRows)) + return fmt.Sprintf("counter export data source: metadata only (%s); no parsed .gpuprofiler_raw counter data found\n", formatRows(summary.metadataOnlyRows)) } } diff --git a/cmd/gputrace/cmd/export_counters_test.go b/cmd/gputrace/cmd/export_counters_test.go index 19f7f32a..0353f6d5 100644 --- a/cmd/gputrace/cmd/export_counters_test.go +++ b/cmd/gputrace/cmd/export_counters_test.go @@ -13,60 +13,29 @@ func TestFormatExportCounterSourceNotice(t *testing.T) { avoid []string }{ { - name: "all parsed", + name: "metadata only without perf counters", summary: exportCounterSourceSummary{ - totalRows: 2, - parsedCounterRows: 2, - syntheticFallbackRows: 0, - perfCountersPresent: true, + totalRows: 2, + metadataOnlyRows: 2, }, want: []string{ - "parsed counter data (2 rows)", - }, - avoid: []string{ - "synthetic fallback", - }, - }, - { - name: "all synthetic without perf counters", - summary: exportCounterSourceSummary{ - totalRows: 2, - parsedCounterRows: 0, - syntheticFallbackRows: 2, - perfCountersPresent: false, - }, - want: []string{ - "synthetic fallback (2 rows)", + "metadata only (2 rows)", "no parsed .gpuprofiler_raw counter data found", }, avoid: []string{ - "parsed counter data (", + "parsed counter data", }, }, { - name: "mixed parsed and synthetic", + name: "metadata only with perf counters", summary: exportCounterSourceSummary{ - totalRows: 3, - parsedCounterRows: 1, - syntheticFallbackRows: 2, - perfCountersPresent: true, + totalRows: 3, + metadataOnlyRows: 3, + perfCountersPresent: true, }, want: []string{ - "parsed counter data (1 row)", - "synthetic fallback (2 rows)", - }, - }, - { - name: "synthetic despite perf counters", - summary: exportCounterSourceSummary{ - totalRows: 1, - parsedCounterRows: 0, - syntheticFallbackRows: 1, - perfCountersPresent: true, - }, - want: []string{ - "synthetic fallback (1 row)", - "performance counter files were present but no parsed row metrics were available", + "metadata only (3 rows)", + "pipeline-scoped and lack an encoder join", }, }, } @@ -88,15 +57,20 @@ func TestFormatExportCounterSourceNotice(t *testing.T) { } } -func TestExportCountersHelpDistinguishesSyntheticFallback(t *testing.T) { +func TestExportCountersHelpWithholdsUnjoinedCounters(t *testing.T) { help := exportCountersCmd.Long for _, want := range []string{ - "parsed counter rows", - "SYNTHETIC FALLBACK", - "reports the row source counts on stderr", + "pipeline-scoped, not encoder-scoped", + "withheld until a stable join exists", + "state on stderr", } { if !strings.Contains(help, want) { t.Fatalf("export-counters help does not contain %q", want) } } + for _, misleading := range []string{"matching the exact format", "matching Xcode's export format exactly"} { + if strings.Contains(help, misleading) { + t.Fatalf("export-counters help makes exactness claim %q", misleading) + } + } } diff --git a/cmd/gputrace/cmd/fences.go b/cmd/gputrace/cmd/fences.go index aa7e68b1..973ad97b 100644 --- a/cmd/gputrace/cmd/fences.go +++ b/cmd/gputrace/cmd/fences.go @@ -23,8 +23,9 @@ func newFencesCommand(opts *fencesOptions) *cobra.Command { Use: "fences ", Short: "List fence operations in the trace", Hidden: true, - Long: `Scans the trace for fence operations (e.g. waitForFence, updateFence) encoded as ICB executions.`, - Args: cobra.ExactArgs(1), + Long: `Scans Culul records for heuristic fence-operation candidates. +The result is not a decoded Metal waitForFence/updateFence API sequence.`, + Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { return runFences(cmd, args, opts) }, @@ -118,7 +119,8 @@ func runFences(cmd *cobra.Command, args []string, opts *fencesOptions) error { return err } - fmt.Fprintln(w, "Scanning for fence operations...") + fmt.Fprintln(w, "Heuristic fence candidates from Culul records") + fmt.Fprintln(w, "Note: operation types marked ? are inferred, not decoded Metal API calls.") fmt.Fprintf(w, "%-10s %-18s %-30s %s\n", "Offset", "Address", "Label", "Details") fmt.Fprintln(w, "--------------------------------------------------------------------------------") for _, f := range fences { diff --git a/cmd/gputrace/cmd/gate.go b/cmd/gputrace/cmd/gate.go new file mode 100644 index 00000000..3826847c --- /dev/null +++ b/cmd/gputrace/cmd/gate.go @@ -0,0 +1,196 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "io" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/gate" +) + +type gateOptions struct { + tokens int + exactTokens bool + invariant string + slack int + stationarityThreshold float64 + blockSize int + json bool + ranges []string + compare bool + timingSidecar string +} + +var gateCmd = newGateCommand(&gateOptions{ + slack: 2, + stationarityThreshold: 0.15, + blockSize: 16, +}) + +func newGateCommand(opts *gateOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "gate [flags] ...", + Short: "Gate a GPU capture against workload invariants and stationarity", + Long: `Gate a GPU capture before trusting anything in it. + +Evaluates three independent checks: + + 1. completeness - scores against a workload invariant (an op the model runs + once per token), not just the tracer's self-reported drop + counter which reads zero when records are stranded. + 2. stationarity - the per-token trajectory must be flat across blocks; a mid-run + excursion leaves per-kernel medians intact while inflating + summed time (requires timing data from streamData or profile-replay). + 3. staging - reports observed data movement (CUDA HtoD transfers or Metal + streamData blit calls) with explicit distinction between + recorded zero and absent data. + +Exit status: + 0: all evaluated gates passed + 1: capture failed a gate (e.g. completeness loss or stationarity excursion) + 2: capture was not evaluable (e.g. 0 invariant matches or missing required flags)`, + Args: cobra.MinimumNArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runGate(cmd, args, opts) + }, + } + + cmd.Flags().IntVarP(&opts.tokens, "tokens", "t", 0, "tokens the run was asked to generate") + cmd.Flags().BoolVar(&opts.exactTokens, "exact-tokens", false, "score want = tokens without prefill +1") + cmd.Flags().StringVarP(&opts.invariant, "invariant", "k", "", "symbol substring for an op that fires once per token (default: arg_reduce on CUDA; required on Metal)") + cmd.Flags().IntVar(&opts.slack, "slack", 2, "tokens allowed missing: flush-window residual, not a loss budget") + cmd.Flags().Float64Var(&opts.stationarityThreshold, "stationarity-threshold", 0.15, "max allowed relative excursion for stationarity (0.15 = 15%)") + cmd.Flags().IntVar(&opts.blockSize, "block-size", 0, "gaps per block for trajectory stationarity (0 = auto; blocks over token gaps, or command-buffer gaps with --timing)") + cmd.Flags().BoolVar(&opts.json, "json", false, "output machine-readable JSON verdict") + cmd.Flags().StringVar(&opts.timingSidecar, "timing", "", "GT_TIMING_OUT sidecar for live command-buffer stationarity (outranks replay-derived streamData timing)") + cmd.Flags().StringSliceVar(&opts.ranges, "ranges", nil, "half-open token ranges lo:hi, one per bundle innermost first; checks invariant counts grow with range width") + cmd.Flags().BoolVar(&opts.compare, "compare", false, "compare staging/residency observations between exactly two bundles") + + return cmd +} + +func init() { + rootCmd.AddCommand(gateCmd) +} + +func runGate(cmd *cobra.Command, args []string, opts *gateOptions) error { + gateOpts := gate.Options{ + Tokens: opts.tokens, + ExactTokens: opts.exactTokens, + InvariantSymbol: opts.invariant, + Slack: opts.slack, + StationarityThreshold: opts.stationarityThreshold, + BlockSize: opts.blockSize, + Ranges: opts.ranges, + TimingSidecar: opts.timingSidecar, + } + + var results []*gate.Result + hasFail := false + hasNotEvaluable := false + + out := cmd.OutOrStdout() + + if len(opts.ranges) > 0 { + rr, err := gate.EvaluateRanges(args, opts.ranges, gateOpts) + if err != nil { + return fmt.Errorf("ranges: %w", err) + } + if opts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + if err := enc.Encode(rr); err != nil { + return fmt.Errorf("write json: %w", err) + } + } else { + fmt.Fprintln(out, rr.Summary) + } + switch rr.Verdict { + case gate.VerdictFail: + return errGateFailed + case gate.VerdictNotEvaluable: + return errGateNotEvaluable + } + return nil + } + + if opts.compare { + if len(args) != 2 { + return fmt.Errorf("--compare requires exactly 2 bundles, got %d", len(args)) + } + cmpRes, err := gate.Compare(args[0], args[1], gateOpts) + if err != nil { + return fmt.Errorf("compare: %w", err) + } + if opts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(cmpRes) + } + fmt.Fprintln(out, cmpRes.Summary) + return nil + } + + for _, bundle := range args { + res, err := gate.Evaluate(bundle, gateOpts) + if err != nil { + return fmt.Errorf("gate %s: %w", bundle, err) + } + results = append(results, res) + + switch res.Verdict { + case gate.VerdictFail: + hasFail = true + case gate.VerdictNotEvaluable: + hasNotEvaluable = true + } + + if !opts.json { + printGateResult(out, res) + } + } + + if opts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + if len(results) == 1 { + if err := enc.Encode(results[0]); err != nil { + return fmt.Errorf("write json: %w", err) + } + } else { + if err := enc.Encode(results); err != nil { + return fmt.Errorf("write json: %w", err) + } + } + } + + if hasFail { + return errGateFailed + } + if hasNotEvaluable { + return errGateNotEvaluable + } + return nil +} + +func printGateResult(w io.Writer, r *gate.Result) { + fmt.Fprintln(w, r.Summary) +} + +type gateFailedError struct{} + +func (gateFailedError) Error() string { return "capture failed gate" } +func (gateFailedError) exitCode() int { return 1 } +func (gateFailedError) alreadyReported() {} + +type gateNotEvaluableError struct{} + +func (gateNotEvaluableError) Error() string { return "capture not evaluable" } +func (gateNotEvaluableError) exitCode() int { return 2 } +func (gateNotEvaluableError) alreadyReported() {} + +var ( + errGateFailed = gateFailedError{} + errGateNotEvaluable = gateNotEvaluableError{} +) diff --git a/cmd/gputrace/cmd/gate_test.go b/cmd/gputrace/cmd/gate_test.go new file mode 100644 index 00000000..8db12710 --- /dev/null +++ b/cmd/gputrace/cmd/gate_test.go @@ -0,0 +1,145 @@ +package cmd + +import ( + "bytes" + "os" + "path/filepath" + "strings" + "testing" +) + +func TestGateCommandHelp(t *testing.T) { + var buf bytes.Buffer + cmd := newGateCommand(&gateOptions{}) + cmd.SetOut(&buf) + cmd.SetErr(&buf) + cmd.SetArgs([]string{"--help"}) + + if err := cmd.Execute(); err != nil { + t.Fatalf("gate --help failed: %v", err) + } + + out := buf.String() + for _, expected := range []string{ + "Gate a GPU capture before trusting anything in it.", + "--tokens", + "--invariant", + "--slack", + "--stationarity-threshold", + "--block-size", + "--json", + "Exit status:", + } { + if !strings.Contains(out, expected) { + t.Errorf("gate --help output missing %q, got:\n%s", expected, out) + } + } +} + +func TestGateCommandCUDAEvents(t *testing.T) { + // Create a synthetic events.jsonl + dir := t.TempDir() + eventsPath := filepath.Join(dir, "events.jsonl") + + var content strings.Builder + // 33 arg_reduce kernels, 10ms apart + for i := 0; i < 33; i++ { + start := uint64(1000000000 + i*10000000) + end := start + 1000000 + content.WriteString(`{"kind":"kernel","raw_symbol":"_Z18arg_reduce_generali","start_ns":` + + strings.TrimSpace(string(intToBytes(start))) + `,"end_ns":` + + strings.TrimSpace(string(intToBytes(end))) + `}` + "\n") + } + // 5 HtoD memcpys + for i := 0; i < 5; i++ { + content.WriteString(`{"kind":"memcpy","src_kind":"host","dst_kind":"device","bytes":1048576,"start_ns":500000,"end_ns":600000}` + "\n") + } + + if err := os.WriteFile(eventsPath, []byte(content.String()), 0o644); err != nil { + t.Fatalf("write events: %v", err) + } + + var buf bytes.Buffer + cmd := newGateCommand(&gateOptions{ + tokens: 32, + invariant: "arg_reduce", + slack: 2, + stationarityThreshold: 0.15, + blockSize: 8, + }) + cmd.SetOut(&buf) + cmd.SetErr(&buf) + cmd.SetArgs([]string{"-t", "32", "-k", "arg_reduce", "--block-size", "8", dir}) + + if err := cmd.Execute(); err != nil { + t.Fatalf("gate execution failed: %v, output: %s", err, buf.String()) + } + + out := buf.String() + if !strings.Contains(out, "completeness ok") { + t.Errorf("expected completeness ok in output, got:\n%s", out) + } + if !strings.Contains(out, "5 HtoD transfers") { + t.Errorf("expected staging info in output, got:\n%s", out) + } + if !strings.Contains(out, "stationarity ok") { + t.Errorf("expected stationarity ok in output, got:\n%s", out) + } +} + +func intToBytes(n uint64) []byte { + return []byte(strings.TrimSpace(string(fmtInt(n)))) +} + +func fmtInt(n uint64) string { + var buf [32]byte + i := len(buf) + for n >= 10 { + i-- + buf[i] = byte('0' + n%10) + n /= 10 + } + i-- + buf[i] = byte('0' + n) + return string(buf[i:]) +} + +func TestGateCommandCompare(t *testing.T) { + dirA := t.TempDir() + eventsPathA := filepath.Join(dirA, "events.jsonl") + var contentA strings.Builder + contentA.WriteString(`{"kind":"kernel","raw_symbol":"_Z18arg_reduce_generali","start_ns":1000000,"end_ns":2000000}` + "\n") + contentA.WriteString(`{"kind":"memcpy","src_kind":"host","dst_kind":"device","bytes":1048576,"start_ns":500000,"end_ns":600000}` + "\n") + if err := os.WriteFile(eventsPathA, []byte(contentA.String()), 0o644); err != nil { + t.Fatalf("write events A: %v", err) + } + + dirB := t.TempDir() + eventsPathB := filepath.Join(dirB, "events.jsonl") + var contentB strings.Builder + contentB.WriteString(`{"kind":"kernel","raw_symbol":"_Z18arg_reduce_generali","start_ns":1000000,"end_ns":2000000}` + "\n") + for i := 0; i < 5; i++ { + contentB.WriteString(`{"kind":"memcpy","src_kind":"host","dst_kind":"device","bytes":1048576,"start_ns":500000,"end_ns":600000}` + "\n") + } + if err := os.WriteFile(eventsPathB, []byte(contentB.String()), 0o644); err != nil { + t.Fatalf("write events B: %v", err) + } + + var buf bytes.Buffer + cmd := newGateCommand(&gateOptions{compare: true}) + cmd.SetOut(&buf) + cmd.SetErr(&buf) + cmd.SetArgs([]string{"--compare", dirA, dirB}) + + if err := cmd.Execute(); err != nil { + t.Fatalf("gate --compare failed: %v", err) + } + + out := buf.String() + if !strings.Contains(out, "Residency / Staging Comparison") { + t.Errorf("expected residency comparison header, got:\n%s", out) + } + if !strings.Contains(out, "+4 transfers") { + t.Errorf("expected +4 transfers delta, got:\n%s", out) + } +} diff --git a/cmd/gputrace/cmd/graph.go b/cmd/gputrace/cmd/graph.go index be169bc9..e2877b74 100644 --- a/cmd/gputrace/cmd/graph.go +++ b/cmd/gputrace/cmd/graph.go @@ -26,8 +26,8 @@ Supported formats: - mermaid: Mermaid diagram format Graph types: - - hierarchy: Command buffer → encoder → shader hierarchy (default) - - flow: Execution flow (temporal order) + - hierarchy: Command buffer → CS-label hierarchy (default); ownership is heuristic + - flow: Observed CS-label order (not verified dispatch flow) - resources: Resource usage and buffer allocations Examples: @@ -71,6 +71,9 @@ func runGraph(cmd *cobra.Command, args []string, opts *graphOptions) error { if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } // Create graph generator based on format var generator graph.Generator diff --git a/cmd/gputrace/cmd/help_test.go b/cmd/gputrace/cmd/help_test.go index 588c3731..1416fa2d 100644 --- a/cmd/gputrace/cmd/help_test.go +++ b/cmd/gputrace/cmd/help_test.go @@ -247,6 +247,11 @@ func TestXcodeProfileExportUsageShowsOptionalOutputPath(t *testing.T) { if err := exportCmd.Args(exportCmd, []string{"out.gputrace"}); err != nil { t.Fatalf("xcode-profile export should accept one arg: %v", err) } + for _, name := range []string{"recover-untitled", "check-recovery", "finalize-workload", "source", "xcode-pid", "xcode-app"} { + if exportCmd.Flags().Lookup(name) == nil { + t.Fatalf("xcode-profile export missing --%s", name) + } + } } func TestTimingProfilerHelpMarksLegacyApproximateFallbacks(t *testing.T) { @@ -299,6 +304,44 @@ func TestTimelineFormatHelpIncludesPerfetto(t *testing.T) { } } +func TestTimelineOwnsPerfettoExportWorkflow(t *testing.T) { + for _, name := range []string{ + "format", + "sidecar", + "open", + "serve", + "max-output-bytes", + "sql-out", + "kernel", + "kernel-occurrence", + "time-start", + "time-end", + } { + if timelineCmd.Flags().Lookup(name) == nil { + t.Errorf("timeline command missing --%s", name) + } + } + + for _, name := range []string{"perfetto", "viewer", "manifest"} { + if visibleSubcommand(rootCmd, name) != nil { + t.Errorf("%s is a separate command; want timeline to own the workflow", name) + } + } + + for _, want := range []string{ + "evidence manifest", + "environment projection", + "resource policy", + "loss receipt", + "part of the timeline export", + "separate commands", + } { + if !strings.Contains(timelineCmd.Long, want) { + t.Errorf("timeline help does not contain %q", want) + } + } +} + func TestGraphHelpMatchesDefaultType(t *testing.T) { flag := graphCmd.Flags().Lookup("type") if flag == nil { diff --git a/cmd/gputrace/cmd/host_receipt.go b/cmd/gputrace/cmd/host_receipt.go new file mode 100644 index 00000000..82d43110 --- /dev/null +++ b/cmd/gputrace/cmd/host_receipt.go @@ -0,0 +1,57 @@ +package cmd + +import ( + "fmt" + "os" + + "github.com/spf13/cobra" + + "github.com/tmc/gputrace/internal/hostevents" +) + +type hostReceiptOptions struct { + output string +} + +var hostReceiptCmd = newHostReceiptCommand(&hostReceiptOptions{}) + +func newHostReceiptCommand(opts *hostReceiptOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "host-receipt ", + Short: "Bind measured host intervals to live GPU timing", + Args: cobra.ExactArgs(2), + RunE: func(cmd *cobra.Command, args []string) error { + return runHostReceipt(cmd, args, opts) + }, + } + cmd.Flags().StringVarP(&opts.output, "output", "o", opts.output, "Write canonical receipt to path instead of stdout") + return cmd +} + +func init() { + rootCmd.AddCommand(hostReceiptCmd) +} + +func runHostReceipt(cmd *cobra.Command, args []string, opts *hostReceiptOptions) error { + receipt, withheld, err := hostevents.Receipt(args[0], args[1]) + for _, event := range withheld { + fmt.Fprintf(cmd.ErrOrStderr(), "withholding %q: outside the sidecar's sampled clock range\n", event.ID) + } + if err != nil { + return fmt.Errorf("build host receipt: %w", err) + } + data, err := receipt.Canonical() + if err != nil { + return fmt.Errorf("encode host receipt: %w", err) + } + if opts.output == "" { + if _, err := cmd.OutOrStdout().Write(data); err != nil { + return fmt.Errorf("write host receipt: %w", err) + } + return nil + } + if err := os.WriteFile(opts.output, data, 0o600); err != nil { + return fmt.Errorf("write host receipt: %w", err) + } + return nil +} diff --git a/cmd/gputrace/cmd/host_receipt_test.go b/cmd/gputrace/cmd/host_receipt_test.go new file mode 100644 index 00000000..2246dd4e --- /dev/null +++ b/cmd/gputrace/cmd/host_receipt_test.go @@ -0,0 +1,43 @@ +package cmd + +import ( + "bytes" + "os" + "path/filepath" + "testing" + + "github.com/tmc/gputrace/internal/hostcorrelation" +) + +func TestHostReceiptCommand(t *testing.T) { + dir := t.TempDir() + host := filepath.Join(dir, "host.jsonl") + timing := filepath.Join(dir, "timing.jsonl") + if err := os.WriteFile(host, []byte(`{"clock_domain":"cpu_uptime_ns","duration_ns":20,"id":"event-1","kind":"interval","name":"Generation","run_id":"run-1","schema":"gputrace.host-event/v1","timestamp_ns":120} +`), 0o600); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(timing, []byte(`{"kind":"clock_sample","cpu_ticks":100,"gpu_ticks":200,"run_id":"run-1"} +{"kind":"clock_sample","cpu_ticks":200,"gpu_ticks":300,"run_id":"run-1"} +{"kind":"clock_sample","cpu_ticks":300,"gpu_ticks":400,"run_id":"run-1"} +{"kind":"command_buffer","id":1,"capture_label":"gputrace.live.cb.1","final_label":"decode","gpu_start_seconds":0.00000025,"gpu_end_seconds":0.00000035,"kernel_start_seconds":0.00000024,"kernel_end_seconds":0.00000034,"status":4,"run_id":"run-1"} +{"kind":"artifact","run_id":"run-1","trace_digest":"sha256:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"} +`), 0o600); err != nil { + t.Fatal(err) + } + + var output bytes.Buffer + cmd := newHostReceiptCommand(&hostReceiptOptions{}) + cmd.SetOut(&output) + cmd.SetArgs([]string{host, timing}) + if err := cmd.Execute(); err != nil { + t.Fatal(err) + } + receiptPath := filepath.Join(dir, "receipt.json") + if err := os.WriteFile(receiptPath, output.Bytes(), 0o600); err != nil { + t.Fatal(err) + } + if _, err := hostcorrelation.Read(receiptPath); err != nil { + t.Fatal(err) + } +} diff --git a/cmd/gputrace/cmd/human_limit.go b/cmd/gputrace/cmd/human_limit.go new file mode 100644 index 00000000..ace9fbda --- /dev/null +++ b/cmd/gputrace/cmd/human_limit.go @@ -0,0 +1,45 @@ +package cmd + +import ( + "fmt" + "io" + "strings" +) + +const defaultHumanLimit = 20 + +func resolveHumanLimit(limit int, all bool) (int, error) { + if all { + return -1, nil + } + if limit <= 0 { + return 0, fmt.Errorf("--limit must be greater than zero") + } + return limit, nil +} + +func limitedCount(total, limit int) int { + if limit < 0 || total <= limit { + return total + } + return limit +} + +func writeLimitedLines(w io.Writer, text string, limit int, noun string) error { + if text == "" { + return nil + } + lines := strings.Split(strings.TrimSuffix(text, "\n"), "\n") + n := limitedCount(len(lines), limit) + for _, line := range lines[:n] { + if _, err := fmt.Fprintln(w, line); err != nil { + return err + } + } + if n < len(lines) { + if _, err := fmt.Fprintf(w, "... %d more %s omitted (use --all)\n", len(lines)-n, noun); err != nil { + return err + } + } + return nil +} diff --git a/cmd/gputrace/cmd/human_limit_test.go b/cmd/gputrace/cmd/human_limit_test.go new file mode 100644 index 00000000..100e0d1b --- /dev/null +++ b/cmd/gputrace/cmd/human_limit_test.go @@ -0,0 +1,37 @@ +package cmd + +import ( + "bytes" + "strings" + "testing" +) + +func TestWriteLimitedLines(t *testing.T) { + var out bytes.Buffer + if err := writeLimitedLines(&out, "one\ntwo\nthree\n", 2, "rows"); err != nil { + t.Fatalf("writeLimitedLines: %v", err) + } + if got, want := out.String(), "one\ntwo\n... 1 more rows omitted (use --all)\n"; got != want { + t.Fatalf("output = %q, want %q", got, want) + } +} + +func TestFormatBuffersTableLimitAndWriter(t *testing.T) { + buffers := []BufferInfo{ + {ID: "1", Filename: "MTLBuffer-1-0", Size: 1}, + {ID: "2", Filename: "MTLBuffer-2-0", Size: 2}, + {ID: "3", Filename: "MTLBuffer-3-0", Size: 3}, + } + var out bytes.Buffer + if err := formatBuffersTable(&out, buffers, nil, 2); err != nil { + t.Fatalf("formatBuffersTable: %v", err) + } + got := out.String() + if !strings.Contains(got, "3 buffers, 6 B") || + !strings.Contains(got, "MTLBuffer-1-0") || + !strings.Contains(got, "MTLBuffer-2-0") || + strings.Contains(got, "MTLBuffer-3-0") || + !strings.Contains(got, "... 1 more buffers omitted (use --all)") { + t.Fatalf("unexpected limited table:\n%s", got) + } +} diff --git a/cmd/gputrace/cmd/insights.go b/cmd/gputrace/cmd/insights.go index a0c3496d..df693565 100644 --- a/cmd/gputrace/cmd/insights.go +++ b/cmd/gputrace/cmd/insights.go @@ -24,7 +24,7 @@ func newInsightsCommand(opts *insightsOptions) *cobra.Command { } cmd := &cobra.Command{ Use: "insights ", - Short: "Generate actionable performance insights from GPU trace", + Short: "Report supported GPU performance hypotheses", Long: `Analyze GPU trace and generate actionable performance insights. This command performs comprehensive analysis to identify: @@ -32,7 +32,7 @@ This command performs comprehensive analysis to identify: * Memory-bound vs compute-bound classification * Dominant shader detection - OPTIMIZATIONS: Opportunities to improve performance - * Low occupancy issues + * Small threadgroup sizes * Excessive dispatch overhead * Work distribution imbalance - ANTI-PATTERNS: Common performance pitfalls @@ -83,6 +83,9 @@ func runInsights(cmd *cobra.Command, args []string, opts *insightsOptions) error if err != nil { return fmt.Errorf("failed to open trace: %w", err) } + if err := trace.RequireCaptureRecords(); err != nil { + return err + } defer trace.Close() // Generate insights @@ -121,20 +124,38 @@ func writeInsightsText(w io.Writer, report *gputrace.InsightsReport) error { // Print summary at the end if len(report.Insights) == 0 { - out.WriteString("✓ No performance issues detected!\n") + if report.TimingApprox { + out.WriteString("No supported issues identified from the available approximate data.\n") + out.WriteString("Measured timing is required for bottleneck ranking.\n") + } else { + out.WriteString("No supported performance issues identified.\n") + } } else { out.WriteString("\n=== Summary ===\n") + attributionLimited := insightsAttributionLimited(report) if report.CriticalCount > 0 { - fmt.Fprintf(&out, "⚠️ %d CRITICAL issues require immediate attention\n", report.CriticalCount) + if attributionLimited { + fmt.Fprintf(&out, "%d CRITICAL attribution hypotheses require corroboration\n", report.CriticalCount) + } else { + fmt.Fprintf(&out, "%d CRITICAL issues require immediate attention\n", report.CriticalCount) + } } if report.HighCount > 0 { - fmt.Fprintf(&out, "⚠️ %d HIGH priority optimizations recommended\n", report.HighCount) + if attributionLimited { + fmt.Fprintf(&out, "%d HIGH-priority attribution hypotheses\n", report.HighCount) + } else { + fmt.Fprintf(&out, "%d HIGH-priority optimizations recommended\n", report.HighCount) + } } if report.MediumCount > 0 { - fmt.Fprintf(&out, "ℹ️ %d MEDIUM priority suggestions available\n", report.MediumCount) + if attributionLimited { + fmt.Fprintf(&out, "%d MEDIUM-priority triage signals\n", report.MediumCount) + } else { + fmt.Fprintf(&out, "%d MEDIUM-priority suggestions available\n", report.MediumCount) + } } if report.LowCount > 0 { - fmt.Fprintf(&out, "ℹ️ %d LOW priority observations noted\n", report.LowCount) + fmt.Fprintf(&out, "%d LOW-priority observations noted\n", report.LowCount) } } @@ -144,6 +165,15 @@ func writeInsightsText(w io.Writer, report *gputrace.InsightsReport) error { return nil } +func insightsAttributionLimited(report *gputrace.InsightsReport) bool { + for _, source := range report.TimingSources { + if strings.Contains(source, "gpuCommandInfoData") { + return true + } + } + return false +} + // filterInsightsBySeverity filters insights by minimum severity level. func filterInsightsBySeverity(report *gputrace.InsightsReport, minLevel string) *gputrace.InsightsReport { // Map severity levels to numeric values @@ -164,6 +194,8 @@ func filterInsightsBySeverity(report *gputrace.InsightsReport, minLevel string) filtered := &gputrace.InsightsReport{ Insights: make([]*gputrace.PerformanceInsight, 0), TotalGPUTimeMs: report.TotalGPUTimeMs, + TimingSources: report.TimingSources, + TimingApprox: report.TimingApprox, TopBottlenecks: report.TopBottlenecks, } diff --git a/cmd/gputrace/cmd/insights_test.go b/cmd/gputrace/cmd/insights_test.go index fe23c328..d47fa5d7 100644 --- a/cmd/gputrace/cmd/insights_test.go +++ b/cmd/gputrace/cmd/insights_test.go @@ -63,6 +63,9 @@ func TestWriteInsightsJSON(t *testing.T) { if got.HighCount != 1 || got.TotalGPUTimeMs != 12.5 { t.Fatalf("json report = %+v", got) } + if !got.TimingApprox || len(got.TimingSources) != 1 || got.TimingSources[0] != "synthetic" { + t.Fatalf("json timing provenance = %q, approximate %t", got.TimingSources, got.TimingApprox) + } if len(got.Insights) != 1 || got.Insights[0].ShaderName != "kernel_a" { t.Fatalf("json insights = %+v", got.Insights) } @@ -78,7 +81,7 @@ func TestWriteInsightsTextPreservesSummaryBytes(t *testing.T) { want := gputrace.FormatInsightsReport(report) + "\n=== Summary ===\n" + - "⚠️ 1 HIGH priority optimizations recommended\n" + "1 HIGH-priority optimizations recommended\n" if got := out.String(); got != want { t.Fatalf("text output mismatch\ngot:\n%s\nwant:\n%s", got, want) } @@ -95,12 +98,49 @@ func TestWriteInsightsTextNoInsights(t *testing.T) { t.Fatalf("writeInsightsText: %v", err) } - want := gputrace.FormatInsightsReport(report) + "✓ No performance issues detected!\n" + want := gputrace.FormatInsightsReport(report) + "No supported performance issues identified.\n" if got := out.String(); got != want { t.Fatalf("text output mismatch\ngot:\n%s\nwant:\n%s", got, want) } } +func TestWriteInsightsTextDoesNotPromoteLimitedAttributionToOptimization(t *testing.T) { + report := testInsightsReport() + report.TimingApprox = false + report.TimingSources = []string{"streamData gpuCommandInfoData dispatch durations"} + + var out bytes.Buffer + if err := writeInsightsText(&out, report); err != nil { + t.Fatalf("writeInsightsText: %v", err) + } + got := out.String() + if !strings.Contains(got, "1 HIGH-priority attribution hypotheses") { + t.Fatalf("limited-attribution summary missing:\n%s", got) + } + if strings.Contains(got, "optimizations recommended") { + t.Fatalf("limited attribution promoted to optimization:\n%s", got) + } +} + +func TestWriteInsightsTextApproximateDoesNotGiveCleanBill(t *testing.T) { + report := &gputrace.InsightsReport{ + Insights: []*gputrace.PerformanceInsight{}, + TimingApprox: true, + } + + var out bytes.Buffer + if err := writeInsightsText(&out, report); err != nil { + t.Fatalf("writeInsightsText: %v", err) + } + got := out.String() + if strings.Contains(got, "✓") || strings.Contains(got, "No performance issues") { + t.Fatalf("approximate report gives a clean bill:\n%s", got) + } + if !strings.Contains(got, "Measured timing is required") { + t.Fatalf("approximate report omits measurement limitation:\n%s", got) + } +} + func testInsightsReport() *gputrace.InsightsReport { return &gputrace.InsightsReport{ Insights: []*gputrace.PerformanceInsight{ @@ -118,6 +158,8 @@ func testInsightsReport() *gputrace.InsightsReport { }, HighCount: 1, TotalGPUTimeMs: 12.5, + TimingSources: []string{"synthetic"}, + TimingApprox: true, TopBottlenecks: []string{"kernel_a"}, } } diff --git a/cmd/gputrace/cmd/kernels.go b/cmd/gputrace/cmd/kernels.go index 62289660..ab424f8d 100644 --- a/cmd/gputrace/cmd/kernels.go +++ b/cmd/gputrace/cmd/kernels.go @@ -19,6 +19,8 @@ type kernelsOptions struct { verbose bool stats bool json bool + limit int + all bool } func newKernelsCommand(opts *kernelsOptions) *cobra.Command { @@ -54,6 +56,8 @@ Examples: cmd.Flags().BoolVarP(&opts.verbose, "verbose", "v", opts.verbose, "Show verbose output with additional details") cmd.Flags().BoolVar(&opts.stats, "stats", opts.stats, "Show detailed statistics (debug groups, encoder labels)") cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output in JSON format") + cmd.Flags().IntVar(&opts.limit, "limit", 50, "Maximum rows to show") + cmd.Flags().BoolVar(&opts.all, "all", false, "Show every row") return cmd } @@ -73,53 +77,55 @@ func runKernels(cmd *cobra.Command, args []string, opts *kernelsOptions) error { return fmt.Errorf("failed to open trace: %w", err) } - // Analyze kernels to get stats - stats, err := trace.AnalyzeKernels() - if err != nil { - return fmt.Errorf("analyze kernels: %w", err) + // Analyze kernels from capture records. A profiler-only bundle has none; + // the streamData dispatch list below replaces them wholesale, so only + // insist on capture records when that list is unavailable. + stats := make(map[string]*gputrace.KernelStat) + if !trace.ProfilerOnly { + stats, err = trace.AnalyzeKernels() + if err != nil { + return fmt.Errorf("analyze kernels: %w", err) + } } - // Try to get timing stats var timingStats map[string]*gputrace.TimingStat - // We check for perf counters availability - if trace.HasPerfCounters() { - // Use extracted timing data - // Note: We need to bridge internal/timing to something usable here without import cycles in core packages. - // Since cmd can import anything, we can implement extraction here or use a helper. - // But gputrace package re-exports ExtractTimingData. - - timings, err := gputrace.ExtractTimingData(trace) - if err == nil { - timingStats = make(map[string]*gputrace.TimingStat) - for _, t := range timings { - name := t.Label - // Normalize name to match kernel stats if possible - // Encoder timing labels usually match encoder labels - - // Clean up name if it's an "Encoder_X_kernel" style - if strings.Contains(name, "_") { - parts := strings.SplitN(name, "_", 3) - if len(parts) >= 3 && parts[0] == "Encoder" { - name = parts[2] - } - } - - if _, exists := timingStats[name]; !exists { - timingStats[name] = &gputrace.TimingStat{ - MinTime: 1e9, - } - } - - s := timingStats[name] - s.TotalTime += t.DurationMs - if t.DurationMs < s.MinTime { - s.MinTime = t.DurationMs - } - if t.DurationMs > s.MaxTime { - s.MaxTime = t.DurationMs + source := "capture records" + if _, profilerStats, err := loadProfilerStats(tracePath); err == nil && len(profilerStats.Dispatches) > 0 { + source = "profiler streamData dispatches" + stats = make(map[string]*gputrace.KernelStat) + timingStats = make(map[string]*gputrace.TimingStat) + captureLabels := trace.ParseComputeEncoders() + for _, dispatch := range profilerStats.Dispatches { + name := dispatch.DisplayName() + k := stats[name] + if k == nil { + k = &gputrace.KernelStat{ + Name: name, + DebugGroups: make(map[string]int), + EncoderLabels: make(map[string]int), } + stats[name] = k + } + k.DispatchCount++ + label := "" + if dispatch.EncoderIndex >= 0 && dispatch.EncoderIndex < len(profilerStats.EncoderTimings) { + label = profilerStats.EncoderTimings[dispatch.EncoderIndex].Label + } + if label == "" && dispatch.EncoderIndex >= 0 && dispatch.EncoderIndex < len(captureLabels) { + label = captureLabels[dispatch.EncoderIndex].Label + } + if label != "" && label != name { + k.EncoderLabels[label]++ + } + s := timingStats[name] + if s == nil { + s = &gputrace.TimingStat{} + timingStats[name] = s } + s.TotalTime += float64(dispatch.DurationUs) / 1000 } + } else if err := trace.RequireCaptureRecords(); err != nil { + return err } // Filter and sort @@ -146,25 +152,64 @@ func runKernels(cmd *cobra.Command, args []string, opts *kernelsOptions) error { } out := cmd.OutOrStdout() + hasTiming := len(timingStats) > 0 + totalDispatches := 0 + attributedDispatches := 0 + for _, k := range stats { + totalDispatches += k.DispatchCount + if k.Name != "unknown" && k.Name != "" { + attributedDispatches += k.DispatchCount + } + } - // Count unique kernels - uniqueKernels := len(kernels) + rows := splitKernelRows(kernels) + namedKernels, unknownBucket := rows.Executed, rows.Unknown + uniqueKernels := len(namedKernels) - // Output header + // Output header. Count only the kernels that ran: a created-but-unrun + // pipeline and a library UUID are both in the inventory and neither is + // evidence of a dispatch. + rowSingular, rowPlural := "dispatched kernel", "dispatched kernels" + if hasTiming { + rowSingular, rowPlural = "timed function", "timed functions" + } if opts.filter != "" { - fmt.Fprintf(out, "%d %s matching %q:\n", uniqueKernels, Pluralize(uniqueKernels, "kernel", "kernels"), opts.filter) + fmt.Fprintf(out, "%d %s matching %q:\n", uniqueKernels, Pluralize(uniqueKernels, rowSingular, rowPlural), opts.filter) } else { - fmt.Fprintf(out, "%d %s:\n", uniqueKernels, Pluralize(uniqueKernels, "kernel", "kernels")) + fmt.Fprintf(out, "%d %s:\n", uniqueKernels, Pluralize(uniqueKernels, rowSingular, rowPlural)) + } + fmt.Fprintf(out, "Source: %s\n", source) + fmt.Fprintf(out, "Dispatch attribution: %d/%d", attributedDispatches, totalDispatches) + if attributedDispatches < totalDispatches { + fmt.Fprint(out, " (unattributed dispatches are reported as unknown)") + } + fmt.Fprintln(out) + if unattributedInventory(attributedDispatches, totalDispatches) { + fmt.Fprint(out, unattributedInventoryNote) + } + if n := countArchiveNamedKernels(namedKernels); n > 0 { + fmt.Fprintf(out, "%d %s named only by shader archive id (archive:...): the capture records\n"+ + "which archive the function came from, not its name. Run 'gputrace profile-replay'\n"+ + "on this trace to get the names.\n", + n, Pluralize(n, "kernel", "kernels")) + } + if hasTiming { + fmt.Fprintln(out, "Timing: cumulative dispatch offsets; spans may include boundary or gap time") } fmt.Fprintln(out) - if uniqueKernels == 0 { + if uniqueKernels == 0 && unknownBucket == nil { + writeInactiveKernelRows(out, rows) return nil } // Determine column widths maxNameLen := 30 - for _, k := range kernels { + shown := namedKernels + if !opts.all && opts.limit >= 0 && len(shown) > opts.limit { + shown = shown[:opts.limit] + } + for _, k := range shown { if len(k.Name) > maxNameLen { maxNameLen = len(k.Name) } @@ -177,9 +222,6 @@ func runKernels(cmd *cobra.Command, args []string, opts *kernelsOptions) error { // Print table header nameFmt := fmt.Sprintf("%%-%ds", maxNameLen) - // Adjust columns if we have timing - hasTiming := len(timingStats) > 0 - fmt.Fprintf(out, nameFmt+" %-18s %-10s", "Name", "Pipeline State", "Dispatches") if hasTiming { fmt.Fprintf(out, " %-10s %-10s", "Total Time", "Avg Time") @@ -199,14 +241,19 @@ func runKernels(cmd *cobra.Command, args []string, opts *kernelsOptions) error { fmt.Fprintln(out, TableSeparator(sepWidth)) // Print rows - for _, k := range kernels { + for _, k := range shown { name := k.Name displayName := name if len(displayName) > maxNameLen { displayName = displayName[:maxNameLen-3] + "..." } - fmt.Fprintf(out, nameFmt+" 0x%-16x %-10d", displayName, k.PipelineAddr, k.DispatchCount) + pipeline := "—" + if k.PipelineAddr != 0 { + pipeline = fmt.Sprintf("0x%x", k.PipelineAddr) + } + fmt.Fprintf(out, nameFmt+" %-18s %-10s", displayName, pipeline, + formatDispatchCount(k.DispatchCount, attributedDispatches, totalDispatches)) if hasTiming { if tStat, ok := timingStats[name]; ok { @@ -268,18 +315,83 @@ func runKernels(cmd *cobra.Command, args []string, opts *kernelsOptions) error { } fmt.Fprintln(out) } + if len(shown) < len(namedKernels) { + fmt.Fprintf(out, "... %d more; use --all to show every row\n", len(namedKernels)-len(shown)) + } - // Print summary of unknown pipelines if any - if k, ok := stats["unknown"]; ok && k.DispatchCount > 0 { - fmt.Fprintf(out, "\nUnknown Pipelines: %d dispatches (encoder: %v)\n", k.DispatchCount, k.EncoderLabels) + writeInactiveKernelRows(out, rows) + + if unknownBucket != nil { + writeUnknownKernelBucket(out, unknownBucket) } return nil } +// An inventory row can mean three different things, and presenting them +// identically invites the reader to count labels as if they were kernels that +// ran. A pipeline is created before it is used, and MLX creates several it +// then fuses away, so the inventory lists kernels that dispatch zero times. +// The library records share the label field with function records, so it also +// lists UUIDs. Reading four zero-dispatch rows as "a whole kernel family" +// happened, and the header saying "named inventory kernel labels" over all +// three kinds is what made it a reasonable reading. +type kernelRows struct { + Executed []*gputrace.KernelStat // dispatched at least once + Unrun []*gputrace.KernelStat // pipeline created, never dispatched + Libraries []*gputrace.KernelStat // library UUIDs, never function names + Unknown *gputrace.KernelStat // the synthetic unattributed bucket +} + +func splitKernelRows(kernels []*gputrace.KernelStat) kernelRows { + var rows kernelRows + for _, k := range kernels { + switch { + case k.Name == "unknown": + rows.Unknown = k + case gputrace.IsLibraryUUID(k.Name): + rows.Libraries = append(rows.Libraries, k) + case k.DispatchCount == 0: + rows.Unrun = append(rows.Unrun, k) + default: + rows.Executed = append(rows.Executed, k) + } + } + return rows +} + +// writeInactiveKernelRows lists the rows that are not evidence a kernel ran, +// under headers that say what they are. +func writeInactiveKernelRows(w io.Writer, rows kernelRows) { + if len(rows.Unrun) > 0 { + fmt.Fprintf(w, "\n%d %s created but never dispatched (a pipeline is created before use, "+ + "and fused-away kernels are created and then not used):\n", + len(rows.Unrun), Pluralize(len(rows.Unrun), "pipeline", "pipelines")) + for _, k := range rows.Unrun { + fmt.Fprintf(w, " %s\n", k.Name) + } + } + if len(rows.Libraries) > 0 { + fmt.Fprintf(w, "\n%d library %s (not kernel names):\n", + len(rows.Libraries), Pluralize(len(rows.Libraries), "UUID", "UUIDs")) + for _, k := range rows.Libraries { + fmt.Fprintf(w, " %s\n", k.Name) + } + } +} + +func writeUnknownKernelBucket(w io.Writer, unknown *gputrace.KernelStat) { + fmt.Fprintf(w, "\nSynthetic unattributed bucket: %d dispatches", unknown.DispatchCount) + if len(unknown.EncoderLabels) > 0 { + fmt.Fprintf(w, " (encoder labels: %v)", unknown.EncoderLabels) + } + fmt.Fprintln(w) +} + func writeKernelsJSON(w io.Writer, kernels []*gputrace.KernelStat, timingStats map[string]*gputrace.TimingStat) error { type kernelJSON struct { Name string `json:"name"` + RowKind string `json:"row_kind"` PipelineAddr string `json:"pipeline_addr"` DispatchCount int `json:"dispatch_count"` DebugGroups map[string]int `json:"debug_groups,omitempty"` @@ -290,8 +402,13 @@ func writeKernelsJSON(w io.Writer, kernels []*gputrace.KernelStat, timingStats m out := make([]kernelJSON, len(kernels)) for i, k := range kernels { + rowKind := "named_inventory" + if k.Name == "unknown" { + rowKind = "synthetic_unattributed_bucket" + } kj := kernelJSON{ Name: k.Name, + RowKind: rowKind, PipelineAddr: fmt.Sprintf("0x%x", k.PipelineAddr), DispatchCount: k.DispatchCount, DebugGroups: k.DebugGroups, diff --git a/cmd/gputrace/cmd/kernels_attribution.go b/cmd/gputrace/cmd/kernels_attribution.go new file mode 100644 index 00000000..1f57028c --- /dev/null +++ b/cmd/gputrace/cmd/kernels_attribution.go @@ -0,0 +1,56 @@ +package cmd + +import ( + "strconv" + + "github.com/tmc/gputrace" +) + +// A kernel row's dispatch count is a join: the inventory supplies the function +// names, the command stream supplies the dispatches, and a Ctt record keyed by +// function address is what connects them. When a capture carries no resolvable +// Ctt mapping the join yields nothing, and every named kernel is left at zero +// beside a trace that plainly ran hundreds of dispatches. +// +// Zero is then the wrong thing to print. It is a number in a count column, so +// it reads as "this kernel never ran" -- a measurement -- when it means "this +// trace cannot say". Print the absence instead, and say so once above the +// table. + +// unattributedDispatchMark stands in for a dispatch count that the trace +// cannot supply. +const unattributedDispatchMark = "—" + +const unattributedInventoryNote = "No dispatch could be joined to a named function, so per-kernel counts are\n" + + "unavailable for this trace and are shown as —. The dispatches happened; this\n" + + "trace cannot say which kernel ran them.\n" + +// unattributedInventory reports whether the trace ran dispatches but resolved +// none of them to a name, which makes every per-kernel count meaningless +// rather than zero. +func unattributedInventory(attributed, total int) bool { + return total > 0 && attributed == 0 +} + +// countArchiveNamedKernels counts the rows whose name is a shader archive id +// rather than a function name. Their dispatch counts are real -- each archive +// id is one pipeline -- but the name is not one a reader can look up, so the +// table says where the name went. +func countArchiveNamedKernels(kernels []*gputrace.KernelStat) int { + n := 0 + for _, k := range kernels { + if gputrace.IsArchiveFunctionName(k.Name) { + n++ + } + } + return n +} + +// formatDispatchCount renders a row's dispatch count, substituting the +// unattributed mark when no count in the table can be trusted as a count. +func formatDispatchCount(count, attributed, total int) string { + if count == 0 && unattributedInventory(attributed, total) { + return unattributedDispatchMark + } + return strconv.Itoa(count) +} diff --git a/cmd/gputrace/cmd/kernels_attribution_test.go b/cmd/gputrace/cmd/kernels_attribution_test.go new file mode 100644 index 00000000..e6bd52b6 --- /dev/null +++ b/cmd/gputrace/cmd/kernels_attribution_test.go @@ -0,0 +1,55 @@ +package cmd + +import "testing" + +func TestUnattributedInventory(t *testing.T) { + tests := []struct { + name string + attributed int + total int + want bool + }{ + {"nothing joined", 0, 490, true}, + {"all joined", 490, 490, false}, + {"partly joined", 12, 490, false}, + {"no dispatches at all", 0, 0, false}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if got := unattributedInventory(tt.attributed, tt.total); got != tt.want { + t.Errorf("unattributedInventory(%d, %d) = %v, want %v", + tt.attributed, tt.total, got, tt.want) + } + }) + } +} + +func TestFormatDispatchCount(t *testing.T) { + tests := []struct { + name string + count int + attributed int + total int + want string + }{ + // The case this exists for: the trace ran 490 dispatches and named + // none of them, so every inventory row sits at zero. + {"unjoinable trace", 0, 0, 490, "—"}, + // A genuine zero, in a trace whose other rows did join. Here the + // kernel really was not dispatched, and zero is the measurement. + {"real zero", 0, 12, 490, "0"}, + {"counted row", 56, 490, 490, "56"}, + // A nonzero count is never suppressed, even in an unjoinable trace: + // something did attribute it, so the number stands. + {"nonzero in unjoinable trace", 3, 0, 490, "3"}, + {"trace with no dispatches", 0, 0, 0, "0"}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + if got := formatDispatchCount(tt.count, tt.attributed, tt.total); got != tt.want { + t.Errorf("formatDispatchCount(%d, %d, %d) = %q, want %q", + tt.count, tt.attributed, tt.total, got, tt.want) + } + }) + } +} diff --git a/cmd/gputrace/cmd/kernels_test.go b/cmd/gputrace/cmd/kernels_test.go index a742a7a2..16268734 100644 --- a/cmd/gputrace/cmd/kernels_test.go +++ b/cmd/gputrace/cmd/kernels_test.go @@ -4,6 +4,7 @@ import ( "bytes" "os" "path/filepath" + "strings" "testing" "github.com/spf13/cobra" @@ -46,6 +47,10 @@ func TestWriteKernelsJSON(t *testing.T) { "encoder": 4, }, }, + { + Name: "unknown", + DispatchCount: 7, + }, } timingStats := map[string]*gputrace.TimingStat{ "copy_kernel": { @@ -61,6 +66,7 @@ func TestWriteKernelsJSON(t *testing.T) { const want = `[ { "name": "copy_kernel", + "row_kind": "named_inventory", "pipeline_addr": "0x1234", "dispatch_count": 4, "debug_groups": { @@ -71,6 +77,12 @@ func TestWriteKernelsJSON(t *testing.T) { }, "total_time_ms": 10, "avg_time_ms": 2.5 + }, + { + "name": "unknown", + "row_kind": "synthetic_unattributed_bucket", + "pipeline_addr": "0x0", + "dispatch_count": 7 } ] ` @@ -79,6 +91,82 @@ func TestWriteKernelsJSON(t *testing.T) { } } +func TestSplitKernelRows(t *testing.T) { + kernels := []*gputrace.KernelStat{ + {Name: "kernel_b", DispatchCount: 56}, + {Name: "unknown", DispatchCount: 435}, + {Name: "kernel_a", DispatchCount: 1}, + } + + rows := splitKernelRows(kernels) + if len(rows.Executed) != 2 || rows.Executed[0].Name != "kernel_b" || rows.Executed[1].Name != "kernel_a" { + t.Fatalf("executed rows = %#v, want kernel_b and kernel_a", rows.Executed) + } + if rows.Unknown == nil || rows.Unknown.DispatchCount != 435 { + t.Fatalf("unknown bucket = %#v, want 435 dispatches", rows.Unknown) + } +} + +// The four zero-dispatch rows below are the ones read as "a whole kernel +// family that exists solely to support static capacity". They were created and +// fused away, and none of them ran. +func TestSplitKernelRowsSeparatesCreatedButUnrun(t *testing.T) { + rows := splitKernelRows([]*gputrace.KernelStat{ + {Name: "g1_Selectbfloat16"}, + {Name: "gemm_bfloat16", DispatchCount: 96}, + {Name: "sv_GreaterEqualint32"}, + {Name: "E0A5F8B1-4C2D-4E7A-9F13-2B6C5D8E0A11"}, + }) + if len(rows.Executed) != 1 || rows.Executed[0].Name != "gemm_bfloat16" { + t.Errorf("executed = %#v, want only the kernel that dispatched", rows.Executed) + } + if len(rows.Unrun) != 2 { + t.Errorf("unrun = %#v, want both zero-dispatch pipelines", rows.Unrun) + } + if len(rows.Libraries) != 1 { + t.Errorf("libraries = %#v, want the UUID row", rows.Libraries) + } +} + +func TestWriteInactiveKernelRowsSaysWhatEachIs(t *testing.T) { + var out bytes.Buffer + writeInactiveKernelRows(&out, splitKernelRows([]*gputrace.KernelStat{ + {Name: "g1_Selectbfloat16"}, + {Name: "E0A5F8B1-4C2D-4E7A-9F13-2B6C5D8E0A11"}, + })) + got := out.String() + if !strings.Contains(got, "created but never dispatched") { + t.Errorf("output does not say unrun pipelines did not run:\n%s", got) + } + if !strings.Contains(got, "library UUID (not kernel names)") { + t.Errorf("output does not distinguish library UUIDs:\n%s", got) + } +} + +// A row that ran must never be filed as inactive: that would remove evidence +// rather than label it. +func TestWriteInactiveKernelRowsOmitsExecuted(t *testing.T) { + var out bytes.Buffer + writeInactiveKernelRows(&out, splitKernelRows([]*gputrace.KernelStat{ + {Name: "gemm_bfloat16", DispatchCount: 96}, + })) + if out.Len() != 0 { + t.Errorf("executed kernel listed as inactive:\n%s", out.String()) + } +} + +func TestWriteUnknownKernelBucket(t *testing.T) { + var out bytes.Buffer + writeUnknownKernelBucket(&out, &gputrace.KernelStat{ + Name: "unknown", + DispatchCount: 435, + }) + const want = "\nSynthetic unattributed bucket: 435 dispatches\n" + if got := out.String(); got != want { + t.Fatalf("output = %q, want %q", got, want) + } +} + func writeKernelsMinimalTraceBundle(t *testing.T) string { t.Helper() diff --git a/cmd/gputrace/cmd/lanepacker_test.go b/cmd/gputrace/cmd/lanepacker_test.go new file mode 100644 index 00000000..3f13f165 --- /dev/null +++ b/cmd/gputrace/cmd/lanepacker_test.go @@ -0,0 +1,94 @@ +package cmd + +import "testing" + +// TestLanePackerSequentialStaysOneLane is the defect this replaced. Dispatches +// run back to back, and the old assignment (3 + index%4) spread them over four +// lanes, drawing concurrency that never happened. +func TestLanePackerSequentialStaysOneLane(t *testing.T) { + p := newLanePacker(3, 4) + for i := uint64(0); i < 10; i++ { + if got := p.assign(i*100, 100); got != 3 { + t.Fatalf("slice %d went to lane %d, want 3: back-to-back slices must share a lane", i, got) + } + } +} + +func TestLanePackerOverlapSeparates(t *testing.T) { + p := newLanePacker(3, 4) + a := p.assign(0, 100) + b := p.assign(50, 100) + c := p.assign(60, 100) + if a == b || b == c || a == c { + t.Fatalf("overlapping slices shared a lane: %d %d %d", a, b, c) + } +} + +// A slice that starts exactly when the previous one ends does not overlap it. +func TestLanePackerTouchingIsNotOverlap(t *testing.T) { + p := newLanePacker(3, 4) + if a, b := p.assign(0, 100), p.assign(100, 100); a != b { + t.Fatalf("touching slices split across lanes %d and %d", a, b) + } +} + +// Beyond the lane count slices stack rather than land on threads the legend +// does not name. +func TestLanePackerStaysInRange(t *testing.T) { + p := newLanePacker(3, 4) + for i := 0; i < 20; i++ { + got := p.assign(0, 1000) + if got < 3 || got > 6 { + t.Fatalf("lane %d outside the named range 3..6", got) + } + } +} + +func TestLanePackerZeroLanes(t *testing.T) { + if got := newLanePacker(3, 0).assign(0, 10); got != 3 { + t.Fatalf("empty packer returned %d, want base 3", got) + } +} + +// TestCommandBuffersKeepIdleGaps guards the worst defect this file has carried. +// +// Command buffers were emitted with a running accumulator that packed each one +// against the end of the last, erasing every idle gap. On the 21-encoder +// capture that compressed 2979 ms of wall time into 8.3 ms and rendered a GPU +// that is 0.28% busy as 99.9% busy -- while the event args asserted +// "real_timing": true. A reader would have concluded the GPU was saturated. +func TestCommandBuffersKeepIdleGaps(t *testing.T) { + // Two command buffers 100 ms apart, each 1 ms long. + const numer, denom = 125, 3 // the 24 MHz timebase these captures use + abs := uint64(1_000_000) + starts := []uint64{abs, abs + 2_400_000} // 2.4M ticks = 100 ms at 24 MHz + + var offsets []uint64 + for _, st := range starts { + offsets = append(offsets, (st-abs)*numer/denom) + } + gapNs := offsets[1] - offsets[0] + if gapNs < 99_000_000 || gapNs > 101_000_000 { + t.Fatalf("offset gap = %d ns, want ~100 ms: real spacing must survive into the timestamp", gapNs) + } + if offsets[0] != 0 { + t.Fatalf("first command buffer offset = %d, want 0", offsets[0]) + } +} + +// TestCounterTrackSignal pins the rule that an all-zero counter track is an +// undecoded counter, not a measured zero. Publishing it drew nine flat lines on +// the 21-encoder capture that read as "no bandwidth used" rather than "unknown". +func TestCounterTrackSignal(t *testing.T) { + zero := CounterTrack{Name: "ALU Utilization", Samples: []CounterSample{{Value: 0}, {Value: 0}}} + if counterTrackHasSignal(zero) { + t.Error("all-zero track reported signal") + } + if counterTrackHasSignal(CounterTrack{Name: "empty"}) { + t.Error("track with no samples reported signal") + } + live := CounterTrack{Name: "Bandwidth", Samples: []CounterSample{{Value: 0}, {Value: 12.5}}} + if !counterTrackHasSignal(live) { + t.Error("track with a nonzero sample reported no signal") + } +} diff --git a/cmd/gputrace/cmd/mtlb.go b/cmd/gputrace/cmd/mtlb.go index e7497f04..5b199f2f 100644 --- a/cmd/gputrace/cmd/mtlb.go +++ b/cmd/gputrace/cmd/mtlb.go @@ -2,6 +2,7 @@ package cmd import ( "fmt" + "io" "os" "github.com/spf13/cobra" @@ -9,12 +10,14 @@ import ( "github.com/tmc/gputrace/internal/trace" ) -type mtlbOptions struct{} +type mtlbOptions struct { + all bool +} var mtlbCmd = newMTLBCommand(new(mtlbOptions)) func newMTLBCommand(opts *mtlbOptions) *cobra.Command { - return &cobra.Command{ + cmd := &cobra.Command{ Use: "mtlb ", Short: "Inspect and analyze Metal Library Binary (MTLB) files", Long: `Inspect and analyze Metal Library Binary (MTLB) files. @@ -29,14 +32,17 @@ Displays header info, function table, and extraction stats.`, return runMTLB(cmd, args, opts) }, } + cmd.Flags().BoolVar(&opts.all, "all", opts.all, "Show every function in direct inspection output") + return cmd } func init() { rootCmd.AddCommand(mtlbCmd) } -func runMTLB(cmd *cobra.Command, args []string, _ *mtlbOptions) error { +func runMTLB(cmd *cobra.Command, args []string, opts *mtlbOptions) error { path := args[0] + w := cmd.OutOrStdout() // Check if it's a trace bundle info, err := os.Stat(path) @@ -47,13 +53,15 @@ func runMTLB(cmd *cobra.Command, args []string, _ *mtlbOptions) error { return err } - fmt.Printf("Trace: %s\n", path) - fmt.Printf("Found %d MTLB Libraries associated with parsing:\n\n", len(t.MTLBLibraries)) + fmt.Fprintf(w, "Trace: %s\n", path) + fmt.Fprintf(w, "Found %d parsed MTLB libraries:\n\n", len(t.MTLBLibraries)) for i, lib := range t.MTLBLibraries { - fmt.Printf("=== Library %d ===\n", i+1) - printMTLBDetails(lib) - fmt.Println() + fmt.Fprintf(w, "=== Library %d ===\n", i+1) + if err := printMTLBDetails(w, lib.Header, lib.ListFunctions, opts.all); err != nil { + return err + } + fmt.Fprintln(w) } return nil } @@ -74,26 +82,32 @@ func runMTLB(cmd *cobra.Command, args []string, _ *mtlbOptions) error { return err } - fmt.Printf("File: %s\n", path) - printMTLBDetails(mtlbFile) - return nil + fmt.Fprintf(w, "File: %s\n", path) + return printMTLBDetails(w, mtlbFile.Header, mtlbFile.ListFunctions, opts.all) } -func printMTLBDetails(lib *metallib.File) { - fmt.Printf("Header:\n") - fmt.Printf(" Version: %d\n", lib.Header.Version) - fmt.Printf(" Total Size: %d bytes\n", lib.Header.TotalSize) - fmt.Printf(" Function Table: 0x%x\n", lib.Header.FunctionTable) - fmt.Printf(" String Table: 0x%x\n", lib.Header.StringTable) +func printMTLBDetails(w io.Writer, header metallib.Header, listFunctions func() ([]string, error), all bool) error { + fmt.Fprintln(w, "Header:") + fmt.Fprintf(w, " Version: %d\n", header.Version) + fmt.Fprintf(w, " Total Size: %d bytes\n", header.TotalSize) + fmt.Fprintf(w, " Function Table: 0x%x\n", header.FunctionTable) + fmt.Fprintf(w, " String Table: 0x%x\n", header.StringTable) - funcs, err := lib.ListFunctions() + funcs, err := listFunctions() if err != nil { - fmt.Printf("Error listing functions: %v\n", err) - return + return fmt.Errorf("list MTLB functions: %w", err) } - fmt.Printf("\nFunctions (%d found):\n", len(funcs)) - for i, f := range funcs { - fmt.Printf(" %d. %s\n", i, f) + fmt.Fprintf(w, "\nFunctions (%d found):\n", len(funcs)) + shown := limitedCount(len(funcs), defaultHumanLimit) + if all { + shown = len(funcs) + } + for i, f := range funcs[:shown] { + fmt.Fprintf(w, " %d. %s\n", i+1, f) } + if shown < len(funcs) { + fmt.Fprintf(w, " ... %d more functions omitted (use --all)\n", len(funcs)-shown) + } + return nil } diff --git a/cmd/gputrace/cmd/mtlb_info.go b/cmd/gputrace/cmd/mtlb_info.go index e5b17edf..94d38b15 100644 --- a/cmd/gputrace/cmd/mtlb_info.go +++ b/cmd/gputrace/cmd/mtlb_info.go @@ -33,41 +33,42 @@ func runMTLBInfo(cmd *cobra.Command, args []string, _ *mtlbInfoOptions) error { } if len(files) == 0 { - fmt.Println("No MTLB files found in trace.") + fmt.Fprintln(cmd.OutOrStdout(), "No MTLB files found in trace.") return nil } + out := cmd.OutOrStdout() for _, f := range files { - fmt.Printf("\n=== Metal Library: %s ===\n", f.Name) + fmt.Fprintf(out, "\n=== Metal Library: %s ===\n", f.Name) data, err := os.ReadFile(f.Path) if err != nil { - fmt.Printf("Error reading file: %v\n", err) + fmt.Fprintf(out, "Error reading file: %v\n", err) continue } lib, err := metallib.Parse(data) if err != nil { - fmt.Printf("Error parsing MTLB: %v\n", err) + fmt.Fprintf(out, "Error parsing MTLB: %v\n", err) continue } - fmt.Printf("\nMagic: %s\n", string(lib.Header.Magic[:])) - fmt.Printf("Version: %d\n", lib.Header.Version) - fmt.Printf("Size: %s\n", fmtutil.FormatBytes(int64(lib.Header.TotalSize), 1)) + fmt.Fprintf(out, "\nMagic: %s\n", string(lib.Header.Magic[:])) + fmt.Fprintf(out, "Version: %d\n", lib.Header.Version) + fmt.Fprintf(out, "Size: %s\n", fmtutil.FormatBytes(int64(lib.Header.TotalSize), 1)) // Assuming flags/reserved might have meaning later // fmt.Printf("Flags: 0x%x\n", lib.Header.Flags) funcs, _ := lib.ListFunctions() - fmt.Println("\nSections:") - fmt.Printf(" Functions: %d\n", len(funcs)) - fmt.Printf(" Bytecode: %s (offset 0x%x)\n", fmtutil.FormatBytes(int64(len(data))-int64(lib.Header.BytecodeOffset), 1), lib.Header.BytecodeOffset) + fmt.Fprintln(out, "\nSections:") + fmt.Fprintf(out, " Functions: %d\n", len(funcs)) + fmt.Fprintf(out, " Bytecode: %s (offset 0x%x)\n", fmtutil.FormatBytes(int64(len(data))-int64(lib.Header.BytecodeOffset), 1), lib.Header.BytecodeOffset) // String table size estimation stringTableSize := int64(lib.Header.BytecodeOffset - lib.Header.StringTable) if stringTableSize > 0 { - fmt.Printf(" Strings: %s\n", fmtutil.FormatBytes(stringTableSize, 1)) + fmt.Fprintf(out, " Strings: %s\n", fmtutil.FormatBytes(stringTableSize, 1)) } } return nil diff --git a/cmd/gputrace/cmd/mtlb_list.go b/cmd/gputrace/cmd/mtlb_list.go index f1dba03c..6da60179 100644 --- a/cmd/gputrace/cmd/mtlb_list.go +++ b/cmd/gputrace/cmd/mtlb_list.go @@ -33,10 +33,11 @@ func runMTLBList(cmd *cobra.Command, args []string, _ *mtlbListOptions) error { return err } - fmt.Println("\n=== Metal Library Files ===") - fmt.Println("") + out := cmd.OutOrStdout() + fmt.Fprintln(out, "\n=== Metal Library Files ===") + fmt.Fprintln(out) - w := tabwriter.NewWriter(os.Stdout, 0, 0, 4, ' ', 0) + w := tabwriter.NewWriter(out, 0, 0, 4, ' ', 0) fmt.Fprintln(w, "File\tSize\tFunctions") fmt.Fprintln(w, "----\t----\t---------") @@ -61,8 +62,8 @@ func runMTLBList(cmd *cobra.Command, args []string, _ *mtlbListOptions) error { } w.Flush() - fmt.Println("") - fmt.Printf("Total: %d libraries, %d functions\n", totalFiles, totalFuncs) + fmt.Fprintln(out) + fmt.Fprintf(out, "Total: %d libraries, %d functions\n", totalFiles, totalFuncs) return nil } diff --git a/cmd/gputrace/cmd/mtlb_stats.go b/cmd/gputrace/cmd/mtlb_stats.go index b72fb3d4..3108cbd7 100644 --- a/cmd/gputrace/cmd/mtlb_stats.go +++ b/cmd/gputrace/cmd/mtlb_stats.go @@ -40,7 +40,8 @@ func runMTLBStats(cmd *cobra.Command, args []string, _ *mtlbStatsOptions) error return fmt.Errorf("open trace: %w", err) } - fmt.Println("\n=== Metal Library Statistics ===") + out := cmd.OutOrStdout() + fmt.Fprintln(out, "\n=== Metal Library Statistics ===") // Collect all functions var allFuncs []string @@ -104,8 +105,8 @@ func runMTLBStats(cmd *cobra.Command, args []string, _ *mtlbStatsOptions) error } } - fmt.Println("\nFunction Categories:") - w := tabwriter.NewWriter(os.Stdout, 0, 0, 4, ' ', 0) + fmt.Fprintln(out, "\nFunction-name Categories (heuristic):") + w := tabwriter.NewWriter(out, 0, 0, 4, ' ', 0) // Sort categories for consistent output (except Other last) var cats []string @@ -122,7 +123,7 @@ func runMTLBStats(cmd *cobra.Command, args []string, _ *mtlbStatsOptions) error } w.Flush() - fmt.Println("\nData Type Coverage:") + fmt.Fprintln(out, "\nFunction-name Data Type Matches:") var types []string for t := range dataTypes { types = append(types, t) @@ -169,14 +170,17 @@ func runMTLBStats(cmd *cobra.Command, args []string, _ *mtlbStatsOptions) error totalFuncs := len(allFuncs) unusedCount := totalFuncs - usedCount usedPct := 0.0 + unusedPct := 0.0 if totalFuncs > 0 { usedPct = float64(usedCount) / float64(totalFuncs) * 100 + unusedPct = float64(unusedCount) / float64(totalFuncs) * 100 } - fmt.Println("\nUsage Analysis:") - fmt.Fprintf(w, " Functions used:\t%d (%.1f%%)\n", usedCount, usedPct) - fmt.Fprintf(w, " Functions unused:\t%d (%.1f%%)\n", unusedCount, 100-usedPct) + fmt.Fprintln(out, "\nObserved Usage Attribution:") + fmt.Fprintf(w, " Functions matched to decoded pipeline records:\t%d (%.1f%%)\n", usedCount, usedPct) + fmt.Fprintf(w, " Functions not matched:\t%d (%.1f%%)\n", unusedCount, unusedPct) w.Flush() + fmt.Fprintln(out, " Note: not matched does not mean unused; pipeline/function decoding is incomplete.") return nil } diff --git a/cmd/gputrace/cmd/mtlb_test.go b/cmd/gputrace/cmd/mtlb_test.go new file mode 100644 index 00000000..478256d4 --- /dev/null +++ b/cmd/gputrace/cmd/mtlb_test.go @@ -0,0 +1,20 @@ +package cmd + +import ( + "bytes" + "errors" + "testing" + + "github.com/tmc/gputrace/internal/metallib" +) + +func TestPrintMTLBDetailsReturnsListError(t *testing.T) { + want := errors.New("invalid function table") + var out bytes.Buffer + err := printMTLBDetails(&out, metallib.Header{}, func() ([]string, error) { + return nil, want + }, false) + if !errors.Is(err, want) { + t.Fatalf("printMTLBDetails error = %v, want %v", err, want) + } +} diff --git a/cmd/gputrace/cmd/ncu.go b/cmd/gputrace/cmd/ncu.go new file mode 100644 index 00000000..4b73e7a5 --- /dev/null +++ b/cmd/gputrace/cmd/ncu.go @@ -0,0 +1,243 @@ +package cmd + +import ( + "bytes" + "encoding/json" + "fmt" + "io" + "os" + "path/filepath" + "strings" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/gpuevent" + "github.com/tmc/gputrace/internal/ncu" +) + +type ncuOptions struct { + top int + launchCount int + metrics string + sudo bool + dryRun bool + json bool + merge bool +} + +var ncuCmd = newNCUCommand(&ncuOptions{top: 3, launchCount: 20, merge: true}) + +func newNCUCommand(opts *ncuOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "ncu [-- command [args...]]", + Short: "Escalate a capture's hottest kernels to Nsight Compute counters", + Long: `Re-run a captured workload under Nsight Compute, profiling only the +kernels the capture says matter. + +ncu replays each kernel many times with different counters armed and +serializes execution, so profiling a whole run is prohibitive. This ranks +the capture's kernels by total GPU time, profiles the top few, and merges +the counter results back into the bundle as ncu.json. + +The workload comes from the bundle's meta.json, so the escalation +profiles the run the timeline described. Pass a command after -- to +override it. + +GPU performance counters are restricted to administrators on many +drivers; use --sudo when 'gputrace doctor' reports them refused. + +Examples: + gputrace ncu run.gpucapture + gputrace ncu run.gpucapture --top 5 --launch-count 10 + gputrace ncu run.gpucapture --dry-run + gputrace ncu run.gpucapture --sudo`, + Args: cobra.MinimumNArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runNCU(cmd, args, opts) + }, + } + cmd.Flags().IntVar(&opts.top, "top", opts.top, "Kernels to profile, ranked by total GPU time") + cmd.Flags().IntVar(&opts.launchCount, "launch-count", opts.launchCount, "Launches replayed per kernel (ncu serializes; keep this small)") + cmd.Flags().StringVar(&opts.metrics, "metrics", "", "Comma-separated ncu metrics (default: a set that exists on current parts)") + cmd.Flags().BoolVar(&opts.sudo, "sudo", opts.sudo, "Run ncu through sudo, for drivers restricting counters to admins") + cmd.Flags().BoolVar(&opts.dryRun, "dry-run", opts.dryRun, "Print the ncu command that would run and exit") + cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output the merged counters as JSON") + cmd.Flags().BoolVar(&opts.merge, "merge", opts.merge, "Write the counters into the bundle as ncu.json") + return cmd +} + +func runNCU(cmd *cobra.Command, args []string, opts *ncuOptions) error { + capturePath := args[0] + override := args[1:] + + rep, err := loadCaptureReport(capturePath) + if err != nil { + return err + } + if len(rep.Kernels) == 0 { + return fmt.Errorf("ncu: %s has no kernels to escalate", capturePath) + } + top := opts.top + if top <= 0 || top > len(rep.Kernels) { + top = len(rep.Kernels) + } + names := make([]string, 0, top) + for _, k := range rep.Kernels[:top] { + names = append(names, k.Name) + } + + workload := override + if len(workload) == 0 { + workload, err = captureWorkload(capturePath) + if err != nil { + return err + } + } + + out := cmd.OutOrStdout() + fmt.Fprintf(out, "Escalating %d of %d kernels by GPU time:\n", top, len(rep.Kernels)) + for _, k := range rep.Kernels[:top] { + fmt.Fprintf(out, " %5.1f%% %5dx mean %-9s %s\n", k.SharePct, k.Count, dur(k.MeanNS), shortKernel(k.Name)) + } + + // One ncu run per kernel. ncu's --launch-count is a budget over every + // launch that matches the name filter, not a budget per name, so a + // single run asking for several kernels spends the whole budget on + // whichever one launches first and silently returns nothing for the + // rest. That crowds out exactly the kernel worth escalating: the hot + // one is rarely the first to launch. Observed on an MLX decode, where + // --top 2 --launch-count 3 profiled three rms_norm_small launches and + // none of gemv_single, which was 86.6% of GPU time [V]. + result := &ncu.Result{Schema: ncu.SchemaV1} + var runErr error + var profiled int + for _, name := range names { + ncuOpts := ncu.Options{ + Kernels: []string{name}, + LaunchCount: opts.launchCount, + Sudo: opts.sudo, + } + if opts.metrics != "" { + ncuOpts.Metrics = strings.Split(opts.metrics, ",") + } + command, err := ncu.Command(ncuOpts, workload) + if err != nil { + return err + } + fmt.Fprintf(out, "\n%s\n\n", strings.Join(command.Args, " ")) + if opts.dryRun { + continue + } + // ncu writes its CSV table and its diagnostics to the same + // streams, so both are captured and the workload's own output is + // forwarded. + var buf bytes.Buffer + command.Stdout = io.MultiWriter(&buf, cmd.ErrOrStderr()) + command.Stderr = io.MultiWriter(&buf, cmd.ErrOrStderr()) + command.Stdin = cmd.InOrStdin() + if err := command.Run(); err != nil && runErr == nil { + runErr = err + } + if ncu.PermissionDenied(buf.String()) { + return fmt.Errorf("ncu: the driver refused GPU performance counters for this user (ERR_NVGPUCTRPERM)\n" + + " re-run with --sudo, or lift the restriction permanently:\n" + + " echo 'options nvidia NVreg_RestrictProfilingToAdminUsers=0' | sudo tee /etc/modprobe.d/nvidia-profiling.conf\n" + + " sudo update-initramfs -u && sudo reboot") + } + one, parseErr := ncu.ParseCSV(&buf) + if parseErr != nil { + if runErr != nil { + return fmt.Errorf("ncu: %v (%w)", runErr, parseErr) + } + return parseErr + } + if len(one.Kernels) > 0 { + profiled++ + } + result.Kernels = append(result.Kernels, one.Kernels...) + result.Command = append(result.Command, command.Args...) + } + if opts.dryRun { + return nil + } + // A kernel the capture ranked but ncu never measured is reported, not + // omitted. Silence here reads as "this kernel is fine". + if profiled < len(names) { + fmt.Fprintf(out, "\nncu returned no counters for %d of the %d kernels requested:\n", len(names)-profiled, len(names)) + for _, name := range names { + var got bool + for _, kc := range result.Kernels { + if ncu.SameKernel(kc.Kernel, name) { + got = true + break + } + } + if !got { + fmt.Fprintf(out, " %s\n", shortKernel(name)) + } + } + fmt.Fprintf(out, " the workload may not have launched them before the replay budget ran out; raise --launch-count\n") + } + + if opts.merge && cupticapture.IsBundle(capturePath) { + path := filepath.Join(capturePath, "ncu.json") + data, err := json.MarshalIndent(result, "", " ") + if err != nil { + return err + } + if err := os.WriteFile(path, append(data, '\n'), 0o644); err != nil { + return fmt.Errorf("ncu: merge into bundle: %w", err) + } + fmt.Fprintf(out, "\nmerged counters -> %s\n", path) + } + if opts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(result) + } + writeNCUResult(out, result, rep) + return runErr +} + +// captureWorkload recovers the command a bundle recorded. +func captureWorkload(path string) ([]string, error) { + if !cupticapture.IsBundle(path) { + return nil, fmt.Errorf("ncu: %s is not a capture bundle; pass the workload after -- ", path) + } + meta, err := cupticapture.ReadMeta(path) + if err != nil { + return nil, fmt.Errorf("ncu: read bundle metadata: %w", err) + } + if len(meta.Command) == 0 { + return nil, fmt.Errorf("ncu: %s records no command; pass the workload after -- ", path) + } + return meta.Command, nil +} + +// writeNCUResult prints measured counters beside the timeline's own +// numbers, so a reader can see the replay agreeing (or not) with what the +// capture measured without replaying anything. +func writeNCUResult(out io.Writer, result *ncu.Result, rep *gpuevent.Report) { + timeline := map[string]gpuevent.KernelStats{} + for _, k := range rep.Kernels { + timeline[k.Name] = k + } + fmt.Fprintf(out, "\nMeasured counters (%d kernel%s profiled):\n", len(result.Kernels), plural(len(result.Kernels))) + for _, kc := range result.Kernels { + fmt.Fprintf(out, "\n %s (%d launches replayed)\n", shortKernel(kc.Kernel), kc.Launches) + if k, ok := timeline[kc.Kernel]; ok { + fmt.Fprintf(out, " capture said: %dx, mean %s, theoretical occupancy %.0f%%\n", + k.Count, dur(k.MeanNS), k.TheoreticalOccupancyPct) + } + for _, m := range kc.Metrics { + fmt.Fprintf(out, " %-56s median %10.2f %s\n", m.Metric, m.Median, m.Unit) + } + if len(kc.Unsupported) > 0 { + fmt.Fprintf(out, " unsupported on this part: %s\n", strings.Join(kc.Unsupported, ", ")) + } + } +} + +func init() { + rootCmd.AddCommand(ncuCmd) +} diff --git a/cmd/gputrace/cmd/nvidia.go b/cmd/gputrace/cmd/nvidia.go new file mode 100644 index 00000000..fda2e831 --- /dev/null +++ b/cmd/gputrace/cmd/nvidia.go @@ -0,0 +1,91 @@ +package cmd + +import ( + "encoding/json" + "errors" + "fmt" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/nvidia" +) + +type nvidiaOptions struct { + json bool +} + +var nvidiaOpts = &nvidiaOptions{} + +var nvidiaCmd = newNVIDIACommand(nvidiaOpts) + +func newNVIDIACommand(opts *nvidiaOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "nvidia", + Short: "Report NVIDIA GPU status via NVML", + Long: `Report NVIDIA GPU status via NVML. + +Lists each visible NVIDIA GPU with its name, UUID, driver version, memory, +utilization, temperature, and power draw. Requires the NVIDIA driver's +libnvidia-ml shared library on Linux.`, + Args: cobra.NoArgs, + RunE: func(cmd *cobra.Command, args []string) error { + devices, err := nvidia.Devices() + if err != nil { + if errors.Is(err, nvidia.ErrNVMLUnavailable) { + return fmt.Errorf("nvidia: %w (is the NVIDIA driver installed?)", err) + } + return err + } + out := cmd.OutOrStdout() + if opts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(devices) + } + driver, _ := nvidia.DriverVersion() + fmt.Fprintf(out, "NVIDIA driver: %s (%d device%s)\n\n", driver, len(devices), plural(len(devices))) + for _, d := range devices { + fmt.Fprintf(out, "GPU %d: %s\n", d.Index, d.Name) + if d.UUID != "" { + fmt.Fprintf(out, " UUID: %s\n", d.UUID) + } + fmt.Fprintf(out, " Memory: %s used / %s total\n", humanBytes(d.MemoryUsed), humanBytes(d.MemoryTotal)) + fmt.Fprintf(out, " Util: gpu %d%%, memory %d%%\n", d.GPUUtilPct, d.MemUtilPct) + if d.TempC > 0 { + fmt.Fprintf(out, " Temp: %d C\n", d.TempC) + } + if d.PowerWatts > 0 { + fmt.Fprintf(out, " Power: %d mW\n", d.PowerWatts) + } + fmt.Fprintln(out) + } + return nil + }, + } + cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output in JSON format") + return cmd +} + +func plural(n int) string { + if n == 1 { + return "" + } + return "s" +} + +func humanBytes(b uint64) string { + const unit = 1024 + if b < unit { + return fmt.Sprintf("%d B", b) + } + div, exp := uint64(unit), 0 + for n := b / unit; n >= unit; n /= unit { + div *= unit + exp++ + } + return fmt.Sprintf("%.1f %ciB", float64(b)/float64(div), "KMGTPE"[exp]) +} + +func init() { + rootCmd.AddCommand(nvidiaCmd) + rootCmd.AddCommand(cuptiCmd) +} diff --git a/cmd/gputrace/cmd/optimize.go b/cmd/gputrace/cmd/optimize.go new file mode 100644 index 00000000..eb0dc508 --- /dev/null +++ b/cmd/gputrace/cmd/optimize.go @@ -0,0 +1,81 @@ +package cmd + +import ( + "encoding/json" + "fmt" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/optimize" +) + +var optimizeRunOpts = struct { + warmups int + iterations int + output string + dir string + json bool +}{warmups: 1, iterations: 5} + +var optimizeCmd = &cobra.Command{ + Use: "optimize", + Short: "Measure, compare, and iterate on GPU workload performance", + Long: `Measure, compare, and iterate on GPU workload performance. + +The optimize subcommands close the agent loop: run a workload +reproducibly, compare two runs with a noise-aware verdict, and cite the +measured deltas that justify each conclusion.`, +} + +var optimizeRunCmd = &cobra.Command{ + Use: "run -- [args...]", + Short: "Run a workload repeatedly and record wall-clock statistics", + Long: `Run a workload repeatedly and record wall-clock statistics. + +Executes the command after '--' warmups+iterations times, discards the +warmups, and reports median/quartile wall time of the measured runs. The +full result persists as JSON when --output is given, in the schema that +'gputrace optimize compare' consumes. + +Child failures are recorded per iteration rather than aborting the +series; check failed_count before trusting a comparison.`, + Args: cobra.ArbitraryArgs, + RunE: func(cmd *cobra.Command, args []string) error { + if len(args) == 0 { + return fmt.Errorf("no command given; use: gputrace optimize run [flags] -- [args...]") + } + res, err := optimize.Run(optimize.Config{ + Command: args, + Dir: optimizeRunOpts.dir, + Warmups: optimizeRunOpts.warmups, + Iterations: optimizeRunOpts.iterations, + OutputPath: optimizeRunOpts.output, + }) + if err != nil && res == nil { + return err + } + out := cmd.OutOrStdout() + if optimizeRunOpts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(res) + } + fmt.Fprintf(out, "%d measured iterations (%d discarded warmups), %d failed\n", + len(res.Iterations), optimizeRunOpts.warmups, res.FailedCount) + fmt.Fprintf(out, "median %s q1 %s q3 %s\n", + dur(res.MedianNS), dur(res.Q1NS), dur(res.Q3NS)) + if optimizeRunOpts.output != "" { + fmt.Fprintf(out, "result written to %s\n", optimizeRunOpts.output) + } + return err + }, +} + +func init() { + optimizeRunCmd.Flags().IntVar(&optimizeRunOpts.warmups, "warmups", optimizeRunOpts.warmups, "Discarded runs before measurement") + optimizeRunCmd.Flags().IntVar(&optimizeRunOpts.iterations, "iterations", optimizeRunOpts.iterations, "Measured runs") + optimizeRunCmd.Flags().StringVarP(&optimizeRunOpts.output, "output", "o", optimizeRunOpts.output, "Persist full JSON result to this path") + optimizeRunCmd.Flags().StringVar(&optimizeRunOpts.dir, "dir", "", "Working directory for the workload") + optimizeRunCmd.Flags().BoolVar(&optimizeRunOpts.json, "json", false, "Print the full result as JSON") + optimizeCmd.AddCommand(optimizeRunCmd) + rootCmd.AddCommand(optimizeCmd) +} diff --git a/cmd/gputrace/cmd/optimize_capture_compare.go b/cmd/gputrace/cmd/optimize_capture_compare.go new file mode 100644 index 00000000..11f6617e --- /dev/null +++ b/cmd/gputrace/cmd/optimize_capture_compare.go @@ -0,0 +1,98 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "os" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/cuptitrace" + "github.com/tmc/gputrace/internal/gpuevent" +) + +var optimizeCaptureCompareCmd = &cobra.Command{ + Use: "compare-captures ", + Short: "Compare kernel activity between two captures (noise-free)", + Long: `Compare kernel activity between two captures. + +Unlike 'optimize compare' (wall-clock, noise-sensitive), this diffs the +kernel-level analyses of two capture bundles or JSONL files: per-kernel +launch counts, mean durations, and total time, with a verdict based on +which kernels moved beyond a 5% threshold. + +This is the right comparison when process startup or unified-memory +overhead dominates wall clock — e.g. integrated-GPU hosts where a real +28% kernel win can read as "equivalent" end to end.`, + Args: cobra.ExactArgs(2), + DisableFlagParsing: false, + RunE: func(cmd *cobra.Command, args []string) error { + base, err := loadCaptureReport(args[0]) + if err != nil { + return err + } + variant, err := loadCaptureReport(args[1]) + if err != nil { + return err + } + cmp := gpuevent.CompareCaptures(base, variant) + out := cmd.OutOrStdout() + if compareCapturesOpts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(cmp) + } + warnCrossHost(out, args[0], args[1]) + fmt.Fprintf(out, "verdict: %s\n", cmp.Verdict) + fmt.Fprintf(out, "%s\n", cmp.Summary) + fmt.Fprintf(out, "total kernel time: %.2f ms -> %.2f ms (%+.1f%%)\n\n", + float64(cmp.BaseTotalNS)/1e6, float64(cmp.VariantTotalNS)/1e6, cmp.TotalDeltaPct) + fmt.Fprintf(out, "Per-kernel deltas (by impact):\n") + for _, d := range cmp.KernelDeltas { + fmt.Fprintf(out, " %6.1f%% %4dx->%-4dx mean %9s -> %-9s %s\n", + d.DeltaPct, d.BaseCount, d.VariantCount, + dur(d.BaseMeanNS), dur(d.VariantMeanNS), shortKernel(d.Name)) + } + return nil + }, +} + +var compareCapturesOpts struct{ json bool } + +// loadCaptureReport reads a bundle or JSONL and analyzes it. +func loadCaptureReport(path string) (*gpuevent.Report, error) { + r, closers, err := cupticapture.OpenEvents(path) + if err != nil { + return nil, err + } + defer closers() + cap, err := gpuevent.DecodeJSONL(r) + if err != nil { + return nil, err + } + samplesPath := cupticapture.ResolveSamples(path, "") + if samplesPath != "" { + sf, sErr := os.Open(samplesPath) + if sErr == nil { + sc, dErr := gpuevent.DecodeJSONL(sf) + if dErr == nil { + cap.Samples = sc.Samples + } + sf.Close() + } + } + for i := range cap.Events { + if cap.Events[i].Kind == gpuevent.KindKernel && cap.Events[i].Name == "" { + cap.Events[i].Name = cuptitrace.Demangle(cap.Events[i].RawSymbol) + } + } + rep := gpuevent.Analyze(cap.Events, cap.Samples) + health := gpuevent.MeasureCompleteness(cap) + rep.Completeness = &health + return rep, nil +} + +func init() { + optimizeCaptureCompareCmd.Flags().BoolVar(&compareCapturesOpts.json, "json", false, "Output machine-readable comparison") + optimizeCmd.AddCommand(optimizeCaptureCompareCmd) +} diff --git a/cmd/gputrace/cmd/optimize_compare.go b/cmd/gputrace/cmd/optimize_compare.go new file mode 100644 index 00000000..e2c54326 --- /dev/null +++ b/cmd/gputrace/cmd/optimize_compare.go @@ -0,0 +1,66 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "os" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/optimize" +) + +var compareOpts = struct{ json bool }{} + +var optimizeCompareCmd = &cobra.Command{ + Use: "compare ", + Short: "Compare two optimize runs with a noise-aware verdict", + Long: `Compare two optimize runs with a noise-aware verdict. + +Reads two result files written by 'gputrace optimize run --output' and +decides whether the variant improved, regressed, is equivalent, or +whether the medians differ inside overlapping noise (noisy-change) — +in which case the only sound action is collecting more iterations. + +The verdict cites its evidence: both IQRs and the median delta. Agents +must treat noisy-change as "not yet known", never as improvement.`, + Args: cobra.ExactArgs(2), + RunE: func(cmd *cobra.Command, args []string) error { + base, err := readRunResult(args[0]) + if err != nil { + return err + } + variant, err := readRunResult(args[1]) + if err != nil { + return err + } + cmp := optimize.Compare(base, variant) + out := cmd.OutOrStdout() + if compareOpts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(cmp) + } + fmt.Fprintf(out, "verdict: %s\n", cmp.Verdict) + fmt.Fprintf(out, "base: median %s\n", dur(cmp.BaseMedianNS)) + fmt.Fprintf(out, "variant: median %s (%+.1f%%)\n", dur(cmp.VariantMedianNS), cmp.DeltaPct) + fmt.Fprintf(out, "%s\n", cmp.Reason) + return nil + }, +} + +func readRunResult(path string) (*optimize.Result, error) { + data, err := os.ReadFile(path) + if err != nil { + return nil, fmt.Errorf("read %s: %w", path, err) + } + var res optimize.Result + if err := json.Unmarshal(data, &res); err != nil { + return nil, fmt.Errorf("parse %s: %w", path, err) + } + return &res, nil +} + +func init() { + optimizeCompareCmd.Flags().BoolVar(&compareOpts.json, "json", false, "Output machine-readable comparison") + optimizeCmd.AddCommand(optimizeCompareCmd) +} diff --git a/cmd/gputrace/cmd/overhead_linux.go b/cmd/gputrace/cmd/overhead_linux.go new file mode 100644 index 00000000..94feb9ca --- /dev/null +++ b/cmd/gputrace/cmd/overhead_linux.go @@ -0,0 +1,186 @@ +//go:build linux + +package cmd + +import ( + "encoding/json" + "fmt" + "io" + "os" + "path/filepath" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/optimize" +) + +type overheadOptions struct { + iterations int + warmups int + effectSize float64 + api bool + nvtx bool + json bool +} + +var overheadCmd = newOverheadCommand(&overheadOptions{iterations: 5, warmups: 1, effectSize: 5}) + +// OverheadReport is how much the capture shim perturbs the workload it +// measures, and whether that perturbation is large enough to invalidate +// the effect the user is studying. +type OverheadReport struct { + Command []string `json:"command"` + Mode string `json:"mode"` + Baseline *optimize.Result `json:"baseline"` + Instrumented *optimize.Result `json:"instrumented"` + Comparison *optimize.Comparison `json:"comparison"` + OverheadPct float64 `json:"overhead_pct"` + EffectSizePct float64 `json:"effect_size_pct"` + Usable bool `json:"usable"` + Verdict string `json:"verdict"` +} + +func newOverheadCommand(opts *overheadOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "overhead [flags] -- [args...]", + Short: "Measure how much the capture shim perturbs a workload", + Long: `Measure the capture shim's own cost by running a workload with and +without it. + +A captured throughput number is only usable if the capture did not move +it. This runs the same command both ways, compares the wall-clock +distributions with the same noise-aware test 'optimize compare' uses, and +says whether the measured overhead is small enough for the effect you are +studying. + +--effect-size names the difference you care about, as a percentage. When +the shim's overhead reaches it, captured timings describe the capture as +much as the workload, and the report says so. + +Examples: + gputrace overhead -- ./workload + gputrace overhead --effect-size 3 -n 10 -- ./workload + gputrace overhead --api -- ./workload`, + Args: cobra.MinimumNArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runOverhead(cmd, args, opts) + }, + } + cmd.Flags().IntVarP(&opts.iterations, "iterations", "n", opts.iterations, "Measured runs per side") + cmd.Flags().IntVar(&opts.warmups, "warmups", opts.warmups, "Discarded runs before measurement, per side") + cmd.Flags().Float64Var(&opts.effectSize, "effect-size", opts.effectSize, "Effect size under study, in percent; overhead at or above it invalidates captured timings") + cmd.Flags().BoolVar(&opts.api, "api", opts.api, "Measure with host-side API records enabled") + cmd.Flags().BoolVar(&opts.nvtx, "nvtx", opts.nvtx, "Measure with NVTX marker records enabled") + cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output machine-readable JSON") + return cmd +} + +func runOverhead(cmd *cobra.Command, args []string, opts *overheadOptions) error { + if opts.iterations < 2 { + return fmt.Errorf("overhead: --iterations must be at least 2 to separate signal from noise") + } + // The instrumented side writes into a throwaway bundle: the records + // are not the point, the cost of producing them is. + bundle, err := os.MkdirTemp("", "gputrace-overhead-*.gpucapture") + if err != nil { + return err + } + defer os.RemoveAll(bundle) + + env, err := cupticapture.PreloadEnv(cupticapture.Options{ + OutputPath: filepath.Join(bundle, cupticapture.EventsFileName), + APIRecords: opts.api, + NVTX: opts.nvtx, + }) + if err != nil { + return err + } + + base := optimize.Config{Command: args, Warmups: opts.warmups, Iterations: opts.iterations} + out := cmd.OutOrStdout() + fmt.Fprintf(out, "Measuring %d runs per side (%d warmup%s)...\n", opts.iterations, opts.warmups, plural(opts.warmups)) + + baseline, err := optimize.Run(base) + if err != nil { + return fmt.Errorf("overhead: baseline: %w", err) + } + instrumented, err := optimize.Run(optimize.Config{ + Command: args, Dir: base.Dir, Warmups: opts.warmups, Iterations: opts.iterations, Env: env, + }) + if err != nil { + return fmt.Errorf("overhead: instrumented: %w", err) + } + + rep := &OverheadReport{ + Command: args, + Mode: overheadMode(opts), + Baseline: baseline, + Instrumented: instrumented, + Comparison: optimize.Compare(baseline, instrumented), + EffectSizePct: opts.effectSize, + } + rep.OverheadPct = rep.Comparison.DeltaPct + rep.Usable, rep.Verdict = overheadVerdict(rep) + + if opts.json { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(rep) + } + writeOverheadReport(out, rep) + return nil +} + +func overheadMode(opts *overheadOptions) string { + mode := "activity" + if opts.api { + mode += "+api" + } + if opts.nvtx { + mode += "+nvtx" + } + return mode +} + +// overheadVerdict decides whether captured timings can carry a claim +// about an effect of the stated size. A change the capture itself could +// have produced is not evidence of anything. +func overheadVerdict(rep *OverheadReport) (bool, string) { + switch rep.Comparison.Verdict { + case optimize.Equivalent: + return true, fmt.Sprintf("the shim's cost is inside run-to-run noise; captured timings carry effects down to %.1f%%", rep.EffectSizePct) + case optimize.NoisyChange, optimize.Inconclusive: + return false, fmt.Sprintf("the measurement itself is too noisy to bound the shim's cost (%s); rerun with more iterations", rep.Comparison.Verdict) + } + magnitude := rep.OverheadPct + if magnitude < 0 { + magnitude = -magnitude + } + if magnitude >= rep.EffectSizePct { + return false, fmt.Sprintf("the shim moves wall time by %.1f%%, at or beyond the %.1f%% effect under study: captured timings describe the capture as much as the workload", + magnitude, rep.EffectSizePct) + } + return true, fmt.Sprintf("the shim moves wall time by %.1f%%, below the %.1f%% effect under study", magnitude, rep.EffectSizePct) +} + +func writeOverheadReport(out io.Writer, rep *OverheadReport) { + fmt.Fprintf(out, "\ncapture mode: %s\n", rep.Mode) + fmt.Fprintf(out, "baseline: median %s IQR [%s..%s]\n", + dur(rep.Baseline.MedianNS), dur(rep.Baseline.Q1NS), dur(rep.Baseline.Q3NS)) + fmt.Fprintf(out, "instrumented: median %s IQR [%s..%s]\n", + dur(rep.Instrumented.MedianNS), dur(rep.Instrumented.Q1NS), dur(rep.Instrumented.Q3NS)) + fmt.Fprintf(out, "overhead: %+.1f%% (%s)\n", rep.OverheadPct, signedDur(rep.Comparison.DeltaNS)) + fmt.Fprintf(out, "separation: %s — %s\n", rep.Comparison.Verdict, rep.Comparison.Reason) + if failed := rep.Baseline.FailedCount + rep.Instrumented.FailedCount; failed > 0 { + fmt.Fprintf(out, "warning: %d run%s exited nonzero; the workload may not be doing equal work on both sides\n", failed, plural(failed)) + } + if rep.Usable { + fmt.Fprintf(out, "\nusable: %s\n", rep.Verdict) + return + } + fmt.Fprintf(out, "\nNOT usable for this effect size: %s\n", rep.Verdict) +} + +func init() { + rootCmd.AddCommand(overheadCmd) +} diff --git a/cmd/gputrace/cmd/overhead_linux_test.go b/cmd/gputrace/cmd/overhead_linux_test.go new file mode 100644 index 00000000..3f05a74c --- /dev/null +++ b/cmd/gputrace/cmd/overhead_linux_test.go @@ -0,0 +1,97 @@ +//go:build linux + +package cmd + +import ( + "strings" + "testing" + + "github.com/tmc/gputrace/internal/optimize" +) + +func TestOverheadVerdict(t *testing.T) { + tests := []struct { + name string + verdict optimize.Verdict + overheadPct float64 + effectSize float64 + wantUsable bool + wantReason string + }{ + { + name: "cost inside noise", + verdict: optimize.Equivalent, + overheadPct: 0.4, + effectSize: 5, + wantUsable: true, + wantReason: "inside run-to-run noise", + }, + { + // The measurement cannot bound the shim's cost, so it cannot + // license a claim either way. + name: "too noisy to bound", + verdict: optimize.NoisyChange, + overheadPct: 7, + effectSize: 5, + wantUsable: false, + wantReason: "too noisy", + }, + { + name: "overhead swamps the effect", + verdict: optimize.Regressed, + overheadPct: 5.7, + effectSize: 5, + wantUsable: false, + wantReason: "describe the capture as much as the workload", + }, + { + name: "overhead below the effect", + verdict: optimize.Regressed, + overheadPct: 1.2, + effectSize: 5, + wantUsable: true, + wantReason: "below the 5.0% effect", + }, + { + // A speedup that large is still a perturbation of this size. + name: "negative overhead still counts", + verdict: optimize.Improved, + overheadPct: -8, + effectSize: 5, + wantUsable: false, + wantReason: "8.0%", + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + rep := &OverheadReport{ + Comparison: &optimize.Comparison{Verdict: tt.verdict}, + OverheadPct: tt.overheadPct, + EffectSizePct: tt.effectSize, + } + usable, reason := overheadVerdict(rep) + if usable != tt.wantUsable { + t.Errorf("usable = %v, want %v (%s)", usable, tt.wantUsable, reason) + } + if !strings.Contains(reason, tt.wantReason) { + t.Errorf("reason = %q, want it to mention %q", reason, tt.wantReason) + } + }) + } +} + +func TestOverheadMode(t *testing.T) { + tests := []struct { + opts overheadOptions + want string + }{ + {overheadOptions{}, "activity"}, + {overheadOptions{api: true}, "activity+api"}, + {overheadOptions{api: true, nvtx: true}, "activity+api+nvtx"}, + } + for _, tt := range tests { + if got := overheadMode(&tt.opts); got != tt.want { + t.Errorf("overheadMode(%+v) = %q, want %q", tt.opts, got, tt.want) + } + } +} diff --git a/cmd/gputrace/cmd/perfcounters_validate.go b/cmd/gputrace/cmd/perfcounters_validate.go index 8d62d0f1..b83e05b8 100644 --- a/cmd/gputrace/cmd/perfcounters_validate.go +++ b/cmd/gputrace/cmd/perfcounters_validate.go @@ -26,7 +26,7 @@ func newPerfcountersValidateCommand(opts *perfcountersValidateOptions) *cobra.Co This command is critical for validating the binary parsing implementation: - Extracts metrics from .gpuprofiler_raw binary files - Compares against known-good Xcode Instruments data -- Reports accuracy for key metrics (Kernel Invocations, ALU Utilization, Occupancy) +- Reports accuracy for key metrics (Kernel Invocations, ALU Utilization) Used to validate replay engine accuracy by cross-checking against ground truth.`, Args: cobra.ExactArgs(2), @@ -73,7 +73,6 @@ type ReferenceCSVData struct { // Key metrics from first data row (index 0) KernelInvocations int ALUUtilization float64 - KernelOccupancy float64 MemoryBandwidthGBs float64 } @@ -128,12 +127,6 @@ func loadReferenceCSV(path string) (*ReferenceCSVData, error) { data.ALUUtilization, _ = strconv.ParseFloat(val, 64) } - // Kernel Occupancy (%) - column ~107 - if idx, ok := colIndex["Kernel Occupancy"]; ok && idx < len(firstRow) { - val := strings.TrimSpace(firstRow[idx]) - data.KernelOccupancy, _ = strconv.ParseFloat(val, 64) - } - // Device Memory Bandwidth (GB/s) - column ~52 if idx, ok := colIndex["Device Memory Bandwidth"]; ok && idx < len(firstRow) { val := strings.ReplaceAll(firstRow[idx], " GB/s", "") @@ -188,19 +181,6 @@ func validateMetrics(stats *gputrace.PerfCounterStats, ref *ReferenceCSVData) er aluDelta, aluStatus) - // Kernel Occupancy - occupancyDelta := firstEncoder.KernelOccupancy - ref.KernelOccupancy - occupancyStatus := "✅ PASS" - if abs(occupancyDelta) > 5.0 { - occupancyStatus = "❌ FAIL" - } - fmt.Printf("%-30s %14.2f%% %14.2f%% %+12.2f%% %s\n", - "Kernel Occupancy", - firstEncoder.KernelOccupancy, - ref.KernelOccupancy, - occupancyDelta, - occupancyStatus) - fmt.Printf("\n=== Summary ===\n") fmt.Printf("Total Encoders: %d\n", len(stats.ShaderMetrics)) fmt.Printf("Total Records: %d\n", stats.TotalRecords) diff --git a/cmd/gputrace/cmd/platform_commands.go b/cmd/gputrace/cmd/platform_commands.go index ec5f001c..e1b08310 100644 --- a/cmd/gputrace/cmd/platform_commands.go +++ b/cmd/gputrace/cmd/platform_commands.go @@ -24,27 +24,27 @@ This command uses Accessibility APIs to control Xcode's UI and extract data. Workflow: run Run full automation (open, replay, export) - open Open a trace file in Xcode - close Close the trace window - export Export the trace with performance data + open Open and bind a trace file in Xcode + close Close and verify removal of a trace window + export Export and verify a trace bundle run-profile Start profiling in Xcode - wait-profile Wait for profiling to complete + wait-profile Wait for Performance data UI readiness Status: check-status Check profiling status (ready, running, complete) check-permissions Check required permissions (Accessibility, Screen Recording) Navigation: - select-tab Select a tab by name - show-performance Click Show Performance button - show-summary Select Summary tab - show-counters Select Counters tab - show-memory Click the Show Memory button - show-dependencies Click Show Dependencies button + select-tab Select and verify a tab by name + show-performance Reveal and verify the Performance view + show-summary Select and verify the Summary tab + show-counters Select and verify the Counters tab + show-memory Select and verify the Memory view + show-dependencies Select and verify the Dependencies view Data Export: - xcode-export-counters Export GPU counters from Performance view to CSV - xcode-export-memory Export memory report from Performance view + xcode-export-counters Export counters to a verified stable CSV + xcode-export-memory Export a verified stable memory report vertex-output Extract vertex shader output from Xcode GPU debugger performance Performance data commands`, Args: cobra.MaximumNArgs(1), @@ -106,12 +106,24 @@ var xcodeProfileCommandSpecs = []platformCommandSpec{ {name: "run", use: "run ", short: "Run full automation (open, replay, export)", args: cobra.ExactArgs(1), silenceUsage: true, flags: outputFlag}, {name: "open", use: "open ", short: "Open a trace file in Xcode", long: `Opens a GPU trace file in Xcode and waits for the window to be ready. By default, opens in background without stealing focus. Use --foreground to bring Xcode to front.`, args: cobra.ExactArgs(1), flags: foregroundFlag}, - {name: "close", use: "close [trace_file]", short: "Close the trace window in Xcode", long: "Closes the Xcode window for the specified trace file, or the first window if no file specified.", args: cobra.MaximumNArgs(1)}, - {name: "export", use: "export [output_path]", short: "Export the trace from Xcode", long: `Triggers File > Export in Xcode and saves to the specified path. -If no path is specified, it defaults to the trace file path with -perfdata suffix, inferred from the Xcode window.`, args: cobra.MaximumNArgs(1)}, + {name: "close", use: "close [trace_file]", short: "Close and verify removal of a selected trace window", long: "Closes the uniquely selected Xcode trace window and verifies that it disappeared. When multiple windows are present, provide trace_file to avoid ambiguity.", args: cobra.MaximumNArgs(1)}, + {name: "export", use: "export [output_path]", short: "Export and verify a trace bundle from Xcode", long: `Triggers File > Export in Xcode, verifies the destination and stable output bundle, and saves to the specified path. +If no path is specified, it defaults to the trace file path with -perfdata suffix, inferred from the Xcode window. + +To recover a Performance workflow left by a combined run, provide all +of --recover-untitled, --source, --xcode-pid, and --xcode-app. Recovery stays +bound to that exact process and verifies the exported UUID against --source. +Use --check-recovery to verify the binding without opening the export sheet. +An exact untitled 95% Summary state is resumed by opening Performance once; +when Xcode omits the AX control, two stable Vision OCR samples are required +inside the selected window's right pane. Recovery never starts Replay. +If the recovered window still has an enabled Stop GPU workload control and +disabled Export, --finalize-workload explicitly presses Stop once and requires +Xcode to restore the exact source-bound Finished state. It then presses Show +Performance once and requires the same window to become export-ready.`, args: cobra.MaximumNArgs(1), flags: standaloneExportFlags}, {name: "run-profile", use: "run-profile [trace_file]", aliases: []string{"run-replay"}, short: "Start profiling in Xcode", long: `Clicks the Profile button if available, otherwise falls back to Replay button. The Profile button starts profiling directly without needing additional checkboxes.`, args: cobra.MaximumNArgs(1)}, - {name: "wait-profile", use: "wait-profile [trace_file]", aliases: []string{"wait-replay"}, short: "Wait for profiling to complete", long: "Polls Xcode until profiling completes (Show Performance button appears or Replay re-enabled).", args: cobra.MaximumNArgs(1)}, + {name: "wait-profile", use: "wait-profile [trace_file]", aliases: []string{"wait-replay"}, short: "Wait for Performance data to become available", long: "Polls the bound trace window until a completion-ready Performance control appears. This verifies UI readiness, not exported-bundle identity.", args: cobra.MaximumNArgs(1)}, {name: "check-status", use: "check-status [trace_file]", short: "Check profiling status", long: `Returns the current profiling status: - initializing: Trace loading, Replay button disabled - replay-ready: Ready to start replay @@ -124,7 +136,7 @@ The Profile button starts profiling directly without needing additional checkbox Use --json for machine-readable output. Use --no-prompt to check without triggering permission dialogs.`, args: cobra.NoArgs}, - {name: "select-tab", use: "select-tab ", short: "Select a tab in the trace viewer", long: `Selects a tab in the Xcode GPU trace viewer. + {name: "select-tab", use: "select-tab ", short: "Select and verify a trace-viewer tab", long: `Selects a tab in the Xcode GPU trace viewer and verifies its selected state. Available tabs: summary - Summary view with overview statistics @@ -133,13 +145,13 @@ Available tabs: encoders - Encoder timeline dependencies - Resource dependencies performance - Performance metrics (same as Show Performance button)`, args: cobra.ExactArgs(1)}, - {name: "show-performance", use: "show-performance", short: "Click the Show Performance button", args: cobra.NoArgs}, - {name: "show-summary", use: "show-summary", short: "Select the Summary tab", args: cobra.NoArgs}, - {name: "show-counters", use: "show-counters", short: "Select the Counters tab", args: cobra.NoArgs}, - {name: "show-memory", use: "show-memory", short: "Click the Show Memory button", args: cobra.NoArgs}, - {name: "show-dependencies", use: "show-dependencies", short: "Click the Show Dependencies button", args: cobra.NoArgs}, - {name: "xcode-export-counters", use: "xcode-export-counters [trace_file]", short: "Export GPU counters from Xcode's Performance view to CSV", long: xcodeExportCountersLong, args: cobra.MaximumNArgs(1), flags: forceFlag}, - {name: "xcode-export-memory", use: "xcode-export-memory [trace_file]", short: "Export memory report from Xcode's Performance view", long: xcodeExportMemoryLong, args: cobra.MaximumNArgs(1), flags: forceFlag}, + {name: "show-performance", use: "show-performance", short: "Reveal and verify the Performance view", args: cobra.NoArgs}, + {name: "show-summary", use: "show-summary", short: "Select and verify the Summary tab", args: cobra.NoArgs}, + {name: "show-counters", use: "show-counters", short: "Select and verify the Counters tab", args: cobra.NoArgs}, + {name: "show-memory", use: "show-memory", short: "Select and verify the Memory view", args: cobra.NoArgs}, + {name: "show-dependencies", use: "show-dependencies", short: "Select and verify the Dependencies view", args: cobra.NoArgs}, + {name: "xcode-export-counters", use: "xcode-export-counters [trace_file]", short: "Export counters to a verified stable CSV", long: xcodeExportCountersLong, args: cobra.MaximumNArgs(1), flags: forceFlag}, + {name: "xcode-export-memory", use: "xcode-export-memory [trace_file]", short: "Export a verified stable memory report", long: xcodeExportMemoryLong, args: cobra.MaximumNArgs(1), flags: forceFlag}, {name: "vertex-output", use: "vertex-output ", short: "Extract vertex shader output from Xcode GPU debugger", long: vertexOutputLong, args: cobra.ExactArgs(1), silenceUsage: true, flags: vertexOutputFlags}, {name: "list-windows", use: "list-windows [trace_file]", short: "List Xcode windows", long: "Lists Xcode windows with their titles, checkboxes, and buttons. Optionally filter by trace filename.", args: cobra.MaximumNArgs(1), hidden: true}, {name: "list-tabs", use: "list-tabs [trace_file]", short: "List available tabs in the trace viewer", args: cobra.MaximumNArgs(1), hidden: true}, @@ -212,6 +224,15 @@ func outputFlag(cmd *cobra.Command) { cmd.Flags().StringP("output", "o", "", "Output path for the exported trace") } +func standaloneExportFlags(cmd *cobra.Command) { + cmd.Flags().Bool("recover-untitled", false, "Recover a Performance workflow using explicit source and Xcode identity") + cmd.Flags().Bool("check-recovery", false, "Verify untitled recovery binding without changing Xcode UI") + cmd.Flags().Bool("finalize-workload", false, "Explicitly stop an unfinalized recovered workload before export") + cmd.Flags().String("source", "", "Source trace used to verify a recovered export") + cmd.Flags().Int("xcode-pid", 0, "Exact Xcode process ID for untitled-window recovery") + cmd.Flags().String("xcode-app", "", "Exact absolute Xcode.app path for untitled-window recovery") +} + func foregroundFlag(cmd *cobra.Command) { cmd.Flags().Bool("foreground", false, "Bring Xcode to foreground (default: open in background)") } @@ -250,7 +271,7 @@ func newPerformanceCommand() *cobra.Command { Long: `Commands for working with GPU performance data in Xcode. Subcommands: - show Click the "Show Performance" button to reveal performance data + show Reveal and verify the Performance view status Check if performance data is available summary Extract visible summary statistics counters Select the Counters tab @@ -267,7 +288,7 @@ Subcommands: func performanceCommandLong(name string) string { switch name { case "show": - return `Clicks the "Show Performance" button in Xcode to reveal GPU performance data.` + return `Reveals Xcode's Performance view and verifies that its navigation controls became available.` case "status": return `Checks whether the "Show Performance" button is available and enabled.` case "summary": @@ -275,14 +296,14 @@ func performanceCommandLong(name string) string { case "memory": return `Extracts memory allocation and usage information from Xcode when visible.` default: - return "Selects the " + name + " tab in Xcode's Performance view." + return "Selects the " + name + " view in Xcode Performance and verifies its selected state." } } func performanceCommandShort(name string) string { switch name { case "show": - return "Click the Show Performance button" + return "Reveal and verify the Performance view" case "status": return "Check if performance data is available" case "summary": @@ -290,7 +311,7 @@ func performanceCommandShort(name string) string { case "memory": return "Extract memory usage info" default: - return "Select the " + name + " tab" + return "Select and verify the " + name + " view" } } diff --git a/cmd/gputrace/cmd/platform_darwin.go b/cmd/gputrace/cmd/platform_darwin.go index 07b53bdb..9b44e506 100644 --- a/cmd/gputrace/cmd/platform_darwin.go +++ b/cmd/gputrace/cmd/platform_darwin.go @@ -60,7 +60,7 @@ func platformXcodeProfileRun(name string) func(*cobra.Command, []string) error { case "select-tab": return runSelectTab(cmd, args) case "show-performance": - return runShowPerformance(cmd, args) + return runPerformanceShow(cmd, args) case "show-summary": return runSelectTab(cmd, []string{"Summary"}) case "show-counters": diff --git a/cmd/gputrace/cmd/platform_unsupported_test.go b/cmd/gputrace/cmd/platform_unsupported_test.go index d636b385..8ac9e891 100644 --- a/cmd/gputrace/cmd/platform_unsupported_test.go +++ b/cmd/gputrace/cmd/platform_unsupported_test.go @@ -67,13 +67,10 @@ func TestDarwinOnlyXcodeProfileSurface(t *testing.T) { {"performance", "memory"}, } { t.Run(strings.Join(args, "_"), func(t *testing.T) { - command, remaining, err := collectXcodeProfileCmd.Find(args) + command, _, err := collectXcodeProfileCmd.Find(args) if err != nil { t.Fatalf("find %v: %v", args, err) } - if len(remaining) > 0 { - t.Fatalf("remaining args = %v, want none", remaining) - } if command == nil || command.RunE == nil { t.Fatalf("command %v missing RunE", args) } diff --git a/cmd/gputrace/cmd/pprof.go b/cmd/gputrace/cmd/pprof.go index 89ca6705..4cff4cae 100644 --- a/cmd/gputrace/cmd/pprof.go +++ b/cmd/gputrace/cmd/pprof.go @@ -8,6 +8,7 @@ import ( "github.com/spf13/cobra" "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/cupticapture" "github.com/tmc/gputrace/internal/export" "github.com/tmc/gputrace/internal/mlxprof" "github.com/tmc/gputrace/internal/timing" @@ -24,6 +25,7 @@ type pprofOptions struct { showStats bool searchPaths []string sourceLines bool + dot string } func newPprofCommand(opts *pprofOptions) *cobra.Command { @@ -60,7 +62,58 @@ The pprof profile shows GPU time organized hierarchically: └─ Encoder └─ Kernel (shader) -This makes it easy to identify which shaders are consuming the most GPU time.`, +This makes it easy to identify which shaders are consuming the most GPU time. + +CUDA captures (a .gpucapture bundle or an activity .jsonl) take a different +path: one sample per kernel launch, with four value types. + + gpu_time nanoseconds kernel end - start + launch_count count 1 per kernel record + queue_delay nanoseconds queued until device start + idle_after nanoseconds device time on this kernel's stream before the + next activity starts + +queue_delay has a structural ceiling: CUDA-graph replays carry no queue +timestamps at all, so on a graph-heavy decode it can only ever describe the +eager remainder — often a few dozen launches out of tens of thousands. The +export prints the coverage per run. Low coverage there is the workload's +shape, not a capture failure. + +Every export also states whether the capture kept every record the run +produced. A capture that dropped activity records loses them uniformly +across kernel names and sizes, so it renders, diffs, and reads as a real +result; the totals from one are a share of the run and not comparable +against anything. + +The last two are what make a comparison conclusive. Diffing two captures at +-sample_index=gpu_time and again at -sample_index=idle_after answers whether +one side's extra time is inside kernels or between them: + + gputrace pprof py.gpucapture -o py.pb.gz + gputrace pprof go.gpucapture -o go.pb.gz + go tool pprof -top -diff_base=py.pb.gz go.pb.gz + go tool pprof -top -sample_index=idle_after -diff_base=py.pb.gz go.pb.gz + +A diff needs both profiles to carry identical sample type lists, so pass +--dot to both sides or to neither. + +Stacks come from the application spans in the capture (the ones a target +writes to GPUTRACE_APP_EVENTS, or NVTX ranges), innermost span last and the +kernel name as the leaf. Span frames are the span name, so every decode step +aggregates into one frame; the step number rides along as the eval_seq +label, sliceable with -tagfocus. A kernel no span encloses is stacked under +an "unattributed" root rather than dropped. + +gpu_time is the SUM of kernel durations, not wall time. Kernels on different +streams overlap, so the total can exceed the capture's wall span, and it is +not rescaled to fit. Sum-of-durations is the right answer to "which kernel +costs most"; a flame graph built from it is not a timeline. For wall span and +occupancy, use gputrace summary — they are deliberately not folded in here. + + --dot join CUDA-graph structure from MLX_SAVE_CUDA_GRAPHS_DOT_FILE + dumps into the same profile, adding kernel_count and + graph_commits joined on kernel name. That turns "+56 kernels" + into "+56 kernels worth X ms". See gputrace dot-pprof.`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { return runPprof(cmd, args, opts) @@ -75,6 +128,7 @@ This makes it easy to identify which shaders are consuming the most GPU time.`, cmd.Flags().BoolVar(&opts.showStats, "stats", opts.showStats, "Show trace statistics only") cmd.Flags().StringSliceVar(&opts.searchPaths, "search-path", opts.searchPaths, "Search paths for shader source files") cmd.Flags().BoolVar(&opts.sourceLines, "source-lines", opts.sourceLines, "Generate pprof with per-source-line samples (enables go tool pprof -list)") + cmd.Flags().StringVar(&opts.dot, "dot", opts.dot, "CUDA-graph DOT dump directory to join into the profile (CUDA captures only)") return cmd } @@ -85,6 +139,16 @@ func init() { func runPprof(cmd *cobra.Command, args []string, opts *pprofOptions) error { tracePath := args[0] + // CUDA captures (.gpucapture bundles or activity JSONL) take the + // cuptiprofile path: per-launch kernel samples instead of Metal + // shader timing. Everything else keeps the Metal flow. + if cupticapture.IsBundle(tracePath) || filepath.Ext(tracePath) == ".jsonl" { + return runCuptiPprof(cmd, args, opts) + } + if opts.dot != "" { + return fmt.Errorf("--dot joins CUDA-graph dumps into a CUDA capture profile; %s is a Metal trace", tracePath) + } + // Verify trace file exists if err := checkTraceFile(tracePath); err != nil { return err @@ -169,7 +233,7 @@ func runPprof(cmd *cobra.Command, args []string, opts *pprofOptions) error { return fmt.Errorf("failed to write profiles: %w", err) } - fmt.Printf("✅ Generated profiles:\n") + fmt.Printf("Generated profiles:\n") fmt.Printf(" %s.gpu.pprof - Hierarchical GPU profile\n", outputPrefix) fmt.Printf(" %s.gpu-flat.pprof - Flat GPU profile\n", outputPrefix) fmt.Printf(" %s.combined.pprof - Combined multi-view profile\n", outputPrefix) @@ -188,7 +252,7 @@ func runPprof(cmd *cobra.Command, args []string, opts *pprofOptions) error { return fmt.Errorf("failed to write text report: %w", err) } - fmt.Fprintf(pprofStatusWriter(outputPath), "✅ Text report written to: %s\n", outputPath) + fmt.Fprintf(pprofStatusWriter(outputPath), "Text report written: %s\n", outputPath) } else { // Generate single pprof file @@ -206,7 +270,7 @@ func runPprof(cmd *cobra.Command, args []string, opts *pprofOptions) error { return fmt.Errorf("failed to write pprof: %w", err) } - fmt.Fprintf(status, "✅ GPU profile written to: %s\n", outputPath) + fmt.Fprintf(status, "GPU profile written: %s\n", outputPath) fmt.Fprintf(status, "\nView with: go tool pprof -top %s\n", outputPath) fmt.Fprintf(status, "Or: go tool pprof -http=:8080 %s\n", outputPath) } @@ -259,6 +323,7 @@ func generateSourceLinesPprof(tracePath string, opts *pprofOptions) error { timingSelection := selectSourceLineTimings(trace) fmt.Fprint(status, formatSourceLineTimingNotice(timingSelection.source, len(timingSelection.timings))) + fmt.Fprint(status, sourceLineGranularityNotice) timings := timingSelection.timings timings = appendSourceMappedEncoderTimings(trace, timings, mapper) @@ -283,8 +348,8 @@ func generateSourceLinesPprof(tracePath string, opts *pprofOptions) error { return fmt.Errorf("failed to write pprof: %w", err) } - fmt.Fprintf(status, "✅ Source-lines pprof written to: %s\n", outputPath) - fmt.Fprintf(status, "\nView per-line costs with:\n") + fmt.Fprintf(status, "Source-lines pprof written: %s\n", outputPath) + fmt.Fprintf(status, "\nLocate a kernel in its source with:\n") fmt.Fprintf(status, " go tool pprof -list %s\n", outputPath) fmt.Fprintf(status, "\nOr interactive mode:\n") fmt.Fprintf(status, " go tool pprof %s\n", outputPath) @@ -298,7 +363,6 @@ type sourceLineTimingSource string const ( sourceLineTimingProfiler sourceLineTimingSource = "profiler" sourceLineTimingEncoderLabels sourceLineTimingSource = "encoder_labels" - sourceLineTimingSynthetic sourceLineTimingSource = "synthetic" ) type sourceLineTimingSelection struct { @@ -323,10 +387,9 @@ func selectSourceLineTimings(trace *gputrace.Trace) sourceLineTimingSelection { } } - return sourceLineTimingSelection{ - timings: timing.GenerateSyntheticTiming(trace), - source: sourceLineTimingSynthetic, - } + // No timing source available. Source-line attribution without durations is + // still useful; invented durations are not. + return sourceLineTimingSelection{} } func sourceLineProfilerTimings(profilerTimings []gputrace.EncoderTimingInfo) []*export.EncoderTiming { @@ -348,16 +411,27 @@ func sourceLineProfilerTimings(profilerTimings []gputrace.EncoderTimingInfo) []* return timings } +// sourceLineGranularityNotice states the granularity of --source-lines output. +// +// A kernel's whole duration lands on the one line where the kernel is +// declared, because that is the only line the mapper can identify. Nothing in +// the trace says how the cost is spread across the kernel body: the archived +// MTLLibrary carries no debug-info section, and no counter record carries a +// program counter or a source line. Without this notice a `pprof -list` +// listing reads as a per-line measurement, which it is not. +// +// See docs/research/SOURCE_LEVEL_COST.md for the evidence. +const sourceLineGranularityNotice = "Granularity: per kernel, not per line. Each kernel's cost is reported at its\n" + + "declaration line; the trace carries no cost breakdown within a kernel body.\n" + func formatSourceLineTimingNotice(source sourceLineTimingSource, count int) string { switch source { case sourceLineTimingProfiler: return fmt.Sprintf("Timing source: profiler .gpuprofiler_raw data (%s)\n", formatTimingRows(count)) case sourceLineTimingEncoderLabels: return fmt.Sprintf("Timing source: encoder label timing data (%s)\n", formatTimingRows(count)) - case sourceLineTimingSynthetic: - return fmt.Sprintf("Timing source: synthetic fallback (%s; no real profiler or encoder label timing found)\n", formatTimingRows(count)) default: - return fmt.Sprintf("Timing source: unknown (%s)\n", formatTimingRows(count)) + return "Timing source: none (no profiler or encoder label timing found); source lines are reported without durations\n" } } @@ -383,11 +457,7 @@ func appendSourceMappedEncoderTimings(trace *gputrace.Trace, timings []*export.E if maxEnd == 0 { maxEnd = 1000000000000000 } - encoders, err := trace.ParseComputeEncoders() - if err != nil { - return timings - } - for _, enc := range encoders { + for _, enc := range trace.ParseComputeEncoders() { if enc.Label == "" || seen[enc.Label] { continue } diff --git a/cmd/gputrace/cmd/pprof_cupti.go b/cmd/gputrace/cmd/pprof_cupti.go new file mode 100644 index 00000000..08a46539 --- /dev/null +++ b/cmd/gputrace/cmd/pprof_cupti.go @@ -0,0 +1,230 @@ +package cmd + +import ( + "fmt" + "os" + "path/filepath" + "strings" + + "github.com/google/pprof/profile" + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/cudagraphdot" + "github.com/tmc/gputrace/internal/cupticapture" + "github.com/tmc/gputrace/internal/cuptiprofile" + "github.com/tmc/gputrace/internal/cuptitrace" + "github.com/tmc/gputrace/internal/gpuevent" +) + +// runCuptiPprof converts a CUDA capture (bundle or JSONL) into a pprof +// profile: one sample per kernel launch carrying GPU time, launch count, +// queue delay, and per-stream idle, stacked under the application spans +// that enclosed it. +func runCuptiPprof(cmd *cobra.Command, args []string, opts *pprofOptions) error { + path := args[0] + r, closers, err := cupticapture.OpenEvents(path) + if err != nil { + return err + } + defer closers() + cap, err := gpuevent.DecodeJSONL(r) + if err != nil { + return err + } + // Demangle for readable pprof function names, keeping each mangled + // symbol so it can be restored as the function's SystemName. + mangled := map[string]string{} + for i := range cap.Events { + e := &cap.Events[i] + if e.Kind != gpuevent.KindKernel || e.RawSymbol == "" { + continue + } + if e.Name == "" { + e.Name = cuptitrace.Demangle(e.RawSymbol) + } + mangled[e.Name] = e.RawSymbol + } + + var buildOpts cuptiprofile.Options + if opts.dot != "" { + nodes, commits, err := readGraphStructure(opts.dot, mangled) + if err != nil { + return err + } + buildOpts.Structure = nodes + buildOpts.Commits = commits + } + + prof, stats, err := cuptiprofile.Build(cap, buildOpts) + if err != nil { + return fmt.Errorf("%s: %w", path, err) + } + cuptiprofile.SetSystemNames(prof, mangled) + + outPath := opts.output + if outPath == "" { + base := filepath.Base(path) + base = strings.TrimSuffix(base, filepath.Ext(base)) + outPath = base + ".pprof" + } + if err := writeProfile(prof, outPath); err != nil { + return err + } + + status := pprofStatusWriter(outPath) + fmt.Fprintf(status, "Wrote %d kernel samples -> %s\n", stats.Kernels, outPath) + fmt.Fprint(status, formatCuptiPprofStats(stats)) + fmt.Fprintf(cmd.ErrOrStderr(), "View with: go tool pprof -top %s\n", outPath) + return nil +} + +// readGraphStructure loads a CUDA-graph dump directory into the structure +// nodes a joined profile carries. Symbols are demangled through the same +// table the activity records use, because the join is on the kernel name: +// a mangled node symbol and a demangled activity symbol are two functions, +// and the profile would show the counts and the time side by side without +// ever adding them up. +func readGraphStructure(path string, mangled map[string]string) ([]cuptiprofile.StructureNode, []string, error) { + files, err := loadGraphDumps(path) + if err != nil { + return nil, nil, err + } + var nodes []cuptiprofile.StructureNode + var commits []string + for _, f := range files { + for _, k := range f.Kernels() { + name := cuptitrace.Demangle(k.Symbol) + mangled[name] = k.Symbol + nodes = append(nodes, cuptiprofile.StructureNode{GraphPath: k.Path, Symbol: name}) + } + commits = append(commits, f.Roots...) + } + if len(nodes) == 0 { + return nil, nil, fmt.Errorf("dot: %s declares no kernel nodes", path) + } + return nodes, commits, nil +} + +// loadGraphDumps accepts a directory of dumps or a single dump file. +func loadGraphDumps(path string) ([]*cudagraphdot.File, error) { + info, err := os.Stat(path) + if err != nil { + return nil, err + } + if info.IsDir() { + return cudagraphdot.ParseDir(path) + } + f, err := cudagraphdot.ParseFile(path) + if err != nil { + return nil, err + } + return []*cudagraphdot.File{f}, nil +} + +// formatCuptiPprofStats reports what the profile's numbers are backed by. +// Span attribution and queue timing are both partial on real captures — +// CUDA-graph launches carry no queue timestamps at all — so the coverage +// is printed rather than left for the reader to assume. +// +// Dropped records are reported first and disqualify the totals rather than +// annotating them. Partial is the harder failure of the two the exporter +// guards against: an empty capture is obvious, while one missing half its +// records renders, diffs, and reads as a finding. +func formatCuptiPprofStats(stats cuptiprofile.Stats) string { + var b strings.Builder + if !stats.Complete() { + fmt.Fprintf(&b, " %s\n", stats.Completeness.Summary()) + fmt.Fprintf(&b, " every total below is a share of the run, not the run, and nothing in the profile\n") + fmt.Fprintf(&b, " itself looks wrong.\n") + for _, line := range strings.Split(stats.Completeness.Remedy(), "\n") { + fmt.Fprintf(&b, " %s\n", line) + } + } + fmt.Fprintf(&b, " gpu_time: %.2f ms summed over %d launches (not wall time; streams overlap)\n", + float64(stats.GPUTimeNS)/1e6, stats.Kernels) + if stats.Spans == 0 { + fmt.Fprintf(&b, " spans: none in capture; stacks are the kernel name alone\n") + } else { + fmt.Fprintf(&b, " spans: %d, enclosing %d of %d kernels (%.1f%%)\n", + stats.Spans, stats.SpanAttributed, stats.Kernels, stats.SpanAttributedPct()) + } + if stats.QueueTimed == 0 { + fmt.Fprintf(&b, " queue_delay: no kernel carries usable queue timestamps (CUDA-graph launches report none); all samples are 0\n") + } else { + fmt.Fprintf(&b, " queue_delay: median %s over the %d of %d kernels that carry queue timestamps\n", + formatDurationNS(stats.MedianQueueNS), stats.QueueTimed, stats.Kernels) + // The uncovered remainder is structural, not a capture failure. + // Saying so here stops the next reader deriving it again from a + // coverage figure that looks alarmingly low on a graph-heavy run. + if stats.QueueTimed < stats.Kernels { + fmt.Fprintf(&b, " the other %d are CUDA-graph replays, which carry no queue timestamps at all; this is a ceiling, not a gap\n", + stats.Kernels-stats.QueueTimed) + } + } + if stats.StructureNodes > 0 { + fmt.Fprintf(&b, " structure: %d graph kernel nodes across %d commits, joined on kernel name\n", + stats.StructureNodes, stats.Commits) + b.WriteString(formatCompleteness(stats)) + } + return b.String() +} + +// completenessFloor is the launches-per-declared-node ratio below which a +// capture is reported as likely incomplete. A graph replayed more than +// once pushes the ratio above 1, so only the low side is evidence; 0.9 +// leaves room for graphs committed near the end of a run that never got +// replayed before it finished. +const completenessFloor = 0.9 + +// formatCompleteness cross-checks declared graph work against measured +// launches. Both halves are already in hand whenever --dot is given, which +// makes this the one completeness check that costs nothing extra: the DOT +// dumps say how many kernel nodes were committed, the activity records say +// how many launched, and a ratio far below 1 is unrecorded work. +func formatCompleteness(stats cuptiprofile.Stats) string { + ratio := stats.NodeToLaunchRatio() + if ratio == 0 { + return "" + } + line := fmt.Sprintf(" completeness: %.2f launches per declared graph kernel node (%d / %d)\n", + ratio, stats.Kernels, stats.StructureNodes) + if ratio >= completenessFloor { + return line + } + return line + fmt.Sprintf( + " below %.2f: the run declared more graph work than it recorded launching. Either records were dropped\n"+ + " (this capture reports %d) or graphs were committed and never replayed — check the capture before comparing.\n", + completenessFloor, stats.Completeness.DroppedRecords) +} + +// formatDurationNS renders a duration at a scale that makes a wrong clock +// domain obvious: the failure this guards against reports seconds where +// microseconds belong. +func formatDurationNS(ns uint64) string { + switch { + case ns >= 1e9: + return fmt.Sprintf("%.2f s", float64(ns)/1e9) + case ns >= 1e6: + return fmt.Sprintf("%.2f ms", float64(ns)/1e6) + case ns >= 1e3: + return fmt.Sprintf("%.2f us", float64(ns)/1e3) + default: + return fmt.Sprintf("%d ns", ns) + } +} + +// writeProfile writes a profile to a path, or to stdout for "-" and +// /dev/stdout so the profile can be piped straight into pprof. +func writeProfile(prof *profile.Profile, outPath string) error { + if outputPathIsExplicitStdout(outPath) { + return prof.Write(os.Stdout) + } + f, err := os.Create(outPath) + if err != nil { + return err + } + if err := prof.Write(f); err != nil { + f.Close() + return fmt.Errorf("write pprof: %w", err) + } + return f.Close() +} diff --git a/cmd/gputrace/cmd/pprof_cupti_test.go b/cmd/gputrace/cmd/pprof_cupti_test.go new file mode 100644 index 00000000..0ab59bff --- /dev/null +++ b/cmd/gputrace/cmd/pprof_cupti_test.go @@ -0,0 +1,360 @@ +package cmd + +import ( + "os" + "path/filepath" + "reflect" + "strings" + "testing" + + "github.com/google/pprof/profile" + "github.com/tmc/gputrace/internal/cuptiprofile" + "github.com/tmc/gputrace/internal/gpuevent" +) + +// writeBundle creates a minimal .gpucapture bundle holding the given JSONL +// records, so the CLI path is exercised end to end without a GPU. +func writeBundle(t *testing.T, records string) string { + t.Helper() + dir := filepath.Join(t.TempDir(), "cap.gpucapture") + if err := os.MkdirAll(dir, 0o755); err != nil { + t.Fatal(err) + } + if err := os.WriteFile(filepath.Join(dir, "events.jsonl"), []byte(records), 0o644); err != nil { + t.Fatal(err) + } + return dir +} + +// spanBundle is a capture whose spans are on the unix clock and whose +// kernels are on the CUPTI clock, joined only by the clock_sync record — +// the arrangement a real MLX capture with GPUTRACE_APP_EVENTS produces. +const spanBundleRecords = `{"kind":"clock_sync","unix_ns":1000000000,"cupti_ns":5000000000} +{"kind":"span","name":"prefill","start_ns":1000000100,"end_ns":1000000400,"clock":"unix","labels":{"phase":"prefill"}} +{"kind":"span","name":"decode","start_ns":1000000400,"end_ns":1000000900,"clock":"unix","labels":{"phase":"decode"}} +{"kind":"span","name":"token","start_ns":1000000400,"end_ns":1000000600,"clock":"unix","eval_seq":1,"labels":{"phase":"decode"}} +{"kind":"kernel","raw_symbol":"_Z5saxpyifPfS_","start_ns":5000000150,"end_ns":5000000250,"stream_id":7,"queued_ns":5000000100,"submitted_ns":5000000120} +{"kind":"kernel","raw_symbol":"_Z4gemvifPfS_","start_ns":5000000450,"end_ns":5000000500,"stream_id":7} +{"kind":"kernel","raw_symbol":"_Z5loosev","start_ns":5000002000,"end_ns":5000002100,"stream_id":7} +` + +func runPprofCmd(t *testing.T, args ...string) error { + t.Helper() + resetPprofTestFlags() + t.Cleanup(resetPprofTestFlags) + rootCmd.SetArgs(args) + return rootCmd.Execute() +} + +func parseProfile(t *testing.T, path string) *profile.Profile { + t.Helper() + f, err := os.Open(path) + if err != nil { + t.Fatalf("open profile: %v", err) + } + defer f.Close() + p, err := profile.Parse(f) + if err != nil { + t.Fatalf("parse profile: %v", err) + } + return p +} + +func profileSampleTypes(p *profile.Profile) []string { + out := make([]string, len(p.SampleType)) + for i, st := range p.SampleType { + out[i] = st.Type + } + return out +} + +// renderStack returns one sample's frames outermost first, the order a +// reader sees them in pprof output. +func renderStack(s *profile.Sample) []string { + out := make([]string, 0, len(s.Location)) + for i := len(s.Location) - 1; i >= 0; i-- { + out = append(out, s.Location[i].Line[0].Function.Name) + } + return out +} + +// TestCuptiPprofEmptyCaptureIsAnError is acceptance criterion 2. A bundle +// that traced spans but flushed no kernel records must fail loudly: a +// valid empty profile parses, renders, and says nothing, which is how a +// missing in-process flush goes unnoticed. +func TestCuptiPprofEmptyCaptureIsAnError(t *testing.T) { + bundle := writeBundle(t, `{"kind":"clock_sync","unix_ns":1,"cupti_ns":2} +{"kind":"span","name":"prefill","start_ns":10,"end_ns":20,"clock":"unix"} +`) + out := filepath.Join(t.TempDir(), "empty.pb.gz") + err := runPprofCmd(t, "pprof", bundle, "-o", out) + if err == nil { + t.Fatal("exporting a capture with no kernel records succeeded") + } + for _, want := range []string{"no kernel records", "flush"} { + if !strings.Contains(err.Error(), want) { + t.Errorf("error %q does not name the likely cause (%q)", err, want) + } + } + if _, statErr := os.Stat(out); statErr == nil { + t.Error("a profile was written for an empty capture") + } +} + +// TestCuptiPprofRendersSpanStacks is acceptance criterion 4, asserted on +// the rendered stacks rather than on the exporter exiting zero. The spans +// arrive on the unix clock, so this also fails if clock_sync is skipped. +func TestCuptiPprofRendersSpanStacks(t *testing.T) { + bundle := writeBundle(t, spanBundleRecords) + out := filepath.Join(t.TempDir(), "spans.pb.gz") + if err := runPprofCmd(t, "pprof", bundle, "-o", out); err != nil { + t.Fatalf("pprof: %v", err) + } + p := parseProfile(t, out) + + if len(p.Sample) != 3 { + t.Fatalf("samples = %d, want 3", len(p.Sample)) + } + deep := 0 + stacks := map[string][]string{} + for _, s := range p.Sample { + frames := renderStack(s) + stacks[frames[len(frames)-1]] = frames + if len(frames) > 1 { + deep++ + } + } + // Two of the three kernels fall inside a span; the third is outside + // every span and must survive under the unattributed root. + if deep != 3 { + t.Errorf("%d of 3 samples have stack depth > 1, want 3", deep) + } + if got, want := stacks["saxpy"], []string{"prefill", "saxpy"}; !reflect.DeepEqual(got, want) { + t.Errorf("prefill stack = %v, want %v", got, want) + } + // decode encloses token encloses the kernel: innermost last. + if got, want := stacks["gemv"], []string{"decode", "token", "gemv"}; !reflect.DeepEqual(got, want) { + t.Errorf("decode stack = %v, want %v", got, want) + } + if got, want := stacks["loose"], []string{"unattributed", "loose"}; !reflect.DeepEqual(got, want) { + t.Errorf("unattributed stack = %v, want %v", got, want) + } +} + +// TestCuptiPprofDiffContract is acceptance criterion 3's precondition: +// pprof -diff_base needs identical sample type lists, and reports a +// confusing mismatch rather than a diff when they differ. +func TestCuptiPprofDiffContract(t *testing.T) { + dir := t.TempDir() + a := filepath.Join(dir, "a.pb.gz") + b := filepath.Join(dir, "b.pb.gz") + if err := runPprofCmd(t, "pprof", writeBundle(t, spanBundleRecords), "-o", a); err != nil { + t.Fatalf("pprof a: %v", err) + } + // A capture with no spans at all still has to produce the same value + // vector, or the two sides cannot be diffed. + noSpans := `{"kind":"kernel","raw_symbol":"_Z5saxpyifPfS_","start_ns":100,"end_ns":200,"stream_id":1} +` + if err := runPprofCmd(t, "pprof", writeBundle(t, noSpans), "-o", b); err != nil { + t.Fatalf("pprof b: %v", err) + } + + pa, pb := parseProfile(t, a), parseProfile(t, b) + want := []string{"gpu_time", "launch_count", "queue_delay", "idle_after"} + if got := profileSampleTypes(pa); !reflect.DeepEqual(got, want) { + t.Errorf("a sample types = %v, want %v", got, want) + } + if got := profileSampleTypes(pb); !reflect.DeepEqual(got, want) { + t.Errorf("b sample types = %v, want %v", got, want) + } + // The operation pprof -diff_base itself performs, which is where an + // incompatible pair fails with a message about the wrong thing. + if _, err := profile.Merge([]*profile.Profile{pa, pb}); err != nil { + t.Errorf("profiles are not diffable: %v", err) + } +} + +// TestCuptiPprofDemanglesAndKeepsTheSymbol checks the name pprof displays +// and the symbol the capture reported are both present: one is readable, +// the other is what a disassembler or nsys report can be matched against. +func TestCuptiPprofDemanglesAndKeepsTheSymbol(t *testing.T) { + bundle := writeBundle(t, spanBundleRecords) + out := filepath.Join(t.TempDir(), "names.pb.gz") + if err := runPprofCmd(t, "pprof", bundle, "-o", out); err != nil { + t.Fatalf("pprof: %v", err) + } + p := parseProfile(t, out) + for _, f := range p.Function { + if f.Name != "saxpy" { + continue + } + if f.SystemName != "_Z5saxpyifPfS_" { + t.Errorf("SystemName = %q, want the mangled symbol", f.SystemName) + } + return + } + // c++filt may be unavailable, in which case the name stays mangled and + // there is nothing to check. + t.Skip("c++filt did not demangle _Z5saxpyifPfS_; nothing to assert") +} + +// TestDotPprofMatchesTheHandCountedDump is acceptance criterion 5 against +// the fixture the cudagraphdot tests count by hand: eleven kernels in the +// root graph plus one reached through two levels of child graph. +func TestDotPprofMatchesTheHandCountedDump(t *testing.T) { + dump := "../../../internal/cudagraphdot/testdata/nested_graph.dot" + if _, err := os.Stat(dump); err != nil { + t.Skipf("fixture missing: %v", err) + } + out := filepath.Join(t.TempDir(), "structure.pb.gz") + resetDotPprofTestFlags() + t.Cleanup(resetDotPprofTestFlags) + rootCmd.SetArgs([]string{"dot-pprof", dump, "-o", out}) + if err := rootCmd.Execute(); err != nil { + t.Fatalf("dot-pprof: %v", err) + } + p := parseProfile(t, out) + if got := profileSampleTypes(p); !reflect.DeepEqual(got, []string{"kernel_count", "graph_commits"}) { + t.Errorf("sample types = %v", got) + } + var kernels, commits int64 + for _, s := range p.Sample { + kernels += s.Value[0] + commits += s.Value[1] + } + if kernels != 12 { + t.Errorf("kernel_count = %d, want 12 (dotdepth.py reports 12 for this dump)", kernels) + } + if commits != 1 { + t.Errorf("graph_commits = %d, want 1", commits) + } + // The grandchild kernel must arrive with the whole descent on its + // stack, not flattened onto the root. + var deepest []string + for _, s := range p.Sample { + if frames := renderStack(s); len(frames) > len(deepest) { + deepest = frames + } + } + want := []string{"graph_130", "graph_131", "graph_132"} + if len(deepest) != 4 || !reflect.DeepEqual(deepest[:3], want) { + t.Errorf("deepest stack = %v, want %v then a kernel", deepest, want) + } +} + +func TestPprofDotRejectedForMetalTraces(t *testing.T) { + tracePath := "../../../testdata/traces/01-single-encoder/01-single-encoder-run1.gputrace" + if _, err := os.Stat(tracePath); os.IsNotExist(err) { + t.Skipf("trace fixture not found: %s", tracePath) + } + err := runPprofCmd(t, "pprof", tracePath, "--dot", t.TempDir(), "-o", filepath.Join(t.TempDir(), "x.pprof")) + if err == nil { + t.Fatal("--dot was accepted for a Metal trace") + } + if !strings.Contains(err.Error(), "CUDA") { + t.Errorf("error %q should explain --dot is for CUDA captures", err) + } +} + +func resetDotPprofTestFlags() { + _ = dotPprofCmd.Flags().Set("output", "") +} + +// TestCuptiPprofStatsDisqualifyAPartialCapture: a capture that dropped +// records still builds a valid profile — which is the problem. Every total +// in it is a share of the run presented as the run, and the loss is uniform +// across kernel names and sizes, so nothing in the profile looks wrong. +// The export has to say so before the numbers. +func TestCuptiPprofStatsDisqualifyAPartialCapture(t *testing.T) { + partial := cuptiprofile.Stats{ + Kernels: 38835, + GPUTimeNS: 943_610_000, + Completeness: gpuevent.Completeness{Records: 38835, DroppedRecords: 31144}, + } + out := formatCuptiPprofStats(partial) + for _, want := range []string{"INCOMPLETE", "31144", "share of the run"} { + if !strings.Contains(out, want) { + t.Errorf("stats do not mention %q:\n%s", want, out) + } + } + // The disqualification must come before the number it disqualifies. + if strings.Index(out, "INCOMPLETE") > strings.Index(out, "gpu_time") { + t.Errorf("the totals are printed before the warning about them:\n%s", out) + } + + clean := cuptiprofile.Stats{ + Kernels: 200, + GPUTimeNS: 1_000_000, + Completeness: gpuevent.Completeness{Records: 200}, + } + if got := formatCuptiPprofStats(clean); strings.Contains(got, "INCOMPLETE") { + t.Errorf("a complete capture was reported incomplete:\n%s", got) + } +} + +// TestCuptiPprofCrossChecksDeclaredWorkAgainstLaunches is the completeness +// alarm that needs no second instrument: with --dot the export already +// holds both halves. The DOT dumps say how many kernel nodes were +// committed, the activity records say how many launched, and a ratio far +// below 1 is work that ran and was not recorded. +func TestCuptiPprofCrossChecksDeclaredWorkAgainstLaunches(t *testing.T) { + short := cuptiprofile.Stats{ + Kernels: 38835, + StructureNodes: 73486, + Commits: 3745, + Completeness: gpuevent.Completeness{Records: 38835}, + } + out := formatCuptiPprofStats(short) + if !strings.Contains(out, "completeness:") { + t.Fatalf("no completeness line for a capture with graph structure:\n%s", out) + } + if !strings.Contains(out, "38835") || !strings.Contains(out, "73486") { + t.Errorf("completeness line does not show both halves:\n%s", out) + } + if !strings.Contains(out, "records were dropped") { + t.Errorf("a ratio of 0.53 is not called out as an alarm:\n%s", out) + } + + // A graph replayed more often than it was committed pushes the ratio + // above 1, which is the ordinary shape of a decode loop, not a defect. + replayed := cuptiprofile.Stats{ + Kernels: 63650, + StructureNodes: 5000, + Commits: 40, + Completeness: gpuevent.Completeness{Records: 63650}, + } + if got := formatCuptiPprofStats(replayed); strings.Contains(got, "records were dropped") { + t.Errorf("a ratio above 1 was reported as loss:\n%s", got) + } +} + +// TestCuptiPprofStatesTheQueueDelayCeiling: on a graph-heavy decode the +// coverage figure looks alarmingly low, and it is structural — CUDA-graph +// replays carry no queue timestamps at all. Saying so where the number is +// printed stops the next reader deriving it again. +func TestCuptiPprofStatesTheQueueDelayCeiling(t *testing.T) { + out := formatCuptiPprofStats(cuptiprofile.Stats{ + Kernels: 63678, + QueueTimed: 28, + MedianQueueNS: 389_070, + Completeness: gpuevent.Completeness{Records: 63678}, + }) + if !strings.Contains(out, "389.07 us") { + t.Errorf("the median is not rendered at a microsecond scale:\n%s", out) + } + for _, want := range []string{"CUDA-graph replays", "ceiling, not a gap"} { + if !strings.Contains(out, want) { + t.Errorf("stats do not explain the uncovered remainder (%q):\n%s", want, out) + } + } + // Full coverage has no remainder to explain. + full := formatCuptiPprofStats(cuptiprofile.Stats{ + Kernels: 200, + QueueTimed: 200, + MedianQueueNS: 1000, + Completeness: gpuevent.Completeness{Records: 200}, + }) + if strings.Contains(full, "ceiling, not a gap") { + t.Errorf("a fully covered capture was given the ceiling note:\n%s", full) + } +} diff --git a/cmd/gputrace/cmd/pprof_test.go b/cmd/gputrace/cmd/pprof_test.go index 36c95a8f..106ad4c9 100644 --- a/cmd/gputrace/cmd/pprof_test.go +++ b/cmd/gputrace/cmd/pprof_test.go @@ -89,7 +89,7 @@ func TestPprofCmd(t *testing.T) { } } -func TestPprofSourceLinesDisclosesSyntheticTimingFallback(t *testing.T) { +func TestPprofSourceLinesDisclosesMissingTiming(t *testing.T) { tmpDir := t.TempDir() tracePath := "../../../testdata/traces/01-single-encoder/01-single-encoder-run1.gputrace" @@ -111,9 +111,12 @@ func TestPprofSourceLinesDisclosesSyntheticTimingFallback(t *testing.T) { if _, err := os.Stat(outputPath); os.IsNotExist(err) { t.Fatal("expected source.pprof to exist") } + // The profile is still written: its non-duration value types are compiler + // statistics that do not depend on timing. What used to be filled in by a + // synthetic fallback is now reported as absent. for _, want := range []string{ - "Timing source: synthetic fallback", - "no real profiler or encoder label timing found", + "Timing source: none", + "source lines are reported without durations", } { if !strings.Contains(stdout, want) { t.Fatalf("stdout does not contain %q:\n%s", want, stdout) @@ -247,10 +250,10 @@ func TestFormatSourceLineTimingNotice(t *testing.T) { want: "Timing source: encoder label timing data (1 encoder)\n", }, { - name: "synthetic", - source: sourceLineTimingSynthetic, + name: "none", + source: "", count: 0, - want: "Timing source: synthetic fallback (0 encoders; no real profiler or encoder label timing found)\n", + want: "Timing source: none (no profiler or encoder label timing found); source lines are reported without durations\n", }, } @@ -273,6 +276,7 @@ func resetPprofTestFlags() { _ = flags.Set("stats", "false") _ = flags.Set("search-path", "") _ = flags.Set("source-lines", "false") + _ = flags.Set("dot", "") } func captureStdout(t *testing.T, run func() error) (string, error) { diff --git a/cmd/gputrace/cmd/profile_replay.go b/cmd/gputrace/cmd/profile_replay.go new file mode 100644 index 00000000..55dad910 --- /dev/null +++ b/cmd/gputrace/cmd/profile_replay.go @@ -0,0 +1,84 @@ +package cmd + +import ( + "fmt" + + "github.com/spf13/cobra" + + "github.com/tmc/gputrace/internal/profilereplay" +) + +var profileReplayCmd = newProfileReplayCommand(&profileReplayOptions{}) + +type profileReplayOptions struct { + output string + embed bool + profilerOnly bool + wait bool +} + +func newProfileReplayCommand(opts *profileReplayOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "profile-replay ", + Short: "Replay a captured trace under the profiler to add performance data", + Long: `Replay a captured .gputrace under Apple's MTLReplayer with the profiler +attached, writing a bundle that carries measured performance data. + +A capture records what a Metal workload did and carries no timing. This replays +it on the GPU and collects streamData plus the Counters, Profiling and Timeline +shards. It is headless -- MTLReplayer is an agent process, so no window opens +and the frontmost application does not change. A small trace takes a few seconds. + +The default output is a self-contained -perfdata.gputrace bundle containing the +original capture and resources plus the profiler payload. Xcode can open it, +and capture-dependent commands such as kernels, buffer bindings, and grid and +threadgroup sizes remain available. + +Use --profiler-only only when the smaller raw payload is sufficient. It writes +a .gpuprofiler_raw directory for profiler, timing, timeline, and pprof. It is +not a .gputrace bundle and cannot be opened by Xcode. + +Only one MTLReplayer profiling job runs at a time. By default, a concurrent +invocation fails with a busy error. Use --wait to queue behind the active job. +This prevents separate replay processes from overlapping; it does not change +the command-buffer or encoder concurrency recorded inside one capture. + +This does not produce derived counters. Utilization, limiter and occupancy +values are not available on this GPU generation; MTLReplayer's counter flags +reach a dispatch branch with no writer, and its raw-counter writer is preempted +by the profiler flags used here. + +Examples: + gputrace profile-replay run.gputrace # run-perfdata.gputrace + gputrace profile-replay run.gputrace -o profiled.gputrace + gputrace profile-replay run.gputrace --profiler-only # .gpuprofiler_raw + gputrace profile-replay run.gputrace --wait # queue serially`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + if opts.embed && opts.profilerOnly { + return fmt.Errorf("--embed and --profiler-only are mutually exclusive") + } + out, err := profilereplay.Profile(cmd.Context(), args[0], profilereplay.Options{ + Output: opts.output, + ProfilerOnly: opts.profilerOnly, + Wait: opts.wait, + }) + if err != nil { + return err + } + fmt.Fprintf(cmd.OutOrStdout(), "wrote %s\n", out) + return nil + }, + } + f := cmd.Flags() + f.StringVarP(&opts.output, "output", "o", "", "path of the bundle to write (default -perfdata.gputrace)") + f.BoolVar(&opts.profilerOnly, "profiler-only", false, "write only a .gpuprofiler_raw payload, not an Xcode-openable trace") + f.BoolVar(&opts.embed, "embed", false, "deprecated compatibility flag; self-contained output is now the default") + _ = f.MarkDeprecated("embed", "self-contained output is now the default; omit --embed") + f.BoolVar(&opts.wait, "wait", false, "wait for another replay instead of reporting that MTLReplayer is busy") + return cmd +} + +func init() { + rootCmd.AddCommand(profileReplayCmd) +} diff --git a/cmd/gputrace/cmd/profile_replay_test.go b/cmd/gputrace/cmd/profile_replay_test.go new file mode 100644 index 00000000..71f4baa8 --- /dev/null +++ b/cmd/gputrace/cmd/profile_replay_test.go @@ -0,0 +1,24 @@ +package cmd + +import ( + "strings" + "testing" +) + +func TestProfileReplayEmbedCompatibility(t *testing.T) { + cmd := newProfileReplayCommand(new(profileReplayOptions)) + cmd.SetArgs([]string{"trace.gputrace", "--embed", "--profiler-only"}) + err := cmd.Execute() + if err == nil || !strings.Contains(err.Error(), "mutually exclusive") { + t.Fatalf("error = %v, want mutually exclusive", err) + } +} + +func TestProfileReplayHelpExplainsOutputShapes(t *testing.T) { + cmd := newProfileReplayCommand(new(profileReplayOptions)) + for _, text := range []string{"self-contained", ".gpuprofiler_raw", "cannot be opened by Xcode"} { + if !strings.Contains(cmd.Long, text) { + t.Errorf("help does not contain %q", text) + } + } +} diff --git a/cmd/gputrace/cmd/profiler.go b/cmd/gputrace/cmd/profiler.go index 391bbb9a..f9b09905 100644 --- a/cmd/gputrace/cmd/profiler.go +++ b/cmd/gputrace/cmd/profiler.go @@ -9,8 +9,11 @@ import ( "path/filepath" "sort" "strings" + "time" "github.com/spf13/cobra" + + "github.com/tmc/gputrace" "github.com/tmc/gputrace/internal/counter" ) @@ -18,19 +21,27 @@ import ( type ProfilerOutputStats struct { *counter.StreamDataStats ExecutionCost []counter.ExecutionCostByFunction `json:"execution_cost,omitempty"` + EncoderCost []counter.EncoderCost `json:"encoder_execution_cost,omitempty"` // TimelineInfo is explicitly included to ensure it appears in JSON output // (StreamDataStats.Timeline is already included via embedding, but this ensures visibility) } -var profilerCmd = newProfilerCommand(new(profilerOptions)) +var profilerCmd = newProfilerCommand(&profilerOptions{limit: 20}) type profilerOptions struct { - json bool - limiters bool - kernels bool + json bool + limiters bool + kernels bool + limit int + minCalls int + benchfmt bool + benchConfig benchfmtConfigFlags } func newProfilerCommand(opts *profilerOptions) *cobra.Command { + if opts.limit == 0 { + opts.limit = 20 + } cmd := &cobra.Command{ Use: "profiler ", Short: "Extract GPU profiler data (timing, dispatches, pipelines) from trace", @@ -57,6 +68,9 @@ Example: cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output in JSON format") cmd.Flags().BoolVar(&opts.limiters, "limiters", opts.limiters, "Show performance limiter data from Counter files") cmd.Flags().BoolVar(&opts.kernels, "kernels", opts.kernels, "Show kernel/function names and per-dispatch details") + cmd.Flags().IntVar(&opts.limit, "limit", opts.limit, "Maximum non-zero limiter rows to show") + cmd.Flags().IntVar(&opts.minCalls, "min-calls", opts.minCalls, "Only table rows for functions dispatched at least N times (off by default; reports what it drops; JSON and benchfmt are never filtered)") + addBenchfmtFlags(cmd, &opts.benchfmt, &opts.benchConfig) return cmd } @@ -65,22 +79,35 @@ func init() { } func runProfiler(cmd *cobra.Command, args []string, opts *profilerOptions) error { + if opts.limit <= 0 { + return fmt.Errorf("--limit must be > 0") + } + if opts.minCalls < 0 { + return fmt.Errorf("--min-calls must be >= 0") + } + if err := validateBenchfmtFlags(opts.benchfmt, opts.benchConfig); err != nil { + return err + } + if opts.benchfmt && opts.json { + return fmt.Errorf("--benchfmt and --json are mutually exclusive") + } tracePath := args[0] profilerDir, stats, err := loadProfilerStats(tracePath) if err != nil { - fmt.Fprintf(os.Stderr, "Hint: To generate performance data, run:\n") - fmt.Fprintf(os.Stderr, " gputrace xcode-profile run %s\n\n", tracePath) return err } - // Parse execution cost from Profiling_f_*.raw files execCost := aggregateExecutionCost(profilerDir, stats) + if opts.benchfmt { + return writeProfilerBenchfmt(cmd.OutOrStdout(), tracePath, stats, execCost, opts.benchConfig) + } if opts.json { output := ProfilerOutputStats{ StreamDataStats: stats, ExecutionCost: execCost, + EncoderCost: stats.CounterArchive.EncoderCosts(), } return writeProfilerJSON(cmd.OutOrStdout(), output) } @@ -105,28 +132,7 @@ func runProfiler(cmd *cobra.Command, args []string, opts *profilerOptions) error totalDeviceStores += p.DeviceStoreCount } - // Aggregate dispatches by function for counts - funcCounts := make(map[string]int) - funcTime := make(map[string]int) - for _, d := range stats.Dispatches { - name := d.DisplayName() - funcCounts[name]++ - funcTime[name] += d.DurationUs - } - - // Sort functions by time - type funcStat struct { - name string - time int - count int - } - var sortedFuncs []funcStat - for name, count := range funcCounts { - sortedFuncs = append(sortedFuncs, funcStat{name, funcTime[name], count}) - } - sort.Slice(sortedFuncs, func(i, j int) bool { - return sortedFuncs[i].time > sortedFuncs[j].time - }) + sortedFuncs := profilerFunctionRows(stats.Dispatches, totalDispatchTime) // === MAIN SUMMARY OUTPUT === // One-line summary @@ -159,46 +165,84 @@ func runProfiler(cmd *cobra.Command, args []string, opts *profilerOptions) error if stats.TimingSource != "" { fmt.Printf(" Timing Source: %s\n", stats.TimingSource) } + if stats.Metadata.NumBlitCalls != nil { + fmt.Printf(" Blit Calls: %s\n", FormatCount(int(*stats.Metadata.NumBlitCalls))) + } if totalThreadgroupMem > 0 { fmt.Printf(" Threadgroup Mem: %s (max per pipeline)\n", FormatBytes(uint64(totalThreadgroupMem))) } if totalDeviceLoads > 0 || totalDeviceStores > 0 { fmt.Printf(" Memory Ops: %s loads, %s stores\n", FormatCount(totalDeviceLoads), FormatCount(totalDeviceStores)) } + // The GRC counter stream is machine wide, so a capture accounts for only + // part of it and the rest belongs to whatever else was on the GPU. The + // share was computed and printed nowhere, which left the Execution Cost + // table below reading as if it covered the whole stream. It covers the + // attributed part. + // + // The encoder counts are spelled out rather than folded into one number + // because they are three different populations: ids carrying counter + // samples, ids declared in Encoder Infos, and the capture's own compute + // encoders printed above. On one trace those are 48, 96, and 3. A single + // unlabeled "encoders" here would read as a contradiction of the line + // above it, which is the same conflation 40a39534 removed from buffers. + if a := stats.CounterArchive; a != nil && a.TotalSamples > 0 { + fmt.Printf(" Counter Samples: %s/%s attributed to this capture (%s), %s machine-wide\n", + FormatCount(a.AttributedSamples), FormatCount(a.TotalSamples), + FormatPercent(100*a.AttributedFraction()), FormatCount(a.MachineWideSamples)) + fmt.Printf(" Counter Encoders: %s with samples, %s declared in Encoder Infos (GPU-wide ids, not the compute encoders above)\n", + FormatCount(len(a.Encoders)), FormatCount(a.KnownEncoderIDs)) + } + + // Compilation is host-side and absent from every duration above. + writeCompileSummary(os.Stdout, summarizeCompilation(stats.Pipelines)) // Show function call counts (always) if len(sortedFuncs) > 0 { fmt.Println() - fmt.Println(Colorize("Function Calls", ColorBold)) - fmt.Println(TableSeparator(80)) - fmt.Printf("%-50s %8s %10s %8s\n", "Function", "Calls", "Span(us)", "Cost") - fmt.Println(TableSeparator(80)) - for _, fs := range sortedFuncs { - pct := 0.0 - if totalDispatchTime > 0 { - pct = float64(fs.time) / float64(totalDispatchTime) * 100 + fmt.Print(formatProfilerFunctionCalls(sortedFuncs, opts.minCalls, + strings.Contains(stats.TimingSource, "gpuCommandInfoData"))) + } + + // Per-encoder execution cost, which is how Xcode groups the column. + if encCost := stats.CounterArchive.EncoderCosts(); len(encCost) > 0 { + fmt.Println() + fmt.Println(Colorize("Execution Cost by Encoder (from APSCounterData GRC_GPU_CYCLES)", ColorBold)) + fmt.Println(TableSeparator(60)) + fmt.Printf("%-10s %10s %14s %10s\n", "Encoder", "Cost", "GPU Cycles", "Reads") + fmt.Println(TableSeparator(60)) + for _, c := range encCost { + mark := "" + if c.Sparse() { + mark = " (few reads)" } - fmt.Printf("%-50s %8s %10s %7s\n", fs.name, FormatCount(fs.count), FormatCount(fs.time), FormatPercent(pct)) + fmt.Printf("%-10d %9s %14s %10d%s\n", + c.Ordinal, FormatPercent(c.CostPercent), FormatCount(int(c.GPUCycles)), c.EndRecords, mark) } + fmt.Println("Differs from Xcode's Execution Cost column by 0.9 to 2.9 pp depending on the") + fmt.Println("capture, worst on whichever encoder dominates the trace, and these are shares") + fmt.Println("that sum to 100%, so understating one encoder overstates the rest. Rank by this") + fmt.Println("column; do not quote it. See internal/counter/encodercost.go.") } // Detailed kernel info only with --kernels flag if opts.kernels { + functionNames := dispatchedFunctionNames(stats.Dispatches) + pipelines := dispatchedPipelines(stats.Pipelines, stats.Dispatches) + // Function names fmt.Println() fmt.Println(Colorize("Kernel Details", ColorBold)) fmt.Println(TableSeparator(40)) - fmt.Printf("%d %s:\n", len(stats.FunctionNames), Pluralize(len(stats.FunctionNames), "function", "functions")) - for i, name := range stats.FunctionNames { - if name != "" { - fmt.Printf(" [%d] %s\n", i, name) - } + fmt.Printf("%d dispatched %s:\n", len(functionNames), Pluralize(len(functionNames), "function", "functions")) + for i, name := range functionNames { + fmt.Printf(" [%d] %s\n", i, name) } // Pipelines with addresses - if len(stats.Pipelines) > 0 { - fmt.Printf("\n%d %s:\n", len(stats.Pipelines), Pluralize(len(stats.Pipelines), "pipeline", "pipelines")) - for i, p := range stats.Pipelines { + if len(pipelines) > 0 { + fmt.Printf("\n%d dispatched %s:\n", len(pipelines), Pluralize(len(pipelines), "pipeline", "pipelines")) + for i, p := range pipelines { if p.PipelineAddress != 0 { fmt.Printf(" [%d] 0x%x ID=%d %s\n", i, p.PipelineAddress, p.PipelineID, p.FunctionName) } else { @@ -218,6 +262,7 @@ func runProfiler(cmd *cobra.Command, args []string, opts *profilerOptions) error fmt.Printf(" Memory Ops: device(load=%d store=%d) threadgroup(load=%d store=%d)\n", p.DeviceLoadCount, p.DeviceStoreCount, p.ThreadgroupLoadCount, p.ThreadgroupStoreCount) } + writePipelineCompileDetail(os.Stdout, &p) } } @@ -409,24 +454,152 @@ func runProfiler(cmd *cobra.Command, args []string, opts *profilerOptions) error if opts.limiters { limiterData := extractLimiterData(profilerDir) if len(limiterData) > 0 { + rows, nonzero, zero := selectLimiterRows(limiterData, opts.limit) + fmt.Println() + fmt.Println(Colorize("Candidate Performance Limiters (heuristic Counter-file decoder)", ColorBold)) + fmt.Printf("Showing %d of %d non-zero rows", len(rows), nonzero) + if zero > 0 { + fmt.Printf(" (%d zero rows omitted)", zero) + } fmt.Println() - fmt.Println(Colorize("Performance Limiters (from Counter files)", ColorBold)) - fmt.Println(TableSeparator(95)) - fmt.Printf("%-5s %-16s %-18s %-16s %-16s %-16s\n", - "Enc", "Occupancy Mgr", "Instr Throughput", "Int & Complex", "F32 Limiter", "L1 Cache") - fmt.Println(TableSeparator(95)) - for _, ld := range limiterData { - fmt.Printf("%-5d %15s %17s %15s %15s %15s\n", - ld.EncoderIndex, FormatPercent(ld.OccupancyManager), FormatPercent(ld.InstructionThroughput), + fmt.Println(TableSeparator(78)) + fmt.Printf("%-5s %-18s %-16s %-16s %-16s\n", + "Record", "Instr Throughput", "Int & Complex", "F32 Limiter", "L1 Cache") + fmt.Println(TableSeparator(78)) + for _, ld := range rows { + fmt.Printf("%-5d %17s %15s %15s %15s\n", + ld.EncoderIndex, FormatPercent(ld.InstructionThroughput), FormatPercent(ld.IntegerComplex), FormatPercent(ld.F32Limiter), FormatPercent(ld.L1Cache)) } - fmt.Println("\nNote: Limiter percentages indicate bottleneck sources (higher = more constrained)") + fmt.Println("\nNote: Values are heuristic candidates, not source-backed bottleneck measurements.") + fmt.Println("Higher values mean more constrained only if the candidate field mapping is correct.") + if nonzero > len(rows) { + fmt.Printf("Use --limit %d or higher to show all non-zero rows.\n", nonzero) + } } } return nil } +// profilerFunctionRows aggregates dispatches by function name and ranks them by +// span, descending. The rows are KernelTiming values so that this table shares +// the low-sample marker and the --min-calls filter with the timing command +// instead of restating either. +func profilerFunctionRows(dispatches []counter.DispatchInfo, totalSpanUs int) []*gputrace.KernelTiming { + byName := make(map[string]*gputrace.KernelTiming) + var rows []*gputrace.KernelTiming + for _, d := range dispatches { + name := d.DisplayName() + kt, ok := byName[name] + if !ok { + kt = &gputrace.KernelTiming{Name: name} + byName[name] = kt + rows = append(rows, kt) + } + kt.InvocationCount++ + kt.TotalDuration += time.Duration(d.DurationUs) * time.Microsecond + } + for _, kt := range rows { + if totalSpanUs > 0 { + kt.PercentOfTotal = float64(kt.TotalDuration.Microseconds()) / float64(totalSpanUs) * 100 + } + } + sort.SliceStable(rows, func(i, j int) bool { + return rows[i].TotalDuration > rows[j].TotalDuration + }) + return rows +} + +// formatProfilerFunctionCalls renders the cost-ranked function table. minCalls +// filters the table only; the JSON and benchfmt outputs are built from the +// unfiltered dispatches, because a filtered export is a partial file that reads +// as a complete one once the flag is forgotten. Filtering never reorders: the +// surviving rows keep their ranking, and the note says the shares no longer sum +// to the whole. +func formatProfilerFunctionCalls(rows []*gputrace.KernelTiming, minCalls int, cumulativeOffsets bool) string { + shown, dropped := gputrace.FilterMinCalls(rows, minCalls) + + var out strings.Builder + out.WriteString(Colorize("Function Calls", ColorBold) + "\n") + out.WriteString(TableSeparator(80) + "\n") + fmt.Fprintf(&out, "%-50s %8s %10s %10s\n", "Function", "Calls", "Span(us)", "Span Share") + out.WriteString(TableSeparator(80) + "\n") + for _, kt := range shown { + marker := "" + if kt.IsLowSample() { + marker = gputrace.LowSampleMarker + } + fmt.Fprintf(&out, "%-50s %8s %10s %7s%s\n", + kt.Name, + FormatCount(kt.InvocationCount), + FormatCount(int(kt.TotalDuration.Microseconds())), + FormatPercent(kt.PercentOfTotal), + marker) + } + out.WriteString(gputrace.LowSampleFootnote(shown)) + out.WriteString(gputrace.MinCallsNote(minCalls, dropped, len(rows))) + if cumulativeOffsets { + out.WriteString("Attribution note: span values are cumulative-offset deltas and may include boundary or gap time.\n") + } + return out.String() +} + +func selectLimiterRows(all []limiterMetrics, limit int) (rows []limiterMetrics, nonzero, zero int) { + for _, row := range all { + if limiterPeak(row) < 0.05 { + zero++ + continue + } + rows = append(rows, row) + } + nonzero = len(rows) + sort.SliceStable(rows, func(i, j int) bool { + return limiterPeak(rows[i]) > limiterPeak(rows[j]) + }) + if limit > 0 && len(rows) > limit { + rows = rows[:limit] + } + return rows, nonzero, zero +} + +func limiterPeak(row limiterMetrics) float64 { + return max(row.InstructionThroughput, row.IntegerComplex, row.F32Limiter, row.L1Cache) +} + +func dispatchedFunctionNames(dispatches []counter.DispatchInfo) []string { + seen := make(map[string]bool) + var names []string + for _, dispatch := range dispatches { + name := dispatch.DisplayName() + if !seen[name] { + seen[name] = true + names = append(names, name) + } + } + return names +} + +func dispatchedPipelines(pipelines []counter.PipelineStats, dispatches []counter.DispatchInfo) []counter.PipelineStats { + ids := make(map[int]bool) + names := make(map[string]bool) + for _, dispatch := range dispatches { + if dispatch.PipelineID != 0 { + ids[dispatch.PipelineID] = true + } + if dispatch.FunctionName != "" { + names[dispatch.FunctionName] = true + } + } + var dispatched []counter.PipelineStats + for _, pipeline := range pipelines { + if ids[pipeline.PipelineID] || names[pipeline.FunctionName] { + dispatched = append(dispatched, pipeline) + } + } + return dispatched +} + func writeProfilerJSON(w io.Writer, output ProfilerOutputStats) error { enc := json.NewEncoder(w) enc.SetIndent("", " ") @@ -436,7 +609,6 @@ func writeProfilerJSON(w io.Writer, output ProfilerOutputStats) error { // limiterMetrics holds extracted performance limiter values per encoder. type limiterMetrics struct { EncoderIndex int - OccupancyManager float64 InstructionThroughput float64 IntegerComplex float64 F32Limiter float64 @@ -507,9 +679,6 @@ func extractLimiterData(profilerDir string) []limiterMetrics { // Map extracted values to limiter types (heuristic based on value ranges) for _, val := range limiters { switch { - case val >= 50 && val <= 100 && ld.OccupancyManager == 0: - // Occupancy Manager typically 50-80% - ld.OccupancyManager = val case val >= 0.01 && val <= 5 && ld.InstructionThroughput == 0: // Instruction throughput limiter (small %) ld.InstructionThroughput = val diff --git a/cmd/gputrace/cmd/profiler_compile.go b/cmd/gputrace/cmd/profiler_compile.go new file mode 100644 index 00000000..af7baf28 --- /dev/null +++ b/cmd/gputrace/cmd/profiler_compile.go @@ -0,0 +1,232 @@ +package cmd + +import ( + "fmt" + "io" + "strings" + + "github.com/tmc/gputrace/internal/counter" +) + +// compileSummary aggregates shader compilation across a capture's pipelines. +// +// Compilation runs on the host before the GPU executes anything, so it lands in +// no dispatch duration, no encoder span, and no execution cost. A run that +// compiles inside its measured window pays a price none of those numbers +// report. That is the whole reason for printing this beside them: a warmup +// either moved compilation out of the window or it did not, and until now the +// only way to check was to export a Perfetto trace and read the arguments. +type compileSummary struct { + // Timed is the number of pipelines carrying a compilation time. It is + // reported against the pipeline count because a partial denominator is + // the usual reason a total looks too small. + Timed int + Pipelines int + TotalMs float64 + SlowestMs float64 + SlowestName string + + // Cached and Compiled count pipelines by "Function was cached". They come + // from a separate dictionary than the times above and are frequently + // absent when the times are present, so they are counted, not derived. + // + // NoCacheRecord is the third bucket: a pipeline that compiled but carries + // no cache flag, because the Compile Performance dictionary is missing or + // does not hold the field. It is counted rather than left as Timed minus + // the other two, since a reader who does that subtraction in their head + // reads the remainder as a cache miss. On one trace 13 cached and 0 + // compiled against 14 pipelines is one absent record, not one miss. + Cached int + Compiled int + NoCacheRecord int + + // Phases sums the compiler phase timings that were recorded as + // non-negative. The archive uses -1 for a phase it did not measure, which + // is why a plain sum would be wrong; NegativePhases counts what was + // dropped so an unexpectedly small total is visible as such. + Phases []compilePhase + NegativePhases int +} + +type compilePhase struct { + Name string + NS int64 + N int +} + +// summarizeCompilation collects compilation timing from pipeline statistics. +// It returns nil when nothing in the capture recorded any, which is the normal +// case for a trace whose shaders were all warm. +func summarizeCompilation(pipelines []counter.PipelineStats) *compileSummary { + s := &compileSummary{Pipelines: len(pipelines)} + phaseNS := map[string]*compilePhase{} + order := []string{ + "translator", "optimization", "backend", + "compiler total", "driver total", "synchronous service", + } + add := func(name string, v *int64) { + if v == nil { + return + } + if *v < 0 { + s.NegativePhases++ + return + } + p := phaseNS[name] + if p == nil { + p = &compilePhase{Name: name} + phaseNS[name] = p + } + p.NS += *v + p.N++ + } + for i := range pipelines { + p := &pipelines[i] + timed := p.CompilationTimeMs > 0 || p.HasRecordedStatistic("Compilation time in milliseconds") + if timed { + s.Timed++ + s.TotalMs += p.CompilationTimeMs + if p.CompilationTimeMs > s.SlowestMs { + s.SlowestMs = p.CompilationTimeMs + s.SlowestName = p.DisplayName() + } + } + cp := p.CompilePerformance + switch { + case cp == nil || cp.FunctionWasCached == nil: + if timed { + s.NoCacheRecord++ + } + case *cp.FunctionWasCached: + s.Cached++ + default: + s.Compiled++ + } + if cp == nil { + continue + } + add("translator", cp.CompilerTranslatorNanoseconds) + add("optimization", cp.CompilerOptimizationNanoseconds) + add("backend", cp.CompilerBackendNanoseconds) + add("compiler total", cp.CompilerTotalNanoseconds) + add("driver total", cp.DriverTotalNanoseconds) + add("synchronous service", cp.SynchronousServiceNanoseconds) + } + for _, name := range order { + if p := phaseNS[name]; p != nil { + s.Phases = append(s.Phases, *p) + } + } + if s.Timed == 0 && s.Cached == 0 && s.Compiled == 0 && len(s.Phases) == 0 && s.NegativePhases == 0 { + return nil + } + return s +} + +// writeCompileSummary prints the shader compilation section. +func writeCompileSummary(w io.Writer, s *compileSummary) { + if s == nil { + return + } + fmt.Fprintln(w) + fmt.Fprintln(w, Colorize("Shader Compilation", ColorBold)) + fmt.Fprintln(w, TableSeparator(60)) + if s.Timed > 0 { + fmt.Fprintf(w, " Total Compile Time: %.3f ms across %d of %d %s\n", + s.TotalMs, s.Timed, s.Pipelines, Pluralize(s.Pipelines, "pipeline", "pipelines")) + if s.SlowestName != "" { + fmt.Fprintf(w, " Slowest: %.3f ms %s\n", s.SlowestMs, s.SlowestName) + } + } else { + fmt.Fprintf(w, " Total Compile Time: (no pipeline recorded one)\n") + } + switch { + case s.Cached > 0 || s.Compiled > 0: + line := fmt.Sprintf("%d cached, %d compiled", s.Cached, s.Compiled) + // Named rather than left as the remainder of the subtraction. A + // pipeline with no cache record did not miss the cache. + if s.NoCacheRecord > 0 { + line += fmt.Sprintf(", %d with no cache record", s.NoCacheRecord) + } + fmt.Fprintf(w, " Compile Cache: %s\n", line) + default: + fmt.Fprintf(w, " Compile Cache: (not recorded in this trace)\n") + } + for _, p := range s.Phases { + fmt.Fprintf(w, " %-18s %s over %d %s\n", + compilePhaseLabel(p.Name), FormatDurationNs(uint64(p.NS)), p.N, + Pluralize(p.N, "pipeline", "pipelines")) + } + switch { + case len(s.Phases) == 0 && s.NegativePhases > 0: + // Every recorded phase is the -1 sentinel. Printing this only as an + // exclusion footnote reads as "no phase data in this section", when + // what happened is that the archive carried the fields and measured + // none of them. That is a fact about the capture, not about the + // section, and no trace on hand has ever recorded otherwise. + fmt.Fprintf(w, " Compiler Phases: none measured (all %d recorded as -1)\n", s.NegativePhases) + case s.NegativePhases > 0: + fmt.Fprintf(w, " %d phase %s recorded as -1 and are excluded; the archive uses -1 for a phase it did not measure.\n", + s.NegativePhases, Pluralize(s.NegativePhases, "timing", "timings")) + } + fmt.Fprintln(w, "Compilation is host-side work. It is in no dispatch, encoder, or execution-cost") + fmt.Fprintln(w, "number above, so a warmup that failed shows up here and nowhere else.") +} + +func compilePhaseLabel(name string) string { + switch name { + case "translator": + return "Translator:" + case "optimization": + return "Optimization:" + case "backend": + return "Backend:" + case "compiler total": + return "Compiler Total:" + case "driver total": + return "Driver Total:" + case "synchronous service": + return "Sync Service:" + } + return name + ":" +} + +// writePipelineCompileDetail prints one pipeline's compilation fields, for the +// detailed kernel listing. Absent fields are omitted rather than printed as +// zero, because zero is a real compile time and absent is not. +func writePipelineCompileDetail(w io.Writer, p *counter.PipelineStats) { + var parts []string + if p.CompilationTimeMs > 0 || p.HasRecordedStatistic("Compilation time in milliseconds") { + parts = append(parts, fmt.Sprintf("%.3f ms", p.CompilationTimeMs)) + } + if cp := p.CompilePerformance; cp != nil { + if cp.FunctionWasCached != nil { + if *cp.FunctionWasCached { + parts = append(parts, "cached") + } else { + parts = append(parts, "not cached") + } + } + for _, f := range []struct { + name string + value *int64 + }{ + {"translator", cp.CompilerTranslatorNanoseconds}, + {"optimization", cp.CompilerOptimizationNanoseconds}, + {"backend", cp.CompilerBackendNanoseconds}, + } { + if f.value == nil { + continue + } + if *f.value < 0 { + parts = append(parts, f.name+"=not recorded") + continue + } + parts = append(parts, fmt.Sprintf("%s=%s", f.name, FormatDurationNs(uint64(*f.value)))) + } + } + if len(parts) == 0 { + return + } + fmt.Fprintf(w, " Compile: %s\n", strings.Join(parts, ", ")) +} diff --git a/cmd/gputrace/cmd/profiler_compile_test.go b/cmd/gputrace/cmd/profiler_compile_test.go new file mode 100644 index 00000000..db11843a --- /dev/null +++ b/cmd/gputrace/cmd/profiler_compile_test.go @@ -0,0 +1,204 @@ +package cmd + +import ( + "bytes" + "strings" + "testing" + + "github.com/tmc/gputrace/internal/counter" +) + +func boolPtr(v bool) *bool { return &v } +func i64Ptr(v int64) *int64 { return &v } +func msPipe(name string, ms float64) counter.PipelineStats { + return counter.PipelineStats{FunctionName: name, CompilationTimeMs: ms} +} + +func TestSummarizeCompilation(t *testing.T) { + tests := []struct { + name string + pipelines []counter.PipelineStats + want *compileSummary + }{ + { + name: "no compilation data returns nil", + pipelines: []counter.PipelineStats{{FunctionName: "warm"}}, + want: nil, + }, + { + name: "times only", + pipelines: []counter.PipelineStats{msPipe("a", 8.262), msPipe("b", 11.172), {FunctionName: "c"}}, + want: &compileSummary{ + Timed: 2, Pipelines: 3, TotalMs: 19.434, + SlowestMs: 11.172, SlowestName: "b", + }, + }, + { + name: "cache flags counted, not derived", + pipelines: []counter.PipelineStats{ + {FunctionName: "a", CompilePerformance: &counter.PipelineCompilePerformance{FunctionWasCached: boolPtr(true)}}, + {FunctionName: "b", CompilePerformance: &counter.PipelineCompilePerformance{FunctionWasCached: boolPtr(false)}}, + {FunctionName: "c", CompilePerformance: &counter.PipelineCompilePerformance{}}, + }, + want: &compileSummary{Pipelines: 3, Cached: 1, Compiled: 1}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := summarizeCompilation(tt.pipelines) + if tt.want == nil { + if got != nil { + t.Fatalf("summarizeCompilation() = %+v, want nil", got) + } + return + } + if got == nil { + t.Fatal("summarizeCompilation() = nil, want a summary") + } + if got.Timed != tt.want.Timed || got.Pipelines != tt.want.Pipelines { + t.Errorf("timed/pipelines = %d/%d, want %d/%d", got.Timed, got.Pipelines, tt.want.Timed, tt.want.Pipelines) + } + if d := got.TotalMs - tt.want.TotalMs; d > 1e-9 || d < -1e-9 { + t.Errorf("TotalMs = %v, want %v", got.TotalMs, tt.want.TotalMs) + } + if got.SlowestName != tt.want.SlowestName { + t.Errorf("SlowestName = %q, want %q", got.SlowestName, tt.want.SlowestName) + } + if got.Cached != tt.want.Cached || got.Compiled != tt.want.Compiled { + t.Errorf("cached/compiled = %d/%d, want %d/%d", got.Cached, got.Compiled, tt.want.Cached, tt.want.Compiled) + } + }) + } +} + +// The archive writes -1 for a phase it did not measure. Summing that as a +// duration would report a total smaller than its own parts, so the sentinel is +// excluded and counted instead. +func TestSummarizeCompilationExcludesNegativePhases(t *testing.T) { + pipelines := []counter.PipelineStats{ + {FunctionName: "a", CompilePerformance: &counter.PipelineCompilePerformance{ + CompilerBackendNanoseconds: i64Ptr(400), + CompilerTotalNanoseconds: i64Ptr(-1), + }}, + {FunctionName: "b", CompilePerformance: &counter.PipelineCompilePerformance{ + CompilerBackendNanoseconds: i64Ptr(600), + }}, + } + s := summarizeCompilation(pipelines) + if s == nil { + t.Fatal("summarizeCompilation() = nil") + } + if s.NegativePhases != 1 { + t.Errorf("NegativePhases = %d, want 1", s.NegativePhases) + } + var backend *compilePhase + for i := range s.Phases { + if s.Phases[i].Name == "backend" { + backend = &s.Phases[i] + } + if s.Phases[i].Name == "compiler total" { + t.Errorf("a -1 phase became a reported total: %+v", s.Phases[i]) + } + } + if backend == nil { + t.Fatal("backend phase missing") + } + if backend.NS != 1000 || backend.N != 2 { + t.Errorf("backend = %d ns over %d, want 1000 over 2", backend.NS, backend.N) + } +} + +func TestWriteCompileSummarySaysWhenCacheIsUnrecorded(t *testing.T) { + var buf bytes.Buffer + writeCompileSummary(&buf, summarizeCompilation([]counter.PipelineStats{msPipe("a", 2.5)})) + out := buf.String() + if !strings.Contains(out, "not recorded in this trace") { + t.Errorf("an absent cache flag must say so rather than read as 0 cached:\n%s", out) + } + if !strings.Contains(out, "2.500 ms") { + t.Errorf("compile time missing from output:\n%s", out) + } +} + +func TestWriteCompileSummaryNilWritesNothing(t *testing.T) { + var buf bytes.Buffer + writeCompileSummary(&buf, nil) + if buf.Len() != 0 { + t.Errorf("a capture with no compilation data printed a section:\n%s", buf.String()) + } +} + +// Zero is a real compile time and absent is not, so the detail line omits +// fields rather than printing them as zero. +func TestWritePipelineCompileDetail(t *testing.T) { + var buf bytes.Buffer + writePipelineCompileDetail(&buf, &counter.PipelineStats{FunctionName: "a"}) + if buf.Len() != 0 { + t.Errorf("pipeline with no compile data printed a line: %q", buf.String()) + } + + buf.Reset() + writePipelineCompileDetail(&buf, &counter.PipelineStats{ + FunctionName: "a", + CompilationTimeMs: 1.25, + CompilePerformance: &counter.PipelineCompilePerformance{ + FunctionWasCached: boolPtr(false), + CompilerBackendNanoseconds: i64Ptr(-1), + }, + }) + out := buf.String() + for _, want := range []string{"1.250 ms", "not cached", "backend=not recorded"} { + if !strings.Contains(out, want) { + t.Errorf("detail line missing %q:\n%s", want, out) + } + } +} + +// A pipeline that compiled but carries no cache flag is a third bucket. On +// qwen25-05b-rotmask-warm-tokens2-4, 13 cached and 0 compiled against 14 timed +// pipelines is one absent Compile Performance dictionary, not one cache miss, +// and the reader must not be left to infer the difference by subtraction. +func TestSummarizeCompilationCountsMissingCacheRecords(t *testing.T) { + cached := true + pipelines := []counter.PipelineStats{ + {FunctionName: "a", CompilationTimeMs: 5, CompilePerformance: &counter.PipelineCompilePerformance{FunctionWasCached: &cached}}, + // Timed, but no dictionary at all: the real shape of the 14th pipeline. + {FunctionName: "v_copybfloat16bfloat16", CompilationTimeMs: 3.598}, + // A dictionary that omits the flag is the same finding. + {FunctionName: "c", CompilationTimeMs: 1, CompilePerformance: &counter.PipelineCompilePerformance{}}, + } + s := summarizeCompilation(pipelines) + if s == nil { + t.Fatal("summarizeCompilation() = nil") + } + if s.Cached != 1 || s.Compiled != 0 || s.NoCacheRecord != 2 { + t.Errorf("cached/compiled/no-record = %d/%d/%d, want 1/0/2", s.Cached, s.Compiled, s.NoCacheRecord) + } + var buf bytes.Buffer + writeCompileSummary(&buf, s) + if !strings.Contains(buf.String(), "2 with no cache record") { + t.Errorf("absent cache records are left to subtraction:\n%s", buf.String()) + } +} + +// Every phase timing on every trace available is the -1 sentinel. Reporting +// that only as an exclusion footnote reads as "this section has no phase +// data", when the archive carried the fields and measured none of them. +func TestWriteCompileSummarySaysWhenNoPhaseWasMeasured(t *testing.T) { + minusOne := int64(-1) + s := summarizeCompilation([]counter.PipelineStats{{ + FunctionName: "a", + CompilationTimeMs: 2, + CompilePerformance: &counter.PipelineCompilePerformance{ + CompilerBackendNanoseconds: &minusOne, + CompilerTotalNanoseconds: &minusOne, + }, + }}) + var buf bytes.Buffer + writeCompileSummary(&buf, s) + out := buf.String() + if !strings.Contains(out, "none measured (all 2 recorded as -1)") { + t.Errorf("a fully unmeasured phase set does not say so:\n%s", out) + } +} diff --git a/cmd/gputrace/cmd/profiler_input.go b/cmd/gputrace/cmd/profiler_input.go index 1f4c79d0..049a99c7 100644 --- a/cmd/gputrace/cmd/profiler_input.go +++ b/cmd/gputrace/cmd/profiler_input.go @@ -2,13 +2,16 @@ package cmd import ( "fmt" + "os" "github.com/tmc/gputrace/internal/counter" + "github.com/tmc/gputrace/internal/profilereplay" ) func loadProfilerStats(tracePath string) (string, *counter.StreamDataStats, error) { profilerDir := findProfilerDir(tracePath) if profilerDir == "" { + fmt.Fprint(os.Stderr, profileReplayHint(tracePath)) return "", nil, fmt.Errorf("no .gpuprofiler_raw directory found in %s", tracePath) } @@ -20,6 +23,26 @@ func loadProfilerStats(tracePath string) (string, *counter.StreamDataStats, erro return profilerDir, stats, nil } +// profileReplayHint returns advice for a trace that holds a capture but no +// performance data, or "" when there is no advice worth giving: a profiler-only +// export has nothing left to replay, so naming a command that would refuse the +// trace is worse than saying nothing. +// +// The Xcode route is the fallback rather than the recommendation. It drives the +// UI and takes minutes with the machine; the replay is headless and takes +// seconds. It is still what remains when MTLReplayer is not installed. +func profileReplayHint(tracePath string) string { + if profilereplay.Replayable(tracePath) != nil { + return "" + } + add := "gputrace profile-replay " + tracePath + if profilereplay.Available() != nil { + add = "gputrace xcode-profile run " + tracePath + } + return fmt.Sprintf("Note: %s holds a capture but no performance data.\n"+ + " Add it with: %s\n\n", tracePath, add) +} + func aggregateExecutionCost(profilerDir string, stats *counter.StreamDataStats) []counter.ExecutionCostByFunction { if stats == nil || len(stats.Pipelines) == 0 { return nil diff --git a/cmd/gputrace/cmd/profiler_input_test.go b/cmd/gputrace/cmd/profiler_input_test.go new file mode 100644 index 00000000..149e08fb --- /dev/null +++ b/cmd/gputrace/cmd/profiler_input_test.go @@ -0,0 +1,56 @@ +package cmd + +import ( + "os" + "path/filepath" + "strings" + "testing" + + "github.com/tmc/gputrace/internal/profilereplay" +) + +func TestProfileReplayHint(t *testing.T) { + if err := profilereplay.Available(); err != nil { + t.Skip(err) + } + tests := []struct { + name string + entries []string + want bool + }{ + {"capture without profiler data", []string{"capture"}, true}, + {"unsorted-capture without profiler data", []string{"unsorted-capture"}, true}, + // Advice that cannot work is worse than silence: a profiler-only + // bundle has no capture stream left to replay, so the hint would send + // the reader to a command that refuses the trace. + {"profiler-only", []string{"trace.gpuprofiler_raw"}, false}, + {"empty", nil, false}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + bundle := filepath.Join(t.TempDir(), "trace.gputrace") + if err := os.Mkdir(bundle, 0o755); err != nil { + t.Fatal(err) + } + for _, entry := range tt.entries { + path := filepath.Join(bundle, entry) + if filepath.Ext(entry) == ".gpuprofiler_raw" { + if err := os.Mkdir(path, 0o755); err != nil { + t.Fatal(err) + } + continue + } + if err := os.WriteFile(path, []byte("x"), 0o644); err != nil { + t.Fatal(err) + } + } + hint := profileReplayHint(bundle) + if got := hint != ""; got != tt.want { + t.Fatalf("profileReplayHint(%v) = %q, want hint = %v", tt.entries, hint, tt.want) + } + if tt.want && !strings.Contains(hint, "gputrace profile-replay "+bundle) { + t.Fatalf("hint does not name a runnable command: %q", hint) + } + }) + } +} diff --git a/cmd/gputrace/cmd/profiler_mincalls_test.go b/cmd/gputrace/cmd/profiler_mincalls_test.go new file mode 100644 index 00000000..c1ed129d --- /dev/null +++ b/cmd/gputrace/cmd/profiler_mincalls_test.go @@ -0,0 +1,182 @@ +package cmd + +import ( + "strings" + "testing" + + "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/counter" +) + +func profilerDispatches() []counter.DispatchInfo { + var out []counter.DispatchInfo + add := func(name string, n, us int) { + for range n { + out = append(out, counter.DispatchInfo{FunctionName: name, DurationUs: us}) + } + } + add("gather_frontbfloat16_uint32_int_2", 1, 487) + add("s_copyint32int32", 1, 364) + add("gemm_bfloat16", 96, 20) + return out +} + +func TestProfilerFunctionRowsRankBySpan(t *testing.T) { + rows := profilerFunctionRows(profilerDispatches(), 487+364+96*20) + want := []struct { + name string + calls int + }{ + {"gemm_bfloat16", 96}, + {"gather_frontbfloat16_uint32_int_2", 1}, + {"s_copyint32int32", 1}, + } + if len(rows) != len(want) { + t.Fatalf("rows = %d, want %d", len(rows), len(want)) + } + for i, w := range want { + if rows[i].Name != w.name || rows[i].InvocationCount != w.calls { + t.Errorf("row %d = %s/%d, want %s/%d", i, rows[i].Name, rows[i].InvocationCount, w.name, w.calls) + } + } + var sum float64 + for _, kt := range rows { + sum += kt.PercentOfTotal + } + if sum < 99.9 || sum > 100.1 { + t.Errorf("shares sum to %.2f%%, want 100%%", sum) + } +} + +// The profiler command prints its own cost-ranked table, so it needs the same +// single-dispatch marker the timing command grew. It shipped without one. +func TestProfilerFunctionCallsMarksSingleDispatchRows(t *testing.T) { + rows := profilerFunctionRows(profilerDispatches(), 487+364+96*20) + out := formatProfilerFunctionCalls(rows, 0, false) + + for _, tt := range []struct { + prefix string + marked bool + }{ + {"gather_frontbfloat16_uint32_int_2", true}, + {"s_copyint32int32", true}, + {"gemm_bfloat16", false}, + } { + line := tableLine(out, tt.prefix) + if line == "" { + t.Fatalf("no row for %s in:\n%s", tt.prefix, out) + } + if got := strings.HasSuffix(line, gputrace.LowSampleMarker); got != tt.marked { + t.Errorf("%s marked = %v, want %v (%q)", tt.prefix, got, tt.marked, line) + } + } + if !strings.Contains(out, "single dispatch (2 of 3)") { + t.Errorf("missing the marker footnote:\n%s", out) + } +} + +func TestProfilerFunctionCallsMinCalls(t *testing.T) { + rows := profilerFunctionRows(profilerDispatches(), 487+364+96*20) + tests := []struct { + name string + minCalls int + wantRows []string + wantNote string + }{ + { + name: "off by default", + minCalls: 0, + wantRows: []string{"gemm_bfloat16", "gather_frontbfloat16_uint32_int_2", "s_copyint32int32"}, + }, + { + name: "one keeps everything", + minCalls: 1, + wantRows: []string{"gemm_bfloat16", "gather_frontbfloat16_uint32_int_2", "s_copyint32int32"}, + }, + { + name: "two drops the single-dispatch rows", + minCalls: 2, + wantRows: []string{"gemm_bfloat16"}, + wantNote: "--min-calls 2 dropped 2 of 3 rows", + }, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + out := formatProfilerFunctionCalls(rows, tt.minCalls, false) + for _, name := range []string{"gemm_bfloat16", "gather_frontbfloat16_uint32_int_2", "s_copyint32int32"} { + want := false + for _, w := range tt.wantRows { + want = want || w == name + } + if got := tableLine(out, name) != ""; got != want { + t.Errorf("row %s present = %v, want %v", name, got, want) + } + } + if tt.wantNote == "" { + if strings.Contains(out, "--min-calls") { + t.Errorf("unfiltered table carries a filter note:\n%s", out) + } + return + } + if !strings.Contains(out, tt.wantNote) { + t.Errorf("missing %q in:\n%s", tt.wantNote, out) + } + if !strings.Contains(out, "no longer sum to 100%") { + t.Errorf("filtered table does not say the shares are partial:\n%s", out) + } + }) + } +} + +// The filter applies to the table only. JSON and benchfmt are built from the +// same dispatch rows, so the filter must not edit them. +func TestProfilerFunctionCallsLeavesRowsUnfiltered(t *testing.T) { + rows := profilerFunctionRows(profilerDispatches(), 487+364+96*20) + before := append([]*gputrace.KernelTiming(nil), rows...) + + formatProfilerFunctionCalls(rows, 2, false) + + if len(rows) != len(before) { + t.Fatalf("--min-calls mutated the shared rows: %d left, want %d", len(rows), len(before)) + } + for i := range rows { + if rows[i] != before[i] { + t.Errorf("row %d changed identity or order", i) + } + } +} + +// Filtering must not reorder: the header still claims a cost ranking. +func TestProfilerFunctionCallsKeepsRankOrder(t *testing.T) { + rows := profilerFunctionRows(profilerDispatches(), 487+364+96*20) + out := formatProfilerFunctionCalls(rows, 0, false) + gather := strings.Index(out, "gather_frontbfloat16_uint32_int_2") + copyIdx := strings.Index(out, "s_copyint32int32") + gemm := strings.Index(out, "gemm_bfloat16") + if !(gemm < gather && gather < copyIdx) { + t.Errorf("rows are not in span order (gemm=%d gather=%d copy=%d):\n%s", gemm, gather, copyIdx, out) + } +} + +// The reported symptom was "unknown flag", so pin the registration itself. +func TestProfilerCommandRegistersMinCalls(t *testing.T) { + opts := &profilerOptions{limit: 20} + cmd := newProfilerCommand(opts) + f := cmd.Flags().Lookup("min-calls") + if f == nil { + t.Fatal("profiler has no --min-calls flag") + } + if f.DefValue != "0" { + t.Errorf("--min-calls defaults to %q, want it off at 0", f.DefValue) + } +} + +// tableLine returns the rendered row starting with prefix, or "". +func tableLine(out, prefix string) string { + for _, line := range strings.Split(out, "\n") { + if strings.HasPrefix(line, prefix) { + return strings.TrimRight(line, " ") + } + } + return "" +} diff --git a/cmd/gputrace/cmd/profiler_test.go b/cmd/gputrace/cmd/profiler_test.go index dd8616ee..ce340312 100644 --- a/cmd/gputrace/cmd/profiler_test.go +++ b/cmd/gputrace/cmd/profiler_test.go @@ -74,3 +74,66 @@ func TestWriteProfilerJSONUsesCommandOutput(t *testing.T) { t.Fatalf("profiler JSON output changed execution_cost shape:\n%s", out.String()) } } + +func TestSelectLimiterRowsSuppressesZerosAndHonorsLimit(t *testing.T) { + all := []limiterMetrics{ + {EncoderIndex: 1}, + {EncoderIndex: 2, F32Limiter: 0.04}, + {EncoderIndex: 3, L1Cache: 10}, + {EncoderIndex: 4, IntegerComplex: 20}, + {EncoderIndex: 5, InstructionThroughput: 5}, + } + + rows, nonzero, zero := selectLimiterRows(all, 2) + if nonzero != 3 || zero != 2 { + t.Fatalf("nonzero=%d zero=%d, want 3 and 2", nonzero, zero) + } + if len(rows) != 2 { + t.Fatalf("rows=%d, want 2", len(rows)) + } + if rows[0].EncoderIndex != 4 || rows[1].EncoderIndex != 3 { + t.Fatalf("rows=%+v, want descending peak limiter", rows) + } +} + +func TestProfilerLimitFlagDefaultsToTwenty(t *testing.T) { + opts := &profilerOptions{limit: 20} + cmd := newProfilerCommand(opts) + if got := cmd.Flag("limit").DefValue; got != "20" { + t.Fatalf("--limit default=%q, want 20", got) + } +} + +func TestDispatchedFunctionNames(t *testing.T) { + dispatches := []counter.DispatchInfo{ + {PipelineID: 7, FunctionName: "used"}, + {PipelineID: 7, FunctionName: "used"}, + {PipelineID: 9}, + } + got := dispatchedFunctionNames(dispatches) + want := []string{"used", "(pipeline_9)"} + if len(got) != len(want) { + t.Fatalf("dispatchedFunctionNames() = %q, want %q", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Fatalf("dispatchedFunctionNames()[%d] = %q, want %q", i, got[i], want[i]) + } + } +} + +func TestDispatchedPipelines(t *testing.T) { + pipelines := []counter.PipelineStats{ + {PipelineID: 1, FunctionName: "unused"}, + {PipelineID: 2, FunctionName: "used_by_id"}, + {PipelineID: 3, FunctionName: "used_by_name"}, + } + dispatches := []counter.DispatchInfo{ + {PipelineID: 2}, + {FunctionName: "used_by_name"}, + } + got := dispatchedPipelines(pipelines, dispatches) + if len(got) != 2 || got[0].PipelineID != 2 || got[1].PipelineID != 3 { + t.Fatalf("dispatchedPipelines() = %+v, want pipeline IDs 2 and 3", got) + } +} diff --git a/cmd/gputrace/cmd/remaining_output_test.go b/cmd/gputrace/cmd/remaining_output_test.go new file mode 100644 index 00000000..ce106cb0 --- /dev/null +++ b/cmd/gputrace/cmd/remaining_output_test.go @@ -0,0 +1,73 @@ +package cmd + +import ( + "bytes" + "testing" + + "github.com/spf13/cobra" +) + +func TestMTLBHumanCommandsUseCommandOutput(t *testing.T) { + tracePath := testCommandBuffersTracePath(t) + tests := []struct { + name string + run func(*cobra.Command) error + }{ + {name: "root", run: func(cmd *cobra.Command) error { + return runMTLB(cmd, []string{tracePath}, &mtlbOptions{}) + }}, + {name: "list", run: func(cmd *cobra.Command) error { + return runMTLBList(cmd, []string{tracePath}, &mtlbListOptions{}) + }}, + {name: "info", run: func(cmd *cobra.Command) error { + return runMTLBInfo(cmd, []string{tracePath}, &mtlbInfoOptions{}) + }}, + {name: "stats", run: func(cmd *cobra.Command) error { + return runMTLBStats(cmd, []string{tracePath}, &mtlbStatsOptions{}) + }}, + } + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + var out bytes.Buffer + cmd := &cobra.Command{} + cmd.SetOut(&out) + stdout, err := captureStdout(t, func() error { return tt.run(cmd) }) + if err != nil { + t.Fatalf("run: %v", err) + } + if stdout != "" { + t.Fatalf("os stdout = %q, want empty", stdout) + } + if out.Len() == 0 { + t.Fatal("command output is empty") + } + }) + } +} + +func TestBufferResourcesUsesCommandOutput(t *testing.T) { + tracePath := testCommandBuffersTracePath(t) + var out bytes.Buffer + cmd := &cobra.Command{} + cmd.SetOut(&out) + opts := &buffersCommandOptions{ + sort: "size", + format: "table", + inspectBytes: 256, + inspectFormat: "hex", + resources: true, + limit: defaultHumanLimit, + } + stdout, err := captureStdout(t, func() error { + return runBuffers(cmd, []string{tracePath}, opts) + }) + if err != nil { + t.Fatalf("runBuffers: %v", err) + } + if stdout != "" { + t.Fatalf("os stdout = %q, want empty", stdout) + } + if out.Len() == 0 { + t.Fatal("command output is empty") + } +} diff --git a/cmd/gputrace/cmd/replay_counters.go b/cmd/gputrace/cmd/replay_counters.go index 3c79733b..15b4b468 100644 --- a/cmd/gputrace/cmd/replay_counters.go +++ b/cmd/gputrace/cmd/replay_counters.go @@ -26,20 +26,35 @@ type replayCountersOptions struct { output string } +func replayCounterConfig(opts *replayCountersOptions) *gputrace.CounterSamplingConfig { + config := &gputrace.CounterSamplingConfig{ + EnabledCounterSets: opts.counterSets, + SampleAtEncoderBoundaries: opts.encoderBoundaries, + SampleAtDispatchBoundaries: opts.dispatchBoundaries, + UseBarriers: opts.useBarriers, + } + if len(config.EnabledCounterSets) == 0 { + config.EnabledCounterSets = []string{"timestamp", "stage_utilization", "statistics"} + } + return config +} + func newReplayCountersCommand(opts *replayCountersOptions) *cobra.Command { cmd := &cobra.Command{ Use: "replay-counters ", - Short: "Plan MTLCounterSampleBuffer sampling; real collection is disabled", + Short: "Collect or plan MTLCounterBuffer samples during replay", Hidden: true, Long: `Plan Metal performance counter sampling for trace replay. -IMPORTANT: This command is fail-closed for real replay counter collection. +On macOS with the metal build tag, this command replays through public Metal +and retains resolved counter bytes without guessing their hardware layout. +Other builds remain simulation-only. Current Behavior: - --simulate builds a sampling plan only - --simulate does not replay GPU work - - Running without --simulate fails closed before trace replay or GPU work - - No replay-time MTLCounterSampleBuffer collection is attempted + - Running without --simulate performs replay-time collection on macOS+metal + - Raw resolved bytes are retained; metric decoding remains explicit and gated Use this command to inspect: - Where counter samples would be taken (encoder/dispatch boundaries) @@ -52,10 +67,7 @@ Use profiler when you need existing profiler data: - No GPU execution required - Binary format undocumented (reverse engineering needed) -Current Status: -Replay-time counter collection currently fails closed until Metal API bindings -are connected and the replay path can collect counters safely. The planned -counter sets are: +Counter sets requested by the replay or simulation plan: - Timestamp counters (GPU cycles) - Stage utilization (vertex/fragment/compute) - Statistics (draw/dispatch counts) @@ -90,9 +102,8 @@ Examples: gputrace replay-counters trace.gputrace --simulate -o counters.json Implementation Status: - This command provides only a planning/simulation path today. Actual replay-time - GPU counter collection is intentionally unavailable and fails closed before - trace replay or GPU work. + Public Metal MTLCounterSampleBuffer collection is available on macOS+metal. + Private APS counters and unverified hardware-byte decoders remain separate. Related Commands: - gputrace profiler: Extract profiler timing data from .gpuprofiler_raw/streamData @@ -131,7 +142,7 @@ func runReplayCounters(cmd *cobra.Command, args []string, opts *replayCountersOp } if !opts.simulate { - return fmt.Errorf("real replay counter collection is unavailable without replay-time Metal bindings; rerun with --simulate to inspect the sampling plan") + return runReplayCountersReal(tracePath, opts) } // Open trace @@ -144,18 +155,7 @@ func runReplayCounters(cmd *cobra.Command, args []string, opts *replayCountersOp engine := gputrace.NewReplayEngine(trace) // Configure counter sampling - config := &gputrace.CounterSamplingConfig{ - EnabledCounterSets: opts.counterSets, - SampleAtEncoderBoundaries: opts.encoderBoundaries, - SampleAtDispatchBoundaries: opts.dispatchBoundaries, - UseBarriers: opts.useBarriers, - GPUFrequency: 0, // Auto-detect - } - - // Use defaults if no counter sets specified - if len(config.EnabledCounterSets) == 0 { - config.EnabledCounterSets = []string{"timestamp", "stage_utilization", "statistics"} - } + config := replayCounterConfig(opts) // Enable counter sampling if err := engine.EnableCounterSampling(config); err != nil { @@ -175,7 +175,8 @@ func runReplayCounters(cmd *cobra.Command, args []string, opts *replayCountersOp if opts.output != "" && isJSONOutput(opts.output) { data = simulation } else { - output = gputrace.FormatCounterSamplingSimulation(simulation) + output = "Mode: SIMULATION — NO GPU WORK EXECUTED; existing profiler data is not used\n\n" + output += gputrace.FormatCounterSamplingSimulation(simulation) } } else { // Perform full analysis with counter sampling @@ -229,7 +230,7 @@ func writeOutput(filename, textOutput string, jsonData interface{}) error { } if filename != "" { - fmt.Fprintf(os.Stderr, "✓ Written to: %s\n", filename) + fmt.Fprintf(os.Stderr, "Written: %s\n", filename) } return nil diff --git a/cmd/gputrace/cmd/replay_counters_darwin.go b/cmd/gputrace/cmd/replay_counters_darwin.go new file mode 100644 index 00000000..8d65e861 --- /dev/null +++ b/cmd/gputrace/cmd/replay_counters_darwin.go @@ -0,0 +1,61 @@ +//go:build darwin && metal + +package cmd + +import ( + "fmt" + + "github.com/tmc/gputrace" +) + +func replayCountersRealAvailable() bool { return true } + +func runReplayCountersReal(tracePath string, opts *replayCountersOptions) error { + trace, err := gputrace.Open(tracePath) + if err != nil { + return fmt.Errorf("open trace: %w", err) + } + defer trace.Close() + + engine, err := gputrace.NewMetalReplayEngine(trace) + if err != nil { + return fmt.Errorf("create Metal replay engine: %w", err) + } + defer engine.Close() + + config := replayCounterConfig(opts) + if len(opts.counterSets) == 0 { + // The generic planning names are not guaranteed to be device counter-set + // names. Timestamp is the only set verified on this host. + config.EnabledCounterSets = []string{"timestamp"} + } + sampler, err := engine.NewCounterSampler(config) + if err != nil { + return fmt.Errorf("create counter sampler: %w", err) + } + plan, err := engine.AnalyzeReplay() + if err != nil { + return fmt.Errorf("analyze replay: %w", err) + } + result, err := engine.ExecuteReplayPlanWithCounters(plan, sampler) + if err != nil { + return fmt.Errorf("execute replay with counters: %w", err) + } + + data := map[string]interface{}{ + "plan": plan, + "result": result, + "sample_count": len(sampler.Samples), + "samples": sampler.Samples, + "raw_data": sampler.RawData, + } + if opts.output != "" && isJSONOutput(opts.output) { + return writeOutput(opts.output, "", data) + } + output := gputrace.FormatMetalReplayResult(result) + output += fmt.Sprintf("Counter samples: %d\n", len(sampler.Samples)) + for name, raw := range sampler.RawData { + output += fmt.Sprintf(" %s: %d raw bytes\n", name, len(raw)) + } + return writeOutput(opts.output, output, nil) +} diff --git a/cmd/gputrace/cmd/replay_counters_stub.go b/cmd/gputrace/cmd/replay_counters_stub.go new file mode 100644 index 00000000..87b6bb38 --- /dev/null +++ b/cmd/gputrace/cmd/replay_counters_stub.go @@ -0,0 +1,11 @@ +//go:build !darwin || !metal + +package cmd + +import "fmt" + +func replayCountersRealAvailable() bool { return false } + +func runReplayCountersReal(_ string, _ *replayCountersOptions) error { + return fmt.Errorf("real replay counter collection requires macOS with the metal build tag; rerun with --simulate") +} diff --git a/cmd/gputrace/cmd/replay_counters_test.go b/cmd/gputrace/cmd/replay_counters_test.go index fe8b6cf2..5f846235 100644 --- a/cmd/gputrace/cmd/replay_counters_test.go +++ b/cmd/gputrace/cmd/replay_counters_test.go @@ -5,10 +5,16 @@ import ( "testing" ) -func TestReplayCountersFailsClosedBeforeOpeningTraceWithoutSimulate(t *testing.T) { +func TestReplayCountersRejectsInvalidTraceWithoutSimulate(t *testing.T) { err := runReplayCounters(nil, []string{t.TempDir()}, &replayCountersOptions{}) if err == nil { - t.Fatal("runReplayCounters succeeded without Metal bindings") + t.Fatal("runReplayCounters succeeded with an invalid trace path") + } + if replayCountersRealAvailable() { + if strings.Contains(err.Error(), "rerun with --simulate") { + t.Fatalf("real Metal build still reports simulation-only error: %q", err) + } + return } if !strings.Contains(err.Error(), "rerun with --simulate") { t.Fatalf("error %q does not mention --simulate", err) @@ -18,13 +24,13 @@ func TestReplayCountersFailsClosedBeforeOpeningTraceWithoutSimulate(t *testing.T } } -func TestReplayCountersHelpDocumentsSimulationGate(t *testing.T) { +func TestReplayCountersHelpDocumentsModes(t *testing.T) { help := replayCountersCmd.Long for _, want := range []string{ "--simulate builds a sampling plan only", "--simulate does not replay GPU work", - "fails closed before trace replay or GPU work", "replay-counters trace.gputrace --simulate", + "Raw resolved bytes are retained", } { if !strings.Contains(help, want) { t.Fatalf("replay-counters help does not contain %q", want) diff --git a/cmd/gputrace/cmd/replay_metal_darwin.go b/cmd/gputrace/cmd/replay_metal_darwin.go new file mode 100644 index 00000000..0562d8ba --- /dev/null +++ b/cmd/gputrace/cmd/replay_metal_darwin.go @@ -0,0 +1,87 @@ +//go:build darwin && metal + +package cmd + +import ( + "fmt" + + "github.com/spf13/cobra" + + "github.com/tmc/gputrace" +) + +type replayMetalOptions struct { + json bool + output string +} + +var replayMetalCmd = newReplayMetalCommand(&replayMetalOptions{}) + +func newReplayMetalCommand(opts *replayMetalOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "replay-metal ", + Short: "Execute a trace through the public Metal replay engine", + Long: `Execute supported trace commands through the public Metal replay engine. + +This command is available only in builds made with the metal build tag. A +successful result means command submission completed for the supported replay +plan; it does not validate replayed buffer contents against the capture. Any +unsupported command fails closed and may leave a partial execution count.`, + Args: cobra.ExactArgs(1), + SilenceUsage: true, + RunE: func(cmd *cobra.Command, args []string) error { + return runReplayMetal(cmd, args[0], opts) + }, + } + cmd.Flags().BoolVar(&opts.json, "json", false, "write the result as JSON") + cmd.Flags().StringVarP(&opts.output, "output", "o", "", "write the result to a file") + return cmd +} + +func init() { + rootCmd.AddCommand(replayMetalCmd) +} + +func runReplayMetal(_ *cobra.Command, path string, opts *replayMetalOptions) error { + if err := checkTraceFile(path); err != nil { + return err + } + trace, err := gputrace.Open(path) + if err != nil { + return fmt.Errorf("open trace: %w", err) + } + defer trace.Close() + + engine, err := gputrace.NewMetalReplayEngine(trace) + if err != nil { + return fmt.Errorf("create Metal replay engine: %w", err) + } + defer engine.Close() + + plan, err := engine.AnalyzeReplay() + if err != nil { + return fmt.Errorf("analyze replay: %w", err) + } + result, replayErr := engine.ExecuteReplayPlan(plan) + if result == nil { + if replayErr != nil { + return fmt.Errorf("execute replay: %w", replayErr) + } + return fmt.Errorf("execute replay: nil result") + } + + var output string + var data interface{} + if opts.json { + data = result + } else { + output = gputrace.FormatMetalReplayResult(result) + } + if err := writeOutput(opts.output, output, data); err != nil { + return err + } + if replayErr != nil { + return fmt.Errorf("execute replay: %w", replayErr) + } + return nil +} diff --git a/cmd/gputrace/cmd/residency.go b/cmd/gputrace/cmd/residency.go new file mode 100644 index 00000000..11e78063 --- /dev/null +++ b/cmd/gputrace/cmd/residency.go @@ -0,0 +1,134 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "io" + "sort" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/fmtutil" + "github.com/tmc/gputrace/internal/trace" +) + +type residencyOptions struct { + json bool +} + +var residencyCmd = newResidencyCommand(&residencyOptions{}) + +func newResidencyCommand(opts *residencyOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "residency ", + Short: "Report allocated footprint by storage mode and whether residency is explicit", + Long: `Report how a capture allocates memory and whether it manages residency. + +Two things are shown together because they are one finding seen from two +directions. An all-shared allocation profile and an uncommitted residency set +both mean the process is leaving placement and residency to the driver. + + - Allocated footprint per MTLStorageMode, from buffer-creation records. + - Counts of newResidencySet, requestResidency, and addResidencySet. + +What is not shown, because the capture format does not record it in any way +this decodes: which buffers belong to which residency set. There is therefore +no wired-bytes figure distinct from the allocated figure. When no residency set +is committed, every allocation is under the driver's automatic residency, and +the allocated total is the working upper bound on what can be made resident.`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runResidency(cmd, args, opts) + }, + } + cmd.Flags().BoolVar(&opts.json, "json", false, "Output in JSON format") + return cmd +} + +func init() { + rootCmd.AddCommand(residencyCmd) +} + +func runResidency(cmd *cobra.Command, args []string, opts *residencyOptions) error { + t, err := trace.Open(args[0]) + if err != nil { + return fmt.Errorf("failed to open trace: %w", err) + } + r, err := t.ResidencyReport() + if err != nil { + return err + } + w := cmd.OutOrStdout() + if opts.json { + enc := json.NewEncoder(w) + enc.SetIndent("", " ") + return enc.Encode(r) + } + writeResidencyReport(w, r) + return nil +} + +func writeResidencyReport(w io.Writer, r *trace.ResidencyReport) { + fmt.Fprintln(w, "Allocated footprint by storage mode") + if len(r.Storage) == 0 { + fmt.Fprintln(w, " (no buffer-creation records in this capture)") + } else { + fmt.Fprintf(w, " %-12s %8s %12s\n", "mode", "records", "bytes") + for _, f := range r.Storage { + fmt.Fprintf(w, " %-12s %8d %12s\n", f.Mode, f.Buffers, fmtutil.FormatBytes(int64(f.Bytes), 1)) + } + fmt.Fprintf(w, " %-12s %8d %12s\n", "total", r.Buffers, fmtutil.FormatBytes(int64(r.Bytes), 1)) + } + + fmt.Fprintln(w, "\nResidency") + fmt.Fprintf(w, " %-20s %6d\n", "newResidencySet", r.Residency.NewResidencySet) + fmt.Fprintf(w, " %-20s %6d\n", "requestResidency", r.Residency.RequestResidency) + fmt.Fprintf(w, " %-20s %6d\n", "addResidencySet", r.Residency.AddResidencySet) + switch { + case r.Residency.Explicit(): + fmt.Fprintln(w, " Residency is managed explicitly.") + case r.Residency.Any(): + fmt.Fprintln(w, " Residency is not managed explicitly; the driver's automatic") + fmt.Fprintln(w, " residency covers every allocation.") + default: + fmt.Fprintln(w, " No residency records were decoded, which is not the same as") + fmt.Fprintln(w, " observing that the program manages no residency.") + } + + if d := r.Disagreements(); len(d) > 0 { + modes := make([]string, 0, len(d)) + for mode := range d { + modes = append(modes, mode) + } + sort.Strings(modes) + fmt.Fprintln(w, "\nScanner disagreement") + fmt.Fprintf(w, " %-12s %10s %10s\n", "mode", "decoded", "scanned") + for _, mode := range modes { + fmt.Fprintf(w, " %-12s %10d %10d\n", mode, d[mode][0], d[mode][1]) + } + fmt.Fprintln(w, " Two independent scans of the same capture disagree, so neither") + fmt.Fprintln(w, " count above is trustworthy. Both are shown rather than one picked.") + } + + if f := r.Finding(); f != "" { + fmt.Fprintf(w, "\n%s\n", f) + } + + fmt.Fprintln(w, "\nWhat this does and does not measure") + fmt.Fprintln(w, " Buffer and residency records are found by scanning the whole capture for") + fmt.Fprintln(w, " record markers. That scan is independent of the dispatch decoding whose") + fmt.Fprintln(w, " coverage \"gputrace api-calls\" reports, so a low decoded-dispatch fraction") + fmt.Fprintln(w, " does not bound these counts. The limit is that the scan finds the record") + fmt.Fprintln(w, " shapes it knows; a shape it does not know is absent, not counted.") + fmt.Fprintln(w, " Residency-set membership is not decoded, so there is no wired-bytes figure") + fmt.Fprintln(w, " separate from the allocated one.") + fmt.Fprintln(w, " The counts above are buffer-creation RECORDS, not distinct buffer resources.") + fmt.Fprintln(w, " \"gputrace buffers\" counts resources and their aliases instead, and reports a") + fmt.Fprintln(w, " number roughly 10x smaller with reconciling bytes. Neither is wrong; they") + fmt.Fprintln(w, " answer different questions under the same word. The cross-check above shares") + fmt.Fprintln(w, " this command's record scan, so it catches a mis-walked stream and not a") + fmt.Fprintln(w, " misread of what the records represent.") + if r.Unsized > 0 { + fmt.Fprintf(w, " %d buffer record(s) carried a zero length, which Metal does not create;\n", r.Unsized) + fmt.Fprintln(w, " the byte totals understate by whatever those held.") + } +} diff --git a/cmd/gputrace/cmd/residency_test.go b/cmd/gputrace/cmd/residency_test.go new file mode 100644 index 00000000..413fbebe --- /dev/null +++ b/cmd/gputrace/cmd/residency_test.go @@ -0,0 +1,94 @@ +package cmd + +import ( + "bytes" + "strings" + "testing" + + "github.com/tmc/gputrace/internal/trace" +) + +func TestWriteResidencyReport(t *testing.T) { + r := &trace.ResidencyReport{ + Storage: []trace.StorageFootprint{ + {Mode: "shared", Buffers: 842, Bytes: 13 << 30}, + {Mode: "private", Buffers: 16, Bytes: 1 << 30}, + }, + Buffers: 858, + Bytes: 14 << 30, + Residency: trace.ResidencyCalls{NewResidencySet: 1}, + } + var b bytes.Buffer + writeResidencyReport(&b, r) + got := b.String() + + for _, want := range []string{ + "shared", "private", "842", "total", + "newResidencySet", "requestResidency", "addResidencySet", + "not managed explicitly", + } { + if !strings.Contains(got, want) { + t.Errorf("report is missing %q:\n%s", want, got) + } + } + // The reason this command exists is that a low decoded-dispatch fraction + // was being read as a limit on these counts, which it is not. Saying so is + // part of the output, not a comment in the source. + if !strings.Contains(got, "does not bound these counts") { + t.Errorf("report does not disclaim the dispatch-coverage confusion:\n%s", got) + } + // There is no residency-set membership in the capture, so a wired-bytes + // figure would be invented. The output has to say that rather than let a + // reader assume the allocated total is a wired total. + if !strings.Contains(got, "no wired-bytes figure") { + t.Errorf("report does not state that wired bytes are not derivable:\n%s", got) + } +} + +// A capture with no buffer records must read as an absent measurement, not as a +// program that allocates nothing. +func TestWriteResidencyReportEmpty(t *testing.T) { + var b bytes.Buffer + writeResidencyReport(&b, &trace.ResidencyReport{}) + got := b.String() + if !strings.Contains(got, "no buffer-creation records") { + t.Errorf("empty report does not disclaim:\n%s", got) + } + if strings.Contains(got, "total") { + t.Errorf("empty report printed a total row:\n%s", got) + } +} + +func TestWriteResidencyReportUnsized(t *testing.T) { + var b bytes.Buffer + writeResidencyReport(&b, &trace.ResidencyReport{ + Storage: []trace.StorageFootprint{{Mode: "shared", Buffers: 2, Bytes: 128}}, + Buffers: 2, Bytes: 128, Unsized: 1, + }) + if got := b.String(); !strings.Contains(got, "understate") { + t.Errorf("unsized records were not disclosed:\n%s", got) + } +} + +func TestResidencyCommandRegistered(t *testing.T) { + for _, c := range rootCmd.Commands() { + if c.Name() == "residency" { + return + } + } + t.Error("residency command is not registered on the root command") +} + +// The empty report must not assert that the driver is managing residency. With +// nothing decoded, that claim has no evidence behind it. +func TestWriteResidencyReportEmptyDoesNotClaimDriverResidency(t *testing.T) { + var b bytes.Buffer + writeResidencyReport(&b, &trace.ResidencyReport{}) + got := b.String() + if strings.Contains(got, "driver's automatic") { + t.Errorf("empty report claims the driver manages residency:\n%s", got) + } + if !strings.Contains(got, "not the same as") { + t.Errorf("empty report does not distinguish absent records from absent residency:\n%s", got) + } +} diff --git a/cmd/gputrace/cmd/root.go b/cmd/gputrace/cmd/root.go index fb6e5ad1..48527e93 100644 --- a/cmd/gputrace/cmd/root.go +++ b/cmd/gputrace/cmd/root.go @@ -2,8 +2,12 @@ package cmd import ( + "context" + "errors" "fmt" "os" + "os/signal" + "syscall" "github.com/spf13/cobra" ) @@ -16,7 +20,8 @@ var rootCmd = &cobra.Command{ Command Groups: Trace Overview: - stats - Comprehensive trace statistics + summary - One-screen structure, timing, and evidence report + stats - Capture structure, resources, and profiler availability api-calls - API call sequences dump - Raw API call dump @@ -26,9 +31,10 @@ Kernel & Shader Analysis: shader-source - Source-level performance attribution Timing & Profiling: - timing - Timing metrics export - profiler - GPU profiler data extraction - pprof - pprof format export + timing - Timing metrics with measured/estimated provenance + profiler - Profiler spans, active time, dispatches, and pipelines + pprof - pprof format export (Metal traces and CUDA captures) + dot-pprof - CUDA-graph DOT dumps as a pprof structure profile correlate - Correlate timing with hardware metrics Command Buffers & Encoders: @@ -42,31 +48,91 @@ Buffer Analysis: Visualization & Export: timeline - Text timeline and Chrome/Perfetto export + host-receipt - Bind measured host intervals to live GPU timing graph - Graph visualization tree - Execution tree view diff - Compare two traces - insights - Actionable performance insights + brief - Compact comparison brief + insights - Diagnostic performance hypotheses Capture & Automation: + capture - Run a Metal workload under the capture interposer + profile-replay - Replay a capture under the profiler to add timing xcode-profile - Xcode GPU profiler automation xcode-bindings - Inspect private Xcode GTShaderProfiler bindings xcode-parity - Audit Xcode metric parity for a trace Utilities: mtlb - Metal Library Binary inspection - clear-buffers - Zero out buffers to reduce trace size + clear-buffers - Destructively zero captured buffers + nvidia - Report NVIDIA GPU status via NVML (Linux) + cupti - Convert CUPTI activity captures to Perfetto traces (Linux) version - Print gputrace build version +Hidden commands are runnable but omitted from Available Commands because their +output is experimental or heuristic: counters, replay-counters, dependencies, +fences, export-counters, perfcounters-validate. + For more information about a specific command: gputrace [command] --help`, } // Execute runs the root command. func Execute() error { - return rootCmd.Execute() + // Cancel the command context on interrupt so commands that launch external + // processes can tear them down. Without this, Go's default SIGINT handling + // ends the process before any Go code runs, and profile-replay leaves the + // MTLReplayer it started behind: LaunchServices, not gputrace, is that + // process's parent, so nothing else reaps it. + // + // Unregistering on the first signal is what keeps a second interrupt able + // to kill gputrace outright. NotifyContext on its own leaves the handler + // installed after it fires, so every later interrupt is swallowed too, and + // a command that does not watch its context becomes unkillable by Ctrl+C -- + // worse than the leak this exists to fix. Verified both ways with a + // sleeping test binary. + ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM) + defer stop() + go func() { + <-ctx.Done() + stop() + }() + return rootCmd.ExecuteContext(ctx) +} + +type exitCoder interface { + error + exitCode() int +} + +// ExitCode returns the process exit code for err. If err implements exitCode(), +// that value is returned; otherwise 1 is returned. +func ExitCode(err error) int { + var coder exitCoder + if errors.As(err, &coder) { + return coder.exitCode() + } + return 1 +} + +type alreadyReportedError interface { + error + alreadyReported() +} + +// ErrorAlreadyReported reports whether err has already been written as +// structured command output. +func ErrorAlreadyReported(err error) bool { + var reported alreadyReportedError + return errors.As(err, &reported) } func init() { + // The command entry point prints ordinary errors. Some commands write a + // structured JSON error before returning, so Cobra must not independently + // print either the error or command usage. + rootCmd.SilenceErrors = true + rootCmd.SilenceUsage = true rootCmd.CompletionOptions.DisableDefaultCmd = true initColorFlag(rootCmd) } diff --git a/cmd/gputrace/cmd/shader_metrics_private.go b/cmd/gputrace/cmd/shader_metrics_private.go new file mode 100644 index 00000000..7e0546c3 --- /dev/null +++ b/cmd/gputrace/cmd/shader_metrics_private.go @@ -0,0 +1,32 @@ +//go:build darwin && gputrace_private_bindings + +package cmd + +import ( + "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/counter" + "github.com/tmc/gputrace/internal/shader" +) + +func applySourceBackedShaderMetrics(streamPath string, stats *counter.StreamDataStats, report *gputrace.ShaderMetricsReport) error { + if stats == nil || report == nil { + return nil + } + // Index the shaders by name first: the nested scan was + // len(Shaders)*len(Pipelines) string compares. Duplicate names keep the + // last shader in report order, which is what the nested loops did. + byName := make(map[string]*shader.ShaderMetrics, len(report.Shaders)) + for _, metric := range report.Shaders { + if metric == nil { + continue + } + byName[metric.Name] = metric + } + byPipeline := make(map[int]*shader.ShaderMetrics, len(stats.Pipelines)) + for _, pipeline := range stats.Pipelines { + if metric, ok := byName[pipeline.FunctionName]; ok { + byPipeline[pipeline.PipelineID] = metric + } + } + return shader.ApplyPipelineShaderMetricsFromStreamData(byPipeline, streamPath) +} diff --git a/cmd/gputrace/cmd/shader_metrics_stub.go b/cmd/gputrace/cmd/shader_metrics_stub.go new file mode 100644 index 00000000..93b91694 --- /dev/null +++ b/cmd/gputrace/cmd/shader_metrics_stub.go @@ -0,0 +1,12 @@ +//go:build !darwin || !gputrace_private_bindings + +package cmd + +import ( + "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/counter" +) + +func applySourceBackedShaderMetrics(_ string, _ *counter.StreamDataStats, _ *gputrace.ShaderMetricsReport) error { + return nil +} diff --git a/cmd/gputrace/cmd/shader_source.go b/cmd/gputrace/cmd/shader_source.go index e6e13590..715f252b 100644 --- a/cmd/gputrace/cmd/shader_source.go +++ b/cmd/gputrace/cmd/shader_source.go @@ -4,6 +4,7 @@ import ( "encoding/json" "fmt" "io" + "strings" "github.com/spf13/cobra" @@ -38,7 +39,7 @@ Features: - Multiple output formats (text, HTML, JSON) The analysis uses: - - Shader performance metrics from trace (timing, invocations, occupancy) + - Shader performance metrics from trace (timing, invocations) - Metal shader source files (.metal) from indexed locations - Static analysis to estimate relative cost of each line - Heuristics to classify instruction types @@ -126,9 +127,19 @@ func runShaderSource(cmd *cobra.Command, args []string, opts *shaderSourceOption switch format { case "text": output = gputrace.FormatShaderSourceAttribution(attribution, opts.hints) + source := "" + approximate := false + if attribution.Metrics == nil || attribution.Metrics.TimingSource == "" { + source = "" + } else { + source = attribution.Metrics.TimingSource + approximate = attribution.Metrics.TimingApprox + } + output = formatShaderSourceProvenance(source, approximate) + output case "html": output = gputrace.FormatShaderSourceAttributionHTML(attribution) + output = strings.Replace(output, "", "\n

Attribution method: static per-line cost heuristic; percentages are estimates, not line-level measurements.

", 1) case "json": data = attribution @@ -154,12 +165,24 @@ func runShaderSource(cmd *cobra.Command, args []string, opts *shaderSourceOption } } if opts.output != "" { - fmt.Fprintf(cmd.ErrOrStderr(), "✓ Written to: %s\n", opts.output) + fmt.Fprintf(cmd.ErrOrStderr(), "Written: %s\n", opts.output) } return nil } +func formatShaderSourceProvenance(source string, approximate bool) string { + note := "Attribution Method: static per-line cost heuristic (estimated, not measured per line)\n" + if source == "" { + return note + "Aggregate Timing Source: unavailable\n\n" + } + kind := "measured" + if approximate { + kind = "approximate" + } + return fmt.Sprintf("%sAggregate Timing Source: %s (%s)\n\n", note, source, kind) +} + func validateShaderSourceFormat(format string) (string, error) { switch format { case "text", "html", "json": diff --git a/cmd/gputrace/cmd/shader_source_test.go b/cmd/gputrace/cmd/shader_source_test.go index 56984fb9..e9fbabce 100644 --- a/cmd/gputrace/cmd/shader_source_test.go +++ b/cmd/gputrace/cmd/shader_source_test.go @@ -2,9 +2,22 @@ package cmd import ( "path/filepath" + "strings" "testing" ) +func TestFormatShaderSourceProvenance(t *testing.T) { + got := formatShaderSourceProvenance("synthetic fallback", true) + for _, want := range []string{ + "static per-line cost heuristic", + "synthetic fallback (approximate)", + } { + if !strings.Contains(got, want) { + t.Fatalf("provenance missing %q:\n%s", want, got) + } + } +} + func TestValidateShaderSourceFormatAcceptsKnownValues(t *testing.T) { for _, format := range []string{"text", "html", "json"} { t.Run(format, func(t *testing.T) { diff --git a/cmd/gputrace/cmd/shaders.go b/cmd/gputrace/cmd/shaders.go index f9f33d14..b7629be3 100644 --- a/cmd/gputrace/cmd/shaders.go +++ b/cmd/gputrace/cmd/shaders.go @@ -11,6 +11,7 @@ import ( "github.com/tmc/gputrace" "github.com/tmc/gputrace/internal/counter" + "github.com/tmc/gputrace/internal/profilerraw" ) var shadersCmd = newShadersCommand(&shadersOptions{ @@ -18,10 +19,11 @@ var shadersCmd = newShadersCommand(&shadersOptions{ }) type shadersOptions struct { - verbose bool - estimate bool - format string - all bool + verbose bool + estimate bool + format string + all bool + xcodeCost bool } func newShadersCommand(opts *shadersOptions) *cobra.Command { @@ -31,21 +33,39 @@ func newShadersCommand(opts *shadersOptions) *cobra.Command { Long: `Display shader/kernel performance statistics. By default shows a simple two-column output: - - Cost % (percentage of total GPU time) + - Share % (SIMD-group share for full traces; dispatch-span share for profiler-only traces) - Shader name +Use --xcode-cost to run Xcode's private stream-data processor and show the +pipeline compute-time share from its All Shaders table. This is slower than the +default parser and requires the matching Xcode framework and GTLLVMHelper. + Use --all for full Xcode Instruments format with additional columns: - Type (Compute) - Pipeline State address - # SIMD Groups (SIMD wavefronts dispatched) - - # Allocated Registers + - Temp Regs (temporary register count) - High Register, shown only when source-backed - Spilled Bytes (register spills to memory) + - Dev Load / Dev Store (device memory load and store instruction counts) + +Temp Regs, Spilled, Dev Load and Dev Store come from the shader compiler's +pipelinePerformanceStatistics and are available for profiler-only traces too. +They read "?" when the trace carries no statistics for that shader; zero is a +real count and is printed as 0. + +The csv and json formats also carry per-shader host-side compilation time and +the compile-cache flag. They are absent from the --all table because that table +reproduces Xcode's All Shaders columns and these are not among them; for a +whole-capture compilation total see "gputrace profiler". Both cells are empty +when the trace recorded nothing, which is not the same as a zero compile time +or a cache miss. Examples: gputrace shaders trace.gputrace # Simple cost + name output gputrace shaders trace.gputrace --all # Full Xcode format gputrace shaders trace.gputrace --estimate # Show estimates for unknown fields + gputrace shaders trace.gputrace --xcode-cost # Match Xcode's All Shaders Cost gputrace shaders trace.gputrace --format csv # Export as CSV gputrace shaders trace.gputrace --format json # Export as JSON`, Args: cobra.ExactArgs(1), @@ -58,6 +78,7 @@ Examples: cmd.Flags().BoolVarP(&opts.estimate, "estimate", "e", opts.estimate, "Show estimated values for uncomputed fields") cmd.Flags().StringVarP(&opts.format, "format", "f", opts.format, "Output format: text, csv, or json") cmd.Flags().BoolVarP(&opts.all, "all", "a", opts.all, "Show all columns (full Xcode Instruments format)") + cmd.Flags().BoolVar(&opts.xcodeCost, "xcode-cost", opts.xcodeCost, "Use Xcode's processed pipeline timing for Cost") return cmd } @@ -93,9 +114,12 @@ func runShaders(cmd *cobra.Command, args []string, opts *shadersOptions) error { fmt.Fprintf(os.Stderr, " gputrace xp run %s -o profiled.gputrace\n\n", tracePath) return fmt.Errorf("profiler data required for shader timing") } + if opts.xcodeCost { + return runShadersXcodeCost(cmd, tracePath, opts) + } if hasUnsortedCapture { - // Full trace with profiler: use SIMD-based cost (matches Xcode) + // Full trace with profiler: use SIMD-based share. return runShadersFromFullTrace(tracePath, opts) } @@ -127,6 +151,7 @@ func checkUnsortedCapture(tracePath string) bool { // runShadersNoCost shows shader names without cost percentages (no profiler data). func runShadersNoCost(tracePath string, opts *shadersOptions) error { + fmt.Fprint(os.Stderr, profileReplayHint(tracePath)) trace, err := gputrace.Open(tracePath) if err != nil { return fmt.Errorf("open trace: %w", err) @@ -142,7 +167,7 @@ func runShadersNoCost(tracePath string, opts *shadersOptions) error { } func writeShadersNoCost(report *gputrace.ShaderMetricsReport, tracePath string, opts *shadersOptions) error { - fmt.Fprintf(os.Stderr, "No profiler data. To get Cost %%, run:\n") + fmt.Fprintf(os.Stderr, "No profiler data. To get a measured shader share, run:\n") fmt.Fprintf(os.Stderr, " gputrace xp run %s -o profiled.gputrace\n\n", tracePath) switch opts.format { @@ -158,15 +183,16 @@ func writeShadersNoCost(report *gputrace.ShaderMetricsReport, tracePath string, } func formatShadersNoCostText(w io.Writer, report *gputrace.ShaderMetricsReport) error { - fmt.Fprintf(w, "Cost Name\n") + libraries := splitShaderLibraryRows(report) + fmt.Fprintf(w, "Share Name\n") for _, shader := range report.Shaders { fmt.Fprintf(w, " ? %s\n", shader.Name) } + writeShaderInventoryNotes(w, report.Shaders, libraries) return nil } -// runShadersFromFullTrace uses full trace parsing for SIMD-based cost calculation. -// This matches Xcode's Cost % = SIMD Groups / Total SIMD Groups × 100 +// runShadersFromFullTrace uses full trace parsing for SIMD-based share. func runShadersFromFullTrace(tracePath string, opts *shadersOptions) error { // Open trace for full parsing trace, err := gputrace.Open(tracePath) @@ -180,6 +206,8 @@ func runShadersFromFullTrace(tracePath string, opts *shadersOptions) error { // Use combined approach: capture file dispatches + profiler function names report, err := extractSIMDBasedMetrics(trace, profilerDir) if err == nil && len(report.Shaders) > 0 { + report.ShareBasis = "simd_groups" + writeShaderShareBasis(opts.format, "SIMD groups") // Output based on format switch opts.format { case "csv": @@ -187,10 +215,7 @@ func runShadersFromFullTrace(tracePath string, opts *shadersOptions) error { case "json": return gputrace.ExportShaderMetricsJSON(os.Stdout, report) case "text": - if opts.all { - return gputrace.FormatShadersXcodeStyle(os.Stdout, report, trace, opts.estimate) - } - return gputrace.FormatShadersSimple(os.Stdout, report) + return writeShadersText(os.Stdout, report, trace, opts) default: return invalidShadersFormatError(opts.format) } @@ -204,7 +229,7 @@ func runShadersFromFullTrace(tracePath string, opts *shadersOptions) error { return fmt.Errorf("extract shader metrics: %w", err) } - // Recalculate Cost % based on SIMD Groups (TotalThreadgroups) to match Xcode + // Recalculate share based on SIMD Groups (TotalThreadgroups). var totalSIMDGroups uint64 for _, shader := range report.Shaders { totalSIMDGroups += shader.TotalThreadgroups @@ -215,10 +240,15 @@ func runShadersFromFullTrace(tracePath string, opts *shadersOptions) error { shader.PercentOfTotal = float64(shader.TotalThreadgroups) / float64(totalSIMDGroups) * 100.0 } } + report.ShareBasis = "simd_groups" + writeShaderShareBasis(opts.format, "SIMD groups") // Re-sort by SIMD-based cost sort.Slice(report.Shaders, func(i, j int) bool { - return report.Shaders[i].PercentOfTotal > report.Shaders[j].PercentOfTotal + if report.Shaders[i].PercentOfTotal != report.Shaders[j].PercentOfTotal { + return report.Shaders[i].PercentOfTotal > report.Shaders[j].PercentOfTotal + } + return report.Shaders[i].Name < report.Shaders[j].Name }) // Output based on format @@ -232,10 +262,8 @@ func runShadersFromFullTrace(tracePath string, opts *shadersOptions) error { return fmt.Errorf("failed to export JSON: %w", err) } case "text": - if opts.all { - gputrace.FormatShadersXcodeStyle(os.Stdout, report, trace, opts.estimate) - } else { - gputrace.FormatShadersSimple(os.Stdout, report) + if err := writeShadersText(os.Stdout, report, trace, opts); err != nil { + return fmt.Errorf("format shaders: %w", err) } default: return invalidShadersFormatError(opts.format) @@ -248,10 +276,7 @@ func runShadersFromFullTrace(tracePath string, opts *shadersOptions) error { // by joining dispatch threadgroup data from capture file with function names from profiler. func extractSIMDBasedMetrics(trace *gputrace.Trace, profilerDir string) (*gputrace.ShaderMetricsReport, error) { // Parse dispatch markers from capture data to get threadgroup dimensions - dispatches, err := trace.ParseDispatchInRegion(trace.CaptureData, 0) - if err != nil { - return nil, fmt.Errorf("parse dispatch markers: %w", err) - } + dispatches := trace.ParseDispatchInRegion(trace.CaptureData, 0) // Parse profiler streamData to get function names per dispatch index stats, err := counter.ParseStreamData(profilerDir, nil) @@ -280,37 +305,15 @@ func extractSIMDBasedMetrics(trace *gputrace.Trace, profilerDir string) (*gputra } } - const simdWidth uint64 = 32 // Apple Silicon SIMD width is 32 threads - for i, dispatch := range dispatches { - // Calculate threadgroups for this dispatch - var tgX, tgY, tgZ uint64 = 1, 1, 1 - if dispatch.ThreadsPerGroupX > 0 { - tgX = (dispatch.ThreadsX + dispatch.ThreadsPerGroupX - 1) / dispatch.ThreadsPerGroupX - } - if dispatch.ThreadsPerGroupY > 0 { - tgY = (dispatch.ThreadsY + dispatch.ThreadsPerGroupY - 1) / dispatch.ThreadsPerGroupY - } - if dispatch.ThreadsPerGroupZ > 0 { - tgZ = (dispatch.ThreadsZ + dispatch.ThreadsPerGroupZ - 1) / dispatch.ThreadsPerGroupZ - } - threadgroups := tgX * tgY * tgZ - - // Calculate SIMD groups (wavefronts) - // Xcode's "# SIMD Groups" = Total Threads / SIMD Width (32) - threadsPerGroup := dispatch.ThreadsPerGroupX * dispatch.ThreadsPerGroupY * dispatch.ThreadsPerGroupZ - totalThreads := threadgroups * threadsPerGroup - simdGroups := (totalThreads + simdWidth - 1) / simdWidth // Round up + simdGroups := dispatch.SIMDGroups() // Get function name from profiler data - funcName := "" + funcName := "(pipeline_unknown)" if i < len(stats.Dispatches) { - funcName = stats.Dispatches[i].FunctionName + funcName = stats.Dispatches[i].DisplayName() funcDurations[funcName] += uint64(stats.Dispatches[i].DurationUs) * 1000 // Convert to ns } - if funcName == "" { - funcName = fmt.Sprintf("(dispatch_%d)", i) - } funcSIMDGroups[funcName] += simdGroups } @@ -336,7 +339,7 @@ func extractSIMDBasedMetrics(trace *gputrace.Trace, profilerDir string) (*gputra TotalDurationNs: funcDurations[funcName], } - // Calculate SIMD-based cost percentage (matches Xcode) + // Calculate SIMD-based share percentage. if totalSIMDGroups > 0 { m.PercentOfTotal = float64(simdGroups) / float64(totalSIMDGroups) * 100.0 } @@ -354,6 +357,13 @@ func extractSIMDBasedMetrics(trace *gputrace.Trace, profilerDir string) (*gputra m.ThreadgroupMemory = ps.ThreadgroupMemory m.AllocatedRegisters = ps.TemporaryRegisterCount m.SpilledBytes = ps.SpilledBytes + m.DeviceLoadCount = ps.DeviceLoadCount + m.DeviceStoreCount = ps.DeviceStoreCount + m.HasPipelineStats = true + m.CompilationTimeMs = ps.CompilationTimeMs + if ps.CompilePerformance != nil { + m.FunctionWasCached = ps.CompilePerformance.FunctionWasCached + } } report.Shaders = append(report.Shaders, m) @@ -361,7 +371,10 @@ func extractSIMDBasedMetrics(trace *gputrace.Trace, profilerDir string) (*gputra // Sort by SIMD-based cost (highest first) sort.Slice(report.Shaders, func(i, j int) bool { - return report.Shaders[i].PercentOfTotal > report.Shaders[j].PercentOfTotal + if report.Shaders[i].PercentOfTotal != report.Shaders[j].PercentOfTotal { + return report.Shaders[i].PercentOfTotal > report.Shaders[j].PercentOfTotal + } + return report.Shaders[i].Name < report.Shaders[j].Name }) report.TotalShaders = len(report.Shaders) @@ -370,59 +383,21 @@ func extractSIMDBasedMetrics(trace *gputrace.Trace, profilerDir string) (*gputra } // findProfilerDir finds the .gpuprofiler_raw directory if it exists. +// Shader metrics come from streamData, so a directory without it is absent. func findProfilerDir(tracePath string) string { - // Check if it's directly a .gpuprofiler_raw directory - if filepath.Ext(tracePath) == ".gpuprofiler_raw" { - if _, err := os.Stat(filepath.Join(tracePath, "streamData")); err == nil { - return tracePath - } - } - // Look inside for .gpuprofiler_raw - entries, err := os.ReadDir(tracePath) - if err != nil { - return "" - } - for _, e := range entries { - if e.IsDir() && filepath.Ext(e.Name()) == ".gpuprofiler_raw" { - dir := filepath.Join(tracePath, e.Name()) - if _, err := os.Stat(filepath.Join(dir, "streamData")); err == nil { - return dir - } - } - } - return "" + return profilerraw.FindDirWithStreamData(tracePath) } // runShadersFromProfiler extracts shader info from .gpuprofiler_raw when unsorted-capture is missing. -// Note: This uses dispatch duration for Cost %, NOT SIMD groups (Xcode uses SIMD groups). -// For Xcode-matching Cost %, use a full trace with unsorted-capture directory. +// Note: This uses dispatch duration for Share %, not Xcode pipeline timing. func runShadersFromProfiler(tracePath string, opts *shadersOptions) error { - fmt.Fprintln(os.Stderr, "Note: Using dispatch duration for Cost % (profiler-only trace).") - fmt.Fprintln(os.Stderr, " Xcode uses SIMD Groups for Cost %. For matching values, use a full trace.") + fmt.Fprintln(os.Stderr, "Note: Share is based on cumulative dispatch span for this profiler-only trace.") + fmt.Fprintln(os.Stderr, " Use --xcode-cost for Xcode's processed pipeline timing.") fmt.Fprintln(os.Stderr, "") - // Find .gpuprofiler_raw directory - profilerDir := "" - - // Check if it's directly a .gpuprofiler_raw directory - if filepath.Ext(tracePath) == ".gpuprofiler_raw" { - profilerDir = tracePath - } else { - // Look inside for .gpuprofiler_raw - entries, err := os.ReadDir(tracePath) - if err != nil { - return fmt.Errorf("read directory: %w", err) - } - for _, e := range entries { - if e.IsDir() && filepath.Ext(e.Name()) == ".gpuprofiler_raw" { - profilerDir = filepath.Join(tracePath, e.Name()) - break - } - } - } + profilerDir := profilerraw.FindDir(tracePath) if profilerDir == "" { - fmt.Fprintf(os.Stderr, "Hint: To generate performance data, run:\n") - fmt.Fprintf(os.Stderr, " gputrace xcode-profile run %s\n\n", tracePath) + fmt.Fprint(os.Stderr, profileReplayHint(tracePath)) return fmt.Errorf("no .gpuprofiler_raw directory found in %s (and unsorted-capture is missing)", tracePath) } @@ -436,6 +411,10 @@ func runShadersFromProfiler(tracePath string, opts *shadersOptions) error { // Note: Uses dispatch duration for Cost %. Statistical sampling from Profiling_f_*.raw // has a complex format that needs further reverse engineering to match Xcode exactly. report := convertPipelineStatsToShaderReport(stats, nil) + report.ShareBasis = "dispatch_span" + if err := applySourceBackedShaderMetrics(filepath.Join(profilerDir, "streamData"), stats, report); err != nil { + fmt.Fprintf(os.Stderr, "Note: source-backed high-register metrics unavailable: %v\n", err) + } // Output based on format switch opts.format { @@ -448,11 +427,9 @@ func runShadersFromProfiler(tracePath string, opts *shadersOptions) error { return fmt.Errorf("failed to export JSON: %w", err) } case "text": - if opts.all { - // Format as Xcode Instruments style output (no trace available) - gputrace.FormatShadersXcodeStyle(os.Stdout, report, nil, opts.estimate) - } else { - gputrace.FormatShadersSimple(os.Stdout, report) + writeShaderShareBasis(opts.format, "dispatch cumulative-offset span") + if err := writeShadersText(os.Stdout, report, nil, opts); err != nil { + return fmt.Errorf("format shaders: %w", err) } default: return invalidShadersFormatError(opts.format) @@ -461,8 +438,75 @@ func runShadersFromProfiler(tracePath string, opts *shadersOptions) error { return nil } +// splitShaderLibraryRows removes the library-UUID rows from report and +// returns them. Library records share the name field with function records, +// so a shaders table that keeps them lists a library UUID as if a shader by +// that name ran. kernels already separates them; this makes shaders say the +// same thing. +func splitShaderLibraryRows(report *gputrace.ShaderMetricsReport) []*gputrace.ShaderMetrics { + var libraries []*gputrace.ShaderMetrics + kept := report.Shaders[:0] + for _, s := range report.Shaders { + if gputrace.IsLibraryUUID(s.Name) { + libraries = append(libraries, s) + continue + } + kept = append(kept, s) + } + report.Shaders = kept + report.TotalShaders = len(kept) + return libraries +} + +// writeShaderInventoryNotes says what the rows that are not shader names are, +// in the same terms kernels uses. +func writeShaderInventoryNotes(w io.Writer, shaders, libraries []*gputrace.ShaderMetrics) { + n := 0 + for _, s := range shaders { + if gputrace.IsArchiveFunctionName(s.Name) { + n++ + } + } + if n > 0 { + fmt.Fprintf(w, "\n%d %s named only by shader archive id (archive:...): the capture records\n"+ + "which archive the function came from, not its name. Run 'gputrace profile-replay'\n"+ + "on this trace to get the names.\n", + n, Pluralize(n, "shader", "shaders")) + } + if len(libraries) > 0 { + fmt.Fprintf(w, "\n%d library %s (not shader names):\n", + len(libraries), Pluralize(len(libraries), "UUID", "UUIDs")) + for _, s := range libraries { + fmt.Fprintf(w, " %s\n", s.Name) + } + } +} + +// writeShadersText renders the shader table and the inventory notes that say +// which rows are not shader names. +func writeShadersText(w io.Writer, report *gputrace.ShaderMetricsReport, t *gputrace.Trace, opts *shadersOptions) error { + libraries := splitShaderLibraryRows(report) + var err error + if opts.all { + err = gputrace.FormatShadersXcodeStyle(w, report, t, opts.estimate) + } else { + err = gputrace.FormatShadersSimple(w, report) + } + if err != nil { + return err + } + writeShaderInventoryNotes(w, report.Shaders, libraries) + return nil +} + +func writeShaderShareBasis(format, basis string) { + if format == "text" { + fmt.Fprintf(os.Stdout, "Share basis: %s\n", basis) + } +} + // convertPipelineStatsToShaderReport converts PipelineStats from streamData to ShaderMetricsReport. -// If execCosts is provided, uses statistical sampling cost for PercentOfTotal (matches Xcode). +// If execCosts is provided, uses statistical sampling cost for PercentOfTotal. // Otherwise falls back to dispatch duration-based cost. func convertPipelineStatsToShaderReport(stats *counter.StreamDataStats, execCosts *counter.ExecutionCostMetrics) *gputrace.ShaderMetricsReport { report := &gputrace.ShaderMetricsReport{ @@ -483,29 +527,21 @@ func convertPipelineStatsToShaderReport(stats *counter.StreamDataStats, execCost funcCounts := make(map[string]int) // function name -> invocation count funcPipeIDs := make(map[string][]int) // function name -> pipeline IDs for _, d := range stats.Dispatches { - name := d.FunctionName - if name == "" { - name = fmt.Sprintf("(pipeline_%d)", d.PipelineIndex) - } + name := d.DisplayName() funcTotals[name] += d.DurationUs funcCounts[name]++ } // Map function names to pipeline IDs for execution cost lookup for _, p := range stats.Pipelines { - name := p.FunctionName - if name == "" { - continue - } - funcPipeIDs[name] = append(funcPipeIDs[name], p.PipelineID) + funcPipeIDs[p.DisplayName()] = append(funcPipeIDs[p.DisplayName()], p.PipelineID) } - // Convert pipelines to shader metrics + // Convert pipelines to shader metrics. An unnamed pipeline still gets a + // row: its time is already in the share denominator, so dropping it hid + // cost the percentages continued to count. for _, p := range stats.Pipelines { - name := p.FunctionName - if name == "" { - continue - } + name := p.DisplayName() m := &gputrace.ShaderMetrics{ Name: name, @@ -523,15 +559,22 @@ func convertPipelineStatsToShaderReport(stats *counter.StreamDataStats, execCost ThreadgroupMemory: p.ThreadgroupMemory, AllocatedRegisters: p.TemporaryRegisterCount, SpilledBytes: p.SpilledBytes, + DeviceLoadCount: p.DeviceLoadCount, + DeviceStoreCount: p.DeviceStoreCount, + HasPipelineStats: true, + CompilationTimeMs: p.CompilationTimeMs, Bottlenecks: make([]string, 0), OptimizationHints: make([]string, 0), } + if p.CompilePerformance != nil { + m.FunctionWasCached = p.CompilePerformance.FunctionWasCached + } if m.InvocationCount > 0 { m.AvgDurationNs = m.TotalDurationNs / uint64(m.InvocationCount) } - // Use execution cost from statistical sampling if available (matches Xcode) + // Use execution cost from statistical sampling if available. if execCosts != nil { // Sum cost across all pipeline IDs for this function var totalCost float64 @@ -549,7 +592,10 @@ func convertPipelineStatsToShaderReport(stats *counter.StreamDataStats, execCost // Sort by cost (highest first) like Xcode does sort.Slice(report.Shaders, func(i, j int) bool { - return report.Shaders[i].PercentOfTotal > report.Shaders[j].PercentOfTotal + if report.Shaders[i].PercentOfTotal != report.Shaders[j].PercentOfTotal { + return report.Shaders[i].PercentOfTotal > report.Shaders[j].PercentOfTotal + } + return report.Shaders[i].Name < report.Shaders[j].Name }) report.TotalShaders = len(report.Shaders) diff --git a/cmd/gputrace/cmd/shaders_xcode_cost_darwin.go b/cmd/gputrace/cmd/shaders_xcode_cost_darwin.go new file mode 100644 index 00000000..223c284b --- /dev/null +++ b/cmd/gputrace/cmd/shaders_xcode_cost_darwin.go @@ -0,0 +1,124 @@ +//go:build darwin + +package cmd + +import ( + "encoding/csv" + "encoding/json" + "fmt" + "io" + "os" + "os/exec" + "path/filepath" + "strconv" + "strings" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/profilerraw" + "github.com/tmc/gputrace/internal/xcodebindings" +) + +func runShadersXcodeCost(cmd *cobra.Command, tracePath string, opts *shadersOptions) error { + if opts.all { + return fmt.Errorf("--xcode-cost cannot be combined with --all") + } + if reexeced, err := ensureXcodeCostFramework(); err != nil || reexeced { + return err + } + profilerDir := profilerraw.FindDirWithStreamData(tracePath) + if profilerDir == "" { + return fmt.Errorf("find profiler archive") + } + + var ( + rows []xcodebindings.ShaderCost + total uint64 + ) + err := withDiscardedXcodeGPUTimeStderr(func() error { + var err error + rows, total, err = xcodebindings.ShaderCosts(filepath.Join(profilerDir, "streamData")) + return err + }) + if err != nil { + return fmt.Errorf("compute Xcode shader costs: %w", err) + } + return writeShadersXcodeCost(cmd.OutOrStdout(), opts.format, rows, total) +} + +const xcodeCostReexecEnv = "GPUTRACE_XCODE_COST_REEXEC" + +// ensureXcodeCostFramework restarts gputrace once with the chosen framework +// fixed before Go package initialization. The generated binding loads during +// init, before Cobra can inspect GPUTRACE_XCODE_APP. +func ensureXcodeCostFramework() (bool, error) { + framework := xcodebindings.FrameworkPath() + if framework == "" { + return false, fmt.Errorf("find GTShaderProfiler framework") + } + if os.Getenv(xcodebindings.FrameworkPathEnv) == framework { + return false, nil + } + if os.Getenv(xcodeCostReexecEnv) != "" { + return false, fmt.Errorf("GTShaderProfiler framework override did not resolve to %s", framework) + } + executable, err := os.Executable() + if err != nil { + return false, fmt.Errorf("locate gputrace executable: %w", err) + } + child := exec.Command(executable, os.Args[1:]...) + child.Env = replaceEnv(os.Environ(), xcodebindings.FrameworkPathEnv, framework) + child.Env = replaceEnv(child.Env, xcodeCostReexecEnv, "1") + child.Stdin = os.Stdin + child.Stdout = os.Stdout + child.Stderr = os.Stderr + if err := child.Run(); err != nil { + return true, err + } + return true, nil +} + +func replaceEnv(env []string, key, value string) []string { + prefix := key + "=" + out := make([]string, 0, len(env)+1) + for _, entry := range env { + if !strings.HasPrefix(entry, prefix) { + out = append(out, entry) + } + } + return append(out, prefix+value) +} + +func writeShadersXcodeCost(w io.Writer, format string, rows []xcodebindings.ShaderCost, total uint64) error { + switch format { + case "text": + fmt.Fprintln(w, "Cost Name") + for _, row := range rows { + fmt.Fprintf(w, "%-12s %s\n", fmt.Sprintf("%.2f%%", row.Cost), row.Name) + } + return nil + case "json": + return json.NewEncoder(w).Encode(struct { + Total uint64 `json:"total_gpu_time"` + Basis string `json:"cost_basis"` + Rows []xcodebindings.ShaderCost `json:"shaders"` + }{total, "pipeline compute time / GPU time", rows}) + case "csv": + out := csv.NewWriter(w) + if err := out.Write([]string{"cost", "name", "compute_time"}); err != nil { + return err + } + for _, row := range rows { + if err := out.Write([]string{ + strconv.FormatFloat(row.Cost, 'f', 2, 64), + row.Name, + strconv.FormatUint(row.ComputeTime, 10), + }); err != nil { + return err + } + } + out.Flush() + return out.Error() + default: + return invalidShadersFormatError(format) + } +} diff --git a/cmd/gputrace/cmd/shaders_xcode_cost_darwin_test.go b/cmd/gputrace/cmd/shaders_xcode_cost_darwin_test.go new file mode 100644 index 00000000..c83c37c8 --- /dev/null +++ b/cmd/gputrace/cmd/shaders_xcode_cost_darwin_test.go @@ -0,0 +1,46 @@ +//go:build darwin + +package cmd + +import ( + "bytes" + "slices" + "strings" + "testing" + + "github.com/tmc/gputrace/internal/xcodebindings" +) + +func TestWriteShadersXcodeCostText(t *testing.T) { + rows := []xcodebindings.ShaderCost{{Name: "kernel", ComputeTime: 3303, Cost: 33.03}} + var out bytes.Buffer + if err := writeShadersXcodeCost(&out, "text", rows, 10000); err != nil { + t.Fatal(err) + } + if got := out.String(); !strings.Contains(got, "33.03%") || !strings.Contains(got, "kernel") { + t.Fatalf("output = %q", got) + } +} + +func TestWriteShadersXcodeCostStructured(t *testing.T) { + rows := []xcodebindings.ShaderCost{{Name: "kernel", ComputeTime: 3303, Cost: 33.03}} + for _, format := range []string{"json", "csv"} { + t.Run(format, func(t *testing.T) { + var out bytes.Buffer + if err := writeShadersXcodeCost(&out, format, rows, 10000); err != nil { + t.Fatal(err) + } + if got := out.String(); !strings.Contains(got, "kernel") || !strings.Contains(got, "33.03") { + t.Fatalf("output = %q", got) + } + }) + } +} + +func TestReplaceEnv(t *testing.T) { + got := replaceEnv([]string{"A=1", "B=2", "A=old"}, "A", "new") + want := []string{"B=2", "A=new"} + if !slices.Equal(got, want) { + t.Fatalf("replaceEnv = %v, want %v", got, want) + } +} diff --git a/cmd/gputrace/cmd/shaders_xcode_cost_unsupported.go b/cmd/gputrace/cmd/shaders_xcode_cost_unsupported.go new file mode 100644 index 00000000..f8b5de72 --- /dev/null +++ b/cmd/gputrace/cmd/shaders_xcode_cost_unsupported.go @@ -0,0 +1,13 @@ +//go:build !darwin + +package cmd + +import ( + "fmt" + + "github.com/spf13/cobra" +) + +func runShadersXcodeCost(*cobra.Command, string, *shadersOptions) error { + return fmt.Errorf("--xcode-cost requires macOS and Xcode") +} diff --git a/cmd/gputrace/cmd/span_table.go b/cmd/gputrace/cmd/span_table.go new file mode 100644 index 00000000..4a143838 --- /dev/null +++ b/cmd/gputrace/cmd/span_table.go @@ -0,0 +1,59 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "text/tabwriter" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace/internal/gpuevent" +) + +// printSpanTable renders the per-span setup/launch/GPU/tail decomposition. +// Every row names its provenance: a pre-span API is shown separately from +// setup, because its launch work cannot be assigned to either adjacent span. +func printSpanTable(cmd *cobra.Command, cap gpuevent.Capture, asJSON bool) error { + spans := gpuevent.AttributeSpans(cap) + decomp := gpuevent.Decompositions(spans, cap.APIs) + out := cmd.OutOrStdout() + + if len(decomp) == 0 { + fmt.Fprintln(out, "No span records in this capture.") + fmt.Fprintln(out, "Spans come from app-events sidecars (GPUTRACE_APP_EVENTS) or in-process capture (mlx-go EvalWithLabel).") + return nil + } + + if asJSON { + enc := json.NewEncoder(out) + enc.SetIndent("", " ") + return enc.Encode(decomp) + } + + w := tabwriter.NewWriter(out, 2, 4, 2, ' ', 0) + fmt.Fprintln(w, "SPAN\tEVAL\tSETUP\tSRC\tPRE-SPAN API [V]\tLAUNCH LAT\tGPU TIME\tTAIL\tKERNELS\tCONF") + for _, d := range decomp { + src := "[D] " + d.SetupSource + " fallback" + if d.SetupSource == "api" { + src = "[V] " + d.SetupSource + } + preSpan := "-" + if d.PreSpanAPINS > 0 { + preSpan = dur(d.PreSpanAPINS) + } + fmt.Fprintf(w, "%s\t%d\t%s\t%s\t%s\t%s\t%s\t%s\t%d\t%s\n", + d.SpanName, d.EvalSeq, + dur(d.SetupNS), src, preSpan, + dur(uint64(max64(d.LaunchLatencyNS, 0))), + dur(d.GPUTimeNS), dur(d.TailNS), + d.KernelCount, d.Confidence) + } + w.Flush() + return nil +} + +func max64(a, b int64) int64 { + if a > b { + return a + } + return b +} diff --git a/cmd/gputrace/cmd/stats.go b/cmd/gputrace/cmd/stats.go index 85e75d6c..3dc707e2 100644 --- a/cmd/gputrace/cmd/stats.go +++ b/cmd/gputrace/cmd/stats.go @@ -4,6 +4,7 @@ import ( "encoding/json" "fmt" "io" + "os" "sort" "strings" @@ -11,13 +12,18 @@ import ( "github.com/tmc/gputrace" "github.com/tmc/gputrace/internal/counter" + "github.com/tmc/gputrace/internal/tracebundle" ) var statsCmd = newStatsCommand(new(statsOptions)) type statsOptions struct { - verbose bool - json bool + verbose bool + json bool + limit int + all bool + benchfmt bool + benchConfig benchfmtConfigFlags } func newStatsCommand(opts *statsOptions) *cobra.Command { @@ -45,6 +51,9 @@ Examples: cmd.Flags().BoolVarP(&opts.verbose, "verbose", "v", opts.verbose, "Show verbose statistics including detailed analysis") cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output statistics in JSON format") + cmd.Flags().IntVar(&opts.limit, "limit", 50, "Maximum rows in each verbose list") + cmd.Flags().BoolVar(&opts.all, "all", false, "Show every row in verbose lists") + addBenchfmtFlags(cmd, &opts.benchfmt, &opts.benchConfig) return cmd } @@ -54,15 +63,33 @@ func init() { func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { tracePath := args[0] + if err := validateBenchfmtFlags(opts.benchfmt, opts.benchConfig); err != nil { + return err + } + if opts.benchfmt && opts.json { + return fmt.Errorf("--benchfmt and --json are mutually exclusive") + } // Verify trace file exists if err := checkTraceFile(tracePath); err != nil { return err } + if opts.benchfmt { + if findProfilerDir(tracePath) != "" { + _, streamStats, err := loadProfilerStats(tracePath) + if err != nil { + return fmt.Errorf("parse streamData: %w", err) + } + return writeProfilerBenchfmt(cmd.OutOrStdout(), tracePath, streamStats, nil, opts.benchConfig) + } + } + payload, payloadErr := tracebundle.InspectPayload(tracePath) // Open trace + // Open now succeeds on a bundle with no capture stream, so ProfilerOnly, + // not the error, is what routes to the profiler-backed report. trace, err := gputrace.Open(tracePath) - if err != nil { + if err != nil || trace.ProfilerOnly { if findProfilerDir(tracePath) != "" { return runStatsFromProfiler(cmd.OutOrStdout(), tracePath, opts) } @@ -74,6 +101,16 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { if err != nil { return fmt.Errorf("failed to extract statistics: %w", err) } + if opts.benchfmt { + config, err := mergeBenchfmtConfig(benchfmtDefaults(tracePath, ""), opts.benchConfig) + if err != nil { + return err + } + return writeBenchfmt(cmd.OutOrStdout(), benchfmtRecord{ + Config: config, + Values: benchfmtStructuralValues(statistics), + }) + } // Handle JSON output if opts.json { @@ -82,9 +119,13 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { // Quick one-liner summary parts := []string{ - fmt.Sprintf("%d %s", statistics.ComputeEncoders, Pluralize(statistics.ComputeEncoders, "encoder", "encoders")), fmt.Sprintf("%d %s", statistics.DispatchCalls, Pluralize(statistics.DispatchCalls, "dispatch", "dispatches")), - fmt.Sprintf("%d %s", statistics.UniqueKernels, Pluralize(statistics.UniqueKernels, "kernel", "kernels")), + fmt.Sprintf("%d observed kernel %s", statistics.ObservedKernelLabels, Pluralize(statistics.ObservedKernelLabels, "label", "labels")), + } + if statistics.ComputeEncodersAvailable { + parts = append([]string{ + fmt.Sprintf("%d %s", statistics.ComputeEncoders, Pluralize(statistics.ComputeEncoders, "encoder", "encoders")), + }, parts...) } if statistics.BufferUsageGB >= 0.001 { parts = append(parts, FormatBytes(statistics.BufferUsageBytes)) @@ -96,6 +137,9 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { fmt.Println(Colorize("Trace Info", ColorBold)) fmt.Println(TableSeparator(40)) fmt.Printf(" Path: %s\n", tracePath) + if payloadErr == nil { + fmt.Printf(" Raw Payload: %s\n", formatPayloadCompleteness(payload)) + } if trace.Metadata != nil { fmt.Printf(" UUID: %s\n", trace.Metadata.UUID) apiName := "Metal" @@ -110,9 +154,15 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { fmt.Println(Colorize("Workload", ColorBold)) fmt.Println(TableSeparator(40)) fmt.Printf(" Command Buffers: %s\n", FormatCount(statistics.CommandBuffers)) - fmt.Printf(" Compute Encoders: %s\n", FormatCount(statistics.ComputeEncoders)) + if statistics.ComputeEncodersAvailable { + fmt.Printf(" Compute Encoders: %s\n", FormatCount(statistics.ComputeEncoders)) + } else { + fmt.Printf(" Compute Encoders: (unavailable)\n") + } + fmt.Printf(" Encoder Count Source: %s\n", statistics.ComputeEncodersSource) fmt.Printf(" Dispatch Calls: %s\n", FormatCount(statistics.DispatchCalls)) - fmt.Printf(" Unique Kernels: %s\n", FormatCount(statistics.UniqueKernels)) + fmt.Printf(" Observed Kernel Labels: %s\n", FormatCount(statistics.ObservedKernelLabels)) + fmt.Printf(" Discovered Functions: %s\n", FormatCount(statistics.DiscoveredFunctions)) fmt.Println() // Memory Section @@ -144,9 +194,10 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { fmt.Println(Colorize("Timing", ColorBold)) fmt.Println(TableSeparator(40)) if gpuTimeUs > 0 { - fmt.Printf(" GPU Time: %s\n", FormatDuration(gpuTimeUs)) + fmt.Printf(" Encoder Span: %s\n", FormatDuration(gpuTimeUs)) } else { - fmt.Printf(" GPU Time: (no profiler data)\n") + fmt.Printf(" Encoder Span: (no profiler data)\n") + defer fmt.Fprint(os.Stderr, profileReplayHint(tracePath)) } if hasProfilerData && profilerDir != "" { if streamStats, err := counter.ParseStreamData(profilerDir, nil); err == nil { @@ -161,6 +212,9 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { if streamStats.CommandBufferWallNs > 0 { fmt.Printf(" CB Wall Time: %s\n", FormatDurationNs(streamStats.CommandBufferWallNs)) } + if streamStats.Metadata.NumBlitCalls != nil { + fmt.Printf(" Blit Calls: %d\n", *streamStats.Metadata.NumBlitCalls) + } if streamStats.TimingSource != "" { fmt.Printf(" Timing Source: %s\n", streamStats.TimingSource) } @@ -172,17 +226,15 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { // Top Kernels Section (if we have data) if hasProfilerData && profilerDir != "" { if streamStats, err := counter.ParseStreamData(profilerDir, nil); err == nil && len(streamStats.Dispatches) > 0 { - fmt.Println(Colorize("Top Kernels (by time)", ColorBold)) + fmt.Println(Colorize("Functions by Dispatch Span", ColorBold)) fmt.Println(TableSeparator(40)) + fmt.Println(" Cumulative offsets may include boundary or gap time.") // Aggregate by function name funcTotals := make(map[string]int) funcCounts := make(map[string]int) for _, d := range streamStats.Dispatches { - name := d.FunctionName - if name == "" { - name = fmt.Sprintf("pipeline_%d", d.PipelineIndex) - } + name := d.DisplayName() funcTotals[name] += d.DurationUs funcCounts[name]++ } @@ -198,7 +250,10 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { sorted = append(sorted, funcStat{name, time, funcCounts[name]}) } sort.Slice(sorted, func(i, j int) bool { - return sorted[i].time > sorted[j].time + if sorted[i].time != sorted[j].time { + return sorted[i].time > sorted[j].time + } + return sorted[i].name < sorted[j].name }) // Show top 5 @@ -226,7 +281,7 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { fmt.Printf(" %5.1f%% %-35s (%dx)\n", pct, name, fs.count) } if len(sorted) > 5 { - fmt.Printf(" ...and %d more kernels\n", len(sorted)-5) + fmt.Printf(" ...and %d more functions\n", len(sorted)-5) } fmt.Println() } @@ -277,27 +332,30 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { // Show all encoder labels if len(trace.EncoderLabels) > 0 { fmt.Printf("%s (%d):\n", Colorize("All Encoder Labels", ColorGreen), len(trace.EncoderLabels)) - for i, label := range trace.EncoderLabels { + for i, label := range limitedStrings(trace.EncoderLabels, opts.limit, opts.all) { fmt.Printf(" [%d] %s\n", i, label) } + printOmittedRows(len(trace.EncoderLabels), opts.limit, opts.all) fmt.Println() } // Show all kernel names if len(trace.KernelNames) > 0 { - fmt.Printf("%s (%d):\n", Colorize("All Kernel Names", ColorGreen), len(trace.KernelNames)) - for i, name := range trace.KernelNames { + fmt.Printf("%s (%d discovered; not necessarily dispatched):\n", Colorize("Library Functions", ColorGreen), len(trace.KernelNames)) + for i, name := range limitedStrings(trace.KernelNames, opts.limit, opts.all) { fmt.Printf(" [%d] %s\n", i, name) } + printOmittedRows(len(trace.KernelNames), opts.limit, opts.all) fmt.Println() } // Show buffer labels if len(trace.BufferLabels) > 0 { fmt.Printf("%s (%d):\n", Colorize("All Buffer Labels", ColorGreen), len(trace.BufferLabels)) - for i, label := range trace.BufferLabels { + for i, label := range limitedStrings(trace.BufferLabels, opts.limit, opts.all) { fmt.Printf(" [%d] %s\n", i, label) } + printOmittedRows(len(trace.BufferLabels), opts.limit, opts.all) fmt.Println() } @@ -323,6 +381,27 @@ func runStats(cmd *cobra.Command, args []string, opts *statsOptions) error { return nil } +func limitedStrings(values []string, limit int, all bool) []string { + if all || limit < 0 || len(values) <= limit { + return values + } + if limit == 0 { + return nil + } + return values[:limit] +} + +func printOmittedRows(total, limit int, all bool) { + if all || limit < 0 || total <= limit { + return + } + shown := limit + if shown < 0 { + shown = 0 + } + fmt.Printf(" ... %d more; use --all to show every row\n", total-shown) +} + type profilerStatsJSONOutput struct { ProfilerOnly bool `json:"profiler_only"` ProfilerDir string `json:"profiler_dir"` @@ -348,6 +427,9 @@ func runStatsFromProfiler(w io.Writer, tracePath string, opts *statsOptions) err if err != nil { return err } + if opts.benchfmt { + return writeProfilerBenchfmt(w, tracePath, streamStats, nil, opts.benchConfig) + } commandBuffers := 0 if streamStats.Timeline != nil { @@ -388,7 +470,10 @@ func runStatsFromProfiler(w io.Writer, tracePath string, opts *statsOptions) err fmt.Println(TableSeparator(40)) fmt.Printf(" Path: %s\n", tracePath) fmt.Printf(" Profiler Data: %s\n", profilerDir) - fmt.Printf(" Note: no MTSP capture data; use profiler for kernel timing details\n") + if payload, err := tracebundle.InspectPayload(tracePath); err == nil { + fmt.Printf(" Raw Payload: %s\n", formatPayloadCompleteness(payload)) + } + fmt.Printf(" Note: aggregate profiler timing is available; structural and threadgroup analysis requires a full raw payload\n") fmt.Println() fmt.Println(Colorize("Workload", ColorBold)) @@ -425,6 +510,17 @@ func runStatsFromProfiler(w io.Writer, tracePath string, opts *statsOptions) err return nil } +func formatPayloadCompleteness(payload tracebundle.Payload) string { + switch payload.Class { + case tracebundle.PayloadFull: + return "full (capture and raw resources present)" + case tracebundle.PayloadProfilerOnly: + return "profiler-only (aggregate timing available; structural/threadgroup data unavailable)" + default: + return "incomplete (structural/threadgroup data unavailable)" + } +} + // StatsJSONOutput represents the JSON output structure for stats command. type StatsJSONOutput struct { Statistics *StatsJSON `json:"statistics"` @@ -434,23 +530,27 @@ type StatsJSONOutput struct { // StatsJSON represents statistics in JSON format. type StatsJSON struct { - BufferUsageBytes uint64 `json:"buffer_usage_bytes"` - BufferUsageGB float64 `json:"buffer_usage_gb"` - BufferSizeSum uint64 `json:"buffer_size_sum"` - UniqueBuffers int `json:"unique_buffers"` - HeapUsageBytes uint64 `json:"heap_usage_bytes"` - HeapUsageMB float64 `json:"heap_usage_mb"` - UniqueHeaps int `json:"unique_heaps"` - UnusedBuffers int `json:"unused_buffers,omitempty"` - UnusedTextures int `json:"unused_textures,omitempty"` - UnusedFunctions int `json:"unused_functions,omitempty"` - UniqueKernels int `json:"unique_kernels"` - CommandBuffers int `json:"command_buffers"` - ComputeEncoders int `json:"compute_encoders"` - DispatchCalls int `json:"dispatch_calls"` - TotalRecords int `json:"total_records"` - RecordTypes map[string]int `json:"record_types"` - MTLBLibraries int `json:"mtlb_libraries"` + BufferUsageBytes uint64 `json:"buffer_usage_bytes"` + BufferUsageGB float64 `json:"buffer_usage_gb"` + BufferSizeSum uint64 `json:"buffer_size_sum"` + UniqueBuffers int `json:"unique_buffers"` + HeapUsageBytes uint64 `json:"heap_usage_bytes"` + HeapUsageMB float64 `json:"heap_usage_mb"` + UniqueHeaps int `json:"unique_heaps"` + UnusedBuffers int `json:"unused_buffers,omitempty"` + UnusedTextures int `json:"unused_textures,omitempty"` + UnusedFunctions int `json:"unused_functions,omitempty"` + UniqueKernels int `json:"unique_kernels"` + ObservedKernelLabels int `json:"observed_kernel_labels"` + DiscoveredFunctions int `json:"discovered_functions"` + CommandBuffers int `json:"command_buffers"` + ComputeEncoders *int `json:"compute_encoders"` + ComputeEncodersAvailable bool `json:"compute_encoders_available"` + ComputeEncodersSource string `json:"compute_encoders_source"` + DispatchCalls int `json:"dispatch_calls"` + TotalRecords int `json:"total_records"` + RecordTypes map[string]int `json:"record_types"` + MTLBLibraries int `json:"mtlb_libraries"` } // MetadataJSON represents trace metadata in JSON format. @@ -481,23 +581,30 @@ type TimingJSON struct { // outputStatsJSON outputs statistics in JSON format. func outputStatsJSON(w io.Writer, stats *gputrace.TraceStatistics, trace *gputrace.Trace, verbose bool) error { s := &StatsJSON{ - BufferUsageBytes: stats.BufferUsageBytes, - BufferUsageGB: stats.BufferUsageGB, - BufferSizeSum: stats.BufferSizeSum, - UniqueBuffers: stats.UniqueBuffers, - HeapUsageBytes: stats.HeapUsageBytes, - HeapUsageMB: stats.HeapUsageMB, - UniqueHeaps: stats.UniqueHeaps, - UnusedBuffers: stats.UnusedBuffers, - UnusedTextures: stats.UnusedTextures, - UnusedFunctions: stats.UnusedFunctions, - UniqueKernels: stats.UniqueKernels, - CommandBuffers: stats.CommandBuffers, - ComputeEncoders: stats.ComputeEncoders, - DispatchCalls: stats.DispatchCalls, - TotalRecords: stats.TotalRecords, - RecordTypes: stats.RecordTypes, - MTLBLibraries: stats.MTLBLibraries, + BufferUsageBytes: stats.BufferUsageBytes, + BufferUsageGB: stats.BufferUsageGB, + BufferSizeSum: stats.BufferSizeSum, + UniqueBuffers: stats.UniqueBuffers, + HeapUsageBytes: stats.HeapUsageBytes, + HeapUsageMB: stats.HeapUsageMB, + UniqueHeaps: stats.UniqueHeaps, + UnusedBuffers: stats.UnusedBuffers, + UnusedTextures: stats.UnusedTextures, + UnusedFunctions: stats.UnusedFunctions, + UniqueKernels: stats.UniqueKernels, + ObservedKernelLabels: stats.ObservedKernelLabels, + DiscoveredFunctions: stats.DiscoveredFunctions, + CommandBuffers: stats.CommandBuffers, + ComputeEncodersAvailable: stats.ComputeEncodersAvailable, + ComputeEncodersSource: stats.ComputeEncodersSource, + DispatchCalls: stats.DispatchCalls, + TotalRecords: stats.TotalRecords, + RecordTypes: stats.RecordTypes, + MTLBLibraries: stats.MTLBLibraries, + } + if stats.ComputeEncodersAvailable { + count := stats.ComputeEncoders + s.ComputeEncoders = &count } output := &StatsJSONOutput{ diff --git a/cmd/gputrace/cmd/stats_test.go b/cmd/gputrace/cmd/stats_test.go index 1b345a76..327a65e9 100644 --- a/cmd/gputrace/cmd/stats_test.go +++ b/cmd/gputrace/cmd/stats_test.go @@ -8,8 +8,19 @@ import ( "testing" "github.com/spf13/cobra" + "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/tracebundle" ) +func TestFormatPayloadCompletenessGatesStructuralClaims(t *testing.T) { + got := formatPayloadCompleteness(tracebundle.Payload{Class: tracebundle.PayloadProfilerOnly, HasProfilerStream: true}) + for _, want := range []string{"profiler-only", "aggregate timing available", "structural/threadgroup data unavailable"} { + if !strings.Contains(got, want) { + t.Fatalf("formatPayloadCompleteness() = %q, want %q", got, want) + } + } +} + func TestRunStatsJSONUsesCommandOutput(t *testing.T) { tracePath := "../../../testdata/traces/01-single-encoder/01-single-encoder-run1.gputrace" if _, err := os.Stat(tracePath); os.IsNotExist(err) { @@ -35,6 +46,9 @@ func TestRunStatsJSONUsesCommandOutput(t *testing.T) { if got.Statistics == nil { t.Fatalf("stats JSON output missing statistics: %s", out.String()) } + if got.Statistics.DiscoveredFunctions < got.Statistics.UniqueKernels { + t.Fatalf("discovered functions = %d, dispatched kernels = %d", got.Statistics.DiscoveredFunctions, got.Statistics.UniqueKernels) + } } func TestWriteStatsJSONProfilerOutput(t *testing.T) { @@ -63,3 +77,34 @@ func TestWriteStatsJSONProfilerOutput(t *testing.T) { t.Fatalf("profiler stats JSON = %+v", got) } } + +func TestOutputStatsJSONReportsEncoderCountAvailability(t *testing.T) { + stats := &gputrace.TraceStatistics{ + CommandBuffers: 10, + DispatchCalls: 435, + ComputeEncodersSource: "unavailable: raw capture lacks command-buffer-scoped encoder lifecycle evidence", + ComputeEncodersAvailable: false, + } + + var out bytes.Buffer + if err := outputStatsJSON(&out, stats, &gputrace.Trace{}, false); err != nil { + t.Fatalf("outputStatsJSON: %v", err) + } + + var got StatsJSONOutput + if err := json.Unmarshal(out.Bytes(), &got); err != nil { + t.Fatalf("decode stats JSON: %v", err) + } + if got.Statistics.CommandBuffers != 10 || got.Statistics.DispatchCalls != 435 { + t.Fatalf("structural totals changed: %+v", got.Statistics) + } + if got.Statistics.ComputeEncodersAvailable { + t.Fatalf("compute_encoders_available = true, want false") + } + if got.Statistics.ComputeEncoders != nil { + t.Fatalf("compute_encoders = %v, want null", got.Statistics.ComputeEncoders) + } + if !strings.Contains(got.Statistics.ComputeEncodersSource, "command-buffer-scoped") { + t.Fatalf("compute_encoders_source = %q", got.Statistics.ComputeEncodersSource) + } +} diff --git a/cmd/gputrace/cmd/summary.go b/cmd/gputrace/cmd/summary.go new file mode 100644 index 00000000..7a9e79d8 --- /dev/null +++ b/cmd/gputrace/cmd/summary.go @@ -0,0 +1,162 @@ +package cmd + +import ( + "encoding/json" + "fmt" + "io" + "time" + + "github.com/spf13/cobra" + "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/evidence" + "github.com/tmc/gputrace/internal/fmtutil" +) + +var summaryCmd = newSummaryCommand(new(summaryOptions)) + +type summaryOptions struct { + json bool + limit int +} + +func newSummaryCommand(opts *summaryOptions) *cobra.Command { + cmd := &cobra.Command{ + Use: "summary ", + Short: "Summarize structure, timing, and evidence gaps", + Long: `Summarize a trace's structure, timing, and evidence gaps. + +Given a .gpucapture bundle (Linux/NVIDIA), it summarizes the capture +instead: kernel time, the GPU busy/idle budget with its largest gaps, and +launch latency when the capture recorded it.`, + Args: cobra.ExactArgs(1), + RunE: func(cmd *cobra.Command, args []string) error { + return runSummary(cmd, args, opts) + }, + } + cmd.Flags().BoolVar(&opts.json, "json", opts.json, "Output JSON") + cmd.Flags().IntVar(&opts.limit, "limit", 5, "Maximum functions to show") + return cmd +} + +func init() { + rootCmd.AddCommand(summaryCmd) +} + +func runSummary(cmd *cobra.Command, args []string, opts *summaryOptions) error { + if opts.limit < 0 { + return fmt.Errorf("--limit must be >= 0") + } + path := args[0] + // A CUDA capture has no command buffers or encoders to summarize; its + // budget is the busy/idle split, which lives in the same place for a + // reader asking the same question. + if isCaptureInput(path) { + return runCaptureSummary(cmd, path, opts) + } + if err := checkTraceFile(path); err != nil { + return err + } + _, stats, err := loadProfilerStats(path) + if err != nil { + return err + } + tr, err := gputrace.Open(path) + if err != nil { + return fmt.Errorf("open trace: %w", err) + } + defer tr.Close() + report, err := evidence.Build(tr, stats) + if err != nil { + return err + } + if opts.json { + encoder := json.NewEncoder(cmd.OutOrStdout()) + encoder.SetIndent("", " ") + return encoder.Encode(report) + } + writeSummary(cmd.OutOrStdout(), report, opts.limit) + return nil +} + +// runCaptureSummary reports the budget of a CUDA capture: where its GPU +// time went, and how much of the span the device spent waiting. +func runCaptureSummary(cmd *cobra.Command, path string, opts *summaryOptions) error { + rep, err := loadCaptureReport(path) + if err != nil { + return err + } + if opts.json { + encoder := json.NewEncoder(cmd.OutOrStdout()) + encoder.SetIndent("", " ") + return encoder.Encode(rep) + } + out := cmd.OutOrStdout() + fmt.Fprintf(out, "%d launches · %d distinct kernels · %s total kernel time\n", + rep.KernelLaunches, len(rep.Kernels), dur(rep.TotalKernelNS)) + if rep.MemcpyCount > 0 || rep.MemsetCount > 0 { + fmt.Fprintf(out, "Transfers: %d copies (%s), %d fills\n", + rep.MemcpyCount, dur(rep.MemcpyNS), rep.MemsetCount) + } else { + fmt.Fprintln(out, "Transfers: 0 copies, 0 fills") + } + + fmt.Fprintln(out, "\nTop work") + rows := rep.Kernels + if opts.limit > 0 && len(rows) > opts.limit { + rows = rows[:opts.limit] + } + for _, k := range rows { + fmt.Fprintf(out, "%-44s %6d launches %9s %5.1f%%\n", + fmtutil.TruncateString(shortKernel(k.Name), 44), k.Count, dur(k.TotalNS), k.SharePct) + } + + printUtilization(out, rep.Utilization) + printGraphs(out, rep.Graphs) + printLaunchLatency(out, rep.LaunchLatency) + return nil +} + +func writeSummary(w io.Writer, report *evidence.Report, limit int) { + fmt.Fprintf(w, "%d command buffers · %d profiler compute encoders · %d dispatches\n", + report.CommandBuffers, report.ComputeEncoders, report.Dispatches) + fmt.Fprintf(w, "Dispatch span %s · CB active %s · CB wall %s\n", + formatSummaryDuration(report.DispatchSpan), formatSummaryDuration(report.CBActiveTime), formatSummaryDuration(report.CBWallSpan)) + fmt.Fprintf(w, "Timing: %s", report.TimingSource) + if report.TimingApproximate { + fmt.Fprint(w, " (approximate)") + } + fmt.Fprintln(w, "; dispatch spans may include boundary or gap time") + fmt.Fprintf(w, "Labels: %d CS/debug label records, %d unique (not encoder instances)\n", + report.CSLabels, report.UniqueCSLabels) + + fmt.Fprintln(w, "\nTop work") + rows := report.Functions + if len(rows) > limit { + rows = rows[:limit] + } + for _, row := range rows { + fmt.Fprintf(w, "%-44s %6d calls %9s %5.1f%%\n", + fmtutil.TruncateString(row.Name, 44), row.Dispatches, formatSummaryDuration(row.Span), row.SpanShare) + } + + fmt.Fprintln(w, "\nPacking") + fmt.Fprintf(w, "median %.1f dispatches/encoder · %.1f dispatches/command buffer\n", + report.Packing.MedianDispatchesPerEncoder, report.Packing.DispatchesPerCommandBuffer) + if len(report.EvidenceGaps) > 0 { + fmt.Fprintln(w, "\nEvidence gaps") + for i, gap := range report.EvidenceGaps { + if i > 0 { + fmt.Fprint(w, " · ") + } + fmt.Fprint(w, gap) + } + fmt.Fprintln(w) + } +} + +func formatSummaryDuration(d time.Duration) string { + if d <= 0 { + return "unavailable" + } + return FormatDurationNs(uint64(d)) +} diff --git a/cmd/gputrace/cmd/summary_ocr_darwin.go b/cmd/gputrace/cmd/summary_ocr_darwin.go new file mode 100644 index 00000000..af5d19f0 --- /dev/null +++ b/cmd/gputrace/cmd/summary_ocr_darwin.go @@ -0,0 +1,319 @@ +//go:build darwin + +package cmd + +import ( + "context" + "fmt" + "math" + "strings" + "time" + + "github.com/tmc/apple/coregraphics" + "github.com/tmc/apple/vision" +) + +type summaryOCRMatch struct { + Text string + Confidence float64 + X float64 + Y float64 + Width float64 + Height float64 +} + +type screenRect struct { + X float64 + Y float64 + Width float64 + Height float64 +} + +func (r screenRect) contains(x, y float64) bool { + return x > r.X && y > r.Y && x < r.X+r.Width && y < r.Y+r.Height +} + +func (m summaryOCRMatch) center() (float64, float64) { + return m.X + m.Width/2, m.Y + m.Height/2 +} + +func clickSummaryPerformanceOCR(ctx context.Context, appAX uintptr, summary standaloneRecoveryWindow, recovery standaloneExportRecovery, geometryKey string) error { + return clickPerformanceOCR(ctx, appAX, summary, recovery, "Summary", + func() (standaloneRecoveryWindow, error) { + return summaryRecoveryTarget(recoveryWindows(appAX), recovery, geometryKey) + }) +} + +func clickFinishedPerformanceOCR(ctx context.Context, appAX uintptr, source standaloneRecoveryWindow, recovery standaloneExportRecovery, geometryKey string) error { + return clickPerformanceOCR(ctx, appAX, source, recovery, "Finished source", + func() (standaloneRecoveryWindow, error) { + return restoredRecoverySourceTarget(recoveryWindows(appAX), recovery, geometryKey) + }) +} + +func clickPerformanceOCR(ctx context.Context, appAX uintptr, selected standaloneRecoveryWindow, recovery standaloneExportRecovery, state string, selectWindow func() (standaloneRecoveryWindow, error)) error { + if err := activateProcessPID(int32(recovery.Identity.PID)); err != nil { + return fmt.Errorf("activate bound Xcode for %s OCR: %w", state, err) + } + if err := axAction(selected.Element, "AXRaise"); err != nil { + return fmt.Errorf("raise selected %s window for OCR: %w", state, err) + } + if err := waitForAutomation(ctx, 200*time.Millisecond); err != nil { + return err + } + var previous summaryOCRMatch + for sample := 0; sample < 2; sample++ { + if err := checkAutomationCanceled(ctx); err != nil { + return err + } + if err := requireRecoveryIdentity(appAX, recovery); err != nil { + return err + } + current, err := selectWindow() + if err != nil { + return fmt.Errorf("revalidate %s before OCR sample %d: %w", state, sample+1, err) + } + if shows := shallowShowPerformanceButtons(current.Element); len(shows) != 0 { + return fmt.Errorf("AX Show Performance controls changed while preparing %s OCR", state) + } + region, err := summaryRightPaneRegion(current.Element) + if err != nil { + return err + } + match, err := recognizeSummaryPerformance(current.Element, region) + if err != nil { + return fmt.Errorf("%s OCR sample %d: %w", state, sample+1, err) + } + if sample > 0 && !stableSummaryOCRMatch(previous, match, 4) { + return fmt.Errorf("%s OCR target moved between stable samples", state) + } + previous = match + selected = current + if sample == 0 { + if err := waitForAutomation(ctx, 250*time.Millisecond); err != nil { + return err + } + } + } + + // The second sample is immediately followed by a final structural and + // hit-test check. No additional OCR or click retry is permitted. + current, err := selectWindow() + if err != nil { + return fmt.Errorf("revalidate %s before OCR click: %w", state, err) + } + if standaloneRecoveryWindowKey(current) != standaloneRecoveryWindowKey(selected) { + return fmt.Errorf("%s window changed after OCR proof", state) + } + cx, cy := previous.center() + region, err := summaryRightPaneRegion(current.Element) + if err != nil { + return err + } + if !region.contains(cx, cy) { + return fmt.Errorf("OCR target center lies outside selected %s right pane", state) + } + selectedWindowID, err := getWindowID(current.Element) + if err != nil { + return fmt.Errorf("read selected Xcode CGWindowID before OCR click: %w", err) + } + hit := axCopyElementAtPosition(appAX, cx, cy) + if hit == 0 { + return fmt.Errorf("cannot hit-test OCR target in selected Xcode window") + } + defer cfRelease(hit) + var hitPID int32 + hitWindowID, hitWindowErr := getWindowID(hit) + if axUIElementGetPid(hit, &hitPID) != kAXErrorSuccess || + int(hitPID) != recovery.Identity.PID || + hitWindowErr != nil || hitWindowID != selectedWindowID { + return fmt.Errorf("OCR target ownership mismatch: role=%q description=%q PID=%d windowID=%d want PID=%d windowID=%d", + axString(hit, "AXRole"), axString(hit, "AXDescription"), + hitPID, hitWindowID, recovery.Identity.PID, selectedWindowID) + } + if err := clickScreenPoint(cx, cy); err != nil { + return fmt.Errorf("click OCR Show Performance target: %w", err) + } + return nil +} + +func summaryRightPaneRegion(window uintptr) (screenRect, error) { + wx, wy := axPosition(window) + ww, wh := axSize(window) + navigator := findElementAtDepth( + window, 3, 96, axChildren, + func(element uintptr) bool { return axString(element, "AXRole") == "AXOutline" }, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXGroup" && + axString(element, "AXDescription") == "navigator" + }, + ) + debugBar := findElementAtDepth( + window, 4, 128, axChildren, + func(element uintptr) bool { return axString(element, "AXRole") == "AXOutline" }, + func(element uintptr) bool { + return axString(element, "AXRole") == "AXGroup" && + axString(element, "AXDescription") == "debug bar" + }, + ) + if ww <= 0 || wh <= 0 || navigator == 0 || debugBar == 0 { + return screenRect{}, fmt.Errorf("cannot establish selected Summary right-pane bounds") + } + nx, _ := axPosition(navigator) + nw, _ := axSize(navigator) + _, debugY := axPosition(debugBar) + left := float64(nx + nw) + top := float64(wy + 52) + right := float64(wx + ww) + bottom := float64(debugY) + if left < float64(wx) || right > float64(wx+ww) || + top < float64(wy) || bottom > float64(wy+wh) || + right-left < 200 || bottom-top < 100 { + return screenRect{}, fmt.Errorf("invalid selected Summary right-pane bounds") + } + return screenRect{X: left, Y: top, Width: right - left, Height: bottom - top}, nil +} + +func recognizeSummaryPerformance(window uintptr, region screenRect) (summaryOCRMatch, error) { + windowID, err := getWindowID(window) + if err != nil { + return summaryOCRMatch{}, err + } + var pid int32 + if axUIElementGetPid(window, &pid) != kAXErrorSuccess || pid == 0 { + return summaryOCRMatch{}, fmt.Errorf("read selected Xcode PID") + } + cgWindow, err := exactCGWindowInfo(pid, windowID) + if err != nil { + return summaryOCRMatch{}, err + } + image := cgWindowListCreateImage( + math.Inf(1), math.Inf(1), 0, 0, + kCGWindowListOptionIncludingWindow, + windowID, + kCGWindowImageBoundsIgnoreFraming|kCGWindowImageBestResolution, + ) + if image == 0 { + return summaryOCRMatch{}, fmt.Errorf("capture selected Xcode window %d", windowID) + } + defer cgImageRelease(image) + imageWidth := float64(coregraphics.CGImageGetWidth(coregraphics.CGImageRef(image))) + imageHeight := float64(coregraphics.CGImageGetHeight(coregraphics.CGImageRef(image))) + wx, wy := axPosition(window) + ww, wh := axSize(window) + if imageWidth <= 0 || imageHeight <= 0 || ww <= 0 || wh <= 0 { + return summaryOCRMatch{}, fmt.Errorf("invalid selected-window image geometry") + } + if !compatibleWindowBounds(screenRect{ + X: float64(wx), Y: float64(wy), Width: float64(ww), Height: float64(wh), + }, screenRect{ + X: cgWindow.bounds.Origin.X, Y: cgWindow.bounds.Origin.Y, + Width: cgWindow.bounds.Size.Width, Height: cgWindow.bounds.Size.Height, + }, 2) { + return summaryOCRMatch{}, fmt.Errorf("selected AX and CG window bounds disagree") + } + + handler := vision.NewImageRequestHandlerWithCGImageOptions(coregraphics.CGImageRef(image), nil) + request := vision.NewVNRecognizeTextRequest() + request.SetRecognitionLevel(vision.VNRequestTextRecognitionLevelAccurate) + request.SetUsesLanguageCorrection(true) + ok, err := handler.PerformRequestsError([]vision.VNRequest{request.VNRequest}) + if err != nil { + return summaryOCRMatch{}, fmt.Errorf("Vision OCR: %w", err) + } + if !ok { + return summaryOCRMatch{}, fmt.Errorf("Vision OCR request failed") + } + + var matches []summaryOCRMatch + for _, observation := range request.Results() { + text := vision.VNRecognizedTextObservationFromID(observation.ID) + candidates := text.TopCandidates(1) + if len(candidates) == 0 { + continue + } + candidate := candidates[0] + if normalizeOCRText(candidate.String()) != "show performance" || + float64(candidate.Confidence()) < 0.8 { + continue + } + bounds := text.BoundingBox() + localX := bounds.Origin.X * cgWindow.bounds.Size.Width + localY := (1 - bounds.Origin.Y - bounds.Size.Height) * cgWindow.bounds.Size.Height + match := summaryOCRMatch{ + Text: candidate.String(), + Confidence: float64(candidate.Confidence()), + X: cgWindow.bounds.Origin.X + localX, + Y: cgWindow.bounds.Origin.Y + localY, + Width: bounds.Size.Width * cgWindow.bounds.Size.Width, + Height: bounds.Size.Height * cgWindow.bounds.Size.Height, + } + cx, cy := match.center() + if match.Width < 40 || match.Height < 8 || + !region.contains(match.X, match.Y) || + !region.contains(match.X+match.Width, match.Y+match.Height) || + !region.contains(cx, cy) { + continue + } + matches = append(matches, match) + } + if len(matches) != 1 { + return summaryOCRMatch{}, fmt.Errorf("want one exact Show Performance OCR match in selected right pane, found %d", len(matches)) + } + return matches[0], nil +} + +func exactCGWindowInfo(pid int32, windowID uint32) (cgWindowInfo, error) { + var matches []cgWindowInfo + for _, window := range cgOnscreenWindowsForPID(pid) { + if window.windowID == windowID { + matches = append(matches, window) + } + } + if len(matches) != 1 { + return cgWindowInfo{}, fmt.Errorf("want one on-screen layer-0 CGWindow ID %d for PID %d, found %d", + windowID, pid, len(matches)) + } + return matches[0], nil +} + +func compatibleWindowBounds(left, right screenRect, tolerance float64) bool { + return math.Abs(left.X-right.X) <= tolerance && + math.Abs(left.Y-right.Y) <= tolerance && + math.Abs(left.Width-right.Width) <= tolerance && + math.Abs(left.Height-right.Height) <= tolerance +} + +func normalizeOCRText(text string) string { + return strings.Join(strings.Fields(strings.ToLower(text)), " ") +} + +func stableSummaryOCRMatch(left, right summaryOCRMatch, tolerance float64) bool { + if normalizeOCRText(left.Text) != "show performance" || + normalizeOCRText(right.Text) != "show performance" { + return false + } + lx, ly := left.center() + rx, ry := right.center() + return math.Abs(lx-rx) <= tolerance && + math.Abs(ly-ry) <= tolerance && + math.Abs(left.Width-right.Width) <= tolerance && + math.Abs(left.Height-right.Height) <= tolerance +} + +func clickScreenPoint(x, y float64) error { + down := cgEventCreateMouseEvent(0, kCGEventLeftMouseDown, x, y, 0) + if down == 0 { + return fmt.Errorf("create mouse down event") + } + defer cfRelease(down) + up := cgEventCreateMouseEvent(0, kCGEventLeftMouseUp, x, y, 0) + if up == 0 { + return fmt.Errorf("create mouse up event") + } + defer cfRelease(up) + cgEventPost(kCGHIDEventTap, down) + time.Sleep(50 * time.Millisecond) + cgEventPost(kCGHIDEventTap, up) + return nil +} diff --git a/cmd/gputrace/cmd/summary_ocr_darwin_test.go b/cmd/gputrace/cmd/summary_ocr_darwin_test.go new file mode 100644 index 00000000..4ff35c12 --- /dev/null +++ b/cmd/gputrace/cmd/summary_ocr_darwin_test.go @@ -0,0 +1,76 @@ +//go:build darwin + +package cmd + +import "testing" + +func TestStableSummaryOCRMatch(t *testing.T) { + base := summaryOCRMatch{ + Text: "Show Performance", + Confidence: 0.98, + X: 900, + Y: 700, + Width: 140, + Height: 22, + } + tests := []struct { + name string + edit func(*summaryOCRMatch) + want bool + }{ + {name: "same", want: true}, + {name: "normalized whitespace", edit: func(m *summaryOCRMatch) { m.Text = " show performance " }, want: true}, + {name: "center within tolerance", edit: func(m *summaryOCRMatch) { m.X += 3; m.Y -= 3 }, want: true}, + {name: "wrong text", edit: func(m *summaryOCRMatch) { m.Text = "Show Dependencies" }}, + {name: "center moved", edit: func(m *summaryOCRMatch) { m.X += 5 }}, + {name: "width changed", edit: func(m *summaryOCRMatch) { m.Width += 5 }}, + {name: "height changed", edit: func(m *summaryOCRMatch) { m.Height += 5 }}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + right := base + if test.edit != nil { + test.edit(&right) + } + if got := stableSummaryOCRMatch(base, right, 4); got != test.want { + t.Fatalf("stable = %v, want %v", got, test.want) + } + }) + } +} + +func TestScreenRectContainsStrictInterior(t *testing.T) { + rect := screenRect{X: 300, Y: 150, Width: 1000, Height: 700} + if !rect.contains(900, 700) { + t.Fatal("interior point rejected") + } + for _, point := range [][2]float64{ + {300, 700}, + {1300, 700}, + {900, 150}, + {900, 850}, + } { + if rect.contains(point[0], point[1]) { + t.Fatalf("edge point accepted: %v", point) + } + } +} + +func TestNormalizeOCRTextRequiresExactPhrase(t *testing.T) { + if got := normalizeOCRText(" Show Performance "); got != "show performance" { + t.Fatalf("normalized text = %q", got) + } + if got := normalizeOCRText("Show Performance Now"); got == "show performance" { + t.Fatalf("substring normalized as exact: %q", got) + } +} + +func TestCompatibleWindowBounds(t *testing.T) { + base := screenRect{X: 0, Y: 100, Width: 1376, Height: 900} + if !compatibleWindowBounds(base, screenRect{X: 1, Y: 99, Width: 1375, Height: 901}, 2) { + t.Fatal("compatible framing difference rejected") + } + if compatibleWindowBounds(base, screenRect{X: 0, Y: 100, Width: 1360, Height: 900}, 2) { + t.Fatal("materially different CG window accepted") + } +} diff --git a/cmd/gputrace/cmd/summary_test.go b/cmd/gputrace/cmd/summary_test.go new file mode 100644 index 00000000..79d64906 --- /dev/null +++ b/cmd/gputrace/cmd/summary_test.go @@ -0,0 +1,45 @@ +package cmd + +import ( + "bytes" + "strings" + "testing" + "time" + + "github.com/tmc/gputrace/internal/evidence" +) + +func TestWriteSummaryUsesCanonicalVocabulary(t *testing.T) { + report := &evidence.Report{ + CommandBuffers: 37, + ComputeEncoders: 23, + Dispatches: 1166, + CSLabels: 997, + UniqueCSLabels: 80, + DispatchSpan: 11 * time.Millisecond, + CBActiveTime: 17 * time.Millisecond, + CBWallSpan: 126 * time.Millisecond, + TimingSource: "profiler offsets", + Functions: []evidence.Function{{ + Name: "kernel", Dispatches: 3, Span: time.Millisecond, SpanShare: 9.1, + }}, + } + var out bytes.Buffer + writeSummary(&out, report, 5) + got := out.String() + for _, want := range []string{ + "23 profiler compute encoders", + "997 CS/debug label records", + "not encoder instances", + "Dispatch span", + "CB active", + "CB wall", + } { + if !strings.Contains(got, want) { + t.Errorf("summary missing %q:\n%s", want, got) + } + } + if lines := strings.Count(got, "\n"); lines > 20 { + t.Fatalf("summary has %d lines, want at most 20:\n%s", lines, got) + } +} diff --git a/cmd/gputrace/cmd/timeline.go b/cmd/gputrace/cmd/timeline.go index e0a6d556..5a0fea3c 100644 --- a/cmd/gputrace/cmd/timeline.go +++ b/cmd/gputrace/cmd/timeline.go @@ -6,24 +6,70 @@ import ( "io" "os" "path/filepath" + "runtime" "sort" + "strconv" + "strings" "github.com/spf13/cobra" "github.com/tmc/gputrace" + "github.com/tmc/gputrace/internal/buildinfo" "github.com/tmc/gputrace/internal/counter" + "github.com/tmc/gputrace/internal/mlxsemantic" + "github.com/tmc/gputrace/internal/perfetto" + "github.com/tmc/gputrace/internal/perfettosql" + "github.com/tmc/gputrace/internal/profilerraw" tracepkg "github.com/tmc/gputrace/internal/trace" ) var timelineCmd = newTimelineCommand(&timelineOptions{ - format: "text", + format: "text", + clock: timelineClockBusy, + kernelOccurrence: -1, + timeStart: -1, + timeEnd: -1, }) type timelineOptions struct { - output string - format string + output string + format string + clock timelineClock + rawProfilerSamples bool + xcodeGPUTime bool + sidecar string + hostCorrelation string + liveTiming string + openViewer bool + serveViewer bool + uiDir string + uiRevision string + remoteUI bool + listen string + maxOutputBytes int64 + sqlOutput string + kernel string + kernelOccurrence int + timeStart float64 + timeEnd float64 + navigationStartNS uint64 + navigationEndNS uint64 + selectionStartNS uint64 + selectionDurationNS uint64 } +// timelineClock selects one measured timestamp domain. The profiler records +// command buffers in wall-clock ticks and encoders in cumulative GPU-busy +// offsets. Those domains have no measured correspondence. +type timelineClock string + +const ( + timelineClockBusy timelineClock = "busy" + timelineClockWall timelineClock = "wall" + timelineClockLive timelineClock = "live" + timelineClockBoth timelineClock = "both" +) + func newTimelineCommand(opts *timelineOptions) *cobra.Command { cmd := &cobra.Command{ Use: "timeline ", @@ -38,10 +84,38 @@ func newTimelineCommand(opts *timelineOptions) *cobra.Command { Output formats: - text: Hierarchical text output to stdout - chrome: Chrome tracing format (chrome://tracing) - - perfetto: Perfetto format (ui.perfetto.dev) - same as chrome + - perfetto: Native Perfetto protobuf format (ui.perfetto.dev) - html: Interactive standalone HTML timeline viewer - json: Raw timeline data in JSON format +Native Perfetto exports include the evidence manifest, environment projection, +resource policy, and loss receipt. These are part of the timeline export, not +separate commands. Use --max-output-bytes for an explicit constrained export +and --sql-out to write the matching PerfettoSQL views. + +Lossless busy-time exports show an Xcode-like Shaders / pipelines group first, +compact encoder spans, a secondary strict encoder sequence, and a separate +group for dispatches whose encoder containment is not proven. These detail +tracks duplicate native GPU slices for presentation only. + +Capture-only launches with no profiler timing are instant track events, not GPU +duration slices. CS/debug labels remain separate observed annotations. + +A --sidecar name that disagrees with the encoder's own Metal label is reported +as a conflict. Both assertions stay visible and neither becomes the canonical +name. + +Clock domains: + - busy (default): cumulative GPU execution offsets for encoders, dispatches, + and counter series only when their clock is established + - wall: APSTimelineData command-buffer scheduling and encoder profiles + - live: original-execution command-buffer intervals from --live-timing + - both: a two-panel or two-section report containing both domains + +There is no measured mapping between cumulative GPU-busy offsets and +command-buffer wall time. The both view preserves the domains separately; it +does not place them on one shared timeline axis. + Examples: # Generate interactive HTML timeline viewer gputrace timeline trace.gputrace -o timeline.html --format html @@ -49,6 +123,15 @@ Examples: # Generate Chrome tracing format gputrace timeline trace.gputrace --format chrome -o timeline.json + # Inspect wall-clock command-buffer scheduling separately + gputrace timeline trace.gputrace --format perfetto --clock wall -o command-buffers.pftrace + + # Add Xcode Overview GPU Time without aligning the two timeline clocks + gputrace timeline trace.gputrace --format perfetto --xcode-gpu-time -o timeline.pftrace + + # Inspect both domains without inventing a clock mapping + gputrace timeline trace.gputrace --format html --clock both -o timeline.html + # View in Chrome # 1. Open chrome://tracing in Chrome # 2. Click "Load" and select timeline.json @@ -56,19 +139,47 @@ Examples: # View in Perfetto UI (recommended) # 1. Open https://ui.perfetto.dev - # 2. Drag and drop timeline.json or click "Open trace file" + # 2. Drag and drop timeline.pftrace or click "Open trace file" # 3. Use keyboard shortcuts: W/S zoom, A/D pan, F fit # Generate raw JSON for custom processing - gputrace timeline trace.gputrace -o timeline.json --format json`, + gputrace timeline trace.gputrace -o timeline.json --format json + + # Emit stable PerfettoSQL views beside a native trace + gputrace timeline trace.gputrace --format perfetto --sql-out gputrace.sql + + # Write a constrained native trace with an embedded loss receipt + gputrace timeline trace.gputrace --format perfetto \ + --max-output-bytes 500000 -o timeline.pftrace + + # Open one exact kernel occurrence + gputrace timeline trace.gputrace --format perfetto --open --remote-ui \ + --kernel rmsbfloat16 --kernel-occurrence 0`, Args: cobra.ExactArgs(1), RunE: func(cmd *cobra.Command, args []string) error { return runTimeline(cmd, args, opts) }, } - cmd.Flags().StringVarP(&opts.output, "output", "o", opts.output, "Output file path (default: stdout for text, timeline.json otherwise)") + cmd.Flags().StringVarP(&opts.output, "output", "o", opts.output, "Output file path (default: stdout for text, timeline.html for html, timeline.pftrace for perfetto, timeline.json otherwise)") cmd.Flags().StringVar(&opts.format, "format", opts.format, "Output format: chrome, perfetto, html, json, text") + cmd.Flags().Var(&opts.clock, "clock", "Timeline clock domain: busy (default), wall, live, or both") + cmd.Flags().BoolVar(&opts.rawProfilerSamples, "include-raw-samples", opts.rawProfilerSamples, "Include GPRWCNTR fixed fields in wall-clock output (hardware counter columns remain uninterpreted)") + cmd.Flags().BoolVar(&opts.xcodeGPUTime, "xcode-gpu-time", opts.xcodeGPUTime, "Read Xcode Overview GPU Time through GTShaderProfiler (Darwin only; runs a private-framework model pass)") + cmd.Flags().StringVar(&opts.sidecar, "sidecar", opts.sidecar, "Attach a strictly trace-identified MLX semantic sidecar") + cmd.Flags().StringVar(&opts.hostCorrelation, "host-correlation", opts.hostCorrelation, "Attach a trace-identified host-event correlation receipt (Perfetto only)") + cmd.Flags().StringVar(&opts.liveTiming, "live-timing", opts.liveTiming, "Attach original-execution command-buffer timing from capture --timing-sidecar") + cmd.Flags().BoolVar(&opts.openViewer, "open", opts.openViewer, "Serve the native trace and open it in Perfetto") + cmd.Flags().BoolVar(&opts.serveViewer, "serve", opts.serveViewer, "Serve the native trace without opening a browser") + cmd.Flags().StringVar(&opts.uiDir, "ui-dir", opts.uiDir, "Pinned local Perfetto UI directory containing perfetto-ui.json (with --open or --serve)") + cmd.Flags().BoolVar(&opts.remoteUI, "remote-ui", opts.remoteUI, "Embed https://ui.perfetto.dev (with --open or --serve)") + cmd.Flags().StringVar(&opts.listen, "listen", "127.0.0.1:0", "Loopback viewer listen address") + cmd.Flags().Int64Var(&opts.maxOutputBytes, "max-output-bytes", opts.maxOutputBytes, "Maximum logical native protobuf bytes; zero is lossless") + cmd.Flags().StringVar(&opts.sqlOutput, "sql-out", opts.sqlOutput, "Write the gputrace PerfettoSQL views (with --format perfetto)") + cmd.Flags().StringVar(&opts.kernel, "kernel", opts.kernel, "Focus an exact kernel name in the viewer") + cmd.Flags().IntVar(&opts.kernelOccurrence, "kernel-occurrence", -1, "Zero-based occurrence for --kernel; required when the name is repeated") + cmd.Flags().Float64Var(&opts.timeStart, "time-start", -1, "Initial viewer range start in seconds") + cmd.Flags().Float64Var(&opts.timeEnd, "time-end", -1, "Initial viewer range end in seconds") return cmd } @@ -81,6 +192,12 @@ func runTimeline(cmd *cobra.Command, args []string, opts *timelineOptions) error if err := validateTimelineFormat(opts.format); err != nil { return err } + if err := validateTimelineClock(opts.clock); err != nil { + return err + } + if err := validateTimelineSQLOutput(opts); err != nil { + return err + } // Verify trace file exists if err := checkTraceFile(tracePath); err != nil { @@ -89,9 +206,11 @@ func runTimeline(cmd *cobra.Command, args []string, opts *timelineOptions) error // Try to open full trace first trace, err := gputrace.Open(tracePath) - if err != nil { - // Fall back to profiler-only mode if unsorted-capture is missing - return runTimelineFromProfiler(tracePath, opts) + if err != nil || trace.ProfilerOnly { + // Fall back to profiler-only mode when there is no capture stream. + // Open now succeeds on such bundles, so the flag, not the error, is + // what distinguishes them. + return runTimelineFromProfiler(cmd, tracePath, opts) } // Generate timeline data @@ -99,31 +218,106 @@ func runTimeline(cmd *cobra.Command, args []string, opts *timelineOptions) error if err != nil { return fmt.Errorf("failed to generate timeline: %w", err) } + if err := enrichTimelineWithXcodeGPUTime(tracePath, timeline, opts.xcodeGPUTime); err != nil { + return err + } + if opts.format == "perfetto" || opts.format == "json" { + attachRawProfilerArtifacts(timeline, tracePath) + } + if trace.Metadata != nil { + timeline.TraceUUID = trace.Metadata.UUID + timeline.DeviceID = trace.Metadata.DeviceID + } + if opts.sidecar != "" { + uuid := "" + if trace.Metadata != nil { + uuid = trace.Metadata.UUID + } + if err := attachMLXSidecar(timeline, tracePath, uuid, opts.sidecar); err != nil { + return err + } + } + if opts.liveTiming != "" { + if err := attachLiveTiming(timeline, trace, opts.liveTiming); err != nil { + return err + } + } + if opts.clock == timelineClockLive && opts.liveTiming == "" { + return fmt.Errorf("--clock live requires --live-timing") + } + if opts.hostCorrelation != "" { + if opts.format != "perfetto" { + return fmt.Errorf("attach host correlation: --format perfetto is required") + } + if err := attachHostCorrelation(timeline, tracePath, opts.clock, opts.hostCorrelation); err != nil { + return err + } + } - // Enhance with raw GPRWCNTR data if available - if err := EnhanceTimelineWithRawData(timeline, tracePath); err != nil { - // Just warn, don't fail as this is optional/experimental - fmt.Fprintf(os.Stderr, "Warning: failed to enhance timeline with raw data: %v\n", err) - } else { - // Check if we actually added samples - sampleCount := 0 - for _, ev := range timeline.Events { - if ev.Category == "gprwcntr" { - sampleCount++ + // Add raw GPRWCNTR records from the exact ShaderProfilerData carriers. + if len(timeline.rawProfilerProfiles) > 0 { + if err := enhanceTimelineWithRawData(timeline); err != nil { + // Raw samples are an optional projection. The transactional enhancer + // publishes none when their wall coordinate cannot be established. + fmt.Fprintf(os.Stderr, "Warning: failed to enhance timeline with raw data: %v\n", err) + } else { + // Check if we actually added samples + sampleCount := 0 + for _, ev := range timeline.Events { + if ev.Category == "gprwcntr" { + sampleCount++ + } + } + if sampleCount > 0 { + if opts.rawProfilerSamples && (opts.clock == timelineClockWall || opts.clock == timelineClockBoth) { + fmt.Fprintf(cmd.ErrOrStderr(), "Raw profiler samples included: %d GPRWCNTR records\n", sampleCount) + } else { + fmt.Fprintf(cmd.ErrOrStderr(), "Raw profiler samples available: %d GPRWCNTR records (excluded by default; use --clock wall --include-raw-samples to export)\n", sampleCount) + } } - } - if sampleCount > 0 { - fmt.Fprintf(os.Stderr, "✓ Enhanced with %d GPRWCNTR samples\n", sampleCount) } } + // Warn if trace timing data is missing or approximate + if opts.clock != timelineClockLive && (timeline.Timing == nil || timeline.Timing.EncoderTimingApproximate || timeline.Timing.TimingSource == "" || timeline.Timing.TimingSource == "unavailable") { + fmt.Fprintf(cmd.ErrOrStderr(), "Warning: trace lacks precise hardware timing data; encoder/dispatch durations are estimated.\n") + fmt.Fprint(cmd.ErrOrStderr(), profileReplayHint(tracePath)) + } + outputPath := timelineOutputPath(opts.format, opts.output) + if err := validateTimelineViewerOptions(opts, outputPath); err != nil { + return err + } + if opts.clock == timelineClockBoth { + if err := exportTimelineBothWithRawSamples(timeline, opts.format, outputPath, opts.rawProfilerSamples); err != nil { + return err + } + if opts.format != "text" || (outputPath != "" && !commandOutputPathIsStdout(outputPath)) { + printTimelineExportStatus(outputPath, opts.format, false) + } + return nil + } + fullTimeline := timeline // Keep pre-clock-filtered timeline for text export wall-time gaps. + timeline = timelineForClockWithRawSamples(timeline, opts.clock, opts.rawProfilerSamples) + if err := resolveTimelineNavigation(timeline, opts); err != nil { + return err + } + if opts.format == "chrome" || opts.format == "perfetto" { + warnDroppedDispatchEvents(cmd.ErrOrStderr(), fullTimeline) + } // Export based on format switch opts.format { - case "chrome", "perfetto": - if err := exportChromeTracing(timeline, outputPath); err != nil { - return fmt.Errorf("failed to export Chrome/Perfetto tracing: %w", err) + case "chrome": + if err := exportChromeTracingForClock(timeline, outputPath, opts.clock); err != nil { + return fmt.Errorf("failed to export Chrome tracing: %w", err) + } + case "perfetto": + if err := exportPerfettoForClockWithBudget(timeline, outputPath, opts.clock, opts.maxOutputBytes); err != nil { + return fmt.Errorf("failed to export Perfetto tracing: %w", err) + } + if err := writeTimelinePerfettoSQL(opts.sqlOutput); err != nil { + return err } case "html": if err := exportHTML(timeline, outputPath); err != nil { @@ -134,7 +328,7 @@ func runTimeline(cmd *cobra.Command, args []string, opts *timelineOptions) error return fmt.Errorf("failed to export JSON: %w", err) } case "text": - if err := exportTextTimeline(timeline, outputPath); err != nil { + if err := exportTextTimeline(timeline, fullTimeline, outputPath); err != nil { return fmt.Errorf("failed to export text: %w", err) } if outputPath != "" && !commandOutputPathIsStdout(outputPath) { @@ -146,9 +340,134 @@ func runTimeline(cmd *cobra.Command, args []string, opts *timelineOptions) error } printTimelineExportStatus(outputPath, opts.format, false) + return serveTimelinePerfetto(cmd, tracePath, outputPath, opts) +} + +func validateTimelineClock(clock timelineClock) error { + switch clock { + case timelineClockBusy, timelineClockWall, timelineClockLive, timelineClockBoth: + return nil + default: + return fmt.Errorf("invalid timeline clock %q (supported: busy, wall, live, both)", clock) + } +} + +func validateTimelineSQLOutput(opts *timelineOptions) error { + if opts.sqlOutput != "" && opts.format != "perfetto" { + return fmt.Errorf("--sql-out requires --format perfetto") + } + return nil +} + +func writeTimelinePerfettoSQL(path string) error { + if path == "" { + return nil + } + w, closeOutput, err := createCommandOutput(path) + if err != nil { + return fmt.Errorf("write PerfettoSQL views: %w", err) + } + if closeOutput != nil { + defer closeOutput() + } + if err := perfettosql.Write(w); err != nil { + return err + } + return nil +} + +// Set implements pflag.Value. +func (c *timelineClock) Set(value string) error { + clock := timelineClock(value) + if err := validateTimelineClock(clock); err != nil { + return err + } + *c = clock return nil } +func (c *timelineClock) Type() string { return "clock" } + +func (c *timelineClock) String() string { return string(*c) } + +// timelineForClock copies the timeline and retains only events whose +// timestamps have the requested meaning. A wall-clock coordinate is never +// inferred for a cumulative GPU-busy event, or vice versa. +func timelineForClock(timeline *Timeline, clock timelineClock) *Timeline { + return timelineForClockWithRawSamples(timeline, clock, true) +} + +// timelineForClockWithRawSamples filters a timeline to one measured clock +// domain. Raw GPRWCNTR records are opt-in because only their fixed GRC fields, +// not the hardware counter columns, have been decoded. +func timelineForClockWithRawSamples(timeline *Timeline, clock timelineClock, rawProfilerSamples bool) *Timeline { + if timeline == nil { + return nil + } + selected := *timeline + if selected.EvidenceInventory == nil { + inventory := timelineEvidenceInventory(timeline) + selected.EvidenceInventory = &inventory + } + selected.ClockDomain = string(clock) + selected.RawProfilerSamples = clock == timelineClockWall && rawProfilerSamples + selected.Events = make([]TimelineEvent, 0, len(timeline.Events)) + for _, event := range timeline.Events { + if timelineEventInClockWithRawSamples(event, clock, rawProfilerSamples) { + selected.Events = append(selected.Events, event) + } + } + if clock == timelineClockWall || clock == timelineClockLive { + selected.Encoders = []EncoderInfo{} + selected.Kernels = []KernelInfo{} + selected.CounterTracks = []CounterTrack{} + } else { + tracks := make([]CounterTrack, 0, len(timeline.CounterTracks)) + for _, track := range timeline.CounterTracks { + if counterTrackHasSignal(track) { + tracks = append(tracks, track) + } + } + selected.CounterTracks = tracks + } + // API calls are not timestamped in this capture, so neither selected clock + // can place them honestly. Keep them out of raw and HTML exports too. + selected.APICallseq = []APICall{} + selected.StartTime = 0 + selected.EndTime = 0 + for _, event := range selected.Events { + if end := (event.Timestamp + event.Duration) * 1000; end > selected.EndTime { + selected.EndTime = end + } + } + for _, track := range selected.CounterTracks { + for _, sample := range track.Samples { + if sample.Timestamp > selected.EndTime { + selected.EndTime = sample.Timestamp + } + } + } + selected.Duration = selected.EndTime + return &selected +} + +func timelineEventInClock(event TimelineEvent, clock timelineClock) bool { + return timelineEventInClockWithRawSamples(event, clock, true) +} + +func timelineEventInClockWithRawSamples(event TimelineEvent, clock timelineClock, rawProfilerSamples bool) bool { + switch clock { + case timelineClockBusy: + return event.Category == "encoder" || event.Category == "kernel" || event.Category == "dispatch" + case timelineClockWall: + return event.Category == "command_buffer" || event.Category == "restore" || (rawProfilerSamples && (event.Category == "profiler_stream" || event.Category == "gprwcntr")) + case timelineClockLive: + return event.Category == "live_command_buffer" + default: + return false + } +} + func validateTimelineFormat(format string) error { switch format { case "chrome", "perfetto", "html", "json", "text": @@ -158,10 +477,19 @@ func validateTimelineFormat(format string) error { } } +// timelineOutputPath picks the default output file for a format. text goes to +// stdout, so it has none. html gets an .html name: writing a whole HTML +// document into timeline.json leaves a file no viewer will open. func timelineOutputPath(format, output string) string { if output != "" || format == "text" { return output } + if format == "html" { + return "timeline.html" + } + if format == "perfetto" { + return "timeline.pftrace" + } return "timeline.json" } @@ -170,7 +498,7 @@ func printTimelineExportStatus(output, format string, profilerOnly bool) { if profilerOnly { suffix = " (profiler-only mode)" } - fmt.Fprintf(os.Stderr, "✓ Timeline written to: %s%s\n", output, suffix) + fmt.Fprintf(os.Stderr, "Timeline written: %s%s\n", output, suffix) if format == "chrome" { fmt.Fprintln(os.Stderr, "\nView in Chrome:") fmt.Fprintln(os.Stderr, " 1. Open chrome://tracing") @@ -188,7 +516,9 @@ func printTimelineExportStatus(output, format string, profilerOnly bool) { } // exportTextTimeline writes the timeline in a hierarchical text format. -func exportTextTimeline(timeline *Timeline, outputPath string) error { +// fullTimeline is the pre-clock-filtered timeline used to extract wall-clock +// command buffer events for showing idle gaps. It may be nil. +func exportTextTimeline(timeline, fullTimeline *Timeline, outputPath string) error { w, closeOutput, err := createCommandOutput(outputPath) if err != nil { return err @@ -196,12 +526,32 @@ func exportTextTimeline(timeline *Timeline, outputPath string) error { if closeOutput != nil { defer closeOutput() } + return writeTextTimeline(w, timeline, fullTimeline) +} +func writeTextTimeline(w io.Writer, timeline *Timeline, fullTimeline ...*Timeline) error { if len(timeline.Encoders) == 0 && len(timeline.Events) == 0 { fmt.Fprintln(w, "No timeline data available.") return nil } + fmt.Fprintln(w, "GPU Timeline") + if timeline.TracePath != "" { + fmt.Fprintf(w, "Trace: %s\n", timeline.TracePath) + } + + // Find command buffer events before printing the summary. + var cbs []TimelineEvent + for _, event := range timeline.Events { + if event.Category == "command_buffer" { + cbs = append(cbs, event) + } + } + fmt.Fprintf(w, "Events: %d %s, %d %s, %d %s\n", + len(cbs), Pluralize(len(cbs), "command buffer", "command buffers"), + len(timeline.Encoders), Pluralize(len(timeline.Encoders), "encoder", "encoders"), + len(timeline.Kernels), Pluralize(len(timeline.Kernels), "kernel dispatch", "kernel dispatches")) + if timeline.Timing != nil && timeline.Timing.EncoderTimingSource != "" { sourceKind := "measured" if timeline.Timing.EncoderTimingApproximate { @@ -209,193 +559,627 @@ func exportTextTimeline(timeline *Timeline, outputPath string) error { } fmt.Fprintf(w, "Timing source: %s (%s)\n", timeline.Timing.EncoderTimingSource, sourceKind) } + if timing := timeline.Timing; timing != nil { + if timing.EncoderSpanNs > 0 { + fmt.Fprintf(w, "Encoder span: %s\n", FormatDurationNs(timing.EncoderSpanNs)) + } + if timing.DispatchSpanNs > 0 { + fmt.Fprintf(w, "Dispatch span: %s\n", FormatDurationNs(timing.DispatchSpanNs)) + } + if timing.CommandBufferActiveNs > 0 { + fmt.Fprintf(w, "Command-buffer active time: %s\n", FormatDurationNs(timing.CommandBufferActiveNs)) + } + if timing.CommandBufferWallNs > 0 { + fmt.Fprintf(w, "Command-buffer wall span: %s\n", FormatDurationNs(timing.CommandBufferWallNs)) + } + if timing.EffectiveGPUTimeNs != nil { + fmt.Fprintf(w, "Xcode Effective GPU Time: %s\n", FormatDurationNs(*timing.EffectiveGPUTimeNs)) + } else if !timing.EncoderTimingApproximate { + fmt.Fprintln(w, "Xcode Effective GPU Time: unavailable") + } + } + fmt.Fprintln(w, "Row units: start and duration are milliseconds; capture-only coordinates are byte offsets.") + fmt.Fprintln(w) - // Find command buffer events - var cbs []TimelineEvent - for _, event := range timeline.Events { - if event.Category == "command_buffer" { - cbs = append(cbs, event) + // Extract wall-clock CB events from the full timeline to annotate gaps. + var wallCBs []TimelineEvent + if len(fullTimeline) > 0 && fullTimeline[0] != nil { + for _, event := range fullTimeline[0].Events { + if event.Category == "command_buffer" { + wallCBs = append(wallCBs, event) + } + } + } + // Build wall-CB lookup by index for gap annotation. + wallCBByIndex := make(map[int]TimelineEvent) + hasRealWallTiming := false + for _, wcb := range wallCBs { + if idx, ok := wcb.Args["index"].(int); ok { + wallCBByIndex[idx] = wcb + } + if wcb.Timestamp > 0 { + hasRealWallTiming = true + } + } + + // When the busy-clock timeline has no native CB events but the full + // timeline has wall-clock CBs, use those as the structural grouping. + // Each encoder maps 1:1 by index to a wall-clock CB. + if len(cbs) == 0 && len(wallCBs) > 0 { + // Build an encoder lookup by index. + encoderByIndex := make(map[int]EncoderInfo) + for _, enc := range timeline.Encoders { + encoderByIndex[enc.Index] = enc + } + + firstTimestamp := timeline.StartTime + + var prevWallEndUs uint64 + for i, wcb := range wallCBs { + cbIndex, ok := wcb.Args["index"].(int) + if !ok { + continue + } + + // Show idle gap between consecutive command buffers. + if hasRealWallTiming && i > 0 && prevWallEndUs > 0 && wcb.Timestamp > prevWallEndUs { + gapUs := wcb.Timestamp - prevWallEndUs + gapMs := float64(gapUs) / 1000.0 + fmt.Fprintf(w, " ⏳ idle gap: %.2fms (wall time)\n", gapMs) + } + if hasRealWallTiming { + prevWallEndUs = wcb.Timestamp + wcb.Duration + } + + if hasRealWallTiming { + wallMs := float64(wcb.Timestamp) / 1000.0 + wallDurMs := float64(wcb.Duration) / 1000.0 + fmt.Fprintf(w, "%s [wall=%.2fms, wall-dur=%.2fms]\n", wcb.Name, wallMs, wallDurMs) + } else { + fmt.Fprintf(w, "%s\n", wcb.Name) + } + + // Find the encoder for this CB index and print it with its kernels. + enc, found := encoderByIndex[cbIndex] + if !found { + continue + } + writeTimelineEncoders(w, timeline, []EncoderInfo{enc}, firstTimestamp) } + + // Show any encoders that didn't map to a wall-clock CB. + mapped := make(map[int]bool) + for _, wcb := range wallCBs { + if idx, ok := wcb.Args["index"].(int); ok { + mapped[idx] = true + } + } + var unmapped []EncoderInfo + for _, enc := range timeline.Encoders { + if !mapped[enc.Index] { + unmapped = append(unmapped, enc) + } + } + if len(unmapped) > 0 { + fmt.Fprintf(w, "\nEncoders not attributed to a command buffer (%d):\n", len(unmapped)) + writeTimelineEncoders(w, timeline, unmapped, timeline.StartTime) + } + return nil } - // If no CB events, create a dummy one + // If no CB events at all, create a dummy one. if len(cbs) == 0 { cbs = append(cbs, TimelineEvent{ Name: "CB#0", Timestamp: timeline.StartTime, - Duration: timeline.Duration, + Duration: timeline.Duration / 1000, // Timeline duration is ns; events use µs. + Args: map[string]interface{}{"index": 0}, }) } + encodersByCB, unattributed := attributeEncodersToCBs(timeline, cbs) + firstTimestamp := timeline.StartTime if len(cbs) > 0 && cbs[0].Timestamp < firstTimestamp { firstTimestamp = cbs[0].Timestamp } - for _, cb := range cbs { - var cbStart float64 - if cb.Timestamp >= firstTimestamp { - cbStart = float64(cb.Timestamp-firstTimestamp) / 1000000.0 - } else { - cbStart = 0.0 - } - // Show duration if available (from APSTimelineData) - if cb.Duration > 0 { - cbDurationMs := float64(cb.Duration) / 1000.0 // Duration is in µs, convert to ms - fmt.Fprintf(w, "%s [%.1fms, duration=%.2fms]\n", cb.Name, cbStart, cbDurationMs) - } else { - fmt.Fprintf(w, "%s [%.1fms]\n", cb.Name, cbStart) - } - + var prevWallEndUs uint64 + for i, cb := range cbs { cbIndex, ok := cb.Args["index"].(int) - if !ok { - continue - } - var cbEncoders []EncoderInfo - for _, encoder := range timeline.Encoders { - belongsToCB := false - for _, k := range timeline.Kernels { - if k.Encoder == encoder.Index { - if kArgCB, ok := getKernelCBIndex(timeline, k); ok && kArgCB == cbIndex { - belongsToCB = true - break - } + // Annotate wall-time gap between consecutive command buffers. + if ok && hasRealWallTiming && len(wallCBs) > 0 { + if wcb, found := wallCBByIndex[cbIndex]; found { + if i > 0 && prevWallEndUs > 0 && wcb.Timestamp > prevWallEndUs { + gapUs := wcb.Timestamp - prevWallEndUs + gapMs := float64(gapUs) / 1000.0 + fmt.Fprintf(w, " ⏳ idle gap: %.2fms (wall time)\n", gapMs) } - } - if belongsToCB { - cbEncoders = append(cbEncoders, encoder) + prevWallEndUs = wcb.Timestamp + wcb.Duration } } - for i, encoder := range cbEncoders { - startMs := float64(encoder.StartTime-firstTimestamp) / 1e6 - durationMs := float64(encoder.Duration) / 1e6 - - label := encoder.Label - if label == "" { - label = "Unknown Encoder" + if source, _ := cb.Args["coordinate_source"].(string); source == "capture byte offset" { + fmt.Fprintf(w, "%s [capture offset %v]\n", cb.Name, cb.Args["offset"]) + } else { + var cbStart float64 + if cb.Timestamp >= firstTimestamp { + cbStart = float64(cb.Timestamp-firstTimestamp) / 1000.0 } - - var encoderKernels []KernelInfo - for _, k := range timeline.Kernels { - if k.Encoder == encoder.Index { - encoderKernels = append(encoderKernels, k) + // Show duration and wall-time anchor if available. + var wallNote string + if ok { + if wcb, found := wallCBByIndex[cbIndex]; found { + wallMs := float64(wcb.Timestamp) / 1000.0 + wallDurMs := float64(wcb.Duration) / 1000.0 + wallNote = fmt.Sprintf(", wall=%.2fms/%.2fms", wallMs, wallDurMs) } } - - prefix := "├─" - if i == len(cbEncoders)-1 { - prefix = "└─" - } - - if len(encoderKernels) > 0 { - for _, k := range encoderKernels { - kStartMs := float64(k.StartTime-firstTimestamp) / 1e6 - kDurationMs := float64(k.Duration) / 1e6 - fmt.Fprintf(w, " %s %.2fms: %s (%.2fms) - %s\n", - prefix, kStartMs, k.Name, kDurationMs, label) - } + if cb.Duration > 0 { + cbDurationMs := float64(cb.Duration) / 1000.0 // Duration is in µs, convert to ms + fmt.Fprintf(w, "%s [%.1fms, duration=%.2fms%s]\n", cb.Name, cbStart, cbDurationMs, wallNote) } else { - fmt.Fprintf(w, " %s %.2fms: %s (%.2fms) - %s\n", prefix, startMs, label, durationMs, "Encoder") + fmt.Fprintf(w, "%s [%.1fms, duration unavailable: no end timestamp%s]\n", cb.Name, cbStart, wallNote) } } + + if !ok { + continue + } + + writeTimelineEncoders(w, timeline, encodersByCB[cbIndex], firstTimestamp) + } + + if len(unattributed) > 0 { + fmt.Fprintf(w, "\nEncoders not attributed to a command buffer (%d):\n", len(unattributed)) + writeTimelineEncoders(w, timeline, unattributed, firstTimestamp) } return nil } -func getKernelCBIndex(timeline *Timeline, k KernelInfo) (int, bool) { - for _, e := range timeline.Events { - if e.Category == "kernel" && e.Name == k.Name && e.Timestamp == k.StartTime/1000 { - if cbIdx, ok := e.Args["cb_index"].(int); ok { - return cbIdx, true +// writeTimelineEncoders prints one command buffer's encoders and the kernels +// each ran, as a nested tree structure under the command buffer line. +func writeTimelineEncoders(w io.Writer, timeline *Timeline, encoders []EncoderInfo, firstTimestamp uint64) { + for i, encoder := range encoders { + startMs := float64(encoder.StartTime-firstTimestamp) / 1e6 + durationMs := float64(encoder.Duration) / 1e6 + + label := encoder.Label + if label == "" { + label = fmt.Sprintf("Encoder#%d", encoder.Index) + } + + var encoderKernels []KernelInfo + for _, k := range timeline.Kernels { + if k.Encoder == encoder.Index { + encoderKernels = append(encoderKernels, k) + } + } + + isLastEncoder := (i == len(encoders)-1) + encPrefix := "├─" + pipePrefix := "│ " + if isLastEncoder { + encPrefix = "└─" + pipePrefix = " " + } + + fmt.Fprintf(w, "%s %s [%.2fms, duration=%.2fms]\n", encPrefix, label, startMs, durationMs) + + for j, k := range encoderKernels { + kStartMs := float64(k.StartTime-firstTimestamp) / 1e6 + kDurationMs := float64(k.Duration) / 1e6 + isLastKernel := (j == len(encoderKernels)-1) + kPrefix := "├─" + if isLastKernel { + kPrefix = "└─" } + fmt.Fprintf(w, "%s %s %.2fms: %s (%.2fms)\n", pipePrefix, kPrefix, kStartMs, k.Name, kDurationMs) } } - return -1, false } -// Timeline represents the complete timeline data. -type Timeline struct { - StartTime uint64 `json:"start_time"` - EndTime uint64 `json:"end_time"` - Duration uint64 `json:"duration"` - Events []TimelineEvent `json:"events"` - Encoders []EncoderInfo `json:"encoders"` - Kernels []KernelInfo `json:"kernels"` - APICallseq []APICall `json:"api_callseq"` - CounterTracks []CounterTrack `json:"counter_tracks,omitempty"` - Timing *TimelineTiming `json:"timing,omitempty"` - XcodeMetrics map[string]any `json:"xcode_metrics,omitempty"` - AbsoluteTime uint64 `json:"absolute_time"` - TimebaseNumer uint64 `json:"timebase_numer"` - TimebaseDenom uint64 `json:"timebase_denom"` +// exportTimelineBoth writes both measured clock domains without assigning a +// timestamp in either domain to data recorded only in the other. +func exportTimelineBoth(timeline *Timeline, format, outputPath string) error { + return exportTimelineBothWithRawSamples(timeline, format, outputPath, false) } -// TimelineTiming summarizes the timing sources that Xcode and gputrace expose. -type TimelineTiming struct { - EncoderSpanNs uint64 `json:"encoder_span_ns,omitempty"` - DispatchSpanNs uint64 `json:"dispatch_span_ns,omitempty"` - EffectiveGPUTimeNs *uint64 `json:"effective_gpu_time_ns,omitempty"` - CommandBufferActiveNs uint64 `json:"command_buffer_active_time_ns,omitempty"` - CommandBufferWallNs uint64 `json:"command_buffer_wall_time_ns,omitempty"` - RestoreActiveNs uint64 `json:"restore_active_time_ns,omitempty"` - RestoreWallNs uint64 `json:"restore_wall_time_ns,omitempty"` - DisplayDurationNs uint64 `json:"display_duration_ns,omitempty"` - DisplayDurationSource string `json:"display_duration_source,omitempty"` - TimingSource string `json:"timing_source,omitempty"` - EncoderTimingSource string `json:"encoder_timing_source,omitempty"` - EncoderTimingApproximate bool `json:"encoder_timing_approximate"` -} +func exportTimelineBothWithRawSamples(timeline *Timeline, format, outputPath string, rawProfilerSamples bool) error { + busy := timelineForClockWithRawSamples(timeline, timelineClockBusy, rawProfilerSamples) + wall := timelineForClockWithRawSamples(timeline, timelineClockWall, rawProfilerSamples) -// TimelineEvent represents a single event in the timeline. -type TimelineEvent struct { - Name string `json:"name"` - Category string `json:"cat,omitempty"` - Phase string `json:"ph"` // B, E, X, i, M - Timestamp uint64 `json:"ts"` - Duration uint64 `json:"dur,omitempty"` - ProcessID int `json:"pid"` - ThreadID int `json:"tid"` - Args map[string]interface{} `json:"args,omitempty"` + switch format { + case "text": + return exportTextTimelineBoth(busy, wall, outputPath) + case "json": + return exportTimelineJSONBoth(busy, wall, outputPath) + case "html": + return exportHTMLBoth(busy, wall, outputPath) + case "chrome", "perfetto": + return fmt.Errorf("--clock both cannot be represented in one %s trace: it has one global time axis; use --format html or json, or export busy and wall separately", format) + default: + return validateTimelineFormat(format) + } } -// EncoderInfo contains information about an encoder. -type EncoderInfo struct { - Index int `json:"index"` - Label string `json:"label"` - Type string `json:"type"` - StartTime uint64 `json:"start_time"` - EndTime uint64 `json:"end_time"` - Duration uint64 `json:"duration"` -} +func exportTextTimelineBoth(busy, wall *Timeline, outputPath string) error { + w, closeOutput, err := createCommandOutput(outputPath) + if err != nil { + return err + } + if closeOutput != nil { + defer closeOutput() + } -// KernelInfo contains information about a kernel execution. -type KernelInfo struct { - Name string `json:"name"` - Encoder int `json:"encoder"` - StartTime uint64 `json:"start_time"` - EndTime uint64 `json:"end_time"` - Duration uint64 `json:"duration"` - Args map[string]interface{} `json:"args,omitempty"` + fmt.Fprintln(w, "GPU Timeline: two independent clock domains") + fmt.Fprintln(w, "Busy time is cumulative GPU execution. Wall time is command-buffer scheduling.") + fmt.Fprintln(w, "No timestamp mapping between them is present in this trace.") + fmt.Fprintln(w) + fmt.Fprintln(w, "=== GPU busy time ===") + if err := writeTextTimeline(w, busy); err != nil { + return err + } + fmt.Fprintln(w, "\n=== Wall-clock scheduling ===") + return writeTextTimeline(w, wall) } -// APICall represents an API call event. -type APICall struct { - Name string `json:"name"` - Timestamp uint64 `json:"timestamp"` - Args map[string]interface{} `json:"args,omitempty"` +type timelineBothJSON struct { + ClockDomain string `json:"clock_domain"` + ClockMapping string `json:"clock_mapping"` + Busy *Timeline `json:"busy"` + Wall *Timeline `json:"wall"` +} + +func exportTimelineJSONBoth(busy, wall *Timeline, outputPath string) error { + f, closeOutput, err := createCommandOutput(outputPath) + if err != nil { + return err + } + if closeOutput != nil { + defer closeOutput() + } + + encoder := json.NewEncoder(f) + encoder.SetIndent("", " ") + return encoder.Encode(timelineBothJSON{ + ClockDomain: string(timelineClockBoth), + ClockMapping: "none: busy and wall timestamps are independently measured and not aligned", + Busy: busy, + Wall: wall, + }) +} + +// attributeEncodersToCBs groups encoders under the command buffer they ran in. +// Encoders that cannot be placed are returned separately so the report still +// accounts for them. +// +// streamData records no encoder-to-command-buffer link. Dispatch and command +// buffer timestamps do share one absolute GPU tick base, so a dispatch that +// starts inside a command buffer's tick window ran in that command buffer, and +// its encoder did too. A trace with a single command buffer needs no ticks: +// every encoder can only belong to it. +func attributeEncodersToCBs(timeline *Timeline, cbs []TimelineEvent) (map[int][]EncoderInfo, []EncoderInfo) { + byCB := make(map[int][]EncoderInfo) + var unattributed []EncoderInfo + + soleCB, hasSoleCB := -1, false + if len(cbs) == 1 { + soleCB, hasSoleCB = timelineEventArgInt(cbs[0].Args, "index") + } + + for _, encoder := range timeline.Encoders { + cbIndex, found := -1, false + for _, k := range timeline.Kernels { + if k.Encoder != encoder.Index { + continue + } + if idx, ok := kernelCBIndex(cbs, k); ok { + cbIndex, found = idx, true + break + } + } + if !found && hasSoleCB { + cbIndex, found = soleCB, true + } + if found { + byCB[cbIndex] = append(byCB[cbIndex], encoder) + } else { + unattributed = append(unattributed, encoder) + } + } + return byCB, unattributed +} + +// kernelCBIndex reports the command buffer whose tick window contains the +// kernel's start tick. Kernels synthesized from an encoder span carry no +// ticks and cannot be placed this way. +func kernelCBIndex(cbs []TimelineEvent, k KernelInfo) (int, bool) { + start, ok := timelineEventArgUint64(k.Args, "start_ticks") + if !ok || start == 0 { + return -1, false + } + for _, cb := range cbs { + cbStart, okStart := timelineEventArgUint64(cb.Args, "start_ticks") + cbEnd, okEnd := timelineEventArgUint64(cb.Args, "end_ticks") + if !okStart || !okEnd || cbEnd < cbStart { + continue + } + if start < cbStart || start > cbEnd { + continue + } + if idx, ok := timelineEventArgInt(cb.Args, "index"); ok { + return idx, true + } + } + return -1, false +} + +// timelineEventArgUint64 reads a tick count from event args. Args round-trip +// through JSON, where every number decodes as float64, so both forms are +// accepted. +func timelineEventArgUint64(args map[string]interface{}, key string) (uint64, bool) { + switch v := args[key].(type) { + case uint64: + return v, true + case int: + if v < 0 { + return 0, false + } + return uint64(v), true + case float64: + if v < 0 { + return 0, false + } + return uint64(v), true + } + return 0, false +} + +func timelineEventArgInt(args map[string]interface{}, key string) (int, bool) { + switch v := args[key].(type) { + case int: + return v, true + case float64: + return int(v), true + } + return 0, false +} + +// Timeline represents the complete timeline data. +type Timeline struct { + TracePath string `json:"trace_path,omitempty"` + ClockDomain string `json:"clock_domain,omitempty"` + RawProfilerSamples bool `json:"raw_profiler_samples,omitempty"` + StartTime uint64 `json:"start_time"` + EndTime uint64 `json:"end_time"` + Duration uint64 `json:"duration"` + Events []TimelineEvent `json:"events"` + Encoders []EncoderInfo `json:"encoders"` + Kernels []KernelInfo `json:"kernels"` + APICallseq []APICall `json:"api_callseq"` + CounterTracks []CounterTrack `json:"counter_tracks,omitempty"` + UnattributedCounters []UnattributedCounterMetric `json:"unattributed_counters,omitempty"` + CounterCatalog []CounterCatalogEntry `json:"counter_catalog,omitempty"` + CounterTraceIDs []CounterTraceIDEntry `json:"counter_trace_ids,omitempty"` + CounterEncoderAggregates []counter.EncoderSamples `json:"counter_encoder_aggregates,omitempty"` + CounterEncoderSamples []counter.AttributedCounterSample `json:"counter_encoder_samples,omitempty"` + UnavailableEvidence []UnavailableEvidence `json:"unavailable_evidence,omitempty"` + Timing *TimelineTiming `json:"timing,omitempty"` + XcodeMetrics map[string]any `json:"xcode_metrics,omitempty"` + AbsoluteTime uint64 `json:"absolute_time"` + ContinuousTime uint64 `json:"continuous_time,omitempty"` + PState *int `json:"pstate,omitempty"` + TimebaseNumer uint64 `json:"timebase_numer"` + TimebaseDenom uint64 `json:"timebase_denom"` + MLXSemantics *mlxsemantic.Sidecar `json:"mlx_semantics,omitempty"` + MLXSemanticReport *mlxsemantic.Report `json:"mlx_semantic_report,omitempty"` + MLXSidecarDigest string `json:"mlx_sidecar_digest,omitempty"` + MLXSemanticLabelConflict []MLXLabelConflict `json:"mlx_semantic_label_conflicts,omitempty"` + HostCorrelation *hostCorrelationProjection `json:"host_correlation,omitempty"` + LiveTiming *liveTimingProjection `json:"live_timing,omitempty"` + TraceUUID string `json:"trace_uuid,omitempty"` + DeviceID int `json:"device_id,omitempty"` + GPUGeneration *uint32 `json:"gpu_generation,omitempty"` + MetalDeviceName string `json:"metal_device_name,omitempty"` + MetalPluginName string `json:"metal_plugin_name,omitempty"` + PipelineCompilerStats []counter.PipelineStats `json:"pipeline_compiler_stats,omitempty"` + PipelineCompilerSource string `json:"pipeline_compiler_source,omitempty"` + StreamDataStrings []string `json:"stream_data_strings,omitempty"` + StreamMetadata *counter.StreamDataMetadata `json:"stream_metadata,omitempty"` + ObservedCSLabels int `json:"observed_cs_labels,omitempty"` + UniqueCSLabels int `json:"unique_cs_labels,omitempty"` + EvidenceInventory *TimelineEvidenceInventory `json:"evidence_inventory,omitempty"` + RawProfilerArtifacts *profilerraw.ArtifactInventory `json:"raw_profiler_artifacts,omitempty"` + RawProfilerArtifactError string `json:"raw_profiler_artifact_error,omitempty"` + rawProfilerProfiles []counter.EncoderProfile + // bundleDispatches is how many dispatches the bundle records, from either + // evidence source. An export that emits no dispatch slice needs it to say + // whether there was anything to emit. + bundleDispatches int +} + +// timelineBundleDispatchCount counts the dispatches the bundle records. +// streamData is preferred because it is what the dispatch lane is built from; +// the capture records are the fallback source. +func timelineBundleDispatchCount(stats *counter.StreamDataStats, trace *gputrace.Trace) int { + if stats != nil && len(stats.Dispatches) > 0 { + return len(stats.Dispatches) + } + if trace != nil { + if dispatches, err := trace.ParseAttributedDispatches(); err == nil { + return len(dispatches) + } + } + return 0 +} + +// warnDroppedDispatchEvents reports an export that carries no dispatch slice +// at all although the bundle recorded dispatches. Such a trace opens in the +// viewer looking like a run with no GPU work rather than one whose work could +// not be placed. +func warnDroppedDispatchEvents(w io.Writer, timeline *Timeline) { + if timeline == nil || timeline.bundleDispatches == 0 { + return + } + for _, event := range timeline.Events { + if event.Category == "kernel" || event.Category == "dispatch" { + return + } + } + fmt.Fprintf(w, "Warning: all %d dispatches were dropped from this export: the bundle carries\n"+ + "no capture dispatch records and no streamData dispatch timing to place them on,\n"+ + "so the exported trace holds encoder and command-buffer spans only.\n", + timeline.bundleDispatches) +} + +func attachRawProfilerArtifacts(timeline *Timeline, tracePath string) { + if timeline == nil { + return + } + dir := profilerraw.FindDir(tracePath) + if dir == "" { + return + } + inventory, err := profilerraw.Inventory(dir) + if err != nil { + timeline.RawProfilerArtifactError = err.Error() + return + } + timeline.RawProfilerArtifacts = inventory +} + +// TimelineEvidenceInventory counts source records before clock filtering. +// Projected event counts are reported separately by each exporter. +type TimelineEvidenceInventory struct { + CommandBuffers int `json:"command_buffers"` + RestoreIntervals int `json:"restore_intervals"` + Encoders int `json:"encoders"` + Dispatches int `json:"dispatches"` + ProfilerStreams int `json:"raw_profiler_streams"` + ProfilerRecords int `json:"raw_profiler_records"` + UntimedDispatches int `json:"untimed_dispatches"` +} + +func timelineEvidenceInventory(timeline *Timeline) TimelineEvidenceInventory { + if timeline == nil { + return TimelineEvidenceInventory{} + } + return TimelineEvidenceInventory{ + CommandBuffers: timelineEventCount(timeline, "command_buffer"), + RestoreIntervals: timelineEventCount(timeline, "restore"), + Encoders: timelineEventCount(timeline, "encoder"), + Dispatches: timelineEventCount(timeline, "kernel") + timelineEventCount(timeline, "dispatch"), + ProfilerStreams: timelineEventCount(timeline, "profiler_stream"), + ProfilerRecords: timelineEventCount(timeline, "gprwcntr"), + UntimedDispatches: timelineUntimedDispatchCount(timeline), + } +} + +// UnavailableEvidence records an evidence family that could not be projected +// without inventing an identity or clock relationship. +type UnavailableEvidence struct { + Family string `json:"family"` + Reason string `json:"reason"` +} + +// TimelineTiming summarizes the timing sources that Xcode and gputrace expose. +type TimelineTiming struct { + EncoderSpanNs uint64 `json:"encoder_span_ns,omitempty"` + DispatchSpanNs uint64 `json:"dispatch_span_ns,omitempty"` + EffectiveGPUTimeNs *uint64 `json:"effective_gpu_time_ns,omitempty"` + CommandBufferActiveNs uint64 `json:"command_buffer_active_time_ns,omitempty"` + CommandBufferWallNs uint64 `json:"command_buffer_wall_time_ns,omitempty"` + RestoreActiveNs uint64 `json:"restore_active_time_ns,omitempty"` + RestoreWallNs uint64 `json:"restore_wall_time_ns,omitempty"` + DisplayDurationNs uint64 `json:"display_duration_ns,omitempty"` + DisplayDurationSource string `json:"display_duration_source,omitempty"` + TimingSource string `json:"timing_source,omitempty"` + EncoderTimingSource string `json:"encoder_timing_source,omitempty"` + EncoderTimingApproximate bool `json:"encoder_timing_approximate"` +} + +// TimelineEvent represents a single event in the timeline. +type TimelineEvent struct { + Name string `json:"name"` + Category string `json:"cat,omitempty"` + Phase string `json:"ph"` // B, E, X, i, M + Timestamp uint64 `json:"ts"` + Duration uint64 `json:"dur,omitempty"` + TimestampNS uint64 `json:"timestamp_ns,omitempty"` + DurationNS uint64 `json:"duration_ns,omitempty"` + ProcessID int `json:"pid"` + ThreadID int `json:"tid"` + Args map[string]interface{} `json:"args,omitempty"` +} + +// EncoderInfo contains information about an encoder. +type EncoderInfo struct { + Index int `json:"index"` + Label string `json:"label"` + Type string `json:"type"` + StartTime uint64 `json:"start_time"` + EndTime uint64 `json:"end_time"` + Duration uint64 `json:"duration"` +} + +// KernelInfo contains information about a kernel execution. +type KernelInfo struct { + Name string `json:"name"` + Encoder int `json:"encoder"` + StartTime uint64 `json:"start_time"` + EndTime uint64 `json:"end_time"` + Duration uint64 `json:"duration"` + Args map[string]interface{} `json:"args,omitempty"` +} + +// APICall represents an API call event. +type APICall struct { + Name string `json:"name"` + Timestamp uint64 `json:"timestamp"` + Args map[string]interface{} `json:"args,omitempty"` } // CounterTrack represents a performance counter track over time. type CounterTrack struct { - Name string `json:"name"` - Unit string `json:"unit"` // %, GB/s, count, etc. - Samples []CounterSample `json:"samples"` - MinValue float64 `json:"min_value"` - MaxValue float64 `json:"max_value"` - AvgValue float64 `json:"avg_value"` + Name string `json:"name"` + Unit string `json:"unit"` // %, GB/s, count, etc. + Description string `json:"description,omitempty"` + XcodeGroups []string `json:"xcode_groups,omitempty"` + XcodeCatalogPath string `json:"xcode_catalog_path,omitempty"` + Samples []CounterSample `json:"samples"` + MinValue float64 `json:"min_value"` + MaxValue float64 `json:"max_value"` + AvgValue float64 `json:"avg_value"` +} + +// UnattributedCounterMetric is a pipeline-scoped counter row for which no +// capture-backed encoder identity exists. +type UnattributedCounterMetric struct { + Label string `json:"label,omitempty"` + Attribution string `json:"attribution"` + Source string `json:"source"` + Values map[string]interface{} `json:"values,omitempty"` +} + +// CounterCatalogEntry is one recorded APSCounterData pass column. Pass-specific +// names are opaque identifiers until a separate catalog proves their meaning. +type CounterCatalogEntry struct { + GroupOrdinal int `json:"group_ordinal"` + ColumnOrdinal int `json:"column_ordinal"` + RecordedName string `json:"recorded_name"` + Classification string `json:"classification"` +} + +// CounterTraceIDEntry is one recorded APSCounterData TraceId-table row. Its +// ordinal relates positionally to encoder execution order; TraceID does not. +type CounterTraceIDEntry struct { + RowOrdinal int `json:"row_ordinal"` + TraceID uint64 `json:"trace_id"` + BatchID int `json:"batch_id"` + SampleIndex int `json:"sample_index"` } // CounterSample represents a single counter measurement at a point in time. @@ -407,6 +1191,7 @@ type CounterSample struct { // generateTimeline creates timeline data from a trace. func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { timeline := &Timeline{ + TracePath: trace.Path, Events: make([]TimelineEvent, 0), Encoders: make([]EncoderInfo, 0), Kernels: make([]KernelInfo, 0), @@ -417,10 +1202,25 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { var profilerDir string if stats, err := counter.ExtractPipelineStatsFromTraceStreamData(trace); err == nil { streamStats = stats + applyStreamIdentity(timeline, streamStats) + if streamStats.Timeline != nil { + timeline.rawProfilerProfiles = streamStats.Timeline.EncoderProfiles + } counter.CorrelateDispatchSamples(streamStats) profilerDir = findProfilerDir(trace.Path) if profilerDir != "" { - annotateDispatchExecutionCosts(streamStats, profilerDir) + annotateDispatchProfilingSampleShares(streamStats, profilerDir) + } + } + + // Capture-only bundles have no streamData, but Xcode archives the same + // shader compilation statistics in the store sections. + var storeStats *counter.StoreStats + if streamStats == nil { + if stats, err := counter.ExtractStoreStats(trace, 0); err == nil { + storeStats = stats + timeline.PipelineCompilerStats = append([]counter.PipelineStats(nil), stats.Pipelines...) + timeline.PipelineCompilerSource = "capture bundle store sections" } } @@ -430,7 +1230,12 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { } var encoderMetrics []counter.EncoderCounterMetrics if perfStats != nil { - encoderMetrics, _ = counter.PopulateEncoderMetricsFromPerfCounterStats(trace, perfStats) + var err error + encoderMetrics, err = counter.PopulateEncoderMetricsFromPerfCounterStats(perfStats) + if err != nil { + return nil, fmt.Errorf("populate counter attribution: %w", err) + } + recordUnattributedCounterMetrics(timeline, encoderMetrics) } var shaderReport *gputrace.ShaderMetricsReport if profilerDir != "" { @@ -438,13 +1243,24 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { shaderReport = report } } - dispatchSIMD := timelineDispatchSIMDGroups(trace, streamStats) + dispatchCapture := timelineDispatchCaptureEvidence(trace, streamStats) sourceMapper := gputrace.NewShaderSourceMapper() _ = sourceMapper.IndexShaderSources() _ = sourceMapper.IndexTraceBundleSources(trace.Path) + if storeStats != nil && storeStats.Source != "" { + _ = sourceMapper.IndexSource(filepath.Join(trace.Path, "store0"), storeStats.Source) + } // Get real encoder labels from ParseComputeEncoders (primary source for labels) - computeEncoders, _ := trace.ParseComputeEncoders() + computeEncoders := trace.ParseComputeEncoders() + timeline.ObservedCSLabels = len(computeEncoders) + uniqueCSLabels := make(map[string]bool) + for _, encoder := range computeEncoders { + if encoder.Label != "" { + uniqueCSLabels[encoder.Label] = true + } + } + timeline.UniqueCSLabels = len(uniqueCSLabels) // Extract timing metrics. This records whether encoder timings came from // measured profiler data or approximate extracted/synthetic fallback data. @@ -466,6 +1282,11 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { timeline.Timing = timelineTimingFromStats(streamStats) if streamStats.Timeline != nil { timeline.AbsoluteTime = streamStats.Timeline.AbsoluteTime + timeline.ContinuousTime = streamStats.Timeline.ContinuousTime + if streamStats.Timeline.PState != nil { + value := *streamStats.Timeline.PState + timeline.PState = &value + } timeline.TimebaseNumer = streamStats.Timeline.TimebaseNumer timeline.TimebaseDenom = streamStats.Timeline.TimebaseDenom } @@ -542,56 +1363,9 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { timeline.Duration = timeline.EndTime - timeline.StartTime // Use compute encoders as primary source for encoder info (better labels) - if len(computeEncoders) > 0 { - avgDuration := timeline.Duration / uint64(len(computeEncoders)) - if avgDuration == 0 { - avgDuration = 1000000 // 1ms default - } - - currentTime := timeline.StartTime - for i, enc := range computeEncoders { - var startTime, endTime, duration uint64 - if timing, ok := timingByLabel[enc.Label]; ok { - startTime = timing.StartTimestamp - endTime = timing.EndTimestamp - duration = timing.DurationNs - } else { - startTime = currentTime - duration = avgDuration - endTime = startTime + duration - currentTime = endTime + 10000 - } - - encoderInfo := EncoderInfo{ - Index: i, - Label: enc.Label, - Type: "compute", - StartTime: startTime, - EndTime: endTime, - Duration: duration, - } - timeline.Encoders = append(timeline.Encoders, encoderInfo) - - // Create timeline event for encoder - event := TimelineEvent{ - Name: enc.Label, - Category: "encoder", - Phase: "X", - Timestamp: startTime / 1000, // Convert to microseconds - Duration: duration / 1000, - ProcessID: 1, - ThreadID: 1, - Args: map[string]interface{}{ - "index": i, - "address": fmt.Sprintf("0x%x", enc.Address), - "duration_ms": float64(duration) / 1e6, - "duration_us": float64(duration) / 1e3, - }, - } - addTimingMetricsEventArgs(event.Args, metrics) - timeline.Events = append(timeline.Events, event) - } - } else { + if len(computeEncoders) > 0 && timelineMetricsSource(metrics) != "unavailable" { + populateUnprofiledEncoderEvents(timeline, computeEncoders, timingByLabel, metrics) + } else if timelineMetricsSource(metrics) != "unavailable" { // Fall back to timing metrics if no compute encoders found for i, encoder := range metrics.EncoderTimings { encoderInfo := EncoderInfo{ @@ -607,15 +1381,14 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { event := TimelineEvent{ Name: encoder.Label, Category: "encoder", - Phase: "X", + Phase: "i", Timestamp: encoder.StartTimestamp / 1000, - Duration: encoder.DurationNs / 1000, + Duration: 0, ProcessID: 1, ThreadID: 1, Args: map[string]interface{}{ - "index": i, - "duration_ms": float64(encoder.DurationNs) / 1e6, - "duration_us": float64(encoder.DurationNs) / 1e3, + "index": i, + "timing_source": "unprofiled (ordering/identity instant)", }, } addTimingMetricsEventArgs(event.Args, metrics) @@ -627,21 +1400,36 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { // Add shader/kernel events. Prefer streamData dispatches so the Shaders lane // matches Xcode's pipeline table instead of duplicating whole encoder spans. - if !addDispatchKernelEvents(timeline, streamStats, dispatchSIMD, shaderReport, perfStats, encoderMetrics, sourceMapper) { - addEncoderKernelEvents(timeline, trace, sourceMapper) + if !addDispatchKernelEvents(timeline, streamStats, dispatchCapture, shaderReport, perfStats, encoderMetrics, sourceMapper) { + if err := addCaptureDispatchEvents(timeline, trace, sourceMapper, storeStats); err != nil { + return nil, fmt.Errorf("add capture dispatches: %w", err) + } } + timeline.bundleDispatches = timelineBundleDispatchCount(streamStats, trace) // Add command buffer events - try to get real timing from APSTimelineData if streamStats != nil && streamStats.Timeline != nil && len(streamStats.Timeline.CommandBufferTimestamps) > 0 { // Use real CB timing from APSTimelineData ti := streamStats.Timeline timeline.AbsoluteTime = ti.AbsoluteTime + timeline.ContinuousTime = ti.ContinuousTime + if ti.PState != nil { + value := *ti.PState + timeline.PState = &value + } timeline.TimebaseNumer = ti.TimebaseNumer timeline.TimebaseDenom = ti.TimebaseDenom - - var displayStartNs uint64 + addRestoreEvents(timeline, ti) + + // Command buffers are placed at their real offset from AbsoluteTime. + // They used to be packed back to back with a running displayStartNs + // accumulator, which erased every idle gap: on the 21-encoder capture + // that compressed 2979 ms of wall time into 8.3 ms and drew a GPU that + // is 0.28% busy as 99.9% busy. The args said "real_timing": true the + // whole time. for _, cb := range ti.CommandBufferTimestamps { durationNs := cb.DurationNs(ti.TimebaseNumer, ti.TimebaseDenom) + durationUs := durationNs / 1000 var rawStartOffsetNs uint64 if cb.StartTicks > ti.AbsoluteTime { rawStartOffsetNs = (cb.StartTicks - ti.AbsoluteTime) * ti.TimebaseNumer / ti.TimebaseDenom @@ -650,9 +1438,9 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { event := TimelineEvent{ Name: fmt.Sprintf("CB#%d", cb.Index), Category: "command_buffer", - Phase: "X", // Duration event - Timestamp: displayStartNs / 1000, // Convert to microseconds for Chrome format - Duration: durationNs / 1000, + Phase: timelineDurationPhase(durationUs), + Timestamp: rawStartOffsetNs / 1000, // Convert to microseconds for Chrome format + Duration: durationUs, ProcessID: 1, ThreadID: 0, Args: map[string]interface{}{ @@ -667,11 +1455,16 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { }, } timeline.Events = append(timeline.Events, event) - displayStartNs += durationNs + if endNs := rawStartOffsetNs + durationNs; endNs > timeline.EndTime { + timeline.EndTime = endNs + } } - // Add encoder profile events from GPRWCNTR ShaderProfilerData + // Preserve aggregates of GPRWCNTR records as raw profiler stream spans. + // These are not established encoder intervals, so they are opt-in with + // the raw records rather than appearing as encoders in the wall view. if len(ti.EncoderProfiles) > 0 { + epLanes := newLanePacker(7, 8) // Raw profiler stream lanes 0..7 for _, ep := range ti.EncoderProfiles { if ep.SampleCount == 0 || ep.StartTicks == 0 { continue @@ -680,13 +1473,13 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { startNs := (ep.StartTicks - ti.AbsoluteTime) * ti.TimebaseNumer / ti.TimebaseDenom event := TimelineEvent{ - Name: fmt.Sprintf("GPRWCNTR Enc#%d", ep.Index), - Category: "encoder_profile", + Name: fmt.Sprintf("Profiler stream %s #%d", ep.Source, ep.Index), + Category: "profiler_stream", Phase: "X", Timestamp: startNs / 1000, // Convert to microseconds Duration: ep.DurationNs / 1000, ProcessID: 1, - ThreadID: 7 + (ep.Index % 8), // 8 Lanes for encoder profiles (7-14) + ThreadID: epLanes.assign(startNs/1000, ep.DurationNs/1000), Args: map[string]interface{}{ "index": ep.Index, "source": ep.Source, @@ -711,12 +1504,14 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { Name: fmt.Sprintf("CommandBuffer %d", i), Category: "command_buffer", Phase: "i", - Timestamp: uint64(cb.Offset), + Timestamp: uint64(i), ProcessID: 1, ThreadID: 0, Args: map[string]interface{}{ - "offset": cb.Offset, - "index": i, + "offset": cb.Offset, + "index": i, + "coordinate_source": "capture byte offset", + "real_timing": false, }, } timeline.Events = append(timeline.Events, event) @@ -821,519 +1616,379 @@ func generateTimeline(trace *gputrace.Trace) (*Timeline, error) { return timeline, nil } -// containsSubstr checks if s contains substr. -func containsSubstr(s, substr string) bool { - if len(substr) > len(s) { - return false +func applyStreamIdentity(timeline *Timeline, stats *counter.StreamDataStats) { + if timeline == nil || stats == nil { + return } - for i := 0; i <= len(s)-len(substr); i++ { - if s[i:i+len(substr)] == substr { - return true - } + if stats.GPUGeneration != nil { + generation := *stats.GPUGeneration + timeline.GPUGeneration = &generation + } + timeline.MetalDeviceName = stats.MetalDeviceName + timeline.MetalPluginName = stats.MetalPluginName + timeline.PipelineCompilerStats = append([]counter.PipelineStats(nil), stats.Pipelines...) + timeline.PipelineCompilerSource = "streamData pipelinePerformanceStatistics" + if stats.FunctionNames != nil { + timeline.StreamDataStrings = make([]string, len(stats.FunctionNames)) + copy(timeline.StreamDataStrings, stats.FunctionNames) + } + if streamDataMetadataPresent(stats.Metadata) { + metadata := cloneStreamDataMetadata(stats.Metadata) + timeline.StreamMetadata = &metadata } - return false } -// generateCounterTracks creates performance counter tracks for the timeline. -// Only returns real data from .gpuprofiler_raw files - no synthetic data. -func generateCounterTracks(trace *gputrace.Trace, timeline *Timeline) []CounterTrack { - tracks := make([]CounterTrack, 0) - - // Skip if no encoders (can't generate meaningful counter data) - if len(timeline.Encoders) == 0 { - return tracks - } +func streamDataMetadataPresent(metadata counter.StreamDataMetadata) bool { + return metadata.Version != nil || metadata.UnixTimestamp != nil || metadata.TraceName != "" || + metadata.ProfiledExecutionMode != nil || metadata.ProfiledPerformanceState != nil || + metadata.ProfiledProfilerMode != nil || metadata.CaptureRangeLocation != nil || + metadata.CaptureRangeLength != nil || metadata.DataSourceHasUnusedResources != nil || + metadata.SupportsSeparateAPSData != nil || metadata.NumBlitCalls != nil || + streamDataTablesPresent(metadata.Tables) || streamDataFamiliesPresent(metadata.Families) || + streamDataDecodedFamiliesPresent(metadata.DecodedFamilies) || metadata.CounterDecode != nil || + metadata.APSDataInventory != nil || len(metadata.ArchiveBlobs) > 0 +} - // Only use real performance counter data - no synthetic fallback - perfStats, err := gputrace.ParsePerfCounters(trace) - if err == nil && len(perfStats.ShaderMetrics) > 0 { - // Also get PipelineStats from streamData for instruction counts - streamStats, _ := gputrace.ExtractPipelineStats(trace) - encoderMetrics, _ := counter.PopulateEncoderMetricsFromPerfCounterStats(trace, perfStats) - return generateCounterTracksFromPerfData(perfStats, streamStats, encoderMetrics, timeline) - } +func streamDataTablesPresent(tables counter.StreamDataTables) bool { + return tables.CommandBuffers != nil || tables.Encoders != nil || tables.GPUCommands != nil || + tables.Pipelines != nil || tables.Functions != nil +} - // No synthetic data - return empty if no real perf data available - return tracks +func streamDataFamiliesPresent(families counter.StreamDataFamilies) bool { + return families.APSData != nil || families.APSTimelineData != nil || families.APSCounterData != nil || + families.ShaderProfilerData != nil || families.GPUTimelineData != nil || + families.BatchIDFilteredCountersData != nil } -// generateCounterTracksFromPerfData creates counter tracks from real performance counter data. -func generateCounterTracksFromPerfData(perfStats *gputrace.PerfCounterStats, streamStats *gputrace.StreamDataStats, encoderMetrics []counter.EncoderCounterMetrics, timeline *Timeline) []CounterTrack { - tracks := make([]CounterTrack, 0) +func streamDataDecodedFamiliesPresent(families counter.StreamDataDecodedFamilies) bool { + return families.APSData != nil || families.APSTimelineData != nil || families.APSCounterData != nil || + families.ShaderProfilerData != nil || families.GPUTimelineData != nil || + families.BatchIDFilteredCountersData != nil +} - // Initialize counter tracks - activeCoresTrack := CounterTrack{ - Name: "Active Cores", - Unit: "count", - Samples: make([]CounterSample, 0), +func cloneStreamDataMetadata(metadata counter.StreamDataMetadata) counter.StreamDataMetadata { + result := metadata + result.Version = cloneInt64(metadata.Version) + result.UnixTimestamp = cloneInt64(metadata.UnixTimestamp) + result.ProfiledExecutionMode = cloneInt64(metadata.ProfiledExecutionMode) + result.ProfiledPerformanceState = cloneInt64(metadata.ProfiledPerformanceState) + result.ProfiledProfilerMode = cloneInt64(metadata.ProfiledProfilerMode) + result.CaptureRangeLocation = cloneInt64(metadata.CaptureRangeLocation) + result.CaptureRangeLength = cloneInt64(metadata.CaptureRangeLength) + result.DataSourceHasUnusedResources = cloneBool(metadata.DataSourceHasUnusedResources) + result.SupportsSeparateAPSData = cloneBool(metadata.SupportsSeparateAPSData) + result.NumBlitCalls = cloneInt64(metadata.NumBlitCalls) + result.Tables.CommandBuffers = cloneStreamDataTable(metadata.Tables.CommandBuffers) + result.Tables.Encoders = cloneStreamDataTable(metadata.Tables.Encoders) + result.Tables.GPUCommands = cloneStreamDataTable(metadata.Tables.GPUCommands) + result.Tables.Pipelines = cloneStreamDataTable(metadata.Tables.Pipelines) + result.Tables.Functions = cloneStreamDataTable(metadata.Tables.Functions) + result.Families.APSData = cloneInt64(metadata.Families.APSData) + result.Families.APSTimelineData = cloneInt64(metadata.Families.APSTimelineData) + result.Families.APSCounterData = cloneInt64(metadata.Families.APSCounterData) + result.Families.ShaderProfilerData = cloneInt64(metadata.Families.ShaderProfilerData) + result.Families.GPUTimelineData = cloneInt64(metadata.Families.GPUTimelineData) + result.Families.BatchIDFilteredCountersData = cloneInt64(metadata.Families.BatchIDFilteredCountersData) + result.DecodedFamilies.APSData = cloneInt64(metadata.DecodedFamilies.APSData) + result.DecodedFamilies.APSTimelineData = cloneInt64(metadata.DecodedFamilies.APSTimelineData) + result.DecodedFamilies.APSCounterData = cloneInt64(metadata.DecodedFamilies.APSCounterData) + result.DecodedFamilies.ShaderProfilerData = cloneInt64(metadata.DecodedFamilies.ShaderProfilerData) + result.DecodedFamilies.GPUTimelineData = cloneInt64(metadata.DecodedFamilies.GPUTimelineData) + result.DecodedFamilies.BatchIDFilteredCountersData = cloneInt64(metadata.DecodedFamilies.BatchIDFilteredCountersData) + if metadata.CounterDecode != nil { + counterDecode := *metadata.CounterDecode + result.CounterDecode = &counterDecode + } + if metadata.APSDataInventory != nil { + inventory := *metadata.APSDataInventory + inventory.BlobRecords = append([]counter.APSDataBlobInventory(nil), metadata.APSDataInventory.BlobRecords...) + for i := range inventory.BlobRecords { + inventory.BlobRecords[i].Keys = cloneStreamDataKeys(metadata.APSDataInventory.BlobRecords[i].Keys) + inventory.BlobRecords[i].Nodes = cloneStreamDataNodes(metadata.APSDataInventory.BlobRecords[i].Nodes) + } + result.APSDataInventory = &inventory } - - occupancyTrack := CounterTrack{ - Name: "Occupancy", - Unit: "%", - Samples: make([]CounterSample, 0), + result.ArchiveBlobs = append([]counter.StreamDataBlobInventory(nil), metadata.ArchiveBlobs...) + for i := range result.ArchiveBlobs { + result.ArchiveBlobs[i].Keys = cloneStreamDataKeys(metadata.ArchiveBlobs[i].Keys) + result.ArchiveBlobs[i].Nodes = cloneStreamDataNodes(metadata.ArchiveBlobs[i].Nodes) } + return result +} - aluTrack := CounterTrack{ - Name: "ALU Utilization", - Unit: "%", - Samples: make([]CounterSample, 0), +func cloneStreamDataNodes(nodes []counter.StreamDataNodeInventory) []counter.StreamDataNodeInventory { + cloned := append([]counter.StreamDataNodeInventory(nil), nodes...) + for i := range cloned { + if nodes[i].ObjectIndex != nil { + value := *nodes[i].ObjectIndex + cloned[i].ObjectIndex = &value + } + if nodes[i].DataBytes != nil { + value := *nodes[i].DataBytes + cloned[i].DataBytes = &value + } + if nodes[i].ContainerCount != nil { + value := *nodes[i].ContainerCount + cloned[i].ContainerCount = &value + } } + return cloned +} - bandwidthTrack := CounterTrack{ - Name: "Bandwidth", - Unit: "GB/s", - Samples: make([]CounterSample, 0), +func cloneStreamDataKeys(keys []counter.StreamDataKeyInventory) []counter.StreamDataKeyInventory { + cloned := append([]counter.StreamDataKeyInventory(nil), keys...) + for i := range cloned { + if keys[i].DataBytes != nil { + value := *keys[i].DataBytes + cloned[i].DataBytes = &value + } + if keys[i].ContainerCount != nil { + value := *keys[i].ContainerCount + cloned[i].ContainerCount = &value + } } + return cloned +} - throughputTrack := CounterTrack{ - Name: "Instruction Throughput", - Unit: "%", - Samples: make([]CounterSample, 0), +func cloneStreamDataTable(table *counter.StreamDataTable) *counter.StreamDataTable { + if table == nil { + return nil } + result := *table + result.RecordSize = cloneInt64(table.RecordSize) + result.RecordCount = cloneInt64(table.RecordCount) + result.RemainderBytes = cloneInt64(table.RemainderBytes) + return &result +} - occupancyManagerTrack := CounterTrack{ - Name: "Occupancy Manager", - Unit: "%", - Samples: make([]CounterSample, 0), +func cloneInt64(value *int64) *int64 { + if value == nil { + return nil } + copy := *value + return © +} - shaderLaunchLimiterTrack := CounterTrack{ - Name: "Shader Launch Limiter", - Unit: "%", - Samples: make([]CounterSample, 0), +func cloneBool(value *bool) *bool { + if value == nil { + return nil } + copy := *value + return © +} - // Create a map of shader name to hardware metrics - shaderMetricsMap := make(map[string]*gputrace.ShaderHardwareMetrics) - for i := range perfStats.ShaderMetrics { - metric := &perfStats.ShaderMetrics[i] - if metric.ShaderName != "" { - shaderMetricsMap[metric.ShaderName] = metric - } - } - encoderMetricsByIndex := make(map[int]*counter.EncoderCounterMetrics) - encoderMetricsByLabel := make(map[string]*counter.EncoderCounterMetrics) - for i := range encoderMetrics { - m := &encoderMetrics[i] - encoderMetricsByIndex[m.EncoderIndex] = m - if m.EncoderLabel != "" { - encoderMetricsByLabel[m.EncoderLabel] = m - } +func populateUnprofiledEncoderEvents(timeline *Timeline, computeEncoders []*tracepkg.ComputeEncoder, timingByLabel map[string]*gputrace.EncoderTiming, metrics *gputrace.TimingMetrics) { + if timeline == nil || len(computeEncoders) == 0 { + return } - - // Build map of function name to PipelineStats for instruction counts - // This provides instruction counts by kernel name directly - pipelineByName := make(map[string]*gputrace.PipelineStats) - if streamStats != nil { - // Index by function name for fuzzy matching - for i, funcName := range streamStats.FunctionNames { - if i < len(streamStats.Pipelines) { - p := &streamStats.Pipelines[i] - pipelineByName[funcName] = p - } - } + avgDuration := timeline.Duration / uint64(len(computeEncoders)) + if avgDuration == 0 { + avgDuration = 1000000 // 1ms default } - // Generate samples for each encoder period using actual hardware metrics - for _, encoder := range timeline.Encoders { - // Look up hardware metrics for this encoder - var metrics *gputrace.ShaderHardwareMetrics - if m, exists := shaderMetricsMap[encoder.Label]; exists { - metrics = m - } - var encoderMetric *counter.EncoderCounterMetrics - if m, exists := encoderMetricsByLabel[encoder.Label]; exists { - encoderMetric = m - } else if m, exists := encoderMetricsByIndex[encoder.Index]; exists { - encoderMetric = m - } - - // Calculate values from real hardware data. - var activeCores float64 - var occupancy float64 - var aluUtil float64 - var bandwidth float64 - var throughput float64 - var occupancyManager float64 - var shaderLaunchLimiter float64 - - if metrics != nil { - // Use real hardware metrics - occupancy = metrics.KernelOccupancy - aluUtil = metrics.ALUUtilization - - // Calculate active cores from SIMD groups - // Typical M-series GPU: 8-10 cores, each core has 128-1024 SIMD lanes - // Heuristic: map SIMD groups to estimated core count - if metrics.SIMDGroups > 0 { - activeCores = float64(metrics.SIMDGroups) / 100.0 // Rough estimate - if activeCores > 8.0 { - activeCores = 8.0 // Cap at typical M-series core count - } - if activeCores < 1.0 { - activeCores = 1.0 - } - } - - // Calculate bandwidth from memory bandwidth counter (convert bytes to GB/s) - if metrics.MemoryBandwidth > 0 && encoder.Duration > 0 { - durationSec := float64(encoder.Duration) / 1e9 - bandwidth = float64(metrics.MemoryBandwidth) / 1e9 / durationSec - } - - // Estimate throughput from occupancy and ALU utilization - if occupancy > 0 && aluUtil > 0 { - throughput = (occupancy + aluUtil) / 2.0 - } - - // Occupancy Manager: Tracks how well the GPU scheduler manages threadgroup dispatch - // High when occupancy is maintained well, low when there are bubbles - if occupancy > 0 { - occupancyManager = occupancy * 0.95 // Typically slightly lower than raw occupancy - } - - // Shader Launch Limiter: Percentage of time shader launches are limited by resources - // High values indicate resource contention (registers, threadgroup memory, etc.) - // Estimate from register pressure and occupancy - if metrics.AllocatedRegs > 0 { - // More registers = more likely to hit launch limits - regPressure := float64(metrics.AllocatedRegs) / 256.0 // 256 max registers typical - if regPressure > 1.0 { - regPressure = 1.0 - } - shaderLaunchLimiter = regPressure * 100.0 - } - } - if encoderMetric != nil { - if occupancy == 0 { - occupancy = encoderMetric.KernelOccupancy - } - if aluUtil == 0 { - aluUtil = encoderMetric.ALUUtilization - } - if bandwidth == 0 { - switch { - case encoderMetric.DeviceMemoryBandwidthGBps > 0: - bandwidth = encoderMetric.DeviceMemoryBandwidthGBps - case encoderMetric.MemoryBandwidth > 0 && encoder.Duration > 0: - durationSec := float64(encoder.Duration) / 1e9 - bandwidth = float64(encoderMetric.MemoryBandwidth) / 1e9 / durationSec - } - } - if throughput == 0 { - throughput = encoderMetric.InstructionThroughputUtil - } - if occupancyManager == 0 { - occupancyManager = encoderMetric.ComputeUtilization - } - if shaderLaunchLimiter == 0 { - shaderLaunchLimiter = encoderMetric.ComputeShaderLaunchLimiter - } + currentTime := timeline.StartTime + for i, enc := range computeEncoders { + var startTime, endTime, duration uint64 + if timing, ok := timingByLabel[enc.Label]; ok { + startTime = timing.StartTimestamp + endTime = timing.EndTimestamp + duration = timing.DurationNs + } else { + startTime = currentTime + duration = avgDuration + endTime = startTime + duration + currentTime = endTime + 10000 } - if metrics == nil && encoderMetric == nil { - // No real data for this encoder - skip it (no synthetic data) - continue + encoderInfo := EncoderInfo{ + Index: i, + Label: enc.Label, + Type: "compute", + StartTime: startTime, + EndTime: endTime, + Duration: duration, } + timeline.Encoders = append(timeline.Encoders, encoderInfo) - // Add samples at start and end of encoder execution. For source-backed - // Xcode counters, zero is a meaningful value and should appear as a - // flat track instead of being reported as unavailable. - appendCounterTrackSample(&activeCoresTrack, encoder, activeCores) - appendCounterTrackSampleValue(&occupancyTrack, encoder, occupancy) - appendCounterTrackSampleValue(&aluTrack, encoder, aluUtil) - appendCounterTrackSampleValue(&bandwidthTrack, encoder, bandwidth) - appendCounterTrackSampleValue(&throughputTrack, encoder, throughput) - appendCounterTrackSampleValue(&occupancyManagerTrack, encoder, occupancyManager) - appendCounterTrackSampleValue(&shaderLaunchLimiterTrack, encoder, shaderLaunchLimiter) + // Create timeline event for encoder in unprofiled fallback: + // Emitted as Phase 'i' zero-duration instant. Synthetic/extracted timestamps + // are retained only for ordering. + event := TimelineEvent{ + Name: enc.Label, + Category: "encoder", + Phase: "i", + Timestamp: startTime / 1000, // Convert to microseconds + Duration: 0, + ProcessID: 1, + ThreadID: 1, + Args: map[string]interface{}{ + "index": i, + "address": fmt.Sprintf("0x%x", enc.Address), + "timing_source": "unprofiled (ordering/identity instant)", + }, + } + addTimingMetricsEventArgs(event.Args, metrics) + timeline.Events = append(timeline.Events, event) } +} - // Calculate statistics for each track - calculateTrackStats(&activeCoresTrack) - calculateTrackStats(&occupancyTrack) - calculateTrackStats(&aluTrack) - calculateTrackStats(&bandwidthTrack) - calculateTrackStats(&throughputTrack) - calculateTrackStats(&occupancyManagerTrack) - calculateTrackStats(&shaderLaunchLimiterTrack) - - tracks = append(tracks, activeCoresTrack, occupancyTrack, aluTrack, bandwidthTrack, throughputTrack, occupancyManagerTrack, shaderLaunchLimiterTrack) - - // Add L1 Cache Miss Rate Track - l1MissTrack := CounterTrack{ - Name: "L1 Cache Miss Rate", - Unit: "%", - Samples: make([]CounterSample, 0), +// generateCounterTracks returns only counter series whose clock is established +// in the selected timeline domain. APSCounterData currently provides useful +// per-encoder aggregates, but its timestamps have no verified mapping to the +// cumulative busy clock, so those aggregates remain encoder details. +func generateCounterTracks(trace *gputrace.Trace, timeline *Timeline) []CounterTrack { + streamStats, _ := gputrace.ExtractPipelineStats(trace) + if streamStats == nil || streamStats.CounterArchive == nil { + return nil } + recordCounterCatalog(timeline, streamStats.CounterArchive) + recordCounterTraceIDs(timeline, streamStats.CounterArchive) + timeline.CounterEncoderAggregates = append([]counter.EncoderSamples(nil), streamStats.CounterArchive.Encoders...) + timeline.CounterEncoderSamples = cloneAttributedCounterSamples(streamStats.CounterArchive.AttributedRecords) + annotateEncoderCounterArchive(timeline, streamStats.CounterArchive) + timeline.UnavailableEvidence = append(timeline.UnavailableEvidence, UnavailableEvidence{ + Family: "APSCounterData time series", + Reason: "counter clock has no verified mapping to cumulative GPU-busy time", + }) + return nil +} - // Add Memory Read/Write Bandwidth Tracks - memReadTrack := CounterTrack{ - Name: "Memory Read BW", - Unit: "GB/s", - Samples: make([]CounterSample, 0), +func cloneAttributedCounterSamples(samples []counter.AttributedCounterSample) []counter.AttributedCounterSample { + if samples == nil { + return nil } - memWriteTrack := CounterTrack{ - Name: "Memory Write BW", - Unit: "GB/s", - Samples: make([]CounterSample, 0), + out := append([]counter.AttributedCounterSample(nil), samples...) + for i := range out { + out[i].Counters = append([]uint64(nil), samples[i].Counters...) } + return out +} - // Add Bottleneck Limiter Tracks - computeLimiterTrack := CounterTrack{ - Name: "Limiter: Compute", - Unit: "%", - Samples: make([]CounterSample, 0), +func recordCounterTraceIDs(timeline *Timeline, archive *counter.CounterArchive) { + if timeline == nil || archive == nil || archive.TraceIDs == nil { + return } - memoryLimiterTrack := CounterTrack{ - Name: "Limiter: Memory", - Unit: "%", - Samples: make([]CounterSample, 0), + for rowOrdinal, row := range archive.TraceIDs.Rows { + timeline.CounterTraceIDs = append(timeline.CounterTraceIDs, CounterTraceIDEntry{ + RowOrdinal: rowOrdinal, TraceID: row.TraceID, + BatchID: row.BatchID, SampleIndex: row.SampleIndex, + }) } +} - // Generate samples for new tracks - only for encoders with real data - for _, encoder := range timeline.Encoders { - metrics := shaderMetricsMap[encoder.Label] - var encoderMetric *counter.EncoderCounterMetrics - if m, exists := encoderMetricsByLabel[encoder.Label]; exists { - encoderMetric = m - } else if m, exists := encoderMetricsByIndex[encoder.Index]; exists { - encoderMetric = m - } - if metrics == nil && encoderMetric == nil { - // No real data for this encoder - skip it (no synthetic data) - continue - } - - var l1Miss float64 - var memRead, memWrite float64 - var compLimit, memLimit float64 - - if metrics != nil { - l1Miss = metrics.BufferL1MissRate - durationSec := float64(encoder.Duration) / 1e9 - if durationSec > 0 { - memRead = float64(metrics.BytesReadFromDeviceMemory) / 1e9 / durationSec - memWrite = float64(metrics.BytesWrittenToDeviceMemory) / 1e9 / durationSec - } - compLimit = metrics.ComputeShaderLaunchLimiter + metrics.ALUUtilization - memLimit = metrics.L1CacheLimiter + metrics.LastLevelCacheLimiter + metrics.TextureReadLimiter - } - if encoderMetric != nil { - if l1Miss == 0 { - l1Miss = encoderMetric.BufferL1MissRate - } - if memRead == 0 { - if encoderMetric.GPUReadBandwidthGBps > 0 { - memRead = encoderMetric.GPUReadBandwidthGBps - } else if encoderMetric.BytesReadFromDeviceMemory > 0 && encoder.Duration > 0 { - durationSec := float64(encoder.Duration) / 1e9 - memRead = float64(encoderMetric.BytesReadFromDeviceMemory) / 1e9 / durationSec - } - } - if memWrite == 0 { - if encoderMetric.GPUWriteBandwidthGBps > 0 { - memWrite = encoderMetric.GPUWriteBandwidthGBps - } else if encoderMetric.BytesWrittenToDeviceMemory > 0 && encoder.Duration > 0 { - durationSec := float64(encoder.Duration) / 1e9 - memWrite = float64(encoderMetric.BytesWrittenToDeviceMemory) / 1e9 / durationSec - } - } - if compLimit == 0 { - compLimit = encoderMetric.ComputeShaderLaunchLimiter - } - if memLimit == 0 { - memLimit = encoderMetric.L1CacheLimiter + encoderMetric.LastLevelCacheLimiter + encoderMetric.TextureReadLimiter +func recordCounterCatalog(timeline *Timeline, archive *counter.CounterArchive) { + if timeline == nil || archive == nil { + return + } + for groupOrdinal, columns := range archive.PassColumns { + for columnOrdinal, name := range columns { + classification := "pass-specific" + if columnOrdinal < len(counter.GRCColumnNames) { + classification = "fixed GRC" } + timeline.CounterCatalog = append(timeline.CounterCatalog, CounterCatalogEntry{ + GroupOrdinal: groupOrdinal, ColumnOrdinal: columnOrdinal, + RecordedName: name, Classification: classification, + }) } - - appendCounterTrackSampleValue(&l1MissTrack, encoder, l1Miss) - appendCounterTrackSampleValue(&memReadTrack, encoder, memRead) - appendCounterTrackSampleValue(&memWriteTrack, encoder, memWrite) - appendCounterTrackSampleValue(&computeLimiterTrack, encoder, compLimit) - appendCounterTrackSampleValue(&memoryLimiterTrack, encoder, memLimit) } +} - calculateTrackStats(&l1MissTrack) - calculateTrackStats(&memReadTrack) - calculateTrackStats(&memWriteTrack) - calculateTrackStats(&computeLimiterTrack) - calculateTrackStats(&memoryLimiterTrack) - - tracks = append(tracks, l1MissTrack, memReadTrack, memWriteTrack, computeLimiterTrack, memoryLimiterTrack) - - // Add Instruction Count Tracks from PipelineStats/streamData - instructionTrack := CounterTrack{ - Name: "Total Instructions", - Unit: "count", - Samples: make([]CounterSample, 0), - } - aluInstrTrack := CounterTrack{ - Name: "ALU Instructions", - Unit: "count", - Samples: make([]CounterSample, 0), - } - fp32InstrTrack := CounterTrack{ - Name: "FP32 Instructions", - Unit: "count", - Samples: make([]CounterSample, 0), - } - fp16InstrTrack := CounterTrack{ - Name: "FP16 Instructions", - Unit: "count", - Samples: make([]CounterSample, 0), - } - int32InstrTrack := CounterTrack{ - Name: "INT32 Instructions", - Unit: "count", - Samples: make([]CounterSample, 0), - } - int16InstrTrack := CounterTrack{ - Name: "INT16 Instructions", - Unit: "count", - Samples: make([]CounterSample, 0), - } - branchInstrTrack := CounterTrack{ - Name: "Branch Instructions", - Unit: "count", - Samples: make([]CounterSample, 0), - } - threadgroupMemTrack := CounterTrack{ - Name: "Threadgroup Memory", - Unit: "bytes", - Samples: make([]CounterSample, 0), - } - allocatedRegsTrack := CounterTrack{ - Name: "Allocated Registers", - Unit: "count", - Samples: make([]CounterSample, 0), - } - uniformRegsTrack := CounterTrack{ - Name: "Uniform Registers", - Unit: "count", - Samples: make([]CounterSample, 0), - } - spilledBytesTrack := CounterTrack{ - Name: "Spilled Bytes", - Unit: "bytes", - Samples: make([]CounterSample, 0), +func recordUnattributedCounterMetrics(timeline *Timeline, metrics []counter.EncoderCounterMetrics) { + if timeline == nil { + return } - - // Generate samples for instruction tracks - use PipelineStats from streamData - // Match by encoder label (which is the kernel/function name) - for _, encoder := range timeline.Encoders { - // Try to find matching PipelineStats by exact or fuzzy match - var pipeline *gputrace.PipelineStats - if p, exists := pipelineByName[encoder.Label]; exists { - pipeline = p - } else { - // Try fuzzy match - encoder label may contain or be contained in function name - for funcName, p := range pipelineByName { - if containsSubstr(encoder.Label, funcName) || containsSubstr(funcName, encoder.Label) { - pipeline = p - break - } - } - } - - if pipeline == nil { + for _, metric := range metrics { + if metric.Attribution == counter.CounterAttributionEncoder && metric.EncoderIndex >= 0 { continue } - - // Add instruction count samples - if pipeline.InstructionCount > 0 { - instructionTrack.Samples = append(instructionTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.InstructionCount)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.InstructionCount)}) + values := make(map[string]interface{}) + if metric.ALUUtilization != 0 { + values["alu_utilization_pct"] = metric.ALUUtilization } - if pipeline.ALUInstructionCount > 0 { - aluInstrTrack.Samples = append(aluInstrTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.ALUInstructionCount)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.ALUInstructionCount)}) + if metric.MemoryBandwidth != 0 { + values["memory_bandwidth_bytes"] = metric.MemoryBandwidth } - if pipeline.FP32InstructionCount > 0 { - fp32InstrTrack.Samples = append(fp32InstrTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.FP32InstructionCount)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.FP32InstructionCount)}) + if metric.DeviceMemoryBandwidthGBps != 0 { + values["device_memory_bandwidth_gbps"] = metric.DeviceMemoryBandwidthGBps } - if pipeline.FP16InstructionCount > 0 { - fp16InstrTrack.Samples = append(fp16InstrTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.FP16InstructionCount)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.FP16InstructionCount)}) + if metric.BytesReadFromDeviceMemory != 0 { + values["device_memory_read_bytes"] = metric.BytesReadFromDeviceMemory } - if pipeline.INT32InstructionCount > 0 { - int32InstrTrack.Samples = append(int32InstrTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.INT32InstructionCount)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.INT32InstructionCount)}) + if metric.BytesWrittenToDeviceMemory != 0 { + values["device_memory_write_bytes"] = metric.BytesWrittenToDeviceMemory } - if pipeline.INT16InstructionCount > 0 { - int16InstrTrack.Samples = append(int16InstrTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.INT16InstructionCount)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.INT16InstructionCount)}) + if metric.InstructionThroughputUtil != 0 { + values["instruction_throughput_utilization_pct"] = metric.InstructionThroughputUtil } - if pipeline.BranchInstructionCount > 0 { - branchInstrTrack.Samples = append(branchInstrTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.BranchInstructionCount)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.BranchInstructionCount)}) + if metric.ComputeShaderLaunchLimiter != 0 { + values["compute_shader_launch_limiter_pct"] = metric.ComputeShaderLaunchLimiter } - if pipeline.ThreadgroupMemory > 0 { - threadgroupMemTrack.Samples = append(threadgroupMemTrack.Samples, - CounterSample{Timestamp: encoder.StartTime, Value: float64(pipeline.ThreadgroupMemory)}, - CounterSample{Timestamp: encoder.EndTime, Value: float64(pipeline.ThreadgroupMemory)}) + if metric.BufferL1MissRate != 0 { + values["buffer_l1_miss_rate_pct"] = metric.BufferL1MissRate } - appendCounterTrackSample(&allocatedRegsTrack, encoder, float64(pipeline.TemporaryRegisterCount)) - appendCounterTrackSample(&uniformRegsTrack, encoder, float64(pipeline.UniformRegisterCount)) - appendCounterTrackSample(&spilledBytesTrack, encoder, float64(pipeline.SpilledBytes)) + timeline.UnattributedCounters = append(timeline.UnattributedCounters, UnattributedCounterMetric{ + Label: metric.EncoderLabel, + Attribution: string(counter.CounterAttributionUnknown), + Source: "PerfCounterStats pipeline row", + Values: values, + }) } +} - // Calculate stats and append tracks that have data - calculateTrackStats(&instructionTrack) - calculateTrackStats(&aluInstrTrack) - calculateTrackStats(&fp32InstrTrack) - calculateTrackStats(&fp16InstrTrack) - calculateTrackStats(&int32InstrTrack) - calculateTrackStats(&int16InstrTrack) - calculateTrackStats(&branchInstrTrack) - calculateTrackStats(&threadgroupMemTrack) - calculateTrackStats(&allocatedRegsTrack) - calculateTrackStats(&uniformRegsTrack) - calculateTrackStats(&spilledBytesTrack) - - // Only add tracks that have samples - if len(instructionTrack.Samples) > 0 { - tracks = append(tracks, instructionTrack) - } - if len(aluInstrTrack.Samples) > 0 { - tracks = append(tracks, aluInstrTrack) - } - if len(fp32InstrTrack.Samples) > 0 { - tracks = append(tracks, fp32InstrTrack) - } - if len(fp16InstrTrack.Samples) > 0 { - tracks = append(tracks, fp16InstrTrack) - } - if len(int32InstrTrack.Samples) > 0 { - tracks = append(tracks, int32InstrTrack) - } - if len(int16InstrTrack.Samples) > 0 { - tracks = append(tracks, int16InstrTrack) - } - if len(branchInstrTrack.Samples) > 0 { - tracks = append(tracks, branchInstrTrack) - } - if len(threadgroupMemTrack.Samples) > 0 { - tracks = append(tracks, threadgroupMemTrack) +// annotateEncoderCounterArchive records capture-backed cycle aggregates on +// encoder events. Encoder Infos guarantees execution order, but does not expose +// a Metal encoder foreign key, so the relationship basis remains explicit. +func annotateEncoderCounterArchive(timeline *Timeline, archive *counter.CounterArchive) { + if archive == nil || timeline == nil { + return } - if len(allocatedRegsTrack.Samples) > 0 { - tracks = append(tracks, allocatedRegsTrack) + costs := archive.EncoderCosts() + if len(costs) == 0 { + return } - if len(uniformRegsTrack.Samples) > 0 { - tracks = append(tracks, uniformRegsTrack) + byOrdinal := make(map[int]counter.EncoderCost, len(costs)) + for _, c := range costs { + byOrdinal[c.Ordinal] = c } - if len(spilledBytesTrack.Samples) > 0 { - tracks = append(tracks, spilledBytesTrack) + for i := range timeline.Events { + event := &timeline.Events[i] + if event.Category != "encoder" { + continue + } + index, ok := timelineEventArgInt(event.Args, "index") + if !ok { + continue + } + c, ok := byOrdinal[index] + if !ok { + continue + } + event.Args["gpu_cycles"] = c.GPUCycles + event.Args["gpu_cycles_source"] = "APSCounterData GRC_GPU_CYCLES end records" + event.Args["execution_cost_pct"] = c.CostPercent + event.Args["execution_cost_formula"] = "100 * encoder GPU cycles / capture GPU cycles" + event.Args["counter_attribution_basis"] = "Encoder Infos execution ordinal" + event.Args["counter_end_records"] = c.EndRecords + event.Args["counter_sample_count"] = c.SampleCount + batchID, hasBatchID := archive.TraceIDs.BatchForOrdinal(index) + if hasBatchID { + event.Args["counter_batch_id"] = batchID + event.Args["counter_batch_id_source"] = "APSCounterData TraceId to BatchId by encoder execution ordinal" + } + sampleIndex, hasSampleIndex := archive.TraceIDs.SampleIndexForOrdinal(index) + if hasSampleIndex { + event.Args["counter_sample_index"] = sampleIndex + event.Args["counter_sample_index_source"] = "APSCounterData TraceId to SampleIndex by encoder execution ordinal" + } + if hasBatchID || hasSampleIndex { + event.Args["counter_trace_id_relation"] = "positional only; TraceId does not equal GRC encoder or kick trace id" + } + if c.Sparse() { + event.Args["counter_coverage"] = "sparse: fewer than 16 end-counter reads" + } else { + event.Args["counter_coverage"] = "at least one end-counter read per replay group" + } } - - return tracks } // calculateTrackStats calculates min, max, and average values for a counter track. @@ -1375,9 +2030,18 @@ func appendCounterTrackSampleValue(track *CounterTrack, encoder EncoderInfo, val CounterSample{Timestamp: encoder.EndTime, Value: value}) } -func annotateDispatchExecutionCosts(stats *counter.StreamDataStats, profilerDir string) { +// annotateDispatchProfilingSampleShares records the share of Profiling_f +// samples assigned to each pipeline. Sampling share is an estimate, not Xcode +// Execution Cost, so it is kept distinct from a validated cost measurement. +func annotateDispatchProfilingSampleShares(stats *counter.StreamDataStats, profilerDir string) { + for i, share := range dispatchProfilingSampleShares(stats, profilerDir) { + stats.Dispatches[i].ProfilingSampleSharePct = share + } +} + +func dispatchProfilingSampleShares(stats *counter.StreamDataStats, profilerDir string) map[int]float64 { if stats == nil || len(stats.Pipelines) == 0 || profilerDir == "" { - return + return nil } pipelineIDs := make([]int, 0, len(stats.Pipelines)) for _, p := range stats.Pipelines { @@ -1385,30 +2049,99 @@ func annotateDispatchExecutionCosts(stats *counter.StreamDataStats, profilerDir } costs, err := counter.ParseExecutionCost(profilerDir, pipelineIDs) if err != nil { + return nil + } + shares := make(map[int]float64, len(stats.Dispatches)) + for i, dispatch := range stats.Dispatches { + if share, ok := costs.PipelineCosts[dispatch.PipelineID]; ok { + shares[i] = share + } + } + return shares +} + +// annotateDispatchExecutionCosts is retained for the legacy Xcode-parity +// report. Its values are sampling-share estimates, not measured Xcode costs. +func annotateDispatchExecutionCosts(stats *counter.StreamDataStats, profilerDir string) { + for i, share := range dispatchProfilingSampleShares(stats, profilerDir) { + stats.Dispatches[i].ExecutionCostPct = share + } +} + +// addPipelineCompilerArgs records static shader compiler statistics. The +// caller supplies an already-attributed pipeline; this function does not infer +// pipeline identity or add dynamic counter measurements. +func addPipelineCompilerArgs(args map[string]interface{}, p *counter.PipelineStats, source string) { + if p == nil { return } - for i := range stats.Dispatches { - if cost, ok := costs.PipelineCosts[stats.Dispatches[i].PipelineID]; ok { - stats.Dispatches[i].ExecutionCostPct = cost + if p.FunctionName != "" { + args["function_name"] = p.FunctionName + } + if p.PipelineID != 0 { + if _, ok := args["pipeline_id"]; !ok { + args["pipeline_id"] = p.PipelineID } } + if p.PipelineAddress != 0 { + args["pipeline_state"] = fmt.Sprintf("0x%x", p.PipelineAddress) + args["pipeline_address"] = p.PipelineAddress + } + addRecordedPipelineInt(args, p, "Temporary register count", "allocated_registers", p.TemporaryRegisterCount) + addRecordedPipelineInt(args, p, "Uniform register count", "uniform_registers", p.UniformRegisterCount) + addRecordedPipelineInt(args, p, "Spilled bytes", "spilled_bytes", p.SpilledBytes) + addRecordedPipelineInt(args, p, "Thread invariant spilled bytes", "thread_invariant_spilled", p.ThreadInvariantSpilled) + addRecordedPipelineInt(args, p, "Threadgroup memory", "threadgroup_memory", p.ThreadgroupMemory) + addRecordedPipelineInt(args, p, "Instruction count", "instruction_count", p.InstructionCount) + addRecordedPipelineInt(args, p, "ALU instruction count", "alu_instruction_count", p.ALUInstructionCount) + addRecordedPipelineInt(args, p, "FP32 instruction count", "fp32_instruction_count", p.FP32InstructionCount) + addRecordedPipelineInt(args, p, "FP16 instruction count", "fp16_instruction_count", p.FP16InstructionCount) + addRecordedPipelineInt(args, p, "INT32 instruction count", "int32_instruction_count", p.INT32InstructionCount) + addRecordedPipelineInt(args, p, "INT16 instruction count", "int16_instruction_count", p.INT16InstructionCount) + addRecordedPipelineInt(args, p, "Branch instruction count", "branch_instruction_count", p.BranchInstructionCount) + addRecordedPipelineInt(args, p, "Device load instruction count", "device_load_instruction_count", p.DeviceLoadCount) + addRecordedPipelineInt(args, p, "Device store instruction count", "device_store_instruction_count", p.DeviceStoreCount) + addRecordedPipelineInt(args, p, "Device atomic instruction count", "device_atomic_instruction_count", p.DeviceAtomicCount) + addRecordedPipelineInt(args, p, "Texture reads instruction count", "texture_reads_instruction_count", p.TextureReadCount) + addRecordedPipelineInt(args, p, "Texture writes instruction count", "texture_writes_instruction_count", p.TextureWriteCount) + addRecordedPipelineInt(args, p, "Threadgroup load instruction count", "threadgroup_load_instruction_count", p.ThreadgroupLoadCount) + addRecordedPipelineInt(args, p, "Threadgroup store instruction count", "threadgroup_store_instruction_count", p.ThreadgroupStoreCount) + addRecordedPipelineInt(args, p, "Threadgroup atomic instruction count", "threadgroup_atomic_instruction_count", p.ThreadgroupAtomicCount) + addRecordedPipelineInt(args, p, "Wait instruction count", "wait_instruction_count", p.WaitInstructionCount) + addRecordedPipelineInt(args, p, "Constant calculation temporary register count", "constant_calculation_temporary_register_count", p.ConstantCalculationTemporaryRegisterCount) + if p.HasRecordedStatistic("Constant calculation phase present") || p.ConstantCalculationPhasePresent { + args["constant_calculation_phase_present"] = p.ConstantCalculationPhasePresent + } + if p.HasRecordedStatistic("Compilation time in milliseconds") || p.CompilationTimeMs != 0 { + args["compilation_time_ms"] = p.CompilationTimeMs + } + args["metrics_source"] = source } -func addEncoderKernelEvents(timeline *Timeline, trace *gputrace.Trace, sourceMapper *gputrace.ShaderSourceMapper) { - computeEncoders, _ := traceComputeEncoders(trace) +func addRecordedPipelineInt(args map[string]interface{}, pipeline *counter.PipelineStats, sourceName, argName string, value int) { + if pipeline.HasRecordedStatistic(sourceName) || value != 0 { + args[argName] = value + } +} + +func addEncoderKernelEvents(timeline *Timeline, trace *gputrace.Trace, sourceMapper *gputrace.ShaderSourceMapper, storeStats *counter.StoreStats) { + computeEncoders := traceComputeEncoders(trace) + lanes := newLanePacker(3, 4) // Kernels Lane 0..3 for i, encoder := range timeline.Encoders { args := map[string]interface{}{ "encoder_index": encoder.Index, - "duration_us": float64(encoder.Duration) / 1e3, "source": "encoder span", } + if encEvent, ok := timelineEncoderEvent(timeline, encoder.Index); !ok || encEvent.Phase != "i" { + args["duration_us"] = float64(encoder.Duration) / 1e3 + } if len(computeEncoders) > 0 && i < len(computeEncoders) { dispatches := parseEncoderDispatches(trace, computeEncoders, i) if len(dispatches) > 0 { var simdGroups uint64 for _, d := range dispatches { - simdGroups += timelineDispatchSIMDGroup(d) + simdGroups += d.SIMDGroups() } if simdGroups > 0 { args["simd_groups"] = simdGroups @@ -1425,6 +2158,7 @@ func addEncoderKernelEvents(timeline *Timeline, trace *gputrace.Trace, sourceMap } } } + addPipelineCompilerArgs(args, storeStats.PipelineForLabel(encoder.Label), "capture bundle store sections") if sourceMapper != nil { if sourceFile, sourceLine := sourceMapper.SourceLocation(encoder.Label); sourceFile != "" { args["source_available"] = true @@ -1442,22 +2176,99 @@ func addEncoderKernelEvents(timeline *Timeline, trace *gputrace.Trace, sourceMap } timeline.Kernels = append(timeline.Kernels, kernelInfo) + threadID := lanes.assign(encoder.StartTime/1000, encoder.Duration/1000) + if id, ok := timelineEncoderThreadID(timeline, encoder.Index); ok { + threadID = id + } + + phase := "X" + eventDuration := encoder.Duration / 1000 + if encEvent, ok := timelineEncoderEvent(timeline, encoder.Index); ok && encEvent.Phase == "i" { + phase = "i" + eventDuration = 0 + } + timeline.Events = append(timeline.Events, TimelineEvent{ Name: encoder.Label, Category: "kernel", - Phase: "X", + Phase: phase, Timestamp: encoder.StartTime / 1000, - Duration: encoder.Duration / 1000, + Duration: eventDuration, + ProcessID: 1, + ThreadID: threadID, + Args: args, + }) + } +} + +func addCaptureDispatchEvents(timeline *Timeline, trace *gputrace.Trace, sourceMapper *gputrace.ShaderSourceMapper, storeStats *counter.StoreStats) error { + if timeline == nil || trace == nil { + return nil + } + dispatches, err := trace.ParseAttributedDispatches() + if err != nil { + return err + } + for _, dispatch := range dispatches { + name := dispatch.FunctionName + if name == "" { + name = fmt.Sprintf("Dispatch #%d", dispatch.Index) + } + args := map[string]interface{}{ + "dispatch_index": dispatch.Index, + "command_buffer_index": dispatch.CommandBuffer, + "capture_offset": dispatch.CaptureOffset, + "pipeline_state": fmt.Sprintf("0x%x", dispatch.PipelineAddr), + "pipeline_address": dispatch.PipelineAddr, + "pipeline_identity_source": "capture dispatch record", + "pipeline_identity_scope": "capture-local", + "source": "capture dispatch record; dispatch geometry", + "coordinate_source": "capture record order", + "timing_source": "unavailable", + "function_attribution": dispatch.AttributionBasis, + "encoder_attribution": "unavailable", + } + addDispatchGeometryArgs(args, dispatch.DispatchThreads) + if dispatch.FunctionName != "" { + addPipelineCompilerArgs(args, storeStats.PipelineForLabel(dispatch.FunctionName), "capture bundle store sections") + if sourceMapper != nil { + if sourceFile, sourceLine := sourceMapper.SourceLocation(dispatch.FunctionName); sourceFile != "" { + args["source_available"] = true + args["source_file"] = sourceFile + args["source_line"] = sourceLine + } + } + } + timeline.Kernels = append(timeline.Kernels, KernelInfo{Name: name, Encoder: -1, Args: args}) + timeline.Events = append(timeline.Events, TimelineEvent{ + Name: name, + Category: "dispatch", + Phase: "i", + Timestamp: uint64(dispatch.Index), ProcessID: 1, - ThreadID: 3 + (encoder.Index % 4), + ThreadID: 3, Args: args, }) } + return nil +} + +func timelineEncoderEvent(timeline *Timeline, index int) (TimelineEvent, bool) { + if timeline == nil { + return TimelineEvent{}, false + } + for _, event := range timeline.Events { + eventIndex, ok := timelineEventArgInt(event.Args, "index") + if event.Category == "encoder" && ok && eventIndex == index { + return event, true + } + } + return TimelineEvent{}, false } -func traceComputeEncoders(trace *gputrace.Trace) ([]*tracepkg.ComputeEncoder, error) { +func traceComputeEncoders(trace *gputrace.Trace) []*tracepkg.ComputeEncoder { if trace == nil { - return nil, fmt.Errorf("nil trace") + return nil } return trace.ParseComputeEncoders() } @@ -1475,32 +2286,37 @@ func parseEncoderDispatches(trace *gputrace.Trace, encoders []*tracepkg.ComputeE if startOffset < 0 || startOffset >= captureLen || endOffset > captureLen || startOffset >= endOffset { return nil } - dispatches, _ := trace.ParseDispatchInRegion(trace.CaptureData[startOffset:endOffset], startOffset) + dispatches := trace.ParseDispatchInRegion(trace.CaptureData[startOffset:endOffset], startOffset) return dispatches } -func addDispatchKernelEvents(timeline *Timeline, stats *counter.StreamDataStats, simd timelineDispatchSIMDStats, shaderReport *gputrace.ShaderMetricsReport, perfStats *gputrace.PerfCounterStats, encoderMetrics []counter.EncoderCounterMetrics, sourceMapper *gputrace.ShaderSourceMapper) bool { +func addDispatchKernelEvents(timeline *Timeline, stats *counter.StreamDataStats, capture timelineDispatchCaptureStats, shaderReport *gputrace.ShaderMetricsReport, perfStats *gputrace.PerfCounterStats, encoderMetrics []counter.EncoderCounterMetrics, sourceMapper *gputrace.ShaderSourceMapper) bool { if timeline == nil || stats == nil || len(stats.Dispatches) == 0 { return false } - pipelineByIndex := make(map[int]*counter.PipelineStats) + // stats.Pipelines comes from pipelinePerformanceStatistics, an NSDictionary, + // so its slice order is unrelated to the pipeline index that + // gpuCommandInfoData records carry. Join on the pipeline ID instead; both + // sides carry it. + pipelineByID := make(map[int]*counter.PipelineStats, len(stats.Pipelines)) for i := range stats.Pipelines { - pipelineByIndex[i] = &stats.Pipelines[i] + pipelineByID[stats.Pipelines[i].PipelineID] = &stats.Pipelines[i] } + lanes := newLanePacker(3, 4) // Kernels Lane 0..3 metrics := shaderMetricLookup(perfStats) shaderMetrics := timelineShaderReportLookup(shaderReport) encoderMetricByIndex := make(map[int]*counter.EncoderCounterMetrics) for i := range encoderMetrics { - encoderMetricByIndex[encoderMetrics[i].EncoderIndex] = &encoderMetrics[i] + metric := &encoderMetrics[i] + if metric.Attribution == counter.CounterAttributionEncoder && metric.EncoderIndex >= 0 { + encoderMetricByIndex[metric.EncoderIndex] = metric + } } encoderOffsets := make(map[int]uint64) var fallbackStartNs uint64 - for _, d := range stats.Dispatches { - name := d.FunctionName - if name == "" { - name = fmt.Sprintf("(pipeline_%d)", d.PipelineID) - } + for dispatchOrdinal, d := range stats.Dispatches { + name := d.DisplayName() durationNs := uint64(d.DurationUs) * 1000 var startNs uint64 @@ -1513,12 +2329,19 @@ func addDispatchKernelEvents(timeline *Timeline, stats *counter.StreamDataStats, fallbackStartNs += durationNs } - pipeline := pipelineByIndex[d.PipelineIndex] + pipeline := pipelineByID[d.PipelineID] metric := metrics.find(name, pipeline) shaderMetric := shaderMetrics.find(name, pipeline) encoderMetric := encoderMetricByIndex[d.EncoderIndex] - simdGroups, simdCostPct := simd.cost(name, d.Index) - args := dispatchKernelArgs(d, pipeline, simdGroups, simdCostPct, shaderMetric, metric, encoderMetric, sourceMapper) + simdGroups, simdGroupSharePct := capture.work(name, dispatchOrdinal) + args := dispatchKernelArgs(d, pipeline, simdGroups, simdGroupSharePct, shaderMetric, metric, encoderMetric, sourceMapper) + if recorded, ok := capture.dispatch(dispatchOrdinal); ok { + addDispatchGeometryArgs(args, recorded.DispatchThreads) + args["geometry_source"] = "capture dispatch record matched by dispatch order after exact count check" + args["command_buffer_index"] = recorded.CommandBuffer + args["capture_offset"] = recorded.CaptureOffset + args["capture_structure_source"] = "capture dispatch record matched by dispatch order after exact count check" + } info := KernelInfo{ Name: name, @@ -1533,6 +2356,26 @@ func addDispatchKernelEvents(timeline *Timeline, stats *counter.StreamDataStats, timeline.EndTime = info.EndTime } + threadID := lanes.assign(startNs/1000, durationNs/1000) + contained := false + if d.EncoderIndex >= 0 && d.EncoderIndex < len(timeline.Encoders) { + encoder := timeline.Encoders[d.EncoderIndex] + if startNs >= encoder.StartTime && info.EndTime <= encoder.EndTime { + contained = true + if id, ok := timelineEncoderThreadID(timeline, d.EncoderIndex); ok { + threadID = id + } + } + } + if contained { + args["encoder_containment"] = "strict" + } else { + // The cumulative-time bucketing can place a dispatch on either + // side of an encoder boundary. Keep the inferred index in args, + // but leave it on a separate track rather than asserting a + // malformed parent/child relationship in Perfetto. + args["encoder_containment"] = "not_strictly_contained" + } timeline.Events = append(timeline.Events, TimelineEvent{ Name: name, Category: "kernel", @@ -1540,14 +2383,30 @@ func addDispatchKernelEvents(timeline *Timeline, stats *counter.StreamDataStats, Timestamp: startNs / 1000, Duration: durationNs / 1000, ProcessID: 1, - ThreadID: 3 + (d.Index % 4), + ThreadID: threadID, Args: args, }) } return true } -func dispatchKernelArgs(d counter.DispatchInfo, p *counter.PipelineStats, simdGroups uint64, simdCostPct float64, shader *gputrace.ShaderMetrics, hardware *counter.ShaderHardwareMetrics, encoderMetric *counter.EncoderCounterMetrics, sourceMapper *gputrace.ShaderSourceMapper) map[string]interface{} { +// timelineEncoderThreadID returns the lane used by an encoder event. Dispatches +// in that encoder use the same lane so Perfetto shows them inside its span. +func timelineEncoderThreadID(timeline *Timeline, index int) (int, bool) { + if timeline == nil { + return 0, false + } + for _, event := range timeline.Events { + eventIndex, ok := timelineEventArgInt(event.Args, "index") + if event.Category != "encoder" || !ok || eventIndex != index { + continue + } + return event.ThreadID, true + } + return 0, false +} + +func dispatchKernelArgs(d counter.DispatchInfo, p *counter.PipelineStats, simdGroups uint64, simdGroupSharePct float64, shader *gputrace.ShaderMetrics, hardware *counter.ShaderHardwareMetrics, encoderMetric *counter.EncoderCounterMetrics, sourceMapper *gputrace.ShaderSourceMapper) map[string]interface{} { args := map[string]interface{}{ "dispatch_index": d.Index, "duration_us": float64(d.DurationUs), @@ -1560,40 +2419,30 @@ func dispatchKernelArgs(d counter.DispatchInfo, p *counter.PipelineStats, simdGr "xcode_view": "Shaders", "timing_source": "streamData gpuCommandInfoData", } - if d.ExecutionCostPct > 0 { - args["xcode_cost_pct"] = d.ExecutionCostPct - args["profiling_cost_pct"] = d.ExecutionCostPct + if d.ProfilingSampleSharePct > 0 { + args["profiling_sample_share_estimate_pct"] = d.ProfilingSampleSharePct } if simdGroups > 0 { args["simd_groups"] = simdGroups } - if simdCostPct > 0 { - args["xcode_cost_pct"] = simdCostPct + if simdGroupSharePct > 0 { + args["simd_group_share_pct"] = simdGroupSharePct + args["simd_group_share_source"] = "captured dispatch geometry" } if d.SampleCount > 0 { args["gprwcntr_sample_count"] = d.SampleCount args["sampling_density"] = d.SamplingDensity + args["sample_attribution_basis"] = "GPRWCNTR samples in a scaled cumulative-dispatch window" } if d.StartTicks != 0 || d.EndTicks != 0 { args["start_ticks"] = d.StartTicks args["end_ticks"] = d.EndTicks + args["sample_window_basis"] = "cumulative dispatch time scaled over the first APSTimelineData command buffer" + args["sample_timestamp_domain"] = "mach absolute ticks" } - if p != nil { - if p.FunctionName != "" { - args["function_name"] = p.FunctionName - } - if p.PipelineAddress != 0 { - args["pipeline_state"] = fmt.Sprintf("0x%x", p.PipelineAddress) - } - args["allocated_registers"] = p.TemporaryRegisterCount - args["uniform_registers"] = p.UniformRegisterCount - args["spilled_bytes"] = p.SpilledBytes - args["threadgroup_memory"] = p.ThreadgroupMemory - args["instruction_count"] = p.InstructionCount - args["alu_instruction_count"] = p.ALUInstructionCount - args["fp32_instruction_count"] = p.FP32InstructionCount - args["fp16_instruction_count"] = p.FP16InstructionCount - } + addPipelineCompilerArgs(args, p, "streamData pipelinePerformanceStatistics") + args["pipeline_identity_source"] = "streamData pipelineStateInfoData" + args["pipeline_identity_scope"] = "capture-local" if sourceMapper != nil { sourceName := d.FunctionName if sourceName == "" && p != nil { @@ -1606,19 +2455,22 @@ func dispatchKernelArgs(d counter.DispatchInfo, p *counter.PipelineStats, simdGr } } if shader != nil { - if shader.PercentOfTotal > 0 && args["xcode_cost_pct"] == nil { - args["xcode_cost_pct"] = shader.PercentOfTotal + if shader.PercentOfTotal > 0 && args["simd_group_share_pct"] == nil { + args["shader_share_pct"] = shader.PercentOfTotal + args["shader_share_source"] = "shader report" } - if shader.TotalThreadgroups > 0 && args["simd_groups"] == nil { - args["simd_groups"] = shader.TotalThreadgroups + if shader.TotalThreadgroups > 0 { + args["function_simd_groups"] = shader.TotalThreadgroups + args["function_simd_groups_source"] = "shader report" } if shader.TotalDurationNs > 0 { args["shader_duration_ns"] = shader.TotalDurationNs } } if hardware != nil { - if hardware.SIMDGroups > 0 && args["simd_groups"] == nil { - args["simd_groups"] = hardware.SIMDGroups + if hardware.SIMDGroups > 0 && args["function_simd_groups"] == nil { + args["function_simd_groups"] = hardware.SIMDGroups + args["function_simd_groups_source"] = "shader hardware metrics" } if hardware.AllocatedRegs > 0 { args["allocated_registers"] = hardware.AllocatedRegs @@ -1629,19 +2481,16 @@ func dispatchKernelArgs(d counter.DispatchInfo, p *counter.PipelineStats, simdGr if hardware.SpilledBytes > 0 { args["spilled_bytes"] = hardware.SpilledBytes } - if hardware.KernelOccupancy > 0 { - args["occupancy_pct"] = hardware.KernelOccupancy - } if hardware.ALUUtilization > 0 { args["alu_utilization_pct"] = hardware.ALUUtilization } } if encoderMetric != nil { - if args["occupancy_pct"] == nil { - args["occupancy_pct"] = encoderMetric.KernelOccupancy - args["occupancy_source"] = "encoder counter fallback" - } - if args["alu_utilization_pct"] == nil { + // Only fall back to a value that was actually read. A zero here is not + // a measurement of zero, it is the absence of one: Xcode reports ALU + // utilization of 1.59 to 3.35 percent for encoders where this stamped + // 0.00 on every dispatch. + if args["alu_utilization_pct"] == nil && encoderMetric.ALUUtilization != 0 { args["alu_utilization_pct"] = encoderMetric.ALUUtilization args["alu_utilization_source"] = "encoder counter fallback" } @@ -1649,63 +2498,61 @@ func dispatchKernelArgs(d counter.DispatchInfo, p *counter.PipelineStats, simdGr return args } -type timelineDispatchSIMDStats struct { - byIndex []uint64 - byName map[string]uint64 - total uint64 +type timelineDispatchCaptureStats struct { + byIndex []uint64 + dispatches []tracepkg.AttributedDispatch + byName map[string]uint64 + total uint64 } -func timelineDispatchSIMDGroups(t *gputrace.Trace, stats *counter.StreamDataStats) timelineDispatchSIMDStats { - out := timelineDispatchSIMDStats{byName: make(map[string]uint64)} +func timelineDispatchCaptureEvidence(t *gputrace.Trace, stats *counter.StreamDataStats) timelineDispatchCaptureStats { + out := timelineDispatchCaptureStats{byName: make(map[string]uint64)} if t == nil || stats == nil || len(stats.Dispatches) == 0 || len(t.CaptureData) == 0 { return out } - dispatches, err := t.ParseDispatchInRegion(t.CaptureData, 0) - if err != nil || len(dispatches) != len(stats.Dispatches) { + dispatches, err := t.ParseAttributedDispatches() + if err != nil { + return out + } + if len(dispatches) != len(stats.Dispatches) { return out } out.byIndex = make([]uint64, len(dispatches)) + out.dispatches = dispatches for i, d := range dispatches { - groups := timelineDispatchSIMDGroup(d) + groups := d.SIMDGroups() out.byIndex[i] = groups out.total += groups - name := stats.Dispatches[i].FunctionName - if name == "" { - name = fmt.Sprintf("(pipeline_%d)", stats.Dispatches[i].PipelineID) - } - out.byName[name] += groups + out.byName[stats.Dispatches[i].DisplayName()] += groups } return out } -func timelineDispatchSIMDGroup(d tracepkg.DispatchThreads) uint64 { - const simdWidth uint64 = 32 - tgX, tgY, tgZ := uint64(1), uint64(1), uint64(1) - if d.ThreadsPerGroupX > 0 { - tgX = (d.ThreadsX + d.ThreadsPerGroupX - 1) / d.ThreadsPerGroupX - } - if d.ThreadsPerGroupY > 0 { - tgY = (d.ThreadsY + d.ThreadsPerGroupY - 1) / d.ThreadsPerGroupY - } - if d.ThreadsPerGroupZ > 0 { - tgZ = (d.ThreadsZ + d.ThreadsPerGroupZ - 1) / d.ThreadsPerGroupZ +func (s timelineDispatchCaptureStats) dispatch(index int) (tracepkg.AttributedDispatch, bool) { + if index < 0 || index >= len(s.dispatches) { + return tracepkg.AttributedDispatch{}, false } - threadsPerGroup := d.ThreadsPerGroupX * d.ThreadsPerGroupY * d.ThreadsPerGroupZ - if threadsPerGroup == 0 { - return 0 + return s.dispatches[index], true +} + +func addDispatchGeometryArgs(args map[string]interface{}, dispatch tracepkg.DispatchThreads) { + args["grid_size"] = fmt.Sprintf("%d,%d,%d", dispatch.ThreadsX, dispatch.ThreadsY, dispatch.ThreadsZ) + args["threadgroup_size"] = fmt.Sprintf("%d,%d,%d", dispatch.ThreadsPerGroupX, dispatch.ThreadsPerGroupY, dispatch.ThreadsPerGroupZ) + if _, ok := args["simd_groups"]; !ok { + args["simd_groups"] = dispatch.SIMDGroups() } - return (tgX*tgY*tgZ*threadsPerGroup + simdWidth - 1) / simdWidth } -func (s timelineDispatchSIMDStats) cost(name string, index int) (uint64, float64) { - groups := s.byName[name] - if groups == 0 && index >= 0 && index < len(s.byIndex) { - groups = s.byIndex[index] +func (s timelineDispatchCaptureStats) work(name string, index int) (uint64, float64) { + var dispatchGroups uint64 + if index >= 0 && index < len(s.byIndex) { + dispatchGroups = s.byIndex[index] } - if groups == 0 || s.total == 0 { - return groups, 0 + functionGroups := s.byName[name] + if functionGroups == 0 || s.total == 0 { + return dispatchGroups, 0 } - return groups, float64(groups) / float64(s.total) * 100 + return dispatchGroups, float64(functionGroups) / float64(s.total) * 100 } type timelineShaderReport struct { @@ -1805,6 +2652,41 @@ func timelineTimingFromStats(stats *counter.StreamDataStats) *TimelineTiming { return timing } +// enrichTimelineWithXcodeGPUTime optionally reads the total Xcode displays in +// Overview. It annotates the export but does not align the busy and wall +// timeline clocks. +func enrichTimelineWithXcodeGPUTime(tracePath string, timeline *Timeline, enabled bool) error { + if !enabled { + return nil + } + gpuTime, err := readXcodeGPUTime(tracePath) + if err != nil { + return fmt.Errorf("read Xcode GPU time: %w", err) + } + if gpuTime == 0 || timeline == nil { + return nil + } + applyXcodeGPUTime(timeline, gpuTime) + return nil +} + +func applyXcodeGPUTime(timeline *Timeline, gpuTime uint64) { + if timeline == nil || gpuTime == 0 { + return + } + if timeline.Timing == nil { + timeline.Timing = &TimelineTiming{} + } + timeline.Timing.EffectiveGPUTimeNs = &gpuTime + timeline.Timing.DisplayDurationNs = gpuTime + timeline.Timing.DisplayDurationSource = "GTMioTraceData.gpuTime (Xcode Overview GPU Time)" + if timeline.Timing.TimingSource == "" { + timeline.Timing.TimingSource = "GTMioTraceData.gpuTime (Xcode Overview GPU Time)" + } else { + timeline.Timing.TimingSource += "; GTMioTraceData.gpuTime (Xcode Overview GPU Time)" + } +} + func annotateTimelineWithTimingMetrics(timeline *Timeline, metrics *gputrace.TimingMetrics) { source := timelineMetricsSource(metrics) if timeline == nil || source == "" { @@ -1837,9 +2719,181 @@ func timelineMetricsSource(metrics *gputrace.TimingMetrics) string { return fmt.Sprint(metrics.TimingSource) } +// timelineDurationPhase returns the Chrome trace phase for a duration expressed +// in microseconds. A zero duration has no dur field in JSON, so it is an +// instant marker rather than a malformed complete event. +func timelineDurationPhase(durationUs uint64) string { + if durationUs == 0 { + return "i" + } + return "X" +} + +// addRestoreEvents retains APSTimelineData Restore Timestamps on the wall +// clock. They describe replay restore activity, not GPU execution, and stay on +// a separate track from command buffers. +func addRestoreEvents(timeline *Timeline, info *counter.TimelineInfo) { + if timeline == nil || info == nil || info.TimebaseNumer == 0 || info.TimebaseDenom == 0 { + return + } + for _, interval := range info.RestoreTimestamps { + if interval.EndTicks < interval.StartTicks { + continue + } + var startNS uint64 + if interval.StartTicks > info.AbsoluteTime { + startNS = (interval.StartTicks - info.AbsoluteTime) * info.TimebaseNumer / info.TimebaseDenom + } + durationNS := interval.DurationNs(info.TimebaseNumer, info.TimebaseDenom) + phase := "i" + if durationNS != 0 { + phase = "X" + } + timeline.Events = append(timeline.Events, TimelineEvent{ + Name: fmt.Sprintf("Restore #%d", interval.Index), + Category: "restore", + Phase: phase, + Timestamp: startNS / 1000, + Duration: durationNS / 1000, + TimestampNS: startNS, + DurationNS: durationNS, + ProcessID: 1, + ThreadID: 2, + Args: map[string]interface{}{ + "index": interval.Index, + "start_ticks": interval.StartTicks, + "end_ticks": interval.EndTicks, + "raw_start_offset_ns": startNS, + "duration_ns": durationNS, + "timing_source": "APSTimelineData Restore Timestamps", + "clock_domain": "wall", + "evidence_kind": "replay_restore_interval", + }, + }) + } +} + // exportChromeTracing exports timeline in Chrome tracing format. func exportChromeTracing(timeline *Timeline, outputPath string) error { - f, closeOutput, err := createCommandOutput(outputPath) + return exportChromeTracingForClock(timeline, outputPath, timelineClockBusy) +} + +func attachMLXSidecar(timeline *Timeline, tracePath, uuid, sidecarPath string) error { + if uuid == "" { + return fmt.Errorf("attach MLX sidecar: trace UUID is unavailable") + } + sidecar, err := mlxsemantic.Read(sidecarPath) + if err != nil { + return err + } + digest, err := mlxsemantic.Digest(tracePath) + if err != nil { + return err + } + counts := map[string]int{ + "dispatch": timelineEventCount(timeline, "kernel"), + "encoder": timelineEventCount(timeline, "encoder"), + "command_buffer": timelineEventCount(timeline, "command_buffer"), + } + report, err := sidecar.Analyze(mlxsemantic.Identity{UUID: uuid, ContentDigest: digest}, counts) + if err != nil { + return err + } + sidecarDigest, err := mlxsemantic.Digest(sidecarPath) + if err != nil { + return err + } + timeline.MLXSemantics = sidecar + timeline.MLXSemanticReport = &report + timeline.MLXSidecarDigest = sidecarDigest + timeline.MLXSemanticLabelConflict = mlxSemanticLabelConflicts(timeline, sidecar) + return nil +} + +// mlxLabelConflictPolicy states what a conflicting pair does and does not +// produce. It is exported into the evidence manifest so a reader does not have +// to infer the rule from the absence of a merged name. +const mlxLabelConflictPolicy = "both assertions retained; no canonical name, parent, or target link is derived from a conflicted pair" + +// MLXLabelConflict records a native Metal label and a sidecar semantic name +// that assert different names for the same source item. +type MLXLabelConflict struct { + LinkID string `json:"link_id"` + SemanticID string `json:"semantic_id"` + TargetKind string `json:"target_kind"` + TargetIndex int `json:"target_index"` + NativeLabel string `json:"native_label"` + SemanticName string `json:"semantic_name"` +} + +// mlxSemanticLabelConflicts reports sidecar links whose semantic name +// disagrees with the native Metal label observed on the same source item. +// Only encoder labels are native semantic carriers: dispatch names come from +// streamData function records and command-buffer names from capture ordering, +// neither of which is an application assertion. +func mlxSemanticLabelConflicts(timeline *Timeline, sidecar *mlxsemantic.Sidecar) []MLXLabelConflict { + if sidecar == nil { + return nil + } + var conflicts []MLXLabelConflict + for _, link := range sidecar.Links { + if link.Target.Kind != "encoder" { + continue + } + target, ok := timelineEventAt(timeline, "encoder", link.Target.Index) + if !ok || target.Name == "" { + continue + } + node := mlxSemanticNode(sidecar, link.SemanticID) + if node.Name == "" || node.Name == target.Name { + continue + } + conflicts = append(conflicts, MLXLabelConflict{ + LinkID: link.ID, + SemanticID: node.ID, + TargetKind: link.Target.Kind, + TargetIndex: link.Target.Index, + NativeLabel: target.Name, + SemanticName: node.Name, + }) + } + return conflicts +} + +func timelineEventCount(timeline *Timeline, category string) int { + count := 0 + for _, event := range timeline.Events { + if event.Category == category { + count++ + } + } + return count +} + +func timelineEventAt(timeline *Timeline, category string, index int) (TimelineEvent, bool) { + for _, event := range timeline.Events { + if event.Category != category { + continue + } + if index == 0 { + return event, true + } + index-- + } + return TimelineEvent{}, false +} + +// exportPerfettoForClock writes one measured clock domain as native Perfetto +// protobuf. Chrome JSON remains available through --format chrome. +func exportPerfettoForClock(timeline *Timeline, outputPath string, clock timelineClock) error { + return exportPerfettoForClockWithBudget(timeline, outputPath, clock, 0) +} + +func exportPerfettoForClockWithBudget(timeline *Timeline, outputPath string, clock timelineClock, maxBytes int64) error { + if timeline == nil { + return fmt.Errorf("write perfetto trace: nil timeline") + } + w, closeOutput, err := createCommandOutput(outputPath) if err != nil { return err } @@ -1847,16 +2901,2653 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { defer closeOutput() } - // Add process and thread name metadata events - metadataEvents := []TimelineEvent{ - { - Name: "process_name", - Category: "__metadata", - Phase: "M", - ProcessID: 1, - ThreadID: 0, - Args: map[string]interface{}{ - "name": "GPU Trace", + gpuName := timeline.MetalDeviceName + if gpuName == "" { + gpuName = "Apple GPU" + } + trace := &perfetto.Trace{ + Identity: timeline.TraceUUID, + ClockDomain: string(clock), + API: "Metal", + GPUName: gpuName, + GPUVendor: "Apple", + GPUModel: timeline.MetalPluginName, + Metadata: map[string]any{ + "schema": "gputrace.perfetto/v1", + "exporter_version": buildinfo.EffectiveVersion(), + "exporter_commit": buildinfo.Commit, + "exporter_build_date": buildinfo.Date, + "clock_domain": string(clock), + "clock_mapping": "none", + "timing_quality": "measured", + "environment_schema": "gputrace.environment/v1", + "environment_os": runtime.GOOS, + "environment_arch": runtime.GOARCH, + "environment_exporter_runtime": runtime.Version(), + "environment_source": "Go runtime and gputrace metadata", + "environment_parser": "gputrace.perfetto/v1", + "environment_driver_availability": "unavailable", + "environment_mlx_runtime_availability": "unavailable", + "environment_workload_availability": "unavailable", + "environment_capability_catalog_availability": "unavailable", + "capture_mode_availability": "unavailable: capture provenance is not recorded in this trace schema", + "replay_mode_availability": "unavailable: replay provenance is not recorded in this trace schema", + "counter_catalog_availability": "unavailable: APSCounterData pass catalog is absent", + "counter_decoder_availability": "unavailable: no clock-aligned decoded hardware counter series is retained", + "counter_encoder_aggregate_availability": "unavailable: no APSCounterData encoder aggregates were decoded", + "counter_encoder_sample_availability": "unavailable: no capture-attributed APSCounterData records were decoded", + "pipeline_compiler_availability": "unavailable: no pipeline compiler diagnostics were decoded", + "pipeline_compiler_remark_availability": "unavailable: no structured compiler remarks were decoded", + "pipeline_compiler_remark_argument_availability": "unavailable: no structured compiler remark arguments were decoded", + "raw_counter_artifact_availability": "unavailable: no separate raw artifact identity and digest were verified", + "perfetto_schema_revision": perfetto.SchemaRevision, + "packet_family_gpu_info": true, + "packet_family_gpu_render_stage_event": true, + "packet_family_track_event": true, + "packet_family_gpu_counter_event": len(timeline.CounterTracks) > 0, + "unavailable_cpu_scheduling": "Metal trace contains no CPU scheduling evidence", + "unavailable_syscalls": "Metal trace contains no syscall evidence", + "unavailable_cpu_frequency": "Metal trace contains no CPU frequency evidence", + "unavailable_system_memory": "Metal trace contains no system-memory evidence", + }, + } + if len(timeline.CounterCatalog) > 0 { + trace.Metadata["counter_catalog_availability"] = "available: recorded APSCounterData pass columns; names remain opaque" + trace.Metadata["counter_catalog_entries"] = len(timeline.CounterCatalog) + trace.Metadata["counter_catalog_source"] = "APSCounterData Subdivided Dictionary passList" + trace.Metadata["counter_catalog_semantics"] = "recorded column identity only; no values, units, derived meaning, encoder attribution, or clock mapping" + } + if len(timeline.CounterTraceIDs) > 0 { + trace.Metadata["counter_trace_id_availability"] = "available: recorded APSCounterData TraceId table" + trace.Metadata["counter_trace_id_rows"] = len(timeline.CounterTraceIDs) + trace.Metadata["counter_trace_id_source"] = "APSCounterData TraceId to BatchId and TraceId to SampleIndex tables" + trace.Metadata["counter_trace_id_semantics"] = "source row identity; only row ordinal has a positional relation to encoder execution order; no GRC equality or clock mapping" + } else { + trace.Metadata["counter_trace_id_availability"] = "unavailable: APSCounterData TraceId table is absent" + } + if len(timeline.CounterEncoderAggregates) > 0 { + trace.Metadata["counter_encoder_aggregate_availability"] = "available: capture-attributed APSCounterData aggregate rows" + trace.Metadata["counter_encoder_aggregate_count"] = len(timeline.CounterEncoderAggregates) + trace.Metadata["counter_encoder_aggregate_count_semantics"] = "source rows across all pass groups; not a distinct Metal encoder count" + trace.Metadata["counter_encoder_aggregate_source"] = "APSCounterData Derived Counter Sample Data joined to Encoder Infos" + trace.Metadata["counter_encoder_aggregate_clock"] = "raw counter ticks; no verified mapping to busy or wall time" + trace.Metadata["counter_encoder_aggregate_semantics"] = "one aggregate per recorded encoder ID; group and ordinal are pass placement, timestamps remain unaligned" + } + if len(timeline.CounterEncoderSamples) > 0 { + trace.Metadata["counter_encoder_sample_availability"] = "available: capture-attributed APSCounterData source records" + trace.Metadata["counter_encoder_sample_count"] = len(timeline.CounterEncoderSamples) + trace.Metadata["counter_encoder_sample_source"] = "APSCounterData Derived Counter Sample Data joined to Encoder Infos" + trace.Metadata["counter_encoder_sample_clock"] = "raw counter ticks; no verified mapping to busy or wall time" + trace.Metadata["counter_encoder_sample_semantics"] = "fixed GRC fields and opaque hardware-counter vectors in recorded order; no passList join, units, derived meaning, Metal encoder foreign key, or clock mapping" + } + if inventory := timeline.RawProfilerArtifacts; inventory != nil { + trace.Metadata["raw_counter_artifact_availability"] = "available: content-identified profiler archive" + trace.Metadata["raw_profiler_artifact_count"] = len(inventory.Artifacts) + trace.Metadata["raw_profiler_artifact_total_bytes"] = inventory.TotalBytes + trace.Metadata["raw_profiler_artifact_inventory_sha256"] = inventory.SHA256 + trace.Metadata["raw_profiler_artifact_digest_algorithm"] = "sha256" + trace.Metadata["raw_profiler_artifact_scope"] = "regular files directly beneath the resolved .gpuprofiler_raw directory" + var timelineHeaders int + for _, artifact := range inventory.Artifacts { + if artifact.TimelineHeader != nil { + timelineHeaders++ + } + } + trace.Metadata["raw_profiler_timeline_header_count"] = timelineHeaders + trace.Metadata["raw_profiler_timeline_header_semantics"] = "fixed raw file header; timestamp domain is private and unaligned" + } else if timeline.RawProfilerArtifactError != "" { + trace.Metadata["raw_counter_artifact_availability"] = "unavailable: " + timeline.RawProfilerArtifactError + } + if timeline.TraceUUID != "" { + trace.Metadata["input_uuid"] = timeline.TraceUUID + trace.Metadata["input_uuid_availability"] = "available" + } else { + trace.Metadata["input_uuid_availability"] = "unavailable" + } + if timeline.MLXSemantics != nil && timeline.MLXSemantics.Trace.ContentDigest != "" { + trace.Metadata["input_content_digest"] = timeline.MLXSemantics.Trace.ContentDigest + trace.Metadata["input_content_digest_availability"] = "available: verified strict sidecar" + } else { + trace.Metadata["input_content_digest_availability"] = "unavailable: exact tree hashing is performed only for strict sidecar validation" + } + if timeline.DeviceID != 0 { + trace.Metadata["environment_device_id"] = timeline.DeviceID + trace.Metadata["environment_device_id_availability"] = "available" + } else { + trace.Metadata["environment_device_id_availability"] = "unavailable" + } + if timeline.MetalDeviceName != "" { + trace.Metadata["environment_device_name"] = timeline.MetalDeviceName + trace.Metadata["environment_device_name_source"] = "streamData metalDeviceName" + trace.Metadata["environment_device_name_availability"] = "available" + trace.Metadata["environment_device_availability"] = "available: streamData archive identity" + } else { + trace.Metadata["environment_device_name_availability"] = "unavailable: streamData metalDeviceName is absent" + trace.Metadata["environment_device_availability"] = "unavailable: streamData metalDeviceName is absent" + } + if timeline.MetalPluginName != "" { + trace.Metadata["environment_metal_plugin_name"] = timeline.MetalPluginName + trace.Metadata["environment_metal_plugin_source"] = "streamData metalPluginName" + trace.Metadata["environment_metal_plugin_availability"] = "available" + } else { + trace.Metadata["environment_metal_plugin_availability"] = "unavailable: streamData metalPluginName is absent" + } + if timeline.GPUGeneration != nil { + trace.Metadata["environment_gpu_generation"] = *timeline.GPUGeneration + trace.Metadata["environment_gpu_generation_source"] = "streamData gpuGeneration" + trace.Metadata["environment_gpu_generation_availability"] = "available" + } else { + trace.Metadata["environment_gpu_generation_availability"] = "unavailable: streamData gpuGeneration is absent" + } + for key, value := range perfettoStreamMetadataArgs(timeline.StreamMetadata) { + trace.Metadata[key] = value + } + appendStreamDataStringArgs(trace.Metadata, timeline.StreamDataStrings) + appendPipelineCompilerArgs(trace.Metadata, timeline) + if timeline != nil { + for key, value := range perfettoClockConversionArgs(timeline) { + trace.Metadata[key] = value + } + inventory := timeline.EvidenceInventory + if inventory == nil { + value := timelineEvidenceInventory(timeline) + inventory = &value + } + trace.Metadata["source_command_buffer_count"] = inventory.CommandBuffers + trace.Metadata["source_restore_interval_count"] = inventory.RestoreIntervals + trace.Metadata["source_encoder_count"] = inventory.Encoders + trace.Metadata["source_dispatch_count"] = inventory.Dispatches + trace.Metadata["source_untimed_dispatch_count"] = inventory.UntimedDispatches + trace.Metadata["source_raw_profiler_stream_count"] = inventory.ProfilerStreams + trace.Metadata["source_raw_profiler_record_count"] = inventory.ProfilerRecords + trace.Metadata["projected_command_buffer_count"] = timelineEventCount(timeline, "command_buffer") + trace.Metadata["projected_restore_interval_count"] = timelineEventCount(timeline, "restore") + trace.Metadata["projected_encoder_count"] = timelineEventCount(timeline, "encoder") + trace.Metadata["projected_dispatch_count"] = timelineEventCount(timeline, "kernel") + timelineEventCount(timeline, "dispatch") + trace.Metadata["projected_untimed_dispatch_count"] = timelineUntimedDispatchCount(timeline) + trace.Metadata["projected_raw_profiler_stream_count"] = timelineEventCount(timeline, "profiler_stream") + trace.Metadata["projected_raw_profiler_record_count"] = timelineEventCount(timeline, "gprwcntr") + trace.Metadata["raw_profiler_samples"] = timeline.RawProfilerSamples + trace.Metadata["dispatch_count"] = len(timeline.Kernels) + trace.Metadata["untimed_dispatch_count"] = timelineUntimedDispatchCount(timeline) + trace.Metadata["encoder_count"] = len(timeline.Encoders) + trace.Metadata["observed_cs_label_count"] = timeline.ObservedCSLabels + trace.Metadata["unique_cs_label_count"] = timeline.UniqueCSLabels + trace.Metadata["cs_label_semantics"] = "observed capture annotations; not dispatch or encoder instances" + trace.Metadata["command_buffer_count"] = timelineEventCount(timeline, "command_buffer") + if timeline.Timing != nil { + for key, value := range perfettoTimingSummaryArgs(timeline.Timing) { + trace.Metadata[key] = value + } + trace.Metadata["timing_source"] = timeline.Timing.TimingSource + trace.Metadata["timing_approximate"] = timeline.Timing.EncoderTimingApproximate + if timeline.Timing.EncoderTimingApproximate { + trace.Metadata["timing_quality"] = "approximate" + } + } else { + trace.Metadata["timing_source"] = "unavailable" + trace.Metadata["timing_quality"] = "unavailable" + } + if timeline.MLXSemantics != nil { + trace.Metadata["mlx_semantic_schema"] = timeline.MLXSemantics.Schema + trace.Metadata["mlx_semantic_producer_name"] = timeline.MLXSemantics.Producer.Name + trace.Metadata["mlx_semantic_producer_version"] = timeline.MLXSemantics.Producer.Version + trace.Metadata["mlx_semantic_nodes"] = len(timeline.MLXSemantics.Nodes) + trace.Metadata["mlx_semantic_links"] = len(timeline.MLXSemantics.Links) + trace.Metadata["mlx_sidecar_digest"] = timeline.MLXSidecarDigest + if report := timeline.MLXSemanticReport; report != nil { + trace.Metadata["mlx_semantic_used_nodes"] = report.UsedNodes + trace.Metadata["mlx_semantic_unused_nodes"] = report.UnusedNodes + for kind, count := range report.MatchedTargets { + trace.Metadata["mlx_semantic_matched_"+kind] = count + } + for kind, count := range report.UnmatchedTargets { + trace.Metadata["mlx_semantic_unmatched_"+kind] = count + } + } + trace.Metadata["mlx_semantic_label_conflicts"] = len(timeline.MLXSemanticLabelConflict) + trace.Metadata["mlx_semantic_label_conflict_policy"] = mlxLabelConflictPolicy + trace.Metadata["mlx_semantic_label_carrier"] = "Metal encoder labels are the only native semantic carrier compared in v1; dispatch and command-buffer names are source records, not application assertions" + projected, unprojected := mlxSemanticProjectionCounts(timeline) + for kind, count := range projected { + trace.Metadata["mlx_semantic_projected_"+kind] = count + } + for kind, count := range unprojected { + trace.Metadata["mlx_semantic_unprojected_"+kind] = count + trace.Metadata["mlx_semantic_unprojected_"+kind+"_reason"] = "target is outside the selected clock domain" + } + } + if correlation := timeline.HostCorrelation; correlation != nil { + trace.Metadata["host_correlation_schema"] = correlation.Schema + trace.Metadata["host_correlation_run_id"] = correlation.RunID + trace.Metadata["host_correlation_host_digest"] = correlation.HostDigest + trace.Metadata["host_correlation_trace_digest"] = correlation.TraceDigest + trace.Metadata["host_correlation_host_clock"] = correlation.HostClock + trace.Metadata["host_correlation_gpu_clock"] = correlation.GPUClock + trace.Metadata["host_correlation_bridge_digest"] = correlation.BridgeDigest + trace.Metadata["host_correlation_max_error_ns"] = correlation.MaxErrorNS + trace.Metadata["host_correlation_event_count"] = len(correlation.Events) + } + if live := timeline.LiveTiming; live != nil { + trace.Metadata["live_timing_run_id"] = live.RunID + trace.Metadata["live_timing_digest"] = live.ContentDigest + trace.Metadata["live_timing_clock_samples"] = live.ClockSamples + trace.Metadata["live_timing_command_buffers"] = live.CommandBuffers + trace.Metadata["live_timing_projected_command_buffers"] = live.Projected + trace.Metadata["live_timing_unmatched_command_buffers"] = live.Unmatched + } + trace.Metadata["unavailable_evidence_count"] = len(timeline.UnavailableEvidence) + trace.Metadata["unattributed_counter_rows"] = len(timeline.UnattributedCounters) + if len(timeline.UnattributedCounters) > 0 { + trace.Metadata["counter_attribution"] = string(counter.CounterAttributionUnknown) + trace.Metadata["counter_attribution_reason"] = "no capture-backed encoder identity" + } + for i, gap := range timeline.UnavailableEvidence { + trace.Metadata[fmt.Sprintf("unavailable_evidence_%d_family", i)] = gap.Family + trace.Metadata[fmt.Sprintf("unavailable_evidence_%d_reason", i)] = gap.Reason + } + } + + trackNames := make(map[[2]int]string) + if clock == timelineClockBusy { + trackNames[[2]int{1, 1}] = "Compute encoders and dispatches (cumulative busy)" + trackNames[[2]int{1, 3}] = "Unattributed compute dispatches (cumulative busy)" + } else if clock == timelineClockLive { + trackNames[[2]int{2, 0}] = "Command buffers (original live GPU clock)" + } else { + trackNames[[2]int{1, 0}] = "Command buffers (wall clock; APSTimelineData)" + trackNames[[2]int{1, 2}] = "Replay restore intervals (wall clock; APSTimelineData)" + } + for _, event := range timeline.Events { + if event.Phase != "M" || event.Name != "thread_name" { + continue + } + if name, ok := event.Args["name"].(string); ok && name != "" { + trackNames[[2]int{event.ProcessID, event.ThreadID}] = name + } + } + + trackIDs := make(map[[2]int]uint64) + for _, event := range timeline.Events { + if event.Phase == "M" || event.Category == "kernel" { + continue + } + key := [2]int{event.ProcessID, event.ThreadID} + if trackIDs[key] != 0 { + continue + } + identity := fmt.Sprintf("%s/%d/%d", clock, key[0], key[1]) + id := perfetto.TrackUUID("gputrace.timeline", identity) + trackIDs[key] = id + name := trackNames[key] + if name == "" { + name = fmt.Sprintf("%s lane %d", event.Category, event.ThreadID) + } + track := perfetto.Track{ + UUID: id, + Name: name, + Description: fmt.Sprintf("gputrace %s-domain evidence", clock), + } + if clock == timelineClockBusy && key == [2]int{1, 1} { + track.ChildOrder = perfetto.ChildTrackOrderChronological + } + trace.Tracks = append(trace.Tracks, track) + } + + for index, event := range timeline.Events { + if event.Phase == "M" { + continue + } + converted := perfetto.Event{ + ID: uint64(index + 1), + Name: event.Name, + Category: event.Category, + StartNS: event.Timestamp * 1000, + DurationNS: event.Duration * 1000, + Args: perfettoEventArgs(timeline, event, clock), + Required: event.Category == "encoder" || event.Category == "command_buffer" || event.Category == "restore" || event.Category == "live_command_buffer", + } + if event.TimestampNS != 0 || event.DurationNS != 0 { + converted.StartNS = event.TimestampNS + converted.DurationNS = event.DurationNS + } + if event.Category == "kernel" { + converted.Kind = perfetto.EventGPUCompute + } else { + converted.TrackUUID = trackIDs[[2]int{event.ProcessID, event.ThreadID}] + if event.Phase == "i" || event.Duration == 0 { + converted.Kind = perfetto.EventInstant + } else { + converted.Kind = perfetto.EventSlice + } + } + trace.Events = append(trace.Events, converted) + } + if includeMetalDispatchDetailProjection(timeline, clock, maxBytes) { + pipelineTracks, pipelineEvents := appendMetalPipelineProjection(trace, timeline, trackIDs[[2]int{1, 1}]) + encoderTracks, encoderEvents, uncertainTracks, uncertainEvents := appendMetalDispatchDetailProjection(trace, timeline, trackIDs[[2]int{1, 1}]) + trace.Metadata["presentation_pipeline_tracks"] = pipelineTracks + trace.Metadata["presentation_pipeline_events"] = pipelineEvents + trace.Metadata["presentation_encoder_tracks"] = encoderTracks + trace.Metadata["presentation_encoder_events"] = encoderEvents + trace.Metadata["presentation_uncertain_tracks"] = uncertainTracks + trace.Metadata["presentation_uncertain_events"] = uncertainEvents + trace.Metadata["presentation_dispatch_tracks"] = pipelineTracks + encoderTracks + uncertainTracks + trace.Metadata["presentation_dispatch_events"] = pipelineEvents + encoderEvents + uncertainEvents + trace.Metadata["presentation_dispatch_accounting"] = "duplicate detail projection; aggregate GPU totals use native gpu_slice only" + } else if clock == timelineClockBusy { + reason := "omitted from constrained export" + if maxBytes == 0 { + reason = "omitted because dispatch timing is unavailable" + } + trace.Metadata["presentation_dispatch_accounting"] = reason + "; aggregate GPU totals use native gpu_slice only" + } + appendMLXSemanticEvents(trace, timeline) + appendHostCorrelationEvents(trace, timeline) + appendEvidenceDetailEvents(trace, timeline) + + counterTracks := append([]CounterTrack(nil), timeline.CounterTracks...) + sort.SliceStable(counterTracks, func(i, j int) bool { return counterTracks[i].Name < counterTracks[j].Name }) + for _, track := range counterTracks { + // Presence and measured zero are different. A native counter series with + // source-backed samples is retained even when every value is zero. + if len(track.Samples) == 0 { + continue + } + counter := perfetto.Counter{ + ID: uint32(len(trace.Counters) + 1), + Name: track.Name, + Description: track.Description, + } + for _, sample := range track.Samples { + counter.Samples = append(counter.Samples, perfetto.CounterSample{ + TimestampNS: sample.Timestamp, + Value: sample.Value, + }) + } + trace.Counters = append(trace.Counters, counter) + } + + receipt, err := perfetto.WriteWithOptions(w, trace, perfetto.WriteOptions{MaxBytes: maxBytes}) + if err != nil { + return err + } + if receipt.EventsDropped > 0 || receipt.SamplesDropped > 0 { + fmt.Fprintf(os.Stderr, "Perfetto output sampled: retained %d/%d events and %d/%d counter samples within %d logical bytes\n", + receipt.EventsRetained, receipt.EventsConsidered, + receipt.SamplesRetained, receipt.SamplesConsidered, receipt.LogicalBytes) + } + return nil +} + +func perfettoStreamMetadataArgs(metadata *counter.StreamDataMetadata) map[string]any { + args := map[string]any{ + "stream_data_metadata_availability": "unavailable: streamData archive metadata is absent", + } + if metadata == nil { + return args + } + args["stream_data_metadata_availability"] = "available: raw streamData archive root fields" + args["stream_data_metadata_source"] = "streamData keyed archive root" + args["stream_data_profile_mode_semantics"] = "raw private enum values; meanings unverified" + args["stream_data_capture_range_semantics"] = "raw private scalar values; units and relationship unverified" + if metadata.Version != nil { + args["stream_data_version"] = *metadata.Version + } + if metadata.UnixTimestamp != nil { + args["stream_data_unix_timestamp"] = *metadata.UnixTimestamp + } + if metadata.TraceName != "" { + args["stream_data_trace_name"] = metadata.TraceName + } + if metadata.ProfiledExecutionMode != nil { + args["stream_data_profiled_execution_mode"] = *metadata.ProfiledExecutionMode + } + if metadata.ProfiledPerformanceState != nil { + args["stream_data_profiled_performance_state"] = *metadata.ProfiledPerformanceState + } + if metadata.ProfiledProfilerMode != nil { + args["stream_data_profiled_profiler_mode"] = *metadata.ProfiledProfilerMode + } + if metadata.CaptureRangeLocation != nil { + args["stream_data_capture_range_location"] = *metadata.CaptureRangeLocation + } + if metadata.CaptureRangeLength != nil { + args["stream_data_capture_range_length"] = *metadata.CaptureRangeLength + } + if metadata.DataSourceHasUnusedResources != nil { + args["stream_data_has_unused_resources"] = *metadata.DataSourceHasUnusedResources + } + if metadata.SupportsSeparateAPSData != nil { + args["stream_data_supports_separate_aps_data"] = *metadata.SupportsSeparateAPSData + } + if metadata.NumBlitCalls != nil { + args["stream_data_num_blit_calls"] = *metadata.NumBlitCalls + } + appendStreamDataTableArgs(args, "command_buffer", metadata.Tables.CommandBuffers) + appendStreamDataTableArgs(args, "encoder", metadata.Tables.Encoders) + appendStreamDataTableArgs(args, "gpu_command", metadata.Tables.GPUCommands) + appendStreamDataTableArgs(args, "pipeline", metadata.Tables.Pipelines) + appendStreamDataTableArgs(args, "function", metadata.Tables.Functions) + args["stream_data_family_count_semantics"] = "top-level archive array entries; not decoded sample counts" + appendStreamDataFamilyArgs(args, "aps_data", metadata.Families.APSData) + appendStreamDataFamilyArgs(args, "aps_timeline_data", metadata.Families.APSTimelineData) + appendStreamDataFamilyArgs(args, "aps_counter_data", metadata.Families.APSCounterData) + appendStreamDataFamilyArgs(args, "shader_profiler_data", metadata.Families.ShaderProfilerData) + appendStreamDataFamilyArgs(args, "gpu_timeline_data", metadata.Families.GPUTimelineData) + appendStreamDataFamilyArgs(args, "batch_id_filtered_counters_data", metadata.Families.BatchIDFilteredCountersData) + args["stream_data_decoded_family_count_semantics"] = "NSData payload blobs recovered from top-level archive arrays; not records or samples" + appendDecodedStreamDataFamilyArgs(args, "aps_data", metadata.Families.APSData, metadata.DecodedFamilies.APSData) + appendDecodedStreamDataFamilyArgs(args, "aps_timeline_data", metadata.Families.APSTimelineData, metadata.DecodedFamilies.APSTimelineData) + appendDecodedStreamDataFamilyArgs(args, "aps_counter_data", metadata.Families.APSCounterData, metadata.DecodedFamilies.APSCounterData) + appendDecodedStreamDataFamilyArgs(args, "shader_profiler_data", metadata.Families.ShaderProfilerData, metadata.DecodedFamilies.ShaderProfilerData) + appendDecodedStreamDataFamilyArgs(args, "gpu_timeline_data", metadata.Families.GPUTimelineData, metadata.DecodedFamilies.GPUTimelineData) + appendDecodedStreamDataFamilyArgs(args, "batch_id_filtered_counters_data", metadata.Families.BatchIDFilteredCountersData, metadata.DecodedFamilies.BatchIDFilteredCountersData) + appendStreamDataCounterDecodeArgs(args, metadata.CounterDecode) + appendAPSDataInventoryArgs(args, metadata.APSDataInventory) + appendStreamDataArchiveArgs(args, metadata.ArchiveBlobs) + return args +} + +func appendStreamDataArchiveArgs(args map[string]any, blobs []counter.StreamDataBlobInventory) { + const prefix = "stream_data_archive_" + if blobs == nil { + args[prefix+"availability"] = "unavailable: no nested streamData archive blobs were decoded" + return + } + args[prefix+"availability"] = "available: content-identified nested streamData archive blobs" + args[prefix+"blob_count"] = len(blobs) + args[prefix+"semantics"] = "source family, ordinal, content identity, and deterministic nested dictionary and array projection; private values remain uninterpreted" + var keys, nodes, bytes, malformed, scalars, dataValues, containers, descriptorErrors int + var expanded, references, depthLimited, nodeTruncated int + byFamily := make(map[string]int) + for _, blob := range blobs { + keys += len(blob.Keys) + nodes += len(blob.Nodes) + bytes += blob.Bytes + byFamily[blob.Family]++ + if blob.DecodeError != "" { + malformed++ + } + if blob.NodesTruncated { + nodeTruncated++ + } + for _, node := range blob.Nodes { + switch node.ExpansionStatus { + case "expanded": + expanded++ + case "reference": + references++ + case "depth_limit": + depthLimited++ + } + } + for _, key := range blob.Keys { + if key.ScalarType != "" { + scalars++ + } + if key.DataSHA256 != "" { + dataValues++ + } + if key.ContainerCount != nil { + containers++ + } + if key.DescriptorError != "" { + descriptorErrors++ + } + } + } + args[prefix+"key_count"] = keys + args[prefix+"node_count"] = nodes + args[prefix+"expanded_node_count"] = expanded + args[prefix+"reference_node_count"] = references + args[prefix+"depth_limited_node_count"] = depthLimited + args[prefix+"node_truncated_blob_count"] = nodeTruncated + args[prefix+"byte_count"] = bytes + args[prefix+"malformed_blob_count"] = malformed + args[prefix+"scalar_value_count"] = scalars + args[prefix+"data_value_count"] = dataValues + args[prefix+"container_value_count"] = containers + args[prefix+"descriptor_error_count"] = descriptorErrors + for family, count := range byFamily { + args[prefix+family+"_blob_count"] = count + } + programs := countStreamDataPrograms(blobs) + args[prefix+"shader_binary_count"] = programs.binaries + args[prefix+"program_address_mapping_count"] = programs.mappings + args[prefix+"program_address_mapping_binary_match_count"] = programs.matches + args[prefix+"program_address_mapping_binary_unmatched_count"] = programs.mappings - programs.matches + args[prefix+"program_address_semantics"] = "recorded Binaries and Program Address Mappings fields joined by exact capture-local binaryUniqueId; no dispatch, function, source, or timing attribution" + var rootScalars, configuration, options, counterInfo, limiterGroups, limiterSamples, profiling int + var carriers, embeddedArtifacts int + var embeddedArtifactBytes int64 + for _, blob := range blobs { + records := streamDataRecordedScalars(blob) + rootScalars += len(records.rootScalars) + configuration += len(records.configuration) + options += len(records.options) + counterInfo += len(records.counterInfo) + limiterGroups += len(records.limiterGroups) + limiterSamples += len(records.limiterSamples) + profiling += len(records.profiling) + if _, artifacts, ok := streamDataCarrier(blob); ok { + carriers++ + embeddedArtifacts += len(artifacts) + for _, artifact := range artifacts { + embeddedArtifactBytes += int64(artifact.Bytes) + } + } + } + args[prefix+"configuration_record_count"] = configuration + args[prefix+"aps_option_record_count"] = options + args[prefix+"counter_info_record_count"] = counterInfo + args[prefix+"limiter_group_record_count"] = limiterGroups + args[prefix+"limiter_sample_counter_record_count"] = limiterSamples + args[prefix+"profiling_configuration_record_count"] = profiling + args[prefix+"root_scalar_record_count"] = rootScalars + args[prefix+"profiler_carrier_record_count"] = carriers + args[prefix+"embedded_profiler_artifact_record_count"] = embeddedArtifacts + args[prefix+"embedded_profiler_artifact_byte_count"] = embeddedArtifactBytes + args[prefix+"configuration_semantics"] = "recorded streamData configuration and profiling options; names and values are preserved without assigning units, clock mappings, or runtime effects" + args[prefix+"limiter_catalog_semantics"] = "recorded Counter Info, Limiter Counter List Map, and limiter sample counters identities; no counter-value, unit, pass, or derived-limiter attribution" + args[prefix+"profiling_configuration_semantics"] = "recorded apsProfilingConfig, Timebase, Perf Info, Frame Consistent Perf Info, and Kick State Trigger Options scalar leaves; no inferred units, clock mapping, or runtime effects" + args[prefix+"profiler_carrier_semantics"] = "same-blob recorded carrier fields and embedded profiler payload identities; source indexes, ring indexes, serials, and file names remain opaque capture-local values" + args[prefix+"root_scalar_semantics"] = "exhaustive decoded root scalar keys with exact names, types, and canonical JSON; no inferred units, clocks, joins, or runtime effects" +} + +type streamDataProgramCounts struct { + binaries int + mappings int + matches int +} + +func countStreamDataPrograms(blobs []counter.StreamDataBlobInventory) streamDataProgramCounts { + var total streamDataProgramCounts + for _, blob := range blobs { + binaries, mappings := streamDataPrograms(blob) + total.binaries += len(binaries) + total.mappings += len(mappings) + for _, mapping := range mappings { + if mapping.BinaryJoinStatus == "matched" { + total.matches++ + } + } + } + return total +} + +type streamDataShaderBinary struct { + Ordinal int `json:"ordinal"` + UniqueID string `json:"unique_id"` + Bytes int `json:"bytes"` + SHA256 string `json:"sha256"` +} + +type streamDataProgramMapping struct { + Ordinal int `json:"ordinal"` + BinaryUniqueID string `json:"binary_unique_id,omitempty"` + Type string `json:"type,omitempty"` + MappedAddressJSON string `json:"mapped_address_json,omitempty"` + MappedSizeJSON string `json:"mapped_size_json,omitempty"` + EncoderIDJSON string `json:"encoder_id_json,omitempty"` + EncoderIndexJSON string `json:"encoder_index_json,omitempty"` + DrawCallIndexJSON string `json:"draw_call_index_json,omitempty"` + DrawFunctionIndexJSON string `json:"draw_function_index_json,omitempty"` + RecordedIndexJSON string `json:"recorded_index_json,omitempty"` + RecordedFieldCount int `json:"recorded_field_count"` + BinaryBytes *int `json:"binary_bytes,omitempty"` + BinarySHA256 string `json:"binary_sha256,omitempty"` + BinaryJoinStatus string `json:"binary_join_status"` +} + +func streamDataPrograms(blob counter.StreamDataBlobInventory) ([]streamDataShaderBinary, []streamDataProgramMapping) { + var binaries []streamDataShaderBinary + byID := make(map[string]int) + var mappings []streamDataProgramMapping + byPath := make(map[string]int) + for _, node := range blob.Nodes { + if node.ParentPath == "/Binaries" && node.Name != "" && node.ValueKind == "data" && node.DataBytes != nil { + byID[node.Name] = len(binaries) + binaries = append(binaries, streamDataShaderBinary{ + Ordinal: node.Ordinal, UniqueID: node.Name, + Bytes: *node.DataBytes, SHA256: node.DataSHA256, + }) + } + if node.ParentPath == "/Program Address Mappings" && node.Relation == "array" && node.ValueKind == "dictionary" { + byPath[node.Path] = len(mappings) + mappings = append(mappings, streamDataProgramMapping{Ordinal: node.Ordinal}) + continue + } + index, ok := byPath[node.ParentPath] + if !ok || node.ScalarJSON == "" { + continue + } + mapping := &mappings[index] + mapping.RecordedFieldCount++ + switch node.Name { + case "binaryUniqueId": + _ = json.Unmarshal([]byte(node.ScalarJSON), &mapping.BinaryUniqueID) + case "type": + _ = json.Unmarshal([]byte(node.ScalarJSON), &mapping.Type) + case "mappedAddress": + mapping.MappedAddressJSON = node.ScalarJSON + case "mappedSize": + mapping.MappedSizeJSON = node.ScalarJSON + case "encID": + mapping.EncoderIDJSON = node.ScalarJSON + case "encIndex": + mapping.EncoderIndexJSON = node.ScalarJSON + case "drawCallIndex": + mapping.DrawCallIndexJSON = node.ScalarJSON + case "drawFunctionIndex": + mapping.DrawFunctionIndexJSON = node.ScalarJSON + case "index": + mapping.RecordedIndexJSON = node.ScalarJSON + } + } + for i := range mappings { + mapping := &mappings[i] + mapping.BinaryJoinStatus = "unmatched" + if index, ok := byID[mapping.BinaryUniqueID]; ok { + binary := &binaries[index] + bytes := binary.Bytes + mapping.BinaryBytes = &bytes + mapping.BinarySHA256 = binary.SHA256 + mapping.BinaryJoinStatus = "matched" + } + } + return binaries, mappings +} + +type streamDataScalarRecord struct { + Path string `json:"path"` + ParentPath string `json:"parent_path"` + Name string `json:"name,omitempty"` + Ordinal int `json:"ordinal"` + ScalarType string `json:"scalar_type"` + ScalarJSON string `json:"scalar_json"` + Group string `json:"group,omitempty"` + Section string `json:"section,omitempty"` +} + +type streamDataScalarRecords struct { + rootScalars []streamDataScalarRecord + configuration []streamDataScalarRecord + options []streamDataScalarRecord + counterInfo []streamDataScalarRecord + limiterGroups []streamDataScalarRecord + limiterSamples []streamDataScalarRecord + profiling []streamDataScalarRecord +} + +type streamDataCarrierRecord struct { + Fields map[string]string `json:"fields,omitempty"` + ArtifactCount int `json:"artifact_count"` + ArtifactBytes int64 `json:"artifact_bytes"` +} + +type streamDataEmbeddedArtifact struct { + Path string `json:"path"` + ParentPath string `json:"parent_path"` + Kind string `json:"kind"` + Ordinal int `json:"ordinal"` + Bytes int `json:"bytes"` + SHA256 string `json:"sha256"` +} + +func streamDataCarrier(blob counter.StreamDataBlobInventory) (streamDataCarrierRecord, []streamDataEmbeddedArtifact, bool) { + record := streamDataCarrierRecord{Fields: make(map[string]string)} + var artifacts []streamDataEmbeddedArtifact + for _, node := range blob.Nodes { + if node.ParentPath == "" && node.ScalarJSON != "" { + switch node.Path { + case "/APSTraceDataFile", "/Source", "/SourceIndex", "/RingBufferIndex", "/Serial": + record.Fields[strings.TrimPrefix(node.Path, "/")] = node.ScalarJSON + } + } + if node.ValueKind != "data" || node.DataBytes == nil || node.DataSHA256 == "" || strings.HasPrefix(node.Path, "/Binaries/") { + continue + } + kind := strings.TrimPrefix(node.Path, "/") + if slash := strings.IndexByte(kind, '/'); slash >= 0 { + kind = kind[:slash] + } + artifact := streamDataEmbeddedArtifact{ + Path: node.Path, ParentPath: node.ParentPath, Kind: kind, + Ordinal: node.Ordinal, Bytes: *node.DataBytes, SHA256: node.DataSHA256, + } + record.ArtifactCount++ + record.ArtifactBytes += int64(artifact.Bytes) + artifacts = append(artifacts, artifact) + } + return record, artifacts, len(record.Fields) != 0 || len(artifacts) != 0 +} + +func streamDataRecordedScalars(blob counter.StreamDataBlobInventory) streamDataScalarRecords { + var records streamDataScalarRecords + for _, key := range blob.Keys { + if key.ScalarJSON == "" { + continue + } + records.rootScalars = append(records.rootScalars, streamDataScalarRecord{ + Path: "/" + streamDataJSONPointerName(key.Name), Name: key.Name, + Ordinal: key.Ordinal, ScalarType: key.ScalarType, ScalarJSON: key.ScalarJSON, + }) + } + for _, node := range blob.Nodes { + if node.ScalarJSON == "" { + continue + } + record := streamDataScalarRecord{ + Path: node.Path, ParentPath: node.ParentPath, Name: node.Name, + Ordinal: node.Ordinal, ScalarType: node.ScalarType, ScalarJSON: node.ScalarJSON, + } + switch { + case strings.HasPrefix(node.Path, "/Configuration Variables/"): + records.configuration = append(records.configuration, record) + case strings.HasPrefix(node.Path, "/APS Options/"): + records.options = append(records.options, record) + case node.ParentPath == "/Counter Info": + records.counterInfo = append(records.counterInfo, record) + case node.ParentPath == "/limiter sample counters": + records.limiterSamples = append(records.limiterSamples, record) + case strings.HasPrefix(node.ParentPath, "/Limiter Counter List Map/"): + group := strings.TrimPrefix(node.ParentPath, "/Limiter Counter List Map/") + if group != "" && !strings.Contains(group, "/") { + record.Group = group + records.limiterGroups = append(records.limiterGroups, record) + } + } + for _, section := range []string{ + "apsProfilingConfig", + "Kick State Trigger Options", + "Perf Info", + "Frame Consistent Perf Info", + "Timebase", + } { + prefix := "/" + section + "/" + if strings.HasPrefix(node.Path, prefix) { + record.Section = section + records.profiling = append(records.profiling, record) + break + } + } + } + return records +} + +func streamDataJSONPointerName(name string) string { + name = strings.ReplaceAll(name, "~", "~0") + return strings.ReplaceAll(name, "/", "~1") +} + +func appendStreamDataStringArgs(args map[string]any, strings []string) { + args["stream_data_string_table_availability"] = "unavailable: streamData strings array is absent" + if strings == nil { + return + } + args["stream_data_string_table_availability"] = "available: exact ordered streamData strings array" + args["stream_data_string_count"] = len(strings) + args["stream_data_string_source"] = "streamData keyed archive strings NSArray" + args["stream_data_string_semantics"] = "source array index and value only; classification and cross-table relationships remain uninterpreted" +} + +func appendPipelineCompilerArgs(args map[string]any, timeline *Timeline) { + if timeline == nil || len(timeline.PipelineCompilerStats) == 0 { + return + } + args["pipeline_compiler_availability"] = "available: recorded static compiler diagnostics" + args["pipeline_compiler_count"] = len(timeline.PipelineCompilerStats) + args["pipeline_compiler_count_semantics"] = "decoded source records; projected SQL rows may be lower under an explicit output budget" + args["pipeline_compiler_source"] = timeline.PipelineCompilerSource + args["pipeline_compiler_semantics"] = "static compilation evidence; remarks are not measured source-line GPU cost; no clock or dispatch join" + var remarks, locations, resolvedLocations, unresolvedLocations, malformed, passed, missed, analysis int + var arguments, malformedArguments int + for _, pipeline := range timeline.PipelineCompilerStats { + for _, remark := range pipeline.CompilerRemarks { + remarks++ + switch remark.ParseStatus { + case "complete": + locations++ + resolvedLocations++ + case "unresolved_source_location": + locations++ + unresolvedLocations++ + case "malformed": + malformed++ + } + switch remark.Kind { + case "Passed": + passed++ + case "Missed": + missed++ + case "Analysis": + analysis++ + } + for _, argument := range remark.Arguments { + arguments++ + if argument.ParseStatus == "malformed" { + malformedArguments++ + } + } + } + } + if remarks > 0 { + args["pipeline_compiler_remark_availability"] = "available: searchable projection of exact compiler Remarks YAML" + args["pipeline_compiler_remark_count"] = remarks + args["pipeline_compiler_remark_source_location_count"] = locations + args["pipeline_compiler_remark_resolved_source_location_count"] = resolvedLocations + args["pipeline_compiler_remark_unresolved_source_location_count"] = unresolvedLocations + args["pipeline_compiler_remark_malformed_count"] = malformed + args["pipeline_compiler_remark_passed_count"] = passed + args["pipeline_compiler_remark_missed_count"] = missed + args["pipeline_compiler_remark_analysis_count"] = analysis + args["pipeline_compiler_remark_count_semantics"] = "decoded source documents; projected SQL rows may be lower under an explicit output budget" + args["pipeline_compiler_remark_semantics"] = "static compiler pass diagnostics; no duration, sample weight, runtime causality, or source-line GPU cost" + } + if arguments > 0 { + args["pipeline_compiler_remark_argument_availability"] = "available: ordered scalar projection of compiler Remarks Args" + args["pipeline_compiler_remark_argument_count"] = arguments + args["pipeline_compiler_remark_argument_malformed_count"] = malformedArguments + args["pipeline_compiler_remark_argument_count_semantics"] = "decoded source scalar entries; projected SQL rows may be lower when their parent remark is omitted under an explicit output budget" + args["pipeline_compiler_remark_argument_semantics"] = "recorded scalar names, order, raw text, and decoded string values only; pass-specific meaning remains uninterpreted" + } +} + +func appendAPSDataInventoryArgs(args map[string]any, inventory *counter.APSDataInventory) { + const prefix = "stream_data_aps_data_inventory_" + if inventory == nil { + args[prefix+"availability"] = "unavailable: no APSData dictionaries were decoded" + return + } + args[prefix+"availability"] = "available" + args[prefix+"count_semantics"] = "independent dictionary key-presence counts; private payloads remain uninterpreted" + args[prefix+"blobs"] = inventory.Blobs + args[prefix+"dictionaries"] = inventory.Dictionaries + args[prefix+"malformed_blobs"] = inventory.MalformedBlobs + args[prefix+"with_counter_info"] = inventory.WithCounterInfo + args[prefix+"with_shader_profiler_data"] = inventory.WithShaderProfilerData + args[prefix+"with_frame_marker"] = inventory.WithFrameMarker + args[prefix+"with_aps_trace_data_file"] = inventory.WithAPSTraceDataFile + args[prefix+"with_trace_id_tables"] = inventory.WithTraceIDTables + args[prefix+"blob_record_count"] = len(inventory.BlobRecords) + var keys int + for _, blob := range inventory.BlobRecords { + keys += len(blob.Keys) + } + args[prefix+"key_record_count"] = keys + args[prefix+"blob_record_semantics"] = "content identity and sorted root dictionary shape; private values remain uninterpreted" +} + +func appendStreamDataCounterDecodeArgs(args map[string]any, decode *counter.StreamDataCounterDecode) { + const prefix = "stream_data_counter_decode_" + if decode == nil { + args[prefix+"availability"] = "unavailable: no APSCounterData counter archive was decoded" + return + } + args[prefix+"availability"] = "available" + args[prefix+"count_semantics"] = "GPRWCNTR records and archive identity tables; no timeline clock mapping" + args[prefix+"gprwcntr_blobs"] = decode.GPRWCNTRBlobs + args[prefix+"decoded_samples"] = decode.DecodedSamples + args[prefix+"attributed_samples"] = decode.AttributedSamples + args[prefix+"machine_wide_samples"] = decode.MachineWideSamples + args[prefix+"unattributed_samples"] = decode.UnattributedSamples + args[prefix+"known_encoder_ids"] = decode.KnownEncoderIDs + args[prefix+"encoder_aggregates"] = decode.EncoderAggregates + args[prefix+"pass_column_groups"] = decode.PassColumnGroups + args[prefix+"trace_id_rows"] = decode.TraceIDRows + args[prefix+"stride_mismatch_blobs"] = decode.StrideMismatchBlobs +} + +func appendDecodedStreamDataFamilyArgs(args map[string]any, name string, entries, blobs *int64) { + prefix := "stream_data_" + name + "_" + if entries == nil || blobs == nil { + args[prefix+"decode_availability"] = "unavailable: archive array key is absent or malformed" + return + } + args[prefix+"decoded_blob_count"] = *blobs + if *blobs > *entries { + args[prefix+"decode_availability"] = "inconsistent: decoded blob count exceeds archive entry count" + return + } + args[prefix+"non_blob_entry_count"] = *entries - *blobs + args[prefix+"decode_availability"] = "available" +} + +func appendStreamDataFamilyArgs(args map[string]any, name string, count *int64) { + prefix := "stream_data_" + name + "_" + if count == nil { + args[prefix+"availability"] = "unavailable: archive array key is absent or malformed" + return + } + args[prefix+"entry_count"] = *count + args[prefix+"availability"] = "available" +} + +func appendStreamDataTableArgs(args map[string]any, name string, table *counter.StreamDataTable) { + prefix := "stream_data_" + name + "_table_" + if table == nil { + args[prefix+"availability"] = "unavailable: archive data key is absent" + return + } + if table.DecodeError != "" { + args[prefix+"availability"] = "unavailable: " + table.DecodeError + args[prefix+"decode_error"] = table.DecodeError + args[prefix+"raw_bytes_availability"] = "unavailable: source table bytes were not recovered" + args[prefix+"integrity"] = "unavailable: source table bytes were not recovered" + return + } + args[prefix+"availability"] = "available" + args[prefix+"bytes"] = table.Bytes + args[prefix+"sha256"] = table.SHA256 + args[prefix+"raw_bytes_availability"] = "available: exact source bytes retained as one untimed table payload" + if table.RecordSize == nil || table.RecordCount == nil || table.RemainderBytes == nil { + args[prefix+"integrity"] = "unknown: record size is absent or invalid" + return + } + args[prefix+"record_size"] = *table.RecordSize + args[prefix+"record_count"] = *table.RecordCount + args[prefix+"remainder_bytes"] = *table.RemainderBytes + if *table.RemainderBytes == 0 { + args[prefix+"integrity"] = "complete: byte length is divisible by record size" + } else { + args[prefix+"integrity"] = "incomplete: trailing bytes do not form a complete record" + } +} + +type namedStreamDataTable struct { + name string + sourceKey string + table *counter.StreamDataTable +} + +func streamDataTables(metadata *counter.StreamDataMetadata) []namedStreamDataTable { + if metadata == nil { + return nil + } + return []namedStreamDataTable{ + {name: "command_buffer", sourceKey: "commandBufferInfoData", table: metadata.Tables.CommandBuffers}, + {name: "encoder", sourceKey: "encoderInfoData", table: metadata.Tables.Encoders}, + {name: "gpu_command", sourceKey: "gpuCommandInfoData", table: metadata.Tables.GPUCommands}, + {name: "pipeline", sourceKey: "pipelineStateInfoData", table: metadata.Tables.Pipelines}, + {name: "function", sourceKey: "functionInfoData", table: metadata.Tables.Functions}, + } +} + +func streamDataTableEvidenceCount(metadata *counter.StreamDataMetadata) int { + n := 0 + for _, named := range streamDataTables(metadata) { + if named.table != nil && named.table.DecodeError == "" { + n++ + } + } + return n +} + +func streamDataArchiveBlobEvidenceCount(metadata *counter.StreamDataMetadata) int { + if metadata == nil { + return 0 + } + return len(metadata.ArchiveBlobs) +} + +func perfettoClockConversionArgs(timeline *Timeline) map[string]any { + args := map[string]any{ + "clock_conversion_domain": "wall", + "clock_conversion_availability": "unavailable: APSTimelineData absolute time and timebase are incomplete", + "continuous_time_availability": "unavailable: APSTimelineData Continuous Time is absent or zero", + "pstate_availability": "unavailable: APSTimelineData PState field is absent", + } + if timeline != nil && timeline.ContinuousTime != 0 { + args["continuous_time"] = timeline.ContinuousTime + args["continuous_time_domain"] = "raw APSTimelineData field; relationship to exported clocks is unverified" + args["continuous_time_availability"] = "available: retained without conversion or clock mapping" + } + if timeline != nil && timeline.PState != nil { + args["pstate"] = *timeline.PState + args["pstate_source"] = "APSTimelineData PState" + args["pstate_semantics"] = "raw replay performance-state value; unit and operating-point mapping are unverified" + args["pstate_availability"] = "available: retained without interpreting frequency or voltage" + } + if timeline == nil || timeline.AbsoluteTime == 0 || timeline.TimebaseNumer == 0 || timeline.TimebaseDenom == 0 { + return args + } + args["absolute_time"] = timeline.AbsoluteTime + args["timebase_numer"] = timeline.TimebaseNumer + args["timebase_denom"] = timeline.TimebaseDenom + args["clock_conversion_source"] = "APSTimelineData Absolute Time and Timebase" + args["clock_conversion_formula"] = "wall nanoseconds = (ticks - absolute_time) * timebase_numer / timebase_denom" + args["clock_conversion_availability"] = "available: raw wall-domain conversion inputs; no busy-to-wall clock mapping" + return args +} + +func includeMetalDispatchDetailProjection(timeline *Timeline, clock timelineClock, maxBytes int64) bool { + if timeline == nil || clock != timelineClockBusy || maxBytes != 0 { + return false + } + for _, event := range timeline.Events { + if event.Category == "kernel" && event.Duration > 0 { + return true + } + } + return false +} + +type metalPipelineLane struct { + identity string + name string + first uint64 + events [][]TimelineEvent +} + +// appendMetalPipelineProjection adds an Xcode-like function and pipeline view. +// Each dispatch keeps its measured busy-time coordinates. Overlapping uses of +// one pipeline get separate lanes rather than being drawn as nested slices. +func appendMetalPipelineProjection(trace *perfetto.Trace, timeline *Timeline, parent uint64) (trackCount, eventCount int) { + if trace == nil || timeline == nil || parent == 0 { + return 0, 0 + } + groupID := perfetto.TrackUUID("gputrace.pipeline-dispatch-detail", "group") + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: groupID, + ParentUUID: parent, + Name: "Shaders / pipelines (measured busy time)", + Description: "Xcode-like presentation duplicate grouped by recorded pipeline identity; native gpu_slice is the accounting source", + ChildOrder: perfetto.ChildTrackOrderChronological, + }) + + byIdentity := make(map[string][]TimelineEvent) + names := make(map[string]string) + for _, event := range timeline.Events { + if event.Category != "kernel" || event.Duration == 0 { + continue + } + identity, display := metalPipelineIdentity(event) + byIdentity[identity] = append(byIdentity[identity], event) + names[identity] = display + } + lanes := make([]metalPipelineLane, 0, len(byIdentity)) + for identity, events := range byIdentity { + sort.SliceStable(events, func(i, j int) bool { + if events[i].Timestamp != events[j].Timestamp { + return events[i].Timestamp < events[j].Timestamp + } + return events[i].Duration < events[j].Duration + }) + packed := packTimelineEventLanes(events) + lanes = append(lanes, metalPipelineLane{identity: identity, name: names[identity], first: events[0].Timestamp, events: packed}) + } + sort.Slice(lanes, func(i, j int) bool { + if lanes[i].first != lanes[j].first { + return lanes[i].first < lanes[j].first + } + return lanes[i].identity < lanes[j].identity + }) + + nextID := nextPerfettoEventID(trace) + for _, pipeline := range lanes { + for lane, events := range pipeline.events { + name := pipeline.name + if len(pipeline.events) > 1 { + name += fmt.Sprintf(" · lane %d", lane+1) + } + trackID := perfetto.TrackUUID("gputrace.pipeline-dispatch-detail", pipeline.identity+"/"+strconv.Itoa(lane)) + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: trackID, + ParentUUID: groupID, + Name: name, + Description: "Presentation duplicate of measured native GPU dispatch slices grouped by pipeline; do not add to gpu_slice totals", + }) + trackCount++ + for _, event := range events { + args := perfettoEventArgs(timeline, event, timelineClockBusy) + args["presentation_projection"] = "pipeline_dispatch_detail" + args["accounting_source"] = "native gpu_slice" + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, Name: event.Name, Category: "kernel_detail", + StartNS: event.Timestamp * 1000, DurationNS: event.Duration * 1000, + Kind: perfetto.EventSlice, Args: args, + }) + nextID++ + eventCount++ + } + } + } + return trackCount, eventCount +} + +func metalPipelineIdentity(event TimelineEvent) (identity, display string) { + display = event.Name + for _, key := range []string{"pipeline_id", "pipeline_idx", "pipeline_state", "pipeline_address"} { + if value, ok := event.Args[key]; ok && fmt.Sprint(value) != "" { + identity = key + "=" + fmt.Sprint(value) + "/function=" + event.Name + return identity, display + } + } + return "function=" + event.Name, display +} + +func packTimelineEventLanes(events []TimelineEvent) [][]TimelineEvent { + var lanes [][]TimelineEvent + var ends []uint64 + for _, event := range events { + lane := -1 + for i, end := range ends { + if end <= event.Timestamp { + lane = i + break + } + } + if lane < 0 { + lane = len(lanes) + lanes = append(lanes, nil) + ends = append(ends, 0) + } + lanes[lane] = append(lanes[lane], event) + ends[lane] = event.Timestamp + event.Duration + } + return lanes +} + +// appendMetalDispatchDetailProjection adds a secondary sequence view for +// strictly contained dispatches. Non-strict encoder associations are placed in +// a separate group so track nesting does not assert parentage the trace lacks. +func appendMetalDispatchDetailProjection(trace *perfetto.Trace, timeline *Timeline, parent uint64) (trackCount, eventCount, uncertainTrackCount, uncertainEventCount int) { + if trace == nil || timeline == nil || parent == 0 { + return 0, 0, 0, 0 + } + groupID := perfetto.TrackUUID("gputrace.encoder-dispatch-detail", "group") + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: groupID, ParentUUID: parent, + Name: "Dispatch sequence by encoder (strict containment)", + Description: "Secondary presentation duplicate containing only dispatches strictly bounded by the recorded encoder interval", + ChildOrder: perfetto.ChildTrackOrderChronological, + }) + byEncoder := make(map[int][]TimelineEvent) + var uncertain []TimelineEvent + for _, event := range timeline.Events { + if event.Category != "kernel" { + continue + } + index, ok := timelineEventArgInt(event.Args, "encoder_index") + if !ok || index < 0 || fmt.Sprint(event.Args["encoder_containment"]) != "strict" { + uncertain = append(uncertain, event) + continue + } + byEncoder[index] = append(byEncoder[index], event) + } + indices := make([]int, 0, len(byEncoder)) + for index := range byEncoder { + indices = append(indices, index) + } + sort.Ints(indices) + nextID := nextPerfettoEventID(trace) + for _, index := range indices { + events := byEncoder[index] + functions := make(map[string]bool) + var duration uint64 + for _, event := range events { + functions[event.Name] = true + duration += event.Duration + } + name := fmt.Sprintf("Encoder %d · %d dispatches · %.3f ms · %d functions", index, len(events), float64(duration)/1000, len(functions)) + if index < len(timeline.Encoders) { + label := timeline.Encoders[index].Label + if label != "" && label != fmt.Sprintf("encoder_%d", index) { + name += " — " + label + } + } + trackID := perfetto.TrackUUID("gputrace.encoder-dispatch-detail", strconv.Itoa(index)) + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: trackID, + ParentUUID: groupID, + Name: name, + Description: "Presentation duplicate of native GPU dispatch slices; do not add to gpu_slice totals", + }) + trackCount++ + for _, event := range events { + args := perfettoEventArgs(timeline, event, timelineClockBusy) + args["presentation_projection"] = "encoder_dispatch_detail" + args["accounting_source"] = "native gpu_slice" + kind := perfetto.EventSlice + if event.Duration == 0 { + kind = perfetto.EventInstant + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, + TrackUUID: trackID, + Name: event.Name, + Category: "kernel_detail", + StartNS: event.Timestamp * 1000, + DurationNS: event.Duration * 1000, + Kind: kind, + Args: args, + }) + nextID++ + eventCount++ + } + } + if len(uncertain) > 0 { + uncertainGroup := perfetto.TrackUUID("gputrace.uncertain-encoder-detail", "group") + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: uncertainGroup, ParentUUID: parent, + Name: "Dispatches without strict encoder containment", + Description: "Reported encoder indices are retained as event arguments but are not represented as track parentage", + ChildOrder: perfetto.ChildTrackOrderChronological, + }) + sort.SliceStable(uncertain, func(i, j int) bool { return uncertain[i].Timestamp < uncertain[j].Timestamp }) + for lane, events := range packTimelineEventLanes(uncertain) { + trackID := perfetto.TrackUUID("gputrace.uncertain-encoder-detail", strconv.Itoa(lane)) + name := "Uncertain encoder association" + if lane > 0 { + name += fmt.Sprintf(" · lane %d", lane+1) + } + trace.Tracks = append(trace.Tracks, perfetto.Track{UUID: trackID, ParentUUID: uncertainGroup, Name: name}) + uncertainTrackCount++ + for _, event := range events { + args := perfettoEventArgs(timeline, event, timelineClockBusy) + args["presentation_projection"] = "uncertain_encoder_detail" + args["accounting_source"] = "native gpu_slice" + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, Name: event.Name, Category: "kernel_detail", + StartNS: event.Timestamp * 1000, DurationNS: event.Duration * 1000, + Kind: perfetto.EventSlice, Args: args, + }) + nextID++ + uncertainEventCount++ + } + } + } + return trackCount, eventCount, uncertainTrackCount, uncertainEventCount +} + +func nextPerfettoEventID(trace *perfetto.Trace) uint64 { + next := uint64(1) + for _, event := range trace.Events { + if event.ID >= next { + next = event.ID + 1 + } + } + return next +} + +func mlxSemanticProjectionCounts(timeline *Timeline) (projected, unprojected map[string]int) { + projected = make(map[string]int) + unprojected = make(map[string]int) + if timeline == nil || timeline.MLXSemantics == nil { + return projected, unprojected + } + for _, link := range timeline.MLXSemantics.Links { + if _, ok := timelineSemanticTargetEvent(timeline, link.Target.Kind, link.Target.Index); ok { + projected[link.Target.Kind]++ + } else { + unprojected[link.Target.Kind]++ + } + } + return projected, unprojected +} + +func timelineUntimedDispatchCount(timeline *Timeline) int { + count := 0 + for _, event := range timeline.Events { + if (event.Category == "kernel" || event.Category == "dispatch") && (event.Phase == "i" || event.Duration == 0) { + count++ + } + } + return count +} + +func perfettoEventArgs(timeline *Timeline, event TimelineEvent, clock timelineClock) map[string]any { + args := make(map[string]any, len(event.Args)+3) + for key, value := range event.Args { + args[key] = value + } + args["clock_domain"] = string(clock) + args["timing_quality"] = perfettoTimingQualityForClock(timeline, clock) + if _, ok := args["timing_source"]; !ok && timeline != nil && timeline.Timing != nil && timeline.Timing.TimingSource != "" { + args["timing_source"] = timeline.Timing.TimingSource + } + return args +} + +func perfettoTimingQualityForClock(timeline *Timeline, clock timelineClock) string { + if clock == timelineClockLive && timelineHasMeasuredClock(timeline, clock) { + return "measured" + } + return perfettoTimingQuality(timeline) +} + +func perfettoTimingQuality(timeline *Timeline) string { + if timeline == nil || timeline.Timing == nil || timeline.Timing.TimingSource == "" || timeline.Timing.TimingSource == "unavailable" { + return "unavailable" + } + if timeline.Timing.EncoderTimingApproximate { + return "approximate" + } + return "measured" +} + +func appendMLXSemanticEvents(trace *perfetto.Trace, timeline *Timeline) { + if timeline.MLXSemantics == nil { + return + } + trackIDs := make(map[string]uint64) + for _, node := range timeline.MLXSemantics.Nodes { + trackIDs[node.ID] = perfetto.TrackUUID("gputrace.mlx", node.ID) + } + for _, node := range timeline.MLXSemantics.Nodes { + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: trackIDs[node.ID], + ParentUUID: trackIDs[node.ParentID], + Name: node.Name, + Description: "MLX " + node.Kind + " semantic evidence", + }) + } + conflicts := make(map[string]MLXLabelConflict, len(timeline.MLXSemanticLabelConflict)) + for _, conflict := range timeline.MLXSemanticLabelConflict { + conflicts[conflict.LinkID] = conflict + } + nextID := nextPerfettoEventID(trace) + for _, node := range timeline.MLXSemantics.Nodes { + args := make(map[string]any, len(node.Attrs)+5) + for key, value := range node.Attrs { + args[key] = value + } + args["semantic_id"] = node.ID + args["semantic_parent_id"] = node.ParentID + args["semantic_kind"] = node.Kind + args["join_basis"] = "sidecar-declaration" + args["clock_domain"] = "none" + args["timing_source"] = "MLX semantic sidecar declaration" + args["timing_quality"] = "unavailable" + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, + TrackUUID: trackIDs[node.ID], + Name: node.Name, + Category: "mlx_semantic_node", + Kind: perfetto.EventInstant, + Required: true, + Args: args, + }) + nextID++ + } + for _, link := range timeline.MLXSemantics.Links { + target, ok := timelineSemanticTargetEvent(timeline, link.Target.Kind, link.Target.Index) + if !ok { + continue // The target belongs to another measured clock domain. + } + node := mlxSemanticNode(timeline.MLXSemantics, link.SemanticID) + args := make(map[string]any, len(node.Attrs)+4) + for key, value := range node.Attrs { + args[key] = value + } + args["semantic_id"] = node.ID + args["semantic_kind"] = node.Kind + args["semantic_link_id"] = link.ID + args["join_basis"] = "sidecar-explicit-id" + args["target_kind"] = link.Target.Kind + args["target_index"] = link.Target.Index + if conflict, ok := conflicts[link.ID]; ok { + args["native_label"] = conflict.NativeLabel + args["native_label_source"] = "Metal encoder label" + args["label_conflict"] = "conflicting_name_assertion" + args["label_conflict_policy"] = mlxLabelConflictPolicy + } else if link.Target.Kind == "encoder" && target.Name != "" { + args["native_label"] = target.Name + args["native_label_source"] = "Metal encoder label" + args["label_conflict"] = "none" + } + args["clock_domain"] = timeline.ClockDomain + args["timing_quality"] = perfettoTimingQuality(timeline) + if target.Args != nil { + if source, ok := target.Args["timing_source"]; ok { + args["timing_source"] = source + } + } + if _, ok := args["timing_source"]; !ok && timeline.Timing != nil { + args["timing_source"] = timeline.Timing.TimingSource + } + kind := perfetto.EventSlice + if target.Duration == 0 { + kind = perfetto.EventInstant + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, + TrackUUID: trackIDs[node.ID], + Name: node.Name, + Category: "mlx_semantic", + StartNS: target.Timestamp * 1000, + DurationNS: target.Duration * 1000, + Kind: kind, + Required: true, + Args: args, + }) + nextID++ + } +} + +func appendEvidenceDetailEvents(trace *perfetto.Trace, timeline *Timeline) { + if len(timeline.UnattributedCounters) == 0 && len(timeline.CounterCatalog) == 0 && len(timeline.CounterTraceIDs) == 0 && len(timeline.CounterEncoderAggregates) == 0 && len(timeline.CounterEncoderSamples) == 0 && len(timeline.UnavailableEvidence) == 0 && + (timeline.RawProfilerArtifacts == nil || len(timeline.RawProfilerArtifacts.Artifacts) == 0) && streamDataTableEvidenceCount(timeline.StreamMetadata) == 0 && streamDataArchiveBlobEvidenceCount(timeline.StreamMetadata) == 0 && len(timeline.StreamDataStrings) == 0 && len(timeline.PipelineCompilerStats) == 0 { + return + } + trackID := perfetto.TrackUUID("gputrace.evidence", "details") + eventStart := len(trace.Events) + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: trackID, + Name: "Evidence details (untimed)", + Description: "Source-backed evidence without a verified timeline coordinate", + }) + defer shardUntimedEvidenceTracks(trace, trackID, eventStart) + nextID := nextPerfettoEventID(trace) + if metadata := timeline.StreamMetadata; metadata != nil { + for _, blob := range metadata.ArchiveBlobs { + args := map[string]any{ + "family": blob.Family, + "blob_ordinal": blob.Ordinal, + "byte_count": blob.Bytes, + "blob_sha256": blob.SHA256, + "dictionary": blob.Dictionary, + "key_count": len(blob.Keys), + "node_count": len(blob.Nodes), + "nodes_truncated": blob.NodesTruncated, + "source": "streamData nested NSData archive entry", + "semantics": "content identity and deterministic nested dictionary and array projection; private values remain uninterpreted", + "clock_domain": "none", + "timing_quality": "unavailable", + } + if blob.DecodeError != "" { + args["decode_error"] = blob.DecodeError + } + for ordinal, node := range blob.Nodes { + data, err := json.Marshal(node) + if err != nil { + args[fmt.Sprintf("archive_node_%06d_error", ordinal)] = err.Error() + continue + } + args[fmt.Sprintf("archive_node_%06d_json", ordinal)] = string(data) + } + binaries, mappings := streamDataPrograms(blob) + for ordinal, binary := range binaries { + data, err := json.Marshal(binary) + if err != nil { + args[fmt.Sprintf("shader_binary_%06d_error", ordinal)] = err.Error() + continue + } + args[fmt.Sprintf("shader_binary_%06d_json", ordinal)] = string(data) + } + for ordinal, mapping := range mappings { + data, err := json.Marshal(mapping) + if err != nil { + args[fmt.Sprintf("program_address_mapping_%06d_error", ordinal)] = err.Error() + continue + } + args[fmt.Sprintf("program_address_mapping_%06d_json", ordinal)] = string(data) + } + records := streamDataRecordedScalars(blob) + appendStreamDataScalarArgs(args, "root_scalar", records.rootScalars) + appendStreamDataScalarArgs(args, "stream_configuration", records.configuration) + appendStreamDataScalarArgs(args, "aps_option", records.options) + appendStreamDataScalarArgs(args, "counter_info", records.counterInfo) + appendStreamDataScalarArgs(args, "limiter_group_counter", records.limiterGroups) + appendStreamDataScalarArgs(args, "limiter_sample_counter", records.limiterSamples) + appendStreamDataScalarArgs(args, "profiling_configuration", records.profiling) + if carrier, artifacts, ok := streamDataCarrier(blob); ok { + data, err := json.Marshal(carrier) + if err != nil { + args["profiler_carrier_error"] = err.Error() + } else { + args["profiler_carrier_json"] = string(data) + } + for ordinal, artifact := range artifacts { + data, err := json.Marshal(artifact) + if err != nil { + args[fmt.Sprintf("embedded_profiler_artifact_%06d_error", ordinal)] = err.Error() + continue + } + args[fmt.Sprintf("embedded_profiler_artifact_%06d_json", ordinal)] = string(data) + } + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: fmt.Sprintf("%s blob %d", blob.Family, blob.Ordinal), + Category: "stream_data_archive_blob", Kind: perfetto.EventInstant, + Args: args, + }) + nextID++ + for _, key := range blob.Keys { + args := map[string]any{ + "family": blob.Family, + "blob_ordinal": blob.Ordinal, + "key_ordinal": key.Ordinal, + "recorded_name": key.Name, + "value_kind": key.ValueKind, + "blob_sha256": blob.SHA256, + "source": "streamData nested archive root NSDictionary", + "semantics": "sorted root key identity and exact non-object value descriptor; private meaning remains uninterpreted", + "clock_domain": "none", + "timing_quality": "unavailable", + } + if key.ScalarType != "" { + args["scalar_type"] = key.ScalarType + args["scalar_json"] = key.ScalarJSON + } + if key.DataBytes != nil { + args["data_bytes"] = *key.DataBytes + args["data_sha256"] = key.DataSHA256 + } + if key.ContainerCount != nil { + args["container_count"] = *key.ContainerCount + } + if key.DescriptorError != "" { + args["descriptor_error"] = key.DescriptorError + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: fmt.Sprintf("%s blob %d key %s", blob.Family, blob.Ordinal, key.Name), + Category: "stream_data_archive_key", Kind: perfetto.EventInstant, + Args: args, + }) + nextID++ + } + } + } + for _, sample := range timeline.CounterEncoderSamples { + values, _ := json.Marshal(sample.Counters) + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: fmt.Sprintf("Counter encoder sample: 0x%x", sample.EncoderID), + Category: "counter_encoder_sample", Kind: perfetto.EventInstant, + Args: map[string]any{ + "blob_ordinal": sample.BlobOrdinal, + "record_ordinal": sample.RecordOrdinal, + "encoder_group": sample.EncoderGroup, + "execution_ordinal": sample.ExecutionOrdinal, + "counter_timestamp": sample.Timestamp, + "gpu_cycles": sample.GPUCycles, + "sample_type": sample.SampleType, + "encoder_id": sample.EncoderID, + "kick_trace_id": sample.KickTraceID, + "kick_slot_index": sample.KickSlotIdx, + "source_id": sample.SourceID, + "counter_value_count": len(sample.Counters), + "counter_values_json": string(values), + "attribution_basis": "GRC encoder ID present in APSCounterData Encoder Infos", + "source": "APSCounterData Derived Counter Sample Data", + "semantics": "source record fixed fields and opaque counter vector in recorded order; no passList join, units, Metal encoder foreign key, or timeline coordinate", + "clock_domain": "counter_raw", + "clock_mapping": "none", + "timing_quality": "measured_unaligned", + }, + }) + nextID++ + } + for _, aggregate := range timeline.CounterEncoderAggregates { + args := map[string]any{ + "encoder_id": aggregate.EncoderID, + "pass_group": aggregate.Group, + "execution_ordinal": aggregate.Ordinal, + "sample_count": aggregate.SampleCount, + "end_sample_count": aggregate.EndSamples, + "gpu_cycles": aggregate.GPUCycles, + "counter_start_ticks": aggregate.StartTicks, + "counter_end_ticks": aggregate.EndTicks, + "counter_duration_ns": aggregate.DurationNs, + "attribution_basis": "GRC encoder ID present in APSCounterData Encoder Infos", + "source": "APSCounterData Derived Counter Sample Data", + "semantics": "capture-attributed counter aggregate; no Metal encoder foreign key or timeline coordinate", + "clock_domain": "counter_raw", + "clock_mapping": "none", + "timing_quality": "measured_unaligned", + } + if aggregate.KickTraceID != 0 { + args["kick_trace_id"] = aggregate.KickTraceID + } + if aggregate.BatchIDRecorded { + args["batch_id"] = aggregate.BatchID + } + if aggregate.SampleIndexRecorded { + args["sample_index"] = aggregate.SampleIndex + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: fmt.Sprintf("Counter encoder aggregate: 0x%x", aggregate.EncoderID), + Category: "counter_encoder_aggregate", Kind: perfetto.EventInstant, + Args: args, + }) + nextID++ + } + for _, column := range timeline.CounterCatalog { + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: fmt.Sprintf("Counter catalog %d/%d", column.GroupOrdinal, column.ColumnOrdinal), + Category: "counter_catalog", Kind: perfetto.EventInstant, Required: true, + Args: map[string]any{ + "group_ordinal": column.GroupOrdinal, + "column_ordinal": column.ColumnOrdinal, + "recorded_name": column.RecordedName, + "classification": column.Classification, + "source": "APSCounterData Subdivided Dictionary passList", + "semantics": "recorded column identity only; no values, units, derived meaning, encoder attribution, or clock mapping", + "clock_domain": "none", + "timing_quality": "unavailable", + }, + }) + nextID++ + } + for _, row := range timeline.CounterTraceIDs { + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: fmt.Sprintf("Counter TraceId row %d", row.RowOrdinal), + Category: "counter_trace_id", Kind: perfetto.EventInstant, Required: true, + Args: map[string]any{ + "row_ordinal": row.RowOrdinal, + "trace_id": row.TraceID, + "batch_id": row.BatchID, + "sample_index": row.SampleIndex, + "source": "APSCounterData TraceId to BatchId and TraceId to SampleIndex tables", + "semantics": "source row identity; only row ordinal has a positional relation to encoder execution order; no GRC equality or clock mapping", + "clock_domain": "none", + "timing_quality": "unavailable", + }, + }) + nextID++ + } + for _, named := range streamDataTables(timeline.StreamMetadata) { + if named.table == nil || named.table.DecodeError != "" { + continue + } + trace.Events = append(trace.Events, streamDataTableEvent(nextID, trackID, named)) + nextID++ + } + for index, value := range timeline.StreamDataStrings { + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: fmt.Sprintf("streamData string %d", index), + Category: "stream_data_string", Kind: perfetto.EventInstant, Required: true, + Args: map[string]any{ + "source_index": index, + "recorded_value": value, + "source": "streamData keyed archive strings NSArray", + "semantics": "source array index and value only; classification and cross-table relationships remain uninterpreted", + "clock_domain": "none", + "timing_quality": "unavailable", + }, + }) + nextID++ + } + for _, pipeline := range timeline.PipelineCompilerStats { + name := pipeline.DisplayName() + args := map[string]any{ + "pipeline_id": pipeline.PipelineID, + "function_name": pipeline.FunctionName, + "source": timeline.PipelineCompilerSource, + "semantics": "static compilation evidence; remarks are not measured source-line GPU cost; no clock or dispatch join", + "clock_domain": "none", + "timing_quality": "unavailable", + } + if pipeline.PipelineAddress != 0 { + args["pipeline_address"] = pipeline.PipelineAddress + args["pipeline_identity_scope"] = "capture-local" + } + if pipeline.Remarks != nil { + args["remarks"] = *pipeline.Remarks + } + if performance := pipeline.CompilePerformance; performance != nil { + appendPipelineCompilePerformanceArgs(args, performance) + } + addPipelineCompilerArgs(args, &pipeline, timeline.PipelineCompilerSource) + if pipeline.RecordedStatistics != nil { + names, _ := json.Marshal(pipeline.RecordedStatistics) + args["recorded_statistic_count"] = len(pipeline.RecordedStatistics) + args["recorded_statistics_json"] = string(names) + args["recorded_statistics_semantics"] = "exact sorted top-level source keys; presence only for opaque values" + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: "Pipeline compiler: " + name, Category: "pipeline_compiler", + Kind: perfetto.EventInstant, Args: args, + }) + nextID++ + for _, remark := range pipeline.CompilerRemarks { + remarkArgs := map[string]any{ + "pipeline_id": pipeline.PipelineID, + "function_name": pipeline.FunctionName, + "remark_index": remark.Index, + "remark_kind": remark.Kind, + "compiler_pass": remark.Pass, + "remark_name": remark.Name, + "remark_function": remark.Function, + "parse_status": remark.ParseStatus, + "source": timeline.PipelineCompilerSource + " Remarks", + "semantics": "static compiler pass diagnostic; no duration, sample weight, runtime causality, or source-line GPU cost", + "clock_domain": "none", + "timing_quality": "unavailable", + } + if pipeline.PipelineAddress != 0 { + remarkArgs["pipeline_address"] = pipeline.PipelineAddress + remarkArgs["pipeline_identity_scope"] = "capture-local" + } + if (remark.ParseStatus == "complete" || remark.ParseStatus == "unresolved_source_location") && + remark.SourceLine != nil && remark.SourceColumn != nil { + remarkArgs["source_file"] = remark.SourceFile + remarkArgs["source_line"] = *remark.SourceLine + remarkArgs["source_column"] = *remark.SourceColumn + } + if len(remark.Arguments) > 0 { + encoded, _ := json.Marshal(remark.Arguments) + remarkArgs["argument_count"] = len(remark.Arguments) + remarkArgs["arguments_json"] = string(encoded) + remarkArgs["argument_semantics"] = "ordered recorded scalar entries; values are strings and pass-specific meaning remains uninterpreted" + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, TrackUUID: trackID, + Name: "Compiler remark: " + remark.Kind + " " + remark.Pass + "/" + remark.Name, + Category: "pipeline_compiler_remark", Kind: perfetto.EventInstant, + Args: remarkArgs, + }) + nextID++ + } + } + for _, metric := range timeline.UnattributedCounters { + label := metric.Label + if label == "" { + label = "(pipeline unknown)" + } + args := make(map[string]any, len(metric.Values)+7) + for key, value := range metric.Values { + args[key] = value + } + args["pipeline_label"] = label + args["attribution"] = metric.Attribution + args["metric_scope"] = "pipeline" + args["source"] = metric.Source + args["clock_domain"] = "none" + args["timing_quality"] = "unavailable" + args["attribution_reason"] = "no capture-backed encoder identity" + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, + TrackUUID: trackID, + Name: "Unattributed counter metrics: " + label, + Category: "counter_attribution", + Kind: perfetto.EventInstant, + Required: true, + Args: args, + }) + nextID++ + } + for _, gap := range timeline.UnavailableEvidence { + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, + TrackUUID: trackID, + Name: "Unavailable evidence: " + gap.Family, + Category: "evidence_gap", + Kind: perfetto.EventInstant, + Required: true, + Args: map[string]any{ + "family": gap.Family, + "reason": gap.Reason, + "clock_domain": "none", + "timing_quality": "unavailable", + }, + }) + nextID++ + } + if timeline.RawProfilerArtifacts == nil { + return + } + for _, artifact := range timeline.RawProfilerArtifacts.Artifacts { + args := map[string]any{ + "name": artifact.Name, + "kind": artifact.Kind, + "size_bytes": artifact.Size, + "sha256": artifact.SHA256, + "digest_algorithm": "sha256", + "path_scope": "basename within resolved .gpuprofiler_raw directory", + "clock_domain": "none", + "timing_quality": "unavailable", + } + if artifact.Index != nil { + args["file_index"] = *artifact.Index + } + if header := artifact.TimelineHeader; header != nil { + args["timeline_header_magic"] = fmt.Sprintf("0x%016x", header.Magic) + args["timeline_counter_count"] = header.CounterCount + args["timeline_data_offset_bytes"] = header.DataOffset + args["timeline_entry_count"] = header.EntryCount + args["timeline_timestamp_raw"] = header.Timestamp + args["timeline_timestamp_semantics"] = "raw private profiler-sampling timestamp; not command-buffer or cumulative GPU-busy time" + } + trace.Events = append(trace.Events, perfetto.Event{ + ID: nextID, + TrackUUID: trackID, + Name: "Raw profiler artifact: " + artifact.Name, + Category: "raw_profiler_artifact", + Kind: perfetto.EventInstant, + Required: true, + Args: args, + }) + nextID++ + } +} + +func appendStreamDataScalarArgs(args map[string]any, prefix string, records []streamDataScalarRecord) { + for ordinal, record := range records { + data, err := json.Marshal(record) + if err != nil { + args[fmt.Sprintf("%s_%06d_error", prefix, ordinal)] = err.Error() + continue + } + args[fmt.Sprintf("%s_%06d_json", prefix, ordinal)] = string(data) + } +} + +// shardUntimedEvidenceTracks preserves debug annotations when many untimed +// instants share one coordinate. Trace Processor stops assigning argument sets +// after a bounded same-timestamp depth on one track; separate tracks retain the +// evidence without manufacturing time between static records. +func shardUntimedEvidenceTracks(trace *perfetto.Trace, firstTrack uint64, eventStart int) { + const eventsPerTrack = 60 + for index := eventStart; index < len(trace.Events); index++ { + shard := (index - eventStart) / eventsPerTrack + if shard == 0 { + continue + } + if (index-eventStart)%eventsPerTrack == 0 { + trace.Tracks = append(trace.Tracks, perfetto.Track{ + UUID: perfetto.TrackUUID("gputrace.evidence", fmt.Sprintf("details-%d", shard+1)), + Name: fmt.Sprintf("Evidence details (untimed) %d", shard+1), + Description: "Source-backed evidence without a verified timeline coordinate", + }) + } + trace.Events[index].TrackUUID = perfetto.TrackUUID("gputrace.evidence", fmt.Sprintf("details-%d", shard+1)) + } +} + +func appendPipelineCompilePerformanceArgs(args map[string]any, performance *counter.PipelineCompilePerformance) { + if performance.FunctionWasCached != nil { + args["function_was_cached"] = *performance.FunctionWasCached + } + for _, field := range []struct { + name string + value *int64 + }{ + {"compiler_backend_ns", performance.CompilerBackendNanoseconds}, + {"compiler_optimization_ns", performance.CompilerOptimizationNanoseconds}, + {"compiler_translator_ns", performance.CompilerTranslatorNanoseconds}, + {"compiler_total_ns", performance.CompilerTotalNanoseconds}, + {"driver_total_ns", performance.DriverTotalNanoseconds}, + {"synchronous_service_ns", performance.SynchronousServiceNanoseconds}, + } { + if field.value == nil { + continue + } + // The archive writes -1 for a phase it did not measure. Emitting that + // into a column named _ns puts a non-duration in a duration field, and + // every aggregate over it is then wrong without saying so: MIN returns + // -1 as the fastest pass, AVG is pulled below zero. On every capture + // measured, all six of these fields are -1 on every pipeline, so the + // column is not merely at risk of a sentinel, it is entirely sentinel. + // + // The three states the archive distinguishes are all still + // distinguishable: absent leaves both columns NULL, a recorded zero + // sets the duration to 0, and a recorded -1 leaves the duration NULL + // and sets _unmeasured. What changes is that the duration column now + // only ever holds durations. + if *field.value < 0 { + args[field.name+"_unmeasured"] = true + continue + } + args[field.name] = *field.value + } +} + +func streamDataTableEvent(id, trackID uint64, named namedStreamDataTable) perfetto.Event { + args := map[string]any{ + "table_name": named.name, + "source_key": named.sourceKey, + "byte_count": named.table.Bytes, + "raw_bytes_hex": named.table.RawBytesHex, + "table_sha256": named.table.SHA256, + "source": "streamData keyed archive fixed-record table", + "semantics": "exact source bytes; record order is byte order; unknown words and cross-table relationships remain uninterpreted", + "clock_domain": "none", + "timing_quality": "unavailable", + } + if named.table.RecordSize != nil { + args["record_size"] = *named.table.RecordSize + } + if named.table.RecordCount != nil { + args["record_count"] = *named.table.RecordCount + } + if named.table.RemainderBytes != nil { + args["remainder_bytes"] = *named.table.RemainderBytes + } + return perfetto.Event{ + ID: id, TrackUUID: trackID, + Name: "streamData table: " + named.name, + Category: "stream_data_table", Kind: perfetto.EventInstant, + Args: args, + } +} + +func timelineSemanticTargetEvent(timeline *Timeline, kind string, index int) (TimelineEvent, bool) { + switch kind { + case "dispatch": + if event, ok := timelineEventAt(timeline, "kernel", index); ok { + return event, true + } + return timelineEventAt(timeline, "dispatch", index) + case "encoder": + return timelineEventAt(timeline, "encoder", index) + case "command_buffer": + return timelineEventAt(timeline, "command_buffer", index) + default: + return TimelineEvent{}, false + } +} + +func mlxSemanticNode(sidecar *mlxsemantic.Sidecar, id string) mlxsemantic.Node { + for _, node := range sidecar.Nodes { + if node.ID == id { + return node + } + } + return mlxsemantic.Node{} +} + +// exportChromeTracingForClock exports one measured timestamp domain. Perfetto +// has one global time axis, so callers must not combine wall-clock command +// buffers and cumulative GPU-busy execution in the same export. +func exportChromeTracingForClock(timeline *Timeline, outputPath string, clock timelineClock) error { + f, closeOutput, err := createCommandOutput(outputPath) + if err != nil { + return err + } + if closeOutput != nil { + defer closeOutput() + } + + processName := "Compute GPU execution (cumulative busy; no wall-clock anchor)" + if clock == timelineClockWall { + processName = "Command buffers (wall clock; APSTimelineData)" + } + + // Add process and thread name metadata events. + metadataEvents := []TimelineEvent{ + { + Name: "process_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 0, + Args: map[string]interface{}{ + "name": processName, + "gputrace_clock_domain": string(clock), + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 0, + Args: map[string]interface{}{ + "name": "Command Buffers (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 1, + Args: map[string]interface{}{ + "name": "Compute encoders and dispatches (cumulative busy)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 2, + Args: map[string]interface{}{ + "name": "Compute encoders and dispatches lane 1 (cumulative busy)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 3, + Args: map[string]interface{}{ + "name": "Unattributed compute dispatches (cumulative busy)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 4, + Args: map[string]interface{}{ + "name": "Unattributed compute dispatches lane 1 (cumulative busy)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 5, + Args: map[string]interface{}{ + "name": "Unattributed compute dispatches lane 2 (cumulative busy)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 6, + Args: map[string]interface{}{ + "name": "Unattributed compute dispatches lane 3 (cumulative busy)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 7, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 0 (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 8, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 1 (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 9, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 2 (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 10, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 3 (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 11, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 4 (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 12, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 5 (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 13, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 6 (wall clock)", + }, + }, + { + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 14, + Args: map[string]interface{}{ + "name": "Raw profiler stream lane 7 (wall clock)", + }, + }, + } + + metadataEvents = append(metadataEvents, + TimelineEvent{ + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 15, + Args: map[string]interface{}{ + "name": "Xcode Parity / Provenance", + }, + }, + TimelineEvent{ + Name: "Xcode Metrics Coverage", + Category: "xcode_metrics", + Phase: "i", + ProcessID: 1, + ThreadID: 15, + Args: timelineCoverageArgs(timeline, clock), + }, + ) + for _, metric := range timeline.UnattributedCounters { + label := metric.Label + if label == "" { + label = "(pipeline unknown)" + } + args := make(map[string]interface{}, len(metric.Values)+4) + for key, value := range metric.Values { + args[key] = value + } + args["attribution"] = metric.Attribution + args["metric_scope"] = "pipeline" + args["pipeline_label"] = label + args["source"] = metric.Source + metadataEvents = append(metadataEvents, TimelineEvent{ + Name: "Unattributed counter metrics: " + label, + Category: "counter_attribution", + Phase: "i", + ProcessID: 1, + ThreadID: 15, + Args: args, + }) + } + for _, gap := range timeline.UnavailableEvidence { + metadataEvents = append(metadataEvents, TimelineEvent{ + Name: "Unavailable evidence: " + gap.Family, + Category: "evidence_gap", + Phase: "i", + ProcessID: 1, + ThreadID: 15, + Args: map[string]interface{}{ + "family": gap.Family, + "reason": gap.Reason, + }, + }) + } + + if timeline.Timing != nil { + metadataEvents = append(metadataEvents, + TimelineEvent{ + Name: "Xcode Timing Summary", + Category: "xcode_timing", + Phase: "i", + ProcessID: 1, + ThreadID: 15, + Args: timelineTimingArgs(timeline.Timing), + }, + ) + } + + // Add counter track metadata and events, grouped by Xcode category group when available. + threadID := 16 // Start after GPRWCNTR lanes (7-14) and provenance lane (15). + counterEvents := make([]TimelineEvent, 0) + groupPIDs := make(map[string]int) + nextGroupPID := 10 + + for _, track := range timeline.CounterTracks { + if !counterTrackHasSignal(track) { + continue + } + pid := 1 + if len(track.XcodeGroups) > 0 && track.XcodeGroups[0] != "" { + groupName := track.XcodeGroups[0] + if existingPID, exists := groupPIDs[groupName]; exists { + pid = existingPID + } else { + pid = nextGroupPID + groupPIDs[groupName] = pid + nextGroupPID++ + metadataEvents = append(metadataEvents, TimelineEvent{ + Name: "process_name", + Category: "__metadata", + Phase: "M", + ProcessID: pid, + ThreadID: 0, + Args: map[string]interface{}{ + "name": fmt.Sprintf("Counters: %s", groupName), + }, + }) + } + } + + // Add thread name for this counter track + metadataEvents = append(metadataEvents, TimelineEvent{ + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: pid, + ThreadID: threadID, + Args: counterTrackMetadataArgs(track), + }) + + // Add counter samples as events + for _, sample := range track.Samples { + // Use counter events (C phase) for Chrome tracing + counterEvent := TimelineEvent{ + Name: track.Name, + Category: "counter", + Phase: "C", // Counter event + Timestamp: sample.Timestamp / 1000, // Convert to microseconds + ProcessID: pid, + ThreadID: threadID, + Args: map[string]interface{}{ + track.Name: sample.Value, + }, + } + counterEvents = append(counterEvents, counterEvent) + } + + threadID++ + } + + // Kernel events retain the tids assigned during timeline construction. + // Strictly contained dispatches therefore share their encoder track; the + // pipeline state remains available in each dispatch's arguments. + allEvents := append(metadataEvents, timeline.Events...) + allEvents = append(allEvents, counterEvents...) + allEvents = timelineMetadataForActiveTracks(allEvents) + + // Chrome tracing format + // Standard format: { "traceEvents": [ ... ] } + // We omit displayTimeUnit and other legacy fields to maximize Perfetto compatibility. + tracing := map[string]interface{}{ + "traceEvents": allEvents, + } + + // Provenance goes under otherData, the Trace Event Format's sanctioned slot + // for producer-specific metadata. It used to sit in gputrace_timing and + // gputrace_xcode_metrics keys at the root, which strict readers are free to + // reject: the format defines the top level, and those names are not in it. + // + // The content is worth keeping rather than dropping. display_duration_source + // says which clock a duration came from, and absent_kernel_arg_fields names + // the metrics we deliberately do not emit, so a reader can tell an absent + // field from one we forgot. + other := map[string]interface{}{} + if timeline.Timing != nil { + other["gputrace_timing"] = timelineTimingArgs(timeline.Timing) + } + other["gputrace_xcode_metrics"] = timelineCoverageArgs(timeline, clock) + rawProfilerSamples := timeline != nil && timeline.RawProfilerSamples + other["gputrace_clock_domain"] = timelineClockProvenanceWithRawSamples(clock, rawProfilerSamples) + tracing["otherData"] = other + + encoder := json.NewEncoder(f) + encoder.SetIndent("", " ") + return encoder.Encode(tracing) +} + +// timelineMetadataForActiveTracks omits names for tracks absent from this +// clock-domain export. Perfetto otherwise renders empty tracks, which makes a +// busy-only trace look as though it also contains wall-clock data. +func timelineMetadataForActiveTracks(events []TimelineEvent) []TimelineEvent { + active := make(map[[2]int]bool) + for _, event := range events { + if event.Phase != "M" { + active[[2]int{event.ProcessID, event.ThreadID}] = true + } + } + + result := events[:0] + for _, event := range events { + if event.Phase == "M" && event.Name == "thread_name" && !active[[2]int{event.ProcessID, event.ThreadID}] { + continue + } + result = append(result, event) + } + return result +} + +func counterTrackMetadataArgs(track CounterTrack) map[string]interface{} { + args := map[string]interface{}{ + "name": fmt.Sprintf("%s (%s)", track.Name, track.Unit), + } + if track.Description != "" { + args["description"] = track.Description + args["xcode_tooltip"] = track.Description + } + if len(track.XcodeGroups) > 0 { + args["xcode_groups"] = track.XcodeGroups + } + if track.XcodeCatalogPath != "" { + args["xcode_catalog_path"] = track.XcodeCatalogPath + } + return args +} + +func timelineCoverageArgs(timeline *Timeline, clock timelineClock) map[string]interface{} { + args := timelineXcodeMetricsArgs(timeline) + rawProfilerSamples := timeline != nil && timeline.RawProfilerSamples + for key, value := range timelineClockProvenanceWithRawSamples(clock, rawProfilerSamples) { + args[key] = value + } + return args +} + +func timelineClockProvenance(clock timelineClock) map[string]interface{} { + return timelineClockProvenanceWithRawSamples(clock, true) +} + +func timelineClockProvenanceWithRawSamples(clock timelineClock, rawProfilerSamples bool) map[string]interface{} { + args := map[string]interface{}{ + "clock_domain": string(clock), + "clock_mapping": "none: trace records no measured correspondence between cumulative GPU-busy offsets and command-buffer wall time", + } + switch clock { + case timelineClockBusy: + args["included_categories"] = []string{"encoder", "kernel", "counter"} + args["excluded_categories"] = []string{"command_buffer", "restore", "profiler_stream", "gprwcntr"} + args["excluded_counter_series"] = "memory-side GTMioCounterData has scope=2/index=0 and a separate tick domain; it is not encoder-attributed or clock-aligned" + case timelineClockWall: + args["included_categories"] = []string{"command_buffer", "restore"} + args["excluded_categories"] = []string{"encoder", "kernel", "counter"} + if rawProfilerSamples { + args["included_categories"] = append(args["included_categories"].([]string), "profiler_stream", "gprwcntr") + } else { + args["excluded_categories"] = append(args["excluded_categories"].([]string), "profiler_stream", "gprwcntr") + args["raw_profiler_samples"] = "excluded by default: GPRWCNTR fixed fields and aggregate profiler streams are not decoded hardware counter series or encoder intervals; use --include-raw-samples to inspect them" + } + case timelineClockLive: + args["included_categories"] = []string{"live_command_buffer"} + args["excluded_categories"] = []string{"encoder", "kernel", "counter", "command_buffer", "profiler_stream", "gprwcntr"} + args["clock_mapping"] = "MTLDevice sampled CPU/GPU timestamp pairs retained by the original-execution timing sidecar" + } + return args +} + +// zeroIsNotAReading names the kernel-event fields that a fallback may stamp +// with zero when nothing was read. For these, only a nonzero value counts as +// evidence that gputrace can produce the field. Every other field in the +// parity list comes from pipeline statistics, which are attached only when +// they were joined, so a zero there is a genuine measurement. +var zeroIsNotAReading = map[string]bool{ + "alu_utilization_pct": true, +} + +func timelineXcodeMetricsArgs(timeline *Timeline) map[string]interface{} { + args := map[string]interface{}{ + "kernel_events": 0, + } + if timeline == nil { + return args + } + + presentFields := make(map[string]bool) + for _, ev := range timeline.Events { + if (ev.Category != "kernel" && ev.Category != "dispatch") || ev.Args == nil { + continue + } + args["kernel_events"] = args["kernel_events"].(int) + 1 + for _, field := range []string{ + "simd_groups", + "simd_group_share_pct", + "allocated_registers", + "uniform_registers", + "high_register", + "spilled_bytes", + "threadgroup_memory", + "instruction_count", + "alu_utilization_pct", + "pipeline_id", + "pipeline_state", + } { + // The counter-derived fields are present only if they carry a + // nonzero value: a zero written by a fallback is not evidence + // that gputrace read the counter, and counting it as presence + // is how alu_utilization_pct came to be reported as a closed + // gap while Xcode reported 1.59% for the same encoder. + // + // The compiler statistics are different. They are set only when + // the pipeline stats were joined, and zero is a real reading + // there: a kernel that spills nothing has spilled_bytes 0. + v, ok := ev.Args[field] + if !ok { + continue + } + if zeroIsNotAReading[field] && isZeroMetricValue(v) { + continue + } + presentFields[field] = true + } + } + + var present, absent []string + for _, field := range []string{ + "simd_groups", + "simd_group_share_pct", + "allocated_registers", + "uniform_registers", + "high_register", + "spilled_bytes", + "threadgroup_memory", + "instruction_count", + "alu_utilization_pct", + "pipeline_id", + "pipeline_state", + } { + if presentFields[field] { + present = append(present, field) + } else { + absent = append(absent, field) + } + } + + var tracks, emptyTracks []string + for _, track := range timeline.CounterTracks { + name := fmt.Sprintf("%s (%s)", track.Name, track.Unit) + if len(track.Samples) == 0 || track.MaxValue == 0 { + // An all-zero track carries no information about whether the + // counter was read. Report it alongside the empty ones. + emptyTracks = append(emptyTracks, name) + } else { + tracks = append(tracks, name) + } + } + sort.Strings(tracks) + sort.Strings(emptyTracks) + + args["kernel_arg_fields"] = present + args["absent_kernel_arg_fields"] = absent + args["binding_candidates"] = xcodeMetricBindingCandidates(absent) + args["counter_tracks"] = tracks + args["empty_counter_tracks"] = emptyTracks + if len(timeline.UnattributedCounters) > 0 { + labels := make([]string, 0, len(timeline.UnattributedCounters)) + for _, metric := range timeline.UnattributedCounters { + if metric.Label != "" { + labels = append(labels, metric.Label) + } + } + sort.Strings(labels) + args["counter_attribution"] = string(counter.CounterAttributionUnknown) + args["unattributed_counter_rows"] = len(timeline.UnattributedCounters) + args["unattributed_counter_labels"] = labels + args["counter_attribution_reason"] = "no capture-backed encoder identity" + } + if len(timeline.UnavailableEvidence) > 0 { + families := make([]string, 0, len(timeline.UnavailableEvidence)) + for _, gap := range timeline.UnavailableEvidence { + families = append(families, gap.Family+": "+gap.Reason) + } + sort.Strings(families) + args["unavailable_evidence"] = families + } + if report := timeline.MLXSemanticReport; report != nil { + args["mlx_semantic_used_nodes"] = report.UsedNodes + args["mlx_semantic_unused_nodes"] = report.UnusedNodes + args["mlx_semantic_matched_targets"] = report.MatchedTargets + args["mlx_semantic_unmatched_targets"] = report.UnmatchedTargets + } + if timeline.Timing != nil { + args["display_duration_source"] = timeline.Timing.DisplayDurationSource + args["timing_source"] = timeline.Timing.TimingSource + if timeline.Timing.EncoderTimingSource != "" { + args["encoder_timing_source"] = timeline.Timing.EncoderTimingSource + args["encoder_timing_approximate"] = timeline.Timing.EncoderTimingApproximate + } + args["has_effective_gpu_time"] = timeline.Timing.EffectiveGPUTimeNs != nil + } else { + args["has_effective_gpu_time"] = false + } + return args +} + +// isZeroMetricValue reports whether v is a numeric zero. Parity accounting uses +// it to tell "gputrace read this counter and it was zero" apart from "gputrace +// wrote a zero because it had nothing to write". +func isZeroMetricValue(v interface{}) bool { + switch n := v.(type) { + case float64: + return n == 0 + case float32: + return n == 0 + case int: + return n == 0 + case int64: + return n == 0 + case uint64: + return n == 0 + default: + return false + } +} + +func xcodeMetricBindingCandidates(fields []string) map[string]string { + candidates := map[string]string{ + "high_register": "GTMioShaderBinaryData.LiveRegisterForInstructionAtIndex", + "alu_utilization_pct": "XRGPUAPSDataProcessor derived counters", + } + result := make(map[string]string) + for _, field := range fields { + if candidate := candidates[field]; candidate != "" { + result[field] = candidate + } + } + return result +} + +func timelineTimingArgs(timing *TimelineTiming) map[string]interface{} { + args := map[string]interface{}{ + "encoder_span_ns": timing.EncoderSpanNs, + "dispatch_span_ns": timing.DispatchSpanNs, + "command_buffer_active_time_ns": timing.CommandBufferActiveNs, + "command_buffer_wall_time_ns": timing.CommandBufferWallNs, + "restore_active_time_ns": timing.RestoreActiveNs, + "restore_wall_time_ns": timing.RestoreWallNs, + "display_duration_ns": timing.DisplayDurationNs, + "display_duration_source": timing.DisplayDurationSource, + "timing_source": timing.TimingSource, + } + if timing.EncoderTimingSource != "" { + args["encoder_timing_source"] = timing.EncoderTimingSource + args["encoder_timing_approximate"] = timing.EncoderTimingApproximate + } + if timing.EffectiveGPUTimeNs != nil { + args["effective_gpu_time_ns"] = *timing.EffectiveGPUTimeNs + } + return args +} + +func perfettoTimingSummaryArgs(timing *TimelineTiming) map[string]interface{} { + args := make(map[string]interface{}) + if timing == nil { + return args + } + for key, value := range map[string]uint64{ + "encoder_span_ns": timing.EncoderSpanNs, + "dispatch_span_ns": timing.DispatchSpanNs, + "command_buffer_active_time_ns": timing.CommandBufferActiveNs, + "command_buffer_wall_time_ns": timing.CommandBufferWallNs, + "restore_active_time_ns": timing.RestoreActiveNs, + "restore_wall_time_ns": timing.RestoreWallNs, + "display_duration_ns": timing.DisplayDurationNs, + } { + if value != 0 { + args[key] = value + } + } + if timing.EffectiveGPUTimeNs != nil { + args["effective_gpu_time_ns"] = *timing.EffectiveGPUTimeNs + } + if timing.DisplayDurationSource != "" { + args["display_duration_source"] = timing.DisplayDurationSource + } + if timing.TimingSource != "" { + args["timing_source"] = timing.TimingSource + } + if timing.EncoderTimingSource != "" { + args["encoder_timing_source"] = timing.EncoderTimingSource + args["encoder_timing_approximate"] = timing.EncoderTimingApproximate + } + return args +} + +// exportTimelineJSON exports raw timeline data as JSON. +func exportTimelineJSON(timeline *Timeline, outputPath string) error { + f, closeOutput, err := createCommandOutput(outputPath) + if err != nil { + return err + } + if closeOutput != nil { + defer closeOutput() + } + + encoder := json.NewEncoder(f) + encoder.SetIndent("", " ") + return encoder.Encode(timeline) +} + +// buildPerfettoTraceForHTML constructs the full trace object (traceEvents + otherData) for embedding in HTML viewer. +func buildPerfettoTraceForHTML(timeline *Timeline) map[string]interface{} { + clock := timelineClockBusy + if timeline != nil && timeline.ClockDomain != "" { + clock = timelineClock(timeline.ClockDomain) + } + + processName := "Compute GPU execution (cumulative busy; no wall-clock anchor)" + if clock == timelineClockWall { + processName = "Command buffers (wall clock; APSTimelineData)" + } + + metadataEvents := []TimelineEvent{ + { + Name: "process_name", + Category: "__metadata", + Phase: "M", + ProcessID: 1, + ThreadID: 0, + Args: map[string]interface{}{ + "name": processName, + "gputrace_clock_domain": string(clock), }, }, { @@ -1866,7 +5557,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 0, Args: map[string]interface{}{ - "name": "Command Buffers", + "name": "Command Buffers (wall clock)", }, }, { @@ -1876,7 +5567,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 1, Args: map[string]interface{}{ - "name": "Encoders Lane 0", + "name": "Compute encoders and dispatches (cumulative busy)", }, }, { @@ -1886,7 +5577,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 2, Args: map[string]interface{}{ - "name": "Encoders Lane 1", + "name": "Compute encoders and dispatches lane 1 (cumulative busy)", }, }, { @@ -1896,7 +5587,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 3, Args: map[string]interface{}{ - "name": "Kernels Lane 0", + "name": "Unattributed compute dispatches (cumulative busy)", }, }, { @@ -1906,7 +5597,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 4, Args: map[string]interface{}{ - "name": "Kernels Lane 1", + "name": "Unattributed compute dispatches lane 1 (cumulative busy)", }, }, { @@ -1916,7 +5607,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 5, Args: map[string]interface{}{ - "name": "Kernels Lane 2", + "name": "Unattributed compute dispatches lane 2 (cumulative busy)", }, }, { @@ -1926,7 +5617,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 6, Args: map[string]interface{}{ - "name": "Kernels Lane 3", + "name": "Unattributed compute dispatches lane 3 (cumulative busy)", }, }, { @@ -1936,7 +5627,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 7, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 0", + "name": "Raw profiler stream lane 0 (wall clock)", }, }, { @@ -1946,7 +5637,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 8, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 1", + "name": "Raw profiler stream lane 1 (wall clock)", }, }, { @@ -1956,7 +5647,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 9, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 2", + "name": "Raw profiler stream lane 2 (wall clock)", }, }, { @@ -1966,7 +5657,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 10, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 3", + "name": "Raw profiler stream lane 3 (wall clock)", }, }, { @@ -1976,7 +5667,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 11, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 4", + "name": "Raw profiler stream lane 4 (wall clock)", }, }, { @@ -1986,7 +5677,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 12, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 5", + "name": "Raw profiler stream lane 5 (wall clock)", }, }, { @@ -1996,7 +5687,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 13, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 6", + "name": "Raw profiler stream lane 6 (wall clock)", }, }, { @@ -2006,7 +5697,7 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { ProcessID: 1, ThreadID: 14, Args: map[string]interface{}{ - "name": "GPRWCNTR Lane 7", + "name": "Raw profiler stream lane 7 (wall clock)", }, }, } @@ -2028,234 +5719,155 @@ func exportChromeTracing(timeline *Timeline, outputPath string) error { Phase: "i", ProcessID: 1, ThreadID: 15, - Args: timelineXcodeMetricsArgs(timeline), + Args: timelineCoverageArgs(timeline, clock), }, ) - if timeline.Timing != nil { - metadataEvents = append(metadataEvents, - TimelineEvent{ - Name: "Xcode Timing Summary", - Category: "xcode_timing", - Phase: "i", - ProcessID: 1, - ThreadID: 15, - Args: timelineTimingArgs(timeline.Timing), - }, - ) - if timeline.Timing.DisplayDurationNs > 0 { - metadataEvents = append(metadataEvents, TimelineEvent{ - Name: "Xcode Display Duration", - Category: "xcode_timing", - Phase: "X", - Timestamp: 0, - Duration: timeline.Timing.DisplayDurationNs / 1000, - ProcessID: 1, - ThreadID: 15, - Args: timelineTimingArgs(timeline.Timing), - }) - } - } - - // Add counter track metadata and events - threadID := 16 // Start after GPRWCNTR lanes (7-14) and provenance lane (15). - counterEvents := make([]TimelineEvent, 0) - for _, track := range timeline.CounterTracks { - // Add thread name for this counter track - metadataEvents = append(metadataEvents, TimelineEvent{ - Name: "thread_name", - Category: "__metadata", - Phase: "M", - ProcessID: 1, - ThreadID: threadID, - Args: map[string]interface{}{ - "name": fmt.Sprintf("%s (%s)", track.Name, track.Unit), - }, - }) - - // Add counter samples as events - for _, sample := range track.Samples { - // Use counter events (C phase) for Chrome tracing - counterEvent := TimelineEvent{ - Name: track.Name, - Category: "counter", - Phase: "C", // Counter event - Timestamp: sample.Timestamp / 1000, // Convert to microseconds - ProcessID: 1, - ThreadID: threadID, - Args: map[string]interface{}{ - track.Name: sample.Value, - }, - } - counterEvents = append(counterEvents, counterEvent) - } - - threadID++ + if timeline != nil && timeline.Timing != nil { + metadataEvents = append(metadataEvents, + TimelineEvent{ + Name: "Xcode Timing Summary", + Category: "xcode_timing", + Phase: "i", + ProcessID: 1, + ThreadID: 15, + Args: timelineTimingArgs(timeline.Timing), + }, + ) } - // Combine metadata events with timeline events - allEvents := append(metadataEvents, timeline.Events...) - allEvents = append(allEvents, counterEvents...) - - // Chrome tracing format - // Standard format: { "traceEvents": [ ... ] } - // We omit displayTimeUnit and other legacy fields to maximize Perfetto compatibility. - tracing := map[string]interface{}{ - "traceEvents": allEvents, - } - if timeline.Timing != nil { - tracing["gputrace_timing"] = timelineTimingArgs(timeline.Timing) - } - tracing["gputrace_xcode_metrics"] = timelineXcodeMetricsArgs(timeline) + threadID := 16 + counterEvents := make([]TimelineEvent, 0) + groupPIDs := make(map[string]int) + nextGroupPID := 10 - encoder := json.NewEncoder(f) - encoder.SetIndent("", " ") - return encoder.Encode(tracing) -} + if timeline != nil { + for _, track := range timeline.CounterTracks { + if !counterTrackHasSignal(track) { + continue + } + pid := 1 + if len(track.XcodeGroups) > 0 && track.XcodeGroups[0] != "" { + groupName := track.XcodeGroups[0] + if existingPID, exists := groupPIDs[groupName]; exists { + pid = existingPID + } else { + pid = nextGroupPID + groupPIDs[groupName] = pid + nextGroupPID++ + metadataEvents = append(metadataEvents, TimelineEvent{ + Name: "process_name", + Category: "__metadata", + Phase: "M", + ProcessID: pid, + ThreadID: 0, + Args: map[string]interface{}{ + "name": fmt.Sprintf("Counters: %s", groupName), + }, + }) + } + } -func timelineXcodeMetricsArgs(timeline *Timeline) map[string]interface{} { - args := map[string]interface{}{ - "kernel_events": 0, - } - if timeline == nil { - return args - } + metadataEvents = append(metadataEvents, TimelineEvent{ + Name: "thread_name", + Category: "__metadata", + Phase: "M", + ProcessID: pid, + ThreadID: threadID, + Args: counterTrackMetadataArgs(track), + }) - presentFields := make(map[string]bool) - for _, ev := range timeline.Events { - if ev.Category != "kernel" || ev.Args == nil { - continue - } - args["kernel_events"] = args["kernel_events"].(int) + 1 - for _, field := range []string{ - "xcode_cost_pct", - "profiling_cost_pct", - "simd_groups", - "allocated_registers", - "uniform_registers", - "high_register", - "spilled_bytes", - "threadgroup_memory", - "instruction_count", - "occupancy_pct", - "alu_utilization_pct", - "pipeline_id", - "pipeline_state", - } { - if _, ok := ev.Args[field]; ok { - presentFields[field] = true + for _, sample := range track.Samples { + counterEvent := TimelineEvent{ + Name: track.Name, + Category: "counter", + Phase: "C", + Timestamp: sample.Timestamp / 1000, + ProcessID: pid, + ThreadID: threadID, + Args: map[string]interface{}{ + track.Name: sample.Value, + }, + } + counterEvents = append(counterEvents, counterEvent) } + threadID++ } } - var present, absent []string - for _, field := range []string{ - "xcode_cost_pct", - "profiling_cost_pct", - "simd_groups", - "allocated_registers", - "uniform_registers", - "high_register", - "spilled_bytes", - "threadgroup_memory", - "instruction_count", - "occupancy_pct", - "alu_utilization_pct", - "pipeline_id", - "pipeline_state", - } { - if presentFields[field] { - present = append(present, field) - } else { - absent = append(absent, field) - } + events := []TimelineEvent{} + if timeline != nil { + events = timeline.Events } + allEvents := append(metadataEvents, events...) + allEvents = append(allEvents, counterEvents...) + allEvents = timelineMetadataForActiveTracks(allEvents) - var tracks, emptyTracks []string - for _, track := range timeline.CounterTracks { - name := fmt.Sprintf("%s (%s)", track.Name, track.Unit) - if len(track.Samples) == 0 { - emptyTracks = append(emptyTracks, name) - } else { - tracks = append(tracks, name) - } + tracing := map[string]interface{}{ + "traceEvents": allEvents, } - sort.Strings(tracks) - sort.Strings(emptyTracks) - args["kernel_arg_fields"] = present - args["absent_kernel_arg_fields"] = absent - args["binding_candidates"] = xcodeMetricBindingCandidates(absent) - args["counter_tracks"] = tracks - args["empty_counter_tracks"] = emptyTracks - if timeline.Timing != nil { - args["display_duration_source"] = timeline.Timing.DisplayDurationSource - args["timing_source"] = timeline.Timing.TimingSource - if timeline.Timing.EncoderTimingSource != "" { - args["encoder_timing_source"] = timeline.Timing.EncoderTimingSource - args["encoder_timing_approximate"] = timeline.Timing.EncoderTimingApproximate - } - args["has_effective_gpu_time"] = timeline.Timing.EffectiveGPUTimeNs != nil - } else { - args["has_effective_gpu_time"] = false + other := map[string]interface{}{} + if timeline != nil && timeline.Timing != nil { + other["gputrace_timing"] = timelineTimingArgs(timeline.Timing) } - return args + other["gputrace_xcode_metrics"] = timelineCoverageArgs(timeline, clock) + rawProfilerSamples := timeline != nil && timeline.RawProfilerSamples + other["gputrace_clock_domain"] = timelineClockProvenanceWithRawSamples(clock, rawProfilerSamples) + tracing["otherData"] = other + + return tracing } -func xcodeMetricBindingCandidates(fields []string) map[string]string { - candidates := map[string]string{ - "high_register": "GTMioShaderBinaryData.LiveRegisterForInstructionAtIndex", - "occupancy_pct": "XRGPUAPSDataProcessor derived counters", - "alu_utilization_pct": "XRGPUAPSDataProcessor derived counters", +// exportHTML exports an interactive standalone HTML timeline viewer. +func exportHTML(timeline *Timeline, outputPath string) error { + f, closeOutput, err := createCommandOutput(outputPath) + if err != nil { + return err } - result := make(map[string]string) - for _, field := range fields { - if candidate := candidates[field]; candidate != "" { - result[field] = candidate - } + if closeOutput != nil { + defer closeOutput() } - return result + html, err := timelineHTML(timeline) + if err != nil { + return err + } + _, err = io.WriteString(f, html) + return err } -func timelineTimingArgs(timing *TimelineTiming) map[string]interface{} { - args := map[string]interface{}{ - "encoder_span_ns": timing.EncoderSpanNs, - "dispatch_span_ns": timing.DispatchSpanNs, - "command_buffer_active_time_ns": timing.CommandBufferActiveNs, - "command_buffer_wall_time_ns": timing.CommandBufferWallNs, - "restore_active_time_ns": timing.RestoreActiveNs, - "restore_wall_time_ns": timing.RestoreWallNs, - "display_duration_ns": timing.DisplayDurationNs, - "display_duration_source": timing.DisplayDurationSource, - "timing_source": timing.TimingSource, - } - if timing.EncoderTimingSource != "" { - args["encoder_timing_source"] = timing.EncoderTimingSource - args["encoder_timing_approximate"] = timing.EncoderTimingApproximate +func timelineHTML(timeline *Timeline) (string, error) { + timelineJSON, err := json.Marshal(timeline) + if err != nil { + return "", fmt.Errorf("marshal timeline: %w", err) } - if timing.EffectiveGPUTimeNs != nil { - args["effective_gpu_time_ns"] = *timing.EffectiveGPUTimeNs + + perfettoTrace := buildPerfettoTraceForHTML(timeline) + perfettoJSON, err := json.Marshal(perfettoTrace) + if err != nil { + return "", fmt.Errorf("marshal perfetto trace: %w", err) } - return args + return generateInteractiveHTML(string(timelineJSON), string(perfettoJSON)), nil } -// exportTimelineJSON exports raw timeline data as JSON. -func exportTimelineJSON(timeline *Timeline, outputPath string) error { - f, closeOutput, err := createCommandOutput(outputPath) +func exportHTMLBoth(busy, wall *Timeline, outputPath string) error { + busyHTML, err := timelineHTML(busy) if err != nil { return err } - if closeOutput != nil { - defer closeOutput() + wallHTML, err := timelineHTML(wall) + if err != nil { + return err + } + busyJSON, err := json.Marshal(busyHTML) + if err != nil { + return fmt.Errorf("marshal busy viewer: %w", err) + } + wallJSON, err := json.Marshal(wallHTML) + if err != nil { + return fmt.Errorf("marshal wall viewer: %w", err) } - encoder := json.NewEncoder(f) - encoder.SetIndent("", " ") - return encoder.Encode(timeline) -} - -// exportHTML exports an interactive standalone HTML timeline viewer. -func exportHTML(timeline *Timeline, outputPath string) error { f, closeOutput, err := createCommandOutput(outputPath) if err != nil { return err @@ -2263,45 +5875,66 @@ func exportHTML(timeline *Timeline, outputPath string) error { if closeOutput != nil { defer closeOutput() } - - // Serialize timeline data to JSON for embedding - timelineJSON, err := json.Marshal(timeline) - if err != nil { - return fmt.Errorf("marshal timeline: %w", err) + wallHeading := "Wall-clock scheduling — command buffers and encoder profiles" + if wall != nil && wall.RawProfilerSamples { + wallHeading += " plus raw profiler records" } - // Generate the HTML content - html := generateInteractiveHTML(string(timelineJSON)) + html := fmt.Sprintf(` + + + + + GPU Timeline: busy and wall time + + + +
+

GPU Timeline: total information view

+

Busy and wall clocks are independently measured. They are displayed separately because this trace has no measured mapping between them.

+
+
+

GPU busy time — encoders, dispatches, and archive-backed counters

+

%s

+
+ + +`, wallHeading, busyJSON, wallJSON) _, err = io.WriteString(f, html) return err } // runTimelineFromProfiler generates timeline from profiler-only traces (.gpuprofiler_raw without unsorted-capture). -func runTimelineFromProfiler(tracePath string, opts *timelineOptions) error { +func runTimelineFromProfiler(cmd *cobra.Command, tracePath string, opts *timelineOptions) error { if err := validateTimelineFormat(opts.format); err != nil { return err } + if err := validateTimelineClock(opts.clock); err != nil { + return err + } + if err := validateTimelineSQLOutput(opts); err != nil { + return err + } // Find .gpuprofiler_raw directory - profilerDir := "" - if filepath.Ext(tracePath) == ".gpuprofiler_raw" { - profilerDir = tracePath - } else { - entries, err := os.ReadDir(tracePath) - if err != nil { - return fmt.Errorf("read directory: %w", err) - } - for _, e := range entries { - if e.IsDir() && filepath.Ext(e.Name()) == ".gpuprofiler_raw" { - profilerDir = filepath.Join(tracePath, e.Name()) - break - } - } - } + profilerDir := profilerraw.FindDir(tracePath) if profilerDir == "" { - fmt.Fprintf(os.Stderr, "Hint: To generate performance data, run:\n") - fmt.Fprintf(os.Stderr, " gputrace xcode-profile run %s\n\n", tracePath) + fmt.Fprint(os.Stderr, profileReplayHint(tracePath)) return fmt.Errorf("no .gpuprofiler_raw directory found in %s (and unsorted-capture is missing)", tracePath) } @@ -2311,18 +5944,59 @@ func runTimelineFromProfiler(tracePath string, opts *timelineOptions) error { return fmt.Errorf("parse streamData: %w", err) } counter.CorrelateDispatchSamples(stats) - annotateDispatchExecutionCosts(stats, profilerDir) + annotateDispatchProfilingSampleShares(stats, profilerDir) // Build timeline from profiler data timeline := buildTimelineFromProfilerData(tracePath, stats) - + if opts.format == "perfetto" || opts.format == "json" { + attachRawProfilerArtifacts(timeline, tracePath) + } + if err := enrichTimelineWithXcodeGPUTime(tracePath, timeline, opts.xcodeGPUTime); err != nil { + return err + } + if opts.sidecar != "" { + return fmt.Errorf("attach MLX sidecar: profiler-only input has no capture UUID; use the self-contained profiled .gputrace") + } + if opts.hostCorrelation != "" { + return fmt.Errorf("attach host correlation: profiler-only input has no trace tree for content identity") + } + if timeline.Timing == nil || timeline.Timing.EncoderTimingApproximate || timeline.Timing.TimingSource == "" || timeline.Timing.TimingSource == "unavailable" { + fmt.Fprintln(os.Stderr, "Warning: trace lacks precise hardware timing data; encoder/dispatch durations are estimated.") + } outputPath := timelineOutputPath(opts.format, opts.output) + if err := validateTimelineViewerOptions(opts, outputPath); err != nil { + return err + } + if opts.clock == timelineClockBoth { + if err := exportTimelineBothWithRawSamples(timeline, opts.format, outputPath, opts.rawProfilerSamples); err != nil { + return err + } + if opts.format != "text" || (outputPath != "" && !commandOutputPathIsStdout(outputPath)) { + printTimelineExportStatus(outputPath, opts.format, true) + } + return nil + } + fullTimeline := timeline // Keep pre-clock-filtered timeline for the dropped-dispatch check. + timeline = timelineForClockWithRawSamples(timeline, opts.clock, opts.rawProfilerSamples) + if err := resolveTimelineNavigation(timeline, opts); err != nil { + return err + } + if opts.format == "chrome" || opts.format == "perfetto" { + warnDroppedDispatchEvents(cmd.ErrOrStderr(), fullTimeline) + } // Export based on format switch opts.format { - case "chrome", "perfetto": - if err := exportChromeTracing(timeline, outputPath); err != nil { - return fmt.Errorf("failed to export Chrome/Perfetto tracing: %w", err) + case "chrome": + if err := exportChromeTracingForClock(timeline, outputPath, opts.clock); err != nil { + return fmt.Errorf("failed to export Chrome tracing: %w", err) + } + case "perfetto": + if err := exportPerfettoForClockWithBudget(timeline, outputPath, opts.clock, opts.maxOutputBytes); err != nil { + return fmt.Errorf("failed to export Perfetto tracing: %w", err) + } + if err := writeTimelinePerfettoSQL(opts.sqlOutput); err != nil { + return err } case "html": if err := exportHTML(timeline, outputPath); err != nil { @@ -2333,7 +6007,7 @@ func runTimelineFromProfiler(tracePath string, opts *timelineOptions) error { return fmt.Errorf("failed to export JSON: %w", err) } case "text": - if err := exportTextTimeline(timeline, outputPath); err != nil { + if err := exportTextTimeline(timeline, nil, outputPath); err != nil { return fmt.Errorf("failed to export text: %w", err) } if outputPath != "" && !commandOutputPathIsStdout(outputPath) { @@ -2345,19 +6019,20 @@ func runTimelineFromProfiler(tracePath string, opts *timelineOptions) error { } printTimelineExportStatus(outputPath, opts.format, true) - - return nil + return serveTimelinePerfetto(cmd, tracePath, outputPath, opts) } // buildTimelineFromProfilerData creates a Timeline from StreamDataStats. func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataStats) *Timeline { timeline := &Timeline{ + TracePath: tracePath, Events: make([]TimelineEvent, 0), Encoders: make([]EncoderInfo, 0), Kernels: make([]KernelInfo, 0), APICallseq: make([]APICall, 0), Timing: timelineTimingFromStats(stats), } + applyStreamIdentity(timeline, stats) if timeline.Timing == nil { timeline.Timing = &TimelineTiming{} } @@ -2371,17 +6046,25 @@ func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataSt timebaseNumer = stats.Timeline.TimebaseNumer timebaseDenom = stats.Timeline.TimebaseDenom absoluteTime = stats.Timeline.AbsoluteTime + timeline.ContinuousTime = stats.Timeline.ContinuousTime + if stats.Timeline.PState != nil { + value := *stats.Timeline.PState + timeline.PState = &value + } } timeline.TimebaseNumer = timebaseNumer timeline.TimebaseDenom = timebaseDenom timeline.AbsoluteTime = absoluteTime + addRestoreEvents(timeline, stats.Timeline) // Add command buffer events with real timing from APSTimelineData if stats.Timeline != nil && len(stats.Timeline.CommandBufferTimestamps) > 0 { - var displayStartNs uint64 + // Real offsets, not a back-to-back accumulator. See the note on the + // other command-buffer emitter: packing erased all idle time. for _, cb := range stats.Timeline.CommandBufferTimestamps { durationNs := cb.DurationNs(timebaseNumer, timebaseDenom) + durationUs := durationNs / 1000 var rawStartOffsetNs uint64 if cb.StartTicks > absoluteTime { rawStartOffsetNs = (cb.StartTicks - absoluteTime) * timebaseNumer / timebaseDenom @@ -2390,9 +6073,9 @@ func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataSt event := TimelineEvent{ Name: fmt.Sprintf("CB#%d", cb.Index), Category: "command_buffer", - Phase: "X", - Timestamp: displayStartNs / 1000, // Convert to µs for Chrome format - Duration: durationNs / 1000, + Phase: timelineDurationPhase(durationUs), + Timestamp: rawStartOffsetNs / 1000, // Convert to µs for Chrome format + Duration: durationUs, ProcessID: 1, ThreadID: 0, Args: map[string]interface{}{ @@ -2407,10 +6090,9 @@ func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataSt }, } timeline.Events = append(timeline.Events, event) - if endNs := displayStartNs + durationNs; endNs > timeline.EndTime { + if endNs := rawStartOffsetNs + durationNs; endNs > timeline.EndTime { timeline.EndTime = endNs } - displayStartNs += durationNs } } @@ -2461,7 +6143,8 @@ func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataSt currentTimeNs = endTimeNs } - // Add GPRWCNTR encoder profile events + // Add raw profiler stream spans. These aggregates describe profiler input, + // not the GPU encoder hierarchy. if stats.Timeline != nil && len(stats.Timeline.EncoderProfiles) > 0 { for _, ep := range stats.Timeline.EncoderProfiles { if ep.SampleCount == 0 || ep.StartTicks == 0 { @@ -2474,8 +6157,8 @@ func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataSt } event := TimelineEvent{ - Name: fmt.Sprintf("GPRWCNTR Enc#%d (%s)", ep.Index, ep.Source), - Category: "encoder_profile", + Name: fmt.Sprintf("Profiler stream %s #%d", ep.Source, ep.Index), + Category: "profiler_stream", Phase: "X", Timestamp: startNs / 1000, Duration: ep.DurationNs / 1000, @@ -2498,9 +6181,10 @@ func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataSt } // Add kernel events from streamData dispatches. - if !addDispatchKernelEvents(timeline, stats, timelineDispatchSIMDStats{}, nil, nil, nil, nil) { - addEncoderKernelEvents(timeline, nil, nil) + if !addDispatchKernelEvents(timeline, stats, timelineDispatchCaptureStats{}, nil, nil, nil, nil) { + addEncoderKernelEvents(timeline, nil, nil, nil) } + timeline.bundleDispatches = timelineBundleDispatchCount(stats, nil) // Set timeline duration if timeline.EndTime > timeline.StartTime { @@ -2577,7 +6261,11 @@ func buildTimelineFromProfilerData(tracePath string, stats *counter.StreamDataSt } // generateInteractiveHTML creates a standalone interactive HTML timeline viewer. -func generateInteractiveHTML(timelineJSON string) string { +func generateInteractiveHTML(timelineJSON string, perfettoJSON ...string) string { + perfJSON := "{}" + if len(perfettoJSON) > 0 && perfettoJSON[0] != "" { + perfJSON = perfettoJSON[0] + } return ` @@ -2729,6 +6417,14 @@ func generateInteractiveHTML(timelineJSON string) string { color: #858585; } + .counter-group { + margin: 10px 0 4px; + font-size: 11px; + text-transform: uppercase; + letter-spacing: 0.5px; + color: #8c8c8c; + } + .counter-track { padding: 6px 10px; margin-bottom: 4px; @@ -2837,12 +6533,37 @@ func generateInteractiveHTML(timelineJSON string) string { pointer-events: none; white-space: nowrap; } + + #warning-banner { + background: #6a4f00; + color: #fff8dc; + padding: 6px 20px; + font-size: 12px; + display: none; + align-items: center; + gap: 8px; + border-bottom: 1px solid #8b6b00; + } + + #warning-banner.visible { + display: flex; + } + + .badge-estimated { + background: #d7ba7d; + color: #1e1e1e; + font-size: 11px; + font-weight: 600; + padding: 2px 6px; + border-radius: 3px; + margin-left: 8px; + }