diff --git a/backends/ze/Makefile.am b/backends/ze/Makefile.am index 942c0947b..275533a43 100644 --- a/backends/ze/Makefile.am +++ b/backends/ze/Makefile.am @@ -175,6 +175,11 @@ CLEANFILES += tracer_ze.c bin_SCRIPTS = \ tracer_ze.sh +if ZE_VALIDATOR + bin_SCRIPTS += validator/ze_validator +endif + + noinst_LTLIBRARIES = libzetracepoints.la nodist_libzetracepoints_la_SOURCES = \ @@ -251,6 +256,23 @@ data_DATA = \ ze_bindings_base.rb \ babeltrace_zeprofiling_apis.txt +# Everything the ze_validator needs +ZE_VALIDATOR_SOURCES = \ + validator/errors.rb \ + validator/allocation_map.rb \ + validator/model.rb \ + validator/callbacks.rb \ + validator/semantics.rb \ + validator/state_object.rb \ + validator/api_metadata.yaml + +EXTRA_DIST += $(ZE_VALIDATOR_SOURCES) + +if ZE_VALIDATOR + ze_validatordir = $(datadir)/ze/validator + ze_validator_DATA = $(ZE_VALIDATOR_SOURCES) +endif + xprof_utils.hpp: $(top_srcdir)/utils/xprof_utils.hpp cp $< $@ diff --git a/backends/ze/validator/.gitignore b/backends/ze/validator/.gitignore new file mode 100644 index 000000000..e2824772f --- /dev/null +++ b/backends/ze/validator/.gitignore @@ -0,0 +1,2 @@ +ze_validator +!api_metadata.yaml diff --git a/backends/ze/validator/allocation_map.rb b/backends/ze/validator/allocation_map.rb new file mode 100644 index 000000000..96f985bb1 --- /dev/null +++ b/backends/ze/validator/allocation_map.rb @@ -0,0 +1,57 @@ +require 'rbtree' + +module ZEModel + + # A collection of pairwise disjoint intervals, keyed on their base pointer. + class AllocationMap + + class OverlapError < StandardError + attr_reader :entries + def initialize(entries) + @entries = entries + super('range overlaps a live entry') + end + end + + def initialize + @tree = RBTree.new + end + + def [](addr) + pair = @tree.upper_bound(addr) + pair && addr < pair[1].base + pair[1].size ? pair[1] : nil + end + + def insert(obj) + clashes = overlapping(obj.base, obj.size) + raise OverlapError, clashes unless clashes.empty? + @tree[obj.base] = obj + end + + def delete(base) + @tree.delete(base) + end + + # The entries meeting [base, base + size), in ascending base order. + def overlapping(base, size) + return [] unless size.positive? + hits = @tree.bound(base, base + size - 1).map { |_base, obj| obj } + pred = @tree.upper_bound(base - 1) + hits.unshift(pred[1]) if pred && pred[1].base + pred[1].size > base + hits + end + + def each(&block) + @tree.each_value(&block) + end + alias each_value each + + def empty? + @tree.empty? + end + + def size + @tree.size + end + end +end diff --git a/backends/ze/validator/api_metadata.yaml b/backends/ze/validator/api_metadata.yaml new file mode 100644 index 000000000..fc1f5e1b7 --- /dev/null +++ b/backends/ze/validator/api_metadata.yaml @@ -0,0 +1,50 @@ +--- +# API -> [version it was deprecated in, replacement]. +deprecated: + zeInit: ['1.10', zeInitDrivers] + zeDriverGet: ['1.10', zeInitDrivers] + zeCommandListImmediateAppendCommandListsExp: ['1.16', zeCommandListImmediateAppendCommandListsWithParameters] + zeImageViewCreateExp: ['', zeImageViewCreateExt] + zesRasGetConfig: ['1.16', zesRasGetConfigExp] + zesRasSetConfig: ['1.16', zesRasSetConfigExp] + +# API -> the [arg name, arg type] +thread_unsafe: + zeCommandListDestroy: + - [hCommandList, command_list] + zeCommandListClose: + - [hCommandList, command_list] + zeCommandListReset: + - [hCommandList, command_list] + zeCommandListAppendWriteGlobalTimestamp: + - [hCommandList, command_list] + zeCommandListAppendLaunchKernel: + - [hCommandList, command_list] + zeCommandListAppendBarrier: + - [hCommandList, command_list] + zeCommandListAppendLaunchCooperativeKernel: + - [hCommandList, command_list] + zeCommandListAppendMemoryCopy: + - [hCommandList, command_list] + zeCommandListAppendMemoryFill: + - [hCommandList, command_list] + zeCommandListAppendMemoryCopyRegion: + - [hCommandList, command_list] + zeCommandListAppendSignalEvent: + - [hCommandList, command_list] + zeCommandListAppendWaitOnEvents: + - [hCommandList, command_list] + zeCommandListAppendEventReset: + - [hCommandList, command_list] + zeCommandQueueExecuteCommandLists: + - [phCommandLists_vals, command_list] + zeKernelSetGroupSize: + - [hKernel, kernel] + zeKernelSetArgumentValue: + - [hKernel, kernel] + zeKernelSetCacheConfig: + - [hKernel, kernel] + zeKernelSetIndirectAccess: + - [hKernel, kernel] + zeKernelSetGlobalOffsetExp: + - [hKernel, kernel] diff --git a/backends/ze/validator/callbacks.rb b/backends/ze/validator/callbacks.rb new file mode 100644 index 000000000..7069cabfd --- /dev/null +++ b/backends/ze/validator/callbacks.rb @@ -0,0 +1,720 @@ +# frozen_string_literal: true + +require 'ze/validator/semantics' +require 'ze/validator/model' +require 'ze_library' + +$upon_entry = {} # called to modify program state on entry +$on_successful_exit = {} # called upon seeing exit functions with a successful return code +$on_erroneous_exit = {} # called upon seeing exit functions with a non-successful return code + +# these two record only that the app asked, for the portability checks +$on_successful_exit['zeDeviceGetProperties'] = lambda { |state, ctx, _payload| + device_ptr = state.find_param(ctx, 'hDevice') + devices = state.find_objects(ctx, 'device') + devices[device_ptr].property_fetched = true +} +# Mark that the queue group property was queried. +$on_successful_exit['zeDeviceGetCommandQueueGroupProperties'] = lambda { |state, ctx, _payload| + device_ptr = state.find_param(ctx, 'hDevice') + devices = state.find_objects(ctx, 'device') + devices[device_ptr].cmd_queue_group_properties_queried = true +} + +# For every append API below: validation at entry (the call may crash), +# recording at exit (only a successful append will ever execute). +$upon_entry['zeCommandListAppendLaunchKernel'] = lambda { |state, ctx, payload| + # Retrieve the compute ordinal from the command list + command_lists = state.find_objects(ctx, 'command_list') + cmd_list = command_lists[payload['hCommandList']] + cqg_ordinal = cmd_list ? cmd_list.queue_group_ordinal : 0 + # both checks must run even if the launch later aborts + check_valid_ordinal(state, ctx, payload, cqg_ordinal, cmd_list) + check_kernel_created(state, ctx, payload) + # the kernel's module must be on the same context as the command list + check_kernel_list_context_match(state, ctx, payload) +} + +$on_successful_exit['zeCommandListAppendLaunchKernel'] = lambda { |state, ctx, _payload| + record_op(state, ctx, state.find_param(ctx, 'hCommandList'), + ZEModel::RecordedOp.new(:launch, + signal: state.find_param(ctx, 'hSignalEvent'), + waits: wait_event_handles(state, ctx), + api: 'zeCommandListAppendLaunchKernel')) +} + +$upon_entry['zeCommandListReset'] = lambda { |state, ctx, payload| + check_command_list_reset(state, ctx, payload) +} + +# on success the list is empty and open again, so clear the recorded ops. +# In-flight executions are unaffected: they snapshotted the ops at submit time. +$on_successful_exit['zeCommandListReset'] = lambda { |state, ctx, _payload| + command_lists = state.find_objects(ctx, 'command_list') + cmd_list = command_lists[state.find_param(ctx, 'hCommandList')] + if cmd_list + cmd_list.ops.clear + cmd_list.status = ZEModel::CommandList.class_variable_get(:@@INITIALIZED) + end +} + +# check_command_list_closed later verifies this happened before any submission +$on_successful_exit['zeCommandListClose'] = lambda { |state, ctx, _payload| + command_lists = state.find_objects(ctx, 'command_list') + command_list_handle = state.find_param(ctx, 'hCommandList') + cmd_list = command_lists[command_list_handle] + cmd_list.status = ZEModel::CommandList.class_variable_get(:@@CLOSED) +} + +$upon_entry['zeCommandListAppendLaunchCooperativeKernel'] = lambda { |state, ctx, payload| + command_lists = state.find_objects(ctx, 'command_list') + cmd_list = command_lists[payload['hCommandList']] + check_group_property_queued(state, ctx, payload, cmd_list.device) if cmd_list + # the kernel's module must be on the same context as the command list + check_kernel_list_context_match(state, ctx, payload) +} + +$on_successful_exit['zeCommandListAppendLaunchCooperativeKernel'] = lambda { |state, ctx, _payload| + record_op(state, ctx, state.find_param(ctx, 'hCommandList'), + ZEModel::RecordedOp.new(:launch, + signal: state.find_param(ctx, 'hSignalEvent'), + waits: wait_event_handles(state, ctx), + api: 'zeCommandListAppendLaunchCooperativeKernel')) +} + +# Copy/event ops are recorded in list order for deferred replay. The +# out-of-bounds check waits until the op's wait-events are satisfied. +$upon_entry['zeCommandListAppendMemoryCopy'] = lambda { |state, ctx, payload| + params = { api: 'zeCommandListAppendMemoryCopy', + ctx_handle: cmd_list_ctx_handle(state, ctx, payload['hCommandList']), + dst: payload['dstptr'], src: payload['srcptr'], size: payload['size'] } + waits = wait_event_handles(state, ctx) + check_null_copy_ptr(state, ctx, 'zeCommandListAppendMemoryCopy', + { 'destination' => payload['dstptr'], 'source' => payload['srcptr'] }) + check_use_after_free_on_append(state, ctx, params, waits) + # an out-of-bounds copy can crash the driver, which emits no _exit + check_oob_copy_on_append(state, ctx, params, waits) + # known memory endpoints must be allocated on the command list's context + check_copy_ptr_list_context(state, ctx, 'zeCommandListAppendMemoryCopy', payload['hCommandList'], + { 'destination' => payload['dstptr'], 'source' => payload['srcptr'] }) +} + +$upon_entry['zeCommandListAppendMemoryFill'] = lambda { |state, ctx, payload| + params = { api: 'zeCommandListAppendMemoryFill', + ctx_handle: cmd_list_ctx_handle(state, ctx, payload['hCommandList']), + dst: payload['ptr'], src: nil, size: payload['size'] } + waits = wait_event_handles(state, ctx) + check_null_copy_ptr(state, ctx, 'zeCommandListAppendMemoryFill', + { 'destination' => payload['ptr'] }) + check_use_after_free_on_append(state, ctx, params, waits) + # an out-of-bounds fill can crash the driver, which emits no _exit + check_oob_copy_on_append(state, ctx, params, waits) + # known memory endpoint must be allocated on the command list's context + check_copy_ptr_list_context(state, ctx, 'zeCommandListAppendMemoryFill', payload['hCommandList'], + { 'destination' => payload['ptr'] }) +} + +$on_successful_exit['zeCommandListAppendMemoryCopy'] = lambda { |state, ctx, _payload| + record_copy_op(state, ctx, 'zeCommandListAppendMemoryCopy', 'dstptr', 'srcptr') +} + +$on_successful_exit['zeCommandListAppendMemoryFill'] = lambda { |state, ctx, _payload| + # a fill only touches the destination; model it as a copy with no source + record_copy_op(state, ctx, 'zeCommandListAppendMemoryFill', 'ptr', nil) +} + +# A failed append is never recorded, so the deferred check would never see it, +# but the copy is out-of-bounds regardless of the error code. Check it here. +$on_erroneous_exit['zeCommandListAppendMemoryCopy'] = lambda { |state, ctx, _payload| + params = { api: 'zeCommandListAppendMemoryCopy', + ctx_handle: cmd_list_ctx_handle(state, ctx, state.find_param(ctx, 'hCommandList')), + dst: state.find_param(ctx, 'dstptr'), + src: state.find_param(ctx, 'srcptr'), + size: state.find_param(ctx, 'size') } + check_oob_copy(state, ctx, params) + check_use_after_free(state, ctx, params) +} + +$on_erroneous_exit['zeCommandListAppendMemoryFill'] = lambda { |state, ctx, _payload| + params = { api: 'zeCommandListAppendMemoryFill', + ctx_handle: cmd_list_ctx_handle(state, ctx, state.find_param(ctx, 'hCommandList')), + dst: state.find_param(ctx, 'ptr'), + src: nil, + size: state.find_param(ctx, 'size') } + check_oob_copy(state, ctx, params) + check_use_after_free(state, ctx, params) +} + +# region copies carry 2D/3D extents, so `size` is not a flat byte count; we only +# record ordering + event effects and skip the flat OOB comparison +$on_successful_exit['zeCommandListAppendMemoryCopyRegion'] = lambda { |state, ctx, _payload| + record_op(state, ctx, state.find_param(ctx, 'hCommandList'), + ZEModel::RecordedOp.new(:launch, + signal: state.find_param(ctx, 'hSignalEvent'), + waits: wait_event_handles(state, ctx), + api: 'zeCommandListAppendMemoryCopyRegion')) +} + +# A device-side signal: the event is signaled when this op executes (after waits). +$on_successful_exit['zeCommandListAppendSignalEvent'] = lambda { |state, ctx, _payload| + record_op(state, ctx, state.find_param(ctx, 'hCommandList'), + ZEModel::RecordedOp.new(:signal, signal: state.find_param(ctx, 'hEvent'), + api: 'zeCommandListAppendSignalEvent')) +} + +# A device-side wait: this op blocks the list until phEvents are signaled. +$on_successful_exit['zeCommandListAppendWaitOnEvents'] = lambda { |state, ctx, _payload| + record_op(state, ctx, state.find_param(ctx, 'hCommandList'), + ZEModel::RecordedOp.new(:wait, waits: wait_event_handles(state, ctx), + api: 'zeCommandListAppendWaitOnEvents')) +} + +# A device-side reset: returns the event to unsignaled when this op executes. +$on_successful_exit['zeCommandListAppendEventReset'] = lambda { |state, ctx, _payload| + record_op(state, ctx, state.find_param(ctx, 'hCommandList'), + ZEModel::RecordedOp.new(:reset, params: { reset_handle: state.find_param(ctx, 'hEvent') })) +} + +# A barrier waits on its events and signals its completion event. +$on_successful_exit['zeCommandListAppendBarrier'] = lambda { |state, ctx, _payload| + record_op(state, ctx, state.find_param(ctx, 'hCommandList'), + ZEModel::RecordedOp.new(:barrier, + signal: state.find_param(ctx, 'hSignalEvent'), + waits: wait_event_handles(state, ctx), + api: 'zeCommandListAppendBarrier')) +} + +# Same event semantics as a plain barrier, plus the memory ranges it names, +# which are validated when the barrier executes (check_uaf_ranges_barrier). +$on_successful_exit['zeCommandListAppendMemoryRangesBarrier'] = lambda { |state, ctx, _payload| + record_ranges_barrier_op(state, ctx) +} + +# host-side event operations, effective immediately in trace order +$on_successful_exit['zeEventHostSignal'] = lambda { |state, ctx, _payload| + handle = state.find_param(ctx, 'hEvent') + check_event_signal_reuse(state, ctx, handle, 'zeEventHostSignal') + state.signal_event(ctx, handle, 'zeEventHostSignal') +} + +$on_successful_exit['zeEventHostReset'] = lambda { |state, ctx, _payload| + state.reset_event(ctx, state.find_param(ctx, 'hEvent')) +} + +# does not signal the event, only records that the signaled state was consumed +$on_successful_exit['zeEventHostSynchronize'] = lambda { |state, ctx, _payload| + state.observe_event(ctx, state.find_param(ctx, 'hEvent')) +} + +# A successful status query also observes the signaled state. +$on_successful_exit['zeEventQueryStatus'] = lambda { |state, ctx, _payload| + state.observe_event(ctx, state.find_param(ctx, 'hEvent')) +} + +# the host waited for all submitted work, so every signaled event was consumed +$on_successful_exit['zeCommandQueueSynchronize'] = lambda { |state, ctx, _payload| + state.observe_all_signaled_events(ctx) +} + +$on_successful_exit['zeCommandListHostSynchronize'] = lambda { |state, ctx, _payload| + state.observe_all_signaled_events(ctx) +} + +# Submission is where the queue, the lists, their events and the fence are +# So the same context checkings between those objects are called here. +$upon_entry['zeCommandQueueExecuteCommandLists'] = lambda { |state, ctx, payload| + command_queues = state.find_objects(ctx, 'command_queue') + command_queue_handle = payload['hCommandQueue'] + command_queue = command_queues[command_queue_handle] + + # check if any command list is null + check_valid_command_lists(state, ctx, payload) + check_valid_command_queue(state, ctx, payload, command_queues, command_queue_handle) + # Check if command list was closed before executing it on the queue + # ignore if it is the first execute call + check_command_list_closed(state, ctx, payload) + check_fence_misuse(state, ctx, payload) + + known_command_lists = state.find_objects(ctx, 'command_list') + command_list_handles = payload['phCommandLists_vals'] || [] + + fences = state.find_objects(ctx, 'fence') + fence_handle = payload['hFence'] + fence = fences[fence_handle] + + if fence + fence.status = fence.in_use # set this at the entry so that other command lists can view it + end + + if command_queue + check_group_property_queued(state, ctx, payload, command_queue.device) + check_fence_and_queue_compatibility(state, ctx, payload, command_queue, fence) + command_list_handles.each do |command_list_handle| + check_list_and_queue_have_matching_context(state, ctx, payload, known_command_lists[command_list_handle], command_queue) + check_list_and_fence_have_matching_context(state, ctx, payload, known_command_lists[command_list_handle], fence) + # a list with a compute kernel launch must not go to a copy-only queue + check_copy_only_queue_submission(state, ctx, command_queue, known_command_lists[command_list_handle]) + # events used by the list must come from a pool on the queue's context + check_event_pool_list_context_match(state, ctx, known_command_lists[command_list_handle]) + end + end + # an unknown queue was already reported by check_valid_command_queue above; + # the deferred execution below still runs so the lists get their checks +} + +# Execute is asynchronous, so each submitted list becomes a deferred unit and +# its copies are checked when their wait-events are signaled, not here. +$on_successful_exit['zeCommandQueueExecuteCommandLists'] = lambda { |state, ctx, _payload| + known_command_lists = state.find_objects(ctx, 'command_list') + command_list_handles = state.find_param(ctx, 'phCommandLists_vals') || [] + command_lists = command_list_handles.map { |h| known_command_lists[h] } + state.enqueue_deferred_execution(ctx, command_lists) +} + +# When a fence signals the host, set the fence's status to signaled +$on_successful_exit['zeFenceHostSynchronize'] = lambda { |state, ctx, _payload| + fence_handle = state.find_param(ctx, 'hFence') + fence = get_fence(state, ctx, fence_handle) + if fence + check_fence_sync_without_reset(state, ctx, fence_handle, fence) + fence.status = fence.signaled + else + state.report(:null_fence_handle, ctx, 'a null fence was used for zeFenceHostSynchronize') + end +} + +# should a double reset be considered as a usage error? +# Also, a fence can be shared throughout the threads and is modeled correctly (if you are wondering about whether the model treats fence associated with different thread-id differently). +$upon_entry['zeFenceReset'] = lambda { |state, ctx, payload| + curr_fence = get_fence(state, ctx, payload['hFence']) + curr_fence.status = curr_fence.not_signaled if curr_fence +} + +# Set the driver for the current context +$on_successful_exit['zeDriverGet'] = lambda { |state, ctx, payload| + drivers = state.get_process(ctx).drivers + payload['phDrivers_vals'].each do |h| + drivers[h] = ZEModel::Driver.new(h) unless drivers[h] + end +} + +$on_successful_exit['zeDeviceGet'] = lambda { |state, ctx, payload| + devices = state.find_objects(ctx, 'device') + driver = state.find_object(ctx, 'driver', 'hDriver') + if driver + payload['phDevices_vals'].each do |h| + unless devices[h] + devices[h] = ZEModel::Device.new(h) + driver.devices.push devices[h] + end + end + end +} + +$on_successful_exit['zeDeviceGetSubDevices'] = lambda { |state, ctx, payload| + devices = state.find_objects(ctx, 'device') + device = state.find_object(ctx, 'device', 'hDevice') + payload['phSubdevices_vals'].each do |h| + unless devices[h] + devices[h] = ZEModel::SubDevice.new(h, device) + device.sub_devices.push devices[h] + end + end +} + +$on_successful_exit['zeContextCreate'] = lambda { |state, ctx, payload| + contexts = state.find_objects(ctx, 'context') + driver = state.find_object(ctx, 'driver', 'hDriver') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZEContextDesc) + handle = payload['phContext_val'] + contexts[handle] = ZEModel::Context.new(handle, driver, desc) + driver.contexts[handle] = contexts[handle] if driver + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_CONTEXT_DESC, desc[:stype]) +} + +# Experimental API, it practically serves the same purpose as zeContextCreate +$on_successful_exit['zeContextCreateEx'] = lambda { |state, ctx, payload| + contexts = state.find_objects(ctx, 'context') + devices = state.find_objects(ctx, 'device') + driver = state.find_object(ctx, 'driver', 'hDriver') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZEContextDesc) + devs = state.find_param(ctx, 'phDevices_vals').collect { |h| devices[h] } + devs = nil unless state.find_param(ctx, 'phDevices') != 0 + handle = payload['phContext_val'] + contexts[handle] = ZEModel::Context.new(handle, driver, desc, devs) + driver.contexts[handle] = contexts[handle] if driver +} + +# A context owns everything created from it, so its destruction is where +# whatever the program did not clean up first gets reported. +$on_successful_exit['zeContextDestroy'] = lambda { |state, ctx, _payload| + contexts = state.find_objects(ctx, 'context') + handle = state.find_param(ctx, 'hContext') + context_obj = contexts.delete(handle) do + state.object_not_found(ctx, 'context', handle) + end + context_obj.driver&.contexts&.delete(handle) + state.report_orphans(ctx, context_obj) +} + +$on_successful_exit['zeEventPoolCreate'] = lambda { |state, ctx, payload| + context_obj = state.find_object(ctx, 'context', 'hContext') + devices = state.find_objects(ctx, 'device') + event_pools = state.find_objects(ctx, 'event_pool') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZEEventPoolDesc) + devs = state.find_param(ctx, 'phDevices_vals').collect { |h| devices[h] } + devs = nil unless state.find_param(ctx, 'phDevices') != 0 + handle = payload['phEventPool_val'] + event_pools[handle] = ZEModel::EventPool.new(handle, context_obj, desc, devs) + context_obj.event_pools[handle] = event_pools[handle] + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_EVENT_POOL_DESC, desc[:stype]) +} + +# Destroying a pool while events are being used should not occur +$on_successful_exit['zeEventPoolDestroy'] = lambda { |state, ctx, _payload| + event_pools = state.find_objects(ctx, 'event_pool') + handle = state.find_param(ctx, 'hEventPool') + event_pool = event_pools.delete(handle) do + state.object_not_found(ctx, 'event_pool', handle) + end + event_pool.context.event_pools.delete(handle) do + state.object_not_found(ctx, 'event_pool', handle, 'context') + end + state.report_orphans(ctx, event_pool) +} + +$upon_entry['zeEventCreate'] = lambda { |state, ctx, payload| + check_valid_event_pool(state, ctx, payload) +} + +$on_successful_exit['zeEventCreate'] = lambda { |state, ctx, payload| + events = state.find_objects(ctx, 'event') + event_pool = state.find_object(ctx, 'event_pool', 'hEventPool') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZEEventDesc) + handle = payload['phEvent_val'] + events[handle] = ZEModel::Event.new(handle, event_pool, desc) + unless event_pool.indices.delete?(desc[:index]) + state.report(:event_pool_index_in_use, ctx, + "event_pool #{state.get_handle_str(event_pool.handle)} index #{desc[:index]} is already used", + key: "pool-index-used-#{state.get_handle_str(event_pool.handle)}-#{desc[:index]}") + end + event_pool.events[handle] = events[handle] + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_EVENT_DESC, desc[:stype]) +} + +$on_successful_exit['zeEventDestroy'] = lambda { |state, ctx, _payload| + events = state.find_objects(ctx, 'event') + handle = state.find_param(ctx, 'hEvent') + event = events.delete(handle) do + state.object_not_found(ctx, 'event', handle) + end + event_pool = event.event_pool + event_pool.events.delete(handle) do + state.object_not_found(ctx, 'event', handle, 'event_pool') + end + unless event_pool.indices.add?(event.desc[:index]) + state.report(:event_pool_index_already_free, ctx, + "event_pool #{state.get_handle_str(event_pool.handle)} index #{event.desc[:index]} is already freed", + key: "pool-index-freed-#{state.get_handle_str(event_pool.handle)}-#{event.desc[:index]}") + end +} + +$on_successful_exit['zeCommandQueueCreate'] = lambda { |state, ctx, payload| + command_queues = state.find_objects(ctx, 'command_queue') + context_obj = state.find_object(ctx, 'context', 'hContext') + device = state.find_object(ctx, 'device', 'hDevice') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZECommandQueueDesc) + handle = payload['phCommandQueue_val'] + command_queues[handle] = ZEModel::CommandQueue.new(handle, context_obj, device, desc) + context_obj.command_queues[handle] = command_queues[handle] + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_COMMAND_QUEUE_DESC, desc[:stype]) +} + +# A failed creation is likely an (ordinal, index) the device does not have, so +# check the index against the real topology to explain the failure. +$on_erroneous_exit['zeCommandQueueCreate'] = lambda { |state, ctx, _payload| + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZECommandQueueDesc) + device_handle = state.find_param(ctx, 'hDevice') + cmd_queue_handle = state.find_param(ctx, 'phCommandQueue') + check_valid_index_for_ordinal(state, ctx, device_handle, cmd_queue_handle, desc[:ordinal], desc[:index]) +} + +$on_successful_exit['zeCommandQueueDestroy'] = lambda { |state, ctx, _payload| + command_queues = state.find_objects(ctx, 'command_queue') + handle = state.find_param(ctx, 'hCommandQueue') + command_queue = command_queues.delete(handle) do + state.object_not_found(ctx, 'command_queue', handle) + end + command_queue.context.command_queues.delete(handle) do + state.object_not_found(ctx, 'command_queue', handle, 'context') + end + state.report_orphans(ctx, command_queue) +} + +$on_successful_exit['zeFenceCreate'] = lambda { |state, ctx, payload| + fences = state.find_objects(ctx, 'fence') + command_queue = state.find_object(ctx, 'command_queue', 'hCommandQueue') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZEFenceDesc) + handle = payload['phFence_val'] + fence = ZEModel::Fence.new(handle, command_queue, desc) + fences[handle] = fence + command_queue.fences[handle] = fence + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_FENCE_DESC, desc[:stype]) +} + +$on_successful_exit['zeFenceDestroy'] = lambda { |state, ctx, _payload| + fences = state.find_objects(ctx, 'fence') + handle = state.find_param(ctx, 'hFence') + fence = fences.delete(handle) do + state.object_not_found(ctx, 'fence', handle) + end + command_queue = fence.command_queue + command_queue.fences.delete(handle) do + state.object_not_found(ctx, 'fence', handle, 'command_queue') + end +} + +$on_successful_exit['zeCommandListCreate'] = lambda { |state, ctx, payload| + command_lists = state.find_objects(ctx, 'command_list') + context_obj = state.find_object(ctx, 'context', 'hContext') + device = state.find_object(ctx, 'device', 'hDevice') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZECommandListDesc) + handle = payload['phCommandList_val'] + command_lists[handle] = ZEModel::CommandList.new(handle, context_obj, device, desc, nil) + command_lists[handle].in_order = desc[:flags].include?(:ZE_COMMAND_LIST_FLAG_IN_ORDER) + context_obj.command_lists[handle] = command_lists[handle] + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_COMMAND_LIST_DESC, desc[:stype]) +} + +$on_successful_exit['zeCommandListCreateImmediate'] = lambda { |state, ctx, payload| + command_lists = state.find_objects(ctx, 'command_list') + context_obj = state.find_object(ctx, 'context', 'hContext') + device = state.find_object(ctx, 'device', 'hDevice') + altdesc_val = state.find_param(ctx, 'altdesc_val') + altdesc = state.to_struct(altdesc_val, ZE::ZECommandQueueDesc) + handle = payload['phCommandList_val'] + check_group_property_queued(state, ctx, payload, device) + command_lists[handle] = ZEModel::CommandList.new(handle, context_obj, device, nil, altdesc) + command_lists[handle].immediate = true + command_lists[handle].in_order = altdesc[:flags].include?(:ZE_COMMAND_QUEUE_FLAG_IN_ORDER) + context_obj.command_lists[handle] = command_lists[handle] + + # immediate command list does not take in the list descriptor as an input + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_COMMAND_QUEUE_DESC, altdesc[:stype]) +} + +$on_successful_exit['zeCommandListDestroy'] = lambda { |state, ctx, _payload| + command_lists = state.find_objects(ctx, 'command_list') + handle = state.find_param(ctx, 'hCommandList') + command_list = command_lists.delete(handle) do + state.object_not_found(ctx, 'command_list', handle) + end + command_list.context.command_lists.delete(handle) do + state.object_not_found(ctx, 'command_list', handle, 'context') + end +} + +$on_successful_exit['zeModuleCreate'] = lambda { |state, ctx, payload| + modules = state.find_objects(ctx, 'module') + context_obj = state.find_object(ctx, 'context', 'hContext') + device = state.find_object(ctx, 'device', 'hDevice') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZEModuleDesc) + handle = payload['phModule_val'] + mod = ZEModel::Module.new(handle, context_obj, device, desc) + modules[handle] = mod + context_obj.modules[handle] = mod + build_log_handle = payload['phBuildLog_val'] + if build_log_handle != 0 + module_build_logs = state.find_objects(ctx, 'module_build_log') + build_log = ZEModel::Module::BuildLog.new(build_log_handle, context_obj, mod) + module_build_logs[build_log_handle] = build_log + context_obj.module_build_logs[build_log_handle] = build_log + modules[handle].build_log = build_log + end + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_MODULE_DESC, desc[:stype]) +} + +# Runs diagnoistics on why the module crete failed +$on_erroneous_exit['zeModuleCreate'] = lambda { |state, ctx, payload| + build_log_handle = payload['phBuildLog_val'] + if build_log_handle != 0 + context_obj = state.find_object(ctx, 'context', 'hContext') + module_build_logs = state.find_objects(ctx, 'module_build_log') + build_log = ZEModel::Module::BuildLog.new(build_log_handle, context_obj) + module_build_logs[build_log_handle] = build_log + context_obj.module_build_logs[build_log_handle] = build_log + end +} + +# Kernel must be destroyed first before module +$on_successful_exit['zeModuleDestroy'] = lambda { |state, ctx, _payload| + modules = state.find_objects(ctx, 'module') + handle = state.find_param(ctx, 'hModule') + mod = modules.delete(handle) do + state.object_not_found(ctx, 'module', handle) + end + mod.context.modules.delete(handle) do + state.object_not_found(ctx, 'module', handle, 'context') + end + state.report_orphans(ctx, mod) +} + +$on_erroneous_exit['zeModuleDynamicLink'] = $on_successful_exit['zeModuleDynamicLink'] = lambda { |state, ctx, payload| + build_log_handle = payload['phLinkLog_val'] + if build_log_handle != 0 + context_obj = state.find_object(ctx, 'context', 'hContext') + module_build_logs = state.find_objects(ctx, 'module_build_log') + build_log = ZEModel::Module::BuildLog.new(build_log_handle, context_obj) + module_build_logs[build_log_handle] = build_log + context_obj.module_build_logs[build_log_handle] = build_log + end +} + +$on_successful_exit['zeModuleBuildLogDestroy'] = lambda { |state, ctx, _payload| + module_build_logs = state.find_objects(ctx, 'module_build_log') + handle = state.find_param(ctx, 'hModuleBuildLog') + module_build_log = module_build_logs.delete(handle) do + state.object_not_found(ctx, 'module_build_log', handle) + end + if module_build_log.module + module_build_log.module.context.module_build_logs.delete(handle) do + state.object_not_found(ctx, 'module_build_log', handle, 'context') + end + module_build_log.module.build_log = nil + end +} + +# upon entering, check if a valid module was passed +$upon_entry['zeKernelCreate'] = lambda { |state, ctx, payload| + check_valid_module(state, ctx, payload) +} + +$on_successful_exit['zeKernelCreate'] = lambda { |state, ctx, payload| + kernels = state.find_objects(ctx, 'kernel') + mod = state.find_object(ctx, 'module', 'hModule') + desc_val = state.find_param(ctx, 'desc_val') + desc = state.to_struct(desc_val, ZE::ZEKernelDesc) + handle = payload['phKernel_val'] + kernelName = state.find_param(ctx, 'desc__pKernelName_val') + kernel = ZEModel::Kernel.new(handle, mod, desc, kernelName) + kernels[handle] = kernel + mod.kernels[handle] = kernel + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_KERNEL_DESC, desc[:stype]) +} + +$on_successful_exit['zeKernelDestroy'] = lambda { |state, ctx, _payload| + kernels = state.find_objects(ctx, 'kernel') + handle = state.find_param(ctx, 'hKernel') + kernel = kernels.delete(handle) do + state.object_not_found(ctx, 'kernel', handle) + end + mod = kernel.module + mod.kernels.delete(handle) do + state.object_not_found(ctx, 'kernel', handle, 'module') + end +} + +$on_successful_exit['zeMemAllocDevice'] = lambda { |state, ctx, payload| + # memory is associated with devices + ze_context = state.find_param(ctx, 'hContext') + context_obj = state.context_from_handle(ctx, ze_context) + device = state.find_object(ctx, 'device', 'hDevice') + size = state.find_param(ctx, 'size') + device_desc_val = state.find_param(ctx, 'device_desc_val') + handle = payload['pptr_val'] + if context_obj + mark_reallocated(context_obj.freed_memory_allocations, handle, size) + memory_allocation = ZEModel::MemoryAllocation.new(handle, context_obj, size, device, 'device') + track_allocation(state, ctx, context_obj.memory_allocations, memory_allocation) + end + device_desc = state.to_struct(device_desc_val, ZE::ZEDeviceMemAllocDesc) + check_struct_stype_misuse(state, ctx, payload, :ZE_STRUCTURE_TYPE_DEVICE_MEM_ALLOC_DESC, device_desc[:stype]) +} + +$on_successful_exit['zeMemAllocShared'] = lambda { |state, ctx, payload| + ze_context = state.find_param(ctx, 'hContext') + # finds the device and context objects associated with the params + context_obj = state.context_from_handle(ctx, ze_context) + device = state.find_object(ctx, 'device', 'hDevice') + # A nullptr device handle shares ownership between the host and all devices + # supporting cross-device shared access. + size = state.find_param(ctx, 'size') + handle = payload['pptr_val'] + return unless context_obj + + mark_reallocated(context_obj.freed_memory_allocations, handle, size) + memory_allocation = ZEModel::MemoryAllocation.new(handle, context_obj, size, device, 'shared') + track_allocation(state, ctx, context_obj.memory_allocations, memory_allocation) +} + +$on_successful_exit['zeMemAllocHost'] = lambda { |state, ctx, payload| + # Host allocations are accessible by the host and all devices within the driver’s context. + ze_context = state.find_param(ctx, 'hContext') + context_obj = state.context_from_handle(ctx, ze_context) + size = state.find_param(ctx, 'size') + handle = payload['pptr_val'] + return unless context_obj + + mark_reallocated(context_obj.freed_memory_allocations, handle, size) + memory_allocation = ZEModel::MemoryAllocation.new(handle, context_obj, size, nil, 'host') + track_allocation(state, ctx, context_obj.memory_allocations, memory_allocation) +} + +# The free is applied at entry: zeMemFree may block until the buffer is idle, so +# by _exit a gated copy could have drained and the in-flight check would miss it. +$upon_entry['zeMemFree'] = lambda { |state, ctx, payload| + context_obj = state.context_from_handle(ctx, payload['hContext']) + return unless context_obj + + memory_allocations = context_obj.memory_allocations + handle = payload['ptr'] + memory_allocation = find_allocation(memory_allocations, handle) + return unless memory_allocation&.base == handle + + # flag if this buffer is still referenced by a copy/fill that has been + # submitted but not yet completed (in-flight device work would touch freed mem) + check_free_in_flight(state, ctx, memory_allocation) + untrack_allocation(memory_allocations, memory_allocation) + # keep the freed allocation in this context's freed registry so a later + # copy/fill/kernel referencing this address is caught as use-after-free + memory_allocation.freed_by = state.get_api_context(ctx) + track_freed_allocation(context_obj.freed_memory_allocations, memory_allocation) +} + +# the free was applied at entry, so restore the allocation if it actually failed. +# The unknown-context case was already reported at entry, hence the plain lookup. +$on_erroneous_exit['zeMemFree'] = lambda { |state, ctx, _payload| + context_obj = state.find_object(ctx, 'context', 'hContext') + handle = state.find_param(ctx, 'ptr') + return unless context_obj + + freed = context_obj.freed_memory_allocations + mem = find_allocation(freed, handle) + if mem&.base == handle + untrack_allocation(freed, mem) + mem.freed_by = nil + track_allocation(state, ctx, context_obj.memory_allocations, mem) + end +} + +$upon_entry['zeMemGetAddressRange'] = lambda { |state, ctx, payload| + context_obj = state.context_from_handle(ctx, payload['hContext']) + return unless context_obj + + handle = payload['ptr'] + memory_allocation = find_allocation(context_obj.memory_allocations, handle) + if !memory_allocation || memory_allocation.base != handle + state.report(:unallocated_address_range, ctx, + "ptr #{state.get_handle_str(handle)} is either out-of-bounds or never got allocated", + key: "addr-range-#{state.get_handle_str(handle)}") + end +} diff --git a/backends/ze/validator/errors.rb b/backends/ze/validator/errors.rb new file mode 100644 index 000000000..05efee6a4 --- /dev/null +++ b/backends/ze/validator/errors.rb @@ -0,0 +1,361 @@ +# frozen_string_literal: true + +require 'set' + +# Error taxonomy and reporting for ze_validator. +module ZEValidator + + # A node of the error tree. + # Severity is inherited from the parent unless the node overrides it + class ErrorNode + attr_reader :id, :title, :parent, :children + + def initialize(id, title, parent: nil, severity: nil) + @id = id + @title = title + @parent = parent + @children = [] + @severity = severity + end + + # :error or :warning, walking up to the root until a node states one. + def severity + @severity || @parent&.severity || :error + end + + def add_child(node) + @children << node + node + end + + def root? + @parent.nil? + end + + def leaf? + @children.empty? + end + + # The root exists only to hold the categories together and carries no information + def ancestors + node = @parent + out = [] + while node && !node.root? + out.unshift(node) + node = node.parent + end + out + end + + # e.g. "MV/portability/command_queue_group_not_queried" + def path + (ancestors + [self]).map(&:id).join('/') + end + + def depth + ancestors.size + end + + def each_node(&block) + yield self unless root? + @children.each { |c| c.each_node(&block) } + self + end + + def each_leaf(&block) + each_node { |n| block.call(n) if n.leaf? } + end + end + + # Builds the tree from a nested declaration and indexes its leaves by id. + class ErrorTreeBuilder + attr_reader :root, :index + + def initialize(id, title) + @root = ErrorNode.new(id, title) + @index = {} + @stack = [@root] + end + + def category(id, title, severity: nil) + node = @stack.last.add_child(ErrorNode.new(id, title, parent: @stack.last, severity: severity)) + @stack.push(node) + yield if block_given? + @stack.pop + node + end + + def leaf(id, title, severity: nil) + raise "duplicate diagnostic id #{id.inspect}" if @index.key?(id) + node = @stack.last.add_child(ErrorNode.new(id, title, parent: @stack.last, severity: severity)) + @index[id] = node + end + end + + # Level Zero error tree + module ErrorTree + def self.build + b = ErrorTreeBuilder.new(:ze, 'Level Zero API usage findings') + + b.category(:PV, 'Progression Violations') do + b.leaf :circular_event_dependency, 'circular event dependency between commands' + b.leaf :in_order_self_deadlock, 'in-order list waits on an event it only signals later' + b.leaf :unsignaled_wait_event, 'command waits on an event that is never signaled' + end + + b.category(:MSV, 'Memory Safety Violations') do + b.leaf :out_of_bounds_copy, 'copy or fill reaches past the end of its allocation' + b.leaf :use_after_free, 'copy or barrier references a freed allocation' + b.leaf :free_while_in_flight, 'allocation freed while in-flight device work still uses it' + b.leaf :null_copy_pointer, 'copy or fill endpoint is a null pointer' + b.leaf :unallocated_address_range, 'address queried was never allocated or is out of range' + b.leaf :overlapping_allocation, 'allocator returned a range overlapping a live allocation' + end + + # The edge letters refer to Figure 7 of the paper, which enumerates the + # object pairs subject to a common-context requirement. + b.category(:HMV, 'Hierarchy-Membership Violations') do + b.leaf :fence_queue_mismatch, 'fence submitted to a queue other than the one it was created on (edge A)' + b.leaf :list_queue_context_mismatch, 'command list and command queue on different contexts (edge B)' + b.leaf :list_fence_context_mismatch, 'command list and fence on different contexts (edge A/B)' + b.leaf :memory_list_context_mismatch, 'copied memory and command list on different contexts (edge C)' + b.leaf :kernel_list_context_mismatch, 'kernel module and command list on different contexts (edge D)' + b.leaf :event_list_context_mismatch, 'event pool and command list on different contexts (edge E)' + end + + b.category(:MV, 'Miscellaneous Violations') do + b.category(:engine, 'engine and ordinal binding') do + b.leaf :kernel_on_copy_only_list, 'kernel appended to a list bound to a copy-only engine' + b.leaf :compute_list_on_copy_only_queue, 'list holding a kernel submitted to a copy-only queue' + b.leaf :queue_index_out_of_range, 'queue index outside the engine group it was created on' + end + + b.category(:lifetime, 'object lifetime') do + b.leaf :object_leak, 'object never destroyed' + b.leaf :object_outlives_owner, 'object not destroyed before the object that owns it' + end + + b.category(:object_state, 'object state') do + b.leaf :command_list_not_closed, 'command list submitted without being closed' + b.leaf :command_list_already_destroyed, 'destroyed command list submitted' + b.leaf :command_list_reset_after_destroy, 'destroyed command list reset' + b.leaf :command_list_reset_immediate, 'immediate command list reset' + b.leaf :command_list_reset_in_flight, 'command list reset while a submission is still in flight' + b.leaf :fence_reuse_without_reset, 'fence reused without being reset' + b.leaf :event_reuse_without_reset, 'event reused as a signal target without being reset' + b.leaf :event_concurrent_signal, 'event signaled again before being reset or consumed' + b.leaf :event_pool_index_in_use, 'event pool index already in use' + b.leaf :event_pool_index_already_free, 'event pool index already free' + end + + b.category(:invalid_argument, 'invalid argument') do + b.leaf :unknown_context, 'context handle was never created, or was already destroyed' + b.leaf :unknown_command_queue, 'command queue handle was never created' + b.leaf :no_command_list_submitted, 'no command list submitted' + b.leaf :unknown_command_list, 'command list handle was never created' + b.leaf :immediate_list_submitted, 'immediate command list submitted to a command queue' + b.leaf :kernel_not_created, 'kernel handle was never created' + b.leaf :null_module_handle, 'null module handle' + b.leaf :null_fence_handle, 'null fence handle' + b.leaf :null_event_pool_handle, 'null event pool handle' + end + + b.category(:concurrency, 'concurrency') do + b.leaf :concurrent_object_access, 'concurrent access to an object that is not thread-safe' + end + + b.category(:api_conformance, 'API conformance') do + b.leaf :descriptor_stype_mismatch, 'descriptor carries the wrong stype' + b.leaf :init_not_called, 'API called before zeInit or zeInitDrivers' + b.leaf :api_never_returned, 'API call never returned' + b.leaf :deprecated_api, 'deprecated API used', severity: :warning + end + + b.category(:portability, 'portability', severity: :warning) do + b.leaf :command_queue_group_not_queried, 'command queue group never queried, ordinals are hardcoded' + end + + # Not an error of the program, but an error from the tracing itself: it is missing ordinals of copy-engines. + # In such case, report a warning + b.category(:coverage, 'analysis coverage', severity: :warning) do + b.leaf :engine_topology_unknown, 'copy-engine checks skipped, no engine topology in the trace' + end + end + + b + end + + BUILDER = build + ROOT = BUILDER.root + LEAVES = BUILDER.index.freeze + + def self.[](id) + LEAVES.fetch(id) { raise ArgumentError, "unknown diagnostic id #{id.inspect}" } + end + end + + # Formats findings, limits how often each kind may be printed, and keeps the + # counts the end-of-trace summary reports. + class Reporter + # Printed lines allowed per diagnostic kind before the rest are suppressed. + # A single kind routinely accounts for tens of thousands of findings in a + # production trace, which drowns everything else. Edit here to change it. + MAX_REPORTS_PER_KIND = 5 + + TOOL = 'ze_validator' + + # Width of the dotted leader in the summary, chosen so the deepest label + # still leaves room for the count. + SUMMARY_WIDTH = 66 + + def initialize(out: $stderr) + @out = out + # leaf -> every detection, including those suppressed below. This is what + # the summary reports: it answers "how much of this is there", which the + # printed lines no longer do once a kind is capped. + @counts = Hash.new(0) + # leaf -> lines actually printed, compared against MAX_REPORTS_PER_KIND + @printed = Hash.new(0) + # dedup keys already reported, for checks that would otherwise repeat + # verbatim on the same object + @seen_keys = Set.new + # leaves whose suppression notice has been printed + @capped = Set.new + end + + # Reports one error. + # + # id error id (leaf in the tree) + # location the location fields, outermost first: [host, pid, tid] for a + # finding attributable to one call, [host, pid] for a process-wide + # message the human-readable description + # key optional dedup key; a finding whose key was already reported is + # counted but not printed, avoiding redundant reports + def report(id, location, message, key: nil) + node = ErrorTree[id] + @counts[node] += 1 + + return if key && !@seen_keys.add?("#{id}\u0000#{key}") + return if capped?(node) + + @printed[node] += 1 + emit(node, location, message) + end + + def error_count + total_for(:error) + end + + def warning_count + total_for(:warning) + end + + # Prints the per-kind error count + def summary + total = @counts.values.sum + @out.puts + @out.puts "===== #{TOOL} summary =====" + if total.zero? + @out.puts 'no findings' + @out.puts '=' * (12 + TOOL.length) + return + end + + @out.puts format('%d %s: %d %s, %d %s, over %d %s', + total, plural(total, 'finding'), + error_count, plural(error_count, 'error'), + warning_count, plural(warning_count, 'warning'), + @counts.size, plural(@counts.size, 'kind')) + @out.puts + visible_children(ErrorTree::ROOT).each_with_index do |category, i| + @out.puts if i.positive? + print_summary_line(category, '') + print_subtree(category, '') + end + @out.puts '=' * (12 + TOOL.length) + end + + # Writes the per-kind counts as CSV. + def export_csv(path) + File.open(path, 'w') do |io| + io.puts csv_row(CSV_HEADER) + ErrorTree::ROOT.each_leaf do |leaf| + io.puts csv_row([leaf.ancestors.first.id, + leaf.ancestors[1]&.id, + leaf.id, + leaf.path, + leaf.severity, + leaf.title, + @counts[leaf], + @printed[leaf]]) + end + end + rescue SystemCallError => e + @out.puts "[#{TOOL}] could not write CSV export to #{path}: #{e.message}" + end + + private + + CSV_HEADER = %w[category subcategory id path severity description detected reported].freeze + + def csv_row(fields) + fields.map { |f| + s = f.to_s + s.match?(/[",\r\n]/) ? %("#{s.gsub('"', '""')}") : s + }.join(',') + end + + # Detections of a node and of everything below it. + def subtree_count(node) + count = @counts[node] + node.children.each { |c| count += subtree_count(c) } + count + end + + def total_for(severity) + @counts.sum { |node, n| node.severity == severity ? n : 0 } + end + + def plural(count, word) + count == 1 ? word : "#{word}s" + end + + def visible_children(node) + node.children.select { |c| subtree_count(c).positive? } + end + + def print_subtree(node, rail) + children = visible_children(node) + children.each_with_index do |child, i| + last = i == children.size - 1 + print_summary_line(child, rail + (last ? '└─ ' : '├─ ')) + print_subtree(child, rail + (last ? ' ' : '│ ')) + end + end + + def print_summary_line(node, prefix) + label = node.depth.zero? && !node.leaf? ? "#{node.title} (#{node.id})" : node.id.to_s + label += ' [warning]' if node.leaf? && node.severity == :warning + count = subtree_count(node) + value = node.leaf? ? count.to_s : "(#{count})" + leader = SUMMARY_WIDTH - prefix.length - label.length - 2 + leader = 1 if leader < 1 + @out.puts format('%s%s %s %8s', prefix, label, '.' * leader, value) + end + + def capped?(node) + return false if @printed[node] < MAX_REPORTS_PER_KIND + + if @capped.add?(node) + @out.puts "[#{TOOL}] [#{node.severity}] [#{node.path}] reported #{MAX_REPORTS_PER_KIND} times; " \ + 'further occurrences are suppressed (see the end-of-trace summary)' + end + true + end + + def emit(node, location, message) + where = Array(location).map { |f| "[#{f}]" }.join(' ') + @out.puts "[#{TOOL}] #{where} [#{node.severity}] [#{node.path}] #{message}" + end + end +end diff --git a/backends/ze/validator/model.rb b/backends/ze/validator/model.rb new file mode 100644 index 000000000..4726145bb --- /dev/null +++ b/backends/ze/validator/model.rb @@ -0,0 +1,456 @@ +# frozen_string_literal: true + +require 'ze/validator/allocation_map' +require 'set' + +module ZEModel + # One of these APIs must be called before any other calls + INIT_API_NAMES = %w[zeInit zeInitDrivers].freeze + + # Which object owns which, following the containment of the specs + OWNERSHIP = { + 'context' => 'driver', + 'memory_allocation' => 'context', + 'event_pool' => 'context', + 'command_queue' => 'context', + 'command_list' => 'context', + 'module' => 'context', + 'module_build_log' => 'context', + 'event' => 'event_pool', + 'fence' => 'command_queue', + 'kernel' => 'module' + }.freeze + + # OWNERSHIP inverted + OWNED_TYPES = OWNERSHIP.each_with_object({}) { |(child, owner), h| + (h[owner] ||= []) << child + }.each_value(&:freeze).freeze + + # This defines the object in which most ze objects (command list, command queue) extend form + class Object + attr_reader :handle + attr_accessor :status + attr_accessor :leak_reported + + # returns the typename of the object: must match exactly 'OWNERSHIP' strings. + # e.g., 'Device' object has :typename = 'device' + class << self + attr_reader :typename + end + + def initialize(handle) + @handle = handle + @lock = nil + @leak_reported = false + end + + # The object that owns this one, or nil when it was never recorded. + def owner + owner_type = OWNERSHIP[self.class.typename] + owner_type && instance_variable_get(:"@#{owner_type}") + end + + # Yields every live child of one type. + def each_child(type, &block) + children(type).each_value(&block) + end + + def child_count(type) + children(type).size + end + + # Lock the object. If it already has been locked, report a race. + # This is typically to be used for thread safety checks on API parameters. + def lock(state, ctx) + if @lock + state.print_race_condition(ctx, @lock, self.class.typename, @handle) + else + @lock = ctx + end + end + + # Unlock the object + def unlock(ctx) + return unless @lock == ctx + + @lock = nil + end + + private + + # The container for a type is the instance variable named after its plural. + # e.g., a 'Driver' has multiple 'Context' in the 'contexts' container member. + def children(type) + instance_variable_get(:"@#{type}s") + end + end + + class Driver < Object + @typename = 'driver' + attr_reader :devices + attr_reader :contexts + + def initialize(handle) + super + @devices = [] + @contexts = {} + end + end + + class Device < Object + @typename = 'device' + attr_reader :sub_devices + attr_accessor :property_fetched + attr_accessor :cmd_queue_group_properties_queried + + def initialize(handle) + super + @sub_devices = [] + @property_fetched = false + @cmd_queue_group_properties_queried = false + end + end + + class SubDevice < Device + attr_reader :parent + + def initialize(handle, parent) + @parent = parent + super(handle) + end + end + + # One allocation returned by zeMemAllocDevice/Shared/Host. `memtypestr` keeps + # device, shared and host allocations apart. + class MemoryAllocation < Object + @typename = 'memory_allocation' + attr_reader :context, :size, :owned_by # the Device for a device allocation; nil for host + attr_accessor :memtypestr, :base # "device" | "host" | "shared" + attr_accessor :freed_by # the zeMemFree that released this allocation, nil while live. + + def initialize(handle, context, size, owned_by, memtypestr) + super(handle) + @context = context + @size = size + @owned_by = owned_by + @memtypestr = memtypestr + @base = handle + @freed_by = nil + end + end + + class Context < Object + @typename = 'context' + attr_reader :driver, :desc, :devices, :event_pools, :command_queues, :command_lists, :modules, :module_build_logs + attr_reader :memory_allocations + attr_reader :freed_memory_allocations + + def initialize(handle, driver, desc, devices = nil) + super(handle) + @driver = driver + @desc = desc + @devices = devices + + @event_pools = {} + @command_queues = {} + @command_lists = {} + @modules = {} # binaries for gpu + @module_build_logs = {} + @memory_allocations = AllocationMap.new + @freed_memory_allocations = AllocationMap.new + end + end + + class EventPool < Object + @typename = 'event_pool' + + attr_reader :context, :desc, :devices, :events + + # slot indices not yet in use: + # - zeEventCreate removes one (double use = error) + # - zeEventDestroy puts it back (double free = error) + attr_reader :indices + + def initialize(handle, context, desc, devices = nil) + super(handle) + @context = context + @desc = desc + @devices = devices + @events = {} + @indices = Set.new(desc[:count].times.to_a) + end + end + + class Event < Object + @typename = 'event' + attr_reader :event_pool, :desc, :signaled_by + attr_accessor :signaled # who last signaled it, for diagnostics + # whether the host observed the signaled state since the last signal. Tells + # a concurrent double-signal (never consumed) from a reuse-without-reset. + attr_reader :observed + + def initialize(handle, event_pool, desc) + super(handle) + @event_pool = event_pool + @desc = desc + # event can have 2 states, not signaled or signaled + @signaled = false + @signaled_by = nil + @observed = false + end + + # `by` records who signaled it, for messages + def signal(by = nil) + @signaled = true + @signaled_by = by + @observed = false + end + + def reset + @signaled = false + @signaled_by = nil + @observed = false + end + + def observe + @observed = true + end + end + + class CommandQueue < Object + @typename = 'command_queue' + attr_reader :context, :device, :fences + # :ordinal and :index are checked against the engine topology recorded by + # the lttng_ust_ze_properties:command_queue_group tracepoint + attr_reader :desc + + def initialize(handle, context, device, desc) + super(handle) + @context = context + @device = device + @desc = desc + @fences = {} + end + end + + class Fence < Object + @typename = 'fence' + attr_reader :command_queue, :desc, :not_signaled, :in_use, :signaled + + # not_signaled -> in_use -> signaled -> not_signaled (zeFenceReset). + attr_accessor :status + + def initialize(handle, command_queue, desc) + super(handle) + @command_queue = command_queue + @desc = desc + @not_signaled = 0 + @in_use = 1 + @signaled = 2 + @status = @not_signaled + end + end + + class CommandList < Object + @typename = 'command_list' + attr_reader :context, :device, :desc, :altdesc + attr_accessor :associated_command_queue, :immediate + attr_accessor :in_order + attr_accessor :ops + + @@INITIALIZED = 0 # created or being properly recycled + @@CLOSED = 1 + @@DESTROYED = 2 + + def initialize(handle, context, device, desc, altdesc) + super(handle) + @context = context + @device = device + @desc = desc + @altdesc = altdesc + @associated_command_queue = nil + @status = @@INITIALIZED + @immediate = false + @in_order = false + @ops = [] + end + + def queue_group_ordinal + return @desc[:commandQueueGroupOrdinal] if @desc + return @altdesc[:ordinal] if @altdesc + + 0 + end + end + + class RecordedOp + # :copy, :wait, :signal, :reset, :barrier, :ranges_barrier or :launch + attr_reader :kind + attr_reader :signal # event this op signals on completion (nil if none) + attr_reader :waits # events that must be signaled before this op may run + attr_reader :params, :api + + def initialize(kind, signal: 0, waits: [], params: {}, api: nil) + @kind = kind + @signal = signal == 0 ? nil : signal # normalize 0 to nil + @waits = waits + @params = params + @api = api || params[:api] + end + end + + class DeferredUnit + attr_reader :ops # snapshot of the list's ops for this execution + attr_reader :context # trace context captured at submit time + attr_reader :label # e.g. "command_list (0x00007f...) + attr_reader :in_order + + attr_accessor :cursor # index of the next op to run; == ops.size means done + attr_accessor :blocked_on # events the current op is still waiting for + + # Events this unit has not signaled yet. If unit U is blocked on an event + # only in V's pending_signals, U waits on V: an edge in the wait-for graph. + attr_accessor :pending_signals + + # the command list this unit came from, so list-scoped checks can find their + # units without matching on the label string + attr_reader :cmd_list_handle + + # true when this unit is the running tail of an immediate list rather than a + # queue submission; only the latter counts as in-flight for a reset + attr_reader :immediate + + def initialize(ops, context, label, in_order: false, cmd_list_handle: nil, immediate: false) + @ops = ops + @context = context + @label = label + @cursor = 0 + @blocked_on = [] + @in_order = in_order + @cmd_list_handle = cmd_list_handle + @immediate = immediate + # every event this unit will eventually signal, for the wait-for graph + @pending_signals = ops.map(&:signal).compact + end + + # Queues one more op behind the ones already here. + def push_op(new_op) + @ops << new_op + @pending_signals << new_op.signal if new_op.signal + end + + # true once every op has executed + def done? + @cursor >= @ops.size + end + + # the op the cursor currently points at (nil when done) + def current_op + @ops[@cursor] + end + end + + class Module < Object + @typename = 'module' + + class BuildLog < Object + @typename = 'module_build_log' + attr_reader :context + attr_reader :module # nil when the build failed and produced no module + + def initialize(handle, context, mod = nil) + super(handle) + @context = context + @module = mod + end + end + + attr_reader :context, :device, :desc, :kernels + attr_accessor :build_log + + def initialize(handle, context, device, desc) + super(handle) + @context = context + @device = device + @desc = desc + @kernels = {} + end + end + + class Kernel < Object + @typename = 'kernel' + attr_reader :module, :desc, :name + + def initialize(handle, mod, desc, name) + super(handle) + @module = mod + @desc = desc + @name = name + end + end + + class ApiCall + attr_reader :name, :params + + def initialize(name, params) + @name = name + @params = params + end + end + + class Thread + attr_reader :vtid + + # call stack: a traced API may call another traced API on the same thread + attr_reader :call_stack + + def initialize(vtid) + @vtid = vtid + @call_stack = [] + end + + # the innermost in-flight ApiCall, or nil if the thread has none + def last_entry + @call_stack.last + end + end + + class Process + + # handle -> object, one table per Level Zero object type + attr_reader :vpid + attr_reader :threads + attr_reader :drivers, :devices, :contexts, :kernels, :event_pools, :events, :command_queues, :fences, :command_lists, :modules, :module_build_logs + + def initialize(vpid) + @vpid = vpid + @threads = Hash.new { |h, k| h[k] = Thread.new(k) } + @drivers = {} + @devices = {} + @contexts = {} + @event_pools = {} + @events = {} + @command_queues = {} + @fences = {} + @command_lists = {} + @modules = {} + @module_build_logs = {} + @kernels = {} + end + + # objects('command_list') returns @command_lists, so callers can iterate + # object types by name (see StateObject#finalize). + def objects(type) + instance_variable_get(:"@#{type}s") + end + end + + class Node + attr_reader :name, :processes # hostname # pid -> Process (auto-created on first sight) + + def initialize(name) + @name = name + @processes = Hash.new { |h, k| h[k] = Process.new(k) } + end + end +end diff --git a/backends/ze/validator/semantics.rb b/backends/ze/validator/semantics.rb new file mode 100644 index 000000000..fe1ede162 --- /dev/null +++ b/backends/ze/validator/semantics.rb @@ -0,0 +1,768 @@ +# frozen_string_literal: true + +require 'ze/validator/model' +require 'ze_library' + +# The engine groups recorded for a device, or nil when the trace carries none. +def device_command_queue_groups(state, device_handle) + return nil unless device_handle + + all = state.device_properties + return nil unless all.key?(device_handle) + + groups = all[device_handle] + groups.empty? ? nil : groups +end + +# Checks for oob index. A command queue is created with an (ordinal, index) +# pair -- which engine group, and which queue within that group. +# Silent when the topology is unknown: this only runs after zeCommandQueueCreate +# already failed, so a "could not check" note would just be noise there. +def check_valid_index_for_ordinal(state, ctx, device_handle, cmd_q_handle, ordinal, index) + groups = device_command_queue_groups(state, device_handle) + return if groups.nil? + + groups.each do |ordinal_key, info| + # find matching ordinal, and check whether the index is oob + next unless ordinal_key == ordinal && (index >= info['numQueues'] || index.negative?) + state.report(:queue_index_out_of_range, ctx, + "command queue (#{state.get_handle_str(cmd_q_handle)}) with ordinal = #{ordinal} was created " \ + "with index = #{index}. Index value should be: 0<= index < #{info['numQueues']}") + end +end + +# Checking whether the application ever called zeDeviceGetCommandQueueGroupProperties +# before calling command queue/list create. Not calling it implies hardcoded ordinals +def check_group_property_queued(state, ctx, _payload, device) + return if device.cmd_queue_group_properties_queried + + state.report(:command_queue_group_not_queried, ctx, + "command queue group wasn't queried. Hardcoded group properties may break the code on different devices", + key: 'check_group_property') +end + +# The copy-only ordinals of the command list's device, or nil when the trace +# does not carry that device's engine topology. +def copy_only_ordinals(state, cmd_list) + groups = device_command_queue_groups(state, cmd_list&.device&.handle) + return nil unless groups + + groups.filter_map do |ordinal, prop| + flags = prop['flags'] + ordinal if flags.include?(:ZE_COMMAND_QUEUE_GROUP_PROPERTY_FLAG_COPY) && + !flags.include?(:ZE_COMMAND_QUEUE_GROUP_PROPERTY_FLAG_COMPUTE) + end +end + +# Records that a copy-engine check could not run for want of topology. Deduped +# per device, so one trace reports each unknown device once whichever check hit +# it first. +def report_unknown_engine_topology(state, ctx, cmd_list) + handle = cmd_list&.device ? state.get_handle_str(cmd_list.device.handle) : 'unknown' + state.report(:engine_topology_unknown, ctx, + "copy-engine checks skipped for device #{handle}: the trace carries no command queue " \ + 'group properties for it (THAPI records them for root devices only, and only when the ' \ + 'properties channel is enabled)', + key: "engine-topology-#{handle}") +end + +# checks whether a command list attached to a copy-only engine receives a kernel +def check_valid_ordinal(state, ctx, _payload, cqg_ordinal, cmd_list) + copy_only = copy_only_ordinals(state, cmd_list) + if copy_only.nil? + report_unknown_engine_topology(state, ctx, cmd_list) + return + end + return unless copy_only.include?(cqg_ordinal) + + kernels = state.find_objects(ctx, 'kernel') + kernel_handle = state.find_param(ctx, 'hKernel') + command_list_handle = state.find_param(ctx, 'hCommandList') + kernel_name = kernels[kernel_handle]&.name || 'UNKNOWN' + state.report(:kernel_on_copy_only_list, ctx, + "launching kernel (#{kernel_name}) on command list #{state.get_handle_str(command_list_handle)}, " \ + "which is bound to copy-only ordinal #{cqg_ordinal}", + key: "copy-ordinal-#{state.get_handle_str(command_list_handle)}-#{kernel_name}") +end + +# list of compute launches +COMPUTE_LAUNCH_APIS = %w[zeCommandListAppendLaunchKernel + zeCommandListAppendLaunchCooperativeKernel].freeze + +def command_list_has_kernel_launch?(cmd_list) + return false unless cmd_list + + cmd_list.ops.any? { |op| op.kind == :launch && COMPUTE_LAUNCH_APIS.include?(op.api) } +end + +# Checks whether a command list that has a compute kernel gets submitted to a command queue that is attached to a copy only engine. +def check_copy_only_queue_submission(state, ctx, queue, cmd_list) + return unless queue.desc && command_list_has_kernel_launch?(cmd_list) + + queue_ordinal = queue.desc[:ordinal] + copy_only = copy_only_ordinals(state, cmd_list) + if copy_only.nil? + report_unknown_engine_topology(state, ctx, cmd_list) + return + end + return unless copy_only.include?(queue_ordinal) + + state.report(:compute_list_on_copy_only_queue, ctx, + "command list #{state.get_handle_str(cmd_list.handle)} contains a compute kernel " \ + "launch but was submitted to command queue #{state.get_handle_str(queue.handle)} " \ + "with copy-only ordinal #{queue_ordinal}", + key: "copyq-submit-#{state.get_handle_str(queue.handle)}-#{state.get_handle_str(cmd_list.handle)}") +end + +# Checks whether the kernel module's context matches that of the command list's. +def check_kernel_list_context_match(state, ctx, payload) + command_lists = state.find_objects(ctx, 'command_list') + kernels = state.find_objects(ctx, 'kernel') + cmd_list = command_lists[payload['hCommandList']] + kernel = kernels[payload['hKernel']] + return unless cmd_list&.context && kernel + + mod = kernel.module + return unless mod&.context && mod.context != cmd_list.context + + state.report(:kernel_list_context_mismatch, ctx, + "kernel #{state.get_handle_str(kernel.handle)} (from module " \ + "#{state.get_handle_str(mod.handle)} on context #{state.get_handle_str(mod.context.handle)}) " \ + "does not share the context of command list #{state.get_handle_str(cmd_list.handle)} " \ + "(context #{state.get_handle_str(cmd_list.context.handle)})", + key: "kernel-list-ctx-#{state.get_handle_str(cmd_list.handle)}-#{state.get_handle_str(kernel.handle)}") +end + +# Checks if the kernel was created +def check_kernel_created(state, ctx, payload) + kernels = state.find_objects(ctx, 'kernel') + kernel_handle = payload['hKernel'] + return if kernels[kernel_handle] + + state.report(:kernel_not_created, ctx, + "kernel #{state.get_handle_str(kernel_handle)} wasn't created. Consider calling zeKernelCreate") +end + +# Checks for using fence without reset +def check_fence_misuse(state, ctx, payload) + fence_handle = payload['hFence'] + fence = get_fence(state, ctx, fence_handle) + return unless fence && (fence.status == fence.signaled || fence.status == fence.in_use) + + state.report(:fence_reuse_without_reset, ctx, + "fence #{state.get_handle_str(fence_handle)} was used twice without being reset") +end + +# Checks for synchronizing on a fence that is already signaled. +def check_fence_sync_without_reset(state, ctx, fence_handle, fence) + return unless fence && fence.status == fence.signaled + + state.report(:fence_reuse_without_reset, ctx, + "fence #{state.get_handle_str(fence_handle)} was synchronized again without being reset") +end + +# Check whether the queue handed to ExecuteCommandLists was never created (or was already destroyed). +def check_valid_command_queue(state, ctx, _payload, cmd_queues, cmd_queue_ptr) + cmd_queue = cmd_queues[cmd_queue_ptr] + return if cmd_queue + + state.report(:unknown_command_queue, ctx, + "command queue #{state.get_handle_str(cmd_queue_ptr)} handed to " \ + 'zeCommandQueueExecuteCommandLists was never created (or was already destroyed)', + key: "unknown-queue-#{state.get_handle_str(cmd_queue_ptr)}") +end + +# Checks for submitting nothing, submitting a handle that was never created, or +# submitting an immediate list, which carries its own queue. +def check_valid_command_lists(state, ctx, payload) + command_queue_handle = payload['hCommandQueue'] + command_lists = payload['phCommandLists_vals'] + known_command_lists = state.find_objects(ctx, 'command_list') + if command_lists.nil? || command_lists.empty? + state.report(:no_command_list_submitted, ctx, + 'no command list was submitted to zeCommandQueueExecuteCommandLists') + return + end + + command_lists.each do |command_list_handle| + cmd_list = known_command_lists[command_list_handle] + if !cmd_list + state.report(:unknown_command_list, ctx, + "command list #{state.get_handle_str(command_list_handle)} handed to " \ + 'zeCommandQueueExecuteCommandLists was never created (or was already destroyed)', + key: "unknown-list-#{state.get_handle_str(command_list_handle)}") + elsif cmd_list.immediate + state.report(:immediate_list_submitted, ctx, + "immediate command list #{state.get_handle_str(command_list_handle)} was submitted to " \ + "command queue #{state.get_handle_str(command_queue_handle)}; an immediate list carries its own queue", + key: "immediate-submit-#{state.get_handle_str(command_list_handle)}") + end + end +end + +# Resolve a fence handle to its model object (nil if unknown). +def get_fence(state, context, fence_handle) + fences = state.find_objects(context, 'fence') + fences[fence_handle] # returns fence +end + +# Resolve the Level Zero context handle that owns a command list. +def cmd_list_ctx_handle(state, ctx, cmd_list_handle) + cmd_list = state.find_objects(ctx, 'command_list')[cmd_list_handle] + cmd_list&.context&.handle +end + +# retrieves the wait event handles at the current state +def wait_event_handles(state, ctx) + handles = state.find_param(ctx, 'phWaitEvents_vals') || + state.find_param(ctx, 'phEvents_vals') || [] + handles.reject(&:zero?) +end + +# Record one op onto a command list +def record_op(state, ctx, cmd_list_handle, op) + cmd_list = state.find_objects(ctx, 'command_list')[cmd_list_handle] + return unless cmd_list + + if cmd_list.immediate + check_event_pool_immediate_list_context_match(state, ctx, cmd_list, op) + state.enqueue_immediate_op(ctx, op, cmd_list_handle, in_order: cmd_list.in_order) + else + cmd_list.ops << op + end +end + +# Record a memory-copy op (zeCommandListAppendMemoryCopy / MemoryFill). +def record_copy_op(state, ctx, api, dst_key, src_key) + cmd_list_handle = state.find_param(ctx, 'hCommandList') + op = ZEModel::RecordedOp.new(:copy, + signal: state.find_param(ctx, 'hSignalEvent'), + waits: wait_event_handles(state, ctx), + params: { api: api, + ctx_handle: cmd_list_ctx_handle(state, ctx, cmd_list_handle), + dst: (dst_key ? state.find_param(ctx, dst_key) : nil), + src: (src_key ? state.find_param(ctx, src_key) : nil), + size: state.find_param(ctx, 'size') }) + record_op(state, ctx, cmd_list_handle, op) +end + +# Records a zeCommandListAppendMemoryRangesBarrier op. +def record_ranges_barrier_op(state, ctx) + cmd_list_handle = state.find_param(ctx, 'hCommandList') + bases = state.find_param(ctx, 'pRanges_vals') || [] + sizes = state.find_param(ctx, 'pRangeSizes_vals') || [] + ranges = bases.each_with_index.map { |base, i| { base: base, size: sizes[i] } } + op = ZEModel::RecordedOp.new(:ranges_barrier, + signal: state.find_param(ctx, 'hSignalEvent'), + waits: wait_event_handles(state, ctx), + params: { api: 'zeCommandListAppendMemoryRangesBarrier', + ctx_handle: cmd_list_ctx_handle(state, ctx, cmd_list_handle), + ranges: ranges }) + record_op(state, ctx, cmd_list_handle, op) +end + +# Check if a command list was closed before launching anything on it (called at the execute command lists, for non-immediate command queues) +def check_command_list_closed(state, ctx, payload) + command_queue_handle = payload['hCommandQueue'] + command_lists = payload['phCommandLists_vals'] || [] + known_command_lists = state.find_objects(ctx, 'command_list') + command_lists.each do |command_list_handle| + cmd_list = known_command_lists[command_list_handle] + next unless cmd_list + + if cmd_list.status == ZEModel::CommandList.class_variable_get(:@@INITIALIZED) + state.report(:command_list_not_closed, ctx, + "command list #{state.get_handle_str(command_list_handle)} wasn't closed before being executed " \ + "on command queue #{state.get_handle_str(command_queue_handle)}", + key: "not-closed-#{state.get_handle_str(command_list_handle)}") + elsif cmd_list.status == ZEModel::CommandList.class_variable_get(:@@DESTROYED) + state.report(:command_list_already_destroyed, ctx, + "command list #{state.get_handle_str(command_list_handle)} was already destroyed when submitted " \ + "to command queue #{state.get_handle_str(command_queue_handle)}", + key: "submit-destroyed-#{state.get_handle_str(command_list_handle)}") + end + end +end + +# check if the command list reset is valid or not. +# Invalid calls: reset on destroyed lists, reset on immeidate lists, and reset on command lists that are already exeucting. +def check_command_list_reset(state, ctx, payload) + handle = payload['hCommandList'] + cmd_list = state.find_objects(ctx, 'command_list')[handle] + return unless cmd_list + + if cmd_list.status == ZEModel::CommandList.class_variable_get(:@@DESTROYED) + state.report(:command_list_reset_after_destroy, ctx, + "command list #{state.get_handle_str(handle)} was already destroyed before zeCommandListReset", + key: "clreset-destroyed-#{state.get_handle_str(handle)}") + return + end + + if cmd_list.immediate + state.report(:command_list_reset_immediate, ctx, + "zeCommandListReset called on immediate command list #{state.get_handle_str(handle)}; " \ + 'immediate command lists cannot be reset', + key: "clreset-immediate-#{state.get_handle_str(handle)}") + end + + return unless state.command_list_in_flight?(ctx, handle) + + state.report(:command_list_reset_in_flight, ctx, + "command list #{state.get_handle_str(handle)} is being reset while a prior " \ + 'zeCommandQueueExecuteCommandLists submission is still in-flight; the device may ' \ + 'still be executing it (undefined behavior)', + key: "clreset-inflight-#{state.get_handle_str(handle)}") +end + +# checks whether zeKernelCreate was given a null module handle. +def check_valid_module(state, ctx, _payload) + module_handle = state.find_param(ctx, 'hModule') + return unless !module_handle || module_handle.zero? + + state.report(:null_module_handle, ctx, 'a null hModule was handed to zeKernelCreate') +end + +# Checks whether zeEventCreate was given an event pool that was never created. +def check_valid_event_pool(state, ctx, payload) + pool_handle = payload['hEventPool'] + return unless !pool_handle || pool_handle.zero? + + state.report(:null_event_pool_handle, ctx, 'a null hEventPool was handed to zeEventCreate') +end + +# Checks if the fence's queue and the command list is on the same context. +def check_list_and_fence_have_matching_context(state, ctx, _payload, cmd_list, fence) + return unless cmd_list&.context && fence&.command_queue&.context && cmd_list.context != fence.command_queue.context + + list_handle = state.get_handle_str(cmd_list.handle) + fence_handle = state.get_handle_str(fence.handle) + state.report(:list_fence_context_mismatch, ctx, + "mismatching context between command list #{list_handle} and fence #{fence_handle}", + key: "list-fence-ctx-#{list_handle}-#{fence_handle}") +end + +# Checks for context between queue and the fence. +# Stronger than a context match, as it checks for the matching of the queue. +def check_fence_and_queue_compatibility(state, ctx, _payload, cmd_queue, fence) + return unless fence && cmd_queue != fence.command_queue + + queue_handle = state.get_handle_str(cmd_queue.handle) + fence_handle = state.get_handle_str(fence.handle) + state.report(:fence_queue_mismatch, ctx, + "fence #{fence_handle} was created on command queue " \ + "#{state.get_handle_str(fence.command_queue.handle)} but was submitted to #{queue_handle}", + key: "fence-queue-#{fence_handle}-#{queue_handle}") +end + +# Check the context between the queue and the list +def check_list_and_queue_have_matching_context(state, ctx, _payload, cmd_list, cmd_queue) + return if cmd_list && cmd_list.context == cmd_queue.context + + queue_handle = state.get_handle_str(cmd_queue.handle) + list_handle = cmd_list ? state.get_handle_str(cmd_list.handle) : 'nullptr' + state.report(:list_queue_context_mismatch, ctx, + "mismatching context between command queue #{queue_handle} and command list #{list_handle}", + key: "list-queue-ctx-#{queue_handle}-#{list_handle}") +end + +# List of operations to collect the events from +EVENT_OP_KINDS = %i[copy launch signal wait reset].freeze + +# retrieves the events in a given op +def event_handles_in_op(op) + handles = [] + if EVENT_OP_KINDS.include?(op.kind) + handles << op.signal if op.signal + handles.concat(op.waits) + end + handles +end + +# returns the distinct event handles a command list references across all of its +# recorded ops that are subject to the same-context requirement. +def event_handles_in_list(cmd_list) + cmd_list.ops.flat_map { |op| event_handles_in_op(op) }.uniq +end + +# Check if all events share the same context +def check_events_share_context(state, ctx, event_handles, ref_context, ref_kind, ref_handle) + return unless ref_context + + events = state.find_objects(ctx, 'event') + event_handles.uniq.each do |h| + ev = events[h] + next if !(ev && ev.event_pool && ev.event_pool.context) || ev.event_pool.context == ref_context + + state.report(:event_list_context_mismatch, ctx, + "event #{state.get_handle_str(h)} (from event pool " \ + "#{state.get_handle_str(ev.event_pool.handle)} on context " \ + "#{state.get_handle_str(ev.event_pool.context.handle)}) does not share the context of " \ + "#{ref_kind} #{state.get_handle_str(ref_handle)} " \ + "(context #{state.get_handle_str(ref_context.handle)})", + key: "evpool-#{ref_kind}-ctx-#{state.get_handle_str(ref_handle)}-#{state.get_handle_str(h)}") + end +end + +# Check if event pool's context matches the command queue's context +def check_event_pool_list_context_match(state, ctx, cmd_list) + return unless cmd_list + + check_events_share_context(state, ctx, event_handles_in_list(cmd_list), + cmd_list.context, 'command list', cmd_list.handle) +end + +# Check if event pool's context matches the immediate command list's context +def check_event_pool_immediate_list_context_match(state, ctx, cmd_list, op) + return unless cmd_list.context + + check_events_share_context(state, ctx, event_handles_in_op(op), + cmd_list.context, 'immediate command list', cmd_list.handle) +end + +# The MemoryAllocation whose range holds 'ptr', or nil. +# - allocations: an AllocationMap +# - ptr: any pointer (i.e., not necessarily a base ptr) +def find_allocation(allocations, ptr) + allocations[ptr] +end + +# Records a live allocation. +def track_allocation(state, ctx, allocations, mem) + allocations.insert(mem) +rescue ZEModel::AllocationMap::OverlapError => e + state.report(:overlapping_allocation, ctx, + "allocation #{state.get_handle_str(mem.base)} of #{mem.size} bytes overlaps a " \ + 'live allocation of the same context; the allocator should return disjoint ranges', + key: "overlap-#{state.get_handle_str(mem.base)}-#{mem.size}") + e.entries.each { |old| allocations.delete(old.base) } + allocations.insert(mem) +end + +# Records a freed allocation. +def track_freed_allocation(allocations, mem) + allocations.insert(mem) +rescue ZEModel::AllocationMap::OverlapError => e + e.entries.each { |old| allocations.delete(old.base) } + allocations.insert(mem) +end + +# Drops an allocation from a map. +def untrack_allocation(allocations, mem) + allocations.delete(mem.base) +end + +# Check whether the copy's endpoints have enough space to support the requested size +# Deduped so an append checked at entry is not reported again when it executes. +def check_copy_endpoint_oob(state, ctx, allocations, ptr, size, api, role) + return unless ptr && ptr != 0 && size + + mem = find_allocation(allocations, ptr) + return unless mem + + offset = ptr - mem.base + available = mem.size - offset + return unless available < size + + state.report(:out_of_bounds_copy, ctx, + "#{api}: #{role} memory #{state.get_handle_str(ptr)} only has #{available} " \ + "bytes available from this offset but the copy needs #{size} bytes", + key: "oob-#{api}-#{role}-#{state.get_handle_str(ptr)}-#{size}") +end + +# Performs the oob check for copy for both endpoints (src and dst) +def check_oob_copy(state, ctx, params) + api = params[:api] + size = params[:size] + allocations = state.memory_allocations(ctx, params[:ctx_handle]) + return unless allocations + + check_copy_endpoint_oob(state, ctx, allocations, params[:dst], size, api, 'destination') + check_copy_endpoint_oob(state, ctx, allocations, params[:src], size, api, 'source') +end + +# Check if the copy is from/to a nullptr +def check_null_copy_ptr(state, ctx, api, endpoints) + endpoints.each do |role, ptr| + next unless ptr.nil? || ptr.zero? + + state.report(:null_copy_pointer, ctx, "#{api}: #{role} pointer is nullptr") + end +end + +# Drops the freed allocations the new one reuses. +def mark_reallocated(freed, handle, size) + freed.overlapping(handle, size).each { |old| freed.delete(old.base) } +end + +# Checks for use-after-free on an address +def check_uaf_endpoint(state, ctx, live, freed, ptr, api, role) + return unless ptr && ptr != 0 && find_allocation(live, ptr).nil? + + mem = find_allocation(freed, ptr) + return unless mem + + offset = ptr - mem.base + where = offset.zero? ? '' : " (offset #{offset} into the freed allocation)" + state.report(:use_after_free, ctx, + "#{api}: #{role} memory #{state.get_handle_str(ptr)}#{where} was already " \ + "freed#{" by Process #{mem.freed_by}" if mem.freed_by}; use-after-free", + key: "uaf-#{api}-#{state.get_handle_str(ptr)}") +end + +# Checks for when an API uses a memory that has been freed +def check_use_after_free(state, ctx, params) + api = params[:api] + live = state.memory_allocations(ctx, params[:ctx_handle]) + freed = state.freed_memory_allocations(ctx, params[:ctx_handle]) + + return unless live && freed + + check_uaf_endpoint(state, ctx, live, freed, params[:dst], api, 'destination') + check_uaf_endpoint(state, ctx, live, freed, params[:src], api, 'source') +end + +# calls the check_use_after_free only if the wait events have been satisfied +def check_use_after_free_on_append(state, ctx, params, waits) + return unless state.waits_satisfied?(ctx, waits) + + check_use_after_free(state, ctx, params) +end + +# Calls check_oob_copy at append time, so an append that crashes the driver (and +# so emits no _exit) is still checked. Gated on the waits like the uaf check. +def check_oob_copy_on_append(state, ctx, params, waits) + return unless state.waits_satisfied?(ctx, waits) + + check_oob_copy(state, ctx, params) +end + +# Checks for uaf on memory ranges barrier +def check_uaf_ranges_barrier(state, ctx, params) + api = params[:api] + live = state.memory_allocations(ctx, params[:ctx_handle]) + freed = state.freed_memory_allocations(ctx, params[:ctx_handle]) + + return unless live && freed + + params[:ranges].each do |r| + check_uaf_endpoint(state, ctx, live, freed, r[:base], api, 'range') + end +end + +# returns true if [a, a+asize) and [b, b+bsize) overlap. +def ranges_overlap?(a, asize, b, bsize) + a < b + bsize && b < a + asize +end + +# Checks for whether memory was deleted during execution of a command list +def check_free_in_flight(state, ctx, mem) + mem_ctx_handle = mem.context.handle + state.each_inflight_copy_op(ctx) do |unit, op| + p = op.params + next unless p[:ctx_handle] == mem_ctx_handle + + hit = [[p[:dst], 'destination'], [p[:src], 'source']].find do |ptr, _role| + ptr && ptr != 0 && ranges_overlap?(mem.base, mem.size, ptr, p[:size]) + end + next unless hit + + _ptr, role = hit + state.report(:free_while_in_flight, ctx, + "memory #{state.get_handle_str(mem.base)} is being freed while still in use as " \ + "the #{role} of an in-flight #{p[:api] || 'copy'} on #{unit.label}; the device " \ + 'may access freed memory', + key: "free-inflight-#{state.get_handle_str(mem.base)}-#{unit.label}-#{role}") + end +end + +# Finds the live allocation holding 'ptr' in one Level Zero context, or the one +# containing it, or nil. An unknown context simply holds nothing. +def find_allocation_in_context(state, ctx, ctx_handle, ptr) + allocations = state.memory_allocations(ctx, ctx_handle) + allocations && find_allocation(allocations, ptr) +end + +# Returns [MemoryAllocation, context_handle] for ptr, preferring the passed context (usually command list's context). +def find_known_memory(state, ctx, ptr, prefer_ctx_handle) + if prefer_ctx_handle + mem = find_allocation_in_context(state, ctx, prefer_ctx_handle, ptr) + return [mem, prefer_ctx_handle] if mem + end + state.get_process(ctx).contexts.each do |cth, context_obj| + next if cth == prefer_ctx_handle + + mem = find_allocation(context_obj.memory_allocations, ptr) + return [mem, cth] if mem + end + [nil, nil] +end + +# Checks that one copy/fill endpoint was allocated on the command list's +# context. Untracked pointers are skipped; deduped per (list, endpoint, ptr). +def check_ptr_endpoint_list_context(state, ctx, list_ctx_handle, list_handle, ptr, api, role) + # unknown command list context -> skip + return unless ptr && ptr != 0 && list_ctx_handle + + mem, found_ctx = find_known_memory(state, ctx, ptr, list_ctx_handle) + # unknown pointer -> skip (no false alarm) + return unless mem && found_ctx != list_ctx_handle + + mem_ctx_str = state.get_handle_str(mem.context.handle) + state.report(:memory_list_context_mismatch, ctx, + "#{api}: #{role} memory #{state.get_handle_str(ptr)} was allocated on context #{mem_ctx_str} " \ + "but command list #{state.get_handle_str(list_handle)} is on context #{state.get_handle_str(list_ctx_handle)}; " \ + 'the command list and copied memory must share a context', + key: "ptr-list-ctx-#{state.get_handle_str(list_handle)}-#{role}-#{state.get_handle_str(ptr)}") +end + +# Checks a copy/fill's endpoints against the command list's context. Runs at +# entry: a cross-context copy can be rejected inside the append. +def check_copy_ptr_list_context(state, ctx, api, list_handle, endpoints) + list_ctx_handle = cmd_list_ctx_handle(state, ctx, list_handle) + endpoints.each do |role, ptr| + check_ptr_endpoint_list_context(state, ctx, list_ctx_handle, list_handle, ptr, api, role) + end +end + +# Checks for an event signaled while already signaled with no reset between: +# reuse-no-reset if the host observed the prior signal, double-signal if not. +def check_event_signal_reuse(state, ctx, handle, who) + ev = state.event_by_handle(ctx, handle) + return unless ev&.signaled + + if ev.observed + state.report(:event_reuse_without_reset, ctx, + "event #{state.get_handle_str(handle)} was reused as a signal target by #{who} " \ + 'without calling zeEventHostReset/zeCommandListAppendEventReset after it was ' \ + "signaled#{" by #{ev.signaled_by}" if ev.signaled_by}", + key: "event-reuse-#{state.get_handle_str(handle)}-#{who}") + else + state.report(:event_concurrent_signal, ctx, + "event #{state.get_handle_str(handle)} was signaled by #{who} before being reset " \ + "or consumed#{" (already signaled by #{ev.signaled_by})" if ev.signaled_by}; " \ + 'concurrent signals of the same event are undefined', + key: "event-double-signal-#{state.get_handle_str(handle)}-#{who}") + end +end + +# Reports wait-events never signaled by end of trace, i.e. a deferred op that +# could never complete. +def report_unsignaled_waits(state, ctx, waits) + waits.each do |h| + ev = state.event_by_handle(ctx, h) + next unless ev && !ev.signaled + + state.report(:unsignaled_wait_event, ctx, + "event #{state.get_handle_str(h)} was never signaled; a deferred command list " \ + 'operation could not complete (possible deadlock or missing signal)', + key: "unsignaled-#{state.get_handle_str(h)}") + end +end + +# The first cycle in a wait-for graph, or nil when there is none. +def first_wait_for_cycle(roots, wait_for) + depth_on_path = {} # unit -> its index in path, while it is on the branch + done = {} # unit -> true once fully explored, so it is never revisited + + roots.each do |root| + next if done[root] + + path = [root] + depth_on_path[root] = 0 + # one frame per unit on the path, holding its not-yet-followed successors + stack = [wait_for[root].dup] + + until stack.empty? + nxt = stack.last.shift + if nxt.nil? # successors exhausted: back out of this unit + finished = path.pop + depth_on_path.delete(finished) + done[finished] = true + stack.pop + elsif (i = depth_on_path[nxt]) # still on the branch: the cycle closes here + return path[i..] + elsif !done[nxt] + depth_on_path[nxt] = path.size + path.push(nxt) + stack.push(wait_for[nxt].dup) + end + end + end + nil +end + +# Checks for a circular event dependency across the units still stuck at end of +# trace, reporting the first cycle found. +def check_circular_deadlock(state, units) + stuck = units.reject { |u| u.blocked_on.empty? } + return if stuck.empty? + + # event handle -> units that may still signal it + signalers = Hash.new { |h, k| h[k] = [] } + stuck.each { |u| u.pending_signals.each { |ev| signalers[ev] << u } } + + # wait-for graph: U -> V if U waits on an event V still owes. + wait_for = stuck.to_h do |u| + [u, u.blocked_on.flat_map { |ev| signalers[ev] }.reject { |v| v.equal?(u) }.uniq] + end + + cycle = first_wait_for_cycle(stuck, wait_for) + report_deadlock_cycle(state, cycle) if cycle +end + +# Labels one node of a deadlock cycle as "::". +def deadlock_node_label(state, unit) + op = unit.current_op + # fall back to the op kind so the label is never blank + api = op ? (op.api || op.kind.to_s) : 'unknown' + waits = unit.blocked_on.map { |h| state.get_handle_str(h) }.join(', ') + "#{unit.label}::#{api} (waiting on event #{waits})" +end + +def report_deadlock_cycle(state, cycle) + ctx = cycle.first.context + desc = cycle.map { |u| deadlock_node_label(state, u) }.join(' -> ') + # close the loop for readability + desc << " -> #{deadlock_node_label(state, cycle.first)}" + state.report_proc(:circular_event_dependency, ctx, + "circular event dependency among command list operations; none can start: #{desc}") +end + +# Checks for an in-order list parked on an event only a later op in the same +# list signals. The cross-list detector misses this since it drops self-edges. +def check_in_order_self_deadlock(state, units) + units.each do |unit| + next unless unit.in_order && !unit.blocked_on.empty? + + self_waits = unit.blocked_on & unit.pending_signals + self_waits.each do |ev| + # the later op in this same list that would signal ev (but never runs) + later = unit.ops[(unit.cursor + 1)..]&.find { |o| o.signal == ev } + report_in_order_self_deadlock(state, unit, ev, later) + end + end +end + +# Reports one intra-list self-deadlock as -> . +def report_in_order_self_deadlock(state, unit, ev, signaling_op) + waiting = unit.current_op + waiting_api = waiting ? (waiting.api || waiting.kind.to_s) : 'unknown' + signaling_api = signaling_op ? (signaling_op.api || signaling_op.kind.to_s) : 'unknown' + ev_str = state.get_handle_str(ev) + desc = "#{unit.label}::#{waiting_api} (waits on event #{ev_str}) -> " \ + "#{unit.label}::#{signaling_api} (signals event #{ev_str} later in the same in-order list)" + state.report_proc(:in_order_self_deadlock, unit.context, + 'in-order command list cannot complete; an earlier command waits on an event a later ' \ + "command in the same list signals: #{desc}", + key: "self-deadlock-#{unit.label}-#{ev_str}") +end + +# Checks a descriptor's stype. Current drivers ignore a wrong one, but it is a +# latent bug a future driver may reject. Reported once per expected stype. +def check_struct_stype_misuse(state, ctx, _payload, expected_stype, observed_stype) + return if expected_stype == observed_stype + + state.report(:descriptor_stype_mismatch, ctx, + "expected stype #{expected_stype} but #{observed_stype} was observed", + key: expected_stype.to_s) +end diff --git a/backends/ze/validator/state_object.rb b/backends/ze/validator/state_object.rb new file mode 100644 index 000000000..30dc51396 --- /dev/null +++ b/backends/ze/validator/state_object.rb @@ -0,0 +1,586 @@ +require 'babeltrace2' +require 'ze_library' +require 'set' +require 'ze/validator/errors' +require 'ze/validator/model' +require 'ze/validator/callbacks' +require 'yaml' + +class StateObject + attr_reader :state + attr_reader :lock_shared_object_on_entry + attr_reader :unlock_shared_object_on_exit + attr_reader :device_properties + attr_reader :reporter + + def initialize(**opts) + metadata = YAML.load_file(File.join(DATADIR, 'ze', 'validator', 'api_metadata.yaml')) + @deprecated = metadata.fetch('deprecated') + @device_properties = Hash.new { |h, k| h[k] = {} } + @reporter = ZEValidator::Reporter.new + + # fullpath to a .csv file for exporting detected errors on 'finalize()' + @csv_export = opts[:csv_export] + + # end-of-trace leak sweep, on unless --no-report-leaks turned it off + @report_leaks = opts.fetch(:report_leaks, true) + + # APIs whose deprecation warning has already been printed. + @deprecation_warned = Set.new + + @state = Hash.new { |h, k| h[k] = ZEModel::Node.new(k) } + @ze_thread_safety = metadata.fetch('thread_unsafe') + @lock_shared_object_on_entry = Hash.new { |h, k| h[k] = [] } + @unlock_shared_object_on_exit = Hash.new { |h, k| h[k] = [] } + @init_called = Hash.new { |h, k| h[k] = false } #pid : init called status + @printed_init_error = false + @deferred_units = [] + @ze_thread_safety.each { |api, objects| + objects.each { |o| + @lock_shared_object_on_entry[api].push( lambda { |state, ctx, payload| + #at entry the input args are in payload directly + handle = payload[o.first] + if handle.kind_of? Array + handle.each { |h| + obj = state.find_object(ctx, o.last, h) + obj.lock(state, ctx) if obj + } + else + obj = state.find_object(ctx, o.last, handle) + obj.lock(state, ctx) if obj + end + }) + @unlock_shared_object_on_exit[api].push( lambda { |state, ctx, payload| + #at exit payload holds only outputs, so the input + #handle comes from the saved entry payload + handle = state.find_param(ctx, o.first) + if handle.kind_of? Array + handle.each { |h| + obj = state.find_object(ctx, o.last, h) + obj.unlock(ctx) if obj + } + else + obj = state.find_object(ctx, o.last, handle) + obj.unlock(ctx) if obj + end + }) + } + } + + end + + + + # The innermost API call currently executing on this thread, or nil. + def get_last_entry(context) + @state[context['hostname']].processes[context['vpid']].threads[context['vtid']].last_entry + end + + def get_thread(context) + @state[context['hostname']].processes[context['vpid']].threads[context['vtid']] + end + + def get_process(context) + @state[context['hostname']].processes[context['vpid']] + end + + # Checks that the call we return from is on top of this thread's stack. A + # mismatch means the model lost sync with the trace, so it aborts. + def check_last_entry(context) + last_entry = get_last_entry(context) + unless last_entry && last_entry.name == context['api'] + raise "Invalid State in #{context['api']}" + end + end + + + # Pushes a call frame, so a traced API calling another traced API on the same + # thread nests correctly. + def set_last_entry(state, context, payload) + get_thread(context).call_stack.push(ZEModel::ApiCall.new(context['api'], payload)) + end + + # Pops the innermost frame on return, exposing the caller's frame. + def reset_last_entry(context) + get_thread(context).call_stack.pop + end + + # Decides whether on_exit runs the success or the error callback. + def validate_result(payload) + ZE::ZEResult.from_native(payload["zeResult"], nil) == :ZE_RESULT_SUCCESS + end + + def get_handle_str(handle) + '0x%016x' % handle + end + + def get_proc_context_str(context) + "#{context['hostname']}:#{context['vpid']}" + end + + def get_api_context(context) + "#{context['vtid']} in #{context['api']}" + end + + def get_context_str(context) + "#{get_proc_context_str(context)}:#{context['vtid']}" + end + + # Reports a finding attributable to one API call. `id` names a leaf of the + # diagnostic tree (see errors.rb); `key`, when given, + # suppresses verbatim repeats about the same object. + def report(id, context, str, key: nil) + @reporter.report(id, [context['hostname'], context['vpid'], context['vtid']], + "in #{context['api']}: #{str}", key: key) + end + + # Reports a process-wide finding, which has no thread or API to attribute to. + def report_proc(id, context, str, key: nil) + @reporter.report(id, [context['hostname'], context['vpid']], str, key: key) + end + + # Warns once per deprecated API actually used. + def print_deprecation_warning(context, old_api) + return unless @deprecation_warned.add?(old_api) + + deprecated_since, new_api = @deprecated[old_api] + since = deprecated_since.to_s.empty? ? '' : " since #{deprecated_since}" + report(:deprecated_api, context, "#{old_api} is deprecated#{since}. Please use #{new_api} instead.") + end + + # Reports a leaked object, naming what it still held. The objects inside it + # are not reported on their own, so the count is where the detail went. + def print_leak_error(context, type, obj) + report_proc(:object_leak, context, + "#{type} #{get_handle_str(obj.handle)} was never destroyed#{held_summary(obj)}") + end + + # " (still held 2 command_lists, 5 memory_allocations)", or "" for an object + # that owns nothing or had nothing left in it. + def held_summary(obj) + held = ZEModel::OWNED_TYPES.fetch(obj.class.typename, []).filter_map { |type| + count = obj.child_count(type) + "#{count} #{type}#{'s' if count > 1}" if count.positive? + } + held.empty? ? '' : " (still held #{held.join(', ')})" + end + + # How an object reads on a report line. Memory is a range rather than a + # handle, and carries the kind the allocator gave it. + def object_label(type, obj) + return "#{obj.memtypestr}-memory #{get_handle_str(obj.base)}" if type == 'memory_allocation' + + "#{type} #{get_handle_str(obj.handle)}" + end + + # Not a finding about the traced program: the validator's own bookkeeping is + # wrong, so further output would be untrustworthy. + def raise_internal_error(context, str) + raise "Invalid state #{get_context_str(context)} in #{context['api']}: #{str}" + end + + # Deduped per (object, other holder) so a racing loop reports once. + def print_race_condition(context, other_context, type, handle) + report(:concurrent_object_access, context, + "concurrent access to #{type} #{get_handle_str(handle)}, already held by #{get_api_context(other_context)}", + key: "#{type}-#{get_handle_str(handle)}-#{get_api_context(other_context)}") + end + + # Passed as the block to Hash#delete, so it fires when a destroy names a + # handle the model never recorded. + def object_not_found(context, type, handle, sub_context = nil) + raise_internal_error(context, "#{type} #{get_handle_str(handle)} not found#{sub_context ? " in #{sub_context}" : ""}") + end + + # Reads one input argument of the call executing on this thread. Works at + # _exit too, since the entry payload is still on the call stack. + def find_param(context, name) + get_last_entry(context).params[name] + end + + # The whole handle -> object table for a type. + def find_objects(context, type) + get_process(context).instance_variable_get("@#{type}s") + end + + # `handle` may be the handle itself or the name of the param carrying it. + def find_object(context, type, handle) + handle = find_param(context, handle) if handle.kind_of? String + find_objects(context, type)[handle] + end + + # AllocationMap for living allocations on a Context + def memory_allocations(context, ctx_handle) + get_process(context).contexts[ctx_handle]&.memory_allocations + end + + # AllocationMap for freed allocations on a Context (for use_after_free detections) + def freed_memory_allocations(context, ctx_handle) + get_process(context).contexts[ctx_handle]&.freed_memory_allocations + end + + # Get a Context of a Process from handles + def context_from_handle(context, ctx_handle) + ctx_obj = get_process(context).contexts[ctx_handle] + return ctx_obj if ctx_obj + + report(:unknown_context, context, + "context #{get_handle_str(ctx_handle)} was never created, or was already destroyed; " \ + 'its allocations cannot be tracked', + key: "unknown-context-#{get_handle_str(ctx_handle)}") + nil + end + + # Yields [unit, op] for every copy op still pending in this process + def each_inflight_copy_op(context) + @deferred_units.each do |unit| + next unless unit.context['hostname'] == context['hostname'] && + unit.context['vpid'] == context['vpid'] + unit.ops[unit.cursor..].each do |op| + next unless op.kind == :copy + yield unit, op + end + end + end + + # True if a prior submission of this command list has not drained yet. + def command_list_in_flight?(context, handle) + @deferred_units.any? do |unit| + !unit.immediate && + unit.cmd_list_handle == handle && + unit.context['hostname'] == context['hostname'] && + unit.context['vpid'] == context['vpid'] && + !unit.done? + end + end + + # Decodes a raw descriptor blob from the trace into a typed FFI struct, or nil + # for a null descriptor. + def to_struct(memory, klass) + memory.size > 0 ? klass.new(FFI::MemoryPointer.from_string(memory)) : nil + end + + # Returns the Event for a handle, nil for a null or unknown one. + def event_by_handle(context, handle) + return nil if handle.nil? || handle == 0 + find_objects(context, 'event')[handle] + end + + # Signals an event, if the handle names one we track. + def signal_event(context, handle, by = nil) + ev = event_by_handle(context, handle) + ev&.signal(by) + ev + end + + # Reset's the given handle's event + def reset_event(context, handle) + event_by_handle(context, handle)&.reset + end + + # Records that the host observed an event's signaled state. + def observe_event(context, handle) + event_by_handle(context, handle)&.observe + end + + # A device-wide host synchronization means every signaled event was consumed. + def observe_all_signaled_events(context) + find_objects(context, 'event').each_value { |ev| ev.observe if ev.signaled } + end + + # True once every wait handle is signaled. Untracked handles count as + # satisfied, so we never invent a deadlock for one. + def waits_satisfied?(context, waits) + waits.all? { |h| ev = event_by_handle(context, h); ev.nil? || ev.signaled } + end + + # Runs the op the cursor points at, applying its deferred checks and signal. + def run_deferred_op(unit) + context = unit.context + op = unit.current_op + if op.kind == :copy + check_oob_copy(self, context, op.params) + #a pointer freed before this copy's turn to execute is a use-after-free + check_use_after_free(self, context, op.params) + end + #a memory-ranges barrier references memory freed before its turn is a UAF + check_uaf_ranges_barrier(self, context, op.params) if op.kind == :ranges_barrier + #a reset takes effect before this op signals its own completion event + reset_event(context, op.params[:reset_handle]) if op.kind == :reset + signaled = false + if op.signal + #the completion event must be unsignaled here: reuse without an intervening + #reset (or a concurrent double-signal) is a misuse + check_event_signal_reuse(self, context, op.signal, op.api || 'a command list append') + signal_event(context, op.signal, op.api) + unit.pending_signals.delete(op.signal) + signaled = true + end + unit.cursor += 1 + unit.blocked_on = [] + signaled + end + + # Advances every unit as far as its wait-events allow, sweeping until a whole + # pass makes no progress since one unit's signal can unblock another. + def pump_deferred + progress = true + while progress + progress = false + @deferred_units.each do |unit| + until unit.done? + op = unit.current_op + if waits_satisfied?(unit.context, op.waits) + run_deferred_op(unit) + progress = true + else + #park the unit on this op and record what it is blocked on so the + #deadlock detector can see the wait-for edges + unit.blocked_on = op.waits.reject { |h| + ev = event_by_handle(unit.context, h); ev.nil? || ev.signaled + } + break + end + end + end + @deferred_units.reject!(&:done?) + end + end + + # Registers a command list's ops as a deferred unit and pumps. + def run_deferred_list(context, ops, label, in_order: false, cmd_list_handle: nil, immediate: false) + @deferred_units << ZEModel::DeferredUnit.new(ops, context, label, in_order: in_order, + cmd_list_handle: cmd_list_handle, + immediate: immediate) + pump_deferred + end + + # Each submitted list becomes its own unit: lists in one submit are ordered by + # events, not by list order, so a cycle between two of them is a real deadlock. + def enqueue_deferred_execution(context, command_lists) + command_lists.each do |cl| + next unless cl + run_deferred_list(context, cl.ops.dup, "command_list (#{get_handle_str(cl.handle)})", + in_order: cl.in_order, cmd_list_handle: cl.handle) + end + end + + # The unit of an in-order immediate list that still has ops to run, if any. + def open_immediate_unit(context, handle) + @deferred_units.find do |unit| + unit.immediate && unit.cmd_list_handle == handle && + unit.context['hostname'] == context['hostname'] && + unit.context['vpid'] == context['vpid'] && + !unit.done? + end + end + + # Immediate lists execute each op as it is appended, but still go through the + # same machinery so they get the same checks. + def enqueue_immediate_op(context, op, handle = nil, in_order: false) + unit = in_order && handle ? open_immediate_unit(context, handle) : nil + if unit + unit.push_op(op) + pump_deferred + return + end + + label = handle ? "immediate command list (#{get_handle_str(handle)})" \ + : 'immediate command list' + run_deferred_list(context, [op], label, in_order: in_order, + cmd_list_handle: handle, immediate: true) + end + + # End-of-trace drain: reports deadlocks among whatever is still stuck, then + # forces each remaining op so its deferred checks run against the final state. + def flush_deferred + pump_deferred + return if @deferred_units.empty? + check_circular_deadlock(self, @deferred_units) + check_in_order_self_deadlock(self, @deferred_units) + until @deferred_units.empty? + unit = @deferred_units.first + #force the op the unit is stuck on: report its unsignaled waits, then run it + report_unsignaled_waits(self, unit.context, unit.current_op.waits) if unit.current_op + run_deferred_op(unit) unless unit.done? + @deferred_units.reject!(&:done?) + #a forced completion may unblock others cleanly + pump_deferred + end + end + + + # Checks for the issues only visible at end of trace + # Notify that all traces had been processed to: + # - perform final checks, that can only be performed after the entire + # traced got read (e.g., deadlocks, leaks, etc.) + # - export to csv if asked by users + def finalize() + flush_deferred + report_unfinished_calls + report_leaks if @report_leaks + @reporter.summary + @reporter.export_csv(@csv_export) if @csv_export + end + + # Any frame still on a thread's call stack is a call that never returned. + def report_unfinished_calls + @state.each { |hostname, node| + node.processes.each { |pid, process| + process.threads.each { |tid, thread| + thread.call_stack.each { |frame| + ctx = {'hostname' => hostname, 'vpid'=> pid, 'vtid' => tid, 'api' => frame.name} + report(:api_never_returned, ctx, 'call did not finish execution') + } + } + } + } + end + + # Objects still registered once the trace is exhausted were never destroyed. + LEAKABLE_OBJECT_TYPES = %w[context event_pool command_queue fence command_list + module module_build_log kernel].freeze + + # Reports what a container still held when it was destroyed, and forgets it. + def report_orphans(context, parent) + parent_label = "#{parent.class.typename} #{get_handle_str(parent.handle)}" + ZEModel::OWNED_TYPES.fetch(parent.class.typename, []).each { |type| + verb = type == 'memory_allocation' ? 'freed' : 'destroyed' + parent.each_child(type) { |child| + report(:object_outlives_owner, context, + "#{object_label(type, child)} was not #{verb} prior to #{parent_label} destruction" \ + "#{held_summary(child)}", + key: "outlives-#{type}-#{get_handle_str(child.handle)}") + mark_leak_reported(type, child) + } + } + end + + # Marks an object and everything below it as already reported, so the end of + # the trace does not name them a second time. + def mark_leak_reported(type, obj) + obj.leak_reported = true + ZEModel::OWNED_TYPES.fetch(type, []).each { |child_type| + obj.each_child(child_type) { |child| mark_leak_reported(child_type, child) } + } + end + + # True when this object is the outermost leaked object of its subtree, and so + # the one worth a report. + def leak_root?(process, type, obj) + return false if obj.leak_reported + + owner_type = ZEModel::OWNERSHIP[type] + return true unless owner_type && LEAKABLE_OBJECT_TYPES.include?(owner_type) + + owner = obj.owner + owner.nil? || !process.objects(owner_type).key?(owner.handle) + end + + def report_leaks + @state.each { |hostname, node| + node.processes.each { |pid, process| + ctx = {'hostname' => hostname, 'vpid'=> pid} + LEAKABLE_OBJECT_TYPES.each { |t| + process.objects(t).each_value { |obj| + print_leak_error(ctx, t, obj) if leak_root?(process, t, obj) + } + } + } + } + end + + + # Checks that zeInit or zeInitDrivers came before any other API call. Keyed by + # pid, and reported once per run. + def check_initialization(context) + if ZEModel::INIT_API_NAMES.include?(context['api']) + @init_called[context['vpid']] = true + end + + if !@init_called[context['vpid']] && !@printed_init_error + report(:init_not_called, context, "zeInit or zeInitDrivers wasn't called before #{context['api']}") + @printed_init_error = true + end + end + + + # Pushes the call frame, takes the thread-safety locks and runs the API's + # entry callback. Checks run here when the call itself might crash. + def on_entry(m,hostname, context,payload) + set_last_entry(self, context, payload) #sets the per-thread callstack of the APIs + @lock_shared_object_on_entry[m[1]].each { |l| + l.call(self, context, payload) + } + #modifies the state based on entry fields. Needed because some fields are easier to access it from the entry + l = $upon_entry[m[1]] + l.call(self,context,payload) if l + end + + def on_exit(m,hostname,context,payload) + #unlock the shared object if the api name matches the predefined in api_metadata.yaml + @unlock_shared_object_on_exit[m[1]].reverse_each { |l| + l.call(self, context, payload) + } + + #check if the return code indicates successful return from the API call + if validate_result(payload) + l = $on_successful_exit[m[1]] #This might be a problem for tracking erroneous exits. + l.call(self, context, payload) if l + else + l = $on_erroneous_exit[m[1]] + l.call(self, context, payload) if l + end + + check_last_entry(context) #When we return from _exit, we need to see what we saw in _entry for the current thread_id + reset_last_entry(context) #Reset the callstack for current thread_id + end + + # The main loop: returns the lambda called with each batch of decoded + # messages. + def consume = lambda { |iterator, _| + iterator.next_messages.each do |m| + next unless m.type == :BT_MESSAGE_TYPE_EVENT + e = m.event + #splits "lttng_ust_ze:zeMemAllocDevice_entry" into the API name (m[1]) + #and the phase (m[2]); anything else is ignored + m = e.name.match(/:(z.*)_(entry|exit)/) + device_property = e.name.match(/lttng_ust_ze_properties:command_queue_group/) + if m + hostname = e.stream.trace.get_environment_entry_value_by_name('hostname').value + context = e.get_common_context_field.value + #the event's own fields: input args at _entry, results at _exit + payload = e.payload_field.value + context['hostname'] = hostname + context['api'] = m[1] + #zeDriversInit or zeInit must be the first one to be called before any api calls + check_initialization(context) + print_deprecation_warning(context, m[1]) if @deprecated[m[1]] + + if m[2] == 'entry' + on_entry(m, hostname, context, payload) + elsif m[2] == 'exit' + on_exit(m, hostname, context, payload) + end + # Runs the blocked commands, if the wait-event(s) are satisfied + pump_deferred + elsif device_property + payload = e.payload_field.value + count = payload["pCount"] + device = payload["hDevice"] + blob = payload["pGroupProperties_vals"] + ptr = FFI::MemoryPointer.from_string(blob) + size = ZE::ZECommandQueueGroupProperties.size + groups = (0...count).map do |i| ZE::ZECommandQueueGroupProperties.new(ptr + i * size) end + + groups.each_with_index do |g, ordinal| + @device_properties[device][ordinal] = {} + @device_properties[device][ordinal]['flags'] = g[:flags] + @device_properties[device][ordinal]['numQueues'] = g[:numQueues] + end + + end + end + } + +end diff --git a/backends/ze/validator/ze_validator.in b/backends/ze/validator/ze_validator.in new file mode 100644 index 000000000..239b4627a --- /dev/null +++ b/backends/ze/validator/ze_validator.in @@ -0,0 +1,127 @@ +#!/usr/bin/env ruby +# coding: utf-8 +DATADIR = File.join("@prefix@", "share") +BINDIR = File.join("@prefix@", "bin") +$:.unshift(DATADIR) if File.directory?(DATADIR) +require 'optparse' +require 'babeltrace2' +require 'find' +require 'ze_library' +require 'set' +require 'ze/validator/errors' +require 'ze/validator/model' +require 'ze/validator/callbacks' +require 'ze/validator/state_object' + +# Don't complain about broken pipe +Signal.trap('SIGPIPE', 'SYSTEM_DEFAULT') + +$options = { live: false, csv_export: nil, report_leaks: true } + +DEFAULT_CSV_EXPORT = 'ze_validator.csv' + +parser = OptionParser.new do |opts| + opts.banner = 'Usage: ze_validator [OPTIONS] trace_directory...' + + opts.on('-h', '--help', 'Prints this help') do + puts opts + exit + end + + opts.on('--live', 'Enable live processing of the trace') do + $options[:live] = true + end + + opts.on('--[no-]report-leaks', + 'Report end-of-trace object and memory leaks (default: enabled)') do |v| + $options[:report_leaks] = v + end + + opts.on('--csv-export[=CSV-FILE]', + 'Write the per-error-type counts to CSV, one row per diagnostic ' \ + "(default: #{DEFAULT_CSV_EXPORT})") do |file| + $options[:csv_export] = file || DEFAULT_CSV_EXPORT + end +end +parser.parse! + +def build_and_run_graph( source_location, sink_object ) + # build graph and set up source + graph = BT2::BTGraph.new + + ctf_fs = BT2::BTPlugin.find('ctf').get_source_component_class_by_name('fs') + ctf_lttng_live = BT2::BTPlugin.find("ctf").get_source_component_class_by_name("lttng-live") + utils_muxer = BT2::BTPlugin.find('utils').get_filter_component_class_by_name('muxer') + + if !$options[:live] + missing = source_location.reject { |path| File.exist?(path) } + unless missing.empty? + warn "ze_validator: no such file or directory: #{missing.join(', ')}" + warn "Try 'ze_validator --help'." + exit 1 + end + + trace_locations = + Find.find(*source_location).reject do |path| + FileTest.directory?(path) + end.select do |path| + File.basename(path) == 'metadata' + end.collect do |path| + File.dirname(path) + end.select do |path| + qe = BT2::BTQueryExecutor.new(component_class: ctf_fs, object_name: 'babeltrace.support-info', + params: { 'input' => path, 'type' => 'directory' }) + qe.query.value['weight'] > 0.5 + end + else + trace_locations = source_location + end + if trace_locations.empty? + warn "ze_validator: no LTTng trace found in: #{Array(source_location).join(', ')}" + warn "Try 'ze_validator --help'." + exit 1 + end + + if !$options[:live] + comp_sources = trace_locations.each_with_index.collect { |trace_location, i| graph.add_component(ctf_fs, "trace_#{i}", params: {"inputs" => [ trace_location ] }) } + else + comp_sources = trace_locations.each_with_index.collect { |trace_location, i| graph.add_component(ctf_lttng_live, "trace_#{i}", params: {"inputs" => [ trace_location ], "session-not-found-action" => "end" }) } + end + + # Muxer + comp_muxer = graph.add_component(utils_muxer, 'mux') + + sink = graph.add_simple_sink('babeltrace_thapi', sink_object.consume) + + # Sources to muxer + comp_sources.flat_map(&:output_ports).each_with_index do |op, i| + ip = comp_muxer.input_port(i) + graph.connect_ports(op, ip) + end + + # Chain the rest + [comp_muxer, sink].flatten.each_cons(2) do |_out, _in| + op = _out.output_port(0) + ip = _in.input_port(0) + graph.connect_ports(op, ip) + end + + graph.run + sink_object.finalize() + +end + +# only executive this code if we launch this as the main +# script. if it's just included with "require" we just want access to the functions. +if __FILE__ == $0 + if ARGV.empty? + puts parser + exit 1 + end + + sink_obj = StateObject.new(csv_export: $options[:csv_export], + report_leaks: $options[:report_leaks]) + ARGV.uniq! + source_location = ARGV + build_and_run_graph(source_location, sink_obj) +end diff --git a/configure.ac b/configure.ac index 275796323..2946064c9 100644 --- a/configure.ac +++ b/configure.ac @@ -133,6 +133,16 @@ AX_RUBY_EXTENSION([nokogiri], [yes]) AX_RUBY_EXTENSION([babeltrace2], [yes]) AX_RUBY_EXTENSION([metababel >= 1.1.3], [yes]) +# ze_validator is opt-in +AC_ARG_ENABLE([ze-validator], + AS_HELP_STRING([--enable-ze-validator], + [Build and install ze_validator (needs the rbtree3 ruby gem)]), + [enable_ze_validator="$enableval"], + [enable_ze_validator="no"]) +AS_IF([test "x$enable_ze_validator" = xyes], + [AX_RUBY_EXTENSION([rbtree3 >= 1.1.0], [yes])]) +AM_CONDITIONAL([ZE_VALIDATOR], [test "x$enable_ze_validator" = xyes]) + AX_CXX_COMPILE_STDCXX([17], [noext], [mandatory]) # Checks for header files. @@ -181,6 +191,8 @@ AC_CONFIG_FILES([utils/test_wrapper_thapi_text_pretty.sh], [chmod +x utils/test_ AC_CONFIG_FILES([backends/opencl/tracer_opencl.sh], [chmod +x backends/opencl/tracer_opencl.sh]) AC_CONFIG_FILES([backends/opencl/extract_enqueues], [chmod +x backends/opencl/extract_enqueues]) AC_CONFIG_FILES([backends/ze/tracer_ze.sh], [chmod +x backends/ze/tracer_ze.sh]) +AC_CONFIG_FILES([backends/ze/validator/ze_validator], + [chmod +x backends/ze/validator/ze_validator]) AC_CONFIG_FILES([backends/cuda/tracer_cuda.sh], [chmod +x backends/cuda/tracer_cuda.sh]) AC_CONFIG_FILES([backends/omp/tracer_omp.sh], [chmod +x backends/omp/tracer_omp.sh]) AC_CONFIG_FILES([backends/hip/tracer_hip.sh], [chmod +x backends/hip/tracer_hip.sh])