diff --git a/projects/rocr-runtime/libhsakmt/CMakeLists.txt b/projects/rocr-runtime/libhsakmt/CMakeLists.txt index 55bd945fd2..115b2926f3 100644 --- a/projects/rocr-runtime/libhsakmt/CMakeLists.txt +++ b/projects/rocr-runtime/libhsakmt/CMakeLists.txt @@ -37,6 +37,7 @@ else() set(HSAKMT_STATIC_DRM_TARGET "${HSAKMT_TARGET}-staticdrm") project ( ${HSAKMT_TARGET} VERSION 1.9.0) + set(BUILD_SHARED_LIBS ON FORCE) # Optionally, build HSAKMT with ccache. set(ROCM_CCACHE_BUILD OFF CACHE BOOL "Set to ON for a ccache enabled build") @@ -79,7 +80,7 @@ else() ## Compiler flags if (UNIX) - set (HSAKMT_C_FLAGS -fPIC -W -Wall -Wextra -Wno-unused-parameter -Wformat-security -Wswitch-default -Wundef -Wshadow -Wpointer-arith -Wbad-function-cast -Wcast-qual -Wstrict-prototypes -Wmissing-prototypes -Wmissing-declarations -Wredundant-decls -Wunreachable-code -std=gnu99 -fvisibility=hidden) + set (HSAKMT_C_FLAGS -fPIC -W -Wall -Wextra -Wno-unused-parameter -Wformat-security -Wswitch-default -Wundef -Wshadow -Wpointer-arith -Wbad-function-cast -Wcast-qual -Wstrict-prototypes -Wmissing-prototypes -Wmissing-declarations -Wredundant-decls -Wunreachable-code -std=gnu99 -fvisibility=default) if ( CMAKE_COMPILER_IS_GNUCC ) set ( HSAKMT_C_FLAGS "${HSAKMT_C_FLAGS}" -Wlogical-op) @@ -170,8 +171,6 @@ else() "src/libhsakmt.c" "src/memory.c" "src/openclose.c" - "src/perfctr.c" - "src/pmc_table.c" "src/queues.c" "src/time.c" "src/topology.c" @@ -182,10 +181,15 @@ else() "src/pc_sampling.c" "src/ais.c" "src/kfdcontext.c") + if (CMAKE_SYSTEM_NAME MATCHES "Linux") + list(APPEND HSAKMT_SRC "src/perfctr.c" "src/pmc_table.c") + elseif (CMAKE_SYSTEM_NAME MATCHES "FreeBSD") + list(APPEND HSAKMT_SRC "src/freebsd/perfctr.c") + endif() endif() ## Declare the library target name - add_library (${HSAKMT_TARGET} STATIC "") + add_library (${HSAKMT_TARGET} SHARED "") ## Add sources target_sources ( ${HSAKMT_TARGET} PRIVATE ${HSAKMT_SRC} ) @@ -205,6 +209,12 @@ else() ${CMAKE_CURRENT_SOURCE_DIR}/../runtime/hsa-runtime) endif() + if (CMAKE_SYSTEM_NAME MATCHES "FreeBSD") + target_include_directories( ${HSAKMT_TARGET} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/include/hsakmt/freebsd ) + elseif (CMAKE_SYSTEM_NAME MATCHES "Linux") + target_include_directories( ${HSAKMT_TARGET} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/include/hsakmt/linux ) + endif() + ## Add headers. The public headers need to point at their location in both build and install ## directory layouts. This declaration allows publishing library use data to downstream clients. target_include_directories( ${HSAKMT_TARGET} @@ -392,9 +402,15 @@ else() if ( NOT BUILD_SHARED_LIBS) ## Create separate target file for static builds ## In static builds, libdrm and libdrm_amdgpu need to be linked statically - add_library (${HSAKMT_STATIC_DRM_TARGET} STATIC "") + add_library (${HSAKMT_STATIC_DRM_TARGET} SHARED "") target_sources (${HSAKMT_STATIC_DRM_TARGET} PRIVATE ${HSAKMT_SRC}) + if (CMAKE_SYSTEM_NAME MATCHES "FreeBSD") + target_include_directories( ${HSAKMT_STATIC_DRM_TARGET} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/include/hsakmt/freebsd ) + elseif (CMAKE_SYSTEM_NAME MATCHES "Linux") + target_include_directories( ${HSAKMT_STATIC_DRM_TARGET} PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}/include/hsakmt/linux ) + endif() + target_include_directories( ${HSAKMT_STATIC_DRM_TARGET} PUBLIC $ diff --git a/projects/rocr-runtime/libhsakmt/include/hsakmt/freebsd/kfd_ioctl.h b/projects/rocr-runtime/libhsakmt/include/hsakmt/freebsd/kfd_ioctl.h new file mode 100644 index 0000000000..bd61243dc4 --- /dev/null +++ b/projects/rocr-runtime/libhsakmt/include/hsakmt/freebsd/kfd_ioctl.h @@ -0,0 +1,1890 @@ +// I have no idea what Liscense to put here? +// Sourojeet Adhikari +// FreeBSD Foundation + +#ifndef KFD_IOCTL_H_INCLUDED +#define KFD_IOCTL_H_INCLUDED + +#include +#include +#include +#include + +/* + * - 1.1 Initial Version +*/ + +#define KFD_IOCTL_MAJOR_VERSION 1 +#define KFD_IOCTL_MINOR_VERSION 1 + +// another small hard coded value +#define _IOC_SIZESHIFT 16 + +// small hack, just remaps the function +#ifndef MADV_DONTFORK +#define MADV_DONTFORK 9999 +#endif + +static inline int upstream_madvice_wrapper(void* addr, size_t len, int advice) +{ + // spidey sense says that smth with page alignment will bite me + if(advice == MADV_DONTFORK) + { + return minherit(addr, len, INHERIT_NONE); + } + /* + Do not use this code unless the above code is broken, I did not think this code + through too well + + if(advice == MADV_DONTFORK) + { + long pagesize = sysconf(_SC_PAGESIZE); + unitptr_t start = (uintptr_t)addr; + uintptr_t end = start + length; + + uintptr_t aligned_start = start & ~(pagesize - 1); + uintptr_t aligned_end = (end + pagesize - 1) & ~(pagesize - 1); + size_t aligned_length = aligned_end - aligned_start; + + return minherit((void *)aligned_start, aligned_length, INHERIT_NONE); + } + + */ + return madvise(addr, len, advice); +} + +#define madvise(addr, len, advice) upstream_madvise_wrapper(addr, len, advice) + +// now theoretically MAP_NORESERVE should do something +// however, we need to consider that Linux & FreeBSD handle stuff +// "differently", Linux is fairly strict, and complains if u over-alloc +// freeBSD follows lazy alloc +// So we can just map this to a "no-op" thingy +#ifndef MAP_NORESERVE +#define MAP_NORESERVE 0 +#endif + +// In the man pages, it literally says (paraphrased) this is optional +// and does not always exist +// Thus implying that mapping to a no-op should be fine. +// I think FreeBSD also does thus auto matically? +#ifndef MADV_HUGEPAGE +#define MADV_HUGEPAGE 0 +#endif + + +struct kfd_ioctl_get_version_args { + __uint32_t major_version; /* from KFD */ + __uint32_t minor_version; /* from KFD */ +}; + +/* For kfd_ioctl_create_queue_args.queue_type. */ +#define KFD_IOC_QUEUE_TYPE_COMPUTE 0x0 +#define KFD_IOC_QUEUE_TYPE_SDMA 0x1 +#define KFD_IOC_QUEUE_TYPE_COMPUTE_AQL 0x2 +#define KFD_IOC_QUEUE_TYPE_SDMA_XGMI 0x3 +#define KFD_IOC_QUEUE_TYPE_SDMA_BY_ENG_ID 0x4 + +#define KFD_MAX_QUEUE_PERCENTAGE 100 +#define KFD_MAX_QUEUE_PRIORITY 15 + +struct kfd_ioctl_create_queue_args { + __uint64_t ring_base_address; /* to KFD */ + __uint64_t write_pointer_address; /* to KFD */ + __uint64_t read_pointer_address; /* to KFD */ + __uint64_t doorbell_offset; /* from KFD */ + + __uint32_t ring_size; /* to KFD */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t queue_type; /* to KFD */ + __uint32_t queue_percentage; /* to KFD */ + __uint32_t queue_priority; /* to KFD */ + __uint32_t queue_id; /* from KFD */ + + __uint64_t eop_buffer_address; /* to KFD */ + __uint64_t eop_buffer_size; /* to KFD */ + __uint64_t ctx_save_restore_address; /* to KFD */ + __uint32_t ctx_save_restore_size; /* to KFD */ + __uint32_t ctl_stack_size; /* to KFD */ + __uint32_t sdma_engine_id; /* to KFD */ + __uint32_t metadata_ring_size; /* to KFD */ +}; + +struct kfd_ioctl_destroy_queue_args { + __uint32_t queue_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_update_queue_args { + __uint64_t ring_base_address; /* to KFD */ + + __uint32_t queue_id; /* to KFD */ + __uint32_t ring_size; /* to KFD */ + __uint32_t queue_percentage; /* to KFD */ + __uint32_t queue_priority; /* to KFD */ +}; + +struct kfd_ioctl_set_cu_mask_args { + __uint32_t queue_id; /* to KFD */ + __uint32_t num_cu_mask; /* to KFD */ + __uint64_t cu_mask_ptr; /* to KFD */ +}; + +struct kfd_ioctl_get_queue_wave_state_args { + __uint64_t ctl_stack_address; /* to KFD */ + __uint32_t ctl_stack_used_size; /* from KFD */ + __uint32_t save_area_used_size; /* from KFD */ + __uint32_t queue_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_queue_snapshot_entry { + __uint64_t exception_status; + __uint64_t ring_base_address; + __uint64_t write_pointer_address; + __uint64_t read_pointer_address; + __uint64_t ctx_save_restore_address; + __uint32_t queue_id; + __uint32_t gpu_id; + __uint32_t ring_size; + __uint32_t queue_type; + __uint32_t ctx_save_restore_area_size; + __uint32_t reserved; +}; + +struct kfd_dbg_device_info_entry { + __uint64_t exception_status; + __uint64_t lds_base; + __uint64_t lds_limit; + __uint64_t scratch_base; + __uint64_t scratch_limit; + __uint64_t gpuvm_base; + __uint64_t gpuvm_limit; + __uint32_t gpu_id; + __uint32_t location_id; + __uint32_t vendor_id; + __uint32_t device_id; + __uint32_t revision_id; + __uint32_t subsystem_vendor_id; + __uint32_t subsystem_device_id; + __uint32_t fw_version; + __uint32_t gfx_target_version; + __uint32_t simd_count; + __uint32_t max_waves_per_simd; + __uint32_t array_count; + __uint32_t simd_arrays_per_engine; + __uint32_t num_xcc; + __uint32_t capability; + __uint32_t debug_prop; +}; + +/* For kfd_ioctl_set_memory_policy_args.default_policy and alternate_policy */ +#define KFD_IOC_CACHE_POLICY_COHERENT 0 +#define KFD_IOC_CACHE_POLICY_NONCOHERENT 1 + +struct kfd_ioctl_set_memory_policy_args { + __uint64_t alternate_aperture_base; /* to KFD */ + __uint64_t alternate_aperture_size; /* to KFD */ + + __uint32_t gpu_id; /* to KFD */ + __uint32_t default_policy; /* to KFD */ + __uint32_t alternate_policy; /* to KFD */ + __uint32_t misc_process_flag; /* to KFD */ +}; + +/* + * All counters are monotonic. They are used for profiling of compute jobs. + * The profiling is done by userspace. + * + * In case of GPU reset, the counter should not be affected. + */ + +struct kfd_ioctl_get_clock_counters_args { + __uint64_t gpu_clock_counter; /* from KFD */ + __uint64_t cpu_clock_counter; /* from KFD */ + __uint64_t system_clock_counter; /* from KFD */ + __uint64_t system_clock_freq; /* from KFD */ + + __uint32_t gpu_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_process_device_apertures { + __uint64_t lds_base; /* from KFD */ + __uint64_t lds_limit; /* from KFD */ + __uint64_t scratch_base; /* from KFD */ + __uint64_t scratch_limit; /* from KFD */ + __uint64_t gpuvm_base; /* from KFD */ + __uint64_t gpuvm_limit; /* from KFD */ + __uint32_t gpu_id; /* from KFD */ + __uint32_t pad; +}; + +/* + * AMDKFD_IOC_GET_PROCESS_APERTURES is deprecated. Use + * AMDKFD_IOC_GET_PROCESS_APERTURES_NEW instead, which supports an + * unlimited number of GPUs. + */ +#define NUM_OF_SUPPORTED_GPUS 7 +struct kfd_ioctl_get_process_apertures_args { + struct kfd_process_device_apertures + process_apertures[NUM_OF_SUPPORTED_GPUS];/* from KFD */ + + /* from KFD, should be in the range [1 - NUM_OF_SUPPORTED_GPUS] */ + __uint32_t num_of_nodes; + __uint32_t pad; +}; + +struct kfd_ioctl_get_process_apertures_new_args { + /* User allocated. Pointer to struct kfd_process_device_apertures + * filled in by Kernel + */ + __uint64_t kfd_process_device_apertures_ptr; + /* to KFD - indicates amount of memory present in + * kfd_process_device_apertures_ptr + * from KFD - Number of entries filled by KFD. + */ + __uint32_t num_of_nodes; + __uint32_t pad; +}; + +#define MAX_ALLOWED_NUM_POINTS 100 +#define MAX_ALLOWED_AW_BUFF_SIZE 4096 +#define MAX_ALLOWED_WAC_BUFF_SIZE 128 + +struct kfd_ioctl_dbg_register_args { + __uint32_t gpu_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_dbg_unregister_args { + __uint32_t gpu_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_dbg_address_watch_args { + __uint64_t content_ptr; /* a pointer to the actual content */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t buf_size_in_bytes; /*including gpu_id and buf_size */ +}; + +struct kfd_ioctl_dbg_wave_control_args { + __uint64_t content_ptr; /* a pointer to the actual content */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t buf_size_in_bytes; /*including gpu_id and buf_size */ +}; +#define KFD_DBG_EV_FLAG_CLEAR_STATUS 1 + +/* queue states for suspend/resume */ +#define KFD_DBG_QUEUE_ERROR_BIT 30 +#define KFD_DBG_QUEUE_INVALID_BIT 31 +#define KFD_DBG_QUEUE_ERROR_MASK (1 << KFD_DBG_QUEUE_ERROR_BIT) +#define KFD_DBG_QUEUE_INVALID_MASK (1 << KFD_DBG_QUEUE_INVALID_BIT) + +#define KFD_INVALID_GPUID 0xffffffff +#define KFD_INVALID_QUEUEID 0xffffffff +#define KFD_INVALID_FD 0xffffffff + +enum kfd_dbg_trap_override_mode { + KFD_DBG_TRAP_OVERRIDE_OR = 0, + KFD_DBG_TRAP_OVERRIDE_REPLACE = 1 +}; +enum kfd_dbg_trap_mask { + KFD_DBG_TRAP_MASK_FP_INVALID = 1, + KFD_DBG_TRAP_MASK_FP_INPUT_DENORMAL = 2, + KFD_DBG_TRAP_MASK_FP_DIVIDE_BY_ZERO = 4, + KFD_DBG_TRAP_MASK_FP_OVERFLOW = 8, + KFD_DBG_TRAP_MASK_FP_UNDERFLOW = 16, + KFD_DBG_TRAP_MASK_FP_INEXACT = 32, + KFD_DBG_TRAP_MASK_INT_DIVIDE_BY_ZERO = 64, + KFD_DBG_TRAP_MASK_DBG_ADDRESS_WATCH = 128, + KFD_DBG_TRAP_MASK_DBG_MEMORY_VIOLATION = 256, + KFD_DBG_TRAP_MASK_TRAP_ON_WAVE_START = (1 << 30), + KFD_DBG_TRAP_MASK_TRAP_ON_WAVE_END = (1 << 31) +}; + +/* Wave launch modes */ +enum kfd_dbg_trap_wave_launch_mode { + KFD_DBG_TRAP_WAVE_LAUNCH_MODE_NORMAL = 0, + KFD_DBG_TRAP_WAVE_LAUNCH_MODE_HALT = 1, + KFD_DBG_TRAP_WAVE_LAUNCH_MODE_DEBUG = 3 +}; + +/* Address watch modes */ +enum kfd_dbg_trap_address_watch_mode { + KFD_DBG_TRAP_ADDRESS_WATCH_MODE_READ = 0, + KFD_DBG_TRAP_ADDRESS_WATCH_MODE_NONREAD = 1, + KFD_DBG_TRAP_ADDRESS_WATCH_MODE_ATOMIC = 2, + KFD_DBG_TRAP_ADDRESS_WATCH_MODE_ALL = 3 +}; + +/* Additional wave settings */ +enum kfd_dbg_trap_flags { + KFD_DBG_TRAP_FLAG_SINGLE_MEM_OP = 1, +}; + +enum kfd_dbg_trap_exception_code { + EC_NONE = 0, + /* per queue */ + EC_QUEUE_WAVE_ABORT = 1, + EC_QUEUE_WAVE_TRAP = 2, + EC_QUEUE_WAVE_MATH_ERROR = 3, + EC_QUEUE_WAVE_ILLEGAL_INSTRUCTION = 4, + EC_QUEUE_WAVE_MEMORY_VIOLATION = 5, + EC_QUEUE_WAVE_APERTURE_VIOLATION = 6, + EC_QUEUE_PACKET_DISPATCH_DIM_INVALID = 16, + EC_QUEUE_PACKET_DISPATCH_GROUP_SEGMENT_SIZE_INVALID = 17, + EC_QUEUE_PACKET_DISPATCH_CODE_INVALID = 18, + EC_QUEUE_PACKET_RESERVED = 19, + EC_QUEUE_PACKET_UNSUPPORTED = 20, + EC_QUEUE_PACKET_DISPATCH_WORK_GROUP_SIZE_INVALID = 21, + EC_QUEUE_PACKET_DISPATCH_REGISTER_INVALID = 22, + EC_QUEUE_PACKET_VENDOR_UNSUPPORTED = 23, + EC_QUEUE_PREEMPTION_ERROR = 30, + EC_QUEUE_NEW = 31, + /* per device */ + EC_DEVICE_QUEUE_DELETE = 32, + EC_DEVICE_MEMORY_VIOLATION = 33, + EC_DEVICE_RAS_ERROR = 34, + EC_DEVICE_FATAL_HALT = 35, + EC_DEVICE_NEW = 36, + /* per process */ + EC_PROCESS_RUNTIME = 48, + EC_PROCESS_DEVICE_REMOVE = 49, + EC_MAX +}; + +/* Mask generated by ecode defined in enum above. */ +#define KFD_EC_MASK(ecode) (1ULL << (ecode - 1)) + +/* Masks for exception code type checks below. */ +#define KFD_EC_MASK_QUEUE (KFD_EC_MASK(EC_QUEUE_WAVE_ABORT) | \ + KFD_EC_MASK(EC_QUEUE_WAVE_TRAP) | \ + KFD_EC_MASK(EC_QUEUE_WAVE_MATH_ERROR) | \ + KFD_EC_MASK(EC_QUEUE_WAVE_ILLEGAL_INSTRUCTION) | \ + KFD_EC_MASK(EC_QUEUE_WAVE_MEMORY_VIOLATION) | \ + KFD_EC_MASK(EC_QUEUE_WAVE_APERTURE_VIOLATION) | \ + KFD_EC_MASK(EC_QUEUE_PACKET_DISPATCH_DIM_INVALID) | \ + KFD_EC_MASK(EC_QUEUE_PACKET_DISPATCH_GROUP_SEGMENT_SIZE_INVALID) | \ + KFD_EC_MASK(EC_QUEUE_PACKET_DISPATCH_CODE_INVALID) | \ + KFD_EC_MASK(EC_QUEUE_PACKET_UNSUPPORTED) | \ + KFD_EC_MASK(EC_QUEUE_PACKET_DISPATCH_WORK_GROUP_SIZE_INVALID) | \ + KFD_EC_MASK(EC_QUEUE_PACKET_DISPATCH_REGISTER_INVALID) | \ + KFD_EC_MASK(EC_QUEUE_PACKET_VENDOR_UNSUPPORTED) | \ + KFD_EC_MASK(EC_QUEUE_PREEMPTION_ERROR) | \ + KFD_EC_MASK(EC_QUEUE_NEW)) +#define KFD_EC_MASK_DEVICE (KFD_EC_MASK(EC_DEVICE_QUEUE_DELETE) | \ + KFD_EC_MASK(EC_DEVICE_RAS_ERROR) | \ + KFD_EC_MASK(EC_DEVICE_FATAL_HALT) | \ + KFD_EC_MASK(EC_DEVICE_MEMORY_VIOLATION) | \ + KFD_EC_MASK(EC_DEVICE_NEW)) +#define KFD_EC_MASK_PROCESS (KFD_EC_MASK(EC_PROCESS_RUNTIME) | \ + KFD_EC_MASK(EC_PROCESS_DEVICE_REMOVE)) + +/* Checks for exception code types for KFD search. */ +#define KFD_DBG_EC_TYPE_IS_QUEUE(ecode) \ + (!!(KFD_EC_MASK(ecode) & KFD_EC_MASK_QUEUE)) +#define KFD_DBG_EC_TYPE_IS_DEVICE(ecode) \ + (!!(KFD_EC_MASK(ecode) & KFD_EC_MASK_DEVICE)) +#define KFD_DBG_EC_TYPE_IS_PROCESS(ecode) \ + (!!(KFD_EC_MASK(ecode) & KFD_EC_MASK_PROCESS)) + +/* Misc. per process flags */ +#define KFD_PROC_FLAG_MFMA_HIGH_PRECISION (1 << 0) + +enum kfd_dbg_runtime_state { + DEBUG_RUNTIME_STATE_DISABLED = 0, + DEBUG_RUNTIME_STATE_ENABLED = 1, + DEBUG_RUNTIME_STATE_ENABLED_BUSY = 2, + DEBUG_RUNTIME_STATE_ENABLED_ERROR = 3 +}; + +struct kfd_runtime_info { + __uint64_t r_debug; + __uint32_t runtime_state; + __uint32_t ttmp_setup; +}; + +/* Enable modes for runtime enable */ +#define KFD_RUNTIME_ENABLE_MODE_ENABLE_MASK 1 +#define KFD_RUNTIME_ENABLE_MODE_TTMP_SAVE_MASK 2 +#define KFD_RUNTIME_ENABLE_CAPS_SUPPORTS_CORE_DUMP_MASK 0x80000000 + +/** + * kfd_ioctl_runtime_enable_args - Arguments for runtime enable + * + * Coordinates debug exception signalling and debug device enablement with runtime. + * + * @r_debug - pointer to user struct for sharing information between ROCr and the debuggger + * @mode_mask - mask to set mode + * KFD_RUNTIME_ENABLE_MODE_ENABLE_MASK - enable runtime for debugging, otherwise disable + * KFD_RUNTIME_ENABLE_MODE_TTMP_SAVE_MASK - enable trap temporary setup (ignore on disable) + * + * Return - 0 on SUCCESS. + * - EBUSY if runtime enable call already pending. + * - EEXIST if user queues already active prior to call. + * If process is debug enabled, runtime enable will enable debug devices and + * wait for debugger process to send runtime exception EC_PROCESS_RUNTIME + * to unblock - see kfd_ioctl_dbg_trap_args. + * + */ +struct kfd_ioctl_runtime_enable_args { + __uint64_t r_debug; + __uint32_t mode_mask; + __uint32_t capabilities_mask; +}; + +/* Context save area header information */ +struct kfd_context_save_area_header { + struct { + __uint32_t control_stack_offset; + __uint32_t control_stack_size; + __uint32_t wave_state_offset; + __uint32_t wave_state_size; + } wave_state; + __uint32_t debug_offset; + __uint32_t debug_size; + __uint64_t err_payload_addr; + __uint32_t err_event_id; + __uint32_t reserved1; +}; + +/* + * Debug operations + * + * For specifics on usage and return values, see documentation per operation + * below. Otherwise, generic error returns apply: + * - ESRCH if the process to debug does not exist. + * + * - EINVAL (with KFD_IOC_DBG_TRAP_ENABLE exempt) if operation + * KFD_IOC_DBG_TRAP_ENABLE has not succeeded prior. + * Also returns this error if GPU hardware scheduling is not supported. + * + * - EPERM (with KFD_IOC_DBG_TRAP_DISABLE exempt) if target process is not + * PTRACE_ATTACHED. KFD_IOC_DBG_TRAP_DISABLE is exempt to allow + * clean up of debug mode as long as process is debug enabled. + * + * - EACCES if any DBG_HW_OP (debug hardware operation) is requested when + * AMDKFD_IOC_RUNTIME_ENABLE has not succeeded prior. + * + * - ENODEV if any GPU does not support debugging on a DBG_HW_OP call. + * + * - Other errors may be returned when a DBG_HW_OP occurs while the GPU + * is in a fatal state. + * + */ +enum kfd_dbg_trap_operations { + KFD_IOC_DBG_TRAP_ENABLE = 0, + KFD_IOC_DBG_TRAP_DISABLE = 1, + KFD_IOC_DBG_TRAP_SEND_RUNTIME_EVENT = 2, + KFD_IOC_DBG_TRAP_SET_EXCEPTIONS_ENABLED = 3, + KFD_IOC_DBG_TRAP_SET_WAVE_LAUNCH_OVERRIDE = 4, /* DBG_HW_OP */ + KFD_IOC_DBG_TRAP_SET_WAVE_LAUNCH_MODE = 5, /* DBG_HW_OP */ + KFD_IOC_DBG_TRAP_SUSPEND_QUEUES = 6, /* DBG_HW_OP */ + KFD_IOC_DBG_TRAP_RESUME_QUEUES = 7, /* DBG_HW_OP */ + KFD_IOC_DBG_TRAP_SET_NODE_ADDRESS_WATCH = 8, /* DBG_HW_OP */ + KFD_IOC_DBG_TRAP_CLEAR_NODE_ADDRESS_WATCH = 9, /* DBG_HW_OP */ + KFD_IOC_DBG_TRAP_SET_FLAGS = 10, + KFD_IOC_DBG_TRAP_QUERY_DEBUG_EVENT = 11, + KFD_IOC_DBG_TRAP_QUERY_EXCEPTION_INFO = 12, + KFD_IOC_DBG_TRAP_GET_QUEUE_SNAPSHOT = 13, + KFD_IOC_DBG_TRAP_GET_DEVICE_SNAPSHOT = 14 +}; + +/** + * kfd_ioctl_dbg_trap_enable_args + * + * Arguments for KFD_IOC_DBG_TRAP_ENABLE. + * + * Enables debug session for target process. Call @op KFD_IOC_DBG_TRAP_DISABLE in + * kfd_ioctl_dbg_trap_args to disable debug session. + * + * @exception_mask (IN) - exceptions to raise to the debugger + * @rinfo_ptr (IN) - pointer to runtime info buffer (see kfd_runtime_info) + * @rinfo_size (IN/OUT) - size of runtime info buffer in bytes + * @dbg_fd (IN) - fd the KFD will nofify the debugger with of raised + * exceptions set in exception_mask. + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * Copies KFD saved kfd_runtime_info to @rinfo_ptr on enable. + * Size of kfd_runtime saved by the KFD returned to @rinfo_size. + * - EBADF if KFD cannot get a reference to dbg_fd. + * - EFAULT if KFD cannot copy runtime info to rinfo_ptr. + * - EINVAL if target process is already debug enabled. + * + */ +struct kfd_ioctl_dbg_trap_enable_args { + __uint64_t exception_mask; + __uint64_t rinfo_ptr; + __uint32_t rinfo_size; + __uint32_t dbg_fd; +}; + +/** + * kfd_ioctl_dbg_trap_send_runtime_event_args + * + * + * Arguments for KFD_IOC_DBG_TRAP_SEND_RUNTIME_EVENT. + * Raises exceptions to runtime. + * + * @exception_mask (IN) - exceptions to raise to runtime + * @gpu_id (IN) - target device id + * @queue_id (IN) - target queue id + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * - ENODEV if gpu_id not found. + * If exception_mask contains EC_PROCESS_RUNTIME, unblocks pending + * AMDKFD_IOC_RUNTIME_ENABLE call - see kfd_ioctl_runtime_enable_args. + * All other exceptions are raised to runtime through err_payload_addr. + * See kfd_context_save_area_header. + */ +struct kfd_ioctl_dbg_trap_send_runtime_event_args { + __uint64_t exception_mask; + __uint32_t gpu_id; + __uint32_t queue_id; +}; + +/** + * kfd_ioctl_dbg_trap_set_exceptions_enabled_args + * + * Arguments for KFD_IOC_SET_EXCEPTIONS_ENABLED + * Set new exceptions to be raised to the debugger. + * + * @exception_mask (IN) - new exceptions to raise the debugger + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + */ +struct kfd_ioctl_dbg_trap_set_exceptions_enabled_args { + __uint64_t exception_mask; +}; + +/** + * kfd_ioctl_dbg_trap_set_wave_launch_override_args + * + * Arguments for KFD_IOC_DBG_TRAP_SET_WAVE_LAUNCH_OVERRIDE + * Enable HW exceptions to raise trap. + * + * @override_mode (IN) - see kfd_dbg_trap_override_mode + * @enable_mask (IN/OUT) - reference kfd_dbg_trap_mask. + * IN is the override modes requested to be enabled. + * OUT is referenced in Return below. + * @support_request_mask (IN/OUT) - reference kfd_dbg_trap_mask. + * IN is the override modes requested for support check. + * OUT is referenced in Return below. + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * Previous enablement is returned in @enable_mask. + * Actual override support is returned in @support_request_mask. + * - EINVAL if override mode is not supported. + * - EACCES if trap support requested is not actually supported. + * i.e. enable_mask (IN) is not a subset of support_request_mask (OUT). + * Otherwise it is considered a generic error (see kfd_dbg_trap_operations). + */ +struct kfd_ioctl_dbg_trap_set_wave_launch_override_args { + __uint32_t override_mode; + __uint32_t enable_mask; + __uint32_t support_request_mask; + __uint32_t pad; +}; + +/** + * kfd_ioctl_dbg_trap_set_wave_launch_mode_args + * + * Arguments for KFD_IOC_DBG_TRAP_SET_WAVE_LAUNCH_MODE + * Set wave launch mode. + * + * @mode (IN) - see kfd_dbg_trap_wave_launch_mode + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + */ +struct kfd_ioctl_dbg_trap_set_wave_launch_mode_args { + __uint32_t launch_mode; + __uint32_t pad; +}; + +/** + * kfd_ioctl_dbg_trap_suspend_queues_ags + * + * Arguments for KFD_IOC_DBG_TRAP_SUSPEND_QUEUES + * Suspend queues. + * + * @exception_mask (IN) - raised exceptions to clear + * @queue_array_ptr (IN) - pointer to array of queue ids (u32 per queue id) + * to suspend + * @num_queues (IN) - number of queues to suspend in @queue_array_ptr + * @grace_period (IN) - wave time allowance before preemption + * per 1K GPU clock cycle unit + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Destruction of a suspended queue is blocked until the queue is + * resumed. This allows the debugger to access queue information and + * the its context save area without running into a race condition on + * queue destruction. + * Automatically copies per queue context save area header information + * into the save area base + * (see kfd_queue_snapshot_entry and kfd_context_save_area_header). + * + * Return - Number of queues suspended on SUCCESS. + * . KFD_DBG_QUEUE_ERROR_MASK and KFD_DBG_QUEUE_INVALID_MASK masked + * for each queue id in @queue_array_ptr array reports unsuccessful + * suspend reason. + * KFD_DBG_QUEUE_ERROR_MASK = HW failure. + * KFD_DBG_QUEUE_INVALID_MASK = queue does not exist, is new or + * is being destroyed. + */ +struct kfd_ioctl_dbg_trap_suspend_queues_args { + __uint64_t exception_mask; + __uint64_t queue_array_ptr; + __uint32_t num_queues; + __uint32_t grace_period; +}; + +/** + * kfd_ioctl_dbg_trap_resume_queues_args + * + * Arguments for KFD_IOC_DBG_TRAP_RESUME_QUEUES + * Resume queues. + * + * @queue_array_ptr (IN) - pointer to array of queue ids (u32 per queue id) + * to resume + * @num_queues (IN) - number of queues to resume in @queue_array_ptr + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - Number of queues resumed on SUCCESS. + * KFD_DBG_QUEUE_ERROR_MASK and KFD_DBG_QUEUE_INVALID_MASK mask + * for each queue id in @queue_array_ptr array reports unsuccessful + * resume reason. + * KFD_DBG_QUEUE_ERROR_MASK = HW failure. + * KFD_DBG_QUEUE_INVALID_MASK = queue does not exist. + */ +struct kfd_ioctl_dbg_trap_resume_queues_args { + __uint64_t queue_array_ptr; + __uint32_t num_queues; + __uint32_t pad; +}; + +/** + * kfd_ioctl_dbg_trap_set_node_address_watch_args + * + * Arguments for KFD_IOC_DBG_TRAP_SET_NODE_ADDRESS_WATCH + * Sets address watch for device. + * + * @address (IN) - watch address to set + * @mode (IN) - see kfd_dbg_trap_address_watch_mode + * @mask (IN) - watch address mask + * @gpu_id (IN) - target gpu to set watch point + * @id (OUT) - watch id allocated + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * Allocated watch ID returned to @id. + * - ENODEV if gpu_id not found. + * - ENOMEM if watch IDs can be allocated + */ +struct kfd_ioctl_dbg_trap_set_node_address_watch_args { + __uint64_t address; + __uint32_t mode; + __uint32_t mask; + __uint32_t gpu_id; + __uint32_t id; +}; + +/** + * kfd_ioctl_dbg_trap_clear_node_address_watch_args + * + * Arguments for KFD_IOC_DBG_TRAP_CLEAR_NODE_ADDRESS_WATCH + * Clear address watch for device. + * + * @gpu_id (IN) - target device to clear watch point + * @id (IN) - allocated watch id to clear + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * - ENODEV if gpu_id not found. + * - EINVAL if watch ID has not been allocated. + */ +struct kfd_ioctl_dbg_trap_clear_node_address_watch_args { + __uint32_t gpu_id; + __uint32_t id; +}; + +/** + * kfd_ioctl_dbg_trap_set_flags_args + * + * Arguments for KFD_IOC_DBG_TRAP_SET_FLAGS + * Sets flags for wave behaviour. + * + * @flags (IN/OUT) - IN = flags to enable, OUT = flags previously enabled + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * - EACCESS if any debug device does not allow flag options. + */ +struct kfd_ioctl_dbg_trap_set_flags_args { + __uint32_t flags; + __uint32_t pad; +}; + +/** + * kfd_ioctl_dbg_trap_query_debug_event_args + * + * Arguments for KFD_IOC_DBG_TRAP_QUERY_DEBUG_EVENT + * + * Find one or more raised exceptions. This function can return multiple + * exceptions from a single queue or a single device with one call. To find + * all raised exceptions, this function must be called repeatedly until it + * returns -EAGAIN. Returned exceptions can optionally be cleared by + * setting the corresponding bit in the @exception_mask input parameter. + * However, clearing an exception prevents retrieving further information + * about it with KFD_IOC_DBG_TRAP_QUERY_EXCEPTION_INFO. + * + * @exception_mask (IN/OUT) - exception to clear (IN) and raised (OUT) + * @gpu_id (OUT) - gpu id of exceptions raised + * @queue_id (OUT) - queue id of exceptions raised + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on raised exception found + * Raised exceptions found are returned in @exception mask + * with reported source id returned in @gpu_id or @queue_id. + * - EAGAIN if no raised exception has been found + */ +struct kfd_ioctl_dbg_trap_query_debug_event_args { + __uint64_t exception_mask; + __uint32_t gpu_id; + __uint32_t queue_id; +}; + +/** + * kfd_ioctl_dbg_trap_query_exception_info_args + * + * Arguments KFD_IOC_DBG_TRAP_QUERY_EXCEPTION_INFO + * Get additional info on raised exception. + * + * @info_ptr (IN) - pointer to exception info buffer to copy to + * @info_size (IN/OUT) - exception info buffer size (bytes) + * @source_id (IN) - target gpu or queue id + * @exception_code (IN) - target exception + * @clear_exception (IN) - clear raised @exception_code exception + * (0 = false, 1 = true) + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * If @exception_code is EC_DEVICE_MEMORY_VIOLATION, copy @info_size(OUT) + * bytes of memory exception data to @info_ptr. + * If @exception_code is EC_PROCESS_RUNTIME, copy saved + * kfd_runtime_info to @info_ptr. + * Actual required @info_ptr size (bytes) is returned in @info_size. + */ +struct kfd_ioctl_dbg_trap_query_exception_info_args { + __uint64_t info_ptr; + __uint32_t info_size; + __uint32_t source_id; + __uint32_t exception_code; + __uint32_t clear_exception; +}; + +/** + * kfd_ioctl_dbg_trap_get_queue_snapshot_args + * + * Arguments KFD_IOC_DBG_TRAP_GET_QUEUE_SNAPSHOT + * Get queue information. + * + * @exception_mask (IN) - exceptions raised to clear + * @snapshot_buf_ptr (IN) - queue snapshot entry buffer (see kfd_queue_snapshot_entry) + * @num_queues (IN/OUT) - number of queue snapshot entries + * The debugger specifies the size of the array allocated in @num_queues. + * KFD returns the number of queues that actually existed. If this is + * larger than the size specified by the debugger, KFD will not overflow + * the array allocated by the debugger. + * + * @entry_size (IN/OUT) - size per entry in bytes + * The debugger specifies sizeof(struct kfd_queue_snapshot_entry) in + * @entry_size. KFD returns the number of bytes actually populated per + * entry. The debugger should use the KFD_IOCTL_MINOR_VERSION to determine, + * which fields in struct kfd_queue_snapshot_entry are valid. This allows + * growing the ABI in a backwards compatible manner. + * Note that entry_size(IN) should still be used to stride the snapshot buffer in the + * event that it's larger than actual kfd_queue_snapshot_entry. + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * Copies @num_queues(IN) queue snapshot entries of size @entry_size(IN) + * into @snapshot_buf_ptr if @num_queues(IN) > 0. + * Otherwise return @num_queues(OUT) queue snapshot entries that exist. + */ +struct kfd_ioctl_dbg_trap_queue_snapshot_args { + __uint64_t exception_mask; + __uint64_t snapshot_buf_ptr; + __uint32_t num_queues; + __uint32_t entry_size; +}; + +/** + * kfd_ioctl_dbg_trap_get_device_snapshot_args + * + * Arguments for KFD_IOC_DBG_TRAP_GET_DEVICE_SNAPSHOT + * Get device information. + * + * @exception_mask (IN) - exceptions raised to clear + * @snapshot_buf_ptr (IN) - pointer to snapshot buffer (see kfd_dbg_device_info_entry) + * @num_devices (IN/OUT) - number of debug devices to snapshot + * The debugger specifies the size of the array allocated in @num_devices. + * KFD returns the number of devices that actually existed. If this is + * larger than the size specified by the debugger, KFD will not overflow + * the array allocated by the debugger. + * + * @entry_size (IN/OUT) - size per entry in bytes + * The debugger specifies sizeof(struct kfd_dbg_device_info_entry) in + * @entry_size. KFD returns the number of bytes actually populated. The + * debugger should use KFD_IOCTL_MINOR_VERSION to determine, which fields + * in struct kfd_dbg_device_info_entry are valid. This allows growing the + * ABI in a backwards compatible manner. + * Note that entry_size(IN) should still be used to stride the snapshot buffer in the + * event that it's larger than actual kfd_dbg_device_info_entry. + * + * Generic errors apply (see kfd_dbg_trap_operations). + * Return - 0 on SUCCESS. + * Copies @num_devices(IN) device snapshot entries of size @entry_size(IN) + * into @snapshot_buf_ptr if @num_devices(IN) > 0. + * Otherwise return @num_devices(OUT) queue snapshot entries that exist. + */ +struct kfd_ioctl_dbg_trap_device_snapshot_args { + __uint64_t exception_mask; + __uint64_t snapshot_buf_ptr; + __uint32_t num_devices; + __uint32_t entry_size; +}; + +/** + * kfd_ioctl_dbg_trap_args + * + * Arguments to debug target process. + * + * @pid - target process to debug + * @op - debug operation (see kfd_dbg_trap_operations) + * + * @op determines which union struct args to use. + * Refer to kern docs for each kfd_ioctl_dbg_trap_*_args struct. + */ +struct kfd_ioctl_dbg_trap_args { + __uint32_t pid; + __uint32_t op; + + union { + struct kfd_ioctl_dbg_trap_enable_args enable; + struct kfd_ioctl_dbg_trap_send_runtime_event_args send_runtime_event; + struct kfd_ioctl_dbg_trap_set_exceptions_enabled_args set_exceptions_enabled; + struct kfd_ioctl_dbg_trap_set_wave_launch_override_args launch_override; + struct kfd_ioctl_dbg_trap_set_wave_launch_mode_args launch_mode; + struct kfd_ioctl_dbg_trap_suspend_queues_args suspend_queues; + struct kfd_ioctl_dbg_trap_resume_queues_args resume_queues; + struct kfd_ioctl_dbg_trap_set_node_address_watch_args set_node_address_watch; + struct kfd_ioctl_dbg_trap_clear_node_address_watch_args clear_node_address_watch; + struct kfd_ioctl_dbg_trap_set_flags_args set_flags; + struct kfd_ioctl_dbg_trap_query_debug_event_args query_debug_event; + struct kfd_ioctl_dbg_trap_query_exception_info_args query_exception_info; + struct kfd_ioctl_dbg_trap_queue_snapshot_args queue_snapshot; + struct kfd_ioctl_dbg_trap_device_snapshot_args device_snapshot; + }; +}; + +/* Matching HSA_EVENTTYPE */ +#define KFD_IOC_EVENT_SIGNAL 0 +#define KFD_IOC_EVENT_NODECHANGE 1 +#define KFD_IOC_EVENT_DEVICESTATECHANGE 2 +#define KFD_IOC_EVENT_HW_EXCEPTION 3 +#define KFD_IOC_EVENT_SYSTEM_EVENT 4 +#define KFD_IOC_EVENT_DEBUG_EVENT 5 +#define KFD_IOC_EVENT_PROFILE_EVENT 6 +#define KFD_IOC_EVENT_QUEUE_EVENT 7 +#define KFD_IOC_EVENT_MEMORY 8 + +#define KFD_IOC_WAIT_RESULT_COMPLETE 0 +#define KFD_IOC_WAIT_RESULT_TIMEOUT 1 +#define KFD_IOC_WAIT_RESULT_FAIL 2 + +#define KFD_SIGNAL_EVENT_LIMIT 4096 + +/* For kfd_event_data.hw_exception_data.reset_type. */ +#define KFD_HW_EXCEPTION_WHOLE_GPU_RESET 0 +#define KFD_HW_EXCEPTION_PER_ENGINE_RESET 1 + +/* For kfd_event_data.hw_exception_data.reset_cause. */ +#define KFD_HW_EXCEPTION_GPU_HANG 0 +#define KFD_HW_EXCEPTION_ECC 1 + +/* For kfd_hsa_memory_exception_data.ErrorType */ +#define KFD_MEM_ERR_NO_RAS 0 +#define KFD_MEM_ERR_SRAM_ECC 1 +#define KFD_MEM_ERR_POISON_CONSUMED 2 +#define KFD_MEM_ERR_GPU_HANG 3 + +struct kfd_ioctl_create_event_args { + __uint64_t event_page_offset; /* from KFD */ + __uint32_t event_trigger_data; /* from KFD - signal events only */ + __uint32_t event_type; /* to KFD */ + __uint32_t auto_reset; /* to KFD */ + __uint32_t node_id; /* to KFD - only valid for certain + event types */ + __uint32_t event_id; /* from KFD */ + __uint32_t event_slot_index; /* from KFD */ +}; + +struct kfd_ioctl_destroy_event_args { + __uint32_t event_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_set_event_args { + __uint32_t event_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_reset_event_args { + __uint32_t event_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_memory_exception_failure { + __uint32_t NotPresent; /* Page not present or supervisor privilege */ + __uint32_t ReadOnly; /* Write access to a read-only page */ + __uint32_t NoExecute; /* Execute access to a page marked NX */ + __uint32_t imprecise; /* Can't determine the exact fault address */ +}; + +/* memory exception data */ +struct kfd_hsa_memory_exception_data { + struct kfd_memory_exception_failure failure; + __uint64_t va; + __uint32_t gpu_id; + __uint32_t ErrorType; /* 0 = no RAS error, + * 1 = ECC_SRAM, + * 2 = Link_SYNFLOOD (poison), + * 3 = GPU hang (not attributable to a specific cause), + * other values reserved + */ +}; + +/* hw exception data */ +struct kfd_hsa_hw_exception_data { + __uint32_t reset_type; + __uint32_t reset_cause; + __uint32_t memory_lost; + __uint32_t gpu_id; +}; + +/* hsa signal event data */ +struct kfd_hsa_signal_event_data { + __uint64_t last_event_age; /* to and from KFD */ +}; + +/* Event data */ +struct kfd_event_data { + union { + /* From KFD */ + struct kfd_hsa_memory_exception_data memory_exception_data; + struct kfd_hsa_hw_exception_data hw_exception_data; + /* To and From KFD */ + struct kfd_hsa_signal_event_data signal_event_data; + }; + __uint64_t kfd_event_data_ext; /* pointer to an extension structure + for future exception types */ + __uint32_t event_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_wait_events_args { + __uint64_t events_ptr; /* pointed to struct + kfd_event_data array, to KFD */ + __uint32_t num_events; /* to KFD */ + __uint32_t wait_for_all; /* to KFD */ + __uint32_t timeout; /* to KFD */ + __uint32_t wait_result; /* from KFD */ +}; + +struct kfd_ioctl_set_scratch_backing_va_args { + __uint64_t va_addr; /* to KFD */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_get_tile_config_args { + /* to KFD: pointer to tile array */ + __uint64_t tile_config_ptr; + /* to KFD: pointer to macro tile array */ + __uint64_t macro_tile_config_ptr; + /* to KFD: array size allocated by user mode + * from KFD: array size filled by kernel + */ + __uint32_t num_tile_configs; + /* to KFD: array size allocated by user mode + * from KFD: array size filled by kernel + */ + __uint32_t num_macro_tile_configs; + + __uint32_t gpu_id; /* to KFD */ + __uint32_t gb_addr_config; /* from KFD */ + __uint32_t num_banks; /* from KFD */ + __uint32_t num_ranks; /* from KFD */ + /* struct size can be extended later if needed + * without breaking ABI compatibility + */ +}; + +struct kfd_ioctl_set_trap_handler_args { + __uint64_t tba_addr; /* to KFD */ + __uint64_t tma_addr; /* to KFD */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_acquire_vm_args { + __uint32_t drm_fd; /* to KFD */ + __uint32_t gpu_id; /* to KFD */ +}; + +/* Allocation flags: memory types */ +#define KFD_IOC_ALLOC_MEM_FLAGS_VRAM (1 << 0) +#define KFD_IOC_ALLOC_MEM_FLAGS_GTT (1 << 1) +#define KFD_IOC_ALLOC_MEM_FLAGS_USERPTR (1 << 2) +#define KFD_IOC_ALLOC_MEM_FLAGS_DOORBELL (1 << 3) +#define KFD_IOC_ALLOC_MEM_FLAGS_MMIO_REMAP (1 << 4) +/* Allocation flags: attributes/access options */ +#define KFD_IOC_ALLOC_MEM_FLAGS_WRITABLE (1 << 31) +#define KFD_IOC_ALLOC_MEM_FLAGS_EXECUTABLE (1 << 30) +#define KFD_IOC_ALLOC_MEM_FLAGS_PUBLIC (1 << 29) +#define KFD_IOC_ALLOC_MEM_FLAGS_NO_SUBSTITUTE (1 << 28) +#define KFD_IOC_ALLOC_MEM_FLAGS_AQL_QUEUE_MEM (1 << 27) +#define KFD_IOC_ALLOC_MEM_FLAGS_COHERENT (1 << 26) +#define KFD_IOC_ALLOC_MEM_FLAGS_UNCACHED (1 << 25) +#define KFD_IOC_ALLOC_MEM_FLAGS_EXT_COHERENT (1 << 24) +#define KFD_IOC_ALLOC_MEM_FLAGS_CONTIGUOUS_BEST_EFFORT (1 << 23) + +/* Allocate memory for later SVM (shared virtual memory) mapping. + * + * @va_addr: virtual address of the memory to be allocated + * all later mappings on all GPUs will use this address + * @size: size in bytes + * @handle: buffer handle returned to user mode, used to refer to + * this allocation for mapping, unmapping and freeing + * @mmap_offset: for CPU-mapping the allocation by mmapping a render node + * for userptrs this is overloaded to specify the CPU address + * @gpu_id: device identifier + * @flags: memory type and attributes. See KFD_IOC_ALLOC_MEM_FLAGS above + */ +struct kfd_ioctl_alloc_memory_of_gpu_args { + __uint64_t va_addr; /* to KFD */ + __uint64_t size; /* to KFD */ + __uint64_t handle; /* from KFD */ + __uint64_t mmap_offset; /* to KFD (userptr), from KFD (mmap offset) */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t flags; +}; + +/* Free memory allocated with kfd_ioctl_alloc_memory_of_gpu + * + * @handle: memory handle returned by alloc + */ +struct kfd_ioctl_free_memory_of_gpu_args { + __uint64_t handle; /* to KFD */ +}; + +/* Inquire available memory with kfd_ioctl_get_available_memory + * + * @available: memory available for alloc + */ +struct kfd_ioctl_get_available_memory_args { + __uint64_t available; /* from KFD */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t pad; +}; + +/* Map memory to one or more GPUs + * + * @handle: memory handle returned by alloc + * @device_ids_array_ptr: array of gpu_ids (__uint32_t per device) + * @n_devices: number of devices in the array + * @n_success: number of devices mapped successfully + * + * @n_success returns information to the caller how many devices from + * the start of the array have mapped the buffer successfully. It can + * be passed into a subsequent retry call to skip those devices. For + * the first call the caller should initialize it to 0. + * + * If the ioctl completes with return code 0 (success), n_success == + * n_devices. + */ +struct kfd_ioctl_map_memory_to_gpu_args { + __uint64_t handle; /* to KFD */ + __uint64_t device_ids_array_ptr; /* to KFD */ + __uint32_t n_devices; /* to KFD */ + __uint32_t n_success; /* to/from KFD */ +}; + +/* Unmap memory from one or more GPUs + * + * same arguments as for mapping + */ +struct kfd_ioctl_unmap_memory_from_gpu_args { + __uint64_t handle; /* to KFD */ + __uint64_t device_ids_array_ptr; /* to KFD */ + __uint32_t n_devices; /* to KFD */ + __uint32_t n_success; /* to/from KFD */ +}; + +/* Allocate GWS for specific queue + * + * @queue_id: queue's id that GWS is allocated for + * @num_gws: how many GWS to allocate + * @first_gws: index of the first GWS allocated. + * only support contiguous GWS allocation + */ +struct kfd_ioctl_alloc_queue_gws_args { + __uint32_t queue_id; /* to KFD */ + __uint32_t num_gws; /* to KFD */ + __uint32_t first_gws; /* from KFD */ + __uint32_t pad; +}; + +struct kfd_ioctl_get_dmabuf_info_args { + __uint64_t size; /* from KFD */ + __uint64_t metadata_ptr; /* to KFD */ + __uint32_t metadata_size; /* to KFD (space allocated by user) + * from KFD (actual metadata size) + */ + __uint32_t gpu_id; /* from KFD */ + __uint32_t flags; /* from KFD (KFD_IOC_ALLOC_MEM_FLAGS) */ + __uint32_t dmabuf_fd; /* to KFD */ +}; + +struct kfd_ioctl_import_dmabuf_args { + __uint64_t va_addr; /* to KFD */ + __uint64_t handle; /* from KFD */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t dmabuf_fd; /* to KFD */ +}; + +struct kfd_ioctl_export_dmabuf_args { + __uint64_t handle; /* to KFD */ + __uint32_t flags; /* to KFD */ + __uint32_t dmabuf_fd; /* from KFD */ +}; + +/* + * KFD SMI(System Management Interface) events + */ +enum kfd_smi_event { + KFD_SMI_EVENT_NONE = 0, /* not used */ + KFD_SMI_EVENT_VMFAULT = 1, /* event start counting at 1 */ + KFD_SMI_EVENT_THERMAL_THROTTLE = 2, + KFD_SMI_EVENT_GPU_PRE_RESET = 3, + KFD_SMI_EVENT_GPU_POST_RESET = 4, +}; + +#define KFD_SMI_EVENT_MASK_FROM_INDEX(i) (1ULL << ((i) - 1)) +#define KFD_SMI_EVENT_MSG_SIZE 96 + +struct kfd_ioctl_smi_events_args { + __uint32_t gpuid; /* to KFD */ + __uint32_t anon_fd; /* from KFD */ +}; + +/** + * kfd_ioctl_spm_op - SPM ioctl operations + * + * @KFD_IOCTL_SPM_OP_ACQUIRE: acquire exclusive access to SPM + * @KFD_IOCTL_SPM_OP_RELEASE: release exclusive access to SPM + * @KFD_IOCTL_SPM_OP_SET_DEST_BUF: set or unset destination buffer for SPM streaming + */ +enum kfd_ioctl_spm_op { + KFD_IOCTL_SPM_OP_ACQUIRE, + KFD_IOCTL_SPM_OP_RELEASE, + KFD_IOCTL_SPM_OP_SET_DEST_BUF +}; + +/** + * kfd_ioctl_spm_args - Arguments for SPM ioctl + * + * @op[in]: specifies the operation to perform + * @gpu_id[in]: GPU ID of the GPU to profile + * @dst_buf[in]: used for the address of the destination buffer + * in @KFD_IOCTL_SPM_SET_DEST_BUFFER + * @buf_size[in]: size of the destination buffer + * @timeout[in/out]: [in]: timeout in milliseconds, [out]: amount of time left + * `in the timeout window + * @bytes_copied[out]: total amount of data that was copied to the previous dest_buf + * @has_data_loss: total count for sub-block which has data loss + * + * This ioctl performs different functions depending on the @op parameter. + * + * KFD_IOCTL_SPM_OP_ACQUIRE + * ------------------------ + * + * Acquires exclusive access of SPM on the specified @gpu_id for the calling process. + * This must be called before using KFD_IOCTL_SPM_OP_SET_DEST_BUF. + * + * KFD_IOCTL_SPM_OP_RELEASE + * ------------------------ + * + * Releases exclusive access of SPM on the specified @gpu_id for the calling process, + * which allows another process to acquire it in the future. + * + * KFD_IOCTL_SPM_OP_SET_DEST_BUF + * ----------------------------- + * + * If @dst_buf is NULL, the destination buffer address is unset and copying of counters + * is stopped. + * + * If @dst_buf is not NULL, it specifies the pointer to a new destination buffer. + * @buf_size specifies the size of the buffer. + * + * If @timeout is non-0, the call will wait for up to @timeout ms for the previous + * buffer to be filled. If previous buffer to be filled before timeout, the @timeout + * will be updated value with the time remaining. If the timeout is exceeded, the function + * copies any partial data available into the previous user buffer and returns success. + * The amount of valid data in the previous user buffer is indicated by @bytes_copied. + * + * If @timeout is 0, the function immediately replaces the previous destination buffer + * without waiting for the previous buffer to be filled. That means the previous buffer + * may only be partially filled, and @bytes_copied will indicate how much data has been + * copied to it. + * + * If data was lost, e.g. due to a ring buffer overflow, @has_data_loss will be non-0. + * + * Returns negative error code on failure, 0 on success. + */ +struct kfd_ioctl_spm_args { + __uint64_t dest_buf; + __uint32_t buf_size; + __uint32_t op; + __uint32_t timeout; + __uint32_t gpu_id; + __uint32_t bytes_copied; + __uint32_t has_data_loss; +}; + +/** + * kfd_ioctl_spm_buffer_header - SPM Buffer header for kfd_ioctl_spm_args->dest_buf + * + * @version [out]: spm versiom + * @bytes_copied [out]: amount of data for each sub-block + * @has_data_loss: [out]: boolean indicating whether data was lost for each sub-block + * (e.g. due to a ring-buffer overflow) + */ +struct kfd_ioctl_spm_buffer_header { + __uint32_t version; /* 0-23: minor 24-31: major */ + __uint32_t bytes_copied; + __uint32_t has_data_loss; + __uint32_t reserved[5]; +}; + +/************************************************************************************************** + * CRIU IOCTLs (Checkpoint Restore In Userspace) + * + * When checkpointing a process, the userspace application will perform: + * 1. PROCESS_INFO op to determine current process information. This pauses execution and evicts + * all the queues. + * 2. CHECKPOINT op to checkpoint process contents (BOs, queues, events, svm-ranges) + * 3. UNPAUSE op to un-evict all the queues + * + * When restoring a process, the CRIU userspace application will perform: + * + * 1. RESTORE op to restore process contents + * 2. RESUME op to start the process + * + * Note: Queues are forced into an evicted state after a successful PROCESS_INFO. User + * application needs to perform an UNPAUSE operation after calling PROCESS_INFO. + */ + +enum kfd_criu_op { + KFD_CRIU_OP_PROCESS_INFO, + KFD_CRIU_OP_CHECKPOINT, + KFD_CRIU_OP_UNPAUSE, + KFD_CRIU_OP_RESTORE, + KFD_CRIU_OP_RESUME, +}; + +/** + * kfd_ioctl_criu_args - Arguments perform CRIU operation + * @devices: [in/out] User pointer to memory location for devices information. + * This is an array of type kfd_criu_device_bucket. + * @bos: [in/out] User pointer to memory location for BOs information + * This is an array of type kfd_criu_bo_bucket. + * @priv_data: [in/out] User pointer to memory location for private data + * @priv_data_size: [in/out] Size of priv_data in bytes + * @num_devices: [in/out] Number of GPUs used by process. Size of @devices array. + * @num_bos [in/out] Number of BOs used by process. Size of @bos array. + * @num_objects: [in/out] Number of objects used by process. Objects are opaque to + * user application. + * @pid: [in/out] PID of the process being checkpointed + * @op [in] Type of operation (kfd_criu_op) + * + * Return: 0 on success, -errno on failure + */ +struct kfd_ioctl_criu_args { + __uint64_t devices; /* Used during ops: CHECKPOINT, RESTORE */ + __uint64_t bos; /* Used during ops: CHECKPOINT, RESTORE */ + __uint64_t priv_data; /* Used during ops: CHECKPOINT, RESTORE */ + __uint64_t priv_data_size; /* Used during ops: PROCESS_INFO, RESTORE */ + __uint32_t num_devices; /* Used during ops: PROCESS_INFO, RESTORE */ + __uint32_t num_bos; /* Used during ops: PROCESS_INFO, RESTORE */ + __uint32_t num_objects; /* Used during ops: PROCESS_INFO, RESTORE */ + __uint32_t pid; /* Used during ops: PROCESS_INFO, RESUME */ + __uint32_t op; +}; + +struct kfd_criu_device_bucket { + __uint32_t user_gpu_id; + __uint32_t actual_gpu_id; + __uint32_t drm_fd; + __uint32_t pad; +}; + +struct kfd_criu_bo_bucket { + __uint64_t addr; + __uint64_t size; + __uint64_t offset; + __uint64_t restored_offset; /* During restore, updated offset for BO */ + __uint32_t gpu_id; /* This is the user_gpu_id */ + __uint32_t alloc_flags; + __uint32_t dmabuf_fd; + __uint32_t pad; +}; + +/* CRIU IOCTLs - END */ +/**************************************************************************************************/ +/* Register offset inside the remapped mmio page + */ +enum kfd_mmio_remap { + KFD_MMIO_REMAP_HDP_MEM_FLUSH_CNTL = 0, + KFD_MMIO_REMAP_HDP_REG_FLUSH_CNTL = 4, +}; + +struct kfd_ioctl_ipc_export_handle_args { + __uint64_t handle; /* to KFD */ + __uint32_t share_handle[4]; /* from KFD */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t flags; /* to KFD */ +}; + +struct kfd_ioctl_ipc_import_handle_args { + __uint64_t handle; /* from KFD */ + __uint64_t va_addr; /* to KFD */ + __uint64_t mmap_offset; /* from KFD */ + __uint32_t share_handle[4]; /* to KFD */ + __uint32_t gpu_id; /* to KFD */ + __uint32_t flags; /* from KFD */ +}; + +struct kfd_memory_range { + __uint64_t va_addr; + __uint64_t size; +}; + +/* flags definitions + * BIT0: 0: read operation, 1: write operation. + * This also identifies if the src or dst array belongs to remote process + */ +#define KFD_CROSS_MEMORY_RW_BIT (1 << 0) +#define KFD_SET_CROSS_MEMORY_READ(flags) (flags &= ~KFD_CROSS_MEMORY_RW_BIT) +#define KFD_SET_CROSS_MEMORY_WRITE(flags) (flags |= KFD_CROSS_MEMORY_RW_BIT) +#define KFD_IS_CROSS_MEMORY_WRITE(flags) (flags & KFD_CROSS_MEMORY_RW_BIT) + +struct kfd_ioctl_cross_memory_copy_args { + /* to KFD: Process ID of the remote process */ + __uint32_t pid; + /* to KFD: See above definition */ + __uint32_t flags; + /* to KFD: Source GPU VM range */ + __uint64_t src_mem_range_array; + /* to KFD: Size of above array */ + __uint64_t src_mem_array_size; + /* to KFD: Destination GPU VM range */ + __uint64_t dst_mem_range_array; + /* to KFD: Size of above array */ + __uint64_t dst_mem_array_size; + /* from KFD: Total amount of bytes copied */ + __uint64_t bytes_copied; +}; + +/* Guarantee host access to memory */ +#define KFD_IOCTL_SVM_FLAG_HOST_ACCESS 0x00000001 +/* Fine grained coherency between all devices with access */ +#define KFD_IOCTL_SVM_FLAG_COHERENT 0x00000002 +/* Use any GPU in same hive as preferred device */ +#define KFD_IOCTL_SVM_FLAG_HIVE_LOCAL 0x00000004 +/* GPUs only read, allows replication */ +#define KFD_IOCTL_SVM_FLAG_GPU_RO 0x00000008 +/* Allow execution on GPU */ +#define KFD_IOCTL_SVM_FLAG_GPU_EXEC 0x00000010 +/* GPUs mostly read, may allow similar optimizations as RO, but writes fault */ +#define KFD_IOCTL_SVM_FLAG_GPU_READ_MOSTLY 0x00000020 +/* Keep GPU memory mapping always valid as if XNACK is disable */ +#define KFD_IOCTL_SVM_FLAG_GPU_ALWAYS_MAPPED 0x00000040 +/* Fine grained coherency between all devices using device-scope atomics */ +#define KFD_IOCTL_SVM_FLAG_EXT_COHERENT 0x00000080 + +/** + * kfd_ioctl_svm_op - SVM ioctl operations + * + * @KFD_IOCTL_SVM_OP_SET_ATTR: Modify one or more attributes + * @KFD_IOCTL_SVM_OP_GET_ATTR: Query one or more attributes + */ +enum kfd_ioctl_svm_op { + KFD_IOCTL_SVM_OP_SET_ATTR, + KFD_IOCTL_SVM_OP_GET_ATTR +}; + +/** kfd_ioctl_svm_location - Enum for preferred and prefetch locations + * + * GPU IDs are used to specify GPUs as preferred and prefetch locations. + * Below definitions are used for system memory or for leaving the preferred + * location unspecified. + */ +enum kfd_ioctl_svm_location { + KFD_IOCTL_SVM_LOCATION_SYSMEM = 0, + KFD_IOCTL_SVM_LOCATION_UNDEFINED = 0xffffffff +}; + +/** + * kfd_ioctl_svm_attr_type - SVM attribute types + * + * @KFD_IOCTL_SVM_ATTR_PREFERRED_LOC: gpuid of the preferred location, 0 for + * system memory + * @KFD_IOCTL_SVM_ATTR_PREFETCH_LOC: gpuid of the prefetch location, 0 for + * system memory. Setting this triggers an + * immediate prefetch (migration). + * @KFD_IOCTL_SVM_ATTR_ACCESS: + * @KFD_IOCTL_SVM_ATTR_ACCESS_IN_PLACE: + * @KFD_IOCTL_SVM_ATTR_NO_ACCESS: specify memory access for the gpuid given + * by the attribute value + * @KFD_IOCTL_SVM_ATTR_SET_FLAGS: bitmask of flags to set (see + * KFD_IOCTL_SVM_FLAG_...) + * @KFD_IOCTL_SVM_ATTR_CLR_FLAGS: bitmask of flags to clear + * @KFD_IOCTL_SVM_ATTR_GRANULARITY: migration granularity + * (log2 num pages) + */ +enum kfd_ioctl_svm_attr_type { + KFD_IOCTL_SVM_ATTR_PREFERRED_LOC, + KFD_IOCTL_SVM_ATTR_PREFETCH_LOC, + KFD_IOCTL_SVM_ATTR_ACCESS, + KFD_IOCTL_SVM_ATTR_ACCESS_IN_PLACE, + KFD_IOCTL_SVM_ATTR_NO_ACCESS, + KFD_IOCTL_SVM_ATTR_SET_FLAGS, + KFD_IOCTL_SVM_ATTR_CLR_FLAGS, + KFD_IOCTL_SVM_ATTR_GRANULARITY +}; + +/** + * kfd_ioctl_svm_attribute - Attributes as pairs of type and value + * + * The meaning of the @value depends on the attribute type. + * + * @type: attribute type (see enum @kfd_ioctl_svm_attr_type) + * @value: attribute value + */ +struct kfd_ioctl_svm_attribute { + __uint32_t type; + __uint32_t value; +}; + +/** + * kfd_ioctl_svm_args - Arguments for SVM ioctl + * + * @op specifies the operation to perform (see enum + * @kfd_ioctl_svm_op). @start_addr and @size are common for all + * operations. + * + * A variable number of attributes can be given in @attrs. + * @nattr specifies the number of attributes. New attributes can be + * added in the future without breaking the ABI. If unknown attributes + * are given, the function returns -EINVAL. + * + * @KFD_IOCTL_SVM_OP_SET_ATTR sets attributes for a virtual address + * range. It may overlap existing virtual address ranges. If it does, + * the existing ranges will be split such that the attribute changes + * only apply to the specified address range. + * + * @KFD_IOCTL_SVM_OP_GET_ATTR returns the intersection of attributes + * over all memory in the given range and returns the result as the + * attribute value. If different pages have different preferred or + * prefetch locations, 0xffffffff will be returned for + * @KFD_IOCTL_SVM_ATTR_PREFERRED_LOC or + * @KFD_IOCTL_SVM_ATTR_PREFETCH_LOC resepctively. For + * @KFD_IOCTL_SVM_ATTR_SET_FLAGS, flags of all pages will be + * aggregated by bitwise AND. That means, a flag will be set in the + * output, if that flag is set for all pages in the range. For + * @KFD_IOCTL_SVM_ATTR_CLR_FLAGS, flags of all pages will be + * aggregated by bitwise NOR. That means, a flag will be set in the + * output, if that flag is clear for all pages in the range. + * The minimum migration granularity throughout the range will be + * returned for @KFD_IOCTL_SVM_ATTR_GRANULARITY. + * + * Querying of accessibility attributes works by initializing the + * attribute type to @KFD_IOCTL_SVM_ATTR_ACCESS and the value to the + * GPUID being queried. Multiple attributes can be given to allow + * querying multiple GPUIDs. The ioctl function overwrites the + * attribute type to indicate the access for the specified GPU. + */ + + // TODO: FIXME: This is not compatible with FreeBSD, but for + // getting a basic Hello World Working, this is fine +struct kfd_ioctl_svm_args { + __uint64_t start_addr; + __uint64_t size; + __uint32_t op; + __uint32_t nattr; + /* Variable length array of attributes */ + struct kfd_ioctl_svm_attribute attrs[]; +}; + +/** + * kfd_ioctl_set_xnack_mode_args - Arguments for set_xnack_mode + * + * @xnack_enabled: [in/out] Whether to enable XNACK mode for this process + * + * @xnack_enabled indicates whether recoverable page faults should be + * enabled for the current process. 0 means disabled, positive means + * enabled, negative means leave unchanged. If enabled, virtual address + * translations on GFXv9 and later AMD GPUs can return XNACK and retry + * the access until a valid PTE is available. This is used to implement + * device page faults. + * + * On output, @xnack_enabled returns the (new) current mode (0 or + * positive). Therefore, a negative input value can be used to query + * the current mode without changing it. + * + * The XNACK mode fundamentally changes the way SVM managed memory works + * in the driver, with subtle effects on application performance and + * functionality. + * + * Enabling XNACK mode requires shader programs to be compiled + * differently. Furthermore, not all GPUs support changing the mode + * per-process. Therefore changing the mode is only allowed while no + * user mode queues exist in the process. This ensure that no shader + * code is running that may be compiled for the wrong mode. And GPUs + * that cannot change to the requested mode will prevent the XNACK + * mode from occurring. All GPUs used by the process must be in the + * same XNACK mode. + * + * GFXv8 or older GPUs do not support 48 bit virtual addresses or SVM. + * Therefore those GPUs are not considered for the XNACK mode switch. + * + * Return: 0 on success, -errno on failure + */ +struct kfd_ioctl_set_xnack_mode_args { + __int32_t xnack_enabled; +}; + +/** + * kfd_ioctl_pc_sample_op - PC Sampling ioctl operations + * + * @KFD_IOCTL_PCS_OP_QUERY_CAPABILITIES: Query device PC Sampling capabilities + * @KFD_IOCTL_PCS_OP_CREATE: Register this process with a per-device PC sampler instance + * @KFD_IOCTL_PCS_OP_DESTROY: Unregister from a previously registered PC sampler instance + * @KFD_IOCTL_PCS_OP_START: Process begins taking samples from a previously registered PC sampler instance + * @KFD_IOCTL_PCS_OP_STOP: Process stops taking samples from a previously registered PC sampler instance + */ +enum kfd_ioctl_pc_sample_op { + KFD_IOCTL_PCS_OP_QUERY_CAPABILITIES, + KFD_IOCTL_PCS_OP_CREATE, + KFD_IOCTL_PCS_OP_DESTROY, + KFD_IOCTL_PCS_OP_START, + KFD_IOCTL_PCS_OP_STOP, +}; + +/* Values have to be a power of 2*/ +#define KFD_IOCTL_PCS_FLAG_POWER_OF_2 0x00000001 + +enum kfd_ioctl_pc_sample_method { + KFD_IOCTL_PCS_METHOD_HOSTTRAP = 1, + KFD_IOCTL_PCS_METHOD_STOCHASTIC, +}; + +enum kfd_ioctl_pc_sample_type { + KFD_IOCTL_PCS_TYPE_TIME_US, + KFD_IOCTL_PCS_TYPE_CLOCK_CYCLES, + KFD_IOCTL_PCS_TYPE_INSTRUCTIONS +}; + +struct kfd_pc_sample_info { + __uint64_t interval; /* [IN] if PCS_TYPE_INTERVAL_US: sample interval in us + * if PCS_TYPE_CLOCK_CYCLES: sample interval in graphics core clk cycles + * if PCS_TYPE_INSTRUCTIONS: sample interval in instructions issued by + * graphics compute units + */ + __uint64_t interval_min; /* [OUT] */ + __uint64_t interval_max; /* [OUT] */ + __uint64_t flags; /* [OUT] indicate potential restrictions e.g FLAG_POWER_OF_2 */ + __uint32_t method; /* [IN/OUT] kfd_ioctl_pc_sample_method */ + __uint32_t type; /* [IN/OUT] kfd_ioctl_pc_sample_type */ +}; + +#define KFD_IOCTL_PCS_QUERY_TYPE_FULL (1 << 0) /* If not set, return current */ + +struct kfd_ioctl_pc_sample_args { + __uint64_t sample_info_ptr; /* array of kfd_pc_sample_info */ + __uint32_t num_sample_info; + __uint32_t op; /* kfd_ioctl_pc_sample_op */ + __uint32_t gpu_id; + __uint32_t trace_id; + __uint32_t flags; /* kfd_ioctl_pcs_query flags */ + __uint32_t reserved; +}; + +#define KFD_IOC_PROFILER_VERSION_NUM 1 +enum kfd_profiler_ops { + KFD_IOC_PROFILER_PMC = 0, + KFD_IOC_PROFILER_PC_SAMPLE = 1, + KFD_IOC_PROFILER_VERSION = 2, +}; + +/** + * Enables/Disables GPU Specific profiler settings + */ +struct kfd_ioctl_pmc_settings { + __uint32_t gpu_id; /* This is the user_gpu_id */ + __uint32_t lock; /* Lock GPU for Profiling */ + __uint32_t perfcount_enable; /* Force Perfcount Enable for queues on GPU */ +}; + +struct kfd_ioctl_profiler_args { + __uint32_t op; /* kfd_profiler_op */ + union { + struct kfd_ioctl_pc_sample_args pc_sample; + struct kfd_ioctl_pmc_settings pmc; + __uint32_t version; /* KFD_IOC_PROFILER_VERSION_NUM */ + }; +}; + +/** + * kfd_ais_ops - AIS ioctl operations + * + * @KFD_IOC_AIS_READ: Direct IO read from a file into VRAM + * @KFD_IOC_AIS_WRITE: Direct IO write into a file from VRAM + */ +enum kfd_ais_ops { + KFD_IOC_AIS_READ = 1, + KFD_IOC_AIS_WRITE = 2, +}; + +/** + * kfd_ais_in_args + * + * @op (IN) - kfd_ais_ops + * @fd (IN) - file descriptor of the file to read/write + * @handle (IN) - memory handle returned by alloc. Should be mapped to + * the GPU with AMDKFD_IOC_MAP_MEMORY_TO_GPU. + * @handle_offset (IN) - offset into the allocated memory to read/write + * @file_offset (IN) - offset from the beginning of the file to read/write + * @size (IN) - size in bytes to read/write + */ + +struct kfd_ais_in_args { + __uint64_t handle; /* to KFD */ + __uint64_t handle_offset; /* to KFD */ + __int64_t file_offset; /* to KFD */ + __uint64_t size; /* to KFD */ + __uint32_t op; /* to KFD */ + __int32_t fd; /* to KFD */ +}; + +/** + * kfd_ais_out_args + * + * @size_copied (OUT) KFD returns number of bytes transferred + * @status (OUT) 0 for success and -ve error values if failure + */ +struct kfd_ais_out_args { + __uint64_t size_copied; /* from KFD */ + __int32_t status; /* from KFD */ + __int32_t pad; /* unused */ +}; + +/** + * Arguments for AMDKFD_IOC_AIS_OP + * AIS (AMD Infinity Storage) operations. + * See @kfd_ais_in_args and @kfd_ais_out_args + */ + +struct kfd_ioctl_ais_args { + union { + struct kfd_ais_in_args in; + struct kfd_ais_out_args out; + }; +}; + +/** + * kfd_ioctl_create_process_args + * Create secondary KFD context ioctl operations + * + * @flags not use at current. + */ +struct kfd_ioctl_create_process_args { + __uint32_t flags; /* [IN] */ + __uint32_t pad; +}; + +#define AMDKFD_IOCTL_BASE 'K' +#define AMDKFD_IO(nr) _IO(AMDKFD_IOCTL_BASE, nr) +#define AMDKFD_IOR(nr, type) _IOR(AMDKFD_IOCTL_BASE, nr, type) +#define AMDKFD_IOW(nr, type) _IOW(AMDKFD_IOCTL_BASE, nr, type) +#define AMDKFD_IOWR(nr, type) _IOWR(AMDKFD_IOCTL_BASE, nr, type) + +#define AMDKFD_IOC_GET_VERSION \ + AMDKFD_IOR(0x01, struct kfd_ioctl_get_version_args) + +#define AMDKFD_IOC_CREATE_QUEUE \ + AMDKFD_IOWR(0x02, struct kfd_ioctl_create_queue_args) + +#define AMDKFD_IOC_DESTROY_QUEUE \ + AMDKFD_IOWR(0x03, struct kfd_ioctl_destroy_queue_args) + +#define AMDKFD_IOC_SET_MEMORY_POLICY \ + AMDKFD_IOW(0x04, struct kfd_ioctl_set_memory_policy_args) + +#define AMDKFD_IOC_GET_CLOCK_COUNTERS \ + AMDKFD_IOWR(0x05, struct kfd_ioctl_get_clock_counters_args) + +#define AMDKFD_IOC_GET_PROCESS_APERTURES \ + AMDKFD_IOR(0x06, struct kfd_ioctl_get_process_apertures_args) + +#define AMDKFD_IOC_UPDATE_QUEUE \ + AMDKFD_IOW(0x07, struct kfd_ioctl_update_queue_args) + +#define AMDKFD_IOC_CREATE_EVENT \ + AMDKFD_IOWR(0x08, struct kfd_ioctl_create_event_args) + +#define AMDKFD_IOC_DESTROY_EVENT \ + AMDKFD_IOW(0x09, struct kfd_ioctl_destroy_event_args) + +#define AMDKFD_IOC_SET_EVENT \ + AMDKFD_IOW(0x0A, struct kfd_ioctl_set_event_args) + +#define AMDKFD_IOC_RESET_EVENT \ + AMDKFD_IOW(0x0B, struct kfd_ioctl_reset_event_args) + +#define AMDKFD_IOC_WAIT_EVENTS \ + AMDKFD_IOWR(0x0C, struct kfd_ioctl_wait_events_args) + +#define AMDKFD_IOC_DBG_REGISTER_DEPRECATED \ + AMDKFD_IOW(0x0D, struct kfd_ioctl_dbg_register_args) + +#define AMDKFD_IOC_DBG_UNREGISTER_DEPRECATED \ + AMDKFD_IOW(0x0E, struct kfd_ioctl_dbg_unregister_args) + +#define AMDKFD_IOC_DBG_ADDRESS_WATCH_DEPRECATED \ + AMDKFD_IOW(0x0F, struct kfd_ioctl_dbg_address_watch_args) + +#define AMDKFD_IOC_DBG_WAVE_CONTROL_DEPRECATED \ + AMDKFD_IOW(0x10, struct kfd_ioctl_dbg_wave_control_args) + +#define AMDKFD_IOC_SET_SCRATCH_BACKING_VA \ + AMDKFD_IOWR(0x11, struct kfd_ioctl_set_scratch_backing_va_args) + +#define AMDKFD_IOC_GET_TILE_CONFIG \ + AMDKFD_IOWR(0x12, struct kfd_ioctl_get_tile_config_args) + +#define AMDKFD_IOC_SET_TRAP_HANDLER \ + AMDKFD_IOW(0x13, struct kfd_ioctl_set_trap_handler_args) + +#define AMDKFD_IOC_GET_PROCESS_APERTURES_NEW \ + AMDKFD_IOWR(0x14, \ + struct kfd_ioctl_get_process_apertures_new_args) + +#define AMDKFD_IOC_ACQUIRE_VM \ + AMDKFD_IOW(0x15, struct kfd_ioctl_acquire_vm_args) + +#define AMDKFD_IOC_ALLOC_MEMORY_OF_GPU \ + AMDKFD_IOWR(0x16, struct kfd_ioctl_alloc_memory_of_gpu_args) + +#define AMDKFD_IOC_FREE_MEMORY_OF_GPU \ + AMDKFD_IOW(0x17, struct kfd_ioctl_free_memory_of_gpu_args) + +#define AMDKFD_IOC_MAP_MEMORY_TO_GPU \ + AMDKFD_IOWR(0x18, struct kfd_ioctl_map_memory_to_gpu_args) + +#define AMDKFD_IOC_UNMAP_MEMORY_FROM_GPU \ + AMDKFD_IOWR(0x19, struct kfd_ioctl_unmap_memory_from_gpu_args) + +#define AMDKFD_IOC_SET_CU_MASK \ + AMDKFD_IOW(0x1A, struct kfd_ioctl_set_cu_mask_args) + +#define AMDKFD_IOC_GET_QUEUE_WAVE_STATE \ + AMDKFD_IOWR(0x1B, struct kfd_ioctl_get_queue_wave_state_args) + +#define AMDKFD_IOC_GET_DMABUF_INFO \ + AMDKFD_IOWR(0x1C, struct kfd_ioctl_get_dmabuf_info_args) + +#define AMDKFD_IOC_IMPORT_DMABUF \ + AMDKFD_IOWR(0x1D, struct kfd_ioctl_import_dmabuf_args) + +#define AMDKFD_IOC_ALLOC_QUEUE_GWS \ + AMDKFD_IOWR(0x1E, struct kfd_ioctl_alloc_queue_gws_args) + +#define AMDKFD_IOC_SMI_EVENTS \ + AMDKFD_IOWR(0x1F, struct kfd_ioctl_smi_events_args) + +#define AMDKFD_IOC_SVM AMDKFD_IOWR(0x20, struct kfd_ioctl_svm_args) + +#define AMDKFD_IOC_SET_XNACK_MODE \ + AMDKFD_IOWR(0x21, struct kfd_ioctl_set_xnack_mode_args) + +#define AMDKFD_IOC_CRIU_OP \ + AMDKFD_IOWR(0x22, struct kfd_ioctl_criu_args) + +#define AMDKFD_IOC_AVAILABLE_MEMORY \ + AMDKFD_IOWR(0x23, struct kfd_ioctl_get_available_memory_args) + +#define AMDKFD_IOC_EXPORT_DMABUF \ + AMDKFD_IOWR(0x24, struct kfd_ioctl_export_dmabuf_args) + +#define AMDKFD_IOC_RUNTIME_ENABLE \ + AMDKFD_IOWR(0x25, struct kfd_ioctl_runtime_enable_args) + +#define AMDKFD_IOC_DBG_TRAP \ + AMDKFD_IOWR(0x26, struct kfd_ioctl_dbg_trap_args) + +#define AMDKFD_IOC_CREATE_PROCESS \ + AMDKFD_IOWR(0x27, struct kfd_ioctl_create_process_args) + +#define AMDKFD_COMMAND_START 0x01 +#define AMDKFD_COMMAND_END 0x28 + +/* non-upstream ioctls */ +#define AMDKFD_IOC_IPC_IMPORT_HANDLE \ + AMDKFD_IOWR(0x80, struct kfd_ioctl_ipc_import_handle_args) + +#define AMDKFD_IOC_IPC_EXPORT_HANDLE \ + AMDKFD_IOWR(0x81, struct kfd_ioctl_ipc_export_handle_args) + +#define AMDKFD_IOC_CROSS_MEMORY_COPY \ + AMDKFD_IOWR(0x83, struct kfd_ioctl_cross_memory_copy_args) + +#define AMDKFD_IOC_RLC_SPM \ + AMDKFD_IOWR(0x84, struct kfd_ioctl_spm_args) + +#define AMDKFD_IOC_PC_SAMPLE \ + AMDKFD_IOWR(0x85, struct kfd_ioctl_pc_sample_args) + +#define AMDKFD_IOC_PROFILER \ + AMDKFD_IOWR(0x86, struct kfd_ioctl_profiler_args) + +#define AMDKFD_IOC_AIS_OP \ + AMDKFD_IOWR(0x87, struct kfd_ioctl_ais_args) + +#define AMDKFD_COMMAND_START_2 0x80 +#define AMDKFD_COMMAND_END_2 0x88 + + +#ifndef IOCPARM_MASK +#define IOCPARM_MASK 0x1fff +#endif +#define AMDKFD_MAX_IOCTL_SIZE IOCPARM_MASK + +#endif //KFD_IOCTL_H_INCLUDED diff --git a/projects/rocr-runtime/libhsakmt/include/hsakmt/freebsd/udmabuf.h b/projects/rocr-runtime/libhsakmt/include/hsakmt/freebsd/udmabuf.h new file mode 100644 index 0000000000..fb62e52891 --- /dev/null +++ b/projects/rocr-runtime/libhsakmt/include/hsakmt/freebsd/udmabuf.h @@ -0,0 +1,37 @@ +// Not sure what liscense yet +// Sourojeet Adhikari +// FreeBSD Foundation + + +#ifndef _THUNK_UDMABUF_H +#define _THUNK_UDMABUF_H + +#include +#include + +#define UDMABUF_FLAGS_CLOEXEC 0x01 + +struct udmabuf_create { + __uint32_t memfd; + __uint32_t flags; + __uint64_t offset; + __uint64_t size; +}; + +struct udmabuf_create_item { + __uint32_t memfd; + __uint32_t __pad; + __uint64_t offset; + __uint64_t size; +}; + +struct udmabuf_create_list { + __uint32_t flags; + __uint32_t count; + struct udmabuf_create_item list[]; +}; + +#define UDMABUF_CREATE _IOW('u', 0x42, struct udmabuf_create) +#define UDMABUF_CREATE_LIST _IOW('u', 0x43, struct udmabuf_create_list) + +#endif // _THUNK_UDMABUF_H diff --git a/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmt_virtio.h b/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmt_virtio.h index d3626c5f56..57053fbb7e 100644 --- a/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmt_virtio.h +++ b/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmt_virtio.h @@ -26,9 +26,7 @@ #ifndef HSAKMT_VIRTIO_H #define HSAKMT_VIRTIO_H -#if defined(__linux__) -#include "hsakmt/linux/kfd_ioctl.h" -#endif +#include // Forward declaration for HsaKFDContext to avoid dependency issues typedef struct _HsaKFDContext HsaKFDContext; diff --git a/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmttypes.h b/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmttypes.h index 77dd5804be..24a541836d 100644 --- a/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmttypes.h +++ b/projects/rocr-runtime/libhsakmt/include/hsakmt/hsakmttypes.h @@ -52,7 +52,7 @@ extern "C" { typedef signed __int64 HSAint64; typedef unsigned __int64 HSAuint64; -#elif defined(__linux__) +#elif defined(__linux__) || defined(__FreeBSD__) #include #include @@ -70,6 +70,7 @@ extern "C" { #endif + typedef void* HSA_HANDLE; typedef HSAuint64 HSA_QUEUEID; // An HSA_QUEUEID that is never a valid queue ID. diff --git a/projects/rocr-runtime/libhsakmt/include/hsakmt/linux/kfd_ioctl.h b/projects/rocr-runtime/libhsakmt/include/hsakmt/linux/kfd_ioctl.h index d173e65c11..7f18e57c5b 100644 --- a/projects/rocr-runtime/libhsakmt/include/hsakmt/linux/kfd_ioctl.h +++ b/projects/rocr-runtime/libhsakmt/include/hsakmt/linux/kfd_ioctl.h @@ -1853,5 +1853,9 @@ struct kfd_ioctl_create_process_args { #define AMDKFD_COMMAND_START_2 0x80 #define AMDKFD_COMMAND_END_2 0x88 +#ifndef _IOC_SIZEMASK +#define _IOC_SIZEMASK ((1 << _IOC_SIZEBITS) - 1) +#endif +#define AMDKFD_MAX_IOCTL_SIZE _IOC_SIZEMASK #endif diff --git a/projects/rocr-runtime/libhsakmt/src/ais.c b/projects/rocr-runtime/libhsakmt/src/ais.c index dc7f6a1abd..661147e565 100644 --- a/projects/rocr-runtime/libhsakmt/src/ais.c +++ b/projects/rocr-runtime/libhsakmt/src/ais.c @@ -24,7 +24,7 @@ */ #include "libhsakmt.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include #include "fmm.h" diff --git a/projects/rocr-runtime/libhsakmt/src/debug.c b/projects/rocr-runtime/libhsakmt/src/debug.c index 043dc1a036..c12e7f1501 100644 --- a/projects/rocr-runtime/libhsakmt/src/debug.c +++ b/projects/rocr-runtime/libhsakmt/src/debug.c @@ -24,7 +24,7 @@ */ #include "libhsakmt.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include #include #include #include diff --git a/projects/rocr-runtime/libhsakmt/src/events.c b/projects/rocr-runtime/libhsakmt/src/events.c index 1877b70275..f59f9a84a9 100644 --- a/projects/rocr-runtime/libhsakmt/src/events.c +++ b/projects/rocr-runtime/libhsakmt/src/events.c @@ -29,7 +29,7 @@ #include #include #include -#include "hsakmt/linux/kfd_ioctl.h" +#include #include "fmm.h" #include "hsakmt/hsakmtmodel.h" #include diff --git a/projects/rocr-runtime/libhsakmt/src/fmm.c b/projects/rocr-runtime/libhsakmt/src/fmm.c index 27ddeea481..418b3ee37d 100644 --- a/projects/rocr-runtime/libhsakmt/src/fmm.c +++ b/projects/rocr-runtime/libhsakmt/src/fmm.c @@ -27,7 +27,7 @@ #include "libhsakmt.h" #include "fmm.h" #include "hsakmt/hsakmtmodel.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include #include #include #include @@ -46,7 +46,7 @@ #include #include -#include "hsakmt/linux/udmabuf.h" +#include #ifndef MPOL_F_STATIC_NODES /* Bug in numaif.h, this should be defined in there. Definition copied diff --git a/projects/rocr-runtime/libhsakmt/src/freebsd/perfctr.c b/projects/rocr-runtime/libhsakmt/src/freebsd/perfctr.c new file mode 100644 index 0000000000..b1cfe8b7e4 --- /dev/null +++ b/projects/rocr-runtime/libhsakmt/src/freebsd/perfctr.c @@ -0,0 +1,66 @@ +#include "libhsakmt.h" + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcGetCounterProperties(HSAuint32 NodeId, + HsaCounterProperties **CounterProperties) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcAcquireTraceAccessCtx(HsaKFDContext* ctx, + HSAuint32 NodeId, + HSATraceId TraceId) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcAcquireTraceAccess(HSAuint32 NodeId, + HSATraceId TraceId) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcReleaseTraceAccess(HSAuint32 NodeId, + HSATraceId TraceId) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcStartTrace(HSATraceId TraceId, + void *TraceBuffer, + HSAuint64 TraceBufferMemorySize) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcQueryTrace(HSATraceId TraceId) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcStopTrace(HSATraceId TraceId) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcRegisterTrace(HSAuint32 NodeId, + HSAuint32 NumberOfCounters, + HsaCounter *Counters, + HsaPmcTraceRoot *TraceRoot) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcUnregisterTrace(HSAuint32 NodeId, + HSATraceId TraceId) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} diff --git a/projects/rocr-runtime/libhsakmt/src/kfdcontext.h b/projects/rocr-runtime/libhsakmt/src/kfdcontext.h index 6d0e60d584..c8f13586b1 100644 --- a/projects/rocr-runtime/libhsakmt/src/kfdcontext.h +++ b/projects/rocr-runtime/libhsakmt/src/kfdcontext.h @@ -26,8 +26,7 @@ #ifndef _KFDCONTEXT_H_ #define _KFDCONTEXT_H_ -#include -#include +#include "hsakmt/hsakmtctx.h" struct hsa_kfd_topology_context; struct hsa_kfd_queue_context; @@ -51,7 +50,7 @@ struct hsa_kfd_perf_context; * context A cannot be used in context B directly. If resources need to be shared between * contexts, they must be explicitly exported and imported using the appropriate APIs. */ -typedef struct _HsaKFDContext +struct _HsaKFDContext { /* File descriptor for the KFD device */ int fd; @@ -86,7 +85,7 @@ typedef struct _HsaKFDContext /* perf context for managing perf operations */ struct hsa_kfd_perf_context *perf_context; -} HsaKFDContext; +}; /* Initialize a pre-allocated HsaKFDContext with the given fd. * Returns 0 on success, -1 on allocation failure. diff --git a/projects/rocr-runtime/libhsakmt/src/libhsakmt.h b/projects/rocr-runtime/libhsakmt/src/libhsakmt.h index 3e948f8023..056846c506 100644 --- a/projects/rocr-runtime/libhsakmt/src/libhsakmt.h +++ b/projects/rocr-runtime/libhsakmt/src/libhsakmt.h @@ -26,7 +26,7 @@ #ifndef LIBHSAKMT_H_INCLUDED #define LIBHSAKMT_H_INCLUDED -#include "hsakmt/linux/kfd_ioctl.h" +#include #include "hsakmt/hsakmt.h" #include "kfdcontext.h" #include "hsakmt/hsakmtctx.h" diff --git a/projects/rocr-runtime/libhsakmt/src/libhsakmt.ver b/projects/rocr-runtime/libhsakmt/src/libhsakmt.ver index d8e9c9491b..5120cd348a 100644 --- a/projects/rocr-runtime/libhsakmt/src/libhsakmt.ver +++ b/projects/rocr-runtime/libhsakmt/src/libhsakmt.ver @@ -19,6 +19,7 @@ hsaKmtQueryEventState; hsaKmtWaitOnEvent; hsaKmtWaitOnMultipleEvents; hsaKmtCreateQueue; +hsaKmtCreateQueueExt; hsaKmtUpdateQueue; hsaKmtDestroyQueue; hsaKmtSetQueueCUMask; @@ -31,6 +32,7 @@ hsaKmtRegisterMemory; hsaKmtRegisterMemoryToNodes; hsaKmtRegisterMemoryWithFlags; hsaKmtRegisterGraphicsHandleToNodes; +hsaKmtRegisterGraphicsHandleToNodesExt; hsaKmtShareMemory; hsaKmtRegisterSharedHandle; hsaKmtRegisterSharedHandleToNodes; @@ -100,6 +102,107 @@ hsaKmtMemoryVaUnmap; hsaKmtMemHandleFree; hsaKmtMemoryGetCpuAddr; hsaKmtMemoryCpuMap; +hsaKmtRegisterGraphicsHandleToNodesExt; +hsaKmtOpenKFD; +hsaKmtCloseKFD; +hsaKmtGetVersion; +hsaKmtAcquireSystemProperties; +hsaKmtReleaseSystemProperties; +hsaKmtGetNodeProperties; +hsaKmtGetNodeMemoryProperties; +hsaKmtGetNodeCacheProperties; +hsaKmtGetNodeIoLinkProperties; +hsaKmtCreateEvent; +hsaKmtDestroyEvent; +hsaKmtSetEvent; +hsaKmtResetEvent; +hsaKmtQueryEventState; +hsaKmtWaitOnEvent; +hsaKmtWaitOnMultipleEvents; +hsaKmtCreateQueue; +hsaKmtCreateQueueExt; +hsaKmtCreateQueueV2; +hsaKmtUpdateQueue; +hsaKmtDestroyQueue; +hsaKmtSetQueueCUMask; +hsaKmtSetMemoryPolicy; +hsaKmtAllocMemory; +hsaKmtAllocMemoryAlign; +hsaKmtFreeMemory; +hsaKmtAvailableMemory; +hsaKmtRegisterMemory; +hsaKmtRegisterMemoryToNodes; +hsaKmtRegisterMemoryWithFlags; +hsaKmtRegisterGraphicsHandleToNodes; +hsaKmtShareMemory; +hsaKmtRegisterSharedHandle; +hsaKmtRegisterSharedHandleToNodes; +hsaKmtProcessVMRead; +hsaKmtProcessVMWrite; +hsaKmtDeregisterMemory; +hsaKmtMapMemoryToGPU; +hsaKmtMapMemoryToGPUNodes; +hsaKmtUnmapMemoryToGPU; +hsaKmtDbgRegister; +hsaKmtDbgUnregister; +hsaKmtDbgWavefrontControl; +hsaKmtDbgAddressWatch; +hsaKmtDbgEnable; +hsaKmtDbgDisable; +hsaKmtDbgGetDeviceData; +hsaKmtDbgGetQueueData; +hsaKmtGetClockCounters; +hsaKmtPmcGetCounterProperties; +hsaKmtPmcRegisterTrace; +hsaKmtPmcUnregisterTrace; +hsaKmtPmcAcquireTraceAccess; +hsaKmtPmcReleaseTraceAccess; +hsaKmtPmcStartTrace; +hsaKmtPmcQueryTrace; +hsaKmtPmcStopTrace; +hsaKmtMapGraphicHandle; +hsaKmtUnmapGraphicHandle; +hsaKmtSetTrapHandler; +hsaKmtGetTileConfig; +hsaKmtQueryPointerInfo; +hsaKmtSetMemoryUserData; +hsaKmtGetQueueInfo; +hsaKmtAllocQueueGWS; +hsaKmtRuntimeEnable; +hsaKmtRuntimeDisable; +hsaKmtCheckRuntimeDebugSupport; +hsaKmtGetRuntimeCapabilities; +hsaKmtDebugTrapIoctl; +hsaKmtSPMAcquire; +hsaKmtSPMRelease; +hsaKmtSPMSetDestBuffer; +hsaKmtSVMSetAttr; +hsaKmtSVMGetAttr; +hsaKmtSetXNACKMode; +hsaKmtGetXNACKMode; +hsaKmtOpenSMI; +hsaKmtExportDMABufHandle; +hsaKmtWaitOnEvent_Ext; +hsaKmtWaitOnMultipleEvents_Ext; +hsaKmtReplaceAsanHeaderPage; +hsaKmtReturnAsanHeaderPage; +hsaKmtGetAMDGPUDeviceHandle; +hsaKmtPcSamplingQueryCapabilities; +hsaKmtPcSamplingCreate; +hsaKmtPcSamplingDestroy; +hsaKmtPcSamplingStart; +hsaKmtPcSamplingStop; +hsaKmtPcSamplingSupport; +hsaKmtModelEnabled; +hsaKmtAisReadWriteFile; +hsaKmtHandleImport; +hsaKmtMemoryVaMap; +hsaKmtMemoryVaUnmap; +hsaKmtMemHandleFree; +hsaKmtMemoryGetCpuAddr; +hsaKmtMemoryCpuMap; +hsaKmtGetNodeWallclockFrequency; + local: *; }; diff --git a/projects/rocr-runtime/libhsakmt/src/memory.c b/projects/rocr-runtime/libhsakmt/src/memory.c index 939edb3926..4c28e566f6 100644 --- a/projects/rocr-runtime/libhsakmt/src/memory.c +++ b/projects/rocr-runtime/libhsakmt/src/memory.c @@ -24,7 +24,7 @@ */ #include "libhsakmt.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include #include #include #include @@ -801,7 +801,8 @@ HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterGraphicsHandleToNodes(HSAuint64 GraphicsRe } -HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterGraphicsHandleToNodesExt(HSAuint64 GraphicsResourceHandle, +// chicken nuggets +__attribute__((visibility("default"))) HSAKMT_STATUS HSAKMTAPI hsaKmtRegisterGraphicsHandleToNodesExt(HSAuint64 GraphicsResourceHandle, HsaGraphicsResourceInfo *GraphicsResourceInfo, HSAuint64 NumberOfNodes, HSAuint32 *NodeArray, @@ -1098,4 +1099,4 @@ HSAKMT_STATUS HSAKMTAPI hsaKmtMemoryGetCpuAddr(HsaAMDGPUDeviceHandle DeviceHandl *fd = (HSAint32)renderFd; *cpu_addr = (HSAuint64)args.out.addr_ptr; return HSAKMT_STATUS_SUCCESS; -} \ No newline at end of file +} diff --git a/projects/rocr-runtime/libhsakmt/src/pc_sampling.c b/projects/rocr-runtime/libhsakmt/src/pc_sampling.c index c36389ba55..3099b9c34e 100644 --- a/projects/rocr-runtime/libhsakmt/src/pc_sampling.c +++ b/projects/rocr-runtime/libhsakmt/src/pc_sampling.c @@ -24,7 +24,7 @@ */ #include "libhsakmt.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include #include #include diff --git a/projects/rocr-runtime/libhsakmt/src/perfctr.c b/projects/rocr-runtime/libhsakmt/src/perfctr.c index 39058914b9..407a02d09e 100644 --- a/projects/rocr-runtime/libhsakmt/src/perfctr.c +++ b/projects/rocr-runtime/libhsakmt/src/perfctr.c @@ -26,11 +26,13 @@ #include #include #include + + #include #include #include "libhsakmt.h" #include "pmc_table.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include #include #include #include @@ -749,4 +751,41 @@ HSAKMT_STATUS HSAKMTAPI hsaKmtPmcAcquireTraceAccess(HSAuint32 NodeId, { return hsaKmtPmcAcquireTraceAccessCtx(&hsakmt_primary_kfd_ctx, NodeId, TraceId); +} + +#else /* !defined(__linux__) */ + +#include "libhsakmt.h" + +HSAKMT_STATUS hsakmt_init_counter_props(HsaKFDContext *ctx, unsigned int NumNodes) +{ + return HSAKMT_STATUS_SUCCESS; +} + +void hsakmt_destroy_counter_props(HsaKFDContext *ctx) +{ +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcGetCounterPropertiesCtx(HsaKFDContext *ctx, + HSAuint32 NodeId, + HsaCounterProperties **CounterProperties) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcRegisterTraceCtx(HsaKFDContext* ctx, + HSAuint32 NodeId, + HSAuint32 NumberOfCounters, + HsaCounter *Counters, + HsaPmcTraceRoot *TraceRoot) +{ + pr_warn_once("not supported\n"); + return HSAKMT_STATUS_NOT_SUPPORTED; +} + +HSAKMT_STATUS HSAKMTAPI hsaKmtPmcUnregisterTraceCtx(HsaKFDContext* ctx, + HSAuint32 NodeId, + HSATraceId TraceId) +{ } \ No newline at end of file diff --git a/projects/rocr-runtime/libhsakmt/src/queues.c b/projects/rocr-runtime/libhsakmt/src/queues.c index 9112b0a086..bfe2fbdb21 100644 --- a/projects/rocr-runtime/libhsakmt/src/queues.c +++ b/projects/rocr-runtime/libhsakmt/src/queues.c @@ -25,7 +25,7 @@ #include "libhsakmt.h" #include "fmm.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include #include #include #include diff --git a/projects/rocr-runtime/libhsakmt/src/spm.c b/projects/rocr-runtime/libhsakmt/src/spm.c index d29d265311..c27c227478 100644 --- a/projects/rocr-runtime/libhsakmt/src/spm.c +++ b/projects/rocr-runtime/libhsakmt/src/spm.c @@ -24,7 +24,7 @@ */ #include "libhsakmt.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include HSAKMT_STATUS HSAKMTAPI hsaKmtSPMAcquire(HSAuint32 PreferredNode) diff --git a/projects/rocr-runtime/libhsakmt/src/svm.c b/projects/rocr-runtime/libhsakmt/src/svm.c index b461ad82bf..d7a13b785a 100644 --- a/projects/rocr-runtime/libhsakmt/src/svm.c +++ b/projects/rocr-runtime/libhsakmt/src/svm.c @@ -25,7 +25,10 @@ #include "libhsakmt.h" #include #include +#include +#if defined(__linux__) // what the fuck #include +#endif #include #include #include @@ -57,6 +60,10 @@ hsaKmtSVMSetAttrCtx(HsaKFDContext *ctx, return HSAKMT_STATUS_INVALID_PARAMETER; s_attr = sizeof(*attrs) * nattr; + + if ((sizeof(*args) + s_attr) > AMDKFD_MAX_IOCTL_SIZE) + return HSAKMT_STATUS_INVALID_PARAMETER; + args = alloca(sizeof(*args) + s_attr); args->start_addr = (uint64_t)start_addr; @@ -125,6 +132,10 @@ hsaKmtSVMGetAttrCtx(HsaKFDContext *ctx, return HSAKMT_STATUS_INVALID_PARAMETER; s_attr = sizeof(*attrs) * nattr; + + if ((sizeof(*args) + s_attr) > AMDKFD_MAX_IOCTL_SIZE) + return HSAKMT_STATUS_INVALID_PARAMETER; + args = alloca(sizeof(*args) + s_attr); args->start_addr = (uint64_t)start_addr; diff --git a/projects/rocr-runtime/libhsakmt/src/test_ioctl_mask b/projects/rocr-runtime/libhsakmt/src/test_ioctl_mask new file mode 100755 index 0000000000..e4ecdbd789 Binary files /dev/null and b/projects/rocr-runtime/libhsakmt/src/test_ioctl_mask differ diff --git a/projects/rocr-runtime/libhsakmt/src/test_ioctl_mask.c b/projects/rocr-runtime/libhsakmt/src/test_ioctl_mask.c new file mode 100644 index 0000000000..61db1e6839 --- /dev/null +++ b/projects/rocr-runtime/libhsakmt/src/test_ioctl_mask.c @@ -0,0 +1,12 @@ +#include +#include + +int main() { +#ifdef IOCPARM_MASK + printf("IOCPARM_MASK: %d\n", IOCPARM_MASK); +#endif +#ifdef _IOC_SIZEMASK + printf("_IOC_SIZEMASK: %d\n", _IOC_SIZEMASK); +#endif + return 0; +} diff --git a/projects/rocr-runtime/libhsakmt/src/time.c b/projects/rocr-runtime/libhsakmt/src/time.c index 222d53f32a..951e7375fc 100644 --- a/projects/rocr-runtime/libhsakmt/src/time.c +++ b/projects/rocr-runtime/libhsakmt/src/time.c @@ -24,7 +24,7 @@ */ #include "libhsakmt.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include HSAKMT_STATUS HSAKMTAPI hsaKmtGetClockCountersCtx(HsaKFDContext *ctx, HSAuint32 NodeId, diff --git a/projects/rocr-runtime/libhsakmt/src/topology.c b/projects/rocr-runtime/libhsakmt/src/topology.c index b622dfad2c..eb4f3e4aa2 100644 --- a/projects/rocr-runtime/libhsakmt/src/topology.c +++ b/projects/rocr-runtime/libhsakmt/src/topology.c @@ -35,7 +35,7 @@ #include #include -#include + #include #include #include @@ -2037,7 +2037,7 @@ HSAKMT_STATUS topology_take_snapshot(HsaKFDContext *ctx) node_props_t *temp_props = 0; HSAKMT_STATUS ret = HSAKMT_STATUS_SUCCESS; struct proc_cpuinfo *cpuinfo; - const uint32_t num_procs = get_nprocs(); + const uint32_t num_procs = sysconf(_SC_NPROCESSORS_ONLN); uint32_t num_ioLinks; bool p2p_links = false; uint32_t num_p2pLinks = 0; diff --git a/projects/rocr-runtime/libhsakmt/src/version.c b/projects/rocr-runtime/libhsakmt/src/version.c index d682c5341d..fe297285b5 100644 --- a/projects/rocr-runtime/libhsakmt/src/version.c +++ b/projects/rocr-runtime/libhsakmt/src/version.c @@ -24,7 +24,7 @@ */ #include "libhsakmt.h" -#include "hsakmt/linux/kfd_ioctl.h" +#include HsaVersionInfo hsakmt_kfd_version_info; diff --git a/projects/rocr-runtime/libhsakmt/src/virtio/hsakmt_virtio_proto.h b/projects/rocr-runtime/libhsakmt/src/virtio/hsakmt_virtio_proto.h index ee23ea7be5..055fbeadcb 100644 --- a/projects/rocr-runtime/libhsakmt/src/virtio/hsakmt_virtio_proto.h +++ b/projects/rocr-runtime/libhsakmt/src/virtio/hsakmt_virtio_proto.h @@ -23,7 +23,7 @@ #ifndef VHSAKMT_VIRTIO_PROTO_H #define VHSAKMT_VIRTIO_PROTO_H -#include "hsakmt/linux/kfd_ioctl.h" +#include // Forward declaration for HsaKFDContext to avoid dependency issues typedef struct _HsaKFDContext HsaKFDContext; diff --git a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/BaseDebug.cpp b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/BaseDebug.cpp index 68a2fb8588..ddee3176be 100644 --- a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/BaseDebug.cpp +++ b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/BaseDebug.cpp @@ -26,7 +26,7 @@ #include #include #include -#include +#include #include #include "unistd.h" diff --git a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDDBGTest.cpp b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDDBGTest.cpp index f9a00b9ab7..caaa16f8e0 100644 --- a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDDBGTest.cpp +++ b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDDBGTest.cpp @@ -25,7 +25,7 @@ #include "KFDDBGTest.hpp" #include #include -#include "hsakmt/linux/kfd_ioctl.h" +#include #include "KFDQMTest.hpp" #include "PM4Queue.hpp" #include "PM4Packet.hpp" diff --git a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDMemoryTest.cpp b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDMemoryTest.cpp index cad1dc948a..8f86c2e9b0 100644 --- a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDMemoryTest.cpp +++ b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDMemoryTest.cpp @@ -37,7 +37,7 @@ #include "PM4Packet.hpp" #include "SDMAQueue.hpp" #include "SDMAPacket.hpp" -#include "hsakmt/linux/kfd_ioctl.h" +#include /* Captures user specified time (seconds) to sleep */ extern unsigned int g_SleepTime; diff --git a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDPCSamplingTest.cpp b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDPCSamplingTest.cpp index 5f43b1d361..d4e75aa868 100644 --- a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDPCSamplingTest.cpp +++ b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDPCSamplingTest.cpp @@ -37,7 +37,7 @@ #include "PM4Packet.hpp" #include "SDMAQueue.hpp" #include "SDMAPacket.hpp" -#include "hsakmt/linux/kfd_ioctl.h" +#include #define N_PROCESSES (2) /* Number of processes running in parallel, must be at least 2 */ diff --git a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDRASTest.cpp b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDRASTest.cpp index d39e801d4e..e4be0a6ad5 100644 --- a/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDRASTest.cpp +++ b/projects/rocr-runtime/libhsakmt/tests/kfdtest/src/KFDRASTest.cpp @@ -24,7 +24,7 @@ #include #include -#include "hsakmt/linux/kfd_ioctl.h" +#include #include "KFDRASTest.hpp" #include "PM4Queue.hpp" diff --git a/projects/rocr-runtime/runtime/hsa-runtime/CMakeLists.txt b/projects/rocr-runtime/runtime/hsa-runtime/CMakeLists.txt index bae0f4b3f1..532cbb9212 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/CMakeLists.txt +++ b/projects/rocr-runtime/runtime/hsa-runtime/CMakeLists.txt @@ -49,7 +49,7 @@ unset ( hsa-runtime64_LIB_DEPENDS CACHE ) set(CMAKE_VERBOSE_MAKEFILE ON) if (UNIX) - set(CMAKE_CXX_STANDARD 17) + set(CMAKE_CXX_STANDARD 20) else() set(CMAKE_CXX_STANDARD 20) endif() @@ -137,7 +137,12 @@ set_property(TARGET ${CORE_RUNTIME_TARGET} PROPERTY OUTPUT_NAME ${CORE_RUNTIME_ ## Compiler preproc definitions. if (UNIX) - target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" __linux__ HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED= + if(CMAKE_SYSTEM_NAME MATCHES "FreeBSD") + set(OS_DEFS __FreeBSD__) + else() + set(OS_DEFS __linux__) + endif() + target_compile_definitions(${CORE_RUNTIME_TARGET} PRIVATE "${HSA_COMMON_DEFS}" ${OS_DEFS} HSA_EXPORT=1 HSA_EXPORT_FINALIZER=1 HSA_EXPORT_IMAGES=1 HSA_DEPRECATED= ROCR_BUILD_ID="${PACKAGE_VERSION_STRING}-${VERSION_JOB}-${VERSION_HASH}" ) ## Check for memfd_create syscall @@ -255,8 +260,13 @@ set ( SRCS core/driver/driver.cpp libamdhsacode/amd_hsa_code.cpp) if(UNIX) - set(SRC_OS core/util/lnx/os_linux.cpp - libamdhsacode/lnx/amd_core_dump.cpp) + if(CMAKE_SYSTEM_NAME MATCHES "FreeBSD") + set(SRC_OS core/util/freebsd/os_freebsd.cpp + libamdhsacode/lnx/amd_core_dump.cpp) + else() + set(SRC_OS core/util/lnx/os_linux.cpp + libamdhsacode/lnx/amd_core_dump.cpp) + endif() set(SRC_XDNA core/driver/xdna/amd_xdna_driver.cpp core/runtime/amd_aie_agent.cpp core/runtime/amd_aie_aql_queue.cpp) diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/driver/kfd/amd_kfd_driver.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/driver/kfd/amd_kfd_driver.cpp index 9c34358b3b..92f3853ae7 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/driver/kfd/amd_kfd_driver.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/driver/kfd/amd_kfd_driver.cpp @@ -46,6 +46,9 @@ #include #include #include +#elif defined(__FreeBSD__) +#include +#include #endif #include "hsakmt/hsakmt.h" diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/amd_xdna_driver.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/amd_xdna_driver.cpp index 382c1d10c4..d4475cdd24 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/amd_xdna_driver.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/amd_xdna_driver.cpp @@ -66,6 +66,10 @@ #include "core/util/utils.h" #include "uapi/amdxdna_accel.h" +#ifdef __FreeBSD__ +#include +#endif + namespace rocr { namespace AMD { diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/uapi/amdxdna_accel.h b/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/uapi/amdxdna_accel.h index 6b20848020..9d6704b397 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/uapi/amdxdna_accel.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/driver/xdna/uapi/amdxdna_accel.h @@ -11,8 +11,14 @@ #else #include #endif + +#ifdef __linux__ #include #include +#elif defined(__FreeBSD__) +#include +#include +#endif #if defined(__cplusplus) extern "C" { diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_hsa_loader.hpp b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_hsa_loader.hpp index 3460b83f24..3dff550fca 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_hsa_loader.hpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/inc/amd_hsa_loader.hpp @@ -75,6 +75,50 @@ /// @brief Descriptive version of the AMD HSA Loader. #define AMD_HSA_LOADER_VERSION "AMD HSA Loader v0.05 (June 16, 2015)" + +#ifndef ElfW + #if defined(__LP64__) || defined(_LP64) + #define ElfW(type) Elf64_##type + #else + #define ElfW(type) Elf32_##type + #endif +#endif + +#ifndef PT_LOAD + #define PT_LOAD 1 +#endif + +#ifndef _GNU_SOURCE + #define _GNU_SOURCE +#endif +#ifndef _BSD_SOURCE + #define _BSD_SOURCE 1 +#endif + +#include +#include +#include + +#ifndef ElfW + #if defined(__LP64__) || defined(_LP64) + #define ElfW(type) Elf64_##type + #else + #define ElfW(type) Elf32_##type + #endif +#endif + +#ifndef Elf64_Addr + typedef uint64_t Elf64_Addr; +#endif +#ifndef Elf32_Addr + typedef uint32_t Elf32_Addr; +#endif + +#ifndef PT_LOAD + #define PT_LOAD 1 +#endif + + enum hsa_ext_symbol_info_t { HSA_EXT_EXECUTABLE_SYMBOL_INFO_KERNEL_OBJECT_SIZE = 100, HSA_EXT_EXECUTABLE_SYMBOL_INFO_KERNEL_OBJECT_ALIGN = 101, diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp index e7a4dd086b..633a98f0ee 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_aql_queue.cpp @@ -673,7 +673,11 @@ int AqlQueue::CreateRingBufferFD(const char* ring_buf_shm_path, #ifdef __linux__ int fd; #ifdef HAVE_MEMFD_CREATE +#ifdef __FreeBSD__ + fd = memfd_create(ring_buf_shm_path, 0); +#else fd = syscall(__NR_memfd_create, ring_buf_shm_path, 0); +#endif if (fd == -1) return -1; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp index 62366e1dc5..793b7679c8 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_blit_kernel.cpp @@ -47,6 +47,11 @@ #include #include +#ifdef __FreeBSD__ +#include +#include +#endif + #include "core/inc/amd_gpu_agent.h" #include "core/inc/hsa_internal.h" #include "core/util/utils.h" diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_hsa_loader.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_hsa_loader.cpp index 97fe9be160..c73b1b31a3 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_hsa_loader.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_hsa_loader.cpp @@ -60,6 +60,49 @@ #include #include +#ifndef ElfW + #if defined(__LP64__) || defined(_LP64) + #define ElfW(type) Elf64_##type + #else + #define ElfW(type) Elf32_##type + #endif +#endif + +#ifndef PT_LOAD + #define PT_LOAD 1 +#endif + +#ifndef _GNU_SOURCE + #define _GNU_SOURCE +#endif +#ifndef _BSD_SOURCE + #define _BSD_SOURCE 1 +#endif + +#include +#include +#include + +#ifndef ElfW + #if defined(__LP64__) || defined(_LP64) + #define ElfW(type) Elf64_##type + #else + #define ElfW(type) Elf32_##type + #endif +#endif + +#ifndef Elf64_Addr + typedef uint64_t Elf64_Addr; +#endif +#ifndef Elf32_Addr + typedef uint32_t Elf32_Addr; +#endif + +#ifndef PT_LOAD + #define PT_LOAD 1 +#endif + + namespace { #if !defined(_WIN32) && !defined(_WIN64) diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_topology.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_topology.cpp index 78095dc250..1a98d73e0f 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_topology.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/amd_topology.cpp @@ -84,11 +84,16 @@ namespace { const std::array&)>, #if _WIN32 - 1 + 1> #elif __linux__ - static_cast(core::DriverType::NUM_DRIVER_TYPES) + static_cast(core::DriverType::NUM_DRIVER_TYPES)> +#elif __FreeBSD__ + static_cast(core::DriverType::NUM_DRIVER_TYPES)> + #endif - > + + + discover_driver_funcs = { KfdDriver::DiscoverDriver #ifdef __linux__ diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/runtime.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/runtime.cpp index 4f6c4b284c..b3fb0e375a 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/runtime.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/runtime.cpp @@ -2375,12 +2375,11 @@ Runtime::Runtime() xnack_enabled_ = false; g_use_interrupt_wait = true; g_use_mwaitx = true; - ::_amdgpu_r_debug = {11, - nullptr, - reinterpret_cast( - &_loader_debug_state), - r_debug::RT_CONSISTENT, - 0}; + ::_amdgpu_r_debug.r_version = 11; + ::_amdgpu_r_debug.r_map = nullptr; + ::_amdgpu_r_debug.r_brk = reinterpret_cast(&_loader_debug_state); + ::_amdgpu_r_debug.r_state = r_debug::RT_CONSISTENT; + ::_amdgpu_r_debug.r_ldbase = 0; log_file = stderr; } diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/thunk_loader.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/thunk_loader.cpp index d48e4641c9..76d09a6acd 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/thunk_loader.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/thunk_loader.cpp @@ -77,7 +77,12 @@ namespace core { is_dxg_ = true; #endif +#if defined(__FreeBSD__) + is_dtif_ = is_dxg_ = false; + return "libhsakmt.so"; +#else return ""; +#endif } ThunkLoader::ThunkLoader() diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/trap_handler/create_trap_handler_header.sh b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/trap_handler/create_trap_handler_header.sh index abd12acb8a..3d424783dd 100755 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/trap_handler/create_trap_handler_header.sh +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/runtime/trap_handler/create_trap_handler_header.sh @@ -1,4 +1,5 @@ -#!/bin/bash -e +#!/usr/local/bin/ksh +set -e ################################################################################ ## ## The University of Illinois/NCSA diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/util/freebsd/os_freebsd.cpp b/projects/rocr-runtime/runtime/hsa-runtime/core/util/freebsd/os_freebsd.cpp new file mode 100644 index 0000000000..a683c132ad --- /dev/null +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/util/freebsd/os_freebsd.cpp @@ -0,0 +1,989 @@ +// Sourojeet Adhikari +// Not sure what Liscense to use +// FreeBSD + + +#ifdef __FreeBSD__ +#include "core/util/os.h" +#include "core/util/utils.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include "core/inc/runtime.h" +#include +#include +#include +#if defined(__i386__) || defined(__x86_64__) +#include +#endif + +#ifndef CLOCK_BOOTTIME +#ifdef CLOCK_UPTIME +#define CLOCK_BOOTTIME CLOCK_UPTIME +#else +#define CLOCK_BOOTTIME CLOCK_MONOTONIC +#endif +#endif + +#ifndef MAP_NORESERVE +#define MAP_NORESERVE 0 +#endif + +#ifndef MADV_HUGEPAGE +#define MADV_HUGEPAGE 0 +#endif + +#ifdef __GLIBC__ +#define ABS_ADDR(base, ptr) (ptr) +#else +#define ABS_ADDR(base, ptr) ((base) + (ptr)) +#endif + +namespace rocr { +namespace os { + +struct ThreadArgs { + void* entry_args; + ThreadEntry entry_function; +}; + +void* __stdcall ThreadTrampoline(void* arg) { + ThreadArgs* ar = (ThreadArgs*)arg; + ThreadEntry CallMe = ar->entry_function; + void* Data = ar->entry_args; + CallMe(Data); + return nullptr; +} + +// Thread container allows multiple waits and separate close (destroy). +class os_thread { + public: + explicit os_thread(ThreadEntry function, + void* threadArgument, + uint stackSize, + int priority) + : thread(0), lock(nullptr), state(RUNNING) { + int err; + lock = CreateMutex(); + if (lock == nullptr) return; + + args.entry_args = threadArgument; + args.entry_function = function; + + pthread_attr_t attrib; + err = pthread_attr_init(&attrib); + if (err != 0) { + fprintf(stderr, "pthread_attr_init failed: %s\n", strerror(err)); + return; + } + + MAKE_SCOPE_GUARD([&]() { + if (pthread_attr_destroy(&attrib)) + fprintf(stderr, "pthread_attr_destroy failed: %s\n", strerror(err)); + }); + + if (stackSize != 0) { + stackSize = Max(uint(PTHREAD_STACK_MIN), stackSize); + stackSize = AlignUp(stackSize, 4096); + err = pthread_attr_setstacksize(&attrib, stackSize); + if (err != 0) { + fprintf(stderr, "pthread_attr_setstacksize failed: %s\n", strerror(err)); + return; + } + } + + int cores = 0; + cpuset_t cpuset; + + if (core::Runtime::runtime_singleton_->flag().override_cpu_affinity()) { + cores = sysconf(_SC_NPROCESSORS_CONF); + CPU_ZERO(&cpuset); + for (int i = 0; i < cores; i++) { + CPU_SET(i, &cpuset); + } +#ifdef HAVE_PTHREAD_ATTR_SETAFFINITY_NP + err = pthread_attr_setaffinity_np(&attrib, sizeof(cpuset), &cpuset); + if (err != 0) { + fprintf(stderr, "pthread_attr_setaffinity_np failed: %s\n", strerror(err)); + return; + } +#endif + } + + do { + err = pthread_create(&thread, &attrib, ThreadTrampoline, &args); + if (!err) break; + + if (err != EINVAL || stackSize == 0) { + fprintf(stderr, "pthread_create failed %d (%s)\n", errno, strerror(errno)); + thread = 0; + return; + } + + // Probably a stack size error since system limits can be different from PTHREAD_STACK_MIN + // Attempt to grow the stack within reason. + stackSize *= 2; + if (pthread_attr_setstacksize(&attrib, stackSize)) { + fprintf(stderr, "pthread_attr_setstacksize failed: %s\n", strerror(err)); + thread = 0; + return; + } + } while (stackSize < 20 * 1024 * 1024); + +#ifndef HAVE_PTHREAD_ATTR_SETAFFINITY_NP + if (cores) { + err = pthread_setaffinity_np(thread, sizeof(cpuset), &cpuset); + if (err != 0) { + fprintf(stderr, "pthread_setaffinity_np failed: %s\n", strerror(err)); + thread = 0; + return; + } + } +#endif + struct sched_param param = {}; + if (priority != OS_THREAD_PRIORITY_DEFAULT) { + int set_priority; + int max_priority = sched_get_priority_max(SCHED_FIFO); + + if (priority == OS_THREAD_PRIORITY_MAX) + set_priority = max_priority; + else if (priority == OS_THREAD_PRIORITY_HIGH) + set_priority = max_priority - 1; + else if (priority > max_priority) + set_priority = max_priority; + else + set_priority = priority; + + param.sched_priority = set_priority; + if (pthread_setschedparam(thread, SCHED_FIFO, ¶m)) { + fprintf(stderr, "pthread_setschedparam failed\n"); + return; + } + + int policy = 0; + if (pthread_getschedparam(thread, &policy, ¶m)) + fprintf(stderr, "pthread_getschedparam failed: %s\n", strerror(err)); + + if (policy != SCHED_FIFO || param.sched_priority != set_priority) + fprintf(stderr, "Failed to adjust thread priority (policy:%s requested:%d current:%d)\n", + policy == SCHED_FIFO ? "FIFO" : + policy == SCHED_OTHER ? "OTHER" : + policy == SCHED_RR ? "RR" : "Unknown", + set_priority, param.sched_priority); + } + } + + os_thread(os_thread&& rhs) { + thread = rhs.thread; + args = rhs.args; + lock = rhs.lock; + state = int(rhs.state); + rhs.thread = 0; + rhs.lock = nullptr; + } + + os_thread(os_thread&) = delete; + + ~os_thread() { + if (lock != nullptr) DestroyMutex(lock); + if ((state == RUNNING) && (thread != 0)) { + int err = pthread_detach(thread); + if (err != 0) fprintf(stderr, "pthread_detach failed: %s\n", strerror(err)); + } + } + + bool Valid() { return (lock != nullptr) && (thread != 0); } + + bool Wait() { + if (state == FINISHED) return true; + AcquireMutex(lock); + if (state == FINISHED) { + ReleaseMutex(lock); + return true; + } + int err = pthread_join(thread, NULL); + bool success = (err == 0); + if (success) state = FINISHED; + ReleaseMutex(lock); + return success; + } + + private: + pthread_t thread; + struct ThreadArgs args; + Mutex lock; + std::atomic state; + enum { FINISHED = 0, RUNNING = 1 }; +}; + +static_assert(sizeof(LibHandle) == sizeof(void*), "OS abstraction size mismatch"); +static_assert(sizeof(Semaphore) == sizeof(sem_t*), "OS abstraction size mismatch"); +static_assert(sizeof(Mutex) == sizeof(pthread_mutex_t*), "OS abstraction size mismatch"); +static_assert(sizeof(SharedMutex) == sizeof(pthread_rwlock_t*), "OS abstraction size mismatch"); +static_assert(sizeof(Thread) == sizeof(os_thread*), "OS abstraction size mismatch"); + +LibHandle LoadLib(std::string filename) { + int dlopen_flags = RTLD_LAZY; +#ifdef RTLD_NODELETE + dlopen_flags |= RTLD_NODELETE; +#endif + void* ret = dlopen(filename.c_str(), dlopen_flags); + if (ret == nullptr) debug_print("LoadLib(%s) failed: %s\n", filename.c_str(), dlerror()); + return ret; +} + +void* GetExportAddress(LibHandle lib, std::string export_name) { + void* ret = dlsym(*(void**)&lib, export_name.c_str()); + + if (ret == NULL) return ret; + + link_map* map; + int err = dlinfo(*(void**)&lib, RTLD_DI_LINKMAP, &map); + if (err == -1) { + fprintf(stderr, "dlinfo failed: %s\n", dlerror()); + return nullptr; + } + + Dl_info info; + err = dladdr(ret, &info); + if (err == 0) { + fprintf(stderr, "dladdr failed.\n"); + return nullptr; + } + + if (strcmp(info.dli_fname, map->l_name) == 0) return ret; + + return NULL; +} + +bool CloseLib(LibHandle lib) { return (dlclose(*(void**)&lib) == 0) ? true : false; } + +#if defined(__has_attribute) +#if __has_attribute(no_sanitize) +__attribute__((no_sanitize("address"))) +#endif +#endif +static int callback(struct dl_phdr_info* info, size_t size, void* data) { + std::vector* loadedToolsLib = (std::vector*)data; + assert(loadedToolsLib != nullptr); + + if ((info) && (info->dlpi_name) && (info->dlpi_name[0] != '\0')) { + if (std::string(info->dlpi_name).find("vdso.so") != std::string::npos) return 0; + + for (int i = 0; i < info->dlpi_phnum; i++) { + if (info->dlpi_phdr[i].p_type == PT_DYNAMIC) { + Elf64_Dyn* dyn_section = (Elf64_Dyn*)(info->dlpi_addr + info->dlpi_phdr[i].p_vaddr); + + char* strings = nullptr; + Elf64_Xword limit = 0; + + for (int j = 0;; j++) { + if (dyn_section[j].d_tag == DT_NULL) break; + + if (dyn_section[j].d_tag == DT_STRTAB) strings = (char*)ABS_ADDR(info->dlpi_addr, dyn_section[j].d_un.d_ptr); + + if (dyn_section[j].d_tag == DT_STRSZ) limit = dyn_section[j].d_un.d_val; + } + + if (strings == nullptr) debug_print("String table not found"); + + if (strings != nullptr) { + char* end = strings + limit; + while (strings < end) { + if (strcmp(strings, "HSA_AMD_TOOL_PRIORITY") == 0) { + loadedToolsLib->push_back(info->dlpi_name); + return 0; + } + strings += (strlen(strings) + 1); + } + } + } + } + } + return 0; +} + +std::vector GetLoadedToolsLib() { + std::vector ret; + std::vector names; + + dl_iterate_phdr(callback, &names); + + if (!names.empty()) { + for (auto& name : names) ret.push_back(LoadLib(name)); + } + + return ret; +} + +std::string GetLibraryName(LibHandle lib) { + link_map *map; + if(dlinfo(lib, RTLD_DI_LINKMAP, &map)!=0) + return ""; + return map->l_name; +} + +Semaphore CreateSemaphore() { + sem_t *sem = new sem_t; + sem_init(sem, 0, 0); + return *(Semaphore*)&sem; +} + +bool WaitSemaphore(Semaphore sem) { + while(sem_wait(*(sem_t**)&sem)) + if (errno != EINTR) return false; + + return true; +} + +void PostSemaphore(Semaphore sem) { + int waitval = 1; + if (sem_getvalue(*(sem_t**)&sem, &waitval)) + assert(false && "Failed to get semaphore waiters"); + + if (waitval > 0) + return; + + if (sem_post(*(sem_t**)&sem)) + assert(false && "Failed to post semaphore"); +} + +void DestroySemaphore(Semaphore sem) { + sem_destroy(*(sem_t**)&sem); + delete *(sem_t**)&sem; +} + +Mutex CreateMutex() { + pthread_mutex_t* mutex = new pthread_mutex_t; + pthread_mutex_init(mutex, NULL); + return *(Mutex*)&mutex; +} + +bool TryAcquireMutex(Mutex lock) { + return pthread_mutex_trylock(*(pthread_mutex_t**)&lock) == 0; +} + +bool AcquireMutex(Mutex lock) { + return pthread_mutex_lock(*(pthread_mutex_t**)&lock) == 0; +} + +void ReleaseMutex(Mutex lock) { + pthread_mutex_unlock(*(pthread_mutex_t**)&lock); +} + +void DestroyMutex(Mutex lock) { + pthread_mutex_destroy(*(pthread_mutex_t**)&lock); + delete *(pthread_mutex_t**)&lock; +} + +void Sleep(int delay_in_millisec) { usleep(delay_in_millisec * 1000); } + +void uSleep(int delayInUs) { usleep(delayInUs); } + +void YieldThread() { sched_yield(); } + +Thread CreateThread(ThreadEntry function, void* threadArgument, uint stackSize, int priority) { + os_thread* result = new os_thread(function, threadArgument, stackSize, priority); + if (!result->Valid()) { + delete result; + return nullptr; + } + + return reinterpret_cast(result); +} + +void CloseThread(Thread thread) { delete reinterpret_cast(thread); } + +bool WaitForThread(Thread thread) { return reinterpret_cast(thread)->Wait(); } + +bool WaitForAllThreads(Thread* threads, uint threadCount) { + for (uint i = 0; i < threadCount; i++) WaitForThread(threads[i]); + return true; +} + +bool IsEnvVarSet(std::string env_var_name) { + char* buff = NULL; + buff = getenv(env_var_name.c_str()); + return (buff != NULL); +} + +void SetEnvVar(std::string env_var_name, std::string env_var_value) { + setenv(env_var_name.c_str(), env_var_value.c_str(), 1); +} + +int GetProcessId() { + return ::getpid(); +} + +std::string GetEnvVar(std::string env_var_name) { + char* buff; + buff = getenv(env_var_name.c_str()); + std::string ret; + if (buff) { + ret = buff; + } + return ret; +} + +size_t GetUserModeVirtualMemorySize() { +#ifdef _LP64 + return (size_t)(0x800000000000); +#else + return (size_t)(0xffffffff); // ~4GB +#endif +} + +size_t GetUsablePhysicalHostMemorySize() { + unsigned long physmem; + size_t len = sizeof(physmem); + int mib[2] = { CTL_HW, HW_PHYSMEM }; + if (sysctl(mib, 2, &physmem, &len, NULL, 0) != 0) { + return 0; + } + return std::min(GetUserModeVirtualMemorySize(), (size_t)physmem); +} + +uintptr_t GetUserModeVirtualMemoryBase() { return (uintptr_t)0; } + +// Os event implementation +typedef struct EventDescriptor_ { + pthread_cond_t event; + pthread_mutex_t mutex; + bool state; + bool auto_reset; +} EventDescriptor; + +EventHandle CreateOsEvent(bool auto_reset, bool init_state) { + EventDescriptor* eventDescrp; + eventDescrp = (EventDescriptor*)malloc(sizeof(EventDescriptor)); + + if(!eventDescrp) { return nullptr; } + + pthread_mutex_init(&eventDescrp->mutex, NULL); + pthread_cond_init(&eventDescrp->event, NULL); + eventDescrp->auto_reset = auto_reset; + eventDescrp->state = init_state; + + EventHandle handle = reinterpret_cast(eventDescrp); + + return handle; +} + +int DestroyOsEvent(EventHandle event) { + if (event == NULL) { + return -1; + } + + EventDescriptor* eventDescrp = reinterpret_cast(event); + int ret_code = pthread_cond_destroy(&eventDescrp->event); + ret_code |= pthread_mutex_destroy(&eventDescrp->mutex); + free(eventDescrp); + return ret_code; +} + +int WaitForOsEvent(EventHandle event, unsigned int milli_seconds) { + if (event == NULL) { + return -1; + } + + EventDescriptor* eventDescrp = reinterpret_cast(event); + if (milli_seconds == 0) { + int tmp_ret = pthread_mutex_trylock(&eventDescrp->mutex); + if (tmp_ret == EBUSY) { + return 1; + } + } else { + pthread_mutex_lock(&eventDescrp->mutex); + } + + int ret_code = 0; + + if (!eventDescrp->state) { + if (milli_seconds == 0) { + ret_code = 1; + } else { + struct timespec ts; + struct timeval tp; + + ret_code = gettimeofday(&tp, NULL); + ts.tv_sec = tp.tv_sec; + ts.tv_nsec = tp.tv_usec * 1000; + + unsigned int sec = milli_seconds / 1000; + unsigned int mSec = milli_seconds % 1000; + + ts.tv_sec += sec; + ts.tv_nsec += mSec * 1000000; + + if (ts.tv_nsec > 1000000000) { + ts.tv_sec += 1; + ts.tv_nsec = ts.tv_nsec - 1000000000; + } + + ret_code = + pthread_cond_timedwait(&eventDescrp->event, &eventDescrp->mutex, &ts); + if (ret_code == 110) { + ret_code = 0x14003; + } + + if (ret_code == 0 && eventDescrp->auto_reset) { + eventDescrp->state = false; + } + } + } else if (eventDescrp->auto_reset) { + eventDescrp->state = false; + } + pthread_mutex_unlock(&eventDescrp->mutex); + + return ret_code; +} + +int SetOsEvent(EventHandle event) { + if (event == NULL) { + return -1; + } + + EventDescriptor* eventDescrp = reinterpret_cast(event); + int ret_code = 0; + ret_code = pthread_mutex_lock(&eventDescrp->mutex); + eventDescrp->state = true; + ret_code = pthread_mutex_unlock(&eventDescrp->mutex); + ret_code |= pthread_cond_signal(&eventDescrp->event); + + return ret_code; +} + +int ResetOsEvent(EventHandle event) { + if (event == NULL) { + return -1; + } + + EventDescriptor* eventDescrp = reinterpret_cast(event); + int ret_code = 0; + ret_code = pthread_mutex_lock(&eventDescrp->mutex); + eventDescrp->state = false; + ret_code = pthread_mutex_unlock(&eventDescrp->mutex); + + return ret_code; +} + +static double invPeriod = 0.0; + +uint64_t ReadAccurateClock() { + if (invPeriod == 0.0) AccurateClockFrequency(); + timespec time; + int err = clock_gettime(CLOCK_MONOTONIC, &time); + if (err != 0) { + perror("clock_gettime(CLOCK_MONOTONIC,...) failed"); + abort(); + } + return (uint64_t(time.tv_sec) * 1000000000ull + uint64_t(time.tv_nsec)) * invPeriod; +} + +uint64_t AccurateClockFrequency() { + static clockid_t clock = CLOCK_MONOTONIC; + timespec time; + int err = clock_getres(clock, &time); + if (err != 0) { + perror("clock_getres failed"); + abort(); + } + if (time.tv_sec != 0 || time.tv_nsec >= 0xFFFFFFFF) { + fprintf(stderr, + "clock_getres(CLOCK_MONOTONIC,...) returned very low " + "frequency (<1Hz).\n"); + abort(); + } + if (invPeriod == 0.0) invPeriod = 1.0 / double(time.tv_nsec); + return 1000000000ull / uint64_t(time.tv_nsec); +} + +SharedMutex CreateSharedMutex() { + pthread_rwlockattr_t attrib; + int err = pthread_rwlockattr_init(&attrib); + if (err != 0) { + fprintf(stderr, "rw lock attribute init failed: %s\n", strerror(err)); + return nullptr; + } + +#ifdef HAVE_PTHREAD_RWLOCKATTR_SETKIND_NP + err = pthread_rwlockattr_setkind_np(&attrib, PTHREAD_RWLOCK_PREFER_WRITER_NONRECURSIVE_NP); + if (err != 0) { + fprintf(stderr, "Set rw lock attribute failure: %s\n", strerror(err)); + return nullptr; + } +#endif + + std::unique_ptr lock(new pthread_rwlock_t); + err = pthread_rwlock_init(lock.get(), &attrib); + if (err != 0) { + fprintf(stderr, "rw lock init failed: %s\n", strerror(err)); + return nullptr; + } + + pthread_rwlockattr_destroy(&attrib); + return lock.release(); +} + +bool TryAcquireSharedMutex(SharedMutex lock) { + int err = pthread_rwlock_trywrlock(*(pthread_rwlock_t**)&lock); + return err == 0; +} + +bool AcquireSharedMutex(SharedMutex lock) { + int err = pthread_rwlock_wrlock(*(pthread_rwlock_t**)&lock); + return err == 0; +} + +void ReleaseSharedMutex(SharedMutex lock) { + int err = pthread_rwlock_unlock(*(pthread_rwlock_t**)&lock); + if (err != 0) { + fprintf(stderr, "SharedMutex unlock failed: %s\n", strerror(err)); + abort(); + } +} + +bool TrySharedAcquireSharedMutex(SharedMutex lock) { + int err = pthread_rwlock_tryrdlock(*(pthread_rwlock_t**)&lock); + return err == 0; +} + +bool SharedAcquireSharedMutex(SharedMutex lock) { + int err = pthread_rwlock_rdlock(*(pthread_rwlock_t**)&lock); + return err == 0; +} + +void SharedReleaseSharedMutex(SharedMutex lock) { + int err = pthread_rwlock_unlock(*(pthread_rwlock_t**)&lock); + if (err != 0) { + fprintf(stderr, "SharedMutex unlock failed: %s\n", strerror(err)); + abort(); + } +} + +void DestroySharedMutex(SharedMutex lock) { + pthread_rwlock_destroy(*(pthread_rwlock_t**)&lock); + delete *(pthread_rwlock_t**)&lock; +} + +static uint64_t sys_clock_period_ = 0; + +uint64_t ReadSystemClock() { + struct timespec ts; + clock_gettime(CLOCK_BOOTTIME, &ts); + uint64_t time = (uint64_t(ts.tv_sec) * 1000000000 + uint64_t(ts.tv_nsec)); + if (sys_clock_period_ != 1) + return time / sys_clock_period_; + else + return time; +} + +uint64_t SystemClockFrequency() { + struct timespec ts; + clock_getres(CLOCK_BOOTTIME, &ts); + sys_clock_period_ = (uint64_t(ts.tv_sec) * 1000000000 + uint64_t(ts.tv_nsec)); + return 1000000000 / sys_clock_period_; +} + +bool ParseCpuID(cpuid_t* cpuinfo) { +#if defined(__i386__) || defined(__x86_64__) + uint32_t eax, ebx, ecx, edx, max_eax = 0; + memset(cpuinfo, 0, sizeof(*cpuinfo)); + + if (!__get_cpuid_max(0x80000004, NULL)) return false; + + if (!__get_cpuid(0, &max_eax, (uint32_t*)&cpuinfo->ManufacturerID[0], + (uint32_t*)&cpuinfo->ManufacturerID[8], + (uint32_t*)&cpuinfo->ManufacturerID[4])) { + return false; + } + + if (!strcmp(cpuinfo->ManufacturerID, "AuthenticAMD")) { + if (__get_cpuid(0x80000001, &eax, &ebx, &ecx, &edx)) { + cpuinfo->mwaitx = !!((ecx >> 29) & 0x1); + } + } + return true; +#else + return false; +#endif +} + +uint64_t TimeNanos() { + struct timespec tp; + ::clock_gettime(CLOCK_MONOTONIC, &tp); + return (uint64_t)tp.tv_sec * (1000ULL * 1000ULL * 1000ULL) + (uint64_t)tp.tv_nsec; +} + +static inline int MemProtToOsProt(MemProt prot) { + switch (prot) { + case MEM_PROT_NONE: + return PROT_NONE; + case MEM_PROT_READ: + return PROT_READ; + case MEM_PROT_RW: + return PROT_READ | PROT_WRITE; + case MEM_PROT_RWX: + return PROT_READ | PROT_WRITE | PROT_EXEC; + default: + break; + } + return -1; +} + +size_t PageSize() { + static size_t g_page_size_ = 0; + if (g_page_size_ == 0) { + g_page_size_ = (size_t)::sysconf(_SC_PAGESIZE); + } + return g_page_size_; +} + +bool UnmapMemory(void* va, size_t size) { return ::munmap(va, size) == 0; } + +bool MapMemory(void* va, size_t size, MemProt perms, int fd, uint64_t cpu_addr) { + void* mapped_ptr = ::mmap(va, size, MemProtToOsProt(perms), + MAP_SHARED | MAP_FIXED, fd, cpu_addr); + if (mapped_ptr != va) + return false; + return true; +} + +void* ReserveMemory(void* start, size_t size, size_t alignment, MemProt prot) { + size = AlignUp(size, PageSize()); + if (size == 0) { + return NULL; + } + alignment = std::max(PageSize(), AlignUp(alignment, PageSize())); + assert(IsPowerOfTwo(alignment) && "not a power of 2"); + + size_t requested = size + alignment - PageSize(); + address mem = (address)::mmap(start, requested, MemProtToOsProt(prot), + MAP_PRIVATE | MAP_NORESERVE | MAP_ANONYMOUS, 0, 0); + + if (mem == MAP_FAILED) return NULL; + + address aligned = AlignUp(mem, alignment); + + if (&aligned[0] != &mem[0]) { + assert(&aligned[0] > &mem[0] && "check this code"); + if (::munmap(&mem[0], &aligned[0] - &mem[0]) != 0) { + assert(!"::munmap failed"); + } + } + if (&aligned[size] != &mem[requested]) { + assert(&aligned[size] < &mem[requested] && "check this code"); + if (::munmap(&aligned[size], &mem[requested] - &aligned[size]) != 0) { + assert(!"::munmap failed"); + } + } + + constexpr size_t kLargePageSize = 2 * 1024 * 1024; + if (size >= kLargePageSize) { + int status = madvise(aligned, size, MADV_HUGEPAGE); + if (status) { + fprintf(stderr, + "madvise with advice MADV_HUGEPAGE" + " starting at address %p and page size 0x%zx, returned %d, errno: %s", + aligned, size, status, strerror(errno)); + } + } + + return aligned; +} + +bool ReleaseMemory(void* addr, size_t size) { + assert(IsMultipleOf(addr, PageSize()) && "not page aligned!"); + size = AlignUp(size, PageSize()); + + return 0 == ::munmap(addr, size); +} + +bool CommitMemory(void* addr, size_t size, MemProt prot) { + assert(IsMultipleOf(addr, PageSize()) && "not page aligned!"); + size = AlignUp(size, PageSize()); + + return ::mmap(addr, size, MemProtToOsProt(prot), MAP_PRIVATE | MAP_FIXED | MAP_ANONYMOUS, -1, + 0) != MAP_FAILED; +} + +bool UncommitMemory(void* addr, size_t size) { + assert(IsMultipleOf(addr, PageSize()) && "not page aligned!"); + size = AlignUp(size, PageSize()); + + return ::mmap(addr, size, PROT_NONE, MAP_PRIVATE | MAP_FIXED | MAP_NORESERVE | MAP_ANONYMOUS, -1, + 0) != MAP_FAILED; +} + +bool ProtectMemory(void* va, size_t size, MemProt perms) { + return ::mprotect(va, size, MemProtToOsProt(perms)) == 0; +} + +uint64_t HostTotalPhysicalMemory() { + static uint64_t totalPhys = 0; + + if (totalPhys != 0) { + return totalPhys; + } + + totalPhys = sysconf(_SC_PAGESIZE) * sysconf(_SC_PHYS_PAGES); + return totalPhys; +} + +int Ffs(int i) { return ffs(i); } + +int Ctz(uint64_t i) { return __builtin_ctz(i); } + +int Popcount(uint32_t i) { return __builtin_popcount(i); } + +char* DlError() { return dlerror(); } + +static inline int IPCSockToFd(IPCSocket sock) { + return static_cast(sock); +} + +static inline IPCSocket FdToIPCSock(int fd) { + return static_cast(fd); +} + +IPCSocket CreateIPCServer(const char* name, int backlog) { + int fd = socket(AF_UNIX, SOCK_STREAM, 0); + if (fd == -1) return INVALID_SOCKET_VALUE; + + struct sockaddr_un address; + memset(&address, 0, sizeof(address)); + address.sun_family = AF_UNIX; + snprintf(address.sun_path, sizeof(address.sun_path), "/tmp/%s", name); + + unlink(address.sun_path); + + if (bind(fd, (struct sockaddr*)&address, sizeof(address)) != 0) { + close(fd); + return INVALID_SOCKET_VALUE; + } + if (listen(fd, backlog) != 0) { + close(fd); + return INVALID_SOCKET_VALUE; + } + return FdToIPCSock(fd); +} + +IPCSocket AcceptIPCConnection(IPCSocket server) { + int fd = accept(IPCSockToFd(server), NULL, NULL); + if (fd == -1) return INVALID_SOCKET_VALUE; + return FdToIPCSock(fd); +} + +IPCSocket ConnectToIPCServer(const char* name, std::chrono::milliseconds timeout, + std::chrono::milliseconds retryInterval) { + int fd = socket(AF_UNIX, SOCK_STREAM, 0); + if (fd == -1) return INVALID_SOCKET_VALUE; + + struct sockaddr_un address; + memset(&address, 0, sizeof(address)); + address.sun_family = AF_UNIX; + snprintf(address.sun_path, sizeof(address.sun_path), "/tmp/%s", name); + + auto deadline = std::chrono::steady_clock::now() + timeout; + while (std::chrono::steady_clock::now() < deadline) { + if (connect(fd, (struct sockaddr*)&address, sizeof(address)) == 0) + return FdToIPCSock(fd); + usleep(static_cast(retryInterval.count()) * 1000); + } + + close(fd); + return INVALID_SOCKET_VALUE; +} + +void SetIPCSocketRecvTimeout(IPCSocket sock, std::chrono::seconds timeout) { + struct timeval tv; + tv.tv_sec = static_cast(timeout.count()); + tv.tv_usec = 0; + setsockopt(IPCSockToFd(sock), SOL_SOCKET, SO_RCVTIMEO, &tv, sizeof(tv)); +} + +int IPCSocketRead(IPCSocket conn, void* buf, size_t len) { + return static_cast(read(IPCSockToFd(conn), buf, len)); +} + +int IPCSocketWrite(IPCSocket conn, const void* buf, size_t len) { + return static_cast(write(IPCSockToFd(conn), buf, len)); +} + +int IPCSendHandle(IPCSocket conn, intptr_t handle) { + int fd = static_cast(handle); + char iov_buf[1] = {'y'}; + struct iovec io = {.iov_base = iov_buf, .iov_len = 1}; + + char cmsg_buf[CMSG_SPACE(sizeof(int))]; + memset(cmsg_buf, 0, sizeof(cmsg_buf)); + + struct msghdr msg = {}; + msg.msg_iov = &io; + msg.msg_iovlen = 1; + msg.msg_control = cmsg_buf; + msg.msg_controllen = sizeof(cmsg_buf); + + struct cmsghdr* cmsg = CMSG_FIRSTHDR(&msg); + if (!cmsg) return -1; + cmsg->cmsg_level = SOL_SOCKET; + cmsg->cmsg_type = SCM_RIGHTS; + cmsg->cmsg_len = CMSG_LEN(sizeof(int)); + memcpy(CMSG_DATA(cmsg), &fd, sizeof(int)); + + msg.msg_controllen = CMSG_SPACE(sizeof(int)); + + return (sendmsg(IPCSockToFd(conn), &msg, 0) < 0) ? -1 : 0; +} + +intptr_t IPCRecvHandle(IPCSocket conn) { + char m_buffer[1]; + struct iovec io = {.iov_base = m_buffer, .iov_len = sizeof(m_buffer)}; + + char c_buffer[256]; + struct msghdr msg = {}; + msg.msg_iov = &io; + msg.msg_iovlen = 1; + msg.msg_control = c_buffer; + msg.msg_controllen = sizeof(c_buffer); + + ssize_t rcv = recvmsg(IPCSockToFd(conn), &msg, MSG_WAITALL); + if (rcv < 0) return -1; + + while (!rcv) + rcv = recvmsg(IPCSockToFd(conn), &msg, MSG_WAITALL); + + struct cmsghdr* cmsg = CMSG_FIRSTHDR(&msg); + if (!cmsg) return -1; + int fd; + memcpy(&fd, CMSG_DATA(cmsg), sizeof(fd)); + return fd; +} + +void CloseIPCSocket(IPCSocket sock) { + if (sock != INVALID_SOCKET_VALUE) + close(IPCSockToFd(sock)); +} + +} // namespace os +} // namespace rocr + +#endif diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/util/memory.h b/projects/rocr-runtime/runtime/hsa-runtime/core/util/memory.h index 4c04b59c50..1f98c67904 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/util/memory.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/util/memory.h @@ -53,7 +53,7 @@ #include "os.h" namespace rocr { -#if defined(__linux__) +#if defined(__linux__) || defined(__FreeBSD__) /// @brief Converts @ref hsa_access_permission_t to mmap memory protection /// flags. __forceinline int PermissionsToMmapFlags(hsa_access_permission_t perms) { diff --git a/projects/rocr-runtime/runtime/hsa-runtime/core/util/os.h b/projects/rocr-runtime/runtime/hsa-runtime/core/util/os.h index 087dcfe12c..339cd0a18a 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/core/util/os.h +++ b/projects/rocr-runtime/runtime/hsa-runtime/core/util/os.h @@ -50,6 +50,18 @@ #include #include "utils.h" +#ifdef __FreeBSD__ +#include +#ifndef MAP_NORESERVE +#define MAP_NORESERVE 0 + +#ifndef MADV_HUGEPAGE +#define MADV_HUGEPAGE 0 +#endif + +#endif +#endif + namespace rocr { namespace os { typedef void* LibHandle; @@ -70,9 +82,10 @@ static __forceinline std::underlying_type::type os_index(os_t val) { return std::underlying_type::type(val); } +// who the fuck uses this #ifdef _WIN32 static const os_t current_os = os_t::OS_WIN; -#elif __linux__ +#elif defined(__linux__) || defined(__FreeBSD__) static const os_t current_os = os_t::OS_LINUX; #else static_assert(false, "Operating System not detected!"); diff --git a/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/amd_elf_image.cpp b/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/amd_elf_image.cpp index 1293bc8d9e..502250861f 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/amd_elf_image.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/amd_elf_image.cpp @@ -86,11 +86,16 @@ #define _write write #define _lseek lseek #define _ftruncate ftruncate -#include #else #define _ftruncate _chsize #endif // !_WIN32 +#ifndef __FreeBSD__ +#ifdef __linux__ +#include +#endif +#endif + #endif // !USE_MEMFILE #if !defined(BSD_LIBELF) @@ -227,6 +232,7 @@ namespace elf { return perror("lseek(3) failed"); } if (_lseek(d, 0L, SEEK_SET) < 0) { return perror("lseek(3) failed"); } +#if defined(__linux__) && !defined(__FreeBSD__) ssize_t written; do { written = sendfile(d, in, NULL, size); @@ -236,6 +242,30 @@ namespace elf { } size -= written; } while (size > 0); +#else + char buffer[4096]; + while (size > 0) { + off_t to_read = (size < (off_t)sizeof(buffer)) ? size : (off_t)sizeof(buffer); + ssize_t bytes_read = _read(in, buffer, to_read); + if (bytes_read < 0) { + _close(in); + return perror("read failed"); + } + if (bytes_read == 0) { + break; // EOF + } + ssize_t bytes_written = 0; + while (bytes_written < bytes_read) { + ssize_t written = _write(d, buffer + bytes_written, bytes_read - bytes_written); + if (written < 0) { + _close(in); + return perror("write failed"); + } + bytes_written += written; + } + size -= bytes_read; + } +#endif _close(in); if (_lseek(d, 0L, SEEK_SET) < 0) { return perror("lseek(0) failed"); } return true; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/lnx/amd_core_dump.cpp b/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/lnx/amd_core_dump.cpp index 5039470dd6..2581a42147 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/lnx/amd_core_dump.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/libamdhsacode/lnx/amd_core_dump.cpp @@ -64,6 +64,11 @@ #include "core/inc/amd_gpu_agent.h" #include "core/inc/amd_aql_queue.h" +#ifdef __FreeBSD__ +#include +#define SYS_gettid thr_self(0) +#endif + constexpr char SNAPSHOT_INFO_ALIGNMENT = 0x8; constexpr uint32_t LOAD_ALIGNMENT_SHIFT = 4; constexpr uint32_t NOTE_ALIGNMENT_SHIFT = 2; diff --git a/projects/rocr-runtime/runtime/hsa-runtime/loader/executable.cpp b/projects/rocr-runtime/runtime/hsa-runtime/loader/executable.cpp index 772cd36722..aef26a7c0d 100644 --- a/projects/rocr-runtime/runtime/hsa-runtime/loader/executable.cpp +++ b/projects/rocr-runtime/runtime/hsa-runtime/loader/executable.cpp @@ -237,7 +237,7 @@ static void RemoveCodeObjectInfoFromDebugMap(link_map* map) { map->l_next->l_prev = map->l_prev; } - free(map->l_name); + free(const_cast(map->l_name)); memset(map, 0, sizeof(link_map)); } @@ -1323,7 +1323,7 @@ hsa_status_t ExecutableImpl::LoadCodeObject( } } - loaded_code_objects.back()->r_debug_info.l_addr = loaded_code_objects.back()->getDelta(); + loaded_code_objects.back()->r_debug_info.l_addr = (decltype(link_map::l_addr))loaded_code_objects.back()->getDelta(); loaded_code_objects.back()->r_debug_info.l_name = strdup(uri.c_str()); loaded_code_objects.back()->r_debug_info.l_prev = nullptr; loaded_code_objects.back()->r_debug_info.l_next = nullptr;